/* * Pure helpers for lifting structured `app.bsky.feed.searchPosts` params out of * a free-text query. Kept free of React Native imports so it can be unit * tested in isolation (the search-posts query hook re-exports these). */ import {type l} from '@atproto/lex' import { filtersToApiParams, type SearchFilters, } from '#/screens/Search/searchParams' import {type app} from '#/lexicons' /** * Input params for `app.bsky.feed.searchPostsV2`. Uses the schema's input type * (not `$Params`, which is the output type with defaults applied and `limit` * required) since these are the params we build up and pass to `client.call`. */ type SearchPostsV2Params = l.InferInput< typeof app.bsky.feed.searchPostsV2.$params > const DATE_RE = /^\d{4}-\d{2}-\d{2}/ /** * Strips a leading `@` from a handle so `from:@alice.bsky.social` and * `from:alice.bsky.social` both resolve to the same author. Mirrors the marker * stripping the advanced-search dialog applies to handles entered in its * filter fields (see `serializeAdvancedSearch`), which otherwise 400s the * appview. */ function stripHandleMarker(value: string): string { return value.startsWith('@') ? value.slice(1) : value } export type ExtractedSearchParams = { q: string author?: string mentions?: string domain?: string url?: string lang?: string since?: string until?: string tag?: string[] } /** * Splits a query into whitespace-delimited tokens, keeping quoted phrases * ("a b") and parenthesized OR groups ((a OR b)) intact so they pass through to * `q` untouched. Shared with the advanced-search dialog's parser (view layer), * which imports it from here so the two stay in sync. */ export function tokenizeQuery(raw: string): string[] { const tokens: string[] = [] let i = 0 const n = raw.length while (i < n) { if (/\s/.test(raw[i])) { i++ continue } const start = i if (raw[i] === '(') { let depth = 0 while (i < n) { if (raw[i] === '(') depth++ else if (raw[i] === ')') { depth-- if (depth === 0) { i++ break } } i++ } tokens.push(raw.slice(start, i)) continue } let buf = '' while (i < n && !/\s/.test(raw[i]) && raw[i] !== '(') { if (raw[i] === '"') { buf += raw[i++] while (i < n && raw[i] !== '"') buf += raw[i++] if (i < n) buf += raw[i++] } else { buf += raw[i++] } } if (buf) tokens.push(buf) } return tokens } /** * Splits a bare `from:me` token out of a query. The "Me" author filter always * travels inside `q` as a `from:me` token (the backend resolves `me` to the * viewer), but the UI never shows it as text: the search input strips it for * display and the advanced-search dialog represents it in the From dropdown. * Tokenization keeps quoted phrases intact, so a `from:me` inside quotes stays * in the query text. */ export function extractFromMe(query: string): {q: string; fromMe: boolean} { const tokens = tokenizeQuery(query) const kept = tokens.filter(token => token !== 'from:me') return {q: kept.join(' '), fromMe: kept.length !== tokens.length} } /** * Re-appends the `from:me` token when the "Me" author filter is active. * Idempotent: a query that already carries a bare `from:me` is returned as-is. */ export function appendFromMe(query: string, fromMe: boolean): string { if (!fromMe) return query if (tokenizeQuery(query).includes('from:me')) return query return query ? `${query} from:me` : 'from:me' } /** * Lifts the operators that `app.bsky.feed.searchPosts` accepts as structured * params out of the free-text query, so the backend filters on them directly. * Recognized operators are stripped from `q`; everything else (free text, * quoted phrases, OR groups, negations, and unsupported operators like * `replies:`, `media:`) is left in `q` verbatim. Singular params keep the first * value seen; `tag` accumulates (the lexicon AND-matches multiple tags). */ export function extractSearchPostsParams(query: string): ExtractedSearchParams { const result: ExtractedSearchParams = {q: ''} const remaining: string[] = [] const tags: string[] = [] for (const token of tokenizeQuery(query)) { if (token.startsWith('#') && token.length > 1 && !token.includes(':')) { tags.push(token.slice(1)) continue } const colonIdx = token.indexOf(':') if (colonIdx === -1) { remaining.push(token) continue } const op = token.slice(0, colonIdx) const value = token.slice(colonIdx + 1) if (!value) { remaining.push(token) continue } switch (op) { case 'from': /* * `me` is resolved to the viewer by the backend, so leave it in the * query text verbatim rather than lifting it into a structured param. */ if (value === 'me') remaining.push(token) else result.author ??= stripHandleMarker(value) break case 'mentions': case 'to': if (value === 'me') remaining.push(token) else result.mentions ??= stripHandleMarker(value) break case 'domain': result.domain ??= value break case 'url': result.url ??= value break case 'lang': result.lang ??= value break case 'since': if (DATE_RE.test(value)) result.since ??= value else remaining.push(token) break case 'until': if (DATE_RE.test(value)) result.until ??= value else remaining.push(token) break default: // Unsupported operator (to:, replies:, media:, etc.) - keep in q. remaining.push(token) } } if (tags.length) result.tag = tags result.q = remaining.join(' ') return result } /** * Concatenates two optional value lists, dropping empties and duplicates while * preserving order. Used to union the back-compat operators embedded in the * query string with the explicit dialog filters so neither source clobbers the * other. */ function mergeList(a?: string[], b?: string[]): string[] | undefined { const merged = [...new Set([...(a ?? []), ...(b ?? [])])] return merged.length ? merged : undefined } /** * Builds the `app.bsky.feed.searchPostsV2` query params (minus q/limit/cursor/ * sort, which the caller owns) from the operators embedded in the query string * plus the structured advanced-search dialog filters. The two sources are * merged rather than overriding each other: list fields union their values, and * scalar fields prefer the explicit dialog filter, falling back to the embedded * operator. v2 renames v1's singular operators to plural arrays, and `lang` to * `language`. */ export function buildSearchPostsV2Filters( embedded: Omit, filters?: SearchFilters, ): SearchPostsV2Params { const apiFilters = filters ? filtersToApiParams(filters) : {} const params: SearchPostsV2Params = {} /* * The values below originate from free-text operators and the advanced-search * dialog as plain strings; the lexicon input params brand them by format * (at-identifier, uri, etc.). The backend validates these, so we assert the * branded types at assignment rather than validating client-side. */ const authors = mergeList( embedded.author ? [embedded.author] : undefined, apiFilters.authors, ) if (authors) params.authors = authors as SearchPostsV2Params['authors'] const mentions = mergeList( embedded.mentions ? [embedded.mentions] : undefined, apiFilters.mentions, ) if (mentions) params.mentions = mentions as SearchPostsV2Params['mentions'] const domains = mergeList( embedded.domain ? [embedded.domain] : undefined, apiFilters.domains, ) if (domains) params.domains = domains const urls = mergeList( embedded.url ? [embedded.url] : undefined, apiFilters.urls, ) if (urls) params.urls = urls as SearchPostsV2Params['urls'] const hashtags = mergeList(embedded.tag, apiFilters.hashtags) if (hashtags) params.hashtags = hashtags const language = apiFilters.language ?? embedded.lang // TODO At the moment, the language selector is single-select. -dsb if (language) params.languages = [language] const since = parseTimestamp(apiFilters.since ?? embedded.since) if (since) params.since = since const until = parseTimestamp(apiFilters.until ?? embedded.until) if (until) params.until = until /* * Exclude lists have no embedded query-string source (operators like `from:` * are always include), so they pass straight through from the dialog filters. */ if (apiFilters.excludeAuthors) params.excludeAuthors = apiFilters.excludeAuthors as SearchPostsV2Params['excludeAuthors'] if (apiFilters.excludeMentions) params.excludeMentions = apiFilters.excludeMentions as SearchPostsV2Params['excludeMentions'] if (apiFilters.excludeDomains) params.excludeDomains = apiFilters.excludeDomains if (apiFilters.excludeUrls) params.excludeUrls = apiFilters.excludeUrls as SearchPostsV2Params['excludeUrls'] if (apiFilters.excludeHashtags) params.excludeHashtags = apiFilters.excludeHashtags if (apiFilters.hasMedia) params.hasMedia = true if (apiFilters.hasVideo) params.hasVideo = true if (apiFilters.following) params.following = true if (apiFilters.excludeReplies) params.excludeReplies = true if (apiFilters.repliesOnly) params.repliesOnly = true return params } /** * Consistent with atproto timestamp parsing. Only the date is used; the time * is appended here since the lexicon expects a datetime value. */ const parseTimestamp = (value: string | undefined): string | undefined => { if (!value) return undefined const date = new Date(value) if (isNaN(date.getTime())) return undefined return date.toISOString().split('.')[0] + 'Z' }