diff options
Diffstat (limited to 'src/search.ts')
| -rw-r--r-- | src/search.ts | 150 |
1 files changed, 105 insertions, 45 deletions
diff --git a/src/search.ts b/src/search.ts index fa7ce35..1306443 100644 --- a/src/search.ts +++ b/src/search.ts @@ -15,8 +15,26 @@ import { describeResolutionFailure, resolveCdpTarget, } from "./endpoint.ts"; -import { BrowserUnavailableError, NoResultsError, SearchChallengeError } from "./errors.ts"; -import { PROBE_SCRIPT, type PageProbe, type SearchResult } from "./extract.ts"; +import { describeEngineFailure, describeEngineSource, resolveEngine } from "./engine-target.ts"; +import { readStoredEngine } from "./engine-store.ts"; +import { + ENGINE_IDS as allEngineIdList, + type EngineDefinition, + type EngineId, + type Recency, + describeRecencySupport, + enginesSupportingRecency, + getEngine, +} from "./engines.ts"; + +const allEngineIds = (): readonly EngineId[] => allEngineIdList; +import { + BrowserUnavailableError, + NoResultsError, + RecencyUnsupportedError, + SearchChallengeError, +} from "./errors.ts"; +import { buildProbeScript, type PageProbe, type SearchResult } from "./extract.ts"; import { readStoredTarget } from "./target-store.ts"; /** Per-search wall clock, covering connect, navigate and extract together. */ @@ -27,12 +45,14 @@ export const MAX_CONCURRENT_SEARCHES = 2; const EXTRACT_RETRIES = 2; const EXTRACT_RETRY_DELAY_MS = 700; -export type Recency = "day" | "week" | "month" | "year"; +export type { Recency } from "./engines.ts"; export type SearchRequest = { query: string; numResults: number; recency?: Recency | undefined; + /** Per-call engine override; falls back to the configured default. */ + engine?: string | null | undefined; }; export type SearchOutcome = { @@ -41,38 +61,59 @@ export type SearchOutcome = { /** The URL the page settled on, which differs from searchUrl after a redirect. */ finalUrl: string; endpoint: string; + engine: EngineDefinition; + /** Where the engine choice came from, for the "via" line in the output. */ + engineSource: string; }; -/** Google's date-restrict vocabulary. */ -const RECENCY_CODE: Record<Recency, string> = { day: "d", week: "w", month: "m", year: "y" }; +/** Build the SERP URL for a request. Exported for tests. */ +export const buildSearchUrl = (engine: EngineDefinition, request: SearchRequest): string => + engine.buildUrl(request.query, request.numResults, request.recency); /** - * `udm=14` asks for the plain "Web" tab: no AI overview, no carousels, just - * ranked links. That is both cheaper to parse and closer to what the agent - * asked for. `hl=en` pins the result language so snippets do not change shape - * with whatever locale the browser profile happens to carry. - * - * Two findings from testing against the real browser, both counter-intuitive - * enough to be worth writing down: - * - * - `tbs=qdr:*` — the parameter the Tools menu puts in the URL — renders an - * empty page in this profile, with `#search` present but no `#rso` and no - * results at all. `as_qdr=*`, the older advanced-search parameter, works and - * genuinely filters (verified: d/w/m/y all return dated results). - * - Any date restriction combined with `udm=14` also renders that empty page. + * Resolve the engine, refusing rather than silently ignoring a recency window + * the chosen engine cannot express. * - * So a time-limited search drops `udm` and uses `as_qdr`. The classic layout it - * falls back to parses fine — the extractor was checked against both. + * Dropping an unsupported filter would be the same failure mode as Google's + * `tbs=qdr:` — the caller asks for recent results, gets whatever the engine + * ranked, and has no way to tell. Naming the engines that *do* support the + * window makes the error actionable, since the model can retry with one. */ -export const buildSearchUrl = (request: SearchRequest): string => { - const params = new URLSearchParams({ - q: request.query, - num: String(request.numResults), - hl: "en", - }); - if (request.recency) params.set("as_qdr", RECENCY_CODE[request.recency]); - else params.set("udm", "14"); - return `https://www.google.com/search?${params.toString()}`; +export const selectEngine = async ( + request: SearchRequest, +): Promise<{ engine: EngineDefinition; source: string }> => { + const resolution = resolveEngine(request.engine, process.env, await readStoredEngine()); + if (resolution.kind === "invalid") { + throw new BrowserUnavailableError(describeEngineFailure(resolution)); + } + const engine = getEngine(resolution.id); + + if (request.recency && engine.recency[request.recency] === undefined) { + const alternatives = enginesSupportingRecency(request.recency); + throw new RecencyUnsupportedError( + `${engine.label} cannot restrict results to the past ${request.recency}. ` + + `It supports: ${describeRecencySupport(engine)}. ` + + (alternatives.length > 0 + ? `Engines that support "${request.recency}": ${alternatives.join(", ")}. ` + + `Retry with engine set to one of those, or drop the recency parameter.` + : `No configured engine supports that window; drop the recency parameter.`), + ); + } + + return { engine, source: describeEngineSource(resolution.source) }; +}; + +/** For status reporting — resolves the configured engine without searching. */ +export const describeConfiguredEngine = async (): Promise<{ id: EngineId; source: string; label: string }> => { + const resolution = resolveEngine(null, process.env, await readStoredEngine()); + if (resolution.kind === "invalid") { + return { id: "google", source: describeEngineFailure(resolution), label: "unresolved" }; + } + return { + id: resolution.id, + source: describeEngineSource(resolution.source), + label: getEngine(resolution.id).label, + }; }; // ── connection lifecycle ─────────────────────────────────────────────────── @@ -190,6 +231,14 @@ const sleep = (ms: number, signal?: AbortSignal): Promise<void> => raceAbort(new Promise<void>((resolve) => setTimeout(resolve, ms)), signal); /** + * The other engines, named in failure messages. A blocked engine is the case + * where the agent most needs to know it has somewhere else to go, and telling + * it inside the error beats hoping it remembers the guidelines. + */ +const otherEngines = (current: EngineId): readonly string[] => + allEngineIds().filter((id) => id !== current); + +/** * Run one search. Throws on every failure — see errors.ts for why a returned * error object would be worse than useless here. */ @@ -203,6 +252,12 @@ export const search = async ( // the caller's and covers the queue wait too — with a cap of two and searches // that take a couple of seconds, waiting is the rare case, and a tool call // that silently takes twice its stated deadline would be worse. + // Resolved before acquiring a slot: an unknown engine name or an unsupported + // recency window is a caller error, and should not wait behind other searches + // or touch the browser at all. + const { engine, source: engineSource } = await selectEngine(request); + const probeScript = buildProbeScript(engine.extraction); + report("queued", "waiting for a search slot"); const release = await acquire(signal); try { @@ -212,18 +267,18 @@ export const search = async ( report("opening", `opening a page on ${endpoint}`); const page = await PageSession.open(conn, signal); - const searchUrl = buildSearchUrl(request); + const searchUrl = buildSearchUrl(engine, request); try { - report("navigating", `searching for "${request.query}"`); + report("navigating", `searching ${engine.label} for "${request.query}"`); await page.navigate(searchUrl, signal); report("extracting", "reading results"); - let probe = await page.evaluate<PageProbe>(PROBE_SCRIPT, signal); + let probe = await page.evaluate<PageProbe>(probeScript, signal); for (let attempt = 0; attempt < EXTRACT_RETRIES; attempt++) { if (probe?.challenge || (probe?.results?.length ?? 0) > 0) break; await sleep(EXTRACT_RETRY_DELAY_MS, signal); - probe = await page.evaluate<PageProbe>(PROBE_SCRIPT, signal); + probe = await page.evaluate<PageProbe>(probeScript, signal); } if (!probe) { @@ -237,26 +292,29 @@ export const search = async ( // after a consent redirect is not the URL we asked for. pageUrl: probe.url || searchUrl, endpoint, + engine: engine.label, + alternatives: otherEngines(engine.id), detail: probe.challenge.detail || undefined, }); } if (probe.results.length === 0) { - // "#rso missing entirely" and "#rso present but empty" are different - // faults, and saying which saves the next person a browser session. + // "container missing entirely" and "container present but empty" are + // different faults, and saying which saves the next person a browser + // session. const diagnosis = - probe.container === "rso" - ? "The results container was rendered but nothing could be parsed out of it, which suggests Google's " + - "result markup changed and castle_cdp_search's extractor needs updating." - : "Google served a page with no results area at all. Most likely this is a soft rate-limit: as a " + - "block decays Google stops serving the /sorry/ challenge page and returns an empty result page " + - "instead, which is indistinguishable from a genuine zero-hit search (observed lasting ~90 minutes " + - "after heavy querying). It can also mean the query truly has no hits, or that this combination of " + - "search parameters is not honoured for this browser profile."; + probe.container === "found" + ? `${engine.label}'s results container was rendered but nothing could be parsed out of it, which ` + + `suggests its markup changed and castle_cdp_search's extractor for ${engine.id} needs updating.` + : `${engine.label} served a page with no results area at all. This is often a soft rate-limit: as a ` + + `block decays an engine stops serving its challenge page and returns an empty result page instead, ` + + `which is indistinguishable from a genuine zero-hit search (observed on Google lasting ~90 minutes ` + + `after heavy querying). It can also mean the query truly has no hits.`; throw new NoResultsError( `No results could be read from ${probe.url || searchUrl} (page title: "${probe.title}"). ${diagnosis} ` + - `Before concluding the extractor is broken, try a broader query and the same query without recency; ` + - `if those are also empty, treat it as rate limiting and wait rather than retrying in a loop.`, + `Before concluding the extractor is broken, retry with a different engine (${otherEngines(engine.id).join(", ")}); ` + + `if those also come back empty the query is the problem, and if only this one does, it is rate limiting — ` + + `wait rather than retrying in a loop.`, ); } @@ -265,6 +323,8 @@ export const search = async ( searchUrl, finalUrl: probe.url || searchUrl, endpoint, + engine, + engineSource, }; } finally { // Always, including the challenge path. Leaving the tab open would let the |
