diff options
| author | Igor Soarez <igor@soarez.org> | 2026-08-03 21:43:56 +0100 |
|---|---|---|
| committer | Igor Soarez <igor@soarez.org> | 2026-08-03 21:43:56 +0100 |
| commit | db69207c06e8d5233bff4996e3ef5b43332d533f (patch) | |
| tree | 032ebbc3983a6eb4f9e3c4ac162c60a9cc1edc75 /src | |
| parent | 495de0d5283dd3e4a6ef715b596c4a2892e95915 (diff) | |
Support DuckDuckGo, Bing and Brave, switchable like the CDP host
Engine resolution mirrors the browser target: CCS_SEARCH_ENGINE, then a
/search-engine choice persisted machine-wide, then Google. A per-call
`engine` parameter sits above both so the agent can fall back when one
engine starts serving captchas — the one case where the model, not the
operator, has to make the call. Unlike the CDP target there is no safety
argument for the environment winning: driving the wrong browser means
automating someone's signed-in Chrome, choosing a different index does not.
An unsupported recency window is refused, naming the engines that support
it, rather than dropped. Silently returning unfiltered results is
indistinguishable from success, which is the failure this whole design is
trying to avoid.
Extraction grows a second mode. DuckDuckGo, Bing and Brave have clean
per-result containers; Google does not, so its heading-walk stays as its
own path rather than being bent into the item shape. DuckDuckGo and Bing
route links through redirectors, unwrapped in the page.
Third silent-failure trap found, alongside Google's two: on Bing, `count`
cancels `filters`. With ex1:"ez1" alone every result is hours old; add
count in either order and months-old results return, looking perfectly
ordinary. Bing now drops count whenever a date filter is present.
Challenge detection widened to Brave's "Verifying you're not a bot" and
"Quick check before you continue searching", which the previous Google-
shaped matcher missed entirely — found by tripping it.
Diffstat (limited to 'src')
| -rw-r--r-- | src/engine-store.ts | 81 | ||||
| -rw-r--r-- | src/engine-target.ts | 79 | ||||
| -rw-r--r-- | src/engines.ts | 227 | ||||
| -rw-r--r-- | src/errors.ts | 37 | ||||
| -rw-r--r-- | src/extract.ts | 276 | ||||
| -rw-r--r-- | src/format.ts | 10 | ||||
| -rw-r--r-- | src/index.ts | 132 | ||||
| -rw-r--r-- | src/search.ts | 150 |
8 files changed, 841 insertions, 151 deletions
diff --git a/src/engine-store.ts b/src/engine-store.ts new file mode 100644 index 0000000..d218d2b --- /dev/null +++ b/src/engine-store.ts @@ -0,0 +1,81 @@ +/** + * Persistence for the `/search-engine` choice. + * + * Deliberately the same shape, and the same atomic write, as the harness's + * browser-target store: a sibling temp file plus rename, so a crash mid-write + * cannot leave a half-written engine name behind. Its own file rather than a + * shared settings blob, so a corrupt engine choice cannot cost anything else. + * + * Like the CDP target, the stored value is the raw string a user would have put + * in the environment variable ("ddg", "bing", …), not a pre-parsed id. One + * vocabulary, one resolution path in engine-target.ts, and a stored value can + * never mean something the variable could not. + * + * Every read failure degrades to "nothing stored" rather than throwing. + */ + +import { mkdir, readFile, rename, unlink, writeFile } from "node:fs/promises"; +import { randomUUID } from "node:crypto"; +import { dirname, join } from "node:path"; + +import { agentDir } from "./paths.ts"; + +const CURRENT_VERSION = 1; + +const engineFilePath = (): string => join(agentDir(), "castle-cdp-search-engine.json"); + +type EngineFile = { + readonly version: 1; + /** Raw CCS_SEARCH_ENGINE-style value, or null when cleared. */ + readonly engine: string | null; + readonly savedAt: string; +}; + +/** + * The persisted engine, or null when nothing is stored, the file is missing, or + * it cannot be parsed. A file written by a newer version is treated as "nothing + * stored" rather than guessed at. + */ +export const readStoredEngine = async (): Promise<string | null> => { + let raw: string; + try { + raw = await readFile(engineFilePath(), "utf8"); + } catch { + return null; + } + try { + const parsed: unknown = JSON.parse(raw); + if (typeof parsed !== "object" || parsed === null) return null; + const file = parsed as Record<string, unknown>; + if (file["version"] !== CURRENT_VERSION) return null; + const engine = file["engine"]; + if (typeof engine !== "string") return null; + const trimmed = engine.trim(); + return trimmed.length > 0 ? trimmed : null; + } catch { + return null; + } +}; + +const write = async (engine: string | null): Promise<void> => { + const path = engineFilePath(); + const tmp = `${path}.${randomUUID()}.tmp`; + const payload: EngineFile = { version: CURRENT_VERSION, engine, savedAt: new Date().toISOString() }; + try { + await mkdir(dirname(path), { recursive: true }); + await writeFile(tmp, `${JSON.stringify(payload, null, 2)}\n`, "utf8"); + await rename(tmp, path); + } catch (e) { + await unlink(tmp).catch(() => {}); + throw new Error(`could not save search engine to ${path}: ${e instanceof Error ? e.message : String(e)}`); + } +}; + +/** Persist the chosen engine. */ +export const writeStoredEngine = (engine: string): Promise<void> => write(engine); + +/** Clear the choice, restoring the built-in default. */ +export const clearStoredEngine = (): Promise<void> => write(null); + +/** Exposed so the command can tell the user where the choice lives. */ +export const storedEngineLocation = (): string => engineFilePath(); diff --git a/src/engine-target.ts b/src/engine-target.ts new file mode 100644 index 0000000..db4d81f --- /dev/null +++ b/src/engine-target.ts @@ -0,0 +1,79 @@ +/** + * Which search engine to use, resolved the same way the CDP host is: + * environment variable, then a persisted choice, then a built-in default. + * + * Pure — no I/O — so the `/search-engine` command can report the configured + * intent without touching disk twice, and so it is testable. + */ + +import { type EngineId, parseEngineId, engineAliases } from "./engines.ts"; + +/** Used when nothing else says otherwise. */ +export const DEFAULT_ENGINE: EngineId = "google"; + +/** The environment variable that pins the engine for one process. */ +export const ENGINE_ENV_VAR = "CCS_SEARCH_ENGINE"; + +export type EngineSource = "parameter" | "env" | "stored" | "default"; + +export type EngineResolution = + | { readonly kind: "ok"; readonly id: EngineId; readonly source: EngineSource } + | { readonly kind: "invalid"; readonly raw: string; readonly source: EngineSource }; + +/** + * Precedence: the per-call tool parameter, then `CCS_SEARCH_ENGINE`, then the + * `/search-engine` choice, then the built-in default. + * + * Note the tool parameter sits *above* the environment variable, which is the + * opposite of how the CDP target treats an explicit endpoint. That difference + * is deliberate. For the browser, an environment variable has to win, because + * quietly driving a different machine means automating the operator's own + * signed-in Chrome — a safety property. Choosing a different search engine has + * no equivalent hazard, and the ability to fall back to another engine when the + * first one is serving a captcha is the single most useful thing the agent can + * do with this tool. So the model is allowed to override the default; the + * operator sets what it starts from. + */ +export const resolveEngine = ( + parameter: string | null | undefined, + env: NodeJS.ProcessEnv = process.env, + stored: string | null = null, +): EngineResolution => { + const candidates: Array<{ raw: string; source: EngineSource }> = []; + + const fromParam = parameter?.trim(); + if (fromParam) candidates.push({ raw: fromParam, source: "parameter" }); + + const fromEnv = env[ENGINE_ENV_VAR]?.trim(); + if (fromEnv) candidates.push({ raw: fromEnv, source: "env" }); + + const fromStore = stored?.trim(); + if (fromStore) candidates.push({ raw: fromStore, source: "stored" }); + + for (const candidate of candidates) { + const id = parseEngineId(candidate.raw); + // An unparseable value is reported rather than skipped over. Falling + // through to the next source would mean a typo in CCS_SEARCH_ENGINE + // silently searches Google instead of saying so. + if (!id) return { kind: "invalid", raw: candidate.raw, source: candidate.source }; + return { kind: "ok", id, source: candidate.source }; + } + + return { kind: "ok", id: DEFAULT_ENGINE, source: "default" }; +}; + +const sourceLabel = (source: EngineSource): string => + source === "parameter" + ? "engine parameter" + : source === "env" + ? ENGINE_ENV_VAR + : source === "stored" + ? "/search-engine" + : "built-in default"; + +export const describeEngineSource = (source: EngineSource): string => sourceLabel(source); + +/** One-line description of a bad engine name, listing what would have worked. */ +export const describeEngineFailure = (failure: Extract<EngineResolution, { kind: "invalid" }>): string => + `unknown search engine "${failure.raw}" from ${sourceLabel(failure.source)} — ` + + `expected one of ${engineAliases().join(", ")}`; diff --git a/src/engines.ts b/src/engines.ts new file mode 100644 index 0000000..79462fc --- /dev/null +++ b/src/engines.ts @@ -0,0 +1,227 @@ +/** + * The search engines this extension knows how to drive. + * + * Every selector and every URL parameter below was read off the real SERP in + * castle's Chrome, not recalled or inferred. That is not pedantry: Google's own + * Tools menu writes `tbs=qdr:*`, which renders an *empty* page on that profile, + * while the undocumented-looking `as_qdr=*` works. Anything in here that was not + * observed is a bug waiting to be reported as "no results". + * + * Two extraction modes, because the engines genuinely differ: + * + * - "items" — the SERP has a clean per-result container (`li.b_algo`, + * `.result`, `.snippet[data-type=web]`). Straightforward. + * - "headings" — Google has no stable result container, so results are found + * from each <h3> outward. Kept as its own mode rather than + * forced into the item shape, because it is the one that has + * been through the most verification. + */ + +export type EngineId = "google" | "duckduckgo" | "bing" | "brave"; + +export type Recency = "day" | "week" | "month" | "year"; + +export const RECENCY_VALUES = ["day", "week", "month", "year"] as const; + +export const ENGINE_IDS: readonly EngineId[] = ["google", "duckduckgo", "bing", "brave"]; + +/** Aliases accepted from humans and from the model. */ +const ENGINE_ALIASES: Record<string, EngineId> = { + google: "google", + g: "google", + duckduckgo: "duckduckgo", + ddg: "duckduckgo", + duck: "duckduckgo", + bing: "bing", + b: "bing", + brave: "brave", +}; + +/** + * How the in-page script should read one engine's results. Passed into the + * browser as JSON, so everything here must be plain data. + */ +export type ExtractionConfig = { + mode: "items" | "headings"; + /** Results container candidates, first match wins. */ + roots: string[]; + /** "items" mode: one result. */ + item?: string; + /** Anchor carrying the outbound link, relative to the item. */ + link?: string; + /** Title element, relative to the item. Falls back to the anchor's text. */ + title?: string; + /** Description element, relative to the item. */ + snippet?: string; + /** + * "items" mode with no snippet selector: text from these is subtracted from + * the item's text to leave the description behind. + */ + subtract?: string[]; + /** Items matching any of these are ads or non-web cards, and are skipped. */ + exclude?: string[]; + /** How outbound links are wrapped, if they are. */ + unwrap: "none" | "ddg" | "bing"; + /** Hosts belonging to the engine itself; links to them are not results. */ + selfHostPattern: string; +}; + +export type EngineDefinition = { + readonly id: EngineId; + readonly label: string; + /** Human-facing note about what makes this engine worth choosing. */ + readonly note: string; + /** + * Recency windows this engine can actually express, mapped to the parameter + * value. A window that is absent is one the engine does not support — never + * one that is silently dropped. + */ + readonly recency: Partial<Record<Recency, string>>; + readonly buildUrl: (query: string, numResults: number, recency: Recency | undefined) => string; + readonly extraction: ExtractionConfig; +}; + +const GOOGLE: EngineDefinition = { + id: "google", + label: "Google", + note: "best result quality; blocks aggressively under repeated automated queries", + recency: { day: "d", week: "w", month: "m", year: "y" }, + buildUrl: (query, numResults, recency) => { + const params = new URLSearchParams({ q: query, num: String(numResults), hl: "en" }); + // `udm=14` is the plain "Web" tab: no AI overview, no carousels. But any + // date restriction combined with it renders an empty page, and `tbs=qdr:*` + // renders an empty page on its own — both observed. So a time-limited + // search drops udm and uses the older as_qdr instead. + if (recency) params.set("as_qdr", GOOGLE.recency[recency] as string); + else params.set("udm", "14"); + return `https://www.google.com/search?${params.toString()}`; + }, + extraction: { + mode: "headings", + roots: ["#rso", "#search"], + unwrap: "none", + selfHostPattern: String.raw`^https?://(www\.)?google\.[a-z.]+/`, + }, +}; + +const DUCKDUCKGO: EngineDefinition = { + id: "duckduckgo", + label: "DuckDuckGo", + note: "no-JS endpoint, the most stable markup here and the least likely to challenge", + // Read off DuckDuckGo's own <select name="df">: "", d, w, m, y. + recency: { day: "d", week: "w", month: "m", year: "y" }, + buildUrl: (query, _numResults, recency) => { + // The html endpoint renders server-side with no JavaScript, which makes it + // both faster and far less fragile than the app at duckduckgo.com. It has + // no result-count parameter — it returns a full page and the caller slices. + const params = new URLSearchParams({ q: query }); + if (recency) params.set("df", DUCKDUCKGO.recency[recency] as string); + return `https://html.duckduckgo.com/html/?${params.toString()}`; + }, + extraction: { + mode: "items", + roots: [".results", "#links"], + item: ".result", + link: "a.result__a[href]", + title: "a.result__a", + snippet: ".result__snippet", + exclude: [".result--ad", ".badge--ad"], + unwrap: "ddg", + selfHostPattern: String.raw`^https?://(html\.|www\.)?duckduckgo\.com/`, + }, +}; + +const BING: EngineDefinition = { + id: "bing", + label: "Bing", + note: "good coverage; supports day/week/month only — it has no year filter", + // ez1/ez2/ez3 verified to filter (results carried "4 hours ago", "1 day ago"). + // There is deliberately no year: Bing's UI offers no such window, and the + // ez5 custom-range form returned nothing when tried. + recency: { day: "ez1", week: "ez2", month: "ez3" }, + buildUrl: (query, numResults, recency) => { + const params = new URLSearchParams({ q: query }); + if (recency) { + // `count` silently cancels `filters` — with ez1 alone every result is + // hours old, and adding count in either order brings back months-old + // ones. Observed directly, and it fails silently: the results look + // perfectly plausible, just unfiltered. Exactly the same trap as Google's + // udm+as_qdr, so the same answer — drop the count parameter and let the + // caller slice the list it gets. + params.set("filters", `ex1:"${BING.recency[recency] as string}"`); + } else { + params.set("count", String(numResults)); + } + return `https://www.bing.com/search?${params.toString()}`; + }, + extraction: { + mode: "items", + // Organic results only. Bing's answer cards also contain <h2>s, which is + // why this targets li.b_algo rather than walking headings. + roots: ["#b_results"], + item: "li.b_algo", + link: "h2 a[href]", + title: "h2", + snippet: ".b_caption p, .b_algoSlug, p", + exclude: [".b_ad", ".b_adBottom"], + unwrap: "bing", + selfHostPattern: String.raw`^https?://(www\.)?bing\.com/`, + }, +}; + +const BRAVE: EngineDefinition = { + id: "brave", + label: "Brave Search", + note: "independent index and unwrapped links; challenges quickly under repeated queries", + // Left empty deliberately — see the note in README. Brave's filter UI is + // client-rendered and the `tf=` values could not be confirmed against a page + // that was not simultaneously serving a bot check, so no window is claimed + // rather than one being guessed at. + recency: {}, + buildUrl: (query, _numResults, _recency) => { + const params = new URLSearchParams({ q: query }); + return `https://search.brave.com/search?${params.toString()}`; + }, + extraction: { + mode: "items", + roots: ["#results"], + // data-type="web" excludes the AI summariser, video and news cards, which + // share the .snippet class but are not ranked web results. + item: '.snippet[data-type="web"]', + link: "a[href]", + title: ".title", + // No dedicated description element; the item's text is + // "source | breadcrumb | title | description", so subtract the first three. + subtract: [".title", "cite", ".sitename", ".netloc"], + unwrap: "none", + selfHostPattern: String.raw`^https?://(search\.)?brave\.com/`, + }, +}; + +const BY_ID: Record<EngineId, EngineDefinition> = { + google: GOOGLE, + duckduckgo: DUCKDUCKGO, + bing: BING, + brave: BRAVE, +}; + +export const getEngine = (id: EngineId): EngineDefinition => BY_ID[id]; + +export const allEngines = (): readonly EngineDefinition[] => ENGINE_IDS.map((id) => BY_ID[id]); + +/** Parse a user- or model-supplied engine name. Null when unrecognised. */ +export const parseEngineId = (raw: string): EngineId | null => + ENGINE_ALIASES[raw.trim().toLowerCase()] ?? null; + +/** The names accepted for an engine, for help text and error messages. */ +export const engineAliases = (): readonly string[] => Object.keys(ENGINE_ALIASES); + +/** Which engines can express a given recency window. */ +export const enginesSupportingRecency = (recency: Recency): readonly EngineId[] => + ENGINE_IDS.filter((id) => BY_ID[id].recency[recency] !== undefined); + +/** Human-readable list of the windows an engine supports, or "none". */ +export const describeRecencySupport = (engine: EngineDefinition): string => { + const windows = RECENCY_VALUES.filter((r) => engine.recency[r] !== undefined); + return windows.length > 0 ? windows.join(", ") : "none"; +}; diff --git a/src/errors.ts b/src/errors.ts index d003e4f..2318b27 100644 --- a/src/errors.ts +++ b/src/errors.ts @@ -19,15 +19,35 @@ export class SearchChallengeError extends Error { readonly pageUrl: string; /** Which browser is blocked, e.g. "10.88.0.25:9223 (built-in default)". */ readonly endpoint: string; + /** Which engine is blocked. */ + readonly engine: string; + /** Engines that are not blocked and could be tried instead. */ + readonly alternatives: readonly string[]; - constructor(options: { challenge: string; pageUrl: string; endpoint: string; detail?: string | undefined }) { + constructor(options: { + challenge: string; + pageUrl: string; + endpoint: string; + engine: string; + alternatives: readonly string[]; + detail?: string | undefined; + }) { super( [ - `Search is blocked by a human verification challenge (${options.challenge}).`, + `${options.engine} is blocked by a human verification challenge (${options.challenge}).`, options.detail ? `Page said: ${options.detail}` : null, - `This cannot be solved by the agent — a person has to clear it in the browser at ${options.endpoint}.`, - `Ask the user to open ${options.pageUrl} in that Chrome, complete the challenge or accept the consent dialog,`, - `then retry the search. The cookie it sets persists in that browser profile, so one pass unblocks later searches.`, + // Retrying elsewhere comes first because it is the action the agent can + // take on its own; a challenge on one engine says nothing about the + // others, and asking the operator for help should be the fallback, not + // the first move. + options.alternatives.length > 0 + ? `A challenge on one engine does not affect the others: retry the same query with engine set to ` + + `${options.alternatives.join(", ")} before asking anyone for help.` + : null, + `If every engine is blocked, this needs a person — the challenge cannot be solved by the agent.`, + `Ask the user to open ${options.pageUrl} in the Chrome at ${options.endpoint}, complete the challenge or`, + `accept the consent dialog, then retry. The cookie it sets persists in that browser profile, so one pass`, + `unblocks later searches.`, ] .filter((line) => line !== null) .join(" "), @@ -35,9 +55,16 @@ export class SearchChallengeError extends Error { this.challenge = options.challenge; this.pageUrl = options.pageUrl; this.endpoint = options.endpoint; + this.engine = options.engine; + this.alternatives = options.alternatives; } } +/** The chosen engine cannot express the requested recency window. */ +export class RecencyUnsupportedError extends Error { + override readonly name = "RecencyUnsupportedError"; +} + /** The browser could not be reached or spoke CDP badly. */ export class BrowserUnavailableError extends Error { override readonly name = "BrowserUnavailableError"; diff --git a/src/extract.ts b/src/extract.ts index c4d263a..282fa07 100644 --- a/src/extract.ts +++ b/src/extract.ts @@ -1,30 +1,45 @@ /** * The script that runs inside the search results page. * - * It is a string rather than a function because it is shipped to Chrome via - * `Runtime.evaluate` — nothing in here is typechecked, so it is kept small, - * defensive, and free of anything that could throw on an unexpected DOM. + * It is built as a string rather than shipped as a function because it goes to + * Chrome via `Runtime.evaluate` — nothing in here is typechecked, so it is kept + * defensive and free of anything that could throw on an unexpected DOM. The + * per-engine configuration is injected as JSON by {@link buildProbeScript}. * - * ## Why it does not select on class names + * ## Two modes, because the engines genuinely differ * - * Google's result classes (`MjjYud`, `kb0PBd`, `yuRUbf`, …) are generated and - * change without notice; the `data-` hooks (`data-snhf` for the title/source - * header, `data-sncf` for the description) are more stable but not promised - * either. So the extractor uses them when present and otherwise falls back to a - * structural walk that only assumes "an <h3> inside a link, with the - * description somewhere in a shared ancestor": + * **"items"** — DuckDuckGo, Bing and Brave each have a clean per-result + * container, so results are read directly out of it. * - * - anchor = nearest enclosing <a href> - * - header = nearest [data-snhf], else the anchor itself + * **"headings"** — Google has no stable result container; its class names + * (`MjjYud`, `kb0PBd`, `yuRUbf`) are generated and change without notice. So + * results are found from each `<h3>` outward: + * + * - anchor = nearest enclosing `<a href>` + * - header = nearest `[data-snhf]`, else the anchor * - container = climb from the header until the text grows past the header's, - * abandoning the climb if a second <h3> comes into scope - * (that would mean we had swallowed the next result) - * - snippet = [data-sncf] if present, else the container's lines minus the - * header's lines, minus URL/breadcrumb noise + * abandoning the climb if a second `<h3>` comes into scope + * (that would mean the next result had been swallowed) + * - snippet = `[data-sncf]` if present, else the container's lines minus the + * header's lines, minus URL and breadcrumb noise + * + * Verified against Google's classic SERP and its `udm=14` layout, DuckDuckGo's + * no-JS endpoint, Bing's `li.b_algo`, and Brave's `.snippet[data-type=web]`. + * + * ## Link unwrapping * - * Verified against both the classic SERP and the `udm=14` "Web" layout. + * DuckDuckGo and Bing both route outbound links through a redirector, so the + * raw href is useless to the agent. Both are unwrapped in the page, where the + * URL and base64 primitives already exist: + * + * - DuckDuckGo: `//duckduckgo.com/l/?uddg=<percent-encoded target>` + * - Bing: `//bing.com/ck/a?…&u=a1<base64url of target>` + * + * Google and Brave link straight out and need no unwrapping. */ +import type { ExtractionConfig } from "./engines.ts"; + /** One search hit, as the page script reports it. */ export type SearchResult = { title: string; @@ -41,113 +56,210 @@ export type PageProbe = { /** Non-null when a human check is in the way. */ challenge: { kind: string; detail: string } | null; /** - * Which results container the page rendered. "none" means Google served a - * shell with no results area at all — a different failure from an empty one, + * Whether the engine's results container was present at all. "none" means a + * page shell with no results area — a different failure from an empty one, * and worth telling apart when diagnosing a zero-hit search. */ - container: "rso" | "search" | "none"; + container: "found" | "none"; results: SearchResult[]; }; /** - * Detection order matters: a `/sorry/` interstitial also contains a recaptcha - * iframe, and reporting the specific page beats reporting the widget on it. + * Challenge detection, shared across engines. + * + * Every phrase here has been seen on a real page during development: Google's + * `/sorry/` interstitial, and Brave's "Verifying you're not a bot / Quick check + * before you continue searching" — the latter is why the wording list is broad + * rather than just matching Google's "unusual traffic". Ordering matters: a + * `/sorry/` page also contains a recaptcha iframe, and naming the page beats + * naming the widget sitting on it. */ -export const PROBE_SCRIPT = String.raw`(() => { - const norm = (s) => (s || "").replace(/\s+/g, " ").trim(); - const linesOf = (el) => - el && el.innerText ? el.innerText.split("\n").map((l) => l.trim()).filter(Boolean) : []; - - const bodyText = document.body ? norm(document.body.innerText).slice(0, 4000) : ""; - +const CHALLENGE_DETECTION = String.raw` const detectChallenge = () => { - const href = location.href; - if (/\/sorry\//.test(href)) { - return { kind: "google-block-page", detail: bodyText.slice(0, 300) }; - } - if (location.hostname.indexOf("consent.") === 0 || /\/consent\b/.test(location.pathname)) { + if (/\/sorry\//.test(location.href)) return { kind: "google-block-page", detail: bodyText.slice(0, 300) }; + if (location.hostname.indexOf("consent.") === 0 || /\/consent\b/.test(location.pathname)) return { kind: "consent-wall", detail: bodyText.slice(0, 300) }; - } - if (document.querySelector("#captcha-form, form#captcha-form")) { + if (document.querySelector("#captcha-form, form#captcha-form")) return { kind: "captcha-form", detail: bodyText.slice(0, 300) }; - } - if (document.querySelector('iframe[src*="recaptcha"], iframe[src*="hcaptcha"], iframe[title*="challenge"]')) { - return { kind: "captcha-widget", detail: bodyText.slice(0, 300) }; - } - if (document.querySelector("#challenge-form, #cf-chl-widget, #cf-challenge-running")) { + if (document.querySelector("#challenge-form, #cf-chl-widget, #cf-challenge-running, [id^=cf-chl]")) return { kind: "cloudflare-challenge", detail: bodyText.slice(0, 300) }; - } - if (/unusual traffic|are you a robot|verify (that )?you('| a)?re human|not a robot|automated queries/i.test(bodyText)) { + if (document.querySelector('iframe[src*="recaptcha"], iframe[src*="hcaptcha"], iframe[src*="turnstile"], iframe[title*="challenge"]')) + return { kind: "captcha-widget", detail: bodyText.slice(0, 300) }; + if (/verifying (that )?you('| a)?re( not)? a? ?(human|bot|robot)|quick check before you continue|verify you are human|are you a robot|not a robot/i.test(bodyText)) return { kind: "bot-check", detail: bodyText.slice(0, 300) }; + if (/unusual traffic|automated queries|suspicious activity from your/i.test(bodyText)) + return { kind: "rate-limit-block", detail: bodyText.slice(0, 300) }; + return null; + }; +`; + +/** + * Build the probe for one engine. The configuration is embedded as a JSON + * literal so the script stays a single self-contained expression. + */ +export const buildProbeScript = (config: ExtractionConfig): string => String.raw`(() => { + var CFG = ${JSON.stringify(config)}; + + var norm = function (s) { return (s || "").replace(/\s+/g, " ").trim(); }; + var linesOf = function (el) { + return el && el.innerText ? el.innerText.split("\n").map(function (l) { return l.trim(); }).filter(Boolean) : []; + }; + var bodyText = document.body ? norm(document.body.innerText).slice(0, 4000) : ""; + var selfHost = new RegExp(CFG.selfHostPattern); + + ${CHALLENGE_DETECTION} + + // DuckDuckGo and Bing both hide the real destination behind a redirector. + var unwrap = function (href) { + try { + var u = new URL(href, location.href); + if (CFG.unwrap === "ddg") { + if (!/(^|\.)duckduckgo\.com$/.test(u.hostname)) return href; + return u.searchParams.get("uddg") || href; + } + if (CFG.unwrap === "bing") { + if (!/(^|\.)bing\.com$/.test(u.hostname)) return href; + var p = u.searchParams.get("u"); + if (!p) return href; + var b64 = p.replace(/^a1/, "").replace(/-/g, "+").replace(/_/g, "/"); + b64 += "=".repeat((4 - (b64.length % 4)) % 4); + var bin = atob(b64); + var bytes = new Uint8Array(bin.length); + for (var i = 0; i < bin.length; i++) bytes[i] = bin.charCodeAt(i); + return new TextDecoder().decode(bytes); + } + return href; + } catch (e) { + // An unwrap that fails leaves the redirector URL in place rather than + // dropping the result: a working link the agent has to follow twice beats + // no link at all. + return href; + } + }; + + var findRoot = function () { + for (var i = 0; i < CFG.roots.length; i++) { + var el = document.querySelector(CFG.roots[i]); + if (el) return el; } return null; }; - const extract = () => { - const root = document.querySelector("#rso") || document.querySelector("#search"); - if (!root) return []; - const out = []; - const seen = new Set(); + var acceptable = function (url) { + return /^https?:/.test(url) && !selfHost.test(url); + }; + + // ── items mode ─────────────────────────────────────────────────────────── + var extractItems = function (root) { + var out = []; + var seen = {}; + var items = root.querySelectorAll(CFG.item); + for (var i = 0; i < items.length; i++) { + var item = items[i]; + + var skip = false; + for (var x = 0; CFG.exclude && x < CFG.exclude.length; x++) { + if (item.matches(CFG.exclude[x]) || item.querySelector(CFG.exclude[x])) { skip = true; break; } + } + if (skip) continue; - const headings = root.querySelectorAll("h3"); - for (let i = 0; i < headings.length; i++) { - const h3 = headings[i]; - const anchor = h3.closest("a[href]") || (h3.parentElement && h3.parentElement.querySelector("a[href]")); + var anchor = item.querySelector(CFG.link); if (!anchor) continue; + var url = unwrap(anchor.href); + if (!acceptable(url) || seen[url]) continue; - const url = anchor.href; - if (!/^https?:/.test(url)) continue; - // Drop Google's own links (image search, cached copies, "more results"). - if (/^https?:\/\/(www\.)?google\.[a-z.]+\//.test(url)) continue; - if (seen.has(url)) continue; + var titleEl = CFG.title ? item.querySelector(CFG.title) : null; + var title = norm(titleEl ? titleEl.innerText : anchor.innerText); + if (!title) continue; - const title = norm(h3.innerText); + var snippet = ""; + var snippetEl = CFG.snippet ? item.querySelector(CFG.snippet) : null; + if (snippetEl) { + snippet = norm(snippetEl.innerText); + } else { + // No dedicated description element: subtract the parts we can name + // (title, breadcrumb, source) and keep what is left. + var drop = {}; + for (var s = 0; CFG.subtract && s < CFG.subtract.length; s++) { + var parts = item.querySelectorAll(CFG.subtract[s]); + for (var p = 0; p < parts.length; p++) { + var pl = linesOf(parts[p]); + for (var q = 0; q < pl.length; q++) drop[norm(pl[q])] = true; + } + } + var kept = linesOf(item).filter(function (l) { + var n = norm(l); + return !drop[n] && n !== title && !/^https?:\/\//.test(n) && n.indexOf("›") === -1; + }); + snippet = norm(kept.join(" ")); + } + + seen[url] = true; + out.push({ title: title, url: url, snippet: snippet.slice(0, 600) }); + } + return out; + }; + + // ── headings mode (Google) ─────────────────────────────────────────────── + var extractHeadings = function (root) { + var out = []; + var seen = {}; + var headings = root.querySelectorAll("h3"); + for (var i = 0; i < headings.length; i++) { + var h3 = headings[i]; + var anchor = h3.closest("a[href]") || (h3.parentElement && h3.parentElement.querySelector("a[href]")); + if (!anchor) continue; + + var url = unwrap(anchor.href); + if (!acceptable(url) || seen[url]) continue; + + var title = norm(h3.innerText); if (!title) continue; - const header = h3.closest("[data-snhf]") || anchor; - const headerLines = new Set(linesOf(header).map(norm)); - const baseline = norm(header.innerText || "").length; + var header = h3.closest("[data-snhf]") || anchor; + var headerLines = {}; + var hl = linesOf(header); + for (var k = 0; k < hl.length; k++) headerLines[norm(hl[k])] = true; + var baseline = norm(header.innerText || "").length; - let container = header.parentElement; - for (let depth = 0; depth < 6 && container && container !== root; depth++) { - if (container.querySelectorAll("h3").length > 1) { - container = null; - break; - } + var container = header.parentElement; + for (var depth = 0; depth < 6 && container && container !== root; depth++) { + if (container.querySelectorAll("h3").length > 1) { container = null; break; } if (norm(container.innerText || "").length > baseline + 40) break; container = container.parentElement; } - let snippet = ""; + var snippet = ""; if (container && container !== root) { - const explicit = container.querySelector("[data-sncf]"); - const body = explicit ? linesOf(explicit) : linesOf(container).filter((l) => !headerLines.has(norm(l))); + var explicit = container.querySelector("[data-sncf]"); + var body = explicit + ? linesOf(explicit) + : linesOf(container).filter(function (l) { return !headerLines[norm(l)]; }); snippet = norm( - body - .filter((l) => !/^https?:\/\//.test(l) && l.indexOf("›") === -1 && l !== "Web results") - .join(" "), - ) - .replace(/Read more$/, "") - .trim(); + body.filter(function (l) { + return !/^https?:\/\//.test(l) && l.indexOf("›") === -1 && l !== "Web results"; + }).join(" "), + ).replace(/Read more$/, "").trim(); } - seen.add(url); + seen[url] = true; out.push({ title: title, url: url, snippet: snippet.slice(0, 600) }); } return out; }; - let challenge = null; - let results = []; - let container = "none"; + var challenge = null; + var results = []; + var container = "none"; try { challenge = detectChallenge(); } catch (e) { challenge = null; } try { - container = document.querySelector("#rso") ? "rso" : document.querySelector("#search") ? "search" : "none"; - results = challenge ? [] : extract(); + var root = findRoot(); + container = root ? "found" : "none"; + if (root && !challenge) results = CFG.mode === "items" ? extractItems(root) : extractHeadings(root); } catch (e) { results = []; } diff --git a/src/format.ts b/src/format.ts index d3872a1..aef6c3b 100644 --- a/src/format.ts +++ b/src/format.ts @@ -14,12 +14,20 @@ export type FormatInput = { searchUrl: string; finalUrl: string; endpoint: string; + /** Engine display name, e.g. "DuckDuckGo". */ + engine: string; + /** Where the engine choice came from, e.g. "/search-engine". */ + engineSource: string; results: SearchResult[]; }; export const formatResults = (input: FormatInput): string => { const lines: string[] = [ - `Search results for "${input.query}" (${input.results.length} hits, via Chrome at ${input.endpoint})`, + // Naming the engine on every result set matters more than it looks: the + // agent may have switched engines mid-conversation to get around a block, + // and results from different indexes are not interchangeable evidence. + `Search results for "${input.query}" — ${input.results.length} hits from ${input.engine} ` + + `(${input.engineSource}), via Chrome at ${input.endpoint}`, `Query URL: ${input.searchUrl}`, ]; if (input.finalUrl !== input.searchUrl) lines.push(`Landed on: ${input.finalUrl}`); diff --git a/src/index.ts b/src/index.ts index ad0ba6f..4462075 100644 --- a/src/index.ts +++ b/src/index.ts @@ -7,6 +7,10 @@ * (`BU_CDP_HTTP` > `/browser-target` > castle), so a single choice governs both * packages and they can never end up driving different browsers. * + * The engine is switchable the same way: `CCS_SEARCH_ENGINE` > `/search-engine` + * > Google, plus a per-call `engine` parameter so the agent can fall back when + * one engine starts serving captchas. + * * The tool is called `castle_cdp_search`, not `web_search`, on purpose: it is * meant to sit alongside pi-web-access rather than shadow it. The model picks * between them, so both stay available in one session. @@ -17,15 +21,27 @@ import type { ExtensionAPI } from "@earendil-works/pi-coding-agent"; import { StringEnum } from "@earendil-works/pi-ai"; import { Type } from "typebox"; +import { clearStoredEngine, storedEngineLocation, writeStoredEngine } from "./engine-store.ts"; +import { DEFAULT_ENGINE, ENGINE_ENV_VAR } from "./engine-target.ts"; +import { + ENGINE_IDS, + RECENCY_VALUES, + type EngineId, + type Recency, + allEngines, + describeRecencySupport, + engineAliases, + parseEngineId, +} from "./engines.ts"; import { messageOf } from "./errors.ts"; import type { SearchResult } from "./extract.ts"; import { formatResults, truncationNote } from "./format.ts"; import { MAX_CONCURRENT_SEARCHES, SEARCH_TIMEOUT_MS, - type Recency, type SearchPhase, describeConfiguredEndpoint, + describeConfiguredEngine, disconnect, search, } from "./search.ts"; @@ -50,8 +66,17 @@ const parameters = Type.Object({ // StringEnum, not Type.Union of Type.Literal: Google's API rejects the // anyOf/const shape typebox emits for unions. recency: Type.Optional( - StringEnum(["day", "week", "month", "year"] as const, { - description: "Restrict results to pages published within this window. Omit for no time limit.", + StringEnum(RECENCY_VALUES, { + description: + "Restrict results to pages published within this window. Omit for no time limit. " + + "Not every engine supports every window — bing has no year, brave has no time filter at all.", + }), + ), + engine: Type.Optional( + StringEnum(ENGINE_IDS, { + description: + "Which search engine to use for this call. Omit to use the configured default. " + + "Switch engines when one returns a challenge or no results — they have independent indexes and blocks.", }), ), }); @@ -61,12 +86,14 @@ export type CastleCdpSearchInput = { query: string; numResults?: number; recency?: Recency; + engine?: EngineId; }; export type CastleCdpSearchDetails = { query: string; numResults: number; recency: Recency | null; + engine: EngineId | null; phase: SearchPhase | "done"; note: string; endpoint: string | null; @@ -75,10 +102,16 @@ export type CastleCdpSearchDetails = { truncated: boolean; }; -const initialDetails = (query: string, numResults: number, recency: Recency | null): CastleCdpSearchDetails => ({ +const initialDetails = ( + query: string, + numResults: number, + recency: Recency | null, + engine: EngineId | null, +): CastleCdpSearchDetails => ({ query, numResults, recency, + engine, phase: "queued", note: "starting", endpoint: null, @@ -87,6 +120,12 @@ const initialDetails = (query: string, numResults: number, recency: Recency | nu truncated: false, }); +/** The engine table, rendered for `/search-engine` and for the tool description. */ +const engineTable = (): string => + allEngines() + .map((e) => ` ${e.id.padEnd(11)} ${e.label.padEnd(14)} recency: ${describeRecencySupport(e).padEnd(24)} ${e.note}`) + .join("\n"); + export default function (pi: ExtensionAPI): void { // Nothing is dialled here. pi runs extension factories in invocations that // never start a session, so opening a socket from a factory would leave one @@ -97,6 +136,50 @@ export default function (pi: ExtensionAPI): void { disconnect(); }); + // The engine analogue of pi-browser-harness's /browser-target. Persisted + // machine-wide for the same reason: the choice should outlive the session + // that made it. + pi.registerCommand("search-engine", { + description: "Show or set the search engine castle_cdp_search uses", + handler: async (args, ctx) => { + const raw = args.trim(); + const current = await describeConfiguredEngine(); + + if (raw === "") { + ctx.ui.notify( + `castle_cdp_search engine: ${current.id} (${current.source})\n\n${engineTable()}\n\n` + + `Set with: /search-engine <name> Clear with: /search-engine default\n` + + `Stored in ${storedEngineLocation()}; ${ENGINE_ENV_VAR} overrides it for one process.`, + "info", + ); + return; + } + + if (raw.toLowerCase() === "default" || raw.toLowerCase() === "clear") { + await clearStoredEngine(); + ctx.ui.notify(`castle_cdp_search engine reset to the built-in default (${DEFAULT_ENGINE}).`, "info"); + return; + } + + const id = parseEngineId(raw); + if (!id) { + ctx.ui.notify(`Unknown search engine "${raw}". Expected one of: ${engineAliases().join(", ")}`, "error"); + return; + } + + await writeStoredEngine(id); + // Saying so explicitly, because the variable silently wins otherwise and + // the user would reasonably assume the command had taken effect. + const pinned = process.env[ENGINE_ENV_VAR]?.trim(); + ctx.ui.notify( + pinned + ? `Saved ${id}, but ${ENGINE_ENV_VAR}=${pinned} is set and overrides it for this process.` + : `castle_cdp_search will use ${id}.`, + pinned ? "warning" : "info", + ); + }, + }); + pi.registerTool({ name: "castle_cdp_search", label: "Castle Search", @@ -105,21 +188,26 @@ export default function (pi: ExtensionAPI): void { "DevTools Protocol. Returns ranked results with titles, URLs and snippets. Because it uses a real " + "logged-in browser rather than a search API, it reaches pages that block datacentre traffic. It " + "returns search results only — it does not fetch or read the linked pages, so follow up with a " + - "fetch/read tool for full page content. Output is truncated at " + - `${formatSize(DEFAULT_MAX_BYTES)} or ${DEFAULT_MAX_LINES} lines.`, + "fetch/read tool for full page content. Supports four engines with independent indexes and " + + `independent rate limits (${ENGINE_IDS.join(", ")}); pass engine to switch when one is blocked. ` + + `Output is truncated at ${formatSize(DEFAULT_MAX_BYTES)} or ${DEFAULT_MAX_LINES} lines.`, promptSnippet: - "Search the web through a real Chrome on the private network (castle_cdp_search); returns titles, URLs and snippets", + "Search the web through a real Chrome on the private network (castle_cdp_search); google/duckduckgo/bing/brave, returns titles, URLs and snippets", promptGuidelines: [ "Use castle_cdp_search to find pages on the open web when you need current information, documentation, " + "or sources you do not already have — it drives a real signed-in Chrome, so it works on sites that " + "reject scripted clients.", "castle_cdp_search returns search results only. To read a result, follow it up with a tool that fetches " + "page content; do not treat the snippet as the full page.", + "If castle_cdp_search fails with a SearchChallengeError or returns no results, retry the same query with " + + "its engine parameter set to a different engine (google, duckduckgo, bing, brave) before giving up — " + + "they have independent indexes and independent blocks, and a challenge on one says nothing about the others.", + "Only ask the user for help with a castle_cdp_search captcha once more than one engine has failed; " + + "when you do, give them the URL from the error so they can clear it in that browser.", "Pass recency to castle_cdp_search when the answer depends on how recent a page is, and keep numResults " + - "small (5-10) unless you genuinely need a wide sweep.", - "If castle_cdp_search fails with a SearchChallengeError, the browser is sitting behind a captcha or " + - "consent wall that only a person can clear: stop searching, tell the user the URL from the error, and " + - "ask them to complete the challenge in that browser before you retry.", + "small (5-10) unless you genuinely need a wide sweep. Not every engine supports every window: bing has " + + "no year, and brave has no time filter, so castle_cdp_search will tell you to switch engines rather than " + + "silently ignoring the request.", ], parameters, @@ -132,8 +220,9 @@ export default function (pi: ExtensionAPI): void { Math.max(MIN_NUM_RESULTS, params.numResults ?? DEFAULT_NUM_RESULTS), ); const recency = (params.recency ?? null) as Recency | null; + const engine = (params.engine ?? null) as EngineId | null; - const details = initialDetails(query, numResults, recency); + const details = initialDetails(query, numResults, recency, engine); // One deadline covering connect + navigate + extract, plus the user's own // abort so Esc drops an in-flight navigation rather than waiting it out. @@ -148,13 +237,19 @@ export default function (pi: ExtensionAPI): void { }; try { - const outcome = await search({ query, numResults, recency: recency ?? undefined }, combined, report); + const outcome = await search( + { query, numResults, recency: recency ?? undefined, engine }, + combined, + report, + ); const body = formatResults({ query, searchUrl: outcome.searchUrl, finalUrl: outcome.finalUrl, endpoint: outcome.endpoint, + engine: outcome.engine.label, + engineSource: outcome.engineSource, results: outcome.results, }); @@ -170,7 +265,8 @@ export default function (pi: ExtensionAPI): void { : truncation.content; details.phase = "done"; - details.note = `${outcome.results.length} results`; + details.note = `${outcome.results.length} results from ${outcome.engine.id}`; + details.engine = outcome.engine.id; details.endpoint = outcome.endpoint; details.searchUrl = outcome.searchUrl; details.results = outcome.results; @@ -181,16 +277,16 @@ export default function (pi: ExtensionAPI): void { // Rethrow, always. pi only marks a tool result as an error when execute() // throws — returning `{ error }` would read to the model as a search that // simply found nothing, which is the one wrong conclusion to invite here. - // The endpoint is worth naming: "which browser failed" is most of the - // diagnosis when the same config can point at three different machines. const endpoint = await describeConfiguredEndpoint().catch(() => "unknown endpoint"); + const configured = await describeConfiguredEngine().catch(() => null); const e2 = e instanceof Error ? e : new Error(messageOf(e)); // Naming the phase turns "it timed out" into something actionable: // "queued" means the cap was the bottleneck, "navigating" means the // browser or the network was. e2.message = - `${e2.message} [castle_cdp_search: failed during "${details.phase}" — browser ${endpoint}, ` + - `up to ${MAX_CONCURRENT_SEARCHES} concurrent searches, ${SEARCH_TIMEOUT_MS / 1000}s deadline covering the whole call]`; + `${e2.message} [castle_cdp_search: failed during "${details.phase}" — engine ` + + `${engine ?? configured?.id ?? "unknown"}, browser ${endpoint}, up to ${MAX_CONCURRENT_SEARCHES} ` + + `concurrent searches, ${SEARCH_TIMEOUT_MS / 1000}s deadline covering the whole call]`; throw e2; } }, diff --git a/src/search.ts b/src/search.ts index fa7ce35..1306443 100644 --- a/src/search.ts +++ b/src/search.ts @@ -15,8 +15,26 @@ import { describeResolutionFailure, resolveCdpTarget, } from "./endpoint.ts"; -import { BrowserUnavailableError, NoResultsError, SearchChallengeError } from "./errors.ts"; -import { PROBE_SCRIPT, type PageProbe, type SearchResult } from "./extract.ts"; +import { describeEngineFailure, describeEngineSource, resolveEngine } from "./engine-target.ts"; +import { readStoredEngine } from "./engine-store.ts"; +import { + ENGINE_IDS as allEngineIdList, + type EngineDefinition, + type EngineId, + type Recency, + describeRecencySupport, + enginesSupportingRecency, + getEngine, +} from "./engines.ts"; + +const allEngineIds = (): readonly EngineId[] => allEngineIdList; +import { + BrowserUnavailableError, + NoResultsError, + RecencyUnsupportedError, + SearchChallengeError, +} from "./errors.ts"; +import { buildProbeScript, type PageProbe, type SearchResult } from "./extract.ts"; import { readStoredTarget } from "./target-store.ts"; /** Per-search wall clock, covering connect, navigate and extract together. */ @@ -27,12 +45,14 @@ export const MAX_CONCURRENT_SEARCHES = 2; const EXTRACT_RETRIES = 2; const EXTRACT_RETRY_DELAY_MS = 700; -export type Recency = "day" | "week" | "month" | "year"; +export type { Recency } from "./engines.ts"; export type SearchRequest = { query: string; numResults: number; recency?: Recency | undefined; + /** Per-call engine override; falls back to the configured default. */ + engine?: string | null | undefined; }; export type SearchOutcome = { @@ -41,38 +61,59 @@ export type SearchOutcome = { /** The URL the page settled on, which differs from searchUrl after a redirect. */ finalUrl: string; endpoint: string; + engine: EngineDefinition; + /** Where the engine choice came from, for the "via" line in the output. */ + engineSource: string; }; -/** Google's date-restrict vocabulary. */ -const RECENCY_CODE: Record<Recency, string> = { day: "d", week: "w", month: "m", year: "y" }; +/** Build the SERP URL for a request. Exported for tests. */ +export const buildSearchUrl = (engine: EngineDefinition, request: SearchRequest): string => + engine.buildUrl(request.query, request.numResults, request.recency); /** - * `udm=14` asks for the plain "Web" tab: no AI overview, no carousels, just - * ranked links. That is both cheaper to parse and closer to what the agent - * asked for. `hl=en` pins the result language so snippets do not change shape - * with whatever locale the browser profile happens to carry. - * - * Two findings from testing against the real browser, both counter-intuitive - * enough to be worth writing down: - * - * - `tbs=qdr:*` — the parameter the Tools menu puts in the URL — renders an - * empty page in this profile, with `#search` present but no `#rso` and no - * results at all. `as_qdr=*`, the older advanced-search parameter, works and - * genuinely filters (verified: d/w/m/y all return dated results). - * - Any date restriction combined with `udm=14` also renders that empty page. + * Resolve the engine, refusing rather than silently ignoring a recency window + * the chosen engine cannot express. * - * So a time-limited search drops `udm` and uses `as_qdr`. The classic layout it - * falls back to parses fine — the extractor was checked against both. + * Dropping an unsupported filter would be the same failure mode as Google's + * `tbs=qdr:` — the caller asks for recent results, gets whatever the engine + * ranked, and has no way to tell. Naming the engines that *do* support the + * window makes the error actionable, since the model can retry with one. */ -export const buildSearchUrl = (request: SearchRequest): string => { - const params = new URLSearchParams({ - q: request.query, - num: String(request.numResults), - hl: "en", - }); - if (request.recency) params.set("as_qdr", RECENCY_CODE[request.recency]); - else params.set("udm", "14"); - return `https://www.google.com/search?${params.toString()}`; +export const selectEngine = async ( + request: SearchRequest, +): Promise<{ engine: EngineDefinition; source: string }> => { + const resolution = resolveEngine(request.engine, process.env, await readStoredEngine()); + if (resolution.kind === "invalid") { + throw new BrowserUnavailableError(describeEngineFailure(resolution)); + } + const engine = getEngine(resolution.id); + + if (request.recency && engine.recency[request.recency] === undefined) { + const alternatives = enginesSupportingRecency(request.recency); + throw new RecencyUnsupportedError( + `${engine.label} cannot restrict results to the past ${request.recency}. ` + + `It supports: ${describeRecencySupport(engine)}. ` + + (alternatives.length > 0 + ? `Engines that support "${request.recency}": ${alternatives.join(", ")}. ` + + `Retry with engine set to one of those, or drop the recency parameter.` + : `No configured engine supports that window; drop the recency parameter.`), + ); + } + + return { engine, source: describeEngineSource(resolution.source) }; +}; + +/** For status reporting — resolves the configured engine without searching. */ +export const describeConfiguredEngine = async (): Promise<{ id: EngineId; source: string; label: string }> => { + const resolution = resolveEngine(null, process.env, await readStoredEngine()); + if (resolution.kind === "invalid") { + return { id: "google", source: describeEngineFailure(resolution), label: "unresolved" }; + } + return { + id: resolution.id, + source: describeEngineSource(resolution.source), + label: getEngine(resolution.id).label, + }; }; // ── connection lifecycle ─────────────────────────────────────────────────── @@ -190,6 +231,14 @@ const sleep = (ms: number, signal?: AbortSignal): Promise<void> => raceAbort(new Promise<void>((resolve) => setTimeout(resolve, ms)), signal); /** + * The other engines, named in failure messages. A blocked engine is the case + * where the agent most needs to know it has somewhere else to go, and telling + * it inside the error beats hoping it remembers the guidelines. + */ +const otherEngines = (current: EngineId): readonly string[] => + allEngineIds().filter((id) => id !== current); + +/** * Run one search. Throws on every failure — see errors.ts for why a returned * error object would be worse than useless here. */ @@ -203,6 +252,12 @@ export const search = async ( // the caller's and covers the queue wait too — with a cap of two and searches // that take a couple of seconds, waiting is the rare case, and a tool call // that silently takes twice its stated deadline would be worse. + // Resolved before acquiring a slot: an unknown engine name or an unsupported + // recency window is a caller error, and should not wait behind other searches + // or touch the browser at all. + const { engine, source: engineSource } = await selectEngine(request); + const probeScript = buildProbeScript(engine.extraction); + report("queued", "waiting for a search slot"); const release = await acquire(signal); try { @@ -212,18 +267,18 @@ export const search = async ( report("opening", `opening a page on ${endpoint}`); const page = await PageSession.open(conn, signal); - const searchUrl = buildSearchUrl(request); + const searchUrl = buildSearchUrl(engine, request); try { - report("navigating", `searching for "${request.query}"`); + report("navigating", `searching ${engine.label} for "${request.query}"`); await page.navigate(searchUrl, signal); report("extracting", "reading results"); - let probe = await page.evaluate<PageProbe>(PROBE_SCRIPT, signal); + let probe = await page.evaluate<PageProbe>(probeScript, signal); for (let attempt = 0; attempt < EXTRACT_RETRIES; attempt++) { if (probe?.challenge || (probe?.results?.length ?? 0) > 0) break; await sleep(EXTRACT_RETRY_DELAY_MS, signal); - probe = await page.evaluate<PageProbe>(PROBE_SCRIPT, signal); + probe = await page.evaluate<PageProbe>(probeScript, signal); } if (!probe) { @@ -237,26 +292,29 @@ export const search = async ( // after a consent redirect is not the URL we asked for. pageUrl: probe.url || searchUrl, endpoint, + engine: engine.label, + alternatives: otherEngines(engine.id), detail: probe.challenge.detail || undefined, }); } if (probe.results.length === 0) { - // "#rso missing entirely" and "#rso present but empty" are different - // faults, and saying which saves the next person a browser session. + // "container missing entirely" and "container present but empty" are + // different faults, and saying which saves the next person a browser + // session. const diagnosis = - probe.container === "rso" - ? "The results container was rendered but nothing could be parsed out of it, which suggests Google's " + - "result markup changed and castle_cdp_search's extractor needs updating." - : "Google served a page with no results area at all. Most likely this is a soft rate-limit: as a " + - "block decays Google stops serving the /sorry/ challenge page and returns an empty result page " + - "instead, which is indistinguishable from a genuine zero-hit search (observed lasting ~90 minutes " + - "after heavy querying). It can also mean the query truly has no hits, or that this combination of " + - "search parameters is not honoured for this browser profile."; + probe.container === "found" + ? `${engine.label}'s results container was rendered but nothing could be parsed out of it, which ` + + `suggests its markup changed and castle_cdp_search's extractor for ${engine.id} needs updating.` + : `${engine.label} served a page with no results area at all. This is often a soft rate-limit: as a ` + + `block decays an engine stops serving its challenge page and returns an empty result page instead, ` + + `which is indistinguishable from a genuine zero-hit search (observed on Google lasting ~90 minutes ` + + `after heavy querying). It can also mean the query truly has no hits.`; throw new NoResultsError( `No results could be read from ${probe.url || searchUrl} (page title: "${probe.title}"). ${diagnosis} ` + - `Before concluding the extractor is broken, try a broader query and the same query without recency; ` + - `if those are also empty, treat it as rate limiting and wait rather than retrying in a loop.`, + `Before concluding the extractor is broken, retry with a different engine (${otherEngines(engine.id).join(", ")}); ` + + `if those also come back empty the query is the problem, and if only this one does, it is rate limiting — ` + + `wait rather than retrying in a loop.`, ); } @@ -265,6 +323,8 @@ export const search = async ( searchUrl, finalUrl: probe.url || searchUrl, endpoint, + engine, + engineSource, }; } finally { // Always, including the challenge path. Leaving the tab open would let the |
