summaryrefslogtreecommitdiff
path: root/src/engines.ts
diff options
context:
space:
mode:
authorIgor Soarez <igor@soarez.org>2026-08-03 21:43:56 +0100
committerIgor Soarez <igor@soarez.org>2026-08-03 21:43:56 +0100
commitdb69207c06e8d5233bff4996e3ef5b43332d533f (patch)
tree032ebbc3983a6eb4f9e3c4ac162c60a9cc1edc75 /src/engines.ts
parent495de0d5283dd3e4a6ef715b596c4a2892e95915 (diff)
Support DuckDuckGo, Bing and Brave, switchable like the CDP host
Engine resolution mirrors the browser target: CCS_SEARCH_ENGINE, then a /search-engine choice persisted machine-wide, then Google. A per-call `engine` parameter sits above both so the agent can fall back when one engine starts serving captchas — the one case where the model, not the operator, has to make the call. Unlike the CDP target there is no safety argument for the environment winning: driving the wrong browser means automating someone's signed-in Chrome, choosing a different index does not. An unsupported recency window is refused, naming the engines that support it, rather than dropped. Silently returning unfiltered results is indistinguishable from success, which is the failure this whole design is trying to avoid. Extraction grows a second mode. DuckDuckGo, Bing and Brave have clean per-result containers; Google does not, so its heading-walk stays as its own path rather than being bent into the item shape. DuckDuckGo and Bing route links through redirectors, unwrapped in the page. Third silent-failure trap found, alongside Google's two: on Bing, `count` cancels `filters`. With ex1:"ez1" alone every result is hours old; add count in either order and months-old results return, looking perfectly ordinary. Bing now drops count whenever a date filter is present. Challenge detection widened to Brave's "Verifying you're not a bot" and "Quick check before you continue searching", which the previous Google- shaped matcher missed entirely — found by tripping it.
Diffstat (limited to 'src/engines.ts')
-rw-r--r--src/engines.ts227
1 files changed, 227 insertions, 0 deletions
diff --git a/src/engines.ts b/src/engines.ts
new file mode 100644
index 0000000..79462fc
--- /dev/null
+++ b/src/engines.ts
@@ -0,0 +1,227 @@
+/**
+ * The search engines this extension knows how to drive.
+ *
+ * Every selector and every URL parameter below was read off the real SERP in
+ * castle's Chrome, not recalled or inferred. That is not pedantry: Google's own
+ * Tools menu writes `tbs=qdr:*`, which renders an *empty* page on that profile,
+ * while the undocumented-looking `as_qdr=*` works. Anything in here that was not
+ * observed is a bug waiting to be reported as "no results".
+ *
+ * Two extraction modes, because the engines genuinely differ:
+ *
+ * - "items" — the SERP has a clean per-result container (`li.b_algo`,
+ * `.result`, `.snippet[data-type=web]`). Straightforward.
+ * - "headings" — Google has no stable result container, so results are found
+ * from each <h3> outward. Kept as its own mode rather than
+ * forced into the item shape, because it is the one that has
+ * been through the most verification.
+ */
+
+export type EngineId = "google" | "duckduckgo" | "bing" | "brave";
+
+export type Recency = "day" | "week" | "month" | "year";
+
+export const RECENCY_VALUES = ["day", "week", "month", "year"] as const;
+
+export const ENGINE_IDS: readonly EngineId[] = ["google", "duckduckgo", "bing", "brave"];
+
+/** Aliases accepted from humans and from the model. */
+const ENGINE_ALIASES: Record<string, EngineId> = {
+ google: "google",
+ g: "google",
+ duckduckgo: "duckduckgo",
+ ddg: "duckduckgo",
+ duck: "duckduckgo",
+ bing: "bing",
+ b: "bing",
+ brave: "brave",
+};
+
+/**
+ * How the in-page script should read one engine's results. Passed into the
+ * browser as JSON, so everything here must be plain data.
+ */
+export type ExtractionConfig = {
+ mode: "items" | "headings";
+ /** Results container candidates, first match wins. */
+ roots: string[];
+ /** "items" mode: one result. */
+ item?: string;
+ /** Anchor carrying the outbound link, relative to the item. */
+ link?: string;
+ /** Title element, relative to the item. Falls back to the anchor's text. */
+ title?: string;
+ /** Description element, relative to the item. */
+ snippet?: string;
+ /**
+ * "items" mode with no snippet selector: text from these is subtracted from
+ * the item's text to leave the description behind.
+ */
+ subtract?: string[];
+ /** Items matching any of these are ads or non-web cards, and are skipped. */
+ exclude?: string[];
+ /** How outbound links are wrapped, if they are. */
+ unwrap: "none" | "ddg" | "bing";
+ /** Hosts belonging to the engine itself; links to them are not results. */
+ selfHostPattern: string;
+};
+
+export type EngineDefinition = {
+ readonly id: EngineId;
+ readonly label: string;
+ /** Human-facing note about what makes this engine worth choosing. */
+ readonly note: string;
+ /**
+ * Recency windows this engine can actually express, mapped to the parameter
+ * value. A window that is absent is one the engine does not support — never
+ * one that is silently dropped.
+ */
+ readonly recency: Partial<Record<Recency, string>>;
+ readonly buildUrl: (query: string, numResults: number, recency: Recency | undefined) => string;
+ readonly extraction: ExtractionConfig;
+};
+
+const GOOGLE: EngineDefinition = {
+ id: "google",
+ label: "Google",
+ note: "best result quality; blocks aggressively under repeated automated queries",
+ recency: { day: "d", week: "w", month: "m", year: "y" },
+ buildUrl: (query, numResults, recency) => {
+ const params = new URLSearchParams({ q: query, num: String(numResults), hl: "en" });
+ // `udm=14` is the plain "Web" tab: no AI overview, no carousels. But any
+ // date restriction combined with it renders an empty page, and `tbs=qdr:*`
+ // renders an empty page on its own — both observed. So a time-limited
+ // search drops udm and uses the older as_qdr instead.
+ if (recency) params.set("as_qdr", GOOGLE.recency[recency] as string);
+ else params.set("udm", "14");
+ return `https://www.google.com/search?${params.toString()}`;
+ },
+ extraction: {
+ mode: "headings",
+ roots: ["#rso", "#search"],
+ unwrap: "none",
+ selfHostPattern: String.raw`^https?://(www\.)?google\.[a-z.]+/`,
+ },
+};
+
+const DUCKDUCKGO: EngineDefinition = {
+ id: "duckduckgo",
+ label: "DuckDuckGo",
+ note: "no-JS endpoint, the most stable markup here and the least likely to challenge",
+ // Read off DuckDuckGo's own <select name="df">: "", d, w, m, y.
+ recency: { day: "d", week: "w", month: "m", year: "y" },
+ buildUrl: (query, _numResults, recency) => {
+ // The html endpoint renders server-side with no JavaScript, which makes it
+ // both faster and far less fragile than the app at duckduckgo.com. It has
+ // no result-count parameter — it returns a full page and the caller slices.
+ const params = new URLSearchParams({ q: query });
+ if (recency) params.set("df", DUCKDUCKGO.recency[recency] as string);
+ return `https://html.duckduckgo.com/html/?${params.toString()}`;
+ },
+ extraction: {
+ mode: "items",
+ roots: [".results", "#links"],
+ item: ".result",
+ link: "a.result__a[href]",
+ title: "a.result__a",
+ snippet: ".result__snippet",
+ exclude: [".result--ad", ".badge--ad"],
+ unwrap: "ddg",
+ selfHostPattern: String.raw`^https?://(html\.|www\.)?duckduckgo\.com/`,
+ },
+};
+
+const BING: EngineDefinition = {
+ id: "bing",
+ label: "Bing",
+ note: "good coverage; supports day/week/month only — it has no year filter",
+ // ez1/ez2/ez3 verified to filter (results carried "4 hours ago", "1 day ago").
+ // There is deliberately no year: Bing's UI offers no such window, and the
+ // ez5 custom-range form returned nothing when tried.
+ recency: { day: "ez1", week: "ez2", month: "ez3" },
+ buildUrl: (query, numResults, recency) => {
+ const params = new URLSearchParams({ q: query });
+ if (recency) {
+ // `count` silently cancels `filters` — with ez1 alone every result is
+ // hours old, and adding count in either order brings back months-old
+ // ones. Observed directly, and it fails silently: the results look
+ // perfectly plausible, just unfiltered. Exactly the same trap as Google's
+ // udm+as_qdr, so the same answer — drop the count parameter and let the
+ // caller slice the list it gets.
+ params.set("filters", `ex1:"${BING.recency[recency] as string}"`);
+ } else {
+ params.set("count", String(numResults));
+ }
+ return `https://www.bing.com/search?${params.toString()}`;
+ },
+ extraction: {
+ mode: "items",
+ // Organic results only. Bing's answer cards also contain <h2>s, which is
+ // why this targets li.b_algo rather than walking headings.
+ roots: ["#b_results"],
+ item: "li.b_algo",
+ link: "h2 a[href]",
+ title: "h2",
+ snippet: ".b_caption p, .b_algoSlug, p",
+ exclude: [".b_ad", ".b_adBottom"],
+ unwrap: "bing",
+ selfHostPattern: String.raw`^https?://(www\.)?bing\.com/`,
+ },
+};
+
+const BRAVE: EngineDefinition = {
+ id: "brave",
+ label: "Brave Search",
+ note: "independent index and unwrapped links; challenges quickly under repeated queries",
+ // Left empty deliberately — see the note in README. Brave's filter UI is
+ // client-rendered and the `tf=` values could not be confirmed against a page
+ // that was not simultaneously serving a bot check, so no window is claimed
+ // rather than one being guessed at.
+ recency: {},
+ buildUrl: (query, _numResults, _recency) => {
+ const params = new URLSearchParams({ q: query });
+ return `https://search.brave.com/search?${params.toString()}`;
+ },
+ extraction: {
+ mode: "items",
+ roots: ["#results"],
+ // data-type="web" excludes the AI summariser, video and news cards, which
+ // share the .snippet class but are not ranked web results.
+ item: '.snippet[data-type="web"]',
+ link: "a[href]",
+ title: ".title",
+ // No dedicated description element; the item's text is
+ // "source | breadcrumb | title | description", so subtract the first three.
+ subtract: [".title", "cite", ".sitename", ".netloc"],
+ unwrap: "none",
+ selfHostPattern: String.raw`^https?://(search\.)?brave\.com/`,
+ },
+};
+
+const BY_ID: Record<EngineId, EngineDefinition> = {
+ google: GOOGLE,
+ duckduckgo: DUCKDUCKGO,
+ bing: BING,
+ brave: BRAVE,
+};
+
+export const getEngine = (id: EngineId): EngineDefinition => BY_ID[id];
+
+export const allEngines = (): readonly EngineDefinition[] => ENGINE_IDS.map((id) => BY_ID[id]);
+
+/** Parse a user- or model-supplied engine name. Null when unrecognised. */
+export const parseEngineId = (raw: string): EngineId | null =>
+ ENGINE_ALIASES[raw.trim().toLowerCase()] ?? null;
+
+/** The names accepted for an engine, for help text and error messages. */
+export const engineAliases = (): readonly string[] => Object.keys(ENGINE_ALIASES);
+
+/** Which engines can express a given recency window. */
+export const enginesSupportingRecency = (recency: Recency): readonly EngineId[] =>
+ ENGINE_IDS.filter((id) => BY_ID[id].recency[recency] !== undefined);
+
+/** Human-readable list of the windows an engine supports, or "none". */
+export const describeRecencySupport = (engine: EngineDefinition): string => {
+ const windows = RECENCY_VALUES.filter((r) => engine.recency[r] !== undefined);
+ return windows.length > 0 ? windows.join(", ") : "none";
+};