diff options
Diffstat (limited to 'src/engines.ts')
| -rw-r--r-- | src/engines.ts | 227 |
1 files changed, 227 insertions, 0 deletions
diff --git a/src/engines.ts b/src/engines.ts new file mode 100644 index 0000000..79462fc --- /dev/null +++ b/src/engines.ts @@ -0,0 +1,227 @@ +/** + * The search engines this extension knows how to drive. + * + * Every selector and every URL parameter below was read off the real SERP in + * castle's Chrome, not recalled or inferred. That is not pedantry: Google's own + * Tools menu writes `tbs=qdr:*`, which renders an *empty* page on that profile, + * while the undocumented-looking `as_qdr=*` works. Anything in here that was not + * observed is a bug waiting to be reported as "no results". + * + * Two extraction modes, because the engines genuinely differ: + * + * - "items" — the SERP has a clean per-result container (`li.b_algo`, + * `.result`, `.snippet[data-type=web]`). Straightforward. + * - "headings" — Google has no stable result container, so results are found + * from each <h3> outward. Kept as its own mode rather than + * forced into the item shape, because it is the one that has + * been through the most verification. + */ + +export type EngineId = "google" | "duckduckgo" | "bing" | "brave"; + +export type Recency = "day" | "week" | "month" | "year"; + +export const RECENCY_VALUES = ["day", "week", "month", "year"] as const; + +export const ENGINE_IDS: readonly EngineId[] = ["google", "duckduckgo", "bing", "brave"]; + +/** Aliases accepted from humans and from the model. */ +const ENGINE_ALIASES: Record<string, EngineId> = { + google: "google", + g: "google", + duckduckgo: "duckduckgo", + ddg: "duckduckgo", + duck: "duckduckgo", + bing: "bing", + b: "bing", + brave: "brave", +}; + +/** + * How the in-page script should read one engine's results. Passed into the + * browser as JSON, so everything here must be plain data. + */ +export type ExtractionConfig = { + mode: "items" | "headings"; + /** Results container candidates, first match wins. */ + roots: string[]; + /** "items" mode: one result. */ + item?: string; + /** Anchor carrying the outbound link, relative to the item. */ + link?: string; + /** Title element, relative to the item. Falls back to the anchor's text. */ + title?: string; + /** Description element, relative to the item. */ + snippet?: string; + /** + * "items" mode with no snippet selector: text from these is subtracted from + * the item's text to leave the description behind. + */ + subtract?: string[]; + /** Items matching any of these are ads or non-web cards, and are skipped. */ + exclude?: string[]; + /** How outbound links are wrapped, if they are. */ + unwrap: "none" | "ddg" | "bing"; + /** Hosts belonging to the engine itself; links to them are not results. */ + selfHostPattern: string; +}; + +export type EngineDefinition = { + readonly id: EngineId; + readonly label: string; + /** Human-facing note about what makes this engine worth choosing. */ + readonly note: string; + /** + * Recency windows this engine can actually express, mapped to the parameter + * value. A window that is absent is one the engine does not support — never + * one that is silently dropped. + */ + readonly recency: Partial<Record<Recency, string>>; + readonly buildUrl: (query: string, numResults: number, recency: Recency | undefined) => string; + readonly extraction: ExtractionConfig; +}; + +const GOOGLE: EngineDefinition = { + id: "google", + label: "Google", + note: "best result quality; blocks aggressively under repeated automated queries", + recency: { day: "d", week: "w", month: "m", year: "y" }, + buildUrl: (query, numResults, recency) => { + const params = new URLSearchParams({ q: query, num: String(numResults), hl: "en" }); + // `udm=14` is the plain "Web" tab: no AI overview, no carousels. But any + // date restriction combined with it renders an empty page, and `tbs=qdr:*` + // renders an empty page on its own — both observed. So a time-limited + // search drops udm and uses the older as_qdr instead. + if (recency) params.set("as_qdr", GOOGLE.recency[recency] as string); + else params.set("udm", "14"); + return `https://www.google.com/search?${params.toString()}`; + }, + extraction: { + mode: "headings", + roots: ["#rso", "#search"], + unwrap: "none", + selfHostPattern: String.raw`^https?://(www\.)?google\.[a-z.]+/`, + }, +}; + +const DUCKDUCKGO: EngineDefinition = { + id: "duckduckgo", + label: "DuckDuckGo", + note: "no-JS endpoint, the most stable markup here and the least likely to challenge", + // Read off DuckDuckGo's own <select name="df">: "", d, w, m, y. + recency: { day: "d", week: "w", month: "m", year: "y" }, + buildUrl: (query, _numResults, recency) => { + // The html endpoint renders server-side with no JavaScript, which makes it + // both faster and far less fragile than the app at duckduckgo.com. It has + // no result-count parameter — it returns a full page and the caller slices. + const params = new URLSearchParams({ q: query }); + if (recency) params.set("df", DUCKDUCKGO.recency[recency] as string); + return `https://html.duckduckgo.com/html/?${params.toString()}`; + }, + extraction: { + mode: "items", + roots: [".results", "#links"], + item: ".result", + link: "a.result__a[href]", + title: "a.result__a", + snippet: ".result__snippet", + exclude: [".result--ad", ".badge--ad"], + unwrap: "ddg", + selfHostPattern: String.raw`^https?://(html\.|www\.)?duckduckgo\.com/`, + }, +}; + +const BING: EngineDefinition = { + id: "bing", + label: "Bing", + note: "good coverage; supports day/week/month only — it has no year filter", + // ez1/ez2/ez3 verified to filter (results carried "4 hours ago", "1 day ago"). + // There is deliberately no year: Bing's UI offers no such window, and the + // ez5 custom-range form returned nothing when tried. + recency: { day: "ez1", week: "ez2", month: "ez3" }, + buildUrl: (query, numResults, recency) => { + const params = new URLSearchParams({ q: query }); + if (recency) { + // `count` silently cancels `filters` — with ez1 alone every result is + // hours old, and adding count in either order brings back months-old + // ones. Observed directly, and it fails silently: the results look + // perfectly plausible, just unfiltered. Exactly the same trap as Google's + // udm+as_qdr, so the same answer — drop the count parameter and let the + // caller slice the list it gets. + params.set("filters", `ex1:"${BING.recency[recency] as string}"`); + } else { + params.set("count", String(numResults)); + } + return `https://www.bing.com/search?${params.toString()}`; + }, + extraction: { + mode: "items", + // Organic results only. Bing's answer cards also contain <h2>s, which is + // why this targets li.b_algo rather than walking headings. + roots: ["#b_results"], + item: "li.b_algo", + link: "h2 a[href]", + title: "h2", + snippet: ".b_caption p, .b_algoSlug, p", + exclude: [".b_ad", ".b_adBottom"], + unwrap: "bing", + selfHostPattern: String.raw`^https?://(www\.)?bing\.com/`, + }, +}; + +const BRAVE: EngineDefinition = { + id: "brave", + label: "Brave Search", + note: "independent index and unwrapped links; challenges quickly under repeated queries", + // Left empty deliberately — see the note in README. Brave's filter UI is + // client-rendered and the `tf=` values could not be confirmed against a page + // that was not simultaneously serving a bot check, so no window is claimed + // rather than one being guessed at. + recency: {}, + buildUrl: (query, _numResults, _recency) => { + const params = new URLSearchParams({ q: query }); + return `https://search.brave.com/search?${params.toString()}`; + }, + extraction: { + mode: "items", + roots: ["#results"], + // data-type="web" excludes the AI summariser, video and news cards, which + // share the .snippet class but are not ranked web results. + item: '.snippet[data-type="web"]', + link: "a[href]", + title: ".title", + // No dedicated description element; the item's text is + // "source | breadcrumb | title | description", so subtract the first three. + subtract: [".title", "cite", ".sitename", ".netloc"], + unwrap: "none", + selfHostPattern: String.raw`^https?://(search\.)?brave\.com/`, + }, +}; + +const BY_ID: Record<EngineId, EngineDefinition> = { + google: GOOGLE, + duckduckgo: DUCKDUCKGO, + bing: BING, + brave: BRAVE, +}; + +export const getEngine = (id: EngineId): EngineDefinition => BY_ID[id]; + +export const allEngines = (): readonly EngineDefinition[] => ENGINE_IDS.map((id) => BY_ID[id]); + +/** Parse a user- or model-supplied engine name. Null when unrecognised. */ +export const parseEngineId = (raw: string): EngineId | null => + ENGINE_ALIASES[raw.trim().toLowerCase()] ?? null; + +/** The names accepted for an engine, for help text and error messages. */ +export const engineAliases = (): readonly string[] => Object.keys(ENGINE_ALIASES); + +/** Which engines can express a given recency window. */ +export const enginesSupportingRecency = (recency: Recency): readonly EngineId[] => + ENGINE_IDS.filter((id) => BY_ID[id].recency[recency] !== undefined); + +/** Human-readable list of the windows an engine supports, or "none". */ +export const describeRecencySupport = (engine: EngineDefinition): string => { + const windows = RECENCY_VALUES.filter((r) => engine.recency[r] !== undefined); + return windows.length > 0 ? windows.join(", ") : "none"; +}; |
