summaryrefslogtreecommitdiff
path: root/src/engines.ts
diff options
context:
space:
mode:
Diffstat (limited to 'src/engines.ts')
-rw-r--r--src/engines.ts227
1 files changed, 227 insertions, 0 deletions
diff --git a/src/engines.ts b/src/engines.ts
new file mode 100644
index 0000000..79462fc
--- /dev/null
+++ b/src/engines.ts
@@ -0,0 +1,227 @@
+/**
+ * The search engines this extension knows how to drive.
+ *
+ * Every selector and every URL parameter below was read off the real SERP in
+ * castle's Chrome, not recalled or inferred. That is not pedantry: Google's own
+ * Tools menu writes `tbs=qdr:*`, which renders an *empty* page on that profile,
+ * while the undocumented-looking `as_qdr=*` works. Anything in here that was not
+ * observed is a bug waiting to be reported as "no results".
+ *
+ * Two extraction modes, because the engines genuinely differ:
+ *
+ * - "items" — the SERP has a clean per-result container (`li.b_algo`,
+ * `.result`, `.snippet[data-type=web]`). Straightforward.
+ * - "headings" — Google has no stable result container, so results are found
+ * from each <h3> outward. Kept as its own mode rather than
+ * forced into the item shape, because it is the one that has
+ * been through the most verification.
+ */
+
+export type EngineId = "google" | "duckduckgo" | "bing" | "brave";
+
+export type Recency = "day" | "week" | "month" | "year";
+
+export const RECENCY_VALUES = ["day", "week", "month", "year"] as const;
+
+export const ENGINE_IDS: readonly EngineId[] = ["google", "duckduckgo", "bing", "brave"];
+
+/** Aliases accepted from humans and from the model. */
+const ENGINE_ALIASES: Record<string, EngineId> = {
+ google: "google",
+ g: "google",
+ duckduckgo: "duckduckgo",
+ ddg: "duckduckgo",
+ duck: "duckduckgo",
+ bing: "bing",
+ b: "bing",
+ brave: "brave",
+};
+
+/**
+ * How the in-page script should read one engine's results. Passed into the
+ * browser as JSON, so everything here must be plain data.
+ */
+export type ExtractionConfig = {
+ mode: "items" | "headings";
+ /** Results container candidates, first match wins. */
+ roots: string[];
+ /** "items" mode: one result. */
+ item?: string;
+ /** Anchor carrying the outbound link, relative to the item. */
+ link?: string;
+ /** Title element, relative to the item. Falls back to the anchor's text. */
+ title?: string;
+ /** Description element, relative to the item. */
+ snippet?: string;
+ /**
+ * "items" mode with no snippet selector: text from these is subtracted from
+ * the item's text to leave the description behind.
+ */
+ subtract?: string[];
+ /** Items matching any of these are ads or non-web cards, and are skipped. */
+ exclude?: string[];
+ /** How outbound links are wrapped, if they are. */
+ unwrap: "none" | "ddg" | "bing";
+ /** Hosts belonging to the engine itself; links to them are not results. */
+ selfHostPattern: string;
+};
+
+export type EngineDefinition = {
+ readonly id: EngineId;
+ readonly label: string;
+ /** Human-facing note about what makes this engine worth choosing. */
+ readonly note: string;
+ /**
+ * Recency windows this engine can actually express, mapped to the parameter
+ * value. A window that is absent is one the engine does not support — never
+ * one that is silently dropped.
+ */
+ readonly recency: Partial<Record<Recency, string>>;
+ readonly buildUrl: (query: string, numResults: number, recency: Recency | undefined) => string;
+ readonly extraction: ExtractionConfig;
+};
+
+const GOOGLE: EngineDefinition = {
+ id: "google",
+ label: "Google",
+ note: "best result quality; blocks aggressively under repeated automated queries",
+ recency: { day: "d", week: "w", month: "m", year: "y" },
+ buildUrl: (query, numResults, recency) => {
+ const params = new URLSearchParams({ q: query, num: String(numResults), hl: "en" });
+ // `udm=14` is the plain "Web" tab: no AI overview, no carousels. But any
+ // date restriction combined with it renders an empty page, and `tbs=qdr:*`
+ // renders an empty page on its own — both observed. So a time-limited
+ // search drops udm and uses the older as_qdr instead.
+ if (recency) params.set("as_qdr", GOOGLE.recency[recency] as string);
+ else params.set("udm", "14");
+ return `https://www.google.com/search?${params.toString()}`;
+ },
+ extraction: {
+ mode: "headings",
+ roots: ["#rso", "#search"],
+ unwrap: "none",
+ selfHostPattern: String.raw`^https?://(www\.)?google\.[a-z.]+/`,
+ },
+};
+
+const DUCKDUCKGO: EngineDefinition = {
+ id: "duckduckgo",
+ label: "DuckDuckGo",
+ note: "no-JS endpoint, the most stable markup here and the least likely to challenge",
+ // Read off DuckDuckGo's own <select name="df">: "", d, w, m, y.
+ recency: { day: "d", week: "w", month: "m", year: "y" },
+ buildUrl: (query, _numResults, recency) => {
+ // The html endpoint renders server-side with no JavaScript, which makes it
+ // both faster and far less fragile than the app at duckduckgo.com. It has
+ // no result-count parameter — it returns a full page and the caller slices.
+ const params = new URLSearchParams({ q: query });
+ if (recency) params.set("df", DUCKDUCKGO.recency[recency] as string);
+ return `https://html.duckduckgo.com/html/?${params.toString()}`;
+ },
+ extraction: {
+ mode: "items",
+ roots: [".results", "#links"],
+ item: ".result",
+ link: "a.result__a[href]",
+ title: "a.result__a",
+ snippet: ".result__snippet",
+ exclude: [".result--ad", ".badge--ad"],
+ unwrap: "ddg",
+ selfHostPattern: String.raw`^https?://(html\.|www\.)?duckduckgo\.com/`,
+ },
+};
+
+const BING: EngineDefinition = {
+ id: "bing",
+ label: "Bing",
+ note: "good coverage; supports day/week/month only — it has no year filter",
+ // ez1/ez2/ez3 verified to filter (results carried "4 hours ago", "1 day ago").
+ // There is deliberately no year: Bing's UI offers no such window, and the
+ // ez5 custom-range form returned nothing when tried.
+ recency: { day: "ez1", week: "ez2", month: "ez3" },
+ buildUrl: (query, numResults, recency) => {
+ const params = new URLSearchParams({ q: query });
+ if (recency) {
+ // `count` silently cancels `filters` — with ez1 alone every result is
+ // hours old, and adding count in either order brings back months-old
+ // ones. Observed directly, and it fails silently: the results look
+ // perfectly plausible, just unfiltered. Exactly the same trap as Google's
+ // udm+as_qdr, so the same answer — drop the count parameter and let the
+ // caller slice the list it gets.
+ params.set("filters", `ex1:"${BING.recency[recency] as string}"`);
+ } else {
+ params.set("count", String(numResults));
+ }
+ return `https://www.bing.com/search?${params.toString()}`;
+ },
+ extraction: {
+ mode: "items",
+ // Organic results only. Bing's answer cards also contain <h2>s, which is
+ // why this targets li.b_algo rather than walking headings.
+ roots: ["#b_results"],
+ item: "li.b_algo",
+ link: "h2 a[href]",
+ title: "h2",
+ snippet: ".b_caption p, .b_algoSlug, p",
+ exclude: [".b_ad", ".b_adBottom"],
+ unwrap: "bing",
+ selfHostPattern: String.raw`^https?://(www\.)?bing\.com/`,
+ },
+};
+
+const BRAVE: EngineDefinition = {
+ id: "brave",
+ label: "Brave Search",
+ note: "independent index and unwrapped links; challenges quickly under repeated queries",
+ // Left empty deliberately — see the note in README. Brave's filter UI is
+ // client-rendered and the `tf=` values could not be confirmed against a page
+ // that was not simultaneously serving a bot check, so no window is claimed
+ // rather than one being guessed at.
+ recency: {},
+ buildUrl: (query, _numResults, _recency) => {
+ const params = new URLSearchParams({ q: query });
+ return `https://search.brave.com/search?${params.toString()}`;
+ },
+ extraction: {
+ mode: "items",
+ roots: ["#results"],
+ // data-type="web" excludes the AI summariser, video and news cards, which
+ // share the .snippet class but are not ranked web results.
+ item: '.snippet[data-type="web"]',
+ link: "a[href]",
+ title: ".title",
+ // No dedicated description element; the item's text is
+ // "source | breadcrumb | title | description", so subtract the first three.
+ subtract: [".title", "cite", ".sitename", ".netloc"],
+ unwrap: "none",
+ selfHostPattern: String.raw`^https?://(search\.)?brave\.com/`,
+ },
+};
+
+const BY_ID: Record<EngineId, EngineDefinition> = {
+ google: GOOGLE,
+ duckduckgo: DUCKDUCKGO,
+ bing: BING,
+ brave: BRAVE,
+};
+
+export const getEngine = (id: EngineId): EngineDefinition => BY_ID[id];
+
+export const allEngines = (): readonly EngineDefinition[] => ENGINE_IDS.map((id) => BY_ID[id]);
+
+/** Parse a user- or model-supplied engine name. Null when unrecognised. */
+export const parseEngineId = (raw: string): EngineId | null =>
+ ENGINE_ALIASES[raw.trim().toLowerCase()] ?? null;
+
+/** The names accepted for an engine, for help text and error messages. */
+export const engineAliases = (): readonly string[] => Object.keys(ENGINE_ALIASES);
+
+/** Which engines can express a given recency window. */
+export const enginesSupportingRecency = (recency: Recency): readonly EngineId[] =>
+ ENGINE_IDS.filter((id) => BY_ID[id].recency[recency] !== undefined);
+
+/** Human-readable list of the windows an engine supports, or "none". */
+export const describeRecencySupport = (engine: EngineDefinition): string => {
+ const windows = RECENCY_VALUES.filter((r) => engine.recency[r] !== undefined);
+ return windows.length > 0 ? windows.join(", ") : "none";
+};