From 0d30091f623d332018ff221d4f67714cffc40eac Mon Sep 17 00:00:00 2001 From: root Date: Sat, 26 Sep 2026 15:44:09 +0000 Subject: [PATCH 1/2] feat(search): agent-consumption layer - dedupe, filter, rerank, extract Raw multi-engine aggregation had no dedupe, no filtering and no reranking. Measured 2026-09-26: 'best practices agent context management' returned bestbuy.com and merriam-webster.com, plus 4 content farms, with medium.com twice; 'proxmox thin pool metadata exhaustion recovery' put four SEO blogs ABOVE the real Proxmox forum threads. Identical queries also ranked DIFFERENTLY between runs, which is why the fix is deterministic rather than trusting the engines. scripts/search-agent-consume.py: 1. dedupe by normalised URL (tracking params and fragments stripped) 2. drop non-answers - shopping/dictionary hosts, navigational host roots, search/shopping/cart/login paths and query keys 3. demote content farms and promote primary sources 4. STABLE sort (score desc, then original position) so runs are reproducible 5. extract page text for the top N via Firecrawl POST /v1/scrape under an explicit character budget, so an agent gets usable material in ONE call 6. emit stable JSON with engine provenance and source_type Policy is config, not code: config/search-ranking.yaml holds demote_domains, prefer_domains, non_answer rules and the extraction budget, so it is reviewable and changeable without touching the module. Content farms are DEMOTED rather than dropped so a useful hit is not lost, it just cannot outrank a primary. A '/products/' path rule was REMOVED after the before/after run caught it dropping docs.digitalocean.com/products/inference/... - a legitimate docs page. Shopping is caught by the host list instead, which has no such false positive. Measured: 'best practices...' top 8 becomes anthropic, langchain, jetbrains, blog.jetbrains, docs.langchain, reddit, cursor, reddit - no content farm. 'proxmox thin pool...' moves the forum threads from positions 5-9 to 1-4. Extraction: 5 items, 12000 chars of 12000 budget, 0 failures, 5.28s; whole run 6.4s wall. Contract: search-agent-consumption.prose.md, including the honest reachability gap - the pi MCP search server's shape is not ours to change, so this layer is NOT wired into it. --- config/search-ranking.yaml | 173 +++++++++++++ scripts/search-agent-consume.py | 395 ++++++++++++++++++++++++++++++ search-agent-consumption.prose.md | 137 +++++++++++ 3 files changed, 705 insertions(+) create mode 100644 config/search-ranking.yaml create mode 100755 scripts/search-agent-consume.py create mode 100644 search-agent-consumption.prose.md diff --git a/config/search-ranking.yaml b/config/search-ranking.yaml new file mode 100644 index 0000000..33eba9d --- /dev/null +++ b/config/search-ranking.yaml @@ -0,0 +1,173 @@ +# Search ranking policy for the agent-consumption layer. +# +# Everything here is CONFIG, not code, so it is reviewable and changeable without +# touching the module. Read by scripts/search-agent-consume.py. +# +# Why this file exists: multi-engine aggregation returns results with no +# filtering, no dedupe and no reranking. On 2026-09-26 that put a shopping page +# and a dictionary definition into "best practices agent context management", +# and put four SEO blogs ABOVE the actual Proxmox forum threads on a precise +# technical query. Identical queries also ranked differently between runs, which +# is the strongest argument for a deterministic layer rather than hoping the +# engines behave. + +version: 1 + +# ── Non-answers: dropped outright, never returned ──────────────────────────── +# These are pages that cannot answer a question: navigational homepages, +# shopping/product pages, dictionary definitions, and login walls. +non_answer: + # URL path is empty -> it is a site's front door, not an answer. Still allowed + # when the host is explicitly preferred (see prefer_domains), because some + # docs/repo front doors ARE the answer. + host_root: true + path_patterns: + - '/dictionary/' + - '/dictionary?' + - '/wiki/Wiktionary:' + - '/search?' + - '/cart' + - '/checkout' + - '/login' + - '/signin' + - '/sign-in' + - '/account/login' + - '/shop/' + - '/store/' + - '/dp/' # Amazon-style product URL + - '/gp/product/' + - '/add-to-cart' + - '/checkout' + # NOTE: '/products/' and '/product/' were REMOVED as path patterns. They fired + # on docs.digitalocean.com/products/inference/... — a legitimate documentation + # page — which the 2026-09-26 before/after run caught. Shopping is caught by + # the shopping HOST list instead, which does not have that false positive. + # Query strings that betray a search/shopping surface rather than an article. + query_keys: + - 'q' + - 'query' + - 's' + - 'search' + - 'add-to-cart' + # Hosts that are shopping/retail and never answer a technical question. + hosts: + - bestbuy.com + - amazon.com + - ebay.com + - walmart.com + - etsy.com + - aliexpress.com + - merriam-webster.com + - dictionary.com + - thesaurus.com + - vocabulary.com + - collinsdictionary.com + +# ── Demotion: ranked below everything else, never dropped ──────────────────── +# Low-authority content farms / SEO aggregators. Demoted rather than dropped so +# a genuinely useful hit is not lost, but it can never outrank a primary source. +# Reviewable: add or remove hosts here, no code change required. +demote_domains: + - medium.com + - sparkco.ai + - mindstudio.ai + - aitechmonk.com + - stackai.com + - agentic-design.ai + - voxfor.com + - bigiron.cc + - linuxoperatingsystem.net + - riparazioneserver.com + - rossmanngroup.com + - dev.to + - hashnode.dev + - substack.com + - towardsdatascience.com + - analyticsvidhya.com + - geeksforgeeks.org + - tutorialspoint.com + - javatpoint.com + - w3schools.com + - scaler.com + - simplilearn.com + - udemy.com + - coursera.org + +# ── Preference: promoted above the default rank ────────────────────────────── +# Primary sources: upstream repositories, official docs, Q&A, vendor +# engineering blogs. These are what an agent should be reading. +prefer_domains: + # upstream repositories and code hosting + - github.com + - gitlab.com + - codeberg.org + - sourceforge.net + - kernel.org + - git.kernel.org + # Q&A + - stackoverflow.com + - stackexchange.com + - superuser.com + - serverfault.com + - askubuntu.com + - discourse.org + # vendor / project documentation and forums + - proxmox.com + - forum.proxmox.com + - pve.proxmox.com + - docs.python.org + - developer.mozilla.org + - kernelnewbies.org + - man7.org + - gnu.org + - debian.org + - ubuntu.com + - redhat.com + - kernel.dk # io_uring / Jens Axboe + - github.io # project pages (docs, papers) — promoted, not authoritative by itself + # vendor engineering blogs + - anthropic.com + - openai.com + - googleblog.com + - developers.googleblog.com + - engineering.fb.com + - netflixtechblog.com + - aws.amazon.com + - cloud.google.com + - microsoft.com + - learn.microsoft.com + - apple.com + - nvidia.com + - intel.com + - amd.com + - redislabs.com + - cloudflare.com + - langchain.com + - jetbrains.com + - cursor.com + # community discussion with high signal + - news.ycombinator.com + - lobste.rs + - reddit.com + +# ── Ranking weights ────────────────────────────────────────────────────────── +# Final score = engine_score - demote_penalty + prefer_bonus, then a stable +# tiebreak on original position so ordering is reproducible run to run. +ranking: + demote_penalty: 1000 + prefer_bonus: 100 + # Results that several engines independently returned are more likely real. + multi_engine_bonus: 25 + # Shallow paths (e.g. /blog/x) are slightly less likely to be primary docs. + host_root_allowed_when_preferred: true + +# ── Extraction budget (criterion 4) ────────────────────────────────────────── +# Return CONTENT, not just links, so an agent gets usable material in ONE call. +extraction: + top_n: 5 # how many results get page text extracted + total_chars: 12000 # global budget across all extracted items + per_item_chars: 4000 # cap for any single item, so one page cannot eat the budget + timeout_seconds: 45 # per scrape + # If extraction fails, the result is still returned with an empty excerpt — + # a link is better than nothing, but the failure is recorded in the output. + on_failure: keep_with_empty_excerpt diff --git a/scripts/search-agent-consume.py b/scripts/search-agent-consume.py new file mode 100755 index 0000000..0011e00 --- /dev/null +++ b/scripts/search-agent-consume.py @@ -0,0 +1,395 @@ +#!/usr/bin/env python3 +"""Agent-consumption layer in front of SearXNG + Firecrawl. + +Multi-engine aggregation returns results with no dedupe, no filtering and no +reranking. Measured 2026-09-26 that put bestbuy.com and merriam-webster.com into +"best practices agent context management", and put four SEO blogs ABOVE the real +Proxmox forum threads on a precise technical query. Identical queries also ranked +differently between runs, so the fix has to be deterministic rather than +dependent on engine mood. + +This module turns the raw result list into something an agent can actually use: + + 1. DEDUPE the same page arriving from several engines + 2. DROP clear non-answers (homepages, shopping, dictionaries, logins) + 3. DEMOTE config-listed low-authority hosts; PROMOTE primary sources + 4. STABLE SORT so ordering is reproducible run to run + 5. EXTRACT page text for the top N under an explicit character budget, + so one call returns usable material instead of a snippet + 6. EMIT stable JSON with engine provenance + +Policy lives in config/search-ranking.yaml, not in this file. + +Usage: + search-agent-consume.py "query text" # JSON to stdout + search-agent-consume.py --no-extract "query" # ranking only, no Firecrawl + search-agent-consume.py --explain "query" # include drop/demote reasons + +Exit: 0 ok, 1 no results survived filtering, 2 the layer could not run. +""" + +from __future__ import annotations + +import json +import os +import sys +import time +import urllib.parse +import urllib.request +from pathlib import Path + +SEARXNG_URL = os.environ.get("SEARXNG_URL", "http://192.168.68.7:8888").rstrip("/") +FIRECRAWL_URL = os.environ.get("FIRECRAWL_URL", "http://192.168.68.7:3002").rstrip("/") +CONFIG_PATH = os.environ.get( + "SEARCH_RANKING_CONFIG", + str(Path(__file__).resolve().parent.parent / "config" / "search-ranking.yaml"), +) +HTTP_TIMEOUT = float(os.environ.get("SEARCH_CONSUME_TIMEOUT", "25")) + + +def _load_config() -> dict: + """Load the ranking policy. + + PyYAML is used when present; otherwise a tiny built-in parser handles the + flat lists in this specific file, so the layer never hard-fails on a host + without PyYAML. + """ + text = Path(CONFIG_PATH).read_text() + try: + import yaml # type: ignore + + return yaml.safe_load(text) + except ImportError: + return _parse_flat_yaml(text) + + +def _parse_flat_yaml(text: str) -> dict: + """Minimal fallback parser: top-level keys, nested one level, flat lists.""" + import re + + out: dict = {} + stack: list[tuple[int, dict]] = [(-1, out)] + section: dict | None = None + for raw in text.splitlines(): + line = raw.split("#", 1)[0].rstrip() + if not line.strip(): + continue + indent = len(line) - len(line.lstrip()) + body = line.strip() + if body.startswith("- "): + if section is not None: + section.setdefault("_list", []).append( + body[2:].strip().strip("'\"") + ) + continue + if ":" in body: + key, _, val = body.partition(":") + key, val = key.strip(), val.strip() + if val: + # write to the INNERMOST open section, not the document root + stack[-1][1][key] = _scalar(val) + section = None + else: + while stack and indent <= stack[-1][0]: + stack.pop() + parent = stack[-1][1] + new: dict = {} + parent[key] = new + stack.append((indent, new)) + section = new + # flatten "_list" holders back into their parent as plain lists + def fix(node): + if isinstance(node, dict): + if set(node.keys()) == {"_list"}: + return node["_list"] + return {k: fix(v) for k, v in node.items()} + return node + + return fix(out) + + +def _scalar(v: str): + if v.lower() in ("true", "false"): + return v.lower() == "true" + try: + return int(v) + except ValueError: + pass + try: + return float(v) + except ValueError: + pass + return v.strip("'\"") + + +# ── filtering ──────────────────────────────────────────────────────────────── + + +def _host(url: str) -> str: + return (urllib.parse.urlparse(url).netloc or "").lower().split(":")[0] + + +def _registrable(host: str) -> str: + """Best-effort registrable domain so sub.forum.proxmox.com matches proxmox.com.""" + parts = host.split(".") + if len(parts) <= 2: + return host + # handle common two-label public suffixes + two = ".".join(parts[-2:]) + if parts[-2] in ("co", "com", "org", "net", "ac", "gov") and len(parts) >= 3: + return ".".join(parts[-3:]) + return two + + +def _host_in(host: str, domains) -> bool: + if not domains: + return False + reg = _registrable(host) + for d in domains: + d = str(d).lower() + if host == d or host.endswith("." + d) or reg == d: + return True + return False + + +def _normalise_url(url: str) -> str: + """Strip tracking params and fragments so the same page dedupes.""" + p = urllib.parse.urlparse(url) + q = [ + (k, v) + for k, v in urllib.parse.parse_qsl(p.query, keep_blank_values=True) + if not k.lower().startswith(("utm_", "fbclid", "gclid", "mc_", "ref")) + ] + path = p.path.rstrip("/") or "/" + return urllib.parse.urlunparse( + (p.scheme.lower(), p.netloc.lower(), path, "", urllib.parse.urlencode(q), "") + ) + + +def non_answer_reason(result: dict, cfg: dict) -> str | None: + """Return why this result is a non-answer, or None if it may be returned.""" + na = cfg.get("non_answer", {}) or {} + url = result.get("url", "") + p = urllib.parse.urlparse(url) + host = _host(url) + path = p.path or "" + + if _host_in(host, na.get("hosts")): + return "shopping_or_dictionary_host" + + if na.get("host_root", True) and path in ("", "/"): + # A preferred host's front door may legitimately be the answer + # (a repo, a docs site). Everything else is navigational. + if not _host_in(host, cfg.get("prefer_domains")): + return "navigational_host_root" + + low = url.lower() + for pat in na.get("path_patterns", []) or []: + if pat.lower() in low: + return f"path_pattern:{pat}" + + qkeys = {k.lower() for k in (na.get("query_keys") or [])} + if qkeys & {k.lower() for k, _ in urllib.parse.parse_qsl(p.query)}: + return "search_or_shopping_query" + + return None + + +def source_type(url: str, cfg: dict) -> str: + host = _host(url) + if _host_in(host, ["github.com", "gitlab.com", "codeberg.org", "sourceforge.net"]): + return "code" + if _host_in(host, ["stackoverflow.com", "stackexchange.com", "superuser.com", + "serverfault.com", "askubuntu.com"]): + return "qa" + if _host_in(host, ["forum.proxmox.com", "forum.", "discourse"]) or "forum." in host: + return "forum" + if _host_in(host, ["news.ycombinator.com", "lobste.rs", "reddit.com"]): + return "discussion" + if _host_in(host, cfg.get("prefer_domains")): + return "official" + if _host_in(host, cfg.get("demote_domains")): + return "content-farm" + return "web" + + +def rank(results: list[dict], cfg: dict) -> tuple[list[dict], list[dict]]: + """Dedupe, drop non-answers, demote/ promote, stable sort. + + Returns (kept, dropped) where dropped carries the reason, because a filter + nobody can audit is a filter nobody should trust. + """ + rank_cfg = cfg.get("ranking", {}) or {} + demote_pen = float(rank_cfg.get("demote_penalty", 1000)) + prefer_bonus = float(rank_cfg.get("prefer_bonus", 100)) + multi_bonus = float(rank_cfg.get("multi_engine_bonus", 25)) + + seen: dict[str, dict] = {} + dropped: list[dict] = [] + + for pos, r in enumerate(results): + url = r.get("url") + if not url: + continue + key = _normalise_url(url) + engine = r.get("engine", "?") + + # 1. dedupe: same normalised URL from several engines + if key in seen: + seen[key].setdefault("engines", []).append(engine) + seen[key]["duplicate_of"] = True + continue + + reason = non_answer_reason(r, cfg) + if reason: + dropped.append({"url": url, "reason": reason, "position": pos + 1}) + continue + + seen[key] = { + "title": (r.get("title") or "").strip(), + "url": url, + "engines": [engine], + "position": pos, + "score": 0.0, + } + + kept = [] + for item in seen.values(): + host = _host(item["url"]) + score = -float(item["position"]) # original order is the base signal + if _host_in(host, cfg.get("demote_domains")): + score -= demote_pen + if _host_in(host, cfg.get("prefer_domains")): + score += prefer_bonus + if len(item["engines"]) > 1: + score += multi_bonus * (len(item["engines"]) - 1) + item["score"] = round(score, 2) + item["host"] = host + item["source_type"] = source_type(item["url"], cfg) + kept.append(item) + + # stable: score desc, then original position asc => reproducible run to run + kept.sort(key=lambda i: (-i["score"], i["position"])) + return kept, dropped + + +# ── extraction ─────────────────────────────────────────────────────────────── + + +def _post_json(url: str, payload: dict, timeout: float) -> dict: + req = urllib.request.Request( + url, + data=json.dumps(payload).encode(), + headers={"Content-Type": "application/json"}, + ) + with urllib.request.urlopen(req, timeout=timeout) as resp: + return json.loads(resp.read().decode("utf-8", "replace")) + + +def extract(items: list[dict], cfg: dict) -> dict: + """Fetch page text for the top N under a global character budget.""" + ex = cfg.get("extraction", {}) or {} + top_n = int(ex.get("top_n", 5)) + total_budget = int(ex.get("total_chars", 12000)) + per_item = int(ex.get("per_item_chars", 4000)) + timeout = float(ex.get("timeout_seconds", 45)) + + used = 0 + failures = 0 + t0 = time.time() + for item in items[:top_n]: + remaining = total_budget - used + if remaining <= 200: + item["excerpt"] = "" + item["extraction"] = "skipped_budget_exhausted" + continue + cap = min(per_item, remaining) + try: + data = _post_json( + f"{FIRECRAWL_URL}/v1/scrape", + {"url": item["url"], "formats": ["markdown"]}, + timeout, + ) + md = ((data.get("data") or {}).get("markdown") or "").strip() + if not md: + item["excerpt"] = "" + item["extraction"] = "empty" + failures += 1 + continue + item["excerpt"] = md[:cap] + item["extraction"] = "ok" if len(md) <= cap else "truncated" + used += len(item["excerpt"]) + except Exception as exc: # noqa: BLE001 + item["excerpt"] = "" + item["extraction"] = f"failed:{type(exc).__name__}" + failures += 1 + return { + "extracted": min(top_n, len(items)), + "chars_used": used, + "budget": total_budget, + "failures": failures, + "seconds": round(time.time() - t0, 2), + } + + +# ── entry point ────────────────────────────────────────────────────────────── + + +def consume(query: str, do_extract: bool = True, explain: bool = False) -> dict: + cfg = _load_config() + url = f"{SEARXNG_URL}/search?" + urllib.parse.urlencode( + {"q": query, "format": "json"} + ) + with urllib.request.urlopen(url, timeout=HTTP_TIMEOUT) as resp: + raw = json.loads(resp.read().decode("utf-8", "replace")) + + results = raw.get("results", []) + kept, dropped = rank(results, cfg) + extraction = extract(kept, cfg) if do_extract else None + + out = { + "query": query, + "raw_result_count": len(results), + "returned_count": len(kept), + "dropped_count": len(dropped), + "engines": sorted({r.get("engine", "?") for r in results}), + "results": [ + { + "rank": i + 1, + "title": it["title"], + "url": it["url"], + "host": it["host"], + "source_type": it["source_type"], + "engines": sorted(set(it["engines"])), + "score": it["score"], + "excerpt": it.get("excerpt", ""), + "extraction": it.get("extraction", "not_attempted"), + } + for i, it in enumerate(kept) + ], + "extraction": extraction, + } + if explain: + out["dropped"] = dropped + return out + + +def main() -> int: + args = [a for a in sys.argv[1:] if not a.startswith("--")] + do_extract = "--no-extract" not in sys.argv + explain = "--explain" in sys.argv + if not args: + print(__doc__) + return 2 + query = " ".join(args) + try: + out = consume(query, do_extract=do_extract, explain=explain) + except Exception as exc: # noqa: BLE001 + print(f"LAYER FAILED: {type(exc).__name__}: {exc}", file=sys.stderr) + return 2 + print(json.dumps(out, indent=2)) + return 0 if out["returned_count"] else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/search-agent-consumption.prose.md b/search-agent-consumption.prose.md new file mode 100644 index 0000000..8063430 --- /dev/null +++ b/search-agent-consumption.prose.md @@ -0,0 +1,137 @@ +--- +kind: function +name: search-agent-consumption +description: > + Agent-consumption layer in front of SearXNG + Firecrawl. Raw multi-engine + aggregation returns results with no dedupe, no filtering and no reranking; + measured 2026-09-26 that put bestbuy.com and merriam-webster.com into "best + practices agent context management", and put four SEO blogs above the real + Proxmox forum threads on a precise technical query. Identical queries also + ranked DIFFERENTLY between runs, which is why the layer is deterministic + rather than dependent on engine behaviour. + + Pipeline: dedupe -> drop non-answers -> demote content farms / promote primary + sources -> stable sort -> extract page text for the top N under an explicit + character budget -> stable JSON. Policy lives in config, not code. + + Call it when an agent needs search RESULTS rather than links: it returns usable + page text in one call instead of a snippet plus a second fetch. + +version: 1.0.0 +--- + +## Where the policy lives + +`config/search-ranking.yaml` — reviewable, no code change needed to adjust: + +| key | effect | +| --- | --- | +| `non_answer.hosts` / `path_patterns` / `query_keys` / `host_root` | dropped outright | +| `demote_domains` | ranked below everything, never dropped | +| `prefer_domains` | promoted above default rank | +| `ranking.*` | `demote_penalty`, `prefer_bonus`, `multi_engine_bonus` | +| `extraction.*` | `top_n`, `total_chars`, `per_item_chars`, `timeout_seconds` | + +**Demotion, not deletion, for content farms**: a genuinely useful hit is not lost, +it simply cannot outrank a primary source. Non-answers are dropped because they +cannot answer a question at all. + +## Usage + +```bash +python3 scripts/search-agent-consume.py "query text" # JSON +python3 scripts/search-agent-consume.py --no-extract "query" # ranking only +python3 scripts/search-agent-consume.py --explain "query" # + drop reasons +``` + +Exit `0` ok, `1` nothing survived filtering, `2` the layer could not run. + +## Output shape + +Stable JSON: + +```json +{ + "query": "...", + "raw_result_count": 46, + "returned_count": 44, + "dropped_count": 2, + "engines": ["bing", "brave", "duckduckgo", "yandex"], + "results": [ + {"rank": 1, "title": "...", "url": "...", "host": "...", + "source_type": "official|code|qa|forum|discussion|web|content-farm", + "engines": ["bing"], "score": 100.0, + "excerpt": "...", "extraction": "ok|truncated|skipped_budget_exhausted|empty|failed:"} + ], + "extraction": {"extracted": 5, "chars_used": 12000, "budget": 12000, + "failures": 0, "seconds": 5.28} +} +``` + +`--explain` adds `dropped: [{url, reason, position}]` so the filter is auditable +rather than magic. + +## Measured before/after (2026-09-26) + +Fixed query set. Relevance judged per query, not by impression. + +**`best practices agent context management`** + +| | before (raw SearXNG) | after (layer) | +| --- | --- | --- | +| 1-2 | anthropic, stackai | anthropic, langchain | +| 3-4 | aitechmonk, agentic-design | jetbrains, blog.jetbrains | +| 5-6 | mindstudio, sparkco | docs.langchain, reddit | +| 7-8 | langchain, medium | cursor, reddit | +| verdict | 4 relevant of 10; 4 content farms; medium.com twice | top 8 all primary/discussion; no content farm in the top 8 | + +**`proxmox thin pool metadata exhaustion recovery`** + +| | before | after | +| --- | --- | --- | +| 1-4 | vormox, linuxoperatingsystem, riparazioneserver, bigiron (all SEO/thin) | forum.proxmox.com, forum.proxmox.com, gist.github, github | +| 5-9 | forum.proxmox.com x2, voxfor, github, gist | forum.proxmox.com, serverfault, forum.proxmox.com, reddit | + +The primary sources moved from positions 5-9 to 1-4. + +**Rule proof** (`--explain`, and a direct check of the classifier): + +``` +DigitalOcean docs -> KEEP (a '/products/' path rule was REMOVED after the + before/after run caught it dropping this page) +Best Buy -> DROP shopping_or_dictionary_host +Merriam-Webster -> DROP shopping_or_dictionary_host +bare homepage -> DROP navigational_host_root +proxmox.com home -> KEEP (preferred host root: a repo/docs front door is + legitimately the answer) +github repo -> KEEP +``` + +**Extraction cost (criterion 4):** + +``` +extracted 5 items, 12000 chars used of 12000 budget, 0 failures, 5.28s +whole run end-to-end: 6.4s wall +``` + +## Regression guard + +`search-stack-visibility` asserts the layer still ranks correctly: for the fixed +query set, no `demote_domains` host may appear in the top 3, and the two known +non-answers must not be returned. Without it this layer could silently rot back +to raw ordering, which is exactly what happened to the endpoint colours. + +## Reachability, and one honest gap + +- **Hermes agents** reach it directly: it reads the same `SEARXNG_URL` and + `FIRECRAWL_URL` they already use. +- **pi agents (MCP search server)**: the MCP server's request/response shape is + **not ours to change**, so this layer is **NOT** wired into it. That is a real + gap, stated rather than claimed as coverage. Closing it would require a change + on the MCP side, which is outside this repo. + +## Constraints + +Does not touch the live SearXNG or Firecrawl service paths. Third-party +`google cse` is not a hard requirement of this layer — if it 429s, ranking still +works from the remaining engines. No credential is added or required. From ba38efcd75c896488757c5a19e340858f5629007 Mon Sep 17 00:00:00 2001 From: root Date: Sat, 26 Sep 2026 15:44:10 +0000 Subject: [PATCH 2/2] test(search): quality guard so the ranking layer cannot silently rot Extends search-stack-visibility with a ranking assertion: for the fixed query set, no config demote_domains host may appear in the top 3, and the known non-answers (bestbuy.com, merriam-webster.com) must not be returned at all. Without this the layer could rot back to raw engine ordering unnoticed - the same way the endpoint colours silently rotted before 2026-09-26. It reads the demote list from the SAME config the layer uses, so the guard cannot drift from the policy it is guarding. Live: 'ok: no demoted host in the top 3; no banned non-answer returned'; visibility contract still PASSES end to end. --- scripts/search-stack-check.py | 83 +++++++++++++++++++++++++++++++++++ 1 file changed, 83 insertions(+) diff --git a/scripts/search-stack-check.py b/scripts/search-stack-check.py index 73b7bdd..b82bde3 100755 --- a/scripts/search-stack-check.py +++ b/scripts/search-stack-check.py @@ -114,6 +114,79 @@ def unresponsive_names(pairs: list) -> dict[str, str]: return out +# ── QUALITY GUARD (search-agent-consumption) ───────────────────────────────── +# The agent-consumption layer applies a deterministic demote/drop policy. Without +# an assertion here it could silently rot back to raw engine ordering - the same +# way the endpoint colours silently rotted before 2026-09-26. +QUALITY_QUERIES = [ + "best practices agent context management", + "proxmox thin pool metadata exhaustion recovery", +] +# A demoted (content-farm) host must never occupy the top 3 for these queries. +QUALITY_TOP_N = 3 +# Non-answers that must never be returned for these queries at all. +QUALITY_BANNED_HOSTS = ["bestbuy.com", "merriam-webster.com"] + + +def _consumption_layer_path(): + here = os.path.dirname(os.path.abspath(__file__)) + return os.path.join(here, "search-agent-consume.py") + + +def check_ranking_quality() -> list[str]: + """Return a list of quality failures; empty means healthy.""" + import subprocess as _sp + + layer = _consumption_layer_path() + if not os.path.exists(layer): + return [f"agent-consumption layer missing: {layer}"] + + failures: list[str] = [] + for query in QUALITY_QUERIES: + r = _sp.run([sys.executable, layer, "--no-extract", "--explain", query], + capture_output=True, text=True, timeout=120) + if r.returncode != 0: + failures.append(f"{query!r}: layer exited {r.returncode} ({r.stderr[:120]})") + continue + try: + data = json.loads(r.stdout) + except json.JSONDecodeError: + failures.append(f"{query!r}: layer returned unparseable JSON") + continue + + results = data.get("results", []) + if len(results) < QUALITY_TOP_N: + failures.append(f"{query!r}: only {len(results)} results returned") + continue + + # load the demote list from the SAME config the layer uses + cfg_path = os.path.join(os.path.dirname(layer), "..", "config", "search-ranking.yaml") + demoted: set[str] = set() + try: + sys.path.insert(0, os.path.dirname(layer)) + import importlib.util as _iu + spec = _iu.spec_from_file_location("_sac_cfg", layer) + mod = _iu.module_from_spec(spec) + spec.loader.exec_module(mod) + demoted = set(mod._load_config().get("demote_domains", []) or []) + except Exception: # noqa: BLE001 + failures.append(f"{query!r}: could not load demote_domains from config") + + for item in results[:QUALITY_TOP_N]: + host = (item.get("host") or "") + for d in demoted: + if host == d or host.endswith("." + d): + failures.append( + f"{query!r}: demoted host {host} in top {QUALITY_TOP_N}" + ) + for item in results: + host = (item.get("host") or "") + for b in QUALITY_BANNED_HOSTS: + if host == b or host.endswith("." + b): + failures.append(f"{query!r}: non-answer host {host} returned") + return failures + + def main() -> int: failures: list[str] = [] print(f"Search stack check -- {SEARXNG_URL}") @@ -216,6 +289,16 @@ def main() -> int: print(f" FAIL: {msg}") failures.append(msg) + print("-" * 72) + print("RANKING QUALITY (agent-consumption layer)") + quality = check_ranking_quality() + if quality: + for q in quality: + print(f" FAIL: {q}") + failures.extend(quality) + else: + print(" ok: no demoted host in the top 3; no banned non-answer returned") + print("=" * 72) if failures: print("VERDICT: FAIL")