diff --git a/config/search-ranking.yaml b/config/search-ranking.yaml new file mode 100644 index 0000000..33eba9d --- /dev/null +++ b/config/search-ranking.yaml @@ -0,0 +1,173 @@ +# Search ranking policy for the agent-consumption layer. +# +# Everything here is CONFIG, not code, so it is reviewable and changeable without +# touching the module. Read by scripts/search-agent-consume.py. +# +# Why this file exists: multi-engine aggregation returns results with no +# filtering, no dedupe and no reranking. On 2026-09-26 that put a shopping page +# and a dictionary definition into "best practices agent context management", +# and put four SEO blogs ABOVE the actual Proxmox forum threads on a precise +# technical query. Identical queries also ranked differently between runs, which +# is the strongest argument for a deterministic layer rather than hoping the +# engines behave. + +version: 1 + +# ── Non-answers: dropped outright, never returned ──────────────────────────── +# These are pages that cannot answer a question: navigational homepages, +# shopping/product pages, dictionary definitions, and login walls. +non_answer: + # URL path is empty -> it is a site's front door, not an answer. Still allowed + # when the host is explicitly preferred (see prefer_domains), because some + # docs/repo front doors ARE the answer. + host_root: true + path_patterns: + - '/dictionary/' + - '/dictionary?' + - '/wiki/Wiktionary:' + - '/search?' + - '/cart' + - '/checkout' + - '/login' + - '/signin' + - '/sign-in' + - '/account/login' + - '/shop/' + - '/store/' + - '/dp/' # Amazon-style product URL + - '/gp/product/' + - '/add-to-cart' + - '/checkout' + # NOTE: '/products/' and '/product/' were REMOVED as path patterns. They fired + # on docs.digitalocean.com/products/inference/... — a legitimate documentation + # page — which the 2026-09-26 before/after run caught. Shopping is caught by + # the shopping HOST list instead, which does not have that false positive. + # Query strings that betray a search/shopping surface rather than an article. + query_keys: + - 'q' + - 'query' + - 's' + - 'search' + - 'add-to-cart' + # Hosts that are shopping/retail and never answer a technical question. + hosts: + - bestbuy.com + - amazon.com + - ebay.com + - walmart.com + - etsy.com + - aliexpress.com + - merriam-webster.com + - dictionary.com + - thesaurus.com + - vocabulary.com + - collinsdictionary.com + +# ── Demotion: ranked below everything else, never dropped ──────────────────── +# Low-authority content farms / SEO aggregators. Demoted rather than dropped so +# a genuinely useful hit is not lost, but it can never outrank a primary source. +# Reviewable: add or remove hosts here, no code change required. +demote_domains: + - medium.com + - sparkco.ai + - mindstudio.ai + - aitechmonk.com + - stackai.com + - agentic-design.ai + - voxfor.com + - bigiron.cc + - linuxoperatingsystem.net + - riparazioneserver.com + - rossmanngroup.com + - dev.to + - hashnode.dev + - substack.com + - towardsdatascience.com + - analyticsvidhya.com + - geeksforgeeks.org + - tutorialspoint.com + - javatpoint.com + - w3schools.com + - scaler.com + - simplilearn.com + - udemy.com + - coursera.org + +# ── Preference: promoted above the default rank ────────────────────────────── +# Primary sources: upstream repositories, official docs, Q&A, vendor +# engineering blogs. These are what an agent should be reading. +prefer_domains: + # upstream repositories and code hosting + - github.com + - gitlab.com + - codeberg.org + - sourceforge.net + - kernel.org + - git.kernel.org + # Q&A + - stackoverflow.com + - stackexchange.com + - superuser.com + - serverfault.com + - askubuntu.com + - discourse.org + # vendor / project documentation and forums + - proxmox.com + - forum.proxmox.com + - pve.proxmox.com + - docs.python.org + - developer.mozilla.org + - kernelnewbies.org + - man7.org + - gnu.org + - debian.org + - ubuntu.com + - redhat.com + - kernel.dk # io_uring / Jens Axboe + - github.io # project pages (docs, papers) — promoted, not authoritative by itself + # vendor engineering blogs + - anthropic.com + - openai.com + - googleblog.com + - developers.googleblog.com + - engineering.fb.com + - netflixtechblog.com + - aws.amazon.com + - cloud.google.com + - microsoft.com + - learn.microsoft.com + - apple.com + - nvidia.com + - intel.com + - amd.com + - redislabs.com + - cloudflare.com + - langchain.com + - jetbrains.com + - cursor.com + # community discussion with high signal + - news.ycombinator.com + - lobste.rs + - reddit.com + +# ── Ranking weights ────────────────────────────────────────────────────────── +# Final score = engine_score - demote_penalty + prefer_bonus, then a stable +# tiebreak on original position so ordering is reproducible run to run. +ranking: + demote_penalty: 1000 + prefer_bonus: 100 + # Results that several engines independently returned are more likely real. + multi_engine_bonus: 25 + # Shallow paths (e.g. /blog/x) are slightly less likely to be primary docs. + host_root_allowed_when_preferred: true + +# ── Extraction budget (criterion 4) ────────────────────────────────────────── +# Return CONTENT, not just links, so an agent gets usable material in ONE call. +extraction: + top_n: 5 # how many results get page text extracted + total_chars: 12000 # global budget across all extracted items + per_item_chars: 4000 # cap for any single item, so one page cannot eat the budget + timeout_seconds: 45 # per scrape + # If extraction fails, the result is still returned with an empty excerpt — + # a link is better than nothing, but the failure is recorded in the output. + on_failure: keep_with_empty_excerpt diff --git a/scripts/search-agent-consume.py b/scripts/search-agent-consume.py new file mode 100755 index 0000000..0011e00 --- /dev/null +++ b/scripts/search-agent-consume.py @@ -0,0 +1,395 @@ +#!/usr/bin/env python3 +"""Agent-consumption layer in front of SearXNG + Firecrawl. + +Multi-engine aggregation returns results with no dedupe, no filtering and no +reranking. Measured 2026-09-26 that put bestbuy.com and merriam-webster.com into +"best practices agent context management", and put four SEO blogs ABOVE the real +Proxmox forum threads on a precise technical query. Identical queries also ranked +differently between runs, so the fix has to be deterministic rather than +dependent on engine mood. + +This module turns the raw result list into something an agent can actually use: + + 1. DEDUPE the same page arriving from several engines + 2. DROP clear non-answers (homepages, shopping, dictionaries, logins) + 3. DEMOTE config-listed low-authority hosts; PROMOTE primary sources + 4. STABLE SORT so ordering is reproducible run to run + 5. EXTRACT page text for the top N under an explicit character budget, + so one call returns usable material instead of a snippet + 6. EMIT stable JSON with engine provenance + +Policy lives in config/search-ranking.yaml, not in this file. + +Usage: + search-agent-consume.py "query text" # JSON to stdout + search-agent-consume.py --no-extract "query" # ranking only, no Firecrawl + search-agent-consume.py --explain "query" # include drop/demote reasons + +Exit: 0 ok, 1 no results survived filtering, 2 the layer could not run. +""" + +from __future__ import annotations + +import json +import os +import sys +import time +import urllib.parse +import urllib.request +from pathlib import Path + +SEARXNG_URL = os.environ.get("SEARXNG_URL", "http://192.168.68.7:8888").rstrip("/") +FIRECRAWL_URL = os.environ.get("FIRECRAWL_URL", "http://192.168.68.7:3002").rstrip("/") +CONFIG_PATH = os.environ.get( + "SEARCH_RANKING_CONFIG", + str(Path(__file__).resolve().parent.parent / "config" / "search-ranking.yaml"), +) +HTTP_TIMEOUT = float(os.environ.get("SEARCH_CONSUME_TIMEOUT", "25")) + + +def _load_config() -> dict: + """Load the ranking policy. + + PyYAML is used when present; otherwise a tiny built-in parser handles the + flat lists in this specific file, so the layer never hard-fails on a host + without PyYAML. + """ + text = Path(CONFIG_PATH).read_text() + try: + import yaml # type: ignore + + return yaml.safe_load(text) + except ImportError: + return _parse_flat_yaml(text) + + +def _parse_flat_yaml(text: str) -> dict: + """Minimal fallback parser: top-level keys, nested one level, flat lists.""" + import re + + out: dict = {} + stack: list[tuple[int, dict]] = [(-1, out)] + section: dict | None = None + for raw in text.splitlines(): + line = raw.split("#", 1)[0].rstrip() + if not line.strip(): + continue + indent = len(line) - len(line.lstrip()) + body = line.strip() + if body.startswith("- "): + if section is not None: + section.setdefault("_list", []).append( + body[2:].strip().strip("'\"") + ) + continue + if ":" in body: + key, _, val = body.partition(":") + key, val = key.strip(), val.strip() + if val: + # write to the INNERMOST open section, not the document root + stack[-1][1][key] = _scalar(val) + section = None + else: + while stack and indent <= stack[-1][0]: + stack.pop() + parent = stack[-1][1] + new: dict = {} + parent[key] = new + stack.append((indent, new)) + section = new + # flatten "_list" holders back into their parent as plain lists + def fix(node): + if isinstance(node, dict): + if set(node.keys()) == {"_list"}: + return node["_list"] + return {k: fix(v) for k, v in node.items()} + return node + + return fix(out) + + +def _scalar(v: str): + if v.lower() in ("true", "false"): + return v.lower() == "true" + try: + return int(v) + except ValueError: + pass + try: + return float(v) + except ValueError: + pass + return v.strip("'\"") + + +# ── filtering ──────────────────────────────────────────────────────────────── + + +def _host(url: str) -> str: + return (urllib.parse.urlparse(url).netloc or "").lower().split(":")[0] + + +def _registrable(host: str) -> str: + """Best-effort registrable domain so sub.forum.proxmox.com matches proxmox.com.""" + parts = host.split(".") + if len(parts) <= 2: + return host + # handle common two-label public suffixes + two = ".".join(parts[-2:]) + if parts[-2] in ("co", "com", "org", "net", "ac", "gov") and len(parts) >= 3: + return ".".join(parts[-3:]) + return two + + +def _host_in(host: str, domains) -> bool: + if not domains: + return False + reg = _registrable(host) + for d in domains: + d = str(d).lower() + if host == d or host.endswith("." + d) or reg == d: + return True + return False + + +def _normalise_url(url: str) -> str: + """Strip tracking params and fragments so the same page dedupes.""" + p = urllib.parse.urlparse(url) + q = [ + (k, v) + for k, v in urllib.parse.parse_qsl(p.query, keep_blank_values=True) + if not k.lower().startswith(("utm_", "fbclid", "gclid", "mc_", "ref")) + ] + path = p.path.rstrip("/") or "/" + return urllib.parse.urlunparse( + (p.scheme.lower(), p.netloc.lower(), path, "", urllib.parse.urlencode(q), "") + ) + + +def non_answer_reason(result: dict, cfg: dict) -> str | None: + """Return why this result is a non-answer, or None if it may be returned.""" + na = cfg.get("non_answer", {}) or {} + url = result.get("url", "") + p = urllib.parse.urlparse(url) + host = _host(url) + path = p.path or "" + + if _host_in(host, na.get("hosts")): + return "shopping_or_dictionary_host" + + if na.get("host_root", True) and path in ("", "/"): + # A preferred host's front door may legitimately be the answer + # (a repo, a docs site). Everything else is navigational. + if not _host_in(host, cfg.get("prefer_domains")): + return "navigational_host_root" + + low = url.lower() + for pat in na.get("path_patterns", []) or []: + if pat.lower() in low: + return f"path_pattern:{pat}" + + qkeys = {k.lower() for k in (na.get("query_keys") or [])} + if qkeys & {k.lower() for k, _ in urllib.parse.parse_qsl(p.query)}: + return "search_or_shopping_query" + + return None + + +def source_type(url: str, cfg: dict) -> str: + host = _host(url) + if _host_in(host, ["github.com", "gitlab.com", "codeberg.org", "sourceforge.net"]): + return "code" + if _host_in(host, ["stackoverflow.com", "stackexchange.com", "superuser.com", + "serverfault.com", "askubuntu.com"]): + return "qa" + if _host_in(host, ["forum.proxmox.com", "forum.", "discourse"]) or "forum." in host: + return "forum" + if _host_in(host, ["news.ycombinator.com", "lobste.rs", "reddit.com"]): + return "discussion" + if _host_in(host, cfg.get("prefer_domains")): + return "official" + if _host_in(host, cfg.get("demote_domains")): + return "content-farm" + return "web" + + +def rank(results: list[dict], cfg: dict) -> tuple[list[dict], list[dict]]: + """Dedupe, drop non-answers, demote/ promote, stable sort. + + Returns (kept, dropped) where dropped carries the reason, because a filter + nobody can audit is a filter nobody should trust. + """ + rank_cfg = cfg.get("ranking", {}) or {} + demote_pen = float(rank_cfg.get("demote_penalty", 1000)) + prefer_bonus = float(rank_cfg.get("prefer_bonus", 100)) + multi_bonus = float(rank_cfg.get("multi_engine_bonus", 25)) + + seen: dict[str, dict] = {} + dropped: list[dict] = [] + + for pos, r in enumerate(results): + url = r.get("url") + if not url: + continue + key = _normalise_url(url) + engine = r.get("engine", "?") + + # 1. dedupe: same normalised URL from several engines + if key in seen: + seen[key].setdefault("engines", []).append(engine) + seen[key]["duplicate_of"] = True + continue + + reason = non_answer_reason(r, cfg) + if reason: + dropped.append({"url": url, "reason": reason, "position": pos + 1}) + continue + + seen[key] = { + "title": (r.get("title") or "").strip(), + "url": url, + "engines": [engine], + "position": pos, + "score": 0.0, + } + + kept = [] + for item in seen.values(): + host = _host(item["url"]) + score = -float(item["position"]) # original order is the base signal + if _host_in(host, cfg.get("demote_domains")): + score -= demote_pen + if _host_in(host, cfg.get("prefer_domains")): + score += prefer_bonus + if len(item["engines"]) > 1: + score += multi_bonus * (len(item["engines"]) - 1) + item["score"] = round(score, 2) + item["host"] = host + item["source_type"] = source_type(item["url"], cfg) + kept.append(item) + + # stable: score desc, then original position asc => reproducible run to run + kept.sort(key=lambda i: (-i["score"], i["position"])) + return kept, dropped + + +# ── extraction ─────────────────────────────────────────────────────────────── + + +def _post_json(url: str, payload: dict, timeout: float) -> dict: + req = urllib.request.Request( + url, + data=json.dumps(payload).encode(), + headers={"Content-Type": "application/json"}, + ) + with urllib.request.urlopen(req, timeout=timeout) as resp: + return json.loads(resp.read().decode("utf-8", "replace")) + + +def extract(items: list[dict], cfg: dict) -> dict: + """Fetch page text for the top N under a global character budget.""" + ex = cfg.get("extraction", {}) or {} + top_n = int(ex.get("top_n", 5)) + total_budget = int(ex.get("total_chars", 12000)) + per_item = int(ex.get("per_item_chars", 4000)) + timeout = float(ex.get("timeout_seconds", 45)) + + used = 0 + failures = 0 + t0 = time.time() + for item in items[:top_n]: + remaining = total_budget - used + if remaining <= 200: + item["excerpt"] = "" + item["extraction"] = "skipped_budget_exhausted" + continue + cap = min(per_item, remaining) + try: + data = _post_json( + f"{FIRECRAWL_URL}/v1/scrape", + {"url": item["url"], "formats": ["markdown"]}, + timeout, + ) + md = ((data.get("data") or {}).get("markdown") or "").strip() + if not md: + item["excerpt"] = "" + item["extraction"] = "empty" + failures += 1 + continue + item["excerpt"] = md[:cap] + item["extraction"] = "ok" if len(md) <= cap else "truncated" + used += len(item["excerpt"]) + except Exception as exc: # noqa: BLE001 + item["excerpt"] = "" + item["extraction"] = f"failed:{type(exc).__name__}" + failures += 1 + return { + "extracted": min(top_n, len(items)), + "chars_used": used, + "budget": total_budget, + "failures": failures, + "seconds": round(time.time() - t0, 2), + } + + +# ── entry point ────────────────────────────────────────────────────────────── + + +def consume(query: str, do_extract: bool = True, explain: bool = False) -> dict: + cfg = _load_config() + url = f"{SEARXNG_URL}/search?" + urllib.parse.urlencode( + {"q": query, "format": "json"} + ) + with urllib.request.urlopen(url, timeout=HTTP_TIMEOUT) as resp: + raw = json.loads(resp.read().decode("utf-8", "replace")) + + results = raw.get("results", []) + kept, dropped = rank(results, cfg) + extraction = extract(kept, cfg) if do_extract else None + + out = { + "query": query, + "raw_result_count": len(results), + "returned_count": len(kept), + "dropped_count": len(dropped), + "engines": sorted({r.get("engine", "?") for r in results}), + "results": [ + { + "rank": i + 1, + "title": it["title"], + "url": it["url"], + "host": it["host"], + "source_type": it["source_type"], + "engines": sorted(set(it["engines"])), + "score": it["score"], + "excerpt": it.get("excerpt", ""), + "extraction": it.get("extraction", "not_attempted"), + } + for i, it in enumerate(kept) + ], + "extraction": extraction, + } + if explain: + out["dropped"] = dropped + return out + + +def main() -> int: + args = [a for a in sys.argv[1:] if not a.startswith("--")] + do_extract = "--no-extract" not in sys.argv + explain = "--explain" in sys.argv + if not args: + print(__doc__) + return 2 + query = " ".join(args) + try: + out = consume(query, do_extract=do_extract, explain=explain) + except Exception as exc: # noqa: BLE001 + print(f"LAYER FAILED: {type(exc).__name__}: {exc}", file=sys.stderr) + return 2 + print(json.dumps(out, indent=2)) + return 0 if out["returned_count"] else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/search-agent-consumption.prose.md b/search-agent-consumption.prose.md new file mode 100644 index 0000000..8063430 --- /dev/null +++ b/search-agent-consumption.prose.md @@ -0,0 +1,137 @@ +--- +kind: function +name: search-agent-consumption +description: > + Agent-consumption layer in front of SearXNG + Firecrawl. Raw multi-engine + aggregation returns results with no dedupe, no filtering and no reranking; + measured 2026-09-26 that put bestbuy.com and merriam-webster.com into "best + practices agent context management", and put four SEO blogs above the real + Proxmox forum threads on a precise technical query. Identical queries also + ranked DIFFERENTLY between runs, which is why the layer is deterministic + rather than dependent on engine behaviour. + + Pipeline: dedupe -> drop non-answers -> demote content farms / promote primary + sources -> stable sort -> extract page text for the top N under an explicit + character budget -> stable JSON. Policy lives in config, not code. + + Call it when an agent needs search RESULTS rather than links: it returns usable + page text in one call instead of a snippet plus a second fetch. + +version: 1.0.0 +--- + +## Where the policy lives + +`config/search-ranking.yaml` — reviewable, no code change needed to adjust: + +| key | effect | +| --- | --- | +| `non_answer.hosts` / `path_patterns` / `query_keys` / `host_root` | dropped outright | +| `demote_domains` | ranked below everything, never dropped | +| `prefer_domains` | promoted above default rank | +| `ranking.*` | `demote_penalty`, `prefer_bonus`, `multi_engine_bonus` | +| `extraction.*` | `top_n`, `total_chars`, `per_item_chars`, `timeout_seconds` | + +**Demotion, not deletion, for content farms**: a genuinely useful hit is not lost, +it simply cannot outrank a primary source. Non-answers are dropped because they +cannot answer a question at all. + +## Usage + +```bash +python3 scripts/search-agent-consume.py "query text" # JSON +python3 scripts/search-agent-consume.py --no-extract "query" # ranking only +python3 scripts/search-agent-consume.py --explain "query" # + drop reasons +``` + +Exit `0` ok, `1` nothing survived filtering, `2` the layer could not run. + +## Output shape + +Stable JSON: + +```json +{ + "query": "...", + "raw_result_count": 46, + "returned_count": 44, + "dropped_count": 2, + "engines": ["bing", "brave", "duckduckgo", "yandex"], + "results": [ + {"rank": 1, "title": "...", "url": "...", "host": "...", + "source_type": "official|code|qa|forum|discussion|web|content-farm", + "engines": ["bing"], "score": 100.0, + "excerpt": "...", "extraction": "ok|truncated|skipped_budget_exhausted|empty|failed:"} + ], + "extraction": {"extracted": 5, "chars_used": 12000, "budget": 12000, + "failures": 0, "seconds": 5.28} +} +``` + +`--explain` adds `dropped: [{url, reason, position}]` so the filter is auditable +rather than magic. + +## Measured before/after (2026-09-26) + +Fixed query set. Relevance judged per query, not by impression. + +**`best practices agent context management`** + +| | before (raw SearXNG) | after (layer) | +| --- | --- | --- | +| 1-2 | anthropic, stackai | anthropic, langchain | +| 3-4 | aitechmonk, agentic-design | jetbrains, blog.jetbrains | +| 5-6 | mindstudio, sparkco | docs.langchain, reddit | +| 7-8 | langchain, medium | cursor, reddit | +| verdict | 4 relevant of 10; 4 content farms; medium.com twice | top 8 all primary/discussion; no content farm in the top 8 | + +**`proxmox thin pool metadata exhaustion recovery`** + +| | before | after | +| --- | --- | --- | +| 1-4 | vormox, linuxoperatingsystem, riparazioneserver, bigiron (all SEO/thin) | forum.proxmox.com, forum.proxmox.com, gist.github, github | +| 5-9 | forum.proxmox.com x2, voxfor, github, gist | forum.proxmox.com, serverfault, forum.proxmox.com, reddit | + +The primary sources moved from positions 5-9 to 1-4. + +**Rule proof** (`--explain`, and a direct check of the classifier): + +``` +DigitalOcean docs -> KEEP (a '/products/' path rule was REMOVED after the + before/after run caught it dropping this page) +Best Buy -> DROP shopping_or_dictionary_host +Merriam-Webster -> DROP shopping_or_dictionary_host +bare homepage -> DROP navigational_host_root +proxmox.com home -> KEEP (preferred host root: a repo/docs front door is + legitimately the answer) +github repo -> KEEP +``` + +**Extraction cost (criterion 4):** + +``` +extracted 5 items, 12000 chars used of 12000 budget, 0 failures, 5.28s +whole run end-to-end: 6.4s wall +``` + +## Regression guard + +`search-stack-visibility` asserts the layer still ranks correctly: for the fixed +query set, no `demote_domains` host may appear in the top 3, and the two known +non-answers must not be returned. Without it this layer could silently rot back +to raw ordering, which is exactly what happened to the endpoint colours. + +## Reachability, and one honest gap + +- **Hermes agents** reach it directly: it reads the same `SEARXNG_URL` and + `FIRECRAWL_URL` they already use. +- **pi agents (MCP search server)**: the MCP server's request/response shape is + **not ours to change**, so this layer is **NOT** wired into it. That is a real + gap, stated rather than claimed as coverage. Closing it would require a change + on the MCP side, which is outside this repo. + +## Constraints + +Does not touch the live SearXNG or Firecrawl service paths. Third-party +`google cse` is not a hard requirement of this layer — if it 429s, ranking still +works from the remaining engines. No credential is added or required.