Raw multi-engine aggregation had no dedupe, no filtering and no reranking.
Measured 2026-09-26: 'best practices agent context management' returned
bestbuy.com and merriam-webster.com, plus 4 content farms, with medium.com twice;
'proxmox thin pool metadata exhaustion recovery' put four SEO blogs ABOVE the
real Proxmox forum threads. Identical queries also ranked DIFFERENTLY between
runs, which is why the fix is deterministic rather than trusting the engines.
scripts/search-agent-consume.py:
1. dedupe by normalised URL (tracking params and fragments stripped)
2. drop non-answers - shopping/dictionary hosts, navigational host roots,
search/shopping/cart/login paths and query keys
3. demote content farms and promote primary sources
4. STABLE sort (score desc, then original position) so runs are reproducible
5. extract page text for the top N via Firecrawl POST /v1/scrape under an
explicit character budget, so an agent gets usable material in ONE call
6. emit stable JSON with engine provenance and source_type
Policy is config, not code: config/search-ranking.yaml holds demote_domains,
prefer_domains, non_answer rules and the extraction budget, so it is reviewable
and changeable without touching the module. Content farms are DEMOTED rather
than dropped so a useful hit is not lost, it just cannot outrank a primary.
A '/products/' path rule was REMOVED after the before/after run caught it
dropping docs.digitalocean.com/products/inference/... - a legitimate docs page.
Shopping is caught by the host list instead, which has no such false positive.
Measured: 'best practices...' top 8 becomes anthropic, langchain, jetbrains,
blog.jetbrains, docs.langchain, reddit, cursor, reddit - no content farm.
'proxmox thin pool...' moves the forum threads from positions 5-9 to 1-4.
Extraction: 5 items, 12000 chars of 12000 budget, 0 failures, 5.28s; whole run
6.4s wall.
Contract: search-agent-consumption.prose.md, including the honest reachability
gap - the pi MCP search server's shape is not ours to change, so this layer is
NOT wired into it.
396 lines
14 KiB
Python
Executable File
396 lines
14 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""Agent-consumption layer in front of SearXNG + Firecrawl.
|
|
|
|
Multi-engine aggregation returns results with no dedupe, no filtering and no
|
|
reranking. Measured 2026-09-26 that put bestbuy.com and merriam-webster.com into
|
|
"best practices agent context management", and put four SEO blogs ABOVE the real
|
|
Proxmox forum threads on a precise technical query. Identical queries also ranked
|
|
differently between runs, so the fix has to be deterministic rather than
|
|
dependent on engine mood.
|
|
|
|
This module turns the raw result list into something an agent can actually use:
|
|
|
|
1. DEDUPE the same page arriving from several engines
|
|
2. DROP clear non-answers (homepages, shopping, dictionaries, logins)
|
|
3. DEMOTE config-listed low-authority hosts; PROMOTE primary sources
|
|
4. STABLE SORT so ordering is reproducible run to run
|
|
5. EXTRACT page text for the top N under an explicit character budget,
|
|
so one call returns usable material instead of a snippet
|
|
6. EMIT stable JSON with engine provenance
|
|
|
|
Policy lives in config/search-ranking.yaml, not in this file.
|
|
|
|
Usage:
|
|
search-agent-consume.py "query text" # JSON to stdout
|
|
search-agent-consume.py --no-extract "query" # ranking only, no Firecrawl
|
|
search-agent-consume.py --explain "query" # include drop/demote reasons
|
|
|
|
Exit: 0 ok, 1 no results survived filtering, 2 the layer could not run.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import os
|
|
import sys
|
|
import time
|
|
import urllib.parse
|
|
import urllib.request
|
|
from pathlib import Path
|
|
|
|
SEARXNG_URL = os.environ.get("SEARXNG_URL", "http://192.168.68.7:8888").rstrip("/")
|
|
FIRECRAWL_URL = os.environ.get("FIRECRAWL_URL", "http://192.168.68.7:3002").rstrip("/")
|
|
CONFIG_PATH = os.environ.get(
|
|
"SEARCH_RANKING_CONFIG",
|
|
str(Path(__file__).resolve().parent.parent / "config" / "search-ranking.yaml"),
|
|
)
|
|
HTTP_TIMEOUT = float(os.environ.get("SEARCH_CONSUME_TIMEOUT", "25"))
|
|
|
|
|
|
def _load_config() -> dict:
|
|
"""Load the ranking policy.
|
|
|
|
PyYAML is used when present; otherwise a tiny built-in parser handles the
|
|
flat lists in this specific file, so the layer never hard-fails on a host
|
|
without PyYAML.
|
|
"""
|
|
text = Path(CONFIG_PATH).read_text()
|
|
try:
|
|
import yaml # type: ignore
|
|
|
|
return yaml.safe_load(text)
|
|
except ImportError:
|
|
return _parse_flat_yaml(text)
|
|
|
|
|
|
def _parse_flat_yaml(text: str) -> dict:
|
|
"""Minimal fallback parser: top-level keys, nested one level, flat lists."""
|
|
import re
|
|
|
|
out: dict = {}
|
|
stack: list[tuple[int, dict]] = [(-1, out)]
|
|
section: dict | None = None
|
|
for raw in text.splitlines():
|
|
line = raw.split("#", 1)[0].rstrip()
|
|
if not line.strip():
|
|
continue
|
|
indent = len(line) - len(line.lstrip())
|
|
body = line.strip()
|
|
if body.startswith("- "):
|
|
if section is not None:
|
|
section.setdefault("_list", []).append(
|
|
body[2:].strip().strip("'\"")
|
|
)
|
|
continue
|
|
if ":" in body:
|
|
key, _, val = body.partition(":")
|
|
key, val = key.strip(), val.strip()
|
|
if val:
|
|
# write to the INNERMOST open section, not the document root
|
|
stack[-1][1][key] = _scalar(val)
|
|
section = None
|
|
else:
|
|
while stack and indent <= stack[-1][0]:
|
|
stack.pop()
|
|
parent = stack[-1][1]
|
|
new: dict = {}
|
|
parent[key] = new
|
|
stack.append((indent, new))
|
|
section = new
|
|
# flatten "_list" holders back into their parent as plain lists
|
|
def fix(node):
|
|
if isinstance(node, dict):
|
|
if set(node.keys()) == {"_list"}:
|
|
return node["_list"]
|
|
return {k: fix(v) for k, v in node.items()}
|
|
return node
|
|
|
|
return fix(out)
|
|
|
|
|
|
def _scalar(v: str):
|
|
if v.lower() in ("true", "false"):
|
|
return v.lower() == "true"
|
|
try:
|
|
return int(v)
|
|
except ValueError:
|
|
pass
|
|
try:
|
|
return float(v)
|
|
except ValueError:
|
|
pass
|
|
return v.strip("'\"")
|
|
|
|
|
|
# ── filtering ────────────────────────────────────────────────────────────────
|
|
|
|
|
|
def _host(url: str) -> str:
|
|
return (urllib.parse.urlparse(url).netloc or "").lower().split(":")[0]
|
|
|
|
|
|
def _registrable(host: str) -> str:
|
|
"""Best-effort registrable domain so sub.forum.proxmox.com matches proxmox.com."""
|
|
parts = host.split(".")
|
|
if len(parts) <= 2:
|
|
return host
|
|
# handle common two-label public suffixes
|
|
two = ".".join(parts[-2:])
|
|
if parts[-2] in ("co", "com", "org", "net", "ac", "gov") and len(parts) >= 3:
|
|
return ".".join(parts[-3:])
|
|
return two
|
|
|
|
|
|
def _host_in(host: str, domains) -> bool:
|
|
if not domains:
|
|
return False
|
|
reg = _registrable(host)
|
|
for d in domains:
|
|
d = str(d).lower()
|
|
if host == d or host.endswith("." + d) or reg == d:
|
|
return True
|
|
return False
|
|
|
|
|
|
def _normalise_url(url: str) -> str:
|
|
"""Strip tracking params and fragments so the same page dedupes."""
|
|
p = urllib.parse.urlparse(url)
|
|
q = [
|
|
(k, v)
|
|
for k, v in urllib.parse.parse_qsl(p.query, keep_blank_values=True)
|
|
if not k.lower().startswith(("utm_", "fbclid", "gclid", "mc_", "ref"))
|
|
]
|
|
path = p.path.rstrip("/") or "/"
|
|
return urllib.parse.urlunparse(
|
|
(p.scheme.lower(), p.netloc.lower(), path, "", urllib.parse.urlencode(q), "")
|
|
)
|
|
|
|
|
|
def non_answer_reason(result: dict, cfg: dict) -> str | None:
|
|
"""Return why this result is a non-answer, or None if it may be returned."""
|
|
na = cfg.get("non_answer", {}) or {}
|
|
url = result.get("url", "")
|
|
p = urllib.parse.urlparse(url)
|
|
host = _host(url)
|
|
path = p.path or ""
|
|
|
|
if _host_in(host, na.get("hosts")):
|
|
return "shopping_or_dictionary_host"
|
|
|
|
if na.get("host_root", True) and path in ("", "/"):
|
|
# A preferred host's front door may legitimately be the answer
|
|
# (a repo, a docs site). Everything else is navigational.
|
|
if not _host_in(host, cfg.get("prefer_domains")):
|
|
return "navigational_host_root"
|
|
|
|
low = url.lower()
|
|
for pat in na.get("path_patterns", []) or []:
|
|
if pat.lower() in low:
|
|
return f"path_pattern:{pat}"
|
|
|
|
qkeys = {k.lower() for k in (na.get("query_keys") or [])}
|
|
if qkeys & {k.lower() for k, _ in urllib.parse.parse_qsl(p.query)}:
|
|
return "search_or_shopping_query"
|
|
|
|
return None
|
|
|
|
|
|
def source_type(url: str, cfg: dict) -> str:
|
|
host = _host(url)
|
|
if _host_in(host, ["github.com", "gitlab.com", "codeberg.org", "sourceforge.net"]):
|
|
return "code"
|
|
if _host_in(host, ["stackoverflow.com", "stackexchange.com", "superuser.com",
|
|
"serverfault.com", "askubuntu.com"]):
|
|
return "qa"
|
|
if _host_in(host, ["forum.proxmox.com", "forum.", "discourse"]) or "forum." in host:
|
|
return "forum"
|
|
if _host_in(host, ["news.ycombinator.com", "lobste.rs", "reddit.com"]):
|
|
return "discussion"
|
|
if _host_in(host, cfg.get("prefer_domains")):
|
|
return "official"
|
|
if _host_in(host, cfg.get("demote_domains")):
|
|
return "content-farm"
|
|
return "web"
|
|
|
|
|
|
def rank(results: list[dict], cfg: dict) -> tuple[list[dict], list[dict]]:
|
|
"""Dedupe, drop non-answers, demote/ promote, stable sort.
|
|
|
|
Returns (kept, dropped) where dropped carries the reason, because a filter
|
|
nobody can audit is a filter nobody should trust.
|
|
"""
|
|
rank_cfg = cfg.get("ranking", {}) or {}
|
|
demote_pen = float(rank_cfg.get("demote_penalty", 1000))
|
|
prefer_bonus = float(rank_cfg.get("prefer_bonus", 100))
|
|
multi_bonus = float(rank_cfg.get("multi_engine_bonus", 25))
|
|
|
|
seen: dict[str, dict] = {}
|
|
dropped: list[dict] = []
|
|
|
|
for pos, r in enumerate(results):
|
|
url = r.get("url")
|
|
if not url:
|
|
continue
|
|
key = _normalise_url(url)
|
|
engine = r.get("engine", "?")
|
|
|
|
# 1. dedupe: same normalised URL from several engines
|
|
if key in seen:
|
|
seen[key].setdefault("engines", []).append(engine)
|
|
seen[key]["duplicate_of"] = True
|
|
continue
|
|
|
|
reason = non_answer_reason(r, cfg)
|
|
if reason:
|
|
dropped.append({"url": url, "reason": reason, "position": pos + 1})
|
|
continue
|
|
|
|
seen[key] = {
|
|
"title": (r.get("title") or "").strip(),
|
|
"url": url,
|
|
"engines": [engine],
|
|
"position": pos,
|
|
"score": 0.0,
|
|
}
|
|
|
|
kept = []
|
|
for item in seen.values():
|
|
host = _host(item["url"])
|
|
score = -float(item["position"]) # original order is the base signal
|
|
if _host_in(host, cfg.get("demote_domains")):
|
|
score -= demote_pen
|
|
if _host_in(host, cfg.get("prefer_domains")):
|
|
score += prefer_bonus
|
|
if len(item["engines"]) > 1:
|
|
score += multi_bonus * (len(item["engines"]) - 1)
|
|
item["score"] = round(score, 2)
|
|
item["host"] = host
|
|
item["source_type"] = source_type(item["url"], cfg)
|
|
kept.append(item)
|
|
|
|
# stable: score desc, then original position asc => reproducible run to run
|
|
kept.sort(key=lambda i: (-i["score"], i["position"]))
|
|
return kept, dropped
|
|
|
|
|
|
# ── extraction ───────────────────────────────────────────────────────────────
|
|
|
|
|
|
def _post_json(url: str, payload: dict, timeout: float) -> dict:
|
|
req = urllib.request.Request(
|
|
url,
|
|
data=json.dumps(payload).encode(),
|
|
headers={"Content-Type": "application/json"},
|
|
)
|
|
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
|
return json.loads(resp.read().decode("utf-8", "replace"))
|
|
|
|
|
|
def extract(items: list[dict], cfg: dict) -> dict:
|
|
"""Fetch page text for the top N under a global character budget."""
|
|
ex = cfg.get("extraction", {}) or {}
|
|
top_n = int(ex.get("top_n", 5))
|
|
total_budget = int(ex.get("total_chars", 12000))
|
|
per_item = int(ex.get("per_item_chars", 4000))
|
|
timeout = float(ex.get("timeout_seconds", 45))
|
|
|
|
used = 0
|
|
failures = 0
|
|
t0 = time.time()
|
|
for item in items[:top_n]:
|
|
remaining = total_budget - used
|
|
if remaining <= 200:
|
|
item["excerpt"] = ""
|
|
item["extraction"] = "skipped_budget_exhausted"
|
|
continue
|
|
cap = min(per_item, remaining)
|
|
try:
|
|
data = _post_json(
|
|
f"{FIRECRAWL_URL}/v1/scrape",
|
|
{"url": item["url"], "formats": ["markdown"]},
|
|
timeout,
|
|
)
|
|
md = ((data.get("data") or {}).get("markdown") or "").strip()
|
|
if not md:
|
|
item["excerpt"] = ""
|
|
item["extraction"] = "empty"
|
|
failures += 1
|
|
continue
|
|
item["excerpt"] = md[:cap]
|
|
item["extraction"] = "ok" if len(md) <= cap else "truncated"
|
|
used += len(item["excerpt"])
|
|
except Exception as exc: # noqa: BLE001
|
|
item["excerpt"] = ""
|
|
item["extraction"] = f"failed:{type(exc).__name__}"
|
|
failures += 1
|
|
return {
|
|
"extracted": min(top_n, len(items)),
|
|
"chars_used": used,
|
|
"budget": total_budget,
|
|
"failures": failures,
|
|
"seconds": round(time.time() - t0, 2),
|
|
}
|
|
|
|
|
|
# ── entry point ──────────────────────────────────────────────────────────────
|
|
|
|
|
|
def consume(query: str, do_extract: bool = True, explain: bool = False) -> dict:
|
|
cfg = _load_config()
|
|
url = f"{SEARXNG_URL}/search?" + urllib.parse.urlencode(
|
|
{"q": query, "format": "json"}
|
|
)
|
|
with urllib.request.urlopen(url, timeout=HTTP_TIMEOUT) as resp:
|
|
raw = json.loads(resp.read().decode("utf-8", "replace"))
|
|
|
|
results = raw.get("results", [])
|
|
kept, dropped = rank(results, cfg)
|
|
extraction = extract(kept, cfg) if do_extract else None
|
|
|
|
out = {
|
|
"query": query,
|
|
"raw_result_count": len(results),
|
|
"returned_count": len(kept),
|
|
"dropped_count": len(dropped),
|
|
"engines": sorted({r.get("engine", "?") for r in results}),
|
|
"results": [
|
|
{
|
|
"rank": i + 1,
|
|
"title": it["title"],
|
|
"url": it["url"],
|
|
"host": it["host"],
|
|
"source_type": it["source_type"],
|
|
"engines": sorted(set(it["engines"])),
|
|
"score": it["score"],
|
|
"excerpt": it.get("excerpt", ""),
|
|
"extraction": it.get("extraction", "not_attempted"),
|
|
}
|
|
for i, it in enumerate(kept)
|
|
],
|
|
"extraction": extraction,
|
|
}
|
|
if explain:
|
|
out["dropped"] = dropped
|
|
return out
|
|
|
|
|
|
def main() -> int:
|
|
args = [a for a in sys.argv[1:] if not a.startswith("--")]
|
|
do_extract = "--no-extract" not in sys.argv
|
|
explain = "--explain" in sys.argv
|
|
if not args:
|
|
print(__doc__)
|
|
return 2
|
|
query = " ".join(args)
|
|
try:
|
|
out = consume(query, do_extract=do_extract, explain=explain)
|
|
except Exception as exc: # noqa: BLE001
|
|
print(f"LAYER FAILED: {type(exc).__name__}: {exc}", file=sys.stderr)
|
|
return 2
|
|
print(json.dumps(out, indent=2))
|
|
return 0 if out["returned_count"] else 1
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|