PR Pipeline — Authorize → Validate → Review → Merge / auth (pull_request) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / validate (pull_request) Successful in 4s
PR Pipeline — Authorize → Validate → Review → Merge / lint (pull_request) Successful in 13s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (pull_request) Successful in 11s
PR Pipeline — Authorize → Validate → Review → Merge / gate (pull_request) Successful in 1s
Extends search-stack-visibility with a ranking assertion: for the fixed query set, no config demote_domains host may appear in the top 3, and the known non-answers (bestbuy.com, merriam-webster.com) must not be returned at all. Without this the layer could rot back to raw engine ordering unnoticed - the same way the endpoint colours silently rotted before 2026-09-26. It reads the demote list from the SAME config the layer uses, so the guard cannot drift from the policy it is guarding. Live: 'ok: no demoted host in the top 3; no banned non-answer returned'; visibility contract still PASSES end to end.
314 lines
12 KiB
Python
Executable File
314 lines
12 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""Search-stack visibility check.
|
|
|
|
The fleet shares one SearXNG instance (search) plus one extraction service
|
|
(Firecrawl). A broken search stack used to fail silently: one engine answered
|
|
and nobody could tell that the other engines had stopped contributing, or that
|
|
an enabled engine was returning nothing at all without reporting an error.
|
|
|
|
This check makes those failures visible and non-zero:
|
|
|
|
* runs two fixed queries against SearXNG; FAILS when fewer than two engines
|
|
contribute to a query, printing the contributing engines and every
|
|
``unresponsive_engines`` entry;
|
|
* FAILS when a known page cannot be extracted to non-empty markdown through
|
|
Firecrawl;
|
|
* reports every *silent zero* engine explicitly -- an engine that is enabled,
|
|
is eligible for the query category, is not listed in
|
|
``unresponsive_engines``, and still contributed no results.
|
|
|
|
Exit code 0 = healthy, 1 = degraded, 2 = the check could not run at all.
|
|
|
|
Environment overrides (all optional):
|
|
SEARXNG_URL default http://192.168.68.7:8888
|
|
FIRECRAWL_URL default http://192.168.68.7:3002
|
|
SEARCH_CHECK_QUERIES comma-separated fixed queries
|
|
SEARCH_CHECK_MIN_ENGINES default 2
|
|
SEARCH_CHECK_TIMEOUT per-request timeout in seconds, default 25
|
|
SEARCH_CHECK_EXTRACT_URL page used for the extraction leg
|
|
SEARCH_CHECK_ENGINES comma-separated engine names the stack is expected to
|
|
run; a silent zero is reported for any of them that is
|
|
enabled but contributes nothing with no error
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import os
|
|
import sys
|
|
import urllib.error
|
|
import urllib.parse
|
|
import urllib.request
|
|
|
|
SEARXNG_URL = os.environ.get("SEARXNG_URL", "http://192.168.68.7:8888").rstrip("/")
|
|
FIRECRAWL_URL = os.environ.get("FIRECRAWL_URL", "http://192.168.68.7:3002").rstrip("/")
|
|
QUERIES = [
|
|
q.strip()
|
|
for q in os.environ.get(
|
|
"SEARCH_CHECK_QUERIES", "proxmox backup server,python asyncio tutorial"
|
|
).split(",")
|
|
if q.strip()
|
|
]
|
|
MIN_ENGINES = int(os.environ.get("SEARCH_CHECK_MIN_ENGINES", "2"))
|
|
TIMEOUT = float(os.environ.get("SEARCH_CHECK_TIMEOUT", "25"))
|
|
EXTRACT_URL = os.environ.get(
|
|
"SEARCH_CHECK_EXTRACT_URL", "https://en.wikipedia.org/wiki/Proxmox_Virtual_Environment"
|
|
)
|
|
|
|
# The general web-search engines this stack intentionally runs. A general query
|
|
# is expected to draw on these; an enabled one that returns nothing without an
|
|
# error is the silent-zero failure this check exists to expose. Specialised
|
|
# engines (images, videos, translate, currency, arxiv, npm, ...) are excluded on
|
|
# purpose -- contributing nothing to a general query is correct for them.
|
|
DEFAULT_EXPECTED_ENGINES = [
|
|
"bing",
|
|
"brave",
|
|
"google cse",
|
|
"yandex",
|
|
"duckduckgo",
|
|
]
|
|
EXPECTED_ENGINES = [
|
|
e.strip()
|
|
for e in os.environ.get(
|
|
"SEARCH_CHECK_ENGINES", ",".join(DEFAULT_EXPECTED_ENGINES)
|
|
).split(",")
|
|
if e.strip()
|
|
]
|
|
|
|
|
|
def _get_json(url: str) -> dict:
|
|
req = urllib.request.Request(url, headers={"User-Agent": "search-stack-check/1.0"})
|
|
with urllib.request.urlopen(req, timeout=TIMEOUT) as resp:
|
|
return json.loads(resp.read().decode("utf-8", "replace"))
|
|
|
|
|
|
def _post_json(url: str, payload: dict) -> dict:
|
|
data = json.dumps(payload).encode("utf-8")
|
|
req = urllib.request.Request(
|
|
url,
|
|
data=data,
|
|
headers={
|
|
"Content-Type": "application/json",
|
|
"User-Agent": "search-stack-check/1.0",
|
|
},
|
|
)
|
|
with urllib.request.urlopen(req, timeout=TIMEOUT) as resp:
|
|
return json.loads(resp.read().decode("utf-8", "replace"))
|
|
|
|
|
|
def enabled_expected_engines() -> set[str]:
|
|
"""Expected engines that SearXNG reports as actually enabled."""
|
|
cfg = _get_json(f"{SEARXNG_URL}/config")
|
|
enabled = {e["name"] for e in cfg.get("engines", []) if e.get("enabled")}
|
|
return {name for name in EXPECTED_ENGINES if name in enabled}
|
|
|
|
|
|
def unresponsive_names(pairs: list) -> dict[str, str]:
|
|
"""``unresponsive_engines`` is a list of [name, reason] pairs (or strings)."""
|
|
out: dict[str, str] = {}
|
|
for item in pairs or []:
|
|
if isinstance(item, (list, tuple)) and len(item) >= 2:
|
|
out[str(item[0])] = str(item[1])
|
|
elif isinstance(item, str):
|
|
out[item] = "unresponsive"
|
|
return out
|
|
|
|
|
|
# ── QUALITY GUARD (search-agent-consumption) ─────────────────────────────────
|
|
# The agent-consumption layer applies a deterministic demote/drop policy. Without
|
|
# an assertion here it could silently rot back to raw engine ordering - the same
|
|
# way the endpoint colours silently rotted before 2026-09-26.
|
|
QUALITY_QUERIES = [
|
|
"best practices agent context management",
|
|
"proxmox thin pool metadata exhaustion recovery",
|
|
]
|
|
# A demoted (content-farm) host must never occupy the top 3 for these queries.
|
|
QUALITY_TOP_N = 3
|
|
# Non-answers that must never be returned for these queries at all.
|
|
QUALITY_BANNED_HOSTS = ["bestbuy.com", "merriam-webster.com"]
|
|
|
|
|
|
def _consumption_layer_path():
|
|
here = os.path.dirname(os.path.abspath(__file__))
|
|
return os.path.join(here, "search-agent-consume.py")
|
|
|
|
|
|
def check_ranking_quality() -> list[str]:
|
|
"""Return a list of quality failures; empty means healthy."""
|
|
import subprocess as _sp
|
|
|
|
layer = _consumption_layer_path()
|
|
if not os.path.exists(layer):
|
|
return [f"agent-consumption layer missing: {layer}"]
|
|
|
|
failures: list[str] = []
|
|
for query in QUALITY_QUERIES:
|
|
r = _sp.run([sys.executable, layer, "--no-extract", "--explain", query],
|
|
capture_output=True, text=True, timeout=120)
|
|
if r.returncode != 0:
|
|
failures.append(f"{query!r}: layer exited {r.returncode} ({r.stderr[:120]})")
|
|
continue
|
|
try:
|
|
data = json.loads(r.stdout)
|
|
except json.JSONDecodeError:
|
|
failures.append(f"{query!r}: layer returned unparseable JSON")
|
|
continue
|
|
|
|
results = data.get("results", [])
|
|
if len(results) < QUALITY_TOP_N:
|
|
failures.append(f"{query!r}: only {len(results)} results returned")
|
|
continue
|
|
|
|
# load the demote list from the SAME config the layer uses
|
|
cfg_path = os.path.join(os.path.dirname(layer), "..", "config", "search-ranking.yaml")
|
|
demoted: set[str] = set()
|
|
try:
|
|
sys.path.insert(0, os.path.dirname(layer))
|
|
import importlib.util as _iu
|
|
spec = _iu.spec_from_file_location("_sac_cfg", layer)
|
|
mod = _iu.module_from_spec(spec)
|
|
spec.loader.exec_module(mod)
|
|
demoted = set(mod._load_config().get("demote_domains", []) or [])
|
|
except Exception: # noqa: BLE001
|
|
failures.append(f"{query!r}: could not load demote_domains from config")
|
|
|
|
for item in results[:QUALITY_TOP_N]:
|
|
host = (item.get("host") or "")
|
|
for d in demoted:
|
|
if host == d or host.endswith("." + d):
|
|
failures.append(
|
|
f"{query!r}: demoted host {host} in top {QUALITY_TOP_N}"
|
|
)
|
|
for item in results:
|
|
host = (item.get("host") or "")
|
|
for b in QUALITY_BANNED_HOSTS:
|
|
if host == b or host.endswith("." + b):
|
|
failures.append(f"{query!r}: non-answer host {host} returned")
|
|
return failures
|
|
|
|
|
|
def main() -> int:
|
|
failures: list[str] = []
|
|
print(f"Search stack check -- {SEARXNG_URL}")
|
|
print(f"Queries: {QUERIES!r} min contributing engines: {MIN_ENGINES}")
|
|
print("=" * 72)
|
|
|
|
try:
|
|
eligible = enabled_expected_engines()
|
|
except Exception as exc: # noqa: BLE001 - report, do not traceback
|
|
print(f"FAIL: could not read /config from SearXNG: {exc!r}")
|
|
return 2
|
|
print(f"Expected engines, enabled ({len(eligible)}): {sorted(eligible)}")
|
|
missing = sorted(set(EXPECTED_ENGINES) - eligible)
|
|
if missing:
|
|
print(f"Expected engines NOT enabled: {missing}")
|
|
failures.append(f"expected engines not enabled in SearXNG: {missing}")
|
|
|
|
contributed: dict[str, int] = {name: 0 for name in eligible}
|
|
silent_zero_all: dict[str, list[str]] = {}
|
|
|
|
for query in QUERIES:
|
|
url = f"{SEARXNG_URL}/search?" + urllib.parse.urlencode(
|
|
{"q": query, "format": "json"}
|
|
)
|
|
print("-" * 72)
|
|
print(f"QUERY: {query!r}")
|
|
try:
|
|
data = _get_json(url)
|
|
except Exception as exc: # noqa: BLE001
|
|
print(f" FAIL: query request failed: {exc!r}")
|
|
failures.append(f"query {query!r} request failed: {exc!r}")
|
|
continue
|
|
|
|
results = data.get("results", [])
|
|
engines: dict[str, int] = {}
|
|
for r in results:
|
|
name = r.get("engine", "?")
|
|
engines[name] = engines.get(name, 0) + 1
|
|
unresponsive = unresponsive_names(data.get("unresponsive_engines", []))
|
|
|
|
print(f" results: {len(results)}")
|
|
print(f" contributing engines: {engines or '(none)'}")
|
|
print(f" unresponsive_engines: {unresponsive or '(none)'}")
|
|
|
|
for name in engines:
|
|
contributed[name] = contributed.get(name, 0) + engines[name]
|
|
|
|
if len(engines) < MIN_ENGINES:
|
|
msg = (
|
|
f"query {query!r} had only {len(engines)} contributing engine(s) "
|
|
f"({sorted(engines)}); need >= {MIN_ENGINES}"
|
|
)
|
|
print(f" FAIL: {msg}")
|
|
failures.append(msg)
|
|
|
|
silent = sorted(
|
|
n for n in eligible if n not in engines and n not in unresponsive
|
|
)
|
|
if silent:
|
|
silent_zero_all[query] = silent
|
|
print(
|
|
" SILENT ZERO (enabled, no error, no results -- reported, "
|
|
f"not fatal): {silent}"
|
|
)
|
|
|
|
print("=" * 72)
|
|
print("Engine contribution across all queries:")
|
|
for name in sorted(contributed):
|
|
status = "ZERO" if contributed[name] == 0 else "ok"
|
|
print(f" {name:<24} {contributed[name]:>4} {status}")
|
|
|
|
if silent_zero_all:
|
|
print("-" * 72)
|
|
print("SILENT-ZERO ENGINES REPORTED (no error raised, no results returned):")
|
|
for query, names in silent_zero_all.items():
|
|
print(f" {query!r}: {names}")
|
|
print(" NOTE: a silent zero is REPORTED, not counted as a failure. These")
|
|
print(" engines are expected to answer a general query, but contributing")
|
|
print(" nothing to one query can be legitimate (result de-duplication, or")
|
|
print(" an engine that only fires on certain query shapes). Only the")
|
|
print(f" <{MIN_ENGINES}-contributing-engine floor and the extraction leg fail the run.")
|
|
|
|
print("-" * 72)
|
|
print(f"EXTRACTION: scraping {EXTRACT_URL} via {FIRECRAWL_URL}/v1/scrape")
|
|
try:
|
|
payload = _post_json(
|
|
f"{FIRECRAWL_URL}/v1/scrape",
|
|
{"url": EXTRACT_URL, "formats": ["markdown"]},
|
|
)
|
|
markdown = ((payload.get("data") or {}).get("markdown") or "").strip()
|
|
if not markdown:
|
|
msg = "extraction returned empty markdown"
|
|
print(f" FAIL: {msg}")
|
|
failures.append(msg)
|
|
else:
|
|
print(f" ok: {len(markdown)} chars of markdown returned")
|
|
print(f" first line: {markdown.splitlines()[0][:120]!r}")
|
|
except Exception as exc: # noqa: BLE001
|
|
msg = f"extraction request failed: {exc!r}"
|
|
print(f" FAIL: {msg}")
|
|
failures.append(msg)
|
|
|
|
print("-" * 72)
|
|
print("RANKING QUALITY (agent-consumption layer)")
|
|
quality = check_ranking_quality()
|
|
if quality:
|
|
for q in quality:
|
|
print(f" FAIL: {q}")
|
|
failures.extend(quality)
|
|
else:
|
|
print(" ok: no demoted host in the top 3; no banned non-answer returned")
|
|
|
|
print("=" * 72)
|
|
if failures:
|
|
print("VERDICT: FAIL")
|
|
for f in failures:
|
|
print(f" - {f}")
|
|
return 1
|
|
print("VERDICT: PASS -- multiple engines contributing, extraction healthy")
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|