PR Pipeline — Authorize → Validate → Review → Merge / auth (pull_request) Successful in 14s
PR Pipeline — Authorize → Validate → Review → Merge / validate (pull_request) Successful in 6s
PR Pipeline — Authorize → Validate → Review → Merge / lint (pull_request) Failing after 10s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (pull_request) Successful in 10s
PR Pipeline — Authorize → Validate → Review → Merge / gate (pull_request) Skipped
The visibility contract's expected-engine set was still the 2026-09-25 list (bing, brave, google cse, yandex, duckduckgo). Three of those five are blocked upstream today, so the check had gone quiet on the engines that DO carry the stack and noisy on ones that cannot. The live stack now runs ten engines enabled: seven that returned real results from this network (bing, yandex, yep, mwmbl, naver, seznam, yahoo) plus the three best-effort canaries kept for recovery visibility (brave, duckduckgo, google cse). The expected set is updated to match, so a silent zero is reported for every engine the stack actually runs. Live proof after the engine expansion (CT 100, 2026-10-03): 'proxmox backup server' -> 7 contributing engines 'python asyncio tutorial' -> 6 contributing engines VERDICT: PASS Before the change both queries contributed from bing alone, one engine above the two-engine floor. Verified broken cases still fail and name the cause: a single-engine floor exits 1 naming the sole contributor, a broken extraction exits 1, and an unreachable SearXNG exits 2. Contract text and version updated to the 2026-10-03 state.
326 lines
12 KiB
Python
Executable File
326 lines
12 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""Search-stack visibility check.
|
|
|
|
The fleet shares one SearXNG instance (search) plus one extraction service
|
|
(Firecrawl). A broken search stack used to fail silently: one engine answered
|
|
and nobody could tell that the other engines had stopped contributing, or that
|
|
an enabled engine was returning nothing at all without reporting an error.
|
|
|
|
This check makes those failures visible and non-zero:
|
|
|
|
* runs two fixed queries against SearXNG; FAILS when fewer than two engines
|
|
contribute to a query, printing the contributing engines and every
|
|
``unresponsive_engines`` entry;
|
|
* FAILS when a known page cannot be extracted to non-empty markdown through
|
|
Firecrawl;
|
|
* reports every *silent zero* engine explicitly -- an engine that is enabled,
|
|
is eligible for the query category, is not listed in
|
|
``unresponsive_engines``, and still contributed no results.
|
|
|
|
Exit code 0 = healthy, 1 = degraded, 2 = the check could not run at all.
|
|
|
|
Environment overrides (all optional):
|
|
SEARXNG_URL default http://192.168.68.7:8888
|
|
FIRECRAWL_URL default http://192.168.68.7:3002
|
|
SEARCH_CHECK_QUERIES comma-separated fixed queries
|
|
SEARCH_CHECK_MIN_ENGINES default 2
|
|
SEARCH_CHECK_TIMEOUT per-request timeout in seconds, default 25
|
|
SEARCH_CHECK_EXTRACT_URL page used for the extraction leg
|
|
SEARCH_CHECK_ENGINES comma-separated engine names the stack is expected to
|
|
run; a silent zero is reported for any of them that is
|
|
enabled but contributes nothing with no error
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import os
|
|
import sys
|
|
import urllib.error
|
|
import urllib.parse
|
|
import urllib.request
|
|
|
|
SEARXNG_URL = os.environ.get("SEARXNG_URL", "http://192.168.68.7:8888").rstrip("/")
|
|
FIRECRAWL_URL = os.environ.get("FIRECRAWL_URL", "http://192.168.68.7:3002").rstrip("/")
|
|
QUERIES = [
|
|
q.strip()
|
|
for q in os.environ.get(
|
|
"SEARCH_CHECK_QUERIES", "proxmox backup server,python asyncio tutorial"
|
|
).split(",")
|
|
if q.strip()
|
|
]
|
|
MIN_ENGINES = int(os.environ.get("SEARCH_CHECK_MIN_ENGINES", "2"))
|
|
TIMEOUT = float(os.environ.get("SEARCH_CHECK_TIMEOUT", "25"))
|
|
EXTRACT_URL = os.environ.get(
|
|
"SEARCH_CHECK_EXTRACT_URL", "https://en.wikipedia.org/wiki/Proxmox_Virtual_Environment"
|
|
)
|
|
|
|
# The general web-search engines this stack intentionally runs. A general query
|
|
# is expected to draw on these; an enabled one that returns nothing without an
|
|
# error is the silent-zero failure this check exists to expose. Specialised
|
|
# engines (images, videos, translate, currency, arxiv, npm, ...) are excluded on
|
|
# purpose -- contributing nothing to a general query is correct for them.
|
|
DEFAULT_EXPECTED_ENGINES = [
|
|
# Multi-engine expansion 2026-10-03. The stack had fallen to Bing-only:
|
|
# brave and google cse are suspended upstream, duckduckgo CAPTCHAs both
|
|
# egresses and yandex flaps. The seven below all returned real results
|
|
# from this network and are the engines a general query must draw on.
|
|
"bing",
|
|
"yandex",
|
|
"yep",
|
|
"mwmbl",
|
|
"naver",
|
|
"seznam",
|
|
"yahoo",
|
|
# Best-effort canaries: intentionally left enabled so a recovery shows up
|
|
# as a contribution and a failure stays visible in unresponsive_engines.
|
|
# All three are blocked upstream today.
|
|
"brave",
|
|
"duckduckgo",
|
|
"google cse",
|
|
]
|
|
EXPECTED_ENGINES = [
|
|
e.strip()
|
|
for e in os.environ.get(
|
|
"SEARCH_CHECK_ENGINES", ",".join(DEFAULT_EXPECTED_ENGINES)
|
|
).split(",")
|
|
if e.strip()
|
|
]
|
|
|
|
|
|
def _get_json(url: str) -> dict:
|
|
req = urllib.request.Request(url, headers={"User-Agent": "search-stack-check/1.0"})
|
|
with urllib.request.urlopen(req, timeout=TIMEOUT) as resp:
|
|
return json.loads(resp.read().decode("utf-8", "replace"))
|
|
|
|
|
|
def _post_json(url: str, payload: dict) -> dict:
|
|
data = json.dumps(payload).encode("utf-8")
|
|
req = urllib.request.Request(
|
|
url,
|
|
data=data,
|
|
headers={
|
|
"Content-Type": "application/json",
|
|
"User-Agent": "search-stack-check/1.0",
|
|
},
|
|
)
|
|
with urllib.request.urlopen(req, timeout=TIMEOUT) as resp:
|
|
return json.loads(resp.read().decode("utf-8", "replace"))
|
|
|
|
|
|
def enabled_expected_engines() -> set[str]:
|
|
"""Expected engines that SearXNG reports as actually enabled."""
|
|
cfg = _get_json(f"{SEARXNG_URL}/config")
|
|
enabled = {e["name"] for e in cfg.get("engines", []) if e.get("enabled")}
|
|
return {name for name in EXPECTED_ENGINES if name in enabled}
|
|
|
|
|
|
def unresponsive_names(pairs: list) -> dict[str, str]:
|
|
"""``unresponsive_engines`` is a list of [name, reason] pairs (or strings)."""
|
|
out: dict[str, str] = {}
|
|
for item in pairs or []:
|
|
if isinstance(item, (list, tuple)) and len(item) >= 2:
|
|
out[str(item[0])] = str(item[1])
|
|
elif isinstance(item, str):
|
|
out[item] = "unresponsive"
|
|
return out
|
|
|
|
|
|
# ── QUALITY GUARD (search-agent-consumption) ─────────────────────────────────
|
|
# The agent-consumption layer applies a deterministic demote/drop policy. Without
|
|
# an assertion here it could silently rot back to raw engine ordering - the same
|
|
# way the endpoint colours silently rotted before 2026-09-26.
|
|
QUALITY_QUERIES = [
|
|
"best practices agent context management",
|
|
"proxmox thin pool metadata exhaustion recovery",
|
|
]
|
|
# A demoted (content-farm) host must never occupy the top 3 for these queries.
|
|
QUALITY_TOP_N = 3
|
|
# Non-answers that must never be returned for these queries at all.
|
|
QUALITY_BANNED_HOSTS = ["bestbuy.com", "merriam-webster.com"]
|
|
|
|
|
|
def _consumption_layer_path():
|
|
here = os.path.dirname(os.path.abspath(__file__))
|
|
return os.path.join(here, "search-agent-consume.py")
|
|
|
|
|
|
def check_ranking_quality() -> list[str]:
|
|
"""Return a list of quality failures; empty means healthy."""
|
|
import subprocess as _sp
|
|
|
|
layer = _consumption_layer_path()
|
|
if not os.path.exists(layer):
|
|
return [f"agent-consumption layer missing: {layer}"]
|
|
|
|
failures: list[str] = []
|
|
for query in QUALITY_QUERIES:
|
|
r = _sp.run([sys.executable, layer, "--no-extract", "--explain", query],
|
|
capture_output=True, text=True, timeout=120)
|
|
if r.returncode != 0:
|
|
failures.append(f"{query!r}: layer exited {r.returncode} ({r.stderr[:120]})")
|
|
continue
|
|
try:
|
|
data = json.loads(r.stdout)
|
|
except json.JSONDecodeError:
|
|
failures.append(f"{query!r}: layer returned unparseable JSON")
|
|
continue
|
|
|
|
results = data.get("results", [])
|
|
if len(results) < QUALITY_TOP_N:
|
|
failures.append(f"{query!r}: only {len(results)} results returned")
|
|
continue
|
|
|
|
# load the demote list from the SAME config the layer uses
|
|
cfg_path = os.path.join(os.path.dirname(layer), "..", "config", "search-ranking.yaml")
|
|
demoted: set[str] = set()
|
|
try:
|
|
sys.path.insert(0, os.path.dirname(layer))
|
|
import importlib.util as _iu
|
|
spec = _iu.spec_from_file_location("_sac_cfg", layer)
|
|
mod = _iu.module_from_spec(spec)
|
|
spec.loader.exec_module(mod)
|
|
demoted = set(mod._load_config().get("demote_domains", []) or [])
|
|
except Exception: # noqa: BLE001
|
|
failures.append(f"{query!r}: could not load demote_domains from config")
|
|
|
|
for item in results[:QUALITY_TOP_N]:
|
|
host = (item.get("host") or "")
|
|
for d in demoted:
|
|
if host == d or host.endswith("." + d):
|
|
failures.append(
|
|
f"{query!r}: demoted host {host} in top {QUALITY_TOP_N}"
|
|
)
|
|
for item in results:
|
|
host = (item.get("host") or "")
|
|
for b in QUALITY_BANNED_HOSTS:
|
|
if host == b or host.endswith("." + b):
|
|
failures.append(f"{query!r}: non-answer host {host} returned")
|
|
return failures
|
|
|
|
|
|
def main() -> int:
|
|
failures: list[str] = []
|
|
print(f"Search stack check -- {SEARXNG_URL}")
|
|
print(f"Queries: {QUERIES!r} min contributing engines: {MIN_ENGINES}")
|
|
print("=" * 72)
|
|
|
|
try:
|
|
eligible = enabled_expected_engines()
|
|
except Exception as exc: # noqa: BLE001 - report, do not traceback
|
|
print(f"FAIL: could not read /config from SearXNG: {exc!r}")
|
|
return 2
|
|
print(f"Expected engines, enabled ({len(eligible)}): {sorted(eligible)}")
|
|
missing = sorted(set(EXPECTED_ENGINES) - eligible)
|
|
if missing:
|
|
print(f"Expected engines NOT enabled: {missing}")
|
|
failures.append(f"expected engines not enabled in SearXNG: {missing}")
|
|
|
|
contributed: dict[str, int] = {name: 0 for name in eligible}
|
|
silent_zero_all: dict[str, list[str]] = {}
|
|
|
|
for query in QUERIES:
|
|
url = f"{SEARXNG_URL}/search?" + urllib.parse.urlencode(
|
|
{"q": query, "format": "json"}
|
|
)
|
|
print("-" * 72)
|
|
print(f"QUERY: {query!r}")
|
|
try:
|
|
data = _get_json(url)
|
|
except Exception as exc: # noqa: BLE001
|
|
print(f" FAIL: query request failed: {exc!r}")
|
|
failures.append(f"query {query!r} request failed: {exc!r}")
|
|
continue
|
|
|
|
results = data.get("results", [])
|
|
engines: dict[str, int] = {}
|
|
for r in results:
|
|
name = r.get("engine", "?")
|
|
engines[name] = engines.get(name, 0) + 1
|
|
unresponsive = unresponsive_names(data.get("unresponsive_engines", []))
|
|
|
|
print(f" results: {len(results)}")
|
|
print(f" contributing engines: {engines or '(none)'}")
|
|
print(f" unresponsive_engines: {unresponsive or '(none)'}")
|
|
|
|
for name in engines:
|
|
contributed[name] = contributed.get(name, 0) + engines[name]
|
|
|
|
if len(engines) < MIN_ENGINES:
|
|
msg = (
|
|
f"query {query!r} had only {len(engines)} contributing engine(s) "
|
|
f"({sorted(engines)}); need >= {MIN_ENGINES}"
|
|
)
|
|
print(f" FAIL: {msg}")
|
|
failures.append(msg)
|
|
|
|
silent = sorted(
|
|
n for n in eligible if n not in engines and n not in unresponsive
|
|
)
|
|
if silent:
|
|
silent_zero_all[query] = silent
|
|
print(
|
|
" SILENT ZERO (enabled, no error, no results -- reported, "
|
|
f"not fatal): {silent}"
|
|
)
|
|
|
|
print("=" * 72)
|
|
print("Engine contribution across all queries:")
|
|
for name in sorted(contributed):
|
|
status = "ZERO" if contributed[name] == 0 else "ok"
|
|
print(f" {name:<24} {contributed[name]:>4} {status}")
|
|
|
|
if silent_zero_all:
|
|
print("-" * 72)
|
|
print("SILENT-ZERO ENGINES REPORTED (no error raised, no results returned):")
|
|
for query, names in silent_zero_all.items():
|
|
print(f" {query!r}: {names}")
|
|
print(" NOTE: a silent zero is REPORTED, not counted as a failure. These")
|
|
print(" engines are expected to answer a general query, but contributing")
|
|
print(" nothing to one query can be legitimate (result de-duplication, or")
|
|
print(" an engine that only fires on certain query shapes). Only the")
|
|
print(f" <{MIN_ENGINES}-contributing-engine floor and the extraction leg fail the run.")
|
|
|
|
print("-" * 72)
|
|
print(f"EXTRACTION: scraping {EXTRACT_URL} via {FIRECRAWL_URL}/v1/scrape")
|
|
try:
|
|
payload = _post_json(
|
|
f"{FIRECRAWL_URL}/v1/scrape",
|
|
{"url": EXTRACT_URL, "formats": ["markdown"]},
|
|
)
|
|
markdown = ((payload.get("data") or {}).get("markdown") or "").strip()
|
|
if not markdown:
|
|
msg = "extraction returned empty markdown"
|
|
print(f" FAIL: {msg}")
|
|
failures.append(msg)
|
|
else:
|
|
print(f" ok: {len(markdown)} chars of markdown returned")
|
|
print(f" first line: {markdown.splitlines()[0][:120]!r}")
|
|
except Exception as exc: # noqa: BLE001
|
|
msg = f"extraction request failed: {exc!r}"
|
|
print(f" FAIL: {msg}")
|
|
failures.append(msg)
|
|
|
|
print("-" * 72)
|
|
print("RANKING QUALITY (agent-consumption layer)")
|
|
quality = check_ranking_quality()
|
|
if quality:
|
|
for q in quality:
|
|
print(f" FAIL: {q}")
|
|
failures.extend(quality)
|
|
else:
|
|
print(" ok: no demoted host in the top 3; no banned non-answer returned")
|
|
|
|
print("=" * 72)
|
|
if failures:
|
|
print("VERDICT: FAIL")
|
|
for f in failures:
|
|
print(f" - {f}")
|
|
return 1
|
|
print("VERDICT: PASS -- multiple engines contributing, extraction healthy")
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|