Files
prose-contracts/scripts/search-stack-check.py
T
root 040fecef3e
PR Pipeline — Authorize → Validate → Review → Merge / auth (pull_request) Successful in 4s
PR Pipeline — Authorize → Validate → Review → Merge / validate (pull_request) Successful in 6s
PR Pipeline — Authorize → Validate → Review → Merge / lint (pull_request) Successful in 10s
PR Pipeline — Authorize → Validate → Review → Merge / ai-review (pull_request) Successful in 5s
PR Pipeline — Authorize → Validate → Review → Merge / gate (pull_request) Successful in 0s
feat: multi-engine search stack + visibility check
Search stack (192.168.68.7) was effectively Bing-only: google served a JS
shell, duckduckgo CAPTCHA'd from the house egress, and every other shipped
engine returned a silent zero. Upgraded SearXNG to 2026.9.23 (same pinned
digest as the image already pulled by other hosts) which uses browser
impersonation, and routed DuckDuckGo through a VPS forward proxy over the
existing WireGuard tunnel via a per-engine 'network'.

Live result: bing, google cse, brave and yandex contribute on every query;
duckduckgo is best-effort via the datacenter egress.

Adds the visibility leg so a future regression cannot be silent:

  scripts/search-stack-check.py
    * two fixed queries; FAILS when fewer than two engines contribute,
      printing contributing engines and every unresponsive_engines entry
    * FAILS when Firecrawl extraction returns empty markdown or errors
    * reports silent-zero engines explicitly

  scripts/contract-run.sh
    * maps search-stack-visibility -> search-stack-check.py

  search-stack-visibility.prose.md
    * contract text, execution model, pass/fail shapes, residual risk

Scheduled hourly at :15 on CT 100 via /etc/cron.d/contract-runner.
2026-09-25 01:14:03 +00:00

223 lines
8.1 KiB
Python
Executable File

#!/usr/bin/env python3
"""Search-stack visibility check.
The fleet shares one SearXNG instance (search) plus one extraction service
(Firecrawl). A broken search stack used to fail silently: one engine answered
and nobody could tell that the other engines had stopped contributing, or that
an enabled engine was returning nothing at all without reporting an error.
This check makes those failures visible and non-zero:
* runs two fixed queries against SearXNG; FAILS when fewer than two engines
contribute to a query, printing the contributing engines and every
``unresponsive_engines`` entry;
* FAILS when a known page cannot be extracted to non-empty markdown through
Firecrawl;
* reports every *silent zero* engine explicitly -- an engine that is enabled,
is eligible for the query category, is not listed in
``unresponsive_engines``, and still contributed no results.
Exit code 0 = healthy, 1 = degraded, 2 = the check could not run at all.
Environment overrides (all optional):
SEARXNG_URL default http://192.168.68.7:8888
FIRECRAWL_URL default http://192.168.68.7:3002
SEARCH_CHECK_QUERIES comma-separated fixed queries
SEARCH_CHECK_MIN_ENGINES default 2
SEARCH_CHECK_TIMEOUT per-request timeout in seconds, default 25
SEARCH_CHECK_EXTRACT_URL page used for the extraction leg
SEARCH_CHECK_ENGINES comma-separated engine names the stack is expected to
run; a silent zero is reported for any of them that is
enabled but contributes nothing with no error
"""
from __future__ import annotations
import json
import os
import sys
import urllib.error
import urllib.parse
import urllib.request
SEARXNG_URL = os.environ.get("SEARXNG_URL", "http://192.168.68.7:8888").rstrip("/")
FIRECRAWL_URL = os.environ.get("FIRECRAWL_URL", "http://192.168.68.7:3002").rstrip("/")
QUERIES = [
q.strip()
for q in os.environ.get(
"SEARCH_CHECK_QUERIES", "proxmox backup server,python asyncio tutorial"
).split(",")
if q.strip()
]
MIN_ENGINES = int(os.environ.get("SEARCH_CHECK_MIN_ENGINES", "2"))
TIMEOUT = float(os.environ.get("SEARCH_CHECK_TIMEOUT", "25"))
EXTRACT_URL = os.environ.get(
"SEARCH_CHECK_EXTRACT_URL", "https://en.wikipedia.org/wiki/Proxmox_Virtual_Environment"
)
# The general web-search engines this stack intentionally runs. A general query
# is expected to draw on these; an enabled one that returns nothing without an
# error is the silent-zero failure this check exists to expose. Specialised
# engines (images, videos, translate, currency, arxiv, npm, ...) are excluded on
# purpose -- contributing nothing to a general query is correct for them.
DEFAULT_EXPECTED_ENGINES = [
"bing",
"brave",
"google cse",
"yandex",
"duckduckgo",
]
EXPECTED_ENGINES = [
e.strip()
for e in os.environ.get(
"SEARCH_CHECK_ENGINES", ",".join(DEFAULT_EXPECTED_ENGINES)
).split(",")
if e.strip()
]
def _get_json(url: str) -> dict:
req = urllib.request.Request(url, headers={"User-Agent": "search-stack-check/1.0"})
with urllib.request.urlopen(req, timeout=TIMEOUT) as resp:
return json.loads(resp.read().decode("utf-8", "replace"))
def _post_json(url: str, payload: dict) -> dict:
data = json.dumps(payload).encode("utf-8")
req = urllib.request.Request(
url,
data=data,
headers={
"Content-Type": "application/json",
"User-Agent": "search-stack-check/1.0",
},
)
with urllib.request.urlopen(req, timeout=TIMEOUT) as resp:
return json.loads(resp.read().decode("utf-8", "replace"))
def enabled_expected_engines() -> set[str]:
"""Expected engines that SearXNG reports as actually enabled."""
cfg = _get_json(f"{SEARXNG_URL}/config")
enabled = {e["name"] for e in cfg.get("engines", []) if e.get("enabled")}
return {name for name in EXPECTED_ENGINES if name in enabled}
def unresponsive_names(pairs: list) -> dict[str, str]:
"""``unresponsive_engines`` is a list of [name, reason] pairs (or strings)."""
out: dict[str, str] = {}
for item in pairs or []:
if isinstance(item, (list, tuple)) and len(item) >= 2:
out[str(item[0])] = str(item[1])
elif isinstance(item, str):
out[item] = "unresponsive"
return out
def main() -> int:
failures: list[str] = []
print(f"Search stack check -- {SEARXNG_URL}")
print(f"Queries: {QUERIES!r} min contributing engines: {MIN_ENGINES}")
print("=" * 72)
try:
eligible = enabled_expected_engines()
except Exception as exc: # noqa: BLE001 - report, do not traceback
print(f"FAIL: could not read /config from SearXNG: {exc!r}")
return 2
print(f"Expected engines, enabled ({len(eligible)}): {sorted(eligible)}")
missing = sorted(set(EXPECTED_ENGINES) - eligible)
if missing:
print(f"Expected engines NOT enabled: {missing}")
failures.append(f"expected engines not enabled in SearXNG: {missing}")
contributed: dict[str, int] = {name: 0 for name in eligible}
silent_zero_all: dict[str, list[str]] = {}
for query in QUERIES:
url = f"{SEARXNG_URL}/search?" + urllib.parse.urlencode(
{"q": query, "format": "json"}
)
print("-" * 72)
print(f"QUERY: {query!r}")
try:
data = _get_json(url)
except Exception as exc: # noqa: BLE001
print(f" FAIL: query request failed: {exc!r}")
failures.append(f"query {query!r} request failed: {exc!r}")
continue
results = data.get("results", [])
engines: dict[str, int] = {}
for r in results:
name = r.get("engine", "?")
engines[name] = engines.get(name, 0) + 1
unresponsive = unresponsive_names(data.get("unresponsive_engines", []))
print(f" results: {len(results)}")
print(f" contributing engines: {engines or '(none)'}")
print(f" unresponsive_engines: {unresponsive or '(none)'}")
for name in engines:
contributed[name] = contributed.get(name, 0) + engines[name]
if len(engines) < MIN_ENGINES:
msg = (
f"query {query!r} had only {len(engines)} contributing engine(s) "
f"({sorted(engines)}); need >= {MIN_ENGINES}"
)
print(f" FAIL: {msg}")
failures.append(msg)
silent = sorted(
n for n in eligible if n not in engines and n not in unresponsive
)
if silent:
silent_zero_all[query] = silent
print(f" SILENT ZERO (enabled, no error, no results): {silent}")
print("=" * 72)
print("Engine contribution across all queries:")
for name in sorted(contributed):
status = "ZERO" if contributed[name] == 0 else "ok"
print(f" {name:<24} {contributed[name]:>4} {status}")
if silent_zero_all:
print("-" * 72)
print("SILENT-ZERO ENGINES REPORTED (no error raised, no results returned):")
for query, names in silent_zero_all.items():
print(f" {query!r}: {names}")
print("-" * 72)
print(f"EXTRACTION: scraping {EXTRACT_URL} via {FIRECRAWL_URL}/v1/scrape")
try:
payload = _post_json(
f"{FIRECRAWL_URL}/v1/scrape",
{"url": EXTRACT_URL, "formats": ["markdown"]},
)
markdown = ((payload.get("data") or {}).get("markdown") or "").strip()
if not markdown:
msg = "extraction returned empty markdown"
print(f" FAIL: {msg}")
failures.append(msg)
else:
print(f" ok: {len(markdown)} chars of markdown returned")
print(f" first line: {markdown.splitlines()[0][:120]!r}")
except Exception as exc: # noqa: BLE001
msg = f"extraction request failed: {exc!r}"
print(f" FAIL: {msg}")
failures.append(msg)
print("=" * 72)
if failures:
print("VERDICT: FAIL")
for f in failures:
print(f" - {f}")
return 1
print("VERDICT: PASS -- multiple engines contributing, extraction healthy")
return 0
if __name__ == "__main__":
sys.exit(main())