# Search ranking policy for the agent-consumption layer. # # Everything here is CONFIG, not code, so it is reviewable and changeable without # touching the module. Read by scripts/search-agent-consume.py. # # Why this file exists: multi-engine aggregation returns results with no # filtering, no dedupe and no reranking. On 2026-09-26 that put a shopping page # and a dictionary definition into "best practices agent context management", # and put four SEO blogs ABOVE the actual Proxmox forum threads on a precise # technical query. Identical queries also ranked differently between runs, which # is the strongest argument for a deterministic layer rather than hoping the # engines behave. version: 1 # ── Non-answers: dropped outright, never returned ──────────────────────────── # These are pages that cannot answer a question: navigational homepages, # shopping/product pages, dictionary definitions, and login walls. non_answer: # URL path is empty -> it is a site's front door, not an answer. Still allowed # when the host is explicitly preferred (see prefer_domains), because some # docs/repo front doors ARE the answer. host_root: true path_patterns: - '/dictionary/' - '/dictionary?' - '/wiki/Wiktionary:' - '/search?' - '/cart' - '/checkout' - '/login' - '/signin' - '/sign-in' - '/account/login' - '/shop/' - '/store/' - '/dp/' # Amazon-style product URL - '/gp/product/' - '/add-to-cart' - '/checkout' # NOTE: '/products/' and '/product/' were REMOVED as path patterns. They fired # on docs.digitalocean.com/products/inference/... — a legitimate documentation # page — which the 2026-09-26 before/after run caught. Shopping is caught by # the shopping HOST list instead, which does not have that false positive. # Query strings that betray a search/shopping surface rather than an article. query_keys: - 'q' - 'query' - 's' - 'search' - 'add-to-cart' # Hosts that are shopping/retail and never answer a technical question. hosts: - bestbuy.com - amazon.com - ebay.com - walmart.com - etsy.com - aliexpress.com - merriam-webster.com - dictionary.com - thesaurus.com - vocabulary.com - collinsdictionary.com # ── Demotion: ranked below everything else, never dropped ──────────────────── # Low-authority content farms / SEO aggregators. Demoted rather than dropped so # a genuinely useful hit is not lost, but it can never outrank a primary source. # Reviewable: add or remove hosts here, no code change required. demote_domains: - medium.com - sparkco.ai - mindstudio.ai - aitechmonk.com - stackai.com - agentic-design.ai - voxfor.com - bigiron.cc - linuxoperatingsystem.net - riparazioneserver.com - rossmanngroup.com - dev.to - hashnode.dev - substack.com - towardsdatascience.com - analyticsvidhya.com - geeksforgeeks.org - tutorialspoint.com - javatpoint.com - w3schools.com - scaler.com - simplilearn.com - udemy.com - coursera.org # ── Preference: promoted above the default rank ────────────────────────────── # Primary sources: upstream repositories, official docs, Q&A, vendor # engineering blogs. These are what an agent should be reading. prefer_domains: # upstream repositories and code hosting - github.com - gitlab.com - codeberg.org - sourceforge.net - kernel.org - git.kernel.org # Q&A - stackoverflow.com - stackexchange.com - superuser.com - serverfault.com - askubuntu.com - discourse.org # vendor / project documentation and forums - proxmox.com - forum.proxmox.com - pve.proxmox.com - docs.python.org - developer.mozilla.org - kernelnewbies.org - man7.org - gnu.org - debian.org - ubuntu.com - redhat.com - kernel.dk # io_uring / Jens Axboe - github.io # project pages (docs, papers) — promoted, not authoritative by itself # vendor engineering blogs - anthropic.com - openai.com - googleblog.com - developers.googleblog.com - engineering.fb.com - netflixtechblog.com - aws.amazon.com - cloud.google.com - microsoft.com - learn.microsoft.com - apple.com - nvidia.com - intel.com - amd.com - redislabs.com - cloudflare.com - langchain.com - jetbrains.com - cursor.com # community discussion with high signal - news.ycombinator.com - lobste.rs - reddit.com # ── Ranking weights ────────────────────────────────────────────────────────── # Final score = engine_score - demote_penalty + prefer_bonus, then a stable # tiebreak on original position so ordering is reproducible run to run. ranking: demote_penalty: 1000 prefer_bonus: 100 # Results that several engines independently returned are more likely real. multi_engine_bonus: 25 # Shallow paths (e.g. /blog/x) are slightly less likely to be primary docs. host_root_allowed_when_preferred: true # ── Extraction budget (criterion 4) ────────────────────────────────────────── # Return CONTENT, not just links, so an agent gets usable material in ONE call. extraction: top_n: 5 # how many results get page text extracted total_chars: 12000 # global budget across all extracted items per_item_chars: 4000 # cap for any single item, so one page cannot eat the budget timeout_seconds: 45 # per scrape # If extraction fails, the result is still returned with an empty excerpt — # a link is better than nothing, but the failure is recorded in the output. on_failure: keep_with_empty_excerpt