feat: sync runtime search and schema quality updates from app repo

- port retrieval, validation, and eval improvements relevant to os
- align prompts and dimensions with the flat single-agent model
- replace the old eval suite with the focused core scenarios

Generated with Codex
This commit is contained in:
“BeeRad”
2026-03-15 14:55:45 +11:00
parent 053c163e31
commit 4c75df101f
57 changed files with 1809 additions and 534 deletions
+11 -3
View File
@@ -1,5 +1,13 @@
{
"id": "golden-v1",
"name": "Golden Dataset v1",
"description": "Baseline eval scenarios for core RA-H helper behavior."
"id": "golden-ra-h-core-v2",
"name": "Golden RA-H Core v2",
"description": "Slim 5-scenario eval set for core RA-H usage: focused graph writes, skill-guided writes, indexed node search, chunk-grounded insight creation, and hub traversal.",
"inherits": "golden-v1",
"version": 2,
"focus": [
"tool-call correctness",
"skill usage only when appropriate",
"core retrieval/write behavior",
"latency/tokens/cost guards"
]
}