feat: sync runtime search and schema quality updates from app repo

- port retrieval, validation, and eval improvements relevant to os
- align prompts and dimensions with the flat single-agent model
- replace the old eval suite with the focused core scenarios

Generated with Codex
This commit is contained in:
“BeeRad”
2026-03-15 14:55:45 +11:00
parent 053c163e31
commit 4c75df101f
57 changed files with 1809 additions and 534 deletions
+19
View File
@@ -0,0 +1,19 @@
import { Scenario } from '../types';
export const scenario: Scenario = {
id: 'hub-traversal',
name: 'Hub traversal',
description: 'Traverse from core hubs and connected nodes to synthesize what the user should focus on next.',
tools: ['queryEdge'],
suites: ['traversal', 'internal'],
input: {
message: 'Traverse from my hub nodes "Building RA-H — Personal Knowledge Graph" and "Nature of Intelligence & Consciousness" and tell me what I should focus on next and why.',
},
expect: {
toolsCalledSoft: ['queryEdge'],
responseContainsSoft: ['focus', 'why'],
maxLatencyMs: 45000,
maxTotalTokens: 16000,
maxEstimatedCostUsd: 0.18,
},
};