feat: sync runtime search and schema quality updates from app repo

- port retrieval, validation, and eval improvements relevant to os
- align prompts and dimensions with the flat single-agent model
- replace the old eval suite with the focused core scenarios

Generated with Codex
This commit is contained in:
“BeeRad”
2026-03-15 14:55:45 +11:00
parent 053c163e31
commit 4c75df101f
57 changed files with 1809 additions and 534 deletions
@@ -0,0 +1,20 @@
import { Scenario } from '../types';
export const scenario: Scenario = {
id: 'focused-graph-write',
name: 'Focused graph write',
description: 'Create one new node from the focused transcript and connect it back without unnecessary retrieval.',
tools: ['createNode', 'createEdge'],
input: {
message: 'Create a new node titled "Lange: Verification bottleneck" for the claim that generating many solutions is easier than verifying them, then connect it to this focused transcript with explanation "Claim extracted from this transcript source."',
focusedNodeQuery: { titleContains: 'When AI Discovers the Next Transformer' },
},
expect: {
toolsCalledSoft: ['createNode', 'createEdge'],
toolsNotCalledSoft: ['readSkill', 'queryNodes', 'searchContentEmbeddings'],
responseContainsSoft: ['Lange: Verification bottleneck'],
maxLatencyMs: 25000,
maxTotalTokens: 9000,
maxEstimatedCostUsd: 0.08,
},
};