feat: sync runtime search and schema quality updates from app repo

- port retrieval, validation, and eval improvements relevant to os
- align prompts and dimensions with the flat single-agent model
- replace the old eval suite with the focused core scenarios

Generated with Codex
This commit is contained in:
“BeeRad”
2026-03-15 14:55:45 +11:00
parent 053c163e31
commit 4c75df101f
57 changed files with 1809 additions and 534 deletions
@@ -0,0 +1,19 @@
import { Scenario } from '../types';
export const scenario: Scenario = {
id: 'chunk-quote-insight',
name: 'Chunk quote to insight',
description: 'Search a focused transcript for a specific quote, then create a grounded insight node from it.',
tools: ['searchContentEmbeddings', 'createNode', 'createEdge'],
input: {
message: 'Search inside this focused transcript for a quote about verification being harder than generating solutions. Quote it briefly, then create a new insight node titled "Lange on verification difficulty" and connect it back to this transcript with explanation "Insight extracted from quoted passage."',
focusedNodeQuery: { titleContains: 'When AI Discovers the Next Transformer' },
},
expect: {
toolsCalledSoft: ['searchContentEmbeddings', 'createNode', 'createEdge'],
responseContainsSoft: ['Lange on verification difficulty'],
maxLatencyMs: 35000,
maxTotalTokens: 14000,
maxEstimatedCostUsd: 0.14,
},
};