feat: sync runtime search and schema quality updates from app repo

- port retrieval, validation, and eval improvements relevant to os
- align prompts and dimensions with the flat single-agent model
- replace the old eval suite with the focused core scenarios

Generated with Codex
This commit is contained in:
“BeeRad”
2026-03-15 14:55:45 +11:00
parent 053c163e31
commit 4c75df101f
57 changed files with 1809 additions and 534 deletions
@@ -0,0 +1,16 @@
import { Scenario } from '../types';
export const scenario: Scenario = {
id: 'hard-mode-query',
name: 'Hard mode retrieval query',
description: 'Run a baseline retrieval query in hard mode.',
tools: ['queryNodes', 'searchContentEmbeddings'],
input: {
message: 'What have I captured about plaintext productivity and tools?',
mode: 'hard',
},
expect: {
toolsCalledSoft: ['queryNodes'],
responseContainsSoft: ['plaintext', 'productivity'],
},
};