feat: sync runtime search and schema quality updates from app repo
- port retrieval, validation, and eval improvements relevant to os - align prompts and dimensions with the flat single-agent model - replace the old eval suite with the focused core scenarios Generated with Codex
This commit is contained in:
@@ -0,0 +1,16 @@
|
||||
import { Scenario } from '../types';
|
||||
|
||||
export const scenario: Scenario = {
|
||||
id: 'hard-mode-query',
|
||||
name: 'Hard mode retrieval query',
|
||||
description: 'Run a baseline retrieval query in hard mode.',
|
||||
tools: ['queryNodes', 'searchContentEmbeddings'],
|
||||
input: {
|
||||
message: 'What have I captured about plaintext productivity and tools?',
|
||||
mode: 'hard',
|
||||
},
|
||||
expect: {
|
||||
toolsCalledSoft: ['queryNodes'],
|
||||
responseContainsSoft: ['plaintext', 'productivity'],
|
||||
},
|
||||
};
|
||||
Reference in New Issue
Block a user