feat: sync runtime search and schema quality updates from app repo
- port retrieval, validation, and eval improvements relevant to os - align prompts and dimensions with the flat single-agent model - replace the old eval suite with the focused core scenarios Generated with Codex
This commit is contained in:
@@ -1,10 +1,18 @@
|
||||
export type ScenarioExpectations = {
|
||||
skillsRead?: string[];
|
||||
skillsReadSoft?: string[];
|
||||
skillsNotRead?: string[];
|
||||
skillsNotReadSoft?: string[];
|
||||
toolsCalled?: string[];
|
||||
toolsCalledSoft?: string[];
|
||||
toolsNotCalled?: string[];
|
||||
toolsNotCalledSoft?: string[];
|
||||
responseContains?: string[];
|
||||
responseContainsSoft?: string[];
|
||||
responseNotContains?: string[];
|
||||
maxLatencyMs?: number;
|
||||
maxTotalTokens?: number;
|
||||
maxEstimatedCostUsd?: number;
|
||||
};
|
||||
|
||||
export type ScenarioInput = {
|
||||
@@ -24,6 +32,7 @@ export type Scenario = {
|
||||
expect?: ScenarioExpectations;
|
||||
description?: string;
|
||||
tools?: string[];
|
||||
suites?: string[];
|
||||
enabled?: boolean;
|
||||
notes?: string;
|
||||
};
|
||||
|
||||
Reference in New Issue
Block a user