From 86c79d78d6fcb8481101d4d6ebbacbfea84d6bb3 Mon Sep 17 00:00:00 2001 From: Krishna Date: Tue, 29 Sep 2026 14:47:37 +0530 Subject: [PATCH 1/7] feat(evals): add agent tool-use evaluation harness Add a deterministic eval layer for the agent harness. Scenarios script the OpenAI-compatible streaming tool loop with model turns and stub tool results, then score tool selection, planning, retrieval, and recovery without a provider key. - apps/sim/evals/agent-tool-use: 8 scenarios, scoring, JSON+Markdown report - `bun run test:evals` from apps/sim runs the suite and writes the report - picked up by the normal vitest run so a regression fails CI - README documents the contract and how to add a case --- apps/sim/evals/README.md | 88 +++++ .../agent-tool-use.eval.test.ts | 34 ++ apps/sim/evals/agent-tool-use/harness.ts | 366 ++++++++++++++++++ apps/sim/evals/agent-tool-use/report.ts | 63 +++ apps/sim/evals/agent-tool-use/scenarios.ts | 352 +++++++++++++++++ apps/sim/evals/agent-tool-use/types.ts | 125 ++++++ apps/sim/package.json | 1 + 7 files changed, 1029 insertions(+) create mode 100644 apps/sim/evals/README.md create mode 100644 apps/sim/evals/agent-tool-use/agent-tool-use.eval.test.ts create mode 100644 apps/sim/evals/agent-tool-use/harness.ts create mode 100644 apps/sim/evals/agent-tool-use/report.ts create mode 100644 apps/sim/evals/agent-tool-use/scenarios.ts create mode 100644 apps/sim/evals/agent-tool-use/types.ts diff --git a/apps/sim/evals/README.md b/apps/sim/evals/README.md new file mode 100644 index 00000000000..604df341037 --- /dev/null +++ b/apps/sim/evals/README.md @@ -0,0 +1,88 @@ +# Agent harness evaluations + +Measurement for the agent harness — the code that turns a model's tool calls +into executed tools, feeds the results back, and keeps the turn alive when a +tool fails. Unit and integration tests prove the harness handles the cases we +already know about; evals measure whether it still behaves across a suite of +scenarios when the harness changes. + +## What runs + +The first suite lives in [`agent-tool-use/`](./agent-tool-use) and drives the +real OpenAI-compatible streaming tool loop +(`apps/sim/providers/openai-compat/streaming-tool-loop.ts`) — the loop that +serves OpenAI, DeepSeek, Groq, Cerebras, and the other OpenAI-compatible +providers. The model is **scripted**: each scenario supplies the assistant turns +(tool calls or a final answer) and the result of each tool call. That keeps the +suite deterministic and runnable in CI with no provider key, while the thing +being measured — tool dispatch, result feedback, error recovery — is real +production code. + +The suites cover four behaviors: + +| Category | What it measures | +| --- | --- | +| `tool-selection` | The loop dispatches the tool the model asked for, including from a set of distractors. | +| `planning` | Multi-turn, dependent and parallel tool calls execute in the right order and all results reach the next turn. | +| `retrieval` | Values returned by a tool survive into the final answer instead of being dropped or invented. | +| `recovery` | Tool errors, unknown tool names, and malformed argument JSON are fed back to the model rather than thrown out of the loop. | + +## Run it + +From `apps/sim`: + +```sh +bun run test:evals +``` + +The command writes a JSON report and a Markdown summary to +`test-results/evals/agent-tool-use.{json,md}` (gitignored) and fails the process +if any scenario fails. To point the report somewhere else, run Vitest directly: + +```sh +EVAL_REPORT_PATH=/tmp/agent-tool-use.json bunx vitest run evals/agent-tool-use +``` + +The suite is also picked up by the normal `bun run test` run, so a regression +fails CI even without the dedicated command. + +## Add a case + +1. Open [`agent-tool-use/scenarios.ts`](./agent-tool-use/scenarios.ts) and add + an entry to `AGENT_TOOL_USE_SCENARIOS`. +2. Declare the `tools` the model may call and the `script` it produces. A + `tools` turn lists the calls the model emits; an `answer` turn ends the run. + Attach each call's stub `result` (or leave it to default to a successful + empty output). +3. Add the assertions you care about under `expect`: the ordered + `toolCallSequence`, `requiredTools`/`forbiddenTools`, `finalContent`, + `maxIterations`, and tool call counts. Every assertion becomes a named check + in the report. +4. Run `bun run test:evals`. + +A scenario is data, not code — there is no harness change needed for a new case. + +### Simulating a failure + +- **Tool error:** give the call `result: { success: false, error: '...' }`. +- **Unknown tool:** call a `name` that is not in `tools`; the loop returns a + tool-not-found error to the model. +- **Malformed arguments:** set `argumentsJson` to an invalid or non-object JSON + string. The loop must not execute the call and must return the parse error to + the model. + +## Report shape + +`report.json` is machine-readable for dashboards and trend tracking; `report.md` +is the same data as a table. Each result carries the scenario id, pass/fail, +every named check with a failure detail, the final content, the executed tool +invocations, and metrics: iterations, tool call counts (success/error), latency, +model/tool time, first-response time, and token usage. + +## Scope and next steps + +This suite evaluates the tool loop directly. The next layer is a scenario that +runs the same scripted model through the full `DAGExecutor` so agent block +wiring, variable resolution, and the executor's retry/fallback policy are +measured alongside the loop. The `AgentToolUseResult` shape is deliberately +independent of the harness entry point so both can share scoring and reporting. diff --git a/apps/sim/evals/agent-tool-use/agent-tool-use.eval.test.ts b/apps/sim/evals/agent-tool-use/agent-tool-use.eval.test.ts new file mode 100644 index 00000000000..43f7cbaae1c --- /dev/null +++ b/apps/sim/evals/agent-tool-use/agent-tool-use.eval.test.ts @@ -0,0 +1,34 @@ +import { providersMock } from '@sim/testing/mocks/providers.mock' +import { providersConversationHistoryMock } from '@sim/testing/mocks/providers-conversation-history.mock' +import { providersUtilsMock } from '@sim/testing/mocks/providers-utils.mock' +import { toolsMock } from '@sim/testing/mocks/tools.mock' +import { afterAll, describe, expect, it, vi } from 'vitest' +import { runScenario } from '@/evals/agent-tool-use/harness' +import { writeEvalReport } from '@/evals/agent-tool-use/report' +import { AGENT_TOOL_USE_SCENARIOS } from '@/evals/agent-tool-use/scenarios' +import type { AgentToolUseResult } from '@/evals/agent-tool-use/types' + +vi.mock('@/providers/conversation-history', () => providersConversationHistoryMock) +vi.mock('@/tools', () => toolsMock) +vi.mock('@/providers/utils', () => providersUtilsMock) +vi.mock('@/providers', () => providersMock) + +const results: AgentToolUseResult[] = [] + +afterAll(() => { + const reportPath = process.env.EVAL_REPORT_PATH + if (reportPath) writeEvalReport(results, reportPath) +}) + +describe('agent tool-use eval suite', () => { + it.each(AGENT_TOOL_USE_SCENARIOS)('$id: $name', async (scenario) => { + const result = await runScenario(scenario) + results.push(result) + + const failed = result.checks.filter((entry) => !entry.passed) + expect( + failed, + failed.map((entry) => `${entry.name}: ${entry.detail}`).join('; ') || undefined + ).toEqual([]) + }) +}) diff --git a/apps/sim/evals/agent-tool-use/harness.ts b/apps/sim/evals/agent-tool-use/harness.ts new file mode 100644 index 00000000000..1bb95744fe0 --- /dev/null +++ b/apps/sim/evals/agent-tool-use/harness.ts @@ -0,0 +1,366 @@ +import { createLogger } from '@sim/logger' +import { collectStream } from '@sim/testing/helpers/async' +import { providersMock } from '@sim/testing/mocks/providers.mock' +import { providersUtilsMockFns } from '@sim/testing/mocks/providers-utils.mock' +import { toolsMockFns } from '@sim/testing/mocks/tools.mock' +import { isRecordLike } from '@sim/utils/object' +import type { ChatCompletionChunk } from 'openai/resources/chat/completions' +import type { CompletionUsage } from 'openai/resources/completions' +import { + createOpenAICompatStreamingToolLoopStream, + type OpenAICompatCreateCompletion, +} from '@/providers/openai-compat/streaming-tool-loop' +import type { AgentStreamEvent } from '@/providers/stream-events' +import type { StreamingToolLoopComplete } from '@/providers/streaming-tool-loop-shared' +import type { ProviderToolConfig, TimeSegment } from '@/providers/types' +import type { ToolResponse } from '@/tools/types' +import type { + AgentToolUseResult, + AgentToolUseScenario, + EvalCheck, + EvalToolDefinition, + EvalToolInvocation, + ScriptedModelTurn, + ScriptedToolCall, +} from './types' + +/** + * Runs one scenario through the real OpenAI-compatible streaming tool loop and + * scores the result. Vitest owns the `@/tools`, `@/providers` and + * `@/providers/utils` module mocks; this module only drives them. + */ + +const EVAL_MODEL = 'eval-model' +const EVAL_PROVIDER = 'Eval' +const MAX_TOOL_ITERATIONS = 20 + +const logger = createLogger('AgentToolUseEval') + +interface CapturedToolCall { + name: string + arguments: Record + success: boolean + result?: unknown + duration?: number +} + +interface ToolCallList { + list: CapturedToolCall[] + count: number +} + +/** Raw call counts as a value the loop never reads; scenarios only assert on it. */ +const COMPLETION_USAGE = (): CompletionUsage => ({ + prompt_tokens: 10, + completion_tokens: 5, + total_tokens: 15, +}) + +let chunkCounter = 0 + +function chunk( + delta: ChatCompletionChunk.Choice['delta'] & { reasoning_content?: string }, + finishReason: ChatCompletionChunk.Choice['finish_reason'] = null, + usage?: CompletionUsage +): ChatCompletionChunk { + chunkCounter += 1 + return { + id: `eval-chunk-${chunkCounter}`, + object: 'chat.completion.chunk', + created: 0, + model: EVAL_MODEL, + choices: [{ index: 0, delta, finish_reason: finishReason, logprobs: null }], + ...(usage ? { usage } : {}), + } +} + +function turnToChunks(turn: ScriptedModelTurn): ChatCompletionChunk[] { + if (turn.kind === 'answer') { + return [ + ...(turn.thinking ? [chunk({ reasoning_content: turn.thinking })] : []), + chunk({ content: turn.content }, 'stop', COMPLETION_USAGE()), + ] + } + + const chunks: ChatCompletionChunk[] = [] + if (turn.thinking) chunks.push(chunk({ reasoning_content: turn.thinking })) + turn.calls.forEach((call, index) => { + chunks.push( + chunk({ + tool_calls: [ + { + index, + id: `call_${index}`, + type: 'function', + function: { + name: call.name, + arguments: call.argumentsJson ?? JSON.stringify(call.args ?? {}), + }, + }, + ], + }) + ) + }) + chunks.push(chunk({}, 'tool_calls', COMPLETION_USAGE())) + return chunks +} + +function createScriptedCompletion(scenario: AgentToolUseScenario): OpenAICompatCreateCompletion { + let turnIndex = 0 + return async () => { + const turn = scenario.script[turnIndex] + turnIndex += 1 + if (!turn) { + throw new Error( + `Scenario "${scenario.id}" requested model turn ${turnIndex} but only ${scenario.script.length} are scripted` + ) + } + return (async function* () { + for (const next of turnToChunks(turn)) yield next + })() + } +} + +function toProviderTools(tools: EvalToolDefinition[]): ProviderToolConfig[] { + return tools.map((tool) => ({ + id: tool.name, + description: tool.description, + params: {}, + parameters: { + type: tool.parameters?.type ?? 'object', + properties: tool.parameters?.properties ?? {}, + required: tool.parameters?.required ?? [], + }, + })) +} + +/** True when the loop will execute the call, so its result must be queued. */ +function isExecutable(call: ScriptedToolCall, toolNames: Set): boolean { + if (!toolNames.has(call.name)) return false + if (call.argumentsJson === undefined) return true + try { + return isRecordLike(JSON.parse(call.argumentsJson)) + } catch { + return false + } +} + +/** + * Results are queued per tool in script order. The loop may execute calls from + * one turn in any completion order, so keying by name keeps every call matched + * to the result the scenario intended. + */ +function buildResultQueues( + scenario: AgentToolUseScenario, + toolNames: Set +): Map { + const queues = new Map() + for (const turn of scenario.script) { + if (turn.kind !== 'tools') continue + for (const call of turn.calls) { + if (!isExecutable(call, toolNames)) continue + const spec = call.result + const queue = queues.get(call.name) ?? [] + queue.push({ + success: spec?.success ?? true, + output: spec?.output ?? {}, + ...(spec?.error ? { error: spec.error } : {}), + }) + queues.set(call.name, queue) + } + } + return queues +} + +function check(name: string, passed: boolean, detail: string): EvalCheck { + return { name, passed, detail } +} + +function sameSequence(actual: string[], expected: string[]): boolean { + return actual.length === expected.length && actual.every((name, i) => name === expected[i]) +} + +function matchesContent(content: string, expected: string | RegExp): boolean { + return typeof expected === 'string' ? content.includes(expected) : expected.test(content) +} + +function score( + scenario: AgentToolUseScenario, + toolCalls: CapturedToolCall[], + finalContent: string, + iterations: number, + error: unknown +): EvalCheck[] { + const expected = scenario.expect + const actualSequence = toolCalls.map((call) => call.name) + const checks: EvalCheck[] = [] + + if (expected.toolCallSequence) { + checks.push( + check( + 'tool-call-sequence', + sameSequence(actualSequence, expected.toolCallSequence), + `expected [${expected.toolCallSequence.join(', ')}], got [${actualSequence.join(', ')}]` + ) + ) + } + + if (expected.requiredTools) { + const missing = expected.requiredTools.filter((name) => !actualSequence.includes(name)) + checks.push( + check( + 'required-tools', + missing.length === 0, + missing.length === 0 ? 'all required tools called' : `missing [${missing.join(', ')}]` + ) + ) + } + + if (expected.forbiddenTools) { + const called = expected.forbiddenTools.filter((name) => actualSequence.includes(name)) + checks.push( + check( + 'forbidden-tools', + called.length === 0, + called.length === 0 ? 'no forbidden tools called' : `called [${called.join(', ')}]` + ) + ) + } + + if (expected.finalContent !== undefined) { + checks.push( + check( + 'final-content', + matchesContent(finalContent, expected.finalContent), + `final content ${JSON.stringify(finalContent)}` + ) + ) + } + + if (expected.maxIterations !== undefined) { + checks.push( + check( + 'max-iterations', + iterations <= expected.maxIterations, + `iterations ${iterations} (max ${expected.maxIterations})` + ) + ) + } + + const successful = toolCalls.filter((call) => call.success).length + if (expected.successfulToolCalls !== undefined) { + checks.push( + check( + 'successful-tool-calls', + successful === expected.successfulToolCalls, + `expected ${expected.successfulToolCalls}, got ${successful}` + ) + ) + } + + const errored = toolCalls.length - successful + if (expected.erroredToolCalls !== undefined) { + checks.push( + check( + 'errored-tool-calls', + errored === expected.erroredToolCalls, + `expected ${expected.erroredToolCalls}, got ${errored}` + ) + ) + } + + if (expected.completesWithoutError !== false) { + checks.push( + check( + 'completes-without-error', + error === undefined, + error === undefined ? 'loop settled' : String(error) + ) + ) + } + + return checks +} + +/** Runs and scores one scenario. */ +export async function runScenario(scenario: AgentToolUseScenario): Promise { + const toolNames = new Set(scenario.tools.map((tool) => tool.name)) + const resultQueues = buildResultQueues(scenario, toolNames) + const invocations: EvalToolInvocation[] = [] + + providersMock.MAX_TOOL_ITERATIONS = MAX_TOOL_ITERATIONS + providersUtilsMockFns.mockCalculateCost.mockReturnValue({ input: 0, output: 0, total: 0 }) + toolsMockFns.mockExecuteTool.mockImplementation( + async (toolId: string, params: Record): Promise => { + const startedAt = Date.now() + const response = resultQueues.get(toolId)?.shift() ?? { success: true, output: {} } + invocations.push({ + name: toolId, + arguments: params ?? {}, + success: response.success, + ...(response.error ? { error: response.error } : {}), + durationMs: Date.now() - startedAt, + }) + return response + } + ) + + const timeSegments: TimeSegment[] = [] + let completed: StreamingToolLoopComplete | undefined + let streamError: unknown + + const startedAt = Date.now() + try { + const stream = createOpenAICompatStreamingToolLoopStream({ + providerName: EVAL_PROVIDER, + request: { + model: EVAL_MODEL, + apiKey: 'eval-key', + messages: [], + tools: toProviderTools(scenario.tools), + }, + basePayload: { model: EVAL_MODEL }, + messages: [{ role: 'user', content: scenario.userMessage }], + createStream: createScriptedCompletion(scenario), + logger, + timeSegments, + onComplete: (result) => { + completed = result + }, + }) + await collectStream(stream as ReadableStream) + } catch (error) { + streamError = error + } + const latencyMs = Date.now() - startedAt + + const toolCalls = ((completed?.toolCalls as ToolCallList | undefined)?.list ?? []).slice() + const finalContent = completed?.content ?? '' + const iterations = completed?.iterations ?? 0 + const checks = score(scenario, toolCalls, finalContent, iterations, streamError) + const successful = toolCalls.filter((call) => call.success).length + + return { + id: scenario.id, + name: scenario.name, + category: scenario.category, + passed: checks.every((entry) => entry.passed), + checks, + finalContent, + toolInvocations: invocations, + metrics: { + iterations, + toolCalls: toolCalls.length, + successfulToolCalls: successful, + erroredToolCalls: toolCalls.length - successful, + latencyMs, + modelTimeMs: completed?.modelTime ?? 0, + toolsTimeMs: completed?.toolsTime ?? 0, + firstResponseTimeMs: completed?.firstResponseTime ?? 0, + inputTokens: completed?.tokens.input ?? 0, + outputTokens: completed?.tokens.output ?? 0, + totalTokens: completed?.tokens.total ?? 0, + }, + ...(streamError ? { error: String(streamError) } : {}), + } +} diff --git a/apps/sim/evals/agent-tool-use/report.ts b/apps/sim/evals/agent-tool-use/report.ts new file mode 100644 index 00000000000..494513d3948 --- /dev/null +++ b/apps/sim/evals/agent-tool-use/report.ts @@ -0,0 +1,63 @@ +import { mkdirSync, writeFileSync } from 'node:fs' +import { dirname } from 'node:path' +import type { AgentToolUseResult } from './types' + +/** Machine-readable summary of one eval run. */ +export interface AgentToolUseEvalReport { + suite: 'agent-tool-use' + generatedAt: string + total: number + passed: number + failed: number + results: AgentToolUseResult[] +} + +export function buildEvalReport(results: AgentToolUseResult[]): AgentToolUseEvalReport { + const passed = results.filter((result) => result.passed).length + return { + suite: 'agent-tool-use', + generatedAt: new Date().toISOString(), + total: results.length, + passed, + failed: results.length - passed, + results, + } +} + +function escapeCell(value: string): string { + return value.replaceAll('|', '\\|').replaceAll('\n', ' ') +} + +function renderMarkdown(report: AgentToolUseEvalReport): string { + const header = [ + '# Agent tool-use eval report', + '', + `Generated: ${report.generatedAt}`, + '', + `**${report.passed}/${report.total} passed**`, + '', + '| Scenario | Category | Status | Iterations | Tools (ok/error) | Latency | Failed checks |', + '| --- | --- | --- | ---: | ---: | ---: | --- |', + ] + + const rows = report.results.map((result) => { + const failed = result.checks + .filter((entry) => !entry.passed) + .map((entry) => entry.name) + .join(', ') + return `| ${escapeCell(result.id)} | ${result.category} | ${result.passed ? '✅ pass' : '❌ fail'} | ${result.metrics.iterations} | ${result.metrics.successfulToolCalls}/${result.metrics.erroredToolCalls} | ${result.metrics.latencyMs}ms | ${failed || '—'} |` + }) + + return [...header, ...rows, ''].join('\n') +} + +/** + * Writes the JSON report to `reportPath` and a sibling Markdown summary. The + * caller supplies the path (`EVAL_REPORT_PATH`) so CI can upload it. + */ +export function writeEvalReport(results: AgentToolUseResult[], reportPath: string): void { + const report = buildEvalReport(results) + mkdirSync(dirname(reportPath), { recursive: true }) + writeFileSync(reportPath, `${JSON.stringify(report, null, 2)}\n`) + writeFileSync(reportPath.replace(/\.json$/, '.md'), renderMarkdown(report)) +} diff --git a/apps/sim/evals/agent-tool-use/scenarios.ts b/apps/sim/evals/agent-tool-use/scenarios.ts new file mode 100644 index 00000000000..a0042b33fb6 --- /dev/null +++ b/apps/sim/evals/agent-tool-use/scenarios.ts @@ -0,0 +1,352 @@ +import type { AgentToolUseScenario, EvalToolDefinition } from './types' + +const searchDocs: EvalToolDefinition = { + name: 'search_docs', + description: 'Search the product documentation for a query and return matching passages.', + parameters: { + type: 'object', + properties: { query: { type: 'string', description: 'Search query' } }, + required: ['query'], + }, +} + +const getWeather: EvalToolDefinition = { + name: 'get_weather', + description: 'Get the current weather for a city.', + parameters: { + type: 'object', + properties: { city: { type: 'string', description: 'City name' } }, + required: ['city'], + }, +} + +const sendEmail: EvalToolDefinition = { + name: 'send_email', + description: 'Send an email to a recipient.', + parameters: { + type: 'object', + properties: { + to: { type: 'string' }, + subject: { type: 'string' }, + body: { type: 'string' }, + }, + required: ['to', 'subject', 'body'], + }, +} + +const lookupOrder: EvalToolDefinition = { + name: 'lookup_order', + description: 'Look up an order by its customer-facing order number.', + parameters: { + type: 'object', + properties: { orderNumber: { type: 'string' } }, + required: ['orderNumber'], + }, +} + +const listFiles: EvalToolDefinition = { + name: 'list_files', + description: 'List files in a directory.', + parameters: { + type: 'object', + properties: { directory: { type: 'string' } }, + required: ['directory'], + }, +} + +const readFile: EvalToolDefinition = { + name: 'read_file', + description: 'Read the contents of a file.', + parameters: { + type: 'object', + properties: { path: { type: 'string' } }, + required: ['path'], + }, +} + +const flakyApi: EvalToolDefinition = { + name: 'flaky_api', + description: 'Fetch a value from an upstream API that intermittently returns 503.', + parameters: { + type: 'object', + properties: { resource: { type: 'string' } }, + required: ['resource'], + }, +} + +const getNews: EvalToolDefinition = { + name: 'get_news', + description: 'Get the top news headlines for a topic.', + parameters: { + type: 'object', + properties: { topic: { type: 'string' } }, + required: ['topic'], + }, +} + +/** + * The first suite: eight agent tool-use reliability cases. Each script is the + * model transcript; the loop, not the model, is what the assertions score. + */ +export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [ + { + id: 'single-tool-lookup', + name: 'calls the one relevant tool and answers from its result', + category: 'tool-selection', + description: + 'A single retrieval tool is available. The loop must dispatch it once and surface the answer built from its result.', + userMessage: 'What is the API rate limit?', + tools: [searchDocs], + script: [ + { + kind: 'tools', + calls: [ + { + name: 'search_docs', + args: { query: 'api rate limit' }, + result: { + success: true, + output: { snippet: 'The API rate limit is 100 requests per minute.' }, + }, + }, + ], + }, + { kind: 'answer', content: 'The API rate limit is 100 requests per minute.' }, + ], + expect: { + toolCallSequence: ['search_docs'], + finalContent: '100 requests per minute', + maxIterations: 3, + successfulToolCalls: 1, + erroredToolCalls: 0, + }, + }, + { + id: 'select-correct-tool', + name: 'selects the relevant tool and leaves the irrelevant ones unused', + category: 'tool-selection', + description: + 'Four tools are exposed; only the weather tool answers the question. The loop must not invoke the distractors.', + userMessage: 'What is the weather in Berlin right now?', + tools: [searchDocs, getWeather, sendEmail, lookupOrder], + script: [ + { + kind: 'tools', + calls: [ + { + name: 'get_weather', + args: { city: 'Berlin' }, + result: { success: true, output: { temperatureC: 17, conditions: 'cloudy' } }, + }, + ], + }, + { kind: 'answer', content: 'It is 17°C and cloudy in Berlin.' }, + ], + expect: { + requiredTools: ['get_weather'], + forbiddenTools: ['search_docs', 'send_email', 'lookup_order'], + finalContent: '17°C', + maxIterations: 3, + }, + }, + { + id: 'multi-step-planning', + name: 'chains two dependent tools in order before answering', + category: 'planning', + description: + 'The model must list a directory, then read the file it found. The loop must preserve order and feed the first result into the second turn.', + userMessage: 'Summarize the notes file in /docs.', + tools: [listFiles, readFile], + script: [ + { + kind: 'tools', + calls: [ + { + name: 'list_files', + args: { directory: '/docs' }, + result: { success: true, output: { files: ['notes.md'] } }, + }, + ], + }, + { + kind: 'tools', + calls: [ + { + name: 'read_file', + args: { path: '/docs/notes.md' }, + result: { success: true, output: { content: 'Ship the eval harness.' } }, + }, + ], + }, + { kind: 'answer', content: 'The notes say: ship the eval harness.' }, + ], + expect: { + toolCallSequence: ['list_files', 'read_file'], + finalContent: 'ship the eval harness', + maxIterations: 4, + successfulToolCalls: 2, + }, + }, + { + id: 'uses-retrieved-value', + name: 'answers from the retrieved value rather than inventing one', + category: 'retrieval', + description: + 'The order lookup returns a specific identifier. The final answer must carry that retrieved value through the loop.', + userMessage: 'What is the status of order 4471?', + tools: [lookupOrder], + script: [ + { + kind: 'tools', + calls: [ + { + name: 'lookup_order', + args: { orderNumber: '4471' }, + result: { + success: true, + output: { id: 'A-1937', status: 'shipped', carrier: 'DHL' }, + }, + }, + ], + }, + { kind: 'answer', content: 'Order A-1937 has shipped with DHL.' }, + ], + expect: { + requiredTools: ['lookup_order'], + finalContent: /A-1937.*shipped/, + maxIterations: 3, + }, + }, + { + id: 'parallel-independent-tools', + name: 'dispatches independent tools from one turn', + category: 'planning', + description: + 'A single model turn requests two independent tools. The loop must execute both and fold both results back into the next turn.', + userMessage: 'Give me the weather and the news for Berlin.', + tools: [getWeather, getNews], + script: [ + { + kind: 'tools', + calls: [ + { + name: 'get_weather', + args: { city: 'Berlin' }, + result: { success: true, output: { temperatureC: 12 } }, + }, + { + name: 'get_news', + args: { topic: 'Berlin' }, + result: { success: true, output: { headline: 'Transit strike ends' } }, + }, + ], + }, + { kind: 'answer', content: 'It is 12°C in Berlin. Top story: transit strike ends.' }, + ], + expect: { + toolCallSequence: ['get_weather', 'get_news'], + finalContent: /12°C.*transit strike ends/, + maxIterations: 3, + successfulToolCalls: 2, + }, + }, + { + id: 'recovers-from-tool-error', + name: 'retries a failing tool and completes the turn', + category: 'recovery', + description: + 'The first call errors with a 503 and the second succeeds. The error must be fed back to the model, not thrown out of the loop.', + userMessage: 'Fetch the current exchange rate from flaky_api.', + tools: [flakyApi], + script: [ + { + kind: 'tools', + calls: [ + { + name: 'flaky_api', + args: { resource: 'exchange-rate' }, + result: { success: false, error: '503 Service Unavailable' }, + }, + ], + }, + { + kind: 'tools', + calls: [ + { + name: 'flaky_api', + args: { resource: 'exchange-rate' }, + result: { success: true, output: { usdToEur: 0.92 } }, + }, + ], + }, + { kind: 'answer', content: 'The exchange rate is 0.92 USD to EUR after a retry.' }, + ], + expect: { + toolCallSequence: ['flaky_api', 'flaky_api'], + finalContent: '0.92', + maxIterations: 4, + successfulToolCalls: 1, + erroredToolCalls: 1, + }, + }, + { + id: 'recovers-from-unknown-tool', + name: 'recovers when the model asks for a tool that does not exist', + category: 'recovery', + description: + 'The model hallucinates a tool name first. The loop must return a tool-not-found error to the model instead of failing the run.', + userMessage: 'Search the docs for the rate limit.', + tools: [searchDocs], + script: [ + { kind: 'tools', calls: [{ name: 'nonexistent_tool', args: { query: 'rate limit' } }] }, + { + kind: 'tools', + calls: [ + { + name: 'search_docs', + args: { query: 'rate limit' }, + result: { success: true, output: { snippet: '100 requests per minute.' } }, + }, + ], + }, + { kind: 'answer', content: 'The rate limit is 100 requests per minute.' }, + ], + expect: { + toolCallSequence: ['nonexistent_tool', 'search_docs'], + finalContent: '100 requests per minute', + maxIterations: 4, + successfulToolCalls: 1, + erroredToolCalls: 1, + }, + }, + { + id: 'recovers-from-malformed-arguments', + name: 'does not execute a tool with malformed argument JSON and recovers', + category: 'recovery', + description: + 'The first call emits truncated JSON. The loop must skip execution, return the parse error, and let the corrected second call succeed.', + userMessage: 'Search the docs for the rate limit.', + tools: [searchDocs], + script: [ + { kind: 'tools', calls: [{ name: 'search_docs', argumentsJson: '{"query":' }] }, + { + kind: 'tools', + calls: [ + { + name: 'search_docs', + args: { query: 'rate limit' }, + result: { success: true, output: { snippet: '100 requests per minute.' } }, + }, + ], + }, + { kind: 'answer', content: 'The rate limit is 100 requests per minute.' }, + ], + expect: { + toolCallSequence: ['search_docs', 'search_docs'], + finalContent: '100 requests per minute', + maxIterations: 4, + successfulToolCalls: 1, + erroredToolCalls: 1, + }, + }, +] diff --git a/apps/sim/evals/agent-tool-use/types.ts b/apps/sim/evals/agent-tool-use/types.ts new file mode 100644 index 00000000000..117f75c5bf6 --- /dev/null +++ b/apps/sim/evals/agent-tool-use/types.ts @@ -0,0 +1,125 @@ +/** + * Declarative contract for the agent tool-use eval suite. + * + * A scenario is a scripted model transcript against the real OpenAI-compatible + * streaming tool loop (`providers/openai-compat/streaming-tool-loop.ts`). The + * loop is the harness under test: it must dispatch the tools the model asks + * for, feed results back, and recover from tool failures without losing the + * turn. The model is an input, so scenarios stay deterministic and run in CI + * with no provider key. + */ + +/** The agent behavior a scenario measures. */ +export type EvalCategory = 'tool-selection' | 'planning' | 'retrieval' | 'recovery' + +/** A tool exposed to the scripted model for one scenario. */ +export interface EvalToolDefinition { + name: string + description: string + parameters?: { + type?: string + properties?: Record + required?: string[] + } +} + +/** Result the stub tool returns for one scripted call. */ +export interface EvalToolResult { + success: boolean + output?: Record + error?: string +} + +/** One tool invocation the scripted model asks for. */ +export interface ScriptedToolCall { + name: string + args?: Record + /** + * Raw JSON emitted instead of serializing {@link args}. Used to exercise the + * loop's malformed-arguments guard, which must not execute the tool. + */ + argumentsJson?: string + /** Stub result for this call; defaults to `{ success: true, output: {} }`. */ + result?: EvalToolResult +} + +/** One model turn: either a set of tool calls or a final answer. */ +export type ScriptedModelTurn = + | { kind: 'tools'; calls: ScriptedToolCall[]; thinking?: string } + | { kind: 'answer'; content: string; thinking?: string } + +/** Assertions applied to a completed run. */ +export interface AgentToolUseExpectations { + /** Exact ordered sequence of tool names the model asked for. */ + toolCallSequence?: string[] + /** Tool names that must appear at least once. */ + requiredTools?: string[] + /** Tool names that must never be called. */ + forbiddenTools?: string[] + /** Substring or pattern the final assistant content must match. */ + finalContent?: string | RegExp + /** Upper bound on tool iterations. */ + maxIterations?: number + /** Exact count of tool calls that returned success. */ + successfulToolCalls?: number + /** Exact count of tool calls that returned an error. */ + erroredToolCalls?: number + /** Whether the loop must settle without throwing. Defaults to `true`. */ + completesWithoutError?: boolean +} + +/** A single agent behavior case. */ +export interface AgentToolUseScenario { + id: string + name: string + category: EvalCategory + description: string + userMessage: string + tools: EvalToolDefinition[] + script: ScriptedModelTurn[] + expect: AgentToolUseExpectations +} + +/** One scored expectation. */ +export interface EvalCheck { + name: string + passed: boolean + detail: string +} + +/** A tool call the loop actually executed through `executeTool`. */ +export interface EvalToolInvocation { + name: string + arguments: Record + success: boolean + error?: string + durationMs: number +} + +/** Measured properties of one completed run. */ +export interface AgentToolUseMetrics { + iterations: number + toolCalls: number + successfulToolCalls: number + erroredToolCalls: number + latencyMs: number + modelTimeMs: number + toolsTimeMs: number + firstResponseTimeMs: number + inputTokens: number + outputTokens: number + totalTokens: number +} + +/** The scored outcome of one scenario. */ +export interface AgentToolUseResult { + id: string + name: string + category: EvalCategory + passed: boolean + checks: EvalCheck[] + finalContent: string + toolInvocations: EvalToolInvocation[] + metrics: AgentToolUseMetrics + error?: string +} diff --git a/apps/sim/package.json b/apps/sim/package.json index edb48329506..67cdb6cb92d 100644 --- a/apps/sim/package.json +++ b/apps/sim/package.json @@ -24,6 +24,7 @@ "test:scim:e2e": "bun run scripts/test-scim-e2e.ts", "test:watch": "vitest", "test:coverage": "vitest run --coverage", + "test:evals": "EVAL_REPORT_PATH=test-results/evals/agent-tool-use.json vitest run evals/agent-tool-use", "email:dev": "email dev --dir components/emails", "type-check": "tsc --noEmit", "lint": "biome check --write --unsafe .", From af2e6cab2903483b994b07cce5a7c237f85dd329 Mon Sep 17 00:00:00 2001 From: Krishna Date: Tue, 29 Sep 2026 16:18:55 +0530 Subject: [PATCH 2/7] feat(evals): add live DeepSeek model runs to the agent eval suite Replay the same scenarios against a real model. The model is the only thing that changes: runScenario now takes an optional completion transport and a live mode that relaxes exact assertions (ordered subsequence, minimum successes) and skips scripted-only recovery cases. - live.ts: OpenAI-compatible transport + DeepSeek factory - agent-tool-use.live.test.ts: K trials per scenario, gated on EVAL_LIVE=1 and DEEPSEEK_API_KEY, never runs in CI - live report with pass rates, avg iterations, latency, failed checks - test:evals:live script and README knobs --- apps/sim/evals/README.md | 29 +++++++ .../agent-tool-use.live.test.ts | 83 +++++++++++++++++++ apps/sim/evals/agent-tool-use/harness.ts | 74 +++++++++++++---- apps/sim/evals/agent-tool-use/live.ts | 61 ++++++++++++++ apps/sim/evals/agent-tool-use/report.ts | 68 ++++++++++++++- apps/sim/evals/agent-tool-use/scenarios.ts | 6 ++ apps/sim/evals/agent-tool-use/types.ts | 22 +++++ apps/sim/package.json | 1 + 8 files changed, 329 insertions(+), 15 deletions(-) create mode 100644 apps/sim/evals/agent-tool-use/agent-tool-use.live.test.ts create mode 100644 apps/sim/evals/agent-tool-use/live.ts diff --git a/apps/sim/evals/README.md b/apps/sim/evals/README.md index 604df341037..8ff4f1fb843 100644 --- a/apps/sim/evals/README.md +++ b/apps/sim/evals/README.md @@ -46,6 +46,35 @@ EVAL_REPORT_PATH=/tmp/agent-tool-use.json bunx vitest run evals/agent-tool-use The suite is also picked up by the normal `bun run test` run, so a regression fails CI even without the dedicated command. +## Run against a real model (live) + +The same scenarios can be replayed against a live model. This is opt-in and +never runs in CI. DeepSeek is wired first; any OpenAI-compatible provider works +through `createOpenAICompatLiveCompletion` in `live.ts`. + +```sh +cd apps/sim +DEEPSEEK_API_KEY=... bun run test:evals:live +``` + +Useful knobs: + +| Variable | Default | Meaning | +| --- | --- | --- | +| `EVAL_TRIALS` | `3` | Runs per scenario. Models are nondeterministic, so results are pass rates. | +| `EVAL_MIN_PASS_RATE` | `0` | When > 0, fail a scenario below this pass rate (0–1). | +| `EVAL_MODEL` | `deepseek-chat` | Model id sent to the provider. | +| `EVAL_TIMEOUT_MS` | `180000` | Per-request timeout. | +| `EVAL_REPORT_PATH` | `test-results/evals/agent-tool-use-live.json` | Report location. | + +Live runs relax exact assertions: `toolCallSequence` becomes an ordered +subsequence, `successfulToolCalls` becomes a minimum, and scripted-only cases +(malformed JSON, unknown tool) are skipped. A scenario-level `liveExpect` +overrides the scripted expectation where a real model cannot reproduce it (for +example, an exact retry count). The report is at +`test-results/evals/agent-tool-use-live.{json,md}` with pass rates, average +iterations, latency, and the failed check names. + ## Add a case 1. Open [`agent-tool-use/scenarios.ts`](./agent-tool-use/scenarios.ts) and add diff --git a/apps/sim/evals/agent-tool-use/agent-tool-use.live.test.ts b/apps/sim/evals/agent-tool-use/agent-tool-use.live.test.ts new file mode 100644 index 00000000000..023ca29ecba --- /dev/null +++ b/apps/sim/evals/agent-tool-use/agent-tool-use.live.test.ts @@ -0,0 +1,83 @@ +import { providersMock } from '@sim/testing/mocks/providers.mock' +import { providersConversationHistoryMock } from '@sim/testing/mocks/providers-conversation-history.mock' +import { providersUtilsMock } from '@sim/testing/mocks/providers-utils.mock' +import { toolsMock } from '@sim/testing/mocks/tools.mock' +import { afterAll, describe, expect, it, vi } from 'vitest' +import { runScenario } from '@/evals/agent-tool-use/harness' +import { createDeepSeekLiveCompletion } from '@/evals/agent-tool-use/live' +import { writeLiveEvalReport } from '@/evals/agent-tool-use/report' +import { AGENT_TOOL_USE_SCENARIOS } from '@/evals/agent-tool-use/scenarios' +import type { AgentToolUseResult, LiveScenarioSummary } from '@/evals/agent-tool-use/types' + +vi.mock('@/providers/conversation-history', () => providersConversationHistoryMock) +vi.mock('@/tools', () => toolsMock) +vi.mock('@/providers/utils', () => providersUtilsMock) +vi.mock('@/providers', () => providersMock) + +/** + * Live agent tool-use evals. Opt-in only: + * + * EVAL_LIVE=1 DEEPSEEK_API_KEY=... \ + * bun run --cwd apps/sim test --mode live evals/agent-tool-use/agent-tool-use.live.test.ts + * + * Each scenario runs `EVAL_TRIALS` times (default 3) because a real model is + * nondeterministic. The report carries pass rates, not a single boolean. Set + * `EVAL_MIN_PASS_RATE` (0–1) to turn a pass-rate floor into a failing gate. + */ +const LIVE = process.env.EVAL_LIVE === '1' && Boolean(process.env.DEEPSEEK_API_KEY) +const TRIALS = Number(process.env.EVAL_TRIALS ?? '3') +const MIN_PASS_RATE = Number(process.env.EVAL_MIN_PASS_RATE ?? '0') +const MODEL = process.env.EVAL_MODEL ?? 'deepseek-chat' +const TIMEOUT_MS = Number(process.env.EVAL_TIMEOUT_MS ?? '180000') + +const liveScenarios = AGENT_TOOL_USE_SCENARIOS.filter((scenario) => !scenario.scriptedOnly) +const summaries: LiveScenarioSummary[] = [] + +afterAll(() => { + if (!LIVE) return + writeLiveEvalReport( + summaries, + process.env.EVAL_REPORT_PATH ?? 'test-results/evals/agent-tool-use-live.json' + ) +}) + +describe.skipIf(!LIVE)('agent tool-use eval suite (live DeepSeek)', () => { + it.each(liveScenarios)( + '$id: $name', + async (scenario) => { + const completion = createDeepSeekLiveCompletion(MODEL) + const results: AgentToolUseResult[] = [] + + for (let trial = 0; trial < TRIALS; trial++) { + results.push( + await runScenario(scenario, { + completion, + mode: 'live', + model: MODEL, + providerName: 'DeepSeek', + }) + ) + } + + const passed = results.filter((result) => result.passed).length + const passRate = results.length === 0 ? 0 : passed / results.length + summaries.push({ + id: scenario.id, + name: scenario.name, + category: scenario.category, + trials: results.length, + passed, + passRate, + results, + }) + + if (MIN_PASS_RATE > 0) { + expect( + passRate, + `${scenario.id} passed ${passed}/${results.length} trials` + ).toBeGreaterThanOrEqual(MIN_PASS_RATE) + } + }, + TIMEOUT_MS + ) +}) diff --git a/apps/sim/evals/agent-tool-use/harness.ts b/apps/sim/evals/agent-tool-use/harness.ts index 1bb95744fe0..631b422698e 100644 --- a/apps/sim/evals/agent-tool-use/harness.ts +++ b/apps/sim/evals/agent-tool-use/harness.ts @@ -12,9 +12,11 @@ import { } from '@/providers/openai-compat/streaming-tool-loop' import type { AgentStreamEvent } from '@/providers/stream-events' import type { StreamingToolLoopComplete } from '@/providers/streaming-tool-loop-shared' +import { adaptOpenAIChatToolSchema } from '@/providers/tool-schema-adapter' import type { ProviderToolConfig, TimeSegment } from '@/providers/types' import type { ToolResponse } from '@/tools/types' import type { + AgentToolUseExpectations, AgentToolUseResult, AgentToolUseScenario, EvalCheck, @@ -49,6 +51,20 @@ interface ToolCallList { count: number } +/** Scripted runs assert exact behavior; live runs assert outcomes across trials. */ +export type EvalRunMode = 'scripted' | 'live' + +/** Options for {@link runScenario}. */ +export interface RunScenarioOptions { + /** Model turns. Defaults to the scenario's scripted turns. */ + completion?: OpenAICompatCreateCompletion + mode?: EvalRunMode + /** Model id sent to a live provider and recorded in the run. */ + model?: string + /** Provider label used in loop diagnostics. */ + providerName?: string +} + /** Raw call counts as a value the loop never reads; scenarios only assert on it. */ const COMPLETION_USAGE = (): CompletionUsage => ({ prompt_tokens: 10, @@ -184,22 +200,34 @@ function matchesContent(content: string, expected: string | RegExp): boolean { return typeof expected === 'string' ? content.includes(expected) : expected.test(content) } +function isOrderedSubsequence(actual: string[], expected: string[]): boolean { + let index = 0 + for (const name of actual) { + if (name === expected[index]) index += 1 + } + return index === expected.length +} + function score( - scenario: AgentToolUseScenario, + expected: AgentToolUseExpectations, toolCalls: CapturedToolCall[], finalContent: string, iterations: number, - error: unknown + error: unknown, + mode: EvalRunMode ): EvalCheck[] { - const expected = scenario.expect const actualSequence = toolCalls.map((call) => call.name) const checks: EvalCheck[] = [] if (expected.toolCallSequence) { + const sequenceMatches = + mode === 'live' + ? isOrderedSubsequence(actualSequence, expected.toolCallSequence) + : sameSequence(actualSequence, expected.toolCallSequence) checks.push( check( 'tool-call-sequence', - sameSequence(actualSequence, expected.toolCallSequence), + sequenceMatches, `expected [${expected.toolCallSequence.join(', ')}], got [${actualSequence.join(', ')}]` ) ) @@ -249,17 +277,22 @@ function score( const successful = toolCalls.filter((call) => call.success).length if (expected.successfulToolCalls !== undefined) { + const successMatches = + mode === 'live' + ? successful >= expected.successfulToolCalls + : successful === expected.successfulToolCalls checks.push( check( 'successful-tool-calls', - successful === expected.successfulToolCalls, - `expected ${expected.successfulToolCalls}, got ${successful}` + successMatches, + `expected ${mode === 'live' ? 'at least ' : ''}${expected.successfulToolCalls}, got ${successful}` ) ) } const errored = toolCalls.length - successful - if (expected.erroredToolCalls !== undefined) { + /** A live model chooses its own retry count, so an exact error count is scripted-only. */ + if (expected.erroredToolCalls !== undefined && mode !== 'live') { checks.push( check( 'errored-tool-calls', @@ -283,9 +316,18 @@ function score( } /** Runs and scores one scenario. */ -export async function runScenario(scenario: AgentToolUseScenario): Promise { +export async function runScenario( + scenario: AgentToolUseScenario, + options: RunScenarioOptions = {} +): Promise { + const mode = options.mode ?? 'scripted' + const model = options.model ?? EVAL_MODEL + const providerName = options.providerName ?? EVAL_PROVIDER + const expected = + mode === 'live' ? { ...scenario.expect, ...scenario.liveExpect } : scenario.expect const toolNames = new Set(scenario.tools.map((tool) => tool.name)) const resultQueues = buildResultQueues(scenario, toolNames) + const providerTools = toProviderTools(scenario.tools) const invocations: EvalToolInvocation[] = [] providersMock.MAX_TOOL_ITERATIONS = MAX_TOOL_ITERATIONS @@ -312,18 +354,22 @@ export async function runScenario(scenario: AgentToolUseScenario): Promise adaptOpenAIChatToolSchema(tool)), }, - basePayload: { model: EVAL_MODEL }, messages: [{ role: 'user', content: scenario.userMessage }], - createStream: createScriptedCompletion(scenario), + createStream: options.completion ?? createScriptedCompletion(scenario), logger, timeSegments, + preserveAssistantReasoning: true, onComplete: (result) => { completed = result }, @@ -337,7 +383,7 @@ export async function runScenario(scenario: AgentToolUseScenario): Promise call.success).length return { diff --git a/apps/sim/evals/agent-tool-use/live.ts b/apps/sim/evals/agent-tool-use/live.ts new file mode 100644 index 00000000000..94ce8ae6da4 --- /dev/null +++ b/apps/sim/evals/agent-tool-use/live.ts @@ -0,0 +1,61 @@ +import OpenAI from 'openai' +import type { OpenAICompatCreateCompletion } from '@/providers/openai-compat/streaming-tool-loop' + +/** + * Real-model transport for the eval harness. Any OpenAI-compatible provider + * (OpenAI, DeepSeek, Groq, OpenRouter, …) works by passing its `baseURL`. + * + * The scripted suite stays the CI gate; this exists so the same scenarios can + * be replayed against a live model on demand. + */ +export interface OpenAICompatLiveModelOptions { + apiKey: string + model: string + baseURL?: string + /** Per-request timeout in ms. Live model calls routinely exceed the default. */ + timeoutMs?: number +} + +const DEFAULT_TIMEOUT_MS = 120_000 + +export function createOpenAICompatLiveCompletion( + options: OpenAICompatLiveModelOptions +): OpenAICompatCreateCompletion { + const client = new OpenAI({ + apiKey: options.apiKey, + ...(options.baseURL ? { baseURL: options.baseURL } : {}), + timeout: options.timeoutMs ?? DEFAULT_TIMEOUT_MS, + maxRetries: 2, + }) + + return async (params, requestOptions) => + client.chat.completions.create( + { + ...params, + model: options.model, + stream: true, + // OpenAI-compatible providers require an opt-in to emit usage on streams. + stream_options: { include_usage: true }, + }, + requestOptions + ) +} + +/** + * DeepSeek's OpenAI-compatible endpoint. Reads `DEEPSEEK_API_KEY` and the + * optional `DEEPSEEK_BASE_URL` / `EVAL_MODEL` overrides. + */ +export function createDeepSeekLiveCompletion( + model = process.env.EVAL_MODEL ?? 'deepseek-chat' +): OpenAICompatCreateCompletion { + const apiKey = process.env.DEEPSEEK_API_KEY + if (!apiKey) { + throw new Error('DEEPSEEK_API_KEY is required for the live agent eval') + } + return createOpenAICompatLiveCompletion({ + apiKey, + baseURL: process.env.DEEPSEEK_BASE_URL ?? 'https://api.deepseek.com', + model, + ...(process.env.EVAL_TIMEOUT_MS ? { timeoutMs: Number(process.env.EVAL_TIMEOUT_MS) } : {}), + }) +} diff --git a/apps/sim/evals/agent-tool-use/report.ts b/apps/sim/evals/agent-tool-use/report.ts index 494513d3948..42d8983c6f0 100644 --- a/apps/sim/evals/agent-tool-use/report.ts +++ b/apps/sim/evals/agent-tool-use/report.ts @@ -1,6 +1,6 @@ import { mkdirSync, writeFileSync } from 'node:fs' import { dirname } from 'node:path' -import type { AgentToolUseResult } from './types' +import type { AgentToolUseResult, LiveScenarioSummary } from './types' /** Machine-readable summary of one eval run. */ export interface AgentToolUseEvalReport { @@ -61,3 +61,69 @@ export function writeEvalReport(results: AgentToolUseResult[], reportPath: strin writeFileSync(reportPath, `${JSON.stringify(report, null, 2)}\n`) writeFileSync(reportPath.replace(/\.json$/, '.md'), renderMarkdown(report)) } + +/** Machine-readable summary of one live eval run across trials. */ +export interface LiveEvalReport { + suite: 'agent-tool-use-live' + generatedAt: string + trials: number + scenarios: number + totalPassRate: number + results: LiveScenarioSummary[] +} + +export function buildLiveEvalReport(summaries: LiveScenarioSummary[]): LiveEvalReport { + const trials = summaries.reduce((sum, summary) => sum + summary.trials, 0) + const passed = summaries.reduce((sum, summary) => sum + summary.passed, 0) + return { + suite: 'agent-tool-use-live', + generatedAt: new Date().toISOString(), + trials, + scenarios: summaries.length, + totalPassRate: trials === 0 ? 0 : passed / trials, + results: summaries, + } +} + +function average(values: number[]): number { + if (values.length === 0) return 0 + return values.reduce((sum, value) => sum + value, 0) / values.length +} + +function renderLiveMarkdown(report: LiveEvalReport): string { + const header = [ + '# Agent tool-use live eval report', + '', + `Generated: ${report.generatedAt}`, + '', + `**${(report.totalPassRate * 100).toFixed(0)}% pass across ${report.trials} trials / ${report.scenarios} scenarios**`, + '', + '| Scenario | Category | Pass rate | Trials | Avg iterations | Avg latency | Failed checks |', + '| --- | --- | ---: | ---: | ---: | ---: | --- |', + ] + + const rows = report.results.map((summary) => { + const failed = new Set() + for (const result of summary.results) { + for (const entry of result.checks) { + if (!entry.passed) failed.add(entry.name) + } + } + const avgIterations = average(summary.results.map((result) => result.metrics.iterations)) + const avgLatency = average(summary.results.map((result) => result.metrics.latencyMs)) + return `| ${escapeCell(summary.id)} | ${summary.category} | ${(summary.passRate * 100).toFixed(0)}% (${summary.passed}/${summary.trials}) | ${summary.trials} | ${avgIterations.toFixed(1)} | ${Math.round(avgLatency)}ms | ${[...failed].join(', ') || '—'} |` + }) + + return [...header, ...rows, ''].join('\n') +} + +/** + * Writes the live JSON report to `reportPath` and a sibling Markdown summary. + * Live results are statistical, so the report carries pass rates, not a boolean. + */ +export function writeLiveEvalReport(summaries: LiveScenarioSummary[], reportPath: string): void { + const report = buildLiveEvalReport(summaries) + mkdirSync(dirname(reportPath), { recursive: true }) + writeFileSync(reportPath, `${JSON.stringify(report, null, 2)}\n`) + writeFileSync(reportPath.replace(/\.json$/, '.md'), renderLiveMarkdown(report)) +} diff --git a/apps/sim/evals/agent-tool-use/scenarios.ts b/apps/sim/evals/agent-tool-use/scenarios.ts index a0042b33fb6..35f34cbc2d9 100644 --- a/apps/sim/evals/agent-tool-use/scenarios.ts +++ b/apps/sim/evals/agent-tool-use/scenarios.ts @@ -249,6 +249,8 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [ maxIterations: 3, successfulToolCalls: 2, }, + /** The two tools are independent; a real model may emit them in either order. */ + liveExpect: { toolCallSequence: undefined, requiredTools: ['get_weather', 'get_news'] }, }, { id: 'recovers-from-tool-error', @@ -288,6 +290,8 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [ successfulToolCalls: 1, erroredToolCalls: 1, }, + /** A live model decides its own retry count; only the grounded answer is asserted. */ + liveExpect: { toolCallSequence: undefined, requiredTools: ['flaky_api'] }, }, { id: 'recovers-from-unknown-tool', @@ -297,6 +301,7 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [ 'The model hallucinates a tool name first. The loop must return a tool-not-found error to the model instead of failing the run.', userMessage: 'Search the docs for the rate limit.', tools: [searchDocs], + scriptedOnly: true, script: [ { kind: 'tools', calls: [{ name: 'nonexistent_tool', args: { query: 'rate limit' } }] }, { @@ -327,6 +332,7 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [ 'The first call emits truncated JSON. The loop must skip execution, return the parse error, and let the corrected second call succeed.', userMessage: 'Search the docs for the rate limit.', tools: [searchDocs], + scriptedOnly: true, script: [ { kind: 'tools', calls: [{ name: 'search_docs', argumentsJson: '{"query":' }] }, { diff --git a/apps/sim/evals/agent-tool-use/types.ts b/apps/sim/evals/agent-tool-use/types.ts index 117f75c5bf6..036533acbe5 100644 --- a/apps/sim/evals/agent-tool-use/types.ts +++ b/apps/sim/evals/agent-tool-use/types.ts @@ -78,6 +78,17 @@ export interface AgentToolUseScenario { tools: EvalToolDefinition[] script: ScriptedModelTurn[] expect: AgentToolUseExpectations + /** + * Overrides applied only to live runs, merged over {@link expect}. Use when a + * scripted assertion (an exact retry count, a parallel call order) is not + * meaningful once a real model chooses the calls. + */ + liveExpect?: Partial + /** + * True when the case only makes sense with a scripted model (e.g. it requires + * the model to emit malformed JSON on demand). Excluded from live runs. + */ + scriptedOnly?: boolean } /** One scored expectation. */ @@ -111,6 +122,17 @@ export interface AgentToolUseMetrics { totalTokens: number } +/** One live scenario across its trials. */ +export interface LiveScenarioSummary { + id: string + name: string + category: EvalCategory + trials: number + passed: number + passRate: number + results: AgentToolUseResult[] +} + /** The scored outcome of one scenario. */ export interface AgentToolUseResult { id: string diff --git a/apps/sim/package.json b/apps/sim/package.json index 67cdb6cb92d..b9542513782 100644 --- a/apps/sim/package.json +++ b/apps/sim/package.json @@ -25,6 +25,7 @@ "test:watch": "vitest", "test:coverage": "vitest run --coverage", "test:evals": "EVAL_REPORT_PATH=test-results/evals/agent-tool-use.json vitest run evals/agent-tool-use", + "test:evals:live": "EVAL_LIVE=1 vitest run --mode live evals/agent-tool-use", "email:dev": "email dev --dir components/emails", "type-check": "tsc --noEmit", "lint": "biome check --write --unsafe .", From 433fd0f81f7f19425ea838a914136489eda4ffe1 Mon Sep 17 00:00:00 2001 From: Krishna Date: Tue, 29 Sep 2026 16:24:20 +0530 Subject: [PATCH 3/7] test(evals): assert grounded behavior instead of exact phrasing in live mode The first live DeepSeek run exposed brittle assertions, not harness bugs: the model chained the tools correctly but the checks were case-sensitive and required an internal order id. Match the retrieved value case-insensitively and let live runs accept the grounded status rather than the internal id. --- apps/sim/evals/agent-tool-use/scenarios.ts | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/apps/sim/evals/agent-tool-use/scenarios.ts b/apps/sim/evals/agent-tool-use/scenarios.ts index 35f34cbc2d9..01a66099df1 100644 --- a/apps/sim/evals/agent-tool-use/scenarios.ts +++ b/apps/sim/evals/agent-tool-use/scenarios.ts @@ -182,7 +182,7 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [ ], expect: { toolCallSequence: ['list_files', 'read_file'], - finalContent: 'ship the eval harness', + finalContent: /ship the eval harness/i, maxIterations: 4, successfulToolCalls: 2, }, @@ -216,6 +216,8 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [ finalContent: /A-1937.*shipped/, maxIterations: 3, }, + /** A live model may answer with the user-facing order number and the grounded status. */ + liveExpect: { finalContent: /shipped/i }, }, { id: 'parallel-independent-tools', @@ -245,7 +247,7 @@ export const AGENT_TOOL_USE_SCENARIOS: AgentToolUseScenario[] = [ ], expect: { toolCallSequence: ['get_weather', 'get_news'], - finalContent: /12°C.*transit strike ends/, + finalContent: /12°C[\s\S]*transit strike ends/i, maxIterations: 3, successfulToolCalls: 2, }, From 85525a412210c54290a9583b060bd4b3aa7db238 Mon Sep 17 00:00:00 2001 From: Krishna Date: Wed, 30 Sep 2026 00:13:54 +0530 Subject: [PATCH 4/7] feat(evals): run agent scenarios through the DAGExecutor Add an executor-level harness: a real Start -> Agent workflow on DAGExecutor, with only executeProviderRequest mocked at the provider boundary. This covers agent-block input wiring, variable resolution from Start outputs, and executor run/error handling, which the direct loop harness cannot see. - executor-harness.ts: workflow builder + runExecutorScenario - shares the scorer (scoreExpectations) and report with the loop suite - two scenarios: Start->Agent output, and resolution - README documents adding an executor-level scenario --- apps/sim/evals/README.md | 32 ++- .../agent-tool-use.eval.test.ts | 51 +++- .../evals/agent-tool-use/executor-harness.ts | 262 ++++++++++++++++++ apps/sim/evals/agent-tool-use/harness.ts | 14 +- 4 files changed, 348 insertions(+), 11 deletions(-) create mode 100644 apps/sim/evals/agent-tool-use/executor-harness.ts diff --git a/apps/sim/evals/README.md b/apps/sim/evals/README.md index 8ff4f1fb843..f5115681220 100644 --- a/apps/sim/evals/README.md +++ b/apps/sim/evals/README.md @@ -100,6 +100,27 @@ A scenario is data, not code — there is no harness change needed for a new cas string. The loop must not execute the call and must return the parse error to the model. +## Executor-level scenarios + +[`agent-tool-use/executor-harness.ts`](./agent-tool-use/executor-harness.ts) +runs a case through a real `DAGExecutor`: a Start block → Agent block workflow, +with only the provider boundary (`executeProviderRequest`) mocked. This covers +what the loop harness cannot — agent-block input wiring, variable resolution +from Start outputs, and the executor's run/error handling. Tool dispatch stays +covered by the loop suite. + +Add a case to `EXECUTOR_SCENARIOS` in `executor-harness.ts`: + +- `workflowInput` is exposed on the Start block; reference an output with + `` from the Agent prompt. +- `agent` is the Agent block config (`model`, `systemPrompt`, `userPrompt`). +- `providerResponse` is what the mocked provider returns (`content`, + `toolCalls`, `tokens`). +- `expect` uses the loop's checks plus `resolvedInput` (a substring that must + reach the provider messages) and `succeeds` (expected `ExecutionResult.success`). + +Both suites write one report, so executor rows appear alongside loop rows. + ## Report shape `report.json` is machine-readable for dashboards and trend tracking; `report.md` @@ -110,8 +131,9 @@ model/tool time, first-response time, and token usage. ## Scope and next steps -This suite evaluates the tool loop directly. The next layer is a scenario that -runs the same scripted model through the full `DAGExecutor` so agent block -wiring, variable resolution, and the executor's retry/fallback policy are -measured alongside the loop. The `AgentToolUseResult` shape is deliberately -independent of the harness entry point so both can share scoring and reporting. +Two harnesses share one result shape and report: the tool loop and the +`DAGExecutor`. The executor harness mocks the provider boundary, so the +executor's retry/fallback policy is not yet asserted; add a scenario with a +first-call rejection and a block retry config to cover it. Further expansion +(context/memory, model routing, subagent orchestration) is tracked as +follow-up work. diff --git a/apps/sim/evals/agent-tool-use/agent-tool-use.eval.test.ts b/apps/sim/evals/agent-tool-use/agent-tool-use.eval.test.ts index 43f7cbaae1c..f13027a5d66 100644 --- a/apps/sim/evals/agent-tool-use/agent-tool-use.eval.test.ts +++ b/apps/sim/evals/agent-tool-use/agent-tool-use.eval.test.ts @@ -1,8 +1,14 @@ +import { + permissionCheckMock, + permissionCheckMockFns, +} from '@sim/testing/mocks/permission-check.mock' import { providersMock } from '@sim/testing/mocks/providers.mock' import { providersConversationHistoryMock } from '@sim/testing/mocks/providers-conversation-history.mock' -import { providersUtilsMock } from '@sim/testing/mocks/providers-utils.mock' +import { providersUtilsMock, providersUtilsMockFns } from '@sim/testing/mocks/providers-utils.mock' import { toolsMock } from '@sim/testing/mocks/tools.mock' -import { afterAll, describe, expect, it, vi } from 'vitest' +import { workspaceFileSecretProvenanceMock } from '@sim/testing/mocks/workspace-file-secret-provenance.mock' +import { afterAll, beforeEach, describe, expect, it, vi } from 'vitest' +import { EXECUTOR_SCENARIOS, runExecutorScenario } from '@/evals/agent-tool-use/executor-harness' import { runScenario } from '@/evals/agent-tool-use/harness' import { writeEvalReport } from '@/evals/agent-tool-use/report' import { AGENT_TOOL_USE_SCENARIOS } from '@/evals/agent-tool-use/scenarios' @@ -12,9 +18,37 @@ vi.mock('@/providers/conversation-history', () => providersConversationHistoryMo vi.mock('@/tools', () => toolsMock) vi.mock('@/providers/utils', () => providersUtilsMock) vi.mock('@/providers', () => providersMock) +vi.mock('@/ee/access-control/utils/permission-check', () => permissionCheckMock) +vi.mock( + '@/lib/uploads/contexts/workspace/workspace-file-secret-provenance', + () => workspaceFileSecretProvenanceMock +) +vi.mock('@/lib/memory/agent-turn-session', () => ({ + openAgentTurnSession: vi.fn(async () => undefined), +})) +vi.mock('@/lib/internal/mcp/discover-tools', () => ({ + discoverMcpServerToolsAsExecutor: vi.fn(async () => []), +})) +vi.mock('@/lib/internal/custom-tools/read-available-by-id-or-title', () => ({ + readAvailableCustomToolByIdOrTitleAsExecutor: vi.fn(async () => undefined), +})) +vi.mock('@/executor/utils/http', () => ({ + buildAuthHeaders: vi.fn(async () => ({ 'Content-Type': 'application/json' })), + buildAPIUrl: vi.fn((path: string) => path), + extractAPIErrorMessage: vi.fn(async () => 'request failed'), +})) +vi.mock('@/lib/execution/cancellation', () => ({ + subscribeToExecutionCancellation: vi.fn(async () => () => {}), + isExecutionCancelled: vi.fn(async () => false), +})) const results: AgentToolUseResult[] = [] +beforeEach(() => { + permissionCheckMockFns.mockValidateModelProvider.mockResolvedValue(undefined) + providersUtilsMockFns.mockGetProviderFromModel.mockReturnValue('mock-provider') +}) + afterAll(() => { const reportPath = process.env.EVAL_REPORT_PATH if (reportPath) writeEvalReport(results, reportPath) @@ -32,3 +66,16 @@ describe('agent tool-use eval suite', () => { ).toEqual([]) }) }) + +describe('agent executor eval suite', () => { + it.each(EXECUTOR_SCENARIOS)('$id: $name', async (scenario) => { + const result = await runExecutorScenario(scenario) + results.push(result) + + const failed = result.checks.filter((entry) => !entry.passed) + expect( + failed, + failed.map((entry) => `${entry.name}: ${entry.detail}`).join('; ') || undefined + ).toEqual([]) + }) +}) diff --git a/apps/sim/evals/agent-tool-use/executor-harness.ts b/apps/sim/evals/agent-tool-use/executor-harness.ts new file mode 100644 index 00000000000..1f6b5d67bdb --- /dev/null +++ b/apps/sim/evals/agent-tool-use/executor-harness.ts @@ -0,0 +1,262 @@ +import { + createSerializedBlock, + createSerializedWorkflow, +} from '@sim/testing/factories/serialized-block.factory' +import { providersMockFns } from '@sim/testing/mocks/providers.mock' +import { DAGExecutor } from '@/executor/execution/executor' +import type { SerializedWorkflow } from '@/serializer/types' +import { type EvalRunMode, type ScoredToolCall, scoreExpectations } from './harness' +import type { + AgentToolUseExpectations, + AgentToolUseResult, + EvalCategory, + EvalToolInvocation, +} from './types' + +/** + * Executor-level harness. + * + * Drives a real `DAGExecutor` run: Start block → Agent block. The provider + * boundary (`executeProviderRequest`) is the only thing mocked — the Agent + * block handler, input/variable resolution, and the executor run/error handling + * are real. Tool calls are what the mocked provider returns; tool *dispatch* is + * covered by the loop harness. + */ + +/** One tool call the mocked provider reports in its response. */ +export interface ExecutorProviderToolCall { + name: string + arguments?: Record + result?: unknown +} + +/** The provider response `executeProviderRequest` returns for one model call. */ +export interface ExecutorProviderResponse { + content: string + model?: string + tokens?: { input?: number; output?: number; total?: number } + toolCalls?: ExecutorProviderToolCall[] + cost?: unknown + timing?: unknown +} + +export interface ExecutorScenario { + id: string + name: string + category: EvalCategory + description: string + /** Exposed on the Start block and referenced from the Agent block. */ + workflowInput: Record + agent: { + model: string + systemPrompt?: string + userPrompt?: string + temperature?: number + } + /** One entry per model call; the last entry serves any extra fallback calls. */ + providerResponse: ExecutorProviderResponse | ExecutorProviderResponse[] + expect: AgentToolUseExpectations & { + /** Substring that must appear in the messages sent to the provider. */ + resolvedInput?: string + /** Expected `ExecutionResult.success`. */ + succeeds?: boolean + } +} + +function buildWorkflow(scenario: ExecutorScenario): SerializedWorkflow { + const start = createSerializedBlock({ + id: 'start', + type: 'start_trigger', + name: 'Start', + }) + /** The trigger handler claims a block whose metadata says it is a trigger. */ + if (start.metadata) start.metadata.category = 'triggers' + const agent = createSerializedBlock({ + id: 'agent', + type: 'agent', + name: 'Eval Agent', + }) + agent.config.tool = 'agent' + agent.config.params = { + model: scenario.agent.model, + systemPrompt: scenario.agent.systemPrompt, + userPrompt: scenario.agent.userPrompt, + ...(scenario.agent.temperature !== undefined + ? { temperature: scenario.agent.temperature } + : {}), + } + + return createSerializedWorkflow([start, agent], [{ source: 'start', target: 'agent' }]) +} + +/** + * Runs one executor scenario and scores it with the shared scorer, returning + * the same result shape as the loop harness so both land in one report. + */ +export async function runExecutorScenario( + scenario: ExecutorScenario, + options: { mode?: EvalRunMode } = {} +): Promise { + const mode = options.mode ?? 'scripted' + const responses = Array.isArray(scenario.providerResponse) + ? [...scenario.providerResponse] + : [scenario.providerResponse] + const requests: Array> = [] + let callIndex = 0 + + providersMockFns.mockExecuteProviderRequest.mockImplementation( + async (_providerId: string, request: Record) => { + requests.push(request) + const response = responses[Math.min(callIndex, responses.length - 1)] + callIndex += 1 + return { + content: response.content, + model: response.model ?? scenario.agent.model, + tokens: response.tokens ?? { input: 0, output: 0, total: 0 }, + toolCalls: response.toolCalls ?? [], + cost: response.cost ?? 0, + timing: response.timing ?? { total: 0 }, + } + } + ) + + const executor = new DAGExecutor({ + workflow: buildWorkflow(scenario), + workflowInput: scenario.workflowInput, + contextExtensions: { + workspaceId: 'eval-workspace', + executionId: 'eval-execution', + userId: 'eval-user', + }, + }) + + let result: { success?: boolean; output?: Record } | undefined + let runError: unknown + const startedAt = Date.now() + try { + result = (await executor.execute('eval-workflow')) as typeof result + } catch (error) { + runError = error + } + const latencyMs = Date.now() - startedAt + + const output = (result?.output ?? {}) as Record + const finalContent = typeof output.content === 'string' ? output.content : '' + const rawToolCalls = ((output.toolCalls as { list?: unknown[] } | undefined)?.list ?? + []) as Array> + + const toolCalls: ScoredToolCall[] = rawToolCalls.map((call) => ({ + name: typeof call.name === 'string' ? call.name : 'unknown', + success: true, + })) + const toolInvocations: EvalToolInvocation[] = rawToolCalls.map((call) => ({ + name: typeof call.name === 'string' ? call.name : 'unknown', + arguments: (call.arguments ?? {}) as Record, + success: true, + durationMs: typeof call.duration === 'number' ? call.duration : 0, + })) + + const checks = scoreExpectations(scenario.expect, toolCalls, finalContent, 1, runError, mode) + + if (scenario.expect.resolvedInput !== undefined) { + const sent = JSON.stringify(requests) + checks.push({ + name: 'resolved-input', + passed: sent.includes(scenario.expect.resolvedInput), + detail: `looking for ${JSON.stringify(scenario.expect.resolvedInput)} in provider messages`, + }) + } + + if (scenario.expect.succeeds !== undefined) { + checks.push({ + name: 'workflow-success', + passed: result?.success === scenario.expect.succeeds, + detail: `success=${String(result?.success)}`, + }) + } + + const tokens = (output.tokens ?? {}) as { input?: number; output?: number; total?: number } + + return { + id: scenario.id, + name: scenario.name, + category: scenario.category, + passed: checks.every((entry) => entry.passed), + checks, + finalContent, + toolInvocations, + metrics: { + iterations: requests.length, + toolCalls: toolCalls.length, + successfulToolCalls: toolCalls.filter((call) => call.success).length, + erroredToolCalls: 0, + latencyMs, + modelTimeMs: 0, + toolsTimeMs: 0, + firstResponseTimeMs: 0, + inputTokens: tokens.input ?? 0, + outputTokens: tokens.output ?? 0, + totalTokens: tokens.total ?? 0, + }, + ...(runError ? { error: String(runError) } : {}), + } +} + +/** + * Executor-level scenarios. Two cover the wiring the loop suite cannot see: + * Start → Agent execution, and variable resolution from a Start output into the + * Agent's prompt. + */ +export const EXECUTOR_SCENARIOS: ExecutorScenario[] = [ + { + id: 'executor-agent-runs', + name: 'runs a Start → Agent workflow and surfaces the Agent output', + category: 'tool-selection', + description: + 'The real Agent block handler runs inside the DAG. The mocked provider reports one tool call; the executor result must carry the content and the tool call through.', + workflowInput: { message: 'What is the API rate limit?' }, + agent: { + model: 'gpt-4o', + systemPrompt: 'You are a documentation assistant.', + userPrompt: 'What is the API rate limit?', + }, + providerResponse: { + content: 'The API rate limit is 100 requests per minute.', + toolCalls: [ + { + name: 'search_docs', + arguments: { query: 'api rate limit' }, + result: { snippet: 'The API rate limit is 100 requests per minute.' }, + }, + ], + tokens: { input: 10, output: 20, total: 30 }, + }, + expect: { + succeeds: true, + finalContent: '100 requests per minute', + toolCallSequence: ['search_docs'], + successfulToolCalls: 1, + }, + }, + { + id: 'executor-resolves-start-input', + name: 'resolves a Start output into the Agent prompt before the provider call', + category: 'planning', + description: + 'The Agent userPrompt references . The value must be resolved by the executor and reach the provider request, not passed through verbatim.', + workflowInput: { message: 'Summarize order A-1937' }, + agent: { + model: 'gpt-4o', + userPrompt: '', + }, + providerResponse: { + content: 'Order A-1937 shipped via DHL.', + tokens: { input: 8, output: 12, total: 20 }, + }, + expect: { + succeeds: true, + resolvedInput: 'Summarize order A-1937', + finalContent: /A-1937/, + }, + }, +] diff --git a/apps/sim/evals/agent-tool-use/harness.ts b/apps/sim/evals/agent-tool-use/harness.ts index 631b422698e..0b369a06992 100644 --- a/apps/sim/evals/agent-tool-use/harness.ts +++ b/apps/sim/evals/agent-tool-use/harness.ts @@ -208,9 +208,15 @@ function isOrderedSubsequence(actual: string[], expected: string[]): boolean { return index === expected.length } -function score( +/** Minimal tool-call shape the scorer needs; both harnesses produce it. */ +export interface ScoredToolCall { + name: string + success: boolean +} + +export function scoreExpectations( expected: AgentToolUseExpectations, - toolCalls: CapturedToolCall[], + toolCalls: ScoredToolCall[], finalContent: string, iterations: number, error: unknown, @@ -307,7 +313,7 @@ function score( check( 'completes-without-error', error === undefined, - error === undefined ? 'loop settled' : String(error) + error === undefined ? 'completed without error' : String(error) ) ) } @@ -383,7 +389,7 @@ export async function runScenario( const toolCalls = ((completed?.toolCalls as ToolCallList | undefined)?.list ?? []).slice() const finalContent = completed?.content ?? '' const iterations = completed?.iterations ?? 0 - const checks = score(expected, toolCalls, finalContent, iterations, streamError, mode) + const checks = scoreExpectations(expected, toolCalls, finalContent, iterations, streamError, mode) const successful = toolCalls.filter((call) => call.success).length return { From fb7f54c73b9a8dbfe36fa951f3a500f3ec4967e2 Mon Sep 17 00:00:00 2001 From: Krishna Date: Wed, 30 Sep 2026 00:48:02 +0530 Subject: [PATCH 5/7] test(evals): assert executor block retry on agent failure Add executor-retries-failed-block: the first provider call rejects, the Agent block has retry enabled, and the executor replays it. The run must complete with the second response. Verifies providerCalls === 2, and fails without the retry policy (checked locally: expected 2, got 1). --- apps/sim/evals/README.md | 15 +++-- .../evals/agent-tool-use/executor-harness.ts | 60 +++++++++++++++++-- 2 files changed, 63 insertions(+), 12 deletions(-) diff --git a/apps/sim/evals/README.md b/apps/sim/evals/README.md index f5115681220..ba9f662705f 100644 --- a/apps/sim/evals/README.md +++ b/apps/sim/evals/README.md @@ -117,7 +117,10 @@ Add a case to `EXECUTOR_SCENARIOS` in `executor-harness.ts`: - `providerResponse` is what the mocked provider returns (`content`, `toolCalls`, `tokens`). - `expect` uses the loop's checks plus `resolvedInput` (a substring that must - reach the provider messages) and `succeeds` (expected `ExecutionResult.success`). + reach the provider messages), `succeeds` (expected `ExecutionResult.success`), + and `providerCalls` (exact provider call count). +- Set `agent.retry` to exercise the executor's per-block retry policy; make the + first `providerResponse` a `reject` and the retry lands on the next one. Both suites write one report, so executor rows appear alongside loop rows. @@ -132,8 +135,8 @@ model/tool time, first-response time, and token usage. ## Scope and next steps Two harnesses share one result shape and report: the tool loop and the -`DAGExecutor`. The executor harness mocks the provider boundary, so the -executor's retry/fallback policy is not yet asserted; add a scenario with a -first-call rejection and a block retry config to cover it. Further expansion -(context/memory, model routing, subagent orchestration) is tracked as -follow-up work. +`DAGExecutor`. The executor suite covers block retry — +`executor-retries-failed-block` makes the first provider call reject, the block +is replayed, and the run completes. Model fallback (`fallbackModels`) is not +asserted yet. Further expansion (context/memory, model routing, subagent +orchestration) is tracked as follow-up work. diff --git a/apps/sim/evals/agent-tool-use/executor-harness.ts b/apps/sim/evals/agent-tool-use/executor-harness.ts index 1f6b5d67bdb..9cae3471870 100644 --- a/apps/sim/evals/agent-tool-use/executor-harness.ts +++ b/apps/sim/evals/agent-tool-use/executor-harness.ts @@ -4,7 +4,7 @@ import { } from '@sim/testing/factories/serialized-block.factory' import { providersMockFns } from '@sim/testing/mocks/providers.mock' import { DAGExecutor } from '@/executor/execution/executor' -import type { SerializedWorkflow } from '@/serializer/types' +import type { SerializedBlock, SerializedWorkflow } from '@/serializer/types' import { type EvalRunMode, type ScoredToolCall, scoreExpectations } from './harness' import type { AgentToolUseExpectations, @@ -30,8 +30,10 @@ export interface ExecutorProviderToolCall { result?: unknown } -/** The provider response `executeProviderRequest` returns for one model call. */ +/** One model call: either a response, or a rejection the block must recover from. */ export interface ExecutorProviderResponse { + /** When set, the call rejects with this message instead of resolving. */ + reject?: string content: string model?: string tokens?: { input?: number; output?: number; total?: number } @@ -52,26 +54,30 @@ export interface ExecutorScenario { systemPrompt?: string userPrompt?: string temperature?: number + /** Enables the executor's per-block retry policy for the Agent block. */ + retry?: { enabled: boolean; maxTries: number; waitBetweenTriesMs: number } } - /** One entry per model call; the last entry serves any extra fallback calls. */ + /** One entry per model call; the last entry serves any extra/retry calls. */ providerResponse: ExecutorProviderResponse | ExecutorProviderResponse[] expect: AgentToolUseExpectations & { /** Substring that must appear in the messages sent to the provider. */ resolvedInput?: string /** Expected `ExecutionResult.success`. */ succeeds?: boolean + /** Exact number of provider calls the executor made. */ + providerCalls?: number } } function buildWorkflow(scenario: ExecutorScenario): SerializedWorkflow { - const start = createSerializedBlock({ + const start: SerializedBlock = createSerializedBlock({ id: 'start', type: 'start_trigger', name: 'Start', }) /** The trigger handler claims a block whose metadata says it is a trigger. */ if (start.metadata) start.metadata.category = 'triggers' - const agent = createSerializedBlock({ + const agent: SerializedBlock = createSerializedBlock({ id: 'agent', type: 'agent', name: 'Eval Agent', @@ -85,6 +91,7 @@ function buildWorkflow(scenario: ExecutorScenario): SerializedWorkflow { ? { temperature: scenario.agent.temperature } : {}), } + if (scenario.agent.retry) agent.retry = scenario.agent.retry return createSerializedWorkflow([start, agent], [{ source: 'start', target: 'agent' }]) } @@ -109,6 +116,7 @@ export async function runExecutorScenario( requests.push(request) const response = responses[Math.min(callIndex, responses.length - 1)] callIndex += 1 + if (response.reject) throw new Error(response.reject) return { content: response.content, model: response.model ?? scenario.agent.model, @@ -156,7 +164,14 @@ export async function runExecutorScenario( durationMs: typeof call.duration === 'number' ? call.duration : 0, })) - const checks = scoreExpectations(scenario.expect, toolCalls, finalContent, 1, runError, mode) + const checks = scoreExpectations( + scenario.expect, + toolCalls, + finalContent, + requests.length, + runError, + mode + ) if (scenario.expect.resolvedInput !== undefined) { const sent = JSON.stringify(requests) @@ -175,6 +190,14 @@ export async function runExecutorScenario( }) } + if (scenario.expect.providerCalls !== undefined) { + checks.push({ + name: 'provider-calls', + passed: requests.length === scenario.expect.providerCalls, + detail: `expected ${scenario.expect.providerCalls}, got ${requests.length}`, + }) + } + const tokens = (output.tokens ?? {}) as { input?: number; output?: number; total?: number } return { @@ -259,4 +282,29 @@ export const EXECUTOR_SCENARIOS: ExecutorScenario[] = [ finalContent: /A-1937/, }, }, + { + id: 'executor-retries-failed-block', + name: 'retries a failed Agent block and completes the run', + category: 'recovery', + description: + 'The first provider call rejects with a 503. The block has retry enabled, so the executor replays it and the second call succeeds — proving the executor retry policy, not the agent handler, recovered the turn.', + workflowInput: { message: 'What is the API rate limit?' }, + agent: { + model: 'gpt-4o', + userPrompt: 'What is the API rate limit?', + retry: { enabled: true, maxTries: 3, waitBetweenTriesMs: 0 }, + }, + providerResponse: [ + { reject: '503 Service Unavailable', content: '' }, + { + content: 'The API rate limit is 100 requests per minute.', + tokens: { input: 10, output: 20, total: 30 }, + }, + ], + expect: { + succeeds: true, + finalContent: '100 requests per minute', + providerCalls: 2, + }, + }, ] From c93426b3a6930d26ef9fdb76f0b1ddb815d09f00 Mon Sep 17 00:00:00 2001 From: Krishna Date: Wed, 30 Sep 2026 00:59:53 +0530 Subject: [PATCH 6/7] test(evals): assert executor model fallback on primary failure Add executor-falls-back-to-secondary-model: the primary call rejects, the Agent block has a fallback model, and the handler serves the answer from gpt-4o-mini. Asserts providerCalls === 2 and lastRequestModel, and fails without the fallback row (checked locally: got gpt-4o, run errored). --- apps/sim/evals/README.md | 15 +++---- .../evals/agent-tool-use/executor-harness.ts | 41 +++++++++++++++++++ 2 files changed, 49 insertions(+), 7 deletions(-) diff --git a/apps/sim/evals/README.md b/apps/sim/evals/README.md index ba9f662705f..6c38a5f14d7 100644 --- a/apps/sim/evals/README.md +++ b/apps/sim/evals/README.md @@ -119,8 +119,10 @@ Add a case to `EXECUTOR_SCENARIOS` in `executor-harness.ts`: - `expect` uses the loop's checks plus `resolvedInput` (a substring that must reach the provider messages), `succeeds` (expected `ExecutionResult.success`), and `providerCalls` (exact provider call count). -- Set `agent.retry` to exercise the executor's per-block retry policy; make the - first `providerResponse` a `reject` and the retry lands on the next one. +- Set `agent.retry` to exercise the executor's per-block retry policy, or + `agent.fallbackModels` to exercise model fallback. Make the first + `providerResponse` a `reject` and the next one succeeds; assert + `providerCalls` and `lastRequestModel` to prove which path recovered. Both suites write one report, so executor rows appear alongside loop rows. @@ -135,8 +137,7 @@ model/tool time, first-response time, and token usage. ## Scope and next steps Two harnesses share one result shape and report: the tool loop and the -`DAGExecutor`. The executor suite covers block retry — -`executor-retries-failed-block` makes the first provider call reject, the block -is replayed, and the run completes. Model fallback (`fallbackModels`) is not -asserted yet. Further expansion (context/memory, model routing, subagent -orchestration) is tracked as follow-up work. +`DAGExecutor`. The executor suite covers both recovery paths — block retry +(`executor-retries-failed-block`) and model fallback +(`executor-falls-back-to-secondary-model`). Further expansion (context/memory, +model routing, subagent orchestration) is tracked as follow-up work. diff --git a/apps/sim/evals/agent-tool-use/executor-harness.ts b/apps/sim/evals/agent-tool-use/executor-harness.ts index 9cae3471870..52bc329a0f3 100644 --- a/apps/sim/evals/agent-tool-use/executor-harness.ts +++ b/apps/sim/evals/agent-tool-use/executor-harness.ts @@ -56,6 +56,8 @@ export interface ExecutorScenario { temperature?: number /** Enables the executor's per-block retry policy for the Agent block. */ retry?: { enabled: boolean; maxTries: number; waitBetweenTriesMs: number } + /** Ordered models the Agent handler tries after the primary fails. */ + fallbackModels?: Array<{ model: string }> } /** One entry per model call; the last entry serves any extra/retry calls. */ providerResponse: ExecutorProviderResponse | ExecutorProviderResponse[] @@ -66,6 +68,8 @@ export interface ExecutorScenario { succeeds?: boolean /** Exact number of provider calls the executor made. */ providerCalls?: number + /** Model id sent on the final provider call (proves which candidate served). */ + lastRequestModel?: string } } @@ -90,6 +94,7 @@ function buildWorkflow(scenario: ExecutorScenario): SerializedWorkflow { ...(scenario.agent.temperature !== undefined ? { temperature: scenario.agent.temperature } : {}), + ...(scenario.agent.fallbackModels ? { fallbackModels: scenario.agent.fallbackModels } : {}), } if (scenario.agent.retry) agent.retry = scenario.agent.retry @@ -198,6 +203,15 @@ export async function runExecutorScenario( }) } + if (scenario.expect.lastRequestModel !== undefined) { + const lastModel = (requests.at(-1) as { model?: string } | undefined)?.model + checks.push({ + name: 'last-request-model', + passed: lastModel === scenario.expect.lastRequestModel, + detail: `expected ${scenario.expect.lastRequestModel}, got ${String(lastModel)}`, + }) + } + const tokens = (output.tokens ?? {}) as { input?: number; output?: number; total?: number } return { @@ -307,4 +321,31 @@ export const EXECUTOR_SCENARIOS: ExecutorScenario[] = [ providerCalls: 2, }, }, + { + id: 'executor-falls-back-to-secondary-model', + name: 'falls back to the secondary model when the primary fails', + category: 'recovery', + description: + 'The primary model call rejects and the Agent block has a fallback model. The handler must serve the answer from the fallback and the run must complete.', + workflowInput: { message: 'What is the API rate limit?' }, + agent: { + model: 'gpt-4o', + userPrompt: 'What is the API rate limit?', + fallbackModels: [{ model: 'gpt-4o-mini' }], + }, + providerResponse: [ + { reject: '429 rate limited', content: '' }, + { + content: 'The API rate limit is 100 requests per minute.', + model: 'gpt-4o-mini', + tokens: { input: 10, output: 20, total: 30 }, + }, + ], + expect: { + succeeds: true, + finalContent: '100 requests per minute', + providerCalls: 2, + lastRequestModel: 'gpt-4o-mini', + }, + }, ] From 26937ff82ed6df64b32a88e1715942216ee03294 Mon Sep 17 00:00:00 2001 From: Krishna Date: Wed, 30 Sep 2026 22:37:57 +0530 Subject: [PATCH 7/7] feat(evals): compare agent tool-use across models Run the same live scenarios across a list of models and write a scenario x model matrix. models.ts resolves provider:model specs (DeepSeek, OpenAI, Groq, OpenRouter) and reads each provider's key from _API_KEY. - agent-tool-use.compare.live.test.ts: EVAL_MODELS x scenarios x trials - report.ts: buildLiveComparisonReport + JSON/Markdown matrix - report.test.ts: key-free aggregation coverage - test:evals:compare script; README documents the spec format --- apps/sim/evals/README.md | 17 +++ .../agent-tool-use.compare.live.test.ts | 92 ++++++++++++++ apps/sim/evals/agent-tool-use/models.ts | 85 +++++++++++++ apps/sim/evals/agent-tool-use/report.test.ts | 71 +++++++++++ apps/sim/evals/agent-tool-use/report.ts | 120 ++++++++++++++++++ apps/sim/package.json | 1 + 6 files changed, 386 insertions(+) create mode 100644 apps/sim/evals/agent-tool-use/agent-tool-use.compare.live.test.ts create mode 100644 apps/sim/evals/agent-tool-use/models.ts create mode 100644 apps/sim/evals/agent-tool-use/report.test.ts diff --git a/apps/sim/evals/README.md b/apps/sim/evals/README.md index 6c38a5f14d7..bbe45a3de6c 100644 --- a/apps/sim/evals/README.md +++ b/apps/sim/evals/README.md @@ -75,6 +75,23 @@ example, an exact retry count). The report is at `test-results/evals/agent-tool-use-live.{json,md}` with pass rates, average iterations, latency, and the failed check names. +### Compare models + +Run the same scenarios across several models and get a scenario × model matrix: + +```sh +cd apps/sim +EVAL_MODELS=deepseek:deepseek-chat,deepseek:deepseek-reasoner \ + DEEPSEEK_API_KEY=... bun run test:evals:compare +``` + +A spec is `provider:model`; a bare model id defaults to DeepSeek. Providers are +DeepSeek, OpenAI, Groq, and OpenRouter, each reading its key from +`_API_KEY`. The report is +`test-results/evals/agent-tool-use-compare.{json,md}`: per-model pass rate, +iterations, latency, and tokens, plus a per-scenario pass-rate matrix. +`EVAL_MIN_PASS_RATE` fails a model below a floor. + ## Add a case 1. Open [`agent-tool-use/scenarios.ts`](./agent-tool-use/scenarios.ts) and add diff --git a/apps/sim/evals/agent-tool-use/agent-tool-use.compare.live.test.ts b/apps/sim/evals/agent-tool-use/agent-tool-use.compare.live.test.ts new file mode 100644 index 00000000000..1c5931a4476 --- /dev/null +++ b/apps/sim/evals/agent-tool-use/agent-tool-use.compare.live.test.ts @@ -0,0 +1,92 @@ +import { providersMock } from '@sim/testing/mocks/providers.mock' +import { providersConversationHistoryMock } from '@sim/testing/mocks/providers-conversation-history.mock' +import { providersUtilsMock } from '@sim/testing/mocks/providers-utils.mock' +import { toolsMock } from '@sim/testing/mocks/tools.mock' +import { afterAll, describe, expect, it, vi } from 'vitest' +import { runScenario } from '@/evals/agent-tool-use/harness' +import { createLiveModelCompletion, parseLiveModels } from '@/evals/agent-tool-use/models' +import { type LiveModelRun, writeLiveComparisonReport } from '@/evals/agent-tool-use/report' +import { AGENT_TOOL_USE_SCENARIOS } from '@/evals/agent-tool-use/scenarios' + +vi.mock('@/providers/conversation-history', () => providersConversationHistoryMock) +vi.mock('@/tools', () => toolsMock) +vi.mock('@/providers/utils', () => providersUtilsMock) +vi.mock('@/providers', () => providersMock) + +/** + * Compare the same scenarios across several models. Opt-in, never in CI: + * + * EVAL_LIVE=1 EVAL_MODELS=deepseek:deepseek-chat,deepseek:deepseek-reasoner \ + * bun run --cwd apps/sim test --mode live evals/agent-tool-use/agent-tool-use.compare.live.test.ts + * + * Each provider reads its key from `_API_KEY`; a bare model id + * defaults to DeepSeek. The report is a scenario × model matrix plus per-model + * pass rate, iterations, latency, and tokens. Set `EVAL_MIN_PASS_RATE` (0–1) to + * fail a model below a pass-rate floor. + */ +const LIVE = process.env.EVAL_LIVE === '1' +const TRIALS = Number(process.env.EVAL_TRIALS ?? '3') +const MIN_PASS_RATE = Number(process.env.EVAL_MIN_PASS_RATE ?? '0') +const TIMEOUT_MS = Number(process.env.EVAL_TIMEOUT_MS ?? '180000') +const models = parseLiveModels(process.env.EVAL_MODELS) +const liveScenarios = AGENT_TOOL_USE_SCENARIOS.filter((scenario) => !scenario.scriptedOnly) + +const runs: LiveModelRun[] = [] + +const cases = models.flatMap((spec) => + liveScenarios.map((scenario) => ({ + id: `${spec.provider.id}/${spec.model} · ${scenario.id}`, + spec, + scenario, + })) +) + +afterAll(() => { + if (!LIVE || runs.length === 0) return + writeLiveComparisonReport( + runs, + process.env.EVAL_COMPARE_REPORT_PATH ?? 'test-results/evals/agent-tool-use-compare.json' + ) +}) + +describe.skipIf(!LIVE)('agent tool-use model comparison', () => { + it.each(cases)( + '$id', + async ({ spec, scenario }) => { + const completion = createLiveModelCompletion(spec) + for (let trial = 0; trial < TRIALS; trial++) { + const result = await runScenario(scenario, { + completion, + mode: 'live', + model: spec.model, + providerName: spec.provider.label, + }) + runs.push({ + provider: spec.provider.id, + model: spec.model, + scenarioId: scenario.id, + result, + }) + } + }, + TIMEOUT_MS + ) + + it('meets the per-model pass-rate floor', () => { + if (MIN_PASS_RATE <= 0 || runs.length === 0) return + const byModel = new Map() + for (const run of runs) { + const key = `${run.provider}/${run.model}` + const entry = byModel.get(key) ?? { passed: 0, total: 0 } + entry.total += 1 + if (run.result.passed) entry.passed += 1 + byModel.set(key, entry) + } + for (const [model, entry] of byModel) { + expect( + entry.passed / entry.total, + `${model} passed ${entry.passed}/${entry.total}` + ).toBeGreaterThanOrEqual(MIN_PASS_RATE) + } + }) +}) diff --git a/apps/sim/evals/agent-tool-use/models.ts b/apps/sim/evals/agent-tool-use/models.ts new file mode 100644 index 00000000000..94fe5b3b238 --- /dev/null +++ b/apps/sim/evals/agent-tool-use/models.ts @@ -0,0 +1,85 @@ +import type { OpenAICompatCreateCompletion } from '@/providers/openai-compat/streaming-tool-loop' +import { createOpenAICompatLiveCompletion } from './live' + +/** + * Provider registry for the model-comparison live suite. + * + * A model spec is `provider:model` (or a bare model id, which defaults to + * DeepSeek). Each provider reads its key from the matching environment + * variable, so adding a provider is one entry plus its key. + */ +export interface LiveProviderSpec { + id: string + label: string + baseURL?: string + keyEnv: string + baseURLEnv?: string +} + +const PROVIDERS: Record = { + deepseek: { + id: 'deepseek', + label: 'DeepSeek', + baseURL: 'https://api.deepseek.com', + baseURLEnv: 'DEEPSEEK_BASE_URL', + keyEnv: 'DEEPSEEK_API_KEY', + }, + openai: { id: 'openai', label: 'OpenAI', keyEnv: 'OPENAI_API_KEY' }, + groq: { + id: 'groq', + label: 'Groq', + baseURL: 'https://api.groq.com/openai/v1', + keyEnv: 'GROQ_API_KEY', + }, + openrouter: { + id: 'openrouter', + label: 'OpenRouter', + baseURL: 'https://openrouter.ai/api/v1', + keyEnv: 'OPENROUTER_API_KEY', + }, +} + +export interface LiveModelSpec { + provider: LiveProviderSpec + model: string +} + +/** Resolves `provider:model`, or a bare model id against DeepSeek. */ +export function resolveLiveModelSpec(spec: string): LiveModelSpec { + const separator = spec.indexOf(':') + if (separator === -1) { + return { provider: PROVIDERS.deepseek, model: spec.trim() } + } + const providerId = spec.slice(0, separator).trim().toLowerCase() + const provider = PROVIDERS[providerId] + if (!provider) { + throw new Error(`Unknown eval provider "${providerId}" in spec "${spec}"`) + } + return { provider, model: spec.slice(separator + 1).trim() } +} + +export function parseLiveModels(value: string | undefined): LiveModelSpec[] { + return (value ?? 'deepseek:deepseek-chat') + .split(',') + .map((spec) => spec.trim()) + .filter(Boolean) + .map(resolveLiveModelSpec) +} + +/** Builds a real completion for one resolved `provider:model` spec. */ +export function createLiveModelCompletion(spec: LiveModelSpec): OpenAICompatCreateCompletion { + const apiKey = process.env[spec.provider.keyEnv] + if (!apiKey) { + throw new Error(`${spec.provider.keyEnv} is required for ${spec.provider.label}/${spec.model}`) + } + const baseURL = spec.provider.baseURLEnv + ? (process.env[spec.provider.baseURLEnv] ?? spec.provider.baseURL) + : spec.provider.baseURL + + return createOpenAICompatLiveCompletion({ + apiKey, + model: spec.model, + ...(baseURL ? { baseURL } : {}), + ...(process.env.EVAL_TIMEOUT_MS ? { timeoutMs: Number(process.env.EVAL_TIMEOUT_MS) } : {}), + }) +} diff --git a/apps/sim/evals/agent-tool-use/report.test.ts b/apps/sim/evals/agent-tool-use/report.test.ts new file mode 100644 index 00000000000..27b4d6ecc2b --- /dev/null +++ b/apps/sim/evals/agent-tool-use/report.test.ts @@ -0,0 +1,71 @@ +import { existsSync, mkdtempSync, rmSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' +import { + buildLiveComparisonReport, + type LiveModelRun, + writeLiveComparisonReport, +} from '@/evals/agent-tool-use/report' +import type { AgentToolUseResult } from '@/evals/agent-tool-use/types' + +function result(passed: boolean): AgentToolUseResult { + return { + id: 'scenario', + name: 'scenario', + category: 'recovery', + passed, + checks: [], + finalContent: '', + toolInvocations: [], + metrics: { + iterations: 1, + toolCalls: 1, + successfulToolCalls: 1, + erroredToolCalls: 0, + latencyMs: 10, + modelTimeMs: 0, + toolsTimeMs: 0, + firstResponseTimeMs: 0, + inputTokens: 1, + outputTokens: 1, + totalTokens: 2, + }, + } +} + +function run(model: string, passed: boolean): LiveModelRun { + return { provider: 'deepseek', model, scenarioId: 'scenario', result: result(passed) } +} + +describe('model comparison report', () => { + it('groups runs by model, sorts by pass rate, and fills the scenario matrix', () => { + const report = buildLiveComparisonReport([ + run('chat', true), + run('chat', true), + run('reasoner', false), + run('reasoner', true), + ]) + + expect(report.models.map((model) => model.label)).toEqual([ + 'deepseek/chat', + 'deepseek/reasoner', + ]) + expect(report.models[0]).toMatchObject({ passRate: 1, passed: 2, trials: 2 }) + expect(report.models[1]).toMatchObject({ passRate: 0.5, passed: 1, trials: 2 }) + expect(report.matrix.scenario['deepseek/chat']).toBe(1) + expect(report.matrix.scenario['deepseek/reasoner']).toBe(0.5) + }) + + it('writes JSON and Markdown reports', () => { + const directory = mkdtempSync(join(tmpdir(), 'eval-compare-')) + try { + const path = join(directory, 'compare.json') + writeLiveComparisonReport([run('chat', true)], path) + expect(existsSync(path)).toBe(true) + expect(existsSync(join(directory, 'compare.md'))).toBe(true) + } finally { + rmSync(directory, { recursive: true, force: true }) + } + }) +}) diff --git a/apps/sim/evals/agent-tool-use/report.ts b/apps/sim/evals/agent-tool-use/report.ts index 42d8983c6f0..504dc5a162a 100644 --- a/apps/sim/evals/agent-tool-use/report.ts +++ b/apps/sim/evals/agent-tool-use/report.ts @@ -127,3 +127,123 @@ export function writeLiveEvalReport(summaries: LiveScenarioSummary[], reportPath writeFileSync(reportPath, `${JSON.stringify(report, null, 2)}\n`) writeFileSync(reportPath.replace(/\.json$/, '.md'), renderLiveMarkdown(report)) } + +/** One trial of one model on one scenario, for the comparison report. */ +export interface LiveModelRun { + provider: string + model: string + scenarioId: string + result: AgentToolUseResult +} + +/** Aggregate behavior of one model across the suite. */ +export interface LiveModelSummary { + provider: string + model: string + label: string + trials: number + passed: number + passRate: number + avgIterations: number + avgLatencyMs: number + totalTokens: number +} + +/** Side-by-side comparison of several models on the same scenarios. */ +export interface LiveComparisonReport { + suite: 'agent-tool-use-compare' + generatedAt: string + scenarios: string[] + models: LiveModelSummary[] + /** scenarioId -> model label -> pass rate. */ + matrix: Record> +} + +function modelLabel(run: LiveModelRun): string { + return `${run.provider}/${run.model}` +} + +export function buildLiveComparisonReport(runs: LiveModelRun[]): LiveComparisonReport { + const byModel = new Map() + for (const run of runs) { + const key = modelLabel(run) + const list = byModel.get(key) ?? [] + list.push(run) + byModel.set(key, list) + } + + const scenarios = [...new Set(runs.map((run) => run.scenarioId))].sort() + const models: LiveModelSummary[] = [] + const matrix: Record> = {} + + for (const [label, modelRuns] of byModel) { + const results = modelRuns.map((run) => run.result) + const passed = results.filter((result) => result.passed).length + const [provider, model] = label.split('/') + models.push({ + provider, + model, + label, + trials: results.length, + passed, + passRate: results.length === 0 ? 0 : passed / results.length, + avgIterations: average(results.map((result) => result.metrics.iterations)), + avgLatencyMs: average(results.map((result) => result.metrics.latencyMs)), + totalTokens: results.reduce((sum, result) => sum + result.metrics.totalTokens, 0), + }) + + for (const scenario of scenarios) { + const scenarioRuns = modelRuns.filter((run) => run.scenarioId === scenario) + const scenarioPassed = scenarioRuns.filter((run) => run.result.passed).length + matrix[scenario] ??= {} + matrix[scenario][label] = scenarioRuns.length === 0 ? 0 : scenarioPassed / scenarioRuns.length + } + } + + models.sort((a, b) => b.passRate - a.passRate) + return { + suite: 'agent-tool-use-compare', + generatedAt: new Date().toISOString(), + scenarios, + models, + matrix, + } +} + +function renderComparisonMarkdown(report: LiveComparisonReport): string { + const labels = report.models.map((model) => model.label) + const lines = [ + '# Agent tool-use model comparison', + '', + `Generated: ${report.generatedAt}`, + '', + '| Model | Pass rate | Trials | Avg iterations | Avg latency | Tokens |', + '| --- | ---: | ---: | ---: | ---: | ---: |', + ...report.models.map( + (model) => + `| ${model.label} | ${(model.passRate * 100).toFixed(0)}% (${model.passed}/${model.trials}) | ${model.trials} | ${model.avgIterations.toFixed(1)} | ${Math.round(model.avgLatencyMs)}ms | ${model.totalTokens} |` + ), + '', + `| Scenario | ${labels.join(' | ')} |`, + `| --- | ${labels.map(() => '---:').join(' | ')} |`, + ...report.scenarios.map( + (scenario) => + `| ${escapeCell(scenario)} | ${labels + .map((label) => `${((report.matrix[scenario]?.[label] ?? 0) * 100).toFixed(0)}%`) + .join(' | ')} |` + ), + '', + ] + return lines.join('\n') +} + +/** + * Writes the model-comparison JSON report and a sibling Markdown table. + * Unlike the single-model report, this is a matrix: scenario × model pass rates. + */ +export function writeLiveComparisonReport(runs: LiveModelRun[], reportPath: string): void { + const report = buildLiveComparisonReport(runs) + mkdirSync(dirname(reportPath), { recursive: true }) + writeFileSync(reportPath, `${JSON.stringify(report, null, 2)}\n`) + writeFileSync(reportPath.replace(/\.json$/, '.md'), renderComparisonMarkdown(report)) +} diff --git a/apps/sim/package.json b/apps/sim/package.json index b9542513782..eaca7259865 100644 --- a/apps/sim/package.json +++ b/apps/sim/package.json @@ -26,6 +26,7 @@ "test:coverage": "vitest run --coverage", "test:evals": "EVAL_REPORT_PATH=test-results/evals/agent-tool-use.json vitest run evals/agent-tool-use", "test:evals:live": "EVAL_LIVE=1 vitest run --mode live evals/agent-tool-use", + "test:evals:compare": "EVAL_LIVE=1 vitest run --mode live evals/agent-tool-use/agent-tool-use.compare.live.test.ts", "email:dev": "email dev --dir components/emails", "type-check": "tsc --noEmit", "lint": "biome check --write --unsafe .",