From 06a1492e923575d17746e8f556acf83a0b250f45 Mon Sep 17 00:00:00 2001 From: Scott Lowe Date: Thu, 1 Oct 2026 15:40:39 -0700 Subject: [PATCH] fix(sim): report failed, canceled and incomplete runs instead of a false pass `npm run sim` read a `results` field the simulation-run API doesn't return and counted `status === "pass"` (items are `passed`/`failed`), so every watched run summarised as 0/0 and exited 0, including failing runs. - src/sim-result.ts: strict, pure verdict. Passed only when the run ended, every expected item exists and passed, and each had a required evaluation that was actually scored; otherwise failed or incomplete with a reason. - src/sim.ts: read run items (paginated or bare array, deduped by id), wait for late item results, print the run link from the create response and the failing judges, cancel the run on --timeout (default 20 min) or Ctrl-C. - src/vapi-client.ts: config-free client; run creation is never retried on a 5xx, since the run may already be queued. - src/sim-cmd.ts: exported simCommandRun; exit 0 passed, 1 failed, 2 usage, 3 incomplete; --no-watch prints the link and exits 0. - Docs: run/item shapes and the verdict in docs/learnings/simulations.md, sim rows in README and AGENTS.md, improvements.md #33. Refs TEST-141 Co-Authored-By: Claude Opus 5.5 --- AGENTS.md | 2 +- README.md | 2 +- docs/learnings/simulations.md | 21 +- improvements.md | 52 +++++ src/sim-cmd.ts | 200 +++++++++++------- src/sim-result.ts | 202 +++++++++++++++++++ src/sim.ts | 309 ++++++++++++++++++++-------- src/vapi-client.ts | 108 ++++++++++ tests/sim-result.test.ts | 178 ++++++++++++++++ tests/sim-run.test.ts | 368 ++++++++++++++++++++++++++++++++++ 10 files changed, 1282 insertions(+), 160 deletions(-) create mode 100644 src/sim-result.ts create mode 100644 src/vapi-client.ts create mode 100644 tests/sim-result.test.ts create mode 100644 tests/sim-run.test.ts diff --git a/AGENTS.md b/AGENTS.md index 2a303f1..05ba2b2 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -902,7 +902,7 @@ npm run promote -- --pipeline --from --to --apply # For npm run validate -- # Lint resources locally (fails fast on schema drift) npm run audit -- # Read-only drift detector: orphan YAML, state ghosts, content-identical clusters, sibling base-slugs, dashboard orphans, inline model.tools. Exit 1 on findings. npm run audit -- --type assistants # Scope audit to a single resource type -npm run sim -- --suite --target # Run a simulation suite against an assistant/squad +npm run sim -- --suite --target # Run a simulation suite against an assistant/squad (exit 0 pass, 1 fail, 3 incomplete; --timeout ) npm run rollback -- --to # Re-apply a snapshot taken before a push npm run rollback -- --list # List available snapshots diff --git a/README.md b/README.md index f3401c4..180da18 100644 --- a/README.md +++ b/README.md @@ -103,7 +103,7 @@ Every command works in two modes: | `npm run cleanup` | ✅ | `npm run cleanup -- [--force --confirm ]` | Inspect (default) or delete orphaned remote resources. Destructive run requires `--confirm `. | | `npm run rollback` | — | `npm run rollback -- --list` or `--to ` | Restore from a snapshot in `.vapi-state..snapshots/` (one is written before every push/apply). | | `npm run call` | ✅ | `npm run call -- -a ` or `-s ` | Start an interactive WebSocket call against an assistant or squad. | -| `npm run sim` | — | `npm run sim -- --suite --target ` | Run a simulation suite (or specific simulations) against a deployed assistant/squad. | +| `npm run sim` | — | `npm run sim -- --suite --target [--timeout ]` | Run a simulation suite (or specific simulations) against a deployed assistant/squad. Prints the run link; exits 0 passed, 1 failed, 3 incomplete (timeout, Ctrl-C, missing results). | | `npm run migrate` | — | `npm run migrate` | One-time, all orgs at once: slim legacy state files to pure `name → uuid` and seed the per-developer `.vapi-state-hash/` baseline store from the old hashes. Required once after upgrading to the hash-store engine — `pull`/`push`/`apply` refuse legacy-shaped state until it runs. Idempotent. | | `npm run build` | — | — | Type-check the codebase (`tsc --noEmit`). | | `npm test` | — | — | Run regression tests (`node:test`). | diff --git a/docs/learnings/simulations.md b/docs/learnings/simulations.md index 96787bc..e94b18b 100644 --- a/docs/learnings/simulations.md +++ b/docs/learnings/simulations.md @@ -471,6 +471,23 @@ This is the unified executor. A "run" is a batch — it expands into many `runIt } ``` +The **create** response (`CreateSimulationRunResponse`) also carries two fields nothing else returns: + +- `url` — the dashboard link to this run. `GET /eval/simulation/run/:id` does **not** return it, so read it from the create response. +- `simulationRunItemIds` — one id per item queued (simulations × iterations), i.e. how many items the run should end with. + +A run has **no `results` field**. Pass/fail lives in `itemCounts` and on the items themselves (`GET …/item`, below). Creating a run queues paid work before the response returns, so a 5xx on create may still have started it — don't blindly retry the POST. + +### How `npm run sim` decides pass/fail + +`src/sim-result.ts` (`simRunVerdict`) reports **passed** only when all of these hold; otherwise **failed** (any failed item) or **incomplete**: + +- the run `ended` and has `itemCounts`, with `total` equal to the number of items created (`simulationRunItemIds`) and greater than 0; +- nothing is `queued`, `running` or `canceled`, and `passed === total`; +- every item was fetched, is `passed`, and has at least one **required** evaluation that wasn't skipped (the all-skipped trap from the chat-mode gotcha above). + +Exit codes: 0 passed, 1 failed, 2 usage error, 3 incomplete (timeout, Ctrl-C, canceled items, results that never arrived). On timeout or Ctrl-C the run is canceled. + ### List runs — `GET /eval/simulation/run` Query params: @@ -486,7 +503,7 @@ Query params: ### Get / Cancel run - `GET /eval/simulation/run/:id` → `SimulationRun` -- `PATCH /eval/simulation/run/:id` → cancels the run **and** all its queued items. No body required. +- `PATCH /eval/simulation/run/:id` → cancels the run **and** all its queued items. No body required. Returns 400 `Run has already ended` for an ended run and 409 when a concurrent cancel wins; both mean "nothing left to cancel". --- @@ -498,6 +515,8 @@ Run items are system-managed — there's no create/update API for users; they're Query params: `limit`, `page`, `simulationId`, `runId`, `status` (`queued` | `running` | `evaluating` | `passed` | `failed` | `canceled`). +Response shape depends on the query: **with** `limit` or `page` it's paginated (`{ results, metadata: { totalItems, itemsPerPage, currentPage } }`, `limit` up to 1000); **without** them it's a bare array. Pages are ordered only by creation time and a run's items share it, so OFFSET pages can overlap — dedupe by `id`. Items can also lag the run: a run can be `ended` while an item is `passed` but its `results` aren't written yet. + ### Get a run item — `GET /eval/simulation/run/:id/item/:itemId` **Response** — `SimulationRunItem` is rich; the highlights: diff --git a/improvements.md b/improvements.md index 44606c4..7472bd7 100644 --- a/improvements.md +++ b/improvements.md @@ -84,6 +84,7 @@ you which stack PR closes the row.** | 30 | Tool-linking pass could PATCH a raw assistant slug | Mid-push 400 naming the wrong resource | None | RESOLVED 2026-08-03 (#51) | | 31 | Unresolved references handled 3 inconsistent ways, no dangling-ref check | Same authoring mistake, three different failure modes | None | Open | | 32 | Test suite never ran in CI; 20 tests rotted after the hash store | Regression guards for #22/#23 silently stopped running | None | RESOLVED 2026-09-30 (#56) | +| 33 | `npm run sim` reported every run as passed | A failing suite exited 0 — false green | None | RESOLVED 2026-10-01 | **Active backlog after cleanup:** `#2`, `#6`, `#8`, `#12`, `#20`, `#24–#26`, `#31`, and the open remainder of `#27` (wiring the listing-completeness verdict into push/delete/audit, and moving `cleanup.ts` onto the shared pager). Resolved entries stay in this file as historical incident notes per the maintenance directive; stale superseded backlog rows are not duplicated. @@ -1721,6 +1722,57 @@ fields. --- +## 33. `npm run sim` reported every run as passed + +**[RESOLVED 2026-10-01]** + +**Discovered:** 2026-10-01, while designing simulation PR checks (TEST-141). + +### Problem + +`npm run sim` scored a run by reading `results[]` on the run and counting +`status === "pass"`. The simulation-run API has no `results` field and +item statuses are `passed` / `failed`, so every watched run summarised +as 0 pass / 0 fail and the command exited 0 — including runs whose +simulations failed. + +### Current behavior (Verified, before the fix) + +- `src/sim.ts` `runSimulation` computed pass/fail from `last.results`, + which `GET /eval/simulation/run/:id` never returns; pass/fail is in + `itemCounts` and on the run items (`GET /eval/simulation/run/:id/item`). +- `src/sim-cmd.ts` exited 1 only when `fail > 0`, which could never happen. +- Polling also stopped on statuses that don't exist (`failed`, + `completed`), never canceled a timed-out run, and didn't print the run + link (only the create response carries `url`). + +### Risk + +Anyone gating on `npm run sim` (locally or in CI) got a green result for a +failing suite. + +### Current mitigation + +None needed once the fix below lands. + +### Possible fix (landed) + +- `src/sim-result.ts` `simRunVerdict`: passed only when the run ended, + every expected item exists, passed, and had a required evaluation that was + actually scored; otherwise failed or incomplete with a reason. +- `src/sim.ts` reads items (paginated or bare-array, deduped by id), + waits for late item results, prints the run link and failing judges, and + cancels the run on `--timeout` (default 20 min) or Ctrl-C. +- `src/vapi-client.ts`: a config-free client that never retries run + creation on a 5xx (the run may already be queued). +- Exit codes: 0 passed, 1 failed, 2 usage, 3 incomplete. + +### Status + +**RESOLVED 2026-10-01.** + +--- + ## Out of scope (intentionally not improvements) - **State file is identity-only and not git-ignored.** It's intentionally diff --git a/src/sim-cmd.ts b/src/sim-cmd.ts index 2e3c531..e0b1fcb 100644 --- a/src/sim-cmd.ts +++ b/src/sim-cmd.ts @@ -2,7 +2,12 @@ // // Wraps `POST /eval/simulation/run` (the simulations API). See AGENTS.md for // usage. +// +// Exit codes: 0 passed, 1 failed, 2 usage/config error, 3 incomplete +// (timed out, interrupted, canceled, or results that never fully arrived). +import { resolve } from "node:path"; +import { fileURLToPath } from "node:url"; import { formatSummary, loadEnvFile, @@ -12,27 +17,28 @@ import { runSimulation, } from "./sim.ts"; -function printUsage(): void { - console.error( - [ - "Usage:", - " npm run sim -- --suite --target ", - " npm run sim -- --simulations , --target ", - "", - "Options:", - " --suite Run an entire simulation suite by local resource name", - " --simulations Run one or more simulations by comma-separated local names", - " --target Local assistant or squad name (resolves to UUID via state)", - " --transport voice|chat Transport (default: voice; chat is faster/cheaper)", - " --iterations N Override default iteration count", - " --watch Tail status until completion (default: on)", - "", - "Examples:", - " npm run sim -- my-org --suite booking-tests --target intake-agent", - " npm run sim -- my-org --simulations happy-path,edge-case --target main-agent --transport chat", - ].join("\n"), - ); -} +const USAGE = [ + "Usage:", + " npm run sim -- --suite --target ", + " npm run sim -- --simulations , --target ", + "", + "Options:", + " --suite Run an entire simulation suite by local resource name", + " --simulations Run one or more simulations by comma-separated local names", + " --target Local assistant or squad name (resolves to UUID via state)", + " --transport voice|chat Transport (default: voice; chat is faster/cheaper)", + " --iterations N Override default iteration count", + " --timeout Give up and cancel the run after this long (default: 20)", + " --no-watch Start the run, print its link, and exit 0 without a verdict", + "", + "Exit codes: 0 passed, 1 failed, 2 usage error, 3 incomplete (timeout, interrupt, missing results)", + "", + "Examples:", + " npm run sim -- my-org --suite booking-tests --target intake-agent", + " npm run sim -- my-org --simulations happy-path,edge-case --target main-agent --transport chat", +].join("\n"); + +class UsageError extends Error {} interface ParsedArgs { env: string; @@ -42,71 +48,83 @@ interface ParsedArgs { squad?: string; transport?: "voice" | "chat"; iterations?: number; + timeoutMinutes?: number; watch: boolean; + help: boolean; } -function parseArgs(): ParsedArgs { - const args = process.argv.slice(2); +function argsParse(args: string[]): ParsedArgs { const env = args[0]; - if (!env) { - printUsage(); - process.exit(1); + if (!env || env === "--help" || env === "-h") { + return { env: env ?? "", watch: true, help: true }; } const SLUG_RE = /^[a-z0-9]([a-z0-9-]*[a-z0-9])?$/; - if (!SLUG_RE.test(env)) { - console.error(`❌ Invalid org name: ${env}`); - process.exit(1); - } - - const parsed: ParsedArgs = { env, watch: true }; + if (!SLUG_RE.test(env)) throw new UsageError(`Invalid org name: ${env}`); + const parsed: ParsedArgs = { env, watch: true, help: false }; for (let i = 1; i < args.length; i++) { const arg = args[i]; if (arg === "--suite") parsed.suite = args[++i]; else if (arg === "--simulations") parsed.simulations = args[++i]; else if (arg === "--target") { // We don't know yet whether target is an assistant or squad — defer - // resolution to the state lookup. Try assistant first; resolveTarget() - // accepts either argument key, so we set the candidate in `assistant` - // and let `resolveTarget` fall through to `squad` if not found. - // For clarity, we accept --assistant / --squad as explicit alternatives. + // resolution to the state lookup (see simCommandRun). parsed.assistant = args[++i]; } else if (arg === "--assistant") parsed.assistant = args[++i]; else if (arg === "--squad") parsed.squad = args[++i]; else if (arg === "--transport") { const v = args[++i]; - if (v === "voice" || v === "chat") parsed.transport = v; - else { - console.error(`❌ --transport must be "voice" or "chat" (got "${v}")`); - process.exit(1); + if (v !== "voice" && v !== "chat") { + throw new UsageError( + `--transport must be "voice" or "chat" (got "${v}")`, + ); } + parsed.transport = v; } else if (arg === "--iterations") { - parsed.iterations = parseInt(args[++i] ?? "", 10); + parsed.iterations = Number.parseInt(args[++i] ?? "", 10); if (Number.isNaN(parsed.iterations)) { - console.error("❌ --iterations requires a number"); - process.exit(1); + throw new UsageError("--iterations requires a number"); + } + } else if (arg === "--timeout") { + parsed.timeoutMinutes = Number(args[++i]); + if ( + !Number.isFinite(parsed.timeoutMinutes) || + parsed.timeoutMinutes <= 0 + ) { + throw new UsageError("--timeout requires a positive number of minutes"); } } else if (arg === "--no-watch") parsed.watch = false; else if (arg === "--watch") parsed.watch = true; - else if (arg === "--help" || arg === "-h") { - printUsage(); - process.exit(0); - } + else if (arg === "--help" || arg === "-h") parsed.help = true; + else throw new UsageError(`Unknown argument: ${arg}`); } - return parsed; } -async function main(): Promise { - const args = parseArgs(); - const cfg = loadEnvFile(args.env); - const state = loadStateFile(args.env); +export async function simCommandRun( + args = process.argv.slice(2), +): Promise { + let parsed: ParsedArgs; + try { + parsed = argsParse(args); + } catch (error) { + if (!(error instanceof UsageError)) throw error; + console.error(`❌ ${error.message}\n\n${USAGE}`); + return 2; + } + if (parsed.help) { + console.log(USAGE); + return parsed.env ? 0 : 2; + } + + const cfg = loadEnvFile(parsed.env); + const state = loadStateFile(parsed.env); // Disambiguate --target: if the bare value matches a squad name in state // and not an assistant, treat it as a squad. Explicit --assistant / --squad // override the heuristic. - let assistant = args.assistant; - let squad = args.squad; + let assistant = parsed.assistant; + let squad = parsed.squad; if (assistant && !squad) { const isSquad = typeof state.squads[assistant] !== "undefined" && @@ -120,39 +138,71 @@ async function main(): Promise { console.log( "═══════════════════════════════════════════════════════════════", ); - console.log(`🧪 Vapi GitOps Sim Runner — Environment: ${args.env}`); + console.log(`🧪 Vapi GitOps Sim Runner — Environment: ${parsed.env}`); console.log(` API: ${cfg.baseUrl}`); console.log( "═══════════════════════════════════════════════════════════════\n", ); const selection = resolveSelection(state, { - suite: args.suite, - simulations: args.simulations, + suite: parsed.suite, + simulations: parsed.simulations, }); const target = resolveTarget(state, { assistant, squad }); - const summary = await runSimulation(cfg, selection, target, { - watch: args.watch, - iterations: args.iterations, - transport: args.transport, - }); + // Ctrl-C / SIGTERM cancel the run instead of leaving it running unwatched. + const controller = new AbortController(); + const stop = () => controller.abort(); + process.on("SIGINT", stop); + process.on("SIGTERM", stop); + let summary; + try { + summary = await runSimulation(cfg, selection, target, { + watch: parsed.watch, + iterations: parsed.iterations, + transport: parsed.transport, + timeoutMs: + parsed.timeoutMinutes === undefined + ? undefined + : parsed.timeoutMinutes * 60_000, + signal: controller.signal, + }); + } finally { + process.off("SIGINT", stop); + process.off("SIGTERM", stop); + } console.log(`\n${formatSummary(summary)}\n`); - - if (summary.fail > 0) { - console.error( - `❌ Simulation run failed (${summary.fail} fail / ${summary.pass} pass)`, - ); - process.exit(1); + if (!summary.verdict) { + console.log("▶️ Run started (not watched)."); + return 0; + } + if (summary.verdict.status === "passed") { + console.log("✅ Simulation run passed."); + return 0; } - console.log("✅ Simulation run passed."); + if (summary.verdict.status === "failed") { + console.error(`❌ Simulation run failed: ${summary.verdict.reason}`); + return 1; + } + console.error(`⚠️ Simulation run incomplete: ${summary.verdict.reason}`); + return 3; } -main().catch((error) => { - console.error( - "\n❌ Sim failed:", - error instanceof Error ? error.message : error, +const isMainModule = + process.argv[1] !== undefined && + resolve(process.argv[1]) === fileURLToPath(import.meta.url); +if (isMainModule) { + simCommandRun().then( + (code) => { + process.exitCode = code; + }, + (error: unknown) => { + console.error( + "\n❌ Sim failed:", + error instanceof Error ? error.message : error, + ); + process.exitCode = 2; + }, ); - process.exit(1); -}); +} diff --git a/src/sim-result.ts b/src/sim-result.ts new file mode 100644 index 0000000..9869c41 --- /dev/null +++ b/src/sim-result.ts @@ -0,0 +1,202 @@ +// Strict pass/fail verdict for a simulation run. +// +// Pure (no I/O, no config.ts) so `npm run sim` and the PR check share it and +// tests can pin every case. The API shapes it reads: +// - GET /eval/simulation/run/:id → a run with `status` (queued | running | +// ended) and `itemCounts` (total, passed, failed, running, queued, +// canceled). There is no `results` field on the run. +// - GET /eval/simulation/run/:id/item → items with `status` (queued | +// running | evaluating | passed | failed | canceled) and, once +// evaluated, `results.evaluations[]`. +// +// A run only passes when every expected item exists, passed, and had at +// least one required evaluation that was actually scored. "All evaluations +// skipped" is not a pass (see docs/learnings/simulations.md). + +export interface SimRunItemCounts { + total: number; + passed: number; + failed: number; + running: number; + queued: number; + canceled: number; +} + +export interface SimRun { + id: string; + status?: string; + itemCounts?: SimRunItemCounts; +} + +export interface SimRunEvaluation { + name?: string; + comparator?: string; + expectedValue?: unknown; + extractedValue?: unknown; + passed?: boolean; + required?: boolean; + isSkipped?: boolean; + skipReason?: string; + error?: string; +} + +export interface SimRunItem { + id: string; + status?: string; + results?: { passed?: boolean; evaluations?: SimRunEvaluation[] }; + metadata?: { simulation?: { name?: string } }; +} + +export type SimRunVerdictStatus = "passed" | "failed" | "incomplete"; + +export interface SimRunFailure { + item: string; + evaluation: string; + comparator?: string; + expected?: unknown; + extracted?: unknown; + reason?: string; +} + +export interface SimRunVerdict { + status: SimRunVerdictStatus; + reason: string; + failures: SimRunFailure[]; +} + +const TERMINAL_ITEM_STATUSES = new Set(["passed", "failed", "canceled"]); + +export function simRunItemTerminal(item: SimRunItem): boolean { + if (!TERMINAL_ITEM_STATUSES.has(item.status ?? "")) return false; + // A passed/failed item without results is still being written. + return item.status === "canceled" || item.results !== undefined; +} + +function itemLabel(item: SimRunItem): string { + return item.metadata?.simulation?.name ?? item.id; +} + +function requiredScored(item: SimRunItem): boolean { + return (item.results?.evaluations ?? []).some( + (evaluation) => + evaluation.required !== false && evaluation.isSkipped !== true, + ); +} + +function failuresList(items: SimRunItem[]): SimRunFailure[] { + const failures: SimRunFailure[] = []; + for (const item of items) { + if (item.status !== "failed") continue; + const failed = (item.results?.evaluations ?? []).filter( + (evaluation) => + evaluation.required !== false && + (evaluation.passed === false || evaluation.error !== undefined), + ); + if (failed.length === 0) { + failures.push({ + item: itemLabel(item), + evaluation: "(no evaluation detail)", + }); + } + for (const evaluation of failed) { + failures.push({ + item: itemLabel(item), + evaluation: evaluation.name ?? "(unnamed)", + comparator: evaluation.comparator, + expected: evaluation.expectedValue, + extracted: evaluation.extractedValue, + reason: evaluation.error ?? evaluation.skipReason, + }); + } + } + return failures; +} + +// `expected` is the number of items the run should have (simulations × +// iterations). Pass `undefined` when it can't be known up front (a suite run +// by ID); the verdict then requires at least one item. +export function simRunVerdict(input: { + run: SimRun; + items: SimRunItem[]; + expected?: number; +}): SimRunVerdict { + const { run, items, expected } = input; + const failures = failuresList(items); + const counts = run.itemCounts; + + if (counts && (counts.failed > 0 || failures.length > 0)) { + return { + status: "failed", + reason: `${counts.failed} of ${counts.total} simulations failed`, + failures, + }; + } + if (run.status !== "ended") { + return { + status: "incomplete", + reason: `run did not end (status: ${run.status ?? "unknown"})`, + failures, + }; + } + if (!counts) { + return { status: "incomplete", reason: "run has no item counts", failures }; + } + if (counts.total === 0) { + return { status: "incomplete", reason: "run has no items", failures }; + } + if (expected !== undefined && counts.total !== expected) { + return { + status: "incomplete", + reason: `expected ${expected} items, run has ${counts.total}`, + failures, + }; + } + if (counts.canceled > 0) { + return { + status: "incomplete", + reason: `${counts.canceled} of ${counts.total} simulations were canceled`, + failures, + }; + } + if ( + counts.queued > 0 || + counts.running > 0 || + counts.passed !== counts.total + ) { + return { + status: "incomplete", + reason: `only ${counts.passed} of ${counts.total} simulations finished`, + failures, + }; + } + if (items.length !== counts.total) { + return { + status: "incomplete", + reason: `fetched ${items.length} of ${counts.total} items`, + failures, + }; + } + const notPassed = items.filter((item) => item.status !== "passed"); + if (notPassed.length > 0) { + return { + status: "incomplete", + reason: `${notPassed.length} items are not marked passed`, + failures, + }; + } + const unscored = items.filter((item) => !requiredScored(item)); + if (unscored.length > 0) { + return { + status: "incomplete", + reason: `${unscored.length} simulations had every required evaluation skipped (${unscored + .map(itemLabel) + .join(", ")})`, + failures, + }; + } + return { + status: "passed", + reason: `${counts.passed} of ${counts.total} simulations passed`, + failures, + }; +} diff --git a/src/sim.ts b/src/sim.ts index c63a4e6..f082f9e 100644 --- a/src/sim.ts +++ b/src/sim.ts @@ -8,8 +8,21 @@ import { missingApiKeyMessage, resolveApiKey } from "./api-key.ts"; import { existsSync, readFileSync } from "fs"; import { dirname, join } from "path"; import { fileURLToPath } from "url"; +import { + simRunItemTerminal, + simRunVerdict, + type SimRun, + type SimRunItem, + type SimRunItemCounts, + type SimRunVerdict, +} from "./sim-result.ts"; import type { StateFile } from "./types.ts"; import { userAgentGet } from "./user-agent.ts"; +import { + VapiApiError, + vapiFetchJson, + type VapiConnection, +} from "./vapi-client.ts"; const __dirname = dirname(fileURLToPath(import.meta.url)); const BASE_DIR = join(__dirname, ".."); @@ -40,14 +53,26 @@ export interface SimRunOptions { watch?: boolean; iterations?: number; transport?: "voice" | "chat"; + // Give up (and cancel the run) after this long. Default 20 minutes. + timeoutMs?: number; + // Aborting cancels the run and reports it as incomplete (Ctrl-C). + signal?: AbortSignal; + pollIntervalMs?: number; + // How long to keep re-reading items after the run ends, while their + // results are still being written. Default 2 minutes. + hydrationMs?: number; } export interface SimRunSummary { runId: string; + // Dashboard link from the create response (GET doesn't return it). + url?: string; status: string; - pass: number; - fail: number; - skipped: number; + // Undefined when the run wasn't watched (`--no-watch`). + verdict?: SimRunVerdict; + counts?: SimRunItemCounts; + // True when this command canceled the run (timeout or interrupt). + canceled: boolean; durationMs: number; } @@ -195,44 +220,95 @@ export function resolveSelection( throw new Error("Must specify --suite or --simulations "); } -interface SimRunResponse { - id?: string; - evalRunId?: string; - status?: string; - results?: Array<{ status?: string; isSkipped?: boolean }>; - endedReason?: string; - endedMessage?: string; - cost?: number; - [key: string]: unknown; +// Fields of `POST /eval/simulation/run`'s response this runner reads. The +// create response is the only place `url` and `simulationRunItemIds` appear. +interface SimRunCreated extends SimRun { + url?: string; + simulationRunItemIds?: string[]; } const POLL_INTERVAL_MS = 3000; -const POLL_TIMEOUT_MS = 600_000; +const DEFAULT_TIMEOUT_MS = 20 * 60_000; +const DEFAULT_HYDRATION_MS = 2 * 60_000; +// The API's maximum page size; one page covers nearly every run. +const ITEM_PAGE_SIZE = 1000; -async function fetchJson( - cfg: SimEnv, - method: "GET" | "POST", - endpoint: string, - body?: unknown, -): Promise { - const response = await fetch(`${cfg.baseUrl}${endpoint}`, { - method, - headers: { - Authorization: `Bearer ${cfg.token}`, - "Content-Type": "application/json", - "User-Agent": userAgentGet("sim"), - }, - ...(body ? { body: JSON.stringify(body) } : {}), +function connectionFor(cfg: SimEnv): VapiConnection { + return { + token: cfg.token, + baseUrl: cfg.baseUrl, + userAgent: userAgentGet("sim"), + }; +} + +// Resolves after `ms`, or early (to "aborted") when the signal fires. +function sleepUnlessAborted( + ms: number, + signal?: AbortSignal, +): Promise<"slept" | "aborted"> { + if (signal?.aborted) return Promise.resolve("aborted"); + return new Promise((resolve) => { + const timer = setTimeout(() => { + signal?.removeEventListener("abort", onAbort); + resolve("slept"); + }, ms); + const onAbort = () => { + clearTimeout(timer); + resolve("aborted"); + }; + signal?.addEventListener("abort", onAbort, { once: true }); }); - if (!response.ok) { - const text = await response.text(); - throw new Error(`API ${method} ${endpoint} → ${response.status}: ${text}`); +} + +// Reads every item of a run. Accepts both the paginated shape +// (`{ results, metadata }`, sent when `limit`/`page` are given) and a bare +// array. Items are deduped by id because the API orders pages only by +// creation time, and a run's items share it, so OFFSET pages can overlap. +export async function simRunItemsFetch( + connection: VapiConnection, + runId: string, +): Promise { + const byId = new Map(); + for (let page = 1; page <= 100; page++) { + const response = await vapiFetchJson< + | SimRunItem[] + | { results?: SimRunItem[]; metadata?: { totalItems?: number } } + >( + connection, + "GET", + `/eval/simulation/run/${runId}/item?page=${page}&limit=${ITEM_PAGE_SIZE}`, + ); + if (Array.isArray(response)) { + for (const item of response) byId.set(item.id, item); + break; + } + const results = response?.results ?? []; + for (const item of results) byId.set(item.id, item); + const total = response?.metadata?.totalItems; + if (results.length < ITEM_PAGE_SIZE) break; + if (total !== undefined && byId.size >= total) break; } - return response.json(); + return [...byId.values()]; } -function sleep(ms: number): Promise { - return new Promise((r) => setTimeout(r, ms)); +// Cancels a run. Returns false (instead of throwing) when the run had +// already ended or a concurrent cancel won the race (400 / 409). +export async function simRunCancel( + connection: VapiConnection, + runId: string, +): Promise { + try { + await vapiFetchJson(connection, "PATCH", `/eval/simulation/run/${runId}`); + return true; + } catch (error) { + if ( + error instanceof VapiApiError && + (error.statusCode === 400 || error.statusCode === 409) + ) { + return false; + } + throw error; + } } export async function runSimulation( @@ -241,6 +317,8 @@ export async function runSimulation( target: SimTarget, options: SimRunOptions = {}, ): Promise { + const connection = connectionFor(cfg); + const pollIntervalMs = options.pollIntervalMs ?? POLL_INTERVAL_MS; const body: Record = { simulations: selection.entries, target: @@ -258,68 +336,135 @@ export async function runSimulation( `🧪 Starting simulation run — ${selection.label} → ${target.type}/${target.resourceName}`, ); const start = Date.now(); - const created = (await fetchJson( - cfg, + // Creating a run queues paid work before the response returns, so a 5xx + // here may still have started it: retry rate limits only. + const created = await vapiFetchJson( + connection, "POST", "/eval/simulation/run", body, - )) as SimRunResponse; - const runId = created.evalRunId ?? created.id; + { retry: "rate-limit-only" }, + ); + const runId = created?.id; if (!runId) { - throw new Error( - `POST /eval/simulation/run returned no runId (keys: ${Object.keys(created).join(", ")})`, - ); + throw new Error("POST /eval/simulation/run returned no run id"); } console.log(` Run ID: ${runId}`); + if (created.url) console.log(` Run: ${created.url}`); - let last: SimRunResponse = created; - if (options.watch ?? true) { - while (Date.now() - start < POLL_TIMEOUT_MS) { - await sleep(POLL_INTERVAL_MS); - last = (await fetchJson( - cfg, - "GET", - `/eval/simulation/run/${runId}`, - )) as SimRunResponse; - const status = last.status ?? "running"; - process.stdout.write(`\r Status: ${status} `); - if (status === "ended" || status === "failed" || status === "completed") { - process.stdout.write("\n"); - break; - } + const summary: SimRunSummary = { + runId, + url: created.url, + status: created.status ?? "queued", + counts: created.itemCounts, + canceled: false, + durationMs: 0, + }; + if (!(options.watch ?? true)) { + summary.durationMs = Date.now() - start; + return summary; + } + + const deadline = start + (options.timeoutMs ?? DEFAULT_TIMEOUT_MS); + let run: SimRun = created; + let stopReason: string | undefined; + while (run.status !== "ended") { + const remaining = deadline - Date.now(); + if (remaining <= 0) { + stopReason = `timed out after ${Math.round((Date.now() - start) / 1000)}s`; + break; } - if (Date.now() - start >= POLL_TIMEOUT_MS) { - throw new Error( - `Simulation run ${runId} timed out after ${POLL_TIMEOUT_MS / 1000}s`, - ); + const slept = await sleepUnlessAborted( + Math.min(pollIntervalMs, remaining), + options.signal, + ); + if (slept === "aborted") { + stopReason = "interrupted"; + break; + } + run = await vapiFetchJson( + connection, + "GET", + `/eval/simulation/run/${runId}`, + ); + if (process.stdout.isTTY) { + process.stdout.write(`\r Status: ${run.status ?? "unknown"} `); } } + if (process.stdout.isTTY) process.stdout.write("\n"); - const results = Array.isArray(last.results) ? last.results : []; - const pass = results.filter( - (r) => r.status === "pass" && !r.isSkipped, - ).length; - const fail = results.filter( - (r) => r.status !== "pass" && !r.isSkipped, - ).length; - const skipped = results.filter((r) => r.isSkipped === true).length; + if (stopReason) { + summary.canceled = await simRunCancel(connection, runId); + summary.status = run.status ?? "unknown"; + summary.counts = run.itemCounts; + summary.verdict = { + status: "incomplete", + reason: `${stopReason}${summary.canceled ? "; run canceled" : ""}`, + failures: [], + }; + summary.durationMs = Date.now() - start; + return summary; + } - return { - runId, - status: last.status ?? "unknown", - pass, - fail, - skipped, - durationMs: Date.now() - start, - }; + // Items can lag the run: keep re-reading until every item is terminal and + // carries its results, or the hydration window closes. + const hydrationDeadline = + Date.now() + (options.hydrationMs ?? DEFAULT_HYDRATION_MS); + let items = await simRunItemsFetch(connection, runId); + while ( + Date.now() < hydrationDeadline && + (items.length < (run.itemCounts?.total ?? 0) || + !items.every(simRunItemTerminal)) + ) { + if ( + (await sleepUnlessAborted(pollIntervalMs, options.signal)) === "aborted" + ) { + break; + } + items = await simRunItemsFetch(connection, runId); + } + + summary.status = run.status ?? "unknown"; + summary.counts = run.itemCounts; + summary.verdict = simRunVerdict({ + run, + items, + expected: created.simulationRunItemIds?.length, + }); + summary.durationMs = Date.now() - start; + return summary; +} + +function valueFormat(value: unknown): string { + return typeof value === "string" ? value : JSON.stringify(value); } export function formatSummary(summary: SimRunSummary): string { - const total = summary.pass + summary.fail + summary.skipped; - return [ - `📊 Simulation summary (run ${summary.runId})`, - ` Status: ${summary.status}`, - ` Results: ${summary.pass}/${total} pass, ${summary.fail} fail${summary.skipped > 0 ? `, ${summary.skipped} skipped` : ""}`, - ` Duration: ${(summary.durationMs / 1000).toFixed(1)}s`, - ].join("\n"); + const lines = [`📊 Simulation summary (run ${summary.runId})`]; + if (summary.url) lines.push(` Run: ${summary.url}`); + lines.push(` Status: ${summary.status}`); + if (summary.counts) { + const c = summary.counts; + lines.push( + ` Items: ${c.passed} passed, ${c.failed} failed, ${c.canceled} canceled, ${c.running + c.queued} unfinished (of ${c.total})`, + ); + } + if (summary.verdict) { + lines.push( + ` Verdict: ${summary.verdict.status} — ${summary.verdict.reason}`, + ); + for (const failure of summary.verdict.failures) { + const detail = + failure.comparator !== undefined + ? ` (expected ${failure.comparator} ${valueFormat(failure.expected)}, got ${valueFormat(failure.extracted)})` + : ""; + lines.push( + ` ✗ ${failure.item}: ${failure.evaluation}${detail}${failure.reason ? ` — ${failure.reason}` : ""}`, + ); + } + } else { + lines.push(" Verdict: not watched (--no-watch)"); + } + lines.push(` Duration: ${(summary.durationMs / 1000).toFixed(1)}s`); + return lines.join("\n"); } diff --git a/src/vapi-client.ts b/src/vapi-client.ts new file mode 100644 index 0000000..7c32959 --- /dev/null +++ b/src/vapi-client.ts @@ -0,0 +1,108 @@ +// Config-free HTTP client for the Vapi API. +// +// api.ts is the engine's client, but it imports config.ts, which parses +// argv and exits at import time and binds a single org. Code that must run +// for several orgs, without a token, or from tests (sim.ts, the check +// runner) uses this client instead. + +export interface VapiConnection { + token: string; + baseUrl: string; + userAgent: string; +} + +// "transient": retry 429 and 5xx. "rate-limit-only": retry 429 only — for +// requests that aren't safe to repeat after a 5xx, because the server may +// have acted before failing (creating a simulation run queues paid work). +export type VapiRetryPolicy = "transient" | "rate-limit-only" | "none"; + +export interface VapiFetchOptions { + retry?: VapiRetryPolicy; + maxRetries?: number; + initialDelayMs?: number; + sleep?: (ms: number) => Promise; +} + +export class VapiApiError extends Error { + constructor( + public readonly method: string, + public readonly endpoint: string, + public readonly statusCode: number, + public readonly apiMessage: string, + public readonly rawBody: string, + ) { + super(`API ${method} ${endpoint} failed (${statusCode}): ${apiMessage}`); + this.name = "VapiApiError"; + } +} + +export function parseApiMessage(body: string): string { + try { + const parsed = JSON.parse(body); + if (typeof parsed.message === "string") return parsed.message; + if (Array.isArray(parsed.message)) return parsed.message.join("; "); + } catch { + /* not JSON, use raw body */ + } + return body; +} + +// 429 = rate limit. 5xx = transient server error (gateway timeout, upstream +// hiccup, deploy in progress). +export function shouldRetry(status: number): boolean { + return status === 429 || (status >= 500 && status < 600); +} + +export const MAX_RETRIES = 5; +export const INITIAL_DELAY_MS = 2000; + +function sleepDefault(ms: number): Promise { + return new Promise((resolve) => setTimeout(resolve, ms)); +} + +function retryable(policy: VapiRetryPolicy, status: number): boolean { + if (policy === "none") return false; + if (policy === "rate-limit-only") return status === 429; + return shouldRetry(status); +} + +export async function vapiFetchJson( + connection: VapiConnection, + method: "GET" | "POST" | "PATCH", + endpoint: string, + body?: unknown, + options: VapiFetchOptions = {}, +): Promise { + const policy = options.retry ?? "transient"; + const maxRetries = options.maxRetries ?? MAX_RETRIES; + const initialDelayMs = options.initialDelayMs ?? INITIAL_DELAY_MS; + const sleep = options.sleep ?? sleepDefault; + + for (let attempt = 0; ; attempt++) { + const response = await fetch(`${connection.baseUrl}${endpoint}`, { + method, + headers: { + Authorization: `Bearer ${connection.token}`, + "Content-Type": "application/json", + "User-Agent": connection.userAgent, + }, + ...(body === undefined ? {} : { body: JSON.stringify(body) }), + }); + if (response.ok) { + const text = await response.text(); + return (text ? JSON.parse(text) : null) as T; + } + if (attempt < maxRetries && retryable(policy, response.status)) { + await sleep(initialDelayMs * 2 ** attempt); + continue; + } + const rawBody = await response.text(); + throw new VapiApiError( + method, + endpoint, + response.status, + parseApiMessage(rawBody), + rawBody, + ); + } +} diff --git a/tests/sim-result.test.ts b/tests/sim-result.test.ts new file mode 100644 index 0000000..4bea397 --- /dev/null +++ b/tests/sim-result.test.ts @@ -0,0 +1,178 @@ +import assert from "node:assert/strict"; +import test from "node:test"; +import { + simRunItemTerminal, + simRunVerdict, + type SimRunItem, + type SimRunItemCounts, +} from "../src/sim-result.ts"; + +// The verdict `npm run sim` and the PR check share. It must never report a +// pass unless every expected simulation ran, passed, and was actually +// scored — the old runner read a `results` field the API doesn't return and +// reported 0/0 as a pass. + +function counts(overrides: Partial = {}): SimRunItemCounts { + return { + total: 2, + passed: 2, + failed: 0, + running: 0, + queued: 0, + canceled: 0, + ...overrides, + }; +} + +function item( + id: string, + status: string, + evaluations: NonNullable["evaluations"] = [ + { name: "goal-met", required: true, passed: status === "passed" }, + ], +): SimRunItem { + return { + id, + status, + results: { passed: status === "passed", evaluations }, + metadata: { simulation: { name: `sim ${id}` } }, + }; +} + +const ended = (itemCounts?: SimRunItemCounts) => ({ + id: "run-1", + status: "ended", + itemCounts, +}); + +test("simRunVerdict: passes when every expected item passed and was scored", () => { + const verdict = simRunVerdict({ + run: ended(counts()), + items: [item("a", "passed"), item("b", "passed")], + expected: 2, + }); + assert.equal(verdict.status, "passed"); +}); + +test("simRunVerdict: the old false-green shape (ended, no results) is not a pass", () => { + // What GET /eval/simulation/run/:id actually returns: no `results` field. + // With no items fetched, the run can't be called passed. + const verdict = simRunVerdict({ run: ended(counts()), items: [] }); + assert.equal(verdict.status, "incomplete"); + assert.match(verdict.reason, /fetched 0 of 2 items/); +}); + +test("simRunVerdict: 0 items is incomplete, never a pass", () => { + const verdict = simRunVerdict({ + run: ended(counts({ total: 0, passed: 0 })), + items: [], + }); + assert.equal(verdict.status, "incomplete"); +}); + +test("simRunVerdict: a failed item fails the run and lists the failing judge", () => { + const verdict = simRunVerdict({ + run: ended(counts({ passed: 1, failed: 1 })), + items: [ + item("a", "passed"), + item("b", "failed", [ + { + name: "booking-confirmed", + required: true, + passed: false, + comparator: "=", + expectedValue: true, + extractedValue: false, + }, + ]), + ], + }); + assert.equal(verdict.status, "failed"); + assert.deepEqual(verdict.failures, [ + { + item: "sim b", + evaluation: "booking-confirmed", + comparator: "=", + expected: true, + extracted: false, + reason: undefined, + }, + ]); +}); + +test("simRunVerdict: canceled items make the run incomplete", () => { + const verdict = simRunVerdict({ + run: ended(counts({ passed: 1, canceled: 1 })), + items: [item("a", "passed"), item("b", "canceled")], + }); + assert.equal(verdict.status, "incomplete"); + assert.match(verdict.reason, /canceled/); +}); + +test("simRunVerdict: all required evaluations skipped is not a pass", () => { + const skipped = [{ name: "audio-check", required: true, isSkipped: true }]; + const verdict = simRunVerdict({ + run: ended(counts()), + items: [item("a", "passed"), item("b", "passed", skipped)], + }); + assert.equal(verdict.status, "incomplete"); + assert.match(verdict.reason, /every required evaluation skipped/); +}); + +test("simRunVerdict: a skipped optional judge doesn't hide a scored required one", () => { + const verdict = simRunVerdict({ + run: ended(counts({ total: 1, passed: 1 })), + items: [ + item("a", "passed", [ + { name: "tone", required: false, isSkipped: true }, + { name: "goal-met", required: true, passed: true }, + ]), + ], + }); + assert.equal(verdict.status, "passed"); +}); + +test("simRunVerdict: missing itemCounts is incomplete", () => { + const verdict = simRunVerdict({ + run: ended(undefined), + items: [item("a", "passed")], + }); + assert.equal(verdict.status, "incomplete"); +}); + +test("simRunVerdict: a count different from the expected one is incomplete", () => { + const verdict = simRunVerdict({ + run: ended(counts()), + items: [item("a", "passed"), item("b", "passed")], + expected: 3, + }); + assert.equal(verdict.status, "incomplete"); + assert.match(verdict.reason, /expected 3 items/); +}); + +test("simRunVerdict: a run that hasn't ended is incomplete", () => { + const verdict = simRunVerdict({ + run: { + id: "run-1", + status: "running", + itemCounts: counts({ passed: 1, running: 1 }), + }, + items: [item("a", "passed")], + }); + assert.equal(verdict.status, "incomplete"); +}); + +test("simRunVerdict: a short item list is incomplete even when counts say passed", () => { + const verdict = simRunVerdict({ + run: ended(counts()), + items: [item("a", "passed")], + }); + assert.equal(verdict.status, "incomplete"); +}); + +test("simRunItemTerminal: passed or failed items need results; canceled doesn't", () => { + assert.equal(simRunItemTerminal({ id: "a", status: "passed" }), false); + assert.equal(simRunItemTerminal(item("a", "passed")), true); + assert.equal(simRunItemTerminal({ id: "a", status: "canceled" }), true); + assert.equal(simRunItemTerminal({ id: "a", status: "evaluating" }), false); +}); diff --git a/tests/sim-run.test.ts b/tests/sim-run.test.ts new file mode 100644 index 0000000..aedc215 --- /dev/null +++ b/tests/sim-run.test.ts @@ -0,0 +1,368 @@ +import assert from "node:assert/strict"; +import { createServer, type IncomingMessage } from "node:http"; +import type { AddressInfo } from "node:net"; +import test from "node:test"; +import { runSimulation, simRunItemsFetch } from "../src/sim.ts"; + +// runSimulation against a local stub of the simulations API. Each stub +// route returns a canned response; requests are recorded so tests can +// assert on what was sent (cancel, retries). + +type Handler = ( + req: IncomingMessage, + body: string, +) => { + status?: number; + json?: unknown; +}; + +async function withServer( + routes: Array<{ method: string; path: RegExp; handle: Handler }>, + fn: (baseUrl: string, seen: string[]) => Promise, +): Promise { + const seen: string[] = []; + const server = createServer((req, res) => { + let body = ""; + req.on("data", (chunk) => (body += chunk)); + req.on("end", () => { + seen.push(`${req.method} ${req.url}`); + const route = routes.find( + (r) => r.method === req.method && r.path.test(req.url ?? ""), + ); + const out = route ? route.handle(req, body) : { status: 404, json: {} }; + res.writeHead(out.status ?? 200, { "Content-Type": "application/json" }); + res.end(JSON.stringify(out.json ?? {})); + }); + }); + await new Promise((resolve) => server.listen(0, resolve)); + const { port } = server.address() as AddressInfo; + const log = console.log; + console.log = () => {}; + try { + await fn(`http://127.0.0.1:${port}`, seen); + } finally { + console.log = log; + await new Promise((resolve) => server.close(() => resolve())); + } +} + +const cfg = (baseUrl: string) => ({ env: "test-org", token: "t", baseUrl }); +const selection = { + entries: [{ type: "simulationSuite" as const, simulationSuiteId: "suite-1" }], + label: "suite test", +}; +const target = { type: "squad" as const, id: "squad-1", resourceName: "s" }; +const fast = { pollIntervalMs: 1, hydrationMs: 50 }; + +const endedRun = (passed: number, failed: number) => ({ + id: "run-1", + status: "ended", + itemCounts: { + total: passed + failed, + passed, + failed, + running: 0, + queued: 0, + canceled: 0, + }, +}); + +const scoredItem = (id: string, status: "passed" | "failed") => ({ + id, + status, + results: { + passed: status === "passed", + evaluations: [ + { + name: "goal-met", + required: true, + passed: status === "passed", + comparator: "=", + expectedValue: true, + extractedValue: status === "passed", + }, + ], + }, + metadata: { simulation: { name: `sim ${id}` } }, +}); + +test("runSimulation: passes when the ended run's items all passed", async () => { + await withServer( + [ + { + method: "POST", + path: /^\/eval\/simulation\/run$/, + handle: () => ({ + status: 201, + json: { + id: "run-1", + status: "queued", + url: "https://dashboard.vapi.ai/simulations/run/run-1", + simulationRunItemIds: ["a", "b"], + }, + }), + }, + { + method: "GET", + path: /^\/eval\/simulation\/run\/run-1$/, + handle: () => ({ json: endedRun(2, 0) }), + }, + { + method: "GET", + path: /^\/eval\/simulation\/run\/run-1\/item/, + handle: () => ({ + json: { + results: [scoredItem("a", "passed"), scoredItem("b", "passed")], + metadata: { totalItems: 2, itemsPerPage: 1000, currentPage: 1 }, + }, + }), + }, + ], + async (baseUrl) => { + const summary = await runSimulation( + cfg(baseUrl), + selection, + target, + fast, + ); + assert.equal(summary.verdict?.status, "passed"); + assert.equal( + summary.url, + "https://dashboard.vapi.ai/simulations/run/run-1", + ); + }, + ); +}); + +test("runSimulation: a failed item fails the run (the old runner reported a pass)", async () => { + await withServer( + [ + { + method: "POST", + path: /^\/eval\/simulation\/run$/, + handle: () => ({ + status: 201, + json: { id: "run-1", status: "queued" }, + }), + }, + { + method: "GET", + path: /^\/eval\/simulation\/run\/run-1$/, + handle: () => ({ json: endedRun(1, 1) }), + }, + { + method: "GET", + path: /^\/eval\/simulation\/run\/run-1\/item/, + // The bare-array shape the API returns without page/limit. + handle: () => ({ + json: [scoredItem("a", "passed"), scoredItem("b", "failed")], + }), + }, + ], + async (baseUrl) => { + const summary = await runSimulation( + cfg(baseUrl), + selection, + target, + fast, + ); + assert.equal(summary.verdict?.status, "failed"); + assert.equal(summary.verdict?.failures[0]?.item, "sim b"); + }, + ); +}); + +test("runSimulation: waits for item results that arrive after the run ends", async () => { + let reads = 0; + await withServer( + [ + { + method: "POST", + path: /^\/eval\/simulation\/run$/, + handle: () => ({ + status: 201, + json: { id: "run-1", status: "queued" }, + }), + }, + { + method: "GET", + path: /^\/eval\/simulation\/run\/run-1$/, + handle: () => ({ json: endedRun(1, 0) }), + }, + { + method: "GET", + path: /^\/eval\/simulation\/run\/run-1\/item/, + handle: () => { + reads++; + // First read: the item is passed but its results aren't written yet. + return { + json: + reads === 1 + ? [{ id: "a", status: "passed" }] + : [scoredItem("a", "passed")], + }; + }, + }, + ], + async (baseUrl) => { + const summary = await runSimulation(cfg(baseUrl), selection, target, { + pollIntervalMs: 1, + hydrationMs: 5_000, + }); + assert.equal(summary.verdict?.status, "passed"); + assert.ok(reads >= 2); + }, + ); +}); + +test("runSimulation: a timeout cancels the run and reports incomplete", async () => { + await withServer( + [ + { + method: "POST", + path: /^\/eval\/simulation\/run$/, + handle: () => ({ + status: 201, + json: { id: "run-1", status: "queued" }, + }), + }, + { + method: "GET", + path: /^\/eval\/simulation\/run\/run-1$/, + handle: () => ({ json: { id: "run-1", status: "running" } }), + }, + { + method: "PATCH", + path: /^\/eval\/simulation\/run\/run-1$/, + handle: () => ({ json: { id: "run-1", status: "ended" } }), + }, + ], + async (baseUrl, seen) => { + const summary = await runSimulation(cfg(baseUrl), selection, target, { + pollIntervalMs: 5, + timeoutMs: 30, + }); + assert.equal(summary.verdict?.status, "incomplete"); + assert.match(summary.verdict?.reason ?? "", /timed out/); + assert.equal(summary.canceled, true); + assert.ok(seen.includes("PATCH /eval/simulation/run/run-1")); + }, + ); +}); + +test("runSimulation: an interrupt cancels; a 400 'already ended' is swallowed", async () => { + const controller = new AbortController(); + await withServer( + [ + { + method: "POST", + path: /^\/eval\/simulation\/run$/, + handle: () => ({ + status: 201, + json: { id: "run-1", status: "queued" }, + }), + }, + { + method: "GET", + path: /^\/eval\/simulation\/run\/run-1$/, + handle: () => { + controller.abort(); + return { json: { id: "run-1", status: "running" } }; + }, + }, + { + method: "PATCH", + path: /^\/eval\/simulation\/run\/run-1$/, + handle: () => ({ + status: 400, + json: { message: "Run has already ended" }, + }), + }, + ], + async (baseUrl) => { + const summary = await runSimulation(cfg(baseUrl), selection, target, { + pollIntervalMs: 1, + signal: controller.signal, + }); + assert.equal(summary.verdict?.status, "incomplete"); + assert.match(summary.verdict?.reason ?? "", /interrupted/); + assert.equal(summary.canceled, false); + }, + ); +}); + +test("runSimulation: a 5xx on run create is not retried (it may already have queued the run)", async () => { + await withServer( + [ + { + method: "POST", + path: /^\/eval\/simulation\/run$/, + handle: () => ({ status: 502, json: { message: "Bad Gateway" } }), + }, + ], + async (baseUrl, seen) => { + await assert.rejects( + runSimulation(cfg(baseUrl), selection, target, fast), + /failed \(502\)/, + ); + assert.equal(seen.filter((s) => s.startsWith("POST")).length, 1); + }, + ); +}); + +test("runSimulation: --no-watch returns after create with the link and no verdict", async () => { + await withServer( + [ + { + method: "POST", + path: /^\/eval\/simulation\/run$/, + handle: () => ({ + status: 201, + json: { id: "run-1", status: "queued", url: "https://x/run-1" }, + }), + }, + ], + async (baseUrl, seen) => { + const summary = await runSimulation(cfg(baseUrl), selection, target, { + watch: false, + }); + assert.equal(summary.verdict, undefined); + assert.equal(summary.url, "https://x/run-1"); + assert.deepEqual(seen, ["POST /eval/simulation/run"]); + }, + ); +}); + +test("simRunItemsFetch: pages until totalItems and dedupes overlapping pages", async () => { + const page = (ids: string[]) => ids.map((id) => ({ id, status: "passed" })); + const first = page(Array.from({ length: 1000 }, (_, i) => `i${i}`)); + await withServer( + [ + { + method: "GET", + path: /\/item\?page=1&/, + handle: () => ({ + json: { results: first, metadata: { totalItems: 1002 } }, + }), + }, + { + method: "GET", + path: /\/item\?page=2&/, + // Overlaps the first page by one item (OFFSET drift). + handle: () => ({ + json: { + results: page(["i999", "i1000", "i1001"]), + metadata: { totalItems: 1002 }, + }, + }), + }, + ], + async (baseUrl) => { + const items = await simRunItemsFetch( + { token: "t", baseUrl, userAgent: "test" }, + "run-1", + ); + assert.equal(items.length, 1002); + }, + ); +});