diff --git a/AGENTS.md b/AGENTS.md index 2a303f1..05ba2b2 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -902,7 +902,7 @@ npm run promote -- --pipeline --from --to --apply # For npm run validate -- # Lint resources locally (fails fast on schema drift) npm run audit -- # Read-only drift detector: orphan YAML, state ghosts, content-identical clusters, sibling base-slugs, dashboard orphans, inline model.tools. Exit 1 on findings. npm run audit -- --type assistants # Scope audit to a single resource type -npm run sim -- --suite --target # Run a simulation suite against an assistant/squad +npm run sim -- --suite --target # Run a simulation suite against an assistant/squad (exit 0 pass, 1 fail, 3 incomplete; --timeout ) npm run rollback -- --to # Re-apply a snapshot taken before a push npm run rollback -- --list # List available snapshots diff --git a/README.md b/README.md index f3401c4..180da18 100644 --- a/README.md +++ b/README.md @@ -103,7 +103,7 @@ Every command works in two modes: | `npm run cleanup` | ✅ | `npm run cleanup -- [--force --confirm ]` | Inspect (default) or delete orphaned remote resources. Destructive run requires `--confirm `. | | `npm run rollback` | — | `npm run rollback -- --list` or `--to ` | Restore from a snapshot in `.vapi-state..snapshots/` (one is written before every push/apply). | | `npm run call` | ✅ | `npm run call -- -a ` or `-s ` | Start an interactive WebSocket call against an assistant or squad. | -| `npm run sim` | — | `npm run sim -- --suite --target ` | Run a simulation suite (or specific simulations) against a deployed assistant/squad. | +| `npm run sim` | — | `npm run sim -- --suite --target [--timeout ]` | Run a simulation suite (or specific simulations) against a deployed assistant/squad. Prints the run link; exits 0 passed, 1 failed, 3 incomplete (timeout, Ctrl-C, missing results). | | `npm run migrate` | — | `npm run migrate` | One-time, all orgs at once: slim legacy state files to pure `name → uuid` and seed the per-developer `.vapi-state-hash/` baseline store from the old hashes. Required once after upgrading to the hash-store engine — `pull`/`push`/`apply` refuse legacy-shaped state until it runs. Idempotent. | | `npm run build` | — | — | Type-check the codebase (`tsc --noEmit`). | | `npm test` | — | — | Run regression tests (`node:test`). | diff --git a/docs/learnings/simulations.md b/docs/learnings/simulations.md index 96787bc..e94b18b 100644 --- a/docs/learnings/simulations.md +++ b/docs/learnings/simulations.md @@ -471,6 +471,23 @@ This is the unified executor. A "run" is a batch — it expands into many `runIt } ``` +The **create** response (`CreateSimulationRunResponse`) also carries two fields nothing else returns: + +- `url` — the dashboard link to this run. `GET /eval/simulation/run/:id` does **not** return it, so read it from the create response. +- `simulationRunItemIds` — one id per item queued (simulations × iterations), i.e. how many items the run should end with. + +A run has **no `results` field**. Pass/fail lives in `itemCounts` and on the items themselves (`GET …/item`, below). Creating a run queues paid work before the response returns, so a 5xx on create may still have started it — don't blindly retry the POST. + +### How `npm run sim` decides pass/fail + +`src/sim-result.ts` (`simRunVerdict`) reports **passed** only when all of these hold; otherwise **failed** (any failed item) or **incomplete**: + +- the run `ended` and has `itemCounts`, with `total` equal to the number of items created (`simulationRunItemIds`) and greater than 0; +- nothing is `queued`, `running` or `canceled`, and `passed === total`; +- every item was fetched, is `passed`, and has at least one **required** evaluation that wasn't skipped (the all-skipped trap from the chat-mode gotcha above). + +Exit codes: 0 passed, 1 failed, 2 usage error, 3 incomplete (timeout, Ctrl-C, canceled items, results that never arrived). On timeout or Ctrl-C the run is canceled. + ### List runs — `GET /eval/simulation/run` Query params: @@ -486,7 +503,7 @@ Query params: ### Get / Cancel run - `GET /eval/simulation/run/:id` → `SimulationRun` -- `PATCH /eval/simulation/run/:id` → cancels the run **and** all its queued items. No body required. +- `PATCH /eval/simulation/run/:id` → cancels the run **and** all its queued items. No body required. Returns 400 `Run has already ended` for an ended run and 409 when a concurrent cancel wins; both mean "nothing left to cancel". --- @@ -498,6 +515,8 @@ Run items are system-managed — there's no create/update API for users; they're Query params: `limit`, `page`, `simulationId`, `runId`, `status` (`queued` | `running` | `evaluating` | `passed` | `failed` | `canceled`). +Response shape depends on the query: **with** `limit` or `page` it's paginated (`{ results, metadata: { totalItems, itemsPerPage, currentPage } }`, `limit` up to 1000); **without** them it's a bare array. Pages are ordered only by creation time and a run's items share it, so OFFSET pages can overlap — dedupe by `id`. Items can also lag the run: a run can be `ended` while an item is `passed` but its `results` aren't written yet. + ### Get a run item — `GET /eval/simulation/run/:id/item/:itemId` **Response** — `SimulationRunItem` is rich; the highlights: diff --git a/improvements.md b/improvements.md index 44606c4..7472bd7 100644 --- a/improvements.md +++ b/improvements.md @@ -84,6 +84,7 @@ you which stack PR closes the row.** | 30 | Tool-linking pass could PATCH a raw assistant slug | Mid-push 400 naming the wrong resource | None | RESOLVED 2026-08-03 (#51) | | 31 | Unresolved references handled 3 inconsistent ways, no dangling-ref check | Same authoring mistake, three different failure modes | None | Open | | 32 | Test suite never ran in CI; 20 tests rotted after the hash store | Regression guards for #22/#23 silently stopped running | None | RESOLVED 2026-09-30 (#56) | +| 33 | `npm run sim` reported every run as passed | A failing suite exited 0 — false green | None | RESOLVED 2026-10-01 | **Active backlog after cleanup:** `#2`, `#6`, `#8`, `#12`, `#20`, `#24–#26`, `#31`, and the open remainder of `#27` (wiring the listing-completeness verdict into push/delete/audit, and moving `cleanup.ts` onto the shared pager). Resolved entries stay in this file as historical incident notes per the maintenance directive; stale superseded backlog rows are not duplicated. @@ -1721,6 +1722,57 @@ fields. --- +## 33. `npm run sim` reported every run as passed + +**[RESOLVED 2026-10-01]** + +**Discovered:** 2026-10-01, while designing simulation PR checks (TEST-141). + +### Problem + +`npm run sim` scored a run by reading `results[]` on the run and counting +`status === "pass"`. The simulation-run API has no `results` field and +item statuses are `passed` / `failed`, so every watched run summarised +as 0 pass / 0 fail and the command exited 0 — including runs whose +simulations failed. + +### Current behavior (Verified, before the fix) + +- `src/sim.ts` `runSimulation` computed pass/fail from `last.results`, + which `GET /eval/simulation/run/:id` never returns; pass/fail is in + `itemCounts` and on the run items (`GET /eval/simulation/run/:id/item`). +- `src/sim-cmd.ts` exited 1 only when `fail > 0`, which could never happen. +- Polling also stopped on statuses that don't exist (`failed`, + `completed`), never canceled a timed-out run, and didn't print the run + link (only the create response carries `url`). + +### Risk + +Anyone gating on `npm run sim` (locally or in CI) got a green result for a +failing suite. + +### Current mitigation + +None needed once the fix below lands. + +### Possible fix (landed) + +- `src/sim-result.ts` `simRunVerdict`: passed only when the run ended, + every expected item exists, passed, and had a required evaluation that was + actually scored; otherwise failed or incomplete with a reason. +- `src/sim.ts` reads items (paginated or bare-array, deduped by id), + waits for late item results, prints the run link and failing judges, and + cancels the run on `--timeout` (default 20 min) or Ctrl-C. +- `src/vapi-client.ts`: a config-free client that never retries run + creation on a 5xx (the run may already be queued). +- Exit codes: 0 passed, 1 failed, 2 usage, 3 incomplete. + +### Status + +**RESOLVED 2026-10-01.** + +--- + ## Out of scope (intentionally not improvements) - **State file is identity-only and not git-ignored.** It's intentionally diff --git a/src/sim-cmd.ts b/src/sim-cmd.ts index 2e3c531..e0b1fcb 100644 --- a/src/sim-cmd.ts +++ b/src/sim-cmd.ts @@ -2,7 +2,12 @@ // // Wraps `POST /eval/simulation/run` (the simulations API). See AGENTS.md for // usage. +// +// Exit codes: 0 passed, 1 failed, 2 usage/config error, 3 incomplete +// (timed out, interrupted, canceled, or results that never fully arrived). +import { resolve } from "node:path"; +import { fileURLToPath } from "node:url"; import { formatSummary, loadEnvFile, @@ -12,27 +17,28 @@ import { runSimulation, } from "./sim.ts"; -function printUsage(): void { - console.error( - [ - "Usage:", - " npm run sim -- --suite --target ", - " npm run sim -- --simulations , --target ", - "", - "Options:", - " --suite Run an entire simulation suite by local resource name", - " --simulations Run one or more simulations by comma-separated local names", - " --target Local assistant or squad name (resolves to UUID via state)", - " --transport voice|chat Transport (default: voice; chat is faster/cheaper)", - " --iterations N Override default iteration count", - " --watch Tail status until completion (default: on)", - "", - "Examples:", - " npm run sim -- my-org --suite booking-tests --target intake-agent", - " npm run sim -- my-org --simulations happy-path,edge-case --target main-agent --transport chat", - ].join("\n"), - ); -} +const USAGE = [ + "Usage:", + " npm run sim -- --suite --target ", + " npm run sim -- --simulations , --target ", + "", + "Options:", + " --suite Run an entire simulation suite by local resource name", + " --simulations Run one or more simulations by comma-separated local names", + " --target Local assistant or squad name (resolves to UUID via state)", + " --transport voice|chat Transport (default: voice; chat is faster/cheaper)", + " --iterations N Override default iteration count", + " --timeout Give up and cancel the run after this long (default: 20)", + " --no-watch Start the run, print its link, and exit 0 without a verdict", + "", + "Exit codes: 0 passed, 1 failed, 2 usage error, 3 incomplete (timeout, interrupt, missing results)", + "", + "Examples:", + " npm run sim -- my-org --suite booking-tests --target intake-agent", + " npm run sim -- my-org --simulations happy-path,edge-case --target main-agent --transport chat", +].join("\n"); + +class UsageError extends Error {} interface ParsedArgs { env: string; @@ -42,71 +48,83 @@ interface ParsedArgs { squad?: string; transport?: "voice" | "chat"; iterations?: number; + timeoutMinutes?: number; watch: boolean; + help: boolean; } -function parseArgs(): ParsedArgs { - const args = process.argv.slice(2); +function argsParse(args: string[]): ParsedArgs { const env = args[0]; - if (!env) { - printUsage(); - process.exit(1); + if (!env || env === "--help" || env === "-h") { + return { env: env ?? "", watch: true, help: true }; } const SLUG_RE = /^[a-z0-9]([a-z0-9-]*[a-z0-9])?$/; - if (!SLUG_RE.test(env)) { - console.error(`❌ Invalid org name: ${env}`); - process.exit(1); - } - - const parsed: ParsedArgs = { env, watch: true }; + if (!SLUG_RE.test(env)) throw new UsageError(`Invalid org name: ${env}`); + const parsed: ParsedArgs = { env, watch: true, help: false }; for (let i = 1; i < args.length; i++) { const arg = args[i]; if (arg === "--suite") parsed.suite = args[++i]; else if (arg === "--simulations") parsed.simulations = args[++i]; else if (arg === "--target") { // We don't know yet whether target is an assistant or squad — defer - // resolution to the state lookup. Try assistant first; resolveTarget() - // accepts either argument key, so we set the candidate in `assistant` - // and let `resolveTarget` fall through to `squad` if not found. - // For clarity, we accept --assistant / --squad as explicit alternatives. + // resolution to the state lookup (see simCommandRun). parsed.assistant = args[++i]; } else if (arg === "--assistant") parsed.assistant = args[++i]; else if (arg === "--squad") parsed.squad = args[++i]; else if (arg === "--transport") { const v = args[++i]; - if (v === "voice" || v === "chat") parsed.transport = v; - else { - console.error(`❌ --transport must be "voice" or "chat" (got "${v}")`); - process.exit(1); + if (v !== "voice" && v !== "chat") { + throw new UsageError( + `--transport must be "voice" or "chat" (got "${v}")`, + ); } + parsed.transport = v; } else if (arg === "--iterations") { - parsed.iterations = parseInt(args[++i] ?? "", 10); + parsed.iterations = Number.parseInt(args[++i] ?? "", 10); if (Number.isNaN(parsed.iterations)) { - console.error("❌ --iterations requires a number"); - process.exit(1); + throw new UsageError("--iterations requires a number"); + } + } else if (arg === "--timeout") { + parsed.timeoutMinutes = Number(args[++i]); + if ( + !Number.isFinite(parsed.timeoutMinutes) || + parsed.timeoutMinutes <= 0 + ) { + throw new UsageError("--timeout requires a positive number of minutes"); } } else if (arg === "--no-watch") parsed.watch = false; else if (arg === "--watch") parsed.watch = true; - else if (arg === "--help" || arg === "-h") { - printUsage(); - process.exit(0); - } + else if (arg === "--help" || arg === "-h") parsed.help = true; + else throw new UsageError(`Unknown argument: ${arg}`); } - return parsed; } -async function main(): Promise { - const args = parseArgs(); - const cfg = loadEnvFile(args.env); - const state = loadStateFile(args.env); +export async function simCommandRun( + args = process.argv.slice(2), +): Promise { + let parsed: ParsedArgs; + try { + parsed = argsParse(args); + } catch (error) { + if (!(error instanceof UsageError)) throw error; + console.error(`❌ ${error.message}\n\n${USAGE}`); + return 2; + } + if (parsed.help) { + console.log(USAGE); + return parsed.env ? 0 : 2; + } + + const cfg = loadEnvFile(parsed.env); + const state = loadStateFile(parsed.env); // Disambiguate --target: if the bare value matches a squad name in state // and not an assistant, treat it as a squad. Explicit --assistant / --squad // override the heuristic. - let assistant = args.assistant; - let squad = args.squad; + let assistant = parsed.assistant; + let squad = parsed.squad; if (assistant && !squad) { const isSquad = typeof state.squads[assistant] !== "undefined" && @@ -120,39 +138,71 @@ async function main(): Promise { console.log( "═══════════════════════════════════════════════════════════════", ); - console.log(`🧪 Vapi GitOps Sim Runner — Environment: ${args.env}`); + console.log(`🧪 Vapi GitOps Sim Runner — Environment: ${parsed.env}`); console.log(` API: ${cfg.baseUrl}`); console.log( "═══════════════════════════════════════════════════════════════\n", ); const selection = resolveSelection(state, { - suite: args.suite, - simulations: args.simulations, + suite: parsed.suite, + simulations: parsed.simulations, }); const target = resolveTarget(state, { assistant, squad }); - const summary = await runSimulation(cfg, selection, target, { - watch: args.watch, - iterations: args.iterations, - transport: args.transport, - }); + // Ctrl-C / SIGTERM cancel the run instead of leaving it running unwatched. + const controller = new AbortController(); + const stop = () => controller.abort(); + process.on("SIGINT", stop); + process.on("SIGTERM", stop); + let summary; + try { + summary = await runSimulation(cfg, selection, target, { + watch: parsed.watch, + iterations: parsed.iterations, + transport: parsed.transport, + timeoutMs: + parsed.timeoutMinutes === undefined + ? undefined + : parsed.timeoutMinutes * 60_000, + signal: controller.signal, + }); + } finally { + process.off("SIGINT", stop); + process.off("SIGTERM", stop); + } console.log(`\n${formatSummary(summary)}\n`); - - if (summary.fail > 0) { - console.error( - `❌ Simulation run failed (${summary.fail} fail / ${summary.pass} pass)`, - ); - process.exit(1); + if (!summary.verdict) { + console.log("▶️ Run started (not watched)."); + return 0; + } + if (summary.verdict.status === "passed") { + console.log("✅ Simulation run passed."); + return 0; } - console.log("✅ Simulation run passed."); + if (summary.verdict.status === "failed") { + console.error(`❌ Simulation run failed: ${summary.verdict.reason}`); + return 1; + } + console.error(`⚠️ Simulation run incomplete: ${summary.verdict.reason}`); + return 3; } -main().catch((error) => { - console.error( - "\n❌ Sim failed:", - error instanceof Error ? error.message : error, +const isMainModule = + process.argv[1] !== undefined && + resolve(process.argv[1]) === fileURLToPath(import.meta.url); +if (isMainModule) { + simCommandRun().then( + (code) => { + process.exitCode = code; + }, + (error: unknown) => { + console.error( + "\n❌ Sim failed:", + error instanceof Error ? error.message : error, + ); + process.exitCode = 2; + }, ); - process.exit(1); -}); +} diff --git a/src/sim-result.ts b/src/sim-result.ts new file mode 100644 index 0000000..9869c41 --- /dev/null +++ b/src/sim-result.ts @@ -0,0 +1,202 @@ +// Strict pass/fail verdict for a simulation run. +// +// Pure (no I/O, no config.ts) so `npm run sim` and the PR check share it and +// tests can pin every case. The API shapes it reads: +// - GET /eval/simulation/run/:id → a run with `status` (queued | running | +// ended) and `itemCounts` (total, passed, failed, running, queued, +// canceled). There is no `results` field on the run. +// - GET /eval/simulation/run/:id/item → items with `status` (queued | +// running | evaluating | passed | failed | canceled) and, once +// evaluated, `results.evaluations[]`. +// +// A run only passes when every expected item exists, passed, and had at +// least one required evaluation that was actually scored. "All evaluations +// skipped" is not a pass (see docs/learnings/simulations.md). + +export interface SimRunItemCounts { + total: number; + passed: number; + failed: number; + running: number; + queued: number; + canceled: number; +} + +export interface SimRun { + id: string; + status?: string; + itemCounts?: SimRunItemCounts; +} + +export interface SimRunEvaluation { + name?: string; + comparator?: string; + expectedValue?: unknown; + extractedValue?: unknown; + passed?: boolean; + required?: boolean; + isSkipped?: boolean; + skipReason?: string; + error?: string; +} + +export interface SimRunItem { + id: string; + status?: string; + results?: { passed?: boolean; evaluations?: SimRunEvaluation[] }; + metadata?: { simulation?: { name?: string } }; +} + +export type SimRunVerdictStatus = "passed" | "failed" | "incomplete"; + +export interface SimRunFailure { + item: string; + evaluation: string; + comparator?: string; + expected?: unknown; + extracted?: unknown; + reason?: string; +} + +export interface SimRunVerdict { + status: SimRunVerdictStatus; + reason: string; + failures: SimRunFailure[]; +} + +const TERMINAL_ITEM_STATUSES = new Set(["passed", "failed", "canceled"]); + +export function simRunItemTerminal(item: SimRunItem): boolean { + if (!TERMINAL_ITEM_STATUSES.has(item.status ?? "")) return false; + // A passed/failed item without results is still being written. + return item.status === "canceled" || item.results !== undefined; +} + +function itemLabel(item: SimRunItem): string { + return item.metadata?.simulation?.name ?? item.id; +} + +function requiredScored(item: SimRunItem): boolean { + return (item.results?.evaluations ?? []).some( + (evaluation) => + evaluation.required !== false && evaluation.isSkipped !== true, + ); +} + +function failuresList(items: SimRunItem[]): SimRunFailure[] { + const failures: SimRunFailure[] = []; + for (const item of items) { + if (item.status !== "failed") continue; + const failed = (item.results?.evaluations ?? []).filter( + (evaluation) => + evaluation.required !== false && + (evaluation.passed === false || evaluation.error !== undefined), + ); + if (failed.length === 0) { + failures.push({ + item: itemLabel(item), + evaluation: "(no evaluation detail)", + }); + } + for (const evaluation of failed) { + failures.push({ + item: itemLabel(item), + evaluation: evaluation.name ?? "(unnamed)", + comparator: evaluation.comparator, + expected: evaluation.expectedValue, + extracted: evaluation.extractedValue, + reason: evaluation.error ?? evaluation.skipReason, + }); + } + } + return failures; +} + +// `expected` is the number of items the run should have (simulations × +// iterations). Pass `undefined` when it can't be known up front (a suite run +// by ID); the verdict then requires at least one item. +export function simRunVerdict(input: { + run: SimRun; + items: SimRunItem[]; + expected?: number; +}): SimRunVerdict { + const { run, items, expected } = input; + const failures = failuresList(items); + const counts = run.itemCounts; + + if (counts && (counts.failed > 0 || failures.length > 0)) { + return { + status: "failed", + reason: `${counts.failed} of ${counts.total} simulations failed`, + failures, + }; + } + if (run.status !== "ended") { + return { + status: "incomplete", + reason: `run did not end (status: ${run.status ?? "unknown"})`, + failures, + }; + } + if (!counts) { + return { status: "incomplete", reason: "run has no item counts", failures }; + } + if (counts.total === 0) { + return { status: "incomplete", reason: "run has no items", failures }; + } + if (expected !== undefined && counts.total !== expected) { + return { + status: "incomplete", + reason: `expected ${expected} items, run has ${counts.total}`, + failures, + }; + } + if (counts.canceled > 0) { + return { + status: "incomplete", + reason: `${counts.canceled} of ${counts.total} simulations were canceled`, + failures, + }; + } + if ( + counts.queued > 0 || + counts.running > 0 || + counts.passed !== counts.total + ) { + return { + status: "incomplete", + reason: `only ${counts.passed} of ${counts.total} simulations finished`, + failures, + }; + } + if (items.length !== counts.total) { + return { + status: "incomplete", + reason: `fetched ${items.length} of ${counts.total} items`, + failures, + }; + } + const notPassed = items.filter((item) => item.status !== "passed"); + if (notPassed.length > 0) { + return { + status: "incomplete", + reason: `${notPassed.length} items are not marked passed`, + failures, + }; + } + const unscored = items.filter((item) => !requiredScored(item)); + if (unscored.length > 0) { + return { + status: "incomplete", + reason: `${unscored.length} simulations had every required evaluation skipped (${unscored + .map(itemLabel) + .join(", ")})`, + failures, + }; + } + return { + status: "passed", + reason: `${counts.passed} of ${counts.total} simulations passed`, + failures, + }; +} diff --git a/src/sim.ts b/src/sim.ts index c63a4e6..f082f9e 100644 --- a/src/sim.ts +++ b/src/sim.ts @@ -8,8 +8,21 @@ import { missingApiKeyMessage, resolveApiKey } from "./api-key.ts"; import { existsSync, readFileSync } from "fs"; import { dirname, join } from "path"; import { fileURLToPath } from "url"; +import { + simRunItemTerminal, + simRunVerdict, + type SimRun, + type SimRunItem, + type SimRunItemCounts, + type SimRunVerdict, +} from "./sim-result.ts"; import type { StateFile } from "./types.ts"; import { userAgentGet } from "./user-agent.ts"; +import { + VapiApiError, + vapiFetchJson, + type VapiConnection, +} from "./vapi-client.ts"; const __dirname = dirname(fileURLToPath(import.meta.url)); const BASE_DIR = join(__dirname, ".."); @@ -40,14 +53,26 @@ export interface SimRunOptions { watch?: boolean; iterations?: number; transport?: "voice" | "chat"; + // Give up (and cancel the run) after this long. Default 20 minutes. + timeoutMs?: number; + // Aborting cancels the run and reports it as incomplete (Ctrl-C). + signal?: AbortSignal; + pollIntervalMs?: number; + // How long to keep re-reading items after the run ends, while their + // results are still being written. Default 2 minutes. + hydrationMs?: number; } export interface SimRunSummary { runId: string; + // Dashboard link from the create response (GET doesn't return it). + url?: string; status: string; - pass: number; - fail: number; - skipped: number; + // Undefined when the run wasn't watched (`--no-watch`). + verdict?: SimRunVerdict; + counts?: SimRunItemCounts; + // True when this command canceled the run (timeout or interrupt). + canceled: boolean; durationMs: number; } @@ -195,44 +220,95 @@ export function resolveSelection( throw new Error("Must specify --suite or --simulations "); } -interface SimRunResponse { - id?: string; - evalRunId?: string; - status?: string; - results?: Array<{ status?: string; isSkipped?: boolean }>; - endedReason?: string; - endedMessage?: string; - cost?: number; - [key: string]: unknown; +// Fields of `POST /eval/simulation/run`'s response this runner reads. The +// create response is the only place `url` and `simulationRunItemIds` appear. +interface SimRunCreated extends SimRun { + url?: string; + simulationRunItemIds?: string[]; } const POLL_INTERVAL_MS = 3000; -const POLL_TIMEOUT_MS = 600_000; +const DEFAULT_TIMEOUT_MS = 20 * 60_000; +const DEFAULT_HYDRATION_MS = 2 * 60_000; +// The API's maximum page size; one page covers nearly every run. +const ITEM_PAGE_SIZE = 1000; -async function fetchJson( - cfg: SimEnv, - method: "GET" | "POST", - endpoint: string, - body?: unknown, -): Promise { - const response = await fetch(`${cfg.baseUrl}${endpoint}`, { - method, - headers: { - Authorization: `Bearer ${cfg.token}`, - "Content-Type": "application/json", - "User-Agent": userAgentGet("sim"), - }, - ...(body ? { body: JSON.stringify(body) } : {}), +function connectionFor(cfg: SimEnv): VapiConnection { + return { + token: cfg.token, + baseUrl: cfg.baseUrl, + userAgent: userAgentGet("sim"), + }; +} + +// Resolves after `ms`, or early (to "aborted") when the signal fires. +function sleepUnlessAborted( + ms: number, + signal?: AbortSignal, +): Promise<"slept" | "aborted"> { + if (signal?.aborted) return Promise.resolve("aborted"); + return new Promise((resolve) => { + const timer = setTimeout(() => { + signal?.removeEventListener("abort", onAbort); + resolve("slept"); + }, ms); + const onAbort = () => { + clearTimeout(timer); + resolve("aborted"); + }; + signal?.addEventListener("abort", onAbort, { once: true }); }); - if (!response.ok) { - const text = await response.text(); - throw new Error(`API ${method} ${endpoint} → ${response.status}: ${text}`); +} + +// Reads every item of a run. Accepts both the paginated shape +// (`{ results, metadata }`, sent when `limit`/`page` are given) and a bare +// array. Items are deduped by id because the API orders pages only by +// creation time, and a run's items share it, so OFFSET pages can overlap. +export async function simRunItemsFetch( + connection: VapiConnection, + runId: string, +): Promise { + const byId = new Map(); + for (let page = 1; page <= 100; page++) { + const response = await vapiFetchJson< + | SimRunItem[] + | { results?: SimRunItem[]; metadata?: { totalItems?: number } } + >( + connection, + "GET", + `/eval/simulation/run/${runId}/item?page=${page}&limit=${ITEM_PAGE_SIZE}`, + ); + if (Array.isArray(response)) { + for (const item of response) byId.set(item.id, item); + break; + } + const results = response?.results ?? []; + for (const item of results) byId.set(item.id, item); + const total = response?.metadata?.totalItems; + if (results.length < ITEM_PAGE_SIZE) break; + if (total !== undefined && byId.size >= total) break; } - return response.json(); + return [...byId.values()]; } -function sleep(ms: number): Promise { - return new Promise((r) => setTimeout(r, ms)); +// Cancels a run. Returns false (instead of throwing) when the run had +// already ended or a concurrent cancel won the race (400 / 409). +export async function simRunCancel( + connection: VapiConnection, + runId: string, +): Promise { + try { + await vapiFetchJson(connection, "PATCH", `/eval/simulation/run/${runId}`); + return true; + } catch (error) { + if ( + error instanceof VapiApiError && + (error.statusCode === 400 || error.statusCode === 409) + ) { + return false; + } + throw error; + } } export async function runSimulation( @@ -241,6 +317,8 @@ export async function runSimulation( target: SimTarget, options: SimRunOptions = {}, ): Promise { + const connection = connectionFor(cfg); + const pollIntervalMs = options.pollIntervalMs ?? POLL_INTERVAL_MS; const body: Record = { simulations: selection.entries, target: @@ -258,68 +336,135 @@ export async function runSimulation( `🧪 Starting simulation run — ${selection.label} → ${target.type}/${target.resourceName}`, ); const start = Date.now(); - const created = (await fetchJson( - cfg, + // Creating a run queues paid work before the response returns, so a 5xx + // here may still have started it: retry rate limits only. + const created = await vapiFetchJson( + connection, "POST", "/eval/simulation/run", body, - )) as SimRunResponse; - const runId = created.evalRunId ?? created.id; + { retry: "rate-limit-only" }, + ); + const runId = created?.id; if (!runId) { - throw new Error( - `POST /eval/simulation/run returned no runId (keys: ${Object.keys(created).join(", ")})`, - ); + throw new Error("POST /eval/simulation/run returned no run id"); } console.log(` Run ID: ${runId}`); + if (created.url) console.log(` Run: ${created.url}`); - let last: SimRunResponse = created; - if (options.watch ?? true) { - while (Date.now() - start < POLL_TIMEOUT_MS) { - await sleep(POLL_INTERVAL_MS); - last = (await fetchJson( - cfg, - "GET", - `/eval/simulation/run/${runId}`, - )) as SimRunResponse; - const status = last.status ?? "running"; - process.stdout.write(`\r Status: ${status} `); - if (status === "ended" || status === "failed" || status === "completed") { - process.stdout.write("\n"); - break; - } + const summary: SimRunSummary = { + runId, + url: created.url, + status: created.status ?? "queued", + counts: created.itemCounts, + canceled: false, + durationMs: 0, + }; + if (!(options.watch ?? true)) { + summary.durationMs = Date.now() - start; + return summary; + } + + const deadline = start + (options.timeoutMs ?? DEFAULT_TIMEOUT_MS); + let run: SimRun = created; + let stopReason: string | undefined; + while (run.status !== "ended") { + const remaining = deadline - Date.now(); + if (remaining <= 0) { + stopReason = `timed out after ${Math.round((Date.now() - start) / 1000)}s`; + break; } - if (Date.now() - start >= POLL_TIMEOUT_MS) { - throw new Error( - `Simulation run ${runId} timed out after ${POLL_TIMEOUT_MS / 1000}s`, - ); + const slept = await sleepUnlessAborted( + Math.min(pollIntervalMs, remaining), + options.signal, + ); + if (slept === "aborted") { + stopReason = "interrupted"; + break; + } + run = await vapiFetchJson( + connection, + "GET", + `/eval/simulation/run/${runId}`, + ); + if (process.stdout.isTTY) { + process.stdout.write(`\r Status: ${run.status ?? "unknown"} `); } } + if (process.stdout.isTTY) process.stdout.write("\n"); - const results = Array.isArray(last.results) ? last.results : []; - const pass = results.filter( - (r) => r.status === "pass" && !r.isSkipped, - ).length; - const fail = results.filter( - (r) => r.status !== "pass" && !r.isSkipped, - ).length; - const skipped = results.filter((r) => r.isSkipped === true).length; + if (stopReason) { + summary.canceled = await simRunCancel(connection, runId); + summary.status = run.status ?? "unknown"; + summary.counts = run.itemCounts; + summary.verdict = { + status: "incomplete", + reason: `${stopReason}${summary.canceled ? "; run canceled" : ""}`, + failures: [], + }; + summary.durationMs = Date.now() - start; + return summary; + } - return { - runId, - status: last.status ?? "unknown", - pass, - fail, - skipped, - durationMs: Date.now() - start, - }; + // Items can lag the run: keep re-reading until every item is terminal and + // carries its results, or the hydration window closes. + const hydrationDeadline = + Date.now() + (options.hydrationMs ?? DEFAULT_HYDRATION_MS); + let items = await simRunItemsFetch(connection, runId); + while ( + Date.now() < hydrationDeadline && + (items.length < (run.itemCounts?.total ?? 0) || + !items.every(simRunItemTerminal)) + ) { + if ( + (await sleepUnlessAborted(pollIntervalMs, options.signal)) === "aborted" + ) { + break; + } + items = await simRunItemsFetch(connection, runId); + } + + summary.status = run.status ?? "unknown"; + summary.counts = run.itemCounts; + summary.verdict = simRunVerdict({ + run, + items, + expected: created.simulationRunItemIds?.length, + }); + summary.durationMs = Date.now() - start; + return summary; +} + +function valueFormat(value: unknown): string { + return typeof value === "string" ? value : JSON.stringify(value); } export function formatSummary(summary: SimRunSummary): string { - const total = summary.pass + summary.fail + summary.skipped; - return [ - `📊 Simulation summary (run ${summary.runId})`, - ` Status: ${summary.status}`, - ` Results: ${summary.pass}/${total} pass, ${summary.fail} fail${summary.skipped > 0 ? `, ${summary.skipped} skipped` : ""}`, - ` Duration: ${(summary.durationMs / 1000).toFixed(1)}s`, - ].join("\n"); + const lines = [`📊 Simulation summary (run ${summary.runId})`]; + if (summary.url) lines.push(` Run: ${summary.url}`); + lines.push(` Status: ${summary.status}`); + if (summary.counts) { + const c = summary.counts; + lines.push( + ` Items: ${c.passed} passed, ${c.failed} failed, ${c.canceled} canceled, ${c.running + c.queued} unfinished (of ${c.total})`, + ); + } + if (summary.verdict) { + lines.push( + ` Verdict: ${summary.verdict.status} — ${summary.verdict.reason}`, + ); + for (const failure of summary.verdict.failures) { + const detail = + failure.comparator !== undefined + ? ` (expected ${failure.comparator} ${valueFormat(failure.expected)}, got ${valueFormat(failure.extracted)})` + : ""; + lines.push( + ` ✗ ${failure.item}: ${failure.evaluation}${detail}${failure.reason ? ` — ${failure.reason}` : ""}`, + ); + } + } else { + lines.push(" Verdict: not watched (--no-watch)"); + } + lines.push(` Duration: ${(summary.durationMs / 1000).toFixed(1)}s`); + return lines.join("\n"); } diff --git a/src/vapi-client.ts b/src/vapi-client.ts new file mode 100644 index 0000000..7c32959 --- /dev/null +++ b/src/vapi-client.ts @@ -0,0 +1,108 @@ +// Config-free HTTP client for the Vapi API. +// +// api.ts is the engine's client, but it imports config.ts, which parses +// argv and exits at import time and binds a single org. Code that must run +// for several orgs, without a token, or from tests (sim.ts, the check +// runner) uses this client instead. + +export interface VapiConnection { + token: string; + baseUrl: string; + userAgent: string; +} + +// "transient": retry 429 and 5xx. "rate-limit-only": retry 429 only — for +// requests that aren't safe to repeat after a 5xx, because the server may +// have acted before failing (creating a simulation run queues paid work). +export type VapiRetryPolicy = "transient" | "rate-limit-only" | "none"; + +export interface VapiFetchOptions { + retry?: VapiRetryPolicy; + maxRetries?: number; + initialDelayMs?: number; + sleep?: (ms: number) => Promise; +} + +export class VapiApiError extends Error { + constructor( + public readonly method: string, + public readonly endpoint: string, + public readonly statusCode: number, + public readonly apiMessage: string, + public readonly rawBody: string, + ) { + super(`API ${method} ${endpoint} failed (${statusCode}): ${apiMessage}`); + this.name = "VapiApiError"; + } +} + +export function parseApiMessage(body: string): string { + try { + const parsed = JSON.parse(body); + if (typeof parsed.message === "string") return parsed.message; + if (Array.isArray(parsed.message)) return parsed.message.join("; "); + } catch { + /* not JSON, use raw body */ + } + return body; +} + +// 429 = rate limit. 5xx = transient server error (gateway timeout, upstream +// hiccup, deploy in progress). +export function shouldRetry(status: number): boolean { + return status === 429 || (status >= 500 && status < 600); +} + +export const MAX_RETRIES = 5; +export const INITIAL_DELAY_MS = 2000; + +function sleepDefault(ms: number): Promise { + return new Promise((resolve) => setTimeout(resolve, ms)); +} + +function retryable(policy: VapiRetryPolicy, status: number): boolean { + if (policy === "none") return false; + if (policy === "rate-limit-only") return status === 429; + return shouldRetry(status); +} + +export async function vapiFetchJson( + connection: VapiConnection, + method: "GET" | "POST" | "PATCH", + endpoint: string, + body?: unknown, + options: VapiFetchOptions = {}, +): Promise { + const policy = options.retry ?? "transient"; + const maxRetries = options.maxRetries ?? MAX_RETRIES; + const initialDelayMs = options.initialDelayMs ?? INITIAL_DELAY_MS; + const sleep = options.sleep ?? sleepDefault; + + for (let attempt = 0; ; attempt++) { + const response = await fetch(`${connection.baseUrl}${endpoint}`, { + method, + headers: { + Authorization: `Bearer ${connection.token}`, + "Content-Type": "application/json", + "User-Agent": connection.userAgent, + }, + ...(body === undefined ? {} : { body: JSON.stringify(body) }), + }); + if (response.ok) { + const text = await response.text(); + return (text ? JSON.parse(text) : null) as T; + } + if (attempt < maxRetries && retryable(policy, response.status)) { + await sleep(initialDelayMs * 2 ** attempt); + continue; + } + const rawBody = await response.text(); + throw new VapiApiError( + method, + endpoint, + response.status, + parseApiMessage(rawBody), + rawBody, + ); + } +} diff --git a/tests/sim-result.test.ts b/tests/sim-result.test.ts new file mode 100644 index 0000000..4bea397 --- /dev/null +++ b/tests/sim-result.test.ts @@ -0,0 +1,178 @@ +import assert from "node:assert/strict"; +import test from "node:test"; +import { + simRunItemTerminal, + simRunVerdict, + type SimRunItem, + type SimRunItemCounts, +} from "../src/sim-result.ts"; + +// The verdict `npm run sim` and the PR check share. It must never report a +// pass unless every expected simulation ran, passed, and was actually +// scored — the old runner read a `results` field the API doesn't return and +// reported 0/0 as a pass. + +function counts(overrides: Partial = {}): SimRunItemCounts { + return { + total: 2, + passed: 2, + failed: 0, + running: 0, + queued: 0, + canceled: 0, + ...overrides, + }; +} + +function item( + id: string, + status: string, + evaluations: NonNullable["evaluations"] = [ + { name: "goal-met", required: true, passed: status === "passed" }, + ], +): SimRunItem { + return { + id, + status, + results: { passed: status === "passed", evaluations }, + metadata: { simulation: { name: `sim ${id}` } }, + }; +} + +const ended = (itemCounts?: SimRunItemCounts) => ({ + id: "run-1", + status: "ended", + itemCounts, +}); + +test("simRunVerdict: passes when every expected item passed and was scored", () => { + const verdict = simRunVerdict({ + run: ended(counts()), + items: [item("a", "passed"), item("b", "passed")], + expected: 2, + }); + assert.equal(verdict.status, "passed"); +}); + +test("simRunVerdict: the old false-green shape (ended, no results) is not a pass", () => { + // What GET /eval/simulation/run/:id actually returns: no `results` field. + // With no items fetched, the run can't be called passed. + const verdict = simRunVerdict({ run: ended(counts()), items: [] }); + assert.equal(verdict.status, "incomplete"); + assert.match(verdict.reason, /fetched 0 of 2 items/); +}); + +test("simRunVerdict: 0 items is incomplete, never a pass", () => { + const verdict = simRunVerdict({ + run: ended(counts({ total: 0, passed: 0 })), + items: [], + }); + assert.equal(verdict.status, "incomplete"); +}); + +test("simRunVerdict: a failed item fails the run and lists the failing judge", () => { + const verdict = simRunVerdict({ + run: ended(counts({ passed: 1, failed: 1 })), + items: [ + item("a", "passed"), + item("b", "failed", [ + { + name: "booking-confirmed", + required: true, + passed: false, + comparator: "=", + expectedValue: true, + extractedValue: false, + }, + ]), + ], + }); + assert.equal(verdict.status, "failed"); + assert.deepEqual(verdict.failures, [ + { + item: "sim b", + evaluation: "booking-confirmed", + comparator: "=", + expected: true, + extracted: false, + reason: undefined, + }, + ]); +}); + +test("simRunVerdict: canceled items make the run incomplete", () => { + const verdict = simRunVerdict({ + run: ended(counts({ passed: 1, canceled: 1 })), + items: [item("a", "passed"), item("b", "canceled")], + }); + assert.equal(verdict.status, "incomplete"); + assert.match(verdict.reason, /canceled/); +}); + +test("simRunVerdict: all required evaluations skipped is not a pass", () => { + const skipped = [{ name: "audio-check", required: true, isSkipped: true }]; + const verdict = simRunVerdict({ + run: ended(counts()), + items: [item("a", "passed"), item("b", "passed", skipped)], + }); + assert.equal(verdict.status, "incomplete"); + assert.match(verdict.reason, /every required evaluation skipped/); +}); + +test("simRunVerdict: a skipped optional judge doesn't hide a scored required one", () => { + const verdict = simRunVerdict({ + run: ended(counts({ total: 1, passed: 1 })), + items: [ + item("a", "passed", [ + { name: "tone", required: false, isSkipped: true }, + { name: "goal-met", required: true, passed: true }, + ]), + ], + }); + assert.equal(verdict.status, "passed"); +}); + +test("simRunVerdict: missing itemCounts is incomplete", () => { + const verdict = simRunVerdict({ + run: ended(undefined), + items: [item("a", "passed")], + }); + assert.equal(verdict.status, "incomplete"); +}); + +test("simRunVerdict: a count different from the expected one is incomplete", () => { + const verdict = simRunVerdict({ + run: ended(counts()), + items: [item("a", "passed"), item("b", "passed")], + expected: 3, + }); + assert.equal(verdict.status, "incomplete"); + assert.match(verdict.reason, /expected 3 items/); +}); + +test("simRunVerdict: a run that hasn't ended is incomplete", () => { + const verdict = simRunVerdict({ + run: { + id: "run-1", + status: "running", + itemCounts: counts({ passed: 1, running: 1 }), + }, + items: [item("a", "passed")], + }); + assert.equal(verdict.status, "incomplete"); +}); + +test("simRunVerdict: a short item list is incomplete even when counts say passed", () => { + const verdict = simRunVerdict({ + run: ended(counts()), + items: [item("a", "passed")], + }); + assert.equal(verdict.status, "incomplete"); +}); + +test("simRunItemTerminal: passed or failed items need results; canceled doesn't", () => { + assert.equal(simRunItemTerminal({ id: "a", status: "passed" }), false); + assert.equal(simRunItemTerminal(item("a", "passed")), true); + assert.equal(simRunItemTerminal({ id: "a", status: "canceled" }), true); + assert.equal(simRunItemTerminal({ id: "a", status: "evaluating" }), false); +}); diff --git a/tests/sim-run.test.ts b/tests/sim-run.test.ts new file mode 100644 index 0000000..aedc215 --- /dev/null +++ b/tests/sim-run.test.ts @@ -0,0 +1,368 @@ +import assert from "node:assert/strict"; +import { createServer, type IncomingMessage } from "node:http"; +import type { AddressInfo } from "node:net"; +import test from "node:test"; +import { runSimulation, simRunItemsFetch } from "../src/sim.ts"; + +// runSimulation against a local stub of the simulations API. Each stub +// route returns a canned response; requests are recorded so tests can +// assert on what was sent (cancel, retries). + +type Handler = ( + req: IncomingMessage, + body: string, +) => { + status?: number; + json?: unknown; +}; + +async function withServer( + routes: Array<{ method: string; path: RegExp; handle: Handler }>, + fn: (baseUrl: string, seen: string[]) => Promise, +): Promise { + const seen: string[] = []; + const server = createServer((req, res) => { + let body = ""; + req.on("data", (chunk) => (body += chunk)); + req.on("end", () => { + seen.push(`${req.method} ${req.url}`); + const route = routes.find( + (r) => r.method === req.method && r.path.test(req.url ?? ""), + ); + const out = route ? route.handle(req, body) : { status: 404, json: {} }; + res.writeHead(out.status ?? 200, { "Content-Type": "application/json" }); + res.end(JSON.stringify(out.json ?? {})); + }); + }); + await new Promise((resolve) => server.listen(0, resolve)); + const { port } = server.address() as AddressInfo; + const log = console.log; + console.log = () => {}; + try { + await fn(`http://127.0.0.1:${port}`, seen); + } finally { + console.log = log; + await new Promise((resolve) => server.close(() => resolve())); + } +} + +const cfg = (baseUrl: string) => ({ env: "test-org", token: "t", baseUrl }); +const selection = { + entries: [{ type: "simulationSuite" as const, simulationSuiteId: "suite-1" }], + label: "suite test", +}; +const target = { type: "squad" as const, id: "squad-1", resourceName: "s" }; +const fast = { pollIntervalMs: 1, hydrationMs: 50 }; + +const endedRun = (passed: number, failed: number) => ({ + id: "run-1", + status: "ended", + itemCounts: { + total: passed + failed, + passed, + failed, + running: 0, + queued: 0, + canceled: 0, + }, +}); + +const scoredItem = (id: string, status: "passed" | "failed") => ({ + id, + status, + results: { + passed: status === "passed", + evaluations: [ + { + name: "goal-met", + required: true, + passed: status === "passed", + comparator: "=", + expectedValue: true, + extractedValue: status === "passed", + }, + ], + }, + metadata: { simulation: { name: `sim ${id}` } }, +}); + +test("runSimulation: passes when the ended run's items all passed", async () => { + await withServer( + [ + { + method: "POST", + path: /^\/eval\/simulation\/run$/, + handle: () => ({ + status: 201, + json: { + id: "run-1", + status: "queued", + url: "https://dashboard.vapi.ai/simulations/run/run-1", + simulationRunItemIds: ["a", "b"], + }, + }), + }, + { + method: "GET", + path: /^\/eval\/simulation\/run\/run-1$/, + handle: () => ({ json: endedRun(2, 0) }), + }, + { + method: "GET", + path: /^\/eval\/simulation\/run\/run-1\/item/, + handle: () => ({ + json: { + results: [scoredItem("a", "passed"), scoredItem("b", "passed")], + metadata: { totalItems: 2, itemsPerPage: 1000, currentPage: 1 }, + }, + }), + }, + ], + async (baseUrl) => { + const summary = await runSimulation( + cfg(baseUrl), + selection, + target, + fast, + ); + assert.equal(summary.verdict?.status, "passed"); + assert.equal( + summary.url, + "https://dashboard.vapi.ai/simulations/run/run-1", + ); + }, + ); +}); + +test("runSimulation: a failed item fails the run (the old runner reported a pass)", async () => { + await withServer( + [ + { + method: "POST", + path: /^\/eval\/simulation\/run$/, + handle: () => ({ + status: 201, + json: { id: "run-1", status: "queued" }, + }), + }, + { + method: "GET", + path: /^\/eval\/simulation\/run\/run-1$/, + handle: () => ({ json: endedRun(1, 1) }), + }, + { + method: "GET", + path: /^\/eval\/simulation\/run\/run-1\/item/, + // The bare-array shape the API returns without page/limit. + handle: () => ({ + json: [scoredItem("a", "passed"), scoredItem("b", "failed")], + }), + }, + ], + async (baseUrl) => { + const summary = await runSimulation( + cfg(baseUrl), + selection, + target, + fast, + ); + assert.equal(summary.verdict?.status, "failed"); + assert.equal(summary.verdict?.failures[0]?.item, "sim b"); + }, + ); +}); + +test("runSimulation: waits for item results that arrive after the run ends", async () => { + let reads = 0; + await withServer( + [ + { + method: "POST", + path: /^\/eval\/simulation\/run$/, + handle: () => ({ + status: 201, + json: { id: "run-1", status: "queued" }, + }), + }, + { + method: "GET", + path: /^\/eval\/simulation\/run\/run-1$/, + handle: () => ({ json: endedRun(1, 0) }), + }, + { + method: "GET", + path: /^\/eval\/simulation\/run\/run-1\/item/, + handle: () => { + reads++; + // First read: the item is passed but its results aren't written yet. + return { + json: + reads === 1 + ? [{ id: "a", status: "passed" }] + : [scoredItem("a", "passed")], + }; + }, + }, + ], + async (baseUrl) => { + const summary = await runSimulation(cfg(baseUrl), selection, target, { + pollIntervalMs: 1, + hydrationMs: 5_000, + }); + assert.equal(summary.verdict?.status, "passed"); + assert.ok(reads >= 2); + }, + ); +}); + +test("runSimulation: a timeout cancels the run and reports incomplete", async () => { + await withServer( + [ + { + method: "POST", + path: /^\/eval\/simulation\/run$/, + handle: () => ({ + status: 201, + json: { id: "run-1", status: "queued" }, + }), + }, + { + method: "GET", + path: /^\/eval\/simulation\/run\/run-1$/, + handle: () => ({ json: { id: "run-1", status: "running" } }), + }, + { + method: "PATCH", + path: /^\/eval\/simulation\/run\/run-1$/, + handle: () => ({ json: { id: "run-1", status: "ended" } }), + }, + ], + async (baseUrl, seen) => { + const summary = await runSimulation(cfg(baseUrl), selection, target, { + pollIntervalMs: 5, + timeoutMs: 30, + }); + assert.equal(summary.verdict?.status, "incomplete"); + assert.match(summary.verdict?.reason ?? "", /timed out/); + assert.equal(summary.canceled, true); + assert.ok(seen.includes("PATCH /eval/simulation/run/run-1")); + }, + ); +}); + +test("runSimulation: an interrupt cancels; a 400 'already ended' is swallowed", async () => { + const controller = new AbortController(); + await withServer( + [ + { + method: "POST", + path: /^\/eval\/simulation\/run$/, + handle: () => ({ + status: 201, + json: { id: "run-1", status: "queued" }, + }), + }, + { + method: "GET", + path: /^\/eval\/simulation\/run\/run-1$/, + handle: () => { + controller.abort(); + return { json: { id: "run-1", status: "running" } }; + }, + }, + { + method: "PATCH", + path: /^\/eval\/simulation\/run\/run-1$/, + handle: () => ({ + status: 400, + json: { message: "Run has already ended" }, + }), + }, + ], + async (baseUrl) => { + const summary = await runSimulation(cfg(baseUrl), selection, target, { + pollIntervalMs: 1, + signal: controller.signal, + }); + assert.equal(summary.verdict?.status, "incomplete"); + assert.match(summary.verdict?.reason ?? "", /interrupted/); + assert.equal(summary.canceled, false); + }, + ); +}); + +test("runSimulation: a 5xx on run create is not retried (it may already have queued the run)", async () => { + await withServer( + [ + { + method: "POST", + path: /^\/eval\/simulation\/run$/, + handle: () => ({ status: 502, json: { message: "Bad Gateway" } }), + }, + ], + async (baseUrl, seen) => { + await assert.rejects( + runSimulation(cfg(baseUrl), selection, target, fast), + /failed \(502\)/, + ); + assert.equal(seen.filter((s) => s.startsWith("POST")).length, 1); + }, + ); +}); + +test("runSimulation: --no-watch returns after create with the link and no verdict", async () => { + await withServer( + [ + { + method: "POST", + path: /^\/eval\/simulation\/run$/, + handle: () => ({ + status: 201, + json: { id: "run-1", status: "queued", url: "https://x/run-1" }, + }), + }, + ], + async (baseUrl, seen) => { + const summary = await runSimulation(cfg(baseUrl), selection, target, { + watch: false, + }); + assert.equal(summary.verdict, undefined); + assert.equal(summary.url, "https://x/run-1"); + assert.deepEqual(seen, ["POST /eval/simulation/run"]); + }, + ); +}); + +test("simRunItemsFetch: pages until totalItems and dedupes overlapping pages", async () => { + const page = (ids: string[]) => ids.map((id) => ({ id, status: "passed" })); + const first = page(Array.from({ length: 1000 }, (_, i) => `i${i}`)); + await withServer( + [ + { + method: "GET", + path: /\/item\?page=1&/, + handle: () => ({ + json: { results: first, metadata: { totalItems: 1002 } }, + }), + }, + { + method: "GET", + path: /\/item\?page=2&/, + // Overlaps the first page by one item (OFFSET drift). + handle: () => ({ + json: { + results: page(["i999", "i1000", "i1001"]), + metadata: { totalItems: 1002 }, + }, + }), + }, + ], + async (baseUrl) => { + const items = await simRunItemsFetch( + { token: "t", baseUrl, userAgent: "test" }, + "run-1", + ); + assert.equal(items.length, 1002); + }, + ); +});