From cb88b94f41acb47bbdbeb593747c340e42f0c616 Mon Sep 17 00:00:00 2001 From: Scott Lowe Date: Thu, 1 Oct 2026 16:26:37 -0700 Subject: [PATCH] feat(check): run checks live and report Vapi Evals statuses `npm run check -- |--all` now runs each built target payload in the check's run org: one POST /eval/simulation/run per target, at most three at once, judged by the same strict verdict as `npm run sim` (the create/poll/cancel/hydrate loop moves into simRunExecute, which both share). - Budget and cancellation: each run's deadline is the earlier of its timeoutMinutes and --budget-minutes; no run starts with under five minutes left; the deadline, SIGINT and SIGTERM cancel in-flight runs. - Keys: VAPI_CHECK_TOKENS, then .env., then VAPI_PRIVATE_API_KEY when every selected check runs in one org. --refresh-bindings runs the same read-only bindings pull promotion uses. - Selection: --changed-since runs only checks affected by changes since the merge base (the org's files and state, the run org's, the config, the engine, and the check's own paths). - Statuses (when GITHUB_TOKEN, GITHUB_REPOSITORY and HEAD_SHA are set): `Vapi Evals / / ` goes pending with the run's link, then success/failure/error; --all runs post the aggregate `Vapi Evals` from a finally block, so config errors and exceptions report too. - Report: a markdown summary (also appended to $GITHUB_STEP_SUMMARY) with failing evaluations' expected vs extracted values and "unmocked tool called" notices from transcripts, plus --json. No PR comments. - Runs send User-Agent vapi-gitops-check/. Exit codes: 0 passed, 1 failed, 2 config or build error, 3 incomplete. Refs TEST-141 Co-Authored-By: Claude Opus 5.5 --- AGENTS.md | 5 +- README.md | 2 +- src/check-build.ts | 80 +++++++ src/check-cmd.ts | 471 +++++++++++++++++++++++++++---------- src/check-report.ts | 111 +++++++++ src/check-run.ts | 177 ++++++++++++++ src/check-select.ts | 56 +++++ src/check-status.ts | 106 +++++++++ src/resource-parse.ts | 4 +- src/sim-result.ts | 16 +- src/sim.ts | 91 ++++--- tests/check-cmd.test.ts | 326 ++++++++++++++++++++++++- tests/check-report.test.ts | 125 ++++++++++ tests/check-run.test.ts | 414 ++++++++++++++++++++++++++++++++ tests/check-select.test.ts | 79 +++++++ tests/check-status.test.ts | 130 ++++++++++ 16 files changed, 2035 insertions(+), 158 deletions(-) create mode 100644 src/check-build.ts create mode 100644 src/check-report.ts create mode 100644 src/check-run.ts create mode 100644 src/check-select.ts create mode 100644 src/check-status.ts create mode 100644 tests/check-report.test.ts create mode 100644 tests/check-run.test.ts create mode 100644 tests/check-select.test.ts create mode 100644 tests/check-status.test.ts diff --git a/AGENTS.md b/AGENTS.md index a147c92..8753eb0 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -90,7 +90,7 @@ If setup reports the org is "already set up locally", do not delete anything to | Push with new resources | `npm run push -- --allow-new-files` — bypass orphan-YAML gate. **AI agents**: do NOT auto-pass this flag; confirm with the human first (see push section below) | | Test a call | `npm run call -- -a ` or `-s ` | | Run a simulation suite | `npm run sim -- --suite --target ` | -| Build PR-check payloads offline | `npm run check -- --dry-run` (or `--all`); `--print-payload` writes the JSON | +| Run PR simulation checks | `npm run check -- ` (or `--all`); `--dry-run` builds the payloads offline, `--print-payload` writes them | | Migrate a legacy state file | `npm run migrate` — one-shot, all orgs; required once after upgrading to the hash-store engine | --- @@ -904,7 +904,8 @@ npm run validate -- # Lint resources locally (fai npm run audit -- # Read-only drift detector: orphan YAML, state ghosts, content-identical clusters, sibling base-slugs, dashboard orphans, inline model.tools. Exit 1 on findings. npm run audit -- --type assistants # Scope audit to a single resource type npm run sim -- --suite --target # Run a simulation suite against an assistant/squad (exit 0 pass, 1 fail, 3 incomplete; --timeout ) -npm run check -- --dry-run # Build a vapi-checks.yml check's inline run payload offline (exit 0 built, 2 config/build error; --print-payload [dir]) +npm run check -- # Run a vapi-checks.yml check inline against the local files (exit 0 pass, 1 fail, 2 config/build error, 3 incomplete; --changed-since , --budget-minutes , --json ) +npm run check -- --dry-run # Build the check's inline run payloads offline, no key needed (--print-payload [dir]) npm run rollback -- --to # Re-apply a snapshot taken before a push npm run rollback -- --list # List available snapshots diff --git a/README.md b/README.md index 61da8da..ca1dc00 100644 --- a/README.md +++ b/README.md @@ -104,7 +104,7 @@ Every command works in two modes: | `npm run rollback` | — | `npm run rollback -- --list` or `--to ` | Restore from a snapshot in `.vapi-state..snapshots/` (one is written before every push/apply). | | `npm run call` | ✅ | `npm run call -- -a ` or `-s ` | Start an interactive WebSocket call against an assistant or squad. | | `npm run sim` | — | `npm run sim -- --suite --target [--timeout ]` | Run a simulation suite (or specific simulations) against a deployed assistant/squad. Prints the run link; exits 0 passed, 1 failed, 3 incomplete (timeout, Ctrl-C, missing results). | -| `npm run check` | — | `npm run check -- \|--all --dry-run [--print-payload [dir]]` | Build each `vapi-checks.yml` check target's inline simulation-run payload from the files on disk, with its tools, handoffs, judges and personalities. Offline: no API key, nothing sent. `--print-payload` writes the JSON (default `tmp/check-payloads/`). Exits 0 when every payload builds, 2 on a config or build error. | +| `npm run check` | — | `npm run check -- \|--all [--dry-run] [--changed-since ] [--budget-minutes ] [--json ]` | Run the `vapi-checks.yml` simulation checks against the files on disk: each target is built inline (tools, handoffs, judges, personalities; tools mocked fail-closed, servers dead-ended) and run in one simulation run per target, so nothing is deployed. Posts `Vapi Evals` commit statuses when run by the PR workflow. `--dry-run` builds the payloads offline (no key, nothing sent; `--print-payload` writes them). Exits 0 passed, 1 failed, 2 config or build error, 3 incomplete (timeout, interrupt, budget, billing). | | `npm run migrate` | — | `npm run migrate` | One-time, all orgs at once: slim legacy state files to pure `name → uuid` and seed the per-developer `.vapi-state-hash/` baseline store from the old hashes. Required once after upgrading to the hash-store engine — `pull`/`push`/`apply` refuse legacy-shaped state until it runs. Idempotent. | | `npm run build` | — | — | Type-check the codebase (`tsc --noEmit`). | | `npm test` | — | — | Run regression tests (`node:test`). | diff --git a/src/check-build.ts b/src/check-build.ts new file mode 100644 index 0000000..05536eb --- /dev/null +++ b/src/check-build.ts @@ -0,0 +1,80 @@ +// Build every target payload of one check from the files under rootDir. +// Shared by the dry run and the live run, so both test the same body. + +import { existsSync, readFileSync } from "node:fs"; +import { join } from "node:path"; +import type { CheckDefinition, CheckTarget } from "./check-config.ts"; +import type { CheckPayloadResult } from "./check-payload.ts"; +import { checkPayloadBuild } from "./check-payload.ts"; +import { promotionBindingsResolve, promotionStateParse } from "./promotion.ts"; +import { orgResourcesRead } from "./resource-parse.ts"; +import type { StateFile } from "./types.ts"; + +export interface CheckJob { + check: CheckDefinition; + target: CheckTarget; + // ` / /` + label: string; + result: CheckPayloadResult; +} + +export function checkTargetLabel(target: CheckTarget): string { + return `${target.type}/${target.id}`; +} + +function stateRead(root: string, org: string, warnings: string[]): StateFile { + const path = join(root, `.vapi-state.${org}.json`); + if (!existsSync(path)) { + warnings.push( + `no .vapi-state.${org}.json: references by UUID and credential bindings can't resolve`, + ); + return promotionStateParse("{}"); + } + return promotionStateParse(readFileSync(path, "utf8")); +} + +export async function checkJobsBuild( + root: string, + check: CheckDefinition, +): Promise { + const job = (target: CheckTarget, result: CheckPayloadResult): CheckJob => ({ + check, + target, + label: `${check.name} / ${checkTargetLabel(target)}`, + result, + }); + if (!existsSync(join(root, "resources", check.org))) { + return check.targets.map((target) => + job(target, { + errors: [`resources/${check.org}/ does not exist`], + warnings: [], + bytes: 0, + }), + ); + } + const warnings: string[] = []; + const resources = await orgResourcesRead(root, check.org); + const sourceState = stateRead(root, check.org, warnings); + const runState = + check.runOrg === check.org + ? sourceState + : stateRead(root, check.runOrg, warnings); + const bindingsResolved = await promotionBindingsResolve( + root, + check.org, + check.runOrg, + sourceState, + ); + return check.targets.map((target) => { + const result = checkPayloadBuild({ + check, + target, + resources, + sourceState, + runState, + bindingsResolved, + }); + result.warnings.unshift(...warnings); + return job(target, result); + }); +} diff --git a/src/check-cmd.ts b/src/check-cmd.ts index e9655bf..5364b08 100644 --- a/src/check-cmd.ts +++ b/src/check-cmd.ts @@ -1,59 +1,113 @@ -// CLI entry: `npm run check -- |--all --dry-run [--print-payload [dir]]` +// CLI entry: `npm run check -- |--all [options]` // // Builds each check target's inline simulation-run payload from the files -// on disk (see check-payload.ts) and reports what would run. Config-free: -// it needs no API key and makes no network calls in --dry-run. +// on disk (check-payload.ts, check-mocks.ts) and, unless --dry-run, runs it +// in the check's run org and reports a strict verdict — as `Vapi Evals` +// commit statuses when the PR workflow's GitHub env is set, and always as +// a markdown report (the job summary in CI). // -// Exit codes: 0 every payload built, 2 usage, config or build error. +// Exit codes: 0 passed (or every payload built, with --dry-run), 1 failed, +// 2 usage, config or build error, 3 incomplete (timeout, interrupt, budget, +// billing, or API trouble). -import { existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs"; -import { join, relative, resolve } from "node:path"; +import { appendFileSync, mkdirSync, writeFileSync } from "node:fs"; +import { dirname, join, relative, resolve } from "node:path"; import { fileURLToPath } from "node:url"; -import type { CheckDefinition } from "./check-config.ts"; +import type { CheckJob } from "./check-build.ts"; +import { checkJobsBuild, checkTargetLabel } from "./check-build.ts"; +import type { CheckDefinition, ChecksConfig } from "./check-config.ts"; import { CHECKS_CONFIG_FILE, checksConfigLoad } from "./check-config.ts"; -import type { CheckPayloadResult } from "./check-payload.ts"; -import { checkPayloadBuild } from "./check-payload.ts"; -import { promotionBindingsResolve, promotionStateParse } from "./promotion.ts"; -import { orgResourcesRead } from "./resource-parse.ts"; -import type { StateFile } from "./types.ts"; +import type { CheckReportSkipped } from "./check-report.ts"; +import { checkReportJson, checkReportMarkdown } from "./check-report.ts"; +import type { CheckTargetResult } from "./check-run.ts"; +import { checkRunAll, outcomeState } from "./check-run.ts"; +import { checkAffectedBy, changedFilesRead } from "./check-select.ts"; +import type { CommitStatus, GitHubStatusEnv } from "./check-status.ts"; +import { + AGGREGATE_CONTEXT, + commitStateWorst, + commitStatusPost, + githubStatusEnvRead, + targetContext, +} from "./check-status.ts"; +import type { OrgConnection } from "./org-connection.ts"; +import { childRun, connectionLoad, tokensParse } from "./org-connection.ts"; +import { userAgentGet } from "./user-agent.ts"; +import type { VapiConnection } from "./vapi-client.ts"; const USAGE = [ "Usage:", - " npm run check -- --dry-run [--print-payload [dir]]", - " npm run check -- --all --dry-run [--print-payload [dir]]", + " npm run check -- [options]", + " npm run check -- --all [options]", "", "Options:", - ` A check name from ${CHECKS_CONFIG_FILE}`, - " --all Every check", - " --dry-run Build the payloads offline; nothing is sent", - " --print-payload [dir] Write each payload as JSON (default: tmp/check-payloads)", + ` A check name from ${CHECKS_CONFIG_FILE}`, + " --all Every check (also posts the aggregate Vapi Evals status)", + " --dry-run Build the payloads offline; nothing is sent", + " --print-payload [dir] Write each payload as JSON (default: tmp/check-payloads)", + " --changed-since Only checks affected by changes since the merge base with ", + " --budget-minutes Overall time budget; runs that can't fit aren't started (default: 60)", + " --refresh-bindings Refresh each run org's credential bindings first (read-only pull)", + " --json Also write the report as JSON", "", - "Exit codes: 0 payloads built, 2 usage, config or build error", + 'Keys (live runs): VAPI_CHECK_TOKENS ({"":""}), then .env.,', + "then VAPI_PRIVATE_API_KEY when every selected check runs in one org.", + "", + "Exit codes: 0 passed, 1 failed, 2 usage/config/build error, 3 incomplete", ].join("\n"); const DEFAULT_PAYLOAD_DIR = "tmp/check-payloads"; +const DEFAULT_BUDGET_MINUTES = 60; +const DEFAULT_BASE_URL = "https://api.vapi.ai"; +const TOKENS_ENV = "VAPI_CHECK_TOKENS"; interface CheckArgs { check?: string; all: boolean; dryRun: boolean; payloadDir?: string; + changedSince?: string; + budgetMinutes: number; + refreshBindings: boolean; + jsonPath?: string; help: boolean; } class UsageError extends Error {} +function valueTake(args: string[], index: number, flag: string): string { + const value = args[index + 1]; + if (!value || value.startsWith("--")) + throw new UsageError(`${flag} needs a value`); + return value; +} + function argsParse(args: string[]): CheckArgs { - const parsed: CheckArgs = { all: false, dryRun: false, help: false }; + const parsed: CheckArgs = { + all: false, + dryRun: false, + budgetMinutes: DEFAULT_BUDGET_MINUTES, + refreshBindings: false, + help: false, + }; for (let index = 0; index < args.length; index++) { const arg = args[index]!; if (arg === "--help" || arg === "-h") parsed.help = true; else if (arg === "--all") parsed.all = true; else if (arg === "--dry-run") parsed.dryRun = true; + else if (arg === "--refresh-bindings") parsed.refreshBindings = true; else if (arg === "--print-payload") { const next = args[index + 1]; parsed.payloadDir = next && !next.startsWith("--") ? args[++index] : DEFAULT_PAYLOAD_DIR; + } else if (arg === "--changed-since") + parsed.changedSince = valueTake(args, index++, arg); + else if (arg === "--json") parsed.jsonPath = valueTake(args, index++, arg); + else if (arg === "--budget-minutes") { + const minutes = Number(valueTake(args, index++, arg)); + if (!(minutes > 0)) + throw new UsageError("--budget-minutes must be a positive number"); + parsed.budgetMinutes = minutes; } else if (arg.startsWith("-")) throw new UsageError(`Unknown option: ${arg}`); else if (parsed.check) throw new UsageError(`Unexpected argument: ${arg}`); @@ -65,93 +119,272 @@ function argsParse(args: string[]): CheckArgs { return parsed; } -function emptyState(): StateFile { - return promotionStateParse("{}"); +function plural(count: number, noun: string): string { + return `${count} ${noun}${count === 1 ? "" : "s"}`; } -function stateRead(root: string, org: string, warnings: string[]): StateFile { - const path = join(root, `.vapi-state.${org}.json`); - if (!existsSync(path)) { - warnings.push( - `no .vapi-state.${org}.json: references by UUID and credential bindings can't resolve`, +function jobPrint(job: CheckJob, root: string, payloadDir?: string): void { + const { result, check } = job; + console.log(`\n${job.label}`); + if (result.body) { + console.log( + ` ✅ ${plural(result.body.simulations.length, "simulation")} × ${plural(check.iterations, "iteration")} over ${check.transport}, ${(result.bytes / 1024).toFixed(1)} KB`, ); - return emptyState(); + } else { + console.log(` ❌ ${plural(result.errors.length, "problem")}`); + } + for (const warning of result.warnings) console.log(` ⚠️ ${warning}`); + for (const error of result.errors) console.log(` ❌ ${error}`); + if (!result.body || !payloadDir) return; + const file = join( + resolve(root, payloadDir), + `${check.name}--${checkTargetLabel(job.target).replace(/\//g, "--")}.json`, + ); + mkdirSync(dirname(file), { recursive: true }); + writeFileSync(file, `${JSON.stringify(result.body, null, 2)}\n`); + console.log(` 📝 ${relative(root, file)}`); +} + +function checksSelect( + root: string, + checks: CheckDefinition[], + changedSince: string | undefined, +): { selected: CheckDefinition[]; skipped: CheckReportSkipped[] } { + if (!changedSince) return { selected: checks, skipped: [] }; + const files = changedFilesRead(root, changedSince); + if (!files) { + console.warn( + `⚠️ Couldn't diff against ${changedSince}; running every selected check`, + ); + return { selected: checks, skipped: [] }; + } + const selected: CheckDefinition[] = []; + const skipped: CheckReportSkipped[] = []; + for (const check of checks) { + const file = checkAffectedBy(check, files); + if (file) { + console.log(`🔎 ${check.name}: affected by ${file}`); + selected.push(check); + } else { + skipped.push({ + check: check.name, + reason: `not affected by changes since ${changedSince}`, + }); + } + } + return { selected, skipped }; +} + +// One connection per run org. Missing keys are a config error, found before +// any run starts. +function connectionsLoad( + root: string, + checks: CheckDefinition[], +): Map { + const tokens = tokensParse(TOKENS_ENV); + // Child processes (the bindings refresh) get one org's key, never the map. + delete process.env[TOKENS_ENV]; + const runOrgs = [...new Set(checks.map((check) => check.runOrg))]; + const envKey = process.env.VAPI_PRIVATE_API_KEY ?? process.env.VAPI_TOKEN; + const connections = new Map(); + for (const runOrg of runOrgs) { + const baseUrl = checks.find( + (check) => check.runOrg === runOrg && check.baseUrl, + )?.baseUrl; + let connection: OrgConnection; + try { + connection = connectionLoad({ + rootDir: root, + org: runOrg, + tokens, + tokensEnvName: TOKENS_ENV, + baseUrl, + }); + } catch (error) { + if (!envKey || runOrgs.length !== 1) throw error; + connection = { token: envKey, baseUrl }; + } + connections.set(runOrg, { + token: connection.token, + baseUrl: + connection.baseUrl ?? process.env.VAPI_BASE_URL ?? DEFAULT_BASE_URL, + }); } - return promotionStateParse(readFileSync(path, "utf8")); + return connections; } -function payloadFileName(check: string, target: string): string { - return `${check}--${target.replace(/\//g, "--")}.json`; +function exitCode(results: CheckTargetResult[]): number { + const outcomes = new Set(results.map((result) => result.outcome)); + if (outcomes.has("error")) return 2; + if (outcomes.has("failed")) return 1; + if (outcomes.has("incomplete")) return 3; + return 0; } -function resultPrint( - label: string, - result: CheckPayloadResult, - check: CheckDefinition, +function aggregateStatus( + results: CheckTargetResult[], + dryRun: boolean, + statusEnv: GitHubStatusEnv, +): CommitStatus { + if (results.length === 0) + return { + context: AGGREGATE_CONTEXT, + state: "success", + description: "No checks affected by this change", + targetUrl: statusEnv.runUrl, + }; + if (dryRun) + return { + context: AGGREGATE_CONTEXT, + state: "error", + description: + "Not run: fork or Dependabot PR — a maintainer must dispatch the check", + targetUrl: statusEnv.runUrl, + }; + const counts = new Map(); + for (const result of results) + counts.set(result.outcome, (counts.get(result.outcome) ?? 0) + 1); + const urls = results.map((result) => result.url).filter(Boolean); + return { + context: AGGREGATE_CONTEXT, + state: commitStateWorst( + results.map((result) => outcomeState(result.outcome)), + ), + description: [...counts] + .map(([outcome, count]) => `${count} ${outcome}`) + .join(", "), + targetUrl: + urls.length === 1 && results.length === 1 ? urls[0] : statusEnv.runUrl, + }; +} + +function reportWrite( + root: string, + parsed: CheckArgs, + results: CheckTargetResult[], + skipped: CheckReportSkipped[], ): void { - console.log(`\n${label}`); - if (result.body) { - const count = result.body.simulations.length; - console.log( - ` ✅ ${count} simulation${count === 1 ? "" : "s"} × ${check.iterations} iteration${check.iterations === 1 ? "" : "s"} over ${check.transport}, ${(result.bytes / 1024).toFixed(1)} KB`, - ); - } else { + const input = { results, skipped, dryRun: parsed.dryRun }; + const markdown = checkReportMarkdown(input); + console.log(`\n${markdown}`); + const summary = process.env.GITHUB_STEP_SUMMARY; + if (summary) appendFileSync(summary, `${markdown}\n`); + if (parsed.jsonPath) { + const path = resolve(root, parsed.jsonPath); + mkdirSync(dirname(path), { recursive: true }); + writeFileSync(path, `${JSON.stringify(checkReportJson(input), null, 2)}\n`); + } +} + +async function liveRun( + parsed: CheckArgs, + jobs: CheckJob[], + connections: Map, + statusEnv: GitHubStatusEnv | undefined, +): Promise { + const controller = new AbortController(); + const abort = () => { + console.log("\n⏹ Interrupted: canceling in-flight runs…"); + controller.abort(); + }; + process.once("SIGINT", abort); + process.once("SIGTERM", abort); + const userAgent = userAgentGet("check"); + const vapi = (job: CheckJob): VapiConnection => { + const connection = connections.get(job.check.runOrg)!; + return { token: connection.token, baseUrl: connection.baseUrl!, userAgent }; + }; + const post = async (status: CommitStatus) => { + if (statusEnv) await commitStatusPost(statusEnv, status); + }; + try { console.log( - ` ❌ ${result.errors.length} problem${result.errors.length === 1 ? "" : "s"}`, + `\n🧪 Running ${plural(jobs.filter((job) => job.result.body).length, "target")}…`, ); + return await checkRunAll({ + jobs, + connectionFor: vapi, + deadline: Date.now() + parsed.budgetMinutes * 60_000, + signal: controller.signal, + onRunCreated: (job, url) => + post({ + context: targetContext(job.check.name, checkTargetLabel(job.target)), + state: "pending", + description: "Simulations running", + targetUrl: url, + }), + onResult: (result) => + post({ + context: targetContext( + result.job.check.name, + checkTargetLabel(result.job.target), + ), + state: outcomeState(result.outcome), + description: result.reason, + targetUrl: result.url ?? statusEnv?.runUrl, + }), + }); + } finally { + process.off("SIGINT", abort); + process.off("SIGTERM", abort); } - for (const warning of result.warnings) console.log(` ⚠️ ${warning}`); - for (const error of result.errors) console.log(` ❌ ${error}`); } -async function checkDryRun( +async function checksRun( root: string, - check: CheckDefinition, - payloadDir: string | undefined, -): Promise { - if (!existsSync(join(root, "resources", check.org))) { - console.log(`\n${check.name}\n ❌ resources/${check.org}/ does not exist`); - return false; + parsed: CheckArgs, + config: ChecksConfig, + statusEnv: GitHubStatusEnv | undefined, +): Promise<{ code: number; results?: CheckTargetResult[] }> { + const names = parsed.all ? Object.keys(config.checks) : [parsed.check!]; + const missing = names.filter((name) => !config.checks[name]); + if (missing.length > 0) { + console.error( + `❌ No check named ${missing.join(", ")} in ${CHECKS_CONFIG_FILE} (checks: ${Object.keys(config.checks).join(", ")})`, + ); + return { code: 2 }; } - const warnings: string[] = []; - const resources = await orgResourcesRead(root, check.org); - const sourceState = stateRead(root, check.org, warnings); - const runState = - check.runOrg === check.org - ? sourceState - : stateRead(root, check.runOrg, warnings); - const bindingsResolved = await promotionBindingsResolve( + const { selected, skipped } = checksSelect( root, - check.org, - check.runOrg, - sourceState, + names.map((name) => config.checks[name]!), + parsed.changedSince, ); - let ok = true; - for (const target of check.targets) { - const targetLabel = `${target.type}/${target.id}`; - const result = checkPayloadBuild({ - check, - target, - resources, - sourceState, - runState, - bindingsResolved, - }); - result.warnings.unshift(...warnings); - resultPrint(`${check.name} / ${targetLabel}`, result, check); - if (!result.body) { - ok = false; - continue; - } - if (payloadDir) { - const dir = resolve(root, payloadDir); - mkdirSync(dir, { recursive: true }); - const file = join(dir, payloadFileName(check.name, targetLabel)); - writeFileSync(file, `${JSON.stringify(result.body, null, 2)}\n`); - console.log(` 📝 ${relative(root, file)}`); + let connections = new Map(); + if (!parsed.dryRun && selected.length > 0) { + try { + connections = connectionsLoad(root, selected); + } catch (error) { + console.error(`❌ ${(error as Error).message}`); + return { code: 2 }; } + if (parsed.refreshBindings) + for (const [org, connection] of connections) + childRun({ + rootDir: root, + script: "src/pull.ts", + org, + connection, + args: ["--bootstrap", "--bindings-only"], + }); } - return ok; + const jobs: CheckJob[] = []; + for (const check of selected) + jobs.push(...(await checkJobsBuild(root, check))); + for (const job of jobs) jobPrint(job, root, parsed.payloadDir); + const results: CheckTargetResult[] = parsed.dryRun + ? jobs.map((job) => ({ + job, + outcome: job.result.body ? "built" : "error", + reason: job.result.body + ? "payload built; not run (--dry-run)" + : `${plural(job.result.errors.length, "problem")} building the payload`, + failures: [], + mockNotices: [], + durationMs: 0, + })) + : await liveRun(parsed, jobs, connections, statusEnv); + reportWrite(root, parsed, results, skipped); + return { code: exitCode(results), results }; } export async function checkCommandRun( @@ -173,42 +406,38 @@ export async function checkCommandRun( console.log(USAGE); return 0; } - let config; + const statusEnv = githubStatusEnvRead(); + let aggregate: CommitStatus | undefined = { + context: AGGREGATE_CONTEXT, + state: "error", + description: "The check failed before it could report", + targetUrl: statusEnv?.runUrl, + }; try { - config = checksConfigLoad(root); - } catch (error) { - console.error(`❌ ${CHECKS_CONFIG_FILE}: ${(error as Error).message}`); - return 2; - } - if (!config) { - console.log( - `No ${CHECKS_CONFIG_FILE} at the repository root; nothing to check.`, - ); - return 0; - } - const names = parsed.all ? Object.keys(config.checks) : [parsed.check!]; - const missing = names.filter((name) => !config.checks[name]); - if (missing.length > 0) { - console.error( - `❌ No check named ${missing.join(", ")} in ${CHECKS_CONFIG_FILE} (checks: ${Object.keys(config.checks).join(", ")})`, - ); - return 2; - } - if (!parsed.dryRun) { - console.error( - "❌ Live runs aren't available yet: pass --dry-run to build the payloads offline.", - ); - return 2; - } - let ok = true; - for (const name of names) { - if (!(await checkDryRun(root, config.checks[name]!, parsed.payloadDir))) - ok = false; + let config: ChecksConfig | null; + try { + config = checksConfigLoad(root); + } catch (error) { + console.error(`❌ ${CHECKS_CONFIG_FILE}: ${(error as Error).message}`); + aggregate.description = `${CHECKS_CONFIG_FILE} is invalid`; + return 2; + } + if (!config) { + console.log( + `No ${CHECKS_CONFIG_FILE} at the repository root; nothing to check.`, + ); + aggregate = undefined; + return 0; + } + const { code, results } = await checksRun(root, parsed, config, statusEnv); + if (results && statusEnv) + aggregate = aggregateStatus(results, parsed.dryRun, statusEnv); + return code; + } finally { + // Only an --all run speaks for the whole PR. + if (parsed.all && statusEnv && aggregate) + await commitStatusPost(statusEnv, aggregate); } - console.log( - ok ? "\n✅ Every payload built." : "\n❌ Some payloads could not be built.", - ); - return ok ? 0 : 2; } const isMainModule = diff --git a/src/check-report.ts b/src/check-report.ts new file mode 100644 index 0000000..93c998b --- /dev/null +++ b/src/check-report.ts @@ -0,0 +1,111 @@ +// The PR check's report: markdown for $GITHUB_STEP_SUMMARY (and the +// terminal), JSON for --json. No PR comments — the commit status links the +// run, and the job summary carries the detail. + +import type { CheckOutcome, CheckTargetResult } from "./check-run.ts"; + +export interface CheckReportSkipped { + check: string; + reason: string; +} + +export interface CheckReportInput { + results: CheckTargetResult[]; + skipped: CheckReportSkipped[]; + dryRun: boolean; +} + +const OUTCOME_LABEL: Record = { + passed: "✅ passed", + failed: "❌ failed", + incomplete: "⚠️ incomplete", + error: "🛑 not run", + built: "📦 built, not run", +}; + +function cell(value: unknown): string { + let text = ""; + if (typeof value === "string") text = value; + else if (value !== undefined) text = JSON.stringify(value); + return text.replace(/\|/g, "\\|").replace(/\n/g, " "); +} + +function countsCell(result: CheckTargetResult): string { + const counts = result.counts; + if (!counts) return ""; + return `${counts.passed}/${counts.total}`; +} + +function detailLines(result: CheckTargetResult): string[] { + const lines: string[] = []; + const { job } = result; + const quiet = result.outcome === "passed" || result.outcome === "built"; + if ( + quiet && + result.mockNotices.length === 0 && + job.result.warnings.length === 0 + ) + return lines; + lines.push(`### ${OUTCOME_LABEL[result.outcome]}: ${job.label}`, ""); + lines.push(result.reason, ""); + if (result.failures.length > 0) { + lines.push( + "| Simulation | Evaluation | Comparator | Expected | Got | Note |", + "|---|---|---|---|---|---|", + ); + for (const failure of result.failures) + lines.push( + `| ${cell(failure.item)} | ${cell(failure.evaluation)} | ${cell(failure.comparator)} | ${cell(failure.expected)} | ${cell(failure.extracted)} | ${cell(failure.reason)} |`, + ); + lines.push(""); + } + for (const error of job.result.errors) lines.push(`- ❌ ${error}`); + for (const notice of result.mockNotices) lines.push(`- 🧪 ${notice}`); + for (const warning of job.result.warnings) lines.push(`- ⚠️ ${warning}`); + if (lines[lines.length - 1] !== "") lines.push(""); + return lines; +} + +export function checkReportMarkdown(input: CheckReportInput): string { + const lines = [ + `## Vapi Evals${input.dryRun ? " (dry run: payloads built, nothing sent)" : ""}`, + "", + ]; + if (input.results.length > 0) { + lines.push( + "| Check | Target | Result | Simulations | Run |", + "|---|---|---|---|---|", + ); + for (const result of input.results) + lines.push( + `| ${cell(result.job.check.name)} | ${cell(`${result.job.target.type}/${result.job.target.id}`)} | ${OUTCOME_LABEL[result.outcome]} | ${countsCell(result)} | ${result.url ? `[open](${result.url})` : ""} |`, + ); + lines.push(""); + } + for (const skipped of input.skipped) + lines.push(`- ⏭️ ${skipped.check}: ${skipped.reason}`); + if (input.skipped.length > 0) lines.push(""); + for (const result of input.results) lines.push(...detailLines(result)); + return `${lines.join("\n").trimEnd()}\n`; +} + +export function checkReportJson(input: CheckReportInput): unknown { + return { + dryRun: input.dryRun, + results: input.results.map((result) => ({ + check: result.job.check.name, + target: `${result.job.target.type}/${result.job.target.id}`, + outcome: result.outcome, + reason: result.reason, + runId: result.runId, + url: result.url, + counts: result.counts, + failures: result.failures, + mockNotices: result.mockNotices, + errors: result.job.result.errors, + warnings: result.job.result.warnings, + durationMs: result.durationMs, + })), + skipped: input.skipped, + }; +} diff --git a/src/check-run.ts b/src/check-run.ts new file mode 100644 index 0000000..5280fa1 --- /dev/null +++ b/src/check-run.ts @@ -0,0 +1,177 @@ +// Run built check payloads live: one simulation run per target, at most +// MAX_CONCURRENT at once, inside one overall budget. Reuses `npm run sim`'s +// create/poll/cancel/verdict loop (simRunExecute), so a check is judged by +// the same strict rules. + +import type { CheckJob } from "./check-build.ts"; +import { MOCK_MARKER } from "./check-mocks.ts"; +import type { CommitState } from "./check-status.ts"; +import type { SimRunExecuteOptions } from "./sim.ts"; +import { simRunExecute } from "./sim.ts"; +import type { + SimRunFailure, + SimRunItem, + SimRunItemCounts, +} from "./sim-result.ts"; +import type { VapiConnection } from "./vapi-client.ts"; +import { VapiApiError } from "./vapi-client.ts"; + +// passed / failed / incomplete come from the run's verdict; error is a +// config or build problem (nothing was sent); built is a dry run's success. +export type CheckOutcome = + "passed" | "failed" | "incomplete" | "error" | "built"; + +export interface CheckTargetResult { + job: CheckJob; + outcome: CheckOutcome; + reason: string; + runId?: string; + url?: string; + counts?: SimRunItemCounts; + failures: SimRunFailure[]; + // "unmocked tool called" notices from the transcripts. + mockNotices: string[]; + durationMs: number; +} + +export interface CheckRunArgs { + jobs: CheckJob[]; + connectionFor: (job: CheckJob) => VapiConnection; + // Absolute time by which every run must have finished. + deadline: number; + signal?: AbortSignal; + // Called once the run exists, with its canonical link. + onRunCreated?: (job: CheckJob, url: string | undefined) => Promise; + onResult?: (result: CheckTargetResult) => Promise; + concurrency?: number; + // Test seams for simRunExecute's polling. + pollIntervalMs?: number; + hydrationMs?: number; +} + +export const MAX_CONCURRENT = 3; +// Don't start a run that couldn't plausibly finish before the deadline. +export const MIN_START_MS = 5 * 60_000; + +export function outcomeState(outcome: CheckOutcome): CommitState { + if (outcome === "passed") return "success"; + if (outcome === "failed") return "failure"; + return "error"; +} + +// Default mocks answer with MOCK_MARKER, so a tool result carrying it means +// the assistant called a tool its scenario didn't mock. Only the call's +// transcript is read: the item also echoes the scenario, default mocks and +// all, which would match whether or not they were called. +export function mockNoticesCollect(items: SimRunItem[]): string[] { + const notices = new Set(); + for (const item of items) { + const label = item.metadata?.simulation?.name ?? item.id; + const messages = item.metadata?.call?.messages ?? []; + const names = new Map(); + for (const message of messages) + for (const call of message.toolCalls ?? []) + if (call.id && call.function?.name) + names.set(call.id, call.function.name); + for (const message of messages) { + if (message.role !== "tool_call_result") continue; + if (!String(message.result ?? "").includes(MOCK_MARKER)) continue; + const id = message.toolCallId ?? message.name ?? ""; + notices.add(`${label}: unmocked tool called: ${names.get(id) ?? id}`); + } + } + return [...notices]; +} + +function errorReason(error: unknown): string { + if (error instanceof VapiApiError && error.statusCode === 402) + return `billing: the run org can't start simulations (402: ${error.apiMessage})`; + return error instanceof Error ? error.message : String(error); +} + +async function jobRun( + args: CheckRunArgs, + job: CheckJob, +): Promise { + const start = Date.now(); + const base = { job, failures: [], mockNotices: [], durationMs: 0 }; + const body = job.result.body; + if (!body) + return { + ...base, + outcome: "error", + reason: `payload could not be built (${job.result.errors.length} problem${job.result.errors.length === 1 ? "" : "s"})`, + }; + if (args.signal?.aborted) + return { + ...base, + outcome: "incomplete", + reason: "not started: interrupted", + }; + const remaining = args.deadline - Date.now(); + if (remaining < MIN_START_MS) + return { + ...base, + outcome: "incomplete", + reason: `not started: ${Math.max(0, Math.round(remaining / 60_000))} min of budget left`, + }; + const options: SimRunExecuteOptions = { + timeoutMs: Math.min(job.check.timeoutMinutes * 60_000, remaining), + signal: args.signal, + progress: false, + pollIntervalMs: args.pollIntervalMs, + hydrationMs: args.hydrationMs, + onCreated: async (created) => { + console.log(` ▶ ${job.label}: ${created.url ?? created.id}`); + await args.onRunCreated?.(job, created.url); + }, + }; + try { + const { summary, items } = await simRunExecute( + args.connectionFor(job), + body, + options, + ); + const verdict = summary.verdict; + return { + job, + outcome: verdict?.status ?? "incomplete", + reason: verdict?.reason ?? "no verdict", + runId: summary.runId, + url: summary.url, + counts: summary.counts, + failures: verdict?.failures ?? [], + mockNotices: mockNoticesCollect(items), + durationMs: Date.now() - start, + }; + } catch (error) { + return { + ...base, + outcome: "incomplete", + reason: errorReason(error), + durationMs: Date.now() - start, + }; + } +} + +// Results come back in job order, whatever order the runs finish in. +export async function checkRunAll( + args: CheckRunArgs, +): Promise { + const results: CheckTargetResult[] = new Array(args.jobs.length); + let next = 0; + const worker = async () => { + while (next < args.jobs.length) { + const index = next++; + const result = await jobRun(args, args.jobs[index]!); + results[index] = result; + await args.onResult?.(result); + } + }; + const workers = Math.min( + args.concurrency ?? MAX_CONCURRENT, + args.jobs.length, + ); + await Promise.all(Array.from({ length: workers }, worker)); + return results; +} diff --git a/src/check-select.ts b/src/check-select.ts new file mode 100644 index 0000000..f3d91da --- /dev/null +++ b/src/check-select.ts @@ -0,0 +1,56 @@ +// Which checks a change affects. `--changed-since ` diffs from the +// merge base (`...HEAD`), so commits that landed on the base branch +// after the PR branched don't count as the PR's changes. + +import { execFileSync } from "node:child_process"; +import type { CheckDefinition } from "./check-config.ts"; +import { CHECKS_CONFIG_FILE } from "./check-config.ts"; +import { compilePattern } from "./resource-parse.ts"; + +// Changes to any of these can change every check's payload or verdict. +const ENGINE_PATTERNS = [ + CHECKS_CONFIG_FILE, + "promotion.yml", + "src/**", + "package.json", + "package-lock.json", +]; + +// Files changed between the merge base of `ref` and HEAD, or undefined when +// git can't answer (shallow clone, unknown ref) — callers then run everything. +export function changedFilesRead( + rootDir: string, + ref: string, +): string[] | undefined { + try { + const output = execFileSync( + "git", + ["diff", "--name-only", `${ref}...HEAD`], + { cwd: rootDir, encoding: "utf8", stdio: ["ignore", "pipe", "pipe"] }, + ); + return output.split("\n").filter((line) => line.length > 0); + } catch { + return undefined; + } +} + +export function checkPatterns(check: CheckDefinition): string[] { + const orgs = [...new Set([check.org, check.runOrg])]; + return [ + ...ENGINE_PATTERNS, + ...orgs.flatMap((org) => [ + `resources/${org}/**`, + `.vapi-state.${org}.json`, + ]), + ...check.paths, + ]; +} + +// The first changed file that affects the check, or undefined. +export function checkAffectedBy( + check: CheckDefinition, + files: string[], +): string | undefined { + const patterns = checkPatterns(check).map(compilePattern); + return files.find((file) => patterns.some((pattern) => pattern.test(file))); +} diff --git a/src/check-status.ts b/src/check-status.ts new file mode 100644 index 0000000..9d61e65 --- /dev/null +++ b/src/check-status.ts @@ -0,0 +1,106 @@ +// GitHub commit statuses for PR checks. Plain fetch, no dependency. +// +// Per check and target: `Vapi Evals / / `, pending with the +// run link as soon as the run exists, then the verdict. The aggregate +// `Vapi Evals` (the name PAL-608 specifies) is what branch protection +// requires: per-target statuses only exist on PRs that touch a check. + +export const AGGREGATE_CONTEXT = "Vapi Evals"; + +export type CommitState = "pending" | "success" | "failure" | "error"; + +export interface GitHubStatusEnv { + token: string; + repo: string; + sha: string; + apiUrl: string; + // The workflow run page, for statuses that cover several runs. + runUrl?: string; +} + +export interface CommitStatus { + context: string; + state: CommitState; + description: string; + targetUrl?: string; +} + +// GitHub caps status descriptions at 140 characters. +const MAX_DESCRIPTION = 140; +const STATE_RANK: Record = { + success: 0, + pending: 1, + failure: 2, + error: 3, +}; + +export function targetContext(check: string, target: string): string { + return `${AGGREGATE_CONTEXT} / ${check} / ${target}`; +} + +// Statuses post only when all three are set (the PR workflow sets them). +export function githubStatusEnvRead( + env: NodeJS.ProcessEnv = process.env, +): GitHubStatusEnv | undefined { + const { GITHUB_TOKEN, GITHUB_REPOSITORY, HEAD_SHA } = env; + if (!GITHUB_TOKEN || !GITHUB_REPOSITORY || !HEAD_SHA) return undefined; + const server = env.GITHUB_SERVER_URL ?? "https://github.com"; + return { + token: GITHUB_TOKEN, + repo: GITHUB_REPOSITORY, + sha: HEAD_SHA, + apiUrl: env.GITHUB_API_URL ?? "https://api.github.com", + runUrl: env.GITHUB_RUN_ID + ? `${server}/${GITHUB_REPOSITORY}/actions/runs/${env.GITHUB_RUN_ID}` + : undefined, + }; +} + +// The worst state wins: error > failure > pending > success. +export function commitStateWorst(states: CommitState[]): CommitState { + return states.reduce( + (worst, state) => (STATE_RANK[state] > STATE_RANK[worst] ? state : worst), + "success", + ); +} + +// Posting a status never fails the check: a read-only token (fork PRs) or a +// GitHub outage only warns. Returns whether the status was accepted. +export async function commitStatusPost( + env: GitHubStatusEnv, + status: CommitStatus, +): Promise { + const description = + status.description.length > MAX_DESCRIPTION + ? `${status.description.slice(0, MAX_DESCRIPTION - 1)}…` + : status.description; + try { + const response = await fetch( + `${env.apiUrl}/repos/${env.repo}/statuses/${env.sha}`, + { + method: "POST", + headers: { + Authorization: `Bearer ${env.token}`, + Accept: "application/vnd.github+json", + "Content-Type": "application/json", + "X-GitHub-Api-Version": "2022-11-28", + }, + body: JSON.stringify({ + context: status.context, + state: status.state, + description, + ...(status.targetUrl ? { target_url: status.targetUrl } : {}), + }), + }, + ); + if (response.ok) return true; + console.warn( + ` ⚠️ Could not post status "${status.context}" (${response.status}); see the job summary instead`, + ); + } catch (error) { + console.warn( + ` ⚠️ Could not post status "${status.context}": ${(error as Error).message}`, + ); + } + return false; +} diff --git a/src/resource-parse.ts b/src/resource-parse.ts index 36b7971..0af7325 100644 --- a/src/resource-parse.ts +++ b/src/resource-parse.ts @@ -359,8 +359,8 @@ export function ignorePatternsRead(resourcesDir: string): string[] { } // Convert a gitignore-flavored glob to a RegExp. We keep the implementation -// intentionally small (no node_modules) since pull.ts is the only consumer. -function compilePattern(pattern: string): RegExp { +// intentionally small (no node_modules). Also matches check trigger paths. +export function compilePattern(pattern: string): RegExp { // Escape regex metacharacters except the glob ones we handle explicitly. // `*` and `?` are translated below; everything else is literal. let regex = ""; diff --git a/src/sim-result.ts b/src/sim-result.ts index 9869c41..cb14fd2 100644 --- a/src/sim-result.ts +++ b/src/sim-result.ts @@ -40,11 +40,25 @@ export interface SimRunEvaluation { error?: string; } +// One transcript message. Tool calls arrive as `tool_calls` (with ids and +// function names) and their answers as `tool_call_result` (by call id). +export interface SimRunMessage { + role?: string; + message?: string; + name?: string; + toolCallId?: string; + result?: unknown; + toolCalls?: Array<{ id?: string; function?: { name?: string } }>; +} + export interface SimRunItem { id: string; status?: string; results?: { passed?: boolean; evaluations?: SimRunEvaluation[] }; - metadata?: { simulation?: { name?: string } }; + metadata?: { + simulation?: { name?: string }; + call?: { messages?: SimRunMessage[] }; + }; } export type SimRunVerdictStatus = "passed" | "failed" | "incomplete"; diff --git a/src/sim.ts b/src/sim.ts index f082f9e..4befa3e 100644 --- a/src/sim.ts +++ b/src/sim.ts @@ -222,7 +222,7 @@ export function resolveSelection( // Fields of `POST /eval/simulation/run`'s response this runner reads. The // create response is the only place `url` and `simulationRunItemIds` appear. -interface SimRunCreated extends SimRun { +export interface SimRunCreated extends SimRun { url?: string; simulationRunItemIds?: string[]; } @@ -311,30 +311,30 @@ export async function simRunCancel( } } -export async function runSimulation( - cfg: SimEnv, - selection: SimSelection, - target: SimTarget, - options: SimRunOptions = {}, -): Promise { - const connection = connectionFor(cfg); - const pollIntervalMs = options.pollIntervalMs ?? POLL_INTERVAL_MS; - const body: Record = { - simulations: selection.entries, - target: - target.type === "assistant" - ? { type: "assistant", assistantId: target.id } - : { type: "squad", squadId: target.id }, - transport: { - provider: - options.transport === "chat" ? "vapi.webchat" : "vapi.websocket", - }, - }; - if (options.iterations !== undefined) body.iterations = options.iterations; +export interface SimRunExecuteOptions extends SimRunOptions { + // Called with the create response, before polling (prints the link, or + // posts a pending commit status pointing at it). + onCreated?: (created: SimRunCreated) => void | Promise; + // Rewrite a "Status: …" line on a TTY while polling. Off when several runs + // poll at once. + progress?: boolean; +} - console.log( - `🧪 Starting simulation run — ${selection.label} → ${target.type}/${target.resourceName}`, - ); +export interface SimRunExecuted { + summary: SimRunSummary; + items: SimRunItem[]; +} + +// Create a run from `body`, poll it until it ends or the deadline passes +// (canceling it then, or on abort), wait for its items, and judge it with +// simRunVerdict. Shared by `npm run sim` and the PR check. +export async function simRunExecute( + connection: VapiConnection, + body: unknown, + options: SimRunExecuteOptions = {}, +): Promise { + const pollIntervalMs = options.pollIntervalMs ?? POLL_INTERVAL_MS; + const progress = options.progress ?? true; const start = Date.now(); // Creating a run queues paid work before the response returns, so a 5xx // here may still have started it: retry rate limits only. @@ -349,8 +349,7 @@ export async function runSimulation( if (!runId) { throw new Error("POST /eval/simulation/run returned no run id"); } - console.log(` Run ID: ${runId}`); - if (created.url) console.log(` Run: ${created.url}`); + await options.onCreated?.(created); const summary: SimRunSummary = { runId, @@ -362,7 +361,7 @@ export async function runSimulation( }; if (!(options.watch ?? true)) { summary.durationMs = Date.now() - start; - return summary; + return { summary, items: [] }; } const deadline = start + (options.timeoutMs ?? DEFAULT_TIMEOUT_MS); @@ -387,11 +386,11 @@ export async function runSimulation( "GET", `/eval/simulation/run/${runId}`, ); - if (process.stdout.isTTY) { + if (progress && process.stdout.isTTY) { process.stdout.write(`\r Status: ${run.status ?? "unknown"} `); } } - if (process.stdout.isTTY) process.stdout.write("\n"); + if (progress && process.stdout.isTTY) process.stdout.write("\n"); if (stopReason) { summary.canceled = await simRunCancel(connection, runId); @@ -403,7 +402,7 @@ export async function runSimulation( failures: [], }; summary.durationMs = Date.now() - start; - return summary; + return { summary, items: [] }; } // Items can lag the run: keep re-reading until every item is terminal and @@ -432,6 +431,38 @@ export async function runSimulation( expected: created.simulationRunItemIds?.length, }); summary.durationMs = Date.now() - start; + return { summary, items }; +} + +export async function runSimulation( + cfg: SimEnv, + selection: SimSelection, + target: SimTarget, + options: SimRunOptions = {}, +): Promise { + const body: Record = { + simulations: selection.entries, + target: + target.type === "assistant" + ? { type: "assistant", assistantId: target.id } + : { type: "squad", squadId: target.id }, + transport: { + provider: + options.transport === "chat" ? "vapi.webchat" : "vapi.websocket", + }, + }; + if (options.iterations !== undefined) body.iterations = options.iterations; + + console.log( + `🧪 Starting simulation run — ${selection.label} → ${target.type}/${target.resourceName}`, + ); + const { summary } = await simRunExecute(connectionFor(cfg), body, { + ...options, + onCreated: (created) => { + console.log(` Run ID: ${created.id}`); + if (created.url) console.log(` Run: ${created.url}`); + }, + }); return summary; } diff --git a/tests/check-cmd.test.ts b/tests/check-cmd.test.ts index 98813fd..82c7d5d 100644 --- a/tests/check-cmd.test.ts +++ b/tests/check-cmd.test.ts @@ -1,4 +1,5 @@ import assert from "node:assert/strict"; +import { execFileSync } from "node:child_process"; import { cpSync, existsSync, @@ -13,6 +14,19 @@ import test from "node:test"; import { fileURLToPath } from "node:url"; import { checkCommandRun } from "../src/check-cmd.ts"; +// Never reach a real org from a developer shell that has a key exported. +for (const name of [ + "VAPI_PRIVATE_API_KEY", + "VAPI_TOKEN", + "VAPI_CHECK_TOKENS", + "VAPI_BASE_URL", + "GITHUB_TOKEN", + "GITHUB_REPOSITORY", + "HEAD_SHA", + "GITHUB_STEP_SUMMARY", +]) + delete process.env[name]; + const PARITY_ROOT = fileURLToPath( new URL("./fixtures/check-parity/", import.meta.url), ); @@ -102,7 +116,7 @@ test("usage, config and selection errors exit 2", async () => { (await run(["core", "--all", "--dry-run"], root)).code, (await run(["core", "--bogus"], root)).code, (await run(["nope", "--dry-run"], root)).code, - // Live runs aren't available in this command yet. + // A live run with no key for the run org. (await run(["core"], root)).code, ]; writeFileSync(join(root, "vapi-checks.yml"), "version: 2\n"); @@ -130,3 +144,313 @@ test("a missing state file is a warning, not an error", async () => { rmSync(root, { recursive: true, force: true }); } }); + +// ── Live runs against a stub serving both the simulations API and GitHub ── + +interface LiveStub { + baseUrl: string; + mode: { failFirstItem: boolean }; + vapiPosts: number; + statuses: Array<{ + context: string; + state: string; + target_url?: string; + description: string; + }>; +} + +async function withLiveStub( + fn: (stub: LiveStub) => Promise, +): Promise { + const { createServer } = await import("node:http"); + const stub: LiveStub = { + baseUrl: "", + mode: { failFirstItem: false }, + vapiPosts: 0, + statuses: [], + }; + let total = 0; + const server = createServer((req, res) => { + let raw = ""; + req.on("data", (chunk) => (raw += chunk)); + req.on("end", () => { + const reply = (status: number, json: unknown) => { + res.writeHead(status, { "Content-Type": "application/json" }); + res.end(JSON.stringify(json)); + }; + const url = req.url ?? ""; + if (url.startsWith("/repos/")) { + stub.statuses.push(JSON.parse(raw)); + return reply(201, {}); + } + const failed = stub.mode.failFirstItem ? 1 : 0; + if (req.method === "POST") { + stub.vapiPosts++; + total = (JSON.parse(raw) as { simulations: unknown[] }).simulations + .length; + return reply(201, { + id: "run-1", + url: "https://dashboard.vapi.ai/run-1", + status: "queued", + simulationRunItemIds: Array.from( + { length: total }, + (_, i) => `i${i}`, + ), + }); + } + if (url.includes("/item")) + return reply(200, { + results: Array.from({ length: total }, (_, i) => ({ + id: `i${i}`, + status: i < failed ? "failed" : "passed", + metadata: { simulation: { name: `S${i + 1}` } }, + results: { + evaluations: [ + { + name: "j", + required: true, + passed: i >= failed, + comparator: "=", + expectedValue: true, + extractedValue: i >= failed, + }, + ], + }, + })), + metadata: { totalItems: total }, + }); + return reply(200, { + id: "run-1", + status: "ended", + itemCounts: { + total, + passed: total - failed, + failed, + running: 0, + queued: 0, + canceled: 0, + }, + }); + }); + }); + await new Promise((resolve) => server.listen(0, resolve)); + stub.baseUrl = `http://127.0.0.1:${(server.address() as import("node:net").AddressInfo).port}`; + try { + await fn(stub); + } finally { + await new Promise((resolve) => server.close(() => resolve())); + } +} + +async function withGitHubEnv( + stub: LiveStub, + summary: string, + fn: () => Promise, +): Promise { + Object.assign(process.env, { + GITHUB_TOKEN: "gh", + GITHUB_REPOSITORY: "acme/gitops", + HEAD_SHA: "abc123", + GITHUB_API_URL: stub.baseUrl, + GITHUB_RUN_ID: "7", + GITHUB_STEP_SUMMARY: summary, + }); + try { + return await fn(); + } finally { + for (const name of [ + "GITHUB_TOKEN", + "GITHUB_REPOSITORY", + "HEAD_SHA", + "GITHUB_API_URL", + "GITHUB_RUN_ID", + "GITHUB_STEP_SUMMARY", + ]) + delete process.env[name]; + } +} + +function liveCopy(stub: LiveStub): string { + const root = parityCopy(); + writeFileSync( + join(root, ".env.parity"), + `VAPI_PRIVATE_API_KEY=test-key\nVAPI_BASE_URL=${stub.baseUrl}\n`, + ); + return root; +} + +test("a live --all run posts pending then success per target, the aggregate, the job summary and JSON", async () => { + await withLiveStub(async (stub) => { + const root = liveCopy(stub); + try { + const summary = join(root, "summary.md"); + const result = await withGitHubEnv(stub, summary, () => + run(["--all", "--json", "tmp/report.json"], root), + ); + const report = JSON.parse( + readFileSync(join(root, "tmp/report.json"), "utf8"), + ); + assert.deepEqual( + { + code: result.code, + statuses: stub.statuses.map((s) => [ + s.context, + s.state, + s.target_url, + ]), + summary: readFileSync(summary, "utf8").includes( + "| core | squads/dental | ✅ passed | 3/3 | [open](https://dashboard.vapi.ai/run-1) |", + ), + json: report.results.map((r: { outcome: string }) => r.outcome), + }, + { + code: 0, + statuses: [ + [ + "Vapi Evals / core / squads/dental", + "pending", + "https://dashboard.vapi.ai/run-1", + ], + [ + "Vapi Evals / core / squads/dental", + "success", + "https://dashboard.vapi.ai/run-1", + ], + ["Vapi Evals", "success", "https://dashboard.vapi.ai/run-1"], + ], + summary: true, + json: ["passed"], + }, + ); + } finally { + rmSync(root, { recursive: true, force: true }); + } + }); +}); + +test("a failing live run exits 1 and turns the aggregate red", async () => { + await withLiveStub(async (stub) => { + stub.mode.failFirstItem = true; + const root = liveCopy(stub); + try { + const result = await withGitHubEnv(stub, join(root, "summary.md"), () => + run(["--all"], root), + ); + assert.deepEqual( + [result.code, stub.statuses.at(-1)], + [ + 1, + { + context: "Vapi Evals", + state: "failure", + description: "1 failed", + target_url: "https://dashboard.vapi.ai/run-1", + }, + ], + ); + } finally { + rmSync(root, { recursive: true, force: true }); + } + }); +}); + +test("a single named check never posts the aggregate", async () => { + await withLiveStub(async (stub) => { + const root = liveCopy(stub); + try { + await withGitHubEnv(stub, join(root, "summary.md"), () => + run(["core"], root), + ); + assert.deepEqual( + stub.statuses.map((s) => s.context), + [ + "Vapi Evals / core / squads/dental", + "Vapi Evals / core / squads/dental", + ], + ); + } finally { + rmSync(root, { recursive: true, force: true }); + } + }); +}); + +test("an unaffected change runs nothing and posts a green aggregate", async () => { + await withLiveStub(async (stub) => { + const root = liveCopy(stub); + try { + const git = (...args: string[]) => + execFileSync( + "git", + ["-c", "user.name=t", "-c", "user.email=t@example.com", ...args], + { cwd: root, stdio: "pipe" }, + ); + writeFileSync(join(root, ".gitignore"), ".env.*\ntmp/\nsummary.md\n"); + git("init", "-q", "-b", "main"); + git("add", "-A"); + git("commit", "-qm", "base"); + git("switch", "-qc", "feature"); + writeFileSync(join(root, "NOTES.md"), "unrelated\n"); + git("add", "-A"); + git("commit", "-qm", "docs"); + const result = await withGitHubEnv(stub, join(root, "summary.md"), () => + run(["--all", "--changed-since", "main"], root), + ); + assert.deepEqual( + [ + result.code, + stub.vapiPosts, + stub.statuses.map((s) => [s.context, s.state, s.description]), + ], + [ + 0, + 0, + [["Vapi Evals", "success", "No checks affected by this change"]], + ], + ); + } finally { + rmSync(root, { recursive: true, force: true }); + } + }); +}); + +test("a dry run of an affected PR (fork or Dependabot) posts an error aggregate and sends nothing", async () => { + await withLiveStub(async (stub) => { + const root = liveCopy(stub); + try { + const result = await withGitHubEnv(stub, join(root, "summary.md"), () => + run(["--all", "--dry-run"], root), + ); + assert.deepEqual( + [ + result.code, + stub.vapiPosts, + stub.statuses.map((s) => [s.context, s.state]), + ], + [0, 0, [["Vapi Evals", "error"]]], + ); + } finally { + rmSync(root, { recursive: true, force: true }); + } + }); +}); + +test("an invalid vapi-checks.yml still posts an error aggregate", async () => { + await withLiveStub(async (stub) => { + const root = liveCopy(stub); + try { + writeFileSync(join(root, "vapi-checks.yml"), "version: 2\n"); + const result = await withGitHubEnv(stub, join(root, "summary.md"), () => + run(["--all"], root), + ); + assert.deepEqual( + [ + result.code, + stub.statuses.map((s) => [s.context, s.state, s.description]), + ], + [2, [["Vapi Evals", "error", "vapi-checks.yml is invalid"]]], + ); + } finally { + rmSync(root, { recursive: true, force: true }); + } + }); +}); diff --git a/tests/check-report.test.ts b/tests/check-report.test.ts new file mode 100644 index 0000000..9be4a71 --- /dev/null +++ b/tests/check-report.test.ts @@ -0,0 +1,125 @@ +import assert from "node:assert/strict"; +import test from "node:test"; +import type { CheckJob } from "../src/check-build.ts"; +import { checksConfigParse } from "../src/check-config.ts"; +import { checkReportJson, checkReportMarkdown } from "../src/check-report.ts"; +import type { CheckTargetResult } from "../src/check-run.ts"; + +const check = checksConfigParse( + "version: 1\nchecks:\n core:\n org: acme\n targets: [squads/main, assistants/solo]\n suites: [core]\n", +).checks.core!; + +function job( + index: number, + errors: string[] = [], + warnings: string[] = [], +): CheckJob { + const target = check.targets[index]!; + return { + check, + target, + label: `core / ${target.type}/${target.id}`, + result: { errors, warnings, bytes: 0 }, + }; +} + +const results: CheckTargetResult[] = [ + { + job: job(0, [], ["a warning"]), + outcome: "failed", + reason: "1 of 3 simulations failed", + url: "https://dashboard.vapi.ai/run-1", + counts: { + total: 3, + passed: 2, + failed: 1, + running: 0, + queued: 0, + canceled: 0, + }, + failures: [ + { + item: "Books | calm", + evaluation: "booked", + comparator: "=", + expected: true, + extracted: false, + }, + ], + mockNotices: ["Books | calm: unmocked tool called: sms_send"], + durationMs: 1, + }, + { + job: job(1, ["no file"]), + outcome: "error", + reason: "payload could not be built (1 problem)", + failures: [], + mockNotices: [], + durationMs: 0, + }, +]; + +test("the markdown report has a summary row per target and detail for anything not clean", () => { + assert.equal( + checkReportMarkdown({ + results, + skipped: [ + { check: "other", reason: "not affected by changes since origin/main" }, + ], + dryRun: false, + }), + [ + "## Vapi Evals", + "", + "| Check | Target | Result | Simulations | Run |", + "|---|---|---|---|---|", + "| core | squads/main | ❌ failed | 2/3 | [open](https://dashboard.vapi.ai/run-1) |", + "| core | assistants/solo | 🛑 not run | | |", + "", + "- ⏭️ other: not affected by changes since origin/main", + "", + "### ❌ failed: core / squads/main", + "", + "1 of 3 simulations failed", + "", + "| Simulation | Evaluation | Comparator | Expected | Got | Note |", + "|---|---|---|---|---|---|", + "| Books \\| calm | booked | = | true | false | |", + "", + "- 🧪 Books | calm: unmocked tool called: sms_send", + "- ⚠️ a warning", + "", + "### 🛑 not run: core / assistants/solo", + "", + "payload could not be built (1 problem)", + "", + "- ❌ no file", + "", + ].join("\n"), + ); +}); + +test("the JSON report carries every result field", () => { + const json = checkReportJson({ results, skipped: [], dryRun: true }) as { + dryRun: boolean; + results: Array<{ + check: string; + target: string; + outcome: string; + errors: string[]; + }>; + }; + assert.deepEqual( + [ + json.dryRun, + json.results.map((r) => [r.check, r.target, r.outcome, r.errors]), + ], + [ + true, + [ + ["core", "squads/main", "failed", []], + ["core", "assistants/solo", "error", ["no file"]], + ], + ], + ); +}); diff --git a/tests/check-run.test.ts b/tests/check-run.test.ts new file mode 100644 index 0000000..eb5e8f6 --- /dev/null +++ b/tests/check-run.test.ts @@ -0,0 +1,414 @@ +import assert from "node:assert/strict"; +import { createServer } from "node:http"; +import type { AddressInfo } from "node:net"; +import test from "node:test"; +import type { CheckJob } from "../src/check-build.ts"; +import { checksConfigParse } from "../src/check-config.ts"; +import { MOCK_MARKER } from "../src/check-mocks.ts"; +import type { CheckRunBody } from "../src/check-payload.ts"; +import type { CheckRunArgs, CheckTargetResult } from "../src/check-run.ts"; +import { checkRunAll, MIN_START_MS } from "../src/check-run.ts"; + +// checkRunAll against a stateful stub of the simulations API. The target +// assistant's name picks the stub's behaviour for that run. + +type Behaviour = + "pass" | "fail" | "skipped" | "hang" | "slow" | "unmocked" | "402" | "502"; + +interface StubRun { + id: string; + behaviour: Behaviour; + total: number; + polls: number; + canceled: boolean; + ended: boolean; +} + +interface Stub { + baseUrl: string; + requests: string[]; + userAgents: Set; + maxActive: number; +} + +const DEADLINE_FAR = () => Date.now() + 60 * 60_000; + +function item(run: StubRun, index: number): Record { + const failed = run.behaviour === "fail" && index === 0; + const skipped = run.behaviour === "skipped"; + return { + id: `${run.id}-item-${index}`, + status: failed ? "failed" : "passed", + metadata: { + simulation: { name: `S${index + 1}` }, + // Items echo the scenario, default mocks included, called or not. + scenario: { + toolMocks: [ + { + toolName: "lookup", + result: JSON.stringify({ + error: `${MOCK_MARKER} lookup is not mocked in this scenario`, + }), + }, + ], + }, + call: { + messages: + run.behaviour === "unmocked" + ? [ + { + role: "tool_calls", + toolCalls: [{ id: "call_1", function: { name: "book" } }], + }, + { + role: "tool_call_result", + name: "call_1", + result: JSON.stringify({ + error: `${MOCK_MARKER} book is not mocked in this scenario`, + }), + }, + ] + : [], + }, + }, + results: { + passed: !failed, + evaluations: [ + { + name: "goal-met", + required: true, + comparator: "=", + expectedValue: true, + extractedValue: !failed, + passed: !failed, + isSkipped: skipped, + }, + ], + }, + }; +} + +async function withStub(fn: (stub: Stub) => Promise): Promise { + const runs = new Map(); + const stub: Stub = { + baseUrl: "", + requests: [], + userAgents: new Set(), + maxActive: 0, + }; + const active = () => [...runs.values()].filter((run) => !run.ended).length; + const server = createServer((req, res) => { + let raw = ""; + req.on("data", (chunk) => (raw += chunk)); + req.on("end", () => { + stub.requests.push(`${req.method} ${req.url}`); + stub.userAgents.add(String(req.headers["user-agent"])); + const reply = (status: number, json: unknown) => { + res.writeHead(status, { "Content-Type": "application/json" }); + res.end(JSON.stringify(json)); + }; + const url = req.url ?? ""; + if (req.method === "POST" && url === "/eval/simulation/run") { + const body = JSON.parse(raw) as CheckRunBody; + const behaviour = ( + body.target.type === "assistant" ? body.target.assistant.name : "pass" + ) as Behaviour; + if (behaviour === "402") + return reply(402, { message: "Insufficient credits" }); + if (behaviour === "502") return reply(502, { message: "Bad gateway" }); + const id = `run-${runs.size + 1}`; + const total = body.simulations.length * body.iterations; + runs.set(id, { + id, + behaviour, + total, + polls: 0, + canceled: false, + ended: false, + }); + stub.maxActive = Math.max(stub.maxActive, active()); + return reply(201, { + id, + url: `https://dashboard.vapi.ai/simulations/runs/${id}`, + status: "queued", + simulationRunItemIds: Array.from( + { length: total }, + (_, i) => `${id}-item-${i}`, + ), + }); + } + const match = url.match(/^\/eval\/simulation\/run\/(run-\d+)(\/item)?/); + const run = match ? runs.get(match[1]!) : undefined; + if (!run) return reply(404, { message: "not found" }); + if (req.method === "PATCH") { + if (run.canceled) + return reply(400, { message: "Run has already ended" }); + run.canceled = true; + run.ended = true; + return reply(200, { id: run.id, status: "ended" }); + } + if (match![2]) { + const items = Array.from({ length: run.total }, (_, i) => item(run, i)); + return reply(200, { + results: items, + metadata: { totalItems: items.length }, + }); + } + run.polls++; + const ended = + run.behaviour !== "hang" && + (run.behaviour !== "slow" || run.polls >= 3); + if (ended) run.ended = true; + const failed = run.behaviour === "fail" ? 1 : 0; + return reply(200, { + id: run.id, + status: ended ? "ended" : "running", + itemCounts: ended + ? { + total: run.total, + passed: run.total - failed, + failed, + running: 0, + queued: 0, + canceled: 0, + } + : { + total: run.total, + passed: 0, + failed: 0, + running: run.total, + queued: 0, + canceled: 0, + }, + }); + }); + }); + await new Promise((resolve) => server.listen(0, resolve)); + stub.baseUrl = `http://127.0.0.1:${(server.address() as AddressInfo).port}`; + const log = console.log; + console.log = () => {}; + try { + await fn(stub); + } finally { + console.log = log; + await new Promise((resolve) => server.close(() => resolve())); + } +} + +function job(name: string, extra = "", entries = 2): CheckJob { + const check = checksConfigParse( + `version: 1\nchecks:\n core:\n org: acme\n targets: [assistants/${name.toLowerCase()}]\n suites: [core]\n${extra}`, + ).checks.core!; + const body: CheckRunBody = { + simulations: Array.from({ length: entries }, (_, i) => ({ + type: "simulation" as const, + name: `S${i + 1}`, + scenario: { name: `S${i + 1}` }, + personalityId: "a0000000-0000-4000-8000-000000000001", + })), + target: { type: "assistant", assistant: { name } }, + transport: { provider: "vapi.webchat" }, + iterations: 1, + }; + return { + check, + target: check.targets[0]!, + label: `core / assistants/${name.toLowerCase()}`, + result: { body, errors: [], warnings: [], bytes: 0 }, + }; +} + +function runArgs( + stub: Stub, + jobs: CheckJob[], + extra: Partial = {}, +): CheckRunArgs { + return { + jobs, + connectionFor: () => ({ + token: "t", + baseUrl: stub.baseUrl, + userAgent: "vapi-gitops-check/test", + }), + deadline: DEADLINE_FAR(), + pollIntervalMs: 1, + hydrationMs: 50, + ...extra, + }; +} + +const summary = (result: CheckTargetResult) => [ + result.outcome, + result.reason, + result.url, +]; + +test("a passing run passes, with its link", async () => { + await withStub(async (stub) => { + const [result] = await checkRunAll(runArgs(stub, [job("pass")])); + assert.deepEqual( + [...summary(result!), [...stub.userAgents]], + [ + "passed", + "2 of 2 simulations passed", + "https://dashboard.vapi.ai/simulations/runs/run-1", + ["vapi-gitops-check/test"], + ], + ); + }); +}); + +test("two targets run, and the failing one names the evaluation", async () => { + await withStub(async (stub) => { + const results = await checkRunAll( + runArgs(stub, [job("pass"), job("fail")]), + ); + assert.deepEqual( + [results.map((r) => r.outcome), results[1]!.failures], + [ + ["passed", "failed"], + [ + { + item: "S1", + evaluation: "goal-met", + comparator: "=", + expected: true, + extracted: false, + reason: undefined, + }, + ], + ], + ); + }); +}); + +test("every required evaluation skipped is incomplete, not a pass", async () => { + await withStub(async (stub) => { + const [result] = await checkRunAll(runArgs(stub, [job("skipped")])); + assert.deepEqual(summary(result!).slice(0, 2), [ + "incomplete", + "2 simulations had every required evaluation skipped (S1, S2)", + ]); + }); +}); + +test("a run past its timeout is canceled and incomplete", async () => { + await withStub(async (stub) => { + const [result] = await checkRunAll( + runArgs(stub, [job("hang", " timeoutMinutes: 0.001\n")]), + ); + assert.deepEqual( + [ + result!.outcome, + /timed out after \d+s; run canceled/.test(result!.reason), + stub.requests.filter((r) => r.startsWith("PATCH")), + ], + ["incomplete", true, ["PATCH /eval/simulation/run/run-1"]], + ); + }); +}); + +test("an abort (SIGINT/SIGTERM) cancels in-flight runs and doesn't start new ones", async () => { + await withStub(async (stub) => { + const controller = new AbortController(); + const results = await checkRunAll( + runArgs(stub, [job("hang"), job("pass")], { + signal: controller.signal, + concurrency: 1, + onRunCreated: async () => { + setTimeout(() => controller.abort(), 20); + }, + }), + ); + assert.deepEqual( + [ + results.map((r) => [r.outcome, r.reason]), + stub.requests.filter((r) => r.startsWith("POST")).length, + ], + [ + [ + ["incomplete", "interrupted; run canceled"], + ["incomplete", "not started: interrupted"], + ], + 1, + ], + ); + }); +}); + +test("runs that can't fit in the remaining budget aren't started", async () => { + await withStub(async (stub) => { + const [result] = await checkRunAll( + runArgs(stub, [job("pass")], { + deadline: Date.now() + MIN_START_MS - 60_000, + }), + ); + assert.deepEqual( + [result!.outcome, result!.reason, stub.requests], + ["incomplete", "not started: 4 min of budget left", []], + ); + }); +}); + +test("402 is reported as a billing problem; a 5xx on create is never retried", async () => { + await withStub(async (stub) => { + const results = await checkRunAll(runArgs(stub, [job("402"), job("502")])); + assert.deepEqual( + [ + results.map((r) => [r.outcome, r.reason]), + stub.requests.filter((r) => r.startsWith("POST")).length, + ], + [ + [ + [ + "incomplete", + "billing: the run org can't start simulations (402: Insufficient credits)", + ], + [ + "incomplete", + "API POST /eval/simulation/run failed (502): Bad gateway", + ], + ], + 2, + ], + ); + }); +}); + +test("a payload that didn't build is an error and sends nothing", async () => { + await withStub(async (stub) => { + const broken = job("pass"); + broken.result = { errors: ["boom", "bang"], warnings: [], bytes: 0 }; + const [result] = await checkRunAll(runArgs(stub, [broken])); + assert.deepEqual( + [result!.outcome, result!.reason, stub.requests], + ["error", "payload could not be built (2 problems)", []], + ); + }); +}); + +test("default-mock answers in transcripts are reported as unmocked tool calls; uncalled defaults aren't", async () => { + await withStub(async (stub) => { + const [clean, result] = await checkRunAll( + runArgs(stub, [job("pass"), job("unmocked")]), + ); + assert.deepEqual(clean!.mockNotices, []); + assert.deepEqual(result!.mockNotices, [ + "S1: unmocked tool called: book", + "S2: unmocked tool called: book", + ]); + }); +}); + +test("at most three runs are in flight, and results come back in job order", async () => { + await withStub(async (stub) => { + const jobs = ["slow", "slow", "slow", "slow", "fail"].map((name) => + job(name), + ); + const results = await checkRunAll(runArgs(stub, jobs)); + assert.deepEqual( + [stub.maxActive, results.map((r) => r.outcome)], + [3, ["passed", "passed", "passed", "passed", "failed"]], + ); + }); +}); diff --git a/tests/check-select.test.ts b/tests/check-select.test.ts new file mode 100644 index 0000000..3647e41 --- /dev/null +++ b/tests/check-select.test.ts @@ -0,0 +1,79 @@ +import assert from "node:assert/strict"; +import { execFileSync } from "node:child_process"; +import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { dirname, join } from "node:path"; +import test from "node:test"; +import { checksConfigParse } from "../src/check-config.ts"; +import { changedFilesRead, checkAffectedBy } from "../src/check-select.ts"; + +const { core, ci } = checksConfigParse( + "version: 1\nchecks:\n core:\n org: acme\n targets: [squads/main]\n suites: [core]\n paths: ['prompts/**']\n ci:\n org: acme-staging\n runOrg: acme-ci\n targets: [squads/main]\n suites: [core]\n", +).checks as Record< + string, + ReturnType["checks"][string] +>; + +test("a check is affected by its org's files, its run org, its state, the engine and its own paths", () => { + const cases: Array<[string, string | undefined, string | undefined]> = [ + [ + "resources/acme/assistants/a.md", + "resources/acme/assistants/a.md", + undefined, + ], + [ + "resources/acme-ci/tools/t.yml", + undefined, + "resources/acme-ci/tools/t.yml", + ], + [".vapi-state.acme.json", ".vapi-state.acme.json", undefined], + ["vapi-checks.yml", "vapi-checks.yml", "vapi-checks.yml"], + ["src/push.ts", "src/push.ts", "src/push.ts"], + ["package-lock.json", "package-lock.json", "package-lock.json"], + ["prompts/shared/tone.md", "prompts/shared/tone.md", undefined], + ["docs/readme.md", undefined, undefined], + ["resources/acme-other/assistants/a.md", undefined, undefined], + ]; + assert.deepEqual( + cases.map(([file]) => [ + checkAffectedBy(core!, [file]), + checkAffectedBy(ci!, [file]), + ]), + cases.map(([, a, b]) => [a, b]), + ); +}); + +test("changed files come from the merge base, so commits that landed on the base don't count", () => { + const root = mkdtempSync(join(tmpdir(), "check-select-")); + const git = (...args: string[]) => + execFileSync( + "git", + ["-c", "user.name=t", "-c", "user.email=t@example.com", ...args], + { cwd: root, stdio: "pipe" }, + ); + const write = (path: string) => { + mkdirSync(dirname(join(root, path)), { recursive: true }); + writeFileSync(join(root, path), path); + }; + try { + git("init", "-q", "-b", "main"); + write("README.md"); + git("add", "-A"); + git("commit", "-qm", "base"); + git("switch", "-qc", "feature"); + write("resources/acme/assistants/a.md"); + git("add", "-A"); + git("commit", "-qm", "feature"); + git("switch", "-q", "main"); + write("src/push.ts"); + git("add", "-A"); + git("commit", "-qm", "main moved on"); + git("switch", "-q", "feature"); + assert.deepEqual( + [changedFilesRead(root, "main"), changedFilesRead(root, "no-such-ref")], + [["resources/acme/assistants/a.md"], undefined], + ); + } finally { + rmSync(root, { recursive: true, force: true }); + } +}); diff --git a/tests/check-status.test.ts b/tests/check-status.test.ts new file mode 100644 index 0000000..d4bbfda --- /dev/null +++ b/tests/check-status.test.ts @@ -0,0 +1,130 @@ +import assert from "node:assert/strict"; +import { createServer } from "node:http"; +import type { AddressInfo } from "node:net"; +import test from "node:test"; +import { + commitStateWorst, + commitStatusPost, + githubStatusEnvRead, + targetContext, +} from "../src/check-status.ts"; + +async function withGitHub( + status: number, + fn: ( + apiUrl: string, + seen: Array<{ url: string; auth: string; body: unknown }>, + ) => Promise, +): Promise { + const seen: Array<{ url: string; auth: string; body: unknown }> = []; + const server = createServer((req, res) => { + let raw = ""; + req.on("data", (chunk) => (raw += chunk)); + req.on("end", () => { + seen.push({ + url: `${req.method} ${req.url}`, + auth: String(req.headers.authorization), + body: JSON.parse(raw), + }); + res.writeHead(status, { "Content-Type": "application/json" }); + res.end("{}"); + }); + }); + await new Promise((resolve) => server.listen(0, resolve)); + const warn = console.warn; + console.warn = () => {}; + try { + await fn( + `http://127.0.0.1:${(server.address() as AddressInfo).port}`, + seen, + ); + } finally { + console.warn = warn; + await new Promise((resolve) => server.close(() => resolve())); + } +} + +test("githubStatusEnvRead needs the token, repository and head SHA", () => { + assert.deepEqual( + [ + githubStatusEnvRead({ + GITHUB_TOKEN: "t", + GITHUB_REPOSITORY: "acme/gitops", + }), + githubStatusEnvRead({ + GITHUB_TOKEN: "t", + GITHUB_REPOSITORY: "acme/gitops", + HEAD_SHA: "abc", + GITHUB_RUN_ID: "42", + }), + ], + [ + undefined, + { + token: "t", + repo: "acme/gitops", + sha: "abc", + apiUrl: "https://api.github.com", + runUrl: "https://github.com/acme/gitops/actions/runs/42", + }, + ], + ); +}); + +test("statuses post to the head SHA with the run link, and long descriptions are cut to 140", async () => { + await withGitHub(201, async (apiUrl, seen) => { + const ok = await commitStatusPost( + { token: "t", repo: "acme/gitops", sha: "abc", apiUrl }, + { + context: targetContext("core", "squads/main"), + state: "failure", + description: "x".repeat(200), + targetUrl: "https://run", + }, + ); + const body = seen[0]!.body as { description: string }; + assert.deepEqual( + [ + ok, + seen[0]!.url, + seen[0]!.auth, + { ...body, description: body.description.length }, + ], + [ + true, + "POST /repos/acme/gitops/statuses/abc", + "Bearer t", + { + context: "Vapi Evals / core / squads/main", + state: "failure", + description: 140, + target_url: "https://run", + }, + ], + ); + }); +}); + +test("a rejected status (read-only token) only warns", async () => { + await withGitHub(403, async (apiUrl) => { + assert.equal( + await commitStatusPost( + { token: "t", repo: "acme/gitops", sha: "abc", apiUrl }, + { context: "Vapi Evals", state: "success", description: "ok" }, + ), + false, + ); + }); +}); + +test("the worst state wins: error > failure > pending > success", () => { + assert.deepEqual( + [ + commitStateWorst([]), + commitStateWorst(["success", "pending"]), + commitStateWorst(["failure", "pending", "success"]), + commitStateWorst(["failure", "error"]), + ], + ["success", "pending", "failure", "error"], + ); +});