UNPKG

jev-evals

Version:

Rubric-based eval harness for LLM/agent outputs, backed by the typesafe-ai/jev evaluation model via Vercel AI Gateway. Cheap enough (~$0.04/1M input tokens, one round trip per case) to run on every PR.

678 lines (661 loc) • 22.8 kB
#!/usr/bin/env node // src/cli/args.ts function parseArgs(argv) { const positionals = []; const flags = {}; let i = 0; while (i < argv.length) { const arg = argv[i]; if (arg.startsWith("--")) { const body = arg.slice(2); const eq = body.indexOf("="); if (eq >= 0) { flags[body.slice(0, eq)] = body.slice(eq + 1); i++; continue; } const next = argv[i + 1]; if (next !== void 0 && !next.startsWith("--")) { flags[body] = next; i += 2; } else { flags[body] = true; i += 1; } continue; } if (arg.startsWith("-") && arg.length === 2) { flags[arg.slice(1)] = true; i += 1; continue; } positionals.push(arg); i += 1; } const command = positionals.shift(); return { command, positionals, flags }; } function flagString(flags, name) { const v = flags[name]; if (v === void 0) return void 0; if (typeof v === "boolean") return void 0; return v; } function flagNumber(flags, name) { const v = flagString(flags, name); if (v === void 0) return void 0; const n = Number(v); if (Number.isNaN(n)) throw new Error(`--${name} must be a number, got "${v}"`); return n; } function flagBoolean(flags, name) { const v = flags[name]; if (v === void 0) return false; if (typeof v === "boolean") return v; return v === "true" || v === "1"; } // src/cli/run.ts import { writeFile } from "fs/promises"; import { pathToFileURL } from "url"; import { resolve } from "path"; // src/runEval.ts import { experimental_evaluate as evaluate } from "ai"; // src/concurrency.ts async function mapWithConcurrency(items, concurrency, fn) { if (items.length === 0) return []; const limit = Math.max(1, Math.floor(concurrency)); const results = new Array(items.length); let nextIndex = 0; async function worker() { while (true) { const i = nextIndex++; if (i >= items.length) return; results[i] = await fn(items[i], i); } } const workers = Array.from({ length: Math.min(limit, items.length) }, () => worker()); await Promise.all(workers); return results; } // src/cost.ts var DEFAULT_PRICING = { inputPerMillion: 0.04, outputPerMillion: 0.04 }; function estimateCostUsd(usage, pricing = DEFAULT_PRICING) { const inputCost = usage.inputTokens / 1e6 * pricing.inputPerMillion; const outputCost = usage.outputTokens / 1e6 * pricing.outputPerMillion; return inputCost + outputCost; } function sumUsage(items) { let inputTokens = 0; let outputTokens = 0; let totalTokens = 0; for (const u of items) { if (!u) continue; inputTokens += u.inputTokens ?? 0; outputTokens += u.outputTokens ?? 0; totalTokens += u.totalTokens ?? (u.inputTokens ?? 0) + (u.outputTokens ?? 0); } return { inputTokens, outputTokens, totalTokens }; } // src/runEval.ts var DEFAULT_MAX_QUESTIONS_PER_CALL = 40; var DEFAULT_CONCURRENCY = 5; var AuthError = class extends Error { constructor() { super( "No credentials found for the AI Gateway. Set one of:\n - AI_GATEWAY_API_KEY (an AI Gateway API key)\n - VERCEL_OIDC_TOKEN (run `vercel env pull` in a Vercel project; lasts 12h)\njev-evals does not implement auth itself \u2014 it relies on the `ai` SDK picking one of these up. See the README for both paths." ); this.name = "AuthError"; } }; function checkAuth(skip) { if (skip) return; if (process.env["AI_GATEWAY_API_KEY"] || process.env["VERCEL_OIDC_TOKEN"]) return; throw new AuthError(); } function chunkEntries(entries, size) { if (entries.length <= size) return [entries]; const chunks = []; for (let i = 0; i < entries.length; i += size) { chunks.push(entries.slice(i, i + size)); } return chunks; } function choiceOrdinal(rubric, choice) { if (rubric.type !== "choice") return 0; const keys = Object.keys(rubric.criteria); const idx = keys.indexOf(choice); if (idx < 0 || keys.length <= 1) return 0; return idx / (keys.length - 1); } function numericValueOf(rubric, answer) { switch (answer.type) { case "boolean": return answer.probability; case "score": return answer.score; case "choice": return choiceOrdinal(rubric, answer.choice); } } function evaluateThreshold(rubric, answer, threshold) { if (answer.type === "choice") { const accepted = Array.isArray(threshold) ? threshold : [threshold]; return accepted.includes(answer.choice); } const numericThreshold = typeof threshold === "number" ? threshold : Number(threshold); const value = answer.type === "boolean" ? answer.probability : answer.score; return value >= numericThreshold; } async function resolveOutput(evalCase, suite) { if (evalCase.output !== void 0) return evalCase.output; const generate = evalCase.generate ?? suite.generate; if (!generate) { throw new Error(`Case "${evalCase.id}" has no \`output\` and no \`generate\` function.`); } return await generate(evalCase.input, evalCase); } async function evaluateOneCase(args) { const { suite, evalCase, model, maxQuestionsPerCall, abortSignal, maxRetries } = args; let output; try { output = await resolveOutput(evalCase, suite); } catch (err) { return { id: evalCase.id, input: evalCase.input, output: void 0, expected: evalCase.expected, metadata: evalCase.metadata, rubrics: {}, passed: false, error: err instanceof Error ? err.message : String(err) }; } const state = { input: evalCase.input, output, ...evalCase.expected !== void 0 ? { expected: evalCase.expected } : {} }; const rubricEntries = Object.entries(suite.rubrics); const chunks = chunkEntries(rubricEntries, maxQuestionsPerCall); try { const chunkResults = await Promise.all( chunks.map( (chunk) => evaluate({ model, state, questions: Object.fromEntries(chunk), ...abortSignal ? { abortSignal } : {}, ...maxRetries !== void 0 ? { maxRetries } : {} }) ) ); const answers = {}; const warnings = []; for (const result of chunkResults) { Object.assign(answers, result.answers); if (result.warnings) warnings.push(...result.warnings); } const usage = sumUsage(chunkResults.map((r) => r.usage)); const rubrics = {}; for (const [id, rubric] of rubricEntries) { const answer = answers[id]; if (!answer) continue; const numericValue = numericValueOf(rubric, answer); const threshold = suite.thresholds?.[id]; const passed2 = threshold === void 0 ? null : evaluateThreshold(rubric, answer, threshold); rubrics[id] = { rubricId: id, type: rubric.type, answer, numericValue, passed: passed2, ...threshold !== void 0 ? { threshold } : {} }; } const passed = Object.values(rubrics).every((r) => r.passed !== false); return { id: evalCase.id, input: evalCase.input, output, expected: evalCase.expected, metadata: evalCase.metadata, rubrics, passed, usage, warnings: warnings.length > 0 ? warnings : void 0 }; } catch (err) { return { id: evalCase.id, input: evalCase.input, output, expected: evalCase.expected, metadata: evalCase.metadata, rubrics: {}, passed: false, error: err instanceof Error ? err.message : String(err) }; } } function aggregate(suite, perCase) { const out = {}; for (const [id, rubric] of Object.entries(suite.rubrics)) { const values = []; let passCount = 0; let thresholdCount = 0; for (const c of perCase) { const r = c.rubrics[id]; if (!r) continue; values.push(r.numericValue); if (r.passed !== null) { thresholdCount++; if (r.passed) passCount++; } } out[id] = { rubricId: id, type: rubric.type, mean: values.length > 0 ? values.reduce((a, b) => a + b, 0) / values.length : NaN, min: values.length > 0 ? Math.min(...values) : NaN, max: values.length > 0 ? Math.max(...values) : NaN, passRate: thresholdCount > 0 ? passCount / thresholdCount : null, count: values.length }; } return out; } async function runEval(suite, options = {}) { checkAuth(options.skipAuthCheck); const model = options.model ?? "typesafe-ai/jev"; const concurrency = options.concurrency ?? DEFAULT_CONCURRENCY; const maxQuestionsPerCall = options.maxQuestionsPerCall ?? suite.maxQuestionsPerCall ?? DEFAULT_MAX_QUESTIONS_PER_CALL; const pricing = { ...DEFAULT_PRICING, ...options.pricing }; const start = Date.now(); const total = suite.cases.length; let completed = 0; const perCase = await mapWithConcurrency(suite.cases, concurrency, async (evalCase) => { const result = await evaluateOneCase({ suite, evalCase, model, maxQuestionsPerCall, ...options.abortSignal ? { abortSignal: options.abortSignal } : {}, ...options.maxRetries !== void 0 ? { maxRetries: options.maxRetries } : {} }); completed++; options.onCaseComplete?.(result, completed, total); return result; }); const ms = Date.now() - start; const passed = perCase.filter((c) => c.passed && !c.error).length; const failed = total - passed; const usage = sumUsage(perCase.map((c) => c.usage)); const rubricTypes = {}; for (const [id, rubric] of Object.entries(suite.rubrics)) { rubricTypes[id] = rubric.type; } return { suite: suite.name, passed, failed, total, success: failed === 0, perCase, perRubric: aggregate(suite, perCase), usage, estimatedCostUsd: estimateCostUsd(usage, pricing), ms, timestamp: (/* @__PURE__ */ new Date()).toISOString(), rubricTypes }; } // src/compareRuns.ts var DEFAULT_TOLERANCE = 0.05; function compareRuns(baseline, current, options = {}) { const tolerance = options.tolerance ?? DEFAULT_TOLERANCE; const rubricIds = /* @__PURE__ */ new Set([ ...Object.keys(baseline.perRubric), ...Object.keys(current.perRubric) ]); const deltas = []; for (const id of rubricIds) { const base = baseline.perRubric[id]; const cur = current.perRubric[id]; if (base && !cur) { deltas.push({ rubricId: id, type: base.type, baselineMean: base.mean, currentMean: null, diff: null, status: "removed" }); continue; } if (cur && !base) { deltas.push({ rubricId: id, type: cur.type, baselineMean: null, currentMean: cur.mean, diff: null, status: "new" }); continue; } if (!base || !cur) continue; const diff = cur.mean - base.mean; let status; const epsilon = 1e-9; if (Math.abs(diff) <= tolerance + epsilon) { status = "unchanged"; } else if (diff < 0) { status = "regressed"; } else { status = "improved"; } deltas.push({ rubricId: id, type: cur.type, baselineMean: base.mean, currentMean: cur.mean, diff, status }); } const order = { regressed: 0, improved: 1, new: 2, removed: 3, unchanged: 4 }; deltas.sort((a, b) => order[a.status] - order[b.status] || a.rubricId.localeCompare(b.rubricId)); return { baselineSuite: baseline.suite, currentSuite: current.suite, tolerance, deltas, regressed: deltas.filter((d) => d.status === "regressed"), improved: deltas.filter((d) => d.status === "improved"), unchanged: deltas.filter((d) => d.status === "unchanged"), hasRegression: deltas.some((d) => d.status === "regressed") }; } // src/cli/format.ts function formatUsd(n) { if (n < 0.01) return `$${n.toFixed(5)}`; return `$${n.toFixed(4)}`; } function formatNum(n) { if (Number.isNaN(n)) return "n/a"; return n.toFixed(3); } function formatPct(n) { if (n === null || Number.isNaN(n)) return "n/a"; return `${(n * 100).toFixed(0)}%`; } function printRunSummary(run) { const lines = []; lines.push(""); lines.push(`jev-evals: ${run.suite}`); lines.push("-".repeat(40)); lines.push(`cases: ${run.passed}/${run.total} passed`); lines.push(`time: ${run.ms}ms`); lines.push( `usage: ${run.usage.inputTokens} in / ${run.usage.outputTokens} out / ${run.usage.totalTokens} total tokens` ); lines.push(`est. cost: ${formatUsd(run.estimatedCostUsd)}`); lines.push(""); lines.push("rubric type mean pass% n"); for (const [id, agg] of Object.entries(run.perRubric)) { lines.push( `${id.padEnd(20)}${agg.type.padEnd(10)}${formatNum(agg.mean).padEnd(9)}${formatPct(agg.passRate).padEnd(8)}${agg.count}` ); } const failing = run.perCase.filter((c) => !c.passed || c.error); if (failing.length > 0) { lines.push(""); lines.push(`failing cases (${failing.length}):`); for (const c of failing) { if (c.error) { lines.push(` - ${c.id}: ERROR: ${c.error}`); } else { const failedRubrics = Object.values(c.rubrics).filter((r) => r.passed === false).map((r) => r.rubricId); lines.push(` - ${c.id}: failed [${failedRubrics.join(", ")}]`); } } } lines.push(""); console.log(lines.join("\n")); } function printCompareSummary(diff) { const lines = []; lines.push(""); lines.push(`jev-evals compare: ${diff.baselineSuite} -> ${diff.currentSuite}`); lines.push("-".repeat(40)); lines.push(`tolerance: ${diff.tolerance}`); lines.push(""); lines.push("rubric status baseline current diff"); for (const d of diff.deltas) { lines.push( `${d.rubricId.padEnd(20)}${d.status.padEnd(12)}${fmtOrNa(d.baselineMean).padEnd(11)}${fmtOrNa(d.currentMean).padEnd(11)}${fmtDiff(d.diff)}` ); } lines.push(""); lines.push(diff.hasRegression ? "RESULT: regression detected" : "RESULT: no regression"); lines.push(""); console.log(lines.join("\n")); } function fmtOrNa(n) { return n === null ? "n/a" : formatNum(n); } function fmtDiff(n) { if (n === null) return "n/a"; const sign = n > 0 ? "+" : ""; return `${sign}${n.toFixed(3)}`; } function buildMarkdownSummary(run, diff) { const lines = []; const badge = run.success ? "\u2705" : "\u274C"; lines.push(`### ${badge} jev-evals: \`${run.suite}\``); lines.push(""); lines.push(`**${run.passed}/${run.total}** cases passed \xB7 **${formatUsd(run.estimatedCostUsd)}** (${run.usage.totalTokens} tokens) \xB7 ${run.ms}ms`); lines.push(""); lines.push("| Rubric | Type | Mean | Pass % |" + (diff ? " \u0394 vs baseline |" : "")); lines.push("|---|---|---|---|" + (diff ? "---|" : "")); for (const [id, agg] of Object.entries(run.perRubric)) { const deltaCell = diff ? renderDeltaCell(diff, id) : ""; lines.push( `| \`${id}\` | ${agg.type} | ${formatNum(agg.mean)} | ${formatPct(agg.passRate)} |` + (diff ? ` ${deltaCell} |` : "") ); } const failing = run.perCase.filter((c) => !c.passed || c.error); if (failing.length > 0) { lines.push(""); lines.push(`<details><summary>${failing.length} failing case(s)</summary>`); lines.push(""); for (const c of failing) { if (c.error) { lines.push(`- \`${c.id}\`: **ERROR** \u2014 ${c.error}`); } else { const failedRubrics = Object.values(c.rubrics).filter((r) => r.passed === false).map((r) => `\`${r.rubricId}\` (${formatNum(r.numericValue)})`); lines.push(`- \`${c.id}\`: ${failedRubrics.join(", ")}`); } } lines.push(""); lines.push("</details>"); } if (diff && diff.hasRegression) { lines.push(""); lines.push(`> \u26A0\uFE0F Regression detected in: ${diff.regressed.map((d) => `\`${d.rubricId}\``).join(", ")}`); } return lines.join("\n"); } function renderDeltaCell(diff, rubricId) { const d = diff.deltas.find((x) => x.rubricId === rubricId); if (!d || d.diff === null) return "n/a"; const icon = d.status === "regressed" ? "\u{1F53B}" : d.status === "improved" ? "\u{1F53A}" : "\u25AA\uFE0F"; const sign = d.diff > 0 ? "+" : ""; return `${icon} ${sign}${d.diff.toFixed(3)}`; } // src/cli/io.ts import { readFile } from "fs/promises"; async function readJsonFile(path) { const raw = await readFile(path, "utf8"); return JSON.parse(raw); } // src/cli/run.ts async function loadSuite(path) { const abs = resolve(process.cwd(), path); const mod = await import(pathToFileURL(abs).href); const candidate = mod.default ?? mod.suite ?? mod; if (!candidate || typeof candidate !== "object" || !("cases" in candidate)) { throw new Error( `"${path}" does not export an EvalSuite. Export it as \`export default suite\` or \`export const suite = ...\`.` ); } return candidate; } async function runCommand(args) { const suitePath = args.positionals[0]; if (!suitePath) { console.error("Usage: jev-evals run <suite-file> [options]"); return 1; } const suite = await loadSuite(suitePath); const concurrency = flagNumber(args.flags, "concurrency"); const maxQuestionsPerCall = flagNumber(args.flags, "max-questions-per-call"); const model = flagString(args.flags, "model"); const maxRetries = flagNumber(args.flags, "max-retries"); const inputPrice = flagNumber(args.flags, "input-price"); const outputPrice = flagNumber(args.flags, "output-price"); const asJson = flagBoolean(args.flags, "json"); const savePath = flagString(args.flags, "save"); const baselinePath = flagString(args.flags, "baseline"); const tolerance = flagNumber(args.flags, "tolerance"); const markdownPath = flagString(args.flags, "markdown"); const run = await runEval(suite, { ...concurrency !== void 0 ? { concurrency } : {}, ...maxQuestionsPerCall !== void 0 ? { maxQuestionsPerCall } : {}, ...model !== void 0 ? { model } : {}, ...maxRetries !== void 0 ? { maxRetries } : {}, ...inputPrice !== void 0 || outputPrice !== void 0 ? { pricing: { ...inputPrice !== void 0 ? { inputPerMillion: inputPrice } : {}, ...outputPrice !== void 0 ? { outputPerMillion: outputPrice } : {} } } : {} }); if (savePath) { await writeFile(resolve(process.cwd(), savePath), JSON.stringify(run, null, 2) + "\n", "utf8"); } let diff; if (baselinePath) { const baseline = await readJsonFile(resolve(process.cwd(), baselinePath)); diff = compareRuns(baseline, run, tolerance !== void 0 ? { tolerance } : {}); } if (markdownPath) { await writeFile(resolve(process.cwd(), markdownPath), buildMarkdownSummary(run, diff) + "\n", "utf8"); } if (asJson) { console.log(JSON.stringify(diff ? { run, compare: diff } : { run }, null, 2)); } else { printRunSummary(run); if (diff) printCompareSummary(diff); } const failed = !run.success || (diff ? diff.hasRegression : false); return failed ? 1 : 0; } // src/cli/compare.ts import { writeFile as writeFile2 } from "fs/promises"; import { resolve as resolve2 } from "path"; async function compareCommand(args) { const [baselinePath, currentPath] = args.positionals; if (!baselinePath || !currentPath) { console.error("Usage: jev-evals compare <baseline.json> <current.json> [options]"); return 1; } const baseline = await readJsonFile(resolve2(process.cwd(), baselinePath)); const current = await readJsonFile(resolve2(process.cwd(), currentPath)); const tolerance = flagNumber(args.flags, "tolerance"); const asJson = flagBoolean(args.flags, "json"); const markdownPath = args.flags["markdown"]; const diff = compareRuns(baseline, current, tolerance !== void 0 ? { tolerance } : {}); if (typeof markdownPath === "string") { await writeFile2(resolve2(process.cwd(), markdownPath), buildMarkdownSummary(current, diff) + "\n", "utf8"); } if (asJson) { console.log(JSON.stringify(diff, null, 2)); } else { printCompareSummary(diff); } return diff.hasRegression ? 1 : 0; } // src/cli.ts var HELP = `jev-evals \u2014 rubric-based evals for LLM/agent outputs, backed by typesafe-ai/jev. Usage: jev-evals run <suite-file> [options] Run a suite and print/save results. jev-evals compare <baseline.json> <current.json> [options] Compare two saved runs. Options for "run": --save <path> Write the full RunResult as JSON to <path>. --baseline <path> Compare this run against a previously saved RunResult JSON and report regressions. --tolerance <n> Regression tolerance passed to compareRuns (only with --baseline). Default 0.05. --concurrency <n> Max cases evaluated in parallel. Default 5. --max-questions-per-call <n> Override the suite's maxQuestionsPerCall. --model <id> Model id passed to evaluate(). Default typesafe-ai/jev. --max-retries <n> Passed through to evaluate(). --input-price <usd/1M> Override input token pricing for cost estimate. --output-price <usd/1M> Override output token pricing for cost estimate. --markdown <path> Write a PR-comment-ready markdown summary to <path>. --json Print machine-readable JSON instead of a table. Options for "compare": --tolerance <n> Regression tolerance. Default 0.05. --markdown <path> Write a markdown summary to <path>. --json Print machine-readable JSON instead of a table. Exit code is non-zero when any case fails its thresholds, or (with --baseline / compare) when a regression is detected \u2014 suitable for use as a CI gate. Auth (not implemented by this package \u2014 see README): AI_GATEWAY_API_KEY an AI Gateway API key, or VERCEL_OIDC_TOKEN from \`vercel env pull\` (12h lifetime) `; async function main() { const args = parseArgs(process.argv.slice(2)); const helpRequested = Boolean(args.flags["help"] || args.flags["h"]); if (helpRequested || !args.command) { console.log(HELP); process.exitCode = helpRequested ? 0 : 1; return; } if (args.flags["version"]) { console.log("jev-evals 0.1.0"); return; } switch (args.command) { case "run": process.exitCode = await runCommand(args); return; case "compare": process.exitCode = await compareCommand(args); return; default: console.error(`Unknown command "${args.command}". `); console.log(HELP); process.exitCode = 1; } } main().catch((err) => { console.error(err instanceof Error ? err.stack ?? err.message : err); process.exitCode = 1; }); //# sourceMappingURL=cli.js.map