jev-evals
Version:
Rubric-based eval harness for LLM/agent outputs, backed by the typesafe-ai/jev evaluation model via Vercel AI Gateway. Cheap enough (~$0.04/1M input tokens, one round trip per case) to run on every PR.
678 lines (661 loc) • 22.8 kB
JavaScript
// src/cli/args.ts
function parseArgs(argv) {
const positionals = [];
const flags = {};
let i = 0;
while (i < argv.length) {
const arg = argv[i];
if (arg.startsWith("--")) {
const body = arg.slice(2);
const eq = body.indexOf("=");
if (eq >= 0) {
flags[body.slice(0, eq)] = body.slice(eq + 1);
i++;
continue;
}
const next = argv[i + 1];
if (next !== void 0 && !next.startsWith("--")) {
flags[body] = next;
i += 2;
} else {
flags[body] = true;
i += 1;
}
continue;
}
if (arg.startsWith("-") && arg.length === 2) {
flags[arg.slice(1)] = true;
i += 1;
continue;
}
positionals.push(arg);
i += 1;
}
const command = positionals.shift();
return { command, positionals, flags };
}
function flagString(flags, name) {
const v = flags[name];
if (v === void 0) return void 0;
if (typeof v === "boolean") return void 0;
return v;
}
function flagNumber(flags, name) {
const v = flagString(flags, name);
if (v === void 0) return void 0;
const n = Number(v);
if (Number.isNaN(n)) throw new Error(`--${name} must be a number, got "${v}"`);
return n;
}
function flagBoolean(flags, name) {
const v = flags[name];
if (v === void 0) return false;
if (typeof v === "boolean") return v;
return v === "true" || v === "1";
}
// src/cli/run.ts
import { writeFile } from "fs/promises";
import { pathToFileURL } from "url";
import { resolve } from "path";
// src/runEval.ts
import { experimental_evaluate as evaluate } from "ai";
// src/concurrency.ts
async function mapWithConcurrency(items, concurrency, fn) {
if (items.length === 0) return [];
const limit = Math.max(1, Math.floor(concurrency));
const results = new Array(items.length);
let nextIndex = 0;
async function worker() {
while (true) {
const i = nextIndex++;
if (i >= items.length) return;
results[i] = await fn(items[i], i);
}
}
const workers = Array.from({ length: Math.min(limit, items.length) }, () => worker());
await Promise.all(workers);
return results;
}
// src/cost.ts
var DEFAULT_PRICING = {
inputPerMillion: 0.04,
outputPerMillion: 0.04
};
function estimateCostUsd(usage, pricing = DEFAULT_PRICING) {
const inputCost = usage.inputTokens / 1e6 * pricing.inputPerMillion;
const outputCost = usage.outputTokens / 1e6 * pricing.outputPerMillion;
return inputCost + outputCost;
}
function sumUsage(items) {
let inputTokens = 0;
let outputTokens = 0;
let totalTokens = 0;
for (const u of items) {
if (!u) continue;
inputTokens += u.inputTokens ?? 0;
outputTokens += u.outputTokens ?? 0;
totalTokens += u.totalTokens ?? (u.inputTokens ?? 0) + (u.outputTokens ?? 0);
}
return { inputTokens, outputTokens, totalTokens };
}
// src/runEval.ts
var DEFAULT_MAX_QUESTIONS_PER_CALL = 40;
var DEFAULT_CONCURRENCY = 5;
var AuthError = class extends Error {
constructor() {
super(
"No credentials found for the AI Gateway. Set one of:\n - AI_GATEWAY_API_KEY (an AI Gateway API key)\n - VERCEL_OIDC_TOKEN (run `vercel env pull` in a Vercel project; lasts 12h)\njev-evals does not implement auth itself \u2014 it relies on the `ai` SDK picking one of these up. See the README for both paths."
);
this.name = "AuthError";
}
};
function checkAuth(skip) {
if (skip) return;
if (process.env["AI_GATEWAY_API_KEY"] || process.env["VERCEL_OIDC_TOKEN"]) return;
throw new AuthError();
}
function chunkEntries(entries, size) {
if (entries.length <= size) return [entries];
const chunks = [];
for (let i = 0; i < entries.length; i += size) {
chunks.push(entries.slice(i, i + size));
}
return chunks;
}
function choiceOrdinal(rubric, choice) {
if (rubric.type !== "choice") return 0;
const keys = Object.keys(rubric.criteria);
const idx = keys.indexOf(choice);
if (idx < 0 || keys.length <= 1) return 0;
return idx / (keys.length - 1);
}
function numericValueOf(rubric, answer) {
switch (answer.type) {
case "boolean":
return answer.probability;
case "score":
return answer.score;
case "choice":
return choiceOrdinal(rubric, answer.choice);
}
}
function evaluateThreshold(rubric, answer, threshold) {
if (answer.type === "choice") {
const accepted = Array.isArray(threshold) ? threshold : [threshold];
return accepted.includes(answer.choice);
}
const numericThreshold = typeof threshold === "number" ? threshold : Number(threshold);
const value = answer.type === "boolean" ? answer.probability : answer.score;
return value >= numericThreshold;
}
async function resolveOutput(evalCase, suite) {
if (evalCase.output !== void 0) return evalCase.output;
const generate = evalCase.generate ?? suite.generate;
if (!generate) {
throw new Error(`Case "${evalCase.id}" has no \`output\` and no \`generate\` function.`);
}
return await generate(evalCase.input, evalCase);
}
async function evaluateOneCase(args) {
const { suite, evalCase, model, maxQuestionsPerCall, abortSignal, maxRetries } = args;
let output;
try {
output = await resolveOutput(evalCase, suite);
} catch (err) {
return {
id: evalCase.id,
input: evalCase.input,
output: void 0,
expected: evalCase.expected,
metadata: evalCase.metadata,
rubrics: {},
passed: false,
error: err instanceof Error ? err.message : String(err)
};
}
const state = {
input: evalCase.input,
output,
...evalCase.expected !== void 0 ? { expected: evalCase.expected } : {}
};
const rubricEntries = Object.entries(suite.rubrics);
const chunks = chunkEntries(rubricEntries, maxQuestionsPerCall);
try {
const chunkResults = await Promise.all(
chunks.map(
(chunk) => evaluate({
model,
state,
questions: Object.fromEntries(chunk),
...abortSignal ? { abortSignal } : {},
...maxRetries !== void 0 ? { maxRetries } : {}
})
)
);
const answers = {};
const warnings = [];
for (const result of chunkResults) {
Object.assign(answers, result.answers);
if (result.warnings) warnings.push(...result.warnings);
}
const usage = sumUsage(chunkResults.map((r) => r.usage));
const rubrics = {};
for (const [id, rubric] of rubricEntries) {
const answer = answers[id];
if (!answer) continue;
const numericValue = numericValueOf(rubric, answer);
const threshold = suite.thresholds?.[id];
const passed2 = threshold === void 0 ? null : evaluateThreshold(rubric, answer, threshold);
rubrics[id] = {
rubricId: id,
type: rubric.type,
answer,
numericValue,
passed: passed2,
...threshold !== void 0 ? { threshold } : {}
};
}
const passed = Object.values(rubrics).every((r) => r.passed !== false);
return {
id: evalCase.id,
input: evalCase.input,
output,
expected: evalCase.expected,
metadata: evalCase.metadata,
rubrics,
passed,
usage,
warnings: warnings.length > 0 ? warnings : void 0
};
} catch (err) {
return {
id: evalCase.id,
input: evalCase.input,
output,
expected: evalCase.expected,
metadata: evalCase.metadata,
rubrics: {},
passed: false,
error: err instanceof Error ? err.message : String(err)
};
}
}
function aggregate(suite, perCase) {
const out = {};
for (const [id, rubric] of Object.entries(suite.rubrics)) {
const values = [];
let passCount = 0;
let thresholdCount = 0;
for (const c of perCase) {
const r = c.rubrics[id];
if (!r) continue;
values.push(r.numericValue);
if (r.passed !== null) {
thresholdCount++;
if (r.passed) passCount++;
}
}
out[id] = {
rubricId: id,
type: rubric.type,
mean: values.length > 0 ? values.reduce((a, b) => a + b, 0) / values.length : NaN,
min: values.length > 0 ? Math.min(...values) : NaN,
max: values.length > 0 ? Math.max(...values) : NaN,
passRate: thresholdCount > 0 ? passCount / thresholdCount : null,
count: values.length
};
}
return out;
}
async function runEval(suite, options = {}) {
checkAuth(options.skipAuthCheck);
const model = options.model ?? "typesafe-ai/jev";
const concurrency = options.concurrency ?? DEFAULT_CONCURRENCY;
const maxQuestionsPerCall = options.maxQuestionsPerCall ?? suite.maxQuestionsPerCall ?? DEFAULT_MAX_QUESTIONS_PER_CALL;
const pricing = { ...DEFAULT_PRICING, ...options.pricing };
const start = Date.now();
const total = suite.cases.length;
let completed = 0;
const perCase = await mapWithConcurrency(suite.cases, concurrency, async (evalCase) => {
const result = await evaluateOneCase({
suite,
evalCase,
model,
maxQuestionsPerCall,
...options.abortSignal ? { abortSignal: options.abortSignal } : {},
...options.maxRetries !== void 0 ? { maxRetries: options.maxRetries } : {}
});
completed++;
options.onCaseComplete?.(result, completed, total);
return result;
});
const ms = Date.now() - start;
const passed = perCase.filter((c) => c.passed && !c.error).length;
const failed = total - passed;
const usage = sumUsage(perCase.map((c) => c.usage));
const rubricTypes = {};
for (const [id, rubric] of Object.entries(suite.rubrics)) {
rubricTypes[id] = rubric.type;
}
return {
suite: suite.name,
passed,
failed,
total,
success: failed === 0,
perCase,
perRubric: aggregate(suite, perCase),
usage,
estimatedCostUsd: estimateCostUsd(usage, pricing),
ms,
timestamp: (/* @__PURE__ */ new Date()).toISOString(),
rubricTypes
};
}
// src/compareRuns.ts
var DEFAULT_TOLERANCE = 0.05;
function compareRuns(baseline, current, options = {}) {
const tolerance = options.tolerance ?? DEFAULT_TOLERANCE;
const rubricIds = /* @__PURE__ */ new Set([
...Object.keys(baseline.perRubric),
...Object.keys(current.perRubric)
]);
const deltas = [];
for (const id of rubricIds) {
const base = baseline.perRubric[id];
const cur = current.perRubric[id];
if (base && !cur) {
deltas.push({
rubricId: id,
type: base.type,
baselineMean: base.mean,
currentMean: null,
diff: null,
status: "removed"
});
continue;
}
if (cur && !base) {
deltas.push({
rubricId: id,
type: cur.type,
baselineMean: null,
currentMean: cur.mean,
diff: null,
status: "new"
});
continue;
}
if (!base || !cur) continue;
const diff = cur.mean - base.mean;
let status;
const epsilon = 1e-9;
if (Math.abs(diff) <= tolerance + epsilon) {
status = "unchanged";
} else if (diff < 0) {
status = "regressed";
} else {
status = "improved";
}
deltas.push({
rubricId: id,
type: cur.type,
baselineMean: base.mean,
currentMean: cur.mean,
diff,
status
});
}
const order = {
regressed: 0,
improved: 1,
new: 2,
removed: 3,
unchanged: 4
};
deltas.sort((a, b) => order[a.status] - order[b.status] || a.rubricId.localeCompare(b.rubricId));
return {
baselineSuite: baseline.suite,
currentSuite: current.suite,
tolerance,
deltas,
regressed: deltas.filter((d) => d.status === "regressed"),
improved: deltas.filter((d) => d.status === "improved"),
unchanged: deltas.filter((d) => d.status === "unchanged"),
hasRegression: deltas.some((d) => d.status === "regressed")
};
}
// src/cli/format.ts
function formatUsd(n) {
if (n < 0.01) return `$${n.toFixed(5)}`;
return `$${n.toFixed(4)}`;
}
function formatNum(n) {
if (Number.isNaN(n)) return "n/a";
return n.toFixed(3);
}
function formatPct(n) {
if (n === null || Number.isNaN(n)) return "n/a";
return `${(n * 100).toFixed(0)}%`;
}
function printRunSummary(run) {
const lines = [];
lines.push("");
lines.push(`jev-evals: ${run.suite}`);
lines.push("-".repeat(40));
lines.push(`cases: ${run.passed}/${run.total} passed`);
lines.push(`time: ${run.ms}ms`);
lines.push(
`usage: ${run.usage.inputTokens} in / ${run.usage.outputTokens} out / ${run.usage.totalTokens} total tokens`
);
lines.push(`est. cost: ${formatUsd(run.estimatedCostUsd)}`);
lines.push("");
lines.push("rubric type mean pass% n");
for (const [id, agg] of Object.entries(run.perRubric)) {
lines.push(
`${id.padEnd(20)}${agg.type.padEnd(10)}${formatNum(agg.mean).padEnd(9)}${formatPct(agg.passRate).padEnd(8)}${agg.count}`
);
}
const failing = run.perCase.filter((c) => !c.passed || c.error);
if (failing.length > 0) {
lines.push("");
lines.push(`failing cases (${failing.length}):`);
for (const c of failing) {
if (c.error) {
lines.push(` - ${c.id}: ERROR: ${c.error}`);
} else {
const failedRubrics = Object.values(c.rubrics).filter((r) => r.passed === false).map((r) => r.rubricId);
lines.push(` - ${c.id}: failed [${failedRubrics.join(", ")}]`);
}
}
}
lines.push("");
console.log(lines.join("\n"));
}
function printCompareSummary(diff) {
const lines = [];
lines.push("");
lines.push(`jev-evals compare: ${diff.baselineSuite} -> ${diff.currentSuite}`);
lines.push("-".repeat(40));
lines.push(`tolerance: ${diff.tolerance}`);
lines.push("");
lines.push("rubric status baseline current diff");
for (const d of diff.deltas) {
lines.push(
`${d.rubricId.padEnd(20)}${d.status.padEnd(12)}${fmtOrNa(d.baselineMean).padEnd(11)}${fmtOrNa(d.currentMean).padEnd(11)}${fmtDiff(d.diff)}`
);
}
lines.push("");
lines.push(diff.hasRegression ? "RESULT: regression detected" : "RESULT: no regression");
lines.push("");
console.log(lines.join("\n"));
}
function fmtOrNa(n) {
return n === null ? "n/a" : formatNum(n);
}
function fmtDiff(n) {
if (n === null) return "n/a";
const sign = n > 0 ? "+" : "";
return `${sign}${n.toFixed(3)}`;
}
function buildMarkdownSummary(run, diff) {
const lines = [];
const badge = run.success ? "\u2705" : "\u274C";
lines.push(`### ${badge} jev-evals: \`${run.suite}\``);
lines.push("");
lines.push(`**${run.passed}/${run.total}** cases passed \xB7 **${formatUsd(run.estimatedCostUsd)}** (${run.usage.totalTokens} tokens) \xB7 ${run.ms}ms`);
lines.push("");
lines.push("| Rubric | Type | Mean | Pass % |" + (diff ? " \u0394 vs baseline |" : ""));
lines.push("|---|---|---|---|" + (diff ? "---|" : ""));
for (const [id, agg] of Object.entries(run.perRubric)) {
const deltaCell = diff ? renderDeltaCell(diff, id) : "";
lines.push(
`| \`${id}\` | ${agg.type} | ${formatNum(agg.mean)} | ${formatPct(agg.passRate)} |` + (diff ? ` ${deltaCell} |` : "")
);
}
const failing = run.perCase.filter((c) => !c.passed || c.error);
if (failing.length > 0) {
lines.push("");
lines.push(`<details><summary>${failing.length} failing case(s)</summary>`);
lines.push("");
for (const c of failing) {
if (c.error) {
lines.push(`- \`${c.id}\`: **ERROR** \u2014 ${c.error}`);
} else {
const failedRubrics = Object.values(c.rubrics).filter((r) => r.passed === false).map((r) => `\`${r.rubricId}\` (${formatNum(r.numericValue)})`);
lines.push(`- \`${c.id}\`: ${failedRubrics.join(", ")}`);
}
}
lines.push("");
lines.push("</details>");
}
if (diff && diff.hasRegression) {
lines.push("");
lines.push(`> \u26A0\uFE0F Regression detected in: ${diff.regressed.map((d) => `\`${d.rubricId}\``).join(", ")}`);
}
return lines.join("\n");
}
function renderDeltaCell(diff, rubricId) {
const d = diff.deltas.find((x) => x.rubricId === rubricId);
if (!d || d.diff === null) return "n/a";
const icon = d.status === "regressed" ? "\u{1F53B}" : d.status === "improved" ? "\u{1F53A}" : "\u25AA\uFE0F";
const sign = d.diff > 0 ? "+" : "";
return `${icon} ${sign}${d.diff.toFixed(3)}`;
}
// src/cli/io.ts
import { readFile } from "fs/promises";
async function readJsonFile(path) {
const raw = await readFile(path, "utf8");
return JSON.parse(raw);
}
// src/cli/run.ts
async function loadSuite(path) {
const abs = resolve(process.cwd(), path);
const mod = await import(pathToFileURL(abs).href);
const candidate = mod.default ?? mod.suite ?? mod;
if (!candidate || typeof candidate !== "object" || !("cases" in candidate)) {
throw new Error(
`"${path}" does not export an EvalSuite. Export it as \`export default suite\` or \`export const suite = ...\`.`
);
}
return candidate;
}
async function runCommand(args) {
const suitePath = args.positionals[0];
if (!suitePath) {
console.error("Usage: jev-evals run <suite-file> [options]");
return 1;
}
const suite = await loadSuite(suitePath);
const concurrency = flagNumber(args.flags, "concurrency");
const maxQuestionsPerCall = flagNumber(args.flags, "max-questions-per-call");
const model = flagString(args.flags, "model");
const maxRetries = flagNumber(args.flags, "max-retries");
const inputPrice = flagNumber(args.flags, "input-price");
const outputPrice = flagNumber(args.flags, "output-price");
const asJson = flagBoolean(args.flags, "json");
const savePath = flagString(args.flags, "save");
const baselinePath = flagString(args.flags, "baseline");
const tolerance = flagNumber(args.flags, "tolerance");
const markdownPath = flagString(args.flags, "markdown");
const run = await runEval(suite, {
...concurrency !== void 0 ? { concurrency } : {},
...maxQuestionsPerCall !== void 0 ? { maxQuestionsPerCall } : {},
...model !== void 0 ? { model } : {},
...maxRetries !== void 0 ? { maxRetries } : {},
...inputPrice !== void 0 || outputPrice !== void 0 ? {
pricing: {
...inputPrice !== void 0 ? { inputPerMillion: inputPrice } : {},
...outputPrice !== void 0 ? { outputPerMillion: outputPrice } : {}
}
} : {}
});
if (savePath) {
await writeFile(resolve(process.cwd(), savePath), JSON.stringify(run, null, 2) + "\n", "utf8");
}
let diff;
if (baselinePath) {
const baseline = await readJsonFile(resolve(process.cwd(), baselinePath));
diff = compareRuns(baseline, run, tolerance !== void 0 ? { tolerance } : {});
}
if (markdownPath) {
await writeFile(resolve(process.cwd(), markdownPath), buildMarkdownSummary(run, diff) + "\n", "utf8");
}
if (asJson) {
console.log(JSON.stringify(diff ? { run, compare: diff } : { run }, null, 2));
} else {
printRunSummary(run);
if (diff) printCompareSummary(diff);
}
const failed = !run.success || (diff ? diff.hasRegression : false);
return failed ? 1 : 0;
}
// src/cli/compare.ts
import { writeFile as writeFile2 } from "fs/promises";
import { resolve as resolve2 } from "path";
async function compareCommand(args) {
const [baselinePath, currentPath] = args.positionals;
if (!baselinePath || !currentPath) {
console.error("Usage: jev-evals compare <baseline.json> <current.json> [options]");
return 1;
}
const baseline = await readJsonFile(resolve2(process.cwd(), baselinePath));
const current = await readJsonFile(resolve2(process.cwd(), currentPath));
const tolerance = flagNumber(args.flags, "tolerance");
const asJson = flagBoolean(args.flags, "json");
const markdownPath = args.flags["markdown"];
const diff = compareRuns(baseline, current, tolerance !== void 0 ? { tolerance } : {});
if (typeof markdownPath === "string") {
await writeFile2(resolve2(process.cwd(), markdownPath), buildMarkdownSummary(current, diff) + "\n", "utf8");
}
if (asJson) {
console.log(JSON.stringify(diff, null, 2));
} else {
printCompareSummary(diff);
}
return diff.hasRegression ? 1 : 0;
}
// src/cli.ts
var HELP = `jev-evals \u2014 rubric-based evals for LLM/agent outputs, backed by typesafe-ai/jev.
Usage:
jev-evals run <suite-file> [options] Run a suite and print/save results.
jev-evals compare <baseline.json> <current.json> [options]
Compare two saved runs.
Options for "run":
--save <path> Write the full RunResult as JSON to <path>.
--baseline <path> Compare this run against a previously saved
RunResult JSON and report regressions.
--tolerance <n> Regression tolerance passed to compareRuns
(only with --baseline). Default 0.05.
--concurrency <n> Max cases evaluated in parallel. Default 5.
--max-questions-per-call <n>
Override the suite's maxQuestionsPerCall.
--model <id> Model id passed to evaluate(). Default typesafe-ai/jev.
--max-retries <n> Passed through to evaluate().
--input-price <usd/1M> Override input token pricing for cost estimate.
--output-price <usd/1M> Override output token pricing for cost estimate.
--markdown <path> Write a PR-comment-ready markdown summary to <path>.
--json Print machine-readable JSON instead of a table.
Options for "compare":
--tolerance <n> Regression tolerance. Default 0.05.
--markdown <path> Write a markdown summary to <path>.
--json Print machine-readable JSON instead of a table.
Exit code is non-zero when any case fails its thresholds, or (with
--baseline / compare) when a regression is detected \u2014 suitable for use as a
CI gate.
Auth (not implemented by this package \u2014 see README):
AI_GATEWAY_API_KEY an AI Gateway API key, or
VERCEL_OIDC_TOKEN from \`vercel env pull\` (12h lifetime)
`;
async function main() {
const args = parseArgs(process.argv.slice(2));
const helpRequested = Boolean(args.flags["help"] || args.flags["h"]);
if (helpRequested || !args.command) {
console.log(HELP);
process.exitCode = helpRequested ? 0 : 1;
return;
}
if (args.flags["version"]) {
console.log("jev-evals 0.1.0");
return;
}
switch (args.command) {
case "run":
process.exitCode = await runCommand(args);
return;
case "compare":
process.exitCode = await compareCommand(args);
return;
default:
console.error(`Unknown command "${args.command}".
`);
console.log(HELP);
process.exitCode = 1;
}
}
main().catch((err) => {
console.error(err instanceof Error ? err.stack ?? err.message : err);
process.exitCode = 1;
});
//# sourceMappingURL=cli.js.map