jev-evals
Version:
Rubric-based eval harness for LLM/agent outputs, backed by the typesafe-ai/jev evaluation model via Vercel AI Gateway. Cheap enough (~$0.04/1M input tokens, one round trip per case) to run on every PR.
429 lines (424 loc) • 13.2 kB
JavaScript
// src/defineEval.ts
var InvalidSuiteError = class extends Error {
constructor(message) {
super(message);
this.name = "InvalidSuiteError";
}
};
function defineEval(suite) {
if (!suite.name || !suite.name.trim()) {
throw new InvalidSuiteError("EvalSuite.name must be a non-empty string.");
}
const rubricIds = Object.keys(suite.rubrics);
if (rubricIds.length === 0) {
throw new InvalidSuiteError(`Suite "${suite.name}" defines no rubrics.`);
}
for (const id of rubricIds) {
const rubric = suite.rubrics[id];
if (!rubric) continue;
validateRubric(suite.name, id, rubric);
}
if (!suite.cases || suite.cases.length === 0) {
throw new InvalidSuiteError(`Suite "${suite.name}" defines no cases.`);
}
const seenIds = /* @__PURE__ */ new Set();
for (const c of suite.cases) {
if (!c.id || !c.id.trim()) {
throw new InvalidSuiteError(`Suite "${suite.name}" has a case with an empty id.`);
}
if (seenIds.has(c.id)) {
throw new InvalidSuiteError(`Suite "${suite.name}" has duplicate case id "${c.id}".`);
}
seenIds.add(c.id);
if (c.output === void 0 && !c.generate && !suite.generate) {
throw new InvalidSuiteError(
`Case "${c.id}" in suite "${suite.name}" has no \`output\` and no \`generate\` function (on the case or the suite) to produce one.`
);
}
}
if (suite.thresholds) {
for (const id of Object.keys(suite.thresholds)) {
if (!(id in suite.rubrics)) {
throw new InvalidSuiteError(
`Suite "${suite.name}" has a threshold for unknown rubric "${id}".`
);
}
}
}
if (suite.maxQuestionsPerCall !== void 0 && suite.maxQuestionsPerCall < 1) {
throw new InvalidSuiteError("maxQuestionsPerCall must be >= 1.");
}
return suite;
}
function validateRubric(suiteName, id, rubric) {
const where = `rubric "${id}" in suite "${suiteName}"`;
switch (rubric.type) {
case "boolean":
if (rubric.instructions === void 0) {
throw new InvalidSuiteError(`${where}: boolean rubric requires \`instructions\`.`);
}
return;
case "choice": {
const keys = Object.keys(rubric.criteria ?? {});
if (keys.length === 0) {
throw new InvalidSuiteError(
`${where}: choice rubric requires a nonempty \`criteria\` map.`
);
}
return;
}
case "score": {
const levels = rubric.criteria ?? [];
if (levels.length < 2) {
throw new InvalidSuiteError(
`${where}: score rubric requires >= 2 \`criteria\` levels (got ${levels.length}).`
);
}
return;
}
default: {
const exhaustive = rubric;
throw new InvalidSuiteError(`${where}: unknown rubric type ${JSON.stringify(exhaustive)}.`);
}
}
}
// src/runEval.ts
import { experimental_evaluate as evaluate } from "ai";
// src/concurrency.ts
async function mapWithConcurrency(items, concurrency, fn) {
if (items.length === 0) return [];
const limit = Math.max(1, Math.floor(concurrency));
const results = new Array(items.length);
let nextIndex = 0;
async function worker() {
while (true) {
const i = nextIndex++;
if (i >= items.length) return;
results[i] = await fn(items[i], i);
}
}
const workers = Array.from({ length: Math.min(limit, items.length) }, () => worker());
await Promise.all(workers);
return results;
}
// src/cost.ts
var DEFAULT_PRICING = {
inputPerMillion: 0.04,
outputPerMillion: 0.04
};
function estimateCostUsd(usage, pricing = DEFAULT_PRICING) {
const inputCost = usage.inputTokens / 1e6 * pricing.inputPerMillion;
const outputCost = usage.outputTokens / 1e6 * pricing.outputPerMillion;
return inputCost + outputCost;
}
function sumUsage(items) {
let inputTokens = 0;
let outputTokens = 0;
let totalTokens = 0;
for (const u of items) {
if (!u) continue;
inputTokens += u.inputTokens ?? 0;
outputTokens += u.outputTokens ?? 0;
totalTokens += u.totalTokens ?? (u.inputTokens ?? 0) + (u.outputTokens ?? 0);
}
return { inputTokens, outputTokens, totalTokens };
}
// src/runEval.ts
var DEFAULT_MAX_QUESTIONS_PER_CALL = 40;
var DEFAULT_CONCURRENCY = 5;
var AuthError = class extends Error {
constructor() {
super(
"No credentials found for the AI Gateway. Set one of:\n - AI_GATEWAY_API_KEY (an AI Gateway API key)\n - VERCEL_OIDC_TOKEN (run `vercel env pull` in a Vercel project; lasts 12h)\njev-evals does not implement auth itself \u2014 it relies on the `ai` SDK picking one of these up. See the README for both paths."
);
this.name = "AuthError";
}
};
function checkAuth(skip) {
if (skip) return;
if (process.env["AI_GATEWAY_API_KEY"] || process.env["VERCEL_OIDC_TOKEN"]) return;
throw new AuthError();
}
function chunkEntries(entries, size) {
if (entries.length <= size) return [entries];
const chunks = [];
for (let i = 0; i < entries.length; i += size) {
chunks.push(entries.slice(i, i + size));
}
return chunks;
}
function choiceOrdinal(rubric, choice) {
if (rubric.type !== "choice") return 0;
const keys = Object.keys(rubric.criteria);
const idx = keys.indexOf(choice);
if (idx < 0 || keys.length <= 1) return 0;
return idx / (keys.length - 1);
}
function numericValueOf(rubric, answer) {
switch (answer.type) {
case "boolean":
return answer.probability;
case "score":
return answer.score;
case "choice":
return choiceOrdinal(rubric, answer.choice);
}
}
function evaluateThreshold(rubric, answer, threshold) {
if (answer.type === "choice") {
const accepted = Array.isArray(threshold) ? threshold : [threshold];
return accepted.includes(answer.choice);
}
const numericThreshold = typeof threshold === "number" ? threshold : Number(threshold);
const value = answer.type === "boolean" ? answer.probability : answer.score;
return value >= numericThreshold;
}
async function resolveOutput(evalCase, suite) {
if (evalCase.output !== void 0) return evalCase.output;
const generate = evalCase.generate ?? suite.generate;
if (!generate) {
throw new Error(`Case "${evalCase.id}" has no \`output\` and no \`generate\` function.`);
}
return await generate(evalCase.input, evalCase);
}
async function evaluateOneCase(args) {
const { suite, evalCase, model, maxQuestionsPerCall, abortSignal, maxRetries } = args;
let output;
try {
output = await resolveOutput(evalCase, suite);
} catch (err) {
return {
id: evalCase.id,
input: evalCase.input,
output: void 0,
expected: evalCase.expected,
metadata: evalCase.metadata,
rubrics: {},
passed: false,
error: err instanceof Error ? err.message : String(err)
};
}
const state = {
input: evalCase.input,
output,
...evalCase.expected !== void 0 ? { expected: evalCase.expected } : {}
};
const rubricEntries = Object.entries(suite.rubrics);
const chunks = chunkEntries(rubricEntries, maxQuestionsPerCall);
try {
const chunkResults = await Promise.all(
chunks.map(
(chunk) => evaluate({
model,
state,
questions: Object.fromEntries(chunk),
...abortSignal ? { abortSignal } : {},
...maxRetries !== void 0 ? { maxRetries } : {}
})
)
);
const answers = {};
const warnings = [];
for (const result of chunkResults) {
Object.assign(answers, result.answers);
if (result.warnings) warnings.push(...result.warnings);
}
const usage = sumUsage(chunkResults.map((r) => r.usage));
const rubrics = {};
for (const [id, rubric] of rubricEntries) {
const answer = answers[id];
if (!answer) continue;
const numericValue = numericValueOf(rubric, answer);
const threshold = suite.thresholds?.[id];
const passed2 = threshold === void 0 ? null : evaluateThreshold(rubric, answer, threshold);
rubrics[id] = {
rubricId: id,
type: rubric.type,
answer,
numericValue,
passed: passed2,
...threshold !== void 0 ? { threshold } : {}
};
}
const passed = Object.values(rubrics).every((r) => r.passed !== false);
return {
id: evalCase.id,
input: evalCase.input,
output,
expected: evalCase.expected,
metadata: evalCase.metadata,
rubrics,
passed,
usage,
warnings: warnings.length > 0 ? warnings : void 0
};
} catch (err) {
return {
id: evalCase.id,
input: evalCase.input,
output,
expected: evalCase.expected,
metadata: evalCase.metadata,
rubrics: {},
passed: false,
error: err instanceof Error ? err.message : String(err)
};
}
}
function aggregate(suite, perCase) {
const out = {};
for (const [id, rubric] of Object.entries(suite.rubrics)) {
const values = [];
let passCount = 0;
let thresholdCount = 0;
for (const c of perCase) {
const r = c.rubrics[id];
if (!r) continue;
values.push(r.numericValue);
if (r.passed !== null) {
thresholdCount++;
if (r.passed) passCount++;
}
}
out[id] = {
rubricId: id,
type: rubric.type,
mean: values.length > 0 ? values.reduce((a, b) => a + b, 0) / values.length : NaN,
min: values.length > 0 ? Math.min(...values) : NaN,
max: values.length > 0 ? Math.max(...values) : NaN,
passRate: thresholdCount > 0 ? passCount / thresholdCount : null,
count: values.length
};
}
return out;
}
async function runEval(suite, options = {}) {
checkAuth(options.skipAuthCheck);
const model = options.model ?? "typesafe-ai/jev";
const concurrency = options.concurrency ?? DEFAULT_CONCURRENCY;
const maxQuestionsPerCall = options.maxQuestionsPerCall ?? suite.maxQuestionsPerCall ?? DEFAULT_MAX_QUESTIONS_PER_CALL;
const pricing = { ...DEFAULT_PRICING, ...options.pricing };
const start = Date.now();
const total = suite.cases.length;
let completed = 0;
const perCase = await mapWithConcurrency(suite.cases, concurrency, async (evalCase) => {
const result = await evaluateOneCase({
suite,
evalCase,
model,
maxQuestionsPerCall,
...options.abortSignal ? { abortSignal: options.abortSignal } : {},
...options.maxRetries !== void 0 ? { maxRetries: options.maxRetries } : {}
});
completed++;
options.onCaseComplete?.(result, completed, total);
return result;
});
const ms = Date.now() - start;
const passed = perCase.filter((c) => c.passed && !c.error).length;
const failed = total - passed;
const usage = sumUsage(perCase.map((c) => c.usage));
const rubricTypes = {};
for (const [id, rubric] of Object.entries(suite.rubrics)) {
rubricTypes[id] = rubric.type;
}
return {
suite: suite.name,
passed,
failed,
total,
success: failed === 0,
perCase,
perRubric: aggregate(suite, perCase),
usage,
estimatedCostUsd: estimateCostUsd(usage, pricing),
ms,
timestamp: (/* @__PURE__ */ new Date()).toISOString(),
rubricTypes
};
}
// src/compareRuns.ts
var DEFAULT_TOLERANCE = 0.05;
function compareRuns(baseline, current, options = {}) {
const tolerance = options.tolerance ?? DEFAULT_TOLERANCE;
const rubricIds = /* @__PURE__ */ new Set([
...Object.keys(baseline.perRubric),
...Object.keys(current.perRubric)
]);
const deltas = [];
for (const id of rubricIds) {
const base = baseline.perRubric[id];
const cur = current.perRubric[id];
if (base && !cur) {
deltas.push({
rubricId: id,
type: base.type,
baselineMean: base.mean,
currentMean: null,
diff: null,
status: "removed"
});
continue;
}
if (cur && !base) {
deltas.push({
rubricId: id,
type: cur.type,
baselineMean: null,
currentMean: cur.mean,
diff: null,
status: "new"
});
continue;
}
if (!base || !cur) continue;
const diff = cur.mean - base.mean;
let status;
const epsilon = 1e-9;
if (Math.abs(diff) <= tolerance + epsilon) {
status = "unchanged";
} else if (diff < 0) {
status = "regressed";
} else {
status = "improved";
}
deltas.push({
rubricId: id,
type: cur.type,
baselineMean: base.mean,
currentMean: cur.mean,
diff,
status
});
}
const order = {
regressed: 0,
improved: 1,
new: 2,
removed: 3,
unchanged: 4
};
deltas.sort((a, b) => order[a.status] - order[b.status] || a.rubricId.localeCompare(b.rubricId));
return {
baselineSuite: baseline.suite,
currentSuite: current.suite,
tolerance,
deltas,
regressed: deltas.filter((d) => d.status === "regressed"),
improved: deltas.filter((d) => d.status === "improved"),
unchanged: deltas.filter((d) => d.status === "unchanged"),
hasRegression: deltas.some((d) => d.status === "regressed")
};
}
export {
AuthError,
DEFAULT_PRICING,
InvalidSuiteError,
compareRuns,
defineEval,
estimateCostUsd,
mapWithConcurrency,
runEval,
sumUsage
};
//# sourceMappingURL=index.js.map