UNPKG

jev-evals

Version:

Rubric-based eval harness for LLM/agent outputs, backed by the typesafe-ai/jev evaluation model via Vercel AI Gateway. Cheap enough (~$0.04/1M input tokens, one round trip per case) to run on every PR.

429 lines (424 loc) • 13.2 kB
// src/defineEval.ts var InvalidSuiteError = class extends Error { constructor(message) { super(message); this.name = "InvalidSuiteError"; } }; function defineEval(suite) { if (!suite.name || !suite.name.trim()) { throw new InvalidSuiteError("EvalSuite.name must be a non-empty string."); } const rubricIds = Object.keys(suite.rubrics); if (rubricIds.length === 0) { throw new InvalidSuiteError(`Suite "${suite.name}" defines no rubrics.`); } for (const id of rubricIds) { const rubric = suite.rubrics[id]; if (!rubric) continue; validateRubric(suite.name, id, rubric); } if (!suite.cases || suite.cases.length === 0) { throw new InvalidSuiteError(`Suite "${suite.name}" defines no cases.`); } const seenIds = /* @__PURE__ */ new Set(); for (const c of suite.cases) { if (!c.id || !c.id.trim()) { throw new InvalidSuiteError(`Suite "${suite.name}" has a case with an empty id.`); } if (seenIds.has(c.id)) { throw new InvalidSuiteError(`Suite "${suite.name}" has duplicate case id "${c.id}".`); } seenIds.add(c.id); if (c.output === void 0 && !c.generate && !suite.generate) { throw new InvalidSuiteError( `Case "${c.id}" in suite "${suite.name}" has no \`output\` and no \`generate\` function (on the case or the suite) to produce one.` ); } } if (suite.thresholds) { for (const id of Object.keys(suite.thresholds)) { if (!(id in suite.rubrics)) { throw new InvalidSuiteError( `Suite "${suite.name}" has a threshold for unknown rubric "${id}".` ); } } } if (suite.maxQuestionsPerCall !== void 0 && suite.maxQuestionsPerCall < 1) { throw new InvalidSuiteError("maxQuestionsPerCall must be >= 1."); } return suite; } function validateRubric(suiteName, id, rubric) { const where = `rubric "${id}" in suite "${suiteName}"`; switch (rubric.type) { case "boolean": if (rubric.instructions === void 0) { throw new InvalidSuiteError(`${where}: boolean rubric requires \`instructions\`.`); } return; case "choice": { const keys = Object.keys(rubric.criteria ?? {}); if (keys.length === 0) { throw new InvalidSuiteError( `${where}: choice rubric requires a nonempty \`criteria\` map.` ); } return; } case "score": { const levels = rubric.criteria ?? []; if (levels.length < 2) { throw new InvalidSuiteError( `${where}: score rubric requires >= 2 \`criteria\` levels (got ${levels.length}).` ); } return; } default: { const exhaustive = rubric; throw new InvalidSuiteError(`${where}: unknown rubric type ${JSON.stringify(exhaustive)}.`); } } } // src/runEval.ts import { experimental_evaluate as evaluate } from "ai"; // src/concurrency.ts async function mapWithConcurrency(items, concurrency, fn) { if (items.length === 0) return []; const limit = Math.max(1, Math.floor(concurrency)); const results = new Array(items.length); let nextIndex = 0; async function worker() { while (true) { const i = nextIndex++; if (i >= items.length) return; results[i] = await fn(items[i], i); } } const workers = Array.from({ length: Math.min(limit, items.length) }, () => worker()); await Promise.all(workers); return results; } // src/cost.ts var DEFAULT_PRICING = { inputPerMillion: 0.04, outputPerMillion: 0.04 }; function estimateCostUsd(usage, pricing = DEFAULT_PRICING) { const inputCost = usage.inputTokens / 1e6 * pricing.inputPerMillion; const outputCost = usage.outputTokens / 1e6 * pricing.outputPerMillion; return inputCost + outputCost; } function sumUsage(items) { let inputTokens = 0; let outputTokens = 0; let totalTokens = 0; for (const u of items) { if (!u) continue; inputTokens += u.inputTokens ?? 0; outputTokens += u.outputTokens ?? 0; totalTokens += u.totalTokens ?? (u.inputTokens ?? 0) + (u.outputTokens ?? 0); } return { inputTokens, outputTokens, totalTokens }; } // src/runEval.ts var DEFAULT_MAX_QUESTIONS_PER_CALL = 40; var DEFAULT_CONCURRENCY = 5; var AuthError = class extends Error { constructor() { super( "No credentials found for the AI Gateway. Set one of:\n - AI_GATEWAY_API_KEY (an AI Gateway API key)\n - VERCEL_OIDC_TOKEN (run `vercel env pull` in a Vercel project; lasts 12h)\njev-evals does not implement auth itself \u2014 it relies on the `ai` SDK picking one of these up. See the README for both paths." ); this.name = "AuthError"; } }; function checkAuth(skip) { if (skip) return; if (process.env["AI_GATEWAY_API_KEY"] || process.env["VERCEL_OIDC_TOKEN"]) return; throw new AuthError(); } function chunkEntries(entries, size) { if (entries.length <= size) return [entries]; const chunks = []; for (let i = 0; i < entries.length; i += size) { chunks.push(entries.slice(i, i + size)); } return chunks; } function choiceOrdinal(rubric, choice) { if (rubric.type !== "choice") return 0; const keys = Object.keys(rubric.criteria); const idx = keys.indexOf(choice); if (idx < 0 || keys.length <= 1) return 0; return idx / (keys.length - 1); } function numericValueOf(rubric, answer) { switch (answer.type) { case "boolean": return answer.probability; case "score": return answer.score; case "choice": return choiceOrdinal(rubric, answer.choice); } } function evaluateThreshold(rubric, answer, threshold) { if (answer.type === "choice") { const accepted = Array.isArray(threshold) ? threshold : [threshold]; return accepted.includes(answer.choice); } const numericThreshold = typeof threshold === "number" ? threshold : Number(threshold); const value = answer.type === "boolean" ? answer.probability : answer.score; return value >= numericThreshold; } async function resolveOutput(evalCase, suite) { if (evalCase.output !== void 0) return evalCase.output; const generate = evalCase.generate ?? suite.generate; if (!generate) { throw new Error(`Case "${evalCase.id}" has no \`output\` and no \`generate\` function.`); } return await generate(evalCase.input, evalCase); } async function evaluateOneCase(args) { const { suite, evalCase, model, maxQuestionsPerCall, abortSignal, maxRetries } = args; let output; try { output = await resolveOutput(evalCase, suite); } catch (err) { return { id: evalCase.id, input: evalCase.input, output: void 0, expected: evalCase.expected, metadata: evalCase.metadata, rubrics: {}, passed: false, error: err instanceof Error ? err.message : String(err) }; } const state = { input: evalCase.input, output, ...evalCase.expected !== void 0 ? { expected: evalCase.expected } : {} }; const rubricEntries = Object.entries(suite.rubrics); const chunks = chunkEntries(rubricEntries, maxQuestionsPerCall); try { const chunkResults = await Promise.all( chunks.map( (chunk) => evaluate({ model, state, questions: Object.fromEntries(chunk), ...abortSignal ? { abortSignal } : {}, ...maxRetries !== void 0 ? { maxRetries } : {} }) ) ); const answers = {}; const warnings = []; for (const result of chunkResults) { Object.assign(answers, result.answers); if (result.warnings) warnings.push(...result.warnings); } const usage = sumUsage(chunkResults.map((r) => r.usage)); const rubrics = {}; for (const [id, rubric] of rubricEntries) { const answer = answers[id]; if (!answer) continue; const numericValue = numericValueOf(rubric, answer); const threshold = suite.thresholds?.[id]; const passed2 = threshold === void 0 ? null : evaluateThreshold(rubric, answer, threshold); rubrics[id] = { rubricId: id, type: rubric.type, answer, numericValue, passed: passed2, ...threshold !== void 0 ? { threshold } : {} }; } const passed = Object.values(rubrics).every((r) => r.passed !== false); return { id: evalCase.id, input: evalCase.input, output, expected: evalCase.expected, metadata: evalCase.metadata, rubrics, passed, usage, warnings: warnings.length > 0 ? warnings : void 0 }; } catch (err) { return { id: evalCase.id, input: evalCase.input, output, expected: evalCase.expected, metadata: evalCase.metadata, rubrics: {}, passed: false, error: err instanceof Error ? err.message : String(err) }; } } function aggregate(suite, perCase) { const out = {}; for (const [id, rubric] of Object.entries(suite.rubrics)) { const values = []; let passCount = 0; let thresholdCount = 0; for (const c of perCase) { const r = c.rubrics[id]; if (!r) continue; values.push(r.numericValue); if (r.passed !== null) { thresholdCount++; if (r.passed) passCount++; } } out[id] = { rubricId: id, type: rubric.type, mean: values.length > 0 ? values.reduce((a, b) => a + b, 0) / values.length : NaN, min: values.length > 0 ? Math.min(...values) : NaN, max: values.length > 0 ? Math.max(...values) : NaN, passRate: thresholdCount > 0 ? passCount / thresholdCount : null, count: values.length }; } return out; } async function runEval(suite, options = {}) { checkAuth(options.skipAuthCheck); const model = options.model ?? "typesafe-ai/jev"; const concurrency = options.concurrency ?? DEFAULT_CONCURRENCY; const maxQuestionsPerCall = options.maxQuestionsPerCall ?? suite.maxQuestionsPerCall ?? DEFAULT_MAX_QUESTIONS_PER_CALL; const pricing = { ...DEFAULT_PRICING, ...options.pricing }; const start = Date.now(); const total = suite.cases.length; let completed = 0; const perCase = await mapWithConcurrency(suite.cases, concurrency, async (evalCase) => { const result = await evaluateOneCase({ suite, evalCase, model, maxQuestionsPerCall, ...options.abortSignal ? { abortSignal: options.abortSignal } : {}, ...options.maxRetries !== void 0 ? { maxRetries: options.maxRetries } : {} }); completed++; options.onCaseComplete?.(result, completed, total); return result; }); const ms = Date.now() - start; const passed = perCase.filter((c) => c.passed && !c.error).length; const failed = total - passed; const usage = sumUsage(perCase.map((c) => c.usage)); const rubricTypes = {}; for (const [id, rubric] of Object.entries(suite.rubrics)) { rubricTypes[id] = rubric.type; } return { suite: suite.name, passed, failed, total, success: failed === 0, perCase, perRubric: aggregate(suite, perCase), usage, estimatedCostUsd: estimateCostUsd(usage, pricing), ms, timestamp: (/* @__PURE__ */ new Date()).toISOString(), rubricTypes }; } // src/compareRuns.ts var DEFAULT_TOLERANCE = 0.05; function compareRuns(baseline, current, options = {}) { const tolerance = options.tolerance ?? DEFAULT_TOLERANCE; const rubricIds = /* @__PURE__ */ new Set([ ...Object.keys(baseline.perRubric), ...Object.keys(current.perRubric) ]); const deltas = []; for (const id of rubricIds) { const base = baseline.perRubric[id]; const cur = current.perRubric[id]; if (base && !cur) { deltas.push({ rubricId: id, type: base.type, baselineMean: base.mean, currentMean: null, diff: null, status: "removed" }); continue; } if (cur && !base) { deltas.push({ rubricId: id, type: cur.type, baselineMean: null, currentMean: cur.mean, diff: null, status: "new" }); continue; } if (!base || !cur) continue; const diff = cur.mean - base.mean; let status; const epsilon = 1e-9; if (Math.abs(diff) <= tolerance + epsilon) { status = "unchanged"; } else if (diff < 0) { status = "regressed"; } else { status = "improved"; } deltas.push({ rubricId: id, type: cur.type, baselineMean: base.mean, currentMean: cur.mean, diff, status }); } const order = { regressed: 0, improved: 1, new: 2, removed: 3, unchanged: 4 }; deltas.sort((a, b) => order[a.status] - order[b.status] || a.rubricId.localeCompare(b.rubricId)); return { baselineSuite: baseline.suite, currentSuite: current.suite, tolerance, deltas, regressed: deltas.filter((d) => d.status === "regressed"), improved: deltas.filter((d) => d.status === "improved"), unchanged: deltas.filter((d) => d.status === "unchanged"), hasRegression: deltas.some((d) => d.status === "regressed") }; } export { AuthError, DEFAULT_PRICING, InvalidSuiteError, compareRuns, defineEval, estimateCostUsd, mapWithConcurrency, runEval, sumUsage }; //# sourceMappingURL=index.js.map