UNPKG

jev-evals

Version:

Rubric-based eval harness for LLM/agent outputs, backed by the typesafe-ai/jev evaluation model via Vercel AI Gateway. Cheap enough (~$0.04/1M input tokens, one round trip per case) to run on every PR.

284 lines (277 loc) • 11.4 kB
/** * Types for jev-evals. * * The `Rubric`/`Question` shapes here mirror the VERIFIED `ai` package * `experimental_evaluate` API exactly (see README for the source). Do not * add fields that aren't part of that API — this package is a thin, * opinionated harness around it, not a reinterpretation of it. */ /** A JSON-serializable object, matching the AI SDK's `JSONObject`. */ type JSONValue = string | number | boolean | null | JSONValue[] | { [key: string]: JSONValue; }; type JSONObject = { [key: string]: JSONValue; }; /** * `Input` matches the `Input` type accepted by jev's `instructions` and * `criteria` fields, and by `evaluate()`'s `state`: a string, a JSON object, * or a JSON array. */ type Input = string | JSONObject | JSONValue[]; interface BooleanRubric { type: 'boolean'; instructions: Input; criteria?: { true?: Input | null; false?: Input | null; }; } interface ChoiceRubric { type: 'choice'; instructions: Input; /** Nonempty map of option name -> description. Key order is significant: * it is treated as low-to-high when a rubric represents an ordered scale * (see `choiceOrdinal` in aggregate.ts). */ criteria: Record<string, Input | null>; } interface ScoreRubric { type: 'score'; instructions: Input; /** >= 2 levels, ordered LOWEST to HIGHEST. */ criteria: (Input | null)[]; } type Rubric = BooleanRubric | ChoiceRubric | ScoreRubric; type RubricType = Rubric['type']; type RubricMap = Record<string, Rubric>; interface BooleanAnswer { type: 'boolean'; /** P(true). ALWAYS present. */ probability: number; } interface ChoiceAnswer { type: 'choice'; choice: string; probabilities?: Record<string, number>; } interface ScoreAnswer { type: 'score'; /** Fractional, in [0, levels - 1]. Never round this. */ score: number; /** Keyed '0', '1', '2', ... */ probabilities?: Record<string, number>; } type Answer = BooleanAnswer | ChoiceAnswer | ScoreAnswer; interface UsageTotals { inputTokens: number; outputTokens: number; totalTokens: number; } /** * A threshold for a rubric. For `boolean` and `score` rubrics this is the * minimum acceptable numeric value (probability, or raw fractional score on * the rubric's own [0, levels-1] scale). For `choice` rubrics it is the * accepted choice (or set of accepted choices). */ type Threshold = number | string | string[]; type ThresholdMap<R extends RubricMap> = Partial<Record<keyof R, Threshold>>; interface EvalCase { id: string; input: Input; /** * Pre-computed output to judge. Omit it (and supply `generate` on the * case or the suite) to have the harness produce it for you. */ output?: Input; /** Optional reference/gold answer, passed to jev alongside input/output. */ expected?: Input; /** * Per-case override for producing `output` when it isn't supplied. * Falls back to the suite-level `generate` when omitted. */ generate?: (input: Input, evalCase: EvalCase) => Promise<Input> | Input; /** Free-form metadata carried through into results, untouched. */ metadata?: JSONObject; } interface EvalSuite<R extends RubricMap = RubricMap> { name: string; rubrics: R; thresholds?: ThresholdMap<R>; cases: EvalCase[]; /** * Suite-level default for producing a case's `output` when the case * doesn't supply one directly. A case's own `generate` takes priority. */ generate?: (input: Input, evalCase: EvalCase) => Promise<Input> | Input; /** * Safety valve for suites with unusually many rubrics: jev answers every * rubric for a case in ONE round trip, which is the whole point (cost and * latency scale with round trips, not with question count). But a single * request still has to fit the provider's context/response budget, so if * a suite defines more rubrics than this, the harness splits that case's * rubrics across multiple `evaluate()` calls (run in parallel) rather * than failing outright. Keep suites under this limit to get the one * -round-trip cost/latency profile the package is built around. * Default: 40. */ maxQuestionsPerCall?: number; } interface RubricResult { rubricId: string; type: RubricType; answer: Answer; /** * A single numeric projection of `answer`, used for thresholds and * aggregation: * - boolean -> probability (0..1) * - score -> the raw fractional score, e.g. 2.97 (0..levels-1) * - choice -> the chosen option's position among `criteria`'s keys, * normalized to 0..1 (0 = first key, 1 = last key). This * only makes sense as an ordinal when the rubric's criteria * are themselves ordered low-to-high, which is a * convention this package encourages but can't enforce. */ numericValue: number; /** null when the rubric has no threshold configured (informational only). */ passed: boolean | null; threshold?: Threshold; } interface CaseResult { id: string; input: Input; output: Input | undefined; expected?: Input; metadata?: JSONObject; rubrics: Record<string, RubricResult>; /** True iff every thresholded rubric passed. True (vacuously) if none. */ passed: boolean; usage?: UsageTotals; warnings?: unknown[]; /** Set instead of `rubrics` succeeding, if generation or evaluation threw. */ error?: string; } interface RubricAggregate { rubricId: string; type: RubricType; /** Mean of `numericValue` across cases that produced an answer for this rubric. */ mean: number; min: number; max: number; /** Fraction of cases that passed this rubric's threshold; null if no threshold. */ passRate: number | null; /** Number of cases with an answer for this rubric (errors excluded). */ count: number; } interface RunResult { suite: string; /** Number of cases where every thresholded rubric passed (and no error). */ passed: number; /** Number of cases with a threshold failure or a runtime error. */ failed: number; total: number; /** `failed === 0`. What the CLI uses for its exit code. */ success: boolean; perCase: CaseResult[]; perRubric: Record<string, RubricAggregate>; usage: UsageTotals; estimatedCostUsd: number; ms: number; timestamp: string; /** Carried through so `compareRuns` can label rubrics without the suite. */ rubricTypes: Record<string, RubricType>; } type DeltaStatus = 'regressed' | 'improved' | 'unchanged' | 'new' | 'removed'; interface RubricDelta { rubricId: string; type: RubricType | undefined; baselineMean: number | null; currentMean: number | null; /** currentMean - baselineMean (null when either side is missing). */ diff: number | null; status: DeltaStatus; } interface CompareResult { baselineSuite: string; currentSuite: string; tolerance: number; deltas: RubricDelta[]; regressed: RubricDelta[]; improved: RubricDelta[]; unchanged: RubricDelta[]; /** True iff at least one rubric regressed beyond tolerance. */ hasRegression: boolean; } declare class InvalidSuiteError extends Error { constructor(message: string); } /** * Defines an eval suite. This is mostly a typed identity function — it * exists so rubric/case shapes are inferred and validated up front (at * suite-definition time) rather than surfacing as a confusing failure deep * inside a jev call. */ declare function defineEval<R extends RubricMap>(suite: EvalSuite<R>): EvalSuite<R>; interface PricingConfig { /** USD per 1,000,000 input tokens. Jev is documented at ~$0.04/1M. */ inputPerMillion: number; /** * USD per 1,000,000 output tokens. Jev's answers are tiny structured * objects, so output tokens are usually a rounding error next to input * tokens (which carry the full state + all rubric instructions/criteria). * Not separately documented for jev; defaults to the same rate as input * so the estimate errs conservative rather than pretending output is * free. Override via `RunEvalOptions.pricing` if you have a better number. */ outputPerMillion: number; } declare const DEFAULT_PRICING: PricingConfig; declare function estimateCostUsd(usage: UsageTotals, pricing?: PricingConfig): number; declare function sumUsage(items: Array<Partial<UsageTotals> | undefined>): UsageTotals; declare class AuthError extends Error { constructor(); } interface RunEvalOptions { /** Max cases evaluated in parallel. Default 5. */ concurrency?: number; /** Overrides `suite.maxQuestionsPerCall`. */ maxQuestionsPerCall?: number; /** Model id passed to `evaluate()`. Default 'typesafe-ai/jev'. */ model?: string; abortSignal?: AbortSignal; /** Passed through to `evaluate()`. Defaults to the AI SDK's own default (2). */ maxRetries?: number; /** Pricing used for `estimatedCostUsd`. Defaults to jev's documented ~$0.04/1M input rate. */ pricing?: Partial<PricingConfig>; /** Called as each case finishes, in completion order (not result order). */ onCaseComplete?: (result: CaseResult, index: number, total: number) => void; /** * Skip the AI Gateway credential preflight check. Intended for tests that * mock `experimental_evaluate` and never actually hit the network. */ skipAuthCheck?: boolean; } declare function runEval(suite: EvalSuite<RubricMap>, options?: RunEvalOptions): Promise<RunResult>; interface CompareOptions { /** * Absolute difference (in the rubric's own numeric scale — 0..1 for * boolean/choice, 0..levels-1 for score) below which a change is * considered noise rather than a real regression or improvement. * Default 0.05. */ tolerance?: number; } /** * Compares a baseline run (e.g. loaded from a JSON file saved by a previous * CI run) against a current run, per rubric, using the rubrics' aggregate * means. A rubric that only exists on one side is reported as 'new' or * 'removed' rather than silently ignored. */ declare function compareRuns(baseline: RunResult, current: RunResult, options?: CompareOptions): CompareResult; /** * Runs `items` through `fn` with at most `concurrency` in flight at once, * returning results in the SAME ORDER as `items` regardless of which * finishes first. This is deliberately hand-rolled (no `p-limit` et al.) to * keep the runtime dependency footprint at just `ai`. */ declare function mapWithConcurrency<T, R>(items: readonly T[], concurrency: number, fn: (item: T, index: number) => Promise<R>): Promise<R[]>; export { type Answer, AuthError, type BooleanAnswer, type BooleanRubric, type CaseResult, type ChoiceAnswer, type ChoiceRubric, type CompareOptions, type CompareResult, DEFAULT_PRICING, type DeltaStatus, type EvalCase, type EvalSuite, type Input, InvalidSuiteError, type JSONObject, type JSONValue, type PricingConfig, type Rubric, type RubricAggregate, type RubricDelta, type RubricMap, type RubricResult, type RubricType, type RunEvalOptions, type RunResult, type ScoreAnswer, type ScoreRubric, type Threshold, type ThresholdMap, type UsageTotals, compareRuns, defineEval, estimateCostUsd, mapWithConcurrency, runEval, sumUsage };