UNPKG

@arizeai/phoenix-evals

Version:

A library for running evaluations for AI use cases

71 lines 1.8 kB
import { LanguageModel } from "ai"; import { WithTelemetry } from "./otel"; /** * A specific AI example that is under evaluation */ export interface ExampleRecord<OutputType, InputType> { output: OutputType; expected?: OutputType; input?: InputType; [key: string]: unknown; } export interface WithLLM { model: LanguageModel; } export interface LLMEvaluationArgs extends WithLLM { } /** * The result of an evaluation */ export interface EvaluationResult { /** * The score of the evaluation. * @example 0.95 */ score?: number; /** * The label of the evaluation. * @example "correct" */ label?: string; /** * The explanation of the evaluation. * @example "The model correctly identified the sentiment of the text." */ explanation?: string; } /** * The result of a classification */ export interface ClassificationResult { label: string; explanation?: string; } /** * The choice (e.g. the label and score mapping) of a classification based evaluation */ export interface ClassificationChoice { label: string; score: number; } /** * A mapping of labels to scores */ export type ClassificationChoicesMap = Record<string, number>; /** * The arguments for creating a classification-based evaluator */ export interface CreateClassifierArgs extends WithTelemetry { model: LanguageModel; /** * The choices to classify the example into. * e.g. { "correct": 1, "incorrect": 0 } */ choices: ClassificationChoicesMap; /** * The prompt template to use for classification */ promptTemplate: string; } export type EvaluatorFn<ExampleType extends Record<string, unknown>> = (args: ExampleType) => Promise<EvaluationResult>; //# sourceMappingURL=evals.d.ts.map