@arizeai/phoenix-evals
Version:
A library for running evaluations for AI use cases
60 lines • 2.81 kB
JavaScript
import { TOOL_INVOCATION_CLASSIFICATION_EVALUATOR_CONFIG } from "../__generated__/default_templates/index.js";
import { createClassificationEvaluator } from "./createClassificationEvaluator.js";
/**
* Creates a tool invocation evaluator function.
*
* This function returns an evaluator that determines whether a tool was invoked
* correctly with proper arguments, formatting, and safe content.
*
* The evaluator checks for:
* - Properly structured JSON (if applicable)
* - All required fields/parameters present
* - No hallucinated or nonexistent fields
* - Argument values matching user query and schema expectations
* - No unsafe content (e.g., PII) in arguments
*
* @param args - The arguments for creating the tool invocation evaluator.
* @param args.model - The model to use for classification.
* @param args.choices - The possible classification choices (defaults to correct/incorrect).
* @param args.promptTemplate - The prompt template to use (defaults to TOOL_INVOCATION_TEMPLATE).
* @param args.telemetry - The telemetry to use for the evaluator.
*
* @returns An evaluator function that takes a {@link ToolInvocationEvaluationRecord} and returns
* a classification result indicating whether the tool invocation is correct or incorrect.
*
* @example
* ```ts
* const evaluator = createToolInvocationEvaluator({ model: openai("gpt-4o-mini") });
*
* // Example with JSON schema format for available tools
* const result = await evaluator.evaluate({
* input: "User: Book a flight from NYC to LA for tomorrow",
* availableTools: JSON.stringify({
* name: "book_flight",
* description: "Book a flight between two cities",
* parameters: {
* type: "object",
* properties: {
* origin: { type: "string", description: "Departure city code" },
* destination: { type: "string", description: "Arrival city code" },
* date: { type: "string", description: "Flight date in YYYY-MM-DD" }
* },
* required: ["origin", "destination", "date"]
* }
* }),
* toolSelection: 'book_flight(origin="NYC", destination="LA", date="2024-01-15")'
* });
* console.log(result.label); // "correct" or "incorrect"
* ```
*/
export function createToolInvocationEvaluator(args) {
const { choices = TOOL_INVOCATION_CLASSIFICATION_EVALUATOR_CONFIG.choices, promptTemplate = TOOL_INVOCATION_CLASSIFICATION_EVALUATOR_CONFIG.template, optimizationDirection = TOOL_INVOCATION_CLASSIFICATION_EVALUATOR_CONFIG.optimizationDirection, name = TOOL_INVOCATION_CLASSIFICATION_EVALUATOR_CONFIG.name, ...rest } = args;
return createClassificationEvaluator({
...rest,
promptTemplate,
choices,
optimizationDirection,
name,
});
}
//# sourceMappingURL=createToolInvocationEvaluator.js.map