@arizeai/phoenix-evals
Version:
A library for running evaluations for AI use cases
70 lines • 3.53 kB
JavaScript
;
var __rest = (this && this.__rest) || function (s, e) {
var t = {};
for (var p in s) if (Object.prototype.hasOwnProperty.call(s, p) && e.indexOf(p) < 0)
t[p] = s[p];
if (s != null && typeof Object.getOwnPropertySymbols === "function")
for (var i = 0, p = Object.getOwnPropertySymbols(s); i < p.length; i++) {
if (e.indexOf(p[i]) < 0 && Object.prototype.propertyIsEnumerable.call(s, p[i]))
t[p[i]] = s[p[i]];
}
return t;
};
Object.defineProperty(exports, "__esModule", { value: true });
exports.createToolResponseHandlingEvaluator = createToolResponseHandlingEvaluator;
const default_templates_1 = require("../__generated__/default_templates");
const createClassificationEvaluator_1 = require("./createClassificationEvaluator");
/**
* Creates a tool response handling evaluator function.
*
* This function returns an evaluator that determines whether an AI agent properly
* handled a tool's response, including error handling, data extraction,
* transformation, and safe information disclosure.
*
* @param args - The arguments for creating the tool response handling evaluator.
* @param args.model - The model to use for classification.
* @param args.choices - The possible classification choices (defaults to correct/incorrect).
* @param args.promptTemplate - The prompt template to use.
* @param args.telemetry - The telemetry to use for the evaluator.
*
* @returns An evaluator function that takes a {@link ToolResponseHandlingEvaluationRecord}
* and returns a classification result indicating whether the tool response handling
* is correct or incorrect.
*
* @example
* ```ts
* const evaluator = createToolResponseHandlingEvaluator({ model: openai("gpt-4o-mini") });
*
* // Example: Correct extraction from tool result
* const result = await evaluator.evaluate({
* input: "What's the weather in Seattle?",
* toolCall: 'get_weather(location="Seattle")',
* toolResult: JSON.stringify({
* temperature: 58,
* unit: "fahrenheit",
* conditions: "partly cloudy"
* }),
* output: "The weather in Seattle is 58°F and partly cloudy."
* });
* console.log(result.label); // "correct"
*
* // Example: Hallucinated data (incorrect)
* const resultHallucinated = await evaluator.evaluate({
* input: "What restaurants are nearby?",
* toolCall: 'search_restaurants(location="downtown")',
* toolResult: JSON.stringify({
* results: [{ name: "Cafe Luna", rating: 4.2 }]
* }),
* output: "I found Cafe Luna (4.2 stars) and Mario's Italian (4.8 stars) nearby."
* });
* console.log(resultHallucinated.label); // "incorrect" - Mario's was hallucinated
* ```
*/
function createToolResponseHandlingEvaluator(args) {
const { choices = default_templates_1.TOOL_RESPONSE_HANDLING_CLASSIFICATION_EVALUATOR_CONFIG.choices, promptTemplate = default_templates_1.TOOL_RESPONSE_HANDLING_CLASSIFICATION_EVALUATOR_CONFIG.template, optimizationDirection = default_templates_1.TOOL_RESPONSE_HANDLING_CLASSIFICATION_EVALUATOR_CONFIG.optimizationDirection, name = default_templates_1.TOOL_RESPONSE_HANDLING_CLASSIFICATION_EVALUATOR_CONFIG.name } = args, rest = __rest(args, ["choices", "promptTemplate", "optimizationDirection", "name"]);
return (0, createClassificationEvaluator_1.createClassificationEvaluator)(Object.assign(Object.assign({}, rest), { promptTemplate,
choices,
optimizationDirection,
name }));
}
//# sourceMappingURL=createToolResponseHandlingEvaluator.js.map