UNPKG

langchain

Version:
197 lines (196 loc) 7.73 kB
import { AgentMiddleware } from "../../types.js"; import { z } from "zod/v3"; import { InferInteropZodInput } from "@langchain/core/utils/types"; //#region src/agents/middleware/provider/anthropic/promptCaching.d.ts declare const contextSchema: z.ZodObject<{ /** * Whether to enable prompt caching. * @default true */ enableCaching: z.ZodOptional<z.ZodBoolean>; /** * The time-to-live for the cached prompt. * @default "5m" */ ttl: z.ZodOptional<z.ZodEnum<["5m", "1h"]>>; /** * The minimum number of messages required before caching is applied. * @default 3 */ minMessagesToCache: z.ZodOptional<z.ZodNumber>; /** * The behavior to take when an unsupported model is used. * - "ignore" will ignore the unsupported model and continue without caching. * - "warn" will warn the user and continue without caching. * - "raise" will raise an error and stop the agent. * @default "warn" */ unsupportedModelBehavior: z.ZodOptional<z.ZodEnum<["ignore", "warn", "raise"]>>; }, "strip", z.ZodTypeAny, { enableCaching?: boolean | undefined; ttl?: "1h" | "5m" | undefined; minMessagesToCache?: number | undefined; unsupportedModelBehavior?: "ignore" | "raise" | "warn" | undefined; }, { enableCaching?: boolean | undefined; ttl?: "1h" | "5m" | undefined; minMessagesToCache?: number | undefined; unsupportedModelBehavior?: "ignore" | "raise" | "warn" | undefined; }>; type PromptCachingMiddlewareConfig = Partial<InferInteropZodInput<typeof contextSchema>>; /** * Creates a prompt caching middleware for Anthropic models to optimize API usage. * * This middleware automatically adds cache control headers to the last messages when using Anthropic models, * enabling their prompt caching feature. This can significantly reduce costs for applications with repetitive * prompts, long system messages, or extensive conversation histories. * * ## How It Works * * The middleware intercepts model requests and adds cache control metadata that tells Anthropic's * API to cache processed prompt prefixes. On subsequent requests with matching prefixes, the * cached representations are reused, skipping redundant token processing. * * ## Benefits * * - **Cost Reduction**: Avoid reprocessing the same tokens repeatedly (up to 90% savings on cached portions) * - **Lower Latency**: Cached prompts are processed faster as embeddings are pre-computed * - **Better Scalability**: Reduced computational load enables handling more requests * - **Consistent Performance**: Stable response times for repetitive queries * * @param middlewareOptions - Configuration options for the caching behavior * @param middlewareOptions.enableCaching - Whether to enable prompt caching (default: `true`) * @param middlewareOptions.ttl - Cache time-to-live: `"5m"` for 5 minutes or `"1h"` for 1 hour (default: `"5m"`) * @param middlewareOptions.minMessagesToCache - Minimum number of messages required before caching is applied (default: `3`) * @param middlewareOptions.unsupportedModelBehavior - The behavior to take when an unsupported model is used (default: `"warn"`) * * @returns A middleware instance that can be passed to `createAgent` * * @throws {Error} If used with non-Anthropic models * * @example * Basic usage with default settings * ```typescript * import { createAgent } from "langchain"; * import { anthropicPromptCachingMiddleware } from "langchain"; * * const agent = createAgent({ * model: "anthropic:claude-3-5-sonnet", * middleware: [ * anthropicPromptCachingMiddleware() * ] * }); * ``` * * @example * Custom configuration for longer conversations * ```typescript * const cachingMiddleware = anthropicPromptCachingMiddleware({ * ttl: "1h", // Cache for 1 hour instead of default 5 minutes * minMessagesToCache: 5 // Only cache after 5 messages * }); * * const agent = createAgent({ * model: "anthropic:claude-3-5-sonnet", * systemPrompt: "You are a helpful assistant with deep knowledge of...", // Long system prompt * middleware: [cachingMiddleware] * }); * ``` * * @example * Conditional caching based on runtime context * ```typescript * const agent = createAgent({ * model: "anthropic:claude-3-5-sonnet", * middleware: [ * anthropicPromptCachingMiddleware({ * enableCaching: true, * ttl: "5m" * }) * ] * }); * * // Disable caching for specific requests * await agent.invoke( * { messages: [new HumanMessage("Process this without caching")] }, * { * configurable: { * middleware_context: { enableCaching: false } * } * } * ); * ``` * * @example * Optimal setup for customer support chatbot * ```typescript * const supportAgent = createAgent({ * model: "anthropic:claude-3-5-sonnet", * systemPrompt: `You are a customer support agent for ACME Corp. * * Company policies: * - Always be polite and professional * - Refer to knowledge base for product information * - Escalate billing issues to human agents * ... (extensive policies and guidelines) * `, * tools: [searchKnowledgeBase, createTicket, checkOrderStatus], * middleware: [ * anthropicPromptCachingMiddleware({ * ttl: "1h", // Long TTL for stable system prompt * minMessagesToCache: 1 // Cache immediately due to large system prompt * }) * ] * }); * ``` * * @remarks * - **Anthropic Only**: This middleware only works with Anthropic models and will throw an error if used with other providers * - **Automatic Application**: Caching is applied automatically when message count exceeds `minMessagesToCache` * - **Cache Scope**: Caches are isolated per API key and cannot be shared across different keys * - **TTL Options**: Only supports "5m" (5 minutes) and "1h" (1 hour) as TTL values per Anthropic's API * - **Best Use Cases**: Long system prompts, multi-turn conversations, repetitive queries, RAG applications * - **Cost Impact**: Cached tokens are billed at 10% of the base input token price, cache writes are billed at 25% of the base * * @see {@link createAgent} for agent creation * @see {@link https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching} Anthropic's prompt caching documentation * @public */ declare function anthropicPromptCachingMiddleware(middlewareOptions?: PromptCachingMiddlewareConfig): AgentMiddleware<undefined, z.ZodObject<{ /** * Whether to enable prompt caching. * @default true */ enableCaching: z.ZodOptional<z.ZodBoolean>; /** * The time-to-live for the cached prompt. * @default "5m" */ ttl: z.ZodOptional<z.ZodEnum<["5m", "1h"]>>; /** * The minimum number of messages required before caching is applied. * @default 3 */ minMessagesToCache: z.ZodOptional<z.ZodNumber>; /** * The behavior to take when an unsupported model is used. * - "ignore" will ignore the unsupported model and continue without caching. * - "warn" will warn the user and continue without caching. * - "raise" will raise an error and stop the agent. * @default "warn" */ unsupportedModelBehavior: z.ZodOptional<z.ZodEnum<["ignore", "warn", "raise"]>>; }, "strip", z.ZodTypeAny, { enableCaching?: boolean | undefined; ttl?: "1h" | "5m" | undefined; minMessagesToCache?: number | undefined; unsupportedModelBehavior?: "ignore" | "raise" | "warn" | undefined; }, { enableCaching?: boolean | undefined; ttl?: "1h" | "5m" | undefined; minMessagesToCache?: number | undefined; unsupportedModelBehavior?: "ignore" | "raise" | "warn" | undefined; }>, any>; //#endregion export { PromptCachingMiddlewareConfig, anthropicPromptCachingMiddleware }; //# sourceMappingURL=promptCaching.d.ts.map