UNPKG

pi-decider

Version:

Decision backends for pi and omp — TypeSafe Jev, OpenRouter's Decisions API, or an OpenAI-compatible chat proxy — exposed as one typed tool (noul / choice / score)

341 lines (324 loc) • 12.9 kB
/** * `chat` transport: proxy the same typed questions to an OpenAI-compatible chat model, and own the * contract around it — fixed prompt shape, JSON-only output, then strict validation in `conformAnswers` * with one repair round trip. * * This backend exists for endpoints that do not serve Jev at all. A proxy answer is a best-effort * judgement from a general model, not a calibrated System One answer: results are labelled with the * backend that produced them, and `confidence` is only relayed when the model actually returned one. */ import { DEFAULT_LLM_MAX_TOKENS, joinUrl, type BackendConfig } from "./config.ts"; import { JevError } from "./errors.ts"; import { parseJsonBody, requestWithRetry } from "./http.ts"; import type { JevQuestion, JsonValue } from "./questions.ts"; import type { TokenCounts } from "./usage.ts"; export interface ChatMessage { role: "system" | "user" | "assistant"; content: string; } export interface ChatResult { text: string; model: string; tokens: TokenCounts; /** Present only when the endpoint reported a cost (OpenRouter does for many routes). */ costUsd?: number; endpoint: string; latencyMs: number; notes: string[]; } const DECISION_SYSTEM_PROMPT = [ "You are a decision engine, not an assistant. You never converse, explain, or add commentary.", "Evaluate STATE and answer every entry in QUESTIONS.", "", "Return exactly one JSON object and nothing else:", '{"answers":{"<question id>":<answer>, ...}}', "", "Answer shapes:", '- noul: {"type":"noul","noul":<number 0..1>} // probability the answer is yes', '- choice: {"type":"choice","choice":"<one option key>","probabilities":{"<option key>":<number>, ...},"confidence":<number 0..1>}', '- score: {"type":"score","score":<number>,"probabilities":{"<level index>":<number>, ...},"confidence":<number 0..1>}', "", "Rules:", "- Answer every question id exactly once, keyed as given. Add no other ids or keys.", "- probabilities must list every option or level index exactly once and sum to 1.", "- choice must be exactly one of the listed option keys.", "- score is 0-indexed over the listed levels and may fall between two levels.", "- confidence is your certainty in 0..1; omit it rather than guessing.", "- Judge only from STATE. Do not invent facts or use outside knowledge.", "- Output raw JSON: no markdown fences, no prose before or after.", ].join("\n"); /** Build the system+user turns handed to a chat backend. */ export function buildDecisionMessages( state: JsonValue, ids: string[], questions: Record<string, JevQuestion>, ): ChatMessage[] { const blocks = ids.map((id) => { const question = questions[id]; const lines = [`${id}: ${question.type}`, ` instructions: ${renderInline(question.instructions)}`]; if (question.type === "noul") { const criteria = question.criteria; if (criteria?.true !== undefined) lines.push(` true means: ${criteria.true}`); if (criteria?.false !== undefined) lines.push(` false means: ${criteria.false}`); } else if (question.type === "choice") { lines.push( ` options: ${Object.entries(question.criteria) .map(([option, rubric]) => (rubric === null ? option : `${option} — ${rubric}`)) .join(" | ")}`, ); } else { lines.push(` levels: ${question.criteria.map((level, index) => `${index}: ${level}`).join(" | ")}`); } return lines.join("\n"); }); const stateText = typeof state === "string" ? state : JSON.stringify(state, null, 2); return [ { role: "system", content: DECISION_SYSTEM_PROMPT }, { role: "user", content: `STATE:\n${stateText}\n\nQUESTIONS:\n${blocks.join("\n")}` }, ]; } /** Append the rejected answer plus the validation failures, and ask for a corrected object. */ export function buildRepairMessages(messages: ChatMessage[], previousText: string, issues: string[]): ChatMessage[] { return [ ...messages, { role: "assistant", content: previousText.slice(0, 4_000) }, { role: "user", content: "That answer was rejected by schema validation:\n" + issues.map((issue) => `- ${issue}`).join("\n") + "\nReturn the corrected JSON object only.", }, ]; } /** Read the answer object out of a model reply that may be fenced or wrapped in prose. */ export function extractAnswerObject(text: string): unknown { const trimmed = text.trim().replace(/^```(?:json)?\s*/i, "").replace(/```$/i, "").trim(); try { return JSON.parse(trimmed) as unknown; } catch { // Fall through to scanning for the first balanced object. } const start = trimmed.indexOf("{"); if (start < 0) return undefined; let depth = 0; let inString = false; let escaped = false; for (let index = start; index < trimmed.length; index += 1) { const char = trimmed[index]; if (inString) { if (escaped) escaped = false; else if (char === "\\") escaped = true; else if (char === '"') inString = false; continue; } if (char === '"') inString = true; else if (char === "{") depth += 1; else if (char === "}") { depth -= 1; if (depth === 0) { try { return JSON.parse(trimmed.slice(start, index + 1)) as unknown; } catch { return undefined; } } } } return undefined; } /** Thrown when a chat model burns its whole output budget without emitting any answer. */ export class ChatBudgetError extends JevError { /** The budget that was exhausted. */ readonly maxTokens: number; /** Reasoning tokens inside that budget, when the endpoint reports them. */ readonly reasoningTokens: number | undefined; /** Tokens the failed attempt was billed for. */ readonly tokens: TokenCounts; readonly costUsd: number | undefined; constructor( message: string, options: { maxTokens: number; reasoningTokens?: number; tokens: TokenCounts; costUsd?: number; hint?: string; body?: string; }, ) { super(message, { hint: options.hint, body: options.body }); this.name = "ChatBudgetError"; this.maxTokens = options.maxTokens; this.reasoningTokens = options.reasoningTokens; this.tokens = options.tokens; this.costUsd = options.costUsd; } } export async function askChat( backend: BackendConfig, model: string, messages: ChatMessage[], signal?: AbortSignal, ): Promise<ChatResult> { if (!backend.apiKey) { throw new JevError(`Backend "${backend.id}" has no API key.`, { hint: "Set llm.apiKey in the config file, or export DECIDER_LLM_API_KEY / OPENROUTER_API_KEY.", }); } const url = joinUrl(backend.baseUrl, backend.path); const notes: string[] = []; const started = Date.now(); let response = await postChat(backend, url, model, messages, backend.jsonMode ?? "json_object", signal); if (response.failedJsonMode) { notes.push("endpoint rejected response_format=json_object; retried without it"); response = await postChat(backend, url, model, messages, "none", signal); } const parsed = parseJsonBody(response.text, url); if (parsed === null || typeof parsed !== "object" || Array.isArray(parsed)) { throw new JevError(`${url} returned an unexpected payload (expected a JSON object).`); } const body = parsed as Record<string, unknown>; const usage = (body.usage ?? {}) as Record<string, unknown>; const tokens: TokenCounts = { input: numberOrUndefined(usage.prompt_tokens ?? usage.input_tokens), output: numberOrUndefined(usage.completion_tokens ?? usage.output_tokens), }; const cost = usage.cost; const costUsd = typeof cost === "number" && Number.isFinite(cost) ? cost : undefined; const choices = body.choices; const first = Array.isArray(choices) && choices.length > 0 ? choices[0] : undefined; const choice = first !== null && typeof first === "object" ? (first as Record<string, unknown>) : undefined; const finishReason = typeof choice?.finish_reason === "string" ? choice.finish_reason : undefined; const text = readContent(choice?.message); if (text.trim() === "") { const maxTokens = backend.maxTokens ?? DEFAULT_LLM_MAX_TOKENS; // A reasoning model can spend the entire budget thinking and never start the JSON object. if (finishReason === "length") { const details = usage.completion_tokens_details ?? usage.completionTokensDetails; const reasoningTokens = details !== null && typeof details === "object" ? numberOrUndefined( (details as Record<string, unknown>).reasoning_tokens ?? (details as Record<string, unknown>).reasoningTokens, ) : undefined; throw new ChatBudgetError( `${url} produced no answer: the model used all ${maxTokens} output tokens` + (reasoningTokens === undefined ? "" : ` (${reasoningTokens} of them on reasoning)`) + " without writing the JSON object", { maxTokens, reasoningTokens, tokens, costUsd, body: response.text, hint: "Raise llm.maxTokens, or route these questions to a decisions backend (typesafe/openrouter).", }, ); } throw new JevError(`${url} returned an empty completion (finish_reason=${finishReason ?? "unknown"}).`, { body: response.text, hint: "Check the model id and the endpoint's reply; a filtered or truncated response carries no answer.", }); } return { text, model: typeof body.model === "string" && body.model.trim() !== "" ? body.model : model, tokens, costUsd, endpoint: url, latencyMs: Date.now() - started, notes, }; } /** Catalogue lookup for chat backends: OpenRouter and most OpenAI-compatible servers expose GET /models. */ export async function listChatCatalog( backend: BackendConfig, modelsPath: string, signal?: AbortSignal, ): Promise<string[]> { const url = joinUrl(backend.baseUrl, modelsPath); const response = await requestWithRetry({ url, method: "GET", headers: backend.apiKey ? { authorization: `Bearer ${backend.apiKey}` } : {}, timeoutMs: Math.min(backend.timeoutMs, 20_000), maxRetries: 0, signal, }); const parsed = parseJsonBody(response.text, url); const entries = parsed !== null && typeof parsed === "object" && !Array.isArray(parsed) ? ((parsed as Record<string, unknown>).data ?? (parsed as Record<string, unknown>).models) : undefined; const names: string[] = []; if (Array.isArray(entries)) { for (const entry of entries) { if (typeof entry === "string") names.push(entry); else if (entry !== null && typeof entry === "object") { const record = entry as Record<string, unknown>; const name = typeof record.id === "string" ? record.id : typeof record.name === "string" ? record.name : undefined; if (name !== undefined) names.push(name); } } } return names; } interface ChatResponse { text: string; failedJsonMode: boolean; } async function postChat( backend: BackendConfig, url: string, model: string, messages: ChatMessage[], jsonMode: "json_object" | "none", signal: AbortSignal | undefined, ): Promise<ChatResponse> { try { const response = await requestWithRetry({ url, method: "POST", headers: { authorization: `Bearer ${backend.apiKey}` }, body: { model, messages, temperature: backend.temperature ?? 0, max_tokens: backend.maxTokens ?? DEFAULT_LLM_MAX_TOKENS, ...(jsonMode === "json_object" ? { response_format: { type: "json_object" } } : {}), }, timeoutMs: backend.timeoutMs, maxRetries: backend.maxRetries, signal, }); return { text: response.text, failedJsonMode: false }; } catch (error) { if (jsonMode === "json_object" && error instanceof JevError && error.status === 400) { return { text: "", failedJsonMode: true }; } throw error; } } /** Some providers return `{type:"text",text}` parts instead of a plain string. */ function readContent(message: unknown): string { if (message === null || typeof message !== "object") return ""; const content = (message as Record<string, unknown>).content; if (typeof content === "string") return content; if (!Array.isArray(content)) return ""; return content .map((part) => part !== null && typeof part === "object" && typeof (part as Record<string, unknown>).text === "string" ? ((part as Record<string, unknown>).text as string) : "", ) .join(""); } function renderInline(value: unknown): string { if (typeof value === "string") return value; return JSON.stringify(value); } function numberOrUndefined(value: unknown): number | undefined { return typeof value === "number" && Number.isFinite(value) ? value : undefined; }