pi-decider
Version:
Decision backends for pi and omp — TypeSafe Jev, OpenRouter's Decisions API, or an OpenAI-compatible chat proxy — exposed as one typed tool (noul / choice / score)
341 lines (324 loc) • 12.9 kB
text/typescript
/**
* `chat` transport: proxy the same typed questions to an OpenAI-compatible chat model, and own the
* contract around it — fixed prompt shape, JSON-only output, then strict validation in `conformAnswers`
* with one repair round trip.
*
* This backend exists for endpoints that do not serve Jev at all. A proxy answer is a best-effort
* judgement from a general model, not a calibrated System One answer: results are labelled with the
* backend that produced them, and `confidence` is only relayed when the model actually returned one.
*/
import { DEFAULT_LLM_MAX_TOKENS, joinUrl, type BackendConfig } from "./config.ts";
import { JevError } from "./errors.ts";
import { parseJsonBody, requestWithRetry } from "./http.ts";
import type { JevQuestion, JsonValue } from "./questions.ts";
import type { TokenCounts } from "./usage.ts";
export interface ChatMessage {
role: "system" | "user" | "assistant";
content: string;
}
export interface ChatResult {
text: string;
model: string;
tokens: TokenCounts;
/** Present only when the endpoint reported a cost (OpenRouter does for many routes). */
costUsd?: number;
endpoint: string;
latencyMs: number;
notes: string[];
}
const DECISION_SYSTEM_PROMPT = [
"You are a decision engine, not an assistant. You never converse, explain, or add commentary.",
"Evaluate STATE and answer every entry in QUESTIONS.",
"",
"Return exactly one JSON object and nothing else:",
'{"answers":{"<question id>":<answer>, ...}}',
"",
"Answer shapes:",
'- noul: {"type":"noul","noul":<number 0..1>} // probability the answer is yes',
'- choice: {"type":"choice","choice":"<one option key>","probabilities":{"<option key>":<number>, ...},"confidence":<number 0..1>}',
'- score: {"type":"score","score":<number>,"probabilities":{"<level index>":<number>, ...},"confidence":<number 0..1>}',
"",
"Rules:",
"- Answer every question id exactly once, keyed as given. Add no other ids or keys.",
"- probabilities must list every option or level index exactly once and sum to 1.",
"- choice must be exactly one of the listed option keys.",
"- score is 0-indexed over the listed levels and may fall between two levels.",
"- confidence is your certainty in 0..1; omit it rather than guessing.",
"- Judge only from STATE. Do not invent facts or use outside knowledge.",
"- Output raw JSON: no markdown fences, no prose before or after.",
].join("\n");
/** Build the system+user turns handed to a chat backend. */
export function buildDecisionMessages(
state: JsonValue,
ids: string[],
questions: Record<string, JevQuestion>,
): ChatMessage[] {
const blocks = ids.map((id) => {
const question = questions[id];
const lines = [`${id}: ${question.type}`, ` instructions: ${renderInline(question.instructions)}`];
if (question.type === "noul") {
const criteria = question.criteria;
if (criteria?.true !== undefined) lines.push(` true means: ${criteria.true}`);
if (criteria?.false !== undefined) lines.push(` false means: ${criteria.false}`);
} else if (question.type === "choice") {
lines.push(
` options: ${Object.entries(question.criteria)
.map(([option, rubric]) => (rubric === null ? option : `${option} — ${rubric}`))
.join(" | ")}`,
);
} else {
lines.push(` levels: ${question.criteria.map((level, index) => `${index}: ${level}`).join(" | ")}`);
}
return lines.join("\n");
});
const stateText = typeof state === "string" ? state : JSON.stringify(state, null, 2);
return [
{ role: "system", content: DECISION_SYSTEM_PROMPT },
{ role: "user", content: `STATE:\n${stateText}\n\nQUESTIONS:\n${blocks.join("\n")}` },
];
}
/** Append the rejected answer plus the validation failures, and ask for a corrected object. */
export function buildRepairMessages(messages: ChatMessage[], previousText: string, issues: string[]): ChatMessage[] {
return [
...messages,
{ role: "assistant", content: previousText.slice(0, 4_000) },
{
role: "user",
content:
"That answer was rejected by schema validation:\n" +
issues.map((issue) => `- ${issue}`).join("\n") +
"\nReturn the corrected JSON object only.",
},
];
}
/** Read the answer object out of a model reply that may be fenced or wrapped in prose. */
export function extractAnswerObject(text: string): unknown {
const trimmed = text.trim().replace(/^```(?:json)?\s*/i, "").replace(/```$/i, "").trim();
try {
return JSON.parse(trimmed) as unknown;
} catch {
// Fall through to scanning for the first balanced object.
}
const start = trimmed.indexOf("{");
if (start < 0) return undefined;
let depth = 0;
let inString = false;
let escaped = false;
for (let index = start; index < trimmed.length; index += 1) {
const char = trimmed[index];
if (inString) {
if (escaped) escaped = false;
else if (char === "\\") escaped = true;
else if (char === '"') inString = false;
continue;
}
if (char === '"') inString = true;
else if (char === "{") depth += 1;
else if (char === "}") {
depth -= 1;
if (depth === 0) {
try {
return JSON.parse(trimmed.slice(start, index + 1)) as unknown;
} catch {
return undefined;
}
}
}
}
return undefined;
}
/** Thrown when a chat model burns its whole output budget without emitting any answer. */
export class ChatBudgetError extends JevError {
/** The budget that was exhausted. */
readonly maxTokens: number;
/** Reasoning tokens inside that budget, when the endpoint reports them. */
readonly reasoningTokens: number | undefined;
/** Tokens the failed attempt was billed for. */
readonly tokens: TokenCounts;
readonly costUsd: number | undefined;
constructor(
message: string,
options: {
maxTokens: number;
reasoningTokens?: number;
tokens: TokenCounts;
costUsd?: number;
hint?: string;
body?: string;
},
) {
super(message, { hint: options.hint, body: options.body });
this.name = "ChatBudgetError";
this.maxTokens = options.maxTokens;
this.reasoningTokens = options.reasoningTokens;
this.tokens = options.tokens;
this.costUsd = options.costUsd;
}
}
export async function askChat(
backend: BackendConfig,
model: string,
messages: ChatMessage[],
signal?: AbortSignal,
): Promise<ChatResult> {
if (!backend.apiKey) {
throw new JevError(`Backend "${backend.id}" has no API key.`, {
hint: "Set llm.apiKey in the config file, or export DECIDER_LLM_API_KEY / OPENROUTER_API_KEY.",
});
}
const url = joinUrl(backend.baseUrl, backend.path);
const notes: string[] = [];
const started = Date.now();
let response = await postChat(backend, url, model, messages, backend.jsonMode ?? "json_object", signal);
if (response.failedJsonMode) {
notes.push("endpoint rejected response_format=json_object; retried without it");
response = await postChat(backend, url, model, messages, "none", signal);
}
const parsed = parseJsonBody(response.text, url);
if (parsed === null || typeof parsed !== "object" || Array.isArray(parsed)) {
throw new JevError(`${url} returned an unexpected payload (expected a JSON object).`);
}
const body = parsed as Record<string, unknown>;
const usage = (body.usage ?? {}) as Record<string, unknown>;
const tokens: TokenCounts = {
input: numberOrUndefined(usage.prompt_tokens ?? usage.input_tokens),
output: numberOrUndefined(usage.completion_tokens ?? usage.output_tokens),
};
const cost = usage.cost;
const costUsd = typeof cost === "number" && Number.isFinite(cost) ? cost : undefined;
const choices = body.choices;
const first = Array.isArray(choices) && choices.length > 0 ? choices[0] : undefined;
const choice = first !== null && typeof first === "object" ? (first as Record<string, unknown>) : undefined;
const finishReason = typeof choice?.finish_reason === "string" ? choice.finish_reason : undefined;
const text = readContent(choice?.message);
if (text.trim() === "") {
const maxTokens = backend.maxTokens ?? DEFAULT_LLM_MAX_TOKENS;
// A reasoning model can spend the entire budget thinking and never start the JSON object.
if (finishReason === "length") {
const details = usage.completion_tokens_details ?? usage.completionTokensDetails;
const reasoningTokens =
details !== null && typeof details === "object"
? numberOrUndefined(
(details as Record<string, unknown>).reasoning_tokens ??
(details as Record<string, unknown>).reasoningTokens,
)
: undefined;
throw new ChatBudgetError(
`${url} produced no answer: the model used all ${maxTokens} output tokens` +
(reasoningTokens === undefined ? "" : ` (${reasoningTokens} of them on reasoning)`) +
" without writing the JSON object",
{
maxTokens,
reasoningTokens,
tokens,
costUsd,
body: response.text,
hint: "Raise llm.maxTokens, or route these questions to a decisions backend (typesafe/openrouter).",
},
);
}
throw new JevError(`${url} returned an empty completion (finish_reason=${finishReason ?? "unknown"}).`, {
body: response.text,
hint: "Check the model id and the endpoint's reply; a filtered or truncated response carries no answer.",
});
}
return {
text,
model: typeof body.model === "string" && body.model.trim() !== "" ? body.model : model,
tokens,
costUsd,
endpoint: url,
latencyMs: Date.now() - started,
notes,
};
}
/** Catalogue lookup for chat backends: OpenRouter and most OpenAI-compatible servers expose GET /models. */
export async function listChatCatalog(
backend: BackendConfig,
modelsPath: string,
signal?: AbortSignal,
): Promise<string[]> {
const url = joinUrl(backend.baseUrl, modelsPath);
const response = await requestWithRetry({
url,
method: "GET",
headers: backend.apiKey ? { authorization: `Bearer ${backend.apiKey}` } : {},
timeoutMs: Math.min(backend.timeoutMs, 20_000),
maxRetries: 0,
signal,
});
const parsed = parseJsonBody(response.text, url);
const entries =
parsed !== null && typeof parsed === "object" && !Array.isArray(parsed)
? ((parsed as Record<string, unknown>).data ?? (parsed as Record<string, unknown>).models)
: undefined;
const names: string[] = [];
if (Array.isArray(entries)) {
for (const entry of entries) {
if (typeof entry === "string") names.push(entry);
else if (entry !== null && typeof entry === "object") {
const record = entry as Record<string, unknown>;
const name = typeof record.id === "string" ? record.id : typeof record.name === "string" ? record.name : undefined;
if (name !== undefined) names.push(name);
}
}
}
return names;
}
interface ChatResponse {
text: string;
failedJsonMode: boolean;
}
async function postChat(
backend: BackendConfig,
url: string,
model: string,
messages: ChatMessage[],
jsonMode: "json_object" | "none",
signal: AbortSignal | undefined,
): Promise<ChatResponse> {
try {
const response = await requestWithRetry({
url,
method: "POST",
headers: { authorization: `Bearer ${backend.apiKey}` },
body: {
model,
messages,
temperature: backend.temperature ?? 0,
max_tokens: backend.maxTokens ?? DEFAULT_LLM_MAX_TOKENS,
...(jsonMode === "json_object" ? { response_format: { type: "json_object" } } : {}),
},
timeoutMs: backend.timeoutMs,
maxRetries: backend.maxRetries,
signal,
});
return { text: response.text, failedJsonMode: false };
} catch (error) {
if (jsonMode === "json_object" && error instanceof JevError && error.status === 400) {
return { text: "", failedJsonMode: true };
}
throw error;
}
}
/** Some providers return `{type:"text",text}` parts instead of a plain string. */
function readContent(message: unknown): string {
if (message === null || typeof message !== "object") return "";
const content = (message as Record<string, unknown>).content;
if (typeof content === "string") return content;
if (!Array.isArray(content)) return "";
return content
.map((part) =>
part !== null && typeof part === "object" && typeof (part as Record<string, unknown>).text === "string"
? ((part as Record<string, unknown>).text as string)
: "",
)
.join("");
}
function renderInline(value: unknown): string {
if (typeof value === "string") return value;
return JSON.stringify(value);
}
function numberOrUndefined(value: unknown): number | undefined {
return typeof value === "number" && Number.isFinite(value) ? value : undefined;
}