pi-decider
Version:
Decision backends for pi and omp — TypeSafe Jev, OpenRouter's Decisions API, or an OpenAI-compatible chat proxy — exposed as one typed tool (noul / choice / score)
230 lines (211 loc) • 8.43 kB
text/typescript
/**
* Question dispatch shared by the tool and the command surface: group questions by backend+model,
* run each batch in parallel, and conform the answers. No harness imports, so it stays testable.
*/
import { askChat, buildDecisionMessages, buildRepairMessages, ChatBudgetError, extractAnswerObject, type ChatResult } from "./chat.ts";
import { DEFAULT_LLM_MAX_TOKENS, selectBackend, type BackendConfig, type BackendId, type DeciderConfig } from "./config.ts";
import { askDecisions } from "./decisions.ts";
import { JevError, messageOf } from "./errors.ts";
import type { GroupMeta } from "./format.ts";
import {
conformAnswers,
type ConformResult,
type JevQuestion,
type JsonValue,
type QuestionRoute,
} from "./questions.ts";
import { inputCostUsd, type TokenCounts } from "./usage.ts";
export interface Group {
backend: BackendConfig;
model: string;
ids: string[];
}
export interface GroupOutcome {
conform: ConformResult;
meta: GroupMeta;
tokens: TokenCounts;
costUsd: number | undefined;
}
export interface GroupPlan {
groups: Group[];
/** Questions whose own backend/model route did not resolve; key = question id, value = the reason. */
failures: Record<string, string>;
}
/** How far one retry may raise a chat backend's output budget after a model exhausts it. */
const BUDGET_ESCALATION_FACTOR = 4;
const BUDGET_ESCALATION_CAP = 16_384;
/**
* Group questions by resolved backend+model, preserving request order inside each group.
*
* A question that names an unusable backend (or a chat backend without a model) becomes an entry in
* `failures` so the rest of the call still runs; only the call-level default throws, because then
* there is no other question to fall back to.
*/
export function buildGroups(
cfg: DeciderConfig,
ids: string[],
routes: Record<string, QuestionRoute>,
callBackend?: BackendId,
callModel?: string,
): GroupPlan {
const groups = new Map<string, Group>();
const failures: Record<string, string> = {};
const needsDefault = ids.some((id) => routes[id]?.backend === undefined);
// An unusable call-level default has no other question to fall back to, so it still throws here.
const fallback = needsDefault ? selectBackend(cfg, callBackend) : undefined;
for (const id of ids) {
const route = routes[id] ?? {};
let backend = fallback;
if (route.backend !== undefined) {
try {
backend = selectBackend(cfg, route.backend);
} catch (error) {
failures[id] = messageOf(error);
continue;
}
}
if (backend === undefined) continue;
const model = route.model ?? callModel ?? backend.model;
if (model === undefined || model.trim() === "") {
const noModel = `No model id for backend "${backend.id}".`;
if (route.backend === undefined) {
throw new JevError(noModel, {
hint: `Set ${backend.id}.model in ${cfg.configPath}, pass "model" in the call, or set model on each question.`,
});
}
failures[id] = noModel;
continue;
}
const key = `${backend.id}\u0000${model}`;
const existing = groups.get(key);
if (existing === undefined) groups.set(key, { backend, model, ids: [id] });
else existing.ids.push(id);
}
return { groups: [...groups.values()], failures };
}
/** Run one backend batch and conform its answers. */
export async function runGroup(
group: Group,
state: JsonValue,
questions: Record<string, JevQuestion>,
signal?: AbortSignal,
): Promise<GroupOutcome> {
const subset: Record<string, JevQuestion> = {};
for (const id of group.ids) subset[id] = questions[id];
const { backend, model } = group;
if (backend.kind === "decisions") {
const result = await askDecisions(backend, { state, model, questions: subset }, signal);
return {
conform: conformAnswers(subset, result.answers),
meta: {
backendId: backend.id,
label: backend.label,
kind: backend.kind,
requestedModel: model,
servedModel: result.model,
provider: result.provider,
endpoint: result.endpoint,
latencyMs: result.latencyMs,
tokens: result.tokens,
costUsd: inputCostUsd(result.tokens, backend.costPerMTokInput, backend.costPerMTokOutput),
ids: group.ids,
},
tokens: result.tokens,
costUsd: inputCostUsd(result.tokens, backend.costPerMTokInput, backend.costPerMTokOutput),
};
}
const messages = buildDecisionMessages(state, group.ids, subset);
let tokens: TokenCounts = { input: 0, output: 0 };
let reportedCostUsd: number | undefined;
const notes: string[] = [];
let latencyMs = 0;
let chat: ChatResult;
try {
chat = await askChat(backend, model, messages, signal);
} catch (error) {
// A reasoning model can spend a small budget entirely on thinking: raise it once, then give up.
const currentMax = error instanceof ChatBudgetError ? error.maxTokens : 0;
const raisedMax = Math.min(currentMax * BUDGET_ESCALATION_FACTOR, BUDGET_ESCALATION_CAP);
if (!(error instanceof ChatBudgetError) || raisedMax <= currentMax) throw error;
tokens = { input: error.tokens.input ?? 0, output: error.tokens.output ?? 0 };
reportedCostUsd = error.costUsd;
notes.push(
`output budget ${currentMax} exhausted` +
(error.reasoningTokens === undefined ? "" : ` (${error.reasoningTokens} on reasoning)`) +
`; retried with ${raisedMax}`,
);
chat = await askChat({ ...backend, maxTokens: raisedMax }, model, messages, signal);
}
// Every attempt is billed, including the ones a repair or a raised budget caused.
const bill = (attempt: ChatResult) => {
tokens = {
input: (tokens.input ?? 0) + (attempt.tokens.input ?? 0),
output: (tokens.output ?? 0) + (attempt.tokens.output ?? 0),
};
if (attempt.costUsd !== undefined) reportedCostUsd = (reportedCostUsd ?? 0) + attempt.costUsd;
};
bill(chat);
latencyMs += chat.latencyMs;
notes.push(...chat.notes);
let conform = conformAnswers(subset, readAnswersMap(extractAnswerObject(chat.text)));
let repairs = 0;
const repairBudget = backend.repairAttempts ?? 0;
while (Object.keys(conform.issues).length > 0 && repairs < repairBudget) {
repairs += 1;
const issues = Object.entries(conform.issues).map(([id, message]) => `${id}: ${message}`);
chat = await askChat(backend, model, buildRepairMessages(messages, chat.text, issues), signal);
bill(chat);
latencyMs += chat.latencyMs;
notes.push(...chat.notes, `validation repair attempt ${repairs}`);
conform = conformAnswers(subset, readAnswersMap(extractAnswerObject(chat.text)));
}
const costUsd = reportedCostUsd ?? configuredChatCost(backend, tokens);
return {
conform,
meta: {
backendId: backend.id,
label: backend.label,
kind: backend.kind,
requestedModel: model,
servedModel: chat.model,
endpoint: chat.endpoint,
latencyMs,
tokens,
costUsd,
ids: group.ids,
notes: [...new Set(notes)],
},
tokens,
costUsd,
};
}
/** Chat endpoints rarely report cost; fall back to configured prices when the user set any. */
function configuredChatCost(backend: BackendConfig, tokens: TokenCounts): number | undefined {
if (backend.costPerMTokInput === 0 && backend.costPerMTokOutput === 0) return undefined;
return inputCostUsd(tokens, backend.costPerMTokInput, backend.costPerMTokOutput);
}
/** Accept `{answers:{...}}` as documented, or the bare id->answer map some models emit. */
function readAnswersMap(parsed: unknown): unknown {
if (parsed !== null && typeof parsed === "object" && !Array.isArray(parsed)) {
const answers = (parsed as Record<string, unknown>).answers;
if (answers !== null && typeof answers === "object" && !Array.isArray(answers)) return answers;
}
return parsed;
}
/** Text, or structured JSON; a JSON-encoded string is decoded when it parses. */
export function normalizeState(raw: unknown): JsonValue {
if (typeof raw === "string") {
const trimmed = raw.trim();
if (trimmed === "") throw new JevError("state must not be empty.");
if (/^[[{]/.test(trimmed)) {
try {
return JSON.parse(trimmed) as JsonValue;
} catch {
return trimmed;
}
}
return trimmed;
}
if (raw === undefined || raw === null) throw new JevError("state is required.");
return raw as JsonValue;
}