UNPKG

kestrel.markets

Version:

A typed, token-efficient language + runtime for agentic trading: agents author bounded plans, the runtime fires them at the tick. CLI + typed library + MCP server.

433 lines (405 loc) 23.5 kB
/** * # session/harness/codex-cli-client — the ZERO-CASH ChatGPT-subscription {@link LlmClient} (subprocess lane) * * A per-turn `LlmClient` that serves `gpt-5.6-sol` (and the other GPT variants) by SHELLING OUT to the * Codex CLI in its non-interactive `codex exec` surface, authenticated by the ChatGPT SUBSCRIPTION OAuth * in `$CODEX_HOME/auth.json` (`auth_mode: "chatgpt"`). No API key, no gateway, no cash: this is the * `codex-cli` lane that `./roster.ts` has declared since day one and that nothing implemented. * * ## Why this is a subprocess and not an AI-SDK provider * There is no OpenAI-compatible HTTP endpoint that accepts a ChatGPT-subscription OAuth token — the * subscription is spendable only through the Codex CLI. So this client is quarantined here rather than in * `./ai-sdk-client.ts` (which owns the SDK-backed lanes and imports the provider packages); the two are * joined by the {@link harnessLlmClient} dispatcher in `./harness-client.ts`. The `LlmClient` contract is * identical, so every existing wrapper (turn cap, timeout, retry, concurrency) composes over this lane * unchanged. * * ## Honest costs (do NOT flatter this lane) * A subprocess-per-turn lane is SLOW — a trivial completion measured **~3.8 s** end-to-end (process spawn + * OAuth + a full agent turn), against ~1 s for an HTTP provider call. {@link CodexCompletion.latencyMs} is * recorded on every call and surfaces on the Pareto latency axis, so "free" shows up as *free-but-slow*. The * CLI also injects its own agent scaffolding, so `input_tokens` carries a **~15k floor** even for a one-line * prompt — irrelevant to CASH (the subscription bills nothing marginal) but it means token counts on this * lane are NOT comparable to an API lane's. Both facts are recorded, never hidden. * * ## Hermetic by construction * The user's `~/.codex/config.toml` is real developer config — it pins `model_reasoning_effort = "ultra"` * and loads MCP servers (a browser, a node REPL, a desktop computer-use bridge). NONE of that may bleed into * a benchmark cell. So every call passes `--ignore-user-config` (auth still resolves from `CODEX_HOME`) and * re-supplies only what the benchmark declares, via explicit `-c` overrides. The sandbox is pinned * `read-only` and the model is given no tools it could use to look ahead at the tape. */ import { spawn } from "node:child_process"; import { existsSync } from "node:fs"; import { mkdtempSync, readFileSync, rmSync } from "node:fs"; import { tmpdir } from "node:os"; import { join } from "node:path"; import { cacheTtlNeverReachesWire, type AgentConfig, type LlmClient, type LlmCompletion, type LlmMessage } from "../agent.ts"; import { scrubSecrets } from "./prompt.ts"; /** The Codex CLI ships INSIDE the ChatGPT desktop app bundle — it is NOT on `PATH` and NOT an npm global * on this machine (`which codex` finds nothing; `npm ls -g` has no `@openai/codex`). The path is the one * the user's own `~/.codex/config.toml` names as `CODEX_CLI_PATH`, so it is the binary the CLI itself * expects to be run. Verified: `codex-cli 0.144.2`. */ export const CODEX_BUNDLED_PATH = "/Applications/ChatGPT.app/Contents/Resources/codex"; /** Where to look for the `codex` binary, in order: an explicit `CODEX_BIN` override, then the ChatGPT.app * bundle, then a plain `codex` on `PATH` (for a machine where it IS installed globally). */ export function codexBinCandidates(): readonly string[] { const explicit = process.env.CODEX_BIN; return [ ...(explicit !== undefined && explicit.trim() !== "" ? [explicit.trim()] : []), CODEX_BUNDLED_PATH, "codex", ]; } /** Raised when no `codex` binary can be found. Fail-closed: a missing CLI must abort the lane loudly, never * silently fall back to a CASH provider (that is exactly the regression this whole lane exists to prevent). */ export class CodexNotFoundError extends Error { constructor(message: string) { super(message); this.name = "CodexNotFoundError"; } } /** Resolve the `codex` binary to invoke. An absolute candidate must EXIST; the bare `codex` fallback is * returned unchecked (the OS resolves it on `PATH`, and a miss surfaces as a spawn ENOENT). */ export function resolveCodexBin(candidates: readonly string[] = codexBinCandidates()): string { for (const c of candidates) { if (!c.includes("/")) return c; // a bare name — let the OS resolve it on PATH if (existsSync(c)) return c; } throw new CodexNotFoundError( `no codex binary found (looked at: ${candidates.join(", ")}). Set CODEX_BIN, or install the Codex CLI. ` + `The ChatGPT desktop app bundles one at ${CODEX_BUNDLED_PATH}.`, ); } /** The `turn.completed` usage payload the CLI emits on its `--json` event stream — the ONLY on-record source * of token counts for this lane (there is no HTTP response to read). Shape verified live against * `codex-cli 0.144.2`: `{"type":"turn.completed","usage":{"input_tokens":…,"cached_input_tokens":…, * "output_tokens":…,"reasoning_output_tokens":…}}`. */ export interface CodexUsageEvent { readonly input_tokens?: number; readonly cached_input_tokens?: number; readonly output_tokens?: number; readonly reasoning_output_tokens?: number; } /** One `codex exec` invocation's raw result — the seam the tests drive. `lastMessage` is the verbatim final * assistant message (the CLI writes it to the `-o` file, so it is the exact reply bytes with no event-stream * framing to parse out of); `stdout` is the `--json` JSONL event stream. */ export interface CodexExecResult { readonly code: number; readonly stdout: string; readonly stderr: string; readonly lastMessage: string; readonly latencyMs: number; } /** The injectable subprocess seam: run `codex exec` with `args`, feeding `prompt` on STDIN. The real one * spawns the CLI; a TEST passes a stand-in so the whole client — flag construction, usage parsing, empty and * failure handling — is exercised with ZERO live calls. */ export type CodexExecFn = (args: readonly string[], prompt: string, timeoutMs: number) => Promise<CodexExecResult>; /** A completion from this lane, with the honest wall-clock the Pareto latency axis reads. */ export interface CodexCompletion extends LlmCompletion { readonly latencyMs: number; } /** The CLI's `model_reasoning_effort` values, in order. Note the dial reaches BOTH ends the OpenAI API enum * does not guarantee (`minimal` below, `ultra` above), which is why this lane gets its own 5→5 mapping * ({@link codexReasoningEffort}) rather than the SDK lanes' clamped 5→3 (`REASONING_EFFORT`). */ export type CodexReasoningEffort = "minimal" | "low" | "medium" | "high" | "ultra"; /** The strategist/watcher reasoning-effort dial, mirrored from {@link AgentConfig.thinkingLevel} (it rides * the ConfigId, so it is run identity) — same alias the SDK lane declares in `./ai-sdk-client.ts`. */ type ThinkingLevel = AgentConfig["thinkingLevel"]; /** * Map the per-request thinking dial onto the CLI's `model_reasoning_effort` knob (kestrel-0drf). * * Until this existed, `thinkingLevel` never reached the argv — the effort flag was frozen at the * CONSTRUCTION-time constant, so every codex-cli row across every thinkingLevel ran at one effort and the * "dial" compared the seat against itself (proven by the p4 injected-exec probe: normalized argv identical * across temp{0,0.7}×thinkingLevel{none,high}). * * The 5→5 granularity decision, explicit: * - `none` (the default EVERYWHERE) ⇒ the construction default — argv byte-identical to before the dial * was wired (the additive-opt-in guarantee, same as every other lane's `none`). * - `low`/`medium`/`high` map 1:1; `max` ⇒ `ultra` (the CLI's top stop — unlike the OpenAI API enum, * the CLI verifiably takes it: the user's own config pins it). * - An explicit level WINS over the construction option: the dial rides the ConfigId (run identity); * the construction option is client plumbing, kept as the `none` default so existing call sites * (author-cells' `reasoningEffort:"high"`) keep their exact behavior. * `minimal` stays reachable only as a construction default — no ThinkingLevel maps to it (a 5-level dial * cannot cover 6 meaningful stops; `none` must stay "as before", never a silent downshift). * * Pure — the fixture asserts the argv varies through the real client seam. */ export function codexReasoningEffort(level: ThinkingLevel, constructionDefault: CodexReasoningEffort): CodexReasoningEffort { switch (level) { case "none": return constructionDefault; case "low": return "low"; case "medium": return "medium"; case "high": return "high"; case "max": return "ultra"; } } export interface CodexCliClientOptions { /** The Codex model to serve, e.g. `gpt-5.6-sol`. Passed straight to `codex exec --model`. */ readonly modelId: string; /** Reasoning effort served when the request's `thinkingLevel` is `none` (the default everywhere) — an * explicit per-request thinkingLevel maps onto the CLI dial and WINS ({@link codexReasoningEffort}, * kestrel-0drf). Defaults to `low` — a benchmark cell declares its own thinking budget, and the user's * config default (`ultra`) would silently make every cell slower and richer than declared. */ readonly reasoningEffort?: CodexReasoningEffort; /** Hard per-call wall-clock ceiling (default 180s — a subprocess turn is slow; see the module doc). */ readonly timeoutMs?: number; /** Extra secret strings to scrub from captured wire evidence (belt-and-suspenders, as the SDK lane does). */ readonly secrets?: readonly string[]; /** Capture scrubbed wire evidence onto each completion (default true). */ readonly captureWire?: boolean; /** Test seam ONLY: a stand-in for the real `codex exec` subprocess. NEVER set on a production client. */ readonly exec?: CodexExecFn; /** Test seam ONLY: skip binary resolution (the injected `exec` does not need a real path). */ readonly bin?: string; } /** Default per-call ceiling. Generous relative to an HTTP lane precisely BECAUSE a subprocess turn is slow — * a 90s HTTP-tuned ceiling would false-positive on a legitimately slow Codex turn. */ export const CODEX_CALL_TIMEOUT_MS = Number(process.env.KESTREL_CODEX_TIMEOUT_MS ?? 180_000); /** * The exact `codex exec` argv this lane invokes — pure, so a test can assert the flags without spawning. * * Every flag earns its place: * `exec` — the non-interactive surface (no TTY, no approval prompts). * `--model <id>` — the model under test; this is what makes `fable`/`opus`-style model selection * work for the GPT family (`gpt-5.6-sol` and friends). * `--json` — JSONL event stream on stdout; the ONLY source of token usage on this lane. * `-o <file>` — the verbatim final assistant message, so the reply needs no de-framing. * `--ignore-user-config` — HERMETIC: do not load `~/.codex/config.toml` (which pins reasoning=ultra and * loads MCP servers). Auth still resolves from `CODEX_HOME` — the subscription * OAuth keeps working, the developer's ambient config does not leak in. * `-c model_reasoning_effort=…` — re-supply ONLY the knob the benchmark declares. * `--sandbox read-only` — the model gets no write access to the workspace. * `--skip-git-repo-check` — the harness may author from a scratch cwd. * `--ephemeral` — do not persist a session file per benchmark turn. * `--color never` — no ANSI escapes in captured evidence. * `-` — read the PROMPT FROM STDIN rather than argv. Deliberate: a rendered trading * briefing is far too large to be safe against `ARG_MAX`, and the CLI documents * `-` as the stdin form. (Passing a prompt on argv AND piping stdin would make * the CLI append stdin as a second `<stdin>` block — a silent prompt corruption.) */ export function codexExecArgs(opts: { readonly modelId: string; readonly reasoningEffort: string; readonly outFile: string }): readonly string[] { return [ "exec", "--model", opts.modelId, "--json", "-o", opts.outFile, "--ignore-user-config", "-c", `model_reasoning_effort="${opts.reasoningEffort}"`, "--sandbox", "read-only", "--skip-git-repo-check", "--ephemeral", "--color", "never", "-", ]; } /** * Flatten the harness's {system, conversation} into the ONE prompt `codex exec` accepts. * * The CLI has no separate system field, so the byte-stable system prefix is rendered as a leading block and * the reused transcript follows it verbatim, role-tagged. This preserves the harness's multi-turn contract * (the growing conversation is re-sent each wake) at the cost of re-sending the prefix as plain text — the * `cached_input_tokens` the CLI reports show the provider still caches it server-side. * * Pure — a test asserts the exact bytes. */ export function renderCodexPrompt(system: string, messages: readonly LlmMessage[]): string { const turns = messages .map((m) => `<turn role="${m.role}">\n${m.content}\n</turn>`) .join("\n\n"); return [ "<instructions>", system, "</instructions>", "", "<conversation>", turns, "</conversation>", "", "Reply with ONLY the assistant's next turn, exactly as the instructions specify. Do not use any tools, do not read or write files, and do not explain — emit the turn and nothing else.", ].join("\n"); } /** Parse the `--json` JSONL event stream for the `turn.completed` usage record. The input/output/thinking * totals return 0 when absent (a degraded but non-fatal outcome — the completion still stands; only its * counts are unknown). The cache-READ counter is preserved ABSENT-vs-REPORTED (kestrel-wa0j.19 §3): it is * present ONLY when the CLI reported `cached_input_tokens` — NEVER coalesced to 0, so the cache-liveness * detector can tell a real 0 (a miss) from "not reported" (no evidence). Pure. */ export function parseCodexUsage(stdout: string): { inputTokens: number; outputTokens: number; thinkingTokens: number; cacheReadTokens?: number } { let usage: CodexUsageEvent | undefined; for (const line of stdout.split("\n")) { const t = line.trim(); if (t === "" || !t.startsWith("{")) continue; try { const ev = JSON.parse(t) as { type?: string; usage?: CodexUsageEvent }; if (ev.type === "turn.completed" && ev.usage !== undefined) usage = ev.usage; } catch { /* not every stdout line is an event — ignore noise */ } } return { inputTokens: usage?.input_tokens ?? 0, outputTokens: usage?.output_tokens ?? 0, thinkingTokens: usage?.reasoning_output_tokens ?? 0, ...(usage?.cached_input_tokens !== undefined ? { cacheReadTokens: usage.cached_input_tokens } : {}), }; } /** The REAL subprocess seam: spawn `codex exec`, write the prompt to stdin, read the `-o` last-message file. * Kills the child on timeout so a hung CLI can never stall the fan (the wrapper's ceiling is enforced HERE, * where the process handle lives, rather than only racing a promise). */ function realExec(bin: string): CodexExecFn { return (args, prompt, timeoutMs) => new Promise<CodexExecResult>((resolve, reject) => { const started = Date.now(); // The `-o` target must be a path only THIS call owns — a fan runs many CLIs concurrently. const dir = mkdtempSync(join(tmpdir(), "kestrel-codex-")); const outFile = args[args.indexOf("-o") + 1] ?? join(dir, "last.txt"); const child = spawn(bin, [...args], { stdio: ["pipe", "pipe", "pipe"] }); let stdout = ""; let stderr = ""; let settled = false; const cleanup = (): void => { try { rmSync(dir, { recursive: true, force: true }); } catch { /* best-effort */ } }; const timer = setTimeout(() => { if (settled) return; settled = true; child.kill("SIGKILL"); cleanup(); // Worded to match `isTransientProviderError`'s `timeout` class, so the existing retry wrapper // reissues it on a fresh process rather than failing the cell outright. reject(new Error(`codex exec call timeout after ${timeoutMs}ms (hung CLI — retrying)`)); }, timeoutMs); child.stdout.on("data", (d: Buffer) => (stdout += d.toString())); child.stderr.on("data", (d: Buffer) => (stderr += d.toString())); child.on("error", (e) => { if (settled) return; settled = true; clearTimeout(timer); cleanup(); reject(e); }); child.on("close", (code) => { if (settled) return; settled = true; clearTimeout(timer); let lastMessage = ""; try { if (existsSync(outFile)) lastMessage = readFileSync(outFile, "utf8"); } catch { /* the reply file is best-effort — an empty read degrades to an empty PASS below */ } cleanup(); resolve({ code: code ?? -1, stdout, stderr, lastMessage, latencyMs: Date.now() - started }); }); child.stdin.write(prompt); child.stdin.end(); }); } /** * Construct the ZERO-CASH Codex-CLI {@link LlmClient} (the ChatGPT-subscription lane). * * Contract parity with `aiSdkLlmClient`: given {system, messages, sampling} it returns {text, usage, wire}. * So every existing wrapper composes over it unchanged — `turnCappedClient` (the 32-turn runaway cap), * `timeoutLlmClient`, `retryingLlmClient`, `concurrencyLimitedClient`. * * Failure semantics, deliberately matching the SDK lane (m9i.1 — never silently turn one config into another): * - a NON-ZERO exit ⇒ THROWS (the cell fails explicitly / the retry wrapper reissues). Never a silent $0 row. * - an EMPTY reply ⇒ returns `text: ""` — a clean PASS, which `live-agent.ts` already treats as a no-text * turn that is NOT appended to the reused transcript, so it cannot poison later wakes. * This is the EXISTING empty-completion hygiene; this lane adds no new path. * - a TIMEOUT ⇒ the child is SIGKILLed and a `timeout`-classed error rejects (transient ⇒ retried). */ export function codexCliClient(opts: CodexCliClientOptions): LlmClient { if (opts.modelId.trim() === "") { // Fail-closed, exactly as `buildModel` does: an empty id is an UNRESOLVED model, and proceeding would let // the CLI substitute its own default — corrupting the run's model identity (kestrel-rul.1). throw new Error(`empty model id for provider "codex-cli" — refusing to construct a client with no model (fail-closed, kestrel-rul.1).`); } const defaultReasoningEffort: CodexReasoningEffort = opts.reasoningEffort ?? "low"; const timeoutMs = opts.timeoutMs ?? CODEX_CALL_TIMEOUT_MS; const captureWire = opts.captureWire ?? true; // Resolve the binary EAGERLY so a missing CLI fails at construction (before a fan is half-spawned), not on // the 40th turn. An injected `exec` seam skips resolution entirely — tests never touch the filesystem. const bin = opts.exec !== undefined ? (opts.bin ?? "codex") : resolveCodexBin(); const exec: CodexExecFn = opts.exec ?? realExec(bin); return { async complete(req): Promise<CodexCompletion> { // FAIL-CLOSED (kestrel-wa0j.1): the CLI exposes no cache-control flag at all — a configured // prompt-cache TTL can never reach the wire on this lane, and silently dropping it would leave a // ConfigId column claiming a TTL the provider never saw (the same honesty rule as `temperature`, // kestrel-gvx). The boolean cache signals stay ignorable (caching here is provider-internal). if (req.cacheTtl !== undefined) { throw new Error(cacheTtlNeverReachesWire(req.cacheTtl, "the codex-cli lane has no cache-control flag")); } // The effort dial, resolved PER REQUEST (kestrel-0drf): the harness's `thinkingLevel` maps onto the // CLI's `model_reasoning_effort`; `none` keeps the construction default so a default request's argv // stays byte-identical to before the dial was wired. const reasoningEffort = codexReasoningEffort(req.sampling.thinkingLevel, defaultReasoningEffort); const outFile = join(mkdtempSync(join(tmpdir(), "kestrel-codex-out-")), "last.txt"); const args = codexExecArgs({ modelId: opts.modelId, reasoningEffort, outFile }); const prompt = renderCodexPrompt(req.system, req.messages); const r = await exec(args, prompt, timeoutMs); if (r.code !== 0) { // Explicit failure — never a silent empty row. The message carries the CLI's own stderr so a real // fault (expired OAuth, unknown model, quota) is diagnosable from the run log alone — SCRUBBED first, // for the same reason the wire evidence is: a CLI error line could echo a token or path we must redact. const stderr = scrubSecrets(r.stderr.trim().slice(0, 400), [...(opts.secrets ?? [])]); throw new Error(`codex exec failed (exit ${r.code}) for model "${opts.modelId}": ${stderr}`); } const u = parseCodexUsage(r.stdout); const usage = { inputTokens: u.inputTokens, outputTokens: u.outputTokens, thinkingTokens: u.thinkingTokens, // The CLI reports a cache-READ counter (`cached_input_tokens`) but NO cache-WRITE counter. Both are // preserved ABSENT-vs-REPORTED (kestrel-wa0j.19 §3): the write is OMITTED (not reported — never the // fabricated 0 that would read as a real miss), and the read rides only when the CLI reported it. ...(u.cacheReadTokens !== undefined ? { cacheReadTokens: u.cacheReadTokens } : {}), }; let wire: string | undefined; if (captureWire) { const raw = JSON.stringify({ provider: "codex-cli", modelId: opts.modelId, lane: "codex-cli/free", // What was ACTUALLY sent. The CLI has no temperature knob, so the harness's declared temperature is // NOT on the wire — recorded as `cli-default` so no artifact can claim a temperature that never // went out (the same honesty rule the SDK lane applies to a rejected `temperature`, kestrel-gvx). // `reasoningEffort` is the RESOLVED per-request effort (the thinkingLevel mapping, kestrel-0drf), // never the construction constant — the evidence records the argv's own value. sampling: { temperature: "cli-default", reasoningEffort, thinkingLevel: req.sampling.thinkingLevel }, argv: args, exitCode: r.code, latencyMs: r.latencyMs, usage, }); wire = scrubSecrets(raw, [...(opts.secrets ?? [])]); } // Trim only the trailing newline the CLI's `-o` file carries; an all-whitespace reply collapses to "" // and rides the existing empty-PASS hygiene (never appended to the reused transcript). const text = r.lastMessage.trim() === "" ? "" : r.lastMessage.replace(/\n+$/, ""); return { text, usage, latencyMs: r.latencyMs, ...(wire !== undefined ? { wire } : {}), }; }, }; }