Revert "feat(ai): claude-cli recipe for native gateway-based subagent dispatch (#2277)"

This reverts commit 5ac81b0d0a.
This commit is contained in:
Garry Tan
2026-07-23 05:03:38 -07:00
parent b0d136ee6d
commit 6ec5261700
6 changed files with 1 additions and 1076 deletions
-22
View File
@@ -1338,10 +1338,6 @@ function instantiateEmbedding(recipe: Recipe, modelId: string, cfg: AIGatewayCon
throw new AIConfigError(
`Anthropic has no embedding model. Use openai or google for embeddings.`,
);
case 'claude-cli':
throw new AIConfigError(
`claude-cli has no embedding model. Use openai or google for embeddings.`,
);
case 'openai-compatible': {
// D12=A: unified auth via Recipe.resolveAuth (or default).
const auth = applyResolveAuth(recipe, cfg, 'embedding');
@@ -2284,15 +2280,6 @@ function instantiateExpansion(recipe: Recipe, modelId: string, cfg: AIGatewayCon
const baseURL = resolveNativeBaseUrl('anthropic', cfg);
return createAnthropic({ apiKey, ...(baseURL ? { baseURL } : {}) }).languageModel(modelId);
}
case 'claude-cli': {
// The CLI handles its own auth (OAuth session); spawn the subprocess
// directly via the same LanguageModelV2 implementation chat uses. There
// is no separate expansion path because claude-cli does not declare a
// separate expansion touchpoint — but routing here keeps the switch
// exhaustive and lets a future expansion touchpoint use the same code.
const { ClaudeCliLanguageModel } = require('./providers/claude-cli-language-model.ts');
return new ClaudeCliLanguageModel(modelId);
}
case 'openai-compatible': {
// D12=A: unified auth via Recipe.resolveAuth (or default).
const auth = applyResolveAuth(recipe, cfg, 'expansion');
@@ -2781,15 +2768,6 @@ function instantiateChat(recipe: Recipe, modelId: string, cfg: AIGatewayConfig):
const baseURL = resolveNativeBaseUrl('anthropic', cfg);
return createAnthropic({ apiKey, ...(baseURL ? { baseURL } : {}) }).languageModel(modelId);
}
case 'claude-cli': {
// The CLI handles its own auth (OAuth session managed by `claude`
// login). Subprocess-based LanguageModelV2 dispatches via the recipe
// path so per-call routing works: `claude-cli:claude-sonnet-4-6` lands
// here, while sibling `litellm:gpt-5.4` continues through the
// openai-compatible path below. No env-var switch, no global flag.
const { ClaudeCliLanguageModel } = require('./providers/claude-cli-language-model.ts');
return new ClaudeCliLanguageModel(modelId);
}
case 'openai-compatible': {
// D12=A: unified auth via Recipe.resolveAuth (or default).
const auth = applyResolveAuth(recipe, cfg, 'chat');
@@ -1,444 +0,0 @@
/**
* ai-sdk LanguageModelV2 implementation that dispatches via the `claude --print`
* CLI subprocess. Used by the `claude-cli` recipe to route gateway.toolLoop /
* gateway.chat calls through Claude Code's OAuth session instead of the
* Anthropic SDK + ANTHROPIC_API_KEY.
*
* Per-call routing is the contract: the gateway resolves the model string
* to this recipe based on the `claude-cli:` prefix, instantiates one of
* these objects per modelId, and dispatches doGenerate. Sibling subagent
* jobs with `litellm:gpt-5.4` continue routing through litellm-proxy in
* the same worker; no env-var switch, no global state.
*
* Tool use is supported via system-prompt-instructed JSON emission:
* The recipe injects a fenced instruction block into the system prompt
* that teaches the model the `<use_tools>[{id,name,input}, ...]</use_tools>`
* emission format. The adapter parses those blocks back into ai-sdk
* `tool-call` content parts. Parallel tool calls (multiple entries in
* the JSON array) round-trip cleanly — this is the case that breaks
* on the codex-proxy / litellm GPT-5.x bridge today.
*
* Context isolation:
* The subprocess is spawned from a dedicated tmpdir so claude-cli's
* CLAUDE.md auto-discovery has no local files to find. `--system-prompt`
* replaces the default system prompt; `--disable-slash-commands` skips
* skill resolution. User-level ~/.claude/CLAUDE.md still loads because
* the only way to skip it is `--bare`, which forces ANTHROPIC_API_KEY
* auth and defeats the whole point of this provider. The ~42k cached
* tokens from user-level instructions are accepted as a cost-trivial
* trade-off on the subscription path.
*
* doStream is not yet implemented; the model declares no streaming. Callers
* (gateway.toolLoop primarily) use doGenerate.
*/
import { spawn } from 'node:child_process';
import { mkdirSync } from 'node:fs';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import type {
LanguageModelV2,
LanguageModelV2CallOptions,
LanguageModelV2Content,
LanguageModelV2FunctionTool,
LanguageModelV2Prompt,
LanguageModelV2Message,
LanguageModelV2ProviderDefinedTool,
} from '@ai-sdk/provider';
function claudeBin(): string {
return process.env.GBRAIN_CLAUDE_CLI_BIN ?? 'claude';
}
const CLAUDE_CWD = join(tmpdir(), `gbrain-claude-cli-cwd-${process.pid}`);
let cwdEnsured = false;
function ensureCleanCwd(): string {
if (!cwdEnsured) {
mkdirSync(CLAUDE_CWD, { recursive: true });
cwdEnsured = true;
}
return CLAUDE_CWD;
}
/** Parsed shape of `claude --print --output-format json`. */
interface ClaudeJsonResult {
type: 'result';
subtype: 'success' | string;
is_error: boolean;
result: string;
stop_reason: string | null;
session_id: string;
num_turns: number;
usage?: {
input_tokens?: number;
output_tokens?: number;
cache_read_input_tokens?: number;
cache_creation_input_tokens?: number;
};
}
/**
* Build the system-prompt addendum that teaches the model the
* `<use_tools>...</use_tools>` emission format. Returns the empty string
* when no tools are registered for this turn so the model gets a normal
* text-completion prompt without protocol noise.
*/
function buildToolUseInstructions(
tools: ReadonlyArray<LanguageModelV2FunctionTool | LanguageModelV2ProviderDefinedTool> | undefined,
): string {
if (!tools || tools.length === 0) return '';
const functionTools = tools.filter((t): t is LanguageModelV2FunctionTool => t.type === 'function');
if (functionTools.length === 0) return '';
const toolSpecs = functionTools.map(t => ({
name: t.name,
description: t.description ?? '',
input_schema: t.inputSchema ?? { type: 'object', properties: {} },
}));
return [
'',
'## Tool Use Protocol',
'',
'You have access to these tools:',
'',
'```json',
JSON.stringify(toolSpecs, null, 2),
'```',
'',
'To call one or more tools in this turn, emit EXACTLY ONE block of this form, ' +
'with no other text outside the block on its own lines:',
'',
'<use_tools>',
'[',
' {"id": "<unique tool call id, like toolu_01ABC>", "name": "<tool name>", "input": <input object matching the tool\'s input_schema>}',
']',
'</use_tools>',
'',
'Multiple tool calls go in the array. Tool results are returned to you on the ' +
'next turn as [tool_result <text>] entries. You may then call more tools or emit a final response.',
'',
'When you are ready to give a final answer instead of calling tools, respond with prose text only — ' +
'do not include a <use_tools> block in that case.',
'',
].join('\n');
}
/**
* Render the ai-sdk message array into a single text prompt for `claude --print`
* stdin. System messages are extracted up-front and concatenated into the
* `--system-prompt` flag value. Tool calls and tool results are rendered as
* placeholders so the model sees the conversation in a coherent shape even
* though the adapter does not natively round-trip tool calls through claude-cli.
*/
function renderPrompt(prompt: LanguageModelV2Prompt): { systemText: string; userPrompt: string } {
const systemParts: string[] = [];
const convo: string[] = [];
for (const msg of prompt as ReadonlyArray<LanguageModelV2Message>) {
if (msg.role === 'system') {
systemParts.push(msg.content);
continue;
}
if (msg.role === 'user') {
const text = msg.content
.map(p => {
if (p.type === 'text') return p.text;
// File parts get a stub — multimodal is not supported via subprocess yet.
if (p.type === 'file') return `[file ${p.mediaType ?? 'unknown'}]`;
return '';
})
.filter(s => s.length > 0)
.join('\n');
if (text) convo.push(`User: ${text}`);
continue;
}
if (msg.role === 'assistant') {
const rendered = msg.content
.map(p => {
if (p.type === 'text') return p.text;
if (p.type === 'reasoning') return ''; // dropped on replay
if (p.type === 'tool-call') {
return `[tool_use ${p.toolName}(${p.input})]`;
}
if (p.type === 'tool-result') {
const out = typeof p.output === 'string' ? p.output : JSON.stringify(p.output);
return `[tool_result ${out}]`;
}
return '';
})
.filter(s => s.length > 0)
.join('\n');
if (rendered) convo.push(`Assistant: ${rendered}`);
continue;
}
if (msg.role === 'tool') {
const rendered = msg.content
.map(p => {
const out = typeof p.output === 'string' ? p.output : JSON.stringify(p.output);
return `[tool_result ${out}]`;
})
.join('\n');
if (rendered) convo.push(`User: ${rendered}`);
continue;
}
}
return { systemText: systemParts.join('\n'), userPrompt: convo.join('\n\n') };
}
/**
* Spawn `claude --print` with the contamination-suppression flags and return
* the parsed `--output-format json` envelope. Aborts propagate to SIGTERM on
* the child.
*/
function runClaude(
systemPrompt: string,
userPrompt: string,
model: string,
signal?: AbortSignal,
): Promise<ClaudeJsonResult> {
return new Promise((resolve, reject) => {
const args = [
'--print',
'--output-format', 'json',
'--model', model,
'--disable-slash-commands',
// Agent isolation: this subprocess must behave like a raw LLM, not a
// full Claude Code agent. `--tools ""` disables every built-in tool
// (Bash/Read/WebSearch/...); `--strict-mcp-config` ignores all user-level
// MCP servers (without it, each call would boot the user's MCP servers —
// including gbrain's own MCP → recursion + PGLite single-writer lock
// contention). Verified against claude CLI 2.1.145 --help.
'--tools', '',
'--strict-mcp-config',
];
if (systemPrompt) {
args.push('--system-prompt', systemPrompt);
}
// Env scrub: guarantee the CLI authenticates via its own OAuth session
// (subscription), never via an inherited API key. Without this, an
// ANTHROPIC_API_KEY in gbrain's env (the exact setup this recipe is meant
// to replace) silently flips billing to per-token API usage.
const env = { ...process.env };
delete env.ANTHROPIC_API_KEY;
delete env.ANTHROPIC_AUTH_TOKEN;
delete env.ANTHROPIC_BASE_URL;
const child = spawn(claudeBin(), args, {
stdio: ['pipe', 'pipe', 'pipe'],
cwd: ensureCleanCwd(),
env,
});
let stdout = '';
let stderr = '';
child.stdout.on('data', chunk => { stdout += String(chunk); });
child.stderr.on('data', chunk => { stderr += String(chunk); });
const onAbort = () => {
child.kill('SIGTERM');
reject(new Error('claude-cli adapter aborted'));
};
if (signal) {
if (signal.aborted) {
onAbort();
return;
}
signal.addEventListener('abort', onAbort, { once: true });
}
child.on('error', err => {
if (signal) signal.removeEventListener('abort', onAbort);
reject(new Error(`claude-cli spawn failed: ${err instanceof Error ? err.message : String(err)}`));
});
child.on('close', code => {
if (signal) signal.removeEventListener('abort', onAbort);
if (code !== 0) {
reject(new Error(`claude-cli exited ${code}: ${stderr.trim() || stdout.trim()}`));
return;
}
try {
let parsed = JSON.parse(stdout) as unknown;
// Compat: when the user has `"verbose": true` in ~/.claude/settings.json,
// `--print --output-format json` emits an ARRAY of events
// ([{type:"system",subtype:"init",...}, ..., {type:"result",...}])
// instead of the bare result object. There is no CLI flag to force it
// off (no --no-verbose; --settings '{}' merges, does not replace), so
// tolerate both shapes and pick the result event. Verified on CLI 2.1.145.
if (Array.isArray(parsed)) {
const resultEvent = parsed.find(
(ev): ev is ClaudeJsonResult =>
!!ev && typeof ev === 'object' && (ev as { type?: unknown }).type === 'result',
);
if (!resultEvent) {
reject(new Error(`claude-cli JSON event array had no "result" event\n--- raw ---\n${stdout.slice(0, 500)}`));
return;
}
parsed = resultEvent;
}
const envelope = parsed as ClaudeJsonResult;
if (envelope.is_error) {
reject(new Error(`claude-cli reported error: ${envelope.result || envelope.subtype}`));
return;
}
resolve(envelope);
} catch (e) {
reject(new Error(`claude-cli output not JSON: ${e instanceof Error ? e.message : String(e)}\n--- raw ---\n${stdout.slice(0, 500)}`));
}
});
// stdin error handler: if the binary does not exist (ENOENT) or the child
// dies before draining stdin, write/end can emit an unhandled 'error'
// (EPIPE) that would crash the worker. The spawn-level 'error' / non-zero
// 'close' handlers above already surface the real failure, so the stdin
// error itself is safe to swallow.
child.stdin.on('error', () => { /* surfaced via child 'error'/'close' */ });
try {
child.stdin.write(userPrompt);
child.stdin.end();
} catch (e) {
if (signal) signal.removeEventListener('abort', onAbort);
reject(new Error(`claude-cli stdin write failed (is the claude binary installed?): ${e instanceof Error ? e.message : String(e)}`));
}
});
}
interface ParsedToolCall {
id: string;
name: string;
/** Stringified JSON, matching the ai-sdk LanguageModelV2ToolCall.input contract. */
input: string;
}
/**
* Locate and parse the `<use_tools>...</use_tools>` block in the assistant's
* raw text response. Returns the parsed tool calls plus whatever prose
* surrounded the block. Returns an empty `toolCalls` array when no block is
* present, malformed, or unterminated — the caller then treats the full
* raw text as a final text response.
*/
function extractToolCalls(raw: string): {
toolCalls: ParsedToolCall[];
beforeText: string;
afterText: string;
} {
const openTag = '<use_tools>';
const closeTag = '</use_tools>';
const openIdx = raw.indexOf(openTag);
if (openIdx === -1) {
return { toolCalls: [], beforeText: raw.trim(), afterText: '' };
}
const closeIdx = raw.indexOf(closeTag, openIdx + openTag.length);
if (closeIdx === -1) {
// Unterminated block — recover gracefully.
return { toolCalls: [], beforeText: raw.trim(), afterText: '' };
}
const beforeText = raw.slice(0, openIdx).trim();
const afterText = raw.slice(closeIdx + closeTag.length).trim();
let inner = raw.slice(openIdx + openTag.length, closeIdx).trim();
if (inner.startsWith('```')) {
inner = inner.replace(/^```(?:json|JSON)?\s*\n?/, '').replace(/\n?```$/, '').trim();
}
let parsed: unknown;
try {
parsed = JSON.parse(inner);
} catch {
return { toolCalls: [], beforeText: raw.trim(), afterText: '' };
}
if (!Array.isArray(parsed)) {
return { toolCalls: [], beforeText: raw.trim(), afterText: '' };
}
const toolCalls: ParsedToolCall[] = [];
for (const entry of parsed) {
if (!entry || typeof entry !== 'object') continue;
const e = entry as Record<string, unknown>;
const name = typeof e.name === 'string' ? e.name : null;
if (!name) continue;
const id = typeof e.id === 'string' && e.id.length > 0
? e.id
: `toolu_claude_cli_${Math.random().toString(36).slice(2, 12)}`;
const inputJson = JSON.stringify(e.input ?? {});
toolCalls.push({ id, name, input: inputJson });
}
return { toolCalls, beforeText, afterText };
}
/**
* Strip provider prefixes (`anthropic:`, `litellm:`, `claude-cli:`) that the
* underlying CLI does not understand. The gateway hands us a bare model id
* via `recipe.aliases` resolution, but defensive normalization here keeps
* direct LanguageModelV2 construction (in tests, for example) ergonomic.
*/
function normalizeModel(model: string): string {
const idx = model.indexOf(':');
return idx >= 0 ? model.slice(idx + 1) : model;
}
export class ClaudeCliLanguageModel implements LanguageModelV2 {
readonly specificationVersion = 'v2' as const;
readonly provider = 'claude-cli';
readonly modelId: string;
readonly supportedUrls = {};
constructor(modelId: string) {
this.modelId = normalizeModel(modelId);
}
async doGenerate(options: LanguageModelV2CallOptions): Promise<{
content: LanguageModelV2Content[];
finishReason: 'stop' | 'length' | 'content-filter' | 'tool-calls' | 'error' | 'other' | 'unknown';
usage: { inputTokens: number | undefined; outputTokens: number | undefined; totalTokens: number | undefined };
warnings: never[];
}> {
const { systemText, userPrompt } = renderPrompt(options.prompt);
const toolInstructions = buildToolUseInstructions(options.tools);
const systemPrompt = [systemText, toolInstructions].filter(s => s.length > 0).join('\n');
const result = await runClaude(systemPrompt, userPrompt, this.modelId, options.abortSignal);
const { toolCalls, beforeText, afterText } = extractToolCalls(result.result);
const content: LanguageModelV2Content[] = [];
if (beforeText) content.push({ type: 'text', text: beforeText });
for (const call of toolCalls) {
content.push({
type: 'tool-call',
toolCallId: call.id,
toolName: call.name,
input: call.input,
});
}
if (afterText) content.push({ type: 'text', text: afterText });
if (content.length === 0) {
// Empty response — still hand the caller a well-formed content array.
content.push({ type: 'text', text: result.result ?? '' });
}
const finishReason = toolCalls.length > 0 ? 'tool-calls' as const : 'stop' as const;
const inputTokens = result.usage?.input_tokens;
const outputTokens = result.usage?.output_tokens;
const totalTokens = (inputTokens ?? 0) + (outputTokens ?? 0);
return {
content,
finishReason,
usage: {
inputTokens,
outputTokens,
totalTokens: inputTokens !== undefined && outputTokens !== undefined ? totalTokens : undefined,
},
warnings: [],
};
}
async doStream(): Promise<never> {
throw new Error(
'claude-cli LanguageModel does not support streaming. Use doGenerate or set ' +
'the model on a non-streaming chat surface (gateway.toolLoop is non-streaming).',
);
}
}
-71
View File
@@ -1,71 +0,0 @@
import type { Recipe } from '../types.ts';
/**
* Claude via the local `claude` CLI binary, using its built-in OAuth session
* (Claude Code / Claude Max subscription). No ANTHROPIC_API_KEY needed — the
* CLI manages its own auth state and the gateway dispatches via subprocess.
*
* Solves the #334 case where Max subscribers want Minions subagent dispatch
* to run against their existing subscription instead of paying per-token API
* charges. The recipe sits alongside the existing `anthropic` recipe so users
* pick per call: `anthropic:claude-sonnet-4-6` (API key + per-token billing)
* vs `claude-cli:claude-sonnet-4-6` (OAuth subscription, no API key).
*
* Chat-only. Claude has no first-party embedding model; users wanting an
* Anthropic chat path with embeddings still combine this with openai/google/
* voyage for embedding the way the existing `anthropic` recipe documents.
*
* Auth: `auth_env.required: []` because the CLI handles auth itself. The
* `claude` binary on PATH (or `GBRAIN_CLAUDE_CLI_BIN`) IS the auth surface;
* there is nothing for the gateway to forward.
*
* Setup expectation: `claude` CLI installed and logged in (Claude Code
* onboarding does this), or `GBRAIN_CLAUDE_CLI_BIN` pointing at the binary.
*/
export const claudeCli: Recipe = {
id: 'claude-cli',
name: 'Claude (via CLI)',
tier: 'native',
implementation: 'claude-cli',
// The CLI owns auth; no env vars are required from the gateway side.
auth_env: {
required: [],
},
touchpoints: {
// No embedding or expansion touchpoints — chat-only.
chat: {
models: [
'claude-opus-4-7',
'claude-sonnet-4-6',
'claude-haiku-4-5-20251001',
],
supports_tools: true,
supports_subagent_loop: true,
// The CLI handles caching internally and does not surface it via the
// standard cache_control control plane. From the gateway's POV the
// model does not support prompt caching.
supports_prompt_cache: false,
max_context_tokens: 200000,
// Cost figures match the underlying Claude API tier, but the actual
// bill is borne by the subscription. We report them for the budget
// ledger's per-call accounting; operators on flat-rate subscriptions
// can treat the numbers as nominal.
cost_per_1m_input_usd: 3.0,
cost_per_1m_output_usd: 15.0,
price_last_verified: '2026-06-17',
},
},
// Friendly aliases mirror the `anthropic` recipe so config strings stay
// portable: switching `anthropic:claude-sonnet-4-6` to `claude-cli:claude-sonnet-4-6`
// is a one-token edit. Reverse aliases rewrite legacy IDs back to canonical.
aliases: {
'claude-haiku-4-5': 'claude-haiku-4-5-20251001',
'claude-sonnet-4-6-20250929': 'claude-sonnet-4-6',
'sonnet': 'claude-sonnet-4-6',
'haiku': 'claude-haiku-4-5-20251001',
'opus': 'claude-opus-4-7',
},
setup_hint:
'Install Claude Code (`claude` CLI) and run `claude` once to log in. ' +
'Set GBRAIN_CLAUDE_CLI_BIN if the binary is not on PATH.',
};
-2
View File
@@ -9,7 +9,6 @@ import type { Recipe } from '../types.ts';
import { openai } from './openai.ts';
import { google } from './google.ts';
import { anthropic } from './anthropic.ts';
import { claudeCli } from './claude-cli.ts';
import { ollama } from './ollama.ts';
import { openrouter } from './openrouter.ts';
import { voyage } from './voyage.ts';
@@ -32,7 +31,6 @@ const ALL: Recipe[] = [
openai,
google,
anthropic,
claudeCli,
ollama,
openrouter,
voyage,
+1 -2
View File
@@ -22,8 +22,7 @@ export type Implementation =
| 'native-openai'
| 'native-google'
| 'native-anthropic'
| 'openai-compatible'
| 'claude-cli';
| 'openai-compatible';
export interface EmbeddingTouchpoint {
models: string[];
-535
View File
@@ -1,535 +0,0 @@
/**
* Tests for the claude-cli LanguageModelV2 implementation that the
* `claude-cli` recipe instantiates.
*
* Strategy: a POSIX shell stub at GBRAIN_CLAUDE_CLI_BIN emits scripted
* --output-format json envelopes. Tests exercise the LanguageModelV2
* doGenerate surface: text round trip, tool-call extraction (single +
* multiple parallel), abort semantics, context-isolation flags. No
* claude-cli installation or API credits required.
*
* Recipe registration is also smoke-tested: getRecipe('claude-cli')
* returns a chat-only Recipe with the right model list.
*
* Env isolation: GBRAIN_CLAUDE_CLI_BIN is set per-test via withEnv(),
* NOT in beforeAll. The provider reads the env var at spawn time so
* withEnv's save/restore in try/finally is sufficient; no leakage to
* sibling test files in the same bun-test process.
*/
import { describe, test, expect, beforeAll, afterAll } from 'bun:test';
import { writeFileSync, chmodSync, mkdirSync, rmSync } from 'node:fs';
import { join } from 'node:path';
import { tmpdir } from 'node:os';
import type { LanguageModelV2CallOptions } from '@ai-sdk/provider';
import { withEnv } from './helpers/with-env.ts';
const stubDir = join(tmpdir(), `claude-cli-recipe-stub-${process.pid}`);
const stubBin = join(stubDir, 'claude');
const stubResponsePath = join(stubDir, 'claude_response.json');
beforeAll(() => {
mkdirSync(stubDir, { recursive: true });
const stub = [
'#!/bin/sh',
'cat > /dev/null',
'case " $* " in',
' *" --print "*) ;;',
' *) echo "missing --print in argv: $*" >&2; exit 64 ;;',
'esac',
`cat "${stubResponsePath}"`,
].join('\n');
writeFileSync(stubBin, stub);
chmodSync(stubBin, 0o755);
});
afterAll(() => {
rmSync(stubDir, { recursive: true, force: true });
});
function withStubEnv<T>(fn: () => T | Promise<T>): Promise<T> {
return withEnv({ GBRAIN_CLAUDE_CLI_BIN: stubBin }, fn);
}
function stageResponse(envelope: Record<string, unknown>): void {
writeFileSync(stubResponsePath, JSON.stringify(envelope));
}
function baseEnvelope(result: string, overrides: Record<string, unknown> = {}): Record<string, unknown> {
return {
type: 'result',
subtype: 'success',
is_error: false,
result,
stop_reason: 'end_turn',
session_id: 'test-session',
num_turns: 1,
usage: {
input_tokens: 12,
output_tokens: 34,
cache_read_input_tokens: 0,
cache_creation_input_tokens: 0,
},
...overrides,
};
}
function userMessage(text: string): LanguageModelV2CallOptions['prompt'][number] {
return { role: 'user', content: [{ type: 'text', text }] };
}
describe('claude-cli recipe registration', () => {
test('getRecipe returns chat-only Recipe with the documented models', async () => {
const { getRecipe } = await import('../src/core/ai/recipes/index.ts');
const recipe = getRecipe('claude-cli');
expect(recipe).toBeDefined();
expect(recipe!.id).toBe('claude-cli');
expect(recipe!.implementation).toBe('claude-cli');
expect(recipe!.touchpoints.chat).toBeDefined();
expect(recipe!.touchpoints.chat!.supports_tools).toBe(true);
expect(recipe!.touchpoints.chat!.supports_subagent_loop).toBe(true);
expect(recipe!.touchpoints.chat!.models).toContain('claude-sonnet-4-6');
expect(recipe!.touchpoints.embedding).toBeUndefined();
expect(recipe!.touchpoints.expansion).toBeUndefined();
});
test('recipe aliases map short names to canonical model ids', async () => {
const { getRecipe } = await import('../src/core/ai/recipes/index.ts');
const recipe = getRecipe('claude-cli');
expect(recipe!.aliases!['sonnet']).toBe('claude-sonnet-4-6');
expect(recipe!.aliases!['haiku']).toBe('claude-haiku-4-5-20251001');
});
});
describe('claude-cli LanguageModel — text-only round trip', () => {
test('returns a single text content block with usage + stop finish reason', async () => {
await withStubEnv(async () => {
stageResponse(baseEnvelope('hello world'));
const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts');
const model = new ClaudeCliLanguageModel('claude-sonnet-4-6');
const result = await model.doGenerate({
prompt: [userMessage('hi')],
} as LanguageModelV2CallOptions);
expect(result.finishReason).toBe('stop');
expect(result.content).toHaveLength(1);
expect(result.content[0]).toEqual({ type: 'text', text: 'hello world' });
expect(result.usage.inputTokens).toBe(12);
expect(result.usage.outputTokens).toBe(34);
});
});
test('strips provider prefixes from the model id', async () => {
const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts');
const model = new ClaudeCliLanguageModel('anthropic:claude-sonnet-4-6');
expect(model.modelId).toBe('claude-sonnet-4-6');
});
});
describe('claude-cli LanguageModel — tool use', () => {
test('parses <use_tools> block into LanguageModelV2 tool-call content', async () => {
await withStubEnv(async () => {
stageResponse(
baseEnvelope(
[
'I will look up the pattern first.',
'<use_tools>',
'[{"id": "toolu_01ABC", "name": "search", "input": {"query": "n+1 query"}}]',
'</use_tools>',
].join('\n'),
),
);
const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts');
const model = new ClaudeCliLanguageModel('claude-sonnet-4-6');
const result = await model.doGenerate({
prompt: [userMessage('find n+1 queries')],
tools: [
{
type: 'function',
name: 'search',
description: 'Search the brain',
inputSchema: { type: 'object', properties: { query: { type: 'string' } } },
},
],
} as LanguageModelV2CallOptions);
expect(result.finishReason).toBe('tool-calls');
expect(result.content).toHaveLength(2);
expect(result.content[0]).toMatchObject({ type: 'text', text: 'I will look up the pattern first.' });
expect(result.content[1]).toMatchObject({
type: 'tool-call',
toolCallId: 'toolu_01ABC',
toolName: 'search',
input: '{"query":"n+1 query"}',
});
});
});
test('parses multiple parallel tool calls in a single block', async () => {
await withStubEnv(async () => {
stageResponse(
baseEnvelope(
[
'<use_tools>',
'[',
' {"id": "toolu_A", "name": "search", "input": {"query": "foo"}},',
' {"id": "toolu_B", "name": "get_page", "input": {"slug": "areas/x"}}',
']',
'</use_tools>',
].join('\n'),
),
);
const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts');
const model = new ClaudeCliLanguageModel('claude-sonnet-4-6');
const result = await model.doGenerate({
prompt: [userMessage('multi')],
tools: [
{ type: 'function', name: 'search', description: 's', inputSchema: { type: 'object', properties: {} } },
{ type: 'function', name: 'get_page', description: 'g', inputSchema: { type: 'object', properties: {} } },
],
} as LanguageModelV2CallOptions);
const calls = result.content.filter(c => c.type === 'tool-call');
expect(calls).toHaveLength(2);
expect(calls.map(c => (c as { toolName: string }).toolName)).toEqual(['search', 'get_page']);
expect(result.finishReason).toBe('tool-calls');
});
});
test('tolerates fenced JSON inside <use_tools>', async () => {
await withStubEnv(async () => {
stageResponse(
baseEnvelope(
[
'<use_tools>',
'```json',
'[{"id": "toolu_F", "name": "search", "input": {"q": "x"}}]',
'```',
'</use_tools>',
].join('\n'),
),
);
const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts');
const model = new ClaudeCliLanguageModel('claude-sonnet-4-6');
const result = await model.doGenerate({
prompt: [userMessage('fenced')],
tools: [{ type: 'function', name: 'search', description: '', inputSchema: { type: 'object', properties: {} } }],
} as LanguageModelV2CallOptions);
const calls = result.content.filter(c => c.type === 'tool-call');
expect(calls).toHaveLength(1);
});
});
test('synthesizes an id when the model omits it', async () => {
await withStubEnv(async () => {
stageResponse(
baseEnvelope(
[
'<use_tools>',
'[{"name": "search", "input": {"q": "x"}}]',
'</use_tools>',
].join('\n'),
),
);
const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts');
const model = new ClaudeCliLanguageModel('claude-sonnet-4-6');
const result = await model.doGenerate({
prompt: [userMessage('no id')],
tools: [{ type: 'function', name: 'search', description: '', inputSchema: { type: 'object', properties: {} } }],
} as LanguageModelV2CallOptions);
const call = result.content.find(c => c.type === 'tool-call') as { toolCallId: string } | undefined;
expect(call).toBeDefined();
expect(call!.toolCallId).toMatch(/^toolu_claude_cli_/);
});
});
test('falls back to text on malformed JSON', async () => {
await withStubEnv(async () => {
stageResponse(
baseEnvelope(
[
'<use_tools>',
'not valid json',
'</use_tools>',
].join('\n'),
),
);
const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts');
const model = new ClaudeCliLanguageModel('claude-sonnet-4-6');
const result = await model.doGenerate({
prompt: [userMessage('malformed')],
tools: [{ type: 'function', name: 'search', description: '', inputSchema: { type: 'object', properties: {} } }],
} as LanguageModelV2CallOptions);
expect(result.content.filter(c => c.type === 'tool-call')).toHaveLength(0);
expect(result.finishReason).toBe('stop');
});
});
test('returns text-only stop when tools are offered but model declines to call any', async () => {
// Real-world case: the model decides the user's request does not require
// a tool call, ignores the use_tools protocol, and answers directly.
// The recipe still must return clean LanguageModelV2 output so the
// caller (gateway.toolLoop) can treat the text as the final answer
// rather than wedge waiting for tool calls that never come.
await withStubEnv(async () => {
stageResponse(
baseEnvelope(
'I do not actually need to call any tools for this. The answer is 42.',
{ stop_reason: 'end_turn' },
),
);
const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts');
const model = new ClaudeCliLanguageModel('claude-sonnet-4-6');
const result = await model.doGenerate({
prompt: [userMessage('what is the meaning of life? you may use tools but do not need to')],
tools: [{ type: 'function', name: 'compute', description: 'Compute things', inputSchema: { type: 'object', properties: {} } }],
} as LanguageModelV2CallOptions);
// No tool-call content blocks; caller treats this as a final answer.
expect(result.content.filter(c => c.type === 'tool-call')).toHaveLength(0);
// Text block present with the full model reply.
const textBlocks = result.content.filter(c => c.type === 'text');
expect(textBlocks).toHaveLength(1);
expect((textBlocks[0] as { text: string }).text).toContain('42');
// finishReason 'stop' tells the gateway-loop this is terminal output,
// not a partial mid-tool-loop state.
expect(result.finishReason).toBe('stop');
});
});
test('drops the block when the close tag is missing', async () => {
await withStubEnv(async () => {
stageResponse(
baseEnvelope(
[
'<use_tools>',
'[{"id": "toolu_X", "name": "search", "input": {}}',
].join('\n'),
),
);
const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts');
const model = new ClaudeCliLanguageModel('claude-sonnet-4-6');
const result = await model.doGenerate({
prompt: [userMessage('unterminated')],
tools: [{ type: 'function', name: 'search', description: '', inputSchema: { type: 'object', properties: {} } }],
} as LanguageModelV2CallOptions);
expect(result.content.filter(c => c.type === 'tool-call')).toHaveLength(0);
expect(result.finishReason).toBe('stop');
});
});
});
describe('claude-cli LanguageModel — context isolation', () => {
test('argv includes --disable-slash-commands + --system-prompt and cwd is the dedicated tmpdir', async () => {
await withStubEnv(async () => {
const argvLog = join(stubDir, 'argv.log');
const cwdLog = join(stubDir, 'cwd.log');
const recordStub = [
'#!/bin/sh',
`printf "%s\\n" "$@" > "${argvLog}"`,
`pwd > "${cwdLog}"`,
'cat > /dev/null',
`cat "${stubResponsePath}"`,
].join('\n');
writeFileSync(stubBin, recordStub);
chmodSync(stubBin, 0o755);
stageResponse(baseEnvelope('ok'));
const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts');
const model = new ClaudeCliLanguageModel('claude-sonnet-4-6');
await model.doGenerate({
prompt: [
{ role: 'system', content: 'You are gbrain subagent.' },
userMessage('hi'),
],
} as LanguageModelV2CallOptions);
const fs = require('node:fs');
const argv = fs.readFileSync(argvLog, 'utf8').split('\n').filter(Boolean);
const cwd = fs.readFileSync(cwdLog, 'utf8').trim();
expect(argv).toContain('--print');
expect(argv).toContain('--output-format');
expect(argv).toContain('json');
expect(argv).toContain('--disable-slash-commands');
// Agent-isolation hardening: no built-in tools, no inherited MCP servers.
expect(argv).toContain('--tools');
expect(argv).toContain('--strict-mcp-config');
expect(argv).toContain('--system-prompt');
expect(argv).toContain('You are gbrain subagent.');
expect(cwd).toMatch(/gbrain-claude-cli-cwd-\d+$/);
const fastStub = [
'#!/bin/sh',
'cat > /dev/null',
`cat "${stubResponsePath}"`,
].join('\n');
writeFileSync(stubBin, fastStub);
chmodSync(stubBin, 0o755);
});
});
test('scrubs ANTHROPIC_* credentials from the child env (subscription-only auth)', async () => {
await withStubEnv(async () => {
await withEnv(
{
ANTHROPIC_API_KEY: 'sk-should-never-leak',
ANTHROPIC_AUTH_TOKEN: 'tok-should-never-leak',
ANTHROPIC_BASE_URL: 'https://proxy.should.never.leak',
},
async () => {
const envLog = join(stubDir, 'env.log');
const envStub = [
'#!/bin/sh',
`printf "key=%s\\ntoken=%s\\nbase=%s\\n" "\${ANTHROPIC_API_KEY:-UNSET}" "\${ANTHROPIC_AUTH_TOKEN:-UNSET}" "\${ANTHROPIC_BASE_URL:-UNSET}" > "${envLog}"`,
'cat > /dev/null',
`cat "${stubResponsePath}"`,
].join('\n');
writeFileSync(stubBin, envStub);
chmodSync(stubBin, 0o755);
stageResponse(baseEnvelope('ok'));
try {
const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts');
const model = new ClaudeCliLanguageModel('claude-sonnet-4-6');
await model.doGenerate({
prompt: [userMessage('hi')],
} as LanguageModelV2CallOptions);
const fs = require('node:fs');
const seen = fs.readFileSync(envLog, 'utf8');
expect(seen).toContain('key=UNSET');
expect(seen).toContain('token=UNSET');
expect(seen).toContain('base=UNSET');
} finally {
const fastStub = [
'#!/bin/sh',
'cat > /dev/null',
`cat "${stubResponsePath}"`,
].join('\n');
writeFileSync(stubBin, fastStub);
chmodSync(stubBin, 0o755);
}
},
);
});
});
});
describe('claude-cli LanguageModel — abort + error envelopes', () => {
test('SIGTERMs the child on AbortSignal', async () => {
await withStubEnv(async () => {
const slowStub = [
'#!/bin/sh',
'cat > /dev/null',
'sleep 30',
'echo "{}"',
].join('\n');
writeFileSync(stubBin, slowStub);
chmodSync(stubBin, 0o755);
try {
const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts');
const model = new ClaudeCliLanguageModel('claude-sonnet-4-6');
const ac = new AbortController();
const promise = model.doGenerate({
prompt: [userMessage('slow')],
abortSignal: ac.signal,
} as LanguageModelV2CallOptions);
setTimeout(() => ac.abort(), 30);
await expect(promise).rejects.toThrow(/aborted/);
} finally {
const fastStub = [
'#!/bin/sh',
'cat > /dev/null',
`cat "${stubResponsePath}"`,
].join('\n');
writeFileSync(stubBin, fastStub);
chmodSync(stubBin, 0o755);
}
});
});
test('rejects when stub reports is_error: true', async () => {
await withStubEnv(async () => {
stageResponse({ ...baseEnvelope('boom'), is_error: true });
const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts');
const model = new ClaudeCliLanguageModel('claude-sonnet-4-6');
await expect(
model.doGenerate({ prompt: [userMessage('x')] } as LanguageModelV2CallOptions),
).rejects.toThrow(/claude-cli reported error/);
});
});
test('rejects on non-JSON output', async () => {
await withStubEnv(async () => {
writeFileSync(stubResponsePath, 'this is not json');
const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts');
const model = new ClaudeCliLanguageModel('claude-sonnet-4-6');
await expect(
model.doGenerate({ prompt: [userMessage('x')] } as LanguageModelV2CallOptions),
).rejects.toThrow(/claude-cli output not JSON/);
});
});
test('accepts a verbose-mode JSON event array and picks the result event', async () => {
// With `"verbose": true` in ~/.claude/settings.json the CLI emits an array
// of events instead of the bare result object (no CLI flag disables it).
await withStubEnv(async () => {
writeFileSync(
stubResponsePath,
JSON.stringify([
{ type: 'system', subtype: 'init', session_id: 'test-session', tools: [], mcp_servers: [] },
baseEnvelope('hello from array'),
]),
);
const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts');
const model = new ClaudeCliLanguageModel('claude-sonnet-4-6');
const result = await model.doGenerate({
prompt: [userMessage('hi')],
} as LanguageModelV2CallOptions);
expect(result.finishReason).toBe('stop');
expect(result.content[0]).toEqual({ type: 'text', text: 'hello from array' });
});
});
test('rejects a verbose-mode event array that lacks a result event', async () => {
// Verbose mode emits an event array; a truncated stream (or one carrying
// only init/system events) has no result event to unwrap.
await withStubEnv(async () => {
writeFileSync(
stubResponsePath,
JSON.stringify([
{ type: 'system', subtype: 'init', session_id: 'test-session', tools: [], mcp_servers: [] },
{ type: 'assistant', message: { role: 'assistant', content: [] } },
]),
);
const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts');
const model = new ClaudeCliLanguageModel('claude-sonnet-4-6');
await expect(
model.doGenerate({ prompt: [userMessage('x')] } as LanguageModelV2CallOptions),
).rejects.toThrow(/had no "result" event/);
});
});
test('rejects cleanly when the claude binary is missing (no worker crash)', async () => {
// A missing binary must surface as a rejected promise via the spawn 'error'
// handler; the child stdin 'error' (EPIPE) handler swallows the pipe failure
// so it never escalates to an unhandled rejection that would down the worker.
await withEnv({ GBRAIN_CLAUDE_CLI_BIN: join(stubDir, 'nonexistent-claude') }, async () => {
const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts');
const model = new ClaudeCliLanguageModel('claude-sonnet-4-6');
await expect(
model.doGenerate({ prompt: [userMessage('x')] } as LanguageModelV2CallOptions),
).rejects.toThrow(/claude-cli spawn failed/);
});
});
test('doStream throws not-supported', async () => {
const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts');
const model = new ClaudeCliLanguageModel('claude-sonnet-4-6');
await expect(model.doStream()).rejects.toThrow(/does not support streaming/);
});
});