diff --git a/src/core/ai/gateway.ts b/src/core/ai/gateway.ts index bfea011b5..970d6b606 100644 --- a/src/core/ai/gateway.ts +++ b/src/core/ai/gateway.ts @@ -1338,10 +1338,6 @@ function instantiateEmbedding(recipe: Recipe, modelId: string, cfg: AIGatewayCon throw new AIConfigError( `Anthropic has no embedding model. Use openai or google for embeddings.`, ); - case 'claude-cli': - throw new AIConfigError( - `claude-cli has no embedding model. Use openai or google for embeddings.`, - ); case 'openai-compatible': { // D12=A: unified auth via Recipe.resolveAuth (or default). const auth = applyResolveAuth(recipe, cfg, 'embedding'); @@ -2284,15 +2280,6 @@ function instantiateExpansion(recipe: Recipe, modelId: string, cfg: AIGatewayCon const baseURL = resolveNativeBaseUrl('anthropic', cfg); return createAnthropic({ apiKey, ...(baseURL ? { baseURL } : {}) }).languageModel(modelId); } - case 'claude-cli': { - // The CLI handles its own auth (OAuth session); spawn the subprocess - // directly via the same LanguageModelV2 implementation chat uses. There - // is no separate expansion path because claude-cli does not declare a - // separate expansion touchpoint — but routing here keeps the switch - // exhaustive and lets a future expansion touchpoint use the same code. - const { ClaudeCliLanguageModel } = require('./providers/claude-cli-language-model.ts'); - return new ClaudeCliLanguageModel(modelId); - } case 'openai-compatible': { // D12=A: unified auth via Recipe.resolveAuth (or default). const auth = applyResolveAuth(recipe, cfg, 'expansion'); @@ -2781,15 +2768,6 @@ function instantiateChat(recipe: Recipe, modelId: string, cfg: AIGatewayConfig): const baseURL = resolveNativeBaseUrl('anthropic', cfg); return createAnthropic({ apiKey, ...(baseURL ? { baseURL } : {}) }).languageModel(modelId); } - case 'claude-cli': { - // The CLI handles its own auth (OAuth session managed by `claude` - // login). Subprocess-based LanguageModelV2 dispatches via the recipe - // path so per-call routing works: `claude-cli:claude-sonnet-4-6` lands - // here, while sibling `litellm:gpt-5.4` continues through the - // openai-compatible path below. No env-var switch, no global flag. - const { ClaudeCliLanguageModel } = require('./providers/claude-cli-language-model.ts'); - return new ClaudeCliLanguageModel(modelId); - } case 'openai-compatible': { // D12=A: unified auth via Recipe.resolveAuth (or default). const auth = applyResolveAuth(recipe, cfg, 'chat'); diff --git a/src/core/ai/providers/claude-cli-language-model.ts b/src/core/ai/providers/claude-cli-language-model.ts deleted file mode 100644 index 4ffaec0af..000000000 --- a/src/core/ai/providers/claude-cli-language-model.ts +++ /dev/null @@ -1,444 +0,0 @@ -/** - * ai-sdk LanguageModelV2 implementation that dispatches via the `claude --print` - * CLI subprocess. Used by the `claude-cli` recipe to route gateway.toolLoop / - * gateway.chat calls through Claude Code's OAuth session instead of the - * Anthropic SDK + ANTHROPIC_API_KEY. - * - * Per-call routing is the contract: the gateway resolves the model string - * to this recipe based on the `claude-cli:` prefix, instantiates one of - * these objects per modelId, and dispatches doGenerate. Sibling subagent - * jobs with `litellm:gpt-5.4` continue routing through litellm-proxy in - * the same worker; no env-var switch, no global state. - * - * Tool use is supported via system-prompt-instructed JSON emission: - * The recipe injects a fenced instruction block into the system prompt - * that teaches the model the `[{id,name,input}, ...]` - * emission format. The adapter parses those blocks back into ai-sdk - * `tool-call` content parts. Parallel tool calls (multiple entries in - * the JSON array) round-trip cleanly — this is the case that breaks - * on the codex-proxy / litellm GPT-5.x bridge today. - * - * Context isolation: - * The subprocess is spawned from a dedicated tmpdir so claude-cli's - * CLAUDE.md auto-discovery has no local files to find. `--system-prompt` - * replaces the default system prompt; `--disable-slash-commands` skips - * skill resolution. User-level ~/.claude/CLAUDE.md still loads because - * the only way to skip it is `--bare`, which forces ANTHROPIC_API_KEY - * auth and defeats the whole point of this provider. The ~42k cached - * tokens from user-level instructions are accepted as a cost-trivial - * trade-off on the subscription path. - * - * doStream is not yet implemented; the model declares no streaming. Callers - * (gateway.toolLoop primarily) use doGenerate. - */ -import { spawn } from 'node:child_process'; -import { mkdirSync } from 'node:fs'; -import { tmpdir } from 'node:os'; -import { join } from 'node:path'; -import type { - LanguageModelV2, - LanguageModelV2CallOptions, - LanguageModelV2Content, - LanguageModelV2FunctionTool, - LanguageModelV2Prompt, - LanguageModelV2Message, - LanguageModelV2ProviderDefinedTool, -} from '@ai-sdk/provider'; - -function claudeBin(): string { - return process.env.GBRAIN_CLAUDE_CLI_BIN ?? 'claude'; -} -const CLAUDE_CWD = join(tmpdir(), `gbrain-claude-cli-cwd-${process.pid}`); -let cwdEnsured = false; -function ensureCleanCwd(): string { - if (!cwdEnsured) { - mkdirSync(CLAUDE_CWD, { recursive: true }); - cwdEnsured = true; - } - return CLAUDE_CWD; -} - -/** Parsed shape of `claude --print --output-format json`. */ -interface ClaudeJsonResult { - type: 'result'; - subtype: 'success' | string; - is_error: boolean; - result: string; - stop_reason: string | null; - session_id: string; - num_turns: number; - usage?: { - input_tokens?: number; - output_tokens?: number; - cache_read_input_tokens?: number; - cache_creation_input_tokens?: number; - }; -} - -/** - * Build the system-prompt addendum that teaches the model the - * `...` emission format. Returns the empty string - * when no tools are registered for this turn so the model gets a normal - * text-completion prompt without protocol noise. - */ -function buildToolUseInstructions( - tools: ReadonlyArray | undefined, -): string { - if (!tools || tools.length === 0) return ''; - - const functionTools = tools.filter((t): t is LanguageModelV2FunctionTool => t.type === 'function'); - if (functionTools.length === 0) return ''; - - const toolSpecs = functionTools.map(t => ({ - name: t.name, - description: t.description ?? '', - input_schema: t.inputSchema ?? { type: 'object', properties: {} }, - })); - - return [ - '', - '## Tool Use Protocol', - '', - 'You have access to these tools:', - '', - '```json', - JSON.stringify(toolSpecs, null, 2), - '```', - '', - 'To call one or more tools in this turn, emit EXACTLY ONE block of this form, ' + - 'with no other text outside the block on its own lines:', - '', - '', - '[', - ' {"id": "", "name": "", "input": }', - ']', - '', - '', - 'Multiple tool calls go in the array. Tool results are returned to you on the ' + - 'next turn as [tool_result ] entries. You may then call more tools or emit a final response.', - '', - 'When you are ready to give a final answer instead of calling tools, respond with prose text only — ' + - 'do not include a block in that case.', - '', - ].join('\n'); -} - -/** - * Render the ai-sdk message array into a single text prompt for `claude --print` - * stdin. System messages are extracted up-front and concatenated into the - * `--system-prompt` flag value. Tool calls and tool results are rendered as - * placeholders so the model sees the conversation in a coherent shape even - * though the adapter does not natively round-trip tool calls through claude-cli. - */ -function renderPrompt(prompt: LanguageModelV2Prompt): { systemText: string; userPrompt: string } { - const systemParts: string[] = []; - const convo: string[] = []; - - for (const msg of prompt as ReadonlyArray) { - if (msg.role === 'system') { - systemParts.push(msg.content); - continue; - } - if (msg.role === 'user') { - const text = msg.content - .map(p => { - if (p.type === 'text') return p.text; - // File parts get a stub — multimodal is not supported via subprocess yet. - if (p.type === 'file') return `[file ${p.mediaType ?? 'unknown'}]`; - return ''; - }) - .filter(s => s.length > 0) - .join('\n'); - if (text) convo.push(`User: ${text}`); - continue; - } - if (msg.role === 'assistant') { - const rendered = msg.content - .map(p => { - if (p.type === 'text') return p.text; - if (p.type === 'reasoning') return ''; // dropped on replay - if (p.type === 'tool-call') { - return `[tool_use ${p.toolName}(${p.input})]`; - } - if (p.type === 'tool-result') { - const out = typeof p.output === 'string' ? p.output : JSON.stringify(p.output); - return `[tool_result ${out}]`; - } - return ''; - }) - .filter(s => s.length > 0) - .join('\n'); - if (rendered) convo.push(`Assistant: ${rendered}`); - continue; - } - if (msg.role === 'tool') { - const rendered = msg.content - .map(p => { - const out = typeof p.output === 'string' ? p.output : JSON.stringify(p.output); - return `[tool_result ${out}]`; - }) - .join('\n'); - if (rendered) convo.push(`User: ${rendered}`); - continue; - } - } - - return { systemText: systemParts.join('\n'), userPrompt: convo.join('\n\n') }; -} - -/** - * Spawn `claude --print` with the contamination-suppression flags and return - * the parsed `--output-format json` envelope. Aborts propagate to SIGTERM on - * the child. - */ -function runClaude( - systemPrompt: string, - userPrompt: string, - model: string, - signal?: AbortSignal, -): Promise { - return new Promise((resolve, reject) => { - const args = [ - '--print', - '--output-format', 'json', - '--model', model, - '--disable-slash-commands', - // Agent isolation: this subprocess must behave like a raw LLM, not a - // full Claude Code agent. `--tools ""` disables every built-in tool - // (Bash/Read/WebSearch/...); `--strict-mcp-config` ignores all user-level - // MCP servers (without it, each call would boot the user's MCP servers — - // including gbrain's own MCP → recursion + PGLite single-writer lock - // contention). Verified against claude CLI 2.1.145 --help. - '--tools', '', - '--strict-mcp-config', - ]; - if (systemPrompt) { - args.push('--system-prompt', systemPrompt); - } - // Env scrub: guarantee the CLI authenticates via its own OAuth session - // (subscription), never via an inherited API key. Without this, an - // ANTHROPIC_API_KEY in gbrain's env (the exact setup this recipe is meant - // to replace) silently flips billing to per-token API usage. - const env = { ...process.env }; - delete env.ANTHROPIC_API_KEY; - delete env.ANTHROPIC_AUTH_TOKEN; - delete env.ANTHROPIC_BASE_URL; - const child = spawn(claudeBin(), args, { - stdio: ['pipe', 'pipe', 'pipe'], - cwd: ensureCleanCwd(), - env, - }); - - let stdout = ''; - let stderr = ''; - child.stdout.on('data', chunk => { stdout += String(chunk); }); - child.stderr.on('data', chunk => { stderr += String(chunk); }); - - const onAbort = () => { - child.kill('SIGTERM'); - reject(new Error('claude-cli adapter aborted')); - }; - if (signal) { - if (signal.aborted) { - onAbort(); - return; - } - signal.addEventListener('abort', onAbort, { once: true }); - } - - child.on('error', err => { - if (signal) signal.removeEventListener('abort', onAbort); - reject(new Error(`claude-cli spawn failed: ${err instanceof Error ? err.message : String(err)}`)); - }); - - child.on('close', code => { - if (signal) signal.removeEventListener('abort', onAbort); - if (code !== 0) { - reject(new Error(`claude-cli exited ${code}: ${stderr.trim() || stdout.trim()}`)); - return; - } - try { - let parsed = JSON.parse(stdout) as unknown; - // Compat: when the user has `"verbose": true` in ~/.claude/settings.json, - // `--print --output-format json` emits an ARRAY of events - // ([{type:"system",subtype:"init",...}, ..., {type:"result",...}]) - // instead of the bare result object. There is no CLI flag to force it - // off (no --no-verbose; --settings '{}' merges, does not replace), so - // tolerate both shapes and pick the result event. Verified on CLI 2.1.145. - if (Array.isArray(parsed)) { - const resultEvent = parsed.find( - (ev): ev is ClaudeJsonResult => - !!ev && typeof ev === 'object' && (ev as { type?: unknown }).type === 'result', - ); - if (!resultEvent) { - reject(new Error(`claude-cli JSON event array had no "result" event\n--- raw ---\n${stdout.slice(0, 500)}`)); - return; - } - parsed = resultEvent; - } - const envelope = parsed as ClaudeJsonResult; - if (envelope.is_error) { - reject(new Error(`claude-cli reported error: ${envelope.result || envelope.subtype}`)); - return; - } - resolve(envelope); - } catch (e) { - reject(new Error(`claude-cli output not JSON: ${e instanceof Error ? e.message : String(e)}\n--- raw ---\n${stdout.slice(0, 500)}`)); - } - }); - - // stdin error handler: if the binary does not exist (ENOENT) or the child - // dies before draining stdin, write/end can emit an unhandled 'error' - // (EPIPE) that would crash the worker. The spawn-level 'error' / non-zero - // 'close' handlers above already surface the real failure, so the stdin - // error itself is safe to swallow. - child.stdin.on('error', () => { /* surfaced via child 'error'/'close' */ }); - try { - child.stdin.write(userPrompt); - child.stdin.end(); - } catch (e) { - if (signal) signal.removeEventListener('abort', onAbort); - reject(new Error(`claude-cli stdin write failed (is the claude binary installed?): ${e instanceof Error ? e.message : String(e)}`)); - } - }); -} - -interface ParsedToolCall { - id: string; - name: string; - /** Stringified JSON, matching the ai-sdk LanguageModelV2ToolCall.input contract. */ - input: string; -} - -/** - * Locate and parse the `...` block in the assistant's - * raw text response. Returns the parsed tool calls plus whatever prose - * surrounded the block. Returns an empty `toolCalls` array when no block is - * present, malformed, or unterminated — the caller then treats the full - * raw text as a final text response. - */ -function extractToolCalls(raw: string): { - toolCalls: ParsedToolCall[]; - beforeText: string; - afterText: string; -} { - const openTag = ''; - const closeTag = ''; - const openIdx = raw.indexOf(openTag); - if (openIdx === -1) { - return { toolCalls: [], beforeText: raw.trim(), afterText: '' }; - } - const closeIdx = raw.indexOf(closeTag, openIdx + openTag.length); - if (closeIdx === -1) { - // Unterminated block — recover gracefully. - return { toolCalls: [], beforeText: raw.trim(), afterText: '' }; - } - - const beforeText = raw.slice(0, openIdx).trim(); - const afterText = raw.slice(closeIdx + closeTag.length).trim(); - let inner = raw.slice(openIdx + openTag.length, closeIdx).trim(); - - if (inner.startsWith('```')) { - inner = inner.replace(/^```(?:json|JSON)?\s*\n?/, '').replace(/\n?```$/, '').trim(); - } - - let parsed: unknown; - try { - parsed = JSON.parse(inner); - } catch { - return { toolCalls: [], beforeText: raw.trim(), afterText: '' }; - } - if (!Array.isArray(parsed)) { - return { toolCalls: [], beforeText: raw.trim(), afterText: '' }; - } - - const toolCalls: ParsedToolCall[] = []; - for (const entry of parsed) { - if (!entry || typeof entry !== 'object') continue; - const e = entry as Record; - const name = typeof e.name === 'string' ? e.name : null; - if (!name) continue; - const id = typeof e.id === 'string' && e.id.length > 0 - ? e.id - : `toolu_claude_cli_${Math.random().toString(36).slice(2, 12)}`; - const inputJson = JSON.stringify(e.input ?? {}); - toolCalls.push({ id, name, input: inputJson }); - } - - return { toolCalls, beforeText, afterText }; -} - -/** - * Strip provider prefixes (`anthropic:`, `litellm:`, `claude-cli:`) that the - * underlying CLI does not understand. The gateway hands us a bare model id - * via `recipe.aliases` resolution, but defensive normalization here keeps - * direct LanguageModelV2 construction (in tests, for example) ergonomic. - */ -function normalizeModel(model: string): string { - const idx = model.indexOf(':'); - return idx >= 0 ? model.slice(idx + 1) : model; -} - -export class ClaudeCliLanguageModel implements LanguageModelV2 { - readonly specificationVersion = 'v2' as const; - readonly provider = 'claude-cli'; - readonly modelId: string; - readonly supportedUrls = {}; - - constructor(modelId: string) { - this.modelId = normalizeModel(modelId); - } - - async doGenerate(options: LanguageModelV2CallOptions): Promise<{ - content: LanguageModelV2Content[]; - finishReason: 'stop' | 'length' | 'content-filter' | 'tool-calls' | 'error' | 'other' | 'unknown'; - usage: { inputTokens: number | undefined; outputTokens: number | undefined; totalTokens: number | undefined }; - warnings: never[]; - }> { - const { systemText, userPrompt } = renderPrompt(options.prompt); - const toolInstructions = buildToolUseInstructions(options.tools); - const systemPrompt = [systemText, toolInstructions].filter(s => s.length > 0).join('\n'); - - const result = await runClaude(systemPrompt, userPrompt, this.modelId, options.abortSignal); - const { toolCalls, beforeText, afterText } = extractToolCalls(result.result); - - const content: LanguageModelV2Content[] = []; - if (beforeText) content.push({ type: 'text', text: beforeText }); - for (const call of toolCalls) { - content.push({ - type: 'tool-call', - toolCallId: call.id, - toolName: call.name, - input: call.input, - }); - } - if (afterText) content.push({ type: 'text', text: afterText }); - if (content.length === 0) { - // Empty response — still hand the caller a well-formed content array. - content.push({ type: 'text', text: result.result ?? '' }); - } - - const finishReason = toolCalls.length > 0 ? 'tool-calls' as const : 'stop' as const; - const inputTokens = result.usage?.input_tokens; - const outputTokens = result.usage?.output_tokens; - const totalTokens = (inputTokens ?? 0) + (outputTokens ?? 0); - - return { - content, - finishReason, - usage: { - inputTokens, - outputTokens, - totalTokens: inputTokens !== undefined && outputTokens !== undefined ? totalTokens : undefined, - }, - warnings: [], - }; - } - - async doStream(): Promise { - throw new Error( - 'claude-cli LanguageModel does not support streaming. Use doGenerate or set ' + - 'the model on a non-streaming chat surface (gateway.toolLoop is non-streaming).', - ); - } -} diff --git a/src/core/ai/recipes/claude-cli.ts b/src/core/ai/recipes/claude-cli.ts deleted file mode 100644 index 2f1accbfe..000000000 --- a/src/core/ai/recipes/claude-cli.ts +++ /dev/null @@ -1,71 +0,0 @@ -import type { Recipe } from '../types.ts'; - -/** - * Claude via the local `claude` CLI binary, using its built-in OAuth session - * (Claude Code / Claude Max subscription). No ANTHROPIC_API_KEY needed — the - * CLI manages its own auth state and the gateway dispatches via subprocess. - * - * Solves the #334 case where Max subscribers want Minions subagent dispatch - * to run against their existing subscription instead of paying per-token API - * charges. The recipe sits alongside the existing `anthropic` recipe so users - * pick per call: `anthropic:claude-sonnet-4-6` (API key + per-token billing) - * vs `claude-cli:claude-sonnet-4-6` (OAuth subscription, no API key). - * - * Chat-only. Claude has no first-party embedding model; users wanting an - * Anthropic chat path with embeddings still combine this with openai/google/ - * voyage for embedding the way the existing `anthropic` recipe documents. - * - * Auth: `auth_env.required: []` because the CLI handles auth itself. The - * `claude` binary on PATH (or `GBRAIN_CLAUDE_CLI_BIN`) IS the auth surface; - * there is nothing for the gateway to forward. - * - * Setup expectation: `claude` CLI installed and logged in (Claude Code - * onboarding does this), or `GBRAIN_CLAUDE_CLI_BIN` pointing at the binary. - */ -export const claudeCli: Recipe = { - id: 'claude-cli', - name: 'Claude (via CLI)', - tier: 'native', - implementation: 'claude-cli', - // The CLI owns auth; no env vars are required from the gateway side. - auth_env: { - required: [], - }, - touchpoints: { - // No embedding or expansion touchpoints — chat-only. - chat: { - models: [ - 'claude-opus-4-7', - 'claude-sonnet-4-6', - 'claude-haiku-4-5-20251001', - ], - supports_tools: true, - supports_subagent_loop: true, - // The CLI handles caching internally and does not surface it via the - // standard cache_control control plane. From the gateway's POV the - // model does not support prompt caching. - supports_prompt_cache: false, - max_context_tokens: 200000, - // Cost figures match the underlying Claude API tier, but the actual - // bill is borne by the subscription. We report them for the budget - // ledger's per-call accounting; operators on flat-rate subscriptions - // can treat the numbers as nominal. - cost_per_1m_input_usd: 3.0, - cost_per_1m_output_usd: 15.0, - price_last_verified: '2026-06-17', - }, - }, - // Friendly aliases mirror the `anthropic` recipe so config strings stay - // portable: switching `anthropic:claude-sonnet-4-6` to `claude-cli:claude-sonnet-4-6` - // is a one-token edit. Reverse aliases rewrite legacy IDs back to canonical. - aliases: { - 'claude-haiku-4-5': 'claude-haiku-4-5-20251001', - 'claude-sonnet-4-6-20250929': 'claude-sonnet-4-6', - 'sonnet': 'claude-sonnet-4-6', - 'haiku': 'claude-haiku-4-5-20251001', - 'opus': 'claude-opus-4-7', - }, - setup_hint: - 'Install Claude Code (`claude` CLI) and run `claude` once to log in. ' + - 'Set GBRAIN_CLAUDE_CLI_BIN if the binary is not on PATH.', -}; diff --git a/src/core/ai/recipes/index.ts b/src/core/ai/recipes/index.ts index 7931323bb..eb751ec61 100644 --- a/src/core/ai/recipes/index.ts +++ b/src/core/ai/recipes/index.ts @@ -9,7 +9,6 @@ import type { Recipe } from '../types.ts'; import { openai } from './openai.ts'; import { google } from './google.ts'; import { anthropic } from './anthropic.ts'; -import { claudeCli } from './claude-cli.ts'; import { ollama } from './ollama.ts'; import { openrouter } from './openrouter.ts'; import { voyage } from './voyage.ts'; @@ -32,7 +31,6 @@ const ALL: Recipe[] = [ openai, google, anthropic, - claudeCli, ollama, openrouter, voyage, diff --git a/src/core/ai/types.ts b/src/core/ai/types.ts index 40bca1b32..8fc785e3d 100644 --- a/src/core/ai/types.ts +++ b/src/core/ai/types.ts @@ -22,8 +22,7 @@ export type Implementation = | 'native-openai' | 'native-google' | 'native-anthropic' - | 'openai-compatible' - | 'claude-cli'; + | 'openai-compatible'; export interface EmbeddingTouchpoint { models: string[]; diff --git a/test/claude-cli-recipe.test.ts b/test/claude-cli-recipe.test.ts deleted file mode 100644 index 26339b457..000000000 --- a/test/claude-cli-recipe.test.ts +++ /dev/null @@ -1,535 +0,0 @@ -/** - * Tests for the claude-cli LanguageModelV2 implementation that the - * `claude-cli` recipe instantiates. - * - * Strategy: a POSIX shell stub at GBRAIN_CLAUDE_CLI_BIN emits scripted - * --output-format json envelopes. Tests exercise the LanguageModelV2 - * doGenerate surface: text round trip, tool-call extraction (single + - * multiple parallel), abort semantics, context-isolation flags. No - * claude-cli installation or API credits required. - * - * Recipe registration is also smoke-tested: getRecipe('claude-cli') - * returns a chat-only Recipe with the right model list. - * - * Env isolation: GBRAIN_CLAUDE_CLI_BIN is set per-test via withEnv(), - * NOT in beforeAll. The provider reads the env var at spawn time so - * withEnv's save/restore in try/finally is sufficient; no leakage to - * sibling test files in the same bun-test process. - */ -import { describe, test, expect, beforeAll, afterAll } from 'bun:test'; -import { writeFileSync, chmodSync, mkdirSync, rmSync } from 'node:fs'; -import { join } from 'node:path'; -import { tmpdir } from 'node:os'; -import type { LanguageModelV2CallOptions } from '@ai-sdk/provider'; -import { withEnv } from './helpers/with-env.ts'; - -const stubDir = join(tmpdir(), `claude-cli-recipe-stub-${process.pid}`); -const stubBin = join(stubDir, 'claude'); -const stubResponsePath = join(stubDir, 'claude_response.json'); - -beforeAll(() => { - mkdirSync(stubDir, { recursive: true }); - const stub = [ - '#!/bin/sh', - 'cat > /dev/null', - 'case " $* " in', - ' *" --print "*) ;;', - ' *) echo "missing --print in argv: $*" >&2; exit 64 ;;', - 'esac', - `cat "${stubResponsePath}"`, - ].join('\n'); - writeFileSync(stubBin, stub); - chmodSync(stubBin, 0o755); -}); - -afterAll(() => { - rmSync(stubDir, { recursive: true, force: true }); -}); - -function withStubEnv(fn: () => T | Promise): Promise { - return withEnv({ GBRAIN_CLAUDE_CLI_BIN: stubBin }, fn); -} - -function stageResponse(envelope: Record): void { - writeFileSync(stubResponsePath, JSON.stringify(envelope)); -} - -function baseEnvelope(result: string, overrides: Record = {}): Record { - return { - type: 'result', - subtype: 'success', - is_error: false, - result, - stop_reason: 'end_turn', - session_id: 'test-session', - num_turns: 1, - usage: { - input_tokens: 12, - output_tokens: 34, - cache_read_input_tokens: 0, - cache_creation_input_tokens: 0, - }, - ...overrides, - }; -} - -function userMessage(text: string): LanguageModelV2CallOptions['prompt'][number] { - return { role: 'user', content: [{ type: 'text', text }] }; -} - -describe('claude-cli recipe registration', () => { - test('getRecipe returns chat-only Recipe with the documented models', async () => { - const { getRecipe } = await import('../src/core/ai/recipes/index.ts'); - const recipe = getRecipe('claude-cli'); - expect(recipe).toBeDefined(); - expect(recipe!.id).toBe('claude-cli'); - expect(recipe!.implementation).toBe('claude-cli'); - expect(recipe!.touchpoints.chat).toBeDefined(); - expect(recipe!.touchpoints.chat!.supports_tools).toBe(true); - expect(recipe!.touchpoints.chat!.supports_subagent_loop).toBe(true); - expect(recipe!.touchpoints.chat!.models).toContain('claude-sonnet-4-6'); - expect(recipe!.touchpoints.embedding).toBeUndefined(); - expect(recipe!.touchpoints.expansion).toBeUndefined(); - }); - - test('recipe aliases map short names to canonical model ids', async () => { - const { getRecipe } = await import('../src/core/ai/recipes/index.ts'); - const recipe = getRecipe('claude-cli'); - expect(recipe!.aliases!['sonnet']).toBe('claude-sonnet-4-6'); - expect(recipe!.aliases!['haiku']).toBe('claude-haiku-4-5-20251001'); - }); -}); - -describe('claude-cli LanguageModel — text-only round trip', () => { - test('returns a single text content block with usage + stop finish reason', async () => { - await withStubEnv(async () => { - stageResponse(baseEnvelope('hello world')); - const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts'); - const model = new ClaudeCliLanguageModel('claude-sonnet-4-6'); - const result = await model.doGenerate({ - prompt: [userMessage('hi')], - } as LanguageModelV2CallOptions); - - expect(result.finishReason).toBe('stop'); - expect(result.content).toHaveLength(1); - expect(result.content[0]).toEqual({ type: 'text', text: 'hello world' }); - expect(result.usage.inputTokens).toBe(12); - expect(result.usage.outputTokens).toBe(34); - }); - }); - - test('strips provider prefixes from the model id', async () => { - const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts'); - const model = new ClaudeCliLanguageModel('anthropic:claude-sonnet-4-6'); - expect(model.modelId).toBe('claude-sonnet-4-6'); - }); -}); - -describe('claude-cli LanguageModel — tool use', () => { - test('parses block into LanguageModelV2 tool-call content', async () => { - await withStubEnv(async () => { - stageResponse( - baseEnvelope( - [ - 'I will look up the pattern first.', - '', - '[{"id": "toolu_01ABC", "name": "search", "input": {"query": "n+1 query"}}]', - '', - ].join('\n'), - ), - ); - const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts'); - const model = new ClaudeCliLanguageModel('claude-sonnet-4-6'); - const result = await model.doGenerate({ - prompt: [userMessage('find n+1 queries')], - tools: [ - { - type: 'function', - name: 'search', - description: 'Search the brain', - inputSchema: { type: 'object', properties: { query: { type: 'string' } } }, - }, - ], - } as LanguageModelV2CallOptions); - - expect(result.finishReason).toBe('tool-calls'); - expect(result.content).toHaveLength(2); - expect(result.content[0]).toMatchObject({ type: 'text', text: 'I will look up the pattern first.' }); - expect(result.content[1]).toMatchObject({ - type: 'tool-call', - toolCallId: 'toolu_01ABC', - toolName: 'search', - input: '{"query":"n+1 query"}', - }); - }); - }); - - test('parses multiple parallel tool calls in a single block', async () => { - await withStubEnv(async () => { - stageResponse( - baseEnvelope( - [ - '', - '[', - ' {"id": "toolu_A", "name": "search", "input": {"query": "foo"}},', - ' {"id": "toolu_B", "name": "get_page", "input": {"slug": "areas/x"}}', - ']', - '', - ].join('\n'), - ), - ); - const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts'); - const model = new ClaudeCliLanguageModel('claude-sonnet-4-6'); - const result = await model.doGenerate({ - prompt: [userMessage('multi')], - tools: [ - { type: 'function', name: 'search', description: 's', inputSchema: { type: 'object', properties: {} } }, - { type: 'function', name: 'get_page', description: 'g', inputSchema: { type: 'object', properties: {} } }, - ], - } as LanguageModelV2CallOptions); - - const calls = result.content.filter(c => c.type === 'tool-call'); - expect(calls).toHaveLength(2); - expect(calls.map(c => (c as { toolName: string }).toolName)).toEqual(['search', 'get_page']); - expect(result.finishReason).toBe('tool-calls'); - }); - }); - - test('tolerates fenced JSON inside ', async () => { - await withStubEnv(async () => { - stageResponse( - baseEnvelope( - [ - '', - '```json', - '[{"id": "toolu_F", "name": "search", "input": {"q": "x"}}]', - '```', - '', - ].join('\n'), - ), - ); - const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts'); - const model = new ClaudeCliLanguageModel('claude-sonnet-4-6'); - const result = await model.doGenerate({ - prompt: [userMessage('fenced')], - tools: [{ type: 'function', name: 'search', description: '', inputSchema: { type: 'object', properties: {} } }], - } as LanguageModelV2CallOptions); - - const calls = result.content.filter(c => c.type === 'tool-call'); - expect(calls).toHaveLength(1); - }); - }); - - test('synthesizes an id when the model omits it', async () => { - await withStubEnv(async () => { - stageResponse( - baseEnvelope( - [ - '', - '[{"name": "search", "input": {"q": "x"}}]', - '', - ].join('\n'), - ), - ); - const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts'); - const model = new ClaudeCliLanguageModel('claude-sonnet-4-6'); - const result = await model.doGenerate({ - prompt: [userMessage('no id')], - tools: [{ type: 'function', name: 'search', description: '', inputSchema: { type: 'object', properties: {} } }], - } as LanguageModelV2CallOptions); - - const call = result.content.find(c => c.type === 'tool-call') as { toolCallId: string } | undefined; - expect(call).toBeDefined(); - expect(call!.toolCallId).toMatch(/^toolu_claude_cli_/); - }); - }); - - test('falls back to text on malformed JSON', async () => { - await withStubEnv(async () => { - stageResponse( - baseEnvelope( - [ - '', - 'not valid json', - '', - ].join('\n'), - ), - ); - const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts'); - const model = new ClaudeCliLanguageModel('claude-sonnet-4-6'); - const result = await model.doGenerate({ - prompt: [userMessage('malformed')], - tools: [{ type: 'function', name: 'search', description: '', inputSchema: { type: 'object', properties: {} } }], - } as LanguageModelV2CallOptions); - - expect(result.content.filter(c => c.type === 'tool-call')).toHaveLength(0); - expect(result.finishReason).toBe('stop'); - }); - }); - - test('returns text-only stop when tools are offered but model declines to call any', async () => { - // Real-world case: the model decides the user's request does not require - // a tool call, ignores the use_tools protocol, and answers directly. - // The recipe still must return clean LanguageModelV2 output so the - // caller (gateway.toolLoop) can treat the text as the final answer - // rather than wedge waiting for tool calls that never come. - await withStubEnv(async () => { - stageResponse( - baseEnvelope( - 'I do not actually need to call any tools for this. The answer is 42.', - { stop_reason: 'end_turn' }, - ), - ); - const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts'); - const model = new ClaudeCliLanguageModel('claude-sonnet-4-6'); - const result = await model.doGenerate({ - prompt: [userMessage('what is the meaning of life? you may use tools but do not need to')], - tools: [{ type: 'function', name: 'compute', description: 'Compute things', inputSchema: { type: 'object', properties: {} } }], - } as LanguageModelV2CallOptions); - - // No tool-call content blocks; caller treats this as a final answer. - expect(result.content.filter(c => c.type === 'tool-call')).toHaveLength(0); - // Text block present with the full model reply. - const textBlocks = result.content.filter(c => c.type === 'text'); - expect(textBlocks).toHaveLength(1); - expect((textBlocks[0] as { text: string }).text).toContain('42'); - // finishReason 'stop' tells the gateway-loop this is terminal output, - // not a partial mid-tool-loop state. - expect(result.finishReason).toBe('stop'); - }); - }); - - test('drops the block when the close tag is missing', async () => { - await withStubEnv(async () => { - stageResponse( - baseEnvelope( - [ - '', - '[{"id": "toolu_X", "name": "search", "input": {}}', - ].join('\n'), - ), - ); - const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts'); - const model = new ClaudeCliLanguageModel('claude-sonnet-4-6'); - const result = await model.doGenerate({ - prompt: [userMessage('unterminated')], - tools: [{ type: 'function', name: 'search', description: '', inputSchema: { type: 'object', properties: {} } }], - } as LanguageModelV2CallOptions); - - expect(result.content.filter(c => c.type === 'tool-call')).toHaveLength(0); - expect(result.finishReason).toBe('stop'); - }); - }); -}); - -describe('claude-cli LanguageModel — context isolation', () => { - test('argv includes --disable-slash-commands + --system-prompt and cwd is the dedicated tmpdir', async () => { - await withStubEnv(async () => { - const argvLog = join(stubDir, 'argv.log'); - const cwdLog = join(stubDir, 'cwd.log'); - const recordStub = [ - '#!/bin/sh', - `printf "%s\\n" "$@" > "${argvLog}"`, - `pwd > "${cwdLog}"`, - 'cat > /dev/null', - `cat "${stubResponsePath}"`, - ].join('\n'); - writeFileSync(stubBin, recordStub); - chmodSync(stubBin, 0o755); - stageResponse(baseEnvelope('ok')); - - const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts'); - const model = new ClaudeCliLanguageModel('claude-sonnet-4-6'); - await model.doGenerate({ - prompt: [ - { role: 'system', content: 'You are gbrain subagent.' }, - userMessage('hi'), - ], - } as LanguageModelV2CallOptions); - - const fs = require('node:fs'); - const argv = fs.readFileSync(argvLog, 'utf8').split('\n').filter(Boolean); - const cwd = fs.readFileSync(cwdLog, 'utf8').trim(); - - expect(argv).toContain('--print'); - expect(argv).toContain('--output-format'); - expect(argv).toContain('json'); - expect(argv).toContain('--disable-slash-commands'); - // Agent-isolation hardening: no built-in tools, no inherited MCP servers. - expect(argv).toContain('--tools'); - expect(argv).toContain('--strict-mcp-config'); - expect(argv).toContain('--system-prompt'); - expect(argv).toContain('You are gbrain subagent.'); - expect(cwd).toMatch(/gbrain-claude-cli-cwd-\d+$/); - - const fastStub = [ - '#!/bin/sh', - 'cat > /dev/null', - `cat "${stubResponsePath}"`, - ].join('\n'); - writeFileSync(stubBin, fastStub); - chmodSync(stubBin, 0o755); - }); - }); - - test('scrubs ANTHROPIC_* credentials from the child env (subscription-only auth)', async () => { - await withStubEnv(async () => { - await withEnv( - { - ANTHROPIC_API_KEY: 'sk-should-never-leak', - ANTHROPIC_AUTH_TOKEN: 'tok-should-never-leak', - ANTHROPIC_BASE_URL: 'https://proxy.should.never.leak', - }, - async () => { - const envLog = join(stubDir, 'env.log'); - const envStub = [ - '#!/bin/sh', - `printf "key=%s\\ntoken=%s\\nbase=%s\\n" "\${ANTHROPIC_API_KEY:-UNSET}" "\${ANTHROPIC_AUTH_TOKEN:-UNSET}" "\${ANTHROPIC_BASE_URL:-UNSET}" > "${envLog}"`, - 'cat > /dev/null', - `cat "${stubResponsePath}"`, - ].join('\n'); - writeFileSync(stubBin, envStub); - chmodSync(stubBin, 0o755); - stageResponse(baseEnvelope('ok')); - - try { - const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts'); - const model = new ClaudeCliLanguageModel('claude-sonnet-4-6'); - await model.doGenerate({ - prompt: [userMessage('hi')], - } as LanguageModelV2CallOptions); - - const fs = require('node:fs'); - const seen = fs.readFileSync(envLog, 'utf8'); - expect(seen).toContain('key=UNSET'); - expect(seen).toContain('token=UNSET'); - expect(seen).toContain('base=UNSET'); - } finally { - const fastStub = [ - '#!/bin/sh', - 'cat > /dev/null', - `cat "${stubResponsePath}"`, - ].join('\n'); - writeFileSync(stubBin, fastStub); - chmodSync(stubBin, 0o755); - } - }, - ); - }); - }); -}); - -describe('claude-cli LanguageModel — abort + error envelopes', () => { - test('SIGTERMs the child on AbortSignal', async () => { - await withStubEnv(async () => { - const slowStub = [ - '#!/bin/sh', - 'cat > /dev/null', - 'sleep 30', - 'echo "{}"', - ].join('\n'); - writeFileSync(stubBin, slowStub); - chmodSync(stubBin, 0o755); - try { - const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts'); - const model = new ClaudeCliLanguageModel('claude-sonnet-4-6'); - const ac = new AbortController(); - const promise = model.doGenerate({ - prompt: [userMessage('slow')], - abortSignal: ac.signal, - } as LanguageModelV2CallOptions); - setTimeout(() => ac.abort(), 30); - await expect(promise).rejects.toThrow(/aborted/); - } finally { - const fastStub = [ - '#!/bin/sh', - 'cat > /dev/null', - `cat "${stubResponsePath}"`, - ].join('\n'); - writeFileSync(stubBin, fastStub); - chmodSync(stubBin, 0o755); - } - }); - }); - - test('rejects when stub reports is_error: true', async () => { - await withStubEnv(async () => { - stageResponse({ ...baseEnvelope('boom'), is_error: true }); - const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts'); - const model = new ClaudeCliLanguageModel('claude-sonnet-4-6'); - await expect( - model.doGenerate({ prompt: [userMessage('x')] } as LanguageModelV2CallOptions), - ).rejects.toThrow(/claude-cli reported error/); - }); - }); - - test('rejects on non-JSON output', async () => { - await withStubEnv(async () => { - writeFileSync(stubResponsePath, 'this is not json'); - const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts'); - const model = new ClaudeCliLanguageModel('claude-sonnet-4-6'); - await expect( - model.doGenerate({ prompt: [userMessage('x')] } as LanguageModelV2CallOptions), - ).rejects.toThrow(/claude-cli output not JSON/); - }); - }); - - test('accepts a verbose-mode JSON event array and picks the result event', async () => { - // With `"verbose": true` in ~/.claude/settings.json the CLI emits an array - // of events instead of the bare result object (no CLI flag disables it). - await withStubEnv(async () => { - writeFileSync( - stubResponsePath, - JSON.stringify([ - { type: 'system', subtype: 'init', session_id: 'test-session', tools: [], mcp_servers: [] }, - baseEnvelope('hello from array'), - ]), - ); - const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts'); - const model = new ClaudeCliLanguageModel('claude-sonnet-4-6'); - const result = await model.doGenerate({ - prompt: [userMessage('hi')], - } as LanguageModelV2CallOptions); - expect(result.finishReason).toBe('stop'); - expect(result.content[0]).toEqual({ type: 'text', text: 'hello from array' }); - }); - }); - - test('rejects a verbose-mode event array that lacks a result event', async () => { - // Verbose mode emits an event array; a truncated stream (or one carrying - // only init/system events) has no result event to unwrap. - await withStubEnv(async () => { - writeFileSync( - stubResponsePath, - JSON.stringify([ - { type: 'system', subtype: 'init', session_id: 'test-session', tools: [], mcp_servers: [] }, - { type: 'assistant', message: { role: 'assistant', content: [] } }, - ]), - ); - const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts'); - const model = new ClaudeCliLanguageModel('claude-sonnet-4-6'); - await expect( - model.doGenerate({ prompt: [userMessage('x')] } as LanguageModelV2CallOptions), - ).rejects.toThrow(/had no "result" event/); - }); - }); - - test('rejects cleanly when the claude binary is missing (no worker crash)', async () => { - // A missing binary must surface as a rejected promise via the spawn 'error' - // handler; the child stdin 'error' (EPIPE) handler swallows the pipe failure - // so it never escalates to an unhandled rejection that would down the worker. - await withEnv({ GBRAIN_CLAUDE_CLI_BIN: join(stubDir, 'nonexistent-claude') }, async () => { - const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts'); - const model = new ClaudeCliLanguageModel('claude-sonnet-4-6'); - await expect( - model.doGenerate({ prompt: [userMessage('x')] } as LanguageModelV2CallOptions), - ).rejects.toThrow(/claude-cli spawn failed/); - }); - }); - - test('doStream throws not-supported', async () => { - const { ClaudeCliLanguageModel } = await import('../src/core/ai/providers/claude-cli-language-model.ts'); - const model = new ClaudeCliLanguageModel('claude-sonnet-4-6'); - await expect(model.doStream()).rejects.toThrow(/does not support streaming/); - }); -});