feat(ai): add Mistral provider recipe (#3001)

Adds an EU-hosted provider covering embedding, expansion and chat on one
OpenAI-compatible endpoint (https://api.mistral.ai/v1), so a brain that must
stay inside EU jurisdiction does not need a US hop for any AI touchpoint.

Every field is measured against the live API, not copied from docs:

- mistral-embed is fixed 1024 dims and accepts no dimension parameter.
  Both spellings are rejected: {"dimensions": N} returns 400 extra_forbidden,
  {"output_dimension": N} returns 400 "does not support output_dimension".
  The generic openai-compatible branch of dimsProviderOptions() already falls
  through to `return undefined` for these model ids, so nothing is emitted.
  Same contract as voyage-4-nano, pinned by a negative assertion in the test.
- max_batch_tokens 65536: a 65,286-token batch is accepted, 66,960 returns
  400 code 3210 "Too many tokens overall, split into more batches."
- chars_per_token 2: the value is a DIVISOR in splitByTokenBudget()
  (estTokens = text.length / charsPerToken), so lower is the conservative
  direction. The module default of 4 assumes English prose; a German-language
  corpus measured 3.58 chars/token, which the default overshoots toward
  overflow.

codestral-embed is deliberately left out: it returns 1536 dims, and a
touchpoint carries a single default_dims. Listing it under a 1024 declaration
is the mixed-dim case embedding-dim-check.ts exists to catch.

Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
bo-developing
2026-07-20 13:42:41 -07:00
committed by GitHub
co-authored by Claude Opus 4.8
parent f3e78fd2fb
commit c873ce3014
4 changed files with 174 additions and 0 deletions
+2
View File
@@ -24,6 +24,7 @@ import { azureOpenAI } from './azure-openai.ts';
import { zeroentropyai } from './zeroentropyai.ts';
import { llamaServerReranker } from './llama-server-reranker.ts';
import { moonshot } from './moonshot.ts';
import { mistral } from './mistral.ts';
const ALL: Recipe[] = [
openai,
@@ -44,6 +45,7 @@ const ALL: Recipe[] = [
azureOpenAI,
zeroentropyai,
moonshot,
mistral,
];
/** Map from `provider:id` key to recipe. */
+84
View File
@@ -0,0 +1,84 @@
import type { Recipe } from '../types.ts';
/**
* Mistral AI exposes an OpenAI-compatible API at https://api.mistral.ai/v1
* (/embeddings + /chat/completions). EU-hosted — the reason this recipe
* exists: a brain that must stay inside EU jurisdiction can run embed +
* expansion + chat on a single provider without a US hop.
*
* Verified against the live API on 2026-07-19 (model catalog, embedding
* dimensions, dimension-parameter rejection, and the batch ceiling — see
* the notes on each field below).
*
* DIMENSIONS — mistral-embed is FIXED 1024 and accepts NO dimension
* parameter at all. Both spellings are rejected upstream:
* {"dimensions": 512} -> 400 extra_forbidden (not in the API schema)
* {"output_dimension": 512} -> 400 "This model does not support output_dimension"
* The generic `openai-compatible` branch of dims.ts:dimsProviderOptions()
* already falls through to `return undefined` for these model ids, so no
* dimension field is emitted. Do NOT add mistral-embed to any of the
* flexible-dim allowlists there — it would 400 every embed call. Same
* contract as voyage-4-nano, for the same reason.
*
* codestral-embed / codestral-embed-2505 are deliberately NOT listed: they
* return 1536 dims, and a touchpoint carries a single `default_dims`.
* Mixing them under a 1024 declaration is the mixed-dim footgun
* embedding-dim-check.ts exists to catch. They are code-retrieval models
* anyway; a prose brain wants mistral-embed.
*/
export const mistral: Recipe = {
id: 'mistral',
name: 'Mistral AI',
tier: 'openai-compat',
implementation: 'openai-compatible',
base_url_default: 'https://api.mistral.ai/v1',
auth_env: {
required: ['MISTRAL_API_KEY'],
setup_url: 'https://console.mistral.ai/api-keys',
},
touchpoints: {
embedding: {
models: ['mistral-embed', 'mistral-embed-2312'],
default_dims: 1024,
// Mistral's published list price. Advisory only — canonical embedding
// spend accounting lives in src/core/embedding-pricing.ts.
cost_per_1m_tokens_usd: 0.1,
price_last_verified: '2026-07-19',
// Measured ceiling, not a doc guess: the /embeddings endpoint accepts a
// 65,286-token batch and rejects 66,960 with
// 400 code 3210 "Too many tokens overall, split into more batches."
// -> the real cap is 65,536 (64K) tokens per request.
max_batch_tokens: 65_536,
// chars_per_token is a DIVISOR in splitByTokenBudget()
// (estTokens = text.length / charsPerToken), so a LOWER value is the
// conservative direction. The module default of 4 is an English-prose
// assumption; German prose measured 3.58 here, and code/JSON/CJK runs
// denser still. 2 keeps the estimate above the real token count for
// every content shape we see.
chars_per_token: 2,
// With safety_factor 0.5 the pre-split budget is 32,768 estimated
// tokens = 65,536 chars. Worst realistic density (~1.5 chars/token)
// puts that at ~43.7K real tokens — still clear of the 64K ceiling.
safety_factor: 0.5,
},
expansion: {
models: ['ministral-3b-latest', 'mistral-small-latest'],
price_last_verified: '2026-07-19',
},
chat: {
models: [
'mistral-small-latest', 'mistral-medium-latest', 'mistral-large-latest',
'ministral-3b-latest', 'ministral-8b-latest', 'magistral-small-latest',
],
supports_tools: true,
// Same call as the Moonshot recipe: ordinary tool calls are fine, but
// gbrain's subagent loop stays Anthropic-pinned for stable tool_use_id
// behavior across crashes/replays.
supports_subagent_loop: false,
supports_prompt_cache: false,
max_context_tokens: 262144,
price_last_verified: '2026-07-19',
},
},
setup_hint: 'Get an API key at https://console.mistral.ai/api-keys, then `export MISTRAL_API_KEY=...` and use `mistral:mistral-embed` (1024 dims) for embeddings.',
};
+3
View File
@@ -37,6 +37,9 @@ export const EMBEDDING_PRICING: Record<string, EmbeddingPricing> = {
'voyage:voyage-4-large': { pricePerMTok: 0.18 },
// ZeroEntropy (https://zeroentropy.dev/pricing — zembed-1)
'zeroentropyai:zembed-1': { pricePerMTok: 0.05 },
// Mistral (https://mistral.ai/pricing/api/, verified 2026-07-19)
'mistral:mistral-embed': { pricePerMTok: 0.10 },
'mistral:mistral-embed-2312': { pricePerMTok: 0.10 },
};
export type PriceLookupResult =
+85
View File
@@ -0,0 +1,85 @@
/**
* Mistral recipe smoke.
*
* The load-bearing assertion here is the negative one: mistral-embed rejects
* every dimension parameter with HTTP 400, so dimsProviderOptions() must emit
* no dimension field for it. Same contract as voyage-4-nano, pinned the same
* way (see the negative regression assertion in test/ai/gateway.test.ts).
*/
import { describe, expect, test } from 'bun:test';
import { getRecipe } from '../../src/core/ai/recipes/index.ts';
import { defaultResolveAuth } from '../../src/core/ai/gateway.ts';
import { assertTouchpoint } from '../../src/core/ai/model-resolver.ts';
import { AIConfigError } from '../../src/core/ai/errors.ts';
import { dimsProviderOptions } from '../../src/core/ai/dims.ts';
import { lookupEmbeddingPrice } from '../../src/core/embedding-pricing.ts';
describe('recipe: mistral', () => {
test('registered with expected OpenAI-compatible shape', () => {
const r = getRecipe('mistral');
expect(r).toBeDefined();
expect(r!.id).toBe('mistral');
expect(r!.tier).toBe('openai-compat');
expect(r!.implementation).toBe('openai-compatible');
expect(r!.base_url_default).toBe('https://api.mistral.ai/v1');
expect(r!.auth_env?.required).toEqual(['MISTRAL_API_KEY']);
});
test('embedding touchpoint pins the measured 1024 dims and 64K batch ceiling', () => {
const e = getRecipe('mistral')!.touchpoints.embedding;
expect(e).toBeDefined();
expect(e!.models).toContain('mistral-embed');
expect(e!.default_dims).toBe(1024);
// Measured: a 65,286-token batch is accepted, 66,960 returns 400 code 3210.
expect(e!.max_batch_tokens).toBe(65_536);
// chars_per_token is a DIVISOR in splitByTokenBudget(), so a lower value
// is the conservative direction. The module default of 4 is an English
// assumption and overshoots on denser prose.
expect(e!.chars_per_token).toBe(2);
});
test('NEGATIVE: no dimension parameter is emitted for mistral-embed', () => {
// Mistral rejects both spellings:
// {"dimensions": N} -> 400 extra_forbidden
// {"output_dimension": N} -> 400 "does not support output_dimension"
// If a future change adds mistral-embed to a flexible-dim allowlist in
// dims.ts, this assertion fails before it reaches users as a 400 on every
// embed call.
expect(dimsProviderOptions('openai-compatible', 'mistral-embed', 1024)).toBeUndefined();
expect(dimsProviderOptions('openai-compatible', 'mistral-embed-2312', 1024)).toBeUndefined();
});
test('embedding models resolve to a known price', () => {
// An unknown price makes the embedding spend cap fail closed.
expect(lookupEmbeddingPrice('mistral:mistral-embed').kind).toBe('known');
expect(lookupEmbeddingPrice('mistral:mistral-embed-2312').kind).toBe('known');
});
test('chat and expansion touchpoints accept their configured models', () => {
const r = getRecipe('mistral')!;
expect(r.touchpoints.chat!.supports_tools).toBe(true);
expect(r.touchpoints.chat!.supports_subagent_loop).toBe(false);
expect(() => assertTouchpoint(r, 'chat', 'mistral-small-latest')).not.toThrow();
expect(() => assertTouchpoint(r, 'expansion', 'ministral-3b-latest')).not.toThrow();
expect(() => assertTouchpoint(r, 'embedding', 'mistral-embed')).not.toThrow();
});
test('codestral-embed is deliberately absent (1536 dims would mix under a 1024 declaration)', () => {
const e = getRecipe('mistral')!.touchpoints.embedding!;
expect(e.models).not.toContain('codestral-embed');
expect(e.models).not.toContain('codestral-embed-2505');
});
test('default auth: MISTRAL_API_KEY set -> Bearer token', () => {
const r = getRecipe('mistral')!;
const auth = defaultResolveAuth(r, { MISTRAL_API_KEY: 'fake-mistral-key' }, 'embedding');
expect(auth.headerName).toBe('Authorization');
expect(auth.token).toBe('Bearer fake-mistral-key');
});
test('default auth: missing MISTRAL_API_KEY -> AIConfigError', () => {
const r = getRecipe('mistral')!;
expect(() => defaultResolveAuth(r, {}, 'embedding')).toThrow(AIConfigError);
});
});