fix(agent): cap max_tokens on the agent chat turn (#4005) (#4031)

This commit is contained in:
oxoxDev
2026-06-24 10:06:38 -07:00
committed by GitHub
parent 02f9769bc5
commit 496292e17c
7 changed files with 82 additions and 15 deletions
+5 -2
View File
@@ -27,7 +27,7 @@ use crate::openhuman::agent::stop_hooks::{current_stop_hooks, StopDecision, Turn
use crate::openhuman::context::guard::{ContextCheckResult, ContextGuard};
use crate::openhuman::context::{summarize_chat_history, EngineAutocompact};
use crate::openhuman::inference::provider::{
ChatMessage, ChatRequest, Provider, ProviderCapabilityError,
ChatMessage, ChatRequest, Provider, ProviderCapabilityError, AGENT_TURN_MAX_OUTPUT_TOKENS,
};
use super::super::parse::build_native_assistant_history;
@@ -528,7 +528,10 @@ pub(crate) async fn run_turn_engine(
messages: &prepared_messages_vec,
tools: request_tools,
stream: delta_tx_opt.as_ref(),
max_tokens: None,
// Cap the turn so reservation-pricing providers price their
// pre-flight against a realistic budget, not the full output
// window (TAURI-RUST-C62).
max_tokens: Some(AGENT_TURN_MAX_OUTPUT_TOKENS),
},
model,
temperature,
@@ -6,7 +6,9 @@ use super::super::types::Agent;
use crate::openhuman::agent::harness;
use crate::openhuman::agent::progress::AgentProgress;
use crate::openhuman::context::ARCHIVIST_EXTRACTION_PROMPT;
use crate::openhuman::inference::provider::{ChatMessage, ChatRequest, ProviderDelta, UsageInfo};
use crate::openhuman::inference::provider::{
ChatMessage, ChatRequest, ProviderDelta, UsageInfo, AGENT_TURN_MAX_OUTPUT_TOKENS,
};
impl Agent {
// ─────────────────────────────────────────────────────────────────
@@ -120,7 +122,8 @@ impl Agent {
messages: &messages,
tools: None,
stream: delta_tx_opt.as_ref(),
max_tokens: None,
// Reservation-pricing pre-flight budget cap (TAURI-RUST-C62).
max_tokens: Some(AGENT_TURN_MAX_OUTPUT_TOKENS),
},
effective_model,
self.temperature,
@@ -46,6 +46,7 @@ use crate::openhuman::agent_tool_policy::ToolPolicySession;
use crate::openhuman::context::ReductionOutcome;
use crate::openhuman::inference::provider::{
ChatMessage, ChatRequest, ConversationMessage, Provider, ProviderDelta, ToolCall, UsageInfo,
AGENT_TURN_MAX_OUTPUT_TOKENS,
};
use crate::openhuman::tools::{Tool, ToolSpec};
@@ -476,7 +477,8 @@ impl CheckpointStrategy for AgentCheckpoint {
messages: &messages,
tools: None,
stream: delta_tx_opt.as_ref(),
max_tokens: None,
// Reservation-pricing pre-flight budget cap (TAURI-RUST-C62).
max_tokens: Some(AGENT_TURN_MAX_OUTPUT_TOKENS),
},
&self.model,
self.temperature,
@@ -5,7 +5,9 @@
//! instead of erroring. Falls back to a deterministic digest summary if the
//! summarization call fails or returns no prose.
use crate::openhuman::inference::provider::{ChatMessage, ChatRequest, Provider};
use crate::openhuman::inference::provider::{
ChatMessage, ChatRequest, Provider, AGENT_TURN_MAX_OUTPUT_TOKENS,
};
/// Sub-agent [`CheckpointStrategy`]: when the iteration cap is hit, summarize
/// the run-so-far into a resumable checkpoint (so the delegating agent can
@@ -44,7 +46,9 @@ impl super::super::super::engine::CheckpointStrategy for SubagentCheckpoint<'_>
messages: &summary_input,
tools: None,
stream: None,
max_tokens: None,
// Bounded progress-summary turn; cap also keeps the
// reservation-pricing pre-flight realistic (TAURI-RUST-C62).
max_tokens: Some(AGENT_TURN_MAX_OUTPUT_TOKENS),
},
&self.model,
self.temperature,
@@ -166,6 +166,39 @@ fn native_request_serializes_max_tokens_only_when_set() {
);
}
#[test]
fn agent_turn_cap_reaches_the_wire() {
// The agent turns now set `max_tokens: Some(AGENT_TURN_MAX_OUTPUT_TOKENS)`
// (#4005); assert that exact cap serializes onto the wire so a careless edit
// to the const — or a regression to `None` on the agent path — fails CI and
// can't silently restore the full-window reservation that 402s low-balance
// BYO users (TAURI-RUST-C62).
let cap = crate::openhuman::inference::provider::AGENT_TURN_MAX_OUTPUT_TOKENS;
assert!(
(8192..=32768).contains(&cap),
"agent cap must stay well above realistic turns yet below the model's full output window; got {cap}"
);
let req = super::NativeChatRequest {
model: "anthropic/claude-fable-5".to_string(),
messages: Vec::new(),
temperature: Some(0.0),
stream: Some(false),
tools: None,
tool_choice: None,
thread_id: None,
stream_options: None,
options: None,
frequency_penalty: None,
max_tokens: Some(cap),
};
let json = serde_json::to_value(&req).unwrap();
assert_eq!(
json.get("max_tokens").and_then(serde_json::Value::as_u64),
Some(u64::from(cap)),
"the agent-turn cap must be forwarded as OpenAI max_tokens"
);
}
#[test]
fn responses_request_serializes_max_output_tokens_only_when_set() {
// The Responses-API branch must carry the cap as `max_output_tokens` so a
+1
View File
@@ -29,6 +29,7 @@ pub mod traits;
pub use traits::{
ChatMessage, ChatRequest, ChatResponse, ConversationMessage, PromptCacheCapabilities, Provider,
ProviderCapabilityError, ProviderDelta, ToolCall, ToolResultMessage, UsageInfo,
AGENT_TURN_MAX_OUTPUT_TOKENS,
};
pub use billing_error::is_budget_exhausted_message;
+29 -8
View File
@@ -142,6 +142,25 @@ pub enum ProviderDelta {
ToolCallArgsDelta { call_id: String, delta: String },
}
/// Upper bound on output tokens requested for an agent chat turn.
///
/// The agent loop used to leave `ChatRequest::max_tokens` `None` ("open-ended
/// generation"), but an unset cap makes reservation-pricing providers (e.g.
/// OpenRouter) reserve credit against the model's *entire* output window
/// (64k+) during their pre-flight balance check — so a modest-balance BYO user
/// can hit a `402` purely from the oversized reservation, a **preventable**
/// condition. Capping every agent turn at a realistic ceiling prices the
/// pre-flight against a budget the user can actually afford; a residual `402`
/// is then the genuine flat-balance case the insufficient-credits demote arm
/// is meant for (TAURI-RUST-C62; mirrors [`EXTRACTION_MAX_OUTPUT_TOKENS`] in
/// `memory_tree::score::extract::llm`).
///
/// `16384` sits comfortably above any realistic single agent turn — `max_tokens`
/// is an upper bound, not a forced length, so the model still stops at its
/// natural end well below the cap on normal turns — while cutting the
/// reservation 4× versus a 64k window.
pub const AGENT_TURN_MAX_OUTPUT_TOKENS: u32 = 16384;
/// Request payload for provider chat calls.
///
/// The system prompt is built once at session start and frozen for the
@@ -161,16 +180,16 @@ pub struct ChatRequest<'a> {
/// Optional upper bound on output tokens to request from the provider
/// (`max_tokens` on the OpenAI-compatible wire).
///
/// Left `None` for open-ended generation (orchestrator, agent turns)
/// where the model should use its full budget. Set to a small concrete
/// value by callers whose output is bounded by construction — notably
/// memory extraction, whose response is a tiny structured-JSON object.
/// Left `None` only for the orchestrator's open-ended generation. Agent
/// turns cap at [`AGENT_TURN_MAX_OUTPUT_TOKENS`] and callers whose output
/// is bounded by construction set a small concrete value — notably memory
/// extraction, whose response is a tiny structured-JSON object.
/// Beyond capping wasted generation, this stops credit-metered providers
/// (e.g. OpenRouter) from reserving the model's *entire* output window
/// during their pre-flight balance check: an unset `max_tokens` makes
/// OpenRouter price the request against the full 64k+ window and 402 a
/// low-balance BYO user who could easily afford the few thousand tokens
/// an extraction actually needs (TAURI-RUST-C62).
/// the turn actually needs (TAURI-RUST-C62).
pub max_tokens: Option<u32>,
}
@@ -479,9 +498,11 @@ pub trait Provider: Send + Sync {
/// OpenAI-compatible provider, which threads it onto the wire for
/// credit-metered backends — TAURI-RUST-C62) override `chat()` directly.
/// The drop is logged below rather than silently swallowed; it is not a
/// hard error because no production caller both sets `max_tokens` and
/// routes through a default-`chat()` provider (agent turns pass `None`;
/// memory extraction uses the compatible provider).
/// hard error because the production callers that set `max_tokens` (agent
/// turns at [`AGENT_TURN_MAX_OUTPUT_TOKENS`], memory extraction) route to
/// the compatible provider, which overrides `chat()` and honors the cap.
/// A provider on this default path simply forgoes the cap — harmless for
/// the non-reservation backends that don't override `chat()`.
async fn chat(
&self,
request: ChatRequest<'_>,