diff --git a/src/openjarvis/agents/hybrid/advisors.py b/src/openjarvis/agents/hybrid/advisors.py index 9ec75fd9..195e8d48 100644 --- a/src/openjarvis/agents/hybrid/advisors.py +++ b/src/openjarvis/agents/hybrid/advisors.py @@ -201,6 +201,8 @@ class AdvisorsAgent(LocalCloudAgent): "cost_usd": cost, "turns": 3, "web_search_uses": n_searches_total, + # GAIA: only the executor passes invoke a tool (web_search). + "tool_calls": int(n_searches_total), "traces": { "initial_response": initial_resp, "advisor_feedback": advisor_text, @@ -305,6 +307,9 @@ class AdvisorsAgent(LocalCloudAgent): "tokens_cloud": tokens_cloud, "cost_usd": cost, "turns": initial_out["turns"] + 1 + final_out["turns"], + # SWE: sum bash turns from both executor passes; advisor pass + # is a local-model critique with no tools. + "tool_calls": int(initial_out["turns"] + final_out["turns"]), "traces": { "swe_mode": True, "initial_summary": initial_out["final_summary"], diff --git a/src/openjarvis/agents/hybrid/archon.py b/src/openjarvis/agents/hybrid/archon.py index 8ab35f75..afad3663 100644 --- a/src/openjarvis/agents/hybrid/archon.py +++ b/src/openjarvis/agents/hybrid/archon.py @@ -501,6 +501,8 @@ class ArchonAgent(LocalCloudAgent): "cost_usd": cost, "turns": (K + 2) if arch == "ensemble_rank_fuse" else 1, "web_search_uses": n_searches, + # GAIA: only ranker/fuser can hit web_search; proposers don't. + "tool_calls": int(n_searches), "traces": { "architecture": arch, "n_samples": K, @@ -613,6 +615,9 @@ class ArchonAgent(LocalCloudAgent): "tokens_cloud": total_tokens_cloud, "cost_usd": total_cost, "turns": sum(c["turns"] for c in candidates) + 1, + # SWE: total bash turns across the K candidate runs; ranker + # is a single text call with no tools. + "tool_calls": int(sum(c["turns"] for c in candidates)), "traces": { "swe_mode": True, "K": K, diff --git a/src/openjarvis/agents/hybrid/baseline_cloud.py b/src/openjarvis/agents/hybrid/baseline_cloud.py index 45ff0bf2..87695e62 100644 --- a/src/openjarvis/agents/hybrid/baseline_cloud.py +++ b/src/openjarvis/agents/hybrid/baseline_cloud.py @@ -90,6 +90,8 @@ class BaselineCloudAgent(LocalCloudAgent): "tokens_cloud": out["tokens_in"] + out["tokens_out"], "cost_usd": out["cost_usd"], "turns": out["turns"], + # SWE-bench: one bash invocation per agent turn. + "tool_calls": int(out["turns"]), "traces": { "backbone": "cloud", "max_turns_hit": out["max_turns_hit"], @@ -125,6 +127,8 @@ class BaselineCloudAgent(LocalCloudAgent): "cost_usd": cost, "turns": turns, "web_search_uses": n_searches, + # GAIA: the only tool is web_search. + "tool_calls": int(n_searches), "traces": { "mode": "anthropic_agent_loop", "is_swe": is_swe, @@ -160,6 +164,8 @@ class BaselineCloudAgent(LocalCloudAgent): "cost_usd": estimate_cost(self._cloud_model, p_tok, c_tok), "turns": 1, "web_search_uses": 0, + # GAIA one-shot: zero tool calls (no bash, no web_search). + "tool_calls": 0, "traces": { "mode": "one_shot", "is_swe": is_swe, diff --git a/src/openjarvis/agents/hybrid/conductor.py b/src/openjarvis/agents/hybrid/conductor.py index 4c1f7bc3..a7ef25de 100644 --- a/src/openjarvis/agents/hybrid/conductor.py +++ b/src/openjarvis/agents/hybrid/conductor.py @@ -563,11 +563,12 @@ def _swe_worker_step( cfg: Dict[str, Any], workdir: Path, step_idx: int, -) -> Tuple[str, int, int, bool, int]: +) -> Tuple[str, int, int, bool, int, int]: """Run one Conductor worker step as a mini-SWE-agent subloop on a shared workdir. Returns (final_summary_or_diff, tokens_in, tokens_out, is_local, - n_web_searches) in the same shape as ``_call_worker``. SWE workers - don't use web_search (the bash tool is the only tool they need).""" + n_web_searches, bash_turns). SWE workers don't use web_search (the bash + tool is the only tool they need); ``bash_turns`` counts the agent-loop + turns so the caller can surface ``tool_calls``.""" ep = (worker.get("endpoint") or "openai").lower() if ep == "vllm": backbone, model, endpoint, is_local = ( @@ -584,7 +585,8 @@ def _swe_worker_step( # backbones today (the loop's tool-call format is Anthropic- or # OpenAI-via-vllm-shaped only). Fall back to one-shot for those — # SWE-bench-wise they were already weak; this preserves behavior. - return _call_worker(worker, prompt, cfg) + text, p, c, is_local, n_searches = _call_worker(worker, prompt, cfg) + return text, p, c, is_local, n_searches, 0 out = run_swe_agent_loop( task, backbone=backbone, @@ -601,7 +603,7 @@ def _swe_worker_step( ) return ( out["final_summary"] or out["answer"], - out["tokens_in"], out["tokens_out"], is_local, 0, + out["tokens_in"], out["tokens_out"], is_local, 0, int(out["turns"]), ) @@ -714,6 +716,9 @@ class ConductorAgent(LocalCloudAgent): final_answer = "" shared_workdir: Optional[Path] = None n_web_searches_total = 0 + # tool_calls aggregator: bash turns on SWE (per worker subloop) + + # web_search uses on GAIA. Conductor planner is text-only. + tool_calls = 0 # Web_search opt-in: when enabled, declare the native server-side # tool on Anthropic worker calls. GAIA-only — SWE workers use bash @@ -754,9 +759,12 @@ class ConductorAgent(LocalCloudAgent): }) if swe_mode: - text, w_in, w_out, is_local, n_searches = _swe_worker_step( - worker, task_meta, prompt, cfg, shared_workdir, i, + text, w_in, w_out, is_local, n_searches, bash_turns = ( + _swe_worker_step( + worker, task_meta, prompt, cfg, shared_workdir, i, + ) ) + tool_calls += bash_turns else: text, w_in, w_out, is_local, n_searches = _call_worker( worker, prompt, cfg, web_search_tool=ws_tool, @@ -769,6 +777,7 @@ class ConductorAgent(LocalCloudAgent): cost += self.cost_usd(worker["model"], w_in, w_out) cost += n_searches * WEB_SEARCH_COST_PER_CALL n_web_searches_total += n_searches + tool_calls += n_searches steps.append({ "step_idx": i, "model_id": mid, @@ -811,6 +820,7 @@ class ConductorAgent(LocalCloudAgent): "cost_usd": cost, "turns": len(steps) + 1, # planner + N execution steps "web_search_uses": n_web_searches_total, + "tool_calls": int(tool_calls), "traces": { "steps": traces, "plan": plan, diff --git a/src/openjarvis/agents/hybrid/minions.py b/src/openjarvis/agents/hybrid/minions.py index bea18a23..0199f8b9 100644 --- a/src/openjarvis/agents/hybrid/minions.py +++ b/src/openjarvis/agents/hybrid/minions.py @@ -494,6 +494,9 @@ class MinionsAgent(LocalCloudAgent): "cost_usd": self.cost_usd(self._cloud_model, rp, rc) + prefetch["cost_usd"], "turns": cfg.get("max_rounds", 3), "web_search_uses": prefetch["n_searches"], + # GAIA: only countable tool surface is the prefetch web_search. + # The Minions protocol itself is supervisor↔worker text, no tools. + "tool_calls": int(prefetch["n_searches"]), "traces": { "mode": mode, "supervisor_messages": out.get("supervisor_messages"), @@ -572,6 +575,8 @@ class MinionsAgent(LocalCloudAgent): "tokens_cloud": p_in + p_out, "cost_usd": supervisor_cost, "turns": 1 + out["turns"], + # SWE: only the worker invokes tools (bash); supervisor is text-only. + "tool_calls": int(out["turns"]), "traces": { "swe_mode": True, "supervisor_plan": plan_text, diff --git a/src/openjarvis/agents/hybrid/runner.py b/src/openjarvis/agents/hybrid/runner.py index ce04248b..eeb02d0d 100644 --- a/src/openjarvis/agents/hybrid/runner.py +++ b/src/openjarvis/agents/hybrid/runner.py @@ -380,6 +380,7 @@ def _run_one(agent, bench: str, task: Dict[str, Any], log_dir: str) -> Dict[str, "cost_usd": float(meta.get("cost_usd", 0.0)), "latency_s": float(meta.get("latency_s", time.time() - t0)), "web_search_uses": int(meta.get("web_search_uses", 0)), + "tool_calls": int(meta.get("tool_calls", 0)), "traces": meta.get("traces", {}), } if "soft_error" in meta: @@ -392,6 +393,7 @@ def _run_one(agent, bench: str, task: Dict[str, Any], log_dir: str) -> Dict[str, "tokens_local": 0, "tokens_cloud": 0, "cost_usd": 0.0, "latency_s": time.time() - t0, "web_search_uses": 0, + "tool_calls": 0, "traces": {}, "error": f"{type(e).__name__}: {e}\n{traceback.format_exc()}", } @@ -434,6 +436,7 @@ def _write_summary( total_local = sum(r.get("tokens_local", 0) for r in rows) total_cloud = sum(r.get("tokens_cloud", 0) for r in rows) total_web_searches = sum(int(r.get("web_search_uses", 0) or 0) for r in rows) + total_tool_calls = sum(int(r.get("tool_calls", 0) or 0) for r in rows) elapsed = time.time() - t_start # Preserve prior wall_time_s on no-op resume so we don't clobber the @@ -467,6 +470,7 @@ def _write_summary( "tokens_local_total": total_local, "tokens_cloud_total": total_cloud, "web_search_uses_total": total_web_searches, + "tool_calls_total": total_tool_calls, "cost_usd_total": total_cost, "wall_time_s": wall, "task_count": len(tasks), diff --git a/src/openjarvis/agents/hybrid/skillorchestra.py b/src/openjarvis/agents/hybrid/skillorchestra.py index 0203e697..510845f2 100644 --- a/src/openjarvis/agents/hybrid/skillorchestra.py +++ b/src/openjarvis/agents/hybrid/skillorchestra.py @@ -460,6 +460,9 @@ class SkillOrchestraAgent(LocalCloudAgent): tokens_cloud = r_in + r_out run_cost = self.cost_usd(router_model, r_in, r_out) self._web_search_uses_last = 0 + # tool_calls: bash turns on SWE, web_search count on GAIA. Router + # itself is text-only — doesn't add to the count. + tool_calls = 0 # 2. Execute via chosen agent task_meta = (context.metadata.get("task") if context is not None else {}) or {} @@ -490,6 +493,7 @@ class SkillOrchestraAgent(LocalCloudAgent): ) ans = out["answer"] tokens_local += out["tokens_in"] + out["tokens_out"] + tool_calls += int(out["turns"]) else: ans, w_in, w_out = self._call_vllm( self._local_model, @@ -520,6 +524,7 @@ class SkillOrchestraAgent(LocalCloudAgent): ans = out["answer"] tokens_cloud += out["tokens_in"] + out["tokens_out"] run_cost += out["cost_usd"] + tool_calls += int(out["turns"]) else: # GAIA cloud worker: opt-in native web_search via the new # method_cfg.web_search schema. Only fires when the chosen @@ -546,6 +551,7 @@ class SkillOrchestraAgent(LocalCloudAgent): run_cost += self.cost_usd(self._cloud_model, w_in, w_out) run_cost += n_searches * WEB_SEARCH_COST_PER_CALL self._web_search_uses_last = n_searches + tool_calls += int(n_searches) worker_model = self._cloud_model meta = { @@ -554,6 +560,7 @@ class SkillOrchestraAgent(LocalCloudAgent): "cost_usd": run_cost, "turns": 2, # router + worker "web_search_uses": self._web_search_uses_last, + "tool_calls": int(tool_calls), "traces": { "chosen_agent": chosen, "worker_model": worker_model, diff --git a/src/openjarvis/agents/hybrid/toolorchestra.py b/src/openjarvis/agents/hybrid/toolorchestra.py index a9864f79..4eec4eb0 100644 --- a/src/openjarvis/agents/hybrid/toolorchestra.py +++ b/src/openjarvis/agents/hybrid/toolorchestra.py @@ -407,14 +407,19 @@ def _swe_call_worker( task: Dict[str, Any], workdir: Path, turn: int, -) -> Tuple[str, int, int, bool, float, int]: +) -> Tuple[str, int, int, bool, float, int, int]: """SWE-bench worker dispatch: route solver workers through run_swe_agent_loop on a shared workdir. Web-search workers fall back - to the regular one-shot dispatch (search isn't an agent loop).""" + to the regular one-shot dispatch (search isn't an agent loop). + + Trailing ``bash_turns`` (last element) counts agent-loop turns so the + caller can surface ``tool_calls`` per row. Fallbacks to one-shot + workers return 0 bash turns (no agent loop ran).""" wtype = worker.get("type", "openai") if wtype == "anthropic-web-search": # Search workers stay one-shot. - return _call_worker(worker, prompt, cfg) + text, p, c, is_local, extra, n_searches = _call_worker(worker, prompt, cfg) + return text, p, c, is_local, extra, n_searches, 0 if wtype == "vllm": backbone = "local" endpoint = worker.get("base_url") @@ -423,7 +428,8 @@ def _swe_call_worker( endpoint = None else: # OpenAI workers fall back to one-shot. - return _call_worker(worker, prompt, cfg) + text, p, c, is_local, extra, n_searches = _call_worker(worker, prompt, cfg) + return text, p, c, is_local, extra, n_searches, 0 out = run_swe_agent_loop( task, backbone=backbone, @@ -442,7 +448,7 @@ def _swe_call_worker( return ( out["final_summary"] or out["answer"], out["tokens_in"], out["tokens_out"], - is_local, 0.0, 0, + is_local, 0.0, 0, int(out["turns"]), ) @@ -531,6 +537,11 @@ class ToolOrchestraAgent(LocalCloudAgent): tokens_cloud = 0 cost = 0.0 n_web_searches_total = 0 + # tool_calls: bash turns from SWE subloops + web_search uses + # from GAIA. Orchestrator dispatch turns are NOT counted (they + # produce text only — calling a worker is one tool call's worth + # of "delegation" but the actual tool action happens inside). + tool_calls = 0 final_answer: Optional[str] = None forced_final = False parse_failures = 0 @@ -584,12 +595,14 @@ class ToolOrchestraAgent(LocalCloudAgent): continue worker = workers[wid] if swe_mode and shared_workdir is not None: - w_text, w_in, w_out, is_local, extra_cost, n_searches = ( + (w_text, w_in, w_out, is_local, extra_cost, + n_searches, bash_turns) = ( _swe_call_worker( worker, str(w_input), cfg, task_meta, shared_workdir, turn, ) ) + tool_calls += bash_turns else: w_text, w_in, w_out, is_local, extra_cost, n_searches = ( _call_worker(worker, str(w_input), cfg) @@ -600,6 +613,7 @@ class ToolOrchestraAgent(LocalCloudAgent): tokens_cloud += w_in + w_out cost += self.cost_usd(worker["model"], w_in, w_out) + extra_cost n_web_searches_total += n_searches + tool_calls += n_searches history.append({ "role": "worker", "turn": turn, @@ -629,10 +643,12 @@ class ToolOrchestraAgent(LocalCloudAgent): key=lambda w: PRICES.get(w.get("model", ""), (0.0, 0.0))[1], ) if swe_mode and shared_workdir is not None: - ans, w_in, w_out, is_local, extra_cost, _ = _swe_call_worker( + (ans, w_in, w_out, is_local, extra_cost, _, + bash_turns) = _swe_call_worker( worker, question, cfg, task_meta, shared_workdir, max_turns + 1, ) + tool_calls += bash_turns else: ans, w_in, w_out, is_local, extra_cost, _ = _call_worker( worker, question, cfg @@ -671,6 +687,7 @@ class ToolOrchestraAgent(LocalCloudAgent): "cost_usd": cost, "turns": len([h for h in history if h["role"] == "orchestrator"]), "web_search_uses": n_web_searches_total, + "tool_calls": int(tool_calls), "traces": { "history": history, "forced_final": forced_final, diff --git a/tests/agents/hybrid/test_tool_calls_tracking.py b/tests/agents/hybrid/test_tool_calls_tracking.py new file mode 100644 index 00000000..a2cd26e3 --- /dev/null +++ b/tests/agents/hybrid/test_tool_calls_tracking.py @@ -0,0 +1,379 @@ +"""Per-row ``tool_calls`` tracking across hybrid paradigms (2026-05-18). + +Confirms each paradigm surfaces a top-level ``tool_calls: int`` in the +``AgentResult.metadata`` so the runner can write it into ``results.jsonl``. +Definition per paradigm: + + - SWE-bench cells: bash turns from ``run_swe_agent_loop``. One tool call + per bash command the agent ran. + - GAIA cells: number of native ``web_search`` invocations the cloud + backbone made. Zero for one-shot GAIA paths and zero for paradigms + whose GAIA path is just text-passing (e.g. Minions w/o prefetch). + +The legacy ``skillorchestra`` tool_calls numbers in +``docs/results-table.md`` came from a defunct telemetry path; this +re-establishes the contract on the actual row-write path. +""" + +from __future__ import annotations + +from types import SimpleNamespace +from unittest.mock import patch + +import pytest + +from openjarvis.agents._stubs import AgentContext + + +# --------------------------------------------------------------------------- +# Anthropic stub (web_search agent loop emits N web_search_requests). +# --------------------------------------------------------------------------- + +def _fake_anthropic_response(text: str = "FINAL ANSWER: 7", n_searches: int = 3): + return SimpleNamespace( + content=[SimpleNamespace(type="text", text=text)], + usage=SimpleNamespace( + input_tokens=100, + output_tokens=20, + server_tool_use=SimpleNamespace(web_search_requests=n_searches), + ), + stop_reason="end_turn", + ) + + +class _FakeMessages: + def __init__(self, n_searches: int = 3): + self.n_searches = n_searches + self.calls = 0 + + def create(self, **kwargs): + self.calls += 1 + # Each anthropic call returns the same shape; total searches + # accumulate via repeated calls (advisors does 2 executor passes, + # each reporting its own n_searches). + return _fake_anthropic_response(n_searches=self.n_searches) + + +class _FakeAnthropic: + last_messages = None + + def __init__(self, *args, **kwargs): + self.messages = _FakeMessages(n_searches=3) + type(self).last_messages = self.messages + + +@pytest.fixture +def fake_anthropic(monkeypatch): + import anthropic + monkeypatch.setattr(anthropic, "Anthropic", _FakeAnthropic) + yield _FakeAnthropic + + +# --------------------------------------------------------------------------- +# 1. baseline_cloud GAIA — tool_calls == n_searches on the agent-loop path. +# --------------------------------------------------------------------------- + +def test_baseline_cloud_gaia_tool_calls_eq_web_searches(fake_anthropic): + from openjarvis.agents.hybrid.baseline_cloud import BaselineCloudAgent + + agent = BaselineCloudAgent( + engine=None, + model="claude-opus-4-7", + cloud_endpoint="anthropic", + cfg={ + "cloud_max_tokens": 1024, + "web_search": {"enabled": True, "max_uses": 4}, + "gaia_max_turns": 2, + }, + ) + ctx = AgentContext(metadata={"task": {"task_id": "t1", "question": "X?"}, "task_id": "t1"}) + result = agent.run("X?", ctx) + tc = result.metadata.get("tool_calls") + assert isinstance(tc, int) + assert tc == result.metadata["web_search_uses"] == 3 + + +def test_baseline_cloud_gaia_oneshot_tool_calls_zero(monkeypatch): + """The one-shot GAIA path makes zero countable tool calls.""" + from openjarvis.agents.hybrid import baseline_cloud as bc_mod + from openjarvis.agents.hybrid.baseline_cloud import BaselineCloudAgent + + def fake_call_cloud(self, *, user, system=None, max_tokens=4096, + temperature=0.0, **kwargs): + return "FINAL ANSWER: x", 10, 5 + + monkeypatch.setattr(bc_mod.BaselineCloudAgent, "_call_cloud", fake_call_cloud) + agent = BaselineCloudAgent( + engine=None, + model="gpt-5", + cloud_endpoint="openai", + cfg={"cloud_max_tokens": 1024}, + ) + ctx = AgentContext(metadata={"task": {"task_id": "t2", "question": "X?"}, "task_id": "t2"}) + result = agent.run("X?", ctx) + assert isinstance(result.metadata.get("tool_calls"), int) + assert result.metadata["tool_calls"] == 0 + + +# --------------------------------------------------------------------------- +# 2. advisors GAIA — tool_calls == n_searches summed across executor passes. +# --------------------------------------------------------------------------- + +def test_advisors_gaia_tool_calls_sums_search_counts(fake_anthropic, monkeypatch): + from openjarvis.agents.hybrid.advisors import AdvisorsAgent + + # Stub the local vLLM advisor call so we don't need a server. + def fake_vllm(model, endpoint, *, user, max_tokens, temperature, + enable_thinking=False): + return "be more careful", 50, 10 + + monkeypatch.setattr( + "openjarvis.agents.hybrid.advisors.AdvisorsAgent._call_vllm", + staticmethod(fake_vllm), + ) + + agent = AdvisorsAgent( + engine=None, + model="claude-opus-4-7", + cloud_endpoint="anthropic", + local_model="qwen", + local_endpoint="http://x", + cfg={ + "executor_max_tokens": 1024, + "advisor_max_tokens": 512, + "web_search": {"enabled": True, "max_uses": 4}, + "gaia_max_turns": 2, + }, + ) + ctx = AgentContext(metadata={"task": {"task_id": "t3", "question": "X?"}, "task_id": "t3"}) + result = agent.run("X?", ctx) + tc = result.metadata.get("tool_calls") + assert isinstance(tc, int) + # Two executor passes, each reports 3 searches → 6 total. + assert tc == result.metadata["web_search_uses"] == 6 + + +# --------------------------------------------------------------------------- +# 3. minions GAIA — tool_calls == prefetch n_searches. +# --------------------------------------------------------------------------- + +def test_minions_gaia_tool_calls_eq_prefetch_searches(monkeypatch): + from openjarvis.agents.hybrid import minions as minions_mod + from openjarvis.agents.hybrid.minions import MinionsAgent + + # Skip the Minions import / patch dance entirely by stubbing _run_paradigm's + # protocol layer. Easiest: stub _prefetch_context + replace Minions class. + def fake_prefetch(question, endpoint, model, max_uses): + return { + "text": "stub digest", + "tokens": 100, + "cost_usd": 0.01, + "n_searches": 2, + } + + class _FakeProtocol: + def __init__(self, **kw): + pass + + def __call__(self, *args, **kw): + return { + "final_answer": "FINAL ANSWER: 42", + "supervisor_messages": [], + "worker_messages": [], + "timing": {}, + "log_file": "/dev/null", + "local_usage": SimpleNamespace(prompt_tokens=10, completion_tokens=5), + "remote_usage": SimpleNamespace(prompt_tokens=20, completion_tokens=10), + } + + fake_minions_module = SimpleNamespace( + clients=SimpleNamespace( + anthropic=SimpleNamespace(AnthropicClient=lambda **kw: None), + openai=SimpleNamespace(OpenAIClient=lambda **kw: None), + ), + minion=SimpleNamespace(Minion=_FakeProtocol), + minions=SimpleNamespace(Minions=_FakeProtocol), + ) + + import sys + monkeypatch.setitem(sys.modules, "minions", fake_minions_module) + monkeypatch.setitem(sys.modules, "minions.clients", fake_minions_module.clients) + monkeypatch.setitem(sys.modules, "minions.clients.anthropic", fake_minions_module.clients.anthropic) + monkeypatch.setitem(sys.modules, "minions.clients.openai", fake_minions_module.clients.openai) + monkeypatch.setitem(sys.modules, "minions.minion", fake_minions_module.minion) + monkeypatch.setitem(sys.modules, "minions.minions", fake_minions_module.minions) + monkeypatch.setattr(minions_mod, "_apply_patches_once", lambda: None) + monkeypatch.setattr(minions_mod, "_prefetch_context", fake_prefetch) + + agent = MinionsAgent( + engine=None, + model="claude-opus-4-7", + cloud_endpoint="anthropic", + local_model="qwen", + local_endpoint="http://x", + cfg={ + "mode": "minion", + "max_rounds": 2, + "web_search": {"enabled": True, "max_uses": 4}, + }, + ) + ctx = AgentContext(metadata={ + "task": {"task_id": "tm", "question": "What is X?"}, + "task_id": "tm", + }) + result = agent.run("What is X?", ctx) + tc = result.metadata.get("tool_calls") + assert isinstance(tc, int) + assert tc == 2 # matches the stubbed prefetch n_searches + + +# --------------------------------------------------------------------------- +# 4. SWE path — tool_calls equals bash turns from run_swe_agent_loop. +# We stub run_swe_agent_loop directly so we don't need a repo / vLLM / +# Anthropic. +# --------------------------------------------------------------------------- + +def _stub_swe_loop(turns: int = 7): + """Return a stub matching ``run_swe_agent_loop``'s output shape.""" + def _stub(task, **kwargs): + return { + "answer": "stub\n\n```diff\n--- a\n+++ b\n```", + "patch": "--- a\n+++ b\n", + "final_summary": "stub", + "tokens_in": 200, + "tokens_out": 100, + "tokens_local": 0, + "tokens_cloud": 300, + "cost_usd": 0.05, + "turns": turns, + "max_turns_hit": False, + "workdir": "/tmp/stub", + } + return _stub + + +def test_baseline_cloud_swe_tool_calls_eq_bash_turns(monkeypatch): + from openjarvis.agents.hybrid import baseline_cloud as bc_mod + from openjarvis.agents.hybrid.baseline_cloud import BaselineCloudAgent + + monkeypatch.setattr(bc_mod, "run_swe_agent_loop", _stub_swe_loop(turns=11)) + + agent = BaselineCloudAgent( + engine=None, + model="claude-opus-4-7", + cloud_endpoint="anthropic", + cfg={"swe_max_turns": 30, "cloud_max_tokens": 4096}, + ) + task = { + "task_id": "swe1", + "problem_statement": "fix it", + "repo": "a/b", + "base_commit": "deadbeef", + } + ctx = AgentContext(metadata={"task": task, "task_id": "swe1"}) + result = agent.run("fix it", ctx) + tc = result.metadata.get("tool_calls") + assert isinstance(tc, int) + assert tc == 11 + + +def test_minions_swe_tool_calls_eq_worker_bash_turns(monkeypatch): + from openjarvis.agents.hybrid import minions as minions_mod + from openjarvis.agents.hybrid.minions import MinionsAgent + + monkeypatch.setattr(minions_mod, "run_swe_agent_loop", _stub_swe_loop(turns=9)) + + def fake_call_cloud(self, *, user, system=None, max_tokens=4096, + temperature=0.0, **kwargs): + return "plan: do X", 50, 25 + + monkeypatch.setattr(minions_mod.MinionsAgent, "_call_cloud", fake_call_cloud) + + agent = MinionsAgent( + engine=None, + model="claude-opus-4-7", + cloud_endpoint="anthropic", + local_model="qwen", + local_endpoint="http://x", + cfg={"swe_use_agent_loop": True, "supervisor_max_tokens": 512}, + ) + task = { + "task_id": "swe2", + "problem_statement": "fix it", + "repo": "a/b", + "base_commit": "deadbeef", + } + ctx = AgentContext(metadata={"task": task, "task_id": "swe2"}) + result = agent.run("fix it", ctx) + tc = result.metadata.get("tool_calls") + assert isinstance(tc, int) + assert tc == 9 # only the worker subloop counts; supervisor is text + + +def test_advisors_swe_tool_calls_sums_both_executor_passes(monkeypatch): + from openjarvis.agents.hybrid import advisors as adv_mod + from openjarvis.agents.hybrid.advisors import AdvisorsAgent + + # Both calls to run_swe_agent_loop return turns=4; sum should be 8. + monkeypatch.setattr(adv_mod, "run_swe_agent_loop", _stub_swe_loop(turns=4)) + + def fake_vllm(model, endpoint, *, user, max_tokens, temperature, + enable_thinking=False): + return "advise", 30, 10 + + monkeypatch.setattr( + adv_mod.AdvisorsAgent, "_call_vllm", staticmethod(fake_vllm), + ) + + agent = AdvisorsAgent( + engine=None, + model="claude-opus-4-7", + cloud_endpoint="anthropic", + local_model="qwen", + local_endpoint="http://x", + cfg={"swe_use_agent_loop": True}, + ) + task = { + "task_id": "swe3", + "problem_statement": "fix it", + "repo": "a/b", + "base_commit": "deadbeef", + } + ctx = AgentContext(metadata={"task": task, "task_id": "swe3"}) + result = agent.run("fix it", ctx) + tc = result.metadata.get("tool_calls") + assert isinstance(tc, int) + assert tc == 8 # initial + final executor passes, each 4 bash turns + + +# --------------------------------------------------------------------------- +# 5. Runner row contract — _run_one writes tool_calls into the row. +# --------------------------------------------------------------------------- + +def test_runner_row_includes_tool_calls(monkeypatch): + """The runner's ``_run_one`` must surface ``tool_calls`` from meta + into the top-level row that ends up in ``results.jsonl``.""" + from openjarvis.agents._stubs import AgentResult + from openjarvis.agents.hybrid import runner + + class _StubAgent: + def run(self, prompt, ctx): + return AgentResult( + content="FINAL ANSWER: x", + metadata={ + "tokens_local": 0, + "tokens_cloud": 100, + "cost_usd": 0.01, + "latency_s": 0.1, + "web_search_uses": 2, + "tool_calls": 5, + "traces": {}, + }, + turns=1, + ) + + task = {"task_id": "row1", "question": "x", "reference": "x", "metadata": {}} + row = runner._run_one(_StubAgent(), "gaia", task, "/tmp/log") + assert "tool_calls" in row + assert isinstance(row["tool_calls"], int) + assert row["tool_calls"] == 5