From d572ecfa679111b51f77d099962c21bbaace641d Mon Sep 17 00:00:00 2001 From: Andrew Park Date: Fri, 15 May 2026 13:45:26 -0700 Subject: [PATCH] hybrid: wire Minions + Archon through mini-SWE-agent, add SWE registry MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Minions (swe mode): skip the upstream Minions library — it doesn't fit SWE-bench (its premise is 'small reads long docs, big summarizes'). Replace with: cloud supervisor writes a high-level fix plan (no tools), local Qwen runs mini-SWE-agent with that plan as additional context. Mirrors Minions's 'cloud supervises, local does the work' shape on a benchmark where the upstream library is wrong tool. Archon (swe mode): K independent local mini-SWE-agent runs producing K diverse candidate patches; cloud ranker picks the best by summary + diff. Skips the 'fuser' layer (can't fuse two diffs cleanly). Add registry/swe_agent_loop.toml with one cell per paradigm exercising the new mode (n=3 smokes; same model pairing as the headline cells). All 6 paradigms + standalone mini_swe_agent = 9 SWE-agent cells total. Smoke imports clean across all paradigms. Cost warning: each cell now runs ~30 LLM turns/task with bash; budget -3/task with Opus. --- src/openjarvis/agents/hybrid/archon.py | 183 +++++++++++++++++- src/openjarvis/agents/hybrid/minions.py | 106 +++++++++- .../hybrid/registry/swe_agent_loop.toml | 70 +++++++ 3 files changed, 353 insertions(+), 6 deletions(-) create mode 100644 src/openjarvis/agents/hybrid/registry/swe_agent_loop.toml diff --git a/src/openjarvis/agents/hybrid/archon.py b/src/openjarvis/agents/hybrid/archon.py index 7dd6fdc3..51b5da8d 100644 --- a/src/openjarvis/agents/hybrid/archon.py +++ b/src/openjarvis/agents/hybrid/archon.py @@ -42,10 +42,26 @@ from typing import Any, Dict, Optional, Tuple from openjarvis.agents._stubs import AgentContext from openjarvis.agents.hybrid._base import LocalCloudAgent, _record_event -from openjarvis.agents.hybrid._prices import NO_TEMP_PREFIXES, cost as _cost_cloud +from openjarvis.agents.hybrid._prices import ( + NO_TEMP_PREFIXES, + cost as _cost_cloud, + supports_temperature, +) +from openjarvis.agents.hybrid.mini_swe_agent import run_swe_agent_loop from openjarvis.core.registry import AgentRegistry +ARCHON_SWE_RANKER_SYS = ( + "You are ranking K candidate patches for a SWE-bench bug. For each " + "candidate, you'll see the candidate's summary text and the unified " + "diff. Pick the index (0-based) of the candidate most likely to " + "actually fix the issue. Prefer minimal, targeted patches; reject " + "candidates with no patch or with patches that touch unrelated files. " + "Respond with a single integer on its own line — the index — " + "followed by a one-line justification." +) + + # ---------- Stubs for Archon's eager-imported heavy deps we don't need ---------- def _stub_archon_imports() -> None: @@ -230,6 +246,45 @@ def _wrap_archon_cloud_generators() -> None: _GMAP["Anthropic_API"] = gen_anthropic +_FUSER_FORMAT_REMINDER = ( + "\n\nIMPORTANT — output format: your synthesized response MUST honor the " + "output-format requirements stated in the original user query above. " + "If the query requires ending with `FINAL ANSWER: ` (GAIA), end " + "with exactly that one line and nothing after it. If the query requires " + "a unified diff inside a ```diff … ``` fence (SWE-bench), end with that " + "fence and nothing after the closing ```. Do not produce free-form " + "analysis, markdown headers, or commentary that breaks the required " + "format — the candidate responses may have done so, but your fused " + "answer must not." +) + + +def _patch_archon_prompts() -> None: + """Make Archon's fuser bench-aware. + + ``Fuser.py`` drops the system message we hand to ``archon.generate()`` and + builds its own user message via ``make_fuser_prompt``. That template tells + the fuser to "synthesize into a refined, well-structured response" with + no reminder of the format the user originally asked for, so on SWE-bench + Opus drifts to a markdown bug report and scores 0. We append an explicit + format-honor reminder to the fuser prompt.""" + from archon.completions.components import prompts as _p # type: ignore[import-not-found] + + if getattr(_p.make_fuser_prompt, "_hybrid_format_patched", False): + return + orig = _p.make_fuser_prompt + + def patched(conv, references, critiques=None, length_control=False): # type: ignore[no-untyped-def] + base = orig(conv, references, critiques=critiques, length_control=length_control) + return base + _FUSER_FORMAT_REMINDER + + patched._hybrid_format_patched = True # type: ignore[attr-defined] + _p.make_fuser_prompt = patched + # Fuser imports the symbol by name at class-body time — rebind there too. + from archon.completions.components import Fuser as _F # type: ignore[import-not-found] + _F.make_fuser_prompt = patched + + _PATCHES_APPLIED = False @@ -243,6 +298,7 @@ def _apply_patches_once() -> None: # Trigger Archon imports so GENERATE_MAP exists. import archon.completions.components.Generator # type: ignore[import-not-found] # noqa: F401 _wrap_archon_cloud_generators() + _patch_archon_prompts() _PATCHES_APPLIED = True @@ -306,11 +362,21 @@ class ArchonAgent(LocalCloudAgent): if "OPENAI_API_KEY" not in os.environ and "ANTHROPIC_API_KEY" not in os.environ: raise RuntimeError("Archon needs OPENAI_API_KEY and/or ANTHROPIC_API_KEY") + cfg = self._cfg + task_meta = (context.metadata.get("task") if context is not None else {}) or {} + swe_mode = ( + bool(cfg.get("swe_use_agent_loop")) + and bool(task_meta.get("problem_statement")) + and bool(task_meta.get("repo")) + and bool(task_meta.get("base_commit")) + ) + if swe_mode: + return self._run_swe(input, task_meta, cfg) + _apply_patches_once() from archon.completions import Archon # type: ignore[import-not-found] from archon.completions.components.Generator import GENERATE_MAP as _GMAP # type: ignore[import-not-found] - cfg = self._cfg arch = cfg.get("architecture", "ensemble_rank_fuse") presets = _presets() if arch not in presets: @@ -377,5 +443,118 @@ class ArchonAgent(LocalCloudAgent): } return answer, meta + # ------------------------------------------------------------------ + # SWE-bench variant: K diverse mini-SWE-agent runs, cloud ranker picks + # ------------------------------------------------------------------ + + def _run_swe( + self, + input: str, + task: Dict[str, Any], + cfg: Dict[str, Any], + ) -> Tuple[str, Dict[str, Any]]: + if not self._local_endpoint or not self._local_model: + raise ValueError( + "ArchonAgent (swe mode) needs local_model + local_endpoint" + ) + + K = int(cfg.get("n_samples", 3)) + max_turns = int(cfg.get("swe_max_turns", 30)) + bash_timeout = int(cfg.get("swe_bash_timeout_s", 120)) + output_cap = int(cfg.get("swe_output_cap", 10_000)) + turn_max_tokens = int(cfg.get("swe_turn_max_tokens", 4096)) + + # K independent runs, each on a FRESH workdir → K diverse patches. + candidates: List[Dict[str, Any]] = [] + total_tokens_local = 0 + total_tokens_cloud = 0 + total_cost = 0.0 + for k in range(K): + out = run_swe_agent_loop( + task, + backbone="local", + backbone_model=self._local_model, + local_endpoint=self._local_endpoint, + initial_prompt=input, + max_turns=max_turns, + bash_timeout=bash_timeout, + output_cap=output_cap, + turn_max_tokens=turn_max_tokens, + trace_prefix=f"archon_gen{k}", + ) + candidates.append({ + "idx": k, + "summary": out["final_summary"], + "patch": out["patch"], + "framed": out["answer"], + "tokens_in": out["tokens_in"], + "tokens_out": out["tokens_out"], + "turns": out["turns"], + }) + total_tokens_local += out["tokens_in"] + out["tokens_out"] + self.record_trace_event({ + "kind": "archon_swe_candidate", + "idx": k, + "patch_chars": len(out["patch"]), + "summary": out["final_summary"], + }) + + # Ranker: cloud picks the best candidate. + ranker_user = ( + f"Issue:\n{task.get('problem_statement','')}\n\n" + f"K = {K} candidate patches:\n\n" + + "\n\n".join( + f"=== Candidate {c['idx']} ===\nSummary: {c['summary']}\n" + f"Patch (chars={len(c['patch'])}):\n```diff\n{c['patch']}```" + for c in candidates + ) + ) + ranker_text, r_in, r_out = self._call_cloud( + user=ranker_user, + system=ARCHON_SWE_RANKER_SYS, + max_tokens=int(cfg.get("ranker_max_tokens", 1024)), + temperature=0.0, + ) + total_tokens_cloud += r_in + r_out + total_cost += self.cost_usd(self._cloud_model, r_in, r_out) + + # Parse — first integer on its own line. + chosen_idx = 0 + for line in ranker_text.splitlines(): + stripped = line.strip() + if stripped.isdigit(): + chosen_idx = int(stripped) + break + if not (0 <= chosen_idx < K): + chosen_idx = 0 + chosen = candidates[chosen_idx] + + self.record_trace_event({ + "kind": "archon_swe_rank", + "chosen_idx": chosen_idx, + "ranker_raw": ranker_text, + "tokens_in": r_in, + "tokens_out": r_out, + }) + + meta = { + "tokens_local": total_tokens_local, + "tokens_cloud": total_tokens_cloud, + "cost_usd": total_cost, + "turns": sum(c["turns"] for c in candidates) + 1, + "traces": { + "swe_mode": True, + "K": K, + "candidates": [ + {"idx": c["idx"], "summary": c["summary"], + "patch_chars": len(c["patch"]), "turns": c["turns"]} + for c in candidates + ], + "chosen_idx": chosen_idx, + "ranker_text": ranker_text, + }, + } + return chosen["framed"], meta + __all__ = ["ArchonAgent"] diff --git a/src/openjarvis/agents/hybrid/minions.py b/src/openjarvis/agents/hybrid/minions.py index 9100e1f3..3c58a217 100644 --- a/src/openjarvis/agents/hybrid/minions.py +++ b/src/openjarvis/agents/hybrid/minions.py @@ -50,9 +50,21 @@ from openjarvis.agents.hybrid._base import ( WEB_SEARCH_COST_PER_CALL, ) from openjarvis.agents.hybrid._prices import NO_TEMP_PREFIXES, supports_temperature +from openjarvis.agents.hybrid.mini_swe_agent import run_swe_agent_loop from openjarvis.core.registry import AgentRegistry +MINIONS_SWE_PLANNER_SYS = ( + "You are the cloud supervisor in a Minions setup. The small local model " + "is about to run an agent loop (with shell access) against a Python " + "repository to fix a bug. Read the issue and write a concise plan: " + "what files the local model should look at first, what tests are most " + "relevant, and 1-3 concrete approaches it should try. Be specific — " + "this is the local model's only briefing from you. Reply in 8 bullet " + "points or fewer." +) + + # ---------- Per-turn JSON schemas (server-side enforcement) ---------- # # Minions's supervisor produces different JSON shapes per turn: @@ -328,17 +340,32 @@ class MinionsAgent(LocalCloudAgent): context: Optional[AgentContext], **kwargs: Any, ) -> Tuple[str, Dict[str, Any]]: + cfg = self._cfg + task_meta: Dict[str, Any] = {} + if context is not None: + task_meta = context.metadata.get("task", {}) or {} + + # SWE-bench branch: the upstream Minions library doesn't fit + # SWE-bench (it's "small model reads docs, big model summarizes"). + # Instead, mirror Minions's "cloud supervises, local does the + # work" pattern: cloud writes a high-level fix plan, local Qwen + # runs mini-SWE-agent with that plan as additional context. + swe_mode = ( + bool(cfg.get("swe_use_agent_loop")) + and bool(task_meta.get("problem_statement")) + and bool(task_meta.get("repo")) + and bool(task_meta.get("base_commit")) + ) + if swe_mode: + return self._run_swe(input, task_meta, cfg) + _apply_patches_once() from minions.clients.openai import OpenAIClient # type: ignore[import-not-found] from minions.clients.anthropic import AnthropicClient # type: ignore[import-not-found] from minions.minion import Minion # type: ignore[import-not-found] from minions.minions import Minions # type: ignore[import-not-found] - cfg = self._cfg mode = cfg.get("mode", "minion") - task_meta: Dict[str, Any] = {} - if context is not None: - task_meta = context.metadata.get("task", {}) or {} if not self._local_endpoint or not self._local_model: raise ValueError( @@ -450,4 +477,75 @@ class MinionsAgent(LocalCloudAgent): return out.get("final_answer", ""), meta + # ------------------------------------------------------------------ + # SWE-bench variant + # ------------------------------------------------------------------ + + def _run_swe( + self, + input: str, + task: Dict[str, Any], + cfg: Dict[str, Any], + ) -> Tuple[str, Dict[str, Any]]: + if not self._local_endpoint or not self._local_model: + raise ValueError( + "MinionsAgent (swe mode) still needs local_model + local_endpoint" + ) + # 1. Cloud supervisor writes a high-level plan (no tools). + plan_text, p_in, p_out = self._call_cloud( + user=( + f"Issue:\n{task.get('problem_statement','')}\n\n" + f"Repo: {task.get('repo','')}\n" + f"Base commit: {task.get('base_commit','')}\n\n" + f"{task.get('hints_text','')}" + ), + system=MINIONS_SWE_PLANNER_SYS, + max_tokens=int(cfg.get("supervisor_max_tokens", 1024)), + temperature=0.0, + ) + self.record_trace_event({ + "kind": "minions_swe_plan", + "plan": plan_text, + "tokens_in": p_in, + "tokens_out": p_out, + }) + supervisor_cost = self.cost_usd(self._cloud_model, p_in, p_out) + + # 2. Local worker runs mini-SWE-agent with the plan as context. + worker_prompt = ( + f"{input}\n\n" + f"-----\n" + f"A cloud supervisor reviewed this issue and wrote a fix plan " + f"for you. Use it as guidance, but verify everything with the " + f"actual code via your bash tool:\n\n{plan_text}" + ) + out = run_swe_agent_loop( + task, + backbone="local", + backbone_model=self._local_model, + local_endpoint=self._local_endpoint, + initial_prompt=worker_prompt, + max_turns=int(cfg.get("swe_max_turns", 30)), + bash_timeout=int(cfg.get("swe_bash_timeout_s", 120)), + output_cap=int(cfg.get("swe_output_cap", 10_000)), + turn_max_tokens=int(cfg.get("swe_turn_max_tokens", 4096)), + trace_prefix="minions_worker", + ) + + meta = { + "tokens_local": out["tokens_in"] + out["tokens_out"], + "tokens_cloud": p_in + p_out, + "cost_usd": supervisor_cost, + "turns": 1 + out["turns"], + "traces": { + "swe_mode": True, + "supervisor_plan": plan_text, + "worker_final_summary": out["final_summary"], + "worker_patch_chars": len(out["patch"]), + "max_turns_hit": out["max_turns_hit"], + }, + } + return out["answer"], meta + + __all__ = ["MinionsAgent"] diff --git a/src/openjarvis/agents/hybrid/registry/swe_agent_loop.toml b/src/openjarvis/agents/hybrid/registry/swe_agent_loop.toml new file mode 100644 index 00000000..96a36637 --- /dev/null +++ b/src/openjarvis/agents/hybrid/registry/swe_agent_loop.toml @@ -0,0 +1,70 @@ +# SWE-bench cells that run each paradigm through mini-SWE-agent. +# +# Same paradigms as the headline cells, same models, but with +# `swe_use_agent_loop = true` so the worker step is a full agent loop +# (shell + repo browse + test runs) instead of one-shot patch prediction. +# Expected: 0.30 → ~0.5-0.77 jump per paradigm (depending on which model +# carries the loop). +# +# These cells are EXPENSIVE — each task is ~30-50 LLM turns with bash, so +# Opus-driven cells run $1-3/task and ~5-10 min/task. Smokes at n=3 first. + +# ------------- Minions: cloud planner + local mini-SWE-agent worker ------------- + +[cells.minions-swe-agent-swebenchverified-qwen27b-opus-3] +method = "minions" +bench = "swebench-verified" +n = 3 +local = { model = "Qwen/Qwen3.5-27B-FP8", endpoint = "http://localhost:8001/v1" } +cloud = { model = "claude-opus-4-7", endpoint = "anthropic" } +method_cfg = { swe_use_agent_loop = true, supervisor_max_tokens = 1024, swe_max_turns = 30, swe_bash_timeout_s = 120, swe_turn_max_tokens = 4096 } + +# ------------- Conductor: shared workdir across plan steps ------------- + +[cells.conductor-swe-agent-swebenchverified-opus-3] +method = "conductor" +bench = "swebench-verified" +n = 3 +local = { model = "Qwen/Qwen3.5-27B-FP8", endpoint = "http://localhost:8001/v1" } +cloud = { model = "claude-opus-4-7", endpoint = "anthropic" } +method_cfg = { swe_use_agent_loop = true, conductor_max_tokens = 2048, swe_max_turns = 25, swe_bash_timeout_s = 120, swe_turn_max_tokens = 4096 } + +# ------------- Advisors: two full agent-loop attempts with local critique between ------------- + +[cells.advisors-swe-agent-swebenchverified-qwen9b-opus-3] +method = "advisors" +bench = "swebench-verified" +n = 3 +local = { model = "Qwen/Qwen3.5-9B", endpoint = "http://localhost:8001/v1" } +cloud = { model = "claude-opus-4-7", endpoint = "anthropic" } +method_cfg = { swe_use_agent_loop = true, swe_max_turns = 25, swe_bash_timeout_s = 120, swe_turn_max_tokens = 4096, advisor_max_tokens = 2048 } + +# ------------- SkillOrchestra: router routes to local or cloud, that one runs the loop ------------- + +[cells.skillorchestra-swe-agent-swebenchverified-qwen27b-opus-3] +method = "skillorchestra" +bench = "swebench-verified" +n = 3 +local = { model = "Qwen/Qwen3.5-27B-FP8", endpoint = "http://localhost:8001/v1" } +cloud = { model = "claude-opus-4-7", endpoint = "anthropic" } +method_cfg = { swe_use_agent_loop = true, router_max_tokens = 1024, swe_max_turns = 30, swe_bash_timeout_s = 120, swe_turn_max_tokens = 4096 } + +# ------------- ToolOrchestra: orchestrator dispatches solver workers through loop on shared workdir ------------- + +[cells.toolorchestra-swe-agent-swebenchverified-qwen27b-opus-3] +method = "toolorchestra" +bench = "swebench-verified" +n = 3 +local = { model = "Qwen/Qwen3.5-27B-FP8", endpoint = "http://localhost:8001/v1" } +cloud = { model = "claude-opus-4-7", endpoint = "anthropic" } +method_cfg = { swe_use_agent_loop = true, max_turns = 4, orchestrator_max_tokens = 1024, swe_max_turns = 25, swe_bash_timeout_s = 120, swe_turn_max_tokens = 4096 } + +# ------------- Archon: K=3 local mini-SWE-agent runs, cloud ranker picks best ------------- + +[cells.archon-swe-agent-swebenchverified-qwen27b-opus-3] +method = "archon" +bench = "swebench-verified" +n = 3 +local = { model = "Qwen/Qwen3.5-27B-FP8", endpoint = "http://localhost:8001/v1" } +cloud = { model = "claude-opus-4-7", endpoint = "anthropic" } +method_cfg = { swe_use_agent_loop = true, n_samples = 3, swe_max_turns = 25, swe_bash_timeout_s = 120, swe_turn_max_tokens = 4096, ranker_max_tokens = 1024 }