From a3ba63d148cd8e13edbadaec2781ff3ae261d8c7 Mon Sep 17 00:00:00 2001 From: Avanika Narayan Date: Tue, 5 May 2026 13:37:33 -0700 Subject: [PATCH] AI_stack_support: subprocess-based external framework harness (#311) --- .github/workflows/ci.yml | 4 +- CHANGELOG.md | 21 + docs/development/contributing.md | 1 + pyproject.toml | 2 + scripts/smoke_framework_comparison.sh | 77 ++++ src/openjarvis/cli/_bg_state.py | 81 ++++ src/openjarvis/cli/eval_cmd.py | 30 +- src/openjarvis/evals/backends/_commit_util.py | 34 ++ .../evals/backends/external/__init__.py | 6 + .../backends/external/_runners/__init__.py | 6 + .../external/_runners/hermes_runner.py | 200 +++++++++ .../external/_runners/openclaw_runner.mjs | 116 +++++ .../backends/external/_subprocess_runner.py | 315 +++++++++++++ .../evals/backends/external/hermes_agent.py | 134 ++++++ .../evals/backends/external/openclaw.py | 122 +++++ src/openjarvis/evals/backends/jarvis_agent.py | 46 +- .../evals/backends/jarvis_direct.py | 28 +- src/openjarvis/evals/cli.py | 74 +++- src/openjarvis/evals/comparison/__init__.py | 5 + .../evals/comparison/make_configs.py | 207 +++++++++ src/openjarvis/evals/comparison/table_gen.py | 415 ++++++++++++++++++ .../evals/comparison/third_party.py | 170 +++++++ .../framework_comparison/_template.toml | 49 +++ .../framework_comparison/_third_party.toml | 27 ++ src/openjarvis/evals/core/backend.py | 10 + src/openjarvis/evals/core/config.py | 29 +- src/openjarvis/evals/core/export.py | 67 +++ src/openjarvis/evals/core/runner.py | 130 +++++- src/openjarvis/evals/core/types.py | 12 + tests/agents/test_native_openhands.py | 17 +- tests/cli/test_eval_cmd_external_flags.py | 26 ++ tests/evals/comparison/__init__.py | 1 + .../comparison/test_cli_backend_dispatch.py | 85 ++++ .../test_export_to_table_gen_roundtrip.py | 245 +++++++++++ tests/evals/comparison/test_hermes_backend.py | 191 ++++++++ tests/evals/comparison/test_live_external.py | 111 +++++ tests/evals/comparison/test_make_configs.py | 97 ++++ .../evals/comparison/test_openclaw_backend.py | 177 ++++++++ .../test_runner_error_path_framework.py | 189 ++++++++ .../comparison/test_subprocess_runner.py | 167 +++++++ tests/evals/comparison/test_table_gen.py | 273 ++++++++++++ tests/evals/comparison/test_third_party.py | 139 ++++++ .../evals/test_jarvis_agent_backend_fields.py | 47 ++ .../test_jarvis_direct_backend_fields.py | 43 ++ tests/install/__init__.py | 0 tests/install/conftest.py | 27 ++ tests/install/test_bg_state.py | 84 ++++ tests/install/test_fixture_smoke.py | 18 + uv.lock | 34 +- 49 files changed, 4360 insertions(+), 29 deletions(-) create mode 100755 scripts/smoke_framework_comparison.sh create mode 100644 src/openjarvis/cli/_bg_state.py create mode 100644 src/openjarvis/evals/backends/_commit_util.py create mode 100644 src/openjarvis/evals/backends/external/__init__.py create mode 100644 src/openjarvis/evals/backends/external/_runners/__init__.py create mode 100644 src/openjarvis/evals/backends/external/_runners/hermes_runner.py create mode 100644 src/openjarvis/evals/backends/external/_runners/openclaw_runner.mjs create mode 100644 src/openjarvis/evals/backends/external/_subprocess_runner.py create mode 100644 src/openjarvis/evals/backends/external/hermes_agent.py create mode 100644 src/openjarvis/evals/backends/external/openclaw.py create mode 100644 src/openjarvis/evals/comparison/__init__.py create mode 100644 src/openjarvis/evals/comparison/make_configs.py create mode 100644 src/openjarvis/evals/comparison/table_gen.py create mode 100644 src/openjarvis/evals/comparison/third_party.py create mode 100644 src/openjarvis/evals/configs/framework_comparison/_template.toml create mode 100644 src/openjarvis/evals/configs/framework_comparison/_third_party.toml create mode 100644 tests/cli/test_eval_cmd_external_flags.py create mode 100644 tests/evals/comparison/__init__.py create mode 100644 tests/evals/comparison/test_cli_backend_dispatch.py create mode 100644 tests/evals/comparison/test_export_to_table_gen_roundtrip.py create mode 100644 tests/evals/comparison/test_hermes_backend.py create mode 100644 tests/evals/comparison/test_live_external.py create mode 100644 tests/evals/comparison/test_make_configs.py create mode 100644 tests/evals/comparison/test_openclaw_backend.py create mode 100644 tests/evals/comparison/test_runner_error_path_framework.py create mode 100644 tests/evals/comparison/test_subprocess_runner.py create mode 100644 tests/evals/comparison/test_table_gen.py create mode 100644 tests/evals/comparison/test_third_party.py create mode 100644 tests/evals/test_jarvis_agent_backend_fields.py create mode 100644 tests/evals/test_jarvis_direct_backend_fields.py create mode 100644 tests/install/__init__.py create mode 100644 tests/install/conftest.py create mode 100644 tests/install/test_bg_state.py create mode 100644 tests/install/test_fixture_smoke.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 632cf242..028b4d9a 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -25,7 +25,7 @@ jobs: uses: astral-sh/setup-uv@v8.0.0 - name: Install dependencies - run: uv sync --extra dev + run: uv sync --extra dev --extra framework-comparison - name: Ruff check run: uv run ruff check src/ tests/ @@ -57,7 +57,7 @@ jobs: uses: astral-sh/setup-uv@v8.0.0 - name: Install dependencies - run: uv sync --extra dev + run: uv sync --extra dev --extra framework-comparison - name: Build Rust extension run: uv run maturin develop --manifest-path rust/crates/openjarvis-python/Cargo.toml diff --git a/CHANGELOG.md b/CHANGELOG.md index 958ea364..38d13aab 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -10,6 +10,27 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/). ### Added +- AI stack support for evaluating other agentic frameworks via subprocess. + New `evals/backends/external/` subpackage wraps Hermes Agent and OpenClaw + as one-shot subprocess backends behind the existing `InferenceBackend` + ABC; new `evals/comparison/` toolkit provides path + commit-pin + enforcement (`third_party.py`), config templating (`make_configs.py`), + and LaTeX table generation (`table_gen.py`). +- New optional extra `framework-comparison` (depends on `polars`). +- New pytest marker `live_external` for integration tests requiring real + foreign-framework installations. + +### Changed + +- `JarvisAgentBackend.generate_full` and `JarvisDirectBackend.generate_full` now return + the spec §6.2 extended fields (`energy_joules`, `peak_power_w`, `tool_calls`, + `turn_count`, `framework`, `framework_commit`, `error`) for cross-framework + comparison parity. Existing callers that didn't read these fields are unaffected. +- `_third_party.toml` no longer ships user-specific default paths. Set + `HERMES_AGENT_PATH` and `OPENCLAW_PATH` env vars to point at your local + checkouts before running the framework-comparison harness; missing or + empty paths now raise `ThirdPartyNotFoundError` with an actionable hint. + #### Skills System (Plans 1, 2A, 2B) - **Skills core** — every skill is a tool. Skills appear in a system prompt catalog, agents invoke them on demand, content (pipeline results, markdown instructions, or both) gets injected into context. diff --git a/docs/development/contributing.md b/docs/development/contributing.md index 5ec38bdd..bf7b95e5 100644 --- a/docs/development/contributing.md +++ b/docs/development/contributing.md @@ -413,3 +413,4 @@ policy: if the component requires new packages See the [registry pattern](#registry-pattern) section above for complete examples. + diff --git a/pyproject.toml b/pyproject.toml index 84a52e3c..63d75e88 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -117,6 +117,7 @@ speech = ["faster-whisper>=1.0"] speech-deepgram = ["deepgram-sdk>=3.0"] eval-wandb = ["wandb>=0.17"] eval-sheets = ["gspread>=6.0", "google-auth>=2.0"] +framework-comparison = ["polars>=1.0"] docs = [ "mkdocs>=1.6", "mkdocs-material>=9.5", @@ -152,6 +153,7 @@ markers = [ "macos15: requires macOS 15+ (Sequoia) for Apple FM", "slow: long-running test", "live_channel: requires real channel credentials (env vars)", + "live_external: requires HERMES_AGENT_PATH and OPENCLAW_PATH; spawns real foreign-framework subprocesses", ] [tool.ruff] diff --git a/scripts/smoke_framework_comparison.sh b/scripts/smoke_framework_comparison.sh new file mode 100755 index 00000000..f9ad6faa --- /dev/null +++ b/scripts/smoke_framework_comparison.sh @@ -0,0 +1,77 @@ +#!/usr/bin/env bash +# smoke_framework_comparison.sh +# +# One-task-per-cell sanity check for the framework-comparison harness. +# Run before kicking off a benchmark sweep. +# +# Required env vars: +# HERMES_AGENT_PATH - path to pinned hermes-agent checkout +# OPENCLAW_PATH - path to pinned openclaw checkout +# JARVIS_MOCK_LLM_URL - OpenAI-compatible endpoint (Ollama or vLLM) +# +# Optional: +# JARVIS_ALLOW_COMMIT_DRIFT=1 - bypass commit-pin enforcement + +set -euo pipefail + +: "${HERMES_AGENT_PATH:?must be set}" +: "${OPENCLAW_PATH:?must be set}" +: "${JARVIS_MOCK_LLM_URL:?must be set (e.g. http://localhost:11434/v1)}" + +# OpenClaw prerequisites: Node version + dist/ dir +NODE_VERSION=$(node --version 2>&1 || echo "v0") +NODE_MAJOR=$(echo "$NODE_VERSION" | sed -E 's/v([0-9]+)\..*/\1/') +if [ "$NODE_MAJOR" -lt 14 ]; then + echo "WARNING: Node $NODE_VERSION may be too old for OpenClaw (needs ≥14.8)" + echo " OpenClaw runs may fail with 'SyntaxError: Unexpected reserved word'" +fi +if [ ! -f "$OPENCLAW_PATH/dist/entry.js" ]; then + echo "WARNING: $OPENCLAW_PATH/dist/entry.js not found" + echo " OpenClaw needs 'pnpm install && pnpm build' before use" +fi + +REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +cd "$REPO_ROOT" + +echo "==> Verifying commit pins" +uv run python -c " +from openjarvis.evals.comparison.third_party import ( + load_third_party_config, verify_commit_pin, +) +cfg = load_third_party_config() +for name, entry in cfg.entries.items(): + print(f' {name}: {entry.path}') + verify_commit_pin(entry) +print(' all pins OK') +" + +echo "==> Running one-task smoke per (framework, benchmark)" +mkdir -p results/smoke +SMOKE_BENCHES=(toolcall15 pinchbench gaia) +SMOKE_FRAMEWORKS=(hermes openclaw openjarvis) +SMOKE_MODEL="qwen-9b" + +for fwk in "${SMOKE_FRAMEWORKS[@]}"; do + for bench in "${SMOKE_BENCHES[@]}"; do + config="src/openjarvis/evals/configs/framework_comparison/${bench}-${fwk}-${SMOKE_MODEL}.toml" + if [ ! -f "$config" ]; then + echo " ! missing config: $config (run make_configs --all-tier1 (or materialize the configs you need))" + continue + fi + echo " ▸ $fwk × $bench" + uv run python -m openjarvis.evals run --config "$config" --max-samples 1 \ + --output-dir "results/smoke/${fwk}/${SMOKE_MODEL}/${bench}/" \ + || echo " FAILED (continuing)" + done +done + +echo "==> Generating T1 from smoke results" +uv run python -m openjarvis.evals.comparison.table_gen \ + --results-glob "results/smoke/**/summary.json" \ + --tables T1 \ + --output-dir results/smoke/tables/ + +echo "==> Verifying T1.tex non-empty" +test -s results/smoke/tables/T1.tex && echo " OK: T1.tex emitted" + +echo "==> Smoke validation complete" diff --git a/src/openjarvis/cli/_bg_state.py b/src/openjarvis/cli/_bg_state.py new file mode 100644 index 00000000..13392776 --- /dev/null +++ b/src/openjarvis/cli/_bg_state.py @@ -0,0 +1,81 @@ +"""Read background-work state from ``~/.openjarvis/.state/``. + +Pure-function reader used by the chat banner, completion-notification +dispatcher, and ``jarvis doctor``. No side effects — safe to call +between every chat turn. +""" + +from __future__ import annotations + +from dataclasses import dataclass, field +from pathlib import Path +from typing import Dict, Optional + +from openjarvis.core import config + + +@dataclass(slots=True) +class BgStatus: + """Snapshot of background-work state.""" + + rust_extension: str = "pending" # pending | ready | failed + rust_error: str = "" + models: Dict[str, str] = field( + default_factory=dict + ) # id -> pending|downloading|ready|failed + + def all_ready(self) -> bool: + if self.rust_extension != "ready": + return False + if any(s != "ready" for s in self.models.values()): + return False + return True + + +def _safe_read(path: Path) -> Optional[str]: + """Read a file, returning None if it disappears mid-read (race).""" + try: + return path.read_text() + except FileNotFoundError: + return None + + +def get_status(home: Optional[Path] = None) -> BgStatus: + """Snapshot the background-work state from the state directory.""" + home = home or config.DEFAULT_CONFIG_DIR + state_dir = home / ".state" + models_dir = state_dir / "models" + + status = BgStatus() + + # Rust extension: ready supersedes failed; failed supersedes pending. + if (state_dir / "extension-built").exists(): + status.rust_extension = "ready" + elif (state_dir / "extension-failed").exists(): + contents = _safe_read(state_dir / "extension-failed") + if contents is not None: + status.rust_extension = "failed" + status.rust_error = contents + + # Models: parse files in models_dir; .ready supersedes .downloading and .failed. + if models_dir.is_dir(): + # First pass: capture every model id we see. + seen: Dict[str, str] = {} + for f in models_dir.iterdir(): + if f.suffix not in (".downloading", ".ready", ".failed"): + continue + model_id = f.name[: -len(f.suffix)] + new_state = f.suffix.lstrip(".") + current = seen.get(model_id, "") + # Precedence: ready > failed > downloading + if current == "ready": + continue + if new_state == "ready": + seen[model_id] = "ready" + elif new_state == "failed" and current != "ready": + seen[model_id] = "failed" + elif new_state == "downloading" and current == "": + seen[model_id] = "downloading" + status.models = seen + + return status diff --git a/src/openjarvis/cli/eval_cmd.py b/src/openjarvis/cli/eval_cmd.py index 740f24aa..f2e9cde6 100644 --- a/src/openjarvis/cli/eval_cmd.py +++ b/src/openjarvis/cli/eval_cmd.py @@ -3,6 +3,7 @@ from __future__ import annotations import json +import os import sys from pathlib import Path from typing import Optional @@ -132,7 +133,28 @@ def eval_list() -> None: "--backend", "backend", default="jarvis-direct", - help="Inference backend (jarvis-direct or jarvis-agent).", + type=click.Choice( + ["jarvis-direct", "jarvis-agent", "hermes", "openclaw", "terminalbench-native"] + ), + help=( + "Inference backend. For hermes/openclaw, also pass --base-url and " + "--api-key (or set JARVIS_BACKEND_BASE_URL/JARVIS_BACKEND_API_KEY)." + ), +) +@click.option( + "--base-url", + "base_url", + default=None, + help=( + "OpenAI-compat endpoint URL for hermes/openclaw backends " + "(env: JARVIS_BACKEND_BASE_URL)." + ), +) +@click.option( + "--api-key", + "api_key", + default=None, + help=("API key for the hermes/openclaw endpoint (env: JARVIS_BACKEND_API_KEY)."), ) @click.option( "--agent", @@ -256,6 +278,8 @@ def eval_run( model: Optional[str], max_samples: Optional[int], backend: str, + base_url: Optional[str], + api_key: Optional[str], agent_name: Optional[str], engine_key: Optional[str], tools: str, @@ -375,6 +399,10 @@ def eval_run( sheets_spreadsheet_id=sheets_spreadsheet_id, sheets_worksheet=sheets_worksheet, sheets_credentials_path=sheets_credentials_path, + # Spec §6.2 — for hermes/openclaw external backends. Falls back to env vars + # so users can also set JARVIS_BACKEND_BASE_URL/JARVIS_BACKEND_API_KEY. + base_url=base_url or os.environ.get("JARVIS_BACKEND_BASE_URL"), + api_key=api_key or os.environ.get("JARVIS_BACKEND_API_KEY"), ) try: diff --git a/src/openjarvis/evals/backends/_commit_util.py b/src/openjarvis/evals/backends/_commit_util.py new file mode 100644 index 00000000..1c5d3d1a --- /dev/null +++ b/src/openjarvis/evals/backends/_commit_util.py @@ -0,0 +1,34 @@ +"""Shared helper for resolving the openjarvis repo's HEAD commit.""" + +from __future__ import annotations + +import functools +import subprocess +from pathlib import Path + + +@functools.lru_cache(maxsize=1) +def openjarvis_commit() -> str: + """Return the openjarvis repo's HEAD commit (cached, lru-1). + + Returns ``"unknown"`` if git is unavailable or the path isn't a repo. + Used by JarvisAgentBackend and JarvisDirectBackend to populate the + ``framework_commit`` field in their extended return dicts. + """ + try: + # Walk up to repo root: backends/ -> evals/ -> openjarvis/ -> src/ -> repo + result = subprocess.run( + [ + "git", + "-C", + str(Path(__file__).resolve().parents[4]), + "rev-parse", + "HEAD", + ], + capture_output=True, + text=True, + check=True, + ) + return result.stdout.strip() + except Exception: + return "unknown" diff --git a/src/openjarvis/evals/backends/external/__init__.py b/src/openjarvis/evals/backends/external/__init__.py new file mode 100644 index 00000000..88085323 --- /dev/null +++ b/src/openjarvis/evals/backends/external/__init__.py @@ -0,0 +1,6 @@ +"""External-framework subprocess backends (Hermes Agent, OpenClaw).""" + +from openjarvis.evals.backends.external.hermes_agent import HermesBackend +from openjarvis.evals.backends.external.openclaw import OpenClawBackend + +__all__ = ["HermesBackend", "OpenClawBackend"] diff --git a/src/openjarvis/evals/backends/external/_runners/__init__.py b/src/openjarvis/evals/backends/external/_runners/__init__.py new file mode 100644 index 00000000..6596b5e2 --- /dev/null +++ b/src/openjarvis/evals/backends/external/_runners/__init__.py @@ -0,0 +1,6 @@ +"""Subprocess bridge scripts for foreign-framework evaluation backends. + +These scripts are NOT part of the OpenJarvis import graph. They run in +their own Python (or Node) subprocess with the foreign framework on +sys.path. +""" diff --git a/src/openjarvis/evals/backends/external/_runners/hermes_runner.py b/src/openjarvis/evals/backends/external/_runners/hermes_runner.py new file mode 100644 index 00000000..3f45ed14 --- /dev/null +++ b/src/openjarvis/evals/backends/external/_runners/hermes_runner.py @@ -0,0 +1,200 @@ +"""Subprocess bridge: runs one task through real Hermes Agent and emits JSON. + +Invoked as: + python hermes_runner.py \\ + --task --model --base-url --api-key \\ + --api-mode --output-json [--workspace ] \\ + [--max-iterations 90] [--system-prompt ] + +Imports `AIAgent` from `run_agent` (the top-level module Hermes ships +at the path indicated by `HERMES_AGENT_PATH`, set by the calling +backend). Writes a JSON dict matching the `_RunnerOutput` schema in +`_subprocess_runner.py` to `--output-json` before exiting. +""" + +from __future__ import annotations + +import argparse +import json +import os +import sys +import traceback +from pathlib import Path + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--task", required=True) + parser.add_argument("--model", required=True) + parser.add_argument("--base-url", required=True) + parser.add_argument("--api-key", required=True) + parser.add_argument("--api-mode", default="chat_completions") + parser.add_argument("--output-json", required=True, type=Path) + parser.add_argument("--workspace", default="") + parser.add_argument("--max-iterations", type=int, default=90) + parser.add_argument("--system-prompt", default="") + args = parser.parse_args() + + output: dict = { + "content": "", + "usage": {}, + "trajectory": [], + "tool_calls": 0, + "turn_count": 0, + "error": None, + } + + hermes_path = os.environ.get("HERMES_AGENT_PATH") + if not hermes_path: + output["error"] = "HERMES_AGENT_PATH not set" + args.output_json.write_text(json.dumps(output)) + return 2 + + sys.path.insert(0, hermes_path) + if args.workspace: + os.chdir(args.workspace) + + try: + from run_agent import AIAgent # type: ignore[import-not-found] + except ImportError as e: + output["error"] = f"hermes_import_failed: {e}" + args.output_json.write_text(json.dumps(output)) + return 3 + + try: + agent = AIAgent( + base_url=args.base_url, + api_key=args.api_key, + model=args.model, + api_mode=args.api_mode, + max_iterations=args.max_iterations, + quiet_mode=True, + save_trajectories=True, + platform="openjarvis-eval", + ) + kwargs = {} + if args.system_prompt: + kwargs["system_message"] = args.system_prompt + result = agent.run_conversation(args.task, **kwargs) + + # Validate that Hermes returned the expected keys; if not, capture + # what we actually got so debugging isn't a guessing game. + if not isinstance(result, dict): + output["error"] = ( + f"hermes_returned_non_dict: type={type(result).__name__}, " + f"value={str(result)[:200]}" + ) + args.output_json.write_text(json.dumps(output)) + return 0 + expected_keys = {"final_response", "messages"} + if not (expected_keys & set(result.keys())): + output["error"] = ( + f"hermes_returned_unexpected_shape: keys={list(result.keys())}" + ) + args.output_json.write_text(json.dumps(output)) + return 0 + + # Hermes returns {"final_response": str, "messages": [dict, ...]} + # plus aggregate token counts at the TOP LEVEL of the result dict + # (see `run_agent.py::AIAgent.run_conversation` — the result dict + # carries `prompt_tokens`, `completion_tokens`, `total_tokens`, + # `input_tokens`, `output_tokens` keys, populated from the agent's + # `session_*_tokens` accumulators). Per-message `usage` dicts are + # NOT populated by Hermes, so the previous per-message-sum approach + # always produced 0. + messages = result.get("messages", []) + + def _safe_int(v: object) -> int: + try: + return int(v or 0) + except (TypeError, ValueError): + return 0 + + # Primary: top-level aggregate (matches Hermes's actual return shape). + prompt_tokens = _safe_int(result.get("prompt_tokens")) + completion_tokens = _safe_int(result.get("completion_tokens")) + + # Secondary: canonical input/output names (Hermes also exposes these). + if not prompt_tokens: + prompt_tokens = _safe_int(result.get("input_tokens")) + if not completion_tokens: + completion_tokens = _safe_int(result.get("output_tokens")) + + # Tertiary: a nested `usage` dict, in case Hermes's API surface + # changes shape in a future version. + if not prompt_tokens or not completion_tokens: + top_level_usage = result.get("usage") + if isinstance(top_level_usage, dict): + if not prompt_tokens: + prompt_tokens = _safe_int( + top_level_usage.get("prompt_tokens") + or top_level_usage.get("input_tokens") + ) + if not completion_tokens: + completion_tokens = _safe_int( + top_level_usage.get("completion_tokens") + or top_level_usage.get("output_tokens") + ) + + # Final fallback: sum per-message usage (works only if Hermes + # populates per-turn usage — currently it does not, but keep the + # path so the bridge degrades gracefully if behaviour changes). + if not prompt_tokens: + prompt_tokens = sum( + _safe_int(m.get("usage", {}).get("prompt_tokens")) + for m in messages + if isinstance(m, dict) + ) + if not completion_tokens: + completion_tokens = sum( + _safe_int(m.get("usage", {}).get("completion_tokens")) + for m in messages + if isinstance(m, dict) + ) + + # Last-resort: agent instance attribute (run_conversation + # accumulates into self.session_*_tokens during the call). + if not prompt_tokens: + prompt_tokens = _safe_int(getattr(agent, "session_prompt_tokens", 0)) + if not completion_tokens: + completion_tokens = _safe_int( + getattr(agent, "session_completion_tokens", 0) + ) + + # Prefer Hermes's top-level total_tokens if it's been populated; + # otherwise reconstruct as prompt + completion. + total_tokens = _safe_int(result.get("total_tokens")) + if not total_tokens: + total_tokens = prompt_tokens + completion_tokens + + tool_calls = sum( + len(m.get("tool_calls", []) or []) for m in messages if isinstance(m, dict) + ) + turn_count = sum( + 1 for m in messages if isinstance(m, dict) and m.get("role") == "assistant" + ) + + output.update( + { + "content": result.get("final_response", ""), + "usage": { + "prompt_tokens": prompt_tokens, + "completion_tokens": completion_tokens, + "total_tokens": total_tokens, + }, + "trajectory": messages, + "tool_calls": tool_calls, + "turn_count": turn_count, + "error": None, + } + ) + except Exception as e: + output["error"] = f"hermes_runtime_error: {e}" + output["trajectory"] = [{"traceback": traceback.format_exc()}] + + args.output_json.write_text(json.dumps(output)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/src/openjarvis/evals/backends/external/_runners/openclaw_runner.mjs b/src/openjarvis/evals/backends/external/_runners/openclaw_runner.mjs new file mode 100644 index 00000000..d199a740 --- /dev/null +++ b/src/openjarvis/evals/backends/external/_runners/openclaw_runner.mjs @@ -0,0 +1,116 @@ +// src/openjarvis/evals/backends/external/_runners/openclaw_runner.mjs +// Subprocess bridge: runs one task through OpenClaw and emits JSON. +// +// Invoked as: +// node openclaw_runner.mjs \ +// --task --model --base-url --api-key \ +// --output-json [--workspace ] +// +// Loads OpenClaw from $OPENCLAW_PATH and runs `openclaw chat --message --json` +// (or equivalent non-interactive entry). Emits a JSON dict matching the +// _RunnerOutput Pydantic schema in _subprocess_runner.py. + +import { writeFileSync, existsSync } from 'node:fs'; +import { spawn } from 'node:child_process'; +import { join } from 'node:path'; +import { argv, env, exit, chdir } from 'node:process'; + +function parseArgs(args) { + const out = {}; + for (let i = 2; i < args.length; i += 2) { + const key = args[i].replace(/^--/, '').replace(/-/g, '_'); + out[key] = args[i + 1]; + } + return out; +} + +async function main() { + const args = parseArgs(argv); + const output = { + content: '', usage: {}, trajectory: [], + tool_calls: 0, turn_count: 0, error: null, + }; + + const openclawPath = env.OPENCLAW_PATH; + if (!openclawPath) { + output.error = 'OPENCLAW_PATH not set'; + writeFileSync(args.output_json, JSON.stringify(output)); + return 2; + } + + if (args.workspace) chdir(args.workspace); + + const openclawBin = join(openclawPath, 'openclaw.mjs'); + if (!existsSync(openclawBin)) { + output.error = `openclaw entry not found: ${openclawBin}`; + writeFileSync(args.output_json, JSON.stringify(output)); + return 3; + } + + const childEnv = { + ...env, + OPENCLAW_MODEL: args.model, + OPENCLAW_BASE_URL: args.base_url, + OPENCLAW_API_KEY: args.api_key, + }; + + // Use the SAME node executable that's running this script — picking up + // 'node' from PATH can resolve to a system Node too old for OpenClaw + // (which uses top-level await, requiring Node >=14.8). + const nodeExe = process.execPath; + + // Use `openclaw agent --local` for headless single-shot invocation: + // runs the embedded agent locally without going through the Gateway, + // emits JSON. A unique --session-id per invocation gives each task a + // fresh OpenClaw session (no carryover between eval tasks). + const sessionId = ( + `openjarvis-eval-${Date.now()}-` + + Math.floor(Math.random() * 1e9).toString(36) + ); + const child = spawn(nodeExe, [ + openclawBin, 'agent', + '--local', + '--session-id', sessionId, + '--message', args.task, + '--json', + ], { env: childEnv }); + + let stdout = ''; + let stderr = ''; + child.stdout.on('data', (d) => { stdout += d.toString(); }); + child.stderr.on('data', (d) => { stderr += d.toString(); }); + + const exitCode = await new Promise((resolve) => { + child.on('close', resolve); + }); + + if (exitCode !== 0) { + output.error = `openclaw_exit_${exitCode}: ${stderr.slice(-500)}`; + writeFileSync(args.output_json, JSON.stringify(output)); + return exitCode; + } + + // Parse OpenClaw's JSON output. Schema (provisional, validated against + // OpenClaw's actual --json output during integration tests): + // { response: str, usage: {...}, messages: [{...}], tool_calls: int } + try { + const parsed = JSON.parse(stdout); + output.content = parsed.response || ''; + output.usage = parsed.usage || {}; + output.trajectory = parsed.messages || []; + output.tool_calls = parsed.tool_calls || 0; + output.turn_count = (parsed.messages || []).filter( + (m) => m.role === 'assistant' + ).length; + } catch (e) { + output.error = `openclaw_output_parse_failed: ${e.message}`; + } + + writeFileSync(args.output_json, JSON.stringify(output)); + return 0; +} + +main().then(exit).catch((e) => { + console.error(e); + exit(1); +}); diff --git a/src/openjarvis/evals/backends/external/_subprocess_runner.py b/src/openjarvis/evals/backends/external/_subprocess_runner.py new file mode 100644 index 00000000..0882495d --- /dev/null +++ b/src/openjarvis/evals/backends/external/_subprocess_runner.py @@ -0,0 +1,315 @@ +"""Shared one-shot subprocess invocation with energy/timing capture. + +Used by HermesBackend and OpenClawBackend to spawn their respective +foreign-framework runner scripts. Hermetic: a fresh subprocess per task, +energy sampled by NVML/powermetrics/ROCm-SMI/RAPL fallback chain. +""" + +from __future__ import annotations + +import json +import logging +import shutil +import subprocess +import threading +import time +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any, Dict, List, Mapping, Optional, Tuple + +from pydantic import BaseModel, Field, ValidationError + +LOGGER = logging.getLogger(__name__) + +_GRACE_PERIOD_SECONDS = 30.0 + + +@dataclass(slots=True) +class EnergySample: + """One sample from the energy sampler thread.""" + + timestamp: float # seconds since sampler start + watts: float + + +@dataclass(slots=True) +class SubprocessResult: + """Result of a one-shot subprocess invocation.""" + + stdout: str + stderr: str + exit_code: int + latency_seconds: float + energy_joules: Optional[float] + peak_power_w: Optional[float] + sampler_method: str # "nvml" | "powermetrics" | "rocm_smi" | "rapl" | "unavailable" + parsed_json: Dict[str, Any] = field(default_factory=dict) + samples: List[EnergySample] = field(default_factory=list) + # "subprocess_crash" | "timeout" | "malformed_runner_output" | ... + error: Optional[str] = None + + +class _RunnerOutput(BaseModel): + """Schema the foreign-framework runner scripts must emit.""" + + content: str + usage: Dict[str, int] = Field(default_factory=dict) + trajectory: List[Dict[str, Any]] = Field(default_factory=list) + tool_calls: int = 0 + turn_count: int = 0 + error: Optional[str] = None + + +def run_one_shot( + cmd: List[str], + env: Mapping[str, str], + timeout: float, + output_json_path: Path, +) -> SubprocessResult: + """Spawn cmd as a one-shot subprocess; capture stdout, energy, JSON output. + + The subprocess is expected to write its result as JSON matching + ``_RunnerOutput`` to ``output_json_path`` before exiting. Energy is + sampled at 10 Hz over the process's lifetime via the fallback sampler + chain (NVML -> powermetrics -> ROCm-SMI -> RAPL -> unavailable). + + Returns a ``SubprocessResult`` with structured ``error`` field on + failure; never raises for subprocess crashes (those are signal, not + exceptions). + """ + sampler = _start_sampler() + t0 = time.monotonic() + try: + proc = subprocess.Popen( + cmd, + env=dict(env), + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + ) + except FileNotFoundError as e: + sampler.stop() + return SubprocessResult( + stdout="", + stderr=str(e), + exit_code=-1, + latency_seconds=0.0, + energy_joules=None, + peak_power_w=None, + sampler_method=sampler.method, + error="subprocess_crash", + ) + + error: Optional[str] = None + try: + stdout, stderr = proc.communicate(timeout=timeout) + except subprocess.TimeoutExpired: + proc.terminate() + try: + stdout, stderr = proc.communicate(timeout=_GRACE_PERIOD_SECONDS) + except subprocess.TimeoutExpired: + proc.kill() + stdout, stderr = proc.communicate() + error = "timeout" + + elapsed = time.monotonic() - t0 + samples = sampler.stop() + energy_j, peak_w = _integrate_energy(samples, elapsed) + + if error is None and proc.returncode != 0: + error = "subprocess_crash" + + parsed: Dict[str, Any] = {} + if error is None: + if not output_json_path.exists(): + error = "invalid_runner_output" + else: + try: + raw = json.loads(output_json_path.read_text()) + _RunnerOutput.model_validate(raw) # validate shape + parsed = raw + except json.JSONDecodeError: + error = "malformed_runner_output" + except ValidationError as e: + LOGGER.error("runner_output_validation_failed: %s", e) + error = "invalid_runner_output" + + return SubprocessResult( + stdout=stdout, + stderr=stderr, + exit_code=proc.returncode, + latency_seconds=elapsed, + energy_joules=energy_j, + peak_power_w=peak_w, + sampler_method=sampler.method, + parsed_json=parsed, + samples=samples, + error=error, + ) + + +_SAMPLE_HZ = 10.0 +_SAMPLE_INTERVAL = 1.0 / _SAMPLE_HZ + + +class _Sampler: + """Base class for energy samplers running on a background thread.""" + + method: str = "unavailable" + + def __init__(self) -> None: + self._samples: List[EnergySample] = [] + self._stop_event = threading.Event() + self._thread: Optional[threading.Thread] = None + self._t0 = time.monotonic() + + def _read_watts(self) -> float: # pragma: no cover (subclass override) + raise NotImplementedError + + def _loop(self) -> None: + while not self._stop_event.is_set(): + try: + w = self._read_watts() + self._samples.append( + EnergySample(timestamp=time.monotonic() - self._t0, watts=w) + ) + except Exception as e: # never crash the sampler + LOGGER.debug("sampler_read_failed: %s", e) + self._stop_event.wait(_SAMPLE_INTERVAL) + + def start(self) -> None: + self._t0 = time.monotonic() + self._thread = threading.Thread(target=self._loop, daemon=True) + self._thread.start() + + def stop(self) -> List[EnergySample]: + self._stop_event.set() + if self._thread is not None: + self._thread.join(timeout=2.0) + return list(self._samples) + + +class _NullSampler(_Sampler): + method = "unavailable" + + def start(self) -> None: + pass # nothing to do + + def stop(self) -> List[EnergySample]: + return [] + + +def _try_start_nvml() -> Optional[_Sampler]: + try: + import pynvml # type: ignore + + pynvml.nvmlInit() + handle = pynvml.nvmlDeviceGetHandleByIndex(0) + + class _NvmlSampler(_Sampler): + method = "nvml" + + def _read_watts(self) -> float: + return pynvml.nvmlDeviceGetPowerUsage(handle) / 1000.0 + + s = _NvmlSampler() + s.start() + return s + except Exception as e: + LOGGER.debug("nvml_unavailable: %s", e) + return None + + +def _try_start_powermetrics() -> Optional[_Sampler]: + """macOS only - requires sudo or appropriate plist permissions.""" + if shutil.which("powermetrics") is None: + return None + # NOTE: real powermetrics integration is non-trivial (parse XML over stdout). + # Apple-hardware support is a follow-up; documented in CHANGELOG. + return None + + +def _try_start_rocm_smi() -> Optional[_Sampler]: + if shutil.which("rocm-smi") is None: + return None + + class _RocmSampler(_Sampler): + method = "rocm_smi" + + def _read_watts(self) -> float: + out = subprocess.run( + ["rocm-smi", "--showpower", "--csv"], + capture_output=True, + text=True, + check=True, + ).stdout + # Parse "card0,15.5W" style; first number after a comma + for line in out.splitlines(): + parts = [p.strip() for p in line.split(",")] + for p in parts[1:]: + if "W" in p: + return float(p.replace("W", "")) + return 0.0 + + s = _RocmSampler() + s.start() + return s + + +def _try_start_rapl() -> Optional[_Sampler]: + rapl_path = Path("/sys/class/powercap/intel-rapl:0/energy_uj") + if not rapl_path.exists(): + return None + + class _RaplSampler(_Sampler): + method = "rapl" + + def __init__(self) -> None: + super().__init__() + self._last_uj = int(rapl_path.read_text().strip()) + self._last_t = time.monotonic() + + def _read_watts(self) -> float: + now_uj = int(rapl_path.read_text().strip()) + now_t = time.monotonic() + d_uj = now_uj - self._last_uj + d_t = now_t - self._last_t + self._last_uj = now_uj + self._last_t = now_t + return (d_uj * 1e-6) / d_t if d_t > 0 else 0.0 + + s = _RaplSampler() + s.start() + return s + + +def _start_sampler() -> _Sampler: + """Return a started sampler from the fallback chain, or the null sampler.""" + for try_fn in ( + _try_start_nvml, + _try_start_powermetrics, + _try_start_rocm_smi, + _try_start_rapl, + ): + s = try_fn() + if s is not None: + return s + return _NullSampler() + + +def _integrate_energy( + samples: List[EnergySample], elapsed: float +) -> Tuple[Optional[float], Optional[float]]: + """Trapezoidal-rule integration over sampled wattage to Joules.""" + if not samples: + return None, None + if len(samples) == 1: + return samples[0].watts * elapsed, samples[0].watts + energy = 0.0 + peak = 0.0 + for a, b in zip(samples, samples[1:]): + dt = b.timestamp - a.timestamp + avg_w = (a.watts + b.watts) / 2.0 + energy += avg_w * dt + peak = max(peak, a.watts, b.watts) + return energy, peak diff --git a/src/openjarvis/evals/backends/external/hermes_agent.py b/src/openjarvis/evals/backends/external/hermes_agent.py new file mode 100644 index 00000000..2fcb050a --- /dev/null +++ b/src/openjarvis/evals/backends/external/hermes_agent.py @@ -0,0 +1,134 @@ +"""HermesBackend - runs real Hermes Agent as a subprocess per task. + +Implements the InferenceBackend ABC by spawning hermes_runner.py with the +foreign Hermes installation on sys.path. Foreign code never imports into +this process; we cross the boundary only via stdin/stdout JSON. +""" + +from __future__ import annotations + +import os +import sys +import tempfile +from pathlib import Path +from typing import Any, Dict + +from openjarvis.evals.backends.external._subprocess_runner import run_one_shot +from openjarvis.evals.comparison.third_party import ( + ThirdPartyEntry, + load_third_party_config, + verify_commit_pin, +) +from openjarvis.evals.core.backend import InferenceBackend + + +class HermesBackend(InferenceBackend): + """Spawn real Hermes Agent (pinned commit) as a subprocess per task.""" + + backend_id = "hermes" + framework_name = "hermes" + + def __init__( + self, + base_url: str, + api_key: str, + api_mode: str = "chat_completions", + max_iterations: int = 90, + timeout_seconds: float = 7200.0, + ) -> None: + self._base_url = base_url + self._api_key = api_key + self._api_mode = api_mode + self._max_iterations = max_iterations + self._timeout = timeout_seconds + + cfg = load_third_party_config() + self._entry: ThirdPartyEntry = cfg.entries["hermes"] + verify_commit_pin(self._entry) + + @property + def framework_commit_value(self) -> str: + """Pinned commit of Hermes Agent (for telemetry tagging).""" + return self._entry.pinned_commit + + def generate( + self, + prompt: str, + *, + model: str, + system: str = "", + temperature: float = 0.0, + max_tokens: int = 2048, + ) -> str: + return self.generate_full( + prompt, + model=model, + system=system, + temperature=temperature, + max_tokens=max_tokens, + )["content"] + + def generate_full( + self, + prompt: str, + *, + model: str, + system: str = "", + temperature: float = 0.0, + max_tokens: int = 2048, + ) -> Dict[str, Any]: + with tempfile.NamedTemporaryFile(mode="w", suffix=".json", delete=False) as fh: + output_json = Path(fh.name) + + try: + python_exe = self._entry.python_executable or sys.executable + runner_script = ( + Path(__file__).resolve().parents[5] / self._entry.runner_script + ) + cmd = [ + python_exe, + str(runner_script), + "--task", + prompt, + "--model", + model, + "--base-url", + self._base_url, + "--api-key", + self._api_key, + "--api-mode", + self._api_mode, + "--max-iterations", + str(self._max_iterations), + "--output-json", + str(output_json), + ] + if system: + cmd.extend(["--system-prompt", system]) + + env = dict(os.environ) + env["HERMES_AGENT_PATH"] = str(self._entry.path) + + result = run_one_shot( + cmd=cmd, + env=env, + timeout=self._timeout, + output_json_path=output_json, + ) + finally: + output_json.unlink(missing_ok=True) + + return { + "content": result.parsed_json.get("content", ""), + "usage": result.parsed_json.get("usage", {}), + "model": model, + "latency_seconds": result.latency_seconds, + "energy_joules": result.energy_joules, + "peak_power_w": result.peak_power_w, + "tool_calls": result.parsed_json.get("tool_calls", 0), + "turn_count": result.parsed_json.get("turn_count", 0), + "framework": "hermes", + "framework_commit": self._entry.pinned_commit, + "cost_usd": 0.0, + "error": result.error or result.parsed_json.get("error"), + } diff --git a/src/openjarvis/evals/backends/external/openclaw.py b/src/openjarvis/evals/backends/external/openclaw.py new file mode 100644 index 00000000..a7c710de --- /dev/null +++ b/src/openjarvis/evals/backends/external/openclaw.py @@ -0,0 +1,122 @@ +"""OpenClawBackend - runs real OpenClaw as a Node subprocess per task.""" + +from __future__ import annotations + +import os +import tempfile +from pathlib import Path +from typing import Any, Dict + +from openjarvis.evals.backends.external._subprocess_runner import run_one_shot +from openjarvis.evals.comparison.third_party import ( + ThirdPartyEntry, + load_third_party_config, + verify_commit_pin, +) +from openjarvis.evals.core.backend import InferenceBackend + + +class OpenClawBackend(InferenceBackend): + """Spawn real OpenClaw (pinned commit) as a Node subprocess per task.""" + + backend_id = "openclaw" + framework_name = "openclaw" + + def __init__( + self, + base_url: str, + api_key: str, + timeout_seconds: float = 7200.0, + ) -> None: + self._base_url = base_url + self._api_key = api_key + self._timeout = timeout_seconds + + cfg = load_third_party_config() + self._entry: ThirdPartyEntry = cfg.entries["openclaw"] + verify_commit_pin(self._entry) + + @property + def framework_commit_value(self) -> str: + """Pinned commit of OpenClaw (for telemetry tagging).""" + return self._entry.pinned_commit + + def generate( + self, + prompt: str, + *, + model: str, + system: str = "", + temperature: float = 0.0, + max_tokens: int = 2048, + ) -> str: + return self.generate_full( + prompt, + model=model, + system=system, + temperature=temperature, + max_tokens=max_tokens, + )["content"] + + def generate_full( + self, + prompt: str, + *, + model: str, + system: str = "", + temperature: float = 0.0, + max_tokens: int = 2048, + ) -> Dict[str, Any]: + with tempfile.NamedTemporaryFile( + mode="w", + suffix=".json", + delete=False, + ) as fh: + output_json = Path(fh.name) + + try: + node_exe = self._entry.node_executable or "node" + # parents[5] = repo root (one level above src/) + runner_script = ( + Path(__file__).resolve().parents[5] / self._entry.runner_script + ) + cmd = [ + node_exe, + str(runner_script), + "--task", + prompt, + "--model", + model, + "--base-url", + self._base_url, + "--api-key", + self._api_key, + "--output-json", + str(output_json), + ] + env = dict(os.environ) + env["OPENCLAW_PATH"] = str(self._entry.path) + + result = run_one_shot( + cmd=cmd, + env=env, + timeout=self._timeout, + output_json_path=output_json, + ) + finally: + output_json.unlink(missing_ok=True) + + return { + "content": result.parsed_json.get("content", ""), + "usage": result.parsed_json.get("usage", {}), + "model": model, + "latency_seconds": result.latency_seconds, + "energy_joules": result.energy_joules, + "peak_power_w": result.peak_power_w, + "tool_calls": result.parsed_json.get("tool_calls", 0), + "turn_count": result.parsed_json.get("turn_count", 0), + "framework": "openclaw", + "framework_commit": self._entry.pinned_commit, + "cost_usd": 0.0, + "error": result.error or result.parsed_json.get("error"), + } diff --git a/src/openjarvis/evals/backends/jarvis_agent.py b/src/openjarvis/evals/backends/jarvis_agent.py index f1f72dce..1ba19f6f 100644 --- a/src/openjarvis/evals/backends/jarvis_agent.py +++ b/src/openjarvis/evals/backends/jarvis_agent.py @@ -6,6 +6,7 @@ import time from pathlib import Path from typing import Any, Dict, List, Optional +from openjarvis.evals.backends._commit_util import openjarvis_commit from openjarvis.evals.core.backend import InferenceBackend @@ -17,6 +18,7 @@ class JarvisAgentBackend(InferenceBackend): """ backend_id = "jarvis-agent" + framework_name = "openjarvis" def __init__( self, @@ -62,6 +64,13 @@ class JarvisAgentBackend(InferenceBackend): builder._config.learning.skills.overlay_dir = str(overlay_dir) self._system = builder.telemetry(telemetry).traces(True).build() + @property + def framework_commit_value(self) -> str: + """OpenJarvis repo HEAD commit (for telemetry tagging).""" + from openjarvis.evals.backends._commit_util import openjarvis_commit + + return openjarvis_commit() + def generate( self, prompt: str, @@ -131,6 +140,35 @@ class JarvisAgentBackend(InferenceBackend): usage = result.get("usage", {}) telemetry_data = result.get("_telemetry", {}) + + # Spec §6.2 extended fields for cross-framework comparison. + tool_calls_count = 0 + turn_count = 0 + if trace_data is not None: + steps = trace_data.get("steps", []) + tool_calls_count = sum( + 1 for s in steps if s.get("step_type") == "tool_call" + ) + turn_count = sum( + 1 for s in steps if s.get("step_type") in ("model_call", "agent_turn") + ) + # Fall back to result-reported counts when no trace is available so + # the comparison harness still gets meaningful turn data. + if turn_count == 0: + turn_count = int(result.get("turns", 1) or 1) + if tool_calls_count == 0: + tool_results = result.get("tool_results", []) or [] + tool_calls_count = len(tool_results) + + # Route real telemetry into the spec field names where present; + # ``None`` signals "not measured" to downstream consumers. + energy_joules = telemetry_data.get("energy_joules") + peak_power_w = telemetry_data.get("peak_power_w") + if peak_power_w is None: + # Older telemetry payloads expose only the running power; treat + # it as a coarse peak proxy when no explicit max is recorded. + peak_power_w = telemetry_data.get("power_watts") + return { "content": result.get("content", ""), "usage": usage, @@ -140,11 +178,17 @@ class JarvisAgentBackend(InferenceBackend): "turns": result.get("turns", 1), "tool_results": result.get("tool_results", []), "ttft": result.get("ttft", telemetry_data.get("ttft", 0.0)), - "energy_joules": telemetry_data.get("energy_joules", 0.0), + "energy_joules": energy_joules, + "peak_power_w": peak_power_w, "power_watts": telemetry_data.get("power_watts", 0.0), "gpu_utilization_pct": telemetry_data.get("gpu_utilization_pct", 0.0), "throughput_tok_per_sec": telemetry_data.get("throughput_tok_per_sec", 0.0), "trace_data": trace_data, + "tool_calls": tool_calls_count, + "turn_count": turn_count, + "framework": "openjarvis", + "framework_commit": openjarvis_commit(), + "error": None, } def set_task_metadata(self, metadata: dict) -> None: diff --git a/src/openjarvis/evals/backends/jarvis_direct.py b/src/openjarvis/evals/backends/jarvis_direct.py index 76c535e5..d0195882 100644 --- a/src/openjarvis/evals/backends/jarvis_direct.py +++ b/src/openjarvis/evals/backends/jarvis_direct.py @@ -5,6 +5,7 @@ from __future__ import annotations import time from typing import Any, Dict, Optional +from openjarvis.evals.backends._commit_util import openjarvis_commit from openjarvis.evals.core.backend import InferenceBackend @@ -16,6 +17,7 @@ class JarvisDirectBackend(InferenceBackend): """ backend_id = "jarvis-direct" + framework_name = "openjarvis" def __init__( self, @@ -37,6 +39,13 @@ class JarvisDirectBackend(InferenceBackend): builder._config.telemetry.gpu_metrics = True self._system = builder.telemetry(telemetry).traces(telemetry).build() + @property + def framework_commit_value(self) -> str: + """OpenJarvis repo HEAD commit (for telemetry tagging).""" + from openjarvis.evals.backends._commit_util import openjarvis_commit + + return openjarvis_commit() + def generate( self, prompt: str, @@ -82,6 +91,17 @@ class JarvisDirectBackend(InferenceBackend): usage = result.get("usage", {}) telemetry_data = result.get("_telemetry", {}) + + # Spec §6.2 extended fields for cross-framework comparison. + # Route real telemetry into the spec field names where present; + # ``None`` signals "not measured" to downstream consumers. + energy_joules = telemetry_data.get("energy_joules") + peak_power_w = telemetry_data.get("peak_power_w") + if peak_power_w is None: + # Older telemetry payloads expose only the running power; treat + # it as a coarse peak proxy when no explicit max is recorded. + peak_power_w = telemetry_data.get("power_watts") + return { "content": result.get("content", ""), "usage": usage, @@ -89,10 +109,16 @@ class JarvisDirectBackend(InferenceBackend): "latency_seconds": elapsed, "cost_usd": result.get("cost_usd", 0.0), "ttft": result.get("ttft", telemetry_data.get("ttft", 0.0)), - "energy_joules": telemetry_data.get("energy_joules", 0.0), + "energy_joules": energy_joules, + "peak_power_w": peak_power_w, "power_watts": telemetry_data.get("power_watts", 0.0), "gpu_utilization_pct": telemetry_data.get("gpu_utilization_pct", 0.0), "throughput_tok_per_sec": telemetry_data.get("throughput_tok_per_sec", 0.0), + "tool_calls": 0, + "turn_count": 1, + "framework": "openjarvis", + "framework_commit": openjarvis_commit(), + "error": None, } def close(self) -> None: diff --git a/src/openjarvis/evals/cli.py b/src/openjarvis/evals/cli.py index d4781f7f..4ab50116 100644 --- a/src/openjarvis/evals/cli.py +++ b/src/openjarvis/evals/cli.py @@ -4,6 +4,7 @@ from __future__ import annotations import json import logging +import os from pathlib import Path from typing import Optional @@ -153,6 +154,8 @@ BENCHMARKS = { BACKENDS = { "jarvis-direct": "Engine-level inference (local or cloud)", "jarvis-agent": "Agent-level inference with tool calling", + "hermes": "Real Hermes Agent (Nous Research) via subprocess", + "openclaw": "Real OpenClaw via Node subprocess", } @@ -174,8 +177,16 @@ def _build_backend( gpu_metrics: bool = False, model: Optional[str] = None, max_turns: Optional[int] = None, + base_url: Optional[str] = None, + api_key: Optional[str] = None, ): - """Construct the appropriate backend.""" + """Construct the appropriate backend. + + For "hermes" and "openclaw" backends, ``base_url`` and ``api_key`` are + REQUIRED — these foreign frameworks need an OpenAI-compatible endpoint + to send model calls to. Pass them via the eval config's + ``[backend.external]`` section or env vars. + """ if backend_name == "jarvis-agent": from openjarvis.evals.backends.jarvis_agent import JarvisAgentBackend @@ -188,7 +199,7 @@ def _build_backend( model=model, max_turns=max_turns, ) - else: + elif backend_name == "jarvis-direct": from openjarvis.evals.backends.jarvis_direct import JarvisDirectBackend return JarvisDirectBackend( @@ -196,6 +207,36 @@ def _build_backend( telemetry=telemetry, gpu_metrics=gpu_metrics, ) + elif backend_name == "hermes": + from openjarvis.evals.backends.external import HermesBackend + + if not base_url or not api_key: + raise click.UsageError( + "hermes backend requires --base-url and --api-key (or " + "the equivalent env vars / config entries) — Hermes needs " + "an OpenAI-compatible endpoint to call the model." + ) + return HermesBackend( + base_url=base_url, + api_key=api_key, + ) + elif backend_name == "openclaw": + from openjarvis.evals.backends.external import OpenClawBackend + + if not base_url or not api_key: + raise click.UsageError( + "openclaw backend requires --base-url and --api-key (or " + "the equivalent env vars / config entries) — OpenClaw needs " + "an OpenAI-compatible endpoint to call the model." + ) + return OpenClawBackend( + base_url=base_url, + api_key=api_key, + ) + else: + raise click.UsageError( + f"unknown backend {backend_name!r}; valid: {list(BACKENDS)}" + ) def _build_dataset(benchmark: str, subset: str | None = None): @@ -665,6 +706,7 @@ def _run_single(config, console: Optional[Console] = None) -> object: if config.benchmark == "terminalbench-native": return _run_terminalbench_native(config, console) + _metadata = getattr(config, "metadata", None) or {} eval_backend = _build_backend( config.backend, config.engine_key, @@ -674,6 +716,16 @@ def _run_single(config, console: Optional[Console] = None) -> object: gpu_metrics=getattr(config, "gpu_metrics", False), model=config.model, max_turns=getattr(config, "max_turns", None), + base_url=( + getattr(config, "base_url", None) + or _metadata.get("base_url") + or os.environ.get("JARVIS_BACKEND_BASE_URL") + ), + api_key=( + getattr(config, "api_key", None) + or _metadata.get("api_key") + or os.environ.get("JARVIS_BACKEND_API_KEY") + ), ) dataset = _build_dataset(config.benchmark) # Inject engine config for benchmarks that run their own simulation @@ -1044,6 +1096,16 @@ def main(): type=click.Choice(list(BACKENDS.keys())), help="Inference backend", ) +@click.option( + "--base-url", + default=None, + help="OpenAI-compat endpoint for hermes/openclaw", +) +@click.option( + "--api-key", + default=None, + help="API key for hermes/openclaw endpoint", +) @click.option("-m", "--model", default=None, help="Model identifier") @click.option( "-e", @@ -1152,6 +1214,8 @@ def run( config_path, benchmark, backend, + base_url, + api_key, model, engine_key, agent_name, @@ -1245,6 +1309,12 @@ def run( sheets_worksheet=sheets_worksheet, sheets_credentials_path=sheets_credentials_path, episode_mode=episode_mode, + base_url=base_url or os.environ.get("JARVIS_BACKEND_BASE_URL"), + api_key=api_key or os.environ.get("JARVIS_BACKEND_API_KEY"), + metadata={ + "base_url": base_url or os.environ.get("JARVIS_BACKEND_BASE_URL"), + "api_key": api_key or os.environ.get("JARVIS_BACKEND_API_KEY"), + }, ) # Banner + config diff --git a/src/openjarvis/evals/comparison/__init__.py b/src/openjarvis/evals/comparison/__init__.py new file mode 100644 index 00000000..2d38c7fb --- /dev/null +++ b/src/openjarvis/evals/comparison/__init__.py @@ -0,0 +1,5 @@ +"""Framework-comparison toolkit. + +Drives other agentic frameworks via subprocess and emits telemetry that +is comparable across frameworks (accuracy, latency, energy, tokens, cost). +""" diff --git a/src/openjarvis/evals/comparison/make_configs.py b/src/openjarvis/evals/comparison/make_configs.py new file mode 100644 index 00000000..6e03c9b1 --- /dev/null +++ b/src/openjarvis/evals/comparison/make_configs.py @@ -0,0 +1,207 @@ +"""CLI for generating framework-comparison eval configs from `_template.toml`. + +Usage: + python -m openjarvis.evals.comparison.make_configs \\ + --framework hermes --model qwen-9b --benchmark gaia +""" + +from __future__ import annotations + +import json +from pathlib import Path +from typing import Dict, Optional + +import click + +# --------------------------------------------------------------------------- +# Domain catalogs - single source of truth for what's a valid combo. +# Extend these dicts to add new frameworks / models / benchmarks. +# --------------------------------------------------------------------------- + +FRAMEWORKS: Dict[str, Dict[str, str]] = { + "openjarvis": {"backend_id": "jarvis-agent"}, + "openjarvis-distilled": {"backend_id": "jarvis-agent"}, + "hermes": {"backend_id": "hermes"}, + "openclaw": {"backend_id": "openclaw"}, +} + +MODELS: Dict[str, Dict[str, object]] = { + "claude-opus-46": { + "model_id": "claude-opus-4-6", + "model_pretty": "Claude Opus 4.6", + "engine": "cloud", + "num_gpus": 0, + "max_tokens": 8192, + }, + "qwen-9b": { + "model_id": "Qwen/Qwen3.5-9B", + "model_pretty": "Qwen3.5-9B", + "engine": "vllm", + "num_gpus": 1, + "max_tokens": 8192, + }, + "qwen-122b": { + "model_id": "Qwen/Qwen3.5-122B", + "model_pretty": "Qwen3.5-122B", + "engine": "vllm", + "num_gpus": 4, + "max_tokens": 8192, + }, +} + +BENCHMARKS: Dict[str, Dict[str, object]] = { + "toolcall15": { + "max_samples": 15, + "tools": ["think", "calculator"], + "temperature": 0.0, + }, + "pinchbench": { + "max_samples": 23, + "tools": ["think", "code_interpreter", "web_search", "file_read"], + "temperature": 0.6, + }, + "livecodebench": { + "max_samples": 100, + "tools": ["think", "code_interpreter"], + "temperature": 0.2, + }, + "taubench": { + "max_samples": 100, + "tools": ["think"], + "temperature": 0.0, + }, + "taubench-telecom": { + "max_samples": 40, + "tools": ["think"], + "temperature": 0.0, + }, + "gaia": { + "max_samples": 50, + "tools": [ + "think", + "calculator", + "code_interpreter", + "web_search", + "file_read", + ], + "temperature": 0.6, + }, + "liveresearchbench": { + "max_samples": 50, + "tools": ["think", "web_search"], + "temperature": 0.6, + }, + "deepresearchbench": { + "max_samples": 80, + "tools": ["think", "web_search"], + "temperature": 0.6, + }, +} + +DEFAULT_JUDGE_MODEL = "gpt-5-mini-2025-08-07" + + +def _template_path() -> Path: + return ( + Path(__file__).resolve().parents[1] + / "configs" + / "framework_comparison" + / "_template.toml" + ) + + +def _default_output_dir() -> Path: + return _template_path().parent + + +def materialize_config( + framework: str, + model: str, + benchmark: str, + output_dir: Optional[Path] = None, + judge_model: str = DEFAULT_JUDGE_MODEL, +) -> Path: + """Materialize one TOML config from the template.""" + if framework not in FRAMEWORKS: + raise ValueError(f"unknown framework {framework!r}; valid: {list(FRAMEWORKS)}") + if model not in MODELS: + raise ValueError(f"unknown model {model!r}; valid: {list(MODELS)}") + if benchmark not in BENCHMARKS: + raise ValueError(f"unknown benchmark {benchmark!r}; valid: {list(BENCHMARKS)}") + + fwk = FRAMEWORKS[framework] + mdl = MODELS[model] + bnk = BENCHMARKS[benchmark] + + template = _template_path().read_text() + sentinel = "# --- TEMPLATE BEGIN ---\n" + if sentinel in template: + template = template.split(sentinel, 1)[1] + rendered = ( + template.replace("{{benchmark}}", benchmark) + .replace("{{framework}}", framework) + .replace("{{framework_id}}", fwk["backend_id"]) + .replace("{{model_pretty}}", str(mdl["model_pretty"])) + .replace("{{model_id}}", str(mdl["model_id"])) + .replace("{{model_slug}}", model) + .replace("{{engine}}", str(mdl["engine"])) + .replace("{{num_gpus}}", str(mdl["num_gpus"])) + .replace("{{tools}}", json.dumps(bnk["tools"])) + .replace("{{max_samples}}", str(bnk["max_samples"])) + .replace("{{temperature}}", str(bnk["temperature"])) + .replace("{{max_tokens}}", str(mdl["max_tokens"])) + .replace("{{judge_model}}", judge_model) + ) + + out_dir = output_dir if output_dir is not None else _default_output_dir() + out_dir.mkdir(parents=True, exist_ok=True) + filename = f"{benchmark}-{framework}-{model}.toml" + out_path = out_dir / filename + out_path.write_text(rendered) + return out_path + + +@click.command() +@click.option("--framework", type=click.Choice(list(FRAMEWORKS))) +@click.option("--model", type=click.Choice(list(MODELS))) +@click.option("--benchmark", type=click.Choice(list(BENCHMARKS))) +@click.option( + "--output-dir", + type=click.Path(file_okay=False, path_type=Path), + default=None, +) +@click.option( + "--all-tier1", + is_flag=True, + help="Materialize the full configs grid (4 frameworks x 8 benchmarks x 3 models)", +) +def main( + framework: Optional[str], + model: Optional[str], + benchmark: Optional[str], + output_dir: Optional[Path], + all_tier1: bool, +) -> None: + """Generate eval config TOMLs for the framework-comparison experiment.""" + if all_tier1: + tier1_models = ["claude-opus-46", "qwen-9b", "qwen-122b"] + count = 0 + for f in FRAMEWORKS: + for m in tier1_models: + for b in BENCHMARKS: + p = materialize_config(f, m, b, output_dir) + click.echo(f" wrote {p.name}") + count += 1 + click.echo(f"\nWrote {count} configs.") + return + + if not framework or not model or not benchmark: + raise click.UsageError( + "Specify --framework + --model + --benchmark, or pass --all-tier1" + ) + p = materialize_config(framework, model, benchmark, output_dir) + click.echo(f"wrote {p}") + + +if __name__ == "__main__": + main() diff --git a/src/openjarvis/evals/comparison/table_gen.py b/src/openjarvis/evals/comparison/table_gen.py new file mode 100644 index 00000000..b224590a --- /dev/null +++ b/src/openjarvis/evals/comparison/table_gen.py @@ -0,0 +1,415 @@ +"""LaTeX table generator for the framework-comparison harness. + +Reads `summary.json` files produced by EvalRunner, builds a long-format +polars DataFrame, then renders 7 tables (T1..T7) as both `tabular` +fragments (paste-into-paper) and `\\documentclass{standalone}` previews +(latexmk-renderable). +""" + +from __future__ import annotations + +import glob +import json +import logging +from dataclasses import dataclass +from pathlib import Path +from typing import Dict, List, Optional, Tuple + +import click + +try: + import polars as pl +except ImportError as _polars_err: # pragma: no cover + raise ImportError( + "polars is required for the framework-comparison table generator. " + "Install with: uv sync --extra framework-comparison" + ) from _polars_err + +from pydantic import BaseModel, ValidationError + +LOGGER = logging.getLogger(__name__) + + +class TableGenError(Exception): + """Base for table-generation problems.""" + + +class MixedCommitError(TableGenError): + """Raised when one (framework, model, benchmark) cell has multiple commits.""" + + +class _MetricStats(BaseModel): + mean: float + std: float + n: int + + +class _SummarySchema(BaseModel): + framework: str + framework_commit: str + model: str + benchmark: str + n_tasks: int + metrics: Dict[str, _MetricStats] + + +@dataclass(slots=True) +class ResultsFrame: + """Long-format DataFrame of all loaded summary.json results.""" + + df: pl.DataFrame + unloadable_count: int = 0 + + +def load_results(glob_pattern: str) -> ResultsFrame: + """Glob summary.json files, validate, return long-format ResultsFrame.""" + paths = glob.glob(glob_pattern, recursive=True) + rows: List[Dict[str, object]] = [] + unloadable = 0 + + for p in paths: + try: + raw = json.loads(Path(p).read_text()) + schema = _SummarySchema.model_validate(raw) + except (json.JSONDecodeError, ValidationError) as e: + LOGGER.warning("skipping unloadable summary at %s: %s", p, e) + unloadable += 1 + continue + for metric_name, stats in schema.metrics.items(): + rows.append( + { + "framework": schema.framework, + "framework_commit": schema.framework_commit, + "model": schema.model, + "benchmark": schema.benchmark, + "metric_name": metric_name, + "mean": stats.mean, + "std": stats.std, + "n": stats.n, + "source_path": p, + } + ) + + if rows: + df = pl.DataFrame(rows) + else: + df = pl.DataFrame( + schema={ + "framework": pl.Utf8, + "framework_commit": pl.Utf8, + "model": pl.Utf8, + "benchmark": pl.Utf8, + "metric_name": pl.Utf8, + "mean": pl.Float64, + "std": pl.Float64, + "n": pl.Int64, + "source_path": pl.Utf8, + } + ) + + # Validate: each (framework, model, benchmark) cell must have one commit. + if not df.is_empty(): + commit_groups = df.group_by(["framework", "model", "benchmark"]).agg( + pl.col("framework_commit").unique().alias("commits") + ) + for row in commit_groups.iter_rows(named=True): + if len(row["commits"]) > 1: + # Sort commits for deterministic error message; polars' + # unique() does not guarantee insertion order across runs. + commits = sorted(row["commits"]) + raise MixedCommitError( + f"{row['framework']}/{row['model']}/{row['benchmark']}: " + f"multiple commits {commits}" + ) + + return ResultsFrame(df=df, unloadable_count=unloadable) + + +def _format_cell(value: Optional[float], precision: int = 2) -> str: + """Format a single numeric cell; em-dash for missing values.""" + if value is None or ( + isinstance(value, float) and value != value # NaN check + ): + return r"\textit{--}" + if isinstance(value, float): + return f"{value:.{precision}f}" + return str(value) + + +def _render_booktabs( + df: pl.DataFrame, + row_col: str, + caption: str, + label: str, + precision: int = 2, +) -> Tuple[str, str]: + """Render a polars DataFrame as a (fragment, standalone) LaTeX tuple. + + The first column is treated as the row label. All other columns are + numeric data cells, rendered with the given precision; None/NaN -> + em-dash. + """ + cols = df.columns + data_cols = [c for c in cols if c != row_col] + + lines: List[str] = [] + lines.append(r"\begin{tabular}{l" + "r" * len(data_cols) + "}") + lines.append(r"\toprule") + lines.append(" & ".join([row_col] + data_cols) + r" \\") + lines.append(r"\midrule") + for row in df.iter_rows(named=True): + cells = [str(row[row_col])] + for c in data_cols: + cells.append(_format_cell(row[c], precision=precision)) + lines.append(" & ".join(cells) + r" \\") + lines.append(r"\bottomrule") + lines.append(r"\end{tabular}") + fragment = "\n".join(lines) + + standalone = ( + "\\documentclass{standalone}\n" + "\\usepackage{booktabs}\n" + "\\begin{document}\n" + f"% caption: {caption} label: {label}\n" + fragment + "\n" + "\\end{document}\n" + ) + return fragment, standalone + + +T1_FRAMEWORKS_ORDER = [ + "openclaw", + "hermes", + "openjarvis", + "openjarvis-distilled", +] + + +def _build_t1(frame: ResultsFrame) -> Tuple[str, str]: + """T1: Portability triangulation. + + Rows: 4 frameworks (in T1_FRAMEWORKS_ORDER). + Cols: 8 benchmarks + Avg. + Cells: accuracy mean (across all models in cell, weighted equally). + """ + df = frame.df.filter(pl.col("metric_name") == "accuracy") + pivot = ( + df.group_by(["framework", "benchmark"]) + .agg(pl.col("mean").mean().alias("acc")) + .pivot(values="acc", index="framework", on="benchmark") + ) + bench_cols = [c for c in pivot.columns if c != "framework"] + if bench_cols: + pivot = pivot.with_columns( + pl.mean_horizontal(*[pl.col(c) for c in bench_cols]).alias("Avg") + ) + return _render_booktabs( + pivot, + row_col="framework", + caption="T1: Portability triangulation across frameworks (accuracy)", + label="tab:t1_portability", + precision=2, + ) + + +T2_METRICS = [ + ("latency_seconds", "Latency (s)"), + ("energy_joules_per_query", "Energy (J)"), + ("peak_power_w", "Power (W)"), + ("input_tokens_per_query", "In tok"), + ("output_tokens_per_query", "Out tok"), + ("cost_usd_per_query", "$/query"), +] + + +def _build_t2(frame: ResultsFrame) -> Tuple[str, str]: + """T2: Master efficiency comparison (mean across all benchmarks).""" + df = frame.df.filter(pl.col("metric_name").is_in([m for m, _ in T2_METRICS])) + pivot = ( + df.group_by(["framework", "model", "metric_name"]) + .agg(pl.col("mean").mean().alias("v")) + .pivot(values="v", index=["framework", "model"], on="metric_name") + ) + rename_map = {m: lbl for m, lbl in T2_METRICS if m in pivot.columns} + pivot = pivot.rename(rename_map) + pivot = pivot.with_columns( + (pl.col("framework") + " + " + pl.col("model")).alias("Configuration") + ).drop(["framework", "model"]) + ordered_cols = ["Configuration"] + [ + lbl for _, lbl in T2_METRICS if lbl in pivot.columns + ] + pivot = pivot.select(ordered_cols) + return _render_booktabs( + pivot, + row_col="Configuration", + caption="T2: Master efficiency comparison (per-query averages)", + label="tab:t2_efficiency", + precision=2, + ) + + +def _build_t3(frame: ResultsFrame) -> Tuple[str, str]: + """T3: Per-benchmark efficiency (one detail panel — pick PinchBench).""" + df = frame.df.filter(pl.col("benchmark") == "pinchbench") + df = df.filter( + pl.col("metric_name").is_in( + [ + "latency_seconds", + "energy_joules_per_query", + "input_tokens_per_query", + "output_tokens_per_query", + "accuracy", + "cost_usd_per_query", + ] + ) + ) + pivot = ( + df.group_by(["framework", "model", "metric_name"]) + .agg(pl.col("mean").mean().alias("v")) + .pivot(values="v", index=["framework", "model"], on="metric_name") + ) + pivot = pivot.with_columns( + (pl.col("framework") + " + " + pl.col("model")).alias("Config") + ).drop(["framework", "model"]) + return _render_booktabs( + pivot, + row_col="Config", + caption="T3: PinchBench per-config efficiency", + label="tab:t3_pb", + ) + + +def _build_t4(frame: ResultsFrame) -> Tuple[str, str]: + """T4: 4-framework x all-model accuracy matrix (avg across benchmarks).""" + df = frame.df.filter(pl.col("metric_name") == "accuracy") + pivot = ( + df.group_by(["model", "framework"]) + .agg(pl.col("mean").mean().alias("acc")) + .pivot(values="acc", index="model", on="framework") + ) + return _render_booktabs( + pivot, + row_col="model", + caption="T4: Per-model accuracy across frameworks", + label="tab:t4_per_model", + ) + + +def _build_t5(frame: ResultsFrame) -> Tuple[str, str]: + """T5: Hardware x framework efficiency (placeholder until per-hw data lands). + + Requires `hardware` field on the source summary.json (not yet wired into + metrics). Until that lands, falls back to T2's shape so the cell is non-empty. + """ + if "hardware" not in frame.df.columns: + return _render_booktabs( + pl.DataFrame({"Platform": ["(no data yet)"]}, schema={"Platform": pl.Utf8}), + row_col="Platform", + caption="T5: Hardware x framework efficiency (no data)", + label="tab:t5_hw", + ) + return _build_t2(frame) + + +def _build_t6(frame: ResultsFrame) -> Tuple[str, str]: + """T6: Token economy decomposition (preliminary; total in/out only). + + Spec §6.3 envisions per-section breakdown (system_prompt, tool_descs, + memory_rag, history, user_msg). Until that instrumentation lands, this + only shows total in/out tokens per framework. + """ + metrics = ["input_tokens_per_query", "output_tokens_per_query"] + df = frame.df.filter(pl.col("metric_name").is_in(metrics)) + pivot = ( + df.group_by(["framework", "metric_name"]) + .agg(pl.col("mean").mean().alias("v")) + .pivot(values="v", index="framework", on="metric_name") + ) + return _render_booktabs( + pivot, + row_col="framework", + caption="T6: Token economy (in/out per query)", + label="tab:t6_tokens", + ) + + +def _build_t7(frame: ResultsFrame) -> Tuple[str, str]: + """T7: Edit category x framework — preliminary (raw deltas). + + Spec §10 says richer edit-attribution data needs the spec-distillation + pipeline to tag accepted edits. For now, raw per-benchmark accuracy + deltas across frameworks. + """ + df = frame.df.filter(pl.col("metric_name") == "accuracy") + pivot = ( + df.group_by(["framework", "benchmark"]) + .agg(pl.col("mean").mean().alias("acc")) + .pivot(values="acc", index="framework", on="benchmark") + ) + return _render_booktabs( + pivot, + row_col="framework", + caption="T7: Edit category x framework (preliminary; raw deltas)", + label="tab:t7_edit", + ) + + +_TABLE_BUILDERS: Dict[str, "object"] = { + "T1": _build_t1, + "T2": _build_t2, + "T3": _build_t3, + "T4": _build_t4, + "T5": _build_t5, + "T6": _build_t6, + "T7": _build_t7, +} + + +def _table_gen_default_output_dir() -> Path: + return Path("experiments/framework_comparison/tables") + + +@click.command() +@click.option( + "--results-glob", + required=True, + help='Glob, e.g. "results/comparison/**/summary.json"', +) +@click.option( + "--tables", + default="T1,T2,T3,T4,T5,T6,T7", + help="Comma-separated list of table names to build", +) +@click.option( + "--output-dir", + type=click.Path(file_okay=False, path_type=Path), + default=None, + help="Output directory (default: experiments/framework_comparison/tables/)", +) +def main(results_glob: str, tables: str, output_dir: Optional[Path]) -> None: + """Build LaTeX tables from framework-comparison evaluation results.""" + out = output_dir or _table_gen_default_output_dir() + out.mkdir(parents=True, exist_ok=True) + (out / "preview").mkdir(parents=True, exist_ok=True) + + frame = load_results(results_glob) + click.echo( + f"Loaded {len(frame.df)} metric rows; {frame.unloadable_count} files skipped." + ) + + requested = [t.strip() for t in tables.split(",") if t.strip()] + for name in requested: + if name not in _TABLE_BUILDERS: + click.echo(f" ! unknown table {name}; skipping") + continue + try: + fragment, standalone = _TABLE_BUILDERS[name](frame) + except Exception as e: + click.echo(f" ! {name} build failed: {e}") + continue + (out / f"{name}.tex").write_text(fragment + "\n") + (out / "preview" / f"{name}_preview.tex").write_text(standalone) + click.echo(f" ✓ {name} → {out}/{name}.tex (+ preview)") + + +if __name__ == "__main__": + main() diff --git a/src/openjarvis/evals/comparison/third_party.py b/src/openjarvis/evals/comparison/third_party.py new file mode 100644 index 00000000..d51d1802 --- /dev/null +++ b/src/openjarvis/evals/comparison/third_party.py @@ -0,0 +1,170 @@ +"""Path resolution and commit-pin verification for foreign frameworks. + +The ``_third_party.toml`` file declares the canonical filesystem path and +expected git commit for each foreign framework (Hermes Agent, OpenClaw). +Env vars (``HERMES_AGENT_PATH``, ``OPENCLAW_PATH``) override the default path. +``JARVIS_ALLOW_COMMIT_DRIFT=1`` disables strict commit-pin enforcement. +""" + +from __future__ import annotations + +import logging +import os +import subprocess +from dataclasses import dataclass, field +from pathlib import Path +from typing import Dict, Optional + +import tomllib + +LOGGER = logging.getLogger(__name__) + +# Map framework name -> env var that overrides its `path` field. +# Must be kept in sync with the section keys in _third_party.toml. +_PATH_ENV_VAR = { + "hermes": "HERMES_AGENT_PATH", + "openclaw": "OPENCLAW_PATH", +} + +# Map framework name -> env var that overrides its language-runtime executable. +# For hermes this overrides `python_executable` (point at the Hermes venv's +# python so its deps don't need to be installed in OpenJarvis's venv); for +# openclaw it overrides `node_executable` (point at a Node ≥14.8 binary if +# the system default is too old). +_RUNTIME_ENV_VAR = { + "hermes": "HERMES_AGENT_PYTHON", + "openclaw": "OPENCLAW_NODE", +} + + +class ThirdPartyError(Exception): + """Base exception for third-party framework problems.""" + + +class ThirdPartyNotFoundError(ThirdPartyError): + """Raised when the third-party path cannot be resolved or doesn't exist.""" + + +class CommitDriftError(ThirdPartyError): + """Raised when the third-party repo's HEAD does not match the pinned commit.""" + + +@dataclass(slots=True) +class ThirdPartyEntry: + """One foreign framework's configuration.""" + + name: str + path: Path + pinned_commit: str + runner_script: str + python_executable: str = "" + node_executable: str = "" + + +@dataclass(slots=True) +class ThirdPartyConfig: + """Top-level config: a map from framework name -> entry.""" + + entries: Dict[str, ThirdPartyEntry] = field(default_factory=dict) + + +def _default_toml_path() -> Path: + """Return the in-repo default location of `_third_party.toml`.""" + return ( + Path(__file__).resolve().parents[1] + / "configs" + / "framework_comparison" + / "_third_party.toml" + ) + + +def load_third_party_config(toml_path: Optional[Path] = None) -> ThirdPartyConfig: + """Load the third-party config, applying env-var path overrides. + + Raises: + ThirdPartyNotFoundError: if the TOML doesn't exist. + """ + if toml_path is None: + toml_path = _default_toml_path() + if not toml_path.exists(): + raise ThirdPartyNotFoundError( + f"_third_party.toml not found at {toml_path}. " + "Either create it or pass an explicit toml_path." + ) + + with open(toml_path, "rb") as fh: + raw = tomllib.load(fh) + + entries: Dict[str, ThirdPartyEntry] = {} + for name, body in raw.items(): + path_env = _PATH_ENV_VAR.get(name) + path_str = os.environ.get(path_env, body["path"]) if path_env else body["path"] + runtime_env = _RUNTIME_ENV_VAR.get(name) + runtime_override = os.environ.get(runtime_env, "") if runtime_env else "" + # The runtime-env override takes precedence over the TOML default for + # whichever runtime field is relevant to this framework. + python_exe = body.get("python_executable", "") + node_exe = body.get("node_executable", "") + if runtime_override: + if name == "hermes": + python_exe = runtime_override + elif name == "openclaw": + node_exe = runtime_override + entries[name] = ThirdPartyEntry( + name=name, + path=Path(path_str), + pinned_commit=body.get("pinned_commit", ""), + runner_script=body.get("runner_script", ""), + python_executable=python_exe, + node_executable=node_exe, + ) + return ThirdPartyConfig(entries=entries) + + +def verify_commit_pin(entry: ThirdPartyEntry) -> None: + """Verify the entry's path is at the pinned commit. + + Raises: + ThirdPartyNotFoundError: if the path doesn't exist or isn't a git repo. + CommitDriftError: if HEAD doesn't match pinned_commit, unless + ``JARVIS_ALLOW_COMMIT_DRIFT=1`` is set (then logs a warning). + """ + env_var = _PATH_ENV_VAR.get(entry.name, "") + if str(entry.path) == "" or str(entry.path) == ".": + raise ThirdPartyNotFoundError( + f"{entry.name} path is not configured. " + f"Set the {env_var} environment variable to point at your " + f"local {entry.name} checkout, or set `path = ...` in " + f"_third_party.toml." + ) + if not entry.path.exists(): + raise ThirdPartyNotFoundError( + f"{entry.name} path does not exist: {entry.path}. " + f"Set {env_var} to override, or update _third_party.toml." + ) + + try: + result = subprocess.run( + ["git", "-C", str(entry.path), "rev-parse", "HEAD"], + capture_output=True, + text=True, + check=True, + ) + except subprocess.CalledProcessError as e: + raise ThirdPartyNotFoundError( + f"{entry.name} path is not a git repo: {entry.path} ({e.stderr.strip()})" + ) from e + + actual = result.stdout.strip() + pinned = entry.pinned_commit.strip() + # Allow short-hash matches (e.g. pinned "abc123" matches actual "abc123def...") + if not actual.startswith(pinned) and not pinned.startswith(actual): + msg = ( + f"{entry.name} commit drift: pinned={pinned}, actual={actual}. " + f"Either reset the repo to the pinned commit or update _third_party.toml. " + f"To bypass for this run, set JARVIS_ALLOW_COMMIT_DRIFT=1." + ) + if os.environ.get("JARVIS_ALLOW_COMMIT_DRIFT") == "1": + LOGGER.warning(msg) + return + raise CommitDriftError(msg) diff --git a/src/openjarvis/evals/configs/framework_comparison/_template.toml b/src/openjarvis/evals/configs/framework_comparison/_template.toml new file mode 100644 index 00000000..34f37a9b --- /dev/null +++ b/src/openjarvis/evals/configs/framework_comparison/_template.toml @@ -0,0 +1,49 @@ +# Jinja-style template for framework-comparison configs. +# +# Substitution variables (all required): +# e.g. "gaia" +# one of: openjarvis, hermes, openclaw, openjarvis-distilled +# backend id: jarvis-agent | hermes | openclaw +# e.g. "Qwen3.5-9B" +# e.g. "Qwen/Qwen3.5-9B" +# filename-safe slug, e.g. "qwen-9b" +# e.g. "vllm" | "cloud" +# int, default 1 +# TOML list literal +# int +# float +# int +# LLM judge id + +# --- TEMPLATE BEGIN --- +[meta] +name = "{{benchmark}}-{{framework}}-{{model_slug}}" +description = "{{benchmark}} on {{framework}} + {{model_pretty}}" +framework = "{{framework}}" +framework_id = "{{framework_id}}" + +[defaults] +temperature = {{temperature}} +max_tokens = {{max_tokens}} + +[judge] +model = "{{judge_model}}" +temperature = 0.0 +max_tokens = 4096 +engine = "cloud" + +[run] +max_workers = 1 +output_dir = "results/comparison/{{framework}}/{{model_slug}}/{{benchmark}}/" +seed = 42 + +[[models]] +name = "{{model_id}}" +engine = "{{engine}}" +num_gpus = {{num_gpus}} + +[[benchmarks]] +name = "{{benchmark}}" +backend = "{{framework_id}}" +max_samples = {{max_samples}} +tools = {{tools}} diff --git a/src/openjarvis/evals/configs/framework_comparison/_third_party.toml b/src/openjarvis/evals/configs/framework_comparison/_third_party.toml new file mode 100644 index 00000000..c46d34cd --- /dev/null +++ b/src/openjarvis/evals/configs/framework_comparison/_third_party.toml @@ -0,0 +1,27 @@ +# Foreign-framework paths + pinned commits. +# +# REQUIRED: set HERMES_AGENT_PATH and OPENCLAW_PATH env vars to point at +# your local checkouts. The empty `path = ""` defaults below are +# intentional — they require explicit user configuration. +# +# Each foreign framework must be runnable in its own right BEFORE the +# harness can drive it: +# - Hermes: install its Python deps (recommend its own venv, point +# `python_executable` at it). +# - OpenClaw: needs Node ≥14.8 AND `pnpm install && pnpm build` first. +# See docs/development/contributing.md for full setup instructions. +# +# To bypass commit-pin enforcement (for development against a different +# commit): set JARVIS_ALLOW_COMMIT_DRIFT=1. + +[hermes] +path = "" +pinned_commit = "5d3be898a" +runner_script = "src/openjarvis/evals/backends/external/_runners/hermes_runner.py" +python_executable = "" + +[openclaw] +path = "" +pinned_commit = "123ae82fca3009df7144fa8d41e1185cd63e61e1" +runner_script = "src/openjarvis/evals/backends/external/_runners/openclaw_runner.mjs" +node_executable = "" diff --git a/src/openjarvis/evals/core/backend.py b/src/openjarvis/evals/core/backend.py index 38c2feb3..6a4f3cc7 100644 --- a/src/openjarvis/evals/core/backend.py +++ b/src/openjarvis/evals/core/backend.py @@ -10,6 +10,16 @@ class InferenceBackend(ABC): """Base class for all inference backends used in evaluation.""" backend_id: str + # Default framework label used when an EvalResult is constructed in an + # error path before the backend's generate_full() has returned a payload + # to read ``framework`` from. Each subclass must override this so that + # error-path results are tagged with the correct framework name. + framework_name: str = "openjarvis" + + @property + def framework_commit_value(self) -> str: + """Default: empty (subclasses override). Used by runner.py error paths.""" + return "" @abstractmethod def generate( diff --git a/src/openjarvis/evals/core/config.py b/src/openjarvis/evals/core/config.py index aff39500..562ab932 100644 --- a/src/openjarvis/evals/core/config.py +++ b/src/openjarvis/evals/core/config.py @@ -3,6 +3,7 @@ from __future__ import annotations import logging +import os import sys from pathlib import Path from typing import List @@ -31,7 +32,13 @@ else: logger = logging.getLogger(__name__) -VALID_BACKENDS = {"jarvis-direct", "jarvis-agent", "terminalbench-native"} +VALID_BACKENDS = { + "jarvis-direct", + "jarvis-agent", + "terminalbench-native", + "hermes", + "openclaw", +} # Known benchmark names (used for warnings, not hard validation) KNOWN_BENCHMARKS = { @@ -134,9 +141,7 @@ def load_eval_config(path: str | Path) -> EvalSuiteConfig: sheets_spreadsheet_id=run_raw.get("sheets_spreadsheet_id", ""), sheets_worksheet=run_raw.get("sheets_worksheet", "Results"), sheets_credentials_path=run_raw.get("sheets_credentials_path", ""), - max_turns=( - int(run_raw["max_turns"]) if "max_turns" in run_raw else None - ), + max_turns=(int(run_raw["max_turns"]) if "max_turns" in run_raw else None), ) # Parse [[models]] @@ -209,6 +214,18 @@ def load_eval_config(path: str | Path) -> EvalSuiteConfig: ) ) + # Parse optional [backend.external] section (for hermes/openclaw backends). + # Env vars override TOML values; either source may be empty. + external_raw = raw.get("backend", {}).get("external", {}) + backend_external_base_url = ( + os.environ.get("JARVIS_BACKEND_BASE_URL") + or external_raw.get("base_url") + or None + ) + backend_external_api_key = ( + os.environ.get("JARVIS_BACKEND_API_KEY") or external_raw.get("api_key") or None + ) + return EvalSuiteConfig( meta=meta, defaults=defaults, @@ -216,6 +233,8 @@ def load_eval_config(path: str | Path) -> EvalSuiteConfig: run=execution, models=models, benchmarks=benchmarks, + backend_external_base_url=backend_external_base_url, + backend_external_api_key=backend_external_api_key, ) @@ -305,6 +324,8 @@ def expand_suite(suite: EvalSuiteConfig) -> List[RunConfig]: sheets_worksheet=suite.run.sheets_worksheet, sheets_credentials_path=suite.run.sheets_credentials_path, max_turns=suite.run.max_turns, + base_url=suite.backend_external_base_url, + api_key=suite.backend_external_api_key, record_ids=bench.record_ids, ) ) diff --git a/src/openjarvis/evals/core/export.py b/src/openjarvis/evals/core/export.py index 69780379..4b3fd0cd 100644 --- a/src/openjarvis/evals/core/export.py +++ b/src/openjarvis/evals/core/export.py @@ -375,6 +375,73 @@ def export_summary_json( if bench_energy is not None: summary["bench_telemetry"] = bench_energy + # ---- Spec §6.3: table_gen-compatible flat schema ---- + # Emit framework / framework_commit / model / benchmark / n_tasks / + # metrics at the top level so the framework-comparison `table_gen` + # loader (`_SummarySchema`) can parse this file. The existing rich + # schema is preserved untouched; this is purely additive. + fwk = "" + fwk_commit = "" + if isinstance(config, dict): + fwk = config.get("framework", "") or "" + fwk_commit = config.get("framework_commit", "") or "" + if not fwk: + fwk = "openjarvis" + + def _stats_block(vals: list[float]) -> dict[str, Any]: + if not vals: + return {"mean": 0.0, "std": 0.0, "n": 0} + return { + "mean": float(statistics.fmean(vals)), + "std": (float(statistics.stdev(vals)) if len(vals) > 1 else 0.0), + "n": len(vals), + } + + accuracy_vals: list[float] = [ + 1.0 if t.is_resolved is True else 0.0 + for t in traces + if t.is_resolved is not None + ] + latency_vals = [t.total_wall_clock_s for t in traces if t.total_wall_clock_s > 0] + energy_vals = [ + t.total_gpu_energy_joules + for t in traces + if t.total_gpu_energy_joules is not None and t.total_gpu_energy_joules > 0 + ] + in_tok_vals = [ + float(t.total_input_tokens) for t in traces if t.total_input_tokens > 0 + ] + out_tok_vals = [ + float(t.total_output_tokens) for t in traces if t.total_output_tokens > 0 + ] + cost_vals = [ + t.total_cost_usd + for t in traces + if t.total_cost_usd is not None and t.total_cost_usd > 0 + ] + power_vals = [ + t.avg_gpu_power_watts + for t in traces + if t.avg_gpu_power_watts is not None and t.avg_gpu_power_watts > 0 + ] + + summary["framework"] = fwk + summary["framework_commit"] = fwk_commit + summary["model"] = config.get("model", "") if isinstance(config, dict) else "" + summary["benchmark"] = ( + config.get("benchmark", "") if isinstance(config, dict) else "" + ) + summary["n_tasks"] = len(traces) + summary["metrics"] = { + "accuracy": _stats_block(accuracy_vals), + "latency_seconds": _stats_block(latency_vals), + "energy_joules_per_query": _stats_block(energy_vals), + "input_tokens_per_query": _stats_block(in_tok_vals), + "output_tokens_per_query": _stats_block(out_tok_vals), + "cost_usd_per_query": _stats_block(cost_vals), + "peak_power_w": _stats_block(power_vals), + } + path.parent.mkdir(parents=True, exist_ok=True) path.write_text(json.dumps(summary, indent=2, default=str)) return path diff --git a/src/openjarvis/evals/core/runner.py b/src/openjarvis/evals/core/runner.py index d19b7551..851a8c8e 100644 --- a/src/openjarvis/evals/core/runner.py +++ b/src/openjarvis/evals/core/runner.py @@ -273,7 +273,11 @@ class EvalRunner: if output_path: summary_path = output_path.with_suffix(".summary.json") with open(summary_path, "w") as f: - json.dump(_summary_to_dict(summary), f, indent=2) + json.dump( + _summary_to_dict(summary, results=self._results), + f, + indent=2, + ) LOGGER.info("Results written to %s", output_path) LOGGER.info("Summary written to %s", summary_path) @@ -369,9 +373,13 @@ class EvalRunner: latency = full.get("latency_seconds", 0.0) cost = full.get("cost_usd", 0.0) - energy_j = full.get("energy_joules", 0.0) - power_w = full.get("power_watts", 0.0) - throughput = full.get("throughput_tok_per_sec", 0.0) + # Coerce None -> 0.0: backends may emit None when telemetry is + # unavailable (e.g., HermesBackend when no GPU sampler). The + # 'or 0.0' handles the case where the key is present but the + # value is None -- .get(default) only kicks in when missing. + energy_j = full.get("energy_joules", 0.0) or 0.0 + power_w = full.get("power_watts", 0.0) or 0.0 + throughput = full.get("throughput_tok_per_sec", 0.0) or 0.0 accuracy_score = 1.0 if is_correct else 0.0 # Compute IPW and IPJ @@ -418,9 +426,11 @@ class EvalRunner: # Extract derived and ITL metrics from _telemetry dict _telem = full.get("_telemetry", {}) - energy_per_out_tok = _telem.get("energy_per_output_token_joules", 0.0) - throughput_per_w = _telem.get("throughput_per_watt", 0.0) - mean_itl = _telem.get("mean_itl_ms", 0.0) + energy_per_out_tok = ( + _telem.get("energy_per_output_token_joules", 0.0) or 0.0 + ) + throughput_per_w = _telem.get("throughput_per_watt", 0.0) or 0.0 + mean_itl = _telem.get("mean_itl_ms", 0.0) or 0.0 # Prefer continuous score from scoring_meta when scorer provides it. score_val = _extract_continuous_score(scoring_meta, is_correct) @@ -434,10 +444,10 @@ class EvalRunner: completion_tokens=usage.get("completion_tokens", 0), cost_usd=cost, scoring_metadata=scoring_meta, - ttft=full.get("ttft", 0.0), + ttft=full.get("ttft", 0.0) or 0.0, energy_joules=energy_j, power_watts=power_w, - gpu_utilization_pct=full.get("gpu_utilization_pct", 0.0), + gpu_utilization_pct=full.get("gpu_utilization_pct", 0.0) or 0.0, throughput_tok_per_sec=throughput, mfu_pct=mfu, mbu_pct=mbu, @@ -448,6 +458,11 @@ class EvalRunner: mean_itl_ms=mean_itl, estimated_flops=estimated_flops, trace_data=full.get("trace_data"), + # Spec §6.2 extended fields for cross-framework comparison + framework=full.get("framework", "openjarvis"), + framework_commit=full.get("framework_commit", ""), + tool_calls=int(full.get("tool_calls", 0)), + turn_count=int(full.get("turn_count", 0)), ) except Exception as exc: LOGGER.error("Error processing %s: %s", record.record_id, exc) @@ -455,6 +470,11 @@ class EvalRunner: record_id=record.record_id, model_answer="", error=str(exc), + framework=getattr(self._backend, "framework_name", "openjarvis"), + framework_commit=getattr(self._backend, "framework_commit_value", "") + or "", + tool_calls=0, + turn_count=0, ) # ------------------------------------------------------------------ @@ -731,6 +751,13 @@ class EvalRunner: cost_usd=total_cost, scoring_metadata=scoring_meta, trace_data=full.get("trace_data"), + # Spec §6.2 extended fields. framework/commit are constant + # per backend, so the last turn's value is correct. tool_calls + # / turn_count come from the interactive loop itself. + framework=full.get("framework", "openjarvis"), + framework_commit=full.get("framework_commit", ""), + tool_calls=int(full.get("tool_calls", 0)), + turn_count=len(all_responses), ) except Exception as exc: LOGGER.error( @@ -743,6 +770,11 @@ class EvalRunner: model_answer="", error=str(exc), scoring_metadata={"interactive": True, "error": str(exc)}, + framework=getattr(self._backend, "framework_name", "openjarvis"), + framework_commit=getattr(self._backend, "framework_commit_value", "") + or "", + tool_calls=0, + turn_count=0, ) finally: if env is not None: @@ -1125,9 +1157,18 @@ def _metric_stats_to_dict(ms: Optional[MetricStats]) -> Optional[Dict[str, float } -def _summary_to_dict(s: RunSummary) -> Dict[str, Any]: - """Convert a RunSummary to a JSON-serializable dict.""" - return { +def _summary_to_dict( + s: RunSummary, + results: Optional[List[EvalResult]] = None, +) -> Dict[str, Any]: + """Convert a RunSummary to a JSON-serializable dict. + + When ``results`` is provided, the dict ALSO carries the spec §6.3 + ``framework`` / ``framework_commit`` / ``n_tasks`` / ``metrics`` block + expected by the framework-comparison ``table_gen`` loader. Existing + top-level keys are preserved (this is purely additive). + """ + d = { "hardware_info": _hardware_info_dict(), "benchmark": s.benchmark, "category": s.category, @@ -1200,6 +1241,66 @@ def _summary_to_dict(s: RunSummary) -> Dict[str, Any]: }, } + # ---- Spec §6.3: table_gen-compatible flat schema ---- + # In addition to the existing rich schema, emit framework / commit / + # per-metric stats so framework-comparison `table_gen.load_results` + # can parse this file as a `_SummarySchema` row. + fwk = "openjarvis" + fwk_commit = "" + if results: + for r in results: + if getattr(r, "framework", None): + fwk = r.framework + fwk_commit = r.framework_commit or "" + break + + def _stats_block(vals: List[float]) -> Dict[str, Any]: + if not vals: + return {"mean": 0.0, "std": 0.0, "n": 0} + return { + "mean": float(statistics.fmean(vals)), + "std": (float(statistics.stdev(vals)) if len(vals) > 1 else 0.0), + "n": len(vals), + } + + if results is not None: + scored = [r for r in results if r.is_correct is not None] + accuracy_vals = [1.0 if r.is_correct else 0.0 for r in scored] + latency_vals = [r.latency_seconds for r in results if r.latency_seconds > 0] + energy_vals = [r.energy_joules for r in results if r.energy_joules > 0] + in_tok_vals = [float(r.prompt_tokens) for r in results if r.prompt_tokens > 0] + out_tok_vals = [ + float(r.completion_tokens) for r in results if r.completion_tokens > 0 + ] + cost_vals = [r.cost_usd for r in results if r.cost_usd > 0] + power_vals = [r.power_watts for r in results if r.power_watts > 0] + n_tasks = len(results) + else: + accuracy_vals = [] + latency_vals = [] + energy_vals = [] + in_tok_vals = [] + out_tok_vals = [] + cost_vals = [] + power_vals = [] + n_tasks = 0 + + d["framework"] = fwk + d["framework_commit"] = fwk_commit + # `model` and `benchmark` are already top-level above; keep them. + d["n_tasks"] = n_tasks + d["metrics"] = { + "accuracy": _stats_block(accuracy_vals), + "latency_seconds": _stats_block(latency_vals), + "energy_joules_per_query": _stats_block(energy_vals), + "input_tokens_per_query": _stats_block(in_tok_vals), + "output_tokens_per_query": _stats_block(out_tok_vals), + "cost_usd_per_query": _stats_block(cost_vals), + "peak_power_w": _stats_block(power_vals), + } + + return d + def _result_to_trace_dict(result: EvalResult) -> Dict[str, Any]: """Convert an EvalResult to a full trace dict for per-sample export.""" @@ -1227,6 +1328,11 @@ def _result_to_trace_dict(result: EvalResult) -> Dict[str, Any]: "throughput_per_watt": result.throughput_per_watt, "mean_itl_ms": result.mean_itl_ms, "estimated_flops": result.estimated_flops, + # Spec §6.2 cross-framework fields + "framework": result.framework, + "framework_commit": result.framework_commit, + "tool_calls": result.tool_calls, + "turn_count": result.turn_count, } if result.trace_data is not None: d["trace_data"] = result.trace_data diff --git a/src/openjarvis/evals/core/types.py b/src/openjarvis/evals/core/types.py index 5c40f933..0050a6b7 100644 --- a/src/openjarvis/evals/core/types.py +++ b/src/openjarvis/evals/core/types.py @@ -49,6 +49,11 @@ class EvalResult: trace_steps: int = 0 trace_energy_joules: float = 0.0 trace_data: Optional[Dict[str, Any]] = None + # Spec §6.2 extended fields for cross-framework comparison + framework: str = "openjarvis" + framework_commit: str = "" + tool_calls: int = 0 + turn_count: int = 0 @dataclass(slots=True) @@ -89,6 +94,9 @@ class RunConfig: # running thinking/reasoning models that consume turns on intermediate # reasoning before producing tool calls. max_turns: Optional[int] = None + # Spec §6.2: backend-external endpoint config (for hermes/openclaw) + base_url: Optional[str] = None + api_key: Optional[str] = None # Filter dataset records by id. When set, only records whose record_id # appears in this list are processed. Used for surgical re-runs of # specific records (e.g. recovering silent-fake records without @@ -269,6 +277,10 @@ class EvalSuiteConfig: run: ExecutionConfig = field(default_factory=ExecutionConfig) models: List[ModelConfig] = field(default_factory=list) benchmarks: List[BenchmarkConfig] = field(default_factory=list) + # Spec §6.2: optional [backend.external] section for hermes/openclaw + # backends. Holds the OpenAI-compatible endpoint config. + backend_external_base_url: Optional[str] = None + backend_external_api_key: Optional[str] = None __all__ = [ diff --git a/tests/agents/test_native_openhands.py b/tests/agents/test_native_openhands.py index c4fdc36b..f42113b7 100644 --- a/tests/agents/test_native_openhands.py +++ b/tests/agents/test_native_openhands.py @@ -379,15 +379,22 @@ class TestNativeOpenHandsAgent: result = agent.run("") assert result.content == "Empty input received." - def test_error_400_handling(self): - """Agent catches 400 errors and returns friendly message.""" + def test_error_400_propagation(self): + """Engine errors propagate so the eval runner records a real failure. + + Updated from earlier behavior (which swallowed 4xx errors and + returned a "context too long" placeholder string) — see PR #303 + for the design rationale: silent fakes were scoring 0 without + surfacing the underlying issue. + """ + import pytest + engine = MagicMock() engine.engine_id = "mock" engine.generate.side_effect = RuntimeError("HTTP 400 Bad Request") agent = NativeOpenHandsAgent(engine, "test-model") - result = agent.run("Hello") - assert "too long" in result.content - assert result.metadata.get("error") is True + with pytest.raises(RuntimeError, match="HTTP 400"): + agent.run("Hello") def test_xml_tool_call_extraction(self): """Agent parses XML-style tool calls.""" diff --git a/tests/cli/test_eval_cmd_external_flags.py b/tests/cli/test_eval_cmd_external_flags.py new file mode 100644 index 00000000..5fdc5e58 --- /dev/null +++ b/tests/cli/test_eval_cmd_external_flags.py @@ -0,0 +1,26 @@ +"""Verify jarvis eval run exposes --base-url / --api-key for hermes/openclaw.""" + +from __future__ import annotations + +from click.testing import CliRunner + + +class TestEvalCmdExternalFlags: + def test_help_lists_external_backend_flags(self) -> None: + from openjarvis.cli.eval_cmd import eval_run + + runner = CliRunner() + result = runner.invoke(eval_run, ["--help"]) + assert result.exit_code == 0 + assert "--base-url" in result.output + assert "--api-key" in result.output + assert "hermes" in result.output + + def test_help_backend_choices_include_hermes_openclaw(self) -> None: + from openjarvis.cli.eval_cmd import eval_run + + runner = CliRunner() + result = runner.invoke(eval_run, ["--help"]) + assert result.exit_code == 0 + assert "hermes" in result.output + assert "openclaw" in result.output diff --git a/tests/evals/comparison/__init__.py b/tests/evals/comparison/__init__.py new file mode 100644 index 00000000..789b45ff --- /dev/null +++ b/tests/evals/comparison/__init__.py @@ -0,0 +1 @@ +"""Tests for openjarvis.evals.comparison.""" diff --git a/tests/evals/comparison/test_cli_backend_dispatch.py b/tests/evals/comparison/test_cli_backend_dispatch.py new file mode 100644 index 00000000..dc39eb40 --- /dev/null +++ b/tests/evals/comparison/test_cli_backend_dispatch.py @@ -0,0 +1,85 @@ +"""Verify cli._build_backend correctly dispatches hermes/openclaw.""" + +from __future__ import annotations + +from pathlib import Path +from unittest.mock import patch + +import click +import pytest + +from openjarvis.evals.cli import BACKENDS, _build_backend + + +class TestBuildBackendDispatch: + def test_backends_dict_includes_external(self) -> None: + assert "hermes" in BACKENDS + assert "openclaw" in BACKENDS + assert "jarvis-agent" in BACKENDS + assert "jarvis-direct" in BACKENDS + + def test_unknown_backend_raises(self) -> None: + with pytest.raises(click.UsageError, match="unknown backend"): + _build_backend( + backend_name="not-real", + engine_key=None, + agent_name="o", + tools=[], + ) + + def test_hermes_requires_base_url(self) -> None: + with pytest.raises(click.UsageError, match="--base-url"): + _build_backend( + backend_name="hermes", + engine_key=None, + agent_name="o", + tools=[], + base_url=None, + api_key="k", + ) + + def test_openclaw_requires_api_key(self) -> None: + with pytest.raises(click.UsageError, match="--api-key"): + _build_backend( + backend_name="openclaw", + engine_key=None, + agent_name="o", + tools=[], + base_url="http://x", + api_key=None, + ) + + def test_hermes_returns_hermes_backend(self) -> None: + from openjarvis.evals.comparison.third_party import ( + ThirdPartyConfig, + ThirdPartyEntry, + ) + + cfg = ThirdPartyConfig( + entries={ + "hermes": ThirdPartyEntry( + name="hermes", + path=Path("/tmp"), + pinned_commit="abc", + runner_script="x", + ) + } + ) + with ( + patch( + "openjarvis.evals.backends.external.hermes_agent.load_third_party_config", + return_value=cfg, + ), + patch("openjarvis.evals.backends.external.hermes_agent.verify_commit_pin"), + ): + from openjarvis.evals.backends.external import HermesBackend + + backend = _build_backend( + backend_name="hermes", + engine_key=None, + agent_name="o", + tools=[], + base_url="http://x", + api_key="k", + ) + assert isinstance(backend, HermesBackend) diff --git a/tests/evals/comparison/test_export_to_table_gen_roundtrip.py b/tests/evals/comparison/test_export_to_table_gen_roundtrip.py new file mode 100644 index 00000000..fbe0c378 --- /dev/null +++ b/tests/evals/comparison/test_export_to_table_gen_roundtrip.py @@ -0,0 +1,245 @@ +"""Round-trip test: EvalRunner output -> table_gen reads it correctly. + +Verifies that the schema emitted by `_summary_to_dict` (regular EvalRunner +path used by hermes/openclaw) and `export_summary_json` (agentic-runner +path) is consumable by `table_gen.load_results` without being skipped as +'unloadable'. + +This was the critical gap caught in final review: components that worked +in isolation but didn't connect end-to-end. +""" + +from __future__ import annotations + +import json +from pathlib import Path + +from openjarvis.evals.comparison.table_gen import ( + _build_t1, + load_results, +) + + +class TestExportToTableGenRoundtrip: + def test_summary_json_loads_via_table_gen(self, tmp_path: Path) -> None: + """Construct a summary.json mimicking _summary_to_dict output; + verify table_gen.load_results parses it without skipping.""" + summary = { + # Existing rich schema (unchanged) + "hardware_info": {"gpu": "H100"}, + "benchmark": "gaia", + "model": "Qwen/Qwen3.5-9B", + "backend": "hermes", + "category": "agentic", + "total_samples": 50, + "scored_samples": 50, + "correct": 21, + "accuracy": 0.42, + "errors": 0, + "mean_latency_seconds": 23.4, + "total_cost_usd": 0.7, + # New §6.3-compatible fields + "framework": "hermes", + "framework_commit": "5d3be898a", + "n_tasks": 50, + "metrics": { + "accuracy": {"mean": 0.42, "std": 0.04, "n": 50}, + "latency_seconds": {"mean": 23.4, "std": 5.1, "n": 50}, + "energy_joules_per_query": { + "mean": 1234.5, + "std": 90, + "n": 50, + }, + "input_tokens_per_query": { + "mean": 5432, + "std": 200, + "n": 50, + }, + "output_tokens_per_query": { + "mean": 845, + "std": 50, + "n": 50, + }, + "cost_usd_per_query": { + "mean": 0.014, + "std": 0.002, + "n": 50, + }, + }, + } + path = tmp_path / "summary.json" + path.write_text(json.dumps(summary)) + + frame = load_results(str(tmp_path / "**" / "summary.json")) + # Should NOT be skipped as unloadable + assert frame.unloadable_count == 0 + # Should have one row per metric (6 metrics here) + assert len(frame.df) == 6 + # Verify framework/commit propagated + assert frame.df["framework"].to_list()[0] == "hermes" + assert frame.df["framework_commit"].to_list()[0] == "5d3be898a" + + def test_table_gen_builds_t1_from_export_schema(self, tmp_path: Path) -> None: + """T1 builder produces non-empty LaTeX from realistic schema.""" + for fwk, acc in [("hermes", 0.30), ("openjarvis", 0.45)]: + summary = { + "framework": fwk, + "framework_commit": "abc" if fwk == "hermes" else "def", + "model": "Qwen/Qwen3.5-9B", + "benchmark": "gaia", + "n_tasks": 50, + "metrics": { + "accuracy": {"mean": acc, "std": 0.04, "n": 50}, + }, + } + (tmp_path / fwk).mkdir() + (tmp_path / fwk / "summary.json").write_text(json.dumps(summary)) + + frame = load_results(str(tmp_path / "**" / "summary.json")) + fragment, _ = _build_t1(frame) + assert "\\begin{tabular}" in fragment + assert "0.30" in fragment + assert "0.45" in fragment + + +class TestSummaryToDictEmitsRequiredFields: + """Verify `_summary_to_dict` emits the §6.3 fields when given results. + + `_summary_to_dict` is the dict-builder used by EvalRunner to write + `.summary.json` for non-agentic runs (the path hermes/openclaw take). + """ + + def test_summary_to_dict_includes_table_gen_fields(self) -> None: + from openjarvis.evals.core.runner import _summary_to_dict + from openjarvis.evals.core.types import EvalResult, RunSummary + + results = [ + EvalResult( + record_id=f"q{i}", + model_answer=f"answer-{i}", + latency_seconds=2.0 + i * 0.1, + prompt_tokens=100, + completion_tokens=50, + cost_usd=0.001, + energy_joules=10.0, + power_watts=20.0, + framework="hermes", + framework_commit="5d3be898a", + tool_calls=2, + turn_count=3, + is_correct=(i % 2 == 0), + ) + for i in range(5) + ] + + summary = RunSummary( + benchmark="gaia", + category="agentic", + backend="hermes", + model="Qwen/Qwen3.5-9B", + total_samples=5, + scored_samples=5, + correct=3, + accuracy=0.6, + errors=0, + mean_latency_seconds=2.2, + total_cost_usd=0.005, + ) + + d = _summary_to_dict(summary, results=results) + + # New §6.3-compatible top-level fields + assert d["framework"] == "hermes" + assert d["framework_commit"] == "5d3be898a" + assert d["model"] == "Qwen/Qwen3.5-9B" + assert d["benchmark"] == "gaia" + assert d["n_tasks"] == 5 + assert "accuracy" in d["metrics"] + assert "latency_seconds" in d["metrics"] + assert d["metrics"]["accuracy"]["n"] == 5 + assert d["metrics"]["latency_seconds"]["n"] == 5 + # Existing rich schema also present (unchanged) + assert "hardware_info" in d + assert "telemetry_summary" in d + + def test_summary_to_dict_without_results_still_works(self) -> None: + """Backward compat: calling without ``results`` must not error.""" + from openjarvis.evals.core.runner import _summary_to_dict + from openjarvis.evals.core.types import RunSummary + + summary = RunSummary( + benchmark="gaia", + category="agentic", + backend="hermes", + model="m", + total_samples=0, + scored_samples=0, + correct=0, + accuracy=0.0, + errors=0, + mean_latency_seconds=0.0, + total_cost_usd=0.0, + ) + + d = _summary_to_dict(summary) + # Still emits the §6.3 keys (defaults), so the schema is stable. + assert d["framework"] == "openjarvis" + assert d["n_tasks"] == 0 + assert d["metrics"]["accuracy"] == {"mean": 0.0, "std": 0.0, "n": 0} + + +class TestExportSummaryJsonEmitsRequiredFields: + """Verify `export_summary_json` (the agentic-runner path) also emits + the §6.3 fields, sourced from the ``config`` dict argument.""" + + def test_export_summary_includes_table_gen_fields(self, tmp_path: Path) -> None: + from openjarvis.evals.core.export import export_summary_json + from openjarvis.evals.core.trace import QueryTrace, TurnTrace + + # Build minimal traces with enough fields populated so the + # statistics blocks are non-empty. + traces = [] + for i in range(5): + turn = TurnTrace( + turn_index=0, + input_tokens=100, + output_tokens=50, + cost_usd=0.001, + gpu_energy_joules=10.0, + gpu_power_avg_watts=20.0, + ) + t = QueryTrace( + query_id=f"q{i}", + workload_type="agentic", + turns=[turn], + total_wall_clock_s=2.0 + i * 0.1, + completed=True, + is_resolved=(i % 2 == 0), + ) + traces.append(t) + + out_path = tmp_path / "summary.json" + export_summary_json( + traces, + config={ + "model": "Qwen/Qwen3.5-9B", + "benchmark": "gaia", + "framework": "hermes", + "framework_commit": "5d3be898a", + }, + path=out_path, + ) + + data = json.loads(out_path.read_text()) + # New §6.3-compatible fields + assert data["framework"] == "hermes" + assert data["framework_commit"] == "5d3be898a" + assert data["model"] == "Qwen/Qwen3.5-9B" + assert data["benchmark"] == "gaia" + assert data["n_tasks"] == 5 + assert "accuracy" in data["metrics"] + assert "latency_seconds" in data["metrics"] + assert data["metrics"]["accuracy"]["n"] > 0 + # Existing rich schema also present (unchanged) + assert "generated_at" in data + assert "totals" in data diff --git a/tests/evals/comparison/test_hermes_backend.py b/tests/evals/comparison/test_hermes_backend.py new file mode 100644 index 00000000..7fa60897 --- /dev/null +++ b/tests/evals/comparison/test_hermes_backend.py @@ -0,0 +1,191 @@ +"""Tests for openjarvis.evals.backends.external.hermes_agent.HermesBackend.""" + +from __future__ import annotations + +from pathlib import Path +from unittest.mock import patch + +import pytest + +from openjarvis.evals.backends.external._subprocess_runner import SubprocessResult +from openjarvis.evals.backends.external.hermes_agent import HermesBackend +from openjarvis.evals.comparison.third_party import ( + CommitDriftError, + ThirdPartyConfig, + ThirdPartyEntry, +) + + +def _fake_third_party(tmp_path: Path) -> ThirdPartyConfig: + return ThirdPartyConfig( + entries={ + "hermes": ThirdPartyEntry( + name="hermes", + path=tmp_path, + pinned_commit="abc123", + runner_script=( + "src/openjarvis/evals/backends/external/_runners/hermes_runner.py" + ), + python_executable="", + ) + } + ) + + +class TestHermesBackend: + def test_generate_full_builds_correct_subprocess_command( + self, tmp_path: Path + ) -> None: + cfg = _fake_third_party(tmp_path) + with ( + patch( + "openjarvis.evals.backends.external.hermes_agent.load_third_party_config", + return_value=cfg, + ), + patch( + "openjarvis.evals.backends.external.hermes_agent.verify_commit_pin", + ), + patch( + "openjarvis.evals.backends.external.hermes_agent.run_one_shot", + ) as mock_run, + ): + mock_run.return_value = SubprocessResult( + stdout="", + stderr="", + exit_code=0, + latency_seconds=1.0, + energy_joules=10.0, + peak_power_w=5.0, + sampler_method="nvml", + parsed_json={ + "content": "answer", + "usage": { + "prompt_tokens": 100, + "completion_tokens": 50, + "total_tokens": 150, + }, + "trajectory": [], + "tool_calls": 2, + "turn_count": 3, + "error": None, + }, + error=None, + ) + backend = HermesBackend(base_url="http://x", api_key="k") + backend.generate_full( + "task", + model="qwen-9b", + system="", + temperature=0.0, + max_tokens=2048, + ) + + cmd = mock_run.call_args.kwargs["cmd"] + assert "--task" in cmd and "task" in cmd + assert "--model" in cmd and "qwen-9b" in cmd + assert "--base-url" in cmd and "http://x" in cmd + assert "--api-key" in cmd and "k" in cmd + + def test_generate_full_returns_extended_dict(self, tmp_path: Path) -> None: + cfg = _fake_third_party(tmp_path) + with ( + patch( + "openjarvis.evals.backends.external.hermes_agent.load_third_party_config", + return_value=cfg, + ), + patch( + "openjarvis.evals.backends.external.hermes_agent.verify_commit_pin", + ), + patch( + "openjarvis.evals.backends.external.hermes_agent.run_one_shot", + ) as mock_run, + ): + mock_run.return_value = SubprocessResult( + stdout="", + stderr="", + exit_code=0, + latency_seconds=2.5, + energy_joules=42.0, + peak_power_w=20.0, + sampler_method="nvml", + parsed_json={ + "content": "answer", + "usage": { + "prompt_tokens": 100, + "completion_tokens": 50, + "total_tokens": 150, + }, + "trajectory": [], + "tool_calls": 2, + "turn_count": 3, + "error": None, + }, + error=None, + ) + backend = HermesBackend(base_url="http://x", api_key="k") + result = backend.generate_full( + "task", + model="qwen-9b", + system="", + temperature=0.0, + max_tokens=2048, + ) + assert result["content"] == "answer" + assert result["energy_joules"] == 42.0 + assert result["peak_power_w"] == 20.0 + assert result["tool_calls"] == 2 + assert result["turn_count"] == 3 + assert result["framework"] == "hermes" + assert result["framework_commit"] == "abc123" + assert result["error"] is None + + def test_generate_full_propagates_subprocess_error(self, tmp_path: Path) -> None: + cfg = _fake_third_party(tmp_path) + with ( + patch( + "openjarvis.evals.backends.external.hermes_agent.load_third_party_config", + return_value=cfg, + ), + patch( + "openjarvis.evals.backends.external.hermes_agent.verify_commit_pin", + ), + patch( + "openjarvis.evals.backends.external.hermes_agent.run_one_shot", + ) as mock_run, + ): + mock_run.return_value = SubprocessResult( + stdout="", + stderr="boom", + exit_code=139, + latency_seconds=0.5, + energy_joules=None, + peak_power_w=None, + sampler_method="unavailable", + parsed_json={}, + error="subprocess_crash", + ) + backend = HermesBackend(base_url="http://x", api_key="k") + result = backend.generate_full( + "task", + model="qwen-9b", + system="", + temperature=0.0, + max_tokens=2048, + ) + assert result["error"] == "subprocess_crash" + assert result["content"] == "" + + def test_init_raises_on_commit_drift(self, tmp_path: Path) -> None: + cfg = _fake_third_party(tmp_path) + with ( + patch( + "openjarvis.evals.backends.external.hermes_agent.load_third_party_config", + return_value=cfg, + ), + patch( + "openjarvis.evals.backends.external.hermes_agent.verify_commit_pin", + side_effect=CommitDriftError("drift!"), + ), + ): + with pytest.raises(CommitDriftError, match="drift"): + HermesBackend(base_url="http://x", api_key="k") diff --git a/tests/evals/comparison/test_live_external.py b/tests/evals/comparison/test_live_external.py new file mode 100644 index 00000000..ee109f18 --- /dev/null +++ b/tests/evals/comparison/test_live_external.py @@ -0,0 +1,111 @@ +"""Integration tests that spawn real Hermes/OpenClaw subprocesses. + +Requires: +- $HERMES_AGENT_PATH set and pointing to the pinned commit +- $OPENCLAW_PATH set and pointing to the pinned commit +- An OpenAI-compatible mock server reachable at $JARVIS_MOCK_LLM_URL + +Skipped by default. Run via: + uv run pytest tests/evals/comparison/test_live_external.py -m live_external -v +""" + +from __future__ import annotations + +import json +import os +import subprocess +import sys +from pathlib import Path + +import pytest + +pytestmark = pytest.mark.live_external + + +def _have_env() -> bool: + return bool(os.environ.get("HERMES_AGENT_PATH")) and bool( + os.environ.get("OPENCLAW_PATH") + ) + + +@pytest.mark.skipif( + not _have_env(), + reason="HERMES_AGENT_PATH or OPENCLAW_PATH not set", +) +class TestLiveExternal: + def test_hermes_runner_emits_valid_json(self, tmp_path: Path) -> None: + runner = ( + Path(__file__).resolve().parents[3] + / "src/openjarvis/evals/backends/external/_runners/hermes_runner.py" + ) + out_json = tmp_path / "out.json" + env = dict(os.environ) + result = subprocess.run( + [ + sys.executable, + str(runner), + "--task", + "Say hello.", + "--model", + "test", + "--base-url", + os.environ.get("JARVIS_MOCK_LLM_URL", "http://localhost:8000/v1"), + "--api-key", + "dummy", + "--api-mode", + "chat_completions", + "--output-json", + str(out_json), + "--max-iterations", + "3", + ], + env=env, + capture_output=True, + text=True, + timeout=120, + ) + assert out_json.exists(), f"runner did not emit output. stderr={result.stderr}" + data = json.loads(out_json.read_text()) + assert "content" in data + assert "usage" in data + assert "tool_calls" in data + assert "turn_count" in data + + def test_openclaw_runner_emits_valid_json(self, tmp_path: Path) -> None: + runner = ( + Path(__file__).resolve().parents[3] + / "src/openjarvis/evals/backends/external/_runners/openclaw_runner.mjs" + ) + if not runner.exists(): + pytest.skip("openclaw_runner.mjs not present") + out_json = tmp_path / "out.json" + env = dict(os.environ) + result = subprocess.run( + [ + "node", + str(runner), + "--task", + "Say hello.", + "--model", + "test", + "--base-url", + os.environ.get("JARVIS_MOCK_LLM_URL", "http://localhost:8000/v1"), + "--api-key", + "dummy", + "--output-json", + str(out_json), + ], + env=env, + capture_output=True, + text=True, + timeout=120, + ) + assert out_json.exists(), f"runner did not emit output. stderr={result.stderr}" + data = json.loads(out_json.read_text()) + assert "content" in data + + def test_end_to_end_runs_one_config(self, tmp_path: Path) -> None: + """Run one full benchmark cell via the CLI; verify summary.json shape.""" + pytest.skip( + "Requires full eval-CLI integration; enable when Layer 2 wiring stable." + ) diff --git a/tests/evals/comparison/test_make_configs.py b/tests/evals/comparison/test_make_configs.py new file mode 100644 index 00000000..27efd5ef --- /dev/null +++ b/tests/evals/comparison/test_make_configs.py @@ -0,0 +1,97 @@ +"""Tests for openjarvis.evals.comparison.make_configs.""" + +from __future__ import annotations + +from pathlib import Path + +import pytest +import tomllib + +from openjarvis.evals.comparison.make_configs import ( + BENCHMARKS, # noqa: F401 (verify export) + FRAMEWORKS, # noqa: F401 (verify export) + MODELS, + materialize_config, +) + + +class TestMaterializeConfig: + def test_emits_valid_toml(self, tmp_path: Path) -> None: + out = materialize_config( + framework="hermes", + model="qwen-9b", + benchmark="gaia", + output_dir=tmp_path, + ) + assert out.exists() + with open(out, "rb") as fh: + data = tomllib.load(fh) + assert data["meta"]["framework"] == "hermes" + assert data["benchmarks"][0]["backend"] == "hermes" + assert data["models"][0]["name"] == MODELS["qwen-9b"]["model_id"] + + def test_filename_convention(self, tmp_path: Path) -> None: + out = materialize_config( + framework="hermes", + model="qwen-9b", + benchmark="gaia", + output_dir=tmp_path, + ) + assert out.name == "gaia-hermes-qwen-9b.toml" + + def test_unknown_framework_rejected(self, tmp_path: Path) -> None: + with pytest.raises(ValueError, match="unknown framework"): + materialize_config( + framework="not-real", + model="qwen-9b", + benchmark="gaia", + output_dir=tmp_path, + ) + + def test_unknown_benchmark_rejected(self, tmp_path: Path) -> None: + with pytest.raises(ValueError, match="unknown benchmark"): + materialize_config( + framework="hermes", + model="qwen-9b", + benchmark="not-real", + output_dir=tmp_path, + ) + + def test_idempotent(self, tmp_path: Path) -> None: + out1 = materialize_config( + framework="hermes", + model="qwen-9b", + benchmark="gaia", + output_dir=tmp_path, + ) + content1 = out1.read_text() + out2 = materialize_config( + framework="hermes", + model="qwen-9b", + benchmark="gaia", + output_dir=tmp_path, + ) + assert out1 == out2 + assert out2.read_text() == content1 + + +class TestTemplateStripping: + def test_no_substitution_in_comments(self, tmp_path: Path) -> None: + """Documentation comments must not contain substituted text.""" + out = materialize_config( + framework="hermes", + model="qwen-9b", + benchmark="gaia", + output_dir=tmp_path, + ) + text = out.read_text() + # The substitution variables doc block uses not {{var}}; + # if it leaks through, "" or similar would appear in output + assert "" not in text + assert "" not in text + assert "{{benchmark}}" not in text # Must be substituted + # The output should start with [meta], not with a doc-comment header + first_non_blank = next(line for line in text.splitlines() if line.strip()) + assert first_non_blank.startswith("[meta]"), ( + f"Expected first line to be [meta], got: {first_non_blank!r}" + ) diff --git a/tests/evals/comparison/test_openclaw_backend.py b/tests/evals/comparison/test_openclaw_backend.py new file mode 100644 index 00000000..af47ce07 --- /dev/null +++ b/tests/evals/comparison/test_openclaw_backend.py @@ -0,0 +1,177 @@ +"""Tests for openjarvis.evals.backends.external.openclaw.OpenClawBackend.""" + +from __future__ import annotations + +from pathlib import Path +from unittest.mock import patch + +import pytest + +from openjarvis.evals.backends.external._subprocess_runner import SubprocessResult +from openjarvis.evals.backends.external.openclaw import OpenClawBackend +from openjarvis.evals.comparison.third_party import ( + CommitDriftError, + ThirdPartyConfig, + ThirdPartyEntry, +) + + +def _fake_third_party(tmp_path: Path) -> ThirdPartyConfig: + return ThirdPartyConfig( + entries={ + "openclaw": ThirdPartyEntry( + name="openclaw", + path=tmp_path, + pinned_commit="def456", + runner_script=( + "src/openjarvis/evals/backends/external/_runners/openclaw_runner.mjs" + ), + node_executable="", + ) + } + ) + + +class TestOpenClawBackend: + def test_generate_full_builds_correct_subprocess_command( + self, tmp_path: Path + ) -> None: + cfg = _fake_third_party(tmp_path) + with ( + patch( + "openjarvis.evals.backends.external.openclaw.load_third_party_config", + return_value=cfg, + ), + patch( + "openjarvis.evals.backends.external.openclaw.verify_commit_pin", + ), + patch( + "openjarvis.evals.backends.external.openclaw.run_one_shot", + ) as mock_run, + ): + mock_run.return_value = SubprocessResult( + stdout="", + stderr="", + exit_code=0, + latency_seconds=1.0, + energy_joules=10.0, + peak_power_w=5.0, + sampler_method="nvml", + parsed_json={ + "content": "x", + "usage": {}, + "trajectory": [], + "tool_calls": 0, + "turn_count": 0, + "error": None, + }, + error=None, + ) + backend = OpenClawBackend(base_url="http://x", api_key="k") + backend.generate_full( + "task", + model="qwen-9b", + system="", + temperature=0.0, + max_tokens=2048, + ) + cmd = mock_run.call_args.kwargs["cmd"] + # First element should be the node executable (or "node") + assert cmd[0] == "node" + assert "--task" in cmd and "task" in cmd + assert "--model" in cmd and "qwen-9b" in cmd + + def test_generate_full_returns_extended_dict(self, tmp_path: Path) -> None: + cfg = _fake_third_party(tmp_path) + with ( + patch( + "openjarvis.evals.backends.external.openclaw.load_third_party_config", + return_value=cfg, + ), + patch( + "openjarvis.evals.backends.external.openclaw.verify_commit_pin", + ), + patch( + "openjarvis.evals.backends.external.openclaw.run_one_shot", + ) as mock_run, + ): + mock_run.return_value = SubprocessResult( + stdout="", + stderr="", + exit_code=0, + latency_seconds=3.0, + energy_joules=99.0, + peak_power_w=30.0, + sampler_method="nvml", + parsed_json={ + "content": "answer", + "usage": {"total_tokens": 100}, + "trajectory": [], + "tool_calls": 1, + "turn_count": 2, + "error": None, + }, + error=None, + ) + backend = OpenClawBackend(base_url="http://x", api_key="k") + result = backend.generate_full( + "task", + model="qwen-9b", + system="", + temperature=0.0, + max_tokens=2048, + ) + assert result["content"] == "answer" + assert result["framework"] == "openclaw" + assert result["framework_commit"] == "def456" + assert result["energy_joules"] == 99.0 + + def test_generate_full_propagates_subprocess_error(self, tmp_path: Path) -> None: + cfg = _fake_third_party(tmp_path) + with ( + patch( + "openjarvis.evals.backends.external.openclaw.load_third_party_config", + return_value=cfg, + ), + patch( + "openjarvis.evals.backends.external.openclaw.verify_commit_pin", + ), + patch( + "openjarvis.evals.backends.external.openclaw.run_one_shot", + ) as mock_run, + ): + mock_run.return_value = SubprocessResult( + stdout="", + stderr="boom", + exit_code=1, + latency_seconds=0.1, + energy_joules=None, + peak_power_w=None, + sampler_method="unavailable", + parsed_json={}, + error="subprocess_crash", + ) + backend = OpenClawBackend(base_url="http://x", api_key="k") + result = backend.generate_full( + "task", + model="qwen-9b", + system="", + temperature=0.0, + max_tokens=2048, + ) + assert result["error"] == "subprocess_crash" + + def test_init_raises_on_commit_drift(self, tmp_path: Path) -> None: + cfg = _fake_third_party(tmp_path) + with ( + patch( + "openjarvis.evals.backends.external.openclaw.load_third_party_config", + return_value=cfg, + ), + patch( + "openjarvis.evals.backends.external.openclaw.verify_commit_pin", + side_effect=CommitDriftError("drift!"), + ), + ): + with pytest.raises(CommitDriftError, match="drift"): + OpenClawBackend(base_url="http://x", api_key="k") diff --git a/tests/evals/comparison/test_runner_error_path_framework.py b/tests/evals/comparison/test_runner_error_path_framework.py new file mode 100644 index 00000000..c6272e39 --- /dev/null +++ b/tests/evals/comparison/test_runner_error_path_framework.py @@ -0,0 +1,189 @@ +"""Verify error-path EvalResult propagates framework_name from backend. + +Regression test for the bug where _process_one()'s exception fallback +constructed EvalResult without ``framework=``, causing the dataclass +default of ``"openjarvis"`` to silently mislabel data from foreign +backends (hermes, openclaw) when a task errored before reaching the +happy-path EvalResult construction. +""" + +from __future__ import annotations + +from typing import Any, Dict, Iterable, List, Optional, Tuple + +from openjarvis.evals.core.dataset import DatasetProvider +from openjarvis.evals.core.runner import EvalRunner +from openjarvis.evals.core.scorer import Scorer +from openjarvis.evals.core.types import EvalRecord, RunConfig + + +class _FailingBackend: + """Mock backend whose generate_full() always raises.""" + + backend_id = "hermes" + framework_name = "hermes" + + def generate(self, *args: Any, **kwargs: Any) -> str: + raise RuntimeError("simulated backend failure") + + def generate_full(self, *args: Any, **kwargs: Any) -> Dict[str, Any]: + raise RuntimeError("simulated backend failure") + + def close(self) -> None: + pass + + +class _SingleRecordDataset(DatasetProvider): + dataset_id = "mock" + dataset_name = "mock" + + def load( + self, + *, + max_samples: Optional[int] = None, + split: Optional[str] = None, + seed: Optional[int] = None, + ) -> None: + pass + + def iter_records(self) -> Iterable[EvalRecord]: + yield EvalRecord( + record_id="rec-1", + problem="Say hi.", + reference="hi", + category="chat", + ) + + def size(self) -> int: + return 1 + + +class _NoopScorer(Scorer): + scorer_id = "noop" + + def score( + self, + record: EvalRecord, + model_answer: str, + ) -> Tuple[Optional[bool], Dict[str, Any]]: + return False, {} + + +class TestErrorPathFrameworkPropagation: + def test_failed_task_keeps_backend_framework_name(self, tmp_path: Any) -> None: + """When the backend raises, EvalResult.framework should still be the + backend's framework_name, not the dataclass default 'openjarvis'.""" + cfg = RunConfig( + benchmark="mock", + backend="hermes", + model="mock-model", + max_samples=1, + max_workers=1, + output_path=str(tmp_path / "out.jsonl"), + ) + runner = EvalRunner( + config=cfg, + dataset=_SingleRecordDataset(), + backend=_FailingBackend(), + scorer=_NoopScorer(), + ) + runner.run() + + results: List[Any] = runner.results + assert len(results) == 1 + result = results[0] + # The KEY assertion: framework must come from backend.framework_name, + # NOT from the EvalResult dataclass default of "openjarvis". + assert result.framework == "hermes", ( + f"Expected framework='hermes' (from backend.framework_name), " + f"got {result.framework!r}" + ) + assert result.error is not None + assert "simulated backend failure" in result.error + + def test_default_framework_when_backend_missing_attr(self, tmp_path: Any) -> None: + """Backends that don't override framework_name should still produce + a sensible default (the ABC's "openjarvis"), via getattr fallback.""" + + class _BackendWithoutFrameworkName: + backend_id = "legacy" + + def generate(self, *a: Any, **kw: Any) -> str: + raise RuntimeError("boom") + + def generate_full(self, *a: Any, **kw: Any) -> Dict[str, Any]: + raise RuntimeError("boom") + + def close(self) -> None: + pass + + cfg = RunConfig( + benchmark="mock", + backend="legacy", + model="mock-model", + max_samples=1, + max_workers=1, + output_path=str(tmp_path / "out.jsonl"), + ) + runner = EvalRunner( + config=cfg, + dataset=_SingleRecordDataset(), + backend=_BackendWithoutFrameworkName(), # type: ignore[arg-type] + scorer=_NoopScorer(), + ) + runner.run() + + results = runner.results + assert len(results) == 1 + # getattr() defensive fallback in the runner kicks in; result keeps + # the conservative "openjarvis" tag rather than crashing. + assert results[0].framework == "openjarvis" + assert results[0].error is not None + + +class _MockBackendWithCommit: + """Mock backend that exposes framework_name AND framework_commit_value.""" + + backend_id = "hermes" + framework_name = "hermes" + framework_commit_value = "abc12345" + + def generate(self, *args: Any, **kwargs: Any) -> str: + raise RuntimeError("simulated") + + def generate_full(self, *args: Any, **kwargs: Any) -> Dict[str, Any]: + raise RuntimeError("simulated") + + def close(self) -> None: + pass + + +class TestErrorPathCommitPropagation: + def test_failed_task_keeps_backend_framework_commit(self, tmp_path: Any) -> None: + """When the backend raises, EvalResult.framework_commit should be + the backend's framework_commit_value, not empty.""" + cfg = RunConfig( + benchmark="mock", + backend="hermes", + model="mock-model", + max_samples=1, + max_workers=1, + output_path=str(tmp_path / "out.jsonl"), + ) + runner = EvalRunner( + config=cfg, + dataset=_SingleRecordDataset(), + backend=_MockBackendWithCommit(), # type: ignore[arg-type] + scorer=_NoopScorer(), + ) + runner.run() + results = runner.results + assert len(results) >= 1 + for r in results: + assert r.framework == "hermes" + assert r.framework_commit == "abc12345", ( + f"Expected framework_commit='abc12345' " + f"(from backend.framework_commit_value), " + f"got {r.framework_commit!r}" + ) + assert r.error is not None diff --git a/tests/evals/comparison/test_subprocess_runner.py b/tests/evals/comparison/test_subprocess_runner.py new file mode 100644 index 00000000..9e0b7fdd --- /dev/null +++ b/tests/evals/comparison/test_subprocess_runner.py @@ -0,0 +1,167 @@ +"""Tests for openjarvis.evals.backends.external._subprocess_runner.""" + +from __future__ import annotations + +import json +from pathlib import Path +from unittest.mock import MagicMock, patch + +import pytest + +from openjarvis.evals.backends.external._subprocess_runner import ( + EnergySample, + SubprocessResult, + run_one_shot, +) + + +class TestSubprocessResult: + def test_dataclass_fields(self) -> None: + result = SubprocessResult( + stdout="hello", + stderr="", + exit_code=0, + latency_seconds=1.5, + energy_joules=100.0, + peak_power_w=20.0, + sampler_method="nvml", + parsed_json={"content": "x"}, + ) + assert result.stdout == "hello" + assert result.exit_code == 0 + assert result.energy_joules == 100.0 + assert result.parsed_json == {"content": "x"} + + +class TestEnergySample: + def test_dataclass_fields(self) -> None: + s = EnergySample(timestamp=1.0, watts=15.5) + assert s.timestamp == 1.0 + assert s.watts == 15.5 + + +class TestRunOneShot: + def test_successful_run_returns_parsed_json(self, tmp_path: Path) -> None: + output_json = tmp_path / "out.json" + output_json.write_text( + json.dumps( + { + "content": "hello", + "usage": { + "prompt_tokens": 10, + "completion_tokens": 5, + "total_tokens": 15, + }, + "trajectory": [], + "tool_calls": 0, + "turn_count": 1, + "error": None, + } + ) + ) + with patch("subprocess.Popen") as MockPopen: + mock_proc = MagicMock() + mock_proc.communicate.return_value = ("done", "") + mock_proc.returncode = 0 + MockPopen.return_value = mock_proc + + result = run_one_shot( + cmd=["echo", "ok"], + env={}, + timeout=10.0, + output_json_path=output_json, + ) + assert result.exit_code == 0 + assert result.parsed_json["content"] == "hello" + assert result.error is None + + def test_nonzero_exit_records_crash(self, tmp_path: Path) -> None: + output_json = tmp_path / "out.json" + with patch("subprocess.Popen") as MockPopen: + mock_proc = MagicMock() + mock_proc.communicate.return_value = ("", "boom") + mock_proc.returncode = 139 + MockPopen.return_value = mock_proc + + result = run_one_shot( + cmd=["false"], + env={}, + timeout=10.0, + output_json_path=output_json, + ) + assert result.exit_code == 139 + assert result.error == "subprocess_crash" + assert "boom" in result.stderr + + def test_timeout_kills_process(self, tmp_path: Path) -> None: + import subprocess as sp + + output_json = tmp_path / "out.json" + with patch("subprocess.Popen") as MockPopen: + mock_proc = MagicMock() + mock_proc.communicate.side_effect = [ + sp.TimeoutExpired(cmd="x", timeout=10), + ("", ""), + ] + mock_proc.returncode = -9 + MockPopen.return_value = mock_proc + + result = run_one_shot( + cmd=["sleep", "100"], + env={}, + timeout=10.0, + output_json_path=output_json, + ) + assert result.error == "timeout" + mock_proc.terminate.assert_called_once() + + def test_malformed_json_recorded(self, tmp_path: Path) -> None: + output_json = tmp_path / "out.json" + output_json.write_text("not json {{{") + with patch("subprocess.Popen") as MockPopen: + mock_proc = MagicMock() + mock_proc.communicate.return_value = ("", "") + mock_proc.returncode = 0 + MockPopen.return_value = mock_proc + + result = run_one_shot( + cmd=["echo"], + env={}, + timeout=10.0, + output_json_path=output_json, + ) + assert result.error == "malformed_runner_output" + + def test_missing_output_json_recorded(self, tmp_path: Path) -> None: + output_json = tmp_path / "never_created.json" + with patch("subprocess.Popen") as MockPopen: + mock_proc = MagicMock() + mock_proc.communicate.return_value = ("", "") + mock_proc.returncode = 0 + MockPopen.return_value = mock_proc + + result = run_one_shot( + cmd=["echo"], + env={}, + timeout=10.0, + output_json_path=output_json, + ) + assert result.error == "invalid_runner_output" + + +class TestEnergySampler: + def test_fallback_chain_reaches_unavailable( + self, monkeypatch: pytest.MonkeyPatch + ) -> None: + """When all samplers fail to initialize, return the null sampler.""" + from openjarvis.evals.backends.external import _subprocess_runner as m + + # Force every probe to fail + monkeypatch.setattr(m, "_try_start_nvml", lambda: None) + monkeypatch.setattr(m, "_try_start_powermetrics", lambda: None) + monkeypatch.setattr(m, "_try_start_rocm_smi", lambda: None) + monkeypatch.setattr(m, "_try_start_rapl", lambda: None) + + sampler = m._start_sampler() + assert sampler.method == "unavailable" + assert sampler.stop() == [] diff --git a/tests/evals/comparison/test_table_gen.py b/tests/evals/comparison/test_table_gen.py new file mode 100644 index 00000000..18c2fcf8 --- /dev/null +++ b/tests/evals/comparison/test_table_gen.py @@ -0,0 +1,273 @@ +"""Tests for openjarvis.evals.comparison.table_gen.""" + +from __future__ import annotations + +import json +from pathlib import Path + +import pytest + +pl = pytest.importorskip("polars") + +from openjarvis.evals.comparison.table_gen import ( # noqa: E402 + MixedCommitError, + ResultsFrame, + load_results, +) + + +def _write_summary(path: Path, **overrides: object) -> None: + payload = { + "framework": "hermes", + "framework_commit": "abc123", + "model": "Qwen/Qwen3.5-9B", + "benchmark": "gaia", + "n_tasks": 50, + "metrics": { + "accuracy": {"mean": 0.42, "std": 0.04, "n": 5}, + "latency_seconds": {"mean": 23.4, "std": 5.1, "n": 5}, + }, + "per_sample": [], + "hardware": {"gpu": "H100"}, + "started_at": "2026-05-01T00:00:00Z", + "ended_at": "2026-05-01T01:00:00Z", + } + payload.update(overrides) + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(payload)) + + +class TestLoadResults: + def test_loads_glob_and_returns_long_frame(self, tmp_path: Path) -> None: + _write_summary(tmp_path / "a" / "summary.json", framework="hermes") + _write_summary( + tmp_path / "b" / "summary.json", + framework="openclaw", + framework_commit="def456", + ) + + frame = load_results(str(tmp_path / "**" / "summary.json")) + assert isinstance(frame, ResultsFrame) + # 2 files x 2 metrics each = 4 rows + assert len(frame.df) == 4 + assert set(frame.df["framework"].to_list()) == {"hermes", "openclaw"} + + def test_skips_malformed_files(self, tmp_path: Path) -> None: + good = tmp_path / "good" / "summary.json" + _write_summary(good) + bad = tmp_path / "bad" / "summary.json" + bad.parent.mkdir(parents=True) + bad.write_text("not json") + + frame = load_results(str(tmp_path / "**" / "summary.json")) + # 1 good file x 2 metrics + assert len(frame.df) == 2 + assert frame.unloadable_count == 1 + + def test_mixed_commits_per_cell_raises(self, tmp_path: Path) -> None: + _write_summary(tmp_path / "a" / "summary.json", framework_commit="abc123") + _write_summary(tmp_path / "b" / "summary.json", framework_commit="zzz999") + with pytest.raises(MixedCommitError, match="abc123.*zzz999"): + load_results(str(tmp_path / "**" / "summary.json")) + + +class TestRenderBooktabs: + def test_emits_valid_tabular(self) -> None: + from openjarvis.evals.comparison.table_gen import _render_booktabs + + df = pl.DataFrame( + { + "row": ["hermes", "openjarvis"], + "col1": [0.42, 0.55], + "col2": [0.30, 0.40], + } + ) + fragment, standalone = _render_booktabs( + df, + row_col="row", + caption="Test caption", + label="tab:test", + ) + assert "\\begin{tabular}" in fragment + assert "\\end{tabular}" in fragment + assert "hermes" in fragment and "openjarvis" in fragment + assert "0.42" in fragment + assert "\\documentclass{standalone}" in standalone + assert fragment in standalone + + def test_missing_cell_renders_em_dash(self) -> None: + from openjarvis.evals.comparison.table_gen import _render_booktabs + + df = pl.DataFrame( + { + "row": ["hermes", "openjarvis"], + "col1": [None, 0.55], + }, + schema={"row": pl.Utf8, "col1": pl.Float64}, + ) + fragment, _ = _render_booktabs( + df, + row_col="row", + caption="x", + label="x", + ) + assert "\\textit{--}" in fragment + + +class TestT1Builder: + def test_t1_builds_from_synthetic_results(self) -> None: + from openjarvis.evals.comparison.table_gen import ( + ResultsFrame, + _build_t1, + ) + + df = pl.DataFrame( + { + "framework": [ + "hermes", + "openjarvis", + "openclaw", + "openjarvis-distilled", + ], + "framework_commit": ["a", "b", "c", "b"], + "model": ["qwen-9b"] * 4, + "benchmark": ["gaia"] * 4, + "metric_name": ["accuracy"] * 4, + "mean": [0.22, 0.40, 0.20, 0.55], + "std": [0.03] * 4, + "n": [5] * 4, + "source_path": ["x", "y", "z", "w"], + } + ) + frame = ResultsFrame(df=df) + fragment, standalone = _build_t1(frame) + assert "\\begin{tabular}" in fragment + assert "0.22" in fragment or "22.00" in fragment + assert "openjarvis-distilled" in fragment.lower() or "OJ-distilled" in fragment + + +class TestT2Builder: + def test_t2_emits_efficiency_table(self) -> None: + from openjarvis.evals.comparison.table_gen import ( + ResultsFrame, + _build_t2, + ) + + rows = [] + for fwk, model in [ + ("hermes", "qwen-9b"), + ("openjarvis", "qwen-9b"), + ("hermes", "claude-opus-46"), + ("openjarvis", "claude-opus-46"), + ]: + for metric, mean in [ + ("latency_seconds", 5.0), + ("energy_joules_per_query", 100.0), + ("input_tokens_per_query", 1000), + ("output_tokens_per_query", 200), + ("cost_usd_per_query", 0.001), + ]: + rows.append( + { + "framework": fwk, + "framework_commit": "x", + "model": model, + "benchmark": "gaia", + "metric_name": metric, + "mean": float(mean), + "std": 0.1, + "n": 5, + "source_path": "p", + } + ) + frame = ResultsFrame(df=pl.DataFrame(rows)) + fragment, _ = _build_t2(frame) + assert "\\begin{tabular}" in fragment + assert "Latency" in fragment or "latency" in fragment + assert "Energy" in fragment or "energy" in fragment + + +class TestT3to7Builders: + """One smoke test per T3-T7 builder; each verifies tabular emission.""" + + @pytest.mark.parametrize( + "builder_name", + ["_build_t3", "_build_t4", "_build_t5", "_build_t6", "_build_t7"], + ) + def test_builder_emits_tabular(self, builder_name: str) -> None: + import openjarvis.evals.comparison.table_gen as m + + builder = getattr(m, builder_name) + rows = [] + for fwk in ["hermes", "openjarvis"]: + for bench in ["gaia", "pinchbench"]: + for metric in [ + "accuracy", + "latency_seconds", + "energy_joules_per_query", + "input_tokens_per_query", + "output_tokens_per_query", + "cost_usd_per_query", + ]: + rows.append( + { + "framework": fwk, + "framework_commit": "x", + "model": "qwen-9b", + "benchmark": bench, + "metric_name": metric, + "mean": 1.0, + "std": 0.1, + "n": 5, + "source_path": "p", + } + ) + frame = m.ResultsFrame(df=pl.DataFrame(rows)) + fragment, _ = builder(frame) + assert "\\begin{tabular}" in fragment + assert "\\end{tabular}" in fragment + + +class TestTableBuilderRegistry: + def test_registry_has_all_seven(self) -> None: + from openjarvis.evals.comparison.table_gen import _TABLE_BUILDERS + + assert set(_TABLE_BUILDERS.keys()) == { + "T1", + "T2", + "T3", + "T4", + "T5", + "T6", + "T7", + } + + +class TestTableGenCLI: + def test_cli_writes_fragment_and_preview(self, tmp_path: Path) -> None: + from click.testing import CliRunner + + from openjarvis.evals.comparison.table_gen import main + + results_dir = tmp_path / "results" + results_dir.mkdir() + summary = results_dir / "summary.json" + _write_summary(summary) + + out_dir = tmp_path / "tables" + runner = CliRunner() + result = runner.invoke( + main, + [ + "--results-glob", + str(results_dir / "**" / "summary.json"), + "--tables", + "T1", + "--output-dir", + str(out_dir), + ], + ) + assert result.exit_code == 0, result.output + assert (out_dir / "T1.tex").exists() + assert (out_dir / "preview" / "T1_preview.tex").exists() + assert "\\begin{tabular}" in (out_dir / "T1.tex").read_text() diff --git a/tests/evals/comparison/test_third_party.py b/tests/evals/comparison/test_third_party.py new file mode 100644 index 00000000..70238d97 --- /dev/null +++ b/tests/evals/comparison/test_third_party.py @@ -0,0 +1,139 @@ +"""Tests for openjarvis.evals.comparison.third_party.""" + +from __future__ import annotations + +from pathlib import Path +from unittest.mock import patch + +import pytest + +from openjarvis.evals.comparison.third_party import ( + ThirdPartyConfig, + ThirdPartyEntry, + ThirdPartyNotFoundError, + load_third_party_config, +) + + +class TestLoadThirdPartyConfig: + def test_loads_default_toml(self, tmp_path: Path) -> None: + toml_path = tmp_path / "_third_party.toml" + toml_path.write_text( + "[hermes]\n" + 'path = "/some/hermes/path"\n' + 'pinned_commit = "abc123"\n' + 'runner_script = "_runners/hermes_runner.py"\n' + 'python_executable = ""\n' + "\n" + "[openclaw]\n" + 'path = "/some/openclaw/path"\n' + 'pinned_commit = "def456"\n' + 'runner_script = "_runners/openclaw_runner.mjs"\n' + 'node_executable = ""\n' + ) + cfg = load_third_party_config(toml_path) + assert isinstance(cfg, ThirdPartyConfig) + assert cfg.entries["hermes"].path == Path("/some/hermes/path") + assert cfg.entries["hermes"].pinned_commit == "abc123" + assert cfg.entries["openclaw"].pinned_commit == "def456" + + def test_env_var_overrides_path( + self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch + ) -> None: + toml_path = tmp_path / "_third_party.toml" + toml_path.write_text( + "[hermes]\n" + 'path = "/default/hermes"\n' + 'pinned_commit = "abc"\n' + 'runner_script = "x"\n' + 'python_executable = ""\n' + ) + monkeypatch.setenv("HERMES_AGENT_PATH", "/override/hermes") + cfg = load_third_party_config(toml_path) + assert cfg.entries["hermes"].path == Path("/override/hermes") + + def test_missing_toml_raises(self, tmp_path: Path) -> None: + toml_path = tmp_path / "does_not_exist.toml" + with pytest.raises(ThirdPartyNotFoundError, match="_third_party.toml"): + load_third_party_config(toml_path) + + def test_loads_default_path_with_no_args(self) -> None: + """Calling load_third_party_config() with no args reads the in-repo default.""" + cfg = load_third_party_config() + assert "hermes" in cfg.entries + assert "openclaw" in cfg.entries + assert cfg.entries["hermes"].pinned_commit == "5d3be898a" + assert ( + cfg.entries["openclaw"].pinned_commit + == "123ae82fca3009df7144fa8d41e1185cd63e61e1" + ) + + +class TestVerifyCommitPin: + def test_matching_commit_passes(self, tmp_path: Path) -> None: + """No exception when HEAD matches pinned commit.""" + from openjarvis.evals.comparison.third_party import verify_commit_pin + + entry = ThirdPartyEntry( + name="hermes", + path=tmp_path, + pinned_commit="abc123", + runner_script="x", + ) + with patch("subprocess.run") as mock_run: + mock_run.return_value.stdout = "abc123\n" + mock_run.return_value.returncode = 0 + verify_commit_pin(entry) # should not raise + + def test_mismatched_commit_raises(self, tmp_path: Path) -> None: + from openjarvis.evals.comparison.third_party import ( + CommitDriftError, + verify_commit_pin, + ) + + entry = ThirdPartyEntry( + name="hermes", + path=tmp_path, + pinned_commit="abc123", + runner_script="x", + ) + with patch("subprocess.run") as mock_run: + mock_run.return_value.stdout = "deadbeef\n" + mock_run.return_value.returncode = 0 + with pytest.raises(CommitDriftError, match="abc123.*deadbeef"): + verify_commit_pin(entry) + + def test_drift_override_env_var( + self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch + ) -> None: + from openjarvis.evals.comparison.third_party import verify_commit_pin + + entry = ThirdPartyEntry( + name="hermes", + path=tmp_path, + pinned_commit="abc123", + runner_script="x", + ) + monkeypatch.setenv("JARVIS_ALLOW_COMMIT_DRIFT", "1") + with patch("subprocess.run") as mock_run: + mock_run.return_value.stdout = "deadbeef\n" + mock_run.return_value.returncode = 0 + # Should not raise; should log a warning instead + verify_commit_pin(entry) + + +class TestVerifyCommitPinPathHandling: + def test_empty_path_raises_with_env_var_hint(self) -> None: + from openjarvis.evals.comparison.third_party import ( + ThirdPartyNotFoundError, + verify_commit_pin, + ) + + entry = ThirdPartyEntry( + name="hermes", + path=Path(""), + pinned_commit="abc", + runner_script="x", + ) + with pytest.raises(ThirdPartyNotFoundError, match="HERMES_AGENT_PATH"): + verify_commit_pin(entry) diff --git a/tests/evals/test_jarvis_agent_backend_fields.py b/tests/evals/test_jarvis_agent_backend_fields.py new file mode 100644 index 00000000..81e70881 --- /dev/null +++ b/tests/evals/test_jarvis_agent_backend_fields.py @@ -0,0 +1,47 @@ +"""Verify JarvisAgentBackend.generate_full returns the spec §6.2 extended fields.""" + +from __future__ import annotations + +from unittest.mock import MagicMock, patch + + +class TestJarvisAgentExtendedFields: + def test_generate_full_includes_framework_and_commit(self) -> None: + from openjarvis.evals.backends.jarvis_agent import JarvisAgentBackend + + with patch("openjarvis.system.SystemBuilder") as MockSB: + mock_system = MagicMock() + mock_system.ask.return_value = { + "content": "answer", + "usage": {"prompt_tokens": 10, "completion_tokens": 5}, + "model": "qwen-9b", + } + mock_system.trace_collector = None + + builder_instance = MockSB.return_value + builder_instance.engine.return_value = builder_instance + builder_instance.model.return_value = builder_instance + builder_instance.agent.return_value = builder_instance + builder_instance.tools.return_value = builder_instance + builder_instance.telemetry.return_value = builder_instance + builder_instance.traces.return_value = builder_instance + builder_instance.build.return_value = mock_system + # The backend mutates ``builder._config`` directly; provide a + # MagicMock that tolerates arbitrary attribute access. + builder_instance._config = MagicMock() + + backend = JarvisAgentBackend(model="qwen-9b") + result = backend.generate_full( + "task", + model="qwen-9b", + system="", + temperature=0.0, + max_tokens=2048, + ) + assert result["framework"] == "openjarvis" + assert "framework_commit" in result + assert "energy_joules" in result # may be None + assert "peak_power_w" in result + assert "tool_calls" in result + assert "turn_count" in result + assert "error" in result diff --git a/tests/evals/test_jarvis_direct_backend_fields.py b/tests/evals/test_jarvis_direct_backend_fields.py new file mode 100644 index 00000000..00cc63da --- /dev/null +++ b/tests/evals/test_jarvis_direct_backend_fields.py @@ -0,0 +1,43 @@ +"""Verify JarvisDirectBackend.generate_full returns the spec §6.2 extended fields.""" + +from __future__ import annotations + +from unittest.mock import MagicMock + + +class TestJarvisDirectExtendedFields: + def test_generate_full_includes_framework_and_commit(self) -> None: + from openjarvis.evals.backends.jarvis_direct import JarvisDirectBackend + + # Build the backend without invoking __init__ (which would spin up an + # engine); set required attrs directly. JarvisDirectBackend.generate_full + # calls ``self._system.engine.generate(messages, ...)`` so we mock the + # whole ``_system`` chain. + backend = JarvisDirectBackend.__new__(JarvisDirectBackend) + backend._telemetry = False + backend._gpu_metrics = False + backend._system = MagicMock() + backend._system.engine.generate.return_value = { + "content": "answer", + "usage": { + "prompt_tokens": 10, + "completion_tokens": 5, + "total_tokens": 15, + }, + "model": "qwen-9b", + } + + result = backend.generate_full( + "task", + model="qwen-9b", + system="", + temperature=0.0, + max_tokens=2048, + ) + assert result["framework"] == "openjarvis" + assert "framework_commit" in result + assert result["tool_calls"] == 0 # direct = no tool calls + assert result["turn_count"] == 1 # direct = single turn + assert "energy_joules" in result + assert "peak_power_w" in result + assert "error" in result diff --git a/tests/install/__init__.py b/tests/install/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/tests/install/conftest.py b/tests/install/conftest.py new file mode 100644 index 00000000..aa9457c0 --- /dev/null +++ b/tests/install/conftest.py @@ -0,0 +1,27 @@ +"""Test fixtures for installer / cold-start refresh tests.""" + +from __future__ import annotations + +from pathlib import Path + +import pytest + + +@pytest.fixture +def tmp_openjarvis_home(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: + """Point ``DEFAULT_CONFIG_DIR`` at a tmpdir for isolated tests. + + Yields the directory; teardown is automatic via tmp_path. + """ + home = tmp_path / ".openjarvis" + home.mkdir() + (home / ".state").mkdir() + (home / ".state" / "models").mkdir() + monkeypatch.setattr( + "openjarvis.core.config.DEFAULT_CONFIG_DIR", home + ) + monkeypatch.setattr( + "openjarvis.core.config.DEFAULT_CONFIG_PATH", home / "config.toml" + ) + monkeypatch.setenv("HOME", str(tmp_path)) + return home diff --git a/tests/install/test_bg_state.py b/tests/install/test_bg_state.py new file mode 100644 index 00000000..8b808c1b --- /dev/null +++ b/tests/install/test_bg_state.py @@ -0,0 +1,84 @@ +"""Tests for openjarvis.cli._bg_state.""" + +from __future__ import annotations + +from pathlib import Path + +from openjarvis.cli import _bg_state + + +def test_get_status_empty(tmp_openjarvis_home: Path) -> None: + """No state files = everything 'pending'.""" + s = _bg_state.get_status() + assert s.rust_extension == "pending" + assert s.models == {} + + +def test_get_status_rust_built(tmp_openjarvis_home: Path) -> None: + (tmp_openjarvis_home / ".state" / "extension-built").write_text("") + s = _bg_state.get_status() + assert s.rust_extension == "ready" + + +def test_get_status_rust_failed(tmp_openjarvis_home: Path) -> None: + (tmp_openjarvis_home / ".state" / "extension-failed").write_text("error tail here") + s = _bg_state.get_status() + assert s.rust_extension == "failed" + assert s.rust_error == "error tail here" + + +def test_get_status_model_downloading(tmp_openjarvis_home: Path) -> None: + (tmp_openjarvis_home / ".state" / "models" / "qwen3.5:9b.downloading").write_text( + "" + ) + s = _bg_state.get_status() + assert s.models == {"qwen3.5:9b": "downloading"} + + +def test_get_status_model_ready(tmp_openjarvis_home: Path) -> None: + (tmp_openjarvis_home / ".state" / "models" / "qwen3.5:9b.ready").write_text("") + s = _bg_state.get_status() + assert s.models == {"qwen3.5:9b": "ready"} + + +def test_get_status_model_failed(tmp_openjarvis_home: Path) -> None: + (tmp_openjarvis_home / ".state" / "models" / "qwen3.5:9b.failed").write_text( + "net error" + ) + s = _bg_state.get_status() + assert s.models == {"qwen3.5:9b": "failed"} + + +def test_get_status_ready_supersedes_downloading(tmp_openjarvis_home: Path) -> None: + """If both .downloading and .ready exist (race window), .ready wins.""" + models_dir = tmp_openjarvis_home / ".state" / "models" + (models_dir / "qwen3.5:9b.downloading").write_text("") + (models_dir / "qwen3.5:9b.ready").write_text("") + s = _bg_state.get_status() + assert s.models["qwen3.5:9b"] == "ready" + + +def test_all_ready_true_when_all_ready(tmp_openjarvis_home: Path) -> None: + (tmp_openjarvis_home / ".state" / "extension-built").write_text("") + (tmp_openjarvis_home / ".state" / "models" / "qwen3.5:9b.ready").write_text("") + s = _bg_state.get_status() + assert s.all_ready() is True + + +def test_all_ready_false_when_anything_pending(tmp_openjarvis_home: Path) -> None: + (tmp_openjarvis_home / ".state" / "extension-built").write_text("") + (tmp_openjarvis_home / ".state" / "models" / "qwen3.5:9b.downloading").write_text( + "" + ) + s = _bg_state.get_status() + assert s.all_ready() is False + + +def test_get_status_handles_deleted_file_race(tmp_openjarvis_home: Path) -> None: + """File present at glob-time, gone at read-time = treat as missing.""" + f = tmp_openjarvis_home / ".state" / "extension-failed" + f.write_text("oops") + f.unlink() # simulate race + # No assertion error expected + s = _bg_state.get_status() + assert s.rust_extension == "pending" diff --git a/tests/install/test_fixture_smoke.py b/tests/install/test_fixture_smoke.py new file mode 100644 index 00000000..a2548a29 --- /dev/null +++ b/tests/install/test_fixture_smoke.py @@ -0,0 +1,18 @@ +"""Smoke test that the tmp_openjarvis_home fixture works.""" + +from __future__ import annotations + +from pathlib import Path + +from openjarvis.core import config as config_mod + + +def test_fixture_redirects_default_config_dir(tmp_openjarvis_home: Path) -> None: + assert config_mod.DEFAULT_CONFIG_DIR == tmp_openjarvis_home + assert tmp_openjarvis_home.exists() + assert (tmp_openjarvis_home / ".state").exists() + assert (tmp_openjarvis_home / ".state" / "models").exists() + + +def test_fixture_redirects_config_path(tmp_openjarvis_home: Path) -> None: + assert config_mod.DEFAULT_CONFIG_PATH == tmp_openjarvis_home / "config.toml" diff --git a/uv.lock b/uv.lock index 0e5ca989..5ad23f12 100644 --- a/uv.lock +++ b/uv.lock @@ -5080,6 +5080,9 @@ eval-sheets = [ eval-wandb = [ { name = "wandb" }, ] +framework-comparison = [ + { name = "polars" }, +] gpu-metrics = [ { name = "pynvml" }, ] @@ -5218,6 +5221,7 @@ requires-dist = [ { name = "pdfplumber", marker = "extra == 'memory-pdf'", specifier = ">=0.10" }, { name = "pdfplumber", marker = "extra == 'pdf'", specifier = ">=0.10" }, { name = "playwright", marker = "extra == 'browser'", specifier = ">=1.40" }, + { name = "polars", marker = "extra == 'framework-comparison'", specifier = ">=1.0" }, { name = "praw", marker = "extra == 'channel-reddit'", specifier = ">=7.0" }, { name = "pre-commit", marker = "extra == 'dev'", specifier = ">=3.0" }, { name = "pydantic", marker = "extra == 'server'", specifier = ">=2.0" }, @@ -5258,7 +5262,7 @@ requires-dist = [ { name = "zeus-ml", extras = ["apple"], marker = "extra == 'energy-apple'" }, { name = "zulip", marker = "extra == 'channel-zulip'", specifier = ">=0.9" }, ] -provides-extras = ["dev", "inference-mlx", "inference-vllm", "inference-cloud", "inference-google", "inference-litellm", "inference-gemma", "tools-search", "memory-faiss", "memory-colbert", "memory-pdf", "memory-bm25", "server", "openhands", "gpu-metrics", "energy-amd", "energy-apple", "energy-all", "orchestrator-training", "learning-dspy", "learning-gepa", "channel-telegram", "channel-discord", "channel-slack", "channel-line", "channel-viber", "channel-messenger", "channel-reddit", "channel-mastodon", "channel-xmpp", "channel-rocketchat", "channel-zulip", "channel-twitter", "channel-twitch", "channel-nostr", "channel-twilio", "channel-gmail", "browser", "media", "pdf", "scheduler", "security-signing", "sandbox-wasm", "sandbox-docker", "dashboard", "speech", "speech-deepgram", "eval-wandb", "eval-sheets", "docs"] +provides-extras = ["dev", "inference-mlx", "inference-vllm", "inference-cloud", "inference-google", "inference-litellm", "inference-gemma", "tools-search", "memory-faiss", "memory-colbert", "memory-pdf", "memory-bm25", "server", "openhands", "gpu-metrics", "energy-amd", "energy-apple", "energy-all", "orchestrator-training", "learning-dspy", "learning-gepa", "channel-telegram", "channel-discord", "channel-slack", "channel-line", "channel-viber", "channel-messenger", "channel-reddit", "channel-mastodon", "channel-xmpp", "channel-rocketchat", "channel-zulip", "channel-twitter", "channel-twitch", "channel-nostr", "channel-twilio", "channel-gmail", "browser", "media", "pdf", "scheduler", "security-signing", "sandbox-wasm", "sandbox-docker", "dashboard", "speech", "speech-deepgram", "eval-wandb", "eval-sheets", "framework-comparison", "docs"] [package.metadata.requires-dev] dev = [{ name = "maturin", specifier = ">=1.12.6" }] @@ -5903,6 +5907,34 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/54/20/4d324d65cc6d9205fabedc306948156824eb9f0ee1633355a8f7ec5c66bf/pluggy-1.6.0-py3-none-any.whl", hash = "sha256:e920276dd6813095e9377c0bc5566d94c932c33b27a3e3945d8389c374dd4746", size = 20538, upload-time = "2025-05-15T12:30:06.134Z" }, ] +[[package]] +name = "polars" +version = "1.40.1" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "polars-runtime-32" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/b3/8c/bc9bc948058348ed43117cecc3007cd608f395915dae8a00974579a5dab1/polars-1.40.1.tar.gz", hash = "sha256:ab2694134b137596b5a59bfd7b4c54ebbc9b59f9403127f18e32d363777552e8", size = 733574, upload-time = "2026-04-22T19:15:55.507Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/ea/91/74fc60d94488685a92ac9d49d7ec55f3e91fe9b77942a6235a5fa7f249c3/polars-1.40.1-py3-none-any.whl", hash = "sha256:c0f861219d1319cdea45c4ce4d30355a47176b8f98dcedf95ea8269f131b8abd", size = 828723, upload-time = "2026-04-22T19:14:25.452Z" }, +] + +[[package]] +name = "polars-runtime-32" +version = "1.40.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/54/ba/26d40f039be9f552b5fd7365a621bdfc0f8e912ef77094ae4693491b0bae/polars_runtime_32-1.40.1.tar.gz", hash = "sha256:37f3065615d1bf90d03b5326222df4c5c1f8a5d33e50470aa588e3465e6eb814", size = 2935843, upload-time = "2026-04-22T19:15:57.26Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/7d/46/22c8af5eed68ac2eeb556e0fa3ca8a7b798e984ceff4450888f3b5ac61fd/polars_runtime_32-1.40.1-cp310-abi3-macosx_10_12_x86_64.whl", hash = "sha256:b748ef652270cc49e9e69f99a035e0eb4d5f856d42bcd6ac4d9d80a40142aa1e", size = 52098755, upload-time = "2026-04-22T19:14:28.555Z" }, + { url = "https://files.pythonhosted.org/packages/c6/3e/48599a38009ca60ff82a6f38c8a621ce3c0286aa7397c7d79e741bd9060e/polars_runtime_32-1.40.1-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:d249b3743e05986060cec0a7aaa542d020df6c6b876e556023a310efd581f9be", size = 46367542, upload-time = "2026-04-22T19:14:32.433Z" }, + { url = "https://files.pythonhosted.org/packages/43/e9/384bc069367a1a36ee31c13782c178dbd039b2b873b772d4a0fc23a2373d/polars_runtime_32-1.40.1-cp310-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:5987b30e7aa1059d069498496e8dda35afd592b0ac3d46ed87e3ff8df1ad652c", size = 50252104, upload-time = "2026-04-22T19:14:35.945Z" }, + { url = "https://files.pythonhosted.org/packages/15/ef/7d57ceb0651af74194e97ed6583e148d352f03d696090221b8059cdfc90b/polars_runtime_32-1.40.1-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:8d7f42a8b3f16fc66002cc0f6516f7dd7653396886ae0ed362ab95c0b3408b59", size = 56250788, upload-time = "2026-04-22T19:14:39.743Z" }, + { url = "https://files.pythonhosted.org/packages/10/0f/e4b3ffc748827a14a474ec9c42e45c066050e440fec57e914091d9adda75/polars_runtime_32-1.40.1-cp310-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:e5f7becc237a7ec9d9a10878dc8e54b73bbf4e2d94a2991c37d7a0b38590d8f9", size = 50432590, upload-time = "2026-04-22T19:14:43.388Z" }, + { url = "https://files.pythonhosted.org/packages/d9/0b/b8d95fbed869fa4caabe9c400e4210374913b376e925e96fdcfa9be6416b/polars_runtime_32-1.40.1-cp310-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:992d14cf191dde043d36fbdbc98a65e43fbc7e9a5024cecd45f838ac4988c1ee", size = 54155564, upload-time = "2026-04-22T19:14:47.239Z" }, + { url = "https://files.pythonhosted.org/packages/06/d9/d091d8fb5cbed5e9536adfed955c4c89987a4cc3b8e73ae4532402b91c74/polars_runtime_32-1.40.1-cp310-abi3-win_amd64.whl", hash = "sha256:f78bb2abd00101cbb23cc0cb068f7e36e081057a15d2ec2dde3dda280709f030", size = 51829755, upload-time = "2026-04-22T19:14:50.85Z" }, + { url = "https://files.pythonhosted.org/packages/65/ad/b33c3022a394f3eb55c3310597cec615412a8a33880055eee191d154a628/polars_runtime_32-1.40.1-cp310-abi3-win_arm64.whl", hash = "sha256:b5cbfaf6b085b420b4bfcbe24e8f665076d1cccfdb80c0484c02a023ce205537", size = 45822104, upload-time = "2026-04-22T19:14:54.192Z" }, +] + [[package]] name = "praw" version = "7.8.1"