"""CLI for the OpenJarvis evaluation framework.""" from __future__ import annotations import json import logging from pathlib import Path from typing import Optional import click from rich.console import Console from rich.progress import ( BarColumn, Progress, SpinnerColumn, TextColumn, TimeRemainingColumn, ) from evals.core.display import ( print_banner, print_completion, print_metrics_table, print_run_header, print_section, print_subject_table, print_suite_summary, ) # Registry of available benchmarks and their metadata BENCHMARKS = { "supergpqa": {"category": "reasoning", "description": "SuperGPQA multiple-choice"}, "gaia": {"category": "agentic", "description": "GAIA agentic benchmark"}, "frames": {"category": "rag", "description": "FRAMES multi-hop RAG"}, "wildchat": {"category": "chat", "description": "WildChat conversation quality"}, } BACKENDS = { "jarvis-direct": "Engine-level inference (local or cloud)", "jarvis-agent": "Agent-level inference with tool calling", } def _setup_logging(verbose: bool) -> None: level = logging.DEBUG if verbose else logging.INFO logging.basicConfig( level=level, format="%(asctime)s %(levelname)s %(name)s: %(message)s", datefmt="%H:%M:%S", ) def _build_backend(backend_name: str, engine_key: Optional[str], agent_name: str, tools: list[str], telemetry: bool = False, gpu_metrics: bool = False): """Construct the appropriate backend.""" if backend_name == "jarvis-agent": from evals.backends.jarvis_agent import JarvisAgentBackend return JarvisAgentBackend( engine_key=engine_key, agent_name=agent_name, tools=tools, telemetry=telemetry, gpu_metrics=gpu_metrics, ) else: from evals.backends.jarvis_direct import JarvisDirectBackend return JarvisDirectBackend( engine_key=engine_key, telemetry=telemetry, gpu_metrics=gpu_metrics, ) def _build_dataset(benchmark: str): """Construct the dataset provider for a benchmark.""" if benchmark == "supergpqa": from evals.datasets.supergpqa import SuperGPQADataset return SuperGPQADataset() elif benchmark == "gaia": from evals.datasets.gaia import GAIADataset return GAIADataset() elif benchmark == "frames": from evals.datasets.frames import FRAMESDataset return FRAMESDataset() elif benchmark == "wildchat": from evals.datasets.wildchat import WildChatDataset return WildChatDataset() else: raise click.ClickException(f"Unknown benchmark: {benchmark}") def _build_scorer(benchmark: str, judge_backend, judge_model: str): """Construct the scorer for a benchmark.""" if benchmark == "supergpqa": from evals.scorers.supergpqa_mcq import SuperGPQAScorer return SuperGPQAScorer(judge_backend, judge_model) elif benchmark == "gaia": from evals.scorers.gaia_exact import GAIAScorer return GAIAScorer(judge_backend, judge_model) elif benchmark == "frames": from evals.scorers.frames_judge import FRAMESScorer return FRAMESScorer(judge_backend, judge_model) elif benchmark == "wildchat": from evals.scorers.wildchat_judge import WildChatScorer return WildChatScorer(judge_backend, judge_model) else: raise click.ClickException(f"Unknown benchmark: {benchmark}") def _build_judge_backend(judge_model: str): """Build the judge backend (always cloud for LLM-as-judge).""" from evals.backends.jarvis_direct import JarvisDirectBackend return JarvisDirectBackend(engine_key="cloud") def _print_summary( summary, console: Optional[Console] = None, output_path: Optional[Path] = None, traces_dir: Optional[Path] = None, ) -> None: """Print a single run summary using Rich display primitives.""" if console is None: console = Console() print_section(console, "Results") print_metrics_table(console, summary) if summary.per_subject and len(summary.per_subject) > 1: print_subject_table(console, summary.per_subject) print_completion(console, summary, output_path, traces_dir) def _run_single(config, console: Optional[Console] = None) -> object: """Run a single eval from a RunConfig and return the summary.""" from evals.core.runner import EvalRunner if console is None: console = Console() eval_backend = _build_backend( config.backend, config.engine_key, config.agent_name or "orchestrator", config.tools, telemetry=getattr(config, "telemetry", False), gpu_metrics=getattr(config, "gpu_metrics", False), ) dataset = _build_dataset(config.benchmark) judge_backend = _build_judge_backend(config.judge_model) scorer = _build_scorer(config.benchmark, judge_backend, config.judge_model) runner = EvalRunner(config, dataset, eval_backend, scorer) try: num_samples = config.max_samples or 0 # Use progress bar if we know the sample count if num_samples > 0: with Progress( SpinnerColumn(), TextColumn("[progress.description]{task.description}"), BarColumn(), TextColumn("[progress.percentage]{task.percentage:>3.0f}%"), TimeRemainingColumn(), console=console, ) as progress: task = progress.add_task("Evaluating samples...", total=num_samples) summary = runner.run( progress_callback=lambda done, total: progress.update( task, completed=done, ), ) else: with console.status("Evaluating samples..."): summary = runner.run() return summary finally: eval_backend.close() judge_backend.close() def _run_from_config(config_path: str, verbose: bool) -> None: """Load a TOML config and run the full models x benchmarks matrix.""" from evals.core.config import expand_suite, load_eval_config console = Console() suite = load_eval_config(config_path) run_configs = expand_suite(suite) suite_name = suite.meta.name or Path(config_path).stem # Banner + configuration print_banner(console) print_section(console, "Suite Configuration") console.print( f" [cyan]Suite:[/cyan] {suite_name}" ) if suite.meta.description: console.print(f" [cyan]Description:[/cyan] {suite.meta.description}") console.print( f" [cyan]Matrix:[/cyan] {len(suite.models)} model(s) x " f"{len(suite.benchmarks)} benchmark(s) = {len(run_configs)} run(s)" ) # Ensure output directory exists output_dir = Path(suite.run.output_dir) output_dir.mkdir(parents=True, exist_ok=True) summaries = [] for i, rc in enumerate(run_configs, 1): print_section( console, f"Run {i}/{len(run_configs)}: {rc.benchmark} / {rc.model}", ) try: summary = _run_single(rc, console=console) summaries.append(summary) console.print( f" [green]{summary.accuracy:.4f}[/green] " f"({summary.correct}/{summary.scored_samples})" ) except Exception as exc: console.print(f" [red bold]FAILED:[/red bold] {exc}") # Print overall summary table if summaries: print_section(console, "Suite Results") print_suite_summary(console, summaries, suite_name) @click.group() def main(): """OpenJarvis Evaluation Framework.""" @main.command() @click.option("-c", "--config", "config_path", default=None, type=click.Path(), help="TOML config file for suite runs") @click.option("-b", "--benchmark", default=None, type=click.Choice(list(BENCHMARKS.keys())), help="Benchmark to run") @click.option("--backend", default="jarvis-direct", type=click.Choice(list(BACKENDS.keys())), help="Inference backend") @click.option("-m", "--model", default=None, help="Model identifier") @click.option("-e", "--engine", "engine_key", default=None, help="Engine key (ollama, vllm, cloud, ...)") @click.option("--agent", "agent_name", default="orchestrator", help="Agent name for jarvis-agent backend") @click.option("--tools", default="", help="Comma-separated tool names") @click.option("-n", "--max-samples", type=int, default=None, help="Maximum samples to evaluate") @click.option("-w", "--max-workers", type=int, default=4, help="Parallel workers") @click.option("--judge-model", default="gpt-5-mini-2025-08-07", help="LLM judge model") @click.option("-o", "--output", "output_path", default=None, help="Output JSONL path") @click.option("--seed", type=int, default=42, help="Random seed") @click.option("--split", "dataset_split", default=None, help="Dataset split override") @click.option("--temperature", type=float, default=0.0, help="Generation temperature") @click.option("--max-tokens", type=int, default=2048, help="Max output tokens") @click.option("--telemetry/--no-telemetry", default=False, help="Enable telemetry collection during eval") @click.option("--gpu-metrics/--no-gpu-metrics", default=False, help="Enable GPU metrics collection") @click.option("-v", "--verbose", is_flag=True, help="Verbose logging") @click.pass_context def run(ctx, config_path, benchmark, backend, model, engine_key, agent_name, tools, max_samples, max_workers, judge_model, output_path, seed, dataset_split, temperature, max_tokens, telemetry, gpu_metrics, verbose): """Run a single benchmark evaluation, or a full suite from a TOML config.""" _setup_logging(verbose) console = Console() # Config-driven mode if config_path is not None: _run_from_config(config_path, verbose) return # CLI-driven mode: validate required args if benchmark is None: raise click.UsageError( "Missing option '-b' / '--benchmark' " "(required when --config is not provided)" ) if model is None: raise click.UsageError( "Missing option '-m' / '--model' " "(required when --config is not provided)" ) from evals.core.types import RunConfig tool_list = [t.strip() for t in tools.split(",") if t.strip()] if tools else [] config = RunConfig( benchmark=benchmark, backend=backend, model=model, max_samples=max_samples, max_workers=max_workers, temperature=temperature, max_tokens=max_tokens, judge_model=judge_model, engine_key=engine_key, agent_name=agent_name, tools=tool_list, output_path=output_path, seed=seed, dataset_split=dataset_split, telemetry=telemetry, gpu_metrics=gpu_metrics, ) # Banner + config print_banner(console) print_section(console, "Configuration") print_run_header( console, benchmark=benchmark, model=model, backend=backend, samples=max_samples, workers=max_workers, ) # Evaluation print_section(console, "Evaluation") summary = _run_single(config, console=console) # Results _output_path = getattr(summary, "_output_path", None) _traces_dir = getattr(summary, "_traces_dir", None) _print_summary( summary, console=console, output_path=_output_path, traces_dir=_traces_dir, ) @main.command("run-all") @click.option("-m", "--model", required=True, help="Model identifier") @click.option("-e", "--engine", "engine_key", default=None, help="Engine key") @click.option("-n", "--max-samples", type=int, default=None, help="Max samples per benchmark") @click.option("-w", "--max-workers", type=int, default=4, help="Parallel workers") @click.option("--judge-model", default="gpt-5-mini-2025-08-07", help="LLM judge model") @click.option("--output-dir", default="results/", help="Output directory for results") @click.option("--seed", type=int, default=42, help="Random seed") @click.option("-v", "--verbose", is_flag=True, help="Verbose logging") def run_all(model, engine_key, max_samples, max_workers, judge_model, output_dir, seed, verbose): """Run all benchmarks.""" _setup_logging(verbose) from evals.core.runner import EvalRunner from evals.core.types import RunConfig console = Console() print_banner(console) print_section(console, "Suite Configuration") console.print( f" [cyan]Model:[/cyan] {model}\n" f" [cyan]Benchmarks:[/cyan] {', '.join(BENCHMARKS.keys())}\n" f" [cyan]Samples:[/cyan] {max_samples if max_samples else 'all'}" ) output_dir_path = Path(output_dir) output_dir_path.mkdir(parents=True, exist_ok=True) model_slug = model.replace("/", "-").replace(":", "-") summaries = [] for i, bench_name in enumerate(BENCHMARKS, 1): print_section(console, f"Run {i}/{len(BENCHMARKS)}: {bench_name}") output_path = output_dir_path / f"{bench_name}_{model_slug}.jsonl" config = RunConfig( benchmark=bench_name, backend="jarvis-direct", model=model, max_samples=max_samples, max_workers=max_workers, judge_model=judge_model, engine_key=engine_key, output_path=str(output_path), seed=seed, ) eval_backend = _build_backend("jarvis-direct", engine_key, "orchestrator", []) dataset = _build_dataset(bench_name) judge_backend = _build_judge_backend(judge_model) scorer = _build_scorer(bench_name, judge_backend, judge_model) runner = EvalRunner(config, dataset, eval_backend, scorer) try: if max_samples and max_samples > 0: with Progress( SpinnerColumn(), TextColumn("[progress.description]{task.description}"), BarColumn(), TextColumn("[progress.percentage]{task.percentage:>3.0f}%"), TimeRemainingColumn(), console=console, ) as progress: task = progress.add_task( f"Evaluating {bench_name}...", total=max_samples, ) summary = runner.run( progress_callback=lambda done, total: progress.update( task, completed=done, ), ) else: with console.status(f"Evaluating {bench_name}..."): summary = runner.run() summaries.append(summary) console.print( f" [green]{summary.accuracy:.4f}[/green] " f"({summary.correct}/{summary.scored_samples})" ) except Exception as exc: console.print(f" [red bold]FAILED:[/red bold] {exc}") finally: eval_backend.close() judge_backend.close() # Print overall summary if summaries: print_section(console, "Suite Results") print_suite_summary(console, summaries, f"All Benchmarks / {model}") @main.command() @click.argument("jsonl_path", type=click.Path(exists=True)) def summarize(jsonl_path): """Summarize results from a JSONL output file.""" records = [] with open(jsonl_path) as f: for line in f: line = line.strip() if line: records.append(json.loads(line)) if not records: click.echo("No records found.") return console = Console() total = len(records) scored = [r for r in records if r.get("is_correct") is not None] correct = [r for r in scored if r["is_correct"]] errors = [r for r in records if r.get("error")] accuracy = len(correct) / len(scored) if scored else 0.0 console.print(f"[cyan]File:[/cyan] {jsonl_path}") console.print(f"[cyan]Benchmark:[/cyan] {records[0].get('benchmark', '?')}") console.print(f"[cyan]Model:[/cyan] {records[0].get('model', '?')}") console.print(f"[cyan]Total:[/cyan] {total}") console.print(f"[cyan]Scored:[/cyan] {len(scored)}") console.print(f"[cyan]Correct:[/cyan] {len(correct)}") console.print(f"[cyan]Accuracy:[/cyan] [bold]{accuracy:.4f}[/bold]") console.print(f"[cyan]Errors:[/cyan] {len(errors)}") @main.command("list") def list_cmd(): """List available benchmarks and backends.""" console = Console() print_banner(console) from rich.table import Table bench_table = Table( title="[bold]Available Benchmarks[/bold]", border_style="bright_blue", title_style="bold cyan", ) bench_table.add_column("Name", style="cyan", no_wrap=True) bench_table.add_column("Category", style="white") bench_table.add_column("Description") for name, info in BENCHMARKS.items(): bench_table.add_row(name, info["category"], info["description"]) console.print(bench_table) backend_table = Table( title="[bold]Available Backends[/bold]", border_style="bright_blue", title_style="bold cyan", ) backend_table.add_column("Name", style="cyan", no_wrap=True) backend_table.add_column("Description") for name, desc in BACKENDS.items(): backend_table.add_row(name, desc) console.print(backend_table) if __name__ == "__main__": main()