mirror of
https://github.com/open-jarvis/OpenJarvis.git
synced 2026-07-30 19:02:16 +00:00
Trace-driven learning pipeline: - TrainingDataMiner: extract SFT/routing/agent pairs from traces - LoRATrainer: fine-tune local models from trace-derived data - AgentConfigEvolver: rewrite agent configs from trace analysis - LearningOrchestrator: coordinate mine→train→evolve cycle, wired into SystemBuilder Eval framework (15 real IPW benchmarks): - Datasets: SuperGPQA, GPQA, MMLU-Pro, MATH-500, Natural Reasoning, HLE, SimpleQA, WildChat, IPW, GAIA, FRAMES, SWE-bench, SWEfficiency, TerminalBench, TerminalBench Native - Scorers: MCQ extraction, LLM-judge, exact match, structural validation - CLI: jarvis eval list|run|compare|report Composable abstractions: - Recipe system: TOML composition of all 5 pillars (3 built-in recipes) - Agent templates: 15 pre-configured TOML manifests with system prompts - Bundled skills: 20 ready-to-use TOML skill manifests - Operator recipes: researcher (4h), correspondent (5min), sentinel (2h) 102 files changed, ~11,500 lines added. 3241 tests pass (44 skipped). Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
68 lines
2.2 KiB
Python
68 lines
2.2 KiB
Python
"""TerminalBench Native scorer — test-result-based evaluation.
|
|
|
|
Reads ``is_resolved`` and ``test_results`` from the record's metadata
|
|
(populated by the native terminal-bench harness) and returns a
|
|
deterministic pass/fail without any LLM judging.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from typing import Any, Dict, Optional, Tuple
|
|
|
|
from evals.core.scorer import Scorer
|
|
from evals.core.types import EvalRecord
|
|
|
|
|
|
class TerminalBenchNativeScorer(Scorer):
|
|
"""Test-result-based scorer for TerminalBench Native tasks.
|
|
|
|
The native terminal-bench package produces ``is_resolved`` and
|
|
``test_results`` fields after executing a task. This scorer reads
|
|
those fields from ``record.metadata`` and translates them into the
|
|
standard ``(is_correct, meta)`` tuple.
|
|
"""
|
|
|
|
scorer_id = "terminalbench-native"
|
|
|
|
def __init__(
|
|
self,
|
|
judge_backend: object = None,
|
|
judge_model: str = "",
|
|
) -> None:
|
|
# Accept judge_backend/judge_model so the CLI factory pattern works,
|
|
# but they are unused — scoring is based on test results.
|
|
self._judge_backend = judge_backend
|
|
self._judge_model = judge_model
|
|
|
|
def score(
|
|
self, record: EvalRecord, model_answer: str,
|
|
) -> Tuple[Optional[bool], Dict[str, Any]]:
|
|
meta = record.metadata
|
|
|
|
is_resolved = meta.get("is_resolved")
|
|
test_results = meta.get("test_results")
|
|
|
|
# If neither field is present, we cannot determine correctness.
|
|
if is_resolved is None and test_results is None:
|
|
return None, {"reason": "no_test_results"}
|
|
|
|
# Build informative metadata from available test output.
|
|
result_meta: Dict[str, Any] = {}
|
|
if test_results is not None:
|
|
result_meta["test_results"] = test_results
|
|
|
|
# Determine pass/fail
|
|
if is_resolved is not None:
|
|
is_correct = bool(is_resolved)
|
|
result_meta["is_resolved"] = is_resolved
|
|
return is_correct, result_meta
|
|
|
|
# Fallback: if only test_results is present, treat as indeterminate.
|
|
return None, {
|
|
"reason": "is_resolved_missing",
|
|
"test_results": test_results,
|
|
}
|
|
|
|
|
|
__all__ = ["TerminalBenchNativeScorer"]
|