mirror of
https://github.com/open-jarvis/OpenJarvis.git
synced 2026-07-30 10:52:15 +00:00
The two eval-dataset suites that download real corpora from the HuggingFace Hub at runtime were running in the default CI lane. When the Hub was unreachable or rate-limited they failed and reddened `main` even though no code changed — confirmed by #506 (docs-only) failing on merge while its own PR run passed an hour earlier. They also dominated CI wall-time (~33 min of downloads + retry backoff on the failing run). Add a `hub` pytest marker, apply it to both suites via module-level `pytestmark`, and exclude it from the default CI lane (`-m "not live and not cloud and not hub"`). The tests stay runnable on demand with `pytest -m hub`. The ADP provider swallows per-config download errors and returns 0 records on a network failure, so a Hub outage surfaced there as `assert 1 <= 0` (not an exception) — it could not be made non-flaky by exception handling alone, only by gating. Coverage holds: removing these from CI drops total from 60.92% to ~60.73% (paranoid worst case 60.18%), still above the 60% gate. Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
96 lines
3.9 KiB
Python
96 lines
3.9 KiB
Python
"""Integration test: each provider's split kwarg produces disjoint train/test slices.
|
|
|
|
Marked ``hub``: every provider here downloads a real corpus from the
|
|
HuggingFace Hub, so the module is excluded from the default CI lane (which
|
|
runs ``-m "not live and not cloud and not hub"``). A transient Hub outage
|
|
or rate-limit raises connectivity errors (``LocalEntryNotFoundError``,
|
|
``HfHubHTTPError``) that the gated/not-found skip branch below does not
|
|
catch — so running these unguarded made ``main`` flaky-red. Run on demand
|
|
with ``pytest -m hub``.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import importlib
|
|
|
|
import pytest
|
|
|
|
pytestmark = pytest.mark.hub
|
|
|
|
PROVIDERS = [
|
|
("openjarvis.evals.datasets.pinchbench", "PinchBenchDataset"),
|
|
("openjarvis.evals.datasets.liveresearch", "LiveResearchBenchDataset"),
|
|
("openjarvis.evals.datasets.gaia", "GAIADataset"),
|
|
("openjarvis.evals.datasets.liveresearchbench", "LiveResearchBenchDataset"),
|
|
("openjarvis.evals.datasets.taubench", "TauBenchDataset"),
|
|
("openjarvis.evals.datasets.toolcall15", "ToolCall15Dataset"),
|
|
("openjarvis.evals.datasets.livecodebench", "LiveCodeBenchDataset"),
|
|
]
|
|
|
|
|
|
@pytest.mark.slow
|
|
@pytest.mark.parametrize("mod_name,cls_name", PROVIDERS)
|
|
def test_train_and_test_are_disjoint_per_provider(mod_name, cls_name):
|
|
mod = importlib.import_module(mod_name)
|
|
ds_cls = getattr(mod, cls_name)
|
|
|
|
# Some providers wrap gated HuggingFace datasets (GAIA,
|
|
# LiveResearchBench) or optional upstream packages (TauBench needs
|
|
# the ``tau2`` package). When the local environment lacks the
|
|
# required HF auth token or the optional install, the dataset
|
|
# raises rather than returns empty — skip cleanly so a CI runner
|
|
# without secrets doesn't fail the suite. The split-disjointness
|
|
# invariant under test is still exercised by every provider whose
|
|
# data IS available.
|
|
try:
|
|
train_ds = ds_cls()
|
|
train_ds.load(split="train", seed=42)
|
|
test_ds = ds_cls()
|
|
test_ds.load(split="test", seed=42)
|
|
except (ImportError, ModuleNotFoundError) as exc:
|
|
pytest.skip(f"{cls_name}: optional upstream not installed ({exc})")
|
|
except Exception as exc: # noqa: BLE001 — HF errors live in many modules
|
|
msg = str(exc)
|
|
if "GatedRepo" in type(exc).__name__ or "gated" in msg.lower():
|
|
pytest.skip(f"{cls_name}: gated dataset, no HF auth ({type(exc).__name__})")
|
|
if "DatasetNotFound" in type(exc).__name__:
|
|
pytest.skip(f"{cls_name}: dataset not accessible ({type(exc).__name__})")
|
|
raise
|
|
|
|
train_ids = {r.record_id for r in train_ds.iter_records()}
|
|
test_ids = {r.record_id for r in test_ds.iter_records()}
|
|
total = len(train_ids) + len(test_ids)
|
|
|
|
# A dataset with fewer than ~10 items can't produce a meaningful 20/80
|
|
# split where both sides are non-empty. In that regime we skip the
|
|
# disjointness check. But always reject a silent zero: if apply_split
|
|
# or an upstream filter produced an empty slice, fail loudly.
|
|
if total == 0:
|
|
pytest.fail(
|
|
f"{cls_name} returned 0 records for both train and test — "
|
|
f"likely a gate regression"
|
|
)
|
|
if total < 10:
|
|
pytest.skip("dataset too small for a 20/80 split")
|
|
|
|
assert train_ids.isdisjoint(test_ids)
|
|
assert total == train_ds.size() + test_ds.size()
|
|
|
|
|
|
@pytest.mark.slow
|
|
def test_toolcall15_split_is_nonempty():
|
|
"""Regression: toolcall15 must not silently return 0 records for split=train."""
|
|
from openjarvis.evals.datasets.toolcall15 import ToolCall15Dataset
|
|
|
|
ds = ToolCall15Dataset()
|
|
ds.load(split="train", seed=42)
|
|
assert ds.size() > 0, "toolcall15 train split should have >0 records"
|
|
|
|
ds_test = ToolCall15Dataset()
|
|
ds_test.load(split="test", seed=42)
|
|
assert ds_test.size() > 0, "toolcall15 test split should have >0 records"
|
|
|
|
train_ids = {r.record_id for r in ds.iter_records()}
|
|
test_ids = {r.record_id for r in ds_test.iter_records()}
|
|
assert train_ids.isdisjoint(test_ids)
|