Files
OpenJarvis/pyproject.toml
T
Jon Saad-FalconandClaude Opus 4.6 2606c283e9 feat(evals): add W&B and Google Sheets result trackers
Introduce ResultTracker ABC with pluggable experiment tracking for the
eval framework, enabling per-sample streaming to W&B and summary-row
appending to Google Sheets — critical for the NeurIPS paper sweep.

- ResultTracker ABC: on_run_start/on_result/on_summary/on_run_end
- WandbTracker: per-sample wandb.log with sample/ prefix, flattened
  MetricStats in run summary, reinit=True for suite mode
- SheetsTracker: summary-only (no per-sample API calls), 29-column
  canonical row, idempotent header, service account + ADC auth
- EvalRunner: trackers wired into lifecycle, all calls try/except
  wrapped so tracker failures never abort an eval run
- CLI: 7 new flags (--wandb-project/entity/tags/group,
  --sheets-id/worksheet/creds) on both jarvis eval run and
  python -m openjarvis.evals run
- Config: tracker fields parsed from TOML [run] section, propagated
  through expand_suite() for suite mode
- pyproject.toml: eval-wandb and eval-sheets optional dependency groups
- 9 new tests (lifecycle, crash resilience, mock W&B/Sheets)

Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
2026-03-04 01:00:47 +00:00

122 lines
3.1 KiB
TOML

[build-system]
requires = ["hatchling"]
build-backend = "hatchling.build"
[project]
name = "openjarvis"
version = "1.0.0"
description = "OpenJarvis — modular AI assistant backend with composable intelligence pillars"
readme = "README.md"
requires-python = ">=3.10"
dependencies = [
"click>=8",
"httpx>=0.27",
"pynvml>=13.0.1",
"rich>=13",
"tomli>=2.0; python_version < '3.11'",
]
[project.optional-dependencies]
dev = [
"pytest>=8",
"pytest-asyncio>=0.24",
"pytest-cov>=5",
"respx>=0.22",
"ruff>=0.4",
]
inference-mlx = ["mlx-lm>=0.19; sys_platform == 'darwin'"]
inference-cloud = [
"openai>=1.30",
"anthropic>=0.30",
]
inference-google = [
"google-genai>=1.0",
]
inference-litellm = ["litellm>=1.40"]
tools-search = [
"tavily-python>=0.3",
]
memory-faiss = [
"faiss-cpu>=1.7",
"sentence-transformers>=2.2",
"numpy>=1.24",
]
memory-colbert = [
"colbert-ai>=0.2",
"torch>=2.0",
]
memory-pdf = ["pdfplumber>=0.10"]
memory-bm25 = ["rank-bm25>=0.2.2"]
server = [
"fastapi>=0.110",
"uvicorn>=0.30",
"pydantic>=2.0",
]
openhands = ["openhands-sdk>=1.0; python_version >= '3.12'"]
gpu-metrics = ["pynvml>=12.0"]
energy-amd = ["amdsmi>=6.1"]
energy-apple = ["zeus-ml[apple]"]
energy-all = ["pynvml>=12.0", "amdsmi>=6.1", "zeus-ml[apple]"]
orchestrator-training = ["torch>=2.0", "transformers>=4.40"]
channel-telegram = ["python-telegram-bot>=21.0"]
channel-discord = ["discord.py>=2.3"]
channel-slack = ["slack-sdk>=3.27"]
channel-line = ["line-bot-sdk>=3.0"]
channel-viber = ["viberbot>=1.0"]
channel-messenger = ["pymessenger>=0.0.7"]
channel-reddit = ["praw>=7.0"]
channel-mastodon = ["Mastodon.py>=1.8"]
channel-xmpp = ["slixmpp>=1.8"]
channel-rocketchat = ["rocketchat-API>=1.30"]
channel-zulip = ["zulip>=0.9"]
channel-twitch = ["twitchio>=2.6"]
channel-nostr = ["pynostr>=0.6"]
browser = ["playwright>=1.40"]
media = ["openai>=1.30"]
pdf = ["pdfplumber>=0.10"]
scheduler = ["croniter>=2.0"]
security-signing = ["cryptography>=43"]
sandbox-wasm = ["wasmtime>=25"]
dashboard = ["textual>=0.80"]
speech = ["faster-whisper>=1.0"]
speech-deepgram = ["deepgram-sdk>=3.0"]
eval-wandb = ["wandb>=0.17"]
eval-sheets = ["gspread>=6.0", "google-auth>=2.0"]
docs = [
"mkdocs>=1.6",
"mkdocs-material>=9.5",
"mkdocstrings[python]>=0.25",
]
[project.scripts]
jarvis = "openjarvis.cli:main"
[tool.hatch.build.targets.wheel]
packages = ["src/openjarvis"]
[tool.hatch.build.targets.wheel.force-include]
"src/openjarvis/agents/claude_code_runner" = "openjarvis/agents/claude_code_runner"
"src/openjarvis/channels/whatsapp_baileys_bridge" = "openjarvis/channels/whatsapp_baileys_bridge"
[tool.pytest.ini_options]
testpaths = ["tests"]
markers = [
"live: requires running inference engine",
"cloud: requires cloud API keys",
"nvidia: requires NVIDIA GPU",
"amd: requires AMD GPU",
"apple: requires Apple Silicon",
"slow: long-running test",
]
[tool.ruff]
target-version = "py310"
src = ["src", "tests"]
[tool.ruff.lint]
select = ["E", "F", "I", "W"]
[tool.ruff.lint.per-file-ignores]
"src/openjarvis/evals/datasets/*.py" = ["E501"]
"src/openjarvis/evals/scorers/*.py" = ["E501"]