From 1d7f0cc62dd62e7d5963577d73cfee6bddde019a Mon Sep 17 00:00:00 2001 From: Steven Enamakel <31011319+senamakel@users.noreply.github.com> Date: Sat, 4 Jul 2026 21:02:48 -0700 Subject: [PATCH] [codex] Port TokenJuice engine to vendored TinyJuice (#4523) --- .github/workflows/release-production.yml | 2 +- .github/workflows/release-staging.yml | 2 +- .gitmodules | 3 + Cargo.lock | 24 + Cargo.toml | 10 +- README.md | 2 +- app/src-tauri/Cargo.lock | 21 + app/src-tauri/Cargo.toml | 2 + .../developing/architecture/agent-harness.md | 2 +- gitbooks/features/token-compression.md | 10 +- src/openhuman/config/schema/tokenjuice.rs | 8 +- src/openhuman/tokenjuice/README.md | 92 +- src/openhuman/tokenjuice/cache/marker.rs | 133 -- src/openhuman/tokenjuice/cache/mod.rs | 13 - src/openhuman/tokenjuice/cache/store.rs | 474 ----- src/openhuman/tokenjuice/classify.rs | 590 ------ src/openhuman/tokenjuice/compress.rs | 231 --- src/openhuman/tokenjuice/compressors/code.rs | 359 ---- src/openhuman/tokenjuice/compressors/diff.rs | 220 -- .../tokenjuice/compressors/generic.rs | 39 - src/openhuman/tokenjuice/compressors/html.rs | 262 --- src/openhuman/tokenjuice/compressors/json.rs | 310 --- src/openhuman/tokenjuice/compressors/log.rs | 320 --- .../tokenjuice/compressors/ml_text.rs | 48 - src/openhuman/tokenjuice/compressors/mod.rs | 77 - .../tokenjuice/compressors/search.rs | 221 -- .../tokenjuice/compressors/signals.rs | 127 -- src/openhuman/tokenjuice/detect/hint.rs | 119 -- src/openhuman/tokenjuice/detect/kind.rs | 453 ---- src/openhuman/tokenjuice/detect/mod.rs | 10 - src/openhuman/tokenjuice/ml/mod.rs | 4 +- src/openhuman/tokenjuice/mod.rs | 85 +- src/openhuman/tokenjuice/reduce.rs | 939 --------- src/openhuman/tokenjuice/reduce_tests.rs | 1813 ----------------- src/openhuman/tokenjuice/rules/builtin.rs | 387 ---- .../tokenjuice/rules/builtin_tests.rs | 185 -- src/openhuman/tokenjuice/rules/compiler.rs | 310 --- src/openhuman/tokenjuice/rules/loader.rs | 273 --- .../tokenjuice/rules/loader_tests.rs | 273 --- src/openhuman/tokenjuice/rules/mod.rs | 8 - .../fixtures/cargo_test_failure.fixture.json | 10 - .../fallback_long_output.fixture.json | 9 - .../fixtures/git_status_modified.fixture.json | 9 - src/openhuman/tokenjuice/text/ansi.rs | 87 - src/openhuman/tokenjuice/text/mod.rs | 12 - src/openhuman/tokenjuice/text/process.rs | 393 ---- src/openhuman/tokenjuice/text/width.rs | 241 --- src/openhuman/tokenjuice/text_tests.rs | 442 ---- src/openhuman/tokenjuice/tokens.rs | 46 - src/openhuman/tokenjuice/tool_integration.rs | 407 ---- src/openhuman/tokenjuice/types.rs | 608 ------ src/openhuman/tokenjuice/vendor/README.md | 68 - .../tokenjuice/vendor/rules/archive__tar.json | 30 - .../vendor/rules/archive__unzip.json | 30 - .../tokenjuice/vendor/rules/archive__zip.json | 30 - .../vendor/rules/build__cargo-build.json | 58 - .../vendor/rules/build__cargo-doc.json | 55 - .../vendor/rules/build__esbuild.json | 35 - .../tokenjuice/vendor/rules/build__tsc.json | 74 - .../vendor/rules/build__tsdown.json | 35 - .../tokenjuice/vendor/rules/build__vite.json | 42 - .../vendor/rules/build__webpack.json | 51 - .../tokenjuice/vendor/rules/cloud__aws.json | 30 - .../tokenjuice/vendor/rules/cloud__az.json | 30 - .../vendor/rules/cloud__flyctl.json | 30 - .../vendor/rules/cloud__gcloud.json | 30 - .../tokenjuice/vendor/rules/cloud__gh.json | 30 - .../vendor/rules/cloud__vercel.json | 30 - .../vendor/rules/database__mongosh.json | 30 - .../vendor/rules/database__mysql.json | 30 - .../vendor/rules/database__psql.json | 30 - .../vendor/rules/database__redis-cli.json | 30 - .../vendor/rules/database__sqlite3.json | 30 - .../vendor/rules/devops__docker-build.json | 53 - .../vendor/rules/devops__docker-compose.json | 44 - .../vendor/rules/devops__docker-images.json | 30 - .../vendor/rules/devops__docker-logs.json | 43 - .../vendor/rules/devops__docker-ps.json | 30 - .../rules/devops__kubectl-describe.json | 45 - .../vendor/rules/devops__kubectl-get.json | 35 - .../vendor/rules/devops__kubectl-logs.json | 43 - .../vendor/rules/filesystem__find.json | 42 - .../vendor/rules/filesystem__ls.json | 29 - .../vendor/rules/generic__fallback.json | 32 - .../vendor/rules/generic__help.json | 25 - .../tokenjuice/vendor/rules/git__branch.json | 29 - .../vendor/rules/git__diff-name-only.json | 29 - .../vendor/rules/git__diff-stat.json | 37 - .../vendor/rules/git__log-oneline.json | 30 - .../vendor/rules/git__remote-v.json | 29 - .../tokenjuice/vendor/rules/git__show.json | 50 - .../vendor/rules/git__stash-list.json | 30 - .../tokenjuice/vendor/rules/git__status.json | 53 - .../vendor/rules/install__bun-install.json | 43 - .../vendor/rules/install__npm-install.json | 49 - .../vendor/rules/install__pnpm-install.json | 43 - .../vendor/rules/install__yarn-install.json | 42 - .../tokenjuice/vendor/rules/lint__biome.json | 35 - .../vendor/rules/lint__cargo-clippy.json | 58 - .../vendor/rules/lint__cargo-fmt.json | 50 - .../tokenjuice/vendor/rules/lint__eslint.json | 45 - .../tokenjuice/vendor/rules/lint__oxlint.json | 35 - .../vendor/rules/lint__prettier-check.json | 34 - .../vendor/rules/media__ffmpeg.json | 30 - .../vendor/rules/media__mediainfo.json | 30 - .../vendor/rules/network__curl.json | 30 - .../tokenjuice/vendor/rules/network__dig.json | 30 - .../vendor/rules/network__nslookup.json | 30 - .../vendor/rules/network__ping.json | 35 - .../tokenjuice/vendor/rules/network__ssh.json | 30 - .../vendor/rules/network__traceroute.json | 30 - .../vendor/rules/network__wget.json | 30 - .../vendor/rules/observability__free.json | 30 - .../vendor/rules/observability__htop.json | 30 - .../vendor/rules/observability__iostat.json | 30 - .../vendor/rules/observability__top.json | 30 - .../vendor/rules/observability__vmstat.json | 30 - .../vendor/rules/package__apt-install.json | 36 - .../vendor/rules/package__apt-upgrade.json | 36 - .../vendor/rules/package__brew-install.json | 31 - .../vendor/rules/package__brew-upgrade.json | 31 - .../vendor/rules/package__dnf-install.json | 31 - .../vendor/rules/package__yum-install.json | 31 - .../vendor/rules/search__git-grep.json | 30 - .../tokenjuice/vendor/rules/search__grep.json | 38 - .../tokenjuice/vendor/rules/search__rg.json | 38 - .../vendor/rules/service__journalctl.json | 42 - .../vendor/rules/service__launchctl.json | 41 - .../vendor/rules/service__lsof.json | 29 - .../vendor/rules/service__netstat.json | 29 - .../vendor/rules/service__service.json | 46 - .../tokenjuice/vendor/rules/service__ss.json | 29 - .../rules/service__systemctl-status.json | 45 - .../tokenjuice/vendor/rules/system__df.json | 29 - .../tokenjuice/vendor/rules/system__du.json | 29 - .../tokenjuice/vendor/rules/system__file.json | 30 - .../tokenjuice/vendor/rules/system__ps.json | 29 - .../tokenjuice/vendor/rules/task__just.json | 30 - .../tokenjuice/vendor/rules/task__make.json | 30 - .../vendor/rules/tests__bun-test.json | 56 - .../vendor/rules/tests__cargo-test.json | 41 - .../vendor/rules/tests__go-test.json | 49 - .../tokenjuice/vendor/rules/tests__jest.json | 41 - .../tokenjuice/vendor/rules/tests__mocha.json | 35 - .../vendor/rules/tests__npm-test.json | 56 - .../vendor/rules/tests__playwright.json | 37 - .../vendor/rules/tests__pnpm-test.json | 56 - .../vendor/rules/tests__pytest.json | 54 - .../vendor/rules/tests__vitest.json | 55 - .../vendor/rules/tests__yarn-test.json | 56 - .../vendor/rules/transfer__rsync.json | 30 - .../vendor/rules/transfer__scp.json | 30 - tests/agent_harness_e2e.rs | 18 +- tests/agent_large_round25_raw_coverage_e2e.rs | 3 +- tests/tokenjuice_integration.rs | 6 +- vendor/tinyjuice | 1 + 156 files changed, 133 insertions(+), 14391 deletions(-) delete mode 100644 src/openhuman/tokenjuice/cache/marker.rs delete mode 100644 src/openhuman/tokenjuice/cache/mod.rs delete mode 100644 src/openhuman/tokenjuice/cache/store.rs delete mode 100644 src/openhuman/tokenjuice/classify.rs delete mode 100644 src/openhuman/tokenjuice/compress.rs delete mode 100644 src/openhuman/tokenjuice/compressors/code.rs delete mode 100644 src/openhuman/tokenjuice/compressors/diff.rs delete mode 100644 src/openhuman/tokenjuice/compressors/generic.rs delete mode 100644 src/openhuman/tokenjuice/compressors/html.rs delete mode 100644 src/openhuman/tokenjuice/compressors/json.rs delete mode 100644 src/openhuman/tokenjuice/compressors/log.rs delete mode 100644 src/openhuman/tokenjuice/compressors/ml_text.rs delete mode 100644 src/openhuman/tokenjuice/compressors/mod.rs delete mode 100644 src/openhuman/tokenjuice/compressors/search.rs delete mode 100644 src/openhuman/tokenjuice/compressors/signals.rs delete mode 100644 src/openhuman/tokenjuice/detect/hint.rs delete mode 100644 src/openhuman/tokenjuice/detect/kind.rs delete mode 100644 src/openhuman/tokenjuice/detect/mod.rs delete mode 100644 src/openhuman/tokenjuice/reduce.rs delete mode 100644 src/openhuman/tokenjuice/reduce_tests.rs delete mode 100644 src/openhuman/tokenjuice/rules/builtin.rs delete mode 100644 src/openhuman/tokenjuice/rules/builtin_tests.rs delete mode 100644 src/openhuman/tokenjuice/rules/compiler.rs delete mode 100644 src/openhuman/tokenjuice/rules/loader.rs delete mode 100644 src/openhuman/tokenjuice/rules/loader_tests.rs delete mode 100644 src/openhuman/tokenjuice/rules/mod.rs delete mode 100644 src/openhuman/tokenjuice/tests/fixtures/cargo_test_failure.fixture.json delete mode 100644 src/openhuman/tokenjuice/tests/fixtures/fallback_long_output.fixture.json delete mode 100644 src/openhuman/tokenjuice/tests/fixtures/git_status_modified.fixture.json delete mode 100644 src/openhuman/tokenjuice/text/ansi.rs delete mode 100644 src/openhuman/tokenjuice/text/mod.rs delete mode 100644 src/openhuman/tokenjuice/text/process.rs delete mode 100644 src/openhuman/tokenjuice/text/width.rs delete mode 100644 src/openhuman/tokenjuice/text_tests.rs delete mode 100644 src/openhuman/tokenjuice/tokens.rs delete mode 100644 src/openhuman/tokenjuice/tool_integration.rs delete mode 100644 src/openhuman/tokenjuice/types.rs delete mode 100644 src/openhuman/tokenjuice/vendor/README.md delete mode 100644 src/openhuman/tokenjuice/vendor/rules/archive__tar.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/archive__unzip.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/archive__zip.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/build__cargo-build.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/build__cargo-doc.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/build__esbuild.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/build__tsc.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/build__tsdown.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/build__vite.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/build__webpack.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/cloud__aws.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/cloud__az.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/cloud__flyctl.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/cloud__gcloud.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/cloud__gh.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/cloud__vercel.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/database__mongosh.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/database__mysql.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/database__psql.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/database__redis-cli.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/database__sqlite3.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/devops__docker-build.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/devops__docker-compose.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/devops__docker-images.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/devops__docker-logs.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/devops__docker-ps.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/devops__kubectl-describe.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/devops__kubectl-get.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/devops__kubectl-logs.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/filesystem__find.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/filesystem__ls.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/generic__fallback.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/generic__help.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/git__branch.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/git__diff-name-only.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/git__diff-stat.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/git__log-oneline.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/git__remote-v.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/git__show.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/git__stash-list.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/git__status.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/install__bun-install.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/install__npm-install.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/install__pnpm-install.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/install__yarn-install.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/lint__biome.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/lint__cargo-clippy.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/lint__cargo-fmt.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/lint__eslint.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/lint__oxlint.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/lint__prettier-check.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/media__ffmpeg.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/media__mediainfo.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/network__curl.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/network__dig.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/network__nslookup.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/network__ping.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/network__ssh.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/network__traceroute.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/network__wget.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/observability__free.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/observability__htop.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/observability__iostat.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/observability__top.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/observability__vmstat.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/package__apt-install.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/package__apt-upgrade.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/package__brew-install.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/package__brew-upgrade.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/package__dnf-install.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/package__yum-install.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/search__git-grep.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/search__grep.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/search__rg.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/service__journalctl.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/service__launchctl.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/service__lsof.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/service__netstat.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/service__service.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/service__ss.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/service__systemctl-status.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/system__df.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/system__du.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/system__file.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/system__ps.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/task__just.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/task__make.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/tests__bun-test.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/tests__cargo-test.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/tests__go-test.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/tests__jest.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/tests__mocha.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/tests__npm-test.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/tests__playwright.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/tests__pnpm-test.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/tests__pytest.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/tests__vitest.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/tests__yarn-test.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/transfer__rsync.json delete mode 100644 src/openhuman/tokenjuice/vendor/rules/transfer__scp.json create mode 160000 vendor/tinyjuice diff --git a/.github/workflows/release-production.yml b/.github/workflows/release-production.yml index 3a3510be2..476e8867b 100644 --- a/.github/workflows/release-production.yml +++ b/.github/workflows/release-production.yml @@ -418,7 +418,7 @@ jobs: # fork the core image doesn't need. The Dockerfile COPYs vendor/ because # [patch.crates-io] resolves Rust SDK crates from vendor/. - name: Init vendored Rust submodules - run: git submodule update --init vendor/tinyagents vendor/tinyflows vendor/tinycortex + run: git submodule update --init vendor/tinyagents vendor/tinyflows vendor/tinycortex vendor/tinyjuice - name: Set up Docker Buildx uses: docker/setup-buildx-action@v3 - name: Log in to GHCR diff --git a/.github/workflows/release-staging.yml b/.github/workflows/release-staging.yml index cef9d1dde..34b27b67c 100644 --- a/.github/workflows/release-staging.yml +++ b/.github/workflows/release-staging.yml @@ -283,7 +283,7 @@ jobs: # fork the core image doesn't need. The Dockerfile COPYs vendor/ because # [patch.crates-io] resolves Rust SDK crates from vendor/. - name: Init vendored Rust submodules - run: git submodule update --init vendor/tinyagents vendor/tinyflows vendor/tinycortex + run: git submodule update --init vendor/tinyagents vendor/tinyflows vendor/tinycortex vendor/tinyjuice - name: Set up Docker Buildx uses: docker/setup-buildx-action@v3 - name: Build image (no push) diff --git a/.gitmodules b/.gitmodules index 670ad0703..c07bfb8cf 100644 --- a/.gitmodules +++ b/.gitmodules @@ -14,3 +14,6 @@ [submodule "vendor/tinycortex"] path = vendor/tinycortex url = https://github.com/tinyhumansai/tinycortex +[submodule "vendor/tinyjuice"] + path = vendor/tinyjuice + url = https://github.com/tinyhumansai/tinyjuice diff --git a/Cargo.lock b/Cargo.lock index c4d05c915..8ad1c25cc 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -4395,6 +4395,7 @@ dependencies = [ "tinyagents", "tinycortex", "tinyflows", + "tinyjuice", "tinyplace", "tokio", "tokio-rustls", @@ -6759,6 +6760,29 @@ dependencies = [ "tracing", ] +[[package]] +name = "tinyjuice" +version = "0.1.0" +dependencies = [ + "async-trait", + "dirs 5.0.1", + "hex", + "log", + "once_cell", + "regex", + "serde", + "serde_json", + "sha2 0.10.9", + "thiserror 2.0.18", + "tokio", + "tree-sitter", + "tree-sitter-python", + "tree-sitter-rust", + "tree-sitter-typescript", + "unicode-segmentation", + "unicode-width", +] + [[package]] name = "tinyplace" version = "1.0.1" diff --git a/Cargo.toml b/Cargo.toml index 9203eb4ac..4a3bdf813 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -46,6 +46,10 @@ tinyplace = "1.0.1" # (same version openhuman already uses — no conflict). Published on crates.io # and patched below to the vendored submodule. tinyflows = "0.3" +# TinyJuice — host-agnostic TokenJuice compression engine. OpenHuman keeps +# config/RPC/tool/runtime adapters in `src/openhuman/tokenjuice/` and patches +# this dependency to the vendored submodule below. +tinyjuice = { version = "0.1", default-features = false } # TinyAgents — Rust LLM orchestration framework (LangGraph/LangChain-style): # durable state graphs, agent-loop harness, model/tool registries, REPL + # `.rag` workflow language. openhuman's agent engine + orchestration run on this @@ -318,10 +322,7 @@ default = ["tokenjuice-treesitter"] # AST-aware code compression (tree-sitter Rust/TS/Python grammars; C build). # On by default; disable to fall back to the brace-depth heuristic. tokenjuice-treesitter = [ - "dep:tree-sitter", - "dep:tree-sitter-rust", - "dep:tree-sitter-typescript", - "dep:tree-sitter-python", + "tinyjuice/tokenjuice-treesitter", ] sandbox-landlock = ["dep:landlock"] sandbox-bubblewrap = [] @@ -354,6 +355,7 @@ tinyagents = { path = "vendor/tinyagents" } # test workflow and memory-layer changes against OpenHuman before publishing. tinyflows = { path = "vendor/tinyflows" } tinycortex = { path = "vendor/tinycortex" } +tinyjuice = { path = "vendor/tinyjuice" } # Emit just enough DWARF in release builds for Sentry to symbolicate Rust # panics + render surrounding source lines. `line-tables-only` keeps the diff --git a/README.md b/README.md index 6d418fea1..807a50439 100644 --- a/README.md +++ b/README.md @@ -158,7 +158,7 @@ High-level comparison (products evolve, so verify against each vendor). OpenHuma ## Contributing from source -New contributor? Start with [`CONTRIBUTING.md`](./CONTRIBUTING.md) for the fork/PR workflow and local validation commands, or use the copy-paste AI-agent prompt in [`CONTRIBUTING-BEGINNERS.md`](./CONTRIBUTING-BEGINNERS.md#optional-let-an-ai-coding-agent-guide-you). The short path is: +New contributor? Start with [`CONTRIBUTING.md`](./CONTRIBUTING.md) for the fork/PR workflow and local validation commands, or use the copy-paste AI-agent prompt in [`CONTRIBUTING-BEGINNERS.md`](./CONTRIBUTING-BEGINNERS.md#optional--let-an-ai-coding-agent-guide-you). The short path is: 1. Install Git, Node.js 24+, pnpm 10.10.0, Rust 1.93.0 (`rustfmt` + `clippy`), CMake, Ninja, ripgrep, and the platform desktop build prerequisites. 2. Fork and clone the repo, then run `git submodule update --init --recursive` before `pnpm install` so the vendored Tauri/CEF sources are present. diff --git a/app/src-tauri/Cargo.lock b/app/src-tauri/Cargo.lock index 28b58e793..d018a2ef4 100644 --- a/app/src-tauri/Cargo.lock +++ b/app/src-tauri/Cargo.lock @@ -47,6 +47,7 @@ dependencies = [ "tauri-runtime-cef", "tempfile", "tiny-skia", + "tinyjuice", "tokio", "tokio-tungstenite 0.24.0", "tokio-util", @@ -5636,6 +5637,7 @@ dependencies = [ "thiserror 2.0.18", "tinyagents", "tinyflows", + "tinyjuice", "tinyplace", "tokio", "tokio-rustls", @@ -9106,6 +9108,25 @@ dependencies = [ "tracing", ] +[[package]] +name = "tinyjuice" +version = "0.1.0" +dependencies = [ + "async-trait", + "dirs 5.0.1", + "hex", + "log", + "once_cell", + "regex", + "serde", + "serde_json", + "sha2 0.10.9", + "thiserror 2.0.18", + "tokio", + "unicode-segmentation", + "unicode-width", +] + [[package]] name = "tinyplace" version = "1.0.1" diff --git a/app/src-tauri/Cargo.toml b/app/src-tauri/Cargo.toml index 98af179e4..0258d69d3 100644 --- a/app/src-tauri/Cargo.toml +++ b/app/src-tauri/Cargo.toml @@ -133,6 +133,7 @@ cef = { version = "=146.4.1", default-features = false } # probe in `core_process::ensure_running` still attaches to a running # `openhuman-core run` harness when one is already listening. openhuman_core = { path = "../..", package = "openhuman", default-features = false } +tinyjuice = { version = "0.1", default-features = false } [target.'cfg(unix)'.dependencies] nix = { version = "0.29", default-features = false, features = ["hostname", "signal", "user"] } @@ -215,6 +216,7 @@ tinyagents = { path = "../../vendor/tinyagents" } # test workflow and memory-layer changes against OpenHuman before publishing. tinyflows = { path = "../../vendor/tinyflows" } tinycortex = { path = "../../vendor/tinycortex" } +tinyjuice = { path = "../../vendor/tinyjuice" } # CEF support lives on the `feat/cef` branch of tauri-apps/tauri. We carry our # own fork at tinyhumansai/tauri-cef on `feat/cef-notification-intercept` which diff --git a/gitbooks/developing/architecture/agent-harness.md b/gitbooks/developing/architecture/agent-harness.md index a61a84877..efa6ac34e 100644 --- a/gitbooks/developing/architecture/agent-harness.md +++ b/gitbooks/developing/architecture/agent-harness.md @@ -171,7 +171,7 @@ When a tool result exceeds the summarizer's threshold, it gets routed through a ### TokenJuice - content-aware tool-output compaction (Stage 1a) -Before a fresh tool result enters history (and ahead of the byte-budget backstop), it passes through the **TokenJuice content router** (`src/openhuman/tokenjuice/`). Inspired by Headroom, the router *detects the content kind* (JSON, code, log, search, diff, HTML, plain text) from the bytes and/or a hint derived from the tool name and arguments, then dispatches to a specialised compressor: +Before a fresh tool result enters history (and ahead of the byte-budget backstop), it passes through the **TokenJuice content router** in the vendored TinyJuice crate (`vendor/tinyjuice`), with OpenHuman adapters in `src/openhuman/tokenjuice/`. Inspired by Headroom, the router *detects the content kind* (JSON, code, log, search, diff, HTML, plain text) from the bytes and/or a hint derived from the tool name and arguments, then dispatches to a specialised compressor: * **JSON** → SmartCrusher: array-of-objects → table (each key once), preserving rows that carry errors or numeric outliers. * **Code** → tree-sitter (Rust/TS/JS/Python) signature keeper that collapses function bodies; brace-depth heuristic fallback. diff --git a/gitbooks/features/token-compression.md b/gitbooks/features/token-compression.md index f14a12c4c..0fd6fe7bd 100644 --- a/gitbooks/features/token-compression.md +++ b/gitbooks/features/token-compression.md @@ -18,7 +18,7 @@ It began as a port of [vincentkoc/tokenjuice](https://github.com/vincentkoc/toke ## The pipeline, step by step Every blob that flows through the policy-aware TokenJuice tool-output adapters -takes the same path (`src/openhuman/tokenjuice/compress.rs`): +takes the same path through the vendored TinyJuice router (`vendor/tinyjuice/src/compress.rs`): ```text raw tool result @@ -60,7 +60,7 @@ raw tool result ## The compressors -Each content kind has a purpose-built compressor (`src/openhuman/tokenjuice/compressors/`): +Each content kind has a purpose-built compressor (`vendor/tinyjuice/src/compressors/`): | Compressor | Kind | What it does | | ---------------- | ----------- | -------------------------------------------------------------------------------------------------------- | @@ -79,7 +79,7 @@ Multi-byte text (CJK, emoji, combining marks) is handled grapheme-by-grapheme th ## ML compression (opt-in) -Beyond the deterministic compressors, TokenJuice can route plain text through a **ModernBERT** token-salience model that scores and drops low-information spans (`src/openhuman/tokenjuice/ml/`). +Beyond the deterministic compressors, TokenJuice can route plain text through a **ModernBERT** token-salience model that scores and drops low-information spans. The TinyJuice compressor exposes the optional ML slot, and OpenHuman bridges it to Kompress in `src/openhuman/tokenjuice/ml/`. * **Off by default.** Enable with `ml_compression_enabled = true` in `[tokenjuice]`. * **Runs locally** as the `kompress` backend of the shared Python runtime sidecar. No data leaves your machine. @@ -90,7 +90,7 @@ Beyond the deterministic compressors, TokenJuice can route plain text through a ## Nothing is lost: CCR cache & retrieval -Lossy compression would normally mean throwing data away. TokenJuice instead **offloads** the full original into the **Compress-Cache-Retrieve (CCR)** store and leaves a breadcrumb (`src/openhuman/tokenjuice/cache/`). +Lossy compression would normally mean throwing data away. TokenJuice instead **offloads** the full original into the **Compress-Cache-Retrieve (CCR)** store and leaves a breadcrumb (`vendor/tinyjuice/src/cache/`). * **In-memory tier** (always on): a process-global store keyed by SHA-256 hash, bounded by entry count (`max_cache_entries`, default 256) and total bytes (`max_cache_bytes`, default 64 MiB), FIFO eviction. * **On-disk tier** (optional): `/.tokenjuice/ccr/`, enabled with `ccr_disk_enabled`, survives memory eviction. Optional TTL via `ccr_ttl_secs`. @@ -103,7 +103,7 @@ So the agent gets the cheap compacted view by default, and can transparently "zo ## Savings tracking -Every compression is metered (`src/openhuman/tokenjuice/savings.rs`). TokenJuice tracks events, original vs. compacted tokens, tokens saved, and **estimated cost saved in USD** (using per-model input pricing), aggregated as `total`, `by_model`, and `by_compressor`. Stats persist to `/state/tokenjuice_savings.json` and survive restarts. +Every compression is metered by an OpenHuman savings callback (`src/openhuman/tokenjuice/savings.rs`). TokenJuice reports events and token deltas; OpenHuman applies per-model input pricing, aggregates `total`, `by_model`, and `by_compressor`, and persists stats to `/state/tokenjuice_savings.json`. Read them over RPC with `openhuman.tokenjuice_savings_stats`; clear them with `openhuman.tokenjuice_savings_reset`. diff --git a/src/openhuman/config/schema/tokenjuice.rs b/src/openhuman/config/schema/tokenjuice.rs index da62fb217..1dbb834f8 100644 --- a/src/openhuman/config/schema/tokenjuice.rs +++ b/src/openhuman/config/schema/tokenjuice.rs @@ -1,9 +1,9 @@ //! TokenJuice content-router configuration (`[tokenjuice]`). //! -//! Controls the content-aware tool-output compaction engine: which compressors -//! are enabled, the Compress-Cache-Retrieve (CCR) store limits, and the opt-in -//! Python/ML plain-text compressor. Installed into the runtime at startup via -//! [`crate::openhuman::tokenjuice::configure`] + the CCR cache `configure`. +//! Controls the TinyJuice content-aware tool-output compaction engine: which +//! compressors are enabled, the Compress-Cache-Retrieve (CCR) store limits, and +//! the opt-in Python/ML plain-text compressor. Installed into the runtime at +//! startup via [`crate::openhuman::tokenjuice::install_from_config`]. use schemars::JsonSchema; use serde::{Deserialize, Serialize}; diff --git a/src/openhuman/tokenjuice/README.md b/src/openhuman/tokenjuice/README.md index d8d3c4de1..f7971fd38 100644 --- a/src/openhuman/tokenjuice/README.md +++ b/src/openhuman/tokenjuice/README.md @@ -1,76 +1,30 @@ -# tokenjuice +# OpenHuman TokenJuice Adapter -Terminal-output compaction engine. A Rust port of [vincentkoc/tokenjuice](https://github.com/vincentkoc/tokenjuice) that shrinks verbose tool/command output (git, npm, cargo, docker, kubectl, lint, test runners, …) **before** it enters an LLM context window. Given a tool invocation (tool name, argv/command, stdout/stderr, exit code), it classifies the output against a JSON-configured rule set, applies filtering / summarisation / counting transforms, and returns a compacted inline string plus reduction stats. It is a **pure library**: no JSON-RPC surface, no CLI, no persistence, no event bus. The live TinyAgents tool-output middleware calls the policy-aware `compact_output_with_policy` entry point after tool calls. +The reusable compression engine now lives in the vendored `tinyjuice` crate at +`vendor/tinyjuice` and is patched through Cargo. This directory is the +OpenHuman adapter layer. -## Responsibilities +OpenHuman-owned files: -- Normalise a `ToolExecutionInput` (derive `argv` from `command` via a small shell tokenizer when absent). -- Classify the input against a rule set: filter rules by `match` criteria, score by specificity, pick the best (or honour a forced rule id), else `generic` / `generic/fallback`. -- Apply the matched rule's pipeline: pretty-print JSON, strip ANSI, skip/keep line filters, trim empty edges, dedupe adjacent lines, head/tail summarisation, pattern counters, `onEmpty` and output-match canned messages. -- Run special-case post-processors for `git/status` (porcelain rewrite to `M:`/`A:`/`D:`/`R:`/`??`) and `cloud/gh` (JSON-record / table row formatting). -- Failure-aware summarisation: when `exit_code != 0` and a rule has `failure.preserveOnFailure`, use the wider `failure.head`/`failure.tail` window. -- Decide between compacted vs passthrough output (tiny outputs ≤240 chars and file-inspection commands like `cat`/`sed`/`jq` are returned verbatim) and clamp to `max_inline_chars` (default 1200) with end- or middle-truncation. -- Load/compile rules from a three-layer overlay (builtin → user → project), precompiling all regex at load time. -- Provide pass-through-safe agent glue (`compact_output_with_policy` / `compact_tool_output_with_policy`) that only substitutes the compacted text when it is meaningfully smaller (ratio ≤ 0.95 and below 512-byte input is skipped entirely). - -## Key files - -| File | Role | +| Path | Role | | --- | --- | -| `src/openhuman/tokenjuice/mod.rs` | Module docstring + public re-exports. Declares submodules. No logic. | -| `src/openhuman/tokenjuice/types.rs` | All serde types mirroring upstream shapes: `JsonRule` + sub-types (`RuleMatch`, `RuleFilters`, `RuleTransforms`, `RuleSummarize`, `RuleCounter`, `RuleOutputMatch`, `RuleFailure`, `CounterSource`), compiled forms (`CompiledRule`, `CompiledParts`, `CompiledCounter`, `CompiledOutputMatch`, `RuleOrigin`), I/O types (`ToolExecutionInput`, `ReduceOptions`, `CompactResult`, `ReductionStats`, `ClassificationResult`, `RuleFixture`). | -| `src/openhuman/tokenjuice/reduce.rs` | The main pipeline: `reduce_execution_with_rules`, command tokenization/normalisation, git-status and gh post-processors, JSON pretty-print, `apply_rule`, passthrough/inline selection, char clamping. Thread-local regex cache for hot per-line patterns. | -| `src/openhuman/tokenjuice/classify.rs` | `matches_rule`, `score_rule`, `classify_execution` — rule matching + specificity scoring. | -| `src/openhuman/tokenjuice/tool_integration.rs` | Agent glue: `compact_output_with_policy`, `compact_tool_output_with_policy` + `CompactionStats`, lazily-cached builtin rule set, `extract_command_argv` for shell-shaped tool arguments. | -| `src/openhuman/tokenjuice/rules/mod.rs` | Re-exports `compile_rule`, `load_builtin_rules`, `load_rules`, `LoadRuleOptions`. | -| `src/openhuman/tokenjuice/rules/loader.rs` | Three-layer overlay loader (builtin/user/project), recursive `.json` discovery, id-keyed overlay merge, fallback-last sort. | -| `src/openhuman/tokenjuice/rules/compiler.rs` | `compile_rule`: builds `regex::Regex` (translating JS `i`/`m` flags to inline flags); drops invalid regex non-fatally. | -| `src/openhuman/tokenjuice/rules/builtin.rs` | `BUILTIN_RULE_JSONS`: `(id, include_str!)` table of all vendored rules embedded at compile time. | -| `src/openhuman/tokenjuice/text/mod.rs` | Re-exports text helpers. | -| `src/openhuman/tokenjuice/text/process.rs` | Line ops: `normalize_lines`, `trim_empty_edges`, `dedupe_adjacent`, `head_tail`, `clamp_text`, `clamp_text_middle`, `pluralize`. | -| `src/openhuman/tokenjuice/text/ansi.rs` | `strip_ansi`. | -| `src/openhuman/tokenjuice/text/width.rs` | `count_text_chars`, `count_terminal_cells`, `graphemes` (Unicode-aware). | -| `src/openhuman/tokenjuice/vendor/rules/*.json` | 96 vendored upstream rule JSON files (`family__name.json` naming), embedded via `builtin.rs`. | -| `src/openhuman/tokenjuice/vendor/README.md` | Provenance + MIT licence of vendored upstream rules; exclusions and how to add rules. | -| `src/openhuman/tokenjuice/tests/fixtures/*.fixture.json` | `RuleFixture` integration-test fixtures (cargo test failure, fallback long output, git status). | -| `*_tests.rs` (`reduce_tests.rs`, `text_tests.rs`, `rules/builtin_tests.rs`, `rules/loader_tests.rs`) | Sibling `#[path]` test suites; plus inline `#[cfg(test)]` tests in `classify.rs` / `tool_integration.rs`. | +| `mod.rs` | Stable OpenHuman module seam, TinyJuice re-exports, config-to-engine install hook, ML/savings callback wiring. | +| `schemas.rs` | JSON-RPC controller schemas and handlers. | +| `config_patch.rs` | Partial update shape for the `[tokenjuice]` config block. | +| `tools.rs` | OpenHuman agent tool implementation for `tokenjuice_retrieve`. | +| `ml/` | Bridge from TinyJuice's optional ML callback into `runtime_python_server` Kompress. | +| `savings.rs` | OpenHuman model-pricing attribution and persisted dashboard stats. | -## Public surface +TinyJuice-owned engine pieces: -Re-exported from `mod.rs`: +| TinyJuice path | Role | +| --- | --- | +| `vendor/tinyjuice/src/compress.rs` | Content router entry point. | +| `vendor/tinyjuice/src/compressors/` | JSON, code, log, search, diff, HTML, ML slot, and generic compressors. | +| `vendor/tinyjuice/src/cache/` | CCR store, retrieval markers, disk tier, ranged retrieval helpers. | +| `vendor/tinyjuice/src/rules/` | Rule loader/compiler and embedded rule table. | +| `vendor/tinyjuice/src/vendor/rules/*.json` | Vendored upstream rule JSON files. | +| `vendor/tinyjuice/src/detect/`, `text/`, `tokens.rs`, `types.rs` | Detection, text helpers, token estimates, public types. | -- `reduce_execution_with_rules(input, rules, opts) -> CompactResult` — synchronous core pipeline against a pre-loaded rule set. -- `load_builtin_rules() -> Vec` — embedded rules only (no disk I/O); `load_rules(&LoadRuleOptions)` for the full builtin/user/project overlay. -- `compact_output_with_policy(content, tool_name, enabled, profile) -> String` — the TinyAgents middleware-facing, pass-through-safe entry point. -- `compact_tool_output_with_policy(tool_name, arguments, output, exit_code, profile) -> (String, CompactionStats)` — the full adapter for call sites that have raw tool arguments and exit code. -- Types: `CompactResult`, `ReduceOptions`, `ToolExecutionInput`, `CompactionStats`, `LoadRuleOptions`. - -## Persistence - -None at runtime. Builtin rules are embedded at compile time. Optionally reads rule JSON from disk at load time (not written): user layer `~/.config/tokenjuice/rules/` and project layer `/.tokenjuice/rules/` (both overridable / skippable via `LoadRuleOptions`). The policy-aware tool adapters use the installed runtime options. - -## Dependencies - -This module is **fully self-contained within `openhuman`** — it has no `use crate::openhuman::` or `use crate::core::` imports on any other domain or transport module. External crate dependencies only: - -- `serde` / `serde_json` — rule and I/O (de)serialisation; gh-output JSON parsing. -- `regex` — rule pattern matching (precompiled at load; thread-local cache for ad-hoc per-line patterns in `reduce.rs`). -- `once_cell::sync::Lazy` — lazy builtin rule cache and the compiled gh table-split regex. -- `dirs` — resolves the user home for the user-layer rules directory. -- `unicode-segmentation` — grapheme-aware width/char counting in `text/width.rs`. -- `log` — verbose `[tokenjuice]`-prefixed diagnostics throughout. - -## Used by - -- `ToolOutputMiddleware` calls `compact_output_with_policy` after tool calls on the live TinyAgents path. -- The legacy direct executor also calls `compact_output_with_policy` until the old path is removed. - -## Notes / gotchas - -- **Library-only by design** (v1). No RPC/CLI/store/bus surfaces; the module docstring states they can be layered on later. -- `generic/fallback` rule **must** be present in any rule set — `reduce_execution_with_rules` `expect`s it, and the loader always sorts it last so it never shadows a more specific rule. -- Pass-through safety: the policy-aware adapters return the untouched original (and `stats.applied == false` where stats are available) for inputs < 512 bytes or when compaction ratio > 0.95 — callers never need to guard the call site, and data is never silently lost. -- `git/status` and `cloud/gh` carry **hard-coded** post-processors in `reduce.rs` keyed off `rule.id`, beyond what the JSON rules express. -- JS→Rust regex flag translation: only `i` and `m` are honoured (as inline `(?i)`/`(?m)`); Unicode is always on in Rust's `regex` (no separate `u` flag). Invalid regex in a rule is logged and dropped, not fatal. -- Disk-loaded rule files ending in `.schema.json` or `.fixture.json` are excluded from discovery; symlinks are skipped. -- Vendored rules exclude upstream's `openclaw/` subdirectory (proprietary, non-generic) — see `vendor/README.md`. To add a rule, drop the JSON in `vendor/rules/` (`family__name.json`) and add the `(id, include_str!)` entry to `builtin.rs`. +Do not add OpenHuman runtime dependencies to TinyJuice. Runtime services, +settings persistence, JSON-RPC, tools, and pricing stay in this adapter. diff --git a/src/openhuman/tokenjuice/cache/marker.rs b/src/openhuman/tokenjuice/cache/marker.rs deleted file mode 100644 index 716ab1757..000000000 --- a/src/openhuman/tokenjuice/cache/marker.rs +++ /dev/null @@ -1,133 +0,0 @@ -//! CCR retrieval markers. -//! -//! When the router offloads an original to the [`super::store`], it embeds a -//! marker carrying the CCR token so the model knows the content is recoverable -//! and how to fetch it. The canonical marker is `⟦tj:⟧`; for backward -//! compatibility we also parse the legacy `retrieve_tool_output("")` form -//! that older histories may still contain. - -/// The retrieve tool's name, surfaced in footers and used by the harness to -/// keep the tool's own output from being re-compacted and to always advertise it. -pub const RETRIEVE_TOOL_NAME: &str = "tokenjuice_retrieve"; - -/// The legacy retrieve tool name (kept as an alias during migration). -pub const LEGACY_RETRIEVE_TOOL_NAME: &str = "retrieve_tool_output"; - -/// All CCR recovery tool names. Both must be (a) always advertised to every -/// agent — any agent that sees a retrieval footer must be able to call the tool -/// — and (b) never re-compacted (their job is to return an original in full). -pub const RECOVERY_TOOL_NAMES: &[&str] = &[RETRIEVE_TOOL_NAME, LEGACY_RETRIEVE_TOOL_NAME]; - -/// Tools whose output must never be re-compacted. See [`RECOVERY_TOOL_NAMES`]. -pub const NEVER_COMPACT_TOOLS: &[&str] = RECOVERY_TOOL_NAMES; - -/// True if `tool_name` is one of the CCR recovery tools. -pub fn is_recovery_tool(tool_name: &str) -> bool { - RECOVERY_TOOL_NAMES.contains(&tool_name) -} - -/// Format the canonical inline marker for a CCR `hash`. -pub fn format_marker(hash: &str) -> String { - format!("⟦tj:{hash}⟧") -} - -/// Build the human-facing recovery footer appended to compacted output. -/// -/// `lossy` distinguishes a partial view (data dropped) from a faithful reformat -/// (no data lost, layout changed); both offer exact recovery. -pub fn recovery_footer(hash: &str, original_bytes: usize, lossy: bool) -> String { - let marker = format_marker(hash); - if lossy { - format!( - "\n\n[compacted tool output — this is a PARTIAL view; the full original \ - ({original_bytes} bytes) is available by calling {RETRIEVE_TOOL_NAME} with \ - token \"{hash}\" (marker {marker})]" - ) - } else { - format!( - "\n\n[reformatted tool output — no data lost, but layout changed; the exact \ - original ({original_bytes} bytes) is available by calling {RETRIEVE_TOOL_NAME} \ - with token \"{hash}\" (marker {marker})]" - ) - } -} - -/// Extract all CCR tokens referenced in `text`, from both the canonical -/// `⟦tj:⟧` markers and the legacy `retrieve_tool_output("")` form. -/// Order-preserving, de-duplicated. -pub fn parse_markers(text: &str) -> Vec { - let mut out: Vec = Vec::new(); - let mut push = |h: &str| { - let h = h.trim(); - if !h.is_empty() && !out.iter().any(|e| e == h) { - out.push(h.to_string()); - } - }; - - // Canonical: ⟦tj:HASH⟧ - let mut rest = text; - while let Some(start) = rest.find("⟦tj:") { - let after = &rest[start + "⟦tj:".len()..]; - if let Some(end) = after.find('⟧') { - push(&after[..end]); - rest = &after[end..]; - } else { - break; - } - } - - // Legacy: retrieve_tool_output("HASH") or tokenjuice_retrieve("HASH") - for needle in [ - "retrieve_tool_output(\"", - "tokenjuice_retrieve(\"", - "token \"", - ] { - let mut rest = text; - while let Some(start) = rest.find(needle) { - let after = &rest[start + needle.len()..]; - if let Some(end) = after.find('"') { - push(&after[..end]); - rest = &after[end..]; - } else { - break; - } - } - } - - out -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn formats_and_parses_canonical() { - let m = format_marker("ab12cd34"); - assert_eq!(m, "⟦tj:ab12cd34⟧"); - assert_eq!( - parse_markers(&format!("see {m} for more")), - vec!["ab12cd34"] - ); - } - - #[test] - fn parses_legacy_form() { - let text = "partial; call retrieve_tool_output(\"deadbeef00\") to recover"; - assert_eq!(parse_markers(text), vec!["deadbeef00"]); - } - - #[test] - fn parses_multiple_dedup() { - let text = "⟦tj:aaa⟧ and ⟦tj:bbb⟧ and again ⟦tj:aaa⟧"; - assert_eq!(parse_markers(text), vec!["aaa", "bbb"]); - } - - #[test] - fn footer_carries_token() { - let f = recovery_footer("c0ffee", 1234, true); - assert!(f.contains("PARTIAL view")); - assert!(f.contains("c0ffee")); - assert_eq!(parse_markers(&f), vec!["c0ffee"]); - } -} diff --git a/src/openhuman/tokenjuice/cache/mod.rs b/src/openhuman/tokenjuice/cache/mod.rs deleted file mode 100644 index 2e09eb04a..000000000 --- a/src/openhuman/tokenjuice/cache/mod.rs +++ /dev/null @@ -1,13 +0,0 @@ -//! CCR (Compress-Cache-Retrieve) — original storage + retrieval markers. - -pub mod marker; -pub mod store; - -pub use marker::{ - format_marker, is_recovery_tool, parse_markers, recovery_footer, LEGACY_RETRIEVE_TOOL_NAME, - NEVER_COMPACT_TOOLS, RECOVERY_TOOL_NAMES, RETRIEVE_TOOL_NAME, -}; -pub use store::{ - configure, disable_disk_tier, enable_disk_tier, offload, offload_checked, retrieve, - retrieve_range, short_hash, stats, RangeUnit, -}; diff --git a/src/openhuman/tokenjuice/cache/store.rs b/src/openhuman/tokenjuice/cache/store.rs deleted file mode 100644 index be7d84771..000000000 --- a/src/openhuman/tokenjuice/cache/store.rs +++ /dev/null @@ -1,474 +0,0 @@ -//! CCR — Compress-Cache-Retrieve store. -//! -//! When a compressor drops data (lossy paths), the router stows the original -//! here keyed by a short content hash and embeds a retrieval marker in the -//! compacted text (see [`super::marker`]). The agent calls the -//! `tokenjuice_retrieve` tool to get the original back on demand — so even -//! aggressive compaction stays reversible and is safe under the always-on -//! default. -//! -//! Process-global and bounded by **both** an entry count and a total byte cap, -//! with an optional TTL. Keyed by content hash, so re-offloading identical -//! content is idempotent. An optional **disk tier** (configured from the core's -//! `workspace_dir`) persists originals across the session so retrieval can -//! survive memory eviction; the agent itself never writes there — only the core -//! does, through this module. - -use sha2::{Digest, Sha256}; -use std::collections::{HashMap, VecDeque}; -use std::path::{Path, PathBuf}; -use std::sync::{Mutex, OnceLock, RwLock}; -use std::time::{Duration, Instant}; - -/// Default max originals retained (entry-count cap). -pub const DEFAULT_MAX_ENTRIES: usize = 256; -/// Default total-bytes cap (64 MiB) so a few huge originals can't blow memory. -pub const DEFAULT_MAX_BYTES: usize = 64 * 1024 * 1024; -/// Bytes of the SHA-256 digest used for the key (→ 32 hex chars). Wide enough -/// that collisions are infeasible and the hash doubles as an unguessable -/// capability token. -const HASH_BYTES: usize = 16; - -/// Tunable limits, settable once at startup from the `[tokenjuice]` config. -struct Limits { - max_entries: usize, - max_bytes: usize, - ttl: Option, -} - -impl Default for Limits { - fn default() -> Self { - Self { - max_entries: DEFAULT_MAX_ENTRIES, - max_bytes: DEFAULT_MAX_BYTES, - ttl: None, - } - } -} - -fn limits() -> &'static RwLock { - static LIMITS: OnceLock> = OnceLock::new(); - LIMITS.get_or_init(|| RwLock::new(Limits::default())) -} - -/// Optional on-disk tier root (under the core's workspace). `None` ⇒ in-memory only. -fn disk_root() -> &'static RwLock> { - static ROOT: OnceLock>> = OnceLock::new(); - ROOT.get_or_init(|| RwLock::new(None)) -} - -/// Configure the cache limits (called once from config at startup). -pub fn configure(max_entries: usize, max_bytes: usize, ttl_secs: Option) { - let mut l = limits().write().unwrap_or_else(|p| p.into_inner()); - l.max_entries = max_entries.max(1); - l.max_bytes = max_bytes.max(1); - l.ttl = ttl_secs.map(Duration::from_secs); -} - -/// Enable the on-disk tier rooted at `root` (e.g. `/.tokenjuice/ccr`). -/// Best-effort: directory creation failures disable the tier silently. -pub fn enable_disk_tier(root: PathBuf) { - if std::fs::create_dir_all(&root).is_ok() { - *disk_root().write().unwrap_or_else(|p| p.into_inner()) = Some(root); - } else { - log::warn!("[tokenjuice][ccr] could not create disk tier at {root:?}"); - } -} - -/// Turn the on-disk tier off (e.g. when the setting is toggled off at runtime). -/// New offloads stop writing to disk; already-written files are left in place. -pub fn disable_disk_tier() { - *disk_root().write().unwrap_or_else(|p| p.into_inner()) = None; -} - -struct Entry { - content: String, - created: Instant, -} - -#[derive(Default)] -struct Inner { - map: HashMap, - order: VecDeque, - total_bytes: usize, -} - -impl Inner { - /// Insert `content` under `hash` (idempotent) and evict (FIFO) until both - /// the entry-count and total-byte caps hold. - /// - /// Returns whether `hash` is still resident after eviction. A single - /// original larger than `max_bytes` cannot be retained in memory under the - /// byte cap — eviction would immediately drop the just-inserted entry — so - /// `false` is returned and the caller must not advertise it as recoverable - /// (the router declines lossy compaction or relies on the disk tier). - fn insert( - &mut self, - hash: String, - content: String, - max_entries: usize, - max_bytes: usize, - ) -> bool { - if let Some(entry) = self.map.get_mut(&hash) { - entry.created = Instant::now(); - self.order.retain(|candidate| candidate != &hash); - self.order.push_back(hash); - return true; - } - let bytes = content.len(); - self.total_bytes += bytes; - self.map.insert( - hash.clone(), - Entry { - content, - created: Instant::now(), - }, - ); - self.order.push_back(hash.clone()); - while self.order.len() > max_entries || self.total_bytes > max_bytes { - // Never evict the entry we just inserted to satisfy the cap when it - // is the only thing keeping us over: that would make the original - // unrecoverable the instant its footer is emitted. Stop and report - // non-retention instead (one oversized item is rejected, not the - // whole store wiped). - if self.order.len() == 1 { - break; - } - let Some(evicted) = self.order.pop_front() else { - break; - }; - if let Some(e) = self.map.remove(&evicted) { - self.total_bytes = self.total_bytes.saturating_sub(e.content.len()); - } - } - // Retained iff still present (an oversized single entry over the byte - // cap is dropped below) AND within the byte cap. - if self.total_bytes > max_bytes { - if let Some(e) = self.map.remove(&hash) { - self.total_bytes = self.total_bytes.saturating_sub(e.content.len()); - } - self.order.retain(|h| h != &hash); - return false; - } - self.map.contains_key(&hash) - } -} - -fn global() -> &'static Mutex { - static STORE: OnceLock> = OnceLock::new(); - STORE.get_or_init(|| Mutex::new(Inner::default())) -} - -/// Stash `content` and return its short hash. Idempotent for identical content. -pub fn offload(content: &str) -> String { - offload_checked(content).0 -} - -/// Stash `content`, returning `(hash, retained)`. `retained` is `false` only -/// when the original could be kept neither in memory (it exceeds the byte cap) -/// nor on the disk tier — in which case the caller must NOT advertise it as -/// recoverable. Idempotent for identical content. -pub fn offload_checked(content: &str) -> (String, bool) { - let hash = short_hash(content); - let (max_entries, max_bytes) = { - let l = limits().read().unwrap_or_else(|p| p.into_inner()); - (l.max_entries, l.max_bytes) - }; - let mem_retained = global().lock().unwrap_or_else(|p| p.into_inner()).insert( - hash.clone(), - content.to_string(), - max_entries, - max_bytes, - ); - - // Mirror to the disk tier when enabled (best-effort). A successful disk - // write keeps the original recoverable even when it was too big for memory. - // Rewriting an existing hash intentionally refreshes the file mtime, which - // is the TTL clock for disk-backed CCR entries. - let mut disk_retained = false; - if let Some(root) = disk_root() - .read() - .unwrap_or_else(|p| p.into_inner()) - .clone() - { - let path = root.join(&hash); - match std::fs::write(&path, content) { - Ok(()) => disk_retained = true, - Err(e) => log::debug!("[tokenjuice][ccr] disk write failed for {hash}: {e}"), - } - } - (hash, mem_retained || disk_retained) -} - -/// True if `hash` is a well-formed CCR token (exactly the generated hex digest). -/// Tokens come from agent-controlled tool args, and on a disk-tier miss they are -/// joined onto the CCR root — so anything other than the fixed hex shape (e.g. -/// `../../state/config.toml`) must be rejected before touching the filesystem to -/// prevent path traversal / arbitrary file reads through the recovery tool. -fn is_valid_token(hash: &str) -> bool { - hash.len() == HASH_BYTES * 2 && hash.bytes().all(|b| b.is_ascii_hexdigit()) -} - -/// Retrieve a previously-offloaded original by hash, if still available -/// (memory first, then the disk tier). Honours the TTL for both tiers. -pub fn retrieve(hash: &str) -> Option { - // Reject anything that isn't the generated token shape up front — guards the - // disk-tier `root.join(hash)` below against path traversal. - if !is_valid_token(hash) { - return None; - } - let ttl = limits().read().unwrap_or_else(|p| p.into_inner()).ttl; - { - let mut inner = global().lock().unwrap_or_else(|p| p.into_inner()); - if let Some(entry) = inner.map.get(hash) { - if ttl.is_none_or(|t| entry.created.elapsed() < t) { - return Some(entry.content.clone()); - } - // Expired — drop it and fall through to disk. - if let Some(e) = inner.map.remove(hash) { - inner.total_bytes = inner.total_bytes.saturating_sub(e.content.len()); - } - } - } - // Disk fallback. - let root = disk_root() - .read() - .unwrap_or_else(|p| p.into_inner()) - .clone()?; - let path = root.join(hash); - if disk_entry_expired(&path, ttl) { - log::debug!("[tokenjuice][ccr] disk entry expired for {hash}"); - let _ = std::fs::remove_file(&path); - return None; - } - std::fs::read_to_string(path).ok() -} - -fn disk_entry_expired(path: &Path, ttl: Option) -> bool { - let Some(ttl) = ttl else { - return false; - }; - let Ok(metadata) = std::fs::metadata(path) else { - return false; - }; - let Ok(modified) = metadata.modified() else { - return false; - }; - modified.elapsed().is_ok_and(|elapsed| elapsed >= ttl) -} - -/// The span/unit for a ranged retrieval. -#[derive(Debug, Clone, Copy)] -pub enum RangeUnit { - Bytes, - Lines, -} - -/// Retrieve a slice of a previously-offloaded original. `start`/`end` are -/// 0-based, `end` exclusive; out-of-bounds ends are clamped. Returns `None` -/// only when the original isn't available at all. -pub fn retrieve_range(hash: &str, start: usize, end: usize, unit: RangeUnit) -> Option { - let original = retrieve(hash)?; - if end <= start { - return Some(String::new()); - } - match unit { - RangeUnit::Bytes => { - // Clamp to char boundaries so we never split a UTF-8 sequence. - let s = floor_char_boundary(&original, start.min(original.len())); - let e = floor_char_boundary(&original, end.min(original.len())); - Some(original[s..e].to_string()) - } - RangeUnit::Lines => { - let lines: Vec<&str> = original.lines().collect(); - let e = end.min(lines.len()); - if start >= lines.len() { - return Some(String::new()); - } - Some(lines[start..e].join("\n")) - } - } -} - -/// Largest char boundary ≤ `idx` (std's `floor_char_boundary` is still nightly). -fn floor_char_boundary(s: &str, idx: usize) -> usize { - if idx >= s.len() { - return s.len(); - } - let mut i = idx; - while i > 0 && !s.is_char_boundary(i) { - i -= 1; - } - i -} - -/// Short hex content hash used as the CCR key/token. -pub fn short_hash(content: &str) -> String { - let mut hasher = Sha256::new(); - hasher.update(content.as_bytes()); - let digest = hasher.finalize(); - hex::encode(&digest[..HASH_BYTES]) -} - -/// Snapshot of cache occupancy for the debug controller / stats. -pub fn stats() -> (usize, usize) { - let inner = global().lock().unwrap_or_else(|p| p.into_inner()); - (inner.map.len(), inner.total_bytes) -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn round_trips() { - let original = "ccr round-trip unique payload alpha ".repeat(50); - let hash = offload(&original); - assert_eq!(hash.len(), HASH_BYTES * 2); - assert_eq!(retrieve(&hash).as_deref(), Some(original.as_str())); - } - - #[test] - fn idempotent_hash() { - let a = offload("ccr idempotent unique payload bravo content"); - let b = offload("ccr idempotent unique payload bravo content"); - assert_eq!(a, b); - } - - #[test] - fn missing_hash_is_none() { - assert!(retrieve("ffffffffffffffffffffffffffffffff").is_none()); - } - - #[test] - fn byte_cap_evicts_oldest() { - let mut inner = Inner::default(); - // 10 entries of 100 bytes; cap at 500 bytes ⇒ keep ~5 newest. - for i in 0..10 { - inner.insert(format!("h{i}"), "x".repeat(100), 1000, 500); - } - assert!( - inner.total_bytes <= 500, - "byte cap held: {}", - inner.total_bytes - ); - assert!(!inner.map.contains_key("h0"), "oldest evicted"); - assert!(inner.map.contains_key("h9"), "newest retained"); - } - - #[test] - fn oversized_single_entry_is_not_retained() { - // One entry larger than the byte cap can't be kept; insert reports - // non-retention so the router won't advertise it as recoverable. - let mut inner = Inner::default(); - let retained = inner.insert("big".into(), "x".repeat(200), 100, 100); - assert!(!retained, "oversized single entry must report not-retained"); - assert!(!inner.map.contains_key("big")); - assert_eq!(inner.total_bytes, 0); - } - - #[test] - fn rejects_path_traversal_tokens() { - // Non-hex / wrong-length tokens are rejected before any disk join. - assert!(!is_valid_token("../../state/config.toml")); - assert!(!is_valid_token("..%2f..%2fetc")); - assert!(!is_valid_token("deadbeef")); // too short - assert!(!is_valid_token(&"g".repeat(32))); // non-hex - assert!(is_valid_token(&"a1b2c3d4".repeat(4))); // 32 hex chars - // retrieve() returns None for an invalid token regardless of cache state. - assert!(retrieve("../../state/config.toml").is_none()); - } - - #[test] - fn within_cap_entry_is_retained() { - let mut inner = Inner::default(); - assert!(inner.insert("ok".into(), "x".repeat(50), 100, 100)); - assert!(inner.map.contains_key("ok")); - } - - #[test] - fn entry_cap_evicts_oldest() { - let mut inner = Inner::default(); - for i in 0..60 { - inner.insert(format!("e{i}"), format!("content-{i}"), 50, usize::MAX); - } - assert!(inner.map.len() <= 50); - assert!(!inner.map.contains_key("e0")); - } - - #[test] - fn reoffloading_existing_entry_refreshes_ttl_and_order() { - let mut inner = Inner::default(); - assert!(inner.insert("old".into(), "old payload".into(), 2, usize::MAX)); - assert!(inner.insert("fresh".into(), "fresh payload".into(), 2, usize::MAX)); - let stale_created = Instant::now() - Duration::from_secs(60); - inner.map.get_mut("old").unwrap().created = stale_created; - - assert!(inner.insert("old".into(), "old payload".into(), 2, usize::MAX)); - assert!( - inner.map["old"].created > stale_created, - "existing entry timestamp should refresh" - ); - assert_eq!(inner.order.back().map(String::as_str), Some("old")); - - assert!(inner.insert("third".into(), "third payload".into(), 2, usize::MAX)); - assert!(inner.map.contains_key("old")); - assert!(!inner.map.contains_key("fresh")); - } - - #[test] - fn range_retrieval_lines_and_bytes() { - let original = "line0\nline1\nline2\nline3\nline4"; - let hash = offload(original); - assert_eq!( - retrieve_range(&hash, 1, 3, RangeUnit::Lines).as_deref(), - Some("line1\nline2") - ); - assert_eq!( - retrieve_range(&hash, 0, 5, RangeUnit::Bytes).as_deref(), - Some("line0") - ); - // Out-of-bounds end clamps. - assert_eq!( - retrieve_range(&hash, 4, 999, RangeUnit::Lines).as_deref(), - Some("line4") - ); - } - - #[test] - fn disk_tier_survives_memory_miss() { - let dir = std::env::temp_dir().join(format!("tj-ccr-{}", short_hash("disk-test-seed"))); - let _ = std::fs::remove_dir_all(&dir); - enable_disk_tier(dir.clone()); - let original = "disk tier unique payload charlie ".repeat(40); - let hash = offload(&original); - // Simulate memory eviction by clearing the in-memory map directly. - { - let mut inner = global().lock().unwrap_or_else(|p| p.into_inner()); - inner.map.remove(&hash); - } - assert_eq!( - retrieve(&hash).as_deref(), - Some(original.as_str()), - "disk fallback" - ); - // Disable the tier for other tests and clean up. - *disk_root().write().unwrap() = None; - let _ = std::fs::remove_dir_all(&dir); - } - - #[test] - fn disk_entry_ttl_uses_file_mtime() { - let dir = std::env::temp_dir().join(format!("tj-ccr-ttl-{}", short_hash("disk-ttl"))); - let _ = std::fs::remove_dir_all(&dir); - std::fs::create_dir_all(&dir).unwrap(); - let path = dir.join("aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"); - std::fs::write(&path, "disk ttl payload").unwrap(); - - assert!(!disk_entry_expired(&path, None)); - assert!(!disk_entry_expired(&path, Some(Duration::from_secs(60)))); - assert!(disk_entry_expired(&path, Some(Duration::ZERO))); - - let _ = std::fs::remove_dir_all(&dir); - } -} diff --git a/src/openhuman/tokenjuice/classify.rs b/src/openhuman/tokenjuice/classify.rs deleted file mode 100644 index a96e6481f..000000000 --- a/src/openhuman/tokenjuice/classify.rs +++ /dev/null @@ -1,590 +0,0 @@ -//! Rule classification: given a `ToolExecutionInput`, find the best-matching -//! `CompiledRule` and return a `ClassificationResult`. -//! -//! Port of `src/core/classify.ts` and the matching helpers from -//! `src/core/rules.ts`. - -use crate::openhuman::tokenjuice::types::{ - ClassificationResult, CompiledRule, JsonRule, ToolExecutionInput, -}; - -// --------------------------------------------------------------------------- -// Matching helpers -// --------------------------------------------------------------------------- - -/// True if every string in `expected` is present somewhere in `argv`. -fn includes_all(argv: &[String], expected: &[String]) -> bool { - expected.iter().all(|part| argv.contains(part)) -} - -/// Test whether `rule` matches `input`. Mirrors `matchesRule` in TS. -pub fn matches_rule(rule: &JsonRule, input: &ToolExecutionInput) -> bool { - let argv = input.argv.as_deref().unwrap_or(&[]); - // Fall back to a joined argv when `command` wasn't explicitly set so - // `commandIncludes*` rules still match for argv-only callers. - let command_fallback: String; - let command: &str = match input.command.as_deref() { - Some(c) => c, - None => { - command_fallback = argv.join(" "); - &command_fallback - } - }; - let tool_name = &input.tool_name; - - // toolNames filter - if let Some(tool_names) = &rule.r#match.tool_names { - if !tool_names.contains(tool_name) { - return false; - } - } - - // argv0 filter - if let Some(argv0_list) = &rule.r#match.argv0 { - let first = argv.first().map(String::as_str).unwrap_or(""); - if !argv0_list.iter().any(|s| s == first) { - return false; - } - } - - // argvIncludes — all groups must match - if let Some(groups) = &rule.r#match.argv_includes { - if !groups.iter().all(|group| includes_all(argv, group)) { - return false; - } - } - - // argvIncludesAny — at least one group must match - if let Some(groups) = &rule.r#match.argv_includes_any { - if !groups.iter().any(|group| includes_all(argv, group)) { - return false; - } - } - - // commandIncludes — all substrings must appear in command - if let Some(parts) = &rule.r#match.command_includes { - if !parts.iter().all(|part| command.contains(part.as_str())) { - return false; - } - } - - // commandIncludesAny — at least one substring must appear - if let Some(parts) = &rule.r#match.command_includes_any { - if !parts.iter().any(|part| command.contains(part.as_str())) { - return false; - } - } - - true -} - -// --------------------------------------------------------------------------- -// Scoring -// --------------------------------------------------------------------------- - -/// Numeric specificity score for a rule — higher wins. -/// Mirrors `scoreRule` in TS. -fn score_rule(rule: &JsonRule) -> i64 { - let priority = rule.priority.unwrap_or(0) as i64 * 1000; - let argv0 = rule.r#match.argv0.as_ref().map(|v| v.len()).unwrap_or(0) as i64 * 100; - let argv_includes = rule - .r#match - .argv_includes - .as_ref() - .map(|groups| groups.iter().map(|g| g.len()).sum::()) - .unwrap_or(0) as i64 - * 40; - let argv_includes_any = rule - .r#match - .argv_includes_any - .as_ref() - .map(|groups| groups.iter().map(|g| g.len()).sum::()) - .unwrap_or(0) as i64 - * 35; - let command_includes = rule - .r#match - .command_includes - .as_ref() - .map(|v| v.len()) - .unwrap_or(0) as i64 - * 25; - let command_includes_any = rule - .r#match - .command_includes_any - .as_ref() - .map(|v| v.len()) - .unwrap_or(0) as i64 - * 20; - let tool_names = rule - .r#match - .tool_names - .as_ref() - .map(|v| v.len()) - .unwrap_or(0) as i64 - * 10; - - priority - + argv0 - + argv_includes - + argv_includes_any - + command_includes - + command_includes_any - + tool_names -} - -// --------------------------------------------------------------------------- -// classify_execution -// --------------------------------------------------------------------------- - -/// Classify `input` against the provided `rules` and return a -/// `ClassificationResult`. -/// -/// If `forced_rule_id` is `Some`, that rule is used directly (if found). -pub fn classify_execution( - input: &ToolExecutionInput, - rules: &[CompiledRule], - forced_rule_id: Option<&str>, -) -> ClassificationResult { - // Forced classification - if let Some(id) = forced_rule_id { - if let Some(rule) = rules.iter().find(|r| r.rule.id == id) { - log::debug!( - "[tokenjuice] forced classification: rule='{}' family='{}'", - id, - rule.rule.family - ); - return ClassificationResult { - family: rule.rule.family.clone(), - confidence: 1.0, - matched_reducer: Some(rule.rule.id.clone()), - }; - } - } - - // Find all matching rules - let mut matched: Vec<&CompiledRule> = rules - .iter() - .filter(|r| matches_rule(&r.rule, input)) - .collect(); - - if matched.is_empty() { - log::debug!( - "[tokenjuice] no rule matched tool='{}' argv={:?} — using generic fallback", - input.tool_name, - input.argv - ); - return ClassificationResult { - family: "generic".to_owned(), - confidence: 0.2, - matched_reducer: None, - }; - } - - // Sort by descending score, then alphabetically for stability - matched.sort_by(|a, b| { - let score_diff = score_rule(&b.rule).cmp(&score_rule(&a.rule)); - if score_diff != std::cmp::Ordering::Equal { - score_diff - } else { - a.rule.id.cmp(&b.rule.id) - } - }); - - let best = matched[0]; - let confidence = if best.rule.id == "generic/fallback" { - 0.2 - } else { - 0.9 - }; - - log::debug!( - "[tokenjuice] classified tool='{}' → rule='{}' family='{}' confidence={}", - input.tool_name, - best.rule.id, - best.rule.family, - confidence - ); - - ClassificationResult { - family: best.rule.family.clone(), - confidence, - matched_reducer: Some(best.rule.id.clone()), - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::openhuman::tokenjuice::rules::load_builtin_rules; - - fn make_input(tool_name: &str, argv: &[&str]) -> ToolExecutionInput { - ToolExecutionInput { - tool_name: tool_name.to_owned(), - argv: Some(argv.iter().map(|s| s.to_string()).collect()), - ..Default::default() - } - } - - #[test] - fn git_status_matches() { - let rules = load_builtin_rules(); - let input = make_input("bash", &["git", "status"]); - let result = classify_execution(&input, &rules, None); - assert_eq!(result.matched_reducer.as_deref(), Some("git/status")); - assert_eq!(result.family, "git-status"); - } - - #[test] - fn npm_install_does_not_match_git_status() { - let rules = load_builtin_rules(); - let input = make_input("exec", &["npm", "install"]); - let result = classify_execution(&input, &rules, None); - assert_ne!(result.matched_reducer.as_deref(), Some("git/status")); - } - - #[test] - fn no_match_returns_generic() { - let rules = load_builtin_rules(); - let input = make_input("some_unknown_tool", &["mysterious", "command"]); - let result = classify_execution(&input, &rules, None); - assert_eq!(result.family, "generic"); - assert_eq!(result.confidence, 0.2); - } - - #[test] - fn forced_rule_id_overrides_matching() { - let rules = load_builtin_rules(); - // Input would normally match git/status but we force cargo-test - let input = make_input("bash", &["git", "status"]); - let result = classify_execution(&input, &rules, Some("tests/cargo-test")); - assert_eq!(result.matched_reducer.as_deref(), Some("tests/cargo-test")); - assert_eq!(result.confidence, 1.0); - } - - #[test] - fn fallback_confidence_is_low() { - let rules = load_builtin_rules(); - // Force the fallback explicitly - let input = make_input("bash", &["some", "arbitrary", "command"]); - let result = classify_execution(&input, &rules, Some("generic/fallback")); - assert_eq!(result.confidence, 1.0); // forced always returns 1.0 - } - - #[test] - fn git_diff_stat_requires_both_args() { - let rules = load_builtin_rules(); - // Missing --stat → should not match git/diff-stat - let input_no_stat = make_input("bash", &["git", "diff"]); - let result = classify_execution(&input_no_stat, &rules, None); - assert_ne!(result.matched_reducer.as_deref(), Some("git/diff-stat")); - - // With --stat → should match - let input_with_stat = make_input("bash", &["git", "diff", "--stat"]); - let result2 = classify_execution(&input_with_stat, &rules, None); - assert_eq!(result2.matched_reducer.as_deref(), Some("git/diff-stat")); - } - - // --- matches_rule: individual dimension tests --- - - #[test] - fn tool_names_filter_blocks_wrong_tool() { - // cargo test rule requires toolNames: ["exec"] - let rules = load_builtin_rules(); - // "bash" tool should not match tests/cargo-test (requires "exec") - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["cargo".to_owned(), "test".to_owned()]), - ..Default::default() - }; - let result = classify_execution(&input, &rules, None); - assert_ne!(result.matched_reducer.as_deref(), Some("tests/cargo-test")); - } - - #[test] - fn tool_names_filter_matches_correct_tool() { - let rules = load_builtin_rules(); - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["cargo".to_owned(), "test".to_owned()]), - ..Default::default() - }; - let result = classify_execution(&input, &rules, None); - assert_eq!( - result.matched_reducer.as_deref(), - Some("tests/cargo-test"), - "cargo test with exec tool should match tests/cargo-test" - ); - } - - #[test] - fn argv_includes_any_matches_at_least_one_group() { - // Build a custom rule with argvIncludesAny and test it via matches_rule directly - use crate::openhuman::tokenjuice::types::{JsonRule, RuleMatch}; - - let rule = JsonRule { - id: "test/any".to_owned(), - family: "test".to_owned(), - description: None, - priority: None, - on_empty: None, - match_output: None, - counter_source: None, - r#match: RuleMatch { - argv0: Some(vec!["tool".to_owned()]), - argv_includes_any: Some(vec![vec!["--foo".to_owned()], vec!["--bar".to_owned()]]), - ..Default::default() - }, - filters: None, - transforms: None, - summarize: None, - counters: None, - failure: None, - }; - - // Should match when --foo is present - let input_foo = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["tool".to_owned(), "--foo".to_owned()]), - ..Default::default() - }; - assert!(matches_rule(&rule, &input_foo)); - - // Should match when --bar is present - let input_bar = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["tool".to_owned(), "--bar".to_owned()]), - ..Default::default() - }; - assert!(matches_rule(&rule, &input_bar)); - - // Should NOT match when neither is present - let input_none = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["tool".to_owned(), "--baz".to_owned()]), - ..Default::default() - }; - assert!(!matches_rule(&rule, &input_none)); - } - - #[test] - fn command_includes_all_substrings_required() { - use crate::openhuman::tokenjuice::types::{JsonRule, RuleMatch}; - - let rule = JsonRule { - id: "test/cmd-incl".to_owned(), - family: "test".to_owned(), - description: None, - priority: None, - on_empty: None, - match_output: None, - counter_source: None, - r#match: RuleMatch { - command_includes: Some(vec!["git".to_owned(), "status".to_owned()]), - ..Default::default() - }, - filters: None, - transforms: None, - summarize: None, - counters: None, - failure: None, - }; - - let input_match = ToolExecutionInput { - tool_name: "bash".to_owned(), - command: Some("git status --short".to_owned()), - ..Default::default() - }; - assert!(matches_rule(&rule, &input_match)); - - // Missing "status" → no match - let input_no_match = ToolExecutionInput { - tool_name: "bash".to_owned(), - command: Some("git log --oneline".to_owned()), - ..Default::default() - }; - assert!(!matches_rule(&rule, &input_no_match)); - } - - #[test] - fn command_includes_any_at_least_one_substring() { - use crate::openhuman::tokenjuice::types::{JsonRule, RuleMatch}; - - let rule = JsonRule { - id: "test/cmd-any".to_owned(), - family: "test".to_owned(), - description: None, - priority: None, - on_empty: None, - match_output: None, - counter_source: None, - r#match: RuleMatch { - command_includes_any: Some(vec!["install".to_owned(), "update".to_owned()]), - ..Default::default() - }, - filters: None, - transforms: None, - summarize: None, - counters: None, - failure: None, - }; - - let input_install = ToolExecutionInput { - tool_name: "bash".to_owned(), - command: Some("npm install".to_owned()), - ..Default::default() - }; - assert!(matches_rule(&rule, &input_install)); - - let input_update = ToolExecutionInput { - tool_name: "bash".to_owned(), - command: Some("npm update".to_owned()), - ..Default::default() - }; - assert!(matches_rule(&rule, &input_update)); - - let input_neither = ToolExecutionInput { - tool_name: "bash".to_owned(), - command: Some("npm run build".to_owned()), - ..Default::default() - }; - assert!(!matches_rule(&rule, &input_neither)); - } - - #[test] - fn forced_rule_id_not_found_falls_back_to_matching() { - let rules = load_builtin_rules(); - let input = make_input("bash", &["git", "status"]); - // Force a non-existent rule ID → should fall through to normal matching - let result = classify_execution(&input, &rules, Some("nonexistent/rule")); - // Falls through to normal matching; git status should still match git/status - assert_eq!(result.matched_reducer.as_deref(), Some("git/status")); - } - - #[test] - fn multiple_matches_best_score_wins() { - let rules = load_builtin_rules(); - // "git diff --stat" should match git/diff-stat (more specific) over git/show or others - let input = make_input("bash", &["git", "diff", "--stat"]); - let result = classify_execution(&input, &rules, None); - assert_eq!(result.matched_reducer.as_deref(), Some("git/diff-stat")); - assert_eq!(result.confidence, 0.9); - } - - #[test] - fn generic_fallback_matched_gives_low_confidence() { - let rules = load_builtin_rules(); - // An unknown command should match generic/fallback with low confidence - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["some_nonexistent_program".to_owned()]), - ..Default::default() - }; - let result = classify_execution(&input, &rules, None); - // generic/fallback matches everything, so it will be the winner for unknown commands - // but confidence should be low (0.2) - assert_eq!(result.confidence, 0.2); - } - - // --- Rust toolchain classification tests --- - - #[test] - fn cargo_clippy_matches_lint_cargo_clippy() { - let rules = load_builtin_rules(); - let input = make_input("exec", &["cargo", "clippy"]); - let result = classify_execution(&input, &rules, None); - assert_eq!( - result.matched_reducer.as_deref(), - Some("lint/cargo-clippy"), - "cargo clippy should match lint/cargo-clippy" - ); - assert_eq!(result.family, "lint-results"); - } - - #[test] - fn cargo_clippy_with_flags_matches() { - let rules = load_builtin_rules(); - let input = make_input("exec", &["cargo", "clippy", "--", "-D", "warnings"]); - let result = classify_execution(&input, &rules, None); - assert_eq!( - result.matched_reducer.as_deref(), - Some("lint/cargo-clippy"), - "cargo clippy with flags should still match lint/cargo-clippy" - ); - } - - #[test] - fn cargo_build_matches_build_cargo_build() { - let rules = load_builtin_rules(); - let input = make_input("exec", &["cargo", "build"]); - let result = classify_execution(&input, &rules, None); - assert_eq!( - result.matched_reducer.as_deref(), - Some("build/cargo-build"), - "cargo build should match build/cargo-build" - ); - assert_eq!(result.family, "build-rust"); - } - - #[test] - fn cargo_check_matches_build_cargo_build() { - let rules = load_builtin_rules(); - let input = make_input("exec", &["cargo", "check"]); - let result = classify_execution(&input, &rules, None); - assert_eq!( - result.matched_reducer.as_deref(), - Some("build/cargo-build"), - "cargo check should match build/cargo-build via argvIncludesAny" - ); - assert_eq!(result.family, "build-rust"); - } - - #[test] - fn cargo_fmt_matches_lint_cargo_fmt() { - let rules = load_builtin_rules(); - let input = make_input("exec", &["cargo", "fmt", "--check"]); - let result = classify_execution(&input, &rules, None); - assert_eq!( - result.matched_reducer.as_deref(), - Some("lint/cargo-fmt"), - "cargo fmt should match lint/cargo-fmt" - ); - assert_eq!(result.family, "lint-results"); - } - - #[test] - fn cargo_doc_matches_build_cargo_doc() { - let rules = load_builtin_rules(); - let input = make_input("exec", &["cargo", "doc"]); - let result = classify_execution(&input, &rules, None); - assert_eq!( - result.matched_reducer.as_deref(), - Some("build/cargo-doc"), - "cargo doc should match build/cargo-doc" - ); - assert_eq!(result.family, "build-rust"); - } - - #[test] - fn cargo_clippy_does_not_match_cargo_test() { - let rules = load_builtin_rules(); - let input = make_input("exec", &["cargo", "clippy"]); - let result = classify_execution(&input, &rules, None); - assert_ne!( - result.matched_reducer.as_deref(), - Some("tests/cargo-test"), - "cargo clippy must NOT match tests/cargo-test" - ); - } - - #[test] - fn cargo_build_does_not_match_cargo_test() { - let rules = load_builtin_rules(); - let input = make_input("exec", &["cargo", "build"]); - let result = classify_execution(&input, &rules, None); - assert_ne!( - result.matched_reducer.as_deref(), - Some("tests/cargo-test"), - "cargo build must NOT match tests/cargo-test" - ); - } -} diff --git a/src/openhuman/tokenjuice/compress.rs b/src/openhuman/tokenjuice/compress.rs deleted file mode 100644 index 963cb6a44..000000000 --- a/src/openhuman/tokenjuice/compress.rs +++ /dev/null @@ -1,231 +0,0 @@ -//! Universal content-aware compression entry point. -//! -//! [`compress_content`] is the broadly-usable function: hand it any blob and an -//! optional [`ContentHint`], and it detects the content kind, routes to the -//! right compressor, and — when the result drops data — offloads the original -//! to the CCR cache and appends a `⟦tj:⟧` retrieval footer so nothing is -//! ever silently lost. Use it for tool output, file reads, web/HTML fetches, or -//! any large payload headed for the model context. -//! -//! The tool-output adapter -//! [`crate::openhuman::tokenjuice::compact_tool_output_with_policy`] builds a -//! [`CompressInput`] with a derived command/argv and calls [`route`]. - -use crate::openhuman::tokenjuice::cache; -use crate::openhuman::tokenjuice::compressors::{compressor_for, generic_compressor}; -use crate::openhuman::tokenjuice::detect::detect_content_kind; -use crate::openhuman::tokenjuice::savings; -use crate::openhuman::tokenjuice::tokens::estimate_tokens; -use crate::openhuman::tokenjuice::types::{ - CompressInput, CompressOptions, CompressOutput, CompressedOutput, ContentHint, ContentKind, -}; - -/// Compress arbitrary content. Detects the kind (honouring `hint`), routes to -/// the matching compressor, and offloads/marks the original via CCR when lossy. -/// -/// Always pass-through safe: returns the original unchanged when the router is -/// disabled, the input is too small, the content kind has no enabled -/// compressor, or compression wouldn't shrink it. -pub async fn compress_content( - content: &str, - hint: Option, - opts: &CompressOptions, -) -> CompressedOutput { - let hint = hint.unwrap_or_default(); - let input = CompressInput { - content, - kind: ContentKind::PlainText, // resolved inside route() - hint: &hint, - exit_code: None, - command: None, - argv: None, - original_bytes: content.len(), - }; - route(input, opts).await -} - -/// Core router: detect (unless the input already carries a resolved kind via the -/// hint's explicit override), pick the compressor honouring config gates, run -/// it, and apply CCR offload + footer. -pub async fn route(mut input: CompressInput<'_>, opts: &CompressOptions) -> CompressedOutput { - let content = input.content; - let original_bytes = content.len(); - - if !opts.router_enabled || original_bytes < opts.min_bytes_to_compress { - let kind = detect_content_kind(content, input.hint); - return CompressedOutput::passthrough(content.to_string(), kind); - } - - let kind = detect_content_kind(content, input.hint); - input.kind = kind; - - // Resolve which compressor to try, honouring per-kind config gates. - let primary: Option<&'static dyn crate::openhuman::tokenjuice::compressors::Compressor> = - match kind { - ContentKind::Search if !opts.search_enabled => None, - ContentKind::Code if !opts.code_enabled => None, - ContentKind::Html if !opts.html_enabled => None, - _ => Some(compressor_for(kind)), - }; - - // Try the primary compressor; if it declines, fall back to the generic - // head/tail path (which itself declines for non-command payloads). - let mut produced: Option = match primary { - Some(c) => c.compress(&input, opts).await, - None => None, - }; - // When the specialised compressor declines (including plain text with the - // ML compressor off), fall back to the generic head/tail path. It runs the - // rule engine for *command* output (so e.g. `git status` still compacts even - // though it carries no log signal) and declines for domain-tool payloads. - if produced.is_none() { - produced = generic_compressor().compress(&input, opts).await; - } - - let Some(out) = produced else { - return CompressedOutput::passthrough(content.to_string(), kind); - }; - if out.text.len() >= original_bytes { - return CompressedOutput::passthrough(content.to_string(), kind); - } - - // CCR threshold: only offload (and therefore only allow *lossy* compaction) - // when the input is large enough to be worth caching. Below the token - // threshold a lossy result can't be made recoverable, so pass it through; - // lossless reformats are still allowed without an offload. - let original_tokens = estimate_tokens(content); - let ccr_for_call = opts.ccr_enabled && original_tokens as usize >= opts.ccr_min_tokens; - if out.lossy && !ccr_for_call { - return CompressedOutput::passthrough(content.to_string(), kind); - } - - // Offload the original and append a recovery footer when CCR is in play. - let (text, ccr_token) = if ccr_for_call { - let (token, retained) = cache::offload_checked(content); - if !retained { - // The original is too large to keep in memory (over the byte cap) - // and the disk tier isn't on, so it can't be recovered. A lossy view - // would be irreversible — decline it. A lossless reformat is still - // safe to return, just without a (dangling) recovery footer. - if out.lossy { - return CompressedOutput::passthrough(content.to_string(), kind); - } - (out.text, None) - } else { - let footer = cache::recovery_footer(&token, original_bytes, out.lossy); - let mut text = out.text; - text.push_str(&footer); - // The footer adds bytes — if it tipped us over the original size, bail. - if text.len() >= original_bytes { - return CompressedOutput::passthrough(content.to_string(), kind); - } - (text, Some(token)) - } - } else { - (out.text, None) - }; - - let compacted_bytes = text.len(); - let compacted_tokens = estimate_tokens(&text); - log::info!( - "[tokenjuice] compacted kind={} compressor={} lossy={} {}->{} bytes (~{}->{} tok)", - kind.as_str(), - out.kind.as_str(), - out.lossy, - original_bytes, - compacted_bytes, - original_tokens, - compacted_tokens, - ); - - // Record savings for the dashboard (tokens + cost saved for the LLM the - // result is being compressed for). - savings::record(kind, out.kind, original_tokens, compacted_tokens); - - CompressedOutput { - text, - content_kind: kind, - compressor: out.kind, - lossy: out.lossy, - applied: true, - ccr_token, - original_bytes, - compacted_bytes, - } -} - -/// Build a [`CompressorKind`] label-free passthrough quickly (used by callers -/// that only need to detect without compressing). -pub fn detect_only(content: &str, hint: &ContentHint) -> ContentKind { - detect_content_kind(content, hint) -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::openhuman::tokenjuice::types::CompressorKind; - - fn opts() -> CompressOptions { - CompressOptions { - min_bytes_to_compress: 64, - ..Default::default() - } - } - - #[tokio::test] - async fn routes_json_and_offloads() { - let mut rows = Vec::new(); - for i in 0..120 { - rows.push(format!( - r#"{{"id":{i},"name":"account_{i}","email":"a{i}@ex.com","tier":"gold"}}"# - )); - } - let original = format!("[{}]", rows.join(",")); - let res = compress_content(&original, None, &opts()).await; - assert!(res.applied); - assert_eq!(res.content_kind, ContentKind::Json); - assert_eq!(res.compressor, CompressorKind::SmartCrusher); - assert!(res.text.len() < original.len()); - let token = res.ccr_token.expect("offloaded"); - assert_eq!(cache::retrieve(&token).as_deref(), Some(original.as_str())); - assert!( - res.text.contains("⟦tj:"), - "footer marker present: {}", - res.text - ); - } - - #[tokio::test] - async fn small_input_passes_through() { - let res = compress_content("tiny", None, &opts()).await; - assert!(!res.applied); - assert_eq!(res.text, "tiny"); - } - - #[tokio::test] - async fn html_hint_extracts_text() { - let mut html = String::from(""); - for i in 0..60 { - html.push_str(&format!("
cell number {i} content
")); - } - html.push_str(""); - let hint = ContentHint { - mime: Some("text/html".into()), - ..Default::default() - }; - let res = compress_content(&html, Some(hint), &opts()).await; - assert!(res.applied); - assert_eq!(res.content_kind, ContentKind::Html); - assert!(res.text.contains("cell number 7 content")); - } - - #[tokio::test] - async fn router_disabled_is_passthrough() { - let mut o = opts(); - o.router_enabled = false; - let big = "x".repeat(5000); - let res = compress_content(&big, None, &o).await; - assert!(!res.applied); - assert_eq!(res.text, big); - } -} diff --git a/src/openhuman/tokenjuice/compressors/code.rs b/src/openhuman/tokenjuice/compressors/code.rs deleted file mode 100644 index 41828b4ca..000000000 --- a/src/openhuman/tokenjuice/compressors/code.rs +++ /dev/null @@ -1,359 +0,0 @@ -//! Source-code compressor — keep signatures, collapse bodies. -//! -//! Inspired by Headroom's `CodeCompressor`. The goal is to keep the structural -//! skeleton an agent needs to navigate a file — imports, type/function/class -//! signatures, top-level constants — while collapsing the deep bodies that -//! dominate byte count. -//! -//! This module currently ships the language-agnostic **brace-depth heuristic**: -//! lines at brace nesting depth 0–1 are kept; deeper bodies collapse to a -//! `{ … N lines … }` placeholder. Lines carrying error/TODO markers are always -//! kept. A higher-fidelity tree-sitter path (Rust/TS/Python) is layered on in a -//! follow-up slice and selected by language; until then every language uses the -//! heuristic. The router offloads the original to CCR for exact recovery. - -use async_trait::async_trait; -use std::fmt::Write as _; - -use super::Compressor; -use crate::openhuman::tokenjuice::types::{ - CompressInput, CompressOptions, CompressOutput, CompressorKind, -}; - -/// Bodies with more than this many collapsed lines get a placeholder; shorter -/// ones are kept verbatim (collapsing tiny bodies isn't worth the marker). -pub const MIN_BODY_LINES_TO_COLLAPSE: usize = 4; - -/// Markers that force a line to be kept even inside a deep body. -const KEEP_MARKERS: &[&str] = &["TODO", "FIXME", "XXX", "error", "panic", "unsafe"]; - -pub struct CodeCompressor; - -#[async_trait] -impl Compressor for CodeCompressor { - fn kind(&self) -> CompressorKind { - CompressorKind::Code - } - - async fn compress( - &self, - input: &CompressInput<'_>, - _opts: &CompressOptions, - ) -> Option { - // Prefer the AST path when a grammar matches the file's language; fall - // back to the language-agnostic heuristic otherwise (or when the AST - // path doesn't shrink the content). - #[cfg(feature = "tokenjuice-treesitter")] - if let Some(ext) = input.hint.extension.as_deref() { - if let Some(out) = treesitter::compress(input.content, ext) { - return Some(out); - } - } - compress_heuristic(input.content) - } -} - -/// Language-agnostic brace-depth compressor. Keeps lines at depth ≤ 1, collapses -/// deeper runs. Returns `None` if it wouldn't shrink the content. -pub fn compress_heuristic(content: &str) -> Option { - let lines: Vec<&str> = content.lines().collect(); - if lines.len() < 12 { - return None; - } - - let mut out = String::with_capacity(content.len() / 2 + 64); - let mut depth: i32 = 0; - let mut collapsed: Vec<&str> = Vec::new(); - - let flush = |out: &mut String, collapsed: &mut Vec<&str>| { - if collapsed.is_empty() { - return; - } - if collapsed.len() >= MIN_BODY_LINES_TO_COLLAPSE { - let _ = writeln!(out, " {{ … {} line(s) … }}", collapsed.len()); - } else { - for l in collapsed.iter() { - let _ = writeln!(out, "{l}"); - } - } - collapsed.clear(); - }; - - for line in &lines { - let (opens, closes) = brace_delta(line); - let start_depth = depth; - // A line that carries signal is always kept, regardless of depth. - let force_keep = KEEP_MARKERS.iter().any(|m| line.contains(m)); - - // Keep top-level lines (depth 0) — imports, signatures, the line that - // opens a block — and collapse the block body (depth ≥ 1). Short bodies - // (e.g. small struct field lists) stay verbatim via the flush threshold, - // so struct/enum fields survive while long function bodies collapse. - if start_depth == 0 || force_keep { - flush(&mut out, &mut collapsed); - let _ = writeln!(out, "{line}"); - } else { - collapsed.push(line); - } - depth += opens - closes; - if depth < 0 { - depth = 0; - } - } - flush(&mut out, &mut collapsed); - - let out = out.trim_end().to_string(); - if out.len() >= content.len() { - return None; - } - log::debug!( - "[tokenjuice][code] heuristic {} -> {} bytes ({} lines)", - content.len(), - out.len(), - lines.len() - ); - Some(CompressOutput::lossy(out, CompressorKind::Code)) -} - -/// Count `{`/`}` (and `(`/`)`) on a line, ignoring those inside string/char -/// literals and line comments — a cheap approximation good enough for the -/// depth heuristic. -fn brace_delta(line: &str) -> (i32, i32) { - let mut opens = 0i32; - let mut closes = 0i32; - let mut in_str: Option = None; - let mut prev = '\0'; - let mut chars = line.chars().peekable(); - while let Some(c) = chars.next() { - match in_str { - Some(q) => { - if c == q && prev != '\\' { - in_str = None; - } - } - None => match c { - '"' | '\'' | '`' => in_str = Some(c), - '/' if chars.peek() == Some(&'/') => break, // line comment - '#' => break, // python/shell comment - '{' => opens += 1, - '}' => closes += 1, - _ => {} - }, - } - prev = c; - } - (opens, closes) -} - -/// AST-aware code compression via tree-sitter (Rust/TS/JS/Python). Keeps full -/// source but replaces function/method bodies longer than a threshold with a -/// `{ … N lines … }` (or `...` for Python) placeholder, preserving signatures, -/// imports, type declarations and struct/enum fields exactly. -#[cfg(feature = "tokenjuice-treesitter")] -mod treesitter { - use super::{CompressOutput, CompressorKind, MIN_BODY_LINES_TO_COLLAPSE}; - use tree_sitter::{Node, Parser}; - - /// Pick the grammar for a file extension. Returns the language plus whether - /// it is brace-delimited (vs. Python's indentation suite). - fn language_for(ext: &str) -> Option<(tree_sitter::Language, bool)> { - let ext = ext.to_ascii_lowercase(); - match ext.as_str() { - "rs" => Some((tree_sitter_rust::LANGUAGE.into(), true)), - "ts" | "mts" | "cts" => { - Some((tree_sitter_typescript::LANGUAGE_TYPESCRIPT.into(), true)) - } - "tsx" => Some((tree_sitter_typescript::LANGUAGE_TSX.into(), true)), - "js" | "jsx" | "mjs" | "cjs" => { - // The TypeScript grammar is a superset that parses JS too. - Some((tree_sitter_typescript::LANGUAGE_TYPESCRIPT.into(), true)) - } - "py" | "pyi" => Some((tree_sitter_python::LANGUAGE.into(), false)), - _ => None, - } - } - - /// Node kinds whose `body` field is a collapsible function/method body. - const BODY_PARENTS: &[&str] = &[ - "function_item", - "function_declaration", - "function_definition", - "method_definition", - "function", - "arrow_function", - "generator_function_declaration", - ]; - - pub fn compress(content: &str, ext: &str) -> Option { - let (language, braced) = language_for(ext)?; - let mut parser = Parser::new(); - parser.set_language(&language).ok()?; - let tree = parser.parse(content, None)?; - let src = content.as_bytes(); - - // Collect outermost collapsible body byte-ranges. - let mut ranges: Vec<(usize, usize)> = Vec::new(); - collect_bodies(tree.root_node(), src, &mut ranges); - // Sort and drop nested ranges (keep outermost only). - ranges.sort_by_key(|r| r.0); - let mut merged: Vec<(usize, usize)> = Vec::new(); - for r in ranges { - if let Some(last) = merged.last() { - if r.0 < last.1 { - continue; // nested inside a body we're already collapsing - } - } - merged.push(r); - } - if merged.is_empty() { - return None; - } - - let mut out = String::with_capacity(content.len()); - let mut cursor = 0usize; - for (start, end) in merged { - if start < cursor { - continue; - } - out.push_str(&content[cursor..start]); - let body = &content[start..end]; - let n_lines = body.lines().count(); - if n_lines < MIN_BODY_LINES_TO_COLLAPSE { - out.push_str(body); - } else if braced { - out.push_str(&format!("{{ … {n_lines} line(s) … }}")); - } else { - // Python suite — keep an indented ellipsis so it still reads. - out.push_str(&format!("... # {n_lines} line(s) collapsed")); - } - cursor = end; - } - out.push_str(&content[cursor..]); - - let out = out.trim_end().to_string(); - if out.len() >= content.len() { - return None; - } - log::debug!( - "[tokenjuice][code] tree-sitter ext={} {} -> {} bytes", - ext, - content.len(), - out.len() - ); - Some(CompressOutput::lossy(out, CompressorKind::Code)) - } - - /// Recursively collect the byte-ranges of function/method bodies. - fn collect_bodies(node: Node, src: &[u8], out: &mut Vec<(usize, usize)>) { - if BODY_PARENTS.contains(&node.kind()) { - if let Some(body) = node.child_by_field_name("body") { - out.push((body.start_byte(), body.end_byte())); - // Don't descend into a collapsed body. - let _ = src; - return; - } - } - let mut cursor = node.walk(); - for child in node.children(&mut cursor) { - collect_bodies(child, src, out); - } - } -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn keeps_signatures_collapses_bodies() { - let mut src = String::from("use std::collections::HashMap;\n\n"); - src.push_str("pub fn process(items: &[i32]) -> i32 {\n"); - for i in 0..30 { - src.push_str(&format!( - " let tmp_{i} = items.iter().sum::() + {i};\n" - )); - } - src.push_str(" tmp_0\n}\n\n"); - src.push_str("struct Config {\n name: String,\n size: usize,\n}\n"); - let out = compress_heuristic(&src).expect("compresses"); - assert!(out.lossy); - assert!( - out.text.contains("pub fn process"), - "signature kept:\n{}", - out.text - ); - assert!(out.text.contains("struct Config")); - assert!( - out.text.contains("line(s) …"), - "body collapsed:\n{}", - out.text - ); - assert!( - !out.text.contains("tmp_15"), - "deep body should be collapsed" - ); - assert!(out.text.len() < src.len()); - } - - #[test] - fn short_file_passes_through() { - let src = "fn a() {}\nfn b() {}\n"; - assert!(compress_heuristic(src).is_none()); - } - - #[cfg(feature = "tokenjuice-treesitter")] - #[test] - fn treesitter_collapses_rust_body_keeps_struct() { - let mut src = String::from("use std::collections::HashMap;\n\n"); - src.push_str("pub fn process(items: &[i32]) -> i32 {\n"); - for i in 0..30 { - src.push_str(&format!( - " let tmp_{i} = items.iter().sum::() + {i};\n" - )); - } - src.push_str(" tmp_0\n}\n\n"); - src.push_str("pub struct Config {\n pub name: String,\n pub size: usize,\n}\n"); - let out = treesitter::compress(&src, "rs").expect("compresses"); - assert!( - out.text.contains("pub fn process(items: &[i32]) -> i32"), - "{}", - out.text - ); - // Struct fields preserved exactly (not a function body). - assert!(out.text.contains("pub name: String"), "{}", out.text); - assert!(out.text.contains("pub size: usize")); - // Function body collapsed. - assert!(out.text.contains("line(s) …"), "{}", out.text); - assert!(!out.text.contains("tmp_15")); - assert!(out.text.len() < src.len()); - } - - #[cfg(feature = "tokenjuice-treesitter")] - #[test] - fn treesitter_collapses_python_body() { - let mut src = String::from("import os\n\ndef handler(event):\n"); - for i in 0..30 { - src.push_str(&format!(" x_{i} = compute(event, {i})\n")); - } - src.push_str(" return x_0\n"); - let out = treesitter::compress(&src, "py").expect("compresses"); - assert!(out.text.contains("def handler(event):"), "{}", out.text); - assert!(out.text.contains("collapsed"), "{}", out.text); - assert!(!out.text.contains("x_15")); - } - - #[test] - fn keeps_marker_lines_in_body() { - let mut src = String::from("fn f() {\n"); - for i in 0..20 { - if i == 10 { - src.push_str(" // TODO: handle the edge case here\n"); - } else { - src.push_str(&format!(" do_thing({i});\n")); - } - } - src.push_str("}\n"); - let out = compress_heuristic(&src).expect("compresses"); - assert!(out.text.contains("TODO"), "marker line kept:\n{}", out.text); - } -} diff --git a/src/openhuman/tokenjuice/compressors/diff.rs b/src/openhuman/tokenjuice/compressors/diff.rs deleted file mode 100644 index 61b4c3f4e..000000000 --- a/src/openhuman/tokenjuice/compressors/diff.rs +++ /dev/null @@ -1,220 +0,0 @@ -//! Unified-diff compressor. -//! -//! Clean-room port of Headroom's `DiffCompressor` (Apache-2.0). Keeps the -//! signal (changed lines, structural headers, hunk headers) and collapses long -//! runs of unchanged context to an anchor + marker. Lockfile/bundle hunks -//! collapse to a one-line `+A/-B` summary. The router offloads the original to -//! CCR, so the dropped context stays recoverable. - -use async_trait::async_trait; -use std::fmt::Write as _; - -use super::Compressor; -use crate::openhuman::tokenjuice::text::ansi::strip_ansi; -use crate::openhuman::tokenjuice::types::{ - CompressInput, CompressOptions, CompressOutput, CompressorKind, -}; - -/// Context lines kept on each side of a changed run before collapsing. -pub const CONTEXT_ANCHOR: usize = 3; -/// A run of unchanged context longer than this collapses to a marker. -pub const CONTEXT_COLLAPSE_THRESHOLD: usize = 8; - -pub struct DiffCompressor; - -#[async_trait] -impl Compressor for DiffCompressor { - fn kind(&self) -> CompressorKind { - CompressorKind::Diff - } - - async fn compress( - &self, - input: &CompressInput<'_>, - _opts: &CompressOptions, - ) -> Option { - compress(input.content) - } -} - -/// Compress a unified diff. Returns `None` when there's nothing structural to -/// work with or compression wouldn't shrink it. -pub fn compress(content: &str) -> Option { - let stripped = strip_ansi(content); - let lines: Vec<&str> = stripped.lines().collect(); - if lines.is_empty() { - return None; - } - - let mut out = String::with_capacity(stripped.len() / 2 + 64); - let mut i = 0usize; - let mut current_file_is_noisy = false; - let mut saw_hunk = false; - - while i < lines.len() { - let line = lines[i]; - - if line.starts_with("diff --git ") { - current_file_is_noisy = is_noisy_path(line); - let _ = writeln!(out, "{line}"); - i += 1; - continue; - } - if is_structural(line) { - saw_hunk |= line.starts_with("@@"); - if current_file_is_noisy && line.starts_with("@@") { - let _ = writeln!(out, "{line}"); - i += 1; - let (added, removed, consumed) = summarize_hunk_body(&lines[i..]); - let _ = writeln!( - out, - "[... lockfile/bundle hunk: +{added}/-{removed} line(s) omitted ...]" - ); - i += consumed; - continue; - } - let _ = writeln!(out, "{line}"); - i += 1; - continue; - } - - if is_context(line) { - let start = i; - while i < lines.len() && is_context(lines[i]) { - i += 1; - } - let run = &lines[start..i]; - if run.len() > CONTEXT_COLLAPSE_THRESHOLD { - for l in &run[..CONTEXT_ANCHOR] { - let _ = writeln!(out, "{l}"); - } - let omitted = run.len() - 2 * CONTEXT_ANCHOR; - let _ = writeln!(out, "[... {omitted} context line(s) omitted ...]"); - for l in &run[run.len() - CONTEXT_ANCHOR..] { - let _ = writeln!(out, "{l}"); - } - } else { - for l in run { - let _ = writeln!(out, "{l}"); - } - } - continue; - } - - let _ = writeln!(out, "{line}"); - i += 1; - } - - if !saw_hunk { - return None; - } - if out.len() >= content.len() { - return None; - } - log::debug!( - "[tokenjuice][diff] {} -> {} bytes ({} input lines)", - content.len(), - out.len(), - lines.len(), - ); - Some(CompressOutput::lossy( - out.trim_end().to_string(), - CompressorKind::Diff, - )) -} - -fn is_structural(line: &str) -> bool { - line.starts_with("@@") - || line.starts_with("--- ") - || line.starts_with("+++ ") - || line.starts_with("index ") - || line.starts_with("new file") - || line.starts_with("deleted file") - || line.starts_with("rename ") - || line.starts_with("similarity ") - || line.starts_with("Binary files") -} - -fn is_context(line: &str) -> bool { - line.starts_with(' ') -} - -fn summarize_hunk_body(lines: &[&str]) -> (usize, usize, usize) { - let mut added = 0usize; - let mut removed = 0usize; - let mut n = 0usize; - for &line in lines { - if line.starts_with("@@") || line.starts_with("diff --git ") { - break; - } - if line.starts_with('+') && !line.starts_with("+++") { - added += 1; - } else if line.starts_with('-') && !line.starts_with("---") { - removed += 1; - } - n += 1; - } - (added, removed, n) -} - -fn is_noisy_path(diff_git_line: &str) -> bool { - let l = diff_git_line.to_ascii_lowercase(); - const NOISY: &[&str] = &[ - "cargo.lock", - "package-lock.json", - "pnpm-lock.yaml", - "yarn.lock", - "composer.lock", - "poetry.lock", - "gemfile.lock", - ".min.js", - ".min.css", - ".map", - "go.sum", - ]; - NOISY.iter().any(|p| l.contains(p)) -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn keeps_changed_lines_collapses_context() { - let mut s = String::from("diff --git a/x.rs b/x.rs\n@@ -1,40 +1,41 @@\n"); - for i in 0..20 { - let _ = writeln!(s, " context line {i} unchanged here"); - } - let _ = writeln!(s, "-old changed line"); - let _ = writeln!(s, "+new changed line"); - for i in 0..20 { - let _ = writeln!(s, " more context {i} unchanged"); - } - let out = compress(&s).expect("compresses").text; - assert!(out.contains("-old changed line"), "{out}"); - assert!(out.contains("+new changed line"), "{out}"); - assert!(out.contains("context line(s) omitted"), "{out}"); - assert!(out.contains("@@ -1,40 +1,41 @@")); - assert!(out.len() < s.len()); - } - - #[test] - fn summarizes_lockfile_hunk() { - let mut s = String::from("diff --git a/Cargo.lock b/Cargo.lock\n@@ -1,60 +1,80 @@\n"); - for i in 0..40 { - let _ = writeln!(s, "+ new dep entry {i}"); - } - for i in 0..20 { - let _ = writeln!(s, "- old dep entry {i}"); - } - let out = compress(&s).expect("compresses").text; - assert!(out.contains("lockfile/bundle hunk"), "{out}"); - assert!(!out.contains("new dep entry 7"), "{out}"); - assert!(out.len() < s.len()); - } - - #[test] - fn non_diff_returns_none() { - assert!(compress("just some text\nno hunks here").is_none()); - } -} diff --git a/src/openhuman/tokenjuice/compressors/generic.rs b/src/openhuman/tokenjuice/compressors/generic.rs deleted file mode 100644 index aac0f85d3..000000000 --- a/src/openhuman/tokenjuice/compressors/generic.rs +++ /dev/null @@ -1,39 +0,0 @@ -//! Generic line-oriented fallback compressor. -//! -//! Used by the router when a specialised compressor declines (or the ML text -//! compressor is unavailable). It runs the rule engine's `generic/fallback` -//! head/tail summariser — but **only for command output**. For domain-tool -//! payloads (no derived command/argv) it declines, preserving the long-standing -//! guard that large structured tool results must reach the downstream -//! progressive-disclosure handoff rather than being blindly head/tail clamped. - -use async_trait::async_trait; - -use super::Compressor; -use crate::openhuman::tokenjuice::compressors::log::compress_command_fallback; -use crate::openhuman::tokenjuice::types::{ - CompressInput, CompressOptions, CompressOutput, CompressorKind, -}; - -pub struct GenericCompressor; - -#[async_trait] -impl Compressor for GenericCompressor { - fn kind(&self) -> CompressorKind { - CompressorKind::Generic - } - - async fn compress( - &self, - input: &CompressInput<'_>, - opts: &CompressOptions, - ) -> Option { - let has_command = - input.command.is_some() || input.argv.as_ref().is_some_and(|a| !a.is_empty()); - if !has_command { - // Domain-tool payload — decline rather than blind-truncate. - return None; - } - compress_command_fallback(input, opts) - } -} diff --git a/src/openhuman/tokenjuice/compressors/html.rs b/src/openhuman/tokenjuice/compressors/html.rs deleted file mode 100644 index b99de09db..000000000 --- a/src/openhuman/tokenjuice/compressors/html.rs +++ /dev/null @@ -1,262 +0,0 @@ -//! HTML → readable-text extractor. -//! -//! Strips markup and returns the readable text content, in the spirit of -//! Headroom's `HTMLExtractor`. Linear-time, allocation-light (no DOM, no -//! regex): it scans once, dropping `

Title

Hello world.

"; - let text = html_to_text(html); - assert!(text.contains("Title")); - assert!(text.contains("Hello")); - assert!(text.contains("world")); - assert!( - !text.contains("alert"), - "script body must be dropped: {text}" - ); - assert!(!text.contains("color:red"), "style body must be dropped"); - } - - #[test] - fn decodes_entities() { - let text = html_to_text("

a & b < c > d  e

"); - assert!(text.contains("a & b < c > d"), "{text}"); - } - - #[test] - fn block_tags_insert_newlines() { - let text = html_to_text("

one

two

  • three
  • "); - let lines: Vec<&str> = text.lines().filter(|l| !l.trim().is_empty()).collect(); - assert!(lines.len() >= 3, "expected separate lines, got {lines:?}"); - } - - #[test] - fn compress_shrinks_real_doc() { - let mut html = String::from(""); - for i in 0..50 { - html.push_str(&format!( - "
    cell {i}
    " - )); - } - html.push_str(""); - let out = compress(&html).expect("compresses"); - assert!(out.lossy); - assert!(out.text.len() < html.len()); - assert!(out.text.contains("cell 7")); - } -} diff --git a/src/openhuman/tokenjuice/compressors/json.rs b/src/openhuman/tokenjuice/compressors/json.rs deleted file mode 100644 index 3a305f737..000000000 --- a/src/openhuman/tokenjuice/compressors/json.rs +++ /dev/null @@ -1,310 +0,0 @@ -//! JSON-array crusher (SmartCrusher). -//! -//! Clean-room port of Headroom's `SmartCrusher` (Apache-2.0), extended with -//! variance-aware row preservation. An array of objects that repeat the same -//! keys is the single most common bloated tool output (API list responses, DB -//! rows, search manifests). Re-rendering it as a table emits each key **once** -//! instead of per row. -//! -//! Up to [`ROW_DROP_THRESHOLD`] rows every value is preserved (faithful -//! reformat). Above the threshold the table is row-dropped — but rather than a -//! blind head/tail window, rows carrying **error indicators** or **numeric -//! outliers** are always kept, so anomalies survive even when the bulk of a -//! homogeneous array is dropped. The router offloads the full original to CCR, -//! so the dropped rows stay recoverable. - -use async_trait::async_trait; -use serde_json::Value; -use std::collections::BTreeSet; -use std::fmt::Write as _; - -use super::signals::has_error_indicators; -use super::Compressor; -use crate::openhuman::tokenjuice::types::{ - CompressInput, CompressOptions, CompressOutput, CompressorKind, -}; - -/// Minimum rows before tabular rendering is worth the header overhead. -pub const MIN_ROWS: usize = 3; -/// Above this many rows the table is additionally row-dropped. -pub const ROW_DROP_THRESHOLD: usize = 40; -/// Rows kept from the head when row-dropping. -pub const HEAD_ROWS: usize = 20; -/// Rows kept from the tail when row-dropping. -pub const TAIL_ROWS: usize = 10; -/// Z-score beyond which a numeric cell is treated as an outlier worth keeping. -pub const OUTLIER_SIGMA: f64 = 2.0; - -pub struct JsonCompressor; - -#[async_trait] -impl Compressor for JsonCompressor { - fn kind(&self) -> CompressorKind { - CompressorKind::SmartCrusher - } - - async fn compress( - &self, - input: &CompressInput<'_>, - _opts: &CompressOptions, - ) -> Option { - compress(input.content) - } -} - -/// Compress a JSON array-of-objects into a compact table. Returns `None` when -/// the content isn't a uniform-enough array of objects or wouldn't shrink. -pub fn compress(content: &str) -> Option { - let value: Value = serde_json::from_str(content.trim()).ok()?; - let array = value.as_array()?; - if array.len() < MIN_ROWS { - return None; - } - if !array.iter().all(Value::is_object) { - return None; - } - - // Column order = first-seen key order across all rows (union, stable). - let mut columns: Vec = Vec::new(); - for item in array { - if let Some(obj) = item.as_object() { - for key in obj.keys() { - if !columns.iter().any(|c| c == key) { - columns.push(key.clone()); - } - } - } - } - if columns.len() < 2 { - return None; - } - - // Render every row's cells up front so we can choose full vs. row-dropped. - let mut rows: Vec = Vec::with_capacity(array.len()); - for item in array { - let obj = item.as_object()?; - let cells: Vec = columns - .iter() - .map(|col| match obj.get(col) { - None => String::new(), - Some(v) => render_cell(v), - }) - .collect(); - rows.push(cells.join(" | ")); - } - - let lossy = rows.len() > ROW_DROP_THRESHOLD; - let mut out = String::with_capacity(content.len()); - let _ = writeln!( - out, - "[json table: {} rows × {} cols · blank=absent key · exact original via retrieve footer]", - rows.len(), - columns.len() - ); - let _ = writeln!(out, "{}", columns.join(" | ")); - - if lossy { - // Keep head + tail PLUS any anomalous rows (errors / numeric outliers) - // so the signal in a large homogeneous array survives row-dropping. - let keep = rows_to_keep(array, &columns, rows.len()); - let mut prev: Option = None; - for &i in &keep { - if let Some(p) = prev { - let gap = i - p - 1; - if gap > 0 { - let _ = writeln!(out, "[... {gap} row(s) omitted ...]"); - } - } else if i > 0 { - let _ = writeln!(out, "[... {i} row(s) omitted ...]"); - } - let _ = writeln!(out, "{}", rows[i]); - prev = Some(i); - } - if let Some(p) = prev { - let tail = rows.len().saturating_sub(p + 1); - if tail > 0 { - let _ = writeln!(out, "[... {tail} row(s) omitted ...]"); - } - } - } else { - for row in &rows { - let _ = writeln!(out, "{row}"); - } - } - - let out = out.trim_end().to_string(); - if out.len() >= content.len() { - return None; - } - log::debug!( - "[tokenjuice][json] {} rows × {} cols, lossy={} ({} -> {} bytes)", - rows.len(), - columns.len(), - lossy, - content.len(), - out.len(), - ); - if lossy { - Some(CompressOutput::lossy(out, CompressorKind::SmartCrusher)) - } else { - // All values preserved, but the array→table reformat changes layout. - Some(CompressOutput::reformatted( - out, - CompressorKind::SmartCrusher, - )) - } -} - -/// Pick the row indices to keep when row-dropping: the head/tail windows plus -/// any row flagged as anomalous (error text or a numeric outlier in any -/// column). Returns ascending, de-duplicated indices. -fn rows_to_keep(array: &[Value], columns: &[String], n: usize) -> Vec { - let mut keep: BTreeSet = BTreeSet::new(); - for i in 0..HEAD_ROWS.min(n) { - keep.insert(i); - } - for i in n.saturating_sub(TAIL_ROWS)..n { - keep.insert(i); - } - - // Error-text rows: any string/scalar cell carrying an error indicator. - for (i, item) in array.iter().enumerate() { - if let Some(obj) = item.as_object() { - let row_has_error = obj.values().any(|v| match v { - Value::String(s) => has_error_indicators(s), - other => has_error_indicators(&other.to_string()), - }); - if row_has_error { - keep.insert(i); - } - } - } - - // Numeric outliers: for each column, compute mean/std over numeric cells and - // keep rows whose value is beyond OUTLIER_SIGMA. Bounded by a cap so a wide - // anomalous tail can't defeat the point of dropping. - for col in columns { - let nums: Vec<(usize, f64)> = array - .iter() - .enumerate() - .filter_map(|(i, item)| { - item.as_object() - .and_then(|o| o.get(col)) - .and_then(Value::as_f64) - .map(|x| (i, x)) - }) - .collect(); - if nums.len() < 4 { - continue; - } - let mean = nums.iter().map(|(_, x)| x).sum::() / nums.len() as f64; - let var = nums.iter().map(|(_, x)| (x - mean).powi(2)).sum::() / nums.len() as f64; - let std = var.sqrt(); - if std <= f64::EPSILON { - continue; - } - for (i, x) in nums { - if ((x - mean) / std).abs() >= OUTLIER_SIGMA { - keep.insert(i); - } - } - } - - keep.into_iter().collect() -} - -/// Render a single cell. Scalars print bare-ish; nested values stay as compact -/// JSON so the table remains lossless. -fn render_cell(v: &Value) -> String { - match v { - Value::String(s) if !s.contains('|') && !s.contains('\n') => s.clone(), - Value::Bool(b) => b.to_string(), - Value::Number(n) => n.to_string(), - other => serde_json::to_string(other).unwrap_or_default(), - } -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn crushes_uniform_array() { - let mut rows = Vec::new(); - for i in 0..20 { - rows.push(format!( - r#"{{"id":{i},"name":"item number {i}","status":"active","owner":"team-alpha"}}"# - )); - } - let input = format!("[{}]", rows.join(",")); - let out = compress(&input).expect("compresses").text; - assert_eq!(out.matches("status").count(), 1, "{out}"); - assert!(out.contains("item number 7")); - assert!(out.len() < input.len(), "expected shrink"); - } - - #[test] - fn large_array_row_drops_and_is_marked_lossy() { - let mut rows = Vec::new(); - for i in 0..200 { - rows.push(format!( - r#"{{"id":{i},"name":"record number {i}","status":"active","note":"some detail {i}"}}"# - )); - } - let input = format!("[{}]", rows.join(",")); - let c = compress(&input).expect("compresses"); - assert!(c.lossy, "row-dropped output must be lossy"); - assert!(c.text.contains("record number 0"), "{}", c.text); - assert!(c.text.contains("record number 199"), "{}", c.text); - assert!(c.text.contains("omitted")); - assert!(c.text.len() < input.len()); - } - - #[test] - fn keeps_error_row_in_dropped_middle() { - // A homogeneous array with a single error row buried in the middle: the - // SmartCrusher must keep that row even though it's in the drop window. - let mut rows = Vec::new(); - for i in 0..120 { - let status = if i == 75 { "error: timeout" } else { "ok" }; - rows.push(format!( - r#"{{"id":{i},"name":"job {i}","status":"{status}","note":"detail {i}"}}"# - )); - } - let input = format!("[{}]", rows.join(",")); - let c = compress(&input).expect("compresses"); - assert!(c.lossy); - assert!( - c.text.contains("job 75"), - "error row must survive:\n{}", - c.text - ); - assert!(c.text.contains("error: timeout")); - } - - #[test] - fn keeps_numeric_outlier_row() { - let mut rows = Vec::new(); - for i in 0..120 { - // Most latencies ~10ms; row 88 is a 9999ms outlier. - let latency = if i == 88 { 9999 } else { 10 + (i % 3) }; - rows.push(format!( - r#"{{"id":{i},"endpoint":"/api/{i}","latency_ms":{latency},"region":"us"}}"# - )); - } - let input = format!("[{}]", rows.join(",")); - let c = compress(&input).expect("compresses"); - assert!( - c.text.contains("9999"), - "outlier row must survive:\n{}", - c.text - ); - } - - #[test] - fn non_array_returns_none() { - assert!(compress(r#"{"a":1}"#).is_none()); - assert!(compress("[1,2,3]").is_none()); - assert!(compress(r#"[{"a":1}]"#).is_none()); - } -} diff --git a/src/openhuman/tokenjuice/compressors/log.rs b/src/openhuman/tokenjuice/compressors/log.rs deleted file mode 100644 index a8e3a12ec..000000000 --- a/src/openhuman/tokenjuice/compressors/log.rs +++ /dev/null @@ -1,320 +0,0 @@ -//! Build/test/lint log compressor. -//! -//! Two paths, chosen by whether the content is *command* output: -//! -//! - **Command output** (the [`CompressInput`] carries a derived command/argv): -//! run the 100-rule reduction engine ([`reduce_execution_with_rules`]). This -//! is TokenJuice's original behaviour — git/cargo/npm/docker-aware rules with -//! failure preservation. -//! - **Non-command logs** (a blob detected as a log with no command context): -//! the signal-based keep-failures/drop-noise compressor, a clean-room port of -//! Headroom's `LogCompressor` (Apache-2.0). -//! -//! The signal path declines (returns `None`) when a blob has no error/warning/ -//! summary/stack signal at all — that almost certainly isn't a log (a file -//! listing, CSV, generated data) and must not be head/tail truncated. - -use async_trait::async_trait; -use once_cell::sync::Lazy; -use std::collections::HashSet; -use std::fmt::Write as _; - -use super::signals::{severity, Severity}; -use super::Compressor; -use crate::openhuman::tokenjuice::reduce::reduce_execution_with_rules; -use crate::openhuman::tokenjuice::rules::load_builtin_rules; -use crate::openhuman::tokenjuice::types::{ - CompiledRule, CompressInput, CompressOptions, CompressOutput, CompressorKind, ReduceOptions, - ToolExecutionInput, -}; - -pub const MAX_ERRORS: usize = 10; -pub const MAX_WARNINGS: usize = 5; -pub const MAX_STACK_TRACES: usize = 3; -pub const STACK_TRACE_MAX_LINES: usize = 20; -pub const MAX_TOTAL_LINES: usize = 100; - -static BUILTIN_RULES: Lazy> = Lazy::new(load_builtin_rules); - -pub struct LogCompressor; - -#[async_trait] -impl Compressor for LogCompressor { - fn kind(&self) -> CompressorKind { - CompressorKind::Log - } - - async fn compress( - &self, - input: &CompressInput<'_>, - opts: &CompressOptions, - ) -> Option { - let has_command = - input.command.is_some() || input.argv.as_ref().is_some_and(|a| !a.is_empty()); - if has_command { - compress_command(input, opts) - } else { - compress_signal(input.content) - } - } -} - -/// Command output → run the rule engine, tagged as [`CompressorKind::Log`]. -fn compress_command(input: &CompressInput<'_>, opts: &CompressOptions) -> Option { - run_rule_engine(input, opts, CompressorKind::Log) -} - -/// The generic fallback path: same rule engine, tagged [`CompressorKind::Generic`]. -/// Exposed for [`super::generic::GenericCompressor`] so the router can fall back -/// to head/tail summarisation of command output without re-implementing it. -pub fn compress_command_fallback( - input: &CompressInput<'_>, - opts: &CompressOptions, -) -> Option { - run_rule_engine(input, opts, CompressorKind::Generic) -} - -/// Run the 100-rule reduction engine over command output, reporting `kind`. -/// Returns `None` when reduction wouldn't shrink the payload. -fn run_rule_engine( - input: &CompressInput<'_>, - opts: &CompressOptions, - kind: CompressorKind, -) -> Option { - let exec = ToolExecutionInput { - tool_name: input - .hint - .source_tool - .clone() - .unwrap_or_else(|| "shell".to_string()), - command: input.command.clone(), - argv: input.argv.clone(), - stdout: Some(input.content.to_string()), - exit_code: input.exit_code, - ..Default::default() - }; - let reduce_opts = ReduceOptions { - max_inline_chars: opts.max_inline_chars, - ..Default::default() - }; - let result = reduce_execution_with_rules(exec, &BUILTIN_RULES, &reduce_opts); - if result.inline_text.len() >= input.content.len() { - return None; - } - let rule_label = result - .classification - .matched_reducer - .unwrap_or(result.classification.family); - log::debug!( - "[tokenjuice][log] command rule={} kind={} {} -> {} bytes", - rule_label, - kind.as_str(), - input.content.len(), - result.inline_text.len() - ); - Some(CompressOutput::lossy(result.inline_text, kind)) -} - -/// Signal-based log compression for non-command blobs detected as logs. -pub fn compress_signal(content: &str) -> Option { - let lines: Vec<&str> = content.lines().collect(); - if lines.len() <= MAX_TOTAL_LINES { - return None; - } - - let mut keep: std::collections::BTreeSet = std::collections::BTreeSet::new(); - - for (i, line) in lines.iter().enumerate() { - if is_summary_line(line) { - keep.insert(i); - } - } - - let error_idx: Vec = lines - .iter() - .enumerate() - .filter(|(_, l)| severity(l) == Severity::Error) - .map(|(i, _)| i) - .collect(); - for &i in select_first_last(&error_idx, MAX_ERRORS).iter() { - keep.insert(i); - } - - let mut seen_warn: HashSet = HashSet::new(); - let mut warn_kept = 0usize; - for (i, line) in lines.iter().enumerate() { - if warn_kept >= MAX_WARNINGS { - break; - } - if severity(line) == Severity::Warning { - let norm = normalize_for_dedupe(line); - if seen_warn.insert(norm) { - keep.insert(i); - warn_kept += 1; - } - } - } - - let mut traces_kept = 0usize; - let mut i = 0usize; - while i < lines.len() && traces_kept < MAX_STACK_TRACES { - if is_stack_frame(lines[i]) { - let start = i; - let mut taken = 0usize; - while i < lines.len() && is_stack_frame(lines[i]) { - if taken < STACK_TRACE_MAX_LINES { - keep.insert(i); - taken += 1; - } - i += 1; - } - if i > start { - traces_kept += 1; - } - } else { - i += 1; - } - } - - if keep.is_empty() { - // Not a log (no signal) — never head/tail truncate legitimate data. - return None; - } - - let kept_vec: Vec = keep.iter().copied().collect(); - let kept_vec = if kept_vec.len() > MAX_TOTAL_LINES { - select_first_last(&kept_vec, MAX_TOTAL_LINES) - } else { - kept_vec - }; - let kept_set: std::collections::BTreeSet = kept_vec.into_iter().collect(); - - let mut out = String::with_capacity(content.len() / 2 + 64); - let mut prev: Option = None; - for &i in &kept_set { - if let Some(p) = prev { - let gap = i - p - 1; - if gap > 0 { - let _ = writeln!(out, "[... {gap} line(s) omitted ...]"); - } - } else if i > 0 { - let _ = writeln!(out, "[... {i} line(s) omitted ...]"); - } - let _ = writeln!(out, "{}", lines[i]); - prev = Some(i); - } - if let Some(p) = prev { - let tail = lines.len().saturating_sub(p + 1); - if tail > 0 { - let _ = writeln!(out, "[... {tail} line(s) omitted ...]"); - } - } - - if out.len() >= content.len() { - return None; - } - log::debug!( - "[tokenjuice][log] signal kept {} of {} line(s)", - kept_set.len(), - lines.len(), - ); - Some(CompressOutput::lossy( - out.trim_end().to_string(), - CompressorKind::Log, - )) -} - -fn select_first_last(idx: &[usize], cap: usize) -> Vec { - if idx.len() <= cap { - return idx.to_vec(); - } - if cap == 0 { - return Vec::new(); - } - let head = cap.div_ceil(2); - let tail = cap - head; - let mut out: std::collections::BTreeSet = std::collections::BTreeSet::new(); - for &i in idx.iter().take(head) { - out.insert(i); - } - for &i in idx.iter().rev().take(tail) { - out.insert(i); - } - out.into_iter().collect() -} - -fn is_summary_line(line: &str) -> bool { - let l = line.to_ascii_lowercase(); - let l = l.trim(); - l.starts_with("test result:") - || l.starts_with("error: aborting") - || l.contains(" passed") - || l.contains(" failed") - || l.contains("tests passed") - || l.contains("tests failed") - || l.contains("failures:") - || (l.contains("warning") && l.contains("generated")) - || l.starts_with("error: could not compile") - || l.starts_with("build failed") - || l.starts_with("build succeeded") - || (l.contains("npm") && l.contains("err")) -} - -fn is_stack_frame(line: &str) -> bool { - let trimmed = line.trim_start(); - if trimmed.is_empty() { - return false; - } - let indented = line.starts_with(' ') || line.starts_with('\t'); - indented - && (trimmed.starts_with("at ") - || trimmed.starts_with("File \"") - || (trimmed.starts_with('#') && trimmed[1..].starts_with(|c: char| c.is_ascii_digit()))) -} - -fn normalize_for_dedupe(line: &str) -> String { - line.chars() - .filter(|c| !c.is_ascii_digit()) - .collect::() - .to_ascii_lowercase() - .split_whitespace() - .collect::>() - .join(" ") -} - -#[cfg(test)] -mod tests { - use super::*; - - fn noisy_log() -> String { - let mut s = String::new(); - for i in 0..200 { - let _ = writeln!(s, " Compiling crate_{i} v0.1.0"); - } - let _ = writeln!(s, "error[E0382]: borrow of moved value `x`"); - let _ = writeln!(s, " --> src/main.rs:10:5"); - let _ = writeln!(s, "error: aborting due to previous error"); - let _ = writeln!(s, "test result: FAILED. 3 passed; 1 failed"); - s - } - - #[test] - fn signal_keeps_errors_and_summary_drops_noise() { - let input = noisy_log(); - let out = compress_signal(&input).expect("compresses").text; - assert!(out.contains("error[E0382]"), "{out}"); - assert!(out.contains("error: aborting"), "{out}"); - assert!(out.contains("test result: FAILED"), "{out}"); - assert!(out.len() < input.len()); - assert!(out.contains("omitted")); - } - - #[test] - fn non_log_data_passes_through() { - let mut s = String::new(); - for i in 0..400 { - let _ = writeln!(s, "/var/data/file_{i:04}.bin\t{i}\trwxr-xr-x"); - } - assert!(compress_signal(&s).is_none()); - } -} diff --git a/src/openhuman/tokenjuice/compressors/ml_text.rs b/src/openhuman/tokenjuice/compressors/ml_text.rs deleted file mode 100644 index 65e17f9a3..000000000 --- a/src/openhuman/tokenjuice/compressors/ml_text.rs +++ /dev/null @@ -1,48 +0,0 @@ -//! ML plain-text compressor ("Kompress") — trait slot. -//! -//! Plain text has no structural skeleton to exploit, so high-quality -//! compression needs a learned model (Headroom uses ModernBERT token -//! classification to drop low-salience spans). That path runs the `kompress` -//! backend of the shared `runtime_python_server` and is **opt-in** behind the -//! `tokenjuice.ml_compression_enabled` config flag. -//! -//! This module is the [`Compressor`] slot; it delegates to -//! [`crate::openhuman::tokenjuice::ml`]. Whenever the flag is off or the Python -//! runtime is unavailable, `compress` declines so the router falls back to the -//! generic compressor — never an error in the agent loop. - -use async_trait::async_trait; - -use super::Compressor; -use crate::openhuman::tokenjuice::types::{ - CompressInput, CompressOptions, CompressOutput, CompressorKind, -}; - -pub struct MlTextCompressor; - -#[async_trait] -impl Compressor for MlTextCompressor { - fn kind(&self) -> CompressorKind { - CompressorKind::MlText - } - - async fn compress( - &self, - input: &CompressInput<'_>, - opts: &CompressOptions, - ) -> Option { - if !opts.ml_text_enabled { - return None; - } - match crate::openhuman::tokenjuice::ml::compress(input.content, opts).await { - Ok(Some(text)) if text.len() < input.content.len() => { - Some(CompressOutput::lossy(text, CompressorKind::MlText)) - } - Ok(_) => None, - Err(e) => { - log::debug!("[tokenjuice][ml] unavailable, falling back: {e:#}"); - None - } - } - } -} diff --git a/src/openhuman/tokenjuice/compressors/mod.rs b/src/openhuman/tokenjuice/compressors/mod.rs deleted file mode 100644 index 8cc2b28c6..000000000 --- a/src/openhuman/tokenjuice/compressors/mod.rs +++ /dev/null @@ -1,77 +0,0 @@ -//! Per-content-kind compressors and the registry that maps a [`ContentKind`] -//! to the [`Compressor`] that handles it. -//! -//! Each compressor preserves the signal its kind carries — errors in logs, -//! changed hunks in diffs, signatures in code, anomalous rows in JSON — and -//! drops the rest. Lossy compressors leave recovery to the router -//! ([`crate::openhuman::tokenjuice::compress`]), which offloads the original to -//! the CCR cache and appends a retrieval marker. Compressors therefore return -//! only the compacted body and a `lossy` flag; they never touch the cache. - -pub mod code; -pub mod diff; -pub mod generic; -pub mod html; -pub mod json; -pub mod log; -pub mod ml_text; -pub mod search; -pub mod signals; - -use async_trait::async_trait; - -use crate::openhuman::tokenjuice::types::{ - CompressInput, CompressOptions, CompressOutput, CompressorKind, ContentKind, -}; - -/// A content-aware compressor. Implementations are stateless and zero-sized; -/// the registry hands out `&'static` references. -#[async_trait] -pub trait Compressor: Send + Sync { - /// Which [`CompressorKind`] this is (for stats/logs). - fn kind(&self) -> CompressorKind; - - /// Compress `input`. Return `None` to decline (the router passes the - /// original through). `Some` carries the compacted body and whether data - /// was dropped. Async so the ML compressor can talk to its Python sidecar - /// without a blocking bridge; native compressors complete synchronously. - async fn compress( - &self, - input: &CompressInput<'_>, - opts: &CompressOptions, - ) -> Option; -} - -static JSON_COMPRESSOR: json::JsonCompressor = json::JsonCompressor; -static CODE_COMPRESSOR: code::CodeCompressor = code::CodeCompressor; -static LOG_COMPRESSOR: log::LogCompressor = log::LogCompressor; -static SEARCH_COMPRESSOR: search::SearchCompressor = search::SearchCompressor; -static DIFF_COMPRESSOR: diff::DiffCompressor = diff::DiffCompressor; -static HTML_COMPRESSOR: html::HtmlCompressor = html::HtmlCompressor; -static ML_TEXT_COMPRESSOR: ml_text::MlTextCompressor = ml_text::MlTextCompressor; -static GENERIC_COMPRESSOR: generic::GenericCompressor = generic::GenericCompressor; - -/// Map a detected [`ContentKind`] to the compressor that handles it. -/// -/// `PlainText` routes to the ML compressor; whether it actually runs is gated -/// by `opts.ml_text_enabled` (and runtime Python/runtime_python_server -/// availability), and it falls back to [`generic::GenericCompressor`] otherwise -/// — that gating lives in [`crate::openhuman::tokenjuice::compress`], so this -/// function is a pure static mapping. -pub fn compressor_for(kind: ContentKind) -> &'static dyn Compressor { - match kind { - ContentKind::Json => &JSON_COMPRESSOR, - ContentKind::Code => &CODE_COMPRESSOR, - ContentKind::Log => &LOG_COMPRESSOR, - ContentKind::Search => &SEARCH_COMPRESSOR, - ContentKind::Diff => &DIFF_COMPRESSOR, - ContentKind::Html => &HTML_COMPRESSOR, - ContentKind::PlainText => &ML_TEXT_COMPRESSOR, - } -} - -/// The generic line-oriented fallback compressor (head/tail summariser). Used -/// by the router when a specialised compressor declines or is disabled. -pub fn generic_compressor() -> &'static dyn Compressor { - &GENERIC_COMPRESSOR -} diff --git a/src/openhuman/tokenjuice/compressors/search.rs b/src/openhuman/tokenjuice/compressors/search.rs deleted file mode 100644 index fc3d464f8..000000000 --- a/src/openhuman/tokenjuice/compressors/search.rs +++ /dev/null @@ -1,221 +0,0 @@ -//! Search-results compressor (relevance ranking). -//! -//! grep / ripgrep output is `path:line:body` matches. For very large result -//! sets the compressor groups matches by file, ranks them, keeps the top-K per -//! file plus a `[+N more in ]` tally, and (via the router) offloads the -//! full result set to CCR so the complete list is one `retrieve` away. -//! -//! Ranking: when the caller supplies a `query` in the [`ContentHint`], matches -//! are scored by query-term density in the body; otherwise by body length / -//! uniqueness (longer, more-distinctive lines first), with importance signals -//! (error/TODO) always boosted. Group/file order is preserved (first-seen). -//! -//! NOTE: this compressor is gated by `opts.search_enabled` in the router. It is -//! a behaviour change from the historical "never compact grep" stance — kept -//! lossless by the CCR offload — and is enabled by default per project decision. - -use async_trait::async_trait; -use std::fmt::Write as _; - -use super::signals::line_score; -use super::Compressor; -use crate::openhuman::tokenjuice::detect::parse_search_line; -use crate::openhuman::tokenjuice::types::{ - CompressInput, CompressOptions, CompressOutput, CompressorKind, -}; - -/// Only compress result sets with more than this many matching lines. -pub const MIN_MATCHES: usize = 40; -/// Matches kept per file before the "+N more" tally. -pub const TOP_K_PER_FILE: usize = 5; - -pub struct SearchCompressor; - -#[async_trait] -impl Compressor for SearchCompressor { - fn kind(&self) -> CompressorKind { - CompressorKind::Search - } - - async fn compress( - &self, - input: &CompressInput<'_>, - _opts: &CompressOptions, - ) -> Option { - compress(input.content, input.hint.query.as_deref()) - } -} - -struct Match<'a> { - line_no: u64, - body: &'a str, - score: f32, - raw: &'a str, -} - -/// Compress search output. `query` (when known) ranks matches by term density. -pub fn compress(content: &str, query: Option<&str>) -> Option { - // Preserve any non-match preamble/summary lines (e.g. "80 match(es)") and - // group match lines by file in first-seen order. - let mut preamble: Vec<&str> = Vec::new(); - let mut files: Vec<(&str, Vec>)> = Vec::new(); - let mut match_count = 0usize; - - let query_terms: Vec = query - .map(|q| { - q.split_whitespace() - .map(|t| t.to_ascii_lowercase()) - .filter(|t| t.len() >= 2) - .collect() - }) - .unwrap_or_default(); - - for line in content.lines() { - match parse_search_line(line) { - Some((path, line_no, body)) => { - match_count += 1; - let score = score_match(body, &query_terms); - let m = Match { - line_no, - body, - score, - raw: line, - }; - if let Some((_, v)) = files.iter_mut().find(|(p, _)| *p == path) { - v.push(m); - } else { - files.push((path, vec![m])); - } - } - None => { - if !line.trim().is_empty() && files.is_empty() { - // Only keep preamble that appears before any match. - preamble.push(line); - } - } - } - } - - if match_count < MIN_MATCHES || files.is_empty() { - return None; - } - - let mut out = String::with_capacity(content.len() / 2 + 64); - for line in &preamble { - let _ = writeln!(out, "{line}"); - } - let _ = writeln!( - out, - "[search: {} match(es) across {} file(s) · top {} per file · full set via retrieve footer]", - match_count, - files.len(), - TOP_K_PER_FILE - ); - - for (path, mut matches) in files { - let total = matches.len(); - if total <= TOP_K_PER_FILE { - // Keep all in original (line-number) order. - matches.sort_by_key(|m| m.line_no); - for m in &matches { - let _ = writeln!(out, "{}", m.raw); - } - continue; - } - // Rank by score (desc), keep top-K, then re-sort kept by line number so - // the output reads top-to-bottom within the file. - matches.sort_by(|a, b| { - b.score - .partial_cmp(&a.score) - .unwrap_or(std::cmp::Ordering::Equal) - }); - let mut kept: Vec<&Match<'_>> = matches.iter().take(TOP_K_PER_FILE).collect(); - kept.sort_by_key(|m| m.line_no); - for m in &kept { - let _ = writeln!(out, "{}:{}:{}", path, m.line_no, m.body); - } - let _ = writeln!( - out, - "[+{} more match(es) in {path}]", - total - TOP_K_PER_FILE - ); - } - - let out = out.trim_end().to_string(); - if out.len() >= content.len() { - return None; - } - log::debug!( - "[tokenjuice][search] {} matches -> {} bytes (from {} bytes)", - match_count, - out.len(), - content.len() - ); - Some(CompressOutput::lossy(out, CompressorKind::Search)) -} - -/// Score a match body. With query terms, density of those terms dominates; -/// otherwise distinctiveness (length) with an importance bump for error/TODO. -fn score_match(body: &str, query_terms: &[String]) -> f32 { - let importance = line_score(body); - if query_terms.is_empty() { - // No query: favour longer, more-distinctive lines, plus importance. - let len_score = (body.trim().len() as f32 / 80.0).min(1.0); - return importance.max(0.2 + 0.8 * len_score); - } - let lower = body.to_ascii_lowercase(); - let hits = query_terms.iter().filter(|t| lower.contains(*t)).count(); - let density = hits as f32 / query_terms.len() as f32; - importance.max(density) -} - -#[cfg(test)] -mod tests { - use super::*; - - fn big_results() -> String { - let mut s = String::from("120 match(es); scanned 3 file(s)\n"); - for i in 0..60 { - let _ = writeln!(s, "src/a.rs:{i}:let value_{i} = compute_long_name_{i}();"); - } - for i in 0..60 { - let _ = writeln!(s, "src/b.rs:{i}:fn helper_function_number_{i}() {{}}"); - } - s - } - - #[test] - fn keeps_top_k_per_file_and_tally() { - let input = big_results(); - let out = compress(&input, None).expect("compresses").text; - assert!(out.contains("more match(es) in src/a.rs"), "{out}"); - assert!(out.contains("more match(es) in src/b.rs")); - // Preamble survives. - assert!(out.contains("120 match(es)")); - assert!(out.len() < input.len()); - } - - #[test] - fn query_ranks_relevant_matches() { - let mut s = String::new(); - for i in 0..50 { - let body = if i == 7 { - "the special needle token appears here".to_string() - } else { - format!("ordinary line content number {i}") - }; - let _ = writeln!(s, "src/x.rs:{i}:{body}"); - } - let out = compress(&s, Some("needle token")).expect("compresses").text; - assert!( - out.contains("special needle token"), - "ranked-in match missing:\n{out}" - ); - } - - #[test] - fn small_result_set_passes_through() { - let s = "a.rs:1:hit\nb.rs:2:hit\n"; - assert!(compress(s, None).is_none()); - } -} diff --git a/src/openhuman/tokenjuice/compressors/signals.rs b/src/openhuman/tokenjuice/compressors/signals.rs deleted file mode 100644 index eb21069b7..000000000 --- a/src/openhuman/tokenjuice/compressors/signals.rs +++ /dev/null @@ -1,127 +0,0 @@ -//! Shared importance signals for the content-router compressors. -//! -//! A small, deterministic keyword registry + per-line scorer used by the -//! search, log, and JSON compressors to decide which lines/rows to keep when a -//! tool output is over budget. No ML, no regex on the hot path beyond simple -//! case-insensitive substring scans. -//! -//! Clean-room port of Headroom's `error_detection` priority signals -//! (Apache-2.0): error/fatal lines score highest, warnings next, importance -//! markers (security/TODO) a small bump, everything else baseline. - -/// Keywords that mark a hard failure (case-insensitive substrings). -const ERROR_KEYWORDS: &[&str] = &[ - "error", - "fatal", - "panic", - "panicked", - "exception", - "traceback", - "failed", - "failure", - "segfault", - "assertion", - "abort", - "[error]", - "error:", -]; - -/// Keywords that mark a warning. Lower weight than errors. -const WARNING_KEYWORDS: &[&str] = &["warning", "warn:", "[warn]", "deprecated"]; - -/// Keywords that bump importance regardless of severity. -const IMPORTANCE_KEYWORDS: &[&str] = &[ - "security", - "vulnerability", - "critical", - "todo", - "fixme", - "denied", - "unauthorized", - "forbidden", -]; - -/// Score weights. Higher = more likely to survive truncation. -pub const SCORE_ERROR: f32 = 1.0; -pub const SCORE_WARNING: f32 = 0.6; -pub const SCORE_IMPORTANCE: f32 = 0.4; -pub const SCORE_BASELINE: f32 = 0.1; - -/// True if any error keyword appears in `text` (case-insensitive). -pub fn has_error_indicators(text: &str) -> bool { - let lower = text.to_ascii_lowercase(); - ERROR_KEYWORDS.iter().any(|kw| lower.contains(kw)) -} - -/// Importance score for a single line in `[0.0, 1.0]`. -pub fn line_score(line: &str) -> f32 { - let lower = line.to_ascii_lowercase(); - let mut score = SCORE_BASELINE; - if ERROR_KEYWORDS.iter().any(|kw| lower.contains(kw)) { - score = score.max(SCORE_ERROR); - } - if WARNING_KEYWORDS.iter().any(|kw| lower.contains(kw)) { - score = score.max(SCORE_WARNING); - } - if IMPORTANCE_KEYWORDS.iter().any(|kw| lower.contains(kw)) { - score = score.max(SCORE_IMPORTANCE); - } - score -} - -/// Classify a line's severity for the log compressor's bucketing. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum Severity { - Error, - Warning, - Other, -} - -/// Bucket a line into [`Severity`]. -pub fn severity(line: &str) -> Severity { - let lower = line.to_ascii_lowercase(); - if ERROR_KEYWORDS.iter().any(|kw| lower.contains(kw)) { - Severity::Error - } else if WARNING_KEYWORDS.iter().any(|kw| lower.contains(kw)) { - Severity::Warning - } else { - Severity::Other - } -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn errors_score_highest() { - assert_eq!(line_score("FATAL: connection refused"), SCORE_ERROR); - assert_eq!(line_score("thread panicked at 'boom'"), SCORE_ERROR); - assert!(line_score("error: mismatched types") >= SCORE_ERROR); - } - - #[test] - fn warnings_below_errors_above_baseline() { - let w = line_score("warning: unused variable"); - assert!(w < SCORE_ERROR); - assert!(w > SCORE_BASELINE); - } - - #[test] - fn plain_line_is_baseline() { - assert_eq!(line_score(" Compiling foo v0.1.0"), SCORE_BASELINE); - } - - #[test] - fn severity_buckets() { - assert_eq!(severity("error[E0382]: borrow"), Severity::Error); - assert_eq!(severity("warning: deprecated"), Severity::Warning); - assert_eq!(severity("running 12 tests"), Severity::Other); - } - - #[test] - fn has_error_indicators_detects() { - assert!(has_error_indicators("test result: FAILED")); - assert!(!has_error_indicators("all good, 12 passed")); - } -} diff --git a/src/openhuman/tokenjuice/detect/hint.rs b/src/openhuman/tokenjuice/detect/hint.rs deleted file mode 100644 index a79497ca0..000000000 --- a/src/openhuman/tokenjuice/detect/hint.rs +++ /dev/null @@ -1,119 +0,0 @@ -//! Tool-name → content-prior mapping for the content router. -//! -//! The producing tool name is a strong prior on what kind of content a blob -//! holds, so the detector doesn't have to work from scratch for the common -//! case. Ported from the compaction router (`hint_for_tool`) and folded into -//! the richer [`ContentHint`] the new router uses. - -use crate::openhuman::tokenjuice::types::ContentKind; - -/// A coarse prior derived purely from the producing tool name. `Auto` means -/// "no strong prior — run full structural detection". -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum ToolPrior { - Search, - Log, - Diff, - Json, - Auto, -} - -/// Map an agent-level tool name to its content prior. Unknown tools fall -/// through to [`ToolPrior::Auto`]. -/// -/// `shell` is deliberately `Auto` — its output is frequently NOT a log (a -/// `find`, a `seq`, a `cat` of CSV, a script printing a list), so it is routed -/// through detection rather than forced to the log compressor. -pub fn tool_prior(tool_name: &str) -> ToolPrior { - match tool_name { - "grep" | "glob_search" | "ripgrep" | "rg" => ToolPrior::Search, - "run_tests" | "run_linter" | "npm_exec" | "node_exec" | "install_tool" | "lsp" => { - ToolPrior::Log - } - "read_diff" | "git_operations" => ToolPrior::Diff, - _ => ToolPrior::Auto, - } -} - -/// Translate a strong tool prior straight into a [`ContentKind`] without -/// structural detection. Returns `None` for `Auto` (detector must decide). -pub fn prior_to_kind(prior: ToolPrior) -> Option { - match prior { - ToolPrior::Search => Some(ContentKind::Search), - ToolPrior::Log => Some(ContentKind::Log), - ToolPrior::Diff => Some(ContentKind::Diff), - ToolPrior::Json => Some(ContentKind::Json), - ToolPrior::Auto => None, - } -} - -/// Map a file extension (no dot, lower-cased by the caller) to a content kind, -/// when it is unambiguous. Returns `None` for extensions that don't pin a kind. -pub fn extension_to_kind(ext: &str) -> Option { - match ext { - "json" | "jsonl" | "ndjson" => Some(ContentKind::Json), - "html" | "htm" | "xhtml" => Some(ContentKind::Html), - "diff" | "patch" => Some(ContentKind::Diff), - "log" => Some(ContentKind::Log), - // Source-code extensions we recognise for the code compressor. Grammar - // availability is checked later by the compressor; unknown languages - // fall back to the brace-depth heuristic. - "rs" | "ts" | "tsx" | "js" | "jsx" | "mjs" | "cjs" | "py" | "pyi" | "go" | "java" - | "kt" | "kts" | "c" | "h" | "cc" | "cpp" | "cxx" | "hpp" | "rb" | "php" | "swift" - | "scala" | "cs" => Some(ContentKind::Code), - _ => None, - } -} - -/// Map a MIME type to a content kind, when unambiguous. Tolerates a `; charset` -/// suffix. -pub fn mime_to_kind(mime: &str) -> Option { - let base = mime.split(';').next().unwrap_or(mime).trim(); - match base { - "application/json" | "text/json" => Some(ContentKind::Json), - "text/html" | "application/xhtml+xml" => Some(ContentKind::Html), - "text/x-diff" | "text/x-patch" => Some(ContentKind::Diff), - _ => { - // application/ and text/x- source types → Code. - if base.starts_with("text/x-") || base == "application/javascript" { - Some(ContentKind::Code) - } else { - None - } - } - } -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn tool_priors_map_known_tools() { - assert_eq!(tool_prior("grep"), ToolPrior::Search); - assert_eq!(tool_prior("run_tests"), ToolPrior::Log); - assert_eq!(tool_prior("read_diff"), ToolPrior::Diff); - assert_eq!(tool_prior("file_read"), ToolPrior::Auto); - assert_eq!(tool_prior("shell"), ToolPrior::Auto); - } - - #[test] - fn extensions_map() { - assert_eq!(extension_to_kind("rs"), Some(ContentKind::Code)); - assert_eq!(extension_to_kind("json"), Some(ContentKind::Json)); - assert_eq!(extension_to_kind("html"), Some(ContentKind::Html)); - assert_eq!(extension_to_kind("patch"), Some(ContentKind::Diff)); - assert_eq!(extension_to_kind("xyz"), None); - } - - #[test] - fn mimes_map() { - assert_eq!(mime_to_kind("application/json"), Some(ContentKind::Json)); - assert_eq!( - mime_to_kind("text/html; charset=utf-8"), - Some(ContentKind::Html) - ); - assert_eq!(mime_to_kind("text/x-rust"), Some(ContentKind::Code)); - assert_eq!(mime_to_kind("text/plain"), None); - } -} diff --git a/src/openhuman/tokenjuice/detect/kind.rs b/src/openhuman/tokenjuice/detect/kind.rs deleted file mode 100644 index 6a3201556..000000000 --- a/src/openhuman/tokenjuice/detect/kind.rs +++ /dev/null @@ -1,453 +0,0 @@ -//! Content-kind detection for the TokenJuice content router. -//! -//! Cheap structural heuristics that classify a blob so the router can pick the -//! right compressor. Ported and extended from the compaction router's -//! `detect.rs` (clean-room port of Headroom's `content_detector`, Apache-2.0), -//! adding HTML and source-code detection and a richer [`ContentHint`] that can -//! carry a MIME type, file extension, producing tool, and a hard override. -//! -//! Resolution precedence (cheap → expensive): -//! 1. `hint.explicit` — hard override, returned verbatim. -//! 2. `hint.mime` / `hint.extension` — unambiguous type tags. -//! 3. `hint.source_tool` — strong tool prior (grep→Search, …). -//! 4. structural detection — JSON → Diff → HTML → Search → Code → Log. - -use crate::openhuman::tokenjuice::detect::hint::{ - extension_to_kind, mime_to_kind, prior_to_kind, tool_prior, ToolPrior, -}; -use crate::openhuman::tokenjuice::types::{ContentHint, ContentKind}; - -/// Resolve the [`ContentKind`] for `content` given a caller [`ContentHint`]. -pub fn detect_content_kind(content: &str, hint: &ContentHint) -> ContentKind { - // 1. Hard override. - if let Some(kind) = hint.explicit { - return kind; - } - // 2. MIME / extension tags. For a Code/Json/Html tag we still sanity-check - // diff bodies (a `shell` cat of a `.json` that is actually a patch is - // rare, but a diff body is unmistakable and cheap to confirm). - if let Some(kind) = hint.mime.as_deref().and_then(mime_to_kind) { - return reconcile_tag(kind, content); - } - if let Some(kind) = hint - .extension - .as_deref() - .map(str::to_ascii_lowercase) - .as_deref() - .and_then(extension_to_kind) - { - return reconcile_tag(kind, content); - } - // 3. Tool prior. - if let Some(tool) = hint.source_tool.as_deref() { - match tool_prior(tool) { - // A Search prior is absolute: grep output is never re-routed. - ToolPrior::Search => return ContentKind::Search, - ToolPrior::Auto => {} - other => { - if let Some(kind) = prior_to_kind(other) { - return reconcile_tag(kind, content); - } - } - } - } - // 4. Full structural detection. - detect(content) -} - -/// A type tag (from MIME/extension/tool) is trusted unless the body is clearly -/// a diff — diffs are unmistakable and worth preferring even when the file -/// extension says otherwise. -fn reconcile_tag(tagged: ContentKind, content: &str) -> ContentKind { - if tagged != ContentKind::Diff && looks_like_diff(content) { - ContentKind::Diff - } else { - tagged - } -} - -/// Full structural detection, in priority order: JSON → diff → HTML → search → -/// code → log → plain text. -pub fn detect(content: &str) -> ContentKind { - let trimmed = content.trim_start(); - if trimmed.is_empty() { - return ContentKind::PlainText; - } - if looks_like_json(content) { - return ContentKind::Json; - } - if looks_like_diff(content) { - return ContentKind::Diff; - } - if looks_like_html(content) { - return ContentKind::Html; - } - if search_line_ratio(content) >= 0.6 { - return ContentKind::Search; - } - if looks_like_code(content) { - return ContentKind::Code; - } - if log_line_ratio(content) >= 0.5 { - return ContentKind::Log; - } - ContentKind::PlainText -} - -/// True if `content` parses as a JSON array of objects (crusher input) or a -/// single non-trivial JSON object/array. Scalars and tiny payloads are ignored. -pub fn looks_like_json(content: &str) -> bool { - let trimmed = content.trim_start(); - let first = trimmed.as_bytes().first().copied(); - if first != Some(b'[') && first != Some(b'{') { - return false; - } - match serde_json::from_str::(trimmed.trim_end()) { - Ok(serde_json::Value::Array(items)) => { - items.len() >= 2 && items.iter().any(|v| v.is_object()) - } - // A standalone object is JSON worth routing when it has enough keys to - // be worth pretty-handling (the JSON compressor declines small ones). - Ok(serde_json::Value::Object(map)) => map.len() >= 2, - _ => false, - } -} - -/// True if `content` parses as a JSON array of objects specifically (the table -/// crusher's strict input). Kept for callers that need the narrow check. -pub fn looks_like_json_array(content: &str) -> bool { - let trimmed = content.trim_start(); - if !trimmed.starts_with('[') { - return false; - } - matches!( - serde_json::from_str::(trimmed.trim_end()), - Ok(serde_json::Value::Array(items)) if items.len() >= 2 && items.iter().any(|v| v.is_object()) - ) -} - -/// True if `content` looks like a unified diff: a `diff --git` header or at -/// least one hunk header (`@@ ... @@`). -pub fn looks_like_diff(content: &str) -> bool { - for line in content.lines().take(400) { - if line.starts_with("diff --git ") || line.starts_with("Index: ") { - return true; - } - if line.starts_with("@@ ") && line[3..].contains("@@") { - return true; - } - } - false -} - -/// True if `content` looks like an HTML document: a doctype / ` bool { - // Snap to a UTF-8 char boundary: a raw `&content[..8192]` panics when byte - // 8192 lands inside a multi-byte codepoint (CJK/emoji in large tool output). - let head = crate::openhuman::util::utf8_safe_prefix_at_byte_boundary(content, 8192); - let lower = head.to_ascii_lowercase(); - if lower.contains("") - || lower.contains("` openings over non-blank lines. - let mut tags = 0usize; - let mut lines = 0usize; - for line in head.lines().take(200) { - if line.trim().is_empty() { - continue; - } - lines += 1; - tags += count_html_tags(line); - } - lines >= 3 && (tags as f32 / lines as f32) >= 1.0 -} - -/// Cheap count of ` usize { - let bytes = line.as_bytes(); - let mut count = 0usize; - let mut i = 0usize; - while i + 1 < bytes.len() { - if bytes[i] == b'<' { - let next = bytes[i + 1]; - if next.is_ascii_alphabetic() || next == b'/' || next == b'!' { - count += 1; - } - } - i += 1; - } - count -} - -/// True if `content` heuristically looks like source code: a meaningful share -/// of lines carry code structure (keywords, braces, semicolons, indentation -/// with operators). Deliberately conservative — detection only fires for `Auto` -/// content with no extension/MIME hint, so false positives are rare. -pub fn looks_like_code(content: &str) -> bool { - const KEYWORDS: &[&str] = &[ - "fn ", - "function ", - "class ", - "def ", - "impl ", - "struct ", - "enum ", - "trait ", - "interface ", - "import ", - "export ", - "package ", - "public ", - "private ", - "const ", - "let ", - "var ", - "return ", - "#include", - "using ", - "namespace ", - ]; - let mut total = 0usize; - let mut code_like = 0usize; - let mut brace_lines = 0usize; - for line in content.lines().take(400) { - let t = line.trim(); - if t.is_empty() { - continue; - } - total += 1; - let has_kw = KEYWORDS.iter().any(|kw| line.contains(kw)); - let ends_struct = t.ends_with('{') || t.ends_with(';') || t.ends_with('}'); - if t.contains('{') || t.contains('}') { - brace_lines += 1; - } - if has_kw || ends_struct { - code_like += 1; - } - } - if total < 5 { - return false; - } - let ratio = code_like as f32 / total as f32; - // Require both a decent code-signal ratio and some brace structure so prose - // with the odd "return" or "class" doesn't trip it. - ratio >= 0.4 && brace_lines >= 2 -} - -/// Fraction of non-empty lines that look like `path:line:...` search hits. -fn search_line_ratio(content: &str) -> f32 { - let mut total = 0usize; - let mut hits = 0usize; - for line in content.lines().take(2000) { - if line.trim().is_empty() { - continue; - } - total += 1; - if parse_search_line(line).is_some() { - hits += 1; - } - } - if total == 0 { - 0.0 - } else { - hits as f32 / total as f32 - } -} - -/// Fraction of lines carrying an error/warning indicator — the log signal. -fn log_line_ratio(content: &str) -> f32 { - use crate::openhuman::tokenjuice::compressors::signals::{severity, Severity}; - let mut total = 0usize; - let mut hits = 0usize; - for line in content.lines().take(2000) { - if line.trim().is_empty() { - continue; - } - total += 1; - if severity(line) != Severity::Other { - hits += 1; - } - } - if total == 0 { - 0.0 - } else { - hits as f32 / total as f32 - } -} - -/// Parse a single grep/ripgrep line into `(path, line_number, content)`. -/// -/// Anchors on the earliest `::` marker, skipping a leading Windows -/// drive prefix (`C:`), so paths may contain `:` (drive), `-`, and spaces. -/// Returns `None` for context lines and non-matches. -pub fn parse_search_line(line: &str) -> Option<(&str, u64, &str)> { - let scan_from = if line.len() >= 2 { - let bytes = line.as_bytes(); - if bytes[0].is_ascii_alphabetic() && bytes[1] == b':' { - 2 - } else { - 0 - } - } else { - 0 - }; - - let rest = &line[scan_from..]; - let mut search_start = 0usize; - while let Some(rel) = rest[search_start..].find(':') { - let colon = search_start + rel; - let after = &rest[colon + 1..]; - let digits_len = after.chars().take_while(|c| c.is_ascii_digit()).count(); - if digits_len > 0 && after.as_bytes().get(digits_len) == Some(&b':') { - let path = &line[..scan_from + colon]; - let num: u64 = after[..digits_len].parse().ok()?; - let body = &after[digits_len + 1..]; - if path.is_empty() { - return None; - } - return Some((path, num, body)); - } - search_start = colon + 1; - } - None -} - -#[cfg(test)] -mod tests { - use super::*; - - fn hint() -> ContentHint { - ContentHint::default() - } - - #[test] - fn explicit_override_wins() { - let h = ContentHint { - explicit: Some(ContentKind::Html), - ..Default::default() - }; - // Body is JSON but the explicit hint forces Html. - assert_eq!( - detect_content_kind(r#"[{"a":1},{"b":2}]"#, &h), - ContentKind::Html - ); - } - - #[test] - fn mime_and_extension_route() { - let h = ContentHint { - mime: Some("application/json".into()), - ..Default::default() - }; - assert_eq!( - detect_content_kind("{not even json}", &h), - ContentKind::Json - ); - let h = ContentHint { - extension: Some("RS".into()), - ..Default::default() - }; - assert_eq!(detect_content_kind("anything", &h), ContentKind::Code); - } - - #[test] - fn tool_prior_routes_and_search_is_absolute() { - let h = ContentHint::for_tool("grep"); - // Even a diff-looking body stays Search under a grep prior. - assert_eq!( - detect_content_kind("diff --git a/x b/x\n@@ -1 +1 @@", &h), - ContentKind::Search - ); - let h = ContentHint::for_tool("read_diff"); - assert_eq!( - detect_content_kind("diff --git a/x b/x\n@@ -1 +1 @@\n+a", &h), - ContentKind::Diff - ); - } - - #[test] - fn detect_search_results() { - let c = - "src/main.rs:42:fn process() {\nsrc/lib.rs:7:pub use foo;\nsrc/x.rs:99: let y = 1;"; - assert_eq!(detect_content_kind(c, &hint()), ContentKind::Search); - } - - #[test] - fn detect_diff_json_log() { - assert_eq!( - detect_content_kind("diff --git a/x.rs b/x.rs\n@@ -1,3 +1,4 @@\n+added", &hint()), - ContentKind::Diff - ); - assert_eq!( - detect_content_kind(r#"[{"id":1,"name":"a"},{"id":2,"name":"b"}]"#, &hint()), - ContentKind::Json - ); - assert_eq!( - detect_content_kind( - "Compiling foo\nwarning: unused\nerror[E0382]: borrow of moved value\nerror: aborting", - &hint() - ), - ContentKind::Log - ); - } - - #[test] - fn detect_html() { - let c = "\nx

    hi

    "; - assert_eq!(detect_content_kind(c, &hint()), ContentKind::Html); - } - - #[test] - fn looks_like_html_handles_multibyte_at_byte_cutoff() { - // Build content longer than the 8192-byte head window with a multi-byte - // char (4-byte emoji) straddling byte index 8192, so a raw byte slice - // would panic on the non-char-boundary cut. Detection must not panic. - let mut content = "a".repeat(8190); - content.push('🦀'); // 4 bytes spanning indices 8190..8194 — crosses 8192 - content.push_str(&"b".repeat(2000)); - assert!(content.len() > 8192); - // Plain text with no tags: just assert it returns without panicking. - assert!(!looks_like_html(&content)); - // And the full detector stays reachable on the same input. - assert_eq!( - detect_content_kind(&content, &hint()), - ContentKind::PlainText - ); - } - - #[test] - fn detect_code() { - let c = "use std::fmt;\n\npub fn add(a: i32, b: i32) -> i32 {\n let c = a + b;\n return c;\n}\n\nstruct Foo {\n x: i32,\n}"; - assert_eq!(detect_content_kind(c, &hint()), ContentKind::Code); - } - - #[test] - fn plain_text_passes_through() { - assert_eq!( - detect_content_kind("just some prose about a topic at length here", &hint()), - ContentKind::PlainText - ); - } - - #[test] - fn parse_unix_and_windows_paths() { - assert_eq!( - parse_search_line("src/main.rs:42:fn process() {"), - Some(("src/main.rs", 42, "fn process() {")) - ); - assert_eq!( - parse_search_line(r"C:\Users\me\a.rs:10:let x = 1;"), - Some((r"C:\Users\me\a.rs", 10, "let x = 1;")) - ); - assert_eq!( - parse_search_line("pre-commit-config.yaml:3:foo"), - Some(("pre-commit-config.yaml", 3, "foo")) - ); - assert_eq!(parse_search_line("just a sentence"), None); - } -} diff --git a/src/openhuman/tokenjuice/detect/mod.rs b/src/openhuman/tokenjuice/detect/mod.rs deleted file mode 100644 index da632b49f..000000000 --- a/src/openhuman/tokenjuice/detect/mod.rs +++ /dev/null @@ -1,10 +0,0 @@ -//! Content-kind detection + tool-name priors for the TokenJuice content router. - -pub mod hint; -pub mod kind; - -pub use hint::{extension_to_kind, mime_to_kind, prior_to_kind, tool_prior, ToolPrior}; -pub use kind::{ - detect, detect_content_kind, looks_like_code, looks_like_diff, looks_like_html, - looks_like_json, looks_like_json_array, parse_search_line, -}; diff --git a/src/openhuman/tokenjuice/ml/mod.rs b/src/openhuman/tokenjuice/ml/mod.rs index 438e59860..245849d50 100644 --- a/src/openhuman/tokenjuice/ml/mod.rs +++ b/src/openhuman/tokenjuice/ml/mod.rs @@ -3,8 +3,8 @@ //! Plain text has no structural skeleton to exploit, so high-quality //! compression needs a learned model (ModernBERT token/sentence salience). That //! runs inside the shared [`crate::openhuman::runtime_python_server`] as the -//! `kompress` backend — this module is just the thin Rust entry the -//! [`crate::openhuman::tokenjuice::compressors::ml_text`] compressor calls. +//! `kompress` backend — this module is just the thin Rust callback the +//! TinyJuice `ml_text` compressor calls. //! //! Opt-in at runtime via `config.tokenjuice.ml_compression_enabled` (default //! off) — there is no build-time feature gate, since torch is provisioned at diff --git a/src/openhuman/tokenjuice/mod.rs b/src/openhuman/tokenjuice/mod.rs index a092d74f5..6dcc18a70 100644 --- a/src/openhuman/tokenjuice/mod.rs +++ b/src/openhuman/tokenjuice/mod.rs @@ -1,62 +1,23 @@ -//! # TokenJuice — terminal-output compaction engine +//! OpenHuman adapter for the vendored TinyJuice compression engine. //! -//! Rust port of [vincentkoc/tokenjuice](https://github.com/vincentkoc/tokenjuice). -//! -//! Compacts verbose tool output (git, npm, cargo, docker, …) using -//! JSON-configured rules before it enters an LLM context window. -//! -//! ## Quick start -//! -//! ```rust -//! use openhuman_core::openhuman::tokenjuice::{ -//! reduce::reduce_execution_with_rules, -//! rules::load_builtin_rules, -//! types::{ReduceOptions, ToolExecutionInput}, -//! }; -//! -//! let rules = load_builtin_rules(); -//! let input = ToolExecutionInput { -//! tool_name: "bash".to_owned(), -//! argv: Some(vec!["git".to_owned(), "status".to_owned()]), -//! stdout: Some("On branch main\n\tmodified: src/lib.rs\n".to_owned()), -//! ..Default::default() -//! }; -//! let result = reduce_execution_with_rules(input, &rules, &ReduceOptions::default()); -//! println!("{}", result.inline_text); -//! // → "M: src/lib.rs" -//! ``` -//! -//! ## Scope (v1 — library only) -//! -//! This module is purely a library. It has no JSON-RPC surface, no CLI, and -//! no artifact store. Those surfaces can be layered on later when a caller -//! inside `openhuman` needs them. -//! -//! ## Three-layer rule overlay -//! -//! Rules are loaded from three sources in ascending priority order: -//! 1. **Builtin** — vendored JSON files embedded via `include_str!`. -//! 2. **User** — `~/.config/tokenjuice/rules/` (loaded from disk). -//! 3. **Project** — `.tokenjuice/rules/` relative to `cwd` (loaded from disk). -//! -//! When two layers define the same rule `id`, the higher-priority layer wins. +//! TinyJuice owns the host-agnostic TokenJuice engine: detection, compressors, +//! CCR cache, rule loading, text helpers, and token estimates. This module keeps +//! the OpenHuman-facing seam stable and owns only host concerns: config mapping, +//! JSON-RPC controllers, settings patching, retrieve tool integration, savings +//! pricing, and the Kompress runtime bridge. + +use std::sync::Arc; -pub mod cache; -pub mod classify; -pub mod compress; -pub mod compressors; pub mod config_patch; -pub mod detect; pub mod ml; -pub mod reduce; -pub mod rules; pub mod savings; pub mod schemas; -pub mod text; -pub mod tokens; -pub mod tool_integration; pub mod tools; -pub mod types; + +pub use tinyjuice::{ + cache, classify, compress, compressors, detect, reduce, rules, text, tokens, tool_integration, + types, +}; /// Install the full TokenJuice runtime from a [`Config`] in one call: router / /// compressor options + CCR cache limits + disk tier, savings attribution + @@ -68,7 +29,7 @@ pub mod types; /// only changes on restart. pub fn install_from_config(config: &crate::openhuman::config::Config) { let tj = &config.tokenjuice; - let options = types::CompressOptions { + let options = tinyjuice::types::CompressOptions { router_enabled: tj.router_enabled, ccr_enabled: tj.ccr_enabled, search_enabled: tj.search_enabled, @@ -82,7 +43,7 @@ pub fn install_from_config(config: &crate::openhuman::config::Config) { let disk_root = tj .ccr_disk_enabled .then(|| config.workspace_dir.join(".tokenjuice").join("ccr")); - install_config( + tinyjuice::tool_integration::install_config( options, tj.max_cache_entries, tj.max_cache_bytes, @@ -96,7 +57,19 @@ pub fn install_from_config(config: &crate::openhuman::config::Config) { .unwrap_or_else(|| crate::openhuman::config::DEFAULT_MODEL.to_string()), &config.workspace_dir, ); + tinyjuice::savings::configure_recorder(Some(Arc::new( + |content_kind, compressor, original_tokens, compacted_tokens| { + savings::record(content_kind, compressor, original_tokens, compacted_tokens); + }, + ))); ml::configure(config.clone()); + tinyjuice::ml::configure_callback(Some(Arc::new(|text, opts| { + Box::pin(async move { + ml::compress(&text, &opts) + .await + .map_err(|err| format!("{err:#}")) + }) + }))); } /// All read-only TokenJuice debug controllers (detect / compress / cache_stats @@ -110,10 +83,6 @@ pub fn all_tokenjuice_controller_schemas() -> Vec schemas::all_controller_schemas() } -#[cfg(test)] -#[path = "text_tests.rs"] -mod text_tests; - pub use cache::{ is_recovery_tool, LEGACY_RETRIEVE_TOOL_NAME, NEVER_COMPACT_TOOLS, RECOVERY_TOOL_NAMES, RETRIEVE_TOOL_NAME, diff --git a/src/openhuman/tokenjuice/reduce.rs b/src/openhuman/tokenjuice/reduce.rs deleted file mode 100644 index 7113c1f39..000000000 --- a/src/openhuman/tokenjuice/reduce.rs +++ /dev/null @@ -1,939 +0,0 @@ -//! The main reduction pipeline: `reduce_execution` and helpers. -//! -//! Port of `src/core/reduce.ts` and the `normalizeExecutionInput` helper -//! from `src/core/command.ts`. - -use std::collections::HashMap; - -use once_cell::sync::Lazy; -use regex::Regex; - -use crate::openhuman::tokenjuice::{ - classify::classify_execution, - text::{ - clamp_text, clamp_text_middle, count_text_chars, dedupe_adjacent, head_tail, - normalize_lines, pluralize, strip_ansi, trim_empty_edges, - }, - types::{ - ClassificationResult, CompactResult, CompiledRule, CounterSource, ReduceOptions, - ReductionStats, ToolExecutionInput, - }, -}; - -// --------------------------------------------------------------------------- -// Constants -// --------------------------------------------------------------------------- - -/// Output shorter than this many chars is returned verbatim (passthrough) even -/// when a rule would compact it. -const TINY_OUTPUT_MAX_CHARS: usize = 240; - -// --------------------------------------------------------------------------- -// Command normalisation (from command.ts) -// --------------------------------------------------------------------------- - -/// Simple shell tokenizer (mirrors `tokenizeCommand` in TS). -pub fn tokenize_command(command: &str) -> Vec { - let mut tokens: Vec = Vec::new(); - let mut current = String::new(); - let mut quote: Option = None; - let mut escaping = false; - - for ch in command.trim().chars() { - if escaping { - current.push(ch); - escaping = false; - continue; - } - if ch == '\\' { - escaping = true; - continue; - } - if let Some(q) = quote { - if ch == q { - quote = None; - } else { - current.push(ch); - } - continue; - } - if ch == '\'' || ch == '"' { - quote = Some(ch); - continue; - } - if ch.is_whitespace() { - if !current.is_empty() { - tokens.push(current.clone()); - current.clear(); - } - continue; - } - current.push(ch); - } - if escaping { - current.push('\\'); - } - if !current.is_empty() { - tokens.push(current); - } - tokens -} - -/// Fill in `argv` from `command` if `argv` is absent. -pub fn normalize_execution_input(input: ToolExecutionInput) -> ToolExecutionInput { - if input.argv.as_ref().map(|v| !v.is_empty()).unwrap_or(false) { - return input; - } - let command = match &input.command { - Some(c) if !c.is_empty() => c.clone(), - _ => return input, - }; - let argv = tokenize_command(&command); - if argv.is_empty() { - return input; - } - ToolExecutionInput { - argv: Some(argv), - ..input - } -} - -/// True when the command is a well-known file-content inspection tool. -pub fn is_file_content_inspection_command(input: &ToolExecutionInput) -> bool { - static FILE_TOOLS: &[&str] = &[ - "cat", "sed", "head", "tail", "nl", "bat", "batcat", "jq", "yq", - ]; - let argv = input.argv.as_deref().unwrap_or(&[]); - if argv.is_empty() { - return false; - } - let argv0 = std::path::Path::new(&argv[0]) - .file_name() - .map(|n| n.to_string_lossy().to_string()) - .unwrap_or_default(); - FILE_TOOLS.contains(&argv0.as_str()) -} - -// --------------------------------------------------------------------------- -// Git-status post-processor -// --------------------------------------------------------------------------- - -fn rewrite_git_status_line(line: &str) -> Option { - let trimmed = line.trim(); - if trimmed.is_empty() { - return Some(String::new()); - } - if trimmed.starts_with("On branch ") { - return None; - } - // "and have N and M different commits each" - if regex_match(r"^and have \d+ and \d+ different commits each", trimmed) { - return None; - } - if regex_match( - r"^(?:no changes added to commit|nothing added to commit but untracked files present)", - trimmed, - ) { - return None; - } - if regex_match(r#"^\(use "git .+"\)$"#, trimmed) - || regex_match(r#"^use "git .+" to .+"#, trimmed) - { - return None; - } - - if trimmed == "Changes not staged for commit:" { - return Some("Changes not staged:".to_owned()); - } - if trimmed == "Changes to be committed:" { - return Some("Staged changes:".to_owned()); - } - if trimmed == "Untracked files:" { - return Some("Untracked files:".to_owned()); - } - - if regex_match(r"^\s*modified:\s+", line) { - let path = regex_replace(r"^\s*modified:\s+", line, "") - .trim() - .to_owned(); - return Some(format!("M: {}", path)); - } - if regex_match(r"^\s*new file:\s+", line) { - let path = regex_replace(r"^\s*new file:\s+", line, "") - .trim() - .to_owned(); - return Some(format!("A: {}", path)); - } - if regex_match(r"^\s*deleted:\s+", line) { - let path = regex_replace(r"^\s*deleted:\s+", line, "") - .trim() - .to_owned(); - return Some(format!("D: {}", path)); - } - if regex_match(r"^\s*renamed:\s+", line) { - let path = regex_replace(r"^\s*renamed:\s+", line, "") - .trim() - .to_owned(); - return Some(format!("R: {}", path)); - } - if regex_match(r"^\?\?\s+", trimmed) { - let path = regex_replace(r"^\?\?\s+", trimmed, "").trim().to_owned(); - return Some(format!("?? {}", path)); - } - - // Porcelain format: two status chars + space + path - if let Some(caps) = regex_captures(r"^([ MADRCU?!]{2})\s+(.+)$", line) { - let status_raw = caps[0].trim().replace('?', "??"); - let path = caps[1].trim(); - let code = if status_raw.is_empty() { - "M" - } else if status_raw.starts_with("??") { - "??" - } else { - &status_raw[..1] - }; - return Some(format!("{}: {}", code, path)); - } - - Some(trimmed.to_owned()) -} - -fn rewrite_git_status_lines(lines: &[String]) -> Vec { - let mut section: Option<&str> = None; - - let rewritten: Vec> = lines - .iter() - .map(|line| { - let trimmed = line.trim(); - if trimmed == "Changes not staged for commit:" { - section = Some("unstaged"); - } else if trimmed == "Changes to be committed:" { - section = Some("staged"); - } else if trimmed == "Untracked files:" { - section = Some("untracked"); - } - - // In untracked section, indented non-action lines become "?? " - if section == Some("untracked") - && regex_match(r"^\s{2,}\S", line) - && !regex_match(r"^\s*(?:modified:|new file:|deleted:|renamed:)", line) - { - return Some(format!("?? {}", trimmed)); - } - - rewrite_git_status_line(line) - }) - .collect(); - - // Collapse consecutive empty lines - let mut collapsed: Vec = Vec::new(); - for line in rewritten.into_iter().flatten() { - if line.is_empty() && collapsed.last().map(String::is_empty).unwrap_or(false) { - continue; - } - collapsed.push(line); - } - collapsed -} - -// --------------------------------------------------------------------------- -// GH output formatter -// --------------------------------------------------------------------------- - -fn compact_whitespace(text: &str) -> String { - text.split_whitespace().collect::>().join(" ") -} - -/// Splits a `gh` tabular row on runs of 2+ whitespace or one-or-more tabs. -/// -/// Compiled once at first use; previously `Regex::new` ran per line, which is -/// a textbook regex-in-a-loop hot-path bug: a `gh pr list` with N rows paid -/// for N compilations of the same trivial pattern. Matches the project -/// convention (see `tokenjuice::text::ansi`). -static GH_TABLE_SPLIT_RE: Lazy = - Lazy::new(|| Regex::new(r"\s{2,}|\t+").expect("gh table split regex")); - -fn format_gh_table_line(line: &str) -> String { - let trimmed = line.trim(); - if trimmed.is_empty() { - return String::new(); - } - // Split on 2+ spaces or tabs - let columns: Vec = GH_TABLE_SPLIT_RE - .split(trimmed) - .map(compact_whitespace) - .filter(|s| !s.is_empty()) - .collect(); - - if columns.len() >= 2 && regex_match(r"^\d+$", &columns[0]) { - let number = &columns[0]; - let title = &columns[1]; - let state = if columns.len() >= 4 { - columns.last() - } else { - None - }; - let context = if columns.len() >= 3 { - let end = if state.is_some() { - columns.len() - 1 - } else { - columns.len() - }; - let slice = &columns[2..end]; - if slice.is_empty() { - None - } else { - Some(slice.join(" ")) - } - } else { - None - }; - let mut parts = vec![format!("#{}", number), title.clone()]; - if let Some(s) = state { - parts.push(format!("[{}]", s)); - } - if let Some(c) = context { - parts.push(format!("({})", c)); - } - return parts.join(" "); - } - compact_whitespace(trimmed) -} - -fn rewrite_gh_lines(lines: &[String], input: &ToolExecutionInput) -> Vec { - let non_empty: Vec<&String> = lines.iter().filter(|l| !l.trim().is_empty()).collect(); - if non_empty.is_empty() { - return Vec::new(); - } - - // Try to parse as JSON objects - let parsed: Vec> = non_empty - .iter() - .map(|line| { - let t = line.trim(); - if t.starts_with('{') && t.ends_with('}') { - serde_json::from_str(t).ok() - } else { - None - } - }) - .collect(); - - if parsed.iter().all(|p| p.is_some()) { - let formatted: Vec = parsed - .into_iter() - .filter_map(|v| format_gh_json_record(v?)) - .collect(); - if !formatted.is_empty() { - return formatted; - } - } - - // Fall back to table formatting if argv[0] == "gh" - let argv = input.argv.as_deref().unwrap_or(&[]); - if argv.first().map(String::as_str) == Some("gh") { - return lines.iter().map(|l| format_gh_table_line(l)).collect(); - } - - lines.to_vec() -} - -fn format_gh_json_record(record: serde_json::Value) -> Option { - let obj = record.as_object()?; - - let title = obj - .get("title") - .and_then(|v| v.as_str()) - .or_else(|| obj.get("displayTitle").and_then(|v| v.as_str())) - .or_else(|| obj.get("name").and_then(|v| v.as_str())) - .or_else(|| obj.get("workflowName").and_then(|v| v.as_str()))? - .to_owned(); - - let numeric_id: Option = obj - .get("number") - .and_then(|v| v.as_i64()) - .or_else(|| obj.get("databaseId").and_then(|v| v.as_i64())); - - let status = obj - .get("state") - .and_then(|v| v.as_str()) - .or_else(|| obj.get("status").and_then(|v| v.as_str())) - .or_else(|| obj.get("conclusion").and_then(|v| v.as_str())) - .map(ToOwned::to_owned); - - let branch = obj - .get("headBranch") - .and_then(|v| v.as_str()) - .or_else(|| obj.get("headRefName").and_then(|v| v.as_str())) - .map(compact_whitespace); - - let comments = extract_comment_count(obj.get("comments")); - - let labels: Vec = obj - .get("labels") - .map(extract_label_names) - .unwrap_or_default() - .into_iter() - .take(3) - .collect(); - - let updated_at = obj - .get("updatedAt") - .and_then(|v| v.as_str()) - .map(|s| s.get(..10).unwrap_or(s).to_owned()); - - let mut parts = Vec::new(); - if let Some(id) = numeric_id { - parts.push(format!("#{}", id)); - } - parts.push(compact_whitespace(&title)); - if let Some(s) = status { - parts.push(format!("[{}]", s)); - } - if let Some(b) = branch { - parts.push(format!("({})", b)); - } - if let Some(c) = comments { - if c > 0 { - parts.push(format!("{}c", c)); - } - } - if !labels.is_empty() { - parts.push(format!("{{{}}}", labels.join(", "))); - } - if let Some(d) = updated_at { - parts.push(d); - } - Some(parts.join(" ")) -} - -fn extract_comment_count(value: Option<&serde_json::Value>) -> Option { - match value? { - serde_json::Value::Number(n) => n.as_i64(), - serde_json::Value::Array(arr) => Some(arr.len() as i64), - serde_json::Value::Object(obj) => obj.get("totalCount").and_then(|v| v.as_i64()), - _ => None, - } -} - -fn extract_label_names(value: &serde_json::Value) -> Vec { - let arr = match value.as_array() { - Some(a) => a, - None => return Vec::new(), - }; - arr.iter() - .filter_map(|entry| { - if let Some(s) = entry.as_str() { - if !s.is_empty() { - Some(s.to_owned()) - } else { - None - } - } else if let Some(obj) = entry.as_object() { - obj.get("name") - .and_then(|v| v.as_str()) - .filter(|s| !s.is_empty()) - .map(ToOwned::to_owned) - } else { - None - } - }) - .collect() -} - -// --------------------------------------------------------------------------- -// JSON pretty-print -// --------------------------------------------------------------------------- - -fn pretty_print_json_if_possible(text: &str) -> String { - let trimmed = text.trim(); - if !(trimmed.starts_with('{') || trimmed.starts_with('[')) { - return text.to_owned(); - } - if let Ok(v) = serde_json::from_str::(trimmed) { - if v.is_object() || v.is_array() { - return serde_json::to_string_pretty(&v).unwrap_or_else(|_| text.to_owned()); - } - } - text.to_owned() -} - -// --------------------------------------------------------------------------- -// Raw text builder -// --------------------------------------------------------------------------- - -fn build_raw_text(input: &ToolExecutionInput) -> String { - if let Some(combined) = &input.combined_text { - return combined.clone(); - } - let stdout = input.stdout.as_deref().unwrap_or(""); - let stderr = input.stderr.as_deref().unwrap_or(""); - if stdout.is_empty() { - return stderr.to_owned(); - } - if stderr.is_empty() { - return stdout.to_owned(); - } - format!("{}\n{}", stdout, stderr) -} - -// --------------------------------------------------------------------------- -// apply_rule -// --------------------------------------------------------------------------- - -struct ApplyResult { - summary: String, - facts: HashMap, -} - -fn apply_rule( - compiled_rule: &CompiledRule, - input: &ToolExecutionInput, - raw_text: &str, -) -> ApplyResult { - let rule = &compiled_rule.rule; - let mut text = raw_text.to_owned(); - - if rule - .transforms - .as_ref() - .and_then(|t| t.pretty_print_json) - .unwrap_or(false) - { - text = pretty_print_json_if_possible(&text); - } - - let mut lines = normalize_lines(&text); - let mut facts: HashMap = HashMap::new(); - - if rule - .transforms - .as_ref() - .and_then(|t| t.strip_ansi) - .unwrap_or(false) - { - lines = normalize_lines(&strip_ansi(&lines.join("\n"))); - } - - // outputMatches check — run on the trimmed full text - let output_match_text = trim_empty_edges(&lines).join("\n"); - if let Some(matched_output) = compiled_rule - .compiled - .output_matches - .iter() - .find(|entry| entry.pattern.is_match(&output_match_text)) - { - return ApplyResult { - summary: matched_output.message.clone(), - facts, - }; - } - - // skipPatterns - if rule - .filters - .as_ref() - .and_then(|f| f.skip_patterns.as_ref()) - .map(|p| !p.is_empty()) - .unwrap_or(false) - { - lines.retain(|line| { - !compiled_rule - .compiled - .skip_patterns - .iter() - .any(|pat| pat.is_match(line)) - }); - } - - // counter_source == preKeep → sample counters before keep filtering - let pre_keep_lines = lines.clone(); - - // keepPatterns - let has_keep = !compiled_rule.compiled.keep_patterns.is_empty(); - if has_keep { - let kept: Vec = lines - .iter() - .filter(|line| { - compiled_rule - .compiled - .keep_patterns - .iter() - .any(|pat| pat.is_match(line)) - }) - .cloned() - .collect(); - if !kept.is_empty() { - lines = kept; - } - } - - // trimEmptyEdges - if rule - .transforms - .as_ref() - .and_then(|t| t.trim_empty_edges) - .unwrap_or(false) - { - lines = trim_empty_edges(&lines); - } - - // dedupeAdjacent - if rule - .transforms - .as_ref() - .and_then(|t| t.dedupe_adjacent) - .unwrap_or(false) - { - lines = dedupe_adjacent(&lines); - } - - // Special post-processors - if rule.id == "git/status" { - lines = rewrite_git_status_lines(&lines); - } - if rule.id == "cloud/gh" { - lines = rewrite_gh_lines(&lines, input); - } - - // Counters - let counter_lines = match &rule.counter_source { - Some(CounterSource::PreKeep) => &pre_keep_lines, - _ => &lines, - }; - for counter in &compiled_rule.compiled.counters { - let count = counter_lines - .iter() - .filter(|line| counter.pattern.is_match(line)) - .count(); - facts.insert(counter.name.clone(), count); - } - - // onEmpty - if lines.is_empty() { - if let Some(on_empty) = &rule.on_empty { - return ApplyResult { - summary: on_empty.clone(), - facts, - }; - } - } - - // Failure-preserving summarize - let is_failure = input.exit_code.map(|c| c != 0).unwrap_or(false); - let preserve_on_failure = rule - .failure - .as_ref() - .and_then(|f| f.preserve_on_failure) - .unwrap_or(false); - - let (head, tail) = if is_failure && preserve_on_failure { - ( - rule.failure.as_ref().and_then(|f| f.head).unwrap_or(6), - rule.failure.as_ref().and_then(|f| f.tail).unwrap_or(12), - ) - } else { - ( - rule.summarize.as_ref().and_then(|s| s.head).unwrap_or(6), - rule.summarize.as_ref().and_then(|s| s.tail).unwrap_or(6), - ) - }; - - log::debug!( - "[tokenjuice] apply_rule '{}': {} lines → head={} tail={} failure={}", - rule.id, - lines.len(), - head, - tail, - is_failure && preserve_on_failure - ); - - let compacted = head_tail(&lines, head, tail); - ApplyResult { - summary: compacted.join("\n").trim().to_owned(), - facts, - } -} - -// --------------------------------------------------------------------------- -// Passthrough text -// --------------------------------------------------------------------------- - -fn build_passthrough_text(input: &ToolExecutionInput, raw_text: &str) -> String { - let normalized = trim_empty_edges(&normalize_lines(&strip_ansi(raw_text))) - .join("\n") - .trim() - .to_owned(); - if normalized.is_empty() { - return "(no output)".to_owned(); - } - if input.exit_code.map(|c| c != 0).unwrap_or(false) { - return format!("exit {}\n{}", input.exit_code.unwrap(), normalized); - } - normalized -} - -// --------------------------------------------------------------------------- -// format_inline -// --------------------------------------------------------------------------- - -fn format_inline( - classification: &ClassificationResult, - input: &ToolExecutionInput, - summary: &str, - facts: &HashMap, -) -> String { - let mut fact_parts: Vec = facts - .iter() - .filter(|(_, &count)| count > 0) - .map(|(name, &count)| pluralize(count, name)) - .collect(); - fact_parts.sort_unstable(); - - let mut lines: Vec = Vec::new(); - if input.exit_code.map(|c| c != 0).unwrap_or(false) { - lines.push(format!("exit {}", input.exit_code.unwrap())); - } - - let include_facts = classification.family == "search" - || (classification.family != "git-status" - && classification.family != "help" - && summary.contains("omitted")) - || (classification.family == "test-results" - && input.exit_code.map(|c| c != 0).unwrap_or(false)); - - if include_facts && !fact_parts.is_empty() { - lines.push(fact_parts.join(", ")); - } - lines.push(summary.to_owned()); - lines.join("\n").trim().to_owned() -} - -// --------------------------------------------------------------------------- -// select_inline_text -// --------------------------------------------------------------------------- - -fn select_inline_text( - classification: &ClassificationResult, - input: &ToolExecutionInput, - raw_text: &str, - compact_text: &str, - max_inline_chars: usize, -) -> String { - if classification.family == "git-status" { - return compact_text.to_owned(); - } - - let passthrough = build_passthrough_text(input, raw_text); - let raw_chars = count_text_chars(&strip_ansi(raw_text)); - let compact_chars = count_text_chars(compact_text); - let passthrough_limit = if classification.family == "help" { - max_inline_chars - } else { - TINY_OUTPUT_MAX_CHARS - }; - - if count_text_chars(&passthrough) > passthrough_limit { - return compact_text.to_owned(); - } - if raw_chars <= max_inline_chars && compact_chars >= raw_chars { - return passthrough; - } - if count_text_chars(&passthrough) <= compact_chars { - return passthrough; - } - compact_text.to_owned() -} - -// --------------------------------------------------------------------------- -// reduce_execution_with_rules (sync, library-only) -// --------------------------------------------------------------------------- - -/// Reduce `input` using a pre-loaded set of compiled rules. -/// -/// This is the synchronous, library-only entry point (no async, no artifact -/// store — those are deferred to v2). -pub fn reduce_execution_with_rules( - input: ToolExecutionInput, - rules: &[CompiledRule], - opts: &ReduceOptions, -) -> CompactResult { - let normalized_input = normalize_execution_input(input); - let raw_text = build_raw_text(&normalized_input); - let measured_raw_chars = count_text_chars(&strip_ansi(&raw_text)); - let classification = classify_execution(&normalized_input, rules, opts.classifier.as_deref()); - - log::debug!( - "[tokenjuice] reduce_execution: tool='{}' raw_chars={} family='{}'", - normalized_input.tool_name, - measured_raw_chars, - classification.family - ); - - // raw pass-through mode - if opts.raw.unwrap_or(false) { - return CompactResult { - inline_text: raw_text, - preview_text: None, - facts: None, - stats: ReductionStats { - raw_chars: measured_raw_chars, - reduced_chars: measured_raw_chars, - ratio: 1.0, - }, - classification, - }; - } - - // File-content inspection commands are never compacted - if classification.matched_reducer.as_deref() == Some("generic/fallback") - && is_file_content_inspection_command(&normalized_input) - { - return CompactResult { - inline_text: raw_text, - preview_text: None, - facts: None, - stats: ReductionStats { - raw_chars: measured_raw_chars, - reduced_chars: measured_raw_chars, - ratio: 1.0, - }, - classification, - }; - } - - // Find the matched rule (fall back to generic/fallback) - let matched_rule = rules - .iter() - .find(|r| Some(r.rule.id.as_str()) == classification.matched_reducer.as_deref()) - .or_else(|| rules.iter().find(|r| r.rule.id == "generic/fallback")) - .expect("generic/fallback rule must be present in the rule set"); - - let ApplyResult { summary, facts } = apply_rule(matched_rule, &normalized_input, &raw_text); - - let compact_text = format_inline( - &classification, - &normalized_input, - &summary.or_empty(), - &facts, - ); - - let max_inline_chars = opts.max_inline_chars.unwrap_or(1200); - let selected = select_inline_text( - &classification, - &normalized_input, - &raw_text, - &compact_text, - max_inline_chars, - ); - - let use_middle_clamp = classification.family == "help" || selected.contains('\n'); - let inline_text = if use_middle_clamp { - clamp_text_middle(&selected, max_inline_chars) - } else { - clamp_text(&selected, max_inline_chars) - }; - - let reduced_chars = count_text_chars(&inline_text); - let ratio = if measured_raw_chars == 0 { - 1.0 - } else { - reduced_chars as f64 / measured_raw_chars as f64 - }; - - log::debug!( - "[tokenjuice] reduce_execution complete: rule='{}' raw={} reduced={} ratio={:.2}", - classification.matched_reducer.as_deref().unwrap_or("?"), - measured_raw_chars, - reduced_chars, - ratio - ); - - CompactResult { - inline_text, - preview_text: if summary.is_empty() { - None - } else { - Some(summary) - }, - facts: if facts.is_empty() { None } else { Some(facts) }, - stats: ReductionStats { - raw_chars: measured_raw_chars, - reduced_chars, - ratio, - }, - classification, - } -} - -// --------------------------------------------------------------------------- -// Convenience trait -// --------------------------------------------------------------------------- - -trait OrEmpty { - fn or_empty(&self) -> String; -} -impl OrEmpty for String { - fn or_empty(&self) -> String { - if self.is_empty() { - "(no output)".to_owned() - } else { - self.clone() - } - } -} - -// --------------------------------------------------------------------------- -// Regex helpers — thread-local cache to avoid repeated compilation -// --------------------------------------------------------------------------- - -use std::cell::RefCell; - -thread_local! { - static REGEX_CACHE: RefCell> = - RefCell::new(HashMap::with_capacity(32)); -} - -/// Get or compile a regex, caching by owned pattern string. Avoids repeated -/// `Regex::new()` calls for patterns used in loops (e.g., per-line processing). -fn get_or_compile(pattern: &str) -> Option { - REGEX_CACHE.with(|cache| { - let mut map = cache.borrow_mut(); - if let Some(re) = map.get(pattern) { - return Some(re.clone()); - } - let re = regex::Regex::new(pattern).ok()?; - map.insert(pattern.to_owned(), re.clone()); - Some(re) - }) -} - -fn regex_match(pattern: &str, text: &str) -> bool { - get_or_compile(pattern) - .map(|re| re.is_match(text)) - .unwrap_or(false) -} - -fn regex_replace(pattern: &str, text: &str, replacement: &str) -> String { - get_or_compile(pattern) - .map(|re| re.replace(text, replacement).into_owned()) - .unwrap_or_else(|| text.to_owned()) -} - -fn regex_captures(pattern: &str, text: &str) -> Option> { - let re = get_or_compile(pattern)?; - let caps = re.captures(text)?; - Some( - (1..caps.len()) - .filter_map(|i| caps.get(i).map(|m| m.as_str().to_owned())) - .collect(), - ) -} - -// --------------------------------------------------------------------------- -// Unit tests -// --------------------------------------------------------------------------- - -#[cfg(test)] -#[path = "reduce_tests.rs"] -mod tests; diff --git a/src/openhuman/tokenjuice/reduce_tests.rs b/src/openhuman/tokenjuice/reduce_tests.rs deleted file mode 100644 index 45ba71164..000000000 --- a/src/openhuman/tokenjuice/reduce_tests.rs +++ /dev/null @@ -1,1813 +0,0 @@ -use super::*; -use crate::openhuman::tokenjuice::rules::load_builtin_rules; - -fn run(input: ToolExecutionInput) -> CompactResult { - let rules = load_builtin_rules(); - reduce_execution_with_rules(input, &rules, &ReduceOptions::default()) -} - -fn make_input(tool_name: &str, argv: &[&str], stdout: &str) -> ToolExecutionInput { - ToolExecutionInput { - tool_name: tool_name.to_owned(), - argv: Some(argv.iter().map(|s| s.to_string()).collect()), - stdout: Some(stdout.to_owned()), - ..Default::default() - } -} - -// --- tokenize_command --- - -#[test] -fn tokenize_basic() { - assert_eq!( - tokenize_command("git status --short"), - vec!["git", "status", "--short"] - ); -} - -#[test] -fn tokenize_quoted() { - assert_eq!( - tokenize_command(r#"echo "hello world""#), - vec!["echo", "hello world"] - ); -} - -// --- failure preservation --- - -#[test] -fn failure_preservation_uses_failure_head_tail() { - let long_stdout: String = (0..50) - .map(|i| format!("line {}", i)) - .collect::>() - .join("\n"); - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["git".to_owned(), "status".to_owned()]), - stdout: Some(long_stdout), - exit_code: Some(1), - ..Default::default() - }; - let rules = load_builtin_rules(); - let result = reduce_execution_with_rules(input.clone(), &rules, &ReduceOptions::default()); - // Should not panic and should produce a result - assert!(!result.inline_text.is_empty()); -} - -#[test] -fn success_uses_summarize_head_tail() { - let long_stdout: String = (0..50) - .map(|i| format!("line {}", i)) - .collect::>() - .join("\n"); - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["git".to_owned(), "status".to_owned()]), - stdout: Some(long_stdout), - exit_code: Some(0), - ..Default::default() - }; - let rules = load_builtin_rules(); - let ok_result = reduce_execution_with_rules(input, &rules, &ReduceOptions::default()); - assert!(!ok_result.inline_text.is_empty()); -} - -// --- git status rewriting --- - -#[test] -fn git_status_rewrites_modified() { - let stdout = "On branch main\n\ - Changes not staged for commit:\n\ - \tmodified: src/foo.rs\n"; - let input = make_input("bash", &["git", "status"], stdout); - let result = run(input); - assert!( - result.inline_text.contains("M: src/foo.rs"), - "got: {}", - result.inline_text - ); -} - -#[test] -fn git_status_rewrites_new_file() { - let stdout = "Changes to be committed:\n\ - \tnew file: src/bar.rs\n"; - let input = make_input("bash", &["git", "status"], stdout); - let result = run(input); - assert!( - result.inline_text.contains("A: src/bar.rs"), - "got: {}", - result.inline_text - ); -} - -// --- raw mode --- - -#[test] -fn raw_mode_returns_unmodified() { - let input = make_input("bash", &["git", "status"], "unchanged text"); - let rules = load_builtin_rules(); - let opts = ReduceOptions { - raw: Some(true), - ..Default::default() - }; - let result = reduce_execution_with_rules(input, &rules, &opts); - assert_eq!(result.inline_text, "unchanged text"); - assert_eq!(result.stats.ratio, 1.0); -} - -// --- clamping --- - -#[test] -fn inline_text_respects_max_inline_chars() { - let long: String = "x\n".repeat(1000); - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["some_tool".to_owned()]), - stdout: Some(long), - ..Default::default() - }; - let rules = load_builtin_rules(); - let opts = ReduceOptions { - max_inline_chars: Some(200), - ..Default::default() - }; - let result = reduce_execution_with_rules(input, &rules, &opts); - // Allow some slack for the truncation suffix - assert!( - count_text_chars(&result.inline_text) <= 300, - "inline_text too long: {} chars", - count_text_chars(&result.inline_text) - ); -} - -// --- tokenize_command edge cases --- - -#[test] -fn tokenize_backslash_escape() { - // backslash before space keeps it as part of the token - let toks = tokenize_command(r"echo hello\ world"); - assert_eq!(toks, vec!["echo", "hello world"]); -} - -#[test] -fn tokenize_trailing_backslash() { - // trailing backslash is emitted as-is - let toks = tokenize_command("echo hello\\"); - assert_eq!(toks, vec!["echo", "hello\\"]); -} - -#[test] -fn tokenize_single_quote() { - let toks = tokenize_command("echo 'hello world'"); - assert_eq!(toks, vec!["echo", "hello world"]); -} - -#[test] -fn tokenize_empty_string() { - assert!(tokenize_command("").is_empty()); - assert!(tokenize_command(" ").is_empty()); -} - -// --- normalize_execution_input --- - -#[test] -fn normalize_fills_argv_from_command() { - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - command: Some("git status --short".to_owned()), - argv: None, - ..Default::default() - }; - let out = normalize_execution_input(input); - let argv: Vec<&str> = out - .argv - .as_ref() - .unwrap() - .iter() - .map(String::as_str) - .collect(); - assert_eq!(argv, vec!["git", "status", "--short"]); -} - -#[test] -fn normalize_skips_when_argv_present() { - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - command: Some("ignored command".to_owned()), - argv: Some(vec!["git".to_owned(), "log".to_owned()]), - ..Default::default() - }; - let out = normalize_execution_input(input); - let argv: Vec<&str> = out - .argv - .as_ref() - .unwrap() - .iter() - .map(String::as_str) - .collect(); - assert_eq!(argv, vec!["git", "log"]); -} - -#[test] -fn normalize_no_op_when_empty_command() { - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - command: Some(String::new()), - argv: None, - ..Default::default() - }; - let out = normalize_execution_input(input); - assert!(out.argv.is_none() || out.argv.as_ref().map(|v| v.is_empty()).unwrap_or(true)); -} - -// --- is_file_content_inspection_command --- - -#[test] -fn cat_is_file_content_inspection() { - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["cat".to_owned(), "foo.txt".to_owned()]), - ..Default::default() - }; - assert!(is_file_content_inspection_command(&input)); -} - -#[test] -fn jq_is_file_content_inspection() { - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec![ - "jq".to_owned(), - ".".to_owned(), - "file.json".to_owned(), - ]), - ..Default::default() - }; - assert!(is_file_content_inspection_command(&input)); -} - -#[test] -fn git_is_not_file_content_inspection() { - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["git".to_owned(), "status".to_owned()]), - ..Default::default() - }; - assert!(!is_file_content_inspection_command(&input)); -} - -#[test] -fn empty_argv_is_not_file_content_inspection() { - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec![]), - ..Default::default() - }; - assert!(!is_file_content_inspection_command(&input)); -} - -#[test] -fn file_inspection_command_with_path_prefix() { - // /usr/bin/cat should still be recognized - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["/usr/bin/cat".to_owned(), "foo.txt".to_owned()]), - ..Default::default() - }; - assert!(is_file_content_inspection_command(&input)); -} - -// --- build_raw_text via reduction pipeline --- - -#[test] -fn combined_text_takes_priority() { - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["some_tool".to_owned()]), - stdout: Some("stdout data".to_owned()), - stderr: Some("stderr data".to_owned()), - combined_text: Some("combined!".to_owned()), - ..Default::default() - }; - let result = run(input); - // Raw text should be the combined_text value - assert!(result.inline_text.contains("combined!")); -} - -#[test] -fn only_stderr_used_when_stdout_empty() { - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["some_tool".to_owned()]), - stdout: Some(String::new()), - stderr: Some("error output".to_owned()), - ..Default::default() - }; - let result = run(input); - assert!(result.inline_text.contains("error output")); -} - -#[test] -fn both_stdout_and_stderr_combined() { - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["some_tool".to_owned()]), - stdout: Some("stdout line".to_owned()), - stderr: Some("stderr line".to_owned()), - ..Default::default() - }; - let result = run(input); - // Both should appear in inline text - assert!( - result.inline_text.contains("stdout line") || result.inline_text.contains("stderr line") - ); -} - -// --- git status additional rewriting --- - -#[test] -fn git_status_rewrites_deleted() { - let stdout = "Changes not staged for commit:\n\ - \tdeleted: src/old.rs\n"; - let input = make_input("bash", &["git", "status"], stdout); - let result = run(input); - assert!( - result.inline_text.contains("D: src/old.rs"), - "got: {}", - result.inline_text - ); -} - -#[test] -fn git_status_rewrites_renamed() { - let stdout = "Changes to be committed:\n\ - \trenamed: old.rs -> new.rs\n"; - let input = make_input("bash", &["git", "status"], stdout); - let result = run(input); - assert!( - result.inline_text.contains("R:"), - "got: {}", - result.inline_text - ); -} - -#[test] -fn git_status_rewrites_untracked_question_marks() { - let stdout = "Untracked files:\n\t\tfoo.txt\n"; - let input = make_input("bash", &["git", "status"], stdout); - let result = run(input); - assert!( - result.inline_text.contains("??"), - "got: {}", - result.inline_text - ); -} - -#[test] -fn git_status_on_branch_line_removed() { - let stdout = "On branch main\nnothing to commit, working tree clean\n"; - let input = make_input("bash", &["git", "status"], stdout); - let result = run(input); - assert!( - !result.inline_text.contains("On branch"), - "got: {}", - result.inline_text - ); -} - -#[test] -fn git_status_section_headers_shortened() { - let stdout = "Changes not staged for commit:\n\tmodified: foo.rs\n\ - Changes to be committed:\n\tnew file: bar.rs\n"; - let input = make_input("bash", &["git", "status"], stdout); - let result = run(input); - assert!( - result.inline_text.contains("Staged changes:") - || result.inline_text.contains("Changes not staged:"), - "got: {}", - result.inline_text - ); -} - -// --- file content inspection passthrough --- - -#[test] -fn cat_command_passes_through_unchanged() { - let content = "line1\nline2\nline3\n"; - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["cat".to_owned(), "foo.txt".to_owned()]), - stdout: Some(content.to_owned()), - ..Default::default() - }; - let rules = load_builtin_rules(); - let result = reduce_execution_with_rules(input, &rules, &ReduceOptions::default()); - // File content inspection always returns raw text (ratio 1.0) - assert_eq!(result.stats.ratio, 1.0); -} - -// --- failure_preservation with exit code non-zero --- - -#[test] -fn non_zero_exit_with_preserve_shows_more_lines() { - // cargo test rule has preserveOnFailure: true with head=18, tail=18 - let long_output: String = (0..60) - .map(|i| format!("test line {}", i)) - .collect::>() - .join("\n"); - let pass_input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["cargo".to_owned(), "test".to_owned()]), - stdout: Some(long_output.clone()), - exit_code: Some(0), - ..Default::default() - }; - let fail_input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["cargo".to_owned(), "test".to_owned()]), - stdout: Some(long_output), - exit_code: Some(1), - ..Default::default() - }; - let rules = load_builtin_rules(); - let pass_result = reduce_execution_with_rules(pass_input, &rules, &ReduceOptions::default()); - let fail_result = reduce_execution_with_rules(fail_input, &rules, &ReduceOptions::default()); - // Failure result should include more content (or at least not be empty) - assert!(!fail_result.inline_text.is_empty()); - assert!(!pass_result.inline_text.is_empty()); -} - -// --- classifier option overrides auto-classification --- - -#[test] -fn classifier_option_forces_rule() { - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["something".to_owned()]), - stdout: Some("output".to_owned()), - ..Default::default() - }; - let rules = load_builtin_rules(); - let opts = ReduceOptions { - classifier: Some("git/status".to_owned()), - ..Default::default() - }; - let result = reduce_execution_with_rules(input, &rules, &opts); - assert_eq!( - result.classification.matched_reducer.as_deref(), - Some("git/status") - ); -} - -// --- stats --- - -#[test] -fn stats_raw_chars_measured_for_empty_output() { - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["some_tool".to_owned()]), - stdout: Some(String::new()), - stderr: Some(String::new()), - ..Default::default() - }; - let result = run(input); - assert_eq!(result.stats.raw_chars, 0); - assert_eq!(result.stats.ratio, 1.0); -} - -// --- counters --- - -#[test] -fn counter_counts_matching_lines() { - // grep rule has a counter for "match" pattern ".+:.+" - let stdout = "file.rs:10: found error\nfile.rs:20: another issue\nno match here\n"; - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["grep".to_owned(), "-r".to_owned(), "error".to_owned()]), - stdout: Some(stdout.to_owned()), - ..Default::default() - }; - let result = run(input); - // Should have facts with the match counter - if let Some(facts) = &result.facts { - assert!(facts.contains_key("match"), "expected 'match' counter"); - } -} - -// --- match_output pattern --- - -#[test] -fn match_output_pattern_returns_canned_message() { - use crate::openhuman::tokenjuice::{ - rules::compiler::compile_rule, - types::{RuleMatch, RuleOutputMatch}, - }; - - // Build a rule with matchOutput that fires when content is "nothing to commit" - let rule = crate::openhuman::tokenjuice::types::JsonRule { - id: "test/match-output".to_owned(), - family: "test".to_owned(), - description: None, - priority: None, - on_empty: None, - match_output: Some(vec![RuleOutputMatch { - pattern: "nothing to commit".to_owned(), - message: "Clean working tree".to_owned(), - flags: None, - }]), - counter_source: None, - r#match: RuleMatch::default(), - filters: None, - transforms: None, - summarize: None, - counters: None, - failure: None, - }; - - let compiled = compile_rule( - rule, - crate::openhuman::tokenjuice::types::RuleOrigin::Builtin, - "builtin:test/match-output".to_owned(), - ); - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["git".to_owned(), "status".to_owned()]), - stdout: Some("nothing to commit, working tree clean".to_owned()), - ..Default::default() - }; - let rules = vec![ - compiled, - // Need fallback to be present - load_builtin_rules() - .into_iter() - .find(|r| r.rule.id == "generic/fallback") - .unwrap(), - ]; - let opts = ReduceOptions { - classifier: Some("test/match-output".to_owned()), - ..Default::default() - }; - let result = reduce_execution_with_rules(input, &rules, &opts); - assert_eq!(result.inline_text, "Clean working tree"); -} - -// --- on_empty --- - -#[test] -fn on_empty_returns_custom_message() { - use crate::openhuman::tokenjuice::{rules::compiler::compile_rule, types::RuleMatch}; - - let rule = crate::openhuman::tokenjuice::types::JsonRule { - id: "test/on-empty".to_owned(), - family: "test".to_owned(), - description: None, - priority: None, - on_empty: Some("(nothing here)".to_owned()), - match_output: None, - counter_source: None, - r#match: RuleMatch::default(), - filters: Some(crate::openhuman::tokenjuice::types::RuleFilters { - // skip everything so lines become empty - skip_patterns: Some(vec![".*".to_owned()]), - keep_patterns: None, - }), - transforms: None, - summarize: None, - counters: None, - failure: None, - }; - let compiled = compile_rule( - rule, - crate::openhuman::tokenjuice::types::RuleOrigin::Builtin, - "builtin:test/on-empty".to_owned(), - ); - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["something".to_owned()]), - stdout: Some("some output that gets filtered out".to_owned()), - ..Default::default() - }; - let fb = load_builtin_rules() - .into_iter() - .find(|r| r.rule.id == "generic/fallback") - .unwrap(); - let rules = vec![compiled, fb]; - let opts = ReduceOptions { - classifier: Some("test/on-empty".to_owned()), - ..Default::default() - }; - let result = reduce_execution_with_rules(input, &rules, &opts); - assert_eq!(result.inline_text, "(nothing here)"); -} - -// --- pretty_print_json transform --- - -#[test] -fn pretty_print_json_transform_works() { - use crate::openhuman::tokenjuice::{rules::compiler::compile_rule, types::RuleMatch}; - - let rule = crate::openhuman::tokenjuice::types::JsonRule { - id: "test/pretty-json".to_owned(), - family: "test".to_owned(), - description: None, - priority: None, - on_empty: None, - match_output: None, - counter_source: None, - r#match: RuleMatch::default(), - filters: None, - transforms: Some(crate::openhuman::tokenjuice::types::RuleTransforms { - pretty_print_json: Some(true), - strip_ansi: None, - trim_empty_edges: None, - dedupe_adjacent: None, - }), - summarize: None, - counters: None, - failure: None, - }; - let compiled = compile_rule( - rule, - crate::openhuman::tokenjuice::types::RuleOrigin::Builtin, - "builtin:test/pretty-json".to_owned(), - ); - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["jq".to_owned()]), - stdout: Some(r#"{"key":"value","num":42}"#.to_owned()), - ..Default::default() - }; - let fb = load_builtin_rules() - .into_iter() - .find(|r| r.rule.id == "generic/fallback") - .unwrap(); - let rules = vec![compiled, fb]; - let opts = ReduceOptions { - classifier: Some("test/pretty-json".to_owned()), - ..Default::default() - }; - let result = reduce_execution_with_rules(input, &rules, &opts); - // Pretty-printed JSON should contain newlines - assert!( - result.inline_text.contains('\n') || result.inline_text.contains("key"), - "got: {}", - result.inline_text - ); -} - -// --- gh output rewriting --- - -#[test] -fn gh_pr_list_json_output_compacted() { - let json_line = - r#"{"number":42,"title":"Fix the bug","state":"open","headRefName":"fix/issue-42"}"#; - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["gh".to_owned(), "pr".to_owned(), "list".to_owned()]), - stdout: Some(json_line.to_owned()), - ..Default::default() - }; - let result = run(input); - assert!( - result.inline_text.contains("#42"), - "got: {}", - result.inline_text - ); - assert!( - result.inline_text.contains("Fix the bug"), - "got: {}", - result.inline_text - ); -} - -#[test] -fn gh_table_format_fallback() { - // Non-JSON gh output falls back to table formatting - let table_output = "42 Fix the bug open fix/issue-42 2024-01-01\n123 Another PR closed main 2024-01-02"; - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["gh".to_owned(), "pr".to_owned(), "list".to_owned()]), - stdout: Some(table_output.to_owned()), - ..Default::default() - }; - let result = run(input); - assert!( - result.inline_text.contains("#42") || result.inline_text.contains("Fix the bug"), - "got: {}", - result.inline_text - ); -} - -// --- keep_patterns --- - -#[test] -fn keep_patterns_filter_lines() { - use crate::openhuman::tokenjuice::{rules::compiler::compile_rule, types::RuleMatch}; - - let rule = crate::openhuman::tokenjuice::types::JsonRule { - id: "test/keep".to_owned(), - family: "test".to_owned(), - description: None, - priority: None, - on_empty: None, - match_output: None, - counter_source: None, - r#match: RuleMatch::default(), - filters: Some(crate::openhuman::tokenjuice::types::RuleFilters { - skip_patterns: None, - keep_patterns: Some(vec!["ERROR".to_owned()]), - }), - transforms: None, - summarize: None, - counters: None, - failure: None, - }; - let compiled = compile_rule( - rule, - crate::openhuman::tokenjuice::types::RuleOrigin::Builtin, - "builtin:test/keep".to_owned(), - ); - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["some_cmd".to_owned()]), - stdout: Some("INFO: all good\nERROR: something failed\nDEBUG: verbose".to_owned()), - ..Default::default() - }; - let fb = load_builtin_rules() - .into_iter() - .find(|r| r.rule.id == "generic/fallback") - .unwrap(); - let rules = vec![compiled, fb]; - let opts = ReduceOptions { - classifier: Some("test/keep".to_owned()), - ..Default::default() - }; - let result = reduce_execution_with_rules(input, &rules, &opts); - assert!( - result.inline_text.contains("ERROR"), - "got: {}", - result.inline_text - ); - // INFO and DEBUG lines should not appear (they don't match keep pattern) - assert!( - !result.inline_text.contains("INFO"), - "got: {}", - result.inline_text - ); -} - -// --- counter_source: pre_keep --- - -#[test] -fn counter_source_pre_keep_counts_before_filtering() { - use crate::openhuman::tokenjuice::{ - rules::compiler::compile_rule, - types::{CounterSource, RuleCounter, RuleMatch}, - }; - - let rule = crate::openhuman::tokenjuice::types::JsonRule { - id: "test/pre-keep".to_owned(), - family: "test".to_owned(), - description: None, - priority: None, - on_empty: None, - match_output: None, - counter_source: Some(CounterSource::PreKeep), - r#match: RuleMatch::default(), - filters: Some(crate::openhuman::tokenjuice::types::RuleFilters { - skip_patterns: None, - keep_patterns: Some(vec!["KEEP".to_owned()]), - }), - transforms: None, - summarize: None, - counters: Some(vec![RuleCounter { - name: "error".to_owned(), - pattern: "ERROR".to_owned(), - flags: None, - }]), - failure: None, - }; - let compiled = compile_rule( - rule, - crate::openhuman::tokenjuice::types::RuleOrigin::Builtin, - "builtin:test/pre-keep".to_owned(), - ); - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["some_cmd".to_owned()]), - // ERROR lines would be filtered out by keep_patterns (only KEEP is kept) - // but pre-keep counters should count them anyway - stdout: Some("ERROR: issue1\nERROR: issue2\nKEEP: this line".to_owned()), - ..Default::default() - }; - let fb = load_builtin_rules() - .into_iter() - .find(|r| r.rule.id == "generic/fallback") - .unwrap(); - let rules = vec![compiled, fb]; - let opts = ReduceOptions { - classifier: Some("test/pre-keep".to_owned()), - ..Default::default() - }; - let result = reduce_execution_with_rules(input, &rules, &opts); - // Counter should have counted the 2 ERROR lines from pre-keep phase - if let Some(facts) = &result.facts { - let error_count = facts.get("error").copied().unwrap_or(0); - assert_eq!(error_count, 2, "pre-keep should count 2 errors"); - } -} - -// --- help family uses middle clamping --- - -#[test] -fn help_family_uses_middle_clamping() { - // The generic/help rule matches --help argument - let long_help: String = "USAGE: tool [OPTIONS]\n".to_owned() - + &" --option-N Description of option N\n".repeat(200); - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["tool".to_owned(), "--help".to_owned()]), - stdout: Some(long_help), - ..Default::default() - }; - let rules = load_builtin_rules(); - let opts = ReduceOptions { - max_inline_chars: Some(400), - ..Default::default() - }; - let result = reduce_execution_with_rules(input, &rules, &opts); - assert!( - count_text_chars(&result.inline_text) <= 500, - "inline_text too long: {} chars", - count_text_chars(&result.inline_text) - ); -} - -// --- git-status family short-circuit in select_inline_text --- - -#[test] -fn git_status_family_returns_compact_text_directly() { - let stdout = "M: src/foo.rs\nA: src/bar.rs\n"; - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["git".to_owned(), "status".to_owned()]), - stdout: Some(stdout.to_owned()), - ..Default::default() - }; - let result = run(input); - // Should produce something - assert!(!result.inline_text.is_empty()); -} - -// --- passthrough for tiny output --- - -#[test] -fn tiny_output_returns_passthrough() { - let tiny = "ok"; - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["some_cmd".to_owned()]), - stdout: Some(tiny.to_owned()), - ..Default::default() - }; - let result = run(input); - assert_eq!(result.inline_text, "ok"); -} - -// --- passthrough with exit code prefix --- - -#[test] -fn passthrough_with_nonzero_exit_prefixes_exit_code() { - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["unknown_tool".to_owned()]), - stdout: Some("tiny output".to_owned()), - exit_code: Some(2), - ..Default::default() - }; - let result = run(input); - // Should include "exit 2" - assert!( - result.inline_text.contains("exit 2"), - "got: {}", - result.inline_text - ); -} - -// --- gh json record with labels and comments --- - -#[test] -fn gh_json_with_labels_and_comments() { - let json_line = r#"{"number":7,"title":"Add feature","state":"open","headRefName":"feat/x","labels":[{"name":"enhancement"},{"name":"help wanted"}],"comments":{"totalCount":3},"updatedAt":"2024-01-15T10:00:00Z"}"#; - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["gh".to_owned(), "issue".to_owned(), "list".to_owned()]), - stdout: Some(json_line.to_owned()), - ..Default::default() - }; - let result = run(input); - assert!( - result.inline_text.contains("#7"), - "got: {}", - result.inline_text - ); - assert!( - result.inline_text.contains("Add feature"), - "got: {}", - result.inline_text - ); -} - -// --- gh json with displayTitle and databaseId --- - -#[test] -fn gh_json_with_display_title_and_database_id() { - let json_line = r#"{"databaseId":999,"displayTitle":"My Workflow Run","status":"completed","conclusion":"success","headBranch":"main"}"#; - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["gh".to_owned(), "run".to_owned(), "list".to_owned()]), - stdout: Some(json_line.to_owned()), - ..Default::default() - }; - let result = run(input); - assert!( - result.inline_text.contains("#999") || result.inline_text.contains("My Workflow Run"), - "got: {}", - result.inline_text - ); -} - -// --- gh empty output --- - -#[test] -fn gh_empty_lines_returns_empty() { - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["gh".to_owned(), "pr".to_owned(), "list".to_owned()]), - stdout: Some(" \n \n".to_owned()), - ..Default::default() - }; - let result = run(input); - // Should produce some output (no output marker or empty) - assert!(!result.inline_text.is_empty() || result.inline_text.is_empty()); -} - -// --- gh table format edge cases --- - -#[test] -fn gh_table_empty_line_returns_empty_string() { - // An empty line in gh table output should produce empty string - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["gh".to_owned(), "pr".to_owned(), "list".to_owned()]), - stdout: Some(" \n42 Fix bug open feat/fix 2024-01-01\n".to_owned()), - ..Default::default() - }; - let result = run(input); - // The non-empty line should be formatted - assert!( - result.inline_text.contains("#42") || result.inline_text.contains("Fix bug"), - "got: {}", - result.inline_text - ); -} - -#[test] -fn gh_table_three_columns_context() { - // Table with 3 cols: number, title, state (no context, no 4th col) - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["gh".to_owned(), "pr".to_owned(), "list".to_owned()]), - stdout: Some("99 My PR open\n".to_owned()), - ..Default::default() - }; - let result = run(input); - assert!( - result.inline_text.contains("#99") || result.inline_text.contains("My PR"), - "got: {}", - result.inline_text - ); -} - -#[test] -fn gh_table_non_numeric_first_column() { - // When first column is not numeric, falls back to compact_whitespace - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["gh".to_owned(), "issue".to_owned(), "list".to_owned()]), - stdout: Some("feature My Issue open\n".to_owned()), - ..Default::default() - }; - let result = run(input); - assert!(!result.inline_text.is_empty()); -} - -// --- gh json: comment count variants --- - -#[test] -fn gh_json_comment_count_as_array() { - // comments field as array (length = comment count) - let json_line = r#"{"number":5,"title":"PR Title","state":"open","comments":[{"body":"comment1"},{"body":"comment2"}]}"#; - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["gh".to_owned(), "pr".to_owned(), "list".to_owned()]), - stdout: Some(json_line.to_owned()), - ..Default::default() - }; - let result = run(input); - assert!( - result.inline_text.contains("#5"), - "got: {}", - result.inline_text - ); - // 2 comments shown as "2c" - assert!( - result.inline_text.contains("2c"), - "got: {}", - result.inline_text - ); -} - -#[test] -fn gh_json_comment_count_as_number() { - // comments as plain number - let json_line = r#"{"number":6,"title":"Another PR","state":"closed","comments":4}"#; - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["gh".to_owned(), "pr".to_owned(), "list".to_owned()]), - stdout: Some(json_line.to_owned()), - ..Default::default() - }; - let result = run(input); - assert!( - result.inline_text.contains("#6"), - "got: {}", - result.inline_text - ); - assert!( - result.inline_text.contains("4c"), - "got: {}", - result.inline_text - ); -} - -// --- gh json: labels as string array --- - -#[test] -fn gh_json_labels_as_string_array() { - // labels as array of strings (not objects) - let json_line = - r#"{"number":8,"title":"Tagged PR","state":"open","labels":["bug","urgent",""]}"#; - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["gh".to_owned(), "pr".to_owned(), "list".to_owned()]), - stdout: Some(json_line.to_owned()), - ..Default::default() - }; - let result = run(input); - assert!( - result.inline_text.contains("#8"), - "got: {}", - result.inline_text - ); - // Should include label names (empty string filtered) - assert!( - result.inline_text.contains("bug") || result.inline_text.contains("{"), - "got: {}", - result.inline_text - ); -} - -#[test] -fn gh_json_labels_non_array_is_ignored() { - // labels as non-array → should not crash - let json_line = r#"{"number":9,"title":"PR no labels","state":"open","labels":"bug"}"#; - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["gh".to_owned(), "pr".to_owned(), "list".to_owned()]), - stdout: Some(json_line.to_owned()), - ..Default::default() - }; - let result = run(input); - assert!( - result.inline_text.contains("#9"), - "got: {}", - result.inline_text - ); -} - -// --- pretty_print_json: array and non-json --- - -#[test] -fn pretty_print_json_array_output() { - use crate::openhuman::tokenjuice::{rules::compiler::compile_rule, types::RuleMatch}; - - let rule = crate::openhuman::tokenjuice::types::JsonRule { - id: "test/ppjson-arr".to_owned(), - family: "test".to_owned(), - description: None, - priority: None, - on_empty: None, - match_output: None, - counter_source: None, - r#match: RuleMatch::default(), - filters: None, - transforms: Some(crate::openhuman::tokenjuice::types::RuleTransforms { - pretty_print_json: Some(true), - strip_ansi: None, - trim_empty_edges: None, - dedupe_adjacent: None, - }), - summarize: None, - counters: None, - failure: None, - }; - let compiled = compile_rule( - rule, - crate::openhuman::tokenjuice::types::RuleOrigin::Builtin, - "builtin:test/ppjson-arr".to_owned(), - ); - // JSON array - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["some_tool".to_owned()]), - stdout: Some(r#"[1,2,3]"#.to_owned()), - ..Default::default() - }; - let fb = load_builtin_rules() - .into_iter() - .find(|r| r.rule.id == "generic/fallback") - .unwrap(); - let rules = vec![compiled, fb]; - let opts = ReduceOptions { - classifier: Some("test/ppjson-arr".to_owned()), - ..Default::default() - }; - let result = reduce_execution_with_rules(input, &rules, &opts); - assert!(!result.inline_text.is_empty()); -} - -#[test] -fn pretty_print_json_non_json_passthrough() { - use crate::openhuman::tokenjuice::{rules::compiler::compile_rule, types::RuleMatch}; - - let rule = crate::openhuman::tokenjuice::types::JsonRule { - id: "test/ppjson-plain".to_owned(), - family: "test".to_owned(), - description: None, - priority: None, - on_empty: None, - match_output: None, - counter_source: None, - r#match: RuleMatch::default(), - filters: None, - transforms: Some(crate::openhuman::tokenjuice::types::RuleTransforms { - pretty_print_json: Some(true), - strip_ansi: None, - trim_empty_edges: None, - dedupe_adjacent: None, - }), - summarize: None, - counters: None, - failure: None, - }; - let compiled = compile_rule( - rule, - crate::openhuman::tokenjuice::types::RuleOrigin::Builtin, - "builtin:test/ppjson-plain".to_owned(), - ); - // Not JSON - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["some_tool".to_owned()]), - stdout: Some("plain text output".to_owned()), - ..Default::default() - }; - let fb = load_builtin_rules() - .into_iter() - .find(|r| r.rule.id == "generic/fallback") - .unwrap(); - let rules = vec![compiled, fb]; - let opts = ReduceOptions { - classifier: Some("test/ppjson-plain".to_owned()), - ..Default::default() - }; - let result = reduce_execution_with_rules(input, &rules, &opts); - assert!(result.inline_text.contains("plain text output")); -} - -// --- normalize_execution_input: empty tokenized argv --- - -#[test] -fn normalize_whitespace_only_command_returns_no_argv() { - // tokenize_command("''") → empty (quotes enclose nothing useful) - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - command: Some("''".to_owned()), // tokenizes to empty because quotes contain nothing - argv: None, - ..Default::default() - }; - let out = normalize_execution_input(input); - // argv should remain None or empty since tokenized form is empty - assert!( - out.argv.as_ref().map(|v| v.is_empty()).unwrap_or(true), - "expected empty or no argv" - ); -} - -// --- select_inline_text: passthrough <= compact_chars branch --- - -#[test] -fn select_inline_text_passthrough_shorter_than_compact() { - // When passthrough is shorter than compact, passthrough is returned - // This happens for short output where compact is longer (rare but possible) - let short_output = "short"; - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["some_tool".to_owned()]), - stdout: Some(short_output.to_owned()), - ..Default::default() - }; - let result = run(input); - // Short output should just be returned as-is - assert_eq!(result.inline_text, "short"); -} - -// --- zero raw_chars gives ratio 1.0 --- - -#[test] -fn zero_raw_chars_ratio_is_one() { - let input = ToolExecutionInput { - tool_name: "bash".to_owned(), - argv: Some(vec!["some_tool".to_owned()]), - stdout: None, - stderr: None, - ..Default::default() - }; - let result = run(input); - assert_eq!(result.stats.ratio, 1.0); - assert_eq!(result.stats.raw_chars, 0); -} - -// --- gh json with workflowName field --- - -#[test] -fn gh_json_workflow_name_field() { - let json_line = r#"{"databaseId":100,"workflowName":"CI/CD Pipeline","status":"in_progress"}"#; - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["gh".to_owned(), "run".to_owned(), "list".to_owned()]), - stdout: Some(json_line.to_owned()), - ..Default::default() - }; - let result = run(input); - assert!( - result.inline_text.contains("CI/CD Pipeline") || result.inline_text.contains("#100"), - "got: {}", - result.inline_text - ); -} - -// --- gh json: no title field returns None (format_gh_json_record returns None) --- - -#[test] -fn gh_json_missing_title_falls_to_table_format() { - // JSON line without any title-like field → format_gh_json_record returns None - // → falls back to table format since argv[0] == "gh" - let json_line = r#"{"number":1,"state":"open"}"#; - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["gh".to_owned(), "pr".to_owned(), "list".to_owned()]), - stdout: Some(json_line.to_owned()), - ..Default::default() - }; - let result = run(input); - // Should not panic, result may be formatted or passthrough - assert!(!result.inline_text.is_empty() || result.inline_text.is_empty()); -} - -// --- skip_patterns --- - -#[test] -fn skip_patterns_remove_matching_lines() { - // cargo test rule skips "Compiling" and "Finished" lines - let stdout = - " Compiling foo v0.1.0\n Finished dev [unoptimized] target(s)\ntest foo ... ok\n"; - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["cargo".to_owned(), "test".to_owned()]), - stdout: Some(stdout.to_owned()), - ..Default::default() - }; - let result = run(input); - assert!( - !result.inline_text.contains("Compiling"), - "got: {}", - result.inline_text - ); -} - -// --- format_inline: search family includes facts --- - -#[test] -fn search_family_includes_fact_counts() { - let output = "file.rs:10: match one\nfile.rs:20: match two\nfile.rs:30: match three\n"; - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["grep".to_owned(), "-r".to_owned(), "match".to_owned()]), - stdout: Some(output.to_owned()), - ..Default::default() - }; - let result = run(input); - // Search family should include fact counts in inline text - // (either via "matches" text or facts map) - assert!(!result.inline_text.is_empty()); -} - -// --- test-results family with failure exits includes facts --- - -#[test] -fn test_results_failure_includes_failed_count() { - let output = "test foo ... ok\ntest bar ... FAILED\ntest baz ... ok\nFAILED\n"; - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["cargo".to_owned(), "test".to_owned()]), - stdout: Some(output.to_owned()), - exit_code: Some(1), - ..Default::default() - }; - let result = run(input); - // Should contain information about the failure - assert!(!result.inline_text.is_empty()); -} - -// --- git/status rewrite: "and have N and M different commits" --- - -#[test] -fn git_status_diverged_message_removed() { - let stdout = "On branch main\nYour branch and 'origin/main' have diverged,\nand have 2 and 3 different commits each.\n"; - let input = make_input("bash", &["git", "status"], stdout); - let result = run(input); - assert!( - !result.inline_text.contains("and have"), - "got: {}", - result.inline_text - ); -} - -#[test] -fn git_status_empty_line_handled() { - // Empty lines in git status output should produce empty strings (not be dropped) - let stdout = "Changes not staged for commit:\n\n\tmodified: foo.rs\n"; - let input = make_input("bash", &["git", "status"], stdout); - let result = run(input); - // Should still have M: foo.rs - assert!( - result.inline_text.contains("M: foo.rs"), - "got: {}", - result.inline_text - ); -} - -#[test] -fn git_status_no_changes_hint_removed() { - let stdout = "nothing added to commit but untracked files present (use \"git add\" to track)\n"; - let input = make_input("bash", &["git", "status"], stdout); - let result = run(input); - // This line should be filtered out - assert!( - !result - .inline_text - .contains("nothing added to commit but untracked"), - "got: {}", - result.inline_text - ); -} - -#[test] -fn git_status_use_git_hint_removed() { - let stdout = "(use \"git add ...\" to update what will be committed)\n"; - let input = make_input("bash", &["git", "status"], stdout); - let result = run(input); - assert!( - !result.inline_text.contains("use \"git add"), - "got: {}", - result.inline_text - ); -} - -#[test] -fn git_status_porcelain_format_mm_code() { - // Two-char porcelain status code - let stdout = "MM src/foo.rs\nA src/bar.rs\n"; - let input = make_input("bash", &["git", "status"], stdout); - let result = run(input); - // Should be parsed somehow (via porcelain fallthrough or direct match) - assert!(!result.inline_text.is_empty()); -} - -#[test] -fn git_status_consecutive_empty_lines_collapsed() { - // Multiple consecutive blank lines should be collapsed to one - let stdout = "Changes not staged for commit:\n\n\n\tmodified: a.rs\n\n\n\tmodified: b.rs\n"; - let input = make_input("bash", &["git", "status"], stdout); - let result = run(input); - assert!( - result.inline_text.contains("M: a.rs"), - "got: {}", - result.inline_text - ); -} - -#[test] -fn git_status_no_changes_to_commit() { - let stdout = "no changes added to commit (use \"git add\" and/or \"git commit -a\")\n"; - let input = make_input("bash", &["git", "status"], stdout); - let result = run(input); - assert!( - !result.inline_text.contains("no changes added to commit"), - "got: {}", - result.inline_text - ); -} - -// --- head_tail with zero counts --- - -#[test] -fn head_tail_zero_head() { - use crate::openhuman::tokenjuice::text::head_tail; - let lines: Vec = (0..5).map(|i| format!("line{}", i)).collect(); - // head=0, tail=2 should return last 2 lines - let result = head_tail(&lines, 0, 2); - assert_eq!(result.len(), 3); // omission marker + 2 tail lines - assert!(result[0].contains("omitted")); -} - -#[test] -fn head_tail_zero_tail() { - use crate::openhuman::tokenjuice::text::head_tail; - let lines: Vec = (0..5).map(|i| format!("line{}", i)).collect(); - let result = head_tail(&lines, 2, 0); - // 2 head + omission marker + 0 tail - assert_eq!(result.len(), 3); -} - -#[test] -fn head_tail_n_greater_than_line_count() { - use crate::openhuman::tokenjuice::text::head_tail; - let lines: Vec = (0..3).map(|i| format!("line{}", i)).collect(); - // head+tail > total, should passthrough unchanged - let result = head_tail(&lines, 5, 5); - assert_eq!(result, lines); -} - -// --- Rust toolchain reduction integration tests --- - -#[test] -fn cargo_clippy_strips_compiling_preserves_warnings() { - let stdout = "\ - Compiling serde v1.0.200 - Compiling serde_json v1.0.120 - Compiling openhuman v0.1.0 (/home/user/project) - Checking openhuman v0.1.0 (/home/user/project) -warning: unused variable: `x` - --> src/main.rs:10:9 - | -10 | let x = 42; - | ^ help: if this is intentional, prefix it with an underscore: `_x` - | - = note: `#[warn(unused_variables)]` on by default - -warning: unused import: `std::collections::HashMap` - --> src/lib.rs:3:5 - | -3 | use std::collections::HashMap; - | ^^^^^^^^^^^^^^^^^^^^^^^^^ - | - = note: `#[warn(unused_imports)]` on by default - -warning: `openhuman` (lib) generated 2 warnings - Finished `dev` profile [unoptimized + debuginfo] target(s) in 4.32s -"; - - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["cargo".to_owned(), "clippy".to_owned()]), - stdout: Some(stdout.to_owned()), - exit_code: Some(0), - ..Default::default() - }; - let result = run(input); - - // Warnings must survive - assert!( - result.inline_text.contains("unused variable"), - "clippy warning text must be preserved, got: {}", - result.inline_text - ); - assert!( - result.inline_text.contains("unused import"), - "clippy warning text must be preserved, got: {}", - result.inline_text - ); - - // Compiling noise must be stripped - assert!( - !result.inline_text.contains("Compiling serde v1.0.200"), - "Compiling lines should be stripped, got: {}", - result.inline_text - ); - assert!( - !result.inline_text.contains("Checking openhuman"), - "Checking lines should be stripped, got: {}", - result.inline_text - ); - - // Classification - assert_eq!( - result.classification.matched_reducer.as_deref(), - Some("lint/cargo-clippy") - ); - assert_eq!(result.classification.family, "lint-results"); - - // Counters should have counted warnings - if let Some(facts) = &result.facts { - if let Some(&warning_count) = facts.get("warning") { - assert!( - warning_count >= 2, - "expected at least 2 warnings counted, got {}", - warning_count - ); - } - } -} - -#[test] -fn cargo_build_failure_preserves_errors() { - let stdout = "\ - Compiling serde v1.0.200 - Compiling serde_json v1.0.120 - Compiling openhuman v0.1.0 (/home/user/project) -error[E0308]: mismatched types - --> src/main.rs:15:20 - | -15 | let x: u32 = \"hello\"; - | --- ^^^^^^^ expected `u32`, found `&str` - | | - | expected due to this - -error[E0425]: cannot find value `undefined_var` in this scope - --> src/main.rs:20:5 - | -20 | undefined_var - | ^^^^^^^^^^^^^ not found in this scope - -error: aborting due to 2 previous errors - -Some errors have detailed explanations: E0308, E0425. -For more information about an error, try `rustc --explain E0308`. -error: could not compile `openhuman` (bin \"openhuman\") due to 2 previous errors -"; - - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["cargo".to_owned(), "build".to_owned()]), - stdout: Some(stdout.to_owned()), - exit_code: Some(101), - ..Default::default() - }; - let result = run(input); - - // Errors must survive - assert!( - result.inline_text.contains("mismatched types"), - "error diagnostic must be preserved, got: {}", - result.inline_text - ); - assert!( - result.inline_text.contains("cannot find value"), - "error diagnostic must be preserved, got: {}", - result.inline_text - ); - - // Compiling noise must be stripped - assert!( - !result.inline_text.contains("Compiling serde v1.0.200"), - "Compiling lines should be stripped, got: {}", - result.inline_text - ); - - // Classification - assert_eq!( - result.classification.matched_reducer.as_deref(), - Some("build/cargo-build") - ); - assert_eq!(result.classification.family, "build-rust"); - - // Error counter - if let Some(facts) = &result.facts { - if let Some(&error_count) = facts.get("error") { - assert!( - error_count >= 2, - "expected at least 2 errors counted, got {}", - error_count - ); - } - } -} - -#[test] -fn cargo_check_classifies_as_cargo_build() { - let stdout = "\ - Checking serde v1.0.200 - Checking openhuman v0.1.0 (/home/user/project) -warning: unused variable: `y` - --> src/lib.rs:5:9 - | -5 | let y = 10; - | ^ - -warning: `openhuman` (lib) generated 1 warning - Finished `dev` profile [unoptimized + debuginfo] target(s) in 1.23s -"; - - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["cargo".to_owned(), "check".to_owned()]), - stdout: Some(stdout.to_owned()), - exit_code: Some(0), - ..Default::default() - }; - let result = run(input); - - // Should classify as build/cargo-build (not cargo-test) - assert_eq!( - result.classification.matched_reducer.as_deref(), - Some("build/cargo-build"), - "cargo check should classify as build/cargo-build" - ); - - // Checking noise stripped - assert!( - !result.inline_text.contains("Checking serde"), - "Checking lines should be stripped, got: {}", - result.inline_text - ); - - // Warning preserved - assert!( - result.inline_text.contains("unused variable"), - "warnings must be preserved, got: {}", - result.inline_text - ); -} - -#[test] -fn cargo_fmt_check_preserves_diff_hunks() { - let stdout = "\ -Diff in /home/user/project/src/main.rs at line 5: -- let x=42; -+ let x = 42; -Diff in /home/user/project/src/lib.rs at line 10: --fn foo(){ -+fn foo() { -"; - - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec![ - "cargo".to_owned(), - "fmt".to_owned(), - "--check".to_owned(), - ]), - stdout: Some(stdout.to_owned()), - exit_code: Some(1), - ..Default::default() - }; - let result = run(input); - - // Classification - assert_eq!( - result.classification.matched_reducer.as_deref(), - Some("lint/cargo-fmt"), - "cargo fmt should classify as lint/cargo-fmt" - ); - assert_eq!(result.classification.family, "lint-results"); - - // Diff headers must survive - assert!( - result.inline_text.contains("Diff in"), - "Diff in lines must be preserved, got: {}", - result.inline_text - ); - - // Counter should count unformatted files - if let Some(facts) = &result.facts { - if let Some(&file_count) = facts.get("unformatted file") { - assert_eq!( - file_count, 2, - "expected 2 unformatted files counted, got {}", - file_count - ); - } - } -} - -#[test] -fn cargo_doc_strips_documenting_noise() { - let stdout = "\ - Compiling serde v1.0.200 - Compiling openhuman v0.1.0 (/home/user/project) - Documenting openhuman v0.1.0 (/home/user/project) -warning: missing documentation for a public function - --> src/lib.rs:10:1 - | -10 | pub fn undocumented() {} - | ^^^^^^^^^^^^^^^^^^^^^ - -warning: unresolved link to `NonExistent` - --> src/lib.rs:5:10 - | -5 | /// See [`NonExistent`] - | ^^^^^^^^^^^^^ no item named `NonExistent` in scope - - Finished `doc` profile [unoptimized + debuginfo] target(s) in 3.45s -"; - - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["cargo".to_owned(), "doc".to_owned()]), - stdout: Some(stdout.to_owned()), - exit_code: Some(0), - ..Default::default() - }; - let result = run(input); - - // Classification - assert_eq!( - result.classification.matched_reducer.as_deref(), - Some("build/cargo-doc"), - "cargo doc should classify as build/cargo-doc" - ); - assert_eq!(result.classification.family, "build-rust"); - - // Documenting noise stripped - assert!( - !result.inline_text.contains("Documenting openhuman"), - "Documenting lines should be stripped, got: {}", - result.inline_text - ); - - // Compiling noise stripped - assert!( - !result.inline_text.contains("Compiling serde"), - "Compiling lines should be stripped, got: {}", - result.inline_text - ); - - // Warnings preserved - assert!( - result.inline_text.contains("missing documentation"), - "doc warnings must be preserved, got: {}", - result.inline_text - ); -} - -#[test] -fn cargo_build_success_compacts_output() { - // Large successful build output with many Compiling lines — should compress heavily - let mut lines = Vec::new(); - for i in 0..80 { - lines.push(format!(" Compiling dep-{} v0.{}.0", i, i)); - } - lines.push(" Compiling my-project v0.1.0 (/home/user/project)".to_owned()); - lines.push( - " Finished `dev` profile [unoptimized + debuginfo] target(s) in 12.34s".to_owned(), - ); - let stdout = lines.join("\n"); - - let input = ToolExecutionInput { - tool_name: "exec".to_owned(), - argv: Some(vec!["cargo".to_owned(), "build".to_owned()]), - stdout: Some(stdout.clone()), - exit_code: Some(0), - ..Default::default() - }; - let result = run(input); - - // All Compiling lines should be stripped - assert!( - !result.inline_text.contains("Compiling dep-"), - "Compiling lines should be stripped, got: {}", - result.inline_text - ); - - // Output should be significantly shorter than input - assert!( - result.stats.reduced_chars < result.stats.raw_chars, - "expected compaction: reduced={} raw={}", - result.stats.reduced_chars, - result.stats.raw_chars - ); -} diff --git a/src/openhuman/tokenjuice/rules/builtin.rs b/src/openhuman/tokenjuice/rules/builtin.rs deleted file mode 100644 index 159e32756..000000000 --- a/src/openhuman/tokenjuice/rules/builtin.rs +++ /dev/null @@ -1,387 +0,0 @@ -//! Embedded built-in rule JSON files. -//! -//! Each rule is embedded at compile time via `include_str!` so the module -//! works with zero external configuration. - -/// All vendored rule JSON files embedded as `(id, json)` pairs. -/// -/// The `generic/fallback` rule MUST be present; the compiler asserts this via -/// `builtin_rules()`. -/// -/// Rules are listed alphabetically by id; `generic/fallback` is placed last -/// because the rule loader sorts it to the end of the compiled list. -pub static BUILTIN_RULE_JSONS: &[(&str, &str)] = &[ - ( - "archive/tar", - include_str!("../vendor/rules/archive__tar.json"), - ), - ( - "archive/unzip", - include_str!("../vendor/rules/archive__unzip.json"), - ), - ( - "archive/zip", - include_str!("../vendor/rules/archive__zip.json"), - ), - ( - "build/cargo-build", - include_str!("../vendor/rules/build__cargo-build.json"), - ), - ( - "build/cargo-doc", - include_str!("../vendor/rules/build__cargo-doc.json"), - ), - ( - "build/esbuild", - include_str!("../vendor/rules/build__esbuild.json"), - ), - ("build/tsc", include_str!("../vendor/rules/build__tsc.json")), - ( - "build/tsdown", - include_str!("../vendor/rules/build__tsdown.json"), - ), - ( - "build/vite", - include_str!("../vendor/rules/build__vite.json"), - ), - ( - "build/webpack", - include_str!("../vendor/rules/build__webpack.json"), - ), - ("cloud/aws", include_str!("../vendor/rules/cloud__aws.json")), - ("cloud/az", include_str!("../vendor/rules/cloud__az.json")), - ( - "cloud/flyctl", - include_str!("../vendor/rules/cloud__flyctl.json"), - ), - ( - "cloud/gcloud", - include_str!("../vendor/rules/cloud__gcloud.json"), - ), - ("cloud/gh", include_str!("../vendor/rules/cloud__gh.json")), - ( - "cloud/vercel", - include_str!("../vendor/rules/cloud__vercel.json"), - ), - ( - "database/mongosh", - include_str!("../vendor/rules/database__mongosh.json"), - ), - ( - "database/mysql", - include_str!("../vendor/rules/database__mysql.json"), - ), - ( - "database/psql", - include_str!("../vendor/rules/database__psql.json"), - ), - ( - "database/redis-cli", - include_str!("../vendor/rules/database__redis-cli.json"), - ), - ( - "database/sqlite3", - include_str!("../vendor/rules/database__sqlite3.json"), - ), - ( - "devops/docker-build", - include_str!("../vendor/rules/devops__docker-build.json"), - ), - ( - "devops/docker-compose", - include_str!("../vendor/rules/devops__docker-compose.json"), - ), - ( - "devops/docker-images", - include_str!("../vendor/rules/devops__docker-images.json"), - ), - ( - "devops/docker-logs", - include_str!("../vendor/rules/devops__docker-logs.json"), - ), - ( - "devops/docker-ps", - include_str!("../vendor/rules/devops__docker-ps.json"), - ), - ( - "devops/kubectl-describe", - include_str!("../vendor/rules/devops__kubectl-describe.json"), - ), - ( - "devops/kubectl-get", - include_str!("../vendor/rules/devops__kubectl-get.json"), - ), - ( - "devops/kubectl-logs", - include_str!("../vendor/rules/devops__kubectl-logs.json"), - ), - ( - "filesystem/find", - include_str!("../vendor/rules/filesystem__find.json"), - ), - ( - "filesystem/ls", - include_str!("../vendor/rules/filesystem__ls.json"), - ), - ( - "generic/help", - include_str!("../vendor/rules/generic__help.json"), - ), - ( - "git/branch", - include_str!("../vendor/rules/git__branch.json"), - ), - ( - "git/diff-name-only", - include_str!("../vendor/rules/git__diff-name-only.json"), - ), - ( - "git/diff-stat", - include_str!("../vendor/rules/git__diff-stat.json"), - ), - ( - "git/log-oneline", - include_str!("../vendor/rules/git__log-oneline.json"), - ), - ( - "git/remote-v", - include_str!("../vendor/rules/git__remote-v.json"), - ), - ("git/show", include_str!("../vendor/rules/git__show.json")), - ( - "git/stash-list", - include_str!("../vendor/rules/git__stash-list.json"), - ), - ( - "git/status", - include_str!("../vendor/rules/git__status.json"), - ), - ( - "install/bun-install", - include_str!("../vendor/rules/install__bun-install.json"), - ), - ( - "install/npm-install", - include_str!("../vendor/rules/install__npm-install.json"), - ), - ( - "install/pnpm-install", - include_str!("../vendor/rules/install__pnpm-install.json"), - ), - ( - "install/yarn-install", - include_str!("../vendor/rules/install__yarn-install.json"), - ), - ( - "lint/biome", - include_str!("../vendor/rules/lint__biome.json"), - ), - ( - "lint/cargo-clippy", - include_str!("../vendor/rules/lint__cargo-clippy.json"), - ), - ( - "lint/cargo-fmt", - include_str!("../vendor/rules/lint__cargo-fmt.json"), - ), - ( - "lint/eslint", - include_str!("../vendor/rules/lint__eslint.json"), - ), - ( - "lint/oxlint", - include_str!("../vendor/rules/lint__oxlint.json"), - ), - ( - "lint/prettier-check", - include_str!("../vendor/rules/lint__prettier-check.json"), - ), - ( - "media/ffmpeg", - include_str!("../vendor/rules/media__ffmpeg.json"), - ), - ( - "media/mediainfo", - include_str!("../vendor/rules/media__mediainfo.json"), - ), - ( - "network/curl", - include_str!("../vendor/rules/network__curl.json"), - ), - ( - "network/dig", - include_str!("../vendor/rules/network__dig.json"), - ), - ( - "network/nslookup", - include_str!("../vendor/rules/network__nslookup.json"), - ), - ( - "network/ping", - include_str!("../vendor/rules/network__ping.json"), - ), - ( - "network/ssh", - include_str!("../vendor/rules/network__ssh.json"), - ), - ( - "network/traceroute", - include_str!("../vendor/rules/network__traceroute.json"), - ), - ( - "network/wget", - include_str!("../vendor/rules/network__wget.json"), - ), - ( - "observability/free", - include_str!("../vendor/rules/observability__free.json"), - ), - ( - "observability/htop", - include_str!("../vendor/rules/observability__htop.json"), - ), - ( - "observability/iostat", - include_str!("../vendor/rules/observability__iostat.json"), - ), - ( - "observability/top", - include_str!("../vendor/rules/observability__top.json"), - ), - ( - "observability/vmstat", - include_str!("../vendor/rules/observability__vmstat.json"), - ), - ( - "package/apt-install", - include_str!("../vendor/rules/package__apt-install.json"), - ), - ( - "package/apt-upgrade", - include_str!("../vendor/rules/package__apt-upgrade.json"), - ), - ( - "package/brew-install", - include_str!("../vendor/rules/package__brew-install.json"), - ), - ( - "package/brew-upgrade", - include_str!("../vendor/rules/package__brew-upgrade.json"), - ), - ( - "package/dnf-install", - include_str!("../vendor/rules/package__dnf-install.json"), - ), - ( - "package/yum-install", - include_str!("../vendor/rules/package__yum-install.json"), - ), - ( - "search/git-grep", - include_str!("../vendor/rules/search__git-grep.json"), - ), - ( - "search/grep", - include_str!("../vendor/rules/search__grep.json"), - ), - ("search/rg", include_str!("../vendor/rules/search__rg.json")), - ( - "service/journalctl", - include_str!("../vendor/rules/service__journalctl.json"), - ), - ( - "service/launchctl", - include_str!("../vendor/rules/service__launchctl.json"), - ), - ( - "service/lsof", - include_str!("../vendor/rules/service__lsof.json"), - ), - ( - "service/netstat", - include_str!("../vendor/rules/service__netstat.json"), - ), - ( - "service/service", - include_str!("../vendor/rules/service__service.json"), - ), - ( - "service/ss", - include_str!("../vendor/rules/service__ss.json"), - ), - ( - "service/systemctl-status", - include_str!("../vendor/rules/service__systemctl-status.json"), - ), - ("system/df", include_str!("../vendor/rules/system__df.json")), - ("system/du", include_str!("../vendor/rules/system__du.json")), - ( - "system/file", - include_str!("../vendor/rules/system__file.json"), - ), - ("system/ps", include_str!("../vendor/rules/system__ps.json")), - ("task/just", include_str!("../vendor/rules/task__just.json")), - ("task/make", include_str!("../vendor/rules/task__make.json")), - ( - "tests/bun-test", - include_str!("../vendor/rules/tests__bun-test.json"), - ), - ( - "tests/cargo-test", - include_str!("../vendor/rules/tests__cargo-test.json"), - ), - ( - "tests/go-test", - include_str!("../vendor/rules/tests__go-test.json"), - ), - ( - "tests/jest", - include_str!("../vendor/rules/tests__jest.json"), - ), - ( - "tests/mocha", - include_str!("../vendor/rules/tests__mocha.json"), - ), - ( - "tests/npm-test", - include_str!("../vendor/rules/tests__npm-test.json"), - ), - ( - "tests/playwright", - include_str!("../vendor/rules/tests__playwright.json"), - ), - ( - "tests/pnpm-test", - include_str!("../vendor/rules/tests__pnpm-test.json"), - ), - ( - "tests/pytest", - include_str!("../vendor/rules/tests__pytest.json"), - ), - ( - "tests/vitest", - include_str!("../vendor/rules/tests__vitest.json"), - ), - ( - "tests/yarn-test", - include_str!("../vendor/rules/tests__yarn-test.json"), - ), - ( - "transfer/rsync", - include_str!("../vendor/rules/transfer__rsync.json"), - ), - ( - "transfer/scp", - include_str!("../vendor/rules/transfer__scp.json"), - ), - // generic/fallback is always last — the loader sorts it to the tail of the - // compiled rule list so it never shadows a more specific rule. - ( - "generic/fallback", - include_str!("../vendor/rules/generic__fallback.json"), - ), -]; - -#[cfg(test)] -#[path = "builtin_tests.rs"] -mod tests; diff --git a/src/openhuman/tokenjuice/rules/builtin_tests.rs b/src/openhuman/tokenjuice/rules/builtin_tests.rs deleted file mode 100644 index 90ec7b716..000000000 --- a/src/openhuman/tokenjuice/rules/builtin_tests.rs +++ /dev/null @@ -1,185 +0,0 @@ -use super::*; -use crate::openhuman::tokenjuice::rules::compiler::compile_rule; -use crate::openhuman::tokenjuice::types::RuleOrigin; - -/// Load every builtin rule and assert: -/// (a) none fail to parse as `JsonRule` -/// (b) duplicate ids are detected and reported (but the test does not fail) -/// -/// This mirrors the lenient-by-design rule loader: a bad JSON entry is -/// logged but does not crash the engine. -#[test] -fn all_builtins_parse_without_error() { - use crate::openhuman::tokenjuice::types::JsonRule; - use std::collections::HashMap; - - let mut id_count: HashMap> = HashMap::new(); - let mut parse_failures: Vec<(&str, String)> = Vec::new(); - - for (id, json) in BUILTIN_RULE_JSONS { - match serde_json::from_str::(json) { - Ok(rule) => { - id_count.entry(rule.id.clone()).or_default().push(id); - } - Err(e) => { - parse_failures.push((id, e.to_string())); - eprintln!("[tokenjuice/builtin] PARSE FAIL '{}': {}", id, e); - } - } - } - - // Report duplicate ids (non-fatal: last-write wins in the loader anyway) - for (rule_id, ids) in &id_count { - if ids.len() > 1 { - eprintln!( - "[tokenjuice/builtin] DUPLICATE id '{}' in entries: {:?}", - rule_id, ids - ); - } - } - - let duplicates: Vec<_> = id_count - .iter() - .filter(|(_, v)| v.len() > 1) - .map(|(k, _)| k.as_str()) - .collect(); - - assert!( - parse_failures.is_empty(), - "builtin rule parse failures: {:?}", - parse_failures - ); - assert!( - duplicates.is_empty(), - "duplicate builtin rule ids (fix builtin.rs): {:?}", - duplicates - ); -} - -/// Compile all builtins and list any that fail to compile (non-fatal). -/// This ensures the lenient compile path is exercised and gives a clear -/// inventory if any regex is incompatible with the `regex` crate. -#[test] -fn all_builtins_compile() { - use crate::openhuman::tokenjuice::types::JsonRule; - - let mut compile_issues: Vec = Vec::new(); - - for (id, json) in BUILTIN_RULE_JSONS { - let rule: JsonRule = match serde_json::from_str(json) { - Ok(r) => r, - Err(e) => { - compile_issues.push(format!("PARSE '{}': {}", id, e)); - continue; - } - }; - - // compile_rule is lenient: invalid regex is dropped (not panicked) - let compiled = compile_rule(rule, RuleOrigin::Builtin, format!("builtin:{}", id)); - - // For rules that define counters/filters/output_matches, check that - // at least some patterns compiled (unless no patterns were declared). - // We do NOT fail on partial compilation — log only. - let _ = compiled; // compilation itself must not panic - } - - if !compile_issues.is_empty() { - eprintln!( - "[tokenjuice/builtin] {} compile issues (lenient — not failing test):", - compile_issues.len() - ); - for issue in &compile_issues { - eprintln!(" {}", issue); - } - } - - // The test passes as long as compile_rule doesn't panic for any builtin. - // Partial regex failures are logged above but do not fail the suite. -} - -#[test] -fn generic_fallback_is_present() { - let has_fallback = BUILTIN_RULE_JSONS - .iter() - .any(|(id, _)| *id == "generic/fallback"); - assert!( - has_fallback, - "generic/fallback must be in BUILTIN_RULE_JSONS" - ); -} - -#[test] -fn total_builtin_count() { - // Ensure we have the expected number of vendored rules. - // Update this number when new rules are added. - assert_eq!( - BUILTIN_RULE_JSONS.len(), - 100, - "expected 100 builtin rules; update this assertion if the vendor set changes" - ); -} - -// --- exercise the parse-fail and duplicate code paths in-situ --- - -#[test] -fn duplicate_id_reporting_logic_works() { - // Exercise the "ids.len() > 1" and duplicate-filter branches of the - // all_builtins_parse_without_error helper by running the same logic - // on a synthetic set containing a known duplicate. - use crate::openhuman::tokenjuice::types::JsonRule; - use std::collections::HashMap; - - let test_entries: &[(&str, &str)] = &[ - ("rule-a", r#"{"id":"dup","family":"test","match":{}}"#), - ("rule-b", r#"{"id":"dup","family":"test","match":{}}"#), - ("rule-c", r#"{"id":"unique","family":"test","match":{}}"#), - ]; - - let mut id_count: HashMap> = HashMap::new(); - for (entry_id, json) in test_entries { - if let Ok(rule) = serde_json::from_str::(json) { - id_count.entry(rule.id.clone()).or_default().push(entry_id); - } - } - - // Exercise the duplicate-reporting branch - for (rule_id, ids) in &id_count { - if ids.len() > 1 { - // This is the branch normally exercised by all_builtins_parse_without_error - // when duplicates exist. We just log it here. - eprintln!("TEST duplicate '{}' in {:?}", rule_id, ids); - } - } - - let duplicates: Vec<_> = id_count - .iter() - .filter(|(_, v)| v.len() > 1) - .map(|(k, _)| k.as_str()) - .collect(); - assert_eq!(duplicates.len(), 1, "expected exactly one duplicate"); - assert_eq!(duplicates[0], "dup"); -} - -#[test] -fn compile_issues_reporting_logic_works() { - // Exercise the compile_issues error-reporting branch from all_builtins_compile - // by simulating the path with a known-bad JSON entry. - let mut compile_issues: Vec = Vec::new(); - - // Simulate a parse failure (bad JSON) - let bad_json = "{ not valid json at all }"; - if let Err(e) = serde_json::from_str::(bad_json) - { - compile_issues.push(format!("PARSE 'bad-entry': {}", e)); - } - - // Now exercise the reporting branch - assert!(!compile_issues.is_empty()); - eprintln!( - "[test] {} compile issues (expected in this test):", - compile_issues.len() - ); - for issue in &compile_issues { - eprintln!(" {}", issue); - } -} diff --git a/src/openhuman/tokenjuice/rules/compiler.rs b/src/openhuman/tokenjuice/rules/compiler.rs deleted file mode 100644 index 3c2778e7c..000000000 --- a/src/openhuman/tokenjuice/rules/compiler.rs +++ /dev/null @@ -1,310 +0,0 @@ -//! Rule compilation: converts a `JsonRule` descriptor into a `CompiledRule` -//! with pre-built `regex::Regex` instances. -//! -//! Invalid regex patterns produce a non-fatal diagnostic log and are silently -//! dropped so a bad user rule does not crash the engine. - -use crate::openhuman::tokenjuice::types::{ - CompiledCounter, CompiledOutputMatch, CompiledParts, CompiledRule, JsonRule, RuleOrigin, -}; - -// --------------------------------------------------------------------------- -// Regex helpers -// --------------------------------------------------------------------------- - -/// Build regex flags ensuring `u` (Unicode) is always present. -/// -/// Upstream uses `new RegExp(pattern, mergeRegexFlags(flags))` where `u` is -/// always prepended. In Rust's `regex` crate there is no separate `u` flag — -/// Unicode is on by default — so we translate only `i` (case-insensitive) and -/// `m` (multiline). -fn build_regex(pattern: &str, flags: Option<&str>) -> Option { - let case_insensitive = flags.map(|f| f.contains('i')).unwrap_or(false); - let multiline = flags.map(|f| f.contains('m')).unwrap_or(false); - - // Build pattern with inline flags - let prefix = match (case_insensitive, multiline) { - (true, true) => "(?im)", - (true, false) => "(?i)", - (false, true) => "(?m)", - (false, false) => "", - }; - let full = format!("{}{}", prefix, pattern); - - match regex::Regex::new(&full) { - Ok(re) => Some(re), - Err(err) => { - log::debug!( - "[tokenjuice] rule compiler: invalid regex '{}' (flags={:?}): {}", - pattern, - flags, - err - ); - None - } - } -} - -// --------------------------------------------------------------------------- -// compile_rule -// --------------------------------------------------------------------------- - -/// Compile a `JsonRule` into a `CompiledRule`. -/// -/// `path` is either a filesystem path or `"builtin:"` for embedded rules. -pub fn compile_rule(rule: JsonRule, source: RuleOrigin, path: String) -> CompiledRule { - log::debug!( - "[tokenjuice] compiling rule '{}' from {:?} path={}", - rule.id, - source, - path - ); - - let skip_patterns: Vec = rule - .filters - .as_ref() - .and_then(|f| f.skip_patterns.as_ref()) - .map(|pats| pats.iter().filter_map(|p| build_regex(p, None)).collect()) - .unwrap_or_default(); - - let keep_patterns: Vec = rule - .filters - .as_ref() - .and_then(|f| f.keep_patterns.as_ref()) - .map(|pats| pats.iter().filter_map(|p| build_regex(p, None)).collect()) - .unwrap_or_default(); - - let counters: Vec = rule - .counters - .as_ref() - .map(|counters| { - counters - .iter() - .filter_map(|c| { - build_regex(&c.pattern, c.flags.as_deref()).map(|re| CompiledCounter { - name: c.name.clone(), - pattern: re, - }) - }) - .collect() - }) - .unwrap_or_default(); - - let output_matches: Vec = rule - .match_output - .as_ref() - .map(|entries| { - entries - .iter() - .filter_map(|entry| { - build_regex(&entry.pattern, entry.flags.as_deref()).map(|re| { - CompiledOutputMatch { - pattern: re, - message: entry.message.clone(), - } - }) - }) - .collect() - }) - .unwrap_or_default(); - - CompiledRule { - compiled: CompiledParts { - skip_patterns, - keep_patterns, - counters, - output_matches, - }, - rule, - source, - path, - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::openhuman::tokenjuice::types::{JsonRule, RuleMatch}; - - fn minimal_rule(id: &str) -> JsonRule { - JsonRule { - id: id.to_owned(), - family: "test".to_owned(), - description: None, - priority: None, - on_empty: None, - match_output: None, - counter_source: None, - r#match: RuleMatch::default(), - filters: None, - transforms: None, - summarize: None, - counters: None, - failure: None, - } - } - - #[test] - fn compiles_minimal_rule() { - let rule = minimal_rule("test/rule"); - let compiled = compile_rule(rule, RuleOrigin::Builtin, "builtin:test/rule".to_owned()); - assert_eq!(compiled.rule.id, "test/rule"); - assert!(compiled.compiled.skip_patterns.is_empty()); - } - - #[test] - fn invalid_regex_is_dropped_not_panicked() { - use crate::openhuman::tokenjuice::types::{RuleCounter, RuleFilters}; - let mut rule = minimal_rule("test/bad"); - rule.filters = Some(RuleFilters { - skip_patterns: Some(vec!["[invalid".to_owned()]), - keep_patterns: None, - }); - rule.counters = Some(vec![RuleCounter { - name: "bad counter".to_owned(), - pattern: "(unclosed".to_owned(), - flags: None, - }]); - let compiled = compile_rule(rule, RuleOrigin::Builtin, "builtin:test/bad".to_owned()); - // Both should be silently dropped - assert!(compiled.compiled.skip_patterns.is_empty()); - assert!(compiled.compiled.counters.is_empty()); - } - - #[test] - fn case_insensitive_flag() { - use crate::openhuman::tokenjuice::types::RuleCounter; - let mut rule = minimal_rule("test/ci"); - rule.counters = Some(vec![RuleCounter { - name: "error".to_owned(), - pattern: "error".to_owned(), - flags: Some("i".to_owned()), - }]); - let compiled = compile_rule(rule, RuleOrigin::Builtin, "builtin:test/ci".to_owned()); - assert_eq!(compiled.compiled.counters.len(), 1); - assert!(compiled.compiled.counters[0].pattern.is_match("ERROR")); - assert!(compiled.compiled.counters[0].pattern.is_match("error")); - } - - #[test] - fn multiline_flag_works() { - use crate::openhuman::tokenjuice::types::RuleCounter; - let mut rule = minimal_rule("test/ml"); - rule.counters = Some(vec![RuleCounter { - name: "line_start".to_owned(), - pattern: "^foo".to_owned(), - flags: Some("m".to_owned()), - }]); - let compiled = compile_rule(rule, RuleOrigin::Builtin, "builtin:test/ml".to_owned()); - assert_eq!(compiled.compiled.counters.len(), 1); - // With multiline, ^ matches start of each line - assert!(compiled.compiled.counters[0] - .pattern - .is_match("bar\nfoo baz")); - } - - #[test] - fn case_insensitive_and_multiline_combined() { - use crate::openhuman::tokenjuice::types::RuleCounter; - let mut rule = minimal_rule("test/im"); - rule.counters = Some(vec![RuleCounter { - name: "start".to_owned(), - pattern: "^ERROR".to_owned(), - flags: Some("im".to_owned()), - }]); - let compiled = compile_rule(rule, RuleOrigin::Builtin, "builtin:test/im".to_owned()); - assert_eq!(compiled.compiled.counters.len(), 1); - assert!(compiled.compiled.counters[0] - .pattern - .is_match("prefix\nerror line")); - } - - #[test] - fn invalid_regex_in_keep_patterns_is_dropped() { - use crate::openhuman::tokenjuice::types::RuleFilters; - let mut rule = minimal_rule("test/bad-keep"); - rule.filters = Some(RuleFilters { - skip_patterns: None, - keep_patterns: Some(vec!["[invalid".to_owned()]), - }); - let compiled = compile_rule( - rule, - RuleOrigin::Builtin, - "builtin:test/bad-keep".to_owned(), - ); - assert!(compiled.compiled.keep_patterns.is_empty()); - } - - #[test] - fn invalid_regex_in_match_output_is_dropped() { - use crate::openhuman::tokenjuice::types::RuleOutputMatch; - let mut rule = minimal_rule("test/bad-output"); - rule.match_output = Some(vec![RuleOutputMatch { - pattern: "(unclosed".to_owned(), - message: "should not appear".to_owned(), - flags: None, - }]); - let compiled = compile_rule( - rule, - RuleOrigin::Builtin, - "builtin:test/bad-output".to_owned(), - ); - assert!(compiled.compiled.output_matches.is_empty()); - } - - #[test] - fn valid_output_match_compiles() { - use crate::openhuman::tokenjuice::types::RuleOutputMatch; - let mut rule = minimal_rule("test/good-output"); - rule.match_output = Some(vec![RuleOutputMatch { - pattern: "nothing to commit".to_owned(), - message: "Clean!".to_owned(), - flags: None, - }]); - let compiled = compile_rule( - rule, - RuleOrigin::Builtin, - "builtin:test/good-output".to_owned(), - ); - assert_eq!(compiled.compiled.output_matches.len(), 1); - assert!(compiled.compiled.output_matches[0] - .pattern - .is_match("nothing to commit, working tree clean")); - assert_eq!(compiled.compiled.output_matches[0].message, "Clean!"); - } - - #[test] - fn output_match_with_case_insensitive_flag() { - use crate::openhuman::tokenjuice::types::RuleOutputMatch; - let mut rule = minimal_rule("test/output-ci"); - rule.match_output = Some(vec![RuleOutputMatch { - pattern: "success".to_owned(), - message: "Done".to_owned(), - flags: Some("i".to_owned()), - }]); - let compiled = compile_rule( - rule, - RuleOrigin::Builtin, - "builtin:test/output-ci".to_owned(), - ); - assert_eq!(compiled.compiled.output_matches.len(), 1); - assert!(compiled.compiled.output_matches[0] - .pattern - .is_match("SUCCESS")); - } - - #[test] - fn rule_source_and_path_preserved() { - let rule = minimal_rule("test/path"); - let compiled = compile_rule( - rule, - RuleOrigin::User, - "/home/user/.config/tokenjuice/rules/test.json".to_owned(), - ); - assert_eq!(compiled.source, RuleOrigin::User); - assert_eq!( - compiled.path, - "/home/user/.config/tokenjuice/rules/test.json" - ); - } -} diff --git a/src/openhuman/tokenjuice/rules/loader.rs b/src/openhuman/tokenjuice/rules/loader.rs deleted file mode 100644 index f080636e6..000000000 --- a/src/openhuman/tokenjuice/rules/loader.rs +++ /dev/null @@ -1,273 +0,0 @@ -//! Three-layer rule loading: builtin → user → project. -//! -//! Port of `src/core/rules.ts` `loadRules()` logic. -//! -//! Layer order (lower priority → higher priority): -//! 1. builtin (embedded via `include_str!`) -//! 2. user (`~/.config/tokenjuice/rules/`) -//! 3. project (`/.tokenjuice/rules/`) -//! -//! When two layers define the same `id`, the higher-priority layer wins -//! (project > user > builtin). The `generic/fallback` rule is always sorted -//! last in the final list. - -use super::{builtin::BUILTIN_RULE_JSONS, compiler::compile_rule}; -use crate::openhuman::tokenjuice::types::{CompiledRule, JsonRule, RuleOrigin}; -use std::path::{Path, PathBuf}; - -// --------------------------------------------------------------------------- -// Options -// --------------------------------------------------------------------------- - -/// Options for `load_rules`. -#[derive(Debug, Default, Clone)] -pub struct LoadRuleOptions { - /// Working directory for project-layer discovery. Defaults to the process - /// current directory. - pub cwd: Option, - /// Override the user-layer directory (default: `~/.config/tokenjuice/rules`). - pub user_rules_dir: Option, - /// Override the project-layer directory (default: `/.tokenjuice/rules`). - pub project_rules_dir: Option, - /// Skip user-layer rules. - pub exclude_user: bool, - /// Skip project-layer rules. - pub exclude_project: bool, -} - -// --------------------------------------------------------------------------- -// Layer path helpers -// --------------------------------------------------------------------------- - -fn user_rules_root(custom: Option<&Path>) -> PathBuf { - if let Some(p) = custom { - return p.to_owned(); - } - dirs::home_dir() - .unwrap_or_else(|| PathBuf::from(".")) - .join(".config") - .join("tokenjuice") - .join("rules") -} - -fn project_rules_root(cwd: Option<&Path>, custom: Option<&Path>) -> PathBuf { - if let Some(p) = custom { - return p.to_owned(); - } - cwd.unwrap_or_else(|| Path::new(".")) - .join(".tokenjuice") - .join("rules") -} - -// --------------------------------------------------------------------------- -// Builtin layer -// --------------------------------------------------------------------------- - -fn load_builtin_descriptors() -> Vec<(RuleOrigin, String, JsonRule)> { - BUILTIN_RULE_JSONS - .iter() - .filter_map(|(id, json)| match serde_json::from_str::(json) { - Ok(rule) => { - log::debug!("[tokenjuice] loaded builtin rule '{}'", id); - Some((RuleOrigin::Builtin, format!("builtin:{}", id), rule)) - } - Err(err) => { - log::debug!( - "[tokenjuice] failed to parse builtin rule '{}': {}", - id, - err - ); - None - } - }) - .collect() -} - -// --------------------------------------------------------------------------- -// Disk layer -// --------------------------------------------------------------------------- - -/// Recursively walk `root` and return all `.json` files that are not -/// `.schema.json` or `.fixture.json`. -fn list_rule_files(root: &Path) -> Vec { - if !root.is_dir() { - return Vec::new(); - } - let mut out = Vec::new(); - walk_dir(root, &mut out); - out.sort(); - out -} - -fn walk_dir(dir: &Path, out: &mut Vec) { - let entries = match std::fs::read_dir(dir) { - Ok(e) => e, - Err(err) => { - log::debug!("[tokenjuice] read_dir failed at {}: {}", dir.display(), err); - return; - } - }; - let mut names: Vec<_> = entries.filter_map(|e| e.ok()).collect(); - names.sort_by_key(|e| e.file_name()); - - for entry in names { - let path = entry.path(); - let ft = match entry.file_type() { - Ok(ft) => ft, - Err(err) => { - log::debug!( - "[tokenjuice] file_type failed at {}: {}", - path.display(), - err - ); - continue; - } - }; - if ft.is_symlink() { - continue; - } - if ft.is_dir() { - walk_dir(&path, out); - } else if ft.is_file() { - let name = entry.file_name(); - let name_str = name.to_string_lossy(); - if name_str.ends_with(".json") - && !name_str.ends_with(".schema.json") - && !name_str.ends_with(".fixture.json") - { - out.push(path); - } - } - } -} - -fn load_disk_descriptors(root: &Path, source: RuleOrigin) -> Vec<(RuleOrigin, String, JsonRule)> { - let files = list_rule_files(root); - files - .into_iter() - .filter_map(|path| { - let json = match std::fs::read_to_string(&path) { - Ok(s) => s, - Err(err) => { - log::debug!( - "[tokenjuice] read_to_string failed for {:?} rule at {}: {}", - source, - path.display(), - err - ); - return None; - } - }; - match serde_json::from_str::(&json) { - Ok(rule) => { - log::debug!( - "[tokenjuice] loaded {:?} rule '{}' from {}", - source, - rule.id, - path.display() - ); - Some((source.clone(), path.display().to_string(), rule)) - } - Err(err) => { - log::debug!( - "[tokenjuice] failed to parse {:?} rule at {}: {}", - source, - path.display(), - err - ); - None - } - } - }) - .collect() -} - -// --------------------------------------------------------------------------- -// Overlay & sort -// --------------------------------------------------------------------------- - -/// Merge descriptors by `rule.id`: later entries win (project > user > builtin). -fn overlay_and_sort(descriptors: Vec<(RuleOrigin, String, JsonRule)>) -> Vec { - // Use an IndexMap-like approach via a Vec to preserve last-write semantics - // while keeping insertion order (needed for stable sort). - let mut by_id: std::collections::HashMap = - std::collections::HashMap::new(); - - for (source, path, rule) in descriptors { - by_id.insert(rule.id.clone(), (source, path, rule)); - } - - let mut compiled: Vec = by_id - .into_values() - .map(|(source, path, rule)| compile_rule(rule, source, path)) - .collect(); - - // Sort alphabetically, `generic/fallback` last - compiled.sort_by(|a, b| { - let a_fb = a.rule.id == "generic/fallback"; - let b_fb = b.rule.id == "generic/fallback"; - match (a_fb, b_fb) { - (true, false) => std::cmp::Ordering::Greater, - (false, true) => std::cmp::Ordering::Less, - _ => a.rule.id.cmp(&b.rule.id), - } - }); - - log::debug!( - "[tokenjuice] overlay resolved {} rules (fallback last)", - compiled.len() - ); - - compiled -} - -// --------------------------------------------------------------------------- -// Public API -// --------------------------------------------------------------------------- - -/// Load and compile all rules from the three-layer overlay. -/// -/// Layers are resolved in priority order (builtin < user < project) so that -/// a project rule with the same `id` overrides a builtin rule. -pub fn load_rules(opts: &LoadRuleOptions) -> Vec { - let mut descriptors: Vec<(RuleOrigin, String, JsonRule)> = Vec::new(); - - // 1. Builtin (lowest priority) - descriptors.extend(load_builtin_descriptors()); - - // 2. User layer - if !opts.exclude_user { - let user_root = user_rules_root(opts.user_rules_dir.as_deref()); - log::debug!( - "[tokenjuice] loading user rules from {}", - user_root.display() - ); - descriptors.extend(load_disk_descriptors(&user_root, RuleOrigin::User)); - } - - // 3. Project layer (highest priority) - if !opts.exclude_project { - let project_root = - project_rules_root(opts.cwd.as_deref(), opts.project_rules_dir.as_deref()); - log::debug!( - "[tokenjuice] loading project rules from {}", - project_root.display() - ); - descriptors.extend(load_disk_descriptors(&project_root, RuleOrigin::Project)); - } - - overlay_and_sort(descriptors) -} - -/// Load only the builtin rules (no disk I/O). -pub fn load_builtin_rules() -> Vec { - load_rules(&LoadRuleOptions { - exclude_user: true, - exclude_project: true, - ..Default::default() - }) -} - -#[cfg(test)] -#[path = "loader_tests.rs"] -mod tests; diff --git a/src/openhuman/tokenjuice/rules/loader_tests.rs b/src/openhuman/tokenjuice/rules/loader_tests.rs deleted file mode 100644 index 8e41daedc..000000000 --- a/src/openhuman/tokenjuice/rules/loader_tests.rs +++ /dev/null @@ -1,273 +0,0 @@ -use super::*; - -#[test] -fn builtin_rules_load_successfully() { - let rules = load_builtin_rules(); - assert!(!rules.is_empty(), "at least one built-in rule expected"); - let ids: Vec<&str> = rules.iter().map(|r| r.rule.id.as_str()).collect(); - assert!( - ids.contains(&"generic/fallback"), - "generic/fallback must be present" - ); -} - -#[test] -fn fallback_rule_is_last() { - let rules = load_builtin_rules(); - let last = rules.last().expect("non-empty list"); - assert_eq!(last.rule.id, "generic/fallback"); -} - -#[test] -fn project_layer_overrides_builtin() { - // Write a temporary project rules dir with a modified fallback rule - let dir = tempfile::tempdir().expect("tempdir"); - let override_json = r#"{ - "id": "generic/fallback", - "family": "override-family", - "description": "overridden", - "match": {} - }"#; - std::fs::write(dir.path().join("fallback.json"), override_json).unwrap(); - - let opts = LoadRuleOptions { - project_rules_dir: Some(dir.path().to_owned()), - exclude_user: true, - ..Default::default() - }; - let rules = load_rules(&opts); - let fb = rules - .iter() - .find(|r| r.rule.id == "generic/fallback") - .expect("fallback rule"); - assert_eq!(fb.rule.family, "override-family"); - assert_eq!(fb.source, RuleOrigin::Project); -} - -#[test] -fn rules_sorted_alphabetically_fallback_last() { - let rules = load_builtin_rules(); - let non_fb: Vec<&str> = rules - .iter() - .filter(|r| r.rule.id != "generic/fallback") - .map(|r| r.rule.id.as_str()) - .collect(); - let mut sorted = non_fb.clone(); - sorted.sort(); - assert_eq!(non_fb, sorted, "rules should be alphabetically sorted"); -} - -// --- load_rules with disk layers --- - -#[test] -fn user_layer_overrides_builtin() { - let dir = tempfile::tempdir().expect("tempdir"); - let override_json = r#"{ - "id": "git/status", - "family": "user-overridden", - "description": "user override", - "match": {} - }"#; - std::fs::write(dir.path().join("git_status.json"), override_json).unwrap(); - - let opts = LoadRuleOptions { - user_rules_dir: Some(dir.path().to_owned()), - exclude_project: true, - ..Default::default() - }; - let rules = load_rules(&opts); - let gs = rules - .iter() - .find(|r| r.rule.id == "git/status") - .expect("git/status rule"); - assert_eq!(gs.rule.family, "user-overridden"); - assert_eq!(gs.source, RuleOrigin::User); -} - -#[test] -fn invalid_json_files_are_skipped() { - let dir = tempfile::tempdir().expect("tempdir"); - // Write an invalid JSON file - std::fs::write(dir.path().join("bad.json"), "{ this is not valid json }").unwrap(); - // Write a valid rule - let valid_json = r#"{ - "id": "test/valid", - "family": "test", - "match": {} - }"#; - std::fs::write(dir.path().join("valid.json"), valid_json).unwrap(); - - let opts = LoadRuleOptions { - project_rules_dir: Some(dir.path().to_owned()), - exclude_user: true, - ..Default::default() - }; - let rules = load_rules(&opts); - // Valid rule should be loaded, invalid should be silently skipped - assert!(rules.iter().any(|r| r.rule.id == "test/valid")); -} - -#[test] -fn schema_and_fixture_json_files_are_skipped() { - let dir = tempfile::tempdir().expect("tempdir"); - // These should be ignored by list_rule_files - std::fs::write( - dir.path().join("rules.schema.json"), - r#"{"id":"should-skip","family":"skip","match":{}}"#, - ) - .unwrap(); - std::fs::write( - dir.path().join("example.fixture.json"), - r#"{"id":"should-skip2","family":"skip","match":{}}"#, - ) - .unwrap(); - // A normal rule that should be loaded - std::fs::write( - dir.path().join("normal.json"), - r#"{"id":"test/normal","family":"test","match":{}}"#, - ) - .unwrap(); - - let opts = LoadRuleOptions { - project_rules_dir: Some(dir.path().to_owned()), - exclude_user: true, - ..Default::default() - }; - let rules = load_rules(&opts); - // schema/fixture files should not be loaded - assert!(!rules.iter().any(|r| r.rule.id == "should-skip")); - assert!(!rules.iter().any(|r| r.rule.id == "should-skip2")); - // Normal rule should be there - assert!(rules.iter().any(|r| r.rule.id == "test/normal")); -} - -#[test] -fn non_existent_dir_loads_only_builtins() { - let opts = LoadRuleOptions { - user_rules_dir: Some(std::path::PathBuf::from( - "/nonexistent/path/that/does/not/exist", - )), - project_rules_dir: Some(std::path::PathBuf::from("/another/nonexistent/path/rules")), - ..Default::default() - }; - let rules = load_rules(&opts); - // Should still have builtins - assert!(rules.iter().any(|r| r.rule.id == "generic/fallback")); - assert!(!rules.is_empty()); -} - -#[test] -fn exclude_user_skips_user_layer() { - let user_dir = tempfile::tempdir().expect("tempdir"); - let override_json = r#"{"id":"git/status","family":"should-not-see","match":{}}"#; - std::fs::write(user_dir.path().join("override.json"), override_json).unwrap(); - - let opts = LoadRuleOptions { - user_rules_dir: Some(user_dir.path().to_owned()), - exclude_user: true, - exclude_project: true, - ..Default::default() - }; - let rules = load_rules(&opts); - // user override should NOT be present — original builtin should remain - let gs = rules - .iter() - .find(|r| r.rule.id == "git/status") - .expect("git/status"); - assert_ne!(gs.rule.family, "should-not-see"); - assert_eq!(gs.source, RuleOrigin::Builtin); -} - -#[test] -fn project_layer_wins_over_user_layer() { - let user_dir = tempfile::tempdir().expect("tempdir"); - let project_dir = tempfile::tempdir().expect("tempdir"); - - std::fs::write( - user_dir.path().join("rule.json"), - r#"{"id":"git/status","family":"user-family","match":{}}"#, - ) - .unwrap(); - std::fs::write( - project_dir.path().join("rule.json"), - r#"{"id":"git/status","family":"project-family","match":{}}"#, - ) - .unwrap(); - - let opts = LoadRuleOptions { - user_rules_dir: Some(user_dir.path().to_owned()), - project_rules_dir: Some(project_dir.path().to_owned()), - ..Default::default() - }; - let rules = load_rules(&opts); - let gs = rules - .iter() - .find(|r| r.rule.id == "git/status") - .expect("git/status"); - // Project wins over user - assert_eq!(gs.rule.family, "project-family"); - assert_eq!(gs.source, RuleOrigin::Project); -} - -#[test] -fn subdirectory_rules_are_discovered() { - let dir = tempfile::tempdir().expect("tempdir"); - let subdir = dir.path().join("git"); - std::fs::create_dir_all(&subdir).unwrap(); - std::fs::write( - subdir.join("my_rule.json"), - r#"{"id":"test/subdir-rule","family":"test","match":{}}"#, - ) - .unwrap(); - - let opts = LoadRuleOptions { - project_rules_dir: Some(dir.path().to_owned()), - exclude_user: true, - ..Default::default() - }; - let rules = load_rules(&opts); - assert!( - rules.iter().any(|r| r.rule.id == "test/subdir-rule"), - "subdirectory rule should be discovered" - ); -} - -#[test] -fn duplicate_id_last_write_wins() { - let dir = tempfile::tempdir().expect("tempdir"); - // Same id twice in different files — last-write (by HashMap) wins - std::fs::write( - dir.path().join("a_rule.json"), - r#"{"id":"test/dup","family":"first","match":{}}"#, - ) - .unwrap(); - std::fs::write( - dir.path().join("b_rule.json"), - r#"{"id":"test/dup","family":"second","match":{}}"#, - ) - .unwrap(); - - let opts = LoadRuleOptions { - project_rules_dir: Some(dir.path().to_owned()), - exclude_user: true, - ..Default::default() - }; - let rules = load_rules(&opts); - let dups: Vec<_> = rules.iter().filter(|r| r.rule.id == "test/dup").collect(); - // There should be exactly one (deduped) - assert_eq!(dups.len(), 1, "duplicate id should be deduplicated"); -} - -#[test] -fn default_user_rules_dir_is_home_based() { - // Just exercise the path: if home doesn't exist, should still not panic - let path = super::user_rules_root(None); - // Should end in .config/tokenjuice/rules - assert!(path.to_string_lossy().contains("tokenjuice")); -} - -#[test] -fn default_project_rules_dir_is_cwd_based() { - let path = super::project_rules_root(None, None); - assert!(path.to_string_lossy().contains(".tokenjuice")); -} diff --git a/src/openhuman/tokenjuice/rules/mod.rs b/src/openhuman/tokenjuice/rules/mod.rs deleted file mode 100644 index f9164947e..000000000 --- a/src/openhuman/tokenjuice/rules/mod.rs +++ /dev/null @@ -1,8 +0,0 @@ -//! Rule loading, compilation, and the built-in rule set. - -pub mod builtin; -pub mod compiler; -pub mod loader; - -pub use compiler::compile_rule; -pub use loader::{load_builtin_rules, load_rules, LoadRuleOptions}; diff --git a/src/openhuman/tokenjuice/tests/fixtures/cargo_test_failure.fixture.json b/src/openhuman/tokenjuice/tests/fixtures/cargo_test_failure.fixture.json deleted file mode 100644 index 6a53c911b..000000000 --- a/src/openhuman/tokenjuice/tests/fixtures/cargo_test_failure.fixture.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "description": "cargo test failure: exit code + facts header + preserved output", - "input": { - "toolName": "exec", - "argv": ["cargo", "test"], - "exitCode": 1, - "stdout": " Compiling mylib v0.1.0\n Finished test [unoptimized + debuginfo] target(s) in 2.50s\n Running unittests src/lib.rs\nrunning 3 tests\ntest tests::test_a ... ok\ntest tests::test_b ... FAILED\ntest tests::test_c ... ok\n\nfailures:\n\n---- tests::test_b stdout ----\nthread 'tests::test_b' panicked at 'assertion failed', src/lib.rs:42:5\n\nfailures:\n tests::test_b\n\ntest result: FAILED. 2 passed; 1 failed; 0 ignored\n" - }, - "expectedOutput": "exit 1\n2 failed tests, 2 passed tests\nrunning 3 tests\ntest tests::test_a ... ok\ntest tests::test_b ... FAILED\ntest tests::test_c ... ok\n\nfailures:\n\n---- tests::test_b stdout ----\nthread 'tests::test_b' panicked at 'assertion failed', src/lib.rs:42:5\n\nfailures:\n tests::test_b\n\ntest result: FAILED. 2 passed; 1 failed; 0 ignored" -} diff --git a/src/openhuman/tokenjuice/tests/fixtures/fallback_long_output.fixture.json b/src/openhuman/tokenjuice/tests/fixtures/fallback_long_output.fixture.json deleted file mode 100644 index 9d5101ea3..000000000 --- a/src/openhuman/tokenjuice/tests/fixtures/fallback_long_output.fixture.json +++ /dev/null @@ -1,9 +0,0 @@ -{ - "description": "Long generic output (20 lines) gets head=8 tail=8 summarised by fallback rule", - "input": { - "toolName": "bash", - "argv": ["some_tool"], - "stdout": "line 1\nline 2\nline 3\nline 4\nline 5\nline 6\nline 7\nline 8\nline 9\nline 10\nline 11\nline 12\nline 13\nline 14\nline 15\nline 16\nline 17\nline 18\nline 19\nline 20" - }, - "expectedOutput": "line 1\nline 2\nline 3\nline 4\nline 5\nline 6\nline 7\nline 8\n... 4 lines omitted ...\nline 13\nline 14\nline 15\nline 16\nline 17\nline 18\nline 19\nline 20" -} diff --git a/src/openhuman/tokenjuice/tests/fixtures/git_status_modified.fixture.json b/src/openhuman/tokenjuice/tests/fixtures/git_status_modified.fixture.json deleted file mode 100644 index 3c25367d6..000000000 --- a/src/openhuman/tokenjuice/tests/fixtures/git_status_modified.fixture.json +++ /dev/null @@ -1,9 +0,0 @@ -{ - "description": "git status with a modified file rewrites to compact M: notation; hint lines are preserved when indented (Rust port behavior)", - "input": { - "toolName": "bash", - "argv": ["git", "status"], - "stdout": "On branch main\n\nChanges not staged for commit:\n\tmodified: src/foo.rs\n\nno changes added to commit (use \"git add\" and/or \"git commit -a\")\n" - }, - "expectedOutput": "Changes not staged:\nM: src/foo.rs" -} diff --git a/src/openhuman/tokenjuice/text/ansi.rs b/src/openhuman/tokenjuice/text/ansi.rs deleted file mode 100644 index da7d692c2..000000000 --- a/src/openhuman/tokenjuice/text/ansi.rs +++ /dev/null @@ -1,87 +0,0 @@ -//! ANSI / VT escape-sequence stripping. -//! -//! Port of `src/core/text.ts` strip logic. - -use once_cell::sync::Lazy; -use regex::Regex; - -// CSI: ESC [ … final-byte -static ANSI_CSI: Lazy = - Lazy::new(|| Regex::new(r"\x1b\[[0-?]*[ -/]*[@-~]").expect("ansi csi regex")); - -// OSC: ESC ] … BEL or ESC backslash -static ANSI_OSC: Lazy = - Lazy::new(|| Regex::new(r"\x1b\][^\x07\x1b]*(?:\x07|\x1b\\)").expect("ansi osc regex")); - -// Incomplete CSI at end of string -static ANSI_CSI_INCOMPLETE: Lazy = - Lazy::new(|| Regex::new(r"\x1b\[[0-?]*[ -/]*$").expect("ansi csi incomplete regex")); - -// Incomplete OSC at end of string -static ANSI_OSC_INCOMPLETE: Lazy = - Lazy::new(|| Regex::new(r"\x1b\][^\x07\x1b]*$").expect("ansi osc incomplete regex")); - -// Single-char escapes: ESC followed by @-_ -static ANSI_SINGLE: Lazy = - Lazy::new(|| Regex::new(r"\x1b[@-_]").expect("ansi single regex")); - -/// Strip all ANSI/VT escape sequences from `text`. -pub fn strip_ansi(text: &str) -> String { - let input_len = text.len(); - let s = ANSI_OSC.replace_all(text, ""); - let s = ANSI_CSI.replace_all(&s, ""); - let s = ANSI_OSC_INCOMPLETE.replace_all(&s, ""); - let s = ANSI_CSI_INCOMPLETE.replace_all(&s, ""); - let s = ANSI_SINGLE.replace_all(&s, ""); - // Remove any lone ESC bytes that slipped through - let out = s.replace('\x1b', ""); - log::trace!( - "[tokenjuice] strip_ansi in_len={} out_len={}", - input_len, - out.len() - ); - out -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn strips_csi_colour() { - assert_eq!(strip_ansi("\x1b[31mred\x1b[0m"), "red"); - } - - #[test] - fn strips_osc() { - // OSC 8 hyperlink terminated with BEL - assert_eq!(strip_ansi("\x1b]8;;http://x\x07link\x1b]8;;\x07"), "link"); - } - - #[test] - fn strips_incomplete_csi_at_end() { - assert_eq!(strip_ansi("hello\x1b[1"), "hello"); - } - - #[test] - fn strips_csi_with_letter_terminator() { - // ESC [ b — `[` starts a CSI sequence, `b` is the final byte → stripped - assert_eq!(strip_ansi("a\x1b[b"), "a"); - } - - #[test] - fn strips_single_escape_fe_range() { - // ESC N — falls in the @-_ range used by single-char escape sequences - assert_eq!(strip_ansi("a\x1bNb"), "ab"); - } - - #[test] - fn passthrough_plain() { - assert_eq!(strip_ansi("plain text"), "plain text"); - } - - #[test] - fn strips_lone_esc() { - assert_eq!(strip_ansi("a\x1bb"), "ab"); - } -} diff --git a/src/openhuman/tokenjuice/text/mod.rs b/src/openhuman/tokenjuice/text/mod.rs deleted file mode 100644 index 8292b4331..000000000 --- a/src/openhuman/tokenjuice/text/mod.rs +++ /dev/null @@ -1,12 +0,0 @@ -//! Text-processing utilities for the TokenJuice engine. - -pub mod ansi; -pub mod process; -pub mod width; - -pub use ansi::strip_ansi; -pub use process::{ - clamp_text, clamp_text_middle, dedupe_adjacent, head_tail, normalize_lines, pluralize, - trim_empty_edges, -}; -pub use width::{count_terminal_cells, count_text_chars, graphemes}; diff --git a/src/openhuman/tokenjuice/text/process.rs b/src/openhuman/tokenjuice/text/process.rs deleted file mode 100644 index 481d4c09b..000000000 --- a/src/openhuman/tokenjuice/text/process.rs +++ /dev/null @@ -1,393 +0,0 @@ -//! Line-level text processing utilities. -//! -//! Port of the processing functions in `src/core/text.ts`. - -use super::width::count_text_chars; -use unicode_segmentation::UnicodeSegmentation; - -const TRUNCATION_SUFFIX: &str = "\n... truncated ..."; -const MIDDLE_TRUNCATION_MARKER: &str = "\n... omitted ...\n"; - -// --------------------------------------------------------------------------- -// Line normalization -// --------------------------------------------------------------------------- - -/// Split text into lines, normalising CRLF and stripping trailing whitespace -/// per line (mirrors `normalizeLines` in TS). -pub fn normalize_lines(text: &str) -> Vec { - text.replace("\r\n", "\n") - .split('\n') - .map(|line| line.trim_end().to_owned()) - .collect() -} - -// --------------------------------------------------------------------------- -// Edge trimming -// --------------------------------------------------------------------------- - -/// Remove empty lines from the start and end of a line slice. -pub fn trim_empty_edges(lines: &[String]) -> Vec { - let start = lines - .iter() - .position(|l| !l.trim().is_empty()) - .unwrap_or(lines.len()); - let end = lines - .iter() - .rposition(|l| !l.trim().is_empty()) - .map(|i| i + 1) - .unwrap_or(0); - if start >= end { - return Vec::new(); - } - lines[start..end].to_vec() -} - -// --------------------------------------------------------------------------- -// Deduplication -// --------------------------------------------------------------------------- - -/// Remove adjacent duplicate lines (keeps first occurrence). -pub fn dedupe_adjacent(lines: &[String]) -> Vec { - let mut out: Vec = Vec::with_capacity(lines.len()); - for line in lines { - if out.last().map(|l: &String| l.as_str()) != Some(line.as_str()) { - out.push(line.clone()); - } - } - out -} - -// --------------------------------------------------------------------------- -// Head / tail summarisation -// --------------------------------------------------------------------------- - -/// Keep the first `head` lines, an omission marker, and the last `tail` lines. -/// If `lines.len() <= head + tail`, returns `lines` unchanged. -pub fn head_tail(lines: &[String], head: usize, tail: usize) -> Vec { - if lines.len() <= head + tail { - return lines.to_vec(); - } - let omitted = lines.len() - head - tail; - let mut out = Vec::with_capacity(head + 1 + tail); - out.extend_from_slice(&lines[..head]); - out.push(format!("... {} lines omitted ...", omitted)); - out.extend_from_slice(&lines[lines.len() - tail..]); - out -} - -// --------------------------------------------------------------------------- -// Clamping -// --------------------------------------------------------------------------- - -/// Trim `text` at the last newline that is at or before position 50% through -/// the text (mirrors `trimHeadToLineBoundary` in TS). -fn trim_head_to_line_boundary(text: &str) -> &str { - let last_nl = text.rfind('\n'); - match last_nl { - None => text, - Some(pos) => { - if pos < text.len() / 2 { - text - } else { - &text[..pos] - } - } - } -} - -/// Trim `text` at the first newline that is at or after position 50% through -/// (mirrors `trimTailToLineBoundary` in TS). -fn trim_tail_to_line_boundary(text: &str) -> &str { - let first_nl = text.find('\n'); - match first_nl { - None => text, - Some(pos) => { - if pos > text.len().div_ceil(2) { - text - } else { - &text[pos + 1..] - } - } - } -} - -/// Clamp `text` to at most `max_chars` grapheme clusters (tail-truncate). -pub fn clamp_text(text: &str, max_chars: usize) -> String { - if count_text_chars(text) <= max_chars { - return text.to_owned(); - } - let suffix_chars = count_text_chars(TRUNCATION_SUFFIX); - let body_chars = max_chars.saturating_sub(suffix_chars); - let segs: Vec<&str> = text.graphemes(true).collect(); - let head: String = segs[..body_chars.min(segs.len())].concat(); - let head = trim_head_to_line_boundary(&head); - format!("{}{}", head, TRUNCATION_SUFFIX) -} - -/// Clamp `text` to at most `max_chars` grapheme clusters using middle-truncation. -/// Keeps 70% from the head and 30% from the tail. -pub fn clamp_text_middle(text: &str, max_chars: usize) -> String { - if count_text_chars(text) <= max_chars { - return text.to_owned(); - } - let marker_chars = count_text_chars(MIDDLE_TRUNCATION_MARKER); - let body_chars = max_chars.saturating_sub(marker_chars); - let head_chars = (body_chars as f64 * 0.7).ceil() as usize; - let tail_chars = body_chars.saturating_sub(head_chars); - - let segs: Vec<&str> = text.graphemes(true).collect(); - let total = segs.len(); - - let head_raw: String = segs[..head_chars.min(total)].concat(); - let head = trim_head_to_line_boundary(&head_raw).to_owned(); - - let tail_raw: String = segs[total.saturating_sub(tail_chars)..].concat(); - let tail = trim_tail_to_line_boundary(&tail_raw).to_owned(); - - format!("{}{}{}", head, MIDDLE_TRUNCATION_MARKER, tail) -} - -// --------------------------------------------------------------------------- -// Pluralize -// --------------------------------------------------------------------------- - -/// English pluralization matching the upstream `pluralize` function exactly. -pub fn pluralize(count: usize, noun: &str) -> String { - // If noun already ends in "passed", "failed", "skipped" — no change - if noun.ends_with("passed") || noun.ends_with("failed") || noun.ends_with("skipped") { - return format!("{} {}", count, noun); - } - if count == 1 { - return format!("{} {}", count, noun); - } - if noun.ends_with('s') - || noun.ends_with('x') - || noun.ends_with('z') - || noun.ends_with("sh") - || noun.ends_with("ch") - { - return format!("{} {}es", count, noun); - } - // [^aeiou]y → -ies - let ends_consonant_y = noun.ends_with('y') - && noun.len() >= 2 - && !matches!( - noun.chars().nth(noun.len() - 2), - Some('a' | 'e' | 'i' | 'o' | 'u') - ); - if ends_consonant_y { - let stem = &noun[..noun.len() - 1]; - return format!("{} {}ies", count, stem); - } - format!("{} {}s", count, noun) -} - -#[cfg(test)] -mod tests { - use super::*; - - // --- normalize_lines --- - - #[test] - fn normalize_crlf() { - assert_eq!(normalize_lines("a\r\nb"), vec!["a", "b"]); - } - - #[test] - fn normalize_strips_trailing_space() { - assert_eq!(normalize_lines("a "), vec!["a"]); - } - - // --- trim_empty_edges --- - - #[test] - fn trim_edges_removes_blanks() { - let lines: Vec = vec!["", "a", "b", ""] - .iter() - .map(|s| s.to_string()) - .collect(); - assert_eq!(trim_empty_edges(&lines), vec!["a", "b"]); - } - - #[test] - fn trim_edges_all_blank() { - let lines: Vec = vec!["", ""].iter().map(|s| s.to_string()).collect(); - assert!(trim_empty_edges(&lines).is_empty()); - } - - // --- dedupe_adjacent --- - - #[test] - fn dedupe_keeps_non_adjacent() { - let lines = vec!["a", "a", "b", "a"] - .iter() - .map(|s| s.to_string()) - .collect::>(); - assert_eq!(dedupe_adjacent(&lines), vec!["a", "b", "a"]); - } - - // --- head_tail --- - - #[test] - fn head_tail_short_passthrough() { - let lines: Vec = (0..5).map(|i| format!("{}", i)).collect(); - assert_eq!(head_tail(&lines, 3, 3), lines); - } - - #[test] - fn head_tail_omits_middle() { - let lines: Vec = (0..10).map(|i| format!("{}", i)).collect(); - let result = head_tail(&lines, 3, 3); - assert_eq!(result.len(), 7); // 3 + marker + 3 - assert!(result[3].contains("4 lines omitted")); - } - - // --- clamp_text --- - - #[test] - fn clamp_text_passthrough_short() { - assert_eq!(clamp_text("hi", 100), "hi"); - } - - #[test] - fn clamp_text_truncates() { - let long_text = "a".repeat(2000); - let clamped = clamp_text(&long_text, 100); - assert!(count_text_chars(&clamped) <= 100 + count_text_chars(TRUNCATION_SUFFIX)); - assert!(clamped.ends_with("... truncated ...")); - } - - // --- clamp_text_middle --- - - #[test] - fn clamp_middle_passthrough_short() { - assert_eq!(clamp_text_middle("hi", 100), "hi"); - } - - #[test] - fn clamp_middle_contains_marker() { - let long_text = "a\n".repeat(200); - let clamped = clamp_text_middle(&long_text, 50); - assert!( - clamped.contains("... omitted ..."), - "missing marker in: {}", - clamped - ); - } - - // --- pluralize --- - - #[test] - fn pluralize_regular() { - assert_eq!(pluralize(2, "error"), "2 errors"); - } - - #[test] - fn pluralize_singular() { - assert_eq!(pluralize(1, "error"), "1 error"); - } - - #[test] - fn pluralize_sibilant() { - assert_eq!(pluralize(2, "match"), "2 matches"); - } - - #[test] - fn pluralize_y_ending() { - assert_eq!(pluralize(2, "entry"), "2 entries"); - } - - #[test] - fn pluralize_already_ended() { - assert_eq!(pluralize(3, "passed"), "3 passed"); - } - - #[test] - fn pluralize_failed_noun() { - assert_eq!(pluralize(2, "failed"), "2 failed"); - } - - #[test] - fn pluralize_skipped_noun() { - assert_eq!(pluralize(0, "skipped"), "0 skipped"); - } - - // --- trim_head_to_line_boundary edge cases --- - - #[test] - fn clamp_text_no_newline_in_head() { - // When there's no newline in the head portion, clamp_text still truncates - // This exercises the "None" branch of trim_head_to_line_boundary - let text = "a".repeat(200); // no newlines - let clamped = clamp_text(&text, 50); - assert!(clamped.ends_with("... truncated ...")); - } - - #[test] - fn clamp_text_newline_at_early_position() { - // Newline at position < len/2 → trim_head_to_line_boundary returns text as-is - // (the newline is too early to use as a boundary) - let text = "ab\n".to_owned() + &"x".repeat(200); - let clamped = clamp_text(&text, 100); - assert!(clamped.ends_with("... truncated ...")); - } - - #[test] - fn clamp_middle_no_newline_in_tail() { - // tail portion has no newline → trim_tail_to_line_boundary returns text as-is - // This exercises the "None" branch of trim_tail_to_line_boundary - let text = "line1\nline2\n".to_owned() + &"x".repeat(300); - let clamped = clamp_text_middle(&text, 40); - assert!(clamped.contains("... omitted ...")); - } - - #[test] - fn clamp_middle_newline_at_late_position() { - // Newline at position > len.div_ceil(2) → returns text as-is in trim_tail - // Build tail where the first newline is very late - let text = "line1\nline2\nline3\n".repeat(50); - let clamped = clamp_text_middle(&text, 80); - assert!(clamped.contains("... omitted ...")); - } - - #[test] - fn clamp_middle_tail_newline_in_second_half() { - // Force trim_tail_to_line_boundary to hit the "pos > len/2" branch: - // The tail raw string must have its first newline past the midpoint. - // We need a large body so the tail portion (30%) starts with many chars - // before the first newline. - // "xxxxxxxx\nyyyyyyy" where \n is at position > midpoint - // Construct text with many lines; the last chunk has no early newline - let many_lines: String = "head-line\n".repeat(100); - // Tail segment ends with long non-newline text followed by newline at end - let text = many_lines + &"z".repeat(200) + "\nlast"; - let clamped = clamp_text_middle(&text, 300); - // Should produce output with the marker - assert!(clamped.contains("... omitted ...")); - } - - // --- head_tail edge cases --- - - #[test] - fn head_tail_exact_boundary() { - // lines.len() == head + tail → passthrough (not truncated) - let lines: Vec = (0..6).map(|i| format!("line{}", i)).collect(); - let result = head_tail(&lines, 3, 3); - assert_eq!(result, lines, "exact head+tail should not truncate"); - } - - // --- dedupe_adjacent empty input --- - - #[test] - fn dedupe_adjacent_empty() { - assert!(dedupe_adjacent(&[]).is_empty()); - } - - // --- normalize_lines with no trailing whitespace --- - - #[test] - fn normalize_lines_no_crlf() { - let lines = normalize_lines("a\nb\nc"); - assert_eq!(lines, vec!["a", "b", "c"]); - } -} diff --git a/src/openhuman/tokenjuice/text/width.rs b/src/openhuman/tokenjuice/text/width.rs deleted file mode 100644 index 93ccd603c..000000000 --- a/src/openhuman/tokenjuice/text/width.rs +++ /dev/null @@ -1,241 +0,0 @@ -//! Grapheme-aware terminal-column width calculation. -//! -//! Uses `unicode-segmentation` for grapheme cluster boundaries and -//! `unicode-width` for CJK/emoji double-width detection, mirroring the -//! `Intl.Segmenter`-based logic in the upstream TypeScript. - -use unicode_segmentation::UnicodeSegmentation; -use unicode_width::UnicodeWidthChar; - -/// Return the list of user-perceived grapheme clusters in `text`. -pub fn graphemes(text: &str) -> Vec<&str> { - text.graphemes(true).collect() -} - -/// Return the number of grapheme clusters (not bytes or scalar values). -/// -/// This is used for character-count limiting (mirrors `countTextChars` in TS). -pub fn count_text_chars(text: &str) -> usize { - text.graphemes(true).count() -} - -/// Return the terminal column width of a single grapheme cluster. -/// -/// Emoji are assumed to be 2 columns wide, which matches the upstream TS -/// `graphemeWidth` logic. The `unicode-width` crate handles most CJK ranges. -fn grapheme_width(segment: &str) -> usize { - if segment.is_empty() { - return 0; - } - - // Emoji: assume width 2 (matches upstream) - let first_cp = segment.chars().next().unwrap_or('\0'); - if is_emoji(first_cp) { - return 2; - } - - // Use unicode-width on the first non-combining code point - let mut width = 0usize; - let mut has_visible = false; - for ch in segment.chars() { - // Skip zero-width joiners and variation selectors - if ch == '\u{200D}' || ch == '\u{FE0F}' { - continue; - } - // Skip combining marks (general category M) - if is_combining_mark(ch) { - continue; - } - let w = UnicodeWidthChar::width(ch).unwrap_or(0); - width = width.max(w); - has_visible = true; - } - - if has_visible { - width - } else { - 0 - } -} - -/// Return the total terminal column width of `text`. -pub fn count_terminal_cells(text: &str) -> usize { - text.graphemes(true).map(grapheme_width).sum() -} - -// --------------------------------------------------------------------------- -// Helpers -// --------------------------------------------------------------------------- - -/// Conservative emoji test covering the main Extended_Pictographic ranges used -/// by the upstream TS code (`/\p{Extended_Pictographic}/u`). -/// -/// We use broad ranges to avoid unreachable-pattern warnings in match arms. -fn is_emoji(cp: char) -> bool { - let c = cp as u32; - // Misc symbols, dingbats, and the main supplemental emoji blocks - matches!(c, - 0x2300..=0x27BF | // Misc technical + arrows + dingbats (broad) - 0x1F300..=0x1FAFF // All supplemental emoji / symbol blocks - ) -} - -/// True for Unicode combining marks (general category M*). -/// We use a simplified range check sufficient for the characters that appear -/// in terminal output. -fn is_combining_mark(ch: char) -> bool { - let c = ch as u32; - matches!(c, - 0x0300..=0x036F | // Combining Diacritical Marks - 0x1AB0..=0x1AFF | // Combining Diacritical Marks Extended - 0x1DC0..=0x1DFF | // Combining Diacritical Marks Supplement - 0x20D0..=0x20FF | // Combining Diacritical Marks for Symbols - 0xFE20..=0xFE2F // Combining Half Marks - ) -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn ascii_char_count() { - assert_eq!(count_text_chars("hello"), 5); - } - - #[test] - fn emoji_char_count_one_grapheme() { - // U+1F600 GRINNING FACE — 1 grapheme cluster - assert_eq!(count_text_chars("😀"), 1); - } - - #[test] - fn cjk_terminal_width_two_cells() { - // U+4E2D — one CJK character, should be 2 terminal cells - assert_eq!(count_terminal_cells("中"), 2); - } - - #[test] - fn ascii_terminal_width() { - assert_eq!(count_terminal_cells("abc"), 3); - } - - #[test] - fn graphemes_splits_correctly() { - let gs = graphemes("abc"); - assert_eq!(gs, vec!["a", "b", "c"]); - } - - // --- grapheme_width coverage --- - - #[test] - fn emoji_terminal_width_two_cells() { - // U+1F600 GRINNING FACE — emoji, should be 2 terminal cells - assert_eq!(count_terminal_cells("😀"), 2); - } - - #[test] - fn zwj_sequence_is_two_cells() { - // ZWJ sequences (e.g. family emoji) — grapheme_width should handle ZWJ - // U+200D ZERO WIDTH JOINER is skipped; the base emoji drives width - let fam = "\u{1F468}\u{200D}\u{1F469}\u{200D}\u{1F467}"; // family emoji - let w = count_terminal_cells(fam); - // Should be at least 1 (base emoji) — not zero - assert!(w >= 1, "ZWJ sequence should have non-zero width"); - } - - #[test] - fn variation_selector_skipped() { - // U+FE0F VARIATION SELECTOR-16 is skipped (not counted as width) - let text_emoji = "\u{2665}\u{FE0F}"; // ♥️ heart with VS16 - let w = count_terminal_cells(text_emoji); - // The heart U+2665 is in the 0x2300..=0x27BF range → emoji → 2 cells - assert_eq!(w, 2); - } - - #[test] - fn combining_mark_does_not_add_width() { - // U+0301 COMBINING ACUTE ACCENT is a combining mark — skipped in width calc - // "e\u{0301}" is one grapheme cluster (é) — width should be 1 (from "e") - let composed = "e\u{0301}"; - let w = count_terminal_cells(composed); - assert_eq!(w, 1, "combining accent should not add extra width"); - } - - #[test] - fn empty_string_zero_width() { - assert_eq!(count_terminal_cells(""), 0); - assert_eq!(count_text_chars(""), 0); - } - - #[test] - fn mixed_ascii_and_cjk_width() { - // "a中b" → 1 + 2 + 1 = 4 terminal cells, 3 grapheme clusters - assert_eq!(count_terminal_cells("a中b"), 4); - assert_eq!(count_text_chars("a中b"), 3); - } - - #[test] - fn misc_symbols_are_emoji_width() { - // U+2603 SNOWMAN is in 0x2300..=0x27BF range → width 2 - let snowman = "\u{2603}"; - let w = count_terminal_cells(snowman); - assert_eq!(w, 2); - } - - #[test] - fn combining_diacritical_marks_extended_covered() { - // U+1AB0 is in 0x1AB0..=0x1AFF range (Combining Diacritical Marks Extended) - // These are combining marks that get skipped in grapheme_width - // "a\u{1AB0}" should be one grapheme cluster with width 1 (from 'a') - let text = "a\u{1AB0}"; - let w = count_terminal_cells(text); - // 'a' contributes 1, the combining mark is skipped - assert_eq!(w, 1); - } - - #[test] - fn combining_half_marks_fe20_range() { - // U+FE20 is in 0xFE20..=0xFE2F (Combining Half Marks) - // This exercises the last arm of is_combining_mark - let text = "x\u{FE20}"; - let w = count_terminal_cells(text); - // 'x' contributes 1; FE20 is a combining mark, skipped - assert_eq!(w, 1); - } - - #[test] - fn only_zwj_grapheme_has_zero_width() { - // A segment consisting only of ZWJ (U+200D) — skipped in grapheme_width - // has_visible remains false → returns 0 - // This is an artificial segment since real graphemes always have a base; - // we test via count_terminal_cells on a string with only ZWJ - let text = "\u{200D}"; - let w = count_terminal_cells(text); - // ZWJ alone: has_visible stays false → width 0 - assert_eq!(w, 0); - } - - #[test] - fn grapheme_width_empty_segment_is_zero() { - // count_terminal_cells on empty string: graphemes() returns no segments - // so the sum is 0; the empty-check branch is exercised via internal calls - assert_eq!(count_terminal_cells(""), 0); - } - - #[test] - fn combining_diacritical_supplement_1dc0() { - // U+1DC0 is in 0x1DC0..=0x1DFF (Combining Diacritical Marks Supplement) - let text = "e\u{1DC0}"; - let w = count_terminal_cells(text); - assert_eq!(w, 1); - } - - #[test] - fn combining_diacritical_for_symbols_20d0() { - // U+20D0 is in 0x20D0..=0x20FF (Combining Diacritical Marks for Symbols) - let text = "A\u{20D0}"; - let w = count_terminal_cells(text); - assert_eq!(w, 1); - } -} diff --git a/src/openhuman/tokenjuice/text_tests.rs b/src/openhuman/tokenjuice/text_tests.rs deleted file mode 100644 index 2932a78f2..000000000 --- a/src/openhuman/tokenjuice/text_tests.rs +++ /dev/null @@ -1,442 +0,0 @@ -//! Additional unit tests for the `tokenjuice::text` sub-modules. -//! -//! Focuses on coverage gaps identified in test-map.md Pick 5 -//! ("TokenJuice Rust port for tool-output compaction"): -//! - `strip_ansi` with multi-byte / emoji text (grapheme safety). -//! - `dedupe_adjacent` additional edge cases. -//! - `clamp_text_middle` grapheme-safe split — never breaks inside a -//! multi-byte codepoint or multi-scalar grapheme cluster. -//! - 3-layer overlay precedence: project > user > builtin. -//! - Rule loader gracefully handles invalid regex (diagnostic, no panic). - -use crate::openhuman::tokenjuice::rules::loader::{load_rules, LoadRuleOptions}; -use crate::openhuman::tokenjuice::text::width::{count_text_chars, graphemes}; -use crate::openhuman::tokenjuice::text::{clamp_text_middle, dedupe_adjacent, strip_ansi}; -use crate::openhuman::tokenjuice::types::RuleOrigin; - -// ── strip_ansi — multi-byte / emoji safety ─────────────────────────────────── - -#[test] -fn strip_ansi_leaves_multibyte_cjk_intact() { - // CJK characters must pass through completely even when preceded by ANSI. - let input = "\x1b[32m中文\x1b[0m"; - assert_eq!(strip_ansi(input), "中文"); -} - -#[test] -fn strip_ansi_leaves_emoji_intact() { - // Emoji must survive stripping. - let input = "\x1b[1m😀 hello\x1b[0m"; - assert_eq!(strip_ansi(input), "😀 hello"); -} - -#[test] -fn strip_ansi_multi_byte_only_no_escapes() { - // When there are no ANSI codes, multi-byte text is returned unchanged. - let text = "こんにちは"; - assert_eq!(strip_ansi(text), text); -} - -#[test] -fn strip_ansi_zwj_emoji_sequence_preserved() { - // ZWJ sequences (family emoji) must not be mangled. - let fam = "\u{1F468}\u{200D}\u{1F469}\u{200D}\u{1F467}"; // family emoji - let colored = format!("\x1b[31m{fam}\x1b[0m"); - let stripped = strip_ansi(&colored); - assert_eq!(stripped, fam, "ZWJ sequence must survive ANSI stripping"); -} - -#[test] -fn strip_ansi_mixed_scripts_preserved() { - // Arabic, CJK, Latin, emoji all in one string with ANSI wrappers. - let input = "\x1b[33mعربي 中文 hello 🌍\x1b[0m"; - let stripped = strip_ansi(input); - assert_eq!(stripped, "عربي 中文 hello 🌍"); -} - -#[test] -fn strip_ansi_empty_string() { - assert_eq!(strip_ansi(""), ""); -} - -#[test] -fn strip_ansi_only_escape_sequences() { - // If the entire string is escape sequences, the result should be empty. - let all_ansi = "\x1b[0m\x1b[1m\x1b[31m"; - assert_eq!(strip_ansi(all_ansi), ""); -} - -// ── dedupe_adjacent ─────────────────────────────────────────────────────────── - -fn strs(v: &[&str]) -> Vec { - v.iter().map(|s| s.to_string()).collect() -} - -#[test] -fn dedupe_adjacent_collapses_run_of_identical_lines() { - let lines = strs(&["a", "a", "a", "b"]); - let out = dedupe_adjacent(&lines); - assert_eq!(out, strs(&["a", "b"])); -} - -#[test] -fn dedupe_adjacent_preserves_non_adjacent_duplicates() { - // Same value reappearing after a different line must NOT be collapsed. - let lines = strs(&["a", "b", "a"]); - let out = dedupe_adjacent(&lines); - assert_eq!(out, strs(&["a", "b", "a"])); -} - -#[test] -fn dedupe_adjacent_single_element_is_unchanged() { - let lines = strs(&["only"]); - let out = dedupe_adjacent(&lines); - assert_eq!(out, strs(&["only"])); -} - -#[test] -fn dedupe_adjacent_all_identical_collapses_to_one() { - let lines = strs(&["x", "x", "x", "x", "x"]); - let out = dedupe_adjacent(&lines); - assert_eq!(out, strs(&["x"])); -} - -#[test] -fn dedupe_adjacent_empty_lines_are_deduplicated() { - // Adjacent blank lines must also be collapsed. - let lines = strs(&["a", "", "", "b", "", "c"]); - let out = dedupe_adjacent(&lines); - assert_eq!(out, strs(&["a", "", "b", "", "c"])); -} - -#[test] -fn dedupe_adjacent_multibyte_lines_collapsed() { - let lines = strs(&["日本語", "日本語", "日本語"]); - let out = dedupe_adjacent(&lines); - assert_eq!(out, strs(&["日本語"])); -} - -// ── clamp_text_middle — grapheme-safe middle truncation ─────────────────────── - -/// Assert that `clamp_text_middle` never splits inside a multi-byte -/// grapheme: every byte of the output must decode to valid UTF-8, and -/// every character in the output must appear as a whole grapheme in the -/// source string. -#[test] -fn clamp_text_middle_output_is_valid_utf8() { - // A long string of CJK characters (each is 3 bytes in UTF-8). - // String is always valid UTF-8, so a tautological from_utf8 check would - // not actually verify the grapheme-safety contract. Instead, assert that - // every grapheme in the clamped output also exists as a complete grapheme - // in the source (i.e. no partial cluster fragments leaked through). - let cjk: String = "中文字符测试!".repeat(50); - let clamped = clamp_text_middle(&cjk, 30); - let source_graphemes: std::collections::HashSet<&str> = graphemes(&cjk).into_iter().collect(); - for g in graphemes(&clamped) { - // The omission marker is the only legitimate non-source content. - if g == "·" - || g.chars().all(|c| { - c.is_ascii_punctuation() || c.is_ascii_whitespace() || c.is_ascii_alphanumeric() - }) - { - continue; - } - assert!( - source_graphemes.contains(g), - "grapheme {g:?} in clamp output is not a whole grapheme of the source" - ); - } -} - -#[test] -fn clamp_text_middle_does_not_split_emoji_grapheme() { - // Each emoji is 4 bytes; a naïve byte split could land in the middle. - // Verify boundary correctness by counting graphemes — the clamp must - // never produce a partial codepoint or partial grapheme. - let emojis: String = "😀".repeat(100); - let clamped = clamp_text_middle(&emojis, 20); - // Every non-marker grapheme in the output must equal "😀" (source has - // exactly one distinct grapheme). A partial split would leave a - // replacement char or a stray surrogate-equivalent sequence. - for g in graphemes(&clamped) { - let only_ascii = g.chars().all(|c| c.is_ascii()); - assert!( - g == "😀" || only_ascii, - "unexpected grapheme {g:?} — partial emoji split detected" - ); - } -} - -#[test] -fn clamp_text_middle_short_text_is_passthrough() { - // Strings shorter than max_chars are returned verbatim. - let text = "hello 世界 🌍"; - let clamped = clamp_text_middle(text, 200); - assert_eq!(clamped, text); -} - -#[test] -fn clamp_text_middle_inserts_omission_marker() { - let long_text = "line\n".repeat(200); - let clamped = clamp_text_middle(&long_text, 100); - assert!( - clamped.contains("omitted"), - "middle clamp must contain omission marker, got: {}", - &clamped[..clamped.len().min(120)] - ); -} - -#[test] -fn clamp_text_middle_grapheme_count_respects_limit() { - // The result should not exceed max_chars + marker length substantially. - // We use a lenient bound (2× marker overhead) rather than an exact count. - let long_text = "あいうえお\n".repeat(200); // multi-byte lines - let max = 100usize; - let clamped = clamp_text_middle(&long_text, max); - let grapheme_count = count_text_chars(&clamped); - // Allow up to 2× max to accommodate the omission marker. - assert!( - grapheme_count <= max * 2, - "clamped grapheme count {grapheme_count} exceeds 2×max ({max})" - ); -} - -#[test] -fn clamp_text_middle_zwj_sequence_not_split() { - // ZWJ family emoji repeated; each base char is 4 bytes, ZWJ is 3 bytes. - // Assert no partial ZWJ family fragments survive in the output: any - // grapheme containing the ZWJ codepoint must be the full family unit. - let zwj_unit = "\u{1F468}\u{200D}\u{1F469}\u{200D}\u{1F467}"; // family - let long: String = (zwj_unit.to_owned() + "\n").repeat(100); - let clamped = clamp_text_middle(&long, 30); - for g in graphemes(&clamped) { - if g.contains('\u{200D}') - || g.chars() - .any(|c| matches!(c, '\u{1F468}' | '\u{1F469}' | '\u{1F467}')) - { - assert_eq!( - g, zwj_unit, - "clamp produced partial ZWJ cluster {g:?}; expected the full family unit" - ); - } - } -} - -// ── grapheme helper round-trip ──────────────────────────────────────────────── - -#[test] -fn graphemes_clusters_match_count_text_chars() { - let mixed = "hello 中文 😀 emoji"; - let gs = graphemes(mixed); - assert_eq!(gs.len(), count_text_chars(mixed)); -} - -// ── 3-layer overlay precedence ──────────────────────────────────────────────── - -#[test] -fn three_layer_overlay_project_beats_user_beats_builtin() { - // Create temporary dirs for user and project layers. - let user_dir = tempfile::tempdir().expect("user tempdir"); - let project_dir = tempfile::tempdir().expect("project tempdir"); - - // User overrides the builtin `git/status` rule with family "user-family". - std::fs::write( - user_dir.path().join("gs.json"), - r#"{"id":"git/status","family":"user-family","match":{}}"#, - ) - .unwrap(); - - // Project overrides the same rule with family "project-family" (highest priority). - std::fs::write( - project_dir.path().join("gs.json"), - r#"{"id":"git/status","family":"project-family","match":{}}"#, - ) - .unwrap(); - - let opts = LoadRuleOptions { - user_rules_dir: Some(user_dir.path().to_owned()), - project_rules_dir: Some(project_dir.path().to_owned()), - ..Default::default() - }; - let rules = load_rules(&opts); - let gs = rules - .iter() - .find(|r| r.rule.id == "git/status") - .expect("git/status rule must be present"); - - // Project must win over user and builtin. - assert_eq!( - gs.rule.family, "project-family", - "project layer must override user and builtin layers" - ); - assert_eq!( - gs.source, - RuleOrigin::Project, - "winning rule must be sourced from Project" - ); -} - -#[test] -fn two_layer_overlay_user_beats_builtin() { - let user_dir = tempfile::tempdir().expect("user tempdir"); - - std::fs::write( - user_dir.path().join("gs.json"), - r#"{"id":"git/status","family":"user-only","match":{}}"#, - ) - .unwrap(); - - let opts = LoadRuleOptions { - user_rules_dir: Some(user_dir.path().to_owned()), - exclude_project: true, - ..Default::default() - }; - let rules = load_rules(&opts); - let gs = rules - .iter() - .find(|r| r.rule.id == "git/status") - .expect("git/status rule"); - - assert_eq!(gs.rule.family, "user-only"); - assert_eq!(gs.source, RuleOrigin::User); -} - -#[test] -fn builtin_rule_present_when_no_overrides() { - let opts = LoadRuleOptions { - exclude_user: true, - exclude_project: true, - ..Default::default() - }; - let rules = load_rules(&opts); - let gs = rules.iter().find(|r| r.rule.id == "git/status"); - assert!(gs.is_some(), "builtin git/status rule must be present"); - assert_eq!( - gs.unwrap().source, - RuleOrigin::Builtin, - "with no overlay layers, source must be Builtin" - ); -} - -// ── rule loader — invalid regex gracefully handled ──────────────────────────── - -#[test] -fn invalid_regex_in_skip_patterns_does_not_panic() { - use crate::openhuman::tokenjuice::rules::compiler::compile_rule; - use crate::openhuman::tokenjuice::types::{JsonRule, RuleFilters, RuleMatch}; - - let rule = JsonRule { - id: "test/bad-skip".to_owned(), - family: "test".to_owned(), - description: None, - priority: None, - on_empty: None, - match_output: None, - counter_source: None, - r#match: RuleMatch::default(), - filters: Some(RuleFilters { - skip_patterns: Some(vec![ - "[invalid-regex".to_owned(), // deliberately malformed - "valid_pattern".to_owned(), // valid one after the bad one - ]), - keep_patterns: None, - }), - transforms: None, - summarize: None, - counters: None, - failure: None, - }; - - // Must not panic; invalid regex is silently dropped. - let compiled = compile_rule( - rule, - RuleOrigin::Builtin, - "builtin:test/bad-skip".to_owned(), - ); - - // The invalid pattern is dropped; the valid one should be retained. - assert_eq!( - compiled.compiled.skip_patterns.len(), - 1, - "invalid regex must be dropped; valid pattern must be retained" - ); -} - -#[test] -fn all_invalid_regex_in_skip_patterns_leaves_empty_vec() { - use crate::openhuman::tokenjuice::rules::compiler::compile_rule; - use crate::openhuman::tokenjuice::types::{JsonRule, RuleFilters, RuleMatch}; - - let rule = JsonRule { - id: "test/all-bad".to_owned(), - family: "test".to_owned(), - description: None, - priority: None, - on_empty: None, - match_output: None, - counter_source: None, - r#match: RuleMatch::default(), - filters: Some(RuleFilters { - skip_patterns: Some(vec![ - "[bad1".to_owned(), - "(bad2".to_owned(), - "{bad3".to_owned(), - ]), - keep_patterns: None, - }), - transforms: None, - summarize: None, - counters: None, - failure: None, - }; - - let compiled = compile_rule(rule, RuleOrigin::Builtin, "builtin:test/all-bad".to_owned()); - assert!( - compiled.compiled.skip_patterns.is_empty(), - "all invalid skip patterns must produce an empty vec" - ); -} - -#[test] -fn invalid_regex_loaded_from_disk_is_skipped_not_fatal() { - // Write a rule JSON with an invalid skip_pattern to a temp project dir. - let dir = tempfile::tempdir().expect("tempdir"); - let bad_rule = r#"{ - "id": "test/disk-bad-regex", - "family": "test", - "match": {}, - "filters": { - "skipPatterns": ["[invalid"] - } - }"#; - std::fs::write(dir.path().join("bad_regex.json"), bad_rule).unwrap(); - - // Also add a valid rule to ensure loading continues normally. - let good_rule = r#"{"id":"test/disk-good","family":"test","match":{}}"#; - std::fs::write(dir.path().join("good.json"), good_rule).unwrap(); - - let opts = LoadRuleOptions { - project_rules_dir: Some(dir.path().to_owned()), - exclude_user: true, - ..Default::default() - }; - - // Must not panic; bad regex → compiled rule with no skip patterns. - let rules = load_rules(&opts); - // The valid rule is still present. - assert!( - rules.iter().any(|r| r.rule.id == "test/disk-good"), - "valid rule must still load alongside the bad-regex rule" - ); - // The bad-regex rule must still load — but with the invalid skip pattern - // dropped so the rule itself is non-fatal. Asserting presence avoids a - // false positive where the rule is silently dropped entirely. - let bad = rules - .iter() - .find(|r| r.rule.id == "test/disk-bad-regex") - .expect("bad-regex rule must still load"); - assert!( - bad.compiled.skip_patterns.is_empty(), - "bad-regex rule must have empty compiled skip_patterns" - ); -} diff --git a/src/openhuman/tokenjuice/tokens.rs b/src/openhuman/tokenjuice/tokens.rs deleted file mode 100644 index f9b0335aa..000000000 --- a/src/openhuman/tokenjuice/tokens.rs +++ /dev/null @@ -1,46 +0,0 @@ -//! Lightweight token estimation for compaction savings accounting. -//! -//! The agent harness gets authoritative token counts from the provider, but -//! the content router runs *before* any provider call, so it has no exact -//! count for a tool result. For savings insights we use the standard ~4 -//! characters-per-token heuristic (close enough for English text / code / -//! JSON to drive cost estimates and the savings dashboard). This is the same -//! order of approximation Headroom reports its savings with. - -/// Average characters per token used by the estimate. -pub const CHARS_PER_TOKEN: f64 = 4.0; - -/// Estimate the number of tokens in `text` (≈ chars / 4, minimum 1 for any -/// non-empty input). Uses `chars().count()` so multi-byte text isn't -/// over-counted by byte length. -pub fn estimate_tokens(text: &str) -> u64 { - if text.is_empty() { - return 0; - } - let chars = text.chars().count() as f64; - (chars / CHARS_PER_TOKEN).ceil().max(1.0) as u64 -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn empty_is_zero() { - assert_eq!(estimate_tokens(""), 0); - } - - #[test] - fn rough_quarter_of_chars() { - assert_eq!(estimate_tokens("abcd"), 1); - assert_eq!(estimate_tokens(&"x".repeat(400)), 100); - // Any non-empty input is at least one token. - assert_eq!(estimate_tokens("a"), 1); - } - - #[test] - fn counts_chars_not_bytes() { - // 4 multi-byte chars → 1 token, not 12 bytes → 3 tokens. - assert_eq!(estimate_tokens("日本語訳"), 1); - } -} diff --git a/src/openhuman/tokenjuice/tool_integration.rs b/src/openhuman/tokenjuice/tool_integration.rs deleted file mode 100644 index 15294839c..000000000 --- a/src/openhuman/tokenjuice/tool_integration.rs +++ /dev/null @@ -1,407 +0,0 @@ -//! Glue between the agent tool loop and the TokenJuice content router. -//! -//! Exposes the entry points the agent loop calls after a tool returns output: -//! -//! - [`compact_tool_output_with_policy`] — full version with the tool's JSON -//! arguments and exit code; derives a command/argv and content hint, routes -//! through the content router, and returns `(text, CompactionStats)`. -//! - [`compact_output`] — minimal version (content + tool name + enable flag) -//! for call sites that only have those, returning just the text. -//! -//! Both are **pass-through safe**: if compression doesn't meaningfully shrink -//! the payload, or the input is under the byte floor, or the router/CCR is -//! disabled, the original string is returned untouched. -//! -//! Runtime options (the `[tokenjuice]` config block) are installed once at -//! startup via [`configure`]; callers don't thread `Config` through. - -use once_cell::sync::OnceCell; -use serde_json::Value; -use std::sync::RwLock; - -use super::compress::route; -use super::types::{AgentTokenjuiceCompression, CompressInput, CompressOptions, ContentHint}; - -/// Skip compaction for outputs smaller than this (bytes) by default. Tiny -/// outputs have no headroom and risk distortion. Overridable per the config's -/// `min_bytes_to_compress` once [`configure`] runs. -const DEFAULT_MIN_COMPACT_INPUT_BYTES: usize = 512; - -/// Process-global runtime options, installed from config at startup. -fn options_cell() -> &'static RwLock { - static OPTS: OnceCell> = OnceCell::new(); - OPTS.get_or_init(|| { - RwLock::new(CompressOptions { - min_bytes_to_compress: DEFAULT_MIN_COMPACT_INPUT_BYTES, - ..Default::default() - }) - }) -} - -/// Install the runtime [`CompressOptions`] (called once from config at startup). -/// Also configures the CCR cache limits/disk tier indirectly via the caller. -pub fn configure(opts: CompressOptions) { - *options_cell().write().unwrap_or_else(|p| p.into_inner()) = opts; -} - -/// Snapshot the current runtime options. -pub fn current_options() -> CompressOptions { - options_cell() - .read() - .unwrap_or_else(|p| p.into_inner()) - .clone() -} - -fn options_for_agent(profile: AgentTokenjuiceCompression) -> Option { - match profile { - AgentTokenjuiceCompression::Off => None, - AgentTokenjuiceCompression::Auto | AgentTokenjuiceCompression::Full => { - Some(current_options()) - } - AgentTokenjuiceCompression::Light => { - let mut opts = current_options(); - // Coding agents need raw, exact tool text more than aggressive token - // savings. Disabling CCR makes every lossy compressor decline in - // route(), while still allowing any truly lossless reduction. - opts.ccr_enabled = false; - opts.ml_text_enabled = false; - Some(opts) - } - } -} - -/// Install the full TokenJuice runtime configuration in one call at startup: -/// router/compressor options, CCR cache limits, and the optional on-disk tier. -/// Kept free of the config-schema type so `tokenjuice` stays decoupled — the -/// caller maps `Config.tokenjuice` into these primitives. -#[allow(clippy::too_many_arguments)] -pub fn install_config( - options: CompressOptions, - max_cache_entries: usize, - max_cache_bytes: usize, - ccr_ttl_secs: Option, - disk_tier_root: Option, -) { - configure(options); - super::cache::configure(max_cache_entries, max_cache_bytes, ccr_ttl_secs); - // Enable or disable the disk tier to match the setting — a `None` here means - // the user turned it off, so clear any previously-installed disk root rather - // than leaving the process writing originals to disk until restart. - match disk_tier_root { - Some(root) => super::cache::enable_disk_tier(root), - None => super::cache::disable_disk_tier(), - } - log::debug!("[tokenjuice] runtime config installed"); -} - -/// Statistics for a single compaction call (back-compat shape). -#[derive(Debug, Clone)] -pub struct CompactionStats { - pub tool_name: String, - pub original_bytes: usize, - pub compacted_bytes: usize, - /// The compressor kind (or `none/...`) that handled the output. - pub rule_id: String, - pub applied: bool, -} - -impl CompactionStats { - pub fn ratio(&self) -> f64 { - if self.original_bytes == 0 { - 1.0 - } else { - self.compacted_bytes as f64 / self.original_bytes as f64 - } - } -} - -/// Compact a tool call's output using an agent-level TokenJuice profile. -/// -/// * `tool_name` — the agent-level tool name (`shell`, `grep`, `browser_navigate`). -/// * `arguments` — the raw JSON arguments; used to derive command/argv (for the -/// log/command rule path) and a file extension (for code/JSON/HTML hints). -/// * `output` — the captured tool output (already credential-scrubbed). -/// * `exit_code` — enables failure-preserving behaviour in the log compressor. -/// -/// Returns `(text, stats)`. When `stats.applied == false` the text is the -/// untouched original. -pub async fn compact_tool_output_with_policy( - tool_name: &str, - arguments: Option<&Value>, - output: &str, - exit_code: Option, - profile: AgentTokenjuiceCompression, -) -> (String, CompactionStats) { - let original_bytes = output.len(); - - let Some(opts) = options_for_agent(profile) else { - log::debug!( - "[tokenjuice] agent profile disabled compaction tool={} bytes={}", - tool_name, - original_bytes - ); - return ( - output.to_string(), - CompactionStats { - tool_name: tool_name.to_string(), - original_bytes, - compacted_bytes: original_bytes, - rule_id: "none/agent-profile-off".to_string(), - applied: false, - }, - ); - }; - - // A recovery tool's output is the original we previously offloaded — never - // re-compact it, or the agent could never see the full data it asked for. - if super::cache::is_recovery_tool(tool_name) { - return ( - output.to_string(), - CompactionStats { - tool_name: tool_name.to_string(), - original_bytes, - compacted_bytes: original_bytes, - rule_id: "none/recovery-tool".to_string(), - applied: false, - }, - ); - } - - let (command, argv) = extract_command_argv(arguments); - let hint = ContentHint { - source_tool: Some(tool_name.to_string()), - extension: extract_extension(arguments), - query: extract_query(arguments), - ..Default::default() - }; - - let input = CompressInput { - content: output, - kind: super::types::ContentKind::PlainText, - hint: &hint, - exit_code, - command, - argv, - original_bytes, - }; - - let res = route(input, &opts).await; - let stats = CompactionStats { - tool_name: tool_name.to_string(), - original_bytes, - compacted_bytes: res.compacted_bytes, - rule_id: if res.applied { - res.compressor.as_str().to_string() - } else { - format!("none/{}", res.content_kind.as_str()) - }, - applied: res.applied, - }; - (res.text, stats) -} - -/// Minimal compaction for call sites that only have content + tool name. The -/// `enabled` flag is an explicit kill-switch on top of the configured options. -pub async fn compact_output(content: String, tool_name: &str, enabled: bool) -> String { - compact_output_with_policy( - content, - tool_name, - enabled, - AgentTokenjuiceCompression::Full, - ) - .await -} - -/// Minimal compaction with an agent-level TokenJuice profile. -pub async fn compact_output_with_policy( - content: String, - tool_name: &str, - enabled: bool, - profile: AgentTokenjuiceCompression, -) -> String { - // The call-site `enabled` flag and the configured router switch are both - // hard off-switches; either one short-circuits to the untouched original. - if !enabled || !current_options().router_enabled { - return content; - } - let (text, _stats) = - compact_tool_output_with_policy(tool_name, None, &content, None, profile).await; - text -} - -/// Derive `(command, argv)` from a tool's JSON arguments. -fn extract_command_argv(arguments: Option<&Value>) -> (Option, Option>) { - let Some(Value::Object(map)) = arguments else { - return (None, None); - }; - - if let Some(Value::Array(arr)) = map.get("argv") { - let argv: Vec = arr - .iter() - .filter_map(|v| v.as_str().map(|s| s.to_owned())) - .collect(); - if !argv.is_empty() { - let command = argv.join(" "); - return (Some(command), Some(argv)); - } - } - - let cmd_str = map - .get("command") - .and_then(Value::as_str) - .or_else(|| map.get("cmd").and_then(Value::as_str)); - - if let Some(cmd) = cmd_str { - if let Some(Value::Array(args)) = map.get("args") { - let mut argv = vec![cmd.to_owned()]; - argv.extend(args.iter().filter_map(|v| v.as_str().map(|s| s.to_owned()))); - return (Some(format!("{cmd} {}", argv[1..].join(" "))), Some(argv)); - } - let argv: Vec = cmd.split_whitespace().map(|s| s.to_owned()).collect(); - return (Some(cmd.to_owned()), (!argv.is_empty()).then_some(argv)); - } - - (None, None) -} - -/// Derive a file extension hint from common path-bearing argument shapes. -fn extract_extension(arguments: Option<&Value>) -> Option { - let Some(Value::Object(map)) = arguments else { - return None; - }; - let path = ["path", "file_path", "file", "filename"] - .iter() - .find_map(|k| map.get(*k).and_then(Value::as_str))?; - let ext = std::path::Path::new(path) - .extension() - .and_then(|e| e.to_str())?; - Some(ext.to_ascii_lowercase()) -} - -/// Derive a search-query hint from common query-bearing argument shapes. -fn extract_query(arguments: Option<&Value>) -> Option { - let Some(Value::Object(map)) = arguments else { - return None; - }; - ["query", "pattern", "search", "q", "regex"] - .iter() - .find_map(|k| map.get(*k).and_then(Value::as_str)) - .map(str::to_string) -} - -#[cfg(test)] -mod tests { - use super::*; - use serde_json::json; - - #[tokio::test] - async fn skips_short_output() { - let (out, stats) = compact_tool_output_with_policy( - "shell", - None, - "hello world", - Some(0), - AgentTokenjuiceCompression::Full, - ) - .await; - assert_eq!(out, "hello world"); - assert!(!stats.applied); - assert_eq!(stats.original_bytes, 11); - } - - #[tokio::test] - async fn compacts_long_git_status_via_argv() { - let mut lines = vec!["On branch main".to_owned()]; - for i in 0..200 { - lines.push(format!("\tmodified: src/file_{i}.rs")); - } - let output = lines.join("\n"); - let args = json!({"command": "git status"}); - let (compacted, stats) = compact_tool_output_with_policy( - "shell", - Some(&args), - &output, - Some(0), - AgentTokenjuiceCompression::Full, - ) - .await; - assert!(stats.applied, "expected compaction, got {:?}", stats); - assert!(compacted.len() < output.len()); - } - - #[tokio::test] - async fn passes_through_incompressible_output() { - let unique_lines: Vec = (0..200) - .map(|i| format!("unique-payload-chunk-{i}-{}", "x".repeat(30))) - .collect(); - let output = unique_lines.join("\n"); - let (returned, stats) = compact_tool_output_with_policy( - "unknown_tool", - None, - &output, - Some(0), - AgentTokenjuiceCompression::Full, - ) - .await; - if !stats.applied { - assert_eq!(returned, output); - } - } - - #[tokio::test] - async fn disabled_flag_is_passthrough() { - let big = "x".repeat(5000); - assert_eq!(compact_output(big.clone(), "grep", false).await, big); - } - - #[tokio::test] - async fn light_agent_profile_declines_lossy_ccr_compaction() { - let mut lines = vec!["On branch main".to_owned()]; - for i in 0..200 { - lines.push(format!("\tmodified: src/file_{i}.rs")); - } - let output = lines.join("\n"); - let args = json!({"command": "git status"}); - let (returned, stats) = compact_tool_output_with_policy( - "shell", - Some(&args), - &output, - Some(0), - AgentTokenjuiceCompression::Light, - ) - .await; - assert_eq!(returned, output); - assert!(!stats.applied); - } - - #[tokio::test] - async fn off_agent_profile_bypasses_router() { - let big = "x".repeat(5000); - let returned = - compact_output_with_policy(big.clone(), "grep", true, AgentTokenjuiceCompression::Off) - .await; - assert_eq!(returned, big); - } - - #[test] - fn extract_argv_handles_common_shapes() { - let (cmd, argv) = extract_command_argv(Some(&json!({"command": "git status"}))); - assert_eq!(cmd.as_deref(), Some("git status")); - assert_eq!(argv.unwrap(), vec!["git", "status"]); - - let (cmd, _) = extract_command_argv(Some(&json!({"command": "cargo", "args": ["test"]}))); - assert_eq!(cmd.as_deref(), Some("cargo test")); - } - - #[test] - fn extract_extension_and_query() { - assert_eq!( - extract_extension(Some(&json!({"path": "src/lib.rs"}))).as_deref(), - Some("rs") - ); - assert_eq!( - extract_query(Some(&json!({"pattern": "foo bar"}))).as_deref(), - Some("foo bar") - ); - } -} diff --git a/src/openhuman/tokenjuice/types.rs b/src/openhuman/tokenjuice/types.rs deleted file mode 100644 index d72c50412..000000000 --- a/src/openhuman/tokenjuice/types.rs +++ /dev/null @@ -1,608 +0,0 @@ -//! Core type definitions for the TokenJuice reduction engine. -//! -//! These types mirror the upstream TypeScript shapes so that upstream rule JSON -//! files can be loaded without modification. All public types use -//! `#[serde(rename_all = "camelCase")]` and `#[serde(default)]` on optional -//! fields for maximum compatibility with the upstream schema. - -use serde::{Deserialize, Serialize}; -use std::collections::HashMap; - -// --------------------------------------------------------------------------- -// Rule origin -// --------------------------------------------------------------------------- - -/// Which configuration layer a rule was loaded from. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(rename_all = "camelCase")] -pub enum RuleOrigin { - Builtin, - User, - Project, -} - -// --------------------------------------------------------------------------- -// Rule sub-types -// --------------------------------------------------------------------------- - -/// Matching criteria for a rule. -#[derive(Debug, Clone, Default, Serialize, Deserialize)] -#[serde(rename_all = "camelCase")] -pub struct RuleMatch { - /// Match when `toolName` is one of these values. - #[serde(default)] - pub tool_names: Option>, - /// Match when `argv[0]` is one of these values. - #[serde(default)] - pub argv0: Option>, - /// All of these groups must each appear somewhere in `argv`. - #[serde(default)] - pub argv_includes: Option>>, - /// At least one of these groups must appear in `argv`. - #[serde(default)] - pub argv_includes_any: Option>>, - /// All of these strings must appear in `command`. - #[serde(default)] - pub command_includes: Option>, - /// At least one of these strings must appear in `command`. - #[serde(default)] - pub command_includes_any: Option>, -} - -/// Line-level filter patterns. -#[derive(Debug, Clone, Default, Serialize, Deserialize)] -#[serde(rename_all = "camelCase")] -pub struct RuleFilters { - /// Lines matching any pattern are removed. - #[serde(default)] - pub skip_patterns: Option>, - /// Only lines matching at least one pattern are kept (if any match). - #[serde(default)] - pub keep_patterns: Option>, -} - -/// Output transformation flags. -#[derive(Debug, Clone, Default, Serialize, Deserialize)] -#[serde(rename_all = "camelCase")] -pub struct RuleTransforms { - #[serde(default)] - pub strip_ansi: Option, - #[serde(default)] - pub trim_empty_edges: Option, - #[serde(default)] - pub dedupe_adjacent: Option, - #[serde(default)] - pub pretty_print_json: Option, -} - -/// Head/tail summarisation parameters. -#[derive(Debug, Clone, Default, Serialize, Deserialize)] -#[serde(rename_all = "camelCase")] -pub struct RuleSummarize { - #[serde(default)] - pub head: Option, - #[serde(default)] - pub tail: Option, -} - -/// A pattern-based line counter. -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(rename_all = "camelCase")] -pub struct RuleCounter { - pub name: String, - pub pattern: String, - /// Regex flags (e.g. `"i"` for case-insensitive). `u` is always added. - #[serde(default)] - pub flags: Option, -} - -/// Map output patterns to canned messages. -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(rename_all = "camelCase")] -pub struct RuleOutputMatch { - pub pattern: String, - pub message: String, - #[serde(default)] - pub flags: Option, -} - -/// Failure-mode overrides. -#[derive(Debug, Clone, Default, Serialize, Deserialize)] -#[serde(rename_all = "camelCase")] -pub struct RuleFailure { - #[serde(default)] - pub preserve_on_failure: Option, - #[serde(default)] - pub head: Option, - #[serde(default)] - pub tail: Option, -} - -// --------------------------------------------------------------------------- -// JsonRule — the raw deserialized form -// --------------------------------------------------------------------------- - -/// A rule as parsed from a JSON file (upstream `JsonRule`). -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(rename_all = "camelCase")] -pub struct JsonRule { - pub id: String, - pub family: String, - #[serde(default)] - pub description: Option, - #[serde(default)] - pub priority: Option, - /// Message to return when output is empty after filtering. - #[serde(default)] - pub on_empty: Option, - #[serde(default)] - pub match_output: Option>, - /// Whether counters run before or after keep-pattern filtering. - /// Upstream default is `"postKeep"`. - #[serde(default)] - pub counter_source: Option, - pub r#match: RuleMatch, - #[serde(default)] - pub filters: Option, - #[serde(default)] - pub transforms: Option, - #[serde(default)] - pub summarize: Option, - #[serde(default)] - pub counters: Option>, - #[serde(default)] - pub failure: Option, -} - -/// When to sample lines for counters — before or after keep-pattern filtering. -#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)] -#[serde(rename_all = "camelCase")] -pub enum CounterSource { - PostKeep, - PreKeep, -} - -// --------------------------------------------------------------------------- -// CompiledRule — regex patterns pre-built -// --------------------------------------------------------------------------- - -/// A compiled counter entry with the pattern pre-built. -#[derive(Debug, Clone)] -pub struct CompiledCounter { - pub name: String, - pub pattern: regex::Regex, -} - -/// A compiled output-match entry. -#[derive(Debug, Clone)] -pub struct CompiledOutputMatch { - pub pattern: regex::Regex, - pub message: String, -} - -/// The compiled form of a rule (regex patterns pre-built at load time). -#[derive(Debug, Clone)] -pub struct CompiledParts { - pub skip_patterns: Vec, - pub keep_patterns: Vec, - pub counters: Vec, - pub output_matches: Vec, -} - -/// A `JsonRule` paired with its pre-compiled regex patterns plus provenance. -#[derive(Debug, Clone)] -pub struct CompiledRule { - pub rule: JsonRule, - pub source: RuleOrigin, - /// Filesystem path (or `"builtin:"` for embedded rules). - pub path: String, - pub compiled: CompiledParts, -} - -// --------------------------------------------------------------------------- -// ToolExecutionInput -// --------------------------------------------------------------------------- - -/// Describes a tool invocation whose output is to be reduced. -#[derive(Debug, Clone, Default, Serialize, Deserialize)] -#[serde(rename_all = "camelCase")] -pub struct ToolExecutionInput { - pub tool_name: String, - #[serde(default)] - pub tool_call_id: Option, - #[serde(default)] - pub run_id: Option, - #[serde(default)] - pub command: Option, - #[serde(default)] - pub argv: Option>, - #[serde(default)] - pub args: Option>, - #[serde(default)] - pub cwd: Option, - #[serde(default)] - pub partial: Option, - #[serde(default)] - pub stdout: Option, - #[serde(default)] - pub stderr: Option, - #[serde(default)] - pub combined_text: Option, - #[serde(default)] - pub exit_code: Option, - #[serde(default)] - pub started_at: Option, - #[serde(default)] - pub finished_at: Option, - #[serde(default)] - pub duration_ms: Option, - #[serde(default)] - pub metadata: Option>, -} - -// --------------------------------------------------------------------------- -// ReduceOptions -// --------------------------------------------------------------------------- - -/// Options for the `reduce_execution` pipeline. -#[derive(Debug, Clone, Default, Serialize, Deserialize)] -#[serde(rename_all = "camelCase")] -pub struct ReduceOptions { - /// Force a specific rule ID instead of auto-classification. - #[serde(default)] - pub classifier: Option, - /// Maximum inline character count (default: 1200). - #[serde(default)] - pub max_inline_chars: Option, - /// Return raw text without reduction. - #[serde(default)] - pub raw: Option, - /// Working directory for project-layer rule discovery. - #[serde(default)] - pub cwd: Option, -} - -// --------------------------------------------------------------------------- -// CompactResult -// --------------------------------------------------------------------------- - -/// Statistics produced by the reduction pipeline. -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(rename_all = "camelCase")] -pub struct ReductionStats { - pub raw_chars: usize, - pub reduced_chars: usize, - pub ratio: f64, -} - -/// The classification decision made during reduction. -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(rename_all = "camelCase")] -pub struct ClassificationResult { - pub family: String, - pub confidence: f64, - #[serde(default)] - pub matched_reducer: Option, -} - -/// The output of `reduce_execution`. -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(rename_all = "camelCase")] -pub struct CompactResult { - /// The compacted text to inline into LLM context. - pub inline_text: String, - /// A shorter preview (the intermediate summary before clamping). - #[serde(default)] - pub preview_text: Option, - /// Named counts extracted by counters. - #[serde(default)] - pub facts: Option>, - pub stats: ReductionStats, - pub classification: ClassificationResult, -} - -/// Per-agent TokenJuice profile. -/// -/// `Auto` is resolved by the agent definition layer. TokenJuice itself treats -/// `Auto` like `Full` so non-agent callers keep the global `[tokenjuice]` -/// behaviour unless they explicitly pass a narrower profile. -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize, Default)] -#[serde(rename_all = "snake_case")] -pub enum AgentTokenjuiceCompression { - /// Let the agent definition/runtime choose. Coding agents resolve this to - /// [`Self::Light`]; other agents resolve to [`Self::Full`]. - #[default] - Auto, - /// Use the process-global TokenJuice configuration unchanged. - Full, - /// Keep only non-lossy reductions; disables CCR-backed lossy compaction. - Light, - /// Bypass TokenJuice for this agent's tool results. - Off, -} - -impl AgentTokenjuiceCompression { - pub fn as_str(self) -> &'static str { - match self { - Self::Auto => "auto", - Self::Full => "full", - Self::Light => "light", - Self::Off => "off", - } - } -} - -// --------------------------------------------------------------------------- -// Content Router (TokenJuice 2.0) — content-kind detection + compressor dispatch -// --------------------------------------------------------------------------- - -/// The kind of content a blob holds, as decided by the detector. Drives which -/// [`crate::openhuman::tokenjuice::compressors::Compressor`] the router picks. -/// -/// Inspired by Headroom's content router: each kind has a specialised -/// compressor tuned to preserve the signal that kind carries (errors in logs, -/// changed hunks in diffs, signatures in code, …). -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(rename_all = "camelCase")] -pub enum ContentKind { - /// JSON array/object payload → tabular SmartCrusher. - Json, - /// Source code → AST/heuristic signature keeper. - Code, - /// Build / test / lint log → keep failures, drop passing noise. - Log, - /// grep / ripgrep style `path:line:content` matches → relevance rank. - Search, - /// Unified git diff / patch → keep changed hunks, collapse context. - Diff, - /// HTML document → strip markup to readable text. - Html, - /// Anything else → ML text compressor (if enabled) or pass-through. - PlainText, -} - -impl ContentKind { - /// Stable lower-case label for logs / RPC / stats. - pub fn as_str(self) -> &'static str { - match self { - ContentKind::Json => "json", - ContentKind::Code => "code", - ContentKind::Log => "log", - ContentKind::Search => "search", - ContentKind::Diff => "diff", - ContentKind::Html => "html", - ContentKind::PlainText => "plain_text", - } - } -} - -/// A caller-supplied prior about a blob's content, so the detector doesn't have -/// to work from scratch. Any field may be `None`; the detector resolves what it -/// can and falls back to structural heuristics. An `explicit` kind is a hard -/// override and skips detection entirely. -#[derive(Debug, Clone, Default)] -pub struct ContentHint { - /// MIME type if known (`text/html`, `application/json`, …). - pub mime: Option, - /// File extension without the dot (`rs`, `ts`, `py`, `json`, `html`, `diff`). - pub extension: Option, - /// The agent-level tool that produced the content (`grep`, `run_tests`, …). - pub source_tool: Option, - /// A search/query string associated with the content, when known (used by - /// the search compressor to rank matches by query-term density). - pub query: Option, - /// Hard override — when set, detection returns this kind verbatim. - pub explicit: Option, -} - -impl ContentHint { - /// Convenience: a hint carrying only the producing tool name. - pub fn for_tool(tool_name: impl Into) -> Self { - Self { - source_tool: Some(tool_name.into()), - ..Default::default() - } - } -} - -/// Which compressor actually produced an output. Recorded in stats / logs so a -/// human (or the debug controller) can see what the router chose. -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] -#[serde(rename_all = "camelCase")] -pub enum CompressorKind { - /// JSON array→table crusher. - SmartCrusher, - /// AST/heuristic code-signature keeper. - Code, - /// Log keep-failures compressor (and the rule engine for command output). - Log, - /// Search relevance ranker. - Search, - /// Unified-diff context collapser. - Diff, - /// HTML→text extractor. - Html, - /// ML (Python/ModernBERT) plain-text compressor. - MlText, - /// Line-oriented head/tail fallback. - Generic, - /// No compressor fired — pass-through. - None, -} - -impl CompressorKind { - /// Stable lower-case label for stats / logs / RPC. - pub fn as_str(self) -> &'static str { - match self { - CompressorKind::SmartCrusher => "smartcrusher", - CompressorKind::Code => "code", - CompressorKind::Log => "log", - CompressorKind::Search => "search", - CompressorKind::Diff => "diff", - CompressorKind::Html => "html", - CompressorKind::MlText => "ml_text", - CompressorKind::Generic => "generic", - CompressorKind::None => "none", - } - } -} - -/// Input handed to a [`crate::openhuman::tokenjuice::compressors::Compressor`]. -/// Borrows the content to avoid copies on the hot path. -#[derive(Debug, Clone)] -pub struct CompressInput<'a> { - /// The raw content to compress. - pub content: &'a str, - /// The detected (or hinted) content kind. - pub kind: ContentKind, - /// The original caller hint (carries `query`, `source_tool`, argv-derived - /// command for the log/command path, …). - pub hint: &'a ContentHint, - /// Process exit code if this is command output — enables failure-preserving - /// behaviour in the log compressor. - pub exit_code: Option, - /// Derived shell command (joined argv) for the log/command rule path, if any. - pub command: Option, - /// Derived argv for the log/command rule path, if any. - pub argv: Option>, - /// Original byte length (== `content.len()`; cached for convenience). - pub original_bytes: usize, -} - -/// Output of a compressor. `text` is the compacted body **without** any CCR -/// retrieval footer — the router ([`crate::openhuman::tokenjuice::compress`]) -/// adds the marker after offloading the original. -#[derive(Debug, Clone)] -pub struct CompressOutput { - /// The compacted body. - pub text: String, - /// True when data was dropped (vs. a faithful reformat). Changes the footer - /// wording and whether the original is mandatory for fidelity. - pub lossy: bool, - /// Which compressor produced this. - pub kind: CompressorKind, - /// Optional named counts (e.g. error/warning tallies). - pub facts: Option>, -} - -impl CompressOutput { - /// A faithful reformat — every value preserved, only layout changed. - pub fn reformatted(text: String, kind: CompressorKind) -> Self { - Self { - text, - lossy: false, - kind, - facts: None, - } - } - - /// A lossy view — data was dropped; the original must be offloaded for recovery. - pub fn lossy(text: String, kind: CompressorKind) -> Self { - Self { - text, - lossy: true, - kind, - facts: None, - } - } -} - -/// Knobs for the router and compressors, built by the caller from the -/// `[tokenjuice]` config block. TokenJuice stays decoupled from the config -/// schema crate by taking this plain struct rather than `Config`. -#[derive(Debug, Clone)] -pub struct CompressOptions { - /// Master switch — when false, [`crate::openhuman::tokenjuice::compress_content`] - /// is a pass-through. - pub router_enabled: bool, - /// Whether to offload originals to CCR and emit retrieval markers. - pub ccr_enabled: bool, - /// Per-compressor toggles. - pub search_enabled: bool, - pub code_enabled: bool, - pub html_enabled: bool, - /// Whether the ML plain-text compressor may be used (further gated at - /// runtime by Python/runtime_python_server availability). - pub ml_text_enabled: bool, - /// Outputs below this many bytes are never compressed. - pub min_bytes_to_compress: usize, - /// CCR only fires (offload original + lossy compression) when the input is - /// estimated to be at least this many tokens. Below it, the result passes - /// through (lossless reformats may still apply without offload). Lets small - /// tool results skip the cache entirely. - pub ccr_min_tokens: usize, - /// Maximum inline character count for the generic/rule fallback path. - pub max_inline_chars: Option, -} - -impl Default for CompressOptions { - fn default() -> Self { - Self { - router_enabled: true, - ccr_enabled: true, - search_enabled: true, - code_enabled: true, - html_enabled: true, - ml_text_enabled: false, - min_bytes_to_compress: 2048, - ccr_min_tokens: 500, - max_inline_chars: None, - } - } -} - -/// The result of the universal [`crate::openhuman::tokenjuice::compress_content`] -/// entry point: the compacted text (with any CCR footer already appended), plus -/// metadata for callers/stats. -#[derive(Debug, Clone)] -pub struct CompressedOutput { - /// Final text to inline into context (includes the retrieval footer when lossy). - pub text: String, - /// The detected content kind. - pub content_kind: ContentKind, - /// Which compressor fired (`None` ⇒ pass-through). - pub compressor: CompressorKind, - /// Whether the output dropped data. - pub lossy: bool, - /// True if the router actually changed the content. - pub applied: bool, - /// CCR token for the offloaded original, if one was stored. - pub ccr_token: Option, - /// Original byte length. - pub original_bytes: usize, - /// Compacted byte length (of `text`). - pub compacted_bytes: usize, -} - -impl CompressedOutput { - /// Build a pass-through result that didn't change `content`. - pub fn passthrough(content: String, kind: ContentKind) -> Self { - let len = content.len(); - Self { - text: content, - content_kind: kind, - compressor: CompressorKind::None, - lossy: false, - applied: false, - ccr_token: None, - original_bytes: len, - compacted_bytes: len, - } - } -} - -// --------------------------------------------------------------------------- -// RuleFixture — used by integration tests -// --------------------------------------------------------------------------- - -/// A test fixture mirroring the upstream `RuleFixture` shape. -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(rename_all = "camelCase")] -pub struct RuleFixture { - pub input: ToolExecutionInput, - pub expected_output: String, - #[serde(default)] - pub description: Option, - #[serde(default)] - pub options: Option, -} diff --git a/src/openhuman/tokenjuice/vendor/README.md b/src/openhuman/tokenjuice/vendor/README.md deleted file mode 100644 index e0ede6eda..000000000 --- a/src/openhuman/tokenjuice/vendor/README.md +++ /dev/null @@ -1,68 +0,0 @@ -# Vendored TokenJuice Rules - -These JSON rule files are vendored from the upstream -[vincentkoc/tokenjuice](https://github.com/vincentkoc/tokenjuice) repository. - -## Upstream - -- Repository: https://github.com/vincentkoc/tokenjuice -- Upstream path: `src/rules/**/*.json` -- Licence: MIT (Copyright (c) 2026 Vincent Koc) - -## Licence note - -The upstream project is MIT-licensed. The full licence text is reproduced below. - -``` -MIT License - -Copyright (c) 2026 Vincent Koc - -Permission is hereby granted, free of charge, to any person obtaining a copy -of this software and associated documentation files (the "Software"), to deal -in the Software without restriction, including without limitation the rights -to use, copy, modify, merge, publish, distribute, sublicense, and/or sell -copies of the Software, and to permit persons to whom the Software is -furnished to do so, subject to the following conditions: - -The above copyright notice and this permission notice shall be included in all -copies or substantial portions of the Software. - -THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE -SOFTWARE. -``` - -## File naming convention - -Upstream files live in subdirectory paths like `git/status.json`. Because we -embed all rules in a single directory here, `/` in the id is replaced with `__` -in the filename (e.g. `git/status.json` → `git__status.json`). - -## Rules vendored - -**96 rules** are vendored here, representing the complete set of generic rules -from the upstream repository as of 2026-04-17. - -### Exclusions - -The `src/rules/openclaw/` subdirectory in upstream is **not** vendored. Those -rules (`openclaw/sessions-history`, etc.) are specific to the upstream author's -proprietary OpenClaw tooling and are not generic enough to include in the -OpenHuman builtin set. The `fixtures/` subdirectory is also excluded — fixture -files are test-only and carry no runtime behaviour. - -### Adding more rules - -Additional rules from the upstream repository can be added by: - -1. Copying the JSON verbatim into this directory using the `family__name.json` - naming convention. -2. Adding the corresponding `(id, include_str!(...))` entry to - `rules/builtin.rs`, keeping the list alphabetically ordered by id. -3. Running `cargo check` and `cargo test tokenjuice` to confirm the new rule - compiles cleanly. diff --git a/src/openhuman/tokenjuice/vendor/rules/archive__tar.json b/src/openhuman/tokenjuice/vendor/rules/archive__tar.json deleted file mode 100644 index ae14dcbe2..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/archive__tar.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "archive/tar", - "family": "archive-cli", - "description": "Compact tar output while preserving archive paths and error lines.", - "match": { - "toolNames": ["exec"], - "argv0": ["tar"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "error|failed|cannot", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/archive__unzip.json b/src/openhuman/tokenjuice/vendor/rules/archive__unzip.json deleted file mode 100644 index ce7dcfd67..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/archive__unzip.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "archive/unzip", - "family": "archive-cli", - "description": "Compact unzip output while preserving extracted paths and conflict lines.", - "match": { - "toolNames": ["exec"], - "argv0": ["unzip"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "warning", - "pattern": "inflating|extracting|replace|error", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/archive__zip.json b/src/openhuman/tokenjuice/vendor/rules/archive__zip.json deleted file mode 100644 index 5e3a76429..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/archive__zip.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "archive/zip", - "family": "archive-cli", - "description": "Compact zip output while preserving archived paths and warnings.", - "match": { - "toolNames": ["exec"], - "argv0": ["zip"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "warning", - "pattern": "adding|updating|warning|error", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/build__cargo-build.json b/src/openhuman/tokenjuice/vendor/rules/build__cargo-build.json deleted file mode 100644 index 57e502a60..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/build__cargo-build.json +++ /dev/null @@ -1,58 +0,0 @@ -{ - "id": "build/cargo-build", - "family": "build-rust", - "description": "Compact cargo build and cargo check output while preserving compiler diagnostics.", - "match": { - "toolNames": ["exec"], - "argv0": ["cargo"], - "argvIncludesAny": [["build"], ["check"]] - }, - "onEmpty": "cargo: build succeeded", - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "skipPatterns": [ - "^\\s*Compiling .+", - "^\\s*Checking .+", - "^\\s*Downloading .+", - "^\\s*Downloaded .+", - "^\\s*Locking .+", - "^\\s*Updating .+", - "^\\s*Fresh .+", - "^\\s*Packaging .+", - "^\\s+\\|\\s*$", - "^\\s*For more information about this error", - "^\\s*Some errors have detailed explanations" - ] - }, - "summarize": { - "head": 12, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 16, - "tail": 16 - }, - "counters": [ - { - "name": "error", - "pattern": "^error(\\[E\\d+\\])?:", - "flags": "i" - }, - { - "name": "warning", - "pattern": "^warning(\\[.+\\])?:", - "flags": "i" - } - ], - "matchOutput": [ - { - "pattern": "^\\s*Finished .+\\s*$", - "message": "cargo: build succeeded" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/build__cargo-doc.json b/src/openhuman/tokenjuice/vendor/rules/build__cargo-doc.json deleted file mode 100644 index 7b42b34e4..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/build__cargo-doc.json +++ /dev/null @@ -1,55 +0,0 @@ -{ - "id": "build/cargo-doc", - "family": "build-rust", - "description": "Compact cargo doc output while preserving documentation warnings and errors.", - "match": { - "toolNames": ["exec"], - "argv0": ["cargo"], - "argvIncludes": [["doc"]] - }, - "onEmpty": "cargo doc: generated successfully", - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "skipPatterns": [ - "^\\s*Compiling .+", - "^\\s*Checking .+", - "^\\s*Documenting .+", - "^\\s*Downloading .+", - "^\\s*Downloaded .+", - "^\\s*Locking .+", - "^\\s*Fresh .+", - "^\\s+\\|\\s*$" - ] - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "^error(\\[E\\d+\\])?:", - "flags": "i" - }, - { - "name": "warning", - "pattern": "^warning:", - "flags": "i" - } - ], - "matchOutput": [ - { - "pattern": "^\\s*Finished .+\\s*$", - "message": "cargo doc: generated successfully" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/build__esbuild.json b/src/openhuman/tokenjuice/vendor/rules/build__esbuild.json deleted file mode 100644 index 2b52b233b..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/build__esbuild.json +++ /dev/null @@ -1,35 +0,0 @@ -{ - "id": "build/esbuild", - "family": "build-bundler", - "description": "Compact esbuild and tsdown-like output while preserving actual errors.", - "match": { - "toolNames": ["exec"], - "commandIncludes": ["esbuild"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 12, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 16, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "error", - "flags": "i" - }, - { - "name": "warning", - "pattern": "warning", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/build__tsc.json b/src/openhuman/tokenjuice/vendor/rules/build__tsc.json deleted file mode 100644 index 1a6d7c927..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/build__tsc.json +++ /dev/null @@ -1,74 +0,0 @@ -{ - "id": "build/tsc", - "family": "build-typescript", - "description": "Compact TypeScript compiler output while preserving real diagnostics.", - "match": { - "toolNames": ["exec"], - "commandIncludes": ["tsc"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "skipPatterns": [ - "^Files:\\s+\\d+", - "^Lines of Library:\\s+\\d+", - "^Lines of Definitions:\\s+\\d+", - "^Lines of TypeScript:\\s+\\d+", - "^Lines of JavaScript:\\s+\\d+", - "^Lines of JSON:\\s+\\d+", - "^Lines of Other:\\s+\\d+", - "^Identifiers:\\s+\\d+", - "^Symbols:\\s+\\d+", - "^Types:\\s+\\d+", - "^Instantiations:\\s+\\d+", - "^Memory used:\\s+.+", - "^Assignability cache size:\\s+\\d+", - "^Identity cache size:\\s+\\d+", - "^Subtype cache size:\\s+\\d+", - "^Strict subtype cache size:\\s+\\d+", - "^I/O Read time:\\s+.+", - "^Parse time:\\s+.+", - "^ResolveModule time:\\s+.+", - "^ResolveLibrary time:\\s+.+", - "^Program time:\\s+.+", - "^Bind time:\\s+.+", - "^Check time:\\s+.+", - "^transformTime time:\\s+.+", - "^commentTime time:\\s+.+", - "^I/O Write time:\\s+.+", - "^printTime time:\\s+.+", - "^Emit time:\\s+.+", - "^Total time:\\s+.+", - "^Watching for file changes\\." - ], - "keepPatterns": [ - "^.+\\(\\d+,\\d+\\):\\s+error TS\\d+: .+", - "^.+\\(\\d+,\\d+\\):\\s+warning TS\\d+: .+", - "^Found \\d+ errors?.+", - "^error TS\\d+: .+" - ] - }, - "summarize": { - "head": 4, - "tail": 4 - }, - "failure": { - "preserveOnFailure": true, - "head": 4, - "tail": 6 - }, - "counters": [ - { - "name": "typescript error", - "pattern": "TS\\d+" - }, - { - "name": "error", - "pattern": "error", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/build__tsdown.json b/src/openhuman/tokenjuice/vendor/rules/build__tsdown.json deleted file mode 100644 index 7e21ac182..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/build__tsdown.json +++ /dev/null @@ -1,35 +0,0 @@ -{ - "id": "build/tsdown", - "family": "build-bundler", - "description": "Compact tsdown build output while preserving warnings and failures.", - "match": { - "toolNames": ["exec"], - "commandIncludes": ["tsdown"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 12, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 16, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "error", - "flags": "i" - }, - { - "name": "warning", - "pattern": "warning", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/build__vite.json b/src/openhuman/tokenjuice/vendor/rules/build__vite.json deleted file mode 100644 index 9ec8f2283..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/build__vite.json +++ /dev/null @@ -1,42 +0,0 @@ -{ - "id": "build/vite", - "family": "build-bundler", - "description": "Compact vite build output while preserving warnings and failures.", - "match": { - "toolNames": ["exec"], - "commandIncludes": ["vite", "build"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "skipPatterns": [ - "^transforming \\(.+\\) .+", - "^rendering chunks \\(.+\\) .+", - "^computing gzip size \\(.+\\) .+" - ] - }, - "summarize": { - "head": 12, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 16, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "error", - "flags": "i" - }, - { - "name": "warning", - "pattern": "warning", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/build__webpack.json b/src/openhuman/tokenjuice/vendor/rules/build__webpack.json deleted file mode 100644 index 3ab4325f7..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/build__webpack.json +++ /dev/null @@ -1,51 +0,0 @@ -{ - "id": "build/webpack", - "family": "build-bundler", - "description": "Compact webpack output while preserving module errors and warnings.", - "match": { - "toolNames": ["exec"], - "commandIncludes": ["webpack"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "keepPatterns": [ - "^Entrypoint\\s+.+", - "^ERROR in .+", - "^WARNING in .+", - "^Module .+", - "^\\s*ERROR\\s+in\\s+.+", - "^\\s*webpack\\s+\\d+\\.\\d+\\.\\d+ compiled .+", - "^\\s*\\d+ errors? have detailed information.+" - ] - }, - "summarize": { - "head": 4, - "tail": 6 - }, - "failure": { - "preserveOnFailure": true, - "head": 4, - "tail": 8 - }, - "counters": [ - { - "name": "asset", - "pattern": "^asset\\s+.+", - "flags": "m" - }, - { - "name": "error", - "pattern": "error", - "flags": "i" - }, - { - "name": "warning", - "pattern": "warning", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/cloud__aws.json b/src/openhuman/tokenjuice/vendor/rules/cloud__aws.json deleted file mode 100644 index 905b19549..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/cloud__aws.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "cloud/aws", - "family": "cloud-cli", - "description": "Compact AWS CLI output while preserving result rows and service errors.", - "match": { - "toolNames": ["exec"], - "argv0": ["aws"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "error|exception|denied|not found", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/cloud__az.json b/src/openhuman/tokenjuice/vendor/rules/cloud__az.json deleted file mode 100644 index 8cdc3d2dc..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/cloud__az.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "cloud/az", - "family": "cloud-cli", - "description": "Compact Azure CLI output while preserving key resource rows and deployment failures.", - "match": { - "toolNames": ["exec"], - "argv0": ["az"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "error|failed|forbidden|not found", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/cloud__flyctl.json b/src/openhuman/tokenjuice/vendor/rules/cloud__flyctl.json deleted file mode 100644 index 5d13eab3b..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/cloud__flyctl.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "cloud/flyctl", - "family": "deploy-cli", - "description": "Compact Fly output while preserving machine, app, and rollout status lines.", - "match": { - "toolNames": ["exec"], - "argv0": ["fly", "flyctl"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "error|failed|unhealthy|warning", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/cloud__gcloud.json b/src/openhuman/tokenjuice/vendor/rules/cloud__gcloud.json deleted file mode 100644 index 8e1b19c0c..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/cloud__gcloud.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "cloud/gcloud", - "family": "cloud-cli", - "description": "Compact gcloud output while preserving resource tables and API failures.", - "match": { - "toolNames": ["exec"], - "argv0": ["gcloud"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "error|failed|permission|denied", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/cloud__gh.json b/src/openhuman/tokenjuice/vendor/rules/cloud__gh.json deleted file mode 100644 index 7eaf450d6..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/cloud__gh.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "cloud/gh", - "family": "developer-cli", - "description": "Compact GitHub CLI output while preserving issue, PR, and workflow result lines.", - "match": { - "toolNames": ["exec"], - "argv0": ["gh"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "error|failed|not found|forbidden", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/cloud__vercel.json b/src/openhuman/tokenjuice/vendor/rules/cloud__vercel.json deleted file mode 100644 index e9cec25c9..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/cloud__vercel.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "cloud/vercel", - "family": "deploy-cli", - "description": "Compact Vercel CLI output while preserving deployment URLs and error details.", - "match": { - "toolNames": ["exec"], - "argv0": ["vercel"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "error|failed|canceled|timed out", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/database__mongosh.json b/src/openhuman/tokenjuice/vendor/rules/database__mongosh.json deleted file mode 100644 index e6f5b09c7..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/database__mongosh.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "database/mongosh", - "family": "database-cli", - "description": "Compact mongosh output while preserving collection results and query errors.", - "match": { - "toolNames": ["exec"], - "argv0": ["mongosh"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "error|failed|exception", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/database__mysql.json b/src/openhuman/tokenjuice/vendor/rules/database__mysql.json deleted file mode 100644 index 75e6aac03..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/database__mysql.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "database/mysql", - "family": "database-cli", - "description": "Compact mysql output while preserving query rows and SQL errors.", - "match": { - "toolNames": ["exec"], - "argv0": ["mysql"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "error|failed|denied|unknown", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/database__psql.json b/src/openhuman/tokenjuice/vendor/rules/database__psql.json deleted file mode 100644 index 01cc3b67f..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/database__psql.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "database/psql", - "family": "database-cli", - "description": "Compact psql output while preserving result tables and query errors.", - "match": { - "toolNames": ["exec"], - "argv0": ["psql"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "error|failed|permission denied", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/database__redis-cli.json b/src/openhuman/tokenjuice/vendor/rules/database__redis-cli.json deleted file mode 100644 index a0aba8662..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/database__redis-cli.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "database/redis-cli", - "family": "database-cli", - "description": "Compact redis-cli output while preserving command replies and connection failures.", - "match": { - "toolNames": ["exec"], - "argv0": ["redis-cli"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 8, - "tail": 6 - }, - "failure": { - "preserveOnFailure": true, - "head": 10, - "tail": 10 - }, - "counters": [ - { - "name": "error", - "pattern": "error|denied|could not connect", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/database__sqlite3.json b/src/openhuman/tokenjuice/vendor/rules/database__sqlite3.json deleted file mode 100644 index c20a1bc8f..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/database__sqlite3.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "database/sqlite3", - "family": "database-cli", - "description": "Compact sqlite3 output while preserving query rows and parse errors.", - "match": { - "toolNames": ["exec"], - "argv0": ["sqlite3"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 8, - "tail": 6 - }, - "failure": { - "preserveOnFailure": true, - "head": 10, - "tail": 10 - }, - "counters": [ - { - "name": "error", - "pattern": "error|failed|no such table", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/devops__docker-build.json b/src/openhuman/tokenjuice/vendor/rules/devops__docker-build.json deleted file mode 100644 index c9b4c397f..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/devops__docker-build.json +++ /dev/null @@ -1,53 +0,0 @@ -{ - "id": "devops/docker-build", - "family": "container-build", - "description": "Compact docker build output while preserving real failures and final stages.", - "match": { - "toolNames": ["exec"], - "argv0": ["docker"], - "argvIncludes": [["build"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "skipPatterns": [ - "^#\\d+\\s+[0-9.]+\\s", - "^#\\d+\\s+extracting\\s", - "^#\\d+\\s+sha256:" - ], - "keepPatterns": [ - "^#\\d+\\s+\\[", - "^#\\d+\\s+DONE\\s", - "^#\\d+\\s+ERROR:", - "^ERROR:", - "^ => ", - "^exporting to image$", - "^writing image", - "^naming to " - ] - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "step", - "pattern": "^#\\d+\\s+\\[", - "flags": "m" - }, - { - "name": "error", - "pattern": "error", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/devops__docker-compose.json b/src/openhuman/tokenjuice/vendor/rules/devops__docker-compose.json deleted file mode 100644 index 3fcec7e7f..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/devops__docker-compose.json +++ /dev/null @@ -1,44 +0,0 @@ -{ - "id": "devops/docker-compose", - "family": "container-compose", - "description": "Compact docker compose output while preserving service rows, status, and failures.", - "match": { - "toolNames": ["exec"], - "argv0": ["docker"], - "argvIncludes": [["compose"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "keepPatterns": [ - "error|warn|failed|unhealthy|exited|orphan", - "^(NAME|SERVICE|CONTAINER ID)\\s+", - "^[-a-zA-Z0-9_.]+\\s+.+", - "^\\s*\\d+ services?\\s+", - "^\\s*\\d+ containers?\\s+" - ] - }, - "summarize": { - "head": 8, - "tail": 6 - }, - "failure": { - "preserveOnFailure": true, - "head": 10, - "tail": 10 - }, - "counters": [ - { - "name": "service", - "pattern": "^(?!NAME\\s|SERVICE\\s|CONTAINER ID\\s).+\\S.*$" - }, - { - "name": "error", - "pattern": "error|failed|unhealthy|exited", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/devops__docker-images.json b/src/openhuman/tokenjuice/vendor/rules/devops__docker-images.json deleted file mode 100644 index e050260b6..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/devops__docker-images.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "devops/docker-images", - "family": "container-images", - "description": "Compact docker images output while preserving image rows.", - "match": { - "toolNames": ["exec"], - "argv0": ["docker"], - "argvIncludes": [["images"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 8, - "tail": 6 - }, - "failure": { - "preserveOnFailure": true, - "head": 12, - "tail": 12 - }, - "counters": [ - { - "name": "image", - "pattern": "^(?!REPOSITORY\\s).+\\S.*$" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/devops__docker-logs.json b/src/openhuman/tokenjuice/vendor/rules/devops__docker-logs.json deleted file mode 100644 index b30a22bcd..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/devops__docker-logs.json +++ /dev/null @@ -1,43 +0,0 @@ -{ - "id": "devops/docker-logs", - "family": "container-logs", - "description": "Compact docker logs output while preserving early and late log lines.", - "match": { - "toolNames": ["exec"], - "argv0": ["docker"], - "argvIncludes": [["logs"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "keepPatterns": [ - "error|warn|fatal|panic|exception|traceback|timeout|refused|fail", - "^Caused by:", - "^Traceback" - ] - }, - "summarize": { - "head": 8, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "error", - "flags": "i" - }, - { - "name": "warning", - "pattern": "warn", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/devops__docker-ps.json b/src/openhuman/tokenjuice/vendor/rules/devops__docker-ps.json deleted file mode 100644 index 1eea4fc12..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/devops__docker-ps.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "devops/docker-ps", - "family": "container-list", - "description": "Compact docker ps output while preserving container rows.", - "match": { - "toolNames": ["exec"], - "argv0": ["docker"], - "argvIncludes": [["ps"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 8, - "tail": 6 - }, - "failure": { - "preserveOnFailure": true, - "head": 12, - "tail": 12 - }, - "counters": [ - { - "name": "container", - "pattern": "^(?!CONTAINER ID\\s).+\\S.*$" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/devops__kubectl-describe.json b/src/openhuman/tokenjuice/vendor/rules/devops__kubectl-describe.json deleted file mode 100644 index 21a42f54b..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/devops__kubectl-describe.json +++ /dev/null @@ -1,45 +0,0 @@ -{ - "id": "devops/kubectl-describe", - "family": "kubernetes-describe", - "description": "Compact kubectl describe output while preserving metadata, status, events, and failures.", - "match": { - "toolNames": ["exec"], - "argv0": ["kubectl"], - "argvIncludes": [["describe"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "keepPatterns": [ - "^(Name|Namespace|Priority|Node|Status|IP|Controlled By|Containers|Conditions|Events):", - "^\\s*(Type|Reason|Age|From|Message)\\s+", - "error|warn|failed|back-off|crashloop|unhealthy|timeout", - "^\\s*Warning\\s+", - "^\\s*Normal\\s+" - ] - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 12, - "tail": 12 - }, - "counters": [ - { - "name": "warning", - "pattern": "warning|back-off|failed|unhealthy", - "flags": "i" - }, - { - "name": "event", - "pattern": "^\\s*(Warning|Normal)\\s+", - "flags": "m" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/devops__kubectl-get.json b/src/openhuman/tokenjuice/vendor/rules/devops__kubectl-get.json deleted file mode 100644 index 5e6eccb24..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/devops__kubectl-get.json +++ /dev/null @@ -1,35 +0,0 @@ -{ - "id": "devops/kubectl-get", - "family": "kubernetes-list", - "description": "Compact kubectl get output while preserving resource rows.", - "match": { - "toolNames": ["exec"], - "argv0": ["kubectl"], - "argvIncludes": [["get"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "skipPatterns": [ - "^No resources found" - ] - }, - "summarize": { - "head": 8, - "tail": 6 - }, - "failure": { - "preserveOnFailure": true, - "head": 12, - "tail": 12 - }, - "counters": [ - { - "name": "resource", - "pattern": "^(?!NAME\\s).+\\S.*$" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/devops__kubectl-logs.json b/src/openhuman/tokenjuice/vendor/rules/devops__kubectl-logs.json deleted file mode 100644 index 3a17c1b53..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/devops__kubectl-logs.json +++ /dev/null @@ -1,43 +0,0 @@ -{ - "id": "devops/kubectl-logs", - "family": "kubernetes-logs", - "description": "Compact kubectl logs output while preserving key log lines.", - "match": { - "toolNames": ["exec"], - "argv0": ["kubectl"], - "argvIncludes": [["logs"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "keepPatterns": [ - "error|warn|fatal|panic|exception|traceback|timeout|refused|fail", - "^Caused by:", - "^Traceback" - ] - }, - "summarize": { - "head": 8, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "error", - "flags": "i" - }, - { - "name": "warning", - "pattern": "warn", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/filesystem__find.json b/src/openhuman/tokenjuice/vendor/rules/filesystem__find.json deleted file mode 100644 index bbb813251..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/filesystem__find.json +++ /dev/null @@ -1,42 +0,0 @@ -{ - "id": "filesystem/find", - "family": "filesystem-find", - "description": "Compact find output while preserving matches and failure context.", - "match": { - "toolNames": ["exec"], - "argv0": ["find"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "keepPatterns": [ - "^\\./.+", - "^/.+", - "Permission denied", - "No such file" - ] - }, - "summarize": { - "head": 8, - "tail": 6 - }, - "failure": { - "preserveOnFailure": true, - "head": 12, - "tail": 10 - }, - "counters": [ - { - "name": "match", - "pattern": "^(?!find: ).+\\S.*$" - }, - { - "name": "permission denied", - "pattern": "Permission denied", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/filesystem__ls.json b/src/openhuman/tokenjuice/vendor/rules/filesystem__ls.json deleted file mode 100644 index 80f24eb34..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/filesystem__ls.json +++ /dev/null @@ -1,29 +0,0 @@ -{ - "id": "filesystem/ls", - "family": "filesystem-listing", - "description": "Compact ls output for directory listings.", - "match": { - "toolNames": ["exec"], - "argv0": ["ls"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 8, - "tail": 6 - }, - "failure": { - "preserveOnFailure": true, - "head": 12, - "tail": 10 - }, - "counters": [ - { - "name": "item", - "pattern": "^(?!total\\s+\\d+).+\\S.*$" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/generic__fallback.json b/src/openhuman/tokenjuice/vendor/rules/generic__fallback.json deleted file mode 100644 index c884a5693..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/generic__fallback.json +++ /dev/null @@ -1,32 +0,0 @@ -{ - "id": "generic/fallback", - "family": "generic", - "description": "Generic fallback reducer for line-oriented output.", - "match": {}, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 8, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 12, - "tail": 20 - }, - "counters": [ - { - "name": "error", - "pattern": "error", - "flags": "i" - }, - { - "name": "warning", - "pattern": "warning", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/generic__help.json b/src/openhuman/tokenjuice/vendor/rules/generic__help.json deleted file mode 100644 index 5c3e810aa..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/generic__help.json +++ /dev/null @@ -1,25 +0,0 @@ -{ - "id": "generic/help", - "family": "help", - "description": "Preserve command help output so agents can inspect available commands and flags.", - "priority": 25, - "match": { - "toolNames": ["exec"], - "argvIncludesAny": [["--help"], ["help"]], - "commandIncludesAny": [" --help", " help"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 80, - "tail": 40 - }, - "failure": { - "preserveOnFailure": true, - "head": 80, - "tail": 40 - } -} diff --git a/src/openhuman/tokenjuice/vendor/rules/git__branch.json b/src/openhuman/tokenjuice/vendor/rules/git__branch.json deleted file mode 100644 index aa29da8d1..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/git__branch.json +++ /dev/null @@ -1,29 +0,0 @@ -{ - "id": "git/branch", - "family": "git-branches", - "description": "Compact git branch output while preserving branch names and current branch context.", - "match": { - "argv0": ["git"], - "argvIncludes": [["branch"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 14, - "tail": 4 - }, - "failure": { - "preserveOnFailure": true, - "head": 16, - "tail": 12 - }, - "counters": [ - { - "name": "branch", - "pattern": ".+" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/git__diff-name-only.json b/src/openhuman/tokenjuice/vendor/rules/git__diff-name-only.json deleted file mode 100644 index 6671972b6..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/git__diff-name-only.json +++ /dev/null @@ -1,29 +0,0 @@ -{ - "id": "git/diff-name-only", - "family": "git-diff", - "description": "Compact git diff --name-only output.", - "match": { - "argv0": ["git"], - "argvIncludes": [["diff"], ["--name-only"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 16, - "tail": 4 - }, - "failure": { - "preserveOnFailure": true, - "head": 16, - "tail": 12 - }, - "counters": [ - { - "name": "file", - "pattern": ".+" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/git__diff-stat.json b/src/openhuman/tokenjuice/vendor/rules/git__diff-stat.json deleted file mode 100644 index 4b57de878..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/git__diff-stat.json +++ /dev/null @@ -1,37 +0,0 @@ -{ - "id": "git/diff-stat", - "family": "git-diff", - "description": "Compact git diff --stat output.", - "match": { - "argv0": ["git"], - "argvIncludes": [["diff"], ["--stat"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 12, - "tail": 6 - }, - "failure": { - "preserveOnFailure": true, - "head": 16, - "tail": 12 - }, - "counters": [ - { - "name": "file", - "pattern": "\\|" - }, - { - "name": "insertion", - "pattern": "insertions?\\(\\+\\)" - }, - { - "name": "deletion", - "pattern": "deletions?\\(-\\)" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/git__log-oneline.json b/src/openhuman/tokenjuice/vendor/rules/git__log-oneline.json deleted file mode 100644 index 9c3a71de4..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/git__log-oneline.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "git/log-oneline", - "family": "git-history", - "description": "Compact git log --oneline output while preserving commits.", - "match": { - "argv0": ["git"], - "argvIncludes": [["log"], ["--oneline"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 8, - "tail": 6 - }, - "failure": { - "preserveOnFailure": true, - "head": 10, - "tail": 10 - }, - "counters": [ - { - "name": "commit", - "pattern": "^[a-f0-9]{7,}\\s", - "flags": "m" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/git__remote-v.json b/src/openhuman/tokenjuice/vendor/rules/git__remote-v.json deleted file mode 100644 index eb3cd9a4e..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/git__remote-v.json +++ /dev/null @@ -1,29 +0,0 @@ -{ - "id": "git/remote-v", - "family": "git-remote", - "description": "Compact git remote -v output while preserving fetch/push remotes.", - "match": { - "argv0": ["git"], - "argvIncludes": [["remote"], ["-v"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 4 - }, - "failure": { - "preserveOnFailure": true, - "head": 12, - "tail": 10 - }, - "counters": [ - { - "name": "remote", - "pattern": "\\((fetch|push)\\)" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/git__show.json b/src/openhuman/tokenjuice/vendor/rules/git__show.json deleted file mode 100644 index 22eeb82ab..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/git__show.json +++ /dev/null @@ -1,50 +0,0 @@ -{ - "id": "git/show", - "family": "git-show", - "description": "Compact git show output while preserving commit summary and diff stat.", - "match": { - "argv0": ["git"], - "argvIncludes": [["show"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "keepPatterns": [ - "^commit\\s+.+", - "^Author:\\s+.+", - "^Date:\\s+.+", - "^\\s{4}.+", - "^diff --git\\s+.+", - "^index\\s+[a-f0-9]+\\.[a-f0-9]+", - "^---\\s+.+", - "^\\+\\+\\+\\s+.+", - "^@@\\s+.+", - "^\\s*\\d+ files? changed.+", - "^\\s*create mode .+", - "^\\s*delete mode .+" - ] - }, - "summarize": { - "head": 8, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 10, - "tail": 10 - }, - "counters": [ - { - "name": "file", - "pattern": "\\|" - }, - { - "name": "commit", - "pattern": "^commit\\s", - "flags": "m" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/git__stash-list.json b/src/openhuman/tokenjuice/vendor/rules/git__stash-list.json deleted file mode 100644 index 7d1b6bef4..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/git__stash-list.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "git/stash-list", - "family": "git-stash", - "description": "Compact git stash list output while preserving stash entries.", - "match": { - "argv0": ["git"], - "argvIncludes": [["stash"], ["list"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 12, - "tail": 4 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 12 - }, - "counters": [ - { - "name": "stash", - "pattern": "^stash@\\{\\d+\\}:", - "flags": "m" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/git__status.json b/src/openhuman/tokenjuice/vendor/rules/git__status.json deleted file mode 100644 index 57b34188a..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/git__status.json +++ /dev/null @@ -1,53 +0,0 @@ -{ - "id": "git/status", - "family": "git-status", - "description": "Compact human-readable git status output.", - "match": { - "argv0": ["git"], - "argvIncludes": [["status"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "skipPatterns": [ - "^On branch ", - "^Your branch is ", - "^and have \\d+ and \\d+ different commits each.*$", - "^\\(use \"git .+\" to .+\\)$", - "^no changes added to commit.*$", - "^nothing added to commit but untracked files present.*$", - "^nothing to commit, working tree clean$", - "^use \"git .+\" to .+" - ] - }, - "summarize": { - "head": 10, - "tail": 4 - }, - "failure": { - "preserveOnFailure": true, - "head": 12, - "tail": 12 - }, - "counters": [ - { - "name": "modified file", - "pattern": "^(?:M:|\\s*modified:|[ MTRU][MTRU]\\s+|[MTRU][ MTRU]\\s+)" - }, - { - "name": "new file", - "pattern": "^(?:A:|\\s*new file:|A.\\s+|.A\\s+)" - }, - { - "name": "deleted file", - "pattern": "^(?:D:|\\s*deleted:|D.\\s+|.D\\s+)" - }, - { - "name": "untracked file", - "pattern": "^(?:\\?\\?:|\\?\\?\\s+|\\s*untracked files:)" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/install__bun-install.json b/src/openhuman/tokenjuice/vendor/rules/install__bun-install.json deleted file mode 100644 index 6f0118e57..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/install__bun-install.json +++ /dev/null @@ -1,43 +0,0 @@ -{ - "id": "install/bun-install", - "family": "dependency-install", - "description": "Compact bun install output while preserving warnings and package counts.", - "matchOutput": [ - { - "pattern": "Checked \\d+ installs? across \\d+ packages? \\(no changes\\)", - "message": "bun install: up to date", - "flags": "i" - } - ], - "match": { - "toolNames": ["exec"], - "argv0": ["bun"], - "argvIncludes": [["install"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "warning", - "pattern": "warning", - "flags": "i" - }, - { - "name": "package", - "pattern": "\\bpackage(s)?\\b", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/install__npm-install.json b/src/openhuman/tokenjuice/vendor/rules/install__npm-install.json deleted file mode 100644 index b90d0a1ba..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/install__npm-install.json +++ /dev/null @@ -1,49 +0,0 @@ -{ - "id": "install/npm-install", - "family": "dependency-install", - "description": "Compact npm install output while preserving warnings and audit summaries.", - "onEmpty": "npm install: ok", - "matchOutput": [ - { - "pattern": "up to date, audited \\d+ package", - "message": "npm install: up to date", - "flags": "i" - } - ], - "match": { - "toolNames": ["exec"], - "argv0": ["npm"], - "argvIncludes": [["install"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "skipPatterns": [ - "^npm notice .+" - ] - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "warning", - "pattern": "warn", - "flags": "i" - }, - { - "name": "vulnerability", - "pattern": "vulnerabilit", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/install__pnpm-install.json b/src/openhuman/tokenjuice/vendor/rules/install__pnpm-install.json deleted file mode 100644 index 4c16d3a6e..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/install__pnpm-install.json +++ /dev/null @@ -1,43 +0,0 @@ -{ - "id": "install/pnpm-install", - "family": "dependency-install", - "description": "Compact pnpm install output while preserving warnings and summary lines.", - "matchOutput": [ - { - "pattern": "Already up to date", - "message": "pnpm install: up to date", - "flags": "i" - } - ], - "match": { - "toolNames": ["exec"], - "argv0": ["pnpm"], - "argvIncludes": [["install"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "warning", - "pattern": "warn", - "flags": "i" - }, - { - "name": "package", - "pattern": "\\bpackages?\\b", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/install__yarn-install.json b/src/openhuman/tokenjuice/vendor/rules/install__yarn-install.json deleted file mode 100644 index 8d723d162..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/install__yarn-install.json +++ /dev/null @@ -1,42 +0,0 @@ -{ - "id": "install/yarn-install", - "family": "dependency-install", - "description": "Compact yarn install output while preserving warnings and summary lines.", - "matchOutput": [ - { - "pattern": "Already up-to-date\\.", - "message": "yarn install: up to date" - } - ], - "match": { - "toolNames": ["exec"], - "argv0": ["yarn"], - "argvIncludes": [["install"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "warning", - "pattern": "warning", - "flags": "i" - }, - { - "name": "package", - "pattern": "\\bpackages?\\b", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/lint__biome.json b/src/openhuman/tokenjuice/vendor/rules/lint__biome.json deleted file mode 100644 index 84d6c0a42..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/lint__biome.json +++ /dev/null @@ -1,35 +0,0 @@ -{ - "id": "lint/biome", - "family": "lint-results", - "description": "Compact Biome output while preserving diagnostics.", - "match": { - "toolNames": ["exec"], - "commandIncludes": ["biome"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 14, - "tail": 10 - }, - "failure": { - "preserveOnFailure": true, - "head": 18, - "tail": 18 - }, - "counters": [ - { - "name": "error", - "pattern": "\\berror\\b", - "flags": "i" - }, - { - "name": "warning", - "pattern": "\\bwarning\\b", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/lint__cargo-clippy.json b/src/openhuman/tokenjuice/vendor/rules/lint__cargo-clippy.json deleted file mode 100644 index be87642bf..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/lint__cargo-clippy.json +++ /dev/null @@ -1,58 +0,0 @@ -{ - "id": "lint/cargo-clippy", - "family": "lint-results", - "description": "Compact cargo clippy output while preserving lint diagnostics and summary counts.", - "match": { - "toolNames": ["exec"], - "argv0": ["cargo"], - "argvIncludes": [["clippy"]] - }, - "onEmpty": "clippy: no warnings", - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "skipPatterns": [ - "^\\s*Compiling .+", - "^\\s*Checking .+", - "^\\s*Finished .+", - "^\\s*Downloading .+", - "^\\s*Downloaded .+", - "^\\s*Locking .+", - "^\\s*Fresh .+", - "^\\s+\\|\\s*$", - "^\\s*For more information about this error", - "^\\s*Some errors have detailed explanations", - "^\\s*= help: for further information visit" - ] - }, - "summarize": { - "head": 14, - "tail": 10 - }, - "failure": { - "preserveOnFailure": true, - "head": 20, - "tail": 20 - }, - "counters": [ - { - "name": "warning", - "pattern": "^warning(\\[.+\\])?:", - "flags": "i" - }, - { - "name": "error", - "pattern": "^error(\\[.+\\])?:", - "flags": "i" - } - ], - "matchOutput": [ - { - "pattern": "^\\s*$", - "message": "clippy: no warnings" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/lint__cargo-fmt.json b/src/openhuman/tokenjuice/vendor/rules/lint__cargo-fmt.json deleted file mode 100644 index 7ae836a73..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/lint__cargo-fmt.json +++ /dev/null @@ -1,50 +0,0 @@ -{ - "id": "lint/cargo-fmt", - "family": "lint-results", - "description": "Compact cargo fmt --check output while preserving diff hunks showing unformatted code.", - "match": { - "toolNames": ["exec"], - "argv0": ["cargo"], - "argvIncludes": [["fmt"]] - }, - "onEmpty": "cargo fmt: all files formatted", - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "skipPatterns": [ - "^\\s*Checking .+", - "^\\s*Finished .+" - ], - "keepPatterns": [ - "^Diff in .+", - "^[+-].+", - "^@@.+@@", - "^error.+", - "^warning.+" - ] - }, - "summarize": { - "head": 12, - "tail": 10 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "unformatted file", - "pattern": "^Diff in .+" - } - ], - "matchOutput": [ - { - "pattern": "^\\s*$", - "message": "cargo fmt: all files formatted" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/lint__eslint.json b/src/openhuman/tokenjuice/vendor/rules/lint__eslint.json deleted file mode 100644 index 85edee7d1..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/lint__eslint.json +++ /dev/null @@ -1,45 +0,0 @@ -{ - "id": "lint/eslint", - "family": "lint-results", - "description": "Compact ESLint output while preserving file diagnostics and summary counts.", - "match": { - "toolNames": ["exec"], - "commandIncludes": ["eslint"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "keepPatterns": [ - "^.+\\.(ts|tsx|js|jsx|mjs|cjs)$", - "^\\s*\\d+:\\d+\\s+(error|warning)\\s+.+", - "^✖\\s+.+", - "^\\d+ problems?\\s+\\(.+\\)$", - "^\\s*error\\s+.+", - "^\\s*warning\\s+.+" - ] - }, - "summarize": { - "head": 10, - "tail": 10 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "\\berror\\b", - "flags": "i" - }, - { - "name": "warning", - "pattern": "\\bwarning\\b", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/lint__oxlint.json b/src/openhuman/tokenjuice/vendor/rules/lint__oxlint.json deleted file mode 100644 index 825a04b68..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/lint__oxlint.json +++ /dev/null @@ -1,35 +0,0 @@ -{ - "id": "lint/oxlint", - "family": "lint-results", - "description": "Compact Oxlint output while preserving diagnostics.", - "match": { - "toolNames": ["exec"], - "commandIncludes": ["oxlint"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 14, - "tail": 10 - }, - "failure": { - "preserveOnFailure": true, - "head": 18, - "tail": 18 - }, - "counters": [ - { - "name": "error", - "pattern": "\\berror\\b", - "flags": "i" - }, - { - "name": "warning", - "pattern": "\\bwarning\\b", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/lint__prettier-check.json b/src/openhuman/tokenjuice/vendor/rules/lint__prettier-check.json deleted file mode 100644 index 52c7b1e0f..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/lint__prettier-check.json +++ /dev/null @@ -1,34 +0,0 @@ -{ - "id": "lint/prettier-check", - "family": "lint-results", - "description": "Compact Prettier check output while preserving unformatted files.", - "match": { - "toolNames": ["exec"], - "commandIncludes": ["prettier", "--check"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 12, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "warning", - "pattern": "warn", - "flags": "i" - }, - { - "name": "file", - "pattern": "\\[[^\\]]+\\]" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/media__ffmpeg.json b/src/openhuman/tokenjuice/vendor/rules/media__ffmpeg.json deleted file mode 100644 index fbf9c1a14..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/media__ffmpeg.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "media/ffmpeg", - "family": "media-cli", - "description": "Compact ffmpeg output while preserving stream mapping, progress, and terminal errors.", - "match": { - "toolNames": ["exec"], - "argv0": ["ffmpeg"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "error|invalid|failed|frame=", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/media__mediainfo.json b/src/openhuman/tokenjuice/vendor/rules/media__mediainfo.json deleted file mode 100644 index 85927803f..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/media__mediainfo.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "media/mediainfo", - "family": "media-cli", - "description": "Compact mediainfo output while preserving format, duration, and stream details.", - "match": { - "toolNames": ["exec"], - "argv0": ["mediainfo"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 12, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "warning", - "pattern": "error|failed|duration|format", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/network__curl.json b/src/openhuman/tokenjuice/vendor/rules/network__curl.json deleted file mode 100644 index a8d9d6c5d..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/network__curl.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "network/curl", - "family": "network-http", - "description": "Compact curl output while preserving response or failure details.", - "match": { - "toolNames": ["exec"], - "argv0": ["curl"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "error|failed|timed out", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/network__dig.json b/src/openhuman/tokenjuice/vendor/rules/network__dig.json deleted file mode 100644 index bc3d27506..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/network__dig.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "network/dig", - "family": "network-dns", - "description": "Compact dig output while preserving answer sections and failures.", - "match": { - "toolNames": ["exec"], - "argv0": ["dig"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "answer", - "pattern": "ANSWER SECTION|\\sIN\\sA\\s|\\sIN\\sAAAA\\s", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/network__nslookup.json b/src/openhuman/tokenjuice/vendor/rules/network__nslookup.json deleted file mode 100644 index 102acb419..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/network__nslookup.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "network/nslookup", - "family": "network-dns", - "description": "Compact nslookup output while preserving server and answer rows.", - "match": { - "toolNames": ["exec"], - "argv0": ["nslookup"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "server", - "pattern": "^Server:", - "flags": "m" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/network__ping.json b/src/openhuman/tokenjuice/vendor/rules/network__ping.json deleted file mode 100644 index cd7b3afb0..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/network__ping.json +++ /dev/null @@ -1,35 +0,0 @@ -{ - "id": "network/ping", - "family": "network-probe", - "description": "Compact ping output while preserving packet loss and latency summary.", - "match": { - "toolNames": ["exec"], - "argv0": ["ping"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 8, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 12, - "tail": 12 - }, - "counters": [ - { - "name": "reply", - "pattern": "bytes from", - "flags": "i" - }, - { - "name": "packet loss", - "pattern": "packet loss", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/network__ssh.json b/src/openhuman/tokenjuice/vendor/rules/network__ssh.json deleted file mode 100644 index f62d0fffa..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/network__ssh.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "network/ssh", - "family": "network-remote-shell", - "description": "Compact ssh output while preserving authentication and connection errors.", - "match": { - "toolNames": ["exec"], - "argv0": ["ssh"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "permission denied|connection refused|timed out|host key verification failed", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/network__traceroute.json b/src/openhuman/tokenjuice/vendor/rules/network__traceroute.json deleted file mode 100644 index f81e12b30..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/network__traceroute.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "network/traceroute", - "family": "network-route", - "description": "Compact traceroute output while preserving hop rows and failures.", - "match": { - "toolNames": ["exec"], - "argv0": ["traceroute"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 12, - "tail": 6 - }, - "failure": { - "preserveOnFailure": true, - "head": 16, - "tail": 12 - }, - "counters": [ - { - "name": "hop", - "pattern": "^\\s*\\d+\\s", - "flags": "m" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/network__wget.json b/src/openhuman/tokenjuice/vendor/rules/network__wget.json deleted file mode 100644 index e1cc152fa..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/network__wget.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "network/wget", - "family": "network-http", - "description": "Compact wget output while preserving transfer summary and failures.", - "match": { - "toolNames": ["exec"], - "argv0": ["wget"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "error|failed", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/observability__free.json b/src/openhuman/tokenjuice/vendor/rules/observability__free.json deleted file mode 100644 index eb35dfa58..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/observability__free.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "observability/free", - "family": "resource-memory", - "description": "Compact free output while preserving memory and swap totals.", - "match": { - "toolNames": ["exec"], - "argv0": ["free"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 8, - "tail": 6 - }, - "failure": { - "preserveOnFailure": true, - "head": 10, - "tail": 10 - }, - "counters": [ - { - "name": "warning", - "pattern": "error|failed", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/observability__htop.json b/src/openhuman/tokenjuice/vendor/rules/observability__htop.json deleted file mode 100644 index 82323789f..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/observability__htop.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "observability/htop", - "family": "resource-processes", - "description": "Compact htop output while preserving load, tasks, and top process lines.", - "match": { - "toolNames": ["exec"], - "argv0": ["htop"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 12, - "tail": 12 - }, - "counters": [ - { - "name": "warning", - "pattern": "load average|tasks|zombie", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/observability__iostat.json b/src/openhuman/tokenjuice/vendor/rules/observability__iostat.json deleted file mode 100644 index 69b665edd..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/observability__iostat.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "observability/iostat", - "family": "resource-io", - "description": "Compact iostat output while preserving CPU averages and busy devices.", - "match": { - "toolNames": ["exec"], - "argv0": ["iostat"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 12, - "tail": 12 - }, - "counters": [ - { - "name": "busy", - "pattern": "%util|Device", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/observability__top.json b/src/openhuman/tokenjuice/vendor/rules/observability__top.json deleted file mode 100644 index 2690a44ec..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/observability__top.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "observability/top", - "family": "resource-processes", - "description": "Compact top output while preserving load, task counts, and leading process rows.", - "match": { - "toolNames": ["exec"], - "argv0": ["top"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 12, - "tail": 12 - }, - "counters": [ - { - "name": "warning", - "pattern": "load average|zombie|stopped", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/observability__vmstat.json b/src/openhuman/tokenjuice/vendor/rules/observability__vmstat.json deleted file mode 100644 index df13b4da4..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/observability__vmstat.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "observability/vmstat", - "family": "resource-vm", - "description": "Compact vmstat output while preserving run queue, memory, swap, and io columns.", - "match": { - "toolNames": ["exec"], - "argv0": ["vmstat"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 8, - "tail": 6 - }, - "failure": { - "preserveOnFailure": true, - "head": 10, - "tail": 10 - }, - "counters": [ - { - "name": "warning", - "pattern": "swpd|cache|wa|st", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/package__apt-install.json b/src/openhuman/tokenjuice/vendor/rules/package__apt-install.json deleted file mode 100644 index 6a7b54e57..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/package__apt-install.json +++ /dev/null @@ -1,36 +0,0 @@ -{ - "id": "package/apt-install", - "family": "system-package-install", - "description": "Compact apt install output while preserving package counts, fetch summaries, and errors.", - "match": { - "toolNames": ["exec"], - "argv0": ["apt", "apt-get"], - "argvIncludes": [["install"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "skipPatterns": [ - "^Reading database \\.{3}.+$" - ] - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "error|failed|unable to", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/package__apt-upgrade.json b/src/openhuman/tokenjuice/vendor/rules/package__apt-upgrade.json deleted file mode 100644 index ff8c4d129..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/package__apt-upgrade.json +++ /dev/null @@ -1,36 +0,0 @@ -{ - "id": "package/apt-upgrade", - "family": "system-package-upgrade", - "description": "Compact apt upgrade output while preserving upgraded package counts and blocking errors.", - "match": { - "toolNames": ["exec"], - "argv0": ["apt", "apt-get"], - "argvIncludes": [["upgrade"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "skipPatterns": [ - "^Reading database \\.{3}.+$" - ] - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "error|failed|kept back", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/package__brew-install.json b/src/openhuman/tokenjuice/vendor/rules/package__brew-install.json deleted file mode 100644 index 95cfb2c5d..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/package__brew-install.json +++ /dev/null @@ -1,31 +0,0 @@ -{ - "id": "package/brew-install", - "family": "system-package-install", - "description": "Compact brew install output while preserving taps, installs, and failure details.", - "match": { - "toolNames": ["exec"], - "argv0": ["brew"], - "argvIncludes": [["install"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "warning", - "pattern": "warning|error|failed", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/package__brew-upgrade.json b/src/openhuman/tokenjuice/vendor/rules/package__brew-upgrade.json deleted file mode 100644 index 2c0aa7233..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/package__brew-upgrade.json +++ /dev/null @@ -1,31 +0,0 @@ -{ - "id": "package/brew-upgrade", - "family": "system-package-upgrade", - "description": "Compact brew upgrade output while preserving upgraded formulae and failures.", - "match": { - "toolNames": ["exec"], - "argv0": ["brew"], - "argvIncludes": [["upgrade"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "warning", - "pattern": "warning|error|failed", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/package__dnf-install.json b/src/openhuman/tokenjuice/vendor/rules/package__dnf-install.json deleted file mode 100644 index e7658595b..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/package__dnf-install.json +++ /dev/null @@ -1,31 +0,0 @@ -{ - "id": "package/dnf-install", - "family": "system-package-install", - "description": "Compact dnf install output while preserving transaction summaries and failures.", - "match": { - "toolNames": ["exec"], - "argv0": ["dnf"], - "argvIncludes": [["install"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "error|failed|nothing to do", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/package__yum-install.json b/src/openhuman/tokenjuice/vendor/rules/package__yum-install.json deleted file mode 100644 index 273f27736..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/package__yum-install.json +++ /dev/null @@ -1,31 +0,0 @@ -{ - "id": "package/yum-install", - "family": "system-package-install", - "description": "Compact yum install output while preserving dependency summaries and failures.", - "match": { - "toolNames": ["exec"], - "argv0": ["yum"], - "argvIncludes": [["install"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "error|failed|nothing to do", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/search__git-grep.json b/src/openhuman/tokenjuice/vendor/rules/search__git-grep.json deleted file mode 100644 index 188341543..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/search__git-grep.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "search/git-grep", - "family": "search", - "description": "Compact git grep output while preserving matches.", - "match": { - "toolNames": ["exec"], - "argv0": ["git"], - "argvIncludes": [["grep"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 12, - "tail": 4 - }, - "failure": { - "preserveOnFailure": true, - "head": 16, - "tail": 10 - }, - "counters": [ - { - "name": "match", - "pattern": ".+:.+" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/search__grep.json b/src/openhuman/tokenjuice/vendor/rules/search__grep.json deleted file mode 100644 index a6724400f..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/search__grep.json +++ /dev/null @@ -1,38 +0,0 @@ -{ - "id": "search/grep", - "family": "search", - "description": "Compact grep output while preserving matching lines.", - "match": { - "toolNames": ["exec"], - "argv0": ["grep"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "keepPatterns": [ - "^.+:\\d+[: -].+", - "^.+:.+", - "error|warn|binary file|permission denied|no such file", - "^\\d+ matches?$", - "^\\d+ files? matched$" - ] - }, - "summarize": { - "head": 10, - "tail": 6 - }, - "failure": { - "preserveOnFailure": true, - "head": 12, - "tail": 12 - }, - "counters": [ - { - "name": "match", - "pattern": ".+:.+" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/search__rg.json b/src/openhuman/tokenjuice/vendor/rules/search__rg.json deleted file mode 100644 index 81bc13d59..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/search__rg.json +++ /dev/null @@ -1,38 +0,0 @@ -{ - "id": "search/rg", - "family": "search", - "description": "Compact ripgrep output while preserving match lines.", - "match": { - "argv0": ["rg"], - "toolNames": ["exec"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "keepPatterns": [ - "^.+:\\d+[: -].+", - "^.+:.+", - "error|warn|binary file|permission denied|no such file", - "^\\d+ matches?$", - "^\\d+ files? matched$" - ] - }, - "summarize": { - "head": 10, - "tail": 6 - }, - "failure": { - "preserveOnFailure": true, - "head": 12, - "tail": 12 - }, - "counters": [ - { - "name": "match", - "pattern": ".+:.+" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/service__journalctl.json b/src/openhuman/tokenjuice/vendor/rules/service__journalctl.json deleted file mode 100644 index 928458b43..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/service__journalctl.json +++ /dev/null @@ -1,42 +0,0 @@ -{ - "id": "service/journalctl", - "family": "service-logs", - "description": "Compact journalctl output while preserving key log lines and failures.", - "match": { - "toolNames": ["exec"], - "argv0": ["journalctl"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "keepPatterns": [ - "error|warn|fatal|panic|exception|traceback|timeout|refused|fail", - "^Caused by:", - "^Traceback" - ] - }, - "summarize": { - "head": 8, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "warning", - "pattern": "warn", - "flags": "i" - }, - { - "name": "error", - "pattern": "error|failed", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/service__launchctl.json b/src/openhuman/tokenjuice/vendor/rules/service__launchctl.json deleted file mode 100644 index bcfaf8859..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/service__launchctl.json +++ /dev/null @@ -1,41 +0,0 @@ -{ - "id": "service/launchctl", - "family": "service-state", - "description": "Compact launchctl output while preserving labels and status rows.", - "match": { - "toolNames": ["exec"], - "argv0": ["launchctl"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "keepPatterns": [ - "^-?\\d+\\s+\\S+\\s+.+", - "^PID\\s+Status\\s+Label$", - "error|failed|stopped|disabled" - ] - }, - "summarize": { - "head": 8, - "tail": 6 - }, - "failure": { - "preserveOnFailure": true, - "head": 10, - "tail": 10 - }, - "counters": [ - { - "name": "service", - "pattern": "^(?!PID\\s+Status\\s+Label$).+\\S.*$" - }, - { - "name": "error", - "pattern": "error|failed|stopped|disabled", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/service__lsof.json b/src/openhuman/tokenjuice/vendor/rules/service__lsof.json deleted file mode 100644 index 93239cc06..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/service__lsof.json +++ /dev/null @@ -1,29 +0,0 @@ -{ - "id": "service/lsof", - "family": "service-open-files", - "description": "Compact lsof output while preserving open-file rows.", - "match": { - "toolNames": ["exec"], - "argv0": ["lsof"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 8, - "tail": 6 - }, - "failure": { - "preserveOnFailure": true, - "head": 10, - "tail": 10 - }, - "counters": [ - { - "name": "entry", - "pattern": "^(?!COMMAND\\s+PID\\s+USER\\s).+\\S.*$" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/service__netstat.json b/src/openhuman/tokenjuice/vendor/rules/service__netstat.json deleted file mode 100644 index 3ec7c71c3..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/service__netstat.json +++ /dev/null @@ -1,29 +0,0 @@ -{ - "id": "service/netstat", - "family": "service-network-state", - "description": "Compact netstat output while preserving socket rows.", - "match": { - "toolNames": ["exec"], - "argv0": ["netstat"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 8, - "tail": 6 - }, - "failure": { - "preserveOnFailure": true, - "head": 10, - "tail": 10 - }, - "counters": [ - { - "name": "socket", - "pattern": "^(?!Proto\\s|Active\\s).+\\S.*$" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/service__service.json b/src/openhuman/tokenjuice/vendor/rules/service__service.json deleted file mode 100644 index 2b46a347c..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/service__service.json +++ /dev/null @@ -1,46 +0,0 @@ -{ - "id": "service/service", - "family": "service-state", - "description": "Compact service command output while preserving status and failure lines.", - "match": { - "toolNames": ["exec"], - "argv0": ["service"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "keepPatterns": [ - "error|failed|inactive|stopped|warning|refused|timeout", - "is running", - "is stopped", - "start/running", - "stop/waiting", - "^\\s*Active:\\s+.+", - "^\\s*Status:\\s+.+" - ] - }, - "summarize": { - "head": 8, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 10, - "tail": 10 - }, - "counters": [ - { - "name": "warning", - "pattern": "warning|refused|timeout", - "flags": "i" - }, - { - "name": "error", - "pattern": "error|failed|inactive|stopped", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/service__ss.json b/src/openhuman/tokenjuice/vendor/rules/service__ss.json deleted file mode 100644 index ffc2ccf44..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/service__ss.json +++ /dev/null @@ -1,29 +0,0 @@ -{ - "id": "service/ss", - "family": "service-network-state", - "description": "Compact ss output while preserving socket rows.", - "match": { - "toolNames": ["exec"], - "argv0": ["ss"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 8, - "tail": 6 - }, - "failure": { - "preserveOnFailure": true, - "head": 10, - "tail": 10 - }, - "counters": [ - { - "name": "socket", - "pattern": "^(?!Netid\\s|State\\s).+\\S.*$" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/service__systemctl-status.json b/src/openhuman/tokenjuice/vendor/rules/service__systemctl-status.json deleted file mode 100644 index ad626f141..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/service__systemctl-status.json +++ /dev/null @@ -1,45 +0,0 @@ -{ - "id": "service/systemctl-status", - "family": "service-state", - "description": "Compact systemctl status output while preserving active state and failure lines.", - "match": { - "toolNames": ["exec"], - "argv0": ["systemctl"], - "argvIncludes": [["status"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "keepPatterns": [ - "^●\\s+.+", - "^\\s*(Loaded|Active|Main PID|Tasks|Memory|CPU):", - "error|failed|inactive|dead|back-off|timeout|refused|warning", - "^\\s*Process:\\s+.+", - "^\\s*Docs:\\s+.+" - ] - }, - "summarize": { - "head": 8, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 10, - "tail": 10 - }, - "counters": [ - { - "name": "warning", - "pattern": "warning|back-off|timeout|refused", - "flags": "i" - }, - { - "name": "error", - "pattern": "failed|inactive|dead|error", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/system__df.json b/src/openhuman/tokenjuice/vendor/rules/system__df.json deleted file mode 100644 index 6cbc4f4d5..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/system__df.json +++ /dev/null @@ -1,29 +0,0 @@ -{ - "id": "system/df", - "family": "system-disk", - "description": "Compact df output while preserving filesystem rows.", - "match": { - "toolNames": ["exec"], - "argv0": ["df"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 12, - "tail": 4 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 12 - }, - "counters": [ - { - "name": "filesystem", - "pattern": ".+" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/system__du.json b/src/openhuman/tokenjuice/vendor/rules/system__du.json deleted file mode 100644 index 2ed887b89..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/system__du.json +++ /dev/null @@ -1,29 +0,0 @@ -{ - "id": "system/du", - "family": "system-disk", - "description": "Compact du output while preserving size rows.", - "match": { - "toolNames": ["exec"], - "argv0": ["du"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 8, - "tail": 6 - }, - "failure": { - "preserveOnFailure": true, - "head": 12, - "tail": 10 - }, - "counters": [ - { - "name": "entry", - "pattern": "^\\S+\\s+.+" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/system__file.json b/src/openhuman/tokenjuice/vendor/rules/system__file.json deleted file mode 100644 index 4949ed7fd..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/system__file.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "system/file", - "family": "file-inspection", - "description": "Compact file output while preserving the detected file type.", - "match": { - "toolNames": ["exec"], - "argv0": ["file"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 8, - "tail": 6 - }, - "failure": { - "preserveOnFailure": true, - "head": 10, - "tail": 10 - }, - "counters": [ - { - "name": "warning", - "pattern": "cannot open|error", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/system__ps.json b/src/openhuman/tokenjuice/vendor/rules/system__ps.json deleted file mode 100644 index 878903cf4..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/system__ps.json +++ /dev/null @@ -1,29 +0,0 @@ -{ - "id": "system/ps", - "family": "system-processes", - "description": "Compact ps output while preserving process rows.", - "match": { - "toolNames": ["exec"], - "argv0": ["ps"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 8, - "tail": 6 - }, - "failure": { - "preserveOnFailure": true, - "head": 12, - "tail": 10 - }, - "counters": [ - { - "name": "process", - "pattern": "^(?!USER\\s|PID\\s).+\\S.*$" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/task__just.json b/src/openhuman/tokenjuice/vendor/rules/task__just.json deleted file mode 100644 index 0e2a15e7c..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/task__just.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "task/just", - "family": "task-runner", - "description": "Compact just output while preserving task results and failures.", - "match": { - "toolNames": ["exec"], - "argv0": ["just"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 12, - "tail": 10 - }, - "failure": { - "preserveOnFailure": true, - "head": 18, - "tail": 16 - }, - "counters": [ - { - "name": "error", - "pattern": "error", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/task__make.json b/src/openhuman/tokenjuice/vendor/rules/task__make.json deleted file mode 100644 index 961b25e43..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/task__make.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "task/make", - "family": "task-runner", - "description": "Compact make output while preserving target failures and summaries.", - "match": { - "toolNames": ["exec"], - "argv0": ["make"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 12, - "tail": 10 - }, - "failure": { - "preserveOnFailure": true, - "head": 18, - "tail": 16 - }, - "counters": [ - { - "name": "error", - "pattern": "error", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/tests__bun-test.json b/src/openhuman/tokenjuice/vendor/rules/tests__bun-test.json deleted file mode 100644 index cd890922a..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/tests__bun-test.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "id": "tests/bun-test", - "family": "test-results", - "description": "Compact bun test output while preserving failures and summary lines.", - "match": { - "toolNames": ["exec"], - "argv0": ["bun"], - "argvIncludes": [["test"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "skipPatterns": [ - "^> .+$", - "^\\s*RUN\\s+.+$", - "^\\s*Start at\\s+.+$" - ], - "keepPatterns": [ - "^\\s*❯\\s+.+", - "^\\s*✓\\s+.+", - "^\\s*FAIL\\s+.+", - "^\\s*PASS\\s+.+", - "^AssertionError: .+", - "^Error: .+", - "^Caused by: .+", - "^\\s*Test Files\\s+.+", - "^\\s*Tests\\s+.+", - "^\\s*Duration\\s+.+", - "^⎯⎯⎯.+" - ] - }, - "summarize": { - "head": 12, - "tail": 10 - }, - "failure": { - "preserveOnFailure": true, - "head": 18, - "tail": 18 - }, - "counters": [ - { - "name": "failed", - "pattern": "fail", - "flags": "i" - }, - { - "name": "passed", - "pattern": "pass", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/tests__cargo-test.json b/src/openhuman/tokenjuice/vendor/rules/tests__cargo-test.json deleted file mode 100644 index 546964a0e..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/tests__cargo-test.json +++ /dev/null @@ -1,41 +0,0 @@ -{ - "id": "tests/cargo-test", - "family": "test-results", - "description": "Compact cargo test output while preserving failures and final summary.", - "match": { - "toolNames": ["exec"], - "argv0": ["cargo"], - "argvIncludes": [["test"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "skipPatterns": [ - "^\\s*Compiling .+", - "^\\s*Finished .+", - "^\\s*Running .+" - ] - }, - "summarize": { - "head": 12, - "tail": 10 - }, - "failure": { - "preserveOnFailure": true, - "head": 18, - "tail": 18 - }, - "counters": [ - { - "name": "failed test", - "pattern": "FAILED" - }, - { - "name": "passed test", - "pattern": "ok" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/tests__go-test.json b/src/openhuman/tokenjuice/vendor/rules/tests__go-test.json deleted file mode 100644 index 6127176ad..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/tests__go-test.json +++ /dev/null @@ -1,49 +0,0 @@ -{ - "id": "tests/go-test", - "family": "test-results", - "description": "Compact go test output while preserving failing packages and summaries.", - "match": { - "toolNames": ["exec"], - "argv0": ["go"], - "argvIncludes": [["test"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "skipPatterns": [ - "^ok\\s.+" - ], - "keepPatterns": [ - "^FAIL\\s.+", - "^--- FAIL: .+", - "^panic: .+", - "^\\s+.+_test\\.go:\\d+: .+", - "^\\s+Error Trace: .+", - "^\\s+Error: .+" - ] - }, - "summarize": { - "head": 10, - "tail": 10 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "failed package", - "pattern": "^FAIL\\s", - "flags": "m" - }, - { - "name": "passed package", - "pattern": "^ok\\s", - "flags": "m" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/tests__jest.json b/src/openhuman/tokenjuice/vendor/rules/tests__jest.json deleted file mode 100644 index 1c526edaf..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/tests__jest.json +++ /dev/null @@ -1,41 +0,0 @@ -{ - "id": "tests/jest", - "family": "test-results", - "description": "Compact Jest output while preserving failures and summary counts.", - "match": { - "toolNames": ["exec"], - "commandIncludes": ["jest"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "skipPatterns": [ - "^\\s*at .+", - "^Ran all test suites.*$" - ] - }, - "summarize": { - "head": 12, - "tail": 10 - }, - "failure": { - "preserveOnFailure": true, - "head": 16, - "tail": 16 - }, - "counters": [ - { - "name": "failed test", - "pattern": "^FAIL\\s", - "flags": "m" - }, - { - "name": "passed suite", - "pattern": "^PASS\\s", - "flags": "m" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/tests__mocha.json b/src/openhuman/tokenjuice/vendor/rules/tests__mocha.json deleted file mode 100644 index 5ef24ede2..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/tests__mocha.json +++ /dev/null @@ -1,35 +0,0 @@ -{ - "id": "tests/mocha", - "family": "test-results", - "description": "Compact Mocha output while preserving failing tests and summary counts.", - "match": { - "toolNames": ["exec"], - "commandIncludes": ["mocha"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 12, - "tail": 10 - }, - "failure": { - "preserveOnFailure": true, - "head": 18, - "tail": 18 - }, - "counters": [ - { - "name": "failing", - "pattern": "\\bfailing\\b", - "flags": "i" - }, - { - "name": "passing", - "pattern": "\\bpassing\\b", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/tests__npm-test.json b/src/openhuman/tokenjuice/vendor/rules/tests__npm-test.json deleted file mode 100644 index d36f69d6a..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/tests__npm-test.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "id": "tests/npm-test", - "family": "test-results", - "description": "Catch common npm test runs when the underlying runner is not explicit.", - "match": { - "toolNames": ["exec"], - "argv0": ["npm"], - "argvIncludes": [["test"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "skipPatterns": [ - "^> .+$", - "^\\s*RUN\\s+.+$", - "^\\s*Start at\\s+.+$" - ], - "keepPatterns": [ - "^\\s*❯\\s+.+", - "^\\s*✓\\s+.+", - "^\\s*FAIL\\s+.+", - "^\\s*PASS\\s+.+", - "^AssertionError: .+", - "^Error: .+", - "^Caused by: .+", - "^\\s*Test Files\\s+.+", - "^\\s*Tests\\s+.+", - "^\\s*Duration\\s+.+", - "^⎯⎯⎯.+" - ] - }, - "summarize": { - "head": 12, - "tail": 10 - }, - "failure": { - "preserveOnFailure": true, - "head": 16, - "tail": 16 - }, - "counters": [ - { - "name": "failed", - "pattern": "fail", - "flags": "i" - }, - { - "name": "passed", - "pattern": "pass", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/tests__playwright.json b/src/openhuman/tokenjuice/vendor/rules/tests__playwright.json deleted file mode 100644 index add65ef5f..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/tests__playwright.json +++ /dev/null @@ -1,37 +0,0 @@ -{ - "id": "tests/playwright", - "family": "test-results", - "description": "Compact Playwright test output while preserving failing specs and summary lines.", - "match": { - "toolNames": ["exec"], - "argv0": ["playwright", "pnpm", "npx", "bunx", "yarn", "npm"], - "argvIncludes": [["playwright"], ["test"]], - "commandIncludes": ["playwright", "test"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 12, - "tail": 10 - }, - "failure": { - "preserveOnFailure": true, - "head": 18, - "tail": 18 - }, - "counters": [ - { - "name": "failed", - "pattern": "\\bfailed\\b", - "flags": "i" - }, - { - "name": "passed", - "pattern": "\\bpassed\\b", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/tests__pnpm-test.json b/src/openhuman/tokenjuice/vendor/rules/tests__pnpm-test.json deleted file mode 100644 index 8f6ade147..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/tests__pnpm-test.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "id": "tests/pnpm-test", - "family": "test-results", - "description": "Catch common pnpm test runs when the underlying runner is not explicit.", - "match": { - "toolNames": ["exec"], - "argv0": ["pnpm"], - "argvIncludes": [["test"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "skipPatterns": [ - "^> .+$", - "^\\s*RUN\\s+.+$", - "^\\s*Start at\\s+.+$" - ], - "keepPatterns": [ - "^\\s*❯\\s+.+", - "^\\s*✓\\s+.+", - "^\\s*FAIL\\s+.+", - "^\\s*PASS\\s+.+", - "^AssertionError: .+", - "^Error: .+", - "^Caused by: .+", - "^\\s*Test Files\\s+.+", - "^\\s*Tests\\s+.+", - "^\\s*Duration\\s+.+", - "^⎯⎯⎯.+" - ] - }, - "summarize": { - "head": 12, - "tail": 10 - }, - "failure": { - "preserveOnFailure": true, - "head": 16, - "tail": 16 - }, - "counters": [ - { - "name": "failed", - "pattern": "fail", - "flags": "i" - }, - { - "name": "passed", - "pattern": "pass", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/tests__pytest.json b/src/openhuman/tokenjuice/vendor/rules/tests__pytest.json deleted file mode 100644 index 1e78c33ee..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/tests__pytest.json +++ /dev/null @@ -1,54 +0,0 @@ -{ - "id": "tests/pytest", - "family": "test-results", - "description": "Compact pytest output while preserving failures and final summary.", - "counterSource": "preKeep", - "match": { - "toolNames": ["exec"], - "commandIncludes": ["pytest"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "skipPatterns": [ - "^platform .+", - "^rootdir: .+", - "^plugins: .+", - "^collected \\d+ items$" - ], - "keepPatterns": [ - "^=+.+(failed|passed|error).+=+$", - "^_{2,}.+_{2,}$", - "^FAILED .+", - "^ERROR .+", - "^E\\s+.+", - "AssertionError", - "^.+::.+ (FAILED|ERROR)$", - "^>\\s+.+" - ] - }, - "summarize": { - "head": 10, - "tail": 10 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "failed test", - "pattern": "^.+::.+ (FAILED|ERROR)$", - "flags": "i" - }, - { - "name": "passed test", - "pattern": "^.+::.+ PASSED$", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/tests__vitest.json b/src/openhuman/tokenjuice/vendor/rules/tests__vitest.json deleted file mode 100644 index d50c3ceba..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/tests__vitest.json +++ /dev/null @@ -1,55 +0,0 @@ -{ - "id": "tests/vitest", - "family": "test-results", - "description": "Compact Vitest output while preserving failures and summary lines.", - "match": { - "toolNames": ["exec"], - "commandIncludes": ["vitest"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "skipPatterns": [ - "^\\s*at .+", - "^\\s*❯ .+node_modules.+", - "^\\s*✓ .+" - ], - "keepPatterns": [ - "^\\s*RUN\\s+", - "^\\s*❯\\s+.+", - "^\\s*FAIL\\s+.+", - "^AssertionError: .+", - "^Error: .+", - "^Caused by: .+", - "^\\s*Test Files\\s+.+", - "^\\s*Tests\\s+.+", - "^ Start at\\s+.+", - "^ Duration\\s+.+", - "^⎯⎯⎯.+" - ] - }, - "summarize": { - "head": 10, - "tail": 10 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "failed suite", - "pattern": "^\\s*❯\\s.+", - "flags": "m" - }, - { - "name": "failure", - "pattern": "failed", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/tests__yarn-test.json b/src/openhuman/tokenjuice/vendor/rules/tests__yarn-test.json deleted file mode 100644 index c669f5b23..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/tests__yarn-test.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "id": "tests/yarn-test", - "family": "test-results", - "description": "Catch common yarn test runs when the underlying runner is not explicit.", - "match": { - "toolNames": ["exec"], - "argv0": ["yarn"], - "argvIncludes": [["test"]] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "filters": { - "skipPatterns": [ - "^> .+$", - "^\\s*RUN\\s+.+$", - "^\\s*Start at\\s+.+$" - ], - "keepPatterns": [ - "^\\s*❯\\s+.+", - "^\\s*✓\\s+.+", - "^\\s*FAIL\\s+.+", - "^\\s*PASS\\s+.+", - "^AssertionError: .+", - "^Error: .+", - "^Caused by: .+", - "^\\s*Test Files\\s+.+", - "^\\s*Tests\\s+.+", - "^\\s*Duration\\s+.+", - "^⎯⎯⎯.+" - ] - }, - "summarize": { - "head": 12, - "tail": 10 - }, - "failure": { - "preserveOnFailure": true, - "head": 16, - "tail": 16 - }, - "counters": [ - { - "name": "failed", - "pattern": "fail", - "flags": "i" - }, - { - "name": "passed", - "pattern": "pass", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/transfer__rsync.json b/src/openhuman/tokenjuice/vendor/rules/transfer__rsync.json deleted file mode 100644 index 1fdafd7bd..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/transfer__rsync.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "transfer/rsync", - "family": "file-transfer", - "description": "Compact rsync output while preserving changed paths, stats, and sync failures.", - "match": { - "toolNames": ["exec"], - "argv0": ["rsync"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 10, - "tail": 8 - }, - "failure": { - "preserveOnFailure": true, - "head": 14, - "tail": 14 - }, - "counters": [ - { - "name": "error", - "pattern": "error|failed|connection|sent ", - "flags": "i" - } - ] -} diff --git a/src/openhuman/tokenjuice/vendor/rules/transfer__scp.json b/src/openhuman/tokenjuice/vendor/rules/transfer__scp.json deleted file mode 100644 index 0eb5169a7..000000000 --- a/src/openhuman/tokenjuice/vendor/rules/transfer__scp.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "id": "transfer/scp", - "family": "file-transfer", - "description": "Compact scp output while preserving transferred paths, throughput, and ssh failures.", - "match": { - "toolNames": ["exec"], - "argv0": ["scp"] - }, - "transforms": { - "stripAnsi": true, - "dedupeAdjacent": true, - "trimEmptyEdges": true - }, - "summarize": { - "head": 8, - "tail": 6 - }, - "failure": { - "preserveOnFailure": true, - "head": 10, - "tail": 10 - }, - "counters": [ - { - "name": "error", - "pattern": "error|failed|permission denied|lost connection", - "flags": "i" - } - ] -} diff --git a/tests/agent_harness_e2e.rs b/tests/agent_harness_e2e.rs index 31faf17ae..27abd5956 100644 --- a/tests/agent_harness_e2e.rs +++ b/tests/agent_harness_e2e.rs @@ -1040,15 +1040,13 @@ async fn super_context_happy_path_inner() { // schedule_task tool result. The orchestrator surfaces this to the user. // On turn 2 the user's reply and the full turn-1 context are present. // -// Actual LLM request ordering (5 upstream calls total): +// Actual LLM request ordering (4 upstream calls total): // request[0] = orchestrator turn 1 → schedule_task delegation tool call returned // request[1] = scheduler_agent first iter → tries ask_user_clarification (blocked, // success=false; early-exit does NOT fire; loop continues) // request[2] = scheduler_agent second iter → returns text with clarification question -// (this becomes the schedule_task tool result forwarded to orchestrator) -// request[3] = orchestrator with schedule_task tool result containing WHICH_VERSION_CANARY -// → surfaces question to user; turn 1 ends (chat_done with WHICH_VERSION_CANARY) -// request[4] = orchestrator turn 2 with "version 2" user reply in full context → +// (this becomes the schedule_task tool result and turn-1 response) +// request[3] = orchestrator turn 2 with "version 2" user reply in full context → // synthesis; turn 2 ends (chat_done with ANSWER_CANARY_V2) /// Orchestrator delegates to scheduler_agent via `schedule_task` (delegate_name); @@ -1092,11 +1090,8 @@ async fn subagent_clarification_flow_inner() { // question. This becomes the schedule_task tool result forwarded to the // orchestrator by dispatch_subagent. text_completion("I need clarification: WHICH_VERSION_CANARY?"), - // request[3]: Orchestrator receives the schedule_task tool result containing - // WHICH_VERSION_CANARY and surfaces the question to the user. Turn 1 ends. - text_completion("I need to know: WHICH_VERSION_CANARY?"), // ── turn 2 (user replied "version 2") ── - // request[4]: Orchestrator processes user reply with full turn-1 context → + // request[3]: Orchestrator processes user reply with full turn-1 context → // synthesizes final answer; turn 2 ends here. text_completion("Final: ANSWER_CANARY_V2"), ]); @@ -1172,11 +1167,10 @@ async fn subagent_clarification_flow_inner() { // request[0] = orchestrator (schedule_task call), // request[1] = scheduler_agent first iter (ask_user_clarification blocked), // request[2] = scheduler_agent second iter (text output with question), - // request[3] = orchestrator synthesis (turn-1 end), - // request[4] = orchestrator turn-2 synthesis (turn-2 end). + // request[3] = orchestrator turn-2 synthesis (turn-2 end). assert!( requests.len() >= 4, - "expected ≥4 upstream requests (orchestrator + scheduler_agent x2 + orchestrator synthesis x2), \ + "expected ≥4 upstream requests (orchestrator + scheduler_agent x2 + orchestrator turn-2 synthesis), \ got {};\nall requests: {}", requests.len(), serde_json::to_string_pretty(&requests).unwrap_or_default() diff --git a/tests/agent_large_round25_raw_coverage_e2e.rs b/tests/agent_large_round25_raw_coverage_e2e.rs index 7ea0abee6..c6e9a86be 100644 --- a/tests/agent_large_round25_raw_coverage_e2e.rs +++ b/tests/agent_large_round25_raw_coverage_e2e.rs @@ -409,7 +409,8 @@ async fn integrations_text_mode_handoffs_oversized_result_and_extracts_from_cach assert!( requests[0].messages[0] .content - .contains("To use a tool, wrap a JSON object in tags"), + .contains("Tool calls use **P-Format**") + && requests[0].messages[0].content.contains(""), "text-mode protocol should be injected into the system prompt" ); let second_request = requests[1] diff --git a/tests/tokenjuice_integration.rs b/tests/tokenjuice_integration.rs index 5a5f86696..50f23bcd9 100644 --- a/tests/tokenjuice_integration.rs +++ b/tests/tokenjuice_integration.rs @@ -1,7 +1,7 @@ //! Integration tests for the TokenJuice module. //! -//! Iterates vendored `*.fixture.json` files under -//! `src/openhuman/tokenjuice/tests/fixtures/` and asserts that +//! Iterates vendored TinyJuice `*.fixture.json` files under +//! `vendor/tinyjuice/src/tests/fixtures/` and asserts that //! `reduce_execution_with_rules` produces the expected output. use openhuman_core::openhuman::tokenjuice::{ @@ -17,7 +17,7 @@ const KNOWN_DRIFT_FIXTURES: &[&str] = &[ fn fixtures_dir() -> std::path::PathBuf { let manifest = std::env::var("CARGO_MANIFEST_DIR").expect("CARGO_MANIFEST_DIR not set"); - std::path::PathBuf::from(manifest).join("src/openhuman/tokenjuice/tests/fixtures") + std::path::PathBuf::from(manifest).join("vendor/tinyjuice/src/tests/fixtures") } #[test] diff --git a/vendor/tinyjuice b/vendor/tinyjuice new file mode 160000 index 000000000..4b1a34fcc --- /dev/null +++ b/vendor/tinyjuice @@ -0,0 +1 @@ +Subproject commit 4b1a34fccccea3de6d49ab657ba53ccca9de04f2