diff --git a/crates/neuromesh-core/src/types.rs b/crates/neuromesh-core/src/types.rs index 2655e6c..adc1b64 100644 --- a/crates/neuromesh-core/src/types.rs +++ b/crates/neuromesh-core/src/types.rs @@ -417,9 +417,23 @@ impl CoverageReport { .filter(|s| s.resolved_id.is_some()) .map(|s| s.query.clone()) .collect(); + // A guess the server made itself (a prose word, an alias, a lone + // token) that resolved nowhere — or was dropped for a better seed — + // is not something the question named: it is not "missing", and an + // agent told to search for `stem:tool` only wastes a call. + const GUESS: &[&str] = &[ + "stem:", + "concept:", + "alias_code:", + "alias_gap_fill:", + "token:", + "fallback:", + "inferred_keyword:", + ]; let seeds_missed: Vec = seeds .iter() .filter(|s| s.resolved_id.is_none()) + .filter(|s| !GUESS.iter().any(|g| s.query.starts_with(g))) .map(|s| s.query.clone()) .collect(); let claim = if seeds_hit.is_empty() { diff --git a/crates/neuromesh-mcp/src/response.rs b/crates/neuromesh-mcp/src/response.rs index 7bd16d9..99c8bfd 100644 --- a/crates/neuromesh-mcp/src/response.rs +++ b/crates/neuromesh-mcp/src/response.rs @@ -150,8 +150,12 @@ fn folds_for_path( file_path: &Path, fold_ids: &[String], ) -> Vec { + // A file reached by several seeds lists its folds once per seed in + // `fold_ids`; the packet names each fold once. + let mut seen = std::collections::HashSet::new(); fold_ids .iter() + .filter(|id| seen.insert(id.as_str())) .filter_map(|id| { let stored = registry.get_fold(id)?; if path_eq(&stored.file_path, file_path) { diff --git a/docs/measured.md b/docs/measured.md index 2f91505..0806b8a 100644 --- a/docs/measured.md +++ b/docs/measured.md @@ -26,6 +26,8 @@ not the project's number.** Only the holdout rows are. | holdout-web (30 q) | fastify/demo (Fastify API), shadcn-ui/taxonomy (Next.js app router) | dev-class for the web domain (fixed on since session 13; 10 blind questions added in G4 scored 0.58 before fixes) | **0.917** | **0.643** | **0** | — | | **private** | one closed-source B2B backend+frontend (Fastify/Drizzle + Next.js, ~1.2k files) | never tuned on; gold and checkout live outside this repo | **1.000** | **0.587** | **0** | — | | concept (14 q) | this repository, plain-language questions (no identifier in the prompt) | dev-class (written 2026-09-30 from the upstream author's report, tuned on in session 16) | 0.679 | 0.392 | 0 | — || **concept-holdout** (12 q) | ripgrep 14.1.1 (Rust), plain-language questions | never tuned on; gold locked before any run; 1.0.0 scored recall **0.042** | **0.500** | **0.152** | **0** | — | +| **concept-holdout** (12 q) | ripgrep 14.1.1 (Rust), plain-language questions | never tuned on; gold locked before any run; 1.0.0 scored recall **0.042** | **0.500** | **0.156** | **0** | — | +| **concept-holdout2** (12 q) | click 8.1.7 (Python, 16 source files), plain-language questions | never tuned on; gold locked before the single run (2026-10-01) | **0.958** | **0.342** | **0** | — | - **recall / precision** are file-level against a hand-written gold (`gold_files`) per question. A forbidden file in the packet zeroes that question's precision. @@ -135,3 +137,32 @@ caller counts, and every ranked candidate with its score breakdown — the tooli 112 gold tasks, 15 repositories, both binaries on the same machine and day, scored by file name from each engine's `optimize` output (`scripts/compare-baseline.sh`; raw rows in `baseline-vs-fork-2026-09-21.txt`). Baseline recall 0.735 / precision 0.206 / 12 forbidden; fork 0.938 / 0.633 / 7. Baseline has no parser for Scala, R, Julia (recall 0 there), skips `.sh`/`.ps1`/`.ipynb`, and answers config questions at 0.5–0.75 recall. Where the baseline already worked (Go, Python), recall is equal and precision is ~3× higher. Both are weak on the Fastify + Next.js set (0.65 recall). Claude Code session A/B (`claude -p`, Sonnet, same task): Django 3.5k files — plain 8 turns / 233k context tokens / $0.239, with the engine 5 turns / 129k / $0.204; fastify/demo 68 files — 244k / $0.408 vs 194k / $0.391. The engine only shrinks the code-context share of a session; system prompt and tool schemas are re-read every turn. + +## Against outside baselines (2026-10-01) + +Same gold, same checkouts, `python scripts/compare_baselines.py --aider`. +Our engine ships a packet of variable size; the baselines are cut at k files. +**bm25** is plain file-level BM25 over the source text (what a search box does). +**aider-repomap** is Aider 0.86.2's own `RepoMap.get_ranked_tags` — PageRank over +definition/reference tags, personalised by the identifiers and file names the question +mentions, the same inputs Aider derives from a chat message. It was built to give a model a +map of the whole repository, not to answer one question, and it shows. + +| set | ours recall / precision | bm25 R@1 / P@1 | bm25 R@3 / P@3 | aider R@5 / P@5 | +|---|---|---|---|---| +| holdout-2 (gin, torchvision) | **1.000 / 0.700** | 0.675 / 0.700 | 0.950 / 0.333 | 0.475 / 0.110 | +| holdout-c (libuv, fmt) | **1.000 / 0.589** | 0.594 / 0.625 | 0.938 / 0.334 | 0.406 / 0.087 | +| holdout-ml2 (peft, keras-hub) | **1.000 / 0.632** | 0.400 / 0.700 | 0.683 / 0.400 | 0.167 / 0.080 | +| holdout-lang (os-lib, cli, Flux.jl) | 1.000 / 0.589 | **0.933 / 0.933** | 1.000 / 0.333 | 0.400 / 0.080 ¹ | +| concept-holdout (ripgrep, plain language) | 0.500 / 0.156 | 0.250 / 0.333 | 0.500 / **0.222** | 0.042 / 0.017 | +| concept-holdout2 (click, plain language) | **0.958 / 0.342** | 0.417 / 0.500 | 0.750 / 0.306 | 0.333 / 0.067 | + +Where we stand, plainly: on questions that name code, the engine gets every gold file at a +precision no fixed cut of BM25 reaches on three of four holdouts. On holdout-lang the single top +BM25 file is right 93% of the time, so our wider packet costs precision there. On plain-language +questions the picture splits: level with BM25 on ripgrep (12 crates, terse names), ahead of it +on click (recall 0.958 at precision 0.342 against 0.750 at 0.306 for three files) — a small +repository, which helps every method. + +¹ Aider's tree-sitter query for Julia fails to load (`Invalid node type: module`); Flux.jl's +five questions are left out of its row. diff --git a/scripts/benchmark-fast.sh b/scripts/benchmark-fast.sh index 5cc1804..7c2ccb7 100644 --- a/scripts/benchmark-fast.sh +++ b/scripts/benchmark-fast.sh @@ -26,6 +26,7 @@ declare -A MANIFEST=( [holdout-cfg]="tests/third_party/holdout-cfg/repos.toml" [holdout-web]="tests/third_party/holdout-web/repos.toml" [concept-holdout]="tests/third_party/concept-holdout/repos.toml" + [concept-holdout2]="tests/third_party/concept-holdout2/repos.toml" ) declare -A TEST=( [dev]="third_party_gold" @@ -39,6 +40,7 @@ declare -A TEST=( [holdout-web]="third_party_web_holdout_gold" [concept]="third_party_private_gold" [concept-holdout]="third_party_private_gold" + [concept-holdout2]="third_party_private_gold" ) sets=("$@") if [ ${#sets[@]} -eq 0 ]; then @@ -73,6 +75,7 @@ run_set() { case "$set" in concept) envs=(NM_PRIVATE_SET_DIR="$root/tests/third_party/concept" NM_PRIVATE_DIR="$root/..") ;; concept-holdout) envs=(NM_PRIVATE_SET_DIR="$root/tests/third_party/concept-holdout" NM_PRIVATE_DIR="$root/target/third_party/concept-holdout") ;; + concept-holdout2) envs=(NM_PRIVATE_SET_DIR="$root/tests/third_party/concept-holdout2" NM_PRIVATE_DIR="$root/target/third_party/concept-holdout2") ;; esac # The harness resolves the workspace from its manifest dir at build time. (cd crates/neuromesh-context && env "${envs[@]}" "$bin" --nocapture >"$root/$out/$set.log" 2>&1 || true) diff --git a/scripts/benchmark-holdout.sh b/scripts/benchmark-holdout.sh index 7d116b2..a83018b 100644 --- a/scripts/benchmark-holdout.sh +++ b/scripts/benchmark-holdout.sh @@ -29,6 +29,7 @@ declare -A MANIFEST=( [holdout-cfg]="tests/third_party/holdout-cfg/repos.toml" [holdout-web]="tests/third_party/holdout-web/repos.toml" [concept-holdout]="tests/third_party/concept-holdout/repos.toml" + [concept-holdout2]="tests/third_party/concept-holdout2/repos.toml" ) declare -A TEST=( [dev]="third_party_gold" @@ -43,6 +44,7 @@ declare -A TEST=( [private]="third_party_private_gold" [concept]="third_party_private_gold" [concept-holdout]="third_party_private_gold" + [concept-holdout2]="third_party_private_gold" ) sets=("$@") if [ ${#sets[@]} -eq 0 ]; then sets=(dev large holdout holdout-c holdout-lang holdout-ml holdout-ml2 holdout-cfg holdout-web); fi @@ -65,6 +67,7 @@ for set in "${sets[@]}"; do case "$set" in concept) envs=(NM_PRIVATE_SET_DIR="$root/tests/third_party/concept" NM_PRIVATE_DIR="$root/..") ;; concept-holdout) envs=(NM_PRIVATE_SET_DIR="$root/tests/third_party/concept-holdout" NM_PRIVATE_DIR="$root/target/third_party/concept-holdout") ;; + concept-holdout2) envs=(NM_PRIVATE_SET_DIR="$root/tests/third_party/concept-holdout2" NM_PRIVATE_DIR="$root/target/third_party/concept-holdout2") ;; esac line=$(env "${envs[@]}" cargo test -q -p neuromesh-context --test "${TEST[$set]}" -- --nocapture 2>&1 \ | grep -E "^third_party" | tail -1 || true) diff --git a/tests/third_party/concept-holdout2/click/gold_tasks.toml b/tests/third_party/concept-holdout2/click/gold_tasks.toml new file mode 100644 index 0000000..7cbe174 --- /dev/null +++ b/tests/third_party/concept-holdout2/click/gold_tasks.toml @@ -0,0 +1,86 @@ +# Locked 2026-10-01 before any engine run on click. Plain-language questions only; +# gold from reading the source (grep-checked). + +[[task]] +id = "click_tab_completion" +prompt = "How does it produce the scripts that let a shell complete commands when the user presses tab?" +gold_files = ["src/click/shell_completion.py"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "click_wrap_help" +prompt = "How is long help text wrapped to fit the width of the terminal?" +gold_files = ["src/click/formatting.py"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "click_progress" +prompt = "How can a program show a progress indicator while it works through many items?" +gold_files = ["src/click/_termui_impl.py", "src/click/termui.py"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "click_value_types" +prompt = "How is a value typed on the command line converted into a number, a file or one of several allowed words?" +gold_files = ["src/click/types.py"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "click_test_runner" +prompt = "How can a test run a command and capture what it printed?" +gold_files = ["src/click/testing.py"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "click_open_editor" +prompt = "How does it open the user's text editor so they can write a longer message?" +gold_files = ["src/click/_termui_impl.py", "src/click/termui.py"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "click_text_color" +prompt = "How is color added to text printed to the terminal?" +gold_files = ["src/click/termui.py"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "click_decorators" +prompt = "How does decorating a function turn it into a command with options attached?" +gold_files = ["src/click/decorators.py"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "click_unknown_option" +prompt = "What happens when the user passes an option the program does not know, and where is that error defined?" +gold_files = ["src/click/exceptions.py", "src/click/parser.py"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "click_windows_console" +prompt = "How does output to the Windows console deal with unicode characters?" +gold_files = ["src/click/_winconsole.py"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "click_find_subcommand" +prompt = "How does a group of commands find the sub command the user typed?" +gold_files = ["src/click/core.py"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "click_env_values" +prompt = "How can option values come from environment variables instead of the command line?" +gold_files = ["src/click/core.py"] +forbidden_files = [] +expect_seeds_missed = false diff --git a/tests/third_party/concept-holdout2/repos.toml b/tests/third_party/concept-holdout2/repos.toml new file mode 100644 index 0000000..1d60ab2 --- /dev/null +++ b/tests/third_party/concept-holdout2/repos.toml @@ -0,0 +1,10 @@ +# Second plain-language holdout (2026-10-01): questions with no identifier, +# on a Python library nobody tuned on. Gold written from the source before +# any engine run (G1: gold files checked by grep); run once. +# +# click — pallets/click 8.1.7: command-line interface library + +[[repo]] +name = "click" +url = "https://github.com/pallets/click" +rev = "874ca2bc1c30d93a4ac6e36a15ed685eafe89097"