From c5e213c12c6d367e0f14164246f1121b2c513aa0 Mon Sep 17 00:00:00 2001 From: ParsaVictor <1.parsa.karkooti@gmail.com> Date: Wed, 30 Sep 2026 23:10:14 +0330 Subject: [PATCH 1/7] gold: concept-question sets (self dev + ripgrep holdout), locked before any run Co-Authored-By: Claude Opus 5.5 --- tests/third_party/concept-holdout/repos.toml | 11 ++ .../concept-holdout/ripgrep/gold_tasks.toml | 86 +++++++++++++++ .../third_party/concept/repo/gold_tasks.toml | 103 ++++++++++++++++++ tests/third_party/concept/repos.toml | 4 + 4 files changed, 204 insertions(+) create mode 100644 tests/third_party/concept-holdout/repos.toml create mode 100644 tests/third_party/concept-holdout/ripgrep/gold_tasks.toml create mode 100644 tests/third_party/concept/repo/gold_tasks.toml create mode 100644 tests/third_party/concept/repos.toml diff --git a/tests/third_party/concept-holdout/repos.toml b/tests/third_party/concept-holdout/repos.toml new file mode 100644 index 0000000..14d6e92 --- /dev/null +++ b/tests/third_party/concept-holdout/repos.toml @@ -0,0 +1,11 @@ +# Concept-question holdout (2026-09-30): plain-language questions, no +# identifier in the prompt, on a Rust workspace we never tuned on. +# Gold written from reading the source before any engine run; run once, +# after the concept work on this repo is frozen. +# +# ripgrep — BurntSushi/ripgrep 14.1.1: multi-crate Rust workspace + +[[repo]] +name = "ripgrep" +url = "https://github.com/BurntSushi/ripgrep" +rev = "4649aa9700619f94cf9c66876e9549d83420e16c" diff --git a/tests/third_party/concept-holdout/ripgrep/gold_tasks.toml b/tests/third_party/concept-holdout/ripgrep/gold_tasks.toml new file mode 100644 index 0000000..c2437c5 --- /dev/null +++ b/tests/third_party/concept-holdout/ripgrep/gold_tasks.toml @@ -0,0 +1,86 @@ +# Locked 2026-09-30 before any engine run on ripgrep. Plain-language +# questions only; gold from reading the source (grep-checked). + +[[task]] +id = "rg_binary_detection" +prompt = "How does it tell that a file is binary while reading it, and what happens then?" +gold_files = ["crates/searcher/src/line_buffer.rs"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "rg_ignore_rules" +prompt = "How are the rules from ignore files matched against paths while walking directories?" +gold_files = ["crates/ignore/src/gitignore.rs", "crates/ignore/src/dir.rs"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "rg_compressed" +prompt = "How does it search inside compressed files such as gzip archives?" +gold_files = ["crates/cli/src/decompress.rs"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "rg_parallel_walk" +prompt = "How is the directory traversal spread across several threads?" +gold_files = ["crates/ignore/src/walk.rs"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "rg_glob_to_regex" +prompt = "How is a shell-style wildcard pattern converted into a regular expression?" +gold_files = ["crates/globset/src/glob.rs"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "rg_hyperlinks" +prompt = "How are clickable links to matching files written into terminal output?" +gold_files = ["crates/printer/src/hyperlink.rs"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "rg_json_output" +prompt = "How are search results emitted in a machine-readable format for other programs?" +gold_files = ["crates/printer/src/json.rs"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "rg_memory_map" +prompt = "When does it read a file through memory mapping instead of normal reads?" +gold_files = ["crates/searcher/src/searcher/mmap.rs"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "rg_smart_case" +prompt = "How does the search become case-insensitive only when the pattern has no capital letters?" +gold_files = ["crates/regex/src/config.rs"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "rg_file_types" +prompt = "How are language names like rust or python mapped to file name extensions?" +gold_files = ["crates/ignore/src/types.rs", "crates/ignore/src/default_types.rs"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "rg_context_lines" +prompt = "How are the lines before and after a match collected for display?" +gold_files = ["crates/searcher/src/searcher/core.rs"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "rg_terminal_colors" +prompt = "How does it decide whether output goes to a terminal so colors can be turned on?" +gold_files = ["crates/cli/src/lib.rs"] +forbidden_files = [] +expect_seeds_missed = false diff --git a/tests/third_party/concept/repo/gold_tasks.toml b/tests/third_party/concept/repo/gold_tasks.toml new file mode 100644 index 0000000..85362b1 --- /dev/null +++ b/tests/third_party/concept/repo/gold_tasks.toml @@ -0,0 +1,103 @@ +# Concept questions about this repository: plain-language, no identifier in +# the prompt. Gold written by reading the code, locked before any packet was +# looked at (2026-09-30). The first four are the upstream author's battery +# (root_fs_safety, reinforcement, max_files_cap, retry); retry's gold is the +# provider files because this fork does implement retry/backoff. + +[[task]] +id = "concept_root_fs_safety" +prompt = "How does this tool prevent indexing dangerous paths like the filesystem root?" +gold_files = ["crates/neuromesh-index/src/confine.rs"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "concept_reinforcement" +prompt = "How does a file's importance get reinforced after repeated edits?" +gold_files = ["crates/neuromesh-graph/src/edge.rs", "crates/neuromesh-graph/src/graph.rs"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "concept_max_files_cap" +prompt = "How does the system determine the maximum number of files to index automatically?" +gold_files = ["crates/neuromesh-index/src/walker.rs", "crates/neuromesh-cli/src/commands/mod.rs"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "concept_retry" +prompt = "Is there retry logic or exponential backoff when calling the AI provider API?" +gold_files = ["crates/neuromesh-provider/src/anthropic.rs", "crates/neuromesh-provider/src/openai.rs"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "concept_project_identity" +prompt = "How is a project's identity computed so the same repository always maps to the same id?" +gold_files = ["crates/neuromesh-core/src/project_id.rs"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "concept_notebooks" +prompt = "How are Jupyter notebooks turned into something the parser can read?" +gold_files = ["crates/neuromesh-index/src/notebook.rs"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "concept_token_estimate" +prompt = "How does the system estimate how many tokens a piece of text costs?" +gold_files = ["crates/neuromesh-core/src/token.rs"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "concept_fold_bodies" +prompt = "How does it decide which function bodies to hide and show only as signatures in the context?" +gold_files = ["crates/neuromesh-context/src/fold.rs"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "concept_unchanged_files" +prompt = "How does re-indexing skip files that have not changed since last time?" +gold_files = ["crates/neuromesh-index/src/tracker.rs"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "concept_stdin_framing" +prompt = "How are incoming protocol messages read and framed from standard input?" +gold_files = ["crates/neuromesh-mcp/src/stdio.rs"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "concept_provider_choice" +prompt = "Which part picks the AI provider implementation from the configuration?" +gold_files = ["crates/neuromesh-provider/src/factory.rs"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "concept_edge_decay" +prompt = "How do connection weights fade over time when a path is not used?" +gold_files = ["crates/neuromesh-graph/src/edge.rs"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "concept_path_escape" +prompt = "How does the indexer stop symlinks from pointing outside the workspace?" +gold_files = ["crates/neuromesh-index/src/confine.rs"] +forbidden_files = [] +expect_seeds_missed = false + +[[task]] +id = "concept_file_watch" +prompt = "How does the server notice a file changed on disk and update the index?" +gold_files = ["crates/neuromesh-index/src/watcher.rs"] +forbidden_files = [] +expect_seeds_missed = false diff --git a/tests/third_party/concept/repos.toml b/tests/third_party/concept/repos.toml new file mode 100644 index 0000000..07c1ae6 --- /dev/null +++ b/tests/third_party/concept/repos.toml @@ -0,0 +1,4 @@ +[[repo]] +name = "repo" +url = "local" +rev = "HEAD" From 08797d5f6cac91e859ea1dff6757aaa34b064827 Mon Sep 17 00:00:00 2001 From: ParsaVictor <1.parsa.karkooti@gmail.com> Date: Thu, 1 Oct 2026 09:16:44 +0330 Subject: [PATCH 2/7] concept questions: BM25F file ranker, comment index, prose words are guesses Plain-language questions (no identifier) were resolved one word at a time and landed on whatever symbol shared a word. New set tests/third_party/concept (14 q on this repo): recall 0.286 -> 0.679. - F89: "How does this tool ..." made `tool` a strong identifier anchor; a lowercase word the prompt never marks as code is retagged stem: - F90: neuromesh-graph/src/file_rank.rs - BM25F over path, defined names, comments and body, Snowball-stemmed (rust-stemmers); new comment_index in the graph (snapshot v4); W3 seeds the best file and up to two runners-up within 70%, and drops guesses far below it - F91: general software thesaurus (thesaurus.txt, weight 0.35), with no topic of the ripgrep holdout in it - NM_RANK_PROBE=1 in the gold harness shows where the ranker puts gold Co-Authored-By: Claude Opus 5.5 --- Cargo.lock | 11 + .../neuromesh-context/src/activator_seed.rs | 118 ++++- .../tests/support/gold_set.rs | 37 ++ crates/neuromesh-graph/Cargo.toml | 1 + crates/neuromesh-graph/src/file_rank.rs | 483 ++++++++++++++++++ crates/neuromesh-graph/src/graph.rs | 44 +- crates/neuromesh-graph/src/intern.rs | 101 ++++ crates/neuromesh-graph/src/lib.rs | 2 + crates/neuromesh-graph/src/quality_tests.rs | 1 + crates/neuromesh-graph/src/thesaurus.txt | 49 ++ docs/planning/stage5-findings.fa.md | 13 + 11 files changed, 857 insertions(+), 3 deletions(-) create mode 100644 crates/neuromesh-graph/src/file_rank.rs create mode 100644 crates/neuromesh-graph/src/thesaurus.txt diff --git a/Cargo.lock b/Cargo.lock index f033e2d..01a6af5 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1889,6 +1889,7 @@ dependencies = [ "neuromesh-parser", "parking_lot", "rayon", + "rust-stemmers", "serde", "serde_json", "sha2", @@ -2787,6 +2788,16 @@ dependencies = [ "windows-sys 0.52.0", ] +[[package]] +name = "rust-stemmers" +version = "1.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e46a2036019fdb888131db7a4c847a1063a7493f971ed94ea82c67eada63ca54" +dependencies = [ + "serde", + "serde_derive", +] + [[package]] name = "rustix" version = "1.1.4" diff --git a/crates/neuromesh-context/src/activator_seed.rs b/crates/neuromesh-context/src/activator_seed.rs index 9ecdd62..a8145e9 100644 --- a/crates/neuromesh-context/src/activator_seed.rs +++ b/crates/neuromesh-context/src/activator_seed.rs @@ -7,6 +7,37 @@ use neuromesh_core::{NodeId, SeedResolutionConfig, TaskIntent, TaskSignature}; use neuromesh_graph::NeuralProjectGraph; use neuromesh_task::{is_prompt_stopword, normalize_prompt_tokens}; +/// An all-lowercase letters-only word the prompt writes as prose: never in +/// backticks, never as `word()`, `word.` member or `::word`. Such a word +/// can name a symbol, but the question gives no sign that it does. +fn is_prose_word(prompt: &str, ident: &str) -> bool { + if ident.len() < 3 || !ident.chars().all(|c| c.is_ascii_lowercase()) { + return false; + } + let marked = [ + format!("`{ident}`"), + format!("{ident}("), + format!("::{ident}"), + format!(".{ident}"), + format!("{ident}."), + ]; + let lower = prompt.to_lowercase(); + // `{ident}.` at the end of a sentence is prose; only count it when a + // word character follows the dot. + !marked.iter().enumerate().any(|(i, m)| { + if i == 4 { + lower.match_indices(m.as_str()).any(|(pos, _)| { + lower[pos + m.len()..] + .chars() + .next() + .is_some_and(|c| c.is_ascii_alphanumeric() || c == '_') + }) + } else { + lower.contains(m.as_str()) + } + }) +} + pub(crate) fn push_anchor_queries( graph: &NeuralProjectGraph, signature: &TaskSignature, @@ -30,7 +61,11 @@ pub(crate) fn push_anchor_queries( let fragment_of_other = signature.identifiers.iter().any(|other| { other != ident && other.to_lowercase().contains(&ident.to_lowercase()) }); - if crate::seed::sink::is_acronym(ident) && !fragment_of_other { + // Same for a plain lowercase word the prompt never marks as code + // ("How does this tool prevent…" → a function called `tool`): + // an English word that happens to be a symbol name is a guess. + let prose_word = is_prose_word(prompt, ident); + if (crate::seed::sink::is_acronym(ident) && !fragment_of_other) || prose_word { let tag = format!("identifier:{ident}"); let buffers = sink.buffers_mut(); let guessed: Vec = buffers @@ -38,7 +73,7 @@ pub(crate) fn push_anchor_queries( .iter() .filter(|s| s.query == tag) .filter_map(|s| s.resolved_id.clone()) - .filter(|id| graph.get_node(id).is_some_and(|n| n.name != *ident)) + .filter(|id| prose_word || graph.get_node(id).is_some_and(|n| n.name != *ident)) .collect(); if !guessed.is_empty() { let retag = format!("stem:{ident}"); @@ -941,6 +976,82 @@ pub(crate) fn push_compound_symbol_seeds( /// elsewhere on one prefix-matched word (`token:pheromone` → /// `PheromoneConfig` in edge.rs) is dropped when the body hit spells two /// or more words and the guess's file spells fewer. +/// W3 by field-weighted BM25 (`NeuralProjectGraph::file_rank`): the whole +/// question against every file's path, defined names, comments and body, +/// stemmed. The best file is seeded, and up to two runners-up that score +/// within 30% of it (recall over precision; a plain-language question +/// often spans two or three files, as with path-word seeds). Guess seeds +/// that landed on a file the ranking puts far below the best — one prose +/// word that happened to be a function name elsewhere — give way. +/// Returns false (legacy W3 runs) when nothing spells two question words. +fn push_ranked_file_seeds( + graph: &NeuralProjectGraph, + prompt: &str, + config: &SeedResolutionConfig, + names_low: bool, + sink: &mut SeedSink<'_, '_, '_>, +) -> bool { + use crate::seed::weak_file_seed::{prefix, WEAK}; + const RUNNER_UP: f32 = 0.7; + const GUESS_FLOOR: f32 = 0.5; + let ranked: Vec = graph + .file_rank(prompt, 200) + .into_iter() + .filter(|r| names_low || !crate::selector::is_noise_path(&r.path)) + .collect(); + let Some(best) = ranked.first() else { + return false; + }; + if best.matched < 2 { + return false; + } + let top = best.score; + let picks: Vec<&neuromesh_graph::RankedFile> = ranked + .iter() + .take(3) + .filter(|r| r.score >= top * RUNNER_UP) + .collect(); + let score_of = |path: &std::path::Path| -> f32 { + ranked + .iter() + .find(|r| r.path == path) + .map(|r| r.score) + .unwrap_or(0.0) + }; + let weaker: Vec = sink + .resolutions() + .iter() + .filter(|s| WEAK.contains(&prefix(&s.query)) || prefix(&s.query) == "stem") + .filter_map(|s| s.resolved_id.clone()) + .filter(|id| { + graph.get_node(id).is_some_and(|n| { + !picks.iter().any(|p| p.path == n.file_path) + && score_of(&n.file_path) < top * GUESS_FLOOR + }) + }) + .collect(); + if !weaker.is_empty() { + let buffers = sink.buffers_mut(); + for s in buffers.resolutions.iter_mut() { + if s.resolved_id.as_ref().is_some_and(|id| weaker.contains(id)) { + s.resolved_id = None; + s.confidence = 0.0; + s.resolution_tier = None; + } + } + for id in &weaker { + buffers.energies.remove(id); + buffers.reasons.remove(id); + } + } + for (pos, r) in picks.iter().enumerate() { + let energy = signal_weight(config, SignalKind::PathHint, pos + 1); + let rel = r.path.to_string_lossy().replace('\\', "/"); + sink.push(graph, prompt, rel, energy, "body"); + } + true +} + pub(crate) fn push_body_word_seeds( graph: &NeuralProjectGraph, prompt: &str, @@ -968,6 +1079,9 @@ pub(crate) fn push_body_word_seeds( if anchored { return; } + if push_ranked_file_seeds(graph, prompt, config, names_low, sink) { + return; + } let words: Vec = { let mut seen = std::collections::HashSet::new(); prompt diff --git a/crates/neuromesh-context/tests/support/gold_set.rs b/crates/neuromesh-context/tests/support/gold_set.rs index da6c73c..ae699ea 100644 --- a/crates/neuromesh-context/tests/support/gold_set.rs +++ b/crates/neuromesh-context/tests/support/gold_set.rs @@ -135,6 +135,43 @@ pub fn run_gold_set(set: &str) -> GoldSetSummary { Vec::new() }; for task in &gold { + // `NM_RANK_PROBE=1`: where the BM25F file ranker puts the gold, + // independent of the seed pipeline. + if std::env::var("NM_RANK_PROBE").is_ok_and(|v| v == "1") { + let ranked: Vec<_> = graph + .file_rank(&task.prompt, 200) + .into_iter() + .filter(|r| !neuromesh_context::selector::is_noise_path(&r.path)) + .collect(); + let pos = |g: &String| { + ranked + .iter() + .position(|r| { + r.path + .to_string_lossy() + .replace('\\', "/") + .ends_with(g.as_str()) + }) + .map(|p| (p + 1).to_string()) + .unwrap_or_else(|| "-".into()) + }; + let gold_pos: Vec = task.gold_files.iter().map(pos).collect(); + let top: Vec = ranked + .iter() + .take(3) + .map(|r| { + format!( + "{}:{:.1}", + r.path.to_string_lossy().replace('\\', "/"), + r.score + ) + }) + .collect(); + lines.push(format!( + " rank {:<28} gold@{:?} top={:?}", + task.id, gold_pos, top + )); + } let activator = ContextActivator::new(Arc::new(ReversibleContextRegistry::new())); let view = activator.activate( &graph, diff --git a/crates/neuromesh-graph/Cargo.toml b/crates/neuromesh-graph/Cargo.toml index 95865a5..20c0ea7 100644 --- a/crates/neuromesh-graph/Cargo.toml +++ b/crates/neuromesh-graph/Cargo.toml @@ -18,6 +18,7 @@ tracing.workspace = true parking_lot.workspace = true rayon.workspace = true sha2 = "0.10" +rust-stemmers = "1.2" simsimd = { version = "6.4", default-features = false } [features] diff --git a/crates/neuromesh-graph/src/file_rank.rs b/crates/neuromesh-graph/src/file_rank.rs new file mode 100644 index 0000000..0dcd092 --- /dev/null +++ b/crates/neuromesh-graph/src/file_rank.rs @@ -0,0 +1,483 @@ +//! Field-weighted BM25 (BM25F) over files, for plain-language questions. +//! +//! A question that names no identifier ("where are uploaded images +//! resized?") used to be resolved one word at a time, and a lone word lands +//! on whatever symbol happens to share it. Here the question is read as a +//! whole against four fields of every file: its *path* (the author's own +//! one-line summary), the *names it defines*, its *comments* (the prose +//! written about the code, closest to how a question is phrased) and its +//! *body*. Words are stemmed (Snowball English), so `resized` meets +//! `resize` and `images` meets `image`. +//! +//! The index is derived from the graph (`word_index`, `comment_index`, +//! `file_to_nodes`) and memoized against the mesh revision; a query is a +//! pass over the postings of its own terms. + +use crate::intern::GraphData; +use neuromesh_core::{NodeId, NodeType}; +use rust_stemmers::{Algorithm, Stemmer}; +use std::collections::{HashMap, HashSet}; + +/// Field weights: the path is the author's own one-line summary of a file, +/// defined names the next best, the body (every word it spells) the weakest. +const W_PATH: f32 = 3.0; +const W_SYMBOL: f32 = 2.0; +const W_BODY: f32 = 1.0; +const W_COMMENT: f32 = 2.0; +const B_COMMENT: f32 = 0.75; +const K1: f32 = 1.2; +const B_BODY: f32 = 0.75; +const B_SYMBOL: f32 = 0.5; + +/// English function words and question scaffolding. Code-generic words +/// (`file`, `code`, `data`) are left to idf. +const STOPWORDS: &[&str] = &[ + "a", + "about", + "after", + "all", + "also", + "an", + "and", + "any", + "are", + "as", + "at", + "be", + "been", + "before", + "being", + "between", + "both", + "but", + "by", + "can", + "could", + "did", + "do", + "does", + "doing", + "done", + "each", + "for", + "from", + "get", + "gets", + "got", + "had", + "has", + "have", + "how", + "i", + "if", + "in", + "into", + "is", + "it", + "its", + "just", + "like", + "many", + "more", + "most", + "much", + "my", + "no", + "not", + "of", + "on", + "once", + "only", + "or", + "other", + "our", + "out", + "over", + "own", + "same", + "should", + "since", + "so", + "some", + "such", + "than", + "that", + "the", + "their", + "them", + "then", + "there", + "these", + "they", + "this", + "those", + "through", + "to", + "too", + "under", + "until", + "up", + "us", + "use", + "used", + "uses", + "using", + "very", + "via", + "was", + "way", + "we", + "were", + "what", + "when", + "where", + "whether", + "which", + "while", + "who", + "why", + "will", + "with", + "within", + "without", + "would", + "you", + "your", + "there", + "here", + "happens", + "happen", + "work", + "works", + "system", + "tool", + "program", + "codebase", + "logic", + "part", + "am", + "go", + "he", + "me", + "piece", + "thing", + "things", + "somewhere", + "someone", + "handled", + "handle", + "handles", + "implemented", + "implement", + "implementation", + "decide", + "decides", + "determine", + "determines", + "make", + "makes", + "made", + "turn", + "turned", + "turns", + "time", + "again", +]; + +/// One ranked file: its node id, relative path, score and how many distinct +/// query terms it matched in any field. +#[derive(Debug, Clone)] +pub struct RankedFile { + pub id: NodeId, + pub path: std::path::PathBuf, + pub score: f32, + pub matched: usize, +} + +#[derive(Default)] +pub(crate) struct FileRankIndex { + ids: Vec, + paths: Vec, + /// term → docs whose path spells it (tf 1). + path: HashMap>, + /// term → (doc, count of defined names spelling it). + symbol: HashMap>, + /// term → docs whose body spells it (binary, as `word_index` is). + body: HashMap>, + /// term → docs whose comments spell it (binary). + comment: HashMap>, + comment_len: Vec, + avg_comment_len: f32, + symbol_len: Vec, + body_len: Vec, + avg_symbol_len: f32, + avg_body_len: f32, +} + +fn stemmer() -> Stemmer { + Stemmer::create(Algorithm::English) +} + +/// Lowercase, split identifiers into words, drop short/non-alphabetic +/// pieces and stopwords, stem. +fn terms_of(text: &str, st: &Stemmer, keep_stopwords: bool) -> Vec { + let mut out = Vec::new(); + for raw in text.split(|c: char| !c.is_ascii_alphanumeric() && c != '_') { + if raw.is_empty() { + continue; + } + for part in neuromesh_parser::tokenize_ident(raw) { + let w = part.to_lowercase(); + if w.len() < 2 || !w.chars().all(|c| c.is_ascii_alphabetic()) { + continue; + } + if !keep_stopwords && STOPWORDS.contains(&w.as_str()) { + continue; + } + out.push(st.stem(&w).into_owned()); + } + } + out +} + +/// Query terms, stemmed and deduplicated in prompt order. +pub fn query_terms(prompt: &str) -> Vec { + weighted_query_terms(prompt) + .into_iter() + .filter(|(_, w)| *w >= 1.0) + .map(|(t, _)| t) + .collect() +} + +/// Weight of a word the question did not say but a thesaurus cluster +/// offers for one it did. +const SYNONYM_WEIGHT: f32 = 0.35; + +/// Question terms (weight 1) plus thesaurus stand-ins (weight +/// [`SYNONYM_WEIGHT`]): a question word in a cluster also searches the +/// rest of its cluster. +pub fn weighted_query_terms(prompt: &str) -> Vec<(String, f32)> { + let st = stemmer(); + let mut out: Vec<(String, f32)> = Vec::new(); + let mut seen = HashSet::new(); + for t in terms_of(prompt, &st, false) { + if seen.insert(t.clone()) { + out.push((t, 1.0)); + } + } + let lower = prompt.to_lowercase(); + let mut extra: Vec = Vec::new(); + if lower.contains("how many") || lower.contains("number of") { + extra.push("count".into()); + } + let words: Vec = lower + .split(|c: char| !c.is_ascii_alphanumeric()) + .filter(|w| !w.is_empty()) + .map(str::to_string) + .collect(); + for cluster in clusters() { + let stems: Vec = cluster.iter().map(|w| st.stem(w).into_owned()).collect(); + let hit = words + .iter() + .any(|w| cluster.contains(&w.as_str()) || stems.contains(&st.stem(w).into_owned())); + if hit { + extra.extend(cluster.iter().map(|w| w.to_string())); + } + } + for w in extra { + let t = st.stem(&w).into_owned(); + if seen.insert(t.clone()) { + out.push((t, SYNONYM_WEIGHT)); + } + } + out +} + +/// How people say it vs how code spells it: one cluster of interchangeable +/// words per line of `thesaurus.txt` (general software vocabulary, not +/// written for any one repository). Kept out of the source so a repository +/// that indexes this crate does not see every synonym in one file. +fn clusters() -> &'static [Vec<&'static str>] { + static CLUSTERS: std::sync::OnceLock>> = std::sync::OnceLock::new(); + CLUSTERS.get_or_init(|| { + include_str!("thesaurus.txt") + .lines() + .map(str::trim) + .filter(|l| !l.is_empty() && !l.starts_with('#')) + .map(|l| { + l.split(',') + .map(str::trim) + .filter(|w| !w.is_empty()) + .collect() + }) + .collect() + }) +} + +impl FileRankIndex { + pub(crate) fn build(data: &GraphData) -> Self { + let st = stemmer(); + let mut idx = FileRankIndex::default(); + let mut doc_of: HashMap = HashMap::new(); + for (path, node_ids) in &data.file_to_nodes { + let Some(file_id) = node_ids.iter().find(|id| { + data.mesh + .node(id) + .is_some_and(|n| n.node_type == NodeType::File) + }) else { + continue; + }; + let doc = idx.ids.len() as u32; + idx.ids.push(file_id.clone()); + idx.paths.push(path.clone()); + doc_of.insert(file_id.clone(), doc); + + let rel = path.to_string_lossy().replace('\\', "/"); + // Path words: every directory and the stem, not the extension. + let without_ext = match rel.rsplit_once('.') { + Some((head, ext)) if !ext.contains('/') => head.to_string(), + _ => rel.clone(), + }; + let mut seen = HashSet::new(); + for t in terms_of(&without_ext, &st, true) { + if seen.insert(t.clone()) { + idx.path.entry(t).or_default().push(doc); + } + } + + let mut counts: HashMap = HashMap::new(); + let mut len = 0u32; + for id in node_ids { + let Some(node) = data.mesh.node(id) else { + continue; + }; + if node.node_type == NodeType::File { + continue; + } + for t in terms_of(&node.name, &st, true) { + len += 1; + let c = counts.entry(t).or_insert(0); + *c = c.saturating_add(1); + } + } + for (t, c) in counts { + idx.symbol.entry(t).or_default().push((doc, c)); + } + idx.symbol_len.push(len); + idx.body_len + .push(data.body_lengths.get(file_id).copied().unwrap_or(0)); + } + for (word, ids) in &data.word_index { + let t = st.stem(word).into_owned(); + let list = idx.body.entry(t).or_default(); + for id in ids { + if let Some(&doc) = doc_of.get(id) { + list.push(doc); + } + } + } + for list in idx.body.values_mut() { + list.sort_unstable(); + list.dedup(); + } + idx.comment_len = vec![0; idx.ids.len()]; + for (word, ids) in &data.comment_index { + let t = st.stem(word).into_owned(); + let list = idx.comment.entry(t).or_default(); + for id in ids { + if let Some(&doc) = doc_of.get(id) { + list.push(doc); + idx.comment_len[doc as usize] += 1; + } + } + } + for list in idx.comment.values_mut() { + list.sort_unstable(); + list.dedup(); + } + let n = idx.ids.len().max(1) as f32; + idx.avg_comment_len = (idx.comment_len.iter().map(|&l| l as f32).sum::() / n).max(1.0); + idx.avg_symbol_len = (idx.symbol_len.iter().map(|&l| l as f32).sum::() / n).max(1.0); + idx.avg_body_len = (idx.body_len.iter().map(|&l| l as f32).sum::() / n).max(1.0); + idx + } + + /// Files ranked by BM25F over `terms` (already stemmed), best first. + pub(crate) fn rank(&self, terms: &[(String, f32)], limit: usize) -> Vec { + let n = self.ids.len() as f32; + if n == 0.0 { + return Vec::new(); + } + // term → doc → weighted, length-normalised tf + let mut scores: HashMap = HashMap::new(); + for (term, weight) in terms { + let mut tf: HashMap = HashMap::new(); + if let Some(docs) = self.path.get(term) { + for &d in docs { + *tf.entry(d).or_insert(0.0) += W_PATH; + } + } + if let Some(docs) = self.symbol.get(term) { + for &(d, c) in docs { + let len = self.symbol_len[d as usize] as f32; + let norm = 1.0 - B_SYMBOL + B_SYMBOL * len / self.avg_symbol_len; + *tf.entry(d).or_insert(0.0) += W_SYMBOL * c as f32 / norm; + } + } + if let Some(docs) = self.body.get(term) { + for &d in docs { + let len = self.body_len[d as usize] as f32; + let norm = 1.0 - B_BODY + B_BODY * len / self.avg_body_len; + *tf.entry(d).or_insert(0.0) += W_BODY / norm; + } + } + if let Some(docs) = self.comment.get(term) { + for &d in docs { + let len = self.comment_len[d as usize] as f32; + let norm = 1.0 - B_COMMENT + B_COMMENT * len / self.avg_comment_len; + *tf.entry(d).or_insert(0.0) += W_COMMENT / norm; + } + } + if tf.is_empty() { + continue; + } + let df = tf.len() as f32; + let idf = ((n - df + 0.5) / (df + 0.5) + 1.0).ln(); + for (d, t) in tf { + let e = scores.entry(d).or_insert((0.0, 0)); + e.0 += weight * idf * t / (K1 + t); + if *weight >= 1.0 { + e.1 += 1; + } + } + } + let mut ranked: Vec = scores + .into_iter() + .map(|(d, (score, matched))| RankedFile { + id: self.ids[d as usize].clone(), + path: self.paths[d as usize].clone(), + score, + matched, + }) + .collect(); + ranked.sort_by(|a, b| { + b.score + .partial_cmp(&a.score) + .unwrap_or(std::cmp::Ordering::Equal) + .then_with(|| a.path.cmp(&b.path)) + }); + ranked.truncate(limit); + ranked + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn query_terms_stem_and_drop_scaffolding() { + let t = query_terms("Where are the uploaded images resized?"); + assert_eq!(t, vec!["upload", "imag", "resiz"]); + } +} diff --git a/crates/neuromesh-graph/src/graph.rs b/crates/neuromesh-graph/src/graph.rs index 9583064..d5322d1 100644 --- a/crates/neuromesh-graph/src/graph.rs +++ b/crates/neuromesh-graph/src/graph.rs @@ -134,6 +134,8 @@ struct DerivedIndexes { learning_boost: Arc>, examples_revision: Option, examples_are_core: bool, + file_rank_key: Option<(u64, u64)>, + file_rank: Arc, } #[derive(Clone)] @@ -637,6 +639,17 @@ impl NeuralProjectGraph { .or_default() .push(file_id.clone()); } + if reingested { + for list in data.comment_index.values_mut() { + list.retain(|existing| *existing != file_id); + } + } + for word in crate::intern::comment_words(src) { + data.comment_index + .entry(word) + .or_default() + .push(file_id.clone()); + } } if keep_source { if let Some(src) = content { @@ -715,6 +728,32 @@ impl NeuralProjectGraph { ranked } + /// Files ranked by field-weighted BM25 (path, defined names, body) over + /// the stemmed words of `prompt` — see [`crate::file_rank`]. + pub fn file_rank(&self, prompt: &str, limit: usize) -> Vec { + let terms = crate::file_rank::weighted_query_terms(prompt); + if terms.is_empty() { + return Vec::new(); + } + self.file_rank_index().rank(&terms, limit) + } + + fn file_rank_index(&self) -> Arc { + let data = self.inner.read(); + let key = (data.mesh.node_revision(), data.generation); + { + let cached = self.derived.read(); + if cached.file_rank_key == Some(key) { + return Arc::clone(&cached.file_rank); + } + } + let index = Arc::new(crate::file_rank::FileRankIndex::build(&data)); + let mut cached = self.derived.write(); + cached.file_rank_key = Some(key); + cached.file_rank = Arc::clone(&index); + index + } + /// How many distinct `words` the body of the file at `path` spells. pub fn body_word_hits(&self, path: &Path, words: &[String]) -> usize { let Some(file_id) = self.file_id_for_path(path) else { @@ -2589,7 +2628,7 @@ impl NeuralProjectGraph { let snapshot = { let data = self.inner.read(); GraphSnapshot { - version: 3, + version: 4, nodes: data .mesh .nodes() @@ -2611,6 +2650,7 @@ impl NeuralProjectGraph { literal_index: data.literal_index.clone(), word_index: data.word_index.clone(), body_lengths: data.body_lengths.clone(), + comment_index: data.comment_index.clone(), } }; if snapshot_structurally_unchanged(path, &snapshot) { @@ -2660,6 +2700,7 @@ impl NeuralProjectGraph { literal_index: HashMap::new(), word_index: HashMap::new(), body_lengths: HashMap::new(), + comment_index: HashMap::new(), }); return Ok(true); } @@ -2683,6 +2724,7 @@ impl NeuralProjectGraph { data.literal_index = snapshot.literal_index; data.word_index = snapshot.word_index; data.body_lengths = snapshot.body_lengths; + data.comment_index = snapshot.comment_index; if snapshot.workspace_root.is_some() { data.workspace_root = snapshot.workspace_root; } diff --git a/crates/neuromesh-graph/src/intern.rs b/crates/neuromesh-graph/src/intern.rs index 7eac7b7..b31ae42 100644 --- a/crates/neuromesh-graph/src/intern.rs +++ b/crates/neuromesh-graph/src/intern.rs @@ -380,6 +380,10 @@ pub(crate) struct GraphData { pub word_index: HashMap>, /// Distinct body words per file — the document length BM25 normalises by. pub body_lengths: HashMap, + /// Comment words → files: the prose an author wrote about the code + /// (line, block and doc comments, docstrings). Plain-language questions + /// match it better than identifiers do — see `file_rank`. + pub comment_index: HashMap>, } #[derive(Serialize, Deserialize)] @@ -412,6 +416,8 @@ pub(crate) struct GraphSnapshot { pub word_index: HashMap>, #[serde(default)] pub body_lengths: HashMap, + #[serde(default)] + pub comment_index: HashMap>, } #[derive(Deserialize)] @@ -582,6 +588,9 @@ pub(crate) fn remove_file_nodes_locked(data: &mut GraphData, path: &Path) { list.retain(|existing| existing != id); } data.body_lengths.remove(id); + for list in data.comment_index.values_mut() { + list.retain(|existing| existing != id); + } if let Some(node) = data.mesh.node(id).cloned() { unindex_name_keys(data, id, &node.name); if let Some(parent) = &node.parent { @@ -741,6 +750,98 @@ pub(crate) fn is_name_like_literal(s: &str) -> bool { /// The distinct words a source file spells: identifiers split into their /// parts, prose in comments and strings, 4+ ASCII letters, lowercased. What /// a grep for a prompt word would hit, minus the noise of two-letter tokens. +/// Distinct lowercase words (2+ letters) of the file's comments: `//`, `#`, +/// `--` line comments (also trailing ones), `/* … */` blocks, and Python +/// triple-quoted docstrings. Preprocessor lines, Rust attributes and +/// shebangs are code, not prose. Language-agnostic by design — a comment +/// marker inside a string occasionally leaks a few words, which is noise +/// BM25 tolerates. +pub(crate) fn comment_words(content: &str) -> Vec { + let mut text = String::new(); + let mut in_block = false; + let mut in_doc: Option<&str> = None; + for line in content.lines() { + let t = line.trim_start(); + if let Some(q) = in_doc { + match t.find(q) { + Some(end) => { + text.push_str(&t[..end]); + in_doc = None; + } + None => text.push_str(t), + } + text.push('\n'); + continue; + } + if in_block { + match t.find("*/") { + Some(end) => { + text.push_str(&t[..end]); + in_block = false; + } + None => text.push_str(t), + } + text.push('\n'); + continue; + } + if let Some(q) = ["\"\"\"", "'''"].into_iter().find(|q| t.starts_with(q)) { + let rest = &t[3..]; + match rest.find(q) { + Some(end) => text.push_str(&rest[..end]), + None => { + text.push_str(rest); + in_doc = Some(q); + } + } + text.push('\n'); + continue; + } + if t.starts_with("#[") + || t.starts_with("#!") + || t.starts_with("#include") + || t.starts_with("#define") + || t.starts_with("#if") + || t.starts_with("#endif") + || t.starts_with("#pragma") + || t.starts_with("#import") + { + continue; + } + if let Some(rest) = t.strip_prefix("/*") { + match rest.find("*/") { + Some(end) => text.push_str(&rest[..end]), + None => { + text.push_str(rest); + in_block = true; + } + } + text.push('\n'); + continue; + } + if t.starts_with("//") || t.starts_with('#') || t.starts_with("--") { + text.push_str(t.trim_start_matches(['/', '#', '-', '!'])); + text.push('\n'); + continue; + } + // Trailing line comment after code. + if let Some(pos) = line.find(" // ").or_else(|| line.find(" # ")) { + text.push_str(&line[pos + 3..]); + text.push('\n'); + } + } + let mut seen: HashSet = HashSet::new(); + for raw in text.split(|c: char| !c.is_ascii_alphanumeric() && c != '_') { + for part in neuromesh_parser::tokenize_ident(raw) { + if part.len() >= 2 && part.chars().all(|c| c.is_ascii_alphabetic()) { + seen.insert(part.to_lowercase()); + } + } + } + let mut words: Vec = seen.into_iter().collect(); + words.sort(); + words +} + pub(crate) fn body_words(content: &str) -> Vec { let mut seen: HashSet = HashSet::new(); for raw in content.split(|c: char| !c.is_ascii_alphanumeric() && c != '_') { diff --git a/crates/neuromesh-graph/src/lib.rs b/crates/neuromesh-graph/src/lib.rs index 3982105..d3ecb2b 100644 --- a/crates/neuromesh-graph/src/lib.rs +++ b/crates/neuromesh-graph/src/lib.rs @@ -2,6 +2,7 @@ pub mod activation; pub mod concept_index; pub mod edge; pub mod embeddings; +pub mod file_rank; pub mod graph; mod intern; pub mod manifest; @@ -28,6 +29,7 @@ pub use embeddings::{ sidecar_tier_stats, stem_union_file_hits, }; pub use embeddings::{load_sidecar, EmbeddingIndex, EmbeddingSidecar}; +pub use file_rank::RankedFile; pub use graph::{ node_learning_bonus, path_echoes_symbol, GraphStats, IndexState, NeuralProjectGraph, NodeLearningProfile, ProjectIdReconciliation, GRAPH_PARSER_EPOCH, diff --git a/crates/neuromesh-graph/src/quality_tests.rs b/crates/neuromesh-graph/src/quality_tests.rs index c25647f..d8bc173 100644 --- a/crates/neuromesh-graph/src/quality_tests.rs +++ b/crates/neuromesh-graph/src/quality_tests.rs @@ -1687,6 +1687,7 @@ class Greeter { literal_index: std::collections::HashMap::new(), word_index: std::collections::HashMap::new(), body_lengths: std::collections::HashMap::new(), + comment_index: std::collections::HashMap::new(), }; let bytes = bincode::serialize(&snapshot).expect("serialize"); std::fs::write(&path, bytes).expect("write"); diff --git a/crates/neuromesh-graph/src/thesaurus.txt b/crates/neuromesh-graph/src/thesaurus.txt new file mode 100644 index 0000000..53c0dc3 --- /dev/null +++ b/crates/neuromesh-graph/src/thesaurus.txt @@ -0,0 +1,49 @@ +limit,cap,max,maximum,ceiling,bound,threshold,quota +min,minimum,floor,least +fade,decay,evaporate,evaporation,expire,expiry,diminish,weaken +hide,fold,collapse,elide,skeleton,stub,omit +estimate,approximate,approximation,heuristic,guess +count,counter,tally,number +cost,price,budget,spend,expense +retry,retries,backoff,reattempt,attempt +dangerous,unsafe,safe,safety,hazard,risky +prevent,guard,protect,forbid,refuse,reject,deny,block,confine,restrict +skip,ignore,exclude,bypass +unchanged,changed,modified,stale,fresh,dirty,fingerprint,checksum,mtime,digest +watch,watcher,monitor,observe,notify,notification +pick,choose,select,factory,resolve,dispatch +identity,id,identifier,uuid,fingerprint +reinforce,strengthen,boost,reward,amplify +importance,weight,score,priority,rank,relevance,salience +connection,edge,link,relation,relationship +parse,decode,deserialize,read,load +save,persist,serialize,store,write,dump,snapshot +message,frame,packet,envelope,payload +start,launch,boot,init,initialize,spawn,bootstrap,startup +stop,shutdown,terminate,kill,exit,teardown +error,failure,fault,exception,panic +log,trace,journal,logger,logging +config,configuration,settings,options,preferences +cache,memo,memoize,memoization +delete,remove,erase,purge,evict +duplicate,dedup,deduplicate,redundant,repeated +split,chunk,segment,partition +merge,combine,join,fuse,union +sort,order,rank,ordering +validate,check,verify,assert,sanitize +convert,transform,translate,map +schedule,timer,cron,interval,periodic +permission,access,authorize,authorization,privilege +user,account,member,profile +database,db,sql,table,query +request,http,endpoint,route,handler +test,spec,fixture,assertion +version,release,semver +dependency,dependencies,import,require +folder,directory,dir +root,top,base +change,edit,modify,update,mutation +provider,backend,vendor,client +summary,digest,overview,outline +signature,declaration,prototype,header +body,implementation,definition diff --git a/docs/planning/stage5-findings.fa.md b/docs/planning/stage5-findings.fa.md index e36c791..0028ee7 100644 --- a/docs/planning/stage5-findings.fa.md +++ b/docs/planning/stage5-findings.fa.md @@ -1329,3 +1329,16 @@ web باقی‌مانده: `fd_login_flow` 1/3 (auth route + password-manager ب | holdout-c 0.589→0.578 (`fmt_parse_format_string`: `format.h` اضافه) | صندلی تمام‌فایل | prompt هم `base.h` را نام برده هم دو تابعش را؛ فایل ظرفِ symbol های نام‌برده است نه آدرس؛ callee های symbol های *دیگر* base.h (stem «format») صندلی گرفتند | فایلی که `identifier:` ای در آن resolve شده از `address_files` بیرون می‌ماند | بعد از هر دو: c 0.589، large 0.675، web 0.856/0.638، self 0.867/0.583 — همه‌ی ۹ مجموعه بدون افت. + +## سؤال‌های مفهومی (session 16، ۲۰۲۶-۰۹-۳۰) — F88–F91 + +منبع: گزارش یوسف (۳/۷ روی ریپوی خودمان). ست جدید `tests/third_party/concept` (۱۴ سؤال، بدون نام شناسه) baseline: recall **0.286**. + +| F | علت ریشه‌ای | اصلاح عمومی | اثر روی concept | +|---|---|---|---| +| F88 | stdout در حالت `mcp`: بنر داشبورد با `println!` | `eprintln!` + تست باینری واقعی (#126) | — | +| F89 | regex «How does [this] X» کلمه‌ی انگلیسی `tool`/`system` را identifier قوی کرد → W3 هرگز اجرا نشد | `is_prose_word`: کلمه‌ی lowercase بدون نشانه‌ی کد (backtick، `()`، `.`، `::`) → `stem:` (مثل acronym در F87) | root_fs درست شد | +| F90 | W3 کلمه‌به‌کلمه: «estimate» → تابعی به همین نام؛ BM25 باینری بدون stem و بدون وزن مسیر/نام/کامنت؛ روی tie بی‌خیال می‌شد | `file_rank.rs`: BM25F (path 3، نام‌ها 2، کامنت‌ها 2، بدنه 1) با Snowball stemmer؛ `comment_index` جدید در گراف؛ تا ۳ فایل با امتیاز ≥۰.۷ بهترین | 0.286 → 0.536 | +| F91 | شکاف واژگانی («fade» ↔ `evaporate`، «how many» ↔ `count`) | thesaurus عمومی ۴۹ خوشه در `thesaurus.txt` (وزن 0.35)؛ عمداً هیچ موضوع ست holdout (ripgrep) در آن نیست | → **0.679** | + +تله: فایل‌های خود ما روی ست self اثر می‌گذارند — thesaurus داخل `.rs` بالای ۴ سؤال آمد؛ به `.txt` (که ایندکس نمی‌شود) منتقل شد. سؤال‌های gold هرگز در کامنت کد نقل نشوند. From b069521b473af6317c06a339ab313a1349d8d13a Mon Sep 17 00:00:00 2001 From: ParsaVictor <1.parsa.karkooti@gmail.com> Date: Thu, 1 Oct 2026 10:20:59 +0330 Subject: [PATCH 3/7] concept: rank cache keyed on structure; prune by unfiltered score; prose halves drop with the compound; word! is code - file_rank index keyed on (generation, file count): learning bumped the node revision every query and rebuilt the index (CI 250 ms gate) - guess pruning scores against every file, tests included - compound-stem pair drops stem: halves too (F89 retag) - `trainstep!` marks code (Julia/Rust macros): holdout-lang back to 0.589 Co-Authored-By: Claude Opus 5.5 --- .../neuromesh-context/src/activator_seed.rs | 20 +++++++++++++------ crates/neuromesh-graph/src/graph.rs | 5 ++++- 2 files changed, 18 insertions(+), 7 deletions(-) diff --git a/crates/neuromesh-context/src/activator_seed.rs b/crates/neuromesh-context/src/activator_seed.rs index a8145e9..1b4ff96 100644 --- a/crates/neuromesh-context/src/activator_seed.rs +++ b/crates/neuromesh-context/src/activator_seed.rs @@ -17,6 +17,8 @@ fn is_prose_word(prompt: &str, ident: &str) -> bool { let marked = [ format!("`{ident}`"), format!("{ident}("), + // Julia mutating functions and Rust macros: `trainstep!`, `vec!`. + format!("{ident}!"), format!("::{ident}"), format!(".{ident}"), format!("{ident}."), @@ -25,7 +27,7 @@ fn is_prose_word(prompt: &str, ident: &str) -> bool { // `{ident}.` at the end of a sentence is prose; only count it when a // word character follows the dot. !marked.iter().enumerate().any(|(i, m)| { - if i == 4 { + if i == 5 { lower.match_indices(m.as_str()).any(|(pos, _)| { lower[pos + m.len()..] .chars() @@ -798,9 +800,12 @@ pub(crate) fn push_compound_stem_seeds( // The halves of the pair are not symbols of their own once the pair // named a file (the bare_owner rule for `owner.member`): the // `identifier:index` that reached some unrelated `index()` goes. + // A half the prompt wrote as prose is retagged `stem:` (F89); it goes too. let halves = [ format!("identifier:{}", pair[0]), format!("identifier:{}", pair[1]), + format!("stem:{}", pair[0]), + format!("stem:{}", pair[1]), ]; let buffers = sink.buffers_mut(); let dropped: Vec = buffers @@ -994,9 +999,10 @@ fn push_ranked_file_seeds( use crate::seed::weak_file_seed::{prefix, WEAK}; const RUNNER_UP: f32 = 0.7; const GUESS_FLOOR: f32 = 0.5; - let ranked: Vec = graph - .file_rank(prompt, 200) - .into_iter() + let all = graph.file_rank(prompt, 200); + let ranked: Vec = all + .iter() + .cloned() .filter(|r| names_low || !crate::selector::is_noise_path(&r.path)) .collect(); let Some(best) = ranked.first() else { @@ -1011,9 +1017,11 @@ fn push_ranked_file_seeds( .take(3) .filter(|r| r.score >= top * RUNNER_UP) .collect(); + // Scored against every file, tests included: a guess that landed on a + // test the prompt spells (`tests/support/index_cache.rs`) is evidence, + // even though tests are not offered as picks. let score_of = |path: &std::path::Path| -> f32 { - ranked - .iter() + all.iter() .find(|r| r.path == path) .map(|r| r.score) .unwrap_or(0.0) diff --git a/crates/neuromesh-graph/src/graph.rs b/crates/neuromesh-graph/src/graph.rs index d5322d1..5614f7a 100644 --- a/crates/neuromesh-graph/src/graph.rs +++ b/crates/neuromesh-graph/src/graph.rs @@ -740,7 +740,10 @@ impl NeuralProjectGraph { fn file_rank_index(&self) -> Arc { let data = self.inner.read(); - let key = (data.mesh.node_revision(), data.generation); + // Structural key only: learning updates bump the node revision on + // every query, and rebuilding (stemming the whole vocabulary) per + // query is what a cache is for avoiding. + let key = (data.generation, data.file_to_nodes.len() as u64); { let cached = self.derived.read(); if cached.file_rank_key == Some(key) { From 860ec8ad0b84c0ee641d4952527aead54797a455 Mon Sep 17 00:00:00 2001 From: ParsaVictor <1.parsa.karkooti@gmail.com> Date: Thu, 1 Oct 2026 10:33:19 +0330 Subject: [PATCH 4/7] concept: a prose word that is part of another prompt identifier is not retagged (predict / predict_proba); clippy holdout-ml back to 0.478. Co-Authored-By: Claude Opus 5.5 --- crates/neuromesh-context/src/activator_seed.rs | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/crates/neuromesh-context/src/activator_seed.rs b/crates/neuromesh-context/src/activator_seed.rs index 1b4ff96..19e14aa 100644 --- a/crates/neuromesh-context/src/activator_seed.rs +++ b/crates/neuromesh-context/src/activator_seed.rs @@ -66,7 +66,7 @@ pub(crate) fn push_anchor_queries( // Same for a plain lowercase word the prompt never marks as code // ("How does this tool prevent…" → a function called `tool`): // an English word that happens to be a symbol name is a guess. - let prose_word = is_prose_word(prompt, ident); + let prose_word = is_prose_word(prompt, ident) && !fragment_of_other; if (crate::seed::sink::is_acronym(ident) && !fragment_of_other) || prose_word { let tag = format!("identifier:{ident}"); let buffers = sink.buffers_mut(); @@ -1002,8 +1002,8 @@ fn push_ranked_file_seeds( let all = graph.file_rank(prompt, 200); let ranked: Vec = all .iter() - .cloned() .filter(|r| names_low || !crate::selector::is_noise_path(&r.path)) + .cloned() .collect(); let Some(best) = ranked.first() else { return false; From 3b089044352c8ef46d39f788bc5973301c0da868 Mon Sep 17 00:00:00 2001 From: ParsaVictor <1.parsa.karkooti@gmail.com> Date: Thu, 1 Oct 2026 11:24:20 +0330 Subject: [PATCH 5/7] latency: no answer from a half-built index; ready before embeddings; big files seeded by their matching definitions Cold start, fresh clone of this repo: first query 3.56 s -> 2.65 s (includes the index build), warm queries 0.1-0.4 s. - spawn_live_sync marks the gate Indexing before the scan thread runs: the gate said Ready over an empty graph and the first question got "no_seed_resolved" - wait_for_index answers from a complete index only (finished scan or loaded snapshot; reset on clear), waits up to 120 s (NEUROMESH_INDEX_WAIT_SECS), never from a graph still filling - reindex_incremental marks Ready before the optional embeddings refresh - rank index built after ingest and snapshot load, not on the first question (CI 250 ms gate) - a ranked file over 8k tokens is seeded by up to three definitions whose name/doc match the question, not whole - NM_TIMING=1: index stages, rank build, seeds/select, MCP stages - benchmark-holdout.sh: `concept` and `concept-holdout` sets - ripgrep holdout run once: recall 0.042 -> 0.500 (F92-F94 open) Co-Authored-By: Claude Opus 5.5 --- crates/neuromesh-cli/src/commands/mod.rs | 3 + crates/neuromesh-context/src/activator.rs | 11 +++- .../neuromesh-context/src/activator_seed.rs | 63 +++++++++++++++++++ crates/neuromesh-graph/src/file_rank.rs | 6 ++ crates/neuromesh-graph/src/graph.rs | 32 +++++++++- crates/neuromesh-graph/src/lib.rs | 13 ++++ crates/neuromesh-mcp/src/tools.rs | 28 ++++++++- docs/planning/stage5-findings.fa.md | 8 +++ scripts/benchmark-holdout.sh | 12 +++- 9 files changed, 171 insertions(+), 5 deletions(-) diff --git a/crates/neuromesh-cli/src/commands/mod.rs b/crates/neuromesh-cli/src/commands/mod.rs index 6d676c5..59e592a 100644 --- a/crates/neuromesh-cli/src/commands/mod.rs +++ b/crates/neuromesh-cli/src/commands/mod.rs @@ -152,6 +152,9 @@ pub fn spawn_live_sync( graph.mark_index_ready(); return; } + // Marked before the scan thread is scheduled: until then the gate still + // said Ready over an empty graph, and a first question got "no seed". + graph.mark_index_indexing(); let bg_graph = graph.clone(); let bg_dir = dir.clone(); let bg_pid = pid.clone(); diff --git a/crates/neuromesh-context/src/activator.rs b/crates/neuromesh-context/src/activator.rs index 51f1e92..5114c36 100644 --- a/crates/neuromesh-context/src/activator.rs +++ b/crates/neuromesh-context/src/activator.rs @@ -173,7 +173,9 @@ impl ContextActivator { ) -> ContextView { #[cfg(feature = "embeddings")] neuromesh_embed::packet_cache_begin(); + let t_all = std::time::Instant::now(); let view = self.activate_with_hops(graph, signature, mode, 0); + neuromesh_graph::timing("query: activate total", t_all); #[cfg(feature = "embeddings")] neuromesh_embed::packet_cache_end(); view @@ -187,7 +189,10 @@ impl ContextActivator { mode: OptimizationMode, hops_override: u8, ) -> ContextView { - self.activate_inner(graph, signature, mode, hops_override) + let t_inner = std::time::Instant::now(); + let view = self.activate_inner(graph, signature, mode, hops_override); + neuromesh_graph::timing("query: activate_inner total", t_inner); + view } /// L1→L2→L3 cost-aware retrieval with conservative sufficiency early exit. @@ -310,6 +315,7 @@ impl ContextActivator { OptimizationMode::MaxSavings => 1, } }; + let t_q = std::time::Instant::now(); let mut seed_resolutions = Vec::new(); let mut seed_energies: HashMap = HashMap::new(); let mut seed_reasons: HashMap = HashMap::new(); @@ -342,6 +348,8 @@ impl ContextActivator { is_style_task(signature), ); + neuromesh_graph::timing("query: seeds", t_q); + let t_q = std::time::Instant::now(); let scaffold_used = seed_result.scaffold_used; let embedding_used = seed_result.embedding_used; @@ -539,6 +547,7 @@ impl ContextActivator { } else { restrict_selection_to_call_graph(graph, &seed_set, &mut selection); } + neuromesh_graph::timing("query: neighborhood+select", t_q); let app_cfg = neuromesh_core::Config::load(); if app_cfg.retrieval.engine == neuromesh_core::RetrievalEngine::Hybrid && effective_mode == OptimizationMode::Balanced diff --git a/crates/neuromesh-context/src/activator_seed.rs b/crates/neuromesh-context/src/activator_seed.rs index 19e14aa..6ec48ca 100644 --- a/crates/neuromesh-context/src/activator_seed.rs +++ b/crates/neuromesh-context/src/activator_seed.rs @@ -1052,14 +1052,77 @@ fn push_ranked_file_seeds( buffers.reasons.remove(id); } } + let q_terms: std::collections::HashSet = + neuromesh_graph::file_rank::query_terms(prompt) + .into_iter() + .collect(); for (pos, r) in picks.iter().enumerate() { let energy = signal_weight(config, SignalKind::PathHint, pos + 1); + // A large file is addressed by the part of it the question is about: + // seeding all of a 40k-token file ships it whole (a 200 KB packet + // and a 20 s answer). Its best-matching definitions stand in for it. + if let Some(symbols) = big_file_symbols(graph, &r.id, &r.path, &q_terms) { + for (id, name) in symbols { + sink.insert(id, energy, format!("body:{name}"), None, None); + } + continue; + } let rel = r.path.to_string_lossy().replace('\\', "/"); sink.push(graph, prompt, rel, energy, "body"); } true } +/// Above this a ranked file is seeded by its definitions, not whole. +const BIG_FILE_TOKENS: usize = 8_000; + +/// For a file over [`BIG_FILE_TOKENS`]: up to three definitions whose name +/// and doc summary share the most stemmed words with the question (at least +/// one). `None` for a small file, or when no definition matches — then the +/// file is seeded whole as before. +fn big_file_symbols( + graph: &NeuralProjectGraph, + file_id: &NodeId, + path: &std::path::Path, + q_terms: &std::collections::HashSet, +) -> Option> { + let file = graph.get_node(file_id)?; + if file.token_cost <= BIG_FILE_TOKENS { + return None; + } + let mut scored: Vec<(usize, usize, NodeId, String)> = graph + .nodes_in_file(path) + .into_iter() + .filter(|n| n.node_type != neuromesh_core::NodeType::File) + .filter_map(|n| { + let mut text = n.name.clone(); + if let Some(doc) = &n.doc_summary { + text.push(' '); + text.push_str(doc); + } + let words: std::collections::HashSet = + neuromesh_graph::file_rank::text_terms(&text) + .into_iter() + .collect(); + let hits = words.intersection(q_terms).count(); + (hits > 0).then_some((hits, n.token_cost, n.id, n.name)) + }) + .collect(); + if scored.is_empty() { + return None; + } + scored.sort_by(|a, b| b.0.cmp(&a.0).then(a.1.cmp(&b.1)).then(a.3.cmp(&b.3))); + let best = scored[0].0; + Some( + scored + .into_iter() + .filter(|s| s.0 == best || s.0 + 1 >= best.max(2)) + .take(3) + .map(|(_, _, id, name)| (id, name)) + .collect(), + ) +} + pub(crate) fn push_body_word_seeds( graph: &NeuralProjectGraph, prompt: &str, diff --git a/crates/neuromesh-graph/src/file_rank.rs b/crates/neuromesh-graph/src/file_rank.rs index 0dcd092..c063b57 100644 --- a/crates/neuromesh-graph/src/file_rank.rs +++ b/crates/neuromesh-graph/src/file_rank.rs @@ -240,6 +240,12 @@ fn terms_of(text: &str, st: &Stemmer, keep_stopwords: bool) -> Vec { out } +/// Stemmed content words of any text (a symbol name, a doc summary), +/// stopwords dropped — comparable with [`query_terms`]. +pub fn text_terms(text: &str) -> Vec { + terms_of(text, &stemmer(), false) +} + /// Query terms, stemmed and deduplicated in prompt order. pub fn query_terms(prompt: &str) -> Vec { weighted_query_terms(prompt) diff --git a/crates/neuromesh-graph/src/graph.rs b/crates/neuromesh-graph/src/graph.rs index 5614f7a..0647076 100644 --- a/crates/neuromesh-graph/src/graph.rs +++ b/crates/neuromesh-graph/src/graph.rs @@ -149,6 +149,10 @@ pub struct NeuralProjectGraph { synaptic_engine: Arc>, physarum_solver: Arc, index_gate: Arc, + /// True once a full index exists (a finished scan, or a loaded snapshot): + /// a query may then run while a refresh is in flight. A graph that is + /// still filling for the first time is never answered from. + complete: Arc, } impl NeuralProjectGraph { @@ -167,6 +171,7 @@ impl NeuralProjectGraph { ))), physarum_solver: Arc::new(PhysarumSolver::new(PhysarumConfig::default())), index_gate: Arc::new(IndexGate::new(IndexState::Ready)), + complete: Arc::new(std::sync::atomic::AtomicBool::new(false)), } } @@ -190,7 +195,13 @@ impl NeuralProjectGraph { self.index_gate.set(IndexState::Indexing); } + pub fn has_complete_index(&self) -> bool { + self.complete.load(std::sync::atomic::Ordering::Acquire) + } + pub fn mark_index_ready(&self) { + self.complete + .store(true, std::sync::atomic::Ordering::Release); self.index_gate.set(IndexState::Ready); } @@ -208,6 +219,8 @@ impl NeuralProjectGraph { } pub fn clear(&self, new_project_id: Option) { + self.complete + .store(false, std::sync::atomic::Ordering::Release); if let Some(new_id) = new_project_id { *self.project_id.write() = new_id; } @@ -750,7 +763,9 @@ impl NeuralProjectGraph { return Arc::clone(&cached.file_rank); } } + let t0 = std::time::Instant::now(); let index = Arc::new(crate::file_rank::FileRankIndex::build(&data)); + crate::timing("file_rank build", t0); let mut cached = self.derived.write(); cached.file_rank_key = Some(key); cached.file_rank = Arc::clone(&index); @@ -2512,6 +2527,10 @@ impl NeuralProjectGraph { self.enforce_single_project(); self.inner.write().parser_epoch = GRAPH_PARSER_EPOCH; let _ = self.save_persisted(workspace); + // The graph answers questions from here on; embeddings are an + // optional extra a query does not wait for (a cold MiniLM + // refresh held every first question for seconds). + self.mark_index_ready(); #[cfg(feature = "embeddings")] { let emb = neuromesh_core::Config::load().embeddings; @@ -2522,7 +2541,6 @@ impl NeuralProjectGraph { ); } } - self.mark_index_ready(); } Err(_) => self.mark_index_failed(), } @@ -2568,6 +2586,7 @@ impl NeuralProjectGraph { } } } + let t0 = std::time::Instant::now(); let parsed: Vec<_> = scanned .par_iter() .filter_map(|(file, content)| { @@ -2583,11 +2602,18 @@ impl NeuralProjectGraph { Some((file, ast, content.as_str())) }) .collect(); + crate::timing("parse", t0); + let t0 = std::time::Instant::now(); for (file, ast, content) in parsed { self.ingest_file_keep(file, &ast, Some(content), keep_source); } + crate::timing("ingest", t0); + let t0 = std::time::Instant::now(); self.apply_manifest_hints(scanned); self.finalize_links(); + crate::timing("manifest+links", t0); + // Built now, not on the first question that needs it. + let _ = self.file_rank_index(); if !keep_source { self.inner.write().source_overlay.clear(); } @@ -2737,6 +2763,10 @@ impl NeuralProjectGraph { } data.source_overlay.clear(); rebuild_indexes(&mut data); + drop(data); + self.complete + .store(true, std::sync::atomic::Ordering::Release); + let _ = self.file_rank_index(); } pub fn apply_stdp_on_path(&self, node_ids: &[NodeId]) { diff --git a/crates/neuromesh-graph/src/lib.rs b/crates/neuromesh-graph/src/lib.rs index d3ecb2b..00dda62 100644 --- a/crates/neuromesh-graph/src/lib.rs +++ b/crates/neuromesh-graph/src/lib.rs @@ -41,3 +41,16 @@ pub use query::{ TraceDirection, TraceHop, TraceResult, }; pub use synapse::{NeuralSpike, StdpConfig, SynapticPlasticityEngine}; + +/// `NM_TIMING=1`: stage timings on stderr (index stages, rank index build). +pub fn timing_enabled() -> bool { + static ON: std::sync::OnceLock = std::sync::OnceLock::new(); + *ON.get_or_init(|| std::env::var("NM_TIMING").is_ok_and(|v| v == "1")) +} + +/// Print `[timing] label elapsed` when [`timing_enabled`]. +pub fn timing(label: &str, since: std::time::Instant) { + if timing_enabled() { + eprintln!("[timing] {label} {:?}", since.elapsed()); + } +} diff --git a/crates/neuromesh-mcp/src/tools.rs b/crates/neuromesh-mcp/src/tools.rs index 894e382..9f65070 100644 --- a/crates/neuromesh-mcp/src/tools.rs +++ b/crates/neuromesh-mcp/src/tools.rs @@ -418,6 +418,7 @@ impl McpToolHandler { } } + let t_tier = std::time::Instant::now(); let view = self.activator .activate_tiered(&self.graph, &signature, gate.effective_mode); @@ -437,7 +438,11 @@ impl McpToolHandler { .record_neural_spike(active.node.id.clone(), false, true); } + neuromesh_graph::timing("mcp: activate_tiered", t_tier); let elapsed_ms = start_time.elapsed().as_millis() as u64; + if neuromesh_graph::timing_enabled() { + eprintln!("[timing] mcp: since tool start {elapsed_ms}ms"); + } let workspace_tokens = self.graph.total_tokens().max(1); let opt_tokens = view.active_tokens; let seeds_missed = view @@ -478,6 +483,7 @@ impl McpToolHandler { server_inferred_keywords: server_inferred, }; + let t_s = std::time::Instant::now(); self.emit_telemetry(ToolTelemetry { tokens_before: workspace_tokens, tokens_after: opt_tokens, @@ -491,6 +497,8 @@ impl McpToolHandler { ) }); + neuromesh_graph::timing("mcp: telemetry", t_s); + let t_s = std::time::Instant::now(); Ok({ let value = cache_and_build( &self.packet_cache, @@ -498,6 +506,8 @@ impl McpToolHandler { &build, detail, ); + neuromesh_graph::timing("mcp: cache_and_build", t_s); + let t_s = std::time::Instant::now(); #[cfg(feature = "embeddings")] { let emb_cfg = Config::load().embeddings; @@ -524,6 +534,7 @@ impl McpToolHandler { } } } + neuromesh_graph::timing("mcp: semantic cache", t_s); value }) } @@ -778,6 +789,9 @@ impl McpToolHandler { let limit = arguments["limit"].as_u64().unwrap_or(20) as usize; let nodes = self.graph.search_symbols(query, limit); let elapsed_ms = start_time.elapsed().as_millis() as u64; + if neuromesh_graph::timing_enabled() { + eprintln!("[timing] mcp: since tool start {elapsed_ms}ms"); + } self.emit_telemetry(ToolTelemetry { nodes_after: nodes.len(), @@ -1094,8 +1108,18 @@ impl McpToolHandler { if self.graph.index_state() == IndexState::Ready { return Ok(()); } - let state = self.graph.wait_until_indexed(Duration::from_secs(5)); - if self.graph.stats().total_nodes > 0 || state == IndexState::Ready { + // A complete index (a loaded snapshot) answers while a refresh runs. + // A graph still filling for the first time does not: a half-linked + // graph answers "no seed" for a file it simply has not reached yet. + if self.graph.has_complete_index() { + return Ok(()); + } + let wait = std::env::var("NEUROMESH_INDEX_WAIT_SECS") + .ok() + .and_then(|v| v.parse().ok()) + .unwrap_or(120); + let state = self.graph.wait_until_indexed(Duration::from_secs(wait)); + if state == IndexState::Ready { return Ok(()); } Err(NeuroMeshError::Config(format!( diff --git a/docs/planning/stage5-findings.fa.md b/docs/planning/stage5-findings.fa.md index 0028ee7..1a22de8 100644 --- a/docs/planning/stage5-findings.fa.md +++ b/docs/planning/stage5-findings.fa.md @@ -1342,3 +1342,11 @@ web باقی‌مانده: `fd_login_flow` 1/3 (auth route + password-manager ب | F91 | شکاف واژگانی («fade» ↔ `evaporate`، «how many» ↔ `count`) | thesaurus عمومی ۴۹ خوشه در `thesaurus.txt` (وزن 0.35)؛ عمداً هیچ موضوع ست holdout (ripgrep) در آن نیست | → **0.679** | تله: فایل‌های خود ما روی ست self اثر می‌گذارند — thesaurus داخل `.rs` بالای ۴ سؤال آمد؛ به `.txt` (که ایندکس نمی‌شود) منتقل شد. سؤال‌های gold هرگز در کامنت کد نقل نشوند. + +**holdout ripgrep (یک بار، بعد از قفل کد):** recall **0.042 → 0.500**، precision 0.152 (v1.0.0 روی همان ست 0.042). رتبه‌بند gold را در ۷ از ۱۲ سؤال در top-3 دارد؛ بقیه شکاف واژگانی‌اند که از سؤال‌های holdout می‌آیند و عمداً روی آن‌ها تیون نشد. + +| F | مشاهده | وضعیت | +|---|---|---| +| F92 | packetهای سؤال مفهومی پهن‌اند (۳ تا ۸ فایل، تا ۴۳k token) — بعد از seed، گسترش گراف فایل‌های زیادی اضافه می‌کند؛ precision 0.15 | باز — کار بعدی | +| F93 | واژه‌ی مرکب: «link» ↔ `hyperlink`، «machine-readable» ↔ `json` — stem کافی نیست؛ embedding یا تجزیه‌ی ترکیب لازم است | باز | +| F94 | فایل‌های hub (`lib.rs`) با doc crate همه‌ی کلمات را دارند و بالا می‌آیند — اما گاهی خودشان جواب‌اند (`cli/src/lib.rs` برای tty) | باز، جریمه‌ی ساده رد شد | diff --git a/scripts/benchmark-holdout.sh b/scripts/benchmark-holdout.sh index b27caf0..7d116b2 100644 --- a/scripts/benchmark-holdout.sh +++ b/scripts/benchmark-holdout.sh @@ -4,6 +4,7 @@ # # bash scripts/benchmark-holdout.sh # all sets # bash scripts/benchmark-holdout.sh holdout # one set: dev|large|holdout|holdout-c|holdout-lang|holdout-ml|holdout-ml2|holdout-cfg|private +# bash scripts/benchmark-holdout.sh concept concept-holdout # plain-language sets (not in the default run) # # "private" is the phase-5b holdout on a repository that is not in this tree: # set NM_PRIVATE_SET_DIR (manifest + gold) and NM_PRIVATE_DIR (checkouts); it is @@ -27,6 +28,7 @@ declare -A MANIFEST=( [holdout-ml2]="tests/third_party/holdout-ml2/repos.toml" [holdout-cfg]="tests/third_party/holdout-cfg/repos.toml" [holdout-web]="tests/third_party/holdout-web/repos.toml" + [concept-holdout]="tests/third_party/concept-holdout/repos.toml" ) declare -A TEST=( [dev]="third_party_gold" @@ -39,10 +41,13 @@ declare -A TEST=( [holdout-cfg]="third_party_cfg_holdout_gold" [holdout-web]="third_party_web_holdout_gold" [private]="third_party_private_gold" + [concept]="third_party_private_gold" + [concept-holdout]="third_party_private_gold" ) sets=("$@") if [ ${#sets[@]} -eq 0 ]; then sets=(dev large holdout holdout-c holdout-lang holdout-ml holdout-ml2 holdout-cfg holdout-web); fi for set in "${sets[@]}"; do + [ "$set" = concept ] && continue if [ "$set" = private ]; then [ -n "${NM_PRIVATE_SET_DIR:-}" ] || { echo "private: NM_PRIVATE_SET_DIR unset; skipping" >&2; } continue @@ -56,7 +61,12 @@ done echo "set recall precision forbidden oracle(reachable/strict)" for set in "${sets[@]}"; do - line=$(cargo test -q -p neuromesh-context --test "${TEST[$set]}" -- --nocapture 2>&1 \ + envs=() + case "$set" in + concept) envs=(NM_PRIVATE_SET_DIR="$root/tests/third_party/concept" NM_PRIVATE_DIR="$root/..") ;; + concept-holdout) envs=(NM_PRIVATE_SET_DIR="$root/tests/third_party/concept-holdout" NM_PRIVATE_DIR="$root/target/third_party/concept-holdout") ;; + esac + line=$(env "${envs[@]}" cargo test -q -p neuromesh-context --test "${TEST[$set]}" -- --nocapture 2>&1 \ | grep -E "^third_party" | tail -1 || true) # third_party_x: N gold cases mean recall R precision P forbidden hits F; N task cases reachable A strict S recall=$(echo "$line" | sed -n 's/.*mean recall \([0-9.]*\).*/\1/p') From 3c2ec4d5a8d3b9f812a9ac07aae334dcb063733a Mon Sep 17 00:00:00 2001 From: ParsaVictor <1.parsa.karkooti@gmail.com> Date: Thu, 1 Oct 2026 11:54:11 +0330 Subject: [PATCH 6/7] revert big-file definition seeding (ripgrep holdout 0.500 -> 0.375, packet not smaller); changelog 1.1.0, reply to upstream, benchmark-fast.sh, roadmap H-M Co-Authored-By: Claude Opus 5.5 --- .../neuromesh-context/src/activator_seed.rs | 63 ----------- docs/CHANGELOG.md | 48 +++++++++ ...1-roadmap-2026-09-30-concept-queries.fa.md | 13 +++ docs/planning/reply-yoosef-2026-10-01.md | 47 ++++++++ scripts/benchmark-fast.sh | 101 ++++++++++++++++++ 5 files changed, 209 insertions(+), 63 deletions(-) create mode 100644 docs/planning/reply-yoosef-2026-10-01.md create mode 100644 scripts/benchmark-fast.sh diff --git a/crates/neuromesh-context/src/activator_seed.rs b/crates/neuromesh-context/src/activator_seed.rs index 6ec48ca..19e14aa 100644 --- a/crates/neuromesh-context/src/activator_seed.rs +++ b/crates/neuromesh-context/src/activator_seed.rs @@ -1052,77 +1052,14 @@ fn push_ranked_file_seeds( buffers.reasons.remove(id); } } - let q_terms: std::collections::HashSet = - neuromesh_graph::file_rank::query_terms(prompt) - .into_iter() - .collect(); for (pos, r) in picks.iter().enumerate() { let energy = signal_weight(config, SignalKind::PathHint, pos + 1); - // A large file is addressed by the part of it the question is about: - // seeding all of a 40k-token file ships it whole (a 200 KB packet - // and a 20 s answer). Its best-matching definitions stand in for it. - if let Some(symbols) = big_file_symbols(graph, &r.id, &r.path, &q_terms) { - for (id, name) in symbols { - sink.insert(id, energy, format!("body:{name}"), None, None); - } - continue; - } let rel = r.path.to_string_lossy().replace('\\', "/"); sink.push(graph, prompt, rel, energy, "body"); } true } -/// Above this a ranked file is seeded by its definitions, not whole. -const BIG_FILE_TOKENS: usize = 8_000; - -/// For a file over [`BIG_FILE_TOKENS`]: up to three definitions whose name -/// and doc summary share the most stemmed words with the question (at least -/// one). `None` for a small file, or when no definition matches — then the -/// file is seeded whole as before. -fn big_file_symbols( - graph: &NeuralProjectGraph, - file_id: &NodeId, - path: &std::path::Path, - q_terms: &std::collections::HashSet, -) -> Option> { - let file = graph.get_node(file_id)?; - if file.token_cost <= BIG_FILE_TOKENS { - return None; - } - let mut scored: Vec<(usize, usize, NodeId, String)> = graph - .nodes_in_file(path) - .into_iter() - .filter(|n| n.node_type != neuromesh_core::NodeType::File) - .filter_map(|n| { - let mut text = n.name.clone(); - if let Some(doc) = &n.doc_summary { - text.push(' '); - text.push_str(doc); - } - let words: std::collections::HashSet = - neuromesh_graph::file_rank::text_terms(&text) - .into_iter() - .collect(); - let hits = words.intersection(q_terms).count(); - (hits > 0).then_some((hits, n.token_cost, n.id, n.name)) - }) - .collect(); - if scored.is_empty() { - return None; - } - scored.sort_by(|a, b| b.0.cmp(&a.0).then(a.1.cmp(&b.1)).then(a.3.cmp(&b.3))); - let best = scored[0].0; - Some( - scored - .into_iter() - .filter(|s| s.0 == best || s.0 + 1 >= best.max(2)) - .take(3) - .map(|(_, _, id, name)| (id, name)) - .collect(), - ) -} - pub(crate) fn push_body_word_seeds( graph: &NeuralProjectGraph, prompt: &str, diff --git a/docs/CHANGELOG.md b/docs/CHANGELOG.md index ad1bc56..3ce815c 100644 --- a/docs/CHANGELOG.md +++ b/docs/CHANGELOG.md @@ -4,6 +4,54 @@ All notable user-facing changes live here. The README stays a product guide, not ## Unreleased +## 1.1.0 — 2026-10-01 + +Plain-language questions, the protocol bug and the cold-start latency the upstream author +reported on 1.0.0 (3 of 7 questions on this repository answered with the wrong file). + +### Headline numbers + +- New set, plain-language questions on this repository (no identifier in the prompt, 14 q): + recall **0.286 → 0.679**. +- New holdout, the same kind of question on ripgrep 14.1.1 (12 q, gold locked before any run, + never tuned on): recall **0.042 → 0.500**. Half the questions still miss; see "Known gaps". +- The nine existing sets: no recall lost; holdout-web precision 0.608 → 0.643, the rest unchanged. +- Cold start on a fresh clone of this repository: first answer 3.6 s → 2.6 s (the index + build included); later answers 0.1–0.4 s. + +### Retrieval + +- **Whole-question file ranking** — a question that names no symbol is read against every file's + path, defined names, comments and body (field-weighted BM25, Snowball-stemmed). The best file is + seeded, plus up to two within 70% of it. + Comments are indexed separately: they are the prose closest to how people ask. +- **A general software thesaurus** (`crates/neuromesh-graph/src/thesaurus.txt`, 49 clusters: + limit/cap/max, fade/decay/expire, retry/backoff, …) searches stand-in words at a third of the weight. +- **English words are not anchors** — "How does this *tool* …" no longer makes a function called + `tool` the strongest seed. A lowercase word counts as code only when the prompt marks it + (backticks, `word()`, `word!`, `a.word`, `::word`) or it is part of another identifier. + +### Fixes + +- `neuromesh mcp` wrote the dashboard banner to stdout, the JSON-RPC channel; strict MCP clients broke. + It goes to stderr now, and a test spawns the real binary and parses every stdout line. +- A question that arrived while the first index was being built got `no_seed_resolved` (the gate said + "ready" over an empty graph). It now waits for a complete index (up to 120 s, + `NEUROMESH_INDEX_WAIT_SECS`), and the index is ready before the optional embedding refresh. +- `NEUROMESH_NO_BROWSER=1` turns off the first-run dashboard tab. + +### Diagnostics + +- `NM_TIMING=1` prints index and query stage timings on stderr. +- `bash scripts/benchmark-holdout.sh concept concept-holdout` runs the two plain-language sets. + +### Known gaps + +- Plain-language packets are wide (ripgrep holdout precision 0.15): after seeding, graph + expansion adds several files. +- Words the code never spells ("clickable link" for `hyperlink`, "machine-readable" for `json`) + still miss; stemming and the thesaurus do not reach them. + ## 1.0.0 — 2026-09-22 The first release of code-context-engine as its own project. Everything below is diff --git a/docs/planning/11-roadmap-2026-09-30-concept-queries.fa.md b/docs/planning/11-roadmap-2026-09-30-concept-queries.fa.md index 87f72ef..9bb6681 100644 --- a/docs/planning/11-roadmap-2026-09-30-concept-queries.fa.md +++ b/docs/planning/11-roadmap-2026-09-30-concept-queries.fa.md @@ -27,3 +27,16 @@ - gold قبل از packet قفل؛ هیچ قانونی که فقط یک سوال را درست کند (hardcode اسم) پذیرفته نیست. - ratchetها فقط بالا می‌روند. - `retry_negative` باگ نیست: این fork واقعاً retry دارد (`crates/neuromesh-provider/src/anthropic.rs`)؛ در ست ما gold آن همان فایل‌های provider است. + +## به‌روزرسانی ۲۰۲۶-۱۰-۰۱ — نتیجه‌ی A–E و فازهای بعدی + +انجام‌شده: A (#126)، B، C، D، E در PR #127 — concept 0.286→0.679، holdout ripgrep 0.042→0.500، cold start 3.6→2.6 s، ۹ ست بدون افت. + +| فاز | کار | الگو | معیار خروج | +|---|---|---|---| +| H | gate نهایی + انتشار 1.1.0 | — | tag + سه باینری | +| I | harness در release + اجرای موازی ست‌ها | — | دور کامل ≤ ۱۰ دقیقه | +| J | ادغام RRF رتبه‌ی BM25F با embedding (MiniLM موجود) فقط برای سؤال بدون anchor | Cody / Continue / claude-context (hybrid) | concept-holdout ≥ 0.7 روی holdout تازه | +| K | packet در سطح تعریف برای سؤال مفهومی + رتبه‌ی PageRank داخل packet | Aider repo-map، chunking در claude-context | precision مفهومی ≥ 0.4 بدون افت recall | +| L | holdout تازه‌ی دوم (Python)، gold قفل قبل از اجرا | — | عدد صادقانه | +| M | مقایسه‌ی مستقیم با رقبا (claude-context، Serena، Aider repo-map) روی همین ست‌ها | — | جدول منتشرشده در `docs/measured.md` | diff --git a/docs/planning/reply-yoosef-2026-10-01.md b/docs/planning/reply-yoosef-2026-10-01.md new file mode 100644 index 0000000..4df98ac --- /dev/null +++ b/docs/planning/reply-yoosef-2026-10-01.md @@ -0,0 +1,47 @@ +# Re: your v1.0.0 test on code-context-engine + +Thank you for testing 1.0.0 against this repository. Your battery found a class of question +none of our sets measured, and a real protocol bug. Here is what we checked and what changed. + +## Your findings, one by one + +| Finding | Verdict | What changed | +|---|---|---| +| Dashboard banner on stdout in `mcp` mode | **Real.** It printed on every start | Banner → stderr. A test spawns the real binary and fails on any non-JSON stdout line (it failed on 1.0.0 with the banner line) | +| `root_fs_safety` → `descriptors.rs` | **Real** | "How does this *tool*…" made the English word `tool` a strong identifier anchor (there is a function called `tool`). Prose words are guesses now → `confine.rs` | +| `reinforcement`, `max_files_cap` → wrong file | **Real** | Same class: plain-language questions. New whole-question file ranking (below) | +| `retry_negative` | **Not a defect in this fork** | This fork implements retry with exponential backoff (`crates/neuromesh-provider/src/anthropic.rs`, `openai.rs`). The expected "no match" came from upstream, which has none | +| ~26 s latency | **Not reproduced as a per-query cost; two cold-start bugs found** | Server-side retrieval is 0.1–0.55 s per query here. The first query in a fresh clone includes the index build (3.6 s → 2.6 s). Fixed: a query during the first index build got `no_seed_resolved` from an empty graph, and "ready" waited for the optional embedding refresh. Note that a shell client reading the reply with `read` is itself slow on 100 KB+ lines; that cost 5–20 s in our first measurements | + +## Why our numbers missed it + +All 30 questions in our self-repo set named an identifier (`resolve_cluster_noun_seeds`, `select()`). +Yours are plain language. So we wrote two new sets, with gold committed before any engine run: + +- **concept**: 14 plain-language questions on this repository (your four included; `retry` with + the provider files as gold). +- **concept-holdout**: 12 plain-language questions on ripgrep 14.1.1, never tuned on and run once + after the code was frozen. + +## Numbers (file-level recall) + +| Set | 1.0.0 | 1.1.0 | +|---|---|---| +| concept (this repo, 14 q) | 0.286 | **0.679** | +| concept-holdout (ripgrep, 12 q, never tuned on) | 0.042 | **0.500** | +| the nine existing sets | — | no recall lost; precision unchanged or better | + +Honest gaps: half the ripgrep questions still miss. Most use words the code never spells ("clickable +link" vs `hyperlink`, "machine-readable" vs `json`). Plain-language packets are also wide +(precision 0.15). Both are the next work. + +## What changed in the engine + +- Field-weighted BM25 over every file's path, defined names, comments and body, Snowball-stemmed; + comments are a new index field. The best file is seeded, plus runners-up within 70%. +- A general software thesaurus (49 clusters) at a third of the weight. None of its clusters came from + the ripgrep questions. +- A lowercase word counts as an identifier only when the prompt marks it as code. + +Reproduce: `bash scripts/benchmark-holdout.sh concept concept-holdout`; `NM_TIMING=1` prints +stage timings. diff --git a/scripts/benchmark-fast.sh b/scripts/benchmark-fast.sh new file mode 100644 index 0000000..b8c7bc7 --- /dev/null +++ b/scripts/benchmark-fast.sh @@ -0,0 +1,101 @@ +#!/usr/bin/env bash +# Same numbers as benchmark-holdout.sh, faster: the gold harnesses are built +# once with optimisations and the sets run side by side (NM_JOBS at a time, +# default 3; each set indexes its own checkouts, so memory is the limit). +# +# bash scripts/benchmark-fast.sh # the nine default sets +# bash scripts/benchmark-fast.sh concept concept-holdout holdout-lang +# +# Output: one summary line per set, in the order given; per-set logs in +# target/bench-fast/.log. +set -euo pipefail +root="$(cd "$(dirname "$0")/.." && pwd)" +cd "$root" +export CARGO_BUILD_JOBS="${CARGO_BUILD_JOBS:-4}" +export NM_THIRD_PARTY=1 +jobs="${NM_JOBS:-3}" + +declare -A MANIFEST=( + [dev]="tests/third_party/repos.toml" + [large]="tests/third_party/large/repos.toml" + [holdout]="tests/third_party/holdout/repos.toml" + [holdout-c]="tests/third_party/holdout-c/repos.toml" + [holdout-lang]="tests/third_party/holdout-lang/repos.toml" + [holdout-ml]="tests/third_party/holdout-ml/repos.toml" + [holdout-ml2]="tests/third_party/holdout-ml2/repos.toml" + [holdout-cfg]="tests/third_party/holdout-cfg/repos.toml" + [holdout-web]="tests/third_party/holdout-web/repos.toml" + [concept-holdout]="tests/third_party/concept-holdout/repos.toml" +) +declare -A TEST=( + [dev]="third_party_gold" + [large]="third_party_large_gold" + [holdout]="third_party_holdout_gold" + [holdout-c]="third_party_c_holdout_gold" + [holdout-lang]="third_party_lang_holdout_gold" + [holdout-ml]="third_party_ml_holdout_gold" + [holdout-ml2]="third_party_ml2_holdout_gold" + [holdout-cfg]="third_party_cfg_holdout_gold" + [holdout-web]="third_party_web_holdout_gold" + [concept]="third_party_private_gold" + [concept-holdout]="third_party_private_gold" +) +sets=("$@") +if [ ${#sets[@]} -eq 0 ]; then + sets=(dev large holdout holdout-c holdout-lang holdout-ml holdout-ml2 holdout-cfg holdout-web) +fi + +for set in "${sets[@]}"; do + case "$set" in + concept) ;; + dev) bash scripts/fetch-third-party.sh >/dev/null ;; + *) bash scripts/fetch-third-party.sh "${MANIFEST[$set]}" "$set" >/dev/null ;; + esac +done + +# Build every needed harness once, optimised; then run the binaries directly. +tests=() +for set in "${sets[@]}"; do tests+=(--test "${TEST[$set]}"); done +cargo test --release -q -p neuromesh-context "${tests[@]}" --no-run 2>/dev/null +bin_of() { + # newest optimised binary for a harness name + ls -t target/release/deps/"$1"-*.exe target/release/deps/"$1"-* 2>/dev/null \ + | grep -v '\.d$\|\.pdb$' | head -1 +} + +out="target/bench-fast" +mkdir -p "$out" +run_set() { + local set="$1" bin + bin="$(bin_of "${TEST[$set]}")" + local envs=() + case "$set" in + concept) envs=(NM_PRIVATE_SET_DIR="$root/tests/third_party/concept" NM_PRIVATE_DIR="$root/..") ;; + concept-holdout) envs=(NM_PRIVATE_SET_DIR="$root/tests/third_party/concept-holdout" NM_PRIVATE_DIR="$root/target/third_party/concept-holdout") ;; + esac + # The harness resolves the workspace from its manifest dir at build time. + (cd crates/neuromesh-context && env "${envs[@]}" "$bin" --nocapture >"$root/$out/$set.log" 2>&1 || true) +} + +running=0 +for set in "${sets[@]}"; do + run_set "$set" & + running=$((running + 1)) + if [ "$running" -ge "$jobs" ]; then + wait -n + running=$((running - 1)) + fi +done +wait + +echo "set recall precision forbidden oracle(reachable/strict)" +for set in "${sets[@]}"; do + line=$(grep -E "^third_party" "$out/$set.log" | tail -1 || true) + recall=$(echo "$line" | sed -n 's/.*mean recall \([0-9.]*\).*/\1/p') + prec=$(echo "$line" | sed -n 's/.*precision \([0-9.]*\).*/\1/p') + forb=$(echo "$line" | sed -n 's/.*forbidden hits \([0-9]*\).*/\1/p') + reach=$(echo "$line" | sed -n 's/.*reachable \([0-9]*\).*/\1/p') + strict=$(echo "$line" | sed -n 's/.*strict \([0-9]*\).*/\1/p') + cases=$(echo "$line" | sed -n 's/.*: \([0-9]*\) gold cases.*/\1/p') + printf "%-19s %-7s %-10s %-10s %s/%s strict %s\n" "$set" "${recall:-?}" "${prec:-?}" "${forb:-?}" "${reach:-?}" "${cases:-?}" "${strict:-?}" +done From 4b3debe982bc7047811b134e5df16937f15bfe22 Mon Sep 17 00:00:00 2001 From: ParsaVictor <1.parsa.karkooti@gmail.com> Date: Thu, 1 Oct 2026 12:04:33 +0330 Subject: [PATCH 7/7] release: 1.1.0 (plain-language questions, stdout fix, cold start) + benchmark-fast.sh fix Nine sets in 78 s warm with benchmark-fast.sh (was ~40 min), same numbers. Co-Authored-By: Claude Opus 5.5 --- CITATION.cff | 4 ++-- Cargo.lock | 34 +++++++++++++-------------- Cargo.toml | 2 +- crates/neuromesh-mcp/src/protocol.rs | 2 +- docs/measured.md | 8 ++++--- editors/vscode-neuromesh/package.json | 2 +- install.ps1 | 2 +- install.sh | 2 +- llms.txt | 2 +- scripts/benchmark-fast.sh | 5 ++-- 10 files changed, 33 insertions(+), 30 deletions(-) diff --git a/CITATION.cff b/CITATION.cff index 6963ba7..4643fd1 100644 --- a/CITATION.cff +++ b/CITATION.cff @@ -9,8 +9,8 @@ authors: email: 1.parsa.karkooti@gmail.com repository-code: https://github.com/ParsaVictor/code-context-engine license: MIT -version: 1.0.0 -date-released: "2026-09-22" +version: 1.1.0 +date-released: "2026-10-01" keywords: - model-context-protocol - code-retrieval diff --git a/Cargo.lock b/Cargo.lock index 01a6af5..604a946 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1759,7 +1759,7 @@ dependencies = [ [[package]] name = "neuromesh-api" -version = "1.0.0" +version = "1.1.0" dependencies = [ "chrono", "futures-util", @@ -1787,7 +1787,7 @@ dependencies = [ [[package]] name = "neuromesh-cache" -version = "1.0.0" +version = "1.1.0" dependencies = [ "chrono", "neuromesh-core", @@ -1800,7 +1800,7 @@ dependencies = [ [[package]] name = "neuromesh-cli" -version = "1.0.0" +version = "1.1.0" dependencies = [ "chrono", "dirs", @@ -1830,7 +1830,7 @@ dependencies = [ [[package]] name = "neuromesh-context" -version = "1.0.0" +version = "1.1.0" dependencies = [ "blake3", "chrono", @@ -1849,7 +1849,7 @@ dependencies = [ [[package]] name = "neuromesh-core" -version = "1.0.0" +version = "1.1.0" dependencies = [ "chrono", "dirs", @@ -1863,7 +1863,7 @@ dependencies = [ [[package]] name = "neuromesh-embed" -version = "1.0.0" +version = "1.1.0" dependencies = [ "dirs", "fastembed", @@ -1878,7 +1878,7 @@ dependencies = [ [[package]] name = "neuromesh-graph" -version = "1.0.0" +version = "1.1.0" dependencies = [ "bincode", "chrono", @@ -1900,7 +1900,7 @@ dependencies = [ [[package]] name = "neuromesh-graph-proxy" -version = "1.0.0" +version = "1.1.0" dependencies = [ "dirs", "neuromesh-core", @@ -1914,7 +1914,7 @@ dependencies = [ [[package]] name = "neuromesh-index" -version = "1.0.0" +version = "1.1.0" dependencies = [ "blake3", "chrono", @@ -1930,7 +1930,7 @@ dependencies = [ [[package]] name = "neuromesh-local-ai" -version = "1.0.0" +version = "1.1.0" dependencies = [ "neuromesh-core", "parking_lot", @@ -1941,7 +1941,7 @@ dependencies = [ [[package]] name = "neuromesh-mcp" -version = "1.0.0" +version = "1.1.0" dependencies = [ "chrono", "neuromesh-cache", @@ -1966,7 +1966,7 @@ dependencies = [ [[package]] name = "neuromesh-memory" -version = "1.0.0" +version = "1.1.0" dependencies = [ "chrono", "dirs", @@ -1980,7 +1980,7 @@ dependencies = [ [[package]] name = "neuromesh-observability" -version = "1.0.0" +version = "1.1.0" dependencies = [ "chrono", "dirs", @@ -1995,7 +1995,7 @@ dependencies = [ [[package]] name = "neuromesh-parser" -version = "1.0.0" +version = "1.1.0" dependencies = [ "neuromesh-core", "neuromesh-index", @@ -2026,7 +2026,7 @@ dependencies = [ [[package]] name = "neuromesh-provider" -version = "1.0.0" +version = "1.1.0" dependencies = [ "futures-util", "neuromesh-core", @@ -2039,7 +2039,7 @@ dependencies = [ [[package]] name = "neuromesh-router" -version = "1.0.0" +version = "1.1.0" dependencies = [ "neuromesh-context", "neuromesh-core", @@ -2052,7 +2052,7 @@ dependencies = [ [[package]] name = "neuromesh-task" -version = "1.0.0" +version = "1.1.0" dependencies = [ "neuromesh-core", "neuromesh-parser", diff --git a/Cargo.toml b/Cargo.toml index 8a6558b..73b5dcb 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -22,7 +22,7 @@ members = [ exclude = ["tests/fixtures/mini-axum"] [workspace.package] -version = "1.0.0" +version = "1.1.0" edition = "2021" rust-version = "1.80" authors = ["NeuroMesh Team "] diff --git a/crates/neuromesh-mcp/src/protocol.rs b/crates/neuromesh-mcp/src/protocol.rs index 1a4576e..83cb2d9 100644 --- a/crates/neuromesh-mcp/src/protocol.rs +++ b/crates/neuromesh-mcp/src/protocol.rs @@ -114,7 +114,7 @@ pub fn canonical_tool_name(name: &str) -> String { } pub fn initialize_instructions() -> &'static str { - "NeuroMesh MCP v1.0.0 (code-context-engine) — agent loop: (1) get_context_packet with the user task as written (default engine: fast — graph + query-side lexical expansion; prompt only; no client keywords; opt in to hybrid/deep for MiniLM embed-primary); (2) if coverage is partial/no_seed_resolved/no_confident_match, neuromesh_search_symbols or neuromesh_expand_gap; (3) neuromesh_expand_fold only when you need a folded body; (4) neuromesh_trace / neuromesh_get_dependencies for callers and blast radius; (5) after a successful edit, neuromesh_record_feedback with touched nodes. Check retrieval.resolution_tier and retrieval.cache_hit. Prefer these tools over reading whole files. neuromesh_get_context is deprecated — use get_context_packet." + "NeuroMesh MCP v1.1.0 (code-context-engine) — agent loop: (1) get_context_packet with the user task as written (default engine: fast — graph + query-side lexical expansion; prompt only; no client keywords; opt in to hybrid/deep for MiniLM embed-primary); (2) if coverage is partial/no_seed_resolved/no_confident_match, neuromesh_search_symbols or neuromesh_expand_gap; (3) neuromesh_expand_fold only when you need a folded body; (4) neuromesh_trace / neuromesh_get_dependencies for callers and blast radius; (5) after a successful edit, neuromesh_record_feedback with touched nodes. Check retrieval.resolution_tier and retrieval.cache_hit. Prefer these tools over reading whole files. neuromesh_get_context is deprecated — use get_context_packet." } pub fn tool_success(id: Option, val: &Value) -> Value { diff --git a/docs/measured.md b/docs/measured.md index 7fa1d68..2f91505 100644 --- a/docs/measured.md +++ b/docs/measured.md @@ -3,14 +3,15 @@ Every number here comes from one command on a pinned checkout: ```bash -bash scripts/benchmark-holdout.sh # all seven public sets, ~30 min on a laptop +bash scripts/benchmark-fast.sh # the nine public sets, optimised and in parallel (~1.5 min warm) +bash scripts/benchmark-holdout.sh # same numbers, debug build, one set at a time (~40 min) bash scripts/benchmark-holdout.sh holdout # one set ``` The rule behind the table: **a number from a repository the engine was tuned on is not the project's number.** Only the holdout rows are. -## Gold sets (2026-09-21, main after G4 — F86) +## Gold sets (2026-10-01, v1.1.0) | set | repos | role | recall | precision | forbidden | oracle reachable / strict | |---|---|---|---|---|---|---| @@ -22,8 +23,9 @@ not the project's number.** Only the holdout rows are. | holdout-ml | keras-io examples (Keras), setfit (Hugging Face) | **tuned on in D-3** (5 iterations read its numbers) — dev-class since then; fresh ML holdout is `holdout-ml2` | **1.000** | **0.478** | **0** | 10/10 · 10 | | **holdout-ml2** | peft (Hugging Face adapter library), keras-hub (Keras 3 model library) | never tuned on; first run after gold lock (session 12) | **1.000** | **0.632** | **0** | 10/10 · 6 | | **holdout-cfg** | lightning-hydra-template (Hydra YAML), detr (argparse) | D-4 gold; F55 (session 12) + F71/F72 (session 13) tuned on it — dev-class for config questions | **1.000** | **0.767** | **0** | — | -| holdout-web (30 q) | fastify/demo (Fastify API), shadcn-ui/taxonomy (Next.js app router) | dev-class for the web domain (fixed on since session 13; 10 blind questions added in G4 scored 0.58 before fixes) | **0.800** | **0.608** | **0** | — | +| holdout-web (30 q) | fastify/demo (Fastify API), shadcn-ui/taxonomy (Next.js app router) | dev-class for the web domain (fixed on since session 13; 10 blind questions added in G4 scored 0.58 before fixes) | **0.917** | **0.643** | **0** | — | | **private** | one closed-source B2B backend+frontend (Fastify/Drizzle + Next.js, ~1.2k files) | never tuned on; gold and checkout live outside this repo | **1.000** | **0.587** | **0** | — | +| concept (14 q) | this repository, plain-language questions (no identifier in the prompt) | dev-class (written 2026-09-30 from the upstream author's report, tuned on in session 16) | 0.679 | 0.392 | 0 | — || **concept-holdout** (12 q) | ripgrep 14.1.1 (Rust), plain-language questions | never tuned on; gold locked before any run; 1.0.0 scored recall **0.042** | **0.500** | **0.152** | **0** | — | - **recall / precision** are file-level against a hand-written gold (`gold_files`) per question. A forbidden file in the packet zeroes that question's precision. diff --git a/editors/vscode-neuromesh/package.json b/editors/vscode-neuromesh/package.json index fd288f7..bb4bb27 100644 --- a/editors/vscode-neuromesh/package.json +++ b/editors/vscode-neuromesh/package.json @@ -2,7 +2,7 @@ "name": "vscode-neuromesh", "displayName": "NeuroMesh", "description": "Evidence packets, reversible folds, and a live mesh sidebar for Cursor & VS Code. Don’t dump files — fold them.", - "version": "1.0.0", + "version": "1.1.0", "publisher": "neuromesh", "license": "MIT", "homepage": "https://github.com/pinoox/neuromesh", diff --git a/install.ps1 b/install.ps1 index 4a84795..a13717a 100644 --- a/install.ps1 +++ b/install.ps1 @@ -31,7 +31,7 @@ Write-Host @" |_| \_|\___|\__,_|_| \___/|_| |_|\___||___/_| |_| "@ -ForegroundColor Cyan -Write-Host "code-context-engine v1.0.0 — MCP context engine (NeuroMesh derivative)`n" -ForegroundColor Green +Write-Host "code-context-engine v1.1.0 — MCP context engine (NeuroMesh derivative)`n" -ForegroundColor Green Write-Host "Fetching latest release…" -ForegroundColor Gray $DownloadUrl = "https://github.com/$Repo/releases/latest/download/neuromesh-windows-x86_64.zip" diff --git a/install.sh b/install.sh index e999b31..1502a7f 100644 --- a/install.sh +++ b/install.sh @@ -28,7 +28,7 @@ cat << 'EOF' |_| \_|\___|\__,_|_| \___/|_| |_|\___||___/_| |_| EOF printf "${NC}\n" -printf "${BOLD}code-context-engine v1.0.0 — MCP context engine (NeuroMesh derivative)${NC}\n\n" +printf "${BOLD}code-context-engine v1.1.0 — MCP context engine (NeuroMesh derivative)${NC}\n\n" OS="$(uname -s)" ARCH="$(uname -m)" diff --git a/llms.txt b/llms.txt index 5df5ba4..2d4de66 100644 --- a/llms.txt +++ b/llms.txt @@ -8,7 +8,7 @@ > in its default engine. Repository: https://github.com/ParsaVictor/code-context-engine -Current release: v1.0.0 (2026-09-22) +Current release: v1.1.0 (2026-10-01) Independent derivative of NeuroMesh (https://github.com/pinoox/neuromesh), MIT. ## Facts worth quoting diff --git a/scripts/benchmark-fast.sh b/scripts/benchmark-fast.sh index b8c7bc7..5cc1804 100644 --- a/scripts/benchmark-fast.sh +++ b/scripts/benchmark-fast.sh @@ -59,8 +59,9 @@ for set in "${sets[@]}"; do tests+=(--test "${TEST[$set]}"); done cargo test --release -q -p neuromesh-context "${tests[@]}" --no-run 2>/dev/null bin_of() { # newest optimised binary for a harness name - ls -t target/release/deps/"$1"-*.exe target/release/deps/"$1"-* 2>/dev/null \ - | grep -v '\.d$\|\.pdb$' | head -1 + # absolute: run_set changes directory before executing it + ls -t "$root"/target/release/deps/"$1"-*.exe "$root"/target/release/deps/"$1"-* 2>/dev/null \ + | grep -Ev '\.(d|pdb|exp|lib|rlib)$' | head -1 } out="target/bench-fast"