diff --git a/crates/neuromesh-cli/src/commands/packet.rs b/crates/neuromesh-cli/src/commands/packet.rs index fd9de31..2001b67 100644 --- a/crates/neuromesh-cli/src/commands/packet.rs +++ b/crates/neuromesh-cli/src/commands/packet.rs @@ -38,6 +38,9 @@ struct PacketJsonOut { reduction_vs_workspace_pct: f32, selected_files: Vec, selected_files_count: usize, + /// Repository-relative paths, best first (highest activation of any + /// node in the file): what an evaluation needs for hit@k. + selected_paths: Vec, identifiers: Vec, seeds_missed: Vec, seed_resolution: Option, @@ -87,6 +90,15 @@ pub fn execute(args: &[String]) -> Result<()> { let view = activator.activate_tiered(&graph, &signature, OptimizationMode::Balanced); let latency_ms = started.elapsed().as_millis() as u64; + let mut best: std::collections::HashMap = std::collections::HashMap::new(); + for n in &view.active_nodes { + let p = n.node.file_path.to_string_lossy().replace('\\', "/"); + let e = best.entry(p).or_insert(f32::MIN); + *e = e.max(n.activation_score); + } + let mut selected_paths: Vec<(String, f32)> = best.into_iter().collect(); + selected_paths.sort_by(|a, b| b.1.total_cmp(&a.1).then_with(|| a.0.cmp(&b.0))); + let selected_paths: Vec = selected_paths.into_iter().map(|(p, _)| p).collect(); let mut files: Vec = packet_file_names(&view).into_iter().collect(); files.sort(); let reduction = if workspace_tokens > 0 { @@ -115,6 +127,7 @@ pub fn execute(args: &[String]) -> Result<()> { reduction_vs_workspace_pct: reduction, selected_files_count: files.len(), selected_files: files.clone(), + selected_paths, identifiers: signature.identifiers.clone(), seeds_missed, seed_resolution: view.seed_resolution_telemetry.clone(), diff --git a/crates/neuromesh-context/src/activator_seed.rs b/crates/neuromesh-context/src/activator_seed.rs index a170b8b..968305f 100644 --- a/crates/neuromesh-context/src/activator_seed.rs +++ b/crates/neuromesh-context/src/activator_seed.rs @@ -1011,7 +1011,14 @@ fn push_ranked_file_seeds( if best.matched < 2 { return false; } - let top = best.score; + // With an embedding sidecar loaded, meaning joins the lexical ranking + // (the lexical gate above still decides whether to seed at all). + #[cfg(feature = "embeddings")] + let ranked: Vec = fuse_with_embeddings(graph, prompt, ranked) + .into_iter() + .filter(|r| names_low || !crate::selector::is_noise_path(&r.path)) + .collect(); + let top = ranked[0].score; let picks: Vec<&neuromesh_graph::RankedFile> = ranked .iter() .take(3) @@ -1069,6 +1076,80 @@ fn push_ranked_file_seeds( true } +/// Lexical file ranking fused with the embedding file tier, when a code-aware +/// model's sidecar is loaded: a word the code never spells (a question about +/// "shrinking pictures" for `resize_image`) reaches a file through +/// meaning; a word it does spell keeps +/// its lexical weight. Convex combination of min-max normalised scores +/// (lexical 0.6, dense 0.4) — it keeps score ratios meaningful for the +/// runner-up rule, which reciprocal-rank fusion flattens (Bruch et al. 2023, +/// "An Analysis of Fusion Functions for Hybrid Retrieval"). +#[cfg(feature = "embeddings")] +fn fuse_with_embeddings( + graph: &NeuralProjectGraph, + prompt: &str, + lexical: Vec, +) -> Vec { + const DEPTH: usize = 50; + const W_LEX: f32 = 0.6; + const W_DENSE: f32 = 0.4; + let index = graph.embedding_index(); + if !index.is_loaded() || lexical.is_empty() { + return lexical; + } + // A loaded sidecar means embeddings are in use, whatever the config + // default says (the query must be embedded with the same model). + let mut cfg = neuromesh_core::Config::load().embeddings; + cfg.enabled = true; + // Only a code-aware model earns a vote: MiniLM, a general paraphrase + // model, lowered every plain-language set it was fused into. + if cfg.model != neuromesh_core::EmbeddingModelId::JinaCodeV2 { + return lexical; + } + let Ok(query) = neuromesh_embed::embed_query_cached(&cfg, prompt) else { + return lexical; + }; + let dense = index.file_ann_search(&query, DEPTH, 0.0); + if dense.is_empty() { + return lexical; + } + let lex_max = lexical[0].score.max(f32::EPSILON); + let (d_min, d_max) = dense.iter().fold((f32::MAX, f32::MIN), |(lo, hi), (_, s)| { + (lo.min(*s), hi.max(*s)) + }); + let d_span = (d_max - d_min).max(f32::EPSILON); + let mut fused: std::collections::HashMap = std::collections::HashMap::new(); + for r in lexical.iter().take(DEPTH) { + *fused.entry(r.id.clone()).or_insert(0.0) += W_LEX * r.score / lex_max; + } + for (id, s) in &dense { + *fused.entry(id.clone()).or_insert(0.0) += W_DENSE * (s - d_min) / d_span; + } + let mut out: Vec = fused + .into_iter() + .filter_map(|(id, score)| { + let lex = lexical.iter().find(|r| r.id == id); + let path = match lex { + Some(r) => r.path.clone(), + None => graph.get_node(&id)?.file_path, + }; + Some(neuromesh_graph::RankedFile { + id, + path, + score, + matched: lex.map(|r| r.matched).unwrap_or(0), + }) + }) + .collect(); + out.sort_by(|a, b| { + b.score + .partial_cmp(&a.score) + .unwrap_or(std::cmp::Ordering::Equal) + .then_with(|| a.path.cmp(&b.path)) + }); + out +} + pub(crate) fn push_body_word_seeds( graph: &NeuralProjectGraph, prompt: &str, diff --git a/crates/neuromesh-core/src/config.rs b/crates/neuromesh-core/src/config.rs index 2e37e85..511e399 100644 --- a/crates/neuromesh-core/src/config.rs +++ b/crates/neuromesh-core/src/config.rs @@ -306,6 +306,11 @@ impl Config { } if let Ok(raw) = std::env::var("NEUROMESH_EMBED_MODEL") { if let Some(model) = crate::EmbeddingModelId::parse(&raw) { + if model != self.embeddings.model { + // Another model's width: 768-d vectors cut to MiniLM's 384 + // are not what the model was trained to produce. + self.embeddings.matryoshka_dim = model.default_matryoshka_dim(); + } self.embeddings.model = model; } } diff --git a/crates/neuromesh-core/src/embedding_config.rs b/crates/neuromesh-core/src/embedding_config.rs index b60c7c8..879d043 100644 --- a/crates/neuromesh-core/src/embedding_config.rs +++ b/crates/neuromesh-core/src/embedding_config.rs @@ -6,6 +6,9 @@ pub enum EmbeddingModelId { Gemma300mQ4, #[default] MiniLmMultilingualQ, + /// jinaai/jina-embeddings-v2-base-code: trained on code and its docs + /// (768-d). Downloaded by fastembed on first use (~640 MB). + JinaCodeV2, } impl EmbeddingModelId { @@ -14,6 +17,7 @@ impl EmbeddingModelId { "gemma300m_q4" | "gemma300m-q4" | "embeddinggemma300m_q4" | "gemma" => { Some(Self::Gemma300mQ4) } + "jina_code_v2" | "jina-code-v2" | "jina_code" | "jina-code" => Some(Self::JinaCodeV2), "minilm_multilingual_q" | "minilm-multilingual-q" | "minilm" => { Some(Self::MiniLmMultilingualQ) } @@ -25,6 +29,7 @@ impl EmbeddingModelId { match self { Self::Gemma300mQ4 => "gemma300m_q4", Self::MiniLmMultilingualQ => "minilm_multilingual_q", + Self::JinaCodeV2 => "jina_code_v2", } } @@ -32,6 +37,7 @@ impl EmbeddingModelId { match self { Self::Gemma300mQ4 => 256, Self::MiniLmMultilingualQ => 384, + Self::JinaCodeV2 => 768, } } } diff --git a/crates/neuromesh-embed/src/embedder.rs b/crates/neuromesh-embed/src/embedder.rs index 3ae7f1a..33c5252 100644 --- a/crates/neuromesh-embed/src/embedder.rs +++ b/crates/neuromesh-embed/src/embedder.rs @@ -83,6 +83,7 @@ pub fn format_query_for_model(model: EmbeddingModelId, prompt: &str) -> String { match model { EmbeddingModelId::Gemma300mQ4 => format_query_gemma(prompt), EmbeddingModelId::MiniLmMultilingualQ => format_query_minilm(prompt), + EmbeddingModelId::JinaCodeV2 => prompt.trim().to_string(), } } @@ -95,6 +96,7 @@ pub fn format_document_for_model( ) -> String { match model { EmbeddingModelId::Gemma300mQ4 => format_document_gemma(title, kind, signature, doc), + EmbeddingModelId::JinaCodeV2 => format_document_minilm(title, kind, signature, doc), EmbeddingModelId::MiniLmMultilingualQ => { format_document_minilm(title, kind, signature, doc) } @@ -116,6 +118,38 @@ fn try_init_text_embedding( try_load_bundled_minilm(config.model, config.intra_threads) .map_err(|e| EmbedderError::Init(format!("{e}. {}", install_hint()))) } + EmbeddingModelId::JinaCodeV2 => { + // Loaded from plain files, not fastembed's hf-hub cache (its + // symlinks fail on Windows without developer mode): + // /jina-code-v2/{model_quantized.onnx, tokenizer*.json, …} + // from huggingface.co/jinaai/jina-embeddings-v2-base-code. + let dir = crate::model_install::default_models_root().join("jina-code-v2"); + let read = |name: &str| { + std::fs::read(dir.join(name)).map_err(|e| { + EmbedderError::Init(format!( + "jina_code_v2: {} missing ({e})", + dir.join(name).display() + )) + }) + }; + let user_model = fastembed::UserDefinedEmbeddingModel::new( + read("model_quantized.onnx")?, + fastembed::TokenizerFiles { + tokenizer_file: read("tokenizer.json")?, + config_file: read("config.json")?, + special_tokens_map_file: read("special_tokens_map.json")?, + tokenizer_config_file: read("tokenizer_config.json")?, + }, + ) + .with_pooling(fastembed::Pooling::Mean) + .with_quantization(fastembed::QuantizationMode::Dynamic); + let mut opts = fastembed::InitOptionsUserDefined::default(); + if let Some(n) = config.intra_threads { + opts = opts.with_intra_threads(n); + } + TextEmbedding::try_new_from_user_defined(user_model, opts) + .map_err(|e| EmbedderError::Init(format!("jina_code_v2: {e}"))) + } EmbeddingModelId::Gemma300mQ4 => Err(EmbedderError::Init(format!( "gemma300m_q4 is not installable yet; use MiniLM ({}). {}", EmbeddingModelId::MiniLmMultilingualQ.as_str(), diff --git a/crates/neuromesh-embed/src/model_install.rs b/crates/neuromesh-embed/src/model_install.rs index 4951c5b..0d0994f 100644 --- a/crates/neuromesh-embed/src/model_install.rs +++ b/crates/neuromesh-embed/src/model_install.rs @@ -33,7 +33,26 @@ pub const MINILM_MULTILINGUAL_Q: EmbedModelSpec = EmbedModelSpec { ], }; -pub static CATALOG: &[EmbedModelSpec] = &[MINILM_MULTILINGUAL_Q]; +/// jinaai/jina-embeddings-v2-base-code, int8 ONNX (768-d): trained on code +/// and its documentation. Fused with the lexical ranking for plain-language +/// questions it lifts the ripgrep holdout from 0.500 to 0.667 recall +/// (`docs/measured.md`); MiniLM, a general paraphrase model, lowered it. +pub const JINA_CODE_V2: EmbedModelSpec = EmbedModelSpec { + id: "jina-code-v2", + dir_name: "jina-code-v2", + label: "Jina embeddings v2 base code, int8 (768-dim, code-aware, ~160 MB)", + aliases: &["jina", "jina-code", "jina_code", "jina_code_v2"], + hf_base: "https://huggingface.co/jinaai/jina-embeddings-v2-base-code/resolve/main", + files: &[ + "onnx/model_quantized.onnx", + TOKENIZER_NAME, + "config.json", + "special_tokens_map.json", + "tokenizer_config.json", + ], +}; + +pub static CATALOG: &[EmbedModelSpec] = &[MINILM_MULTILINGUAL_Q, JINA_CODE_V2]; #[derive(Debug, Clone, Copy, Default)] pub struct InstallOptions { @@ -86,11 +105,18 @@ pub fn model_install_dir(spec: &EmbedModelSpec) -> PathBuf { } pub fn is_model_installed(spec: &EmbedModelSpec) -> bool { - model_dir_ready(&model_install_dir(spec)) + spec_ready(spec, &model_install_dir(spec)) } -fn model_dir_ready(dir: &Path) -> bool { - dir.join(ONNX_NAME).is_file() && dir.join(TOKENIZER_NAME).is_file() +/// Local file name of a spec entry: `onnx/model_quantized.onnx` is saved +/// as `model_quantized.onnx` next to the tokenizer. +fn local_name(name: &str) -> &str { + name.rsplit('/').next().unwrap_or(name) +} + +/// Every file of the spec is on disk. +fn spec_ready(spec: &EmbedModelSpec, dir: &Path) -> bool { + spec.files.iter().all(|f| dir.join(local_name(f)).is_file()) } pub fn list_installed() -> Vec<(EmbedModelSpec, PathBuf)> { @@ -98,7 +124,7 @@ pub fn list_installed() -> Vec<(EmbedModelSpec, PathBuf)> { .iter() .filter_map(|spec| { let dir = model_install_dir(spec); - if model_dir_ready(&dir) { + if spec_ready(spec, &dir) { Some((*spec, dir)) } else { None @@ -147,9 +173,9 @@ fn install_model_inner( let dest = model_install_dir(spec); std::fs::create_dir_all(&dest)?; - if !opts.force && model_dir_ready(&dest) { + if !opts.force && spec_ready(spec, &dest) { if !opts.quiet { - eprintln!("MiniLM already installed at {}", dest.display()); + eprintln!("{} already installed at {}", spec.id, dest.display()); } return Ok(dest); } @@ -160,43 +186,63 @@ fn install_model_inner( .map_err(|e| ModelInstallError::Download(e.to_string()))?; for name in spec.files { - let out = dest.join(name); + let local = local_name(name); + let out = dest.join(local); if !opts.force && out.is_file() { if !opts.quiet { - eprintln!(" skip {name} (exists)"); + eprintln!(" skip {local} (exists)"); } continue; } let url = format!("{}/{}", spec.hf_base, name); if !opts.quiet { - eprintln!(" fetch {name}…"); + eprintln!(" fetch {local}…"); } - let response = client - .get(&url) - .send() - .map_err(|e| ModelInstallError::Download(format!("{name}: {e}")))?; - if !response.status().is_success() { - return Err(ModelInstallError::Download(format!( - "{name}: HTTP {}", - response.status() - ))); + // A slow or flaky link drops large files mid-body: retry the whole + // file a few times before giving up (the .download temp never + // becomes the real file unless it arrived complete). + let mut last_err = String::new(); + let mut bytes = None; + for attempt in 1..=3 { + let result = client + .get(&url) + .send() + .map_err(|e| e.to_string()) + .and_then(|r| { + if r.status().is_success() { + r.bytes().map_err(|e| e.to_string()) + } else { + Err(format!("HTTP {}", r.status())) + } + }); + match result { + Ok(b) => { + bytes = Some(b); + break; + } + Err(e) => { + if !opts.quiet { + eprintln!(" {local}: attempt {attempt} failed ({e})"); + } + last_err = e; + } + } } - let bytes = response - .bytes() - .map_err(|e| ModelInstallError::Download(format!("{name}: {e}")))?; - let tmp = dest.join(format!(".{name}.download")); + let bytes = + bytes.ok_or_else(|| ModelInstallError::Download(format!("{local}: {last_err}")))?; + let tmp = dest.join(format!(".{local}.download")); std::fs::write(&tmp, &bytes)?; std::fs::rename(&tmp, &out)?; } - if !model_dir_ready(&dest) { + if !spec_ready(spec, &dest) { return Err(ModelInstallError::Download( "install incomplete after download".into(), )); } if !opts.quiet { - eprintln!("MiniLM installed at {}", dest.display()); + eprintln!("{} installed at {}", spec.id, dest.display()); } Ok(dest) } diff --git a/docs/CHANGELOG.md b/docs/CHANGELOG.md index 3ce815c..8305836 100644 --- a/docs/CHANGELOG.md +++ b/docs/CHANGELOG.md @@ -4,6 +4,31 @@ All notable user-facing changes live here. The README stays a product guide, not ## Unreleased +### Retrieval + +- **Optional code-aware embeddings** — `neuromesh install embed jina-code` fetches Jina embeddings v2 + base code (int8 ONNX, ~160 MB); select it with `NEUROMESH_EMBED_MODEL=jina_code_v2` (or + `embeddings.model` in the config). For plain-language questions the lexical file ranking is then + fused with the model's file vectors (convex combination, lexical 0.6 / dense 0.4). Measured with the + model loaded, fusion off → on: ripgrep holdout 0.500 / 0.156 → **0.667 / 0.347**, click holdout + 0.958 / 0.342 → 0.958 / **0.439**, this repository 0.679 / 0.404 → 0.786 / 0.429. MiniLM is never + fused: it lowered every plain-language set. Without the model nothing changes. +- **Plain-language questions trust the whole-question ranking** — word-by-word guesses outside its + picks are dropped. + +### Fixes + +- A symbol whose line range starts past the end of its file no longer panics packet rendering. +- Each fold is listed once per file (it was repeated once per seed). +- Guesses the server made itself (a prose word, an alias) are no longer reported as missing seeds, so + coverage does not say `partial` and send the agent to search for them. +- `neuromesh install embed` retries a dropped download up to three times. + +### Measured + +- `docs/measured.md`: outside baselines (plain BM25, Aider's RepoMap) on five holdouts, and a second + plain-language holdout (click 8.1.7, 0.958 / 0.342 without embeddings). + ## 1.1.0 — 2026-10-01 Plain-language questions, the protocol bug and the cold-start latency the upstream author diff --git a/docs/configuration.md b/docs/configuration.md index 4f3087b..eefd6b3 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -111,9 +111,17 @@ Sidecar v4/v5 requires `neuromesh embed rebuild` after upgrading to v0.9.0. neuromesh doctor --embed # sidecar status + model install check neuromesh doctor --embed --bench # p50/p95 embed latency (model required) neuromesh install embed minilm # download MiniLM Q (hybrid/deep prerequisite) +neuromesh install embed jina-code # code-aware model (~160 MB), see below neuromesh embed prefetch # warm installed MiniLM ``` +**Code-aware embeddings (recommended for plain-language questions).** With +`NEUROMESH_EMBED_MODEL=jina_code_v2` (or `"embeddings": {"model": "jina_code_v2"}`) and the model +installed, the background embedding build uses Jina embeddings v2 base code, and a question that names +no identifier is ranked by the lexical file ranking fused with the model's file vectors. Measured on +the ripgrep plain-language holdout: recall 0.500 → 0.667, precision 0.156 → 0.347 +(`docs/measured.md`). MiniLM is not fused (it made those questions worse). + Release tarballs ship the binary only; install MiniLM before hybrid/deep. --- diff --git a/docs/measured.md b/docs/measured.md index 0806b8a..14d8d32 100644 --- a/docs/measured.md +++ b/docs/measured.md @@ -156,6 +156,8 @@ map of the whole repository, not to answer one question, and it shows. | holdout-lang (os-lib, cli, Flux.jl) | 1.000 / 0.589 | **0.933 / 0.933** | 1.000 / 0.333 | 0.400 / 0.080 ¹ | | concept-holdout (ripgrep, plain language) | 0.500 / 0.156 | 0.250 / 0.333 | 0.500 / **0.222** | 0.042 / 0.017 | | concept-holdout2 (click, plain language) | **0.958 / 0.342** | 0.417 / 0.500 | 0.750 / 0.306 | 0.333 / 0.067 | +| concept-holdout, ours **+ jina-code** | **0.667 / 0.347** | | | | +| concept-holdout2, ours **+ jina-code** | **0.958 / 0.439** | | | | Where we stand, plainly: on questions that name code, the engine gets every gold file at a precision no fixed cut of BM25 reaches on three of four holdouts. On holdout-lang the single top @@ -166,3 +168,12 @@ repository, which helps every method. ¹ Aider's tree-sitter query for Julia fails to load (`Invalid node type: module`); Flux.jl's five questions are left out of its row. + +**With the optional code-aware model** (`neuromesh install embed jina-code`, +`NEUROMESH_EMBED_MODEL=jina_code_v2`): plain-language questions fuse the lexical file ranking with +the model's file vectors. Same build, model loaded, fusion off → on: ripgrep 0.500 / 0.156 → +0.667 / 0.347, click 0.958 / 0.342 → 0.958 / 0.439, this repository 0.679 / 0.404 → 0.786 / 0.429. +The fusion weights (0.6 lexical / 0.4 dense) were set from the literature before the run, not fitted +to these sets; both holdouts had been run before (ripgrep several times while the lexical ranker was +built), so read them as dev-adjacent, not blind. MiniLM fused the same way lowered every set (ripgrep +0.375, this repository 0.607) and is never fused. diff --git a/docs/planning/handoff-2026-10-01-session16.fa.md b/docs/planning/handoff-2026-10-01-session16.fa.md new file mode 100644 index 0000000..3c2dd8a --- /dev/null +++ b/docs/planning/handoff-2026-10-01-session16.fa.md @@ -0,0 +1,44 @@ +# Handoff — session 16 (۲۰۲۶-۰۹-۳۰ تا ۲۰۲۶-۱۰-۰۱) + +ورودی: گزارش یوسف (۳ از ۷ روی ریپوی خودمان، ۲۶ ثانیه، باگ stdout). نقشه: `11-roadmap-2026-09-30-concept-queries.fa.md`. + +## چه چیزی merge شد + +| PR | محتوا | +|---|---| +| #126 | بنر داشبورد از stdout به stderr؛ تستی که باینری واقعی را اجرا می‌کند | +| #127 (v1.1.0) | رتبه‌بند BM25F کل‌سؤال (`neuromesh-graph/src/file_rank.rs`)، `comment_index`، thesaurus در `thesaurus.txt`، F89 (کلمه‌ی انگلیسی anchor نیست)، اصلاح cold start، `NM_TIMING=1` | +| #128 | سؤال ساده فقط به رتبه‌بند اعتماد می‌کند؛ رفع panic در `skeleton.rs`؛ `scripts/compare_baselines.py` | +| #129 | جدول مقایسه با BM25 و Aider در `docs/measured.md`؛ holdout دوم (click)؛ folds تکراری؛ حدس‌های سرور «missing» نیستند | + +انتشار: tag `v1.1.0` با سه باینری. MCP دسکتاپ پارسا روی v1.1.0 است (backup: `neuromesh-v1.0.0-backup.exe`). + +## اعداد (صادقانه) + +| ست | v1.0.0 | الان | +|---|---|---| +| concept (ریپوی خودمان، dev) | 0.286 | 0.679 / 0.404 | +| concept-holdout (ripgrep، یک بار) | 0.042 | 0.500 / 0.156 | +| concept-holdout2 (click، تازه، یک بار) | — | 0.958 / 0.342 | +| ۹ ست قدیمی | — | بدون افت؛ web 0.608→0.643 | +| query اول روی کلون تازه | 3.6 s | 2.6 s؛ بعدی‌ها 0.1–0.4 s | + +مقایسه با BM25: روی سؤال‌هایی که اسم کد دارند جلوییم (recall 1.0 با دقت 0.59–0.70)، جز holdout-lang که BM25@1 دقت 0.93 دارد. روی ripgrep (سؤال ساده) مساوی، روی click جلوتر. + +## رد شده با عدد (تکرار نکنید) + +- seed کردن تعریف‌های فایل بزرگ به‌جای کل فایل: ripgrep 0.500→0.375 +- ترکیب با MiniLM: ripgrep 0.375، self 0.607 (MiniLM کد نمی‌فهمد؛ F79 دوباره تایید شد) +- gate روی fillهای packet مفهومی: فایل‌های اضافه خود seedها هستند، نه fill + +## باز + +- J2: مدل `jina_code_v2` (مخصوص کد) به‌عنوان گزینه اضافه شد (`NEUROMESH_EMBED_MODEL=jina_code_v2`، فایل‌ها در `%LOCALAPPDATA%/neuromesh/models/jina-code-v2`). نتیجه‌ی آزمایش fusion در بخش پایین. +- F92 دقت packet سؤال مفهومی (0.15–0.40)؛ F93 واژه‌هایی که در کد نیست؛ F94 فایل‌های hub. +- holdout-lang: fillهای connector (utility:16–41) دقت را پایین می‌آورند — تیون precision دو بار بسته شده؛ فقط با holdout جدید. + +## روش کار سریع + +- `bash scripts/benchmark-fast.sh` — ۹ ست در ۱.۵ تا ۵ دقیقه (release، موازی) +- `bash scripts/benchmark-fast.sh concept concept-holdout concept-holdout2` +- یک build در هر لحظه؛ دیسک یک بار پر شد. diff --git a/docs/planning/reply-yoosef-2026-10-01.md b/docs/planning/reply-yoosef-2026-10-01.md index 4df98ac..120cf33 100644 --- a/docs/planning/reply-yoosef-2026-10-01.md +++ b/docs/planning/reply-yoosef-2026-10-01.md @@ -29,8 +29,17 @@ Yours are plain language. So we wrote two new sets, with gold committed before a |---|---|---| | concept (this repo, 14 q) | 0.286 | **0.679** | | concept-holdout (ripgrep, 12 q, never tuned on) | 0.042 | **0.500** | +| concept-holdout2 (click 8.1.7, 12 q, written and run once after 1.1.0) | — | **0.958** (precision 0.342) | | the nine existing sets | — | no recall lost; precision unchanged or better | +Against outside baselines on the same gold (`scripts/compare_baselines.py`): plain file-level +BM25 cut at three files gets 0.500 / 0.222 on ripgrep (level with us on recall, better on +precision) and 0.750 / 0.306 on click (we are ahead on both). Aider's RepoMap, used the way +Aider feeds it a chat message, reaches 0.042 and 0.333 at five files: it maps a repository, it +does not answer a question. On the code-naming holdouts we keep recall 1.000 at precision +0.59–0.70, which no fixed BM25 cut reaches, except holdout-lang where BM25's single top file is +right 93% of the time. Full table: `docs/measured.md`. + Honest gaps: half the ripgrep questions still miss. Most use words the code never spells ("clickable link" vs `hyperlink`, "machine-readable" vs `json`). Plain-language packets are also wide (precision 0.15). Both are the next work. diff --git a/docs/research/contributions-log.md b/docs/research/contributions-log.md new file mode 100644 index 0000000..b759231 --- /dev/null +++ b/docs/research/contributions-log.md @@ -0,0 +1,103 @@ +# Research log — contributions, evidence, and the paper plan + +One place for every idea this project tried, kept with its numbers, so a paper can be written from +it later. Rule for this file: **every claim carries the measurement that supports it, and every +rejected idea stays here with the number that rejected it.** Negative results are part of the paper. + +Working title: *Task-conditioned context packets for coding agents: retrieving the files a question +needs, measured on holdouts*. + +--- + +## 1. Problem and thesis + +Coding agents (Claude Code, Cursor, Copilot agents) spend most of their tokens reading code to find +what a task needs. The question this project answers: **given a task in plain text and a repository, +which minimal set of files/definitions does the model need, and how do we prove the answer on +repositories we never tuned on?** + +Thesis: a graph-and-lexicon engine with explicit evidence rules (a file enters the packet only for a +word the prompt wrote), plus hybrid lexical/dense ranking for plain-language questions, reaches +recall 1.0 at precision 0.6–0.7 on code-naming questions and beats plain BM25 and Aider's RepoMap on +plain-language ones — at 97–99% fewer tokens than the workspace. + +## 2. Contributions (candidate list for the paper) + +| # | Contribution | Where | Evidence | +|---|---|---|---| +| C1 | **Holdout discipline for retrieval engines**: gold written from source before any run, locked by commit, run once; a set that is looked at becomes "dev-class" and is labelled so | `docs/measured.md`, `tests/third_party/*` | 11 sets, 9 languages; dev vs holdout gap reported per set | +| C2 | **Prompt-evidence seeding**: every packet file must be justified by a word the prompt wrote (identifier, quoted literal, route, config key, stem, directory convention, body word) | `neuromesh-context/src/activator_seed.rs`, `seed/` | holdouts recall 1.000, precision 0.589–0.700 | +| C3 | **Prose words are guesses, not anchors** (F89): an English word that happens to be a symbol name ("How does this *tool*…") is demoted unless the prompt marks it as code (backticks, `()`, `!`, `.`, `::`, fragment of another identifier) | `is_prose_word` | fixed upstream author's `root_fs_safety` miss; no regression on 9 sets | +| C4 | **Whole-question BM25F file ranking** with a separate **comment field** (comments are the prose closest to how people ask), Snowball stemming, general software thesaurus kept out of the indexed source | `neuromesh-graph/src/file_rank.rs`, `thesaurus.txt` | plain-language self set 0.286 → 0.679; ripgrep holdout 0.042 → 0.500 | +| C5 | **Code-aware hybrid fusion** for plain-language questions only: convex combination (0.6 lexical / 0.4 dense, Bruch et al. 2023) of the BM25F ranking with Jina code v2 file vectors; the general-purpose model (MiniLM) is shown to *hurt* and is excluded | `fuse_with_embeddings` | ripgrep 0.500/0.156 → 0.667/0.347; click 0.958/0.342 → 0.958/0.439; MiniLM: ripgrep 0.375 | +| C6 | **Self-measurement contamination** finding: an engine measured on its own repository is biased by its own source (a thesaurus in a `.rs` file ranked first on 4 of 14 questions; doc comments quoting gold questions) | session 16 notes | 4/14 questions flipped until the thesaurus moved to `.txt` | +| C7 | **Cold-start correctness**: an MCP server must never answer from a half-built index ("no seed" from an empty graph); "ready" must not wait for optional embeddings | `wait_for_index`, `has_complete_index` | first answer 3.6 s → 2.6 s; empty answers eliminated | +| C8 | **Task-level isolation** (P0): one graph per project id, no cross-project leakage, CI leak test | `docs/isolation.md` | merged upstream (pinoox/neuromesh#34) | +| C9 | **Fast, reproducible benchmark loop**: optimised harnesses run in parallel | `scripts/benchmark-fast.sh` | 9 sets in 1.5–5 min instead of ~40 | + +## 3. Results to report (current) + +### 3.1 Code-naming questions (holdouts never tuned on) + +| set | ours R / P | BM25 R@1/P@1 | BM25 R@3/P@3 | Aider RepoMap R@5/P@5 | +|---|---|---|---|---| +| gin + torchvision | 1.000 / 0.700 | 0.675 / 0.700 | 0.950 / 0.333 | 0.475 / 0.110 | +| libuv + fmt | 1.000 / 0.589 | 0.594 / 0.625 | 0.938 / 0.334 | 0.406 / 0.087 | +| peft + keras-hub | 1.000 / 0.632 | 0.400 / 0.700 | 0.683 / 0.400 | 0.167 / 0.080 | +| os-lib + cli + Flux.jl | 1.000 / 0.589 | 0.933 / 0.933 | 1.000 / 0.333 | 0.400 / 0.080 | + +### 3.2 Plain-language questions + +| set | BM25 R@3/P@3 | ours lexical | ours + jina-code fusion | +|---|---|---|---| +| ripgrep (Rust, 12 q) | 0.500 / 0.222 | 0.500 / 0.156 | **0.667 / 0.347** | +| click (Python, 12 q, fresh) | 0.750 / 0.306 | 0.958 / 0.342 | **0.958 / 0.439** | +| this repository (dev) | 0.357 / 0.143 | 0.679 / 0.404 | **0.786 / 0.429** | + +### 3.3 Earlier measured (sessions 9–15) + +- Task success with a real model (DeepSeek-V4-Flash, GLM judge): all gates pass on two holdouts. +- Token reduction vs whole workspace: 97.5–99.8%. +- Against upstream NeuroMesh v0.9.0 on 112 tasks: recall 0.735 → 0.938, ~3× precision. + +## 4. Negative results (keep — they are part of the paper) + +| Idea | Result | Why it failed | +|---|---|---| +| MiniLM as primary retrieval (F79) | worse than lexical on every set | general paraphrase model; code vocabulary | +| MiniLM fused with BM25F | ripgrep 0.500 → 0.375, self 0.679 → 0.607 | same | +| Seed big files by their best-matching definitions | ripgrep 0.500 → 0.375 | definition seeds lost downstream; packet not smaller | +| Lexical-relevance gate on packet fill | recall 0.679 → 0.643, precision flat | extra files are seeds, not fills | +| Precision tuning on dev sets (twice) | gains did not transfer to holdouts | overfitting to dev; why C1 exists | +| Aider RepoMap as a question→file retriever | R@5 0.04–0.48 | built as a whole-repo map, not a retriever | + +## 5. What a top-venue paper still needs (honest gap list) + +1. **Standard benchmarks**, not only ours: file-level localisation on **SWE-bench Lite / Verified** + (gold = files touched by the reference patch), plus RepoBench or Long Code Arena retrieval tasks. +2. **Strong published baselines** on the same data: Agentless localisation, LocAgent / CodeRAG-style + retrievers, SWE-agent's search tools, a dense code retriever (e.g. Jina/Voyage code) alone, BM25. +3. **End-to-end task success per token** with at least two models (the project's actual claim): + resolve rate on SWE-bench Lite subset with vs without the packet, tokens spent. +4. **Ablations** for every component in §2 (C2–C5), each on holdouts. +5. **Statistical treatment**: confidence intervals (bootstrap over questions), more questions per set + (≥ 50 per holdout), inter-annotator agreement on gold (two people write gold independently). +6. **Latency and cost** table (index time, p50/p95 per query, model sizes). + +Realistic venues: now — a workshop or industry track (LLM4Code, MSR/FSE industry, ICSE SEIP). +Top-tier (ICSE/FSE/ASE main track, or NeurIPS/ICLR datasets & benchmarks track for the holdout +methodology) once items 1–5 are done. + +## 6. Timeline of sessions (for the paper's "development" section) + +- Sessions 1–5: fork, P0 isolation, task-success harness, real-repo gold (precision 0.146 → baseline). +- Sessions 6–8: stage 4 precision 0.146 → 0.856 on dev; ratchets. +- Sessions 9–15: holdout phases, generality (C/C++/Scala/R/Julia grammars), model task success, v1.0.0. +- Session 16 (2026-09-30 → 10-01): upstream author's report → plain-language class discovered; + C3–C7, C9; v1.1.0; baselines; jina-code fusion. + +## 7. Next experiments queued + +- Cross-encoder reranking (jina-reranker-v1-turbo, the Continue/Cody pattern) on top of C5, aimed at + precision on plain-language packets (F92). +- SWE-bench Lite file localisation run (item 5.1) — the first step toward a top-venue paper. diff --git a/scripts/swebench_localize.py b/scripts/swebench_localize.py new file mode 100644 index 0000000..911f73e --- /dev/null +++ b/scripts/swebench_localize.py @@ -0,0 +1,116 @@ +"""File-level localisation on SWE-bench Lite: does the packet contain the file the +reference patch edits? + + python scripts/swebench_localize.py --data lite.json --repos + --bin [--limit N] [--repo django/django] [--out results.jsonl] + +For every instance: check out `base_commit` into a scratch worktree, run +`neuromesh packet --json --query ` there (fresh NEUROMESH_HOME), and +record the packet's files in order; plain file-level BM25 over the same checkout ranks +the same query as the baseline. Gold = the files `patch` edits (exactly one per Lite +instance). Reports hit@k (k = 1, 3, 5, packet) — the "file-level Acc@k" used by +Agentless and LocAgent — plus packet size and tokens. + +Results stream to --out as JSON lines, so an interrupted run resumes where it stopped. +""" +import argparse +import json +import os +import re +import shutil +import subprocess +import sys +import tempfile + +sys.path.insert(0, os.path.dirname(__file__)) +from compare_baselines import Bm25, repo_files # noqa: E402 + + +def gold_files(patch): + return sorted(set(re.findall(r"^diff --git a/(\S+)", patch, re.M))) + + +def checkout(clone, commit, dest): + if os.path.isdir(dest): + subprocess.run(["git", "-C", clone, "worktree", "remove", "--force", dest], capture_output=True) + shutil.rmtree(dest, ignore_errors=True) + subprocess.run(["git", "-C", clone, "worktree", "prune"], capture_output=True) + r = subprocess.run(["git", "-C", clone, "worktree", "add", "--detach", dest, commit], capture_output=True, text=True) + if r.returncode != 0: + raise RuntimeError(r.stderr.strip()[-300:]) + + +def run_packet(binary, workspace, query): + home = tempfile.mkdtemp(prefix="nmhome-") + env = dict(os.environ, NEUROMESH_HOME=home, NEUROMESH_NO_BROWSER="1", PWD=workspace) + try: + r = subprocess.run( + [binary, "packet", "--json", "--query", query], + cwd=workspace, env=env, capture_output=True, text=True, timeout=900, + ) + out = r.stdout + start = out.find("{") + data = json.loads(out[start:]) if start >= 0 else {} + return data + finally: + shutil.rmtree(home, ignore_errors=True) + + +def hit(files, gold, k=None): + top = files if k is None else files[:k] + top = [f.replace("\\", "/") for f in top] + return int(all(any(t.endswith(g) or g.endswith(t) for t in top) for g in gold)) + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--data", required=True) + ap.add_argument("--repos", required=True) + ap.add_argument("--bin", required=True) + ap.add_argument("--out", default="results.jsonl") + ap.add_argument("--limit", type=int, default=0) + ap.add_argument("--repo", default="") + ap.add_argument("--work", default=os.path.join(tempfile.gettempdir(), "swe-wt")) + args = ap.parse_args() + + rows = json.load(open(args.data, encoding="utf-8")) + if args.repo: + rows = [r for r in rows if r["repo"] == args.repo] + done = set() + if os.path.exists(args.out): + done = {json.loads(l)["instance_id"] for l in open(args.out, encoding="utf-8") if l.strip()} + todo = [r for r in rows if r["instance_id"] not in done] + if args.limit: + todo = todo[: args.limit] + os.makedirs(args.work, exist_ok=True) + with open(args.out, "a", encoding="utf-8") as out: + for i, row in enumerate(todo, 1): + name = row["repo"].split("/")[1] + clone = os.path.join(args.repos, name) + dest = os.path.join(args.work, name) + gold = gold_files(row["patch"]) + rec = {"instance_id": row["instance_id"], "repo": row["repo"], "gold": gold} + try: + checkout(clone, row["base_commit"], dest) + pkt = run_packet(args.bin, dest, row["problem_statement"]) + files = pkt.get("selected_paths") or pkt.get("selected_files", []) + rec.update( + ours_files=files, + ours_tokens=pkt.get("packet_tokens"), + ours_latency_ms=pkt.get("latency_ms"), + ours_hit=hit(files, gold), + **{f"ours_hit@{k}": hit(files, gold, k) for k in (1, 3, 5)}, + ) + bm = Bm25(dest, repo_files(dest)).rank(row["problem_statement"]) + rec.update(**{f"bm25_hit@{k}": hit(bm, gold, k) for k in (1, 3, 5, 10)}) + except Exception as e: # recorded, not fatal: one bad checkout must not stop the run + rec["error"] = str(e)[-300:] + out.write(json.dumps(rec) + "\n") + out.flush() + print(f"[{i}/{len(todo)}] {rec['instance_id']} ours={rec.get('ours_hit')} " + f"files={len(rec.get('ours_files', []))} bm25@1={rec.get('bm25_hit@1')} {rec.get('error', '')[:80]}", + flush=True) + + +if __name__ == "__main__": + main()