From 5d4c9cbef65e2b35841e55b9983f899661a9c484 Mon Sep 17 00:00:00 2001 From: ParsaVictor <1.parsa.karkooti@gmail.com> Date: Thu, 1 Oct 2026 16:53:50 +0330 Subject: [PATCH 1/5] issue mode: long reports also seed the whole-question ranking and order the packet by it SWE-bench Lite, dev split (flask, requests, seaborn, xarray, pylint; 24 issues the parameters were chosen on): hit@1 0.125 -> 0.333, hit@3 0.250 -> 0.583, hit@5 0.417 -> 0.667, in-packet 0.583 -> 0.667 (plain BM25: 0.250 / 0.625 / 0.750). The other seven repositories are the test split, measured once afterwards. - a prompt of 60+ words with anchors also seeds the BM25F ranking's best file and runner-up within 70%; nothing is pruned - gold::packet_file_order: packet files best first; for long reports reciprocal-rank fusion of activation order and the file ranking (lexical weight 2, k = 60); CLI selected_paths uses it - packet --json: ranked_paths (BM25F top 10) for evaluation - nine gold sets unchanged Co-Authored-By: Claude Opus 5.5 --- crates/neuromesh-cli/src/commands/packet.rs | 17 ++++--- .../neuromesh-context/src/activator_seed.rs | 42 +++++++++++++++++ crates/neuromesh-context/src/gold.rs | 45 +++++++++++++++++++ docs/research/contributions-log.md | 17 +++++++ scripts/swebench_localize.py | 3 ++ 5 files changed, 115 insertions(+), 9 deletions(-) diff --git a/crates/neuromesh-cli/src/commands/packet.rs b/crates/neuromesh-cli/src/commands/packet.rs index 2001b67..83847d5 100644 --- a/crates/neuromesh-cli/src/commands/packet.rs +++ b/crates/neuromesh-cli/src/commands/packet.rs @@ -41,6 +41,8 @@ struct PacketJsonOut { /// Repository-relative paths, best first (highest activation of any /// node in the file): what an evaluation needs for hit@k. selected_paths: Vec, + /// The whole-question file ranking (BM25F), top 10, for evaluation. + ranked_paths: Vec, identifiers: Vec, seeds_missed: Vec, seed_resolution: Option, @@ -90,15 +92,7 @@ pub fn execute(args: &[String]) -> Result<()> { let view = activator.activate_tiered(&graph, &signature, OptimizationMode::Balanced); let latency_ms = started.elapsed().as_millis() as u64; - let mut best: std::collections::HashMap = std::collections::HashMap::new(); - for n in &view.active_nodes { - let p = n.node.file_path.to_string_lossy().replace('\\', "/"); - let e = best.entry(p).or_insert(f32::MIN); - *e = e.max(n.activation_score); - } - let mut selected_paths: Vec<(String, f32)> = best.into_iter().collect(); - selected_paths.sort_by(|a, b| b.1.total_cmp(&a.1).then_with(|| a.0.cmp(&b.0))); - let selected_paths: Vec = selected_paths.into_iter().map(|(p, _)| p).collect(); + let selected_paths = neuromesh_context::gold::packet_file_order(&graph, &prompt, &view); let mut files: Vec = packet_file_names(&view).into_iter().collect(); files.sort(); let reduction = if workspace_tokens > 0 { @@ -128,6 +122,11 @@ pub fn execute(args: &[String]) -> Result<()> { selected_files_count: files.len(), selected_files: files.clone(), selected_paths, + ranked_paths: graph + .file_rank(&prompt, 10) + .into_iter() + .map(|r| r.path.to_string_lossy().replace('\\', "/")) + .collect(), identifiers: signature.identifiers.clone(), seeds_missed, seed_resolution: view.seed_resolution_telemetry.clone(), diff --git a/crates/neuromesh-context/src/activator_seed.rs b/crates/neuromesh-context/src/activator_seed.rs index 968305f..08eb1b0 100644 --- a/crates/neuromesh-context/src/activator_seed.rs +++ b/crates/neuromesh-context/src/activator_seed.rs @@ -1150,6 +1150,47 @@ fn fuse_with_embeddings( out } +/// Words from which a prompt reads as a report (an issue with a traceback, +/// a pasted snippet) rather than a question. +const LONG_REPORT_WORDS: usize = 60; + +/// A long report names many identifiers — frames of a traceback, names in a +/// pasted snippet, the reporter's own code — and each one seeds where it +/// resolves, so the packet fills with whatever the report happened to +/// mention. Read as a whole, the same text points at the file it is about +/// (SWE-bench Lite: the whole-question ranking alone puts the edited file +/// first 36% of the time against 12% for the anchor-seeded packet). Its best +/// file, and the runner-up within 70%, join the anchors; nothing is pruned. +fn push_long_report_seeds( + graph: &NeuralProjectGraph, + prompt: &str, + config: &SeedResolutionConfig, + names_low: bool, + sink: &mut SeedSink<'_, '_, '_>, +) { + if prompt.split_whitespace().count() < LONG_REPORT_WORDS { + return; + } + let ranked: Vec = graph + .file_rank(prompt, 50) + .into_iter() + .filter(|r| names_low || !crate::selector::is_noise_path(&r.path)) + .collect(); + let Some(top) = ranked.first().map(|r| r.score) else { + return; + }; + for (pos, r) in ranked + .iter() + .take(2) + .filter(|r| r.score >= top * 0.7) + .enumerate() + { + let energy = signal_weight(config, SignalKind::PathHint, pos + 1); + let rel = r.path.to_string_lossy().replace('\\', "/"); + sink.push(graph, prompt, rel, energy, "body"); + } +} + pub(crate) fn push_body_word_seeds( graph: &NeuralProjectGraph, prompt: &str, @@ -1175,6 +1216,7 @@ pub(crate) fn push_body_word_seeds( }) }); if anchored { + push_long_report_seeds(graph, prompt, config, names_low, sink); return; } if push_ranked_file_seeds(graph, prompt, config, names_low, sink) { diff --git a/crates/neuromesh-context/src/gold.rs b/crates/neuromesh-context/src/gold.rs index 7005e26..05c8c3b 100644 --- a/crates/neuromesh-context/src/gold.rs +++ b/crates/neuromesh-context/src/gold.rs @@ -1504,3 +1504,48 @@ forbidden_files = ["src/directive/clipboard.js", "src/views/profile/UserCard.vue ); } } + +/// Packet files, best first. By default the highest activation of any node +/// in the file orders them. For a long report (an issue with a traceback), +/// that order follows whichever identifiers the report happened to name, so +/// it is fused (reciprocal rank, k = 60) with the whole-question file +/// ranking, which reads the report as one text. +pub fn packet_file_order( + graph: &neuromesh_graph::NeuralProjectGraph, + prompt: &str, + view: &ContextView, +) -> Vec { + let mut best: std::collections::HashMap = std::collections::HashMap::new(); + for n in &view.active_nodes { + let p = n.node.file_path.to_string_lossy().replace('\\', "/"); + let e = best.entry(p).or_insert(f32::MIN); + *e = e.max(n.activation_score); + } + let mut by_activation: Vec<(String, f32)> = best.into_iter().collect(); + by_activation.sort_by(|a, b| b.1.total_cmp(&a.1).then_with(|| a.0.cmp(&b.0))); + let order: Vec = by_activation.into_iter().map(|(p, _)| p).collect(); + if prompt.split_whitespace().count() < 60 { + return order; + } + let ranked: std::collections::HashMap = graph + .file_rank(prompt, 400) + .into_iter() + .enumerate() + .map(|(i, r)| (r.path.to_string_lossy().replace('\\', "/"), i)) + .collect(); + const K: f32 = 60.0; + const LEX_WEIGHT: f32 = 2.0; + let mut fused: Vec<(String, f32)> = order + .iter() + .enumerate() + .map(|(i, p)| { + let lex = ranked + .get(p) + .map(|r| LEX_WEIGHT / (K + *r as f32 + 1.0)) + .unwrap_or(0.0); + (p.clone(), 1.0 / (K + i as f32 + 1.0) + lex) + }) + .collect(); + fused.sort_by(|a, b| b.1.total_cmp(&a.1).then_with(|| a.0.cmp(&b.0))); + fused.into_iter().map(|(p, _)| p).collect() +} diff --git a/docs/research/contributions-log.md b/docs/research/contributions-log.md index b759231..e63a631 100644 --- a/docs/research/contributions-log.md +++ b/docs/research/contributions-log.md @@ -70,6 +70,7 @@ plain-language ones — at 97–99% fewer tokens than the workspace. | Lexical-relevance gate on packet fill | recall 0.679 → 0.643, precision flat | extra files are seeds, not fills | | Precision tuning on dev sets (twice) | gains did not transfer to holdouts | overfitting to dev; why C1 exists | | Aider RepoMap as a question→file retriever | R@5 0.04–0.48 | built as a whole-repo map, not a retriever | +| Cross-encoder rerank (jina-reranker-v1-turbo-en) on top of jina-code fusion | ripgrep 0.667/0.347 → 0.375/0.215; click 0.958/0.439 → 0.917/0.550; self 0.786/0.429 → 0.857/0.500 | English-text reranker; precision up, recall collapses on the true holdout. Code-trained v2 reranker queued | ## 5. What a top-venue paper still needs (honest gap list) @@ -101,3 +102,19 @@ methodology) once items 1–5 are done. - Cross-encoder reranking (jina-reranker-v1-turbo, the Continue/Cody pattern) on top of C5, aimed at precision on plain-language packets (F92). - SWE-bench Lite file localisation run (item 5.1) — the first step toward a top-venue paper. + +## 8. SWE-bench Lite file localisation (in progress, 2026-10-01) + +First 61 instances (flask, requests, seaborn, xarray, pylint, sphinx, astropy partial), v1.1.0+ lexical +engine, issue text as the query, gold = the one file the reference patch edits: + +| method | hit@1 | hit@3 | hit@5 | whole packet | +|---|---|---|---|---| +| ours (packet, best-first) | 0.180 | 0.328 | 0.410 | 0.492 (4.5 files, ~9.4k tokens) | +| plain BM25 over the checkout | 0.295 | 0.525 | 0.639 | — | + +Finding: on long issue reports (tracebacks, code snippets) the identifier-seeding pipeline scatters +across many anchors and fills the packet with the wrong files (including test fixtures); reading the +whole text (BM25) does better. This is the first standard-benchmark result and it is a weakness — +the next engine work targets it (whole-question ranking for long reports, traceback file paths as +anchors). Published reference points to beat: Agentless and LocAgent file-level Acc@k (LLM-based). diff --git a/scripts/swebench_localize.py b/scripts/swebench_localize.py index 911f73e..5da643a 100644 --- a/scripts/swebench_localize.py +++ b/scripts/swebench_localize.py @@ -101,6 +101,9 @@ def main(): ours_hit=hit(files, gold), **{f"ours_hit@{k}": hit(files, gold, k) for k in (1, 3, 5)}, ) + ranked = pkt.get("ranked_paths") or [] + if ranked: + rec.update(**{f"bm25f_hit@{k}": hit(ranked, gold, k) for k in (1, 3, 5, 10)}) bm = Bm25(dest, repo_files(dest)).rank(row["problem_statement"]) rec.update(**{f"bm25_hit@{k}": hit(bm, gold, k) for k in (1, 3, 5, 10)}) except Exception as e: # recorded, not fatal: one bad checkout must not stop the run From cd5f69d8e90bc30eca2d448dff2c8de33a798a5f Mon Sep 17 00:00:00 2001 From: ParsaVictor <1.parsa.karkooti@gmail.com> Date: Thu, 1 Oct 2026 19:56:58 +0330 Subject: [PATCH 2/5] gold: packet_file_order before the test module (clippy); harness --shard/--only for parallel runs Co-Authored-By: Claude Opus 5.5 --- crates/neuromesh-context/src/gold.rs | 90 ++++++++++++++-------------- scripts/swebench_localize.py | 8 +++ 2 files changed, 53 insertions(+), 45 deletions(-) diff --git a/crates/neuromesh-context/src/gold.rs b/crates/neuromesh-context/src/gold.rs index 05c8c3b..3bdd97c 100644 --- a/crates/neuromesh-context/src/gold.rs +++ b/crates/neuromesh-context/src/gold.rs @@ -1000,6 +1000,51 @@ pub fn workspace_gold_path() -> Option { None } +/// Packet files, best first. By default the highest activation of any node +/// in the file orders them. For a long report (an issue with a traceback), +/// that order follows whichever identifiers the report happened to name, so +/// it is fused (reciprocal rank, k = 60) with the whole-question file +/// ranking, which reads the report as one text. +pub fn packet_file_order( + graph: &neuromesh_graph::NeuralProjectGraph, + prompt: &str, + view: &ContextView, +) -> Vec { + let mut best: std::collections::HashMap = std::collections::HashMap::new(); + for n in &view.active_nodes { + let p = n.node.file_path.to_string_lossy().replace('\\', "/"); + let e = best.entry(p).or_insert(f32::MIN); + *e = e.max(n.activation_score); + } + let mut by_activation: Vec<(String, f32)> = best.into_iter().collect(); + by_activation.sort_by(|a, b| b.1.total_cmp(&a.1).then_with(|| a.0.cmp(&b.0))); + let order: Vec = by_activation.into_iter().map(|(p, _)| p).collect(); + if prompt.split_whitespace().count() < 60 { + return order; + } + let ranked: std::collections::HashMap = graph + .file_rank(prompt, 400) + .into_iter() + .enumerate() + .map(|(i, r)| (r.path.to_string_lossy().replace('\\', "/"), i)) + .collect(); + const K: f32 = 60.0; + const LEX_WEIGHT: f32 = 2.0; + let mut fused: Vec<(String, f32)> = order + .iter() + .enumerate() + .map(|(i, p)| { + let lex = ranked + .get(p) + .map(|r| LEX_WEIGHT / (K + *r as f32 + 1.0)) + .unwrap_or(0.0); + (p.clone(), 1.0 / (K + i as f32 + 1.0) + lex) + }) + .collect(); + fused.sort_by(|a, b| b.1.total_cmp(&a.1).then_with(|| a.0.cmp(&b.0))); + fused.into_iter().map(|(p, _)| p).collect() +} + #[cfg(test)] mod tests { use super::*; @@ -1504,48 +1549,3 @@ forbidden_files = ["src/directive/clipboard.js", "src/views/profile/UserCard.vue ); } } - -/// Packet files, best first. By default the highest activation of any node -/// in the file orders them. For a long report (an issue with a traceback), -/// that order follows whichever identifiers the report happened to name, so -/// it is fused (reciprocal rank, k = 60) with the whole-question file -/// ranking, which reads the report as one text. -pub fn packet_file_order( - graph: &neuromesh_graph::NeuralProjectGraph, - prompt: &str, - view: &ContextView, -) -> Vec { - let mut best: std::collections::HashMap = std::collections::HashMap::new(); - for n in &view.active_nodes { - let p = n.node.file_path.to_string_lossy().replace('\\', "/"); - let e = best.entry(p).or_insert(f32::MIN); - *e = e.max(n.activation_score); - } - let mut by_activation: Vec<(String, f32)> = best.into_iter().collect(); - by_activation.sort_by(|a, b| b.1.total_cmp(&a.1).then_with(|| a.0.cmp(&b.0))); - let order: Vec = by_activation.into_iter().map(|(p, _)| p).collect(); - if prompt.split_whitespace().count() < 60 { - return order; - } - let ranked: std::collections::HashMap = graph - .file_rank(prompt, 400) - .into_iter() - .enumerate() - .map(|(i, r)| (r.path.to_string_lossy().replace('\\', "/"), i)) - .collect(); - const K: f32 = 60.0; - const LEX_WEIGHT: f32 = 2.0; - let mut fused: Vec<(String, f32)> = order - .iter() - .enumerate() - .map(|(i, p)| { - let lex = ranked - .get(p) - .map(|r| LEX_WEIGHT / (K + *r as f32 + 1.0)) - .unwrap_or(0.0); - (p.clone(), 1.0 / (K + i as f32 + 1.0) + lex) - }) - .collect(); - fused.sort_by(|a, b| b.1.total_cmp(&a.1).then_with(|| a.0.cmp(&b.0))); - fused.into_iter().map(|(p, _)| p).collect() -} diff --git a/scripts/swebench_localize.py b/scripts/swebench_localize.py index 5da643a..7aa15c7 100644 --- a/scripts/swebench_localize.py +++ b/scripts/swebench_localize.py @@ -70,16 +70,24 @@ def main(): ap.add_argument("--out", default="results.jsonl") ap.add_argument("--limit", type=int, default=0) ap.add_argument("--repo", default="") + ap.add_argument("--only", default="", help="comma-separated repos to keep") + ap.add_argument("--shard", default="", help="i/n: this process takes every n-th instance starting at i") ap.add_argument("--work", default=os.path.join(tempfile.gettempdir(), "swe-wt")) args = ap.parse_args() rows = json.load(open(args.data, encoding="utf-8")) + if args.only: + keep = set(args.only.split(",")) + rows = [r for r in rows if r["repo"] in keep] if args.repo: rows = [r for r in rows if r["repo"] == args.repo] done = set() if os.path.exists(args.out): done = {json.loads(l)["instance_id"] for l in open(args.out, encoding="utf-8") if l.strip()} todo = [r for r in rows if r["instance_id"] not in done] + if args.shard: + i, n = (int(x) for x in args.shard.split("/")) + todo = todo[i::n] if args.limit: todo = todo[: args.limit] os.makedirs(args.work, exist_ok=True) From 58dd3de17cb5852db2a11346f231aa2774b031d9 Mon Sep 17 00:00:00 2001 From: ParsaVictor <1.parsa.karkooti@gmail.com> Date: Thu, 1 Oct 2026 21:22:58 +0330 Subject: [PATCH 3/5] escalation: never build the embedding file tier inside a question With an embedding model on disk and the fast engine, a question that escalated to L3 built the whole workspace's file-tier sidecar synchronously before answering: django (3.5k files) held one SWE-bench issue for more than 10 minutes. The build now starts once on a background thread and the question is answered without it; a later question uses the sidecar. Same issue: >600 s -> 171 s, of which the query is ~19 s and the rest the cold index under 4 parallel jobs. NM_TIMING=1 also times the escalation stages. Co-Authored-By: Claude Opus 5.5 --- .../src/retrieval/escalate.rs | 49 +++++++++++++++---- 1 file changed, 40 insertions(+), 9 deletions(-) diff --git a/crates/neuromesh-context/src/retrieval/escalate.rs b/crates/neuromesh-context/src/retrieval/escalate.rs index 3f52768..ce12587 100644 --- a/crates/neuromesh-context/src/retrieval/escalate.rs +++ b/crates/neuromesh-context/src/retrieval/escalate.rs @@ -92,7 +92,9 @@ pub fn run_incremental( ); levels_attempted.push(RetrievalTier::L1.as_str().into()); + let t_e = Instant::now(); let mut est = estimator.estimate(&view, signature); + neuromesh_graph::timing("escalate: L1 estimate", t_e); let mut final_tier = RetrievalTier::L1; let l1_budget = budget.for_tier(RetrievalTier::L1); @@ -115,10 +117,16 @@ pub fn run_incremental( } // L2: pattern expand + 2 hops — critical gaps or low embedding confidence - if should_escalate_to_l2(&est, &view, activator, graph, signature, &embedding_config) { + let t_e = Instant::now(); + let go_l2 = should_escalate_to_l2(&est, &view, activator, graph, signature, &embedding_config); + neuromesh_graph::timing("escalate: should_escalate_to_l2", t_e); + if go_l2 { let l2_start = Instant::now(); let seed_ids = activator.seed_node_ids(&view); + let t_e = Instant::now(); let pattern_files = pattern_expand(graph, &seed_ids, plan.intent); + neuromesh_graph::timing("escalate: pattern_expand", t_e); + let t_l2 = Instant::now(); sig.engine_override = Some(RetrievalTier::L2.seed_engine( configured_engine, retrieval_engine, @@ -136,6 +144,7 @@ pub fn run_incremental( &plan, Some(view), ); + neuromesh_graph::timing("escalate: L2 activate", t_l2); latency_ms.insert( RetrievalTier::L2.as_str().into(), l2_start.elapsed().as_millis() as u64, @@ -172,14 +181,16 @@ pub fn run_incremental( #[cfg(feature = "embeddings")] if retrieval_engine == RetrievalEngine::Fast && !l3_sidecar_loaded { if let Some(workspace) = graph.workspace_root() { - let l3_emb = fast_l3_embedding_config(retrieval_engine, &embedding_config); - if let Err(e) = - neuromesh_graph::ensure_file_tier_sidecar(graph, &workspace, &l3_emb) - { - tracing::warn!("fast L3 sidecar build failed: {e}"); - } else { - l3_sidecar_loaded = graph.embedding_index().is_loaded(); - } + // Never inside the question: building the file tier embeds + // every file of the workspace (django: 3.5k files, many + // minutes on a laptop CPU) and the answer waited for it. + // Built once in the background; a later question uses it. + spawn_l3_sidecar_build( + graph.clone(), + workspace, + fast_l3_embedding_config(retrieval_engine, &embedding_config), + ); + l3_sidecar_loaded = graph.embedding_index().is_loaded(); } } #[cfg(not(feature = "embeddings"))] @@ -312,6 +323,26 @@ fn needs_embedding_escalation( low_embedding_confidence(graph, prompt, embedding_config, &seed_ids) } +/// Builds the file-tier sidecar on a background thread, once per process. +#[cfg(feature = "embeddings")] +fn spawn_l3_sidecar_build( + graph: NeuralProjectGraph, + workspace: std::path::PathBuf, + config: EmbeddingConfig, +) { + static STARTED: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false); + if STARTED.swap(true, std::sync::atomic::Ordering::AcqRel) { + return; + } + let _ = std::thread::Builder::new() + .name("neuromesh-l3-sidecar".into()) + .spawn(move || { + if let Err(e) = neuromesh_graph::ensure_file_tier_sidecar(&graph, &workspace, &config) { + tracing::warn!("fast L3 sidecar build failed: {e}"); + } + }); +} + fn fast_l3_embedding_config( retrieval_engine: RetrievalEngine, base: &EmbeddingConfig, From 4ce7a5717f9bf9a838d762265df28a8bb3a271fb Mon Sep 17 00:00:00 2001 From: ParsaVictor <1.parsa.karkooti@gmail.com> Date: Thu, 1 Oct 2026 21:55:28 +0330 Subject: [PATCH 4/5] file_rank: a prompt cannot buy unbounded work (first 32 KB, 256 terms) Issue mode ranked the whole 2 MB hostile prompt of the stage-4 security gate: 57 s against the 30 s bound. Same cap as the seed pipeline. Co-Authored-By: Claude Opus 5.5 --- crates/neuromesh-graph/src/graph.rs | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/crates/neuromesh-graph/src/graph.rs b/crates/neuromesh-graph/src/graph.rs index 0647076..8ed8c63 100644 --- a/crates/neuromesh-graph/src/graph.rs +++ b/crates/neuromesh-graph/src/graph.rs @@ -744,7 +744,14 @@ impl NeuralProjectGraph { /// Files ranked by field-weighted BM25 (path, defined names, body) over /// the stemmed words of `prompt` — see [`crate::file_rank`]. pub fn file_rank(&self, prompt: &str, limit: usize) -> Vec { - let terms = crate::file_rank::weighted_query_terms(prompt); + // A prompt cannot buy unbounded work: the first 32 KB (the seed + // pipeline's own cap), and at most 256 distinct terms. + let mut end = prompt.len().min(32 * 1024); + while !prompt.is_char_boundary(end) { + end -= 1; + } + let mut terms = crate::file_rank::weighted_query_terms(&prompt[..end]); + terms.truncate(256); if terms.is_empty() { return Vec::new(); } From e1d149cca871fe1734bff5edbfda2dcbc7d815e5 Mon Sep 17 00:00:00 2001 From: ParsaVictor <1.parsa.karkooti@gmail.com> Date: Thu, 1 Oct 2026 21:56:09 +0330 Subject: [PATCH 5/5] handoff: session 16 final (open PR #131, SWE-bench test split in progress, next steps) Co-Authored-By: Claude Opus 5.5 --- .../handoff-2026-10-01-session16.fa.md | 67 +++++++++++-------- 1 file changed, 39 insertions(+), 28 deletions(-) diff --git a/docs/planning/handoff-2026-10-01-session16.fa.md b/docs/planning/handoff-2026-10-01-session16.fa.md index 3c2dd8a..795e204 100644 --- a/docs/planning/handoff-2026-10-01-session16.fa.md +++ b/docs/planning/handoff-2026-10-01-session16.fa.md @@ -1,44 +1,55 @@ -# Handoff — session 16 (۲۰۲۶-۰۹-۳۰ تا ۲۰۲۶-۱۰-۰۱) +# Handoff — session 16 (۲۰۲۶-۰۹-۳۰ تا ۲۰۲۶-۱۰-۰۱) — نسخه‌ی نهایی -ورودی: گزارش یوسف (۳ از ۷ روی ریپوی خودمان، ۲۶ ثانیه، باگ stdout). نقشه: `11-roadmap-2026-09-30-concept-queries.fa.md`. +ورودی: گزارش یوسف (۳ از ۷ روی ریپوی خودمان، ۲۶ ثانیه، باگ stdout). نقشه: `11-roadmap-2026-09-30-concept-queries.fa.md`. لاگ پژوهشی (برای مقاله): `docs/research/contributions-log.md`. -## چه چیزی merge شد +## merge شده | PR | محتوا | |---|---| -| #126 | بنر داشبورد از stdout به stderr؛ تستی که باینری واقعی را اجرا می‌کند | -| #127 (v1.1.0) | رتبه‌بند BM25F کل‌سؤال (`neuromesh-graph/src/file_rank.rs`)، `comment_index`، thesaurus در `thesaurus.txt`، F89 (کلمه‌ی انگلیسی anchor نیست)، اصلاح cold start، `NM_TIMING=1` | -| #128 | سؤال ساده فقط به رتبه‌بند اعتماد می‌کند؛ رفع panic در `skeleton.rs`؛ `scripts/compare_baselines.py` | -| #129 | جدول مقایسه با BM25 و Aider در `docs/measured.md`؛ holdout دوم (click)؛ folds تکراری؛ حدس‌های سرور «missing» نیستند | +| #126 | بنر stdout → stderr؛ تست باینری واقعی | +| #127 = v1.1.0 | رتبه‌بند BM25F کل‌سؤال، comment_index، thesaurus، F89، اصلاح cold start، NM_TIMING | +| #128 | سؤال ساده فقط به رتبه‌بند اعتماد می‌کند؛ panic در skeleton | +| #129 | جدول BM25/Aider، holdout click، folds تکراری، حدس‌ها «missing» نیستند | +| #130 | مدل اختیاری **jina-code v2** + fusion (فقط برای سؤال ساده؛ MiniLM هرگز)؛ `install embed jina-code`؛ `selected_paths`؛ harness SWE-bench | -انتشار: tag `v1.1.0` با سه باینری. MCP دسکتاپ پارسا روی v1.1.0 است (backup: `neuromesh-v1.0.0-backup.exe`). +## باز — PR #131 (`phase-n/issue-mode`) -## اعداد (صادقانه) +۱. **حالت issue**: پرامپت ≥۶۰ کلمه رتبه‌ی BM25F را هم seed می‌کند و ترتیب packet با RRF (activation + BM25F×2) — `gold::packet_file_order`. SWE-bench Lite **dev split** (flask/requests/seaborn/xarray/pylint، ۲۴ issue): hit@1 0.125→0.333، @3 0.250→0.583، @5 0.417→0.667 (BM25: 0.250/0.625/0.750). +۲. **باگ escalation (مهم)**: L3 با موتور fast وقتی مدل embedding روی دیسک بود، کل sidecar را **داخل سؤال** می‌ساخت (django >۱۰ دقیقه). حالا یک‌بار در پس‌زمینه. همان issue: >600s → 171s. +۳. **cap روی file_rank** (۳۲KB، ۲۵۶ واژه) — gate امنیتی stage4 (57s > 30s) را درست کرد؛ محلی پاس شد (7.5s). +۴. `--shard/--only` در harness. CI آخرین commit (4ce7a57) هنوز دیده نشده — **اول این را چک کن.** + +## در حال اجرا (ممکن است تمام شده باشد) + +SWE-bench Lite **test split** (astropy, sphinx, pytest, sklearn, matplotlib, sympy, django — ۲۷۶ issue) با باینری اصلاح‌شده، ۴ shard: +`C:\1\1_پروژه\5_neuromesh\swebench\results-new-{0..3}.jsonl` (resumable؛ همان دستور را دوباره اجرا کن تا فقط باقی‌مانده‌ها اجرا شوند). سرعت ~۱ issue/دقیقه کل (ایندکس سرد django در هر issue ۱–۳ دقیقه). نتایج قبلی آلوده به MiniLM بودند → `old-runs/`، استفاده نکن. dev split هم باید با `nm-fix` دوباره اجرا شود (اعداد بالا با باینری قبل از اصلاح escalation گرفته شده). + +## اعداد فعلی (صادقانه) | ست | v1.0.0 | الان | |---|---|---| -| concept (ریپوی خودمان، dev) | 0.286 | 0.679 / 0.404 | -| concept-holdout (ripgrep، یک بار) | 0.042 | 0.500 / 0.156 | -| concept-holdout2 (click، تازه، یک بار) | — | 0.958 / 0.342 | -| ۹ ست قدیمی | — | بدون افت؛ web 0.608→0.643 | -| query اول روی کلون تازه | 3.6 s | 2.6 s؛ بعدی‌ها 0.1–0.4 s | - -مقایسه با BM25: روی سؤال‌هایی که اسم کد دارند جلوییم (recall 1.0 با دقت 0.59–0.70)، جز holdout-lang که BM25@1 دقت 0.93 دارد. روی ripgrep (سؤال ساده) مساوی، روی click جلوتر. +| concept (dev، ریپوی خودمان) | 0.286 | 0.679/0.404؛ با jina 0.786/0.429 | +| ripgrep holdout | 0.042 | 0.500/0.156؛ با jina **0.667/0.347** (BM25@3 0.500/0.222) | +| click holdout | — | 0.958/0.342؛ با jina 0.958/0.439 | +| ۹ ست قدیمی | — | بدون افت | +| SWE-bench Lite dev | 0.125@1 | 0.333@1، 0.583@3 | -## رد شده با عدد (تکرار نکنید) +## رد شده با عدد (تکرار نکن) -- seed کردن تعریف‌های فایل بزرگ به‌جای کل فایل: ripgrep 0.500→0.375 -- ترکیب با MiniLM: ripgrep 0.375، self 0.607 (MiniLM کد نمی‌فهمد؛ F79 دوباره تایید شد) -- gate روی fillهای packet مفهومی: فایل‌های اضافه خود seedها هستند، نه fill +seed تعریف‌های فایل بزرگ؛ fusion با MiniLM؛ gate روی fill؛ reranker jina-v1-turbo (ripgrep 0.667→0.375)؛ Aider RepoMap به‌عنوان retriever. -## باز +## گام‌های بعدی به ترتیب -- J2: مدل `jina_code_v2` (مخصوص کد) به‌عنوان گزینه اضافه شد (`NEUROMESH_EMBED_MODEL=jina_code_v2`، فایل‌ها در `%LOCALAPPDATA%/neuromesh/models/jina-code-v2`). نتیجه‌ی آزمایش fusion در بخش پایین. -- F92 دقت packet سؤال مفهومی (0.15–0.40)؛ F93 واژه‌هایی که در کد نیست؛ F94 فایل‌های hub. -- holdout-lang: fillهای connector (utility:16–41) دقت را پایین می‌آورند — تیون precision دو بار بسته شده؛ فقط با holdout جدید. +1. CI PR #131 → merge. نتایج test split را جمع کن (اسکریپت خلاصه در handoff بالا) و با BM25 مقایسه کن؛ در `contributions-log.md` §8 ثبت کن. +2. **سرعت ایندکس**: `manifest+links` روی django ۶۰–۹۷ ثانیه (روی ریپوی خودمان ۲.۴s) — رشد فوق‌خطی در `finalize_links`؛ با `NM_TIMING=1` پروفایل کن. این همان تأخیر cold start یوسف است. +3. **BM25F با tf واقعی** در فیلد body (الان باینری) — BM25 ساده در @3/@5 به همین دلیل جلوتر است. +4. reranker v2 (کد-آموزش‌دیده، `%LOCALAPPDATA%\neuromesh\models\jina-reranker-v2`، اگر دانلود کامل شده) روی concept-holdout. +5. انتشار 1.2.0 (jina-code، issue mode، اصلاح escalation). +6. مسیر مقاله: §5 در `contributions-log.md` (SWE-bench کامل + Agentless/LocAgent، task success per token، ablation، ≥۵۰ سؤال در هر holdout با دو نفر). -## روش کار سریع +## تله‌ها -- `bash scripts/benchmark-fast.sh` — ۹ ست در ۱.۵ تا ۵ دقیقه (release، موازی) -- `bash scripts/benchmark-fast.sh concept concept-holdout concept-holdout2` -- یک build در هر لحظه؛ دیسک یک بار پر شد. +- یک build در هر لحظه؛ دیسک یک بار پر شد (الان ~۶ GB آزاد). `sed` با `\\` و `a\`/`i\` در MSYS خراب می‌کند — Edit یا head/tail. +- `benchmark-fast.sh` ۹ ست در ۱.۵–۵ دقیقه. private harness مسیر مطلق می‌خواهد. +- MCP دسکتاپ پارسا روی v1.1.0 (backup v1.0.0 کنارش). +- مدل‌ها در `%LOCALAPPDATA%\neuromesh\models\` (jina-code-v2، jina-reranker-v1-turbo)؛ MiniLM در `repo/crates/neuromesh-embed/models` (gitignored) — وجود آن روی این ماشین رفتار L3 را تغییر می‌دهد.