Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
17 changes: 8 additions & 9 deletions crates/neuromesh-cli/src/commands/packet.rs
Original file line number Diff line number Diff line change
Expand Up @@ -41,6 +41,8 @@ struct PacketJsonOut {
/// Repository-relative paths, best first (highest activation of any
/// node in the file): what an evaluation needs for hit@k.
selected_paths: Vec<String>,
/// The whole-question file ranking (BM25F), top 10, for evaluation.
ranked_paths: Vec<String>,
identifiers: Vec<String>,
seeds_missed: Vec<String>,
seed_resolution: Option<neuromesh_core::SeedResolutionTelemetry>,
Expand Down Expand Up @@ -90,15 +92,7 @@ pub fn execute(args: &[String]) -> Result<()> {
let view = activator.activate_tiered(&graph, &signature, OptimizationMode::Balanced);
let latency_ms = started.elapsed().as_millis() as u64;

let mut best: std::collections::HashMap<String, f32> = std::collections::HashMap::new();
for n in &view.active_nodes {
let p = n.node.file_path.to_string_lossy().replace('\\', "/");
let e = best.entry(p).or_insert(f32::MIN);
*e = e.max(n.activation_score);
}
let mut selected_paths: Vec<(String, f32)> = best.into_iter().collect();
selected_paths.sort_by(|a, b| b.1.total_cmp(&a.1).then_with(|| a.0.cmp(&b.0)));
let selected_paths: Vec<String> = selected_paths.into_iter().map(|(p, _)| p).collect();
let selected_paths = neuromesh_context::gold::packet_file_order(&graph, &prompt, &view);
let mut files: Vec<String> = packet_file_names(&view).into_iter().collect();
files.sort();
let reduction = if workspace_tokens > 0 {
Expand Down Expand Up @@ -128,6 +122,11 @@ pub fn execute(args: &[String]) -> Result<()> {
selected_files_count: files.len(),
selected_files: files.clone(),
selected_paths,
ranked_paths: graph
.file_rank(&prompt, 10)
.into_iter()
.map(|r| r.path.to_string_lossy().replace('\\', "/"))
.collect(),
identifiers: signature.identifiers.clone(),
seeds_missed,
seed_resolution: view.seed_resolution_telemetry.clone(),
Expand Down
42 changes: 42 additions & 0 deletions crates/neuromesh-context/src/activator_seed.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1150,6 +1150,47 @@ fn fuse_with_embeddings(
out
}

/// Words from which a prompt reads as a report (an issue with a traceback,
/// a pasted snippet) rather than a question.
const LONG_REPORT_WORDS: usize = 60;

/// A long report names many identifiers — frames of a traceback, names in a
/// pasted snippet, the reporter's own code — and each one seeds where it
/// resolves, so the packet fills with whatever the report happened to
/// mention. Read as a whole, the same text points at the file it is about
/// (SWE-bench Lite: the whole-question ranking alone puts the edited file
/// first 36% of the time against 12% for the anchor-seeded packet). Its best
/// file, and the runner-up within 70%, join the anchors; nothing is pruned.
fn push_long_report_seeds(
graph: &NeuralProjectGraph,
prompt: &str,
config: &SeedResolutionConfig,
names_low: bool,
sink: &mut SeedSink<'_, '_, '_>,
) {
if prompt.split_whitespace().count() < LONG_REPORT_WORDS {
return;
}
let ranked: Vec<neuromesh_graph::RankedFile> = graph
.file_rank(prompt, 50)
.into_iter()
.filter(|r| names_low || !crate::selector::is_noise_path(&r.path))
.collect();
let Some(top) = ranked.first().map(|r| r.score) else {
return;
};
for (pos, r) in ranked
.iter()
.take(2)
.filter(|r| r.score >= top * 0.7)
.enumerate()
{
let energy = signal_weight(config, SignalKind::PathHint, pos + 1);
let rel = r.path.to_string_lossy().replace('\\', "/");
sink.push(graph, prompt, rel, energy, "body");
}
}

pub(crate) fn push_body_word_seeds(
graph: &NeuralProjectGraph,
prompt: &str,
Expand All @@ -1175,6 +1216,7 @@ pub(crate) fn push_body_word_seeds(
})
});
if anchored {
push_long_report_seeds(graph, prompt, config, names_low, sink);
return;
}
if push_ranked_file_seeds(graph, prompt, config, names_low, sink) {
Expand Down
45 changes: 45 additions & 0 deletions crates/neuromesh-context/src/gold.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1000,6 +1000,51 @@ pub fn workspace_gold_path() -> Option<std::path::PathBuf> {
None
}

/// Packet files, best first. By default the highest activation of any node
/// in the file orders them. For a long report (an issue with a traceback),
/// that order follows whichever identifiers the report happened to name, so
/// it is fused (reciprocal rank, k = 60) with the whole-question file
/// ranking, which reads the report as one text.
pub fn packet_file_order(
graph: &neuromesh_graph::NeuralProjectGraph,
prompt: &str,
view: &ContextView,
) -> Vec<String> {
let mut best: std::collections::HashMap<String, f32> = std::collections::HashMap::new();
for n in &view.active_nodes {
let p = n.node.file_path.to_string_lossy().replace('\\', "/");
let e = best.entry(p).or_insert(f32::MIN);
*e = e.max(n.activation_score);
}
let mut by_activation: Vec<(String, f32)> = best.into_iter().collect();
by_activation.sort_by(|a, b| b.1.total_cmp(&a.1).then_with(|| a.0.cmp(&b.0)));
let order: Vec<String> = by_activation.into_iter().map(|(p, _)| p).collect();
if prompt.split_whitespace().count() < 60 {
return order;
}
let ranked: std::collections::HashMap<String, usize> = graph
.file_rank(prompt, 400)
.into_iter()
.enumerate()
.map(|(i, r)| (r.path.to_string_lossy().replace('\\', "/"), i))
.collect();
const K: f32 = 60.0;
const LEX_WEIGHT: f32 = 2.0;
let mut fused: Vec<(String, f32)> = order
.iter()
.enumerate()
.map(|(i, p)| {
let lex = ranked
.get(p)
.map(|r| LEX_WEIGHT / (K + *r as f32 + 1.0))
.unwrap_or(0.0);
(p.clone(), 1.0 / (K + i as f32 + 1.0) + lex)
})
.collect();
fused.sort_by(|a, b| b.1.total_cmp(&a.1).then_with(|| a.0.cmp(&b.0)));
fused.into_iter().map(|(p, _)| p).collect()
}

#[cfg(test)]
mod tests {
use super::*;
Expand Down
49 changes: 40 additions & 9 deletions crates/neuromesh-context/src/retrieval/escalate.rs
Original file line number Diff line number Diff line change
Expand Up @@ -92,7 +92,9 @@ pub fn run_incremental(
);
levels_attempted.push(RetrievalTier::L1.as_str().into());

let t_e = Instant::now();
let mut est = estimator.estimate(&view, signature);
neuromesh_graph::timing("escalate: L1 estimate", t_e);
let mut final_tier = RetrievalTier::L1;

let l1_budget = budget.for_tier(RetrievalTier::L1);
Expand All @@ -115,10 +117,16 @@ pub fn run_incremental(
}

// L2: pattern expand + 2 hops — critical gaps or low embedding confidence
if should_escalate_to_l2(&est, &view, activator, graph, signature, &embedding_config) {
let t_e = Instant::now();
let go_l2 = should_escalate_to_l2(&est, &view, activator, graph, signature, &embedding_config);
neuromesh_graph::timing("escalate: should_escalate_to_l2", t_e);
if go_l2 {
let l2_start = Instant::now();
let seed_ids = activator.seed_node_ids(&view);
let t_e = Instant::now();
let pattern_files = pattern_expand(graph, &seed_ids, plan.intent);
neuromesh_graph::timing("escalate: pattern_expand", t_e);
let t_l2 = Instant::now();
sig.engine_override = Some(RetrievalTier::L2.seed_engine(
configured_engine,
retrieval_engine,
Expand All @@ -136,6 +144,7 @@ pub fn run_incremental(
&plan,
Some(view),
);
neuromesh_graph::timing("escalate: L2 activate", t_l2);
latency_ms.insert(
RetrievalTier::L2.as_str().into(),
l2_start.elapsed().as_millis() as u64,
Expand Down Expand Up @@ -172,14 +181,16 @@ pub fn run_incremental(
#[cfg(feature = "embeddings")]
if retrieval_engine == RetrievalEngine::Fast && !l3_sidecar_loaded {
if let Some(workspace) = graph.workspace_root() {
let l3_emb = fast_l3_embedding_config(retrieval_engine, &embedding_config);
if let Err(e) =
neuromesh_graph::ensure_file_tier_sidecar(graph, &workspace, &l3_emb)
{
tracing::warn!("fast L3 sidecar build failed: {e}");
} else {
l3_sidecar_loaded = graph.embedding_index().is_loaded();
}
// Never inside the question: building the file tier embeds
// every file of the workspace (django: 3.5k files, many
// minutes on a laptop CPU) and the answer waited for it.
// Built once in the background; a later question uses it.
spawn_l3_sidecar_build(
graph.clone(),
workspace,
fast_l3_embedding_config(retrieval_engine, &embedding_config),
);
l3_sidecar_loaded = graph.embedding_index().is_loaded();
}
}
#[cfg(not(feature = "embeddings"))]
Expand Down Expand Up @@ -312,6 +323,26 @@ fn needs_embedding_escalation(
low_embedding_confidence(graph, prompt, embedding_config, &seed_ids)
}

/// Builds the file-tier sidecar on a background thread, once per process.
#[cfg(feature = "embeddings")]
fn spawn_l3_sidecar_build(
graph: NeuralProjectGraph,
workspace: std::path::PathBuf,
config: EmbeddingConfig,
) {
static STARTED: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false);
if STARTED.swap(true, std::sync::atomic::Ordering::AcqRel) {
return;
}
let _ = std::thread::Builder::new()
.name("neuromesh-l3-sidecar".into())
.spawn(move || {
if let Err(e) = neuromesh_graph::ensure_file_tier_sidecar(&graph, &workspace, &config) {
tracing::warn!("fast L3 sidecar build failed: {e}");
}
});
}

fn fast_l3_embedding_config(
retrieval_engine: RetrievalEngine,
base: &EmbeddingConfig,
Expand Down
9 changes: 8 additions & 1 deletion crates/neuromesh-graph/src/graph.rs
Original file line number Diff line number Diff line change
Expand Up @@ -744,7 +744,14 @@ impl NeuralProjectGraph {
/// Files ranked by field-weighted BM25 (path, defined names, body) over
/// the stemmed words of `prompt` — see [`crate::file_rank`].
pub fn file_rank(&self, prompt: &str, limit: usize) -> Vec<crate::file_rank::RankedFile> {
let terms = crate::file_rank::weighted_query_terms(prompt);
// A prompt cannot buy unbounded work: the first 32 KB (the seed
// pipeline's own cap), and at most 256 distinct terms.
let mut end = prompt.len().min(32 * 1024);
while !prompt.is_char_boundary(end) {
end -= 1;
}
let mut terms = crate::file_rank::weighted_query_terms(&prompt[..end]);
terms.truncate(256);
if terms.is_empty() {
return Vec::new();
}
Expand Down
67 changes: 39 additions & 28 deletions docs/planning/handoff-2026-10-01-session16.fa.md
Original file line number Diff line number Diff line change
@@ -1,44 +1,55 @@
# Handoff — session 16 (۲۰۲۶-۰۹-۳۰ تا ۲۰۲۶-۱۰-۰۱)
# Handoff — session 16 (۲۰۲۶-۰۹-۳۰ تا ۲۰۲۶-۱۰-۰۱) — نسخه‌ی نهایی

ورودی: گزارش یوسف (۳ از ۷ روی ریپوی خودمان، ۲۶ ثانیه، باگ stdout). نقشه: `11-roadmap-2026-09-30-concept-queries.fa.md`.
ورودی: گزارش یوسف (۳ از ۷ روی ریپوی خودمان، ۲۶ ثانیه، باگ stdout). نقشه: `11-roadmap-2026-09-30-concept-queries.fa.md`. لاگ پژوهشی (برای مقاله): `docs/research/contributions-log.md`.

## چه چیزی merge شد
## merge شده

| PR | محتوا |
|---|---|
| #126 | بنر داشبورد از stdout به stderr؛ تستی که باینری واقعی را اجرا می‌کند |
| #127 (v1.1.0) | رتبه‌بند BM25F کل‌سؤال (`neuromesh-graph/src/file_rank.rs`)، `comment_index`، thesaurus در `thesaurus.txt`، F89 (کلمه‌ی انگلیسی anchor نیست)، اصلاح cold start، `NM_TIMING=1` |
| #128 | سؤال ساده فقط به رتبه‌بند اعتماد می‌کند؛ رفع panic در `skeleton.rs`؛ `scripts/compare_baselines.py` |
| #129 | جدول مقایسه با BM25 و Aider در `docs/measured.md`؛ holdout دوم (click)؛ folds تکراری؛ حدس‌های سرور «missing» نیستند |
| #126 | بنر stdout → stderr؛ تست باینری واقعی |
| #127 = v1.1.0 | رتبه‌بند BM25F کل‌سؤال، comment_index، thesaurus، F89، اصلاح cold start، NM_TIMING |
| #128 | سؤال ساده فقط به رتبه‌بند اعتماد می‌کند؛ panic در skeleton |
| #129 | جدول BM25/Aider، holdout click، folds تکراری، حدس‌ها «missing» نیستند |
| #130 | مدل اختیاری **jina-code v2** + fusion (فقط برای سؤال ساده؛ MiniLM هرگز)؛ `install embed jina-code`؛ `selected_paths`؛ harness SWE-bench |

انتشار: tag `v1.1.0` با سه باینری. MCP دسکتاپ پارسا روی v1.1.0 است (backup: `neuromesh-v1.0.0-backup.exe`).
## باز — PR #131 (`phase-n/issue-mode`)

## اعداد (صادقانه)
۱. **حالت issue**: پرامپت ≥۶۰ کلمه رتبه‌ی BM25F را هم seed می‌کند و ترتیب packet با RRF (activation + BM25F×2) — `gold::packet_file_order`. SWE-bench Lite **dev split** (flask/requests/seaborn/xarray/pylint، ۲۴ issue): hit@1 0.125→0.333، @3 0.250→0.583، @5 0.417→0.667 (BM25: 0.250/0.625/0.750).
۲. **باگ escalation (مهم)**: L3 با موتور fast وقتی مدل embedding روی دیسک بود، کل sidecar را **داخل سؤال** می‌ساخت (django >۱۰ دقیقه). حالا یک‌بار در پس‌زمینه. همان issue: >600s → 171s.
۳. **cap روی file_rank** (۳۲KB، ۲۵۶ واژه) — gate امنیتی stage4 (57s > 30s) را درست کرد؛ محلی پاس شد (7.5s).
۴. `--shard/--only` در harness. CI آخرین commit (4ce7a57) هنوز دیده نشده — **اول این را چک کن.**

## در حال اجرا (ممکن است تمام شده باشد)

SWE-bench Lite **test split** (astropy, sphinx, pytest, sklearn, matplotlib, sympy, django — ۲۷۶ issue) با باینری اصلاح‌شده، ۴ shard:
`C:\1\1_پروژه\5_neuromesh\swebench\results-new-{0..3}.jsonl` (resumable؛ همان دستور را دوباره اجرا کن تا فقط باقی‌مانده‌ها اجرا شوند). سرعت ~۱ issue/دقیقه کل (ایندکس سرد django در هر issue ۱–۳ دقیقه). نتایج قبلی آلوده به MiniLM بودند → `old-runs/`، استفاده نکن. dev split هم باید با `nm-fix` دوباره اجرا شود (اعداد بالا با باینری قبل از اصلاح escalation گرفته شده).

## اعداد فعلی (صادقانه)

| ست | v1.0.0 | الان |
|---|---|---|
| concept (ریپوی خودمان، dev) | 0.286 | 0.679 / 0.404 |
| concept-holdout (ripgrep، یک بار) | 0.042 | 0.500 / 0.156 |
| concept-holdout2 (click، تازه، یک بار) | — | 0.958 / 0.342 |
| ۹ ست قدیمی | — | بدون افت؛ web 0.608→0.643 |
| query اول روی کلون تازه | 3.6 s | 2.6 s؛ بعدی‌ها 0.1–0.4 s |

مقایسه با BM25: روی سؤال‌هایی که اسم کد دارند جلوییم (recall 1.0 با دقت 0.59–0.70)، جز holdout-lang که BM25@1 دقت 0.93 دارد. روی ripgrep (سؤال ساده) مساوی، روی click جلوتر.
| concept (dev، ریپوی خودمان) | 0.286 | 0.679/0.404؛ با jina 0.786/0.429 |
| ripgrep holdout | 0.042 | 0.500/0.156؛ با jina **0.667/0.347** (BM25@3 0.500/0.222) |
| click holdout | — | 0.958/0.342؛ با jina 0.958/0.439 |
| ۹ ست قدیمی | — | بدون افت |
| SWE-bench Lite dev | 0.125@1 | 0.333@1، 0.583@3 |

## رد شده با عدد (تکرار نکنید)
## رد شده با عدد (تکرار نکن)

- seed کردن تعریف‌های فایل بزرگ به‌جای کل فایل: ripgrep 0.500→0.375
- ترکیب با MiniLM: ripgrep 0.375، self 0.607 (MiniLM کد نمی‌فهمد؛ F79 دوباره تایید شد)
- gate روی fillهای packet مفهومی: فایل‌های اضافه خود seedها هستند، نه fill
seed تعریف‌های فایل بزرگ؛ fusion با MiniLM؛ gate روی fill؛ reranker jina-v1-turbo (ripgrep 0.667→0.375)؛ Aider RepoMap به‌عنوان retriever.

## باز
## گام‌های بعدی به ترتیب

- J2: مدل `jina_code_v2` (مخصوص کد) به‌عنوان گزینه اضافه شد (`NEUROMESH_EMBED_MODEL=jina_code_v2`، فایل‌ها در `%LOCALAPPDATA%/neuromesh/models/jina-code-v2`). نتیجه‌ی آزمایش fusion در بخش پایین.
- F92 دقت packet سؤال مفهومی (0.15–0.40)؛ F93 واژه‌هایی که در کد نیست؛ F94 فایل‌های hub.
- holdout-lang: fillهای connector (utility:16–41) دقت را پایین می‌آورند — تیون precision دو بار بسته شده؛ فقط با holdout جدید.
1. CI PR #131 → merge. نتایج test split را جمع کن (اسکریپت خلاصه در handoff بالا) و با BM25 مقایسه کن؛ در `contributions-log.md` §8 ثبت کن.
2. **سرعت ایندکس**: `manifest+links` روی django ۶۰–۹۷ ثانیه (روی ریپوی خودمان ۲.۴s) — رشد فوق‌خطی در `finalize_links`؛ با `NM_TIMING=1` پروفایل کن. این همان تأخیر cold start یوسف است.
3. **BM25F با tf واقعی** در فیلد body (الان باینری) — BM25 ساده در @3/@5 به همین دلیل جلوتر است.
4. reranker v2 (کد-آموزش‌دیده، `%LOCALAPPDATA%\neuromesh\models\jina-reranker-v2`، اگر دانلود کامل شده) روی concept-holdout.
5. انتشار 1.2.0 (jina-code، issue mode، اصلاح escalation).
6. مسیر مقاله: §5 در `contributions-log.md` (SWE-bench کامل + Agentless/LocAgent، task success per token، ablation، ≥۵۰ سؤال در هر holdout با دو نفر).

## روش کار سریع
## تله‌ها

- `bash scripts/benchmark-fast.sh` — ۹ ست در ۱.۵ تا ۵ دقیقه (release، موازی)
- `bash scripts/benchmark-fast.sh concept concept-holdout concept-holdout2`
- یک build در هر لحظه؛ دیسک یک بار پر شد.
- یک build در هر لحظه؛ دیسک یک بار پر شد (الان ~۶ GB آزاد). `sed` با `\\` و `a\`/`i\` در MSYS خراب می‌کند — Edit یا head/tail.
- `benchmark-fast.sh` ۹ ست در ۱.۵–۵ دقیقه. private harness مسیر مطلق می‌خواهد.
- MCP دسکتاپ پارسا روی v1.1.0 (backup v1.0.0 کنارش).
- مدل‌ها در `%LOCALAPPDATA%\neuromesh\models\` (jina-code-v2، jina-reranker-v1-turbo)؛ MiniLM در `repo/crates/neuromesh-embed/models` (gitignored) — وجود آن روی این ماشین رفتار L3 را تغییر می‌دهد.
Loading
Loading