Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 13 additions & 0 deletions crates/neuromesh-cli/src/commands/packet.rs
Original file line number Diff line number Diff line change
Expand Up @@ -38,6 +38,9 @@ struct PacketJsonOut {
reduction_vs_workspace_pct: f32,
selected_files: Vec<String>,
selected_files_count: usize,
/// Repository-relative paths, best first (highest activation of any
/// node in the file): what an evaluation needs for hit@k.
selected_paths: Vec<String>,
identifiers: Vec<String>,
seeds_missed: Vec<String>,
seed_resolution: Option<neuromesh_core::SeedResolutionTelemetry>,
Expand Down Expand Up @@ -87,6 +90,15 @@ pub fn execute(args: &[String]) -> Result<()> {
let view = activator.activate_tiered(&graph, &signature, OptimizationMode::Balanced);
let latency_ms = started.elapsed().as_millis() as u64;

let mut best: std::collections::HashMap<String, f32> = std::collections::HashMap::new();
for n in &view.active_nodes {
let p = n.node.file_path.to_string_lossy().replace('\\', "/");
let e = best.entry(p).or_insert(f32::MIN);
*e = e.max(n.activation_score);
}
let mut selected_paths: Vec<(String, f32)> = best.into_iter().collect();
selected_paths.sort_by(|a, b| b.1.total_cmp(&a.1).then_with(|| a.0.cmp(&b.0)));
let selected_paths: Vec<String> = selected_paths.into_iter().map(|(p, _)| p).collect();
let mut files: Vec<String> = packet_file_names(&view).into_iter().collect();
files.sort();
let reduction = if workspace_tokens > 0 {
Expand Down Expand Up @@ -115,6 +127,7 @@ pub fn execute(args: &[String]) -> Result<()> {
reduction_vs_workspace_pct: reduction,
selected_files_count: files.len(),
selected_files: files.clone(),
selected_paths,
identifiers: signature.identifiers.clone(),
seeds_missed,
seed_resolution: view.seed_resolution_telemetry.clone(),
Expand Down
83 changes: 82 additions & 1 deletion crates/neuromesh-context/src/activator_seed.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1011,7 +1011,14 @@ fn push_ranked_file_seeds(
if best.matched < 2 {
return false;
}
let top = best.score;
// With an embedding sidecar loaded, meaning joins the lexical ranking
// (the lexical gate above still decides whether to seed at all).
#[cfg(feature = "embeddings")]
let ranked: Vec<neuromesh_graph::RankedFile> = fuse_with_embeddings(graph, prompt, ranked)
.into_iter()
.filter(|r| names_low || !crate::selector::is_noise_path(&r.path))
.collect();
let top = ranked[0].score;
let picks: Vec<&neuromesh_graph::RankedFile> = ranked
.iter()
.take(3)
Expand Down Expand Up @@ -1069,6 +1076,80 @@ fn push_ranked_file_seeds(
true
}

/// Lexical file ranking fused with the embedding file tier, when a code-aware
/// model's sidecar is loaded: a word the code never spells (a question about
/// "shrinking pictures" for `resize_image`) reaches a file through
/// meaning; a word it does spell keeps
/// its lexical weight. Convex combination of min-max normalised scores
/// (lexical 0.6, dense 0.4) — it keeps score ratios meaningful for the
/// runner-up rule, which reciprocal-rank fusion flattens (Bruch et al. 2023,
/// "An Analysis of Fusion Functions for Hybrid Retrieval").
#[cfg(feature = "embeddings")]
fn fuse_with_embeddings(
graph: &NeuralProjectGraph,
prompt: &str,
lexical: Vec<neuromesh_graph::RankedFile>,
) -> Vec<neuromesh_graph::RankedFile> {
const DEPTH: usize = 50;
const W_LEX: f32 = 0.6;
const W_DENSE: f32 = 0.4;
let index = graph.embedding_index();
if !index.is_loaded() || lexical.is_empty() {
return lexical;
}
// A loaded sidecar means embeddings are in use, whatever the config
// default says (the query must be embedded with the same model).
let mut cfg = neuromesh_core::Config::load().embeddings;
cfg.enabled = true;
// Only a code-aware model earns a vote: MiniLM, a general paraphrase
// model, lowered every plain-language set it was fused into.
if cfg.model != neuromesh_core::EmbeddingModelId::JinaCodeV2 {
return lexical;
}
let Ok(query) = neuromesh_embed::embed_query_cached(&cfg, prompt) else {
return lexical;
};
let dense = index.file_ann_search(&query, DEPTH, 0.0);
if dense.is_empty() {
return lexical;
}
let lex_max = lexical[0].score.max(f32::EPSILON);
let (d_min, d_max) = dense.iter().fold((f32::MAX, f32::MIN), |(lo, hi), (_, s)| {
(lo.min(*s), hi.max(*s))
});
let d_span = (d_max - d_min).max(f32::EPSILON);
let mut fused: std::collections::HashMap<NodeId, f32> = std::collections::HashMap::new();
for r in lexical.iter().take(DEPTH) {
*fused.entry(r.id.clone()).or_insert(0.0) += W_LEX * r.score / lex_max;
}
for (id, s) in &dense {
*fused.entry(id.clone()).or_insert(0.0) += W_DENSE * (s - d_min) / d_span;
}
let mut out: Vec<neuromesh_graph::RankedFile> = fused
.into_iter()
.filter_map(|(id, score)| {
let lex = lexical.iter().find(|r| r.id == id);
let path = match lex {
Some(r) => r.path.clone(),
None => graph.get_node(&id)?.file_path,
};
Some(neuromesh_graph::RankedFile {
id,
path,
score,
matched: lex.map(|r| r.matched).unwrap_or(0),
})
})
.collect();
out.sort_by(|a, b| {
b.score
.partial_cmp(&a.score)
.unwrap_or(std::cmp::Ordering::Equal)
.then_with(|| a.path.cmp(&b.path))
});
out
}

pub(crate) fn push_body_word_seeds(
graph: &NeuralProjectGraph,
prompt: &str,
Expand Down
5 changes: 5 additions & 0 deletions crates/neuromesh-core/src/config.rs
Original file line number Diff line number Diff line change
Expand Up @@ -306,6 +306,11 @@ impl Config {
}
if let Ok(raw) = std::env::var("NEUROMESH_EMBED_MODEL") {
if let Some(model) = crate::EmbeddingModelId::parse(&raw) {
if model != self.embeddings.model {
// Another model's width: 768-d vectors cut to MiniLM's 384
// are not what the model was trained to produce.
self.embeddings.matryoshka_dim = model.default_matryoshka_dim();
}
self.embeddings.model = model;
}
}
Expand Down
6 changes: 6 additions & 0 deletions crates/neuromesh-core/src/embedding_config.rs
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,9 @@ pub enum EmbeddingModelId {
Gemma300mQ4,
#[default]
MiniLmMultilingualQ,
/// jinaai/jina-embeddings-v2-base-code: trained on code and its docs
/// (768-d). Downloaded by fastembed on first use (~640 MB).
JinaCodeV2,
}

impl EmbeddingModelId {
Expand All @@ -14,6 +17,7 @@ impl EmbeddingModelId {
"gemma300m_q4" | "gemma300m-q4" | "embeddinggemma300m_q4" | "gemma" => {
Some(Self::Gemma300mQ4)
}
"jina_code_v2" | "jina-code-v2" | "jina_code" | "jina-code" => Some(Self::JinaCodeV2),
"minilm_multilingual_q" | "minilm-multilingual-q" | "minilm" => {
Some(Self::MiniLmMultilingualQ)
}
Expand All @@ -25,13 +29,15 @@ impl EmbeddingModelId {
match self {
Self::Gemma300mQ4 => "gemma300m_q4",
Self::MiniLmMultilingualQ => "minilm_multilingual_q",
Self::JinaCodeV2 => "jina_code_v2",
}
}

pub fn default_matryoshka_dim(self) -> usize {
match self {
Self::Gemma300mQ4 => 256,
Self::MiniLmMultilingualQ => 384,
Self::JinaCodeV2 => 768,
}
}
}
Expand Down
34 changes: 34 additions & 0 deletions crates/neuromesh-embed/src/embedder.rs
Original file line number Diff line number Diff line change
Expand Up @@ -83,6 +83,7 @@ pub fn format_query_for_model(model: EmbeddingModelId, prompt: &str) -> String {
match model {
EmbeddingModelId::Gemma300mQ4 => format_query_gemma(prompt),
EmbeddingModelId::MiniLmMultilingualQ => format_query_minilm(prompt),
EmbeddingModelId::JinaCodeV2 => prompt.trim().to_string(),
}
}

Expand All @@ -95,6 +96,7 @@ pub fn format_document_for_model(
) -> String {
match model {
EmbeddingModelId::Gemma300mQ4 => format_document_gemma(title, kind, signature, doc),
EmbeddingModelId::JinaCodeV2 => format_document_minilm(title, kind, signature, doc),
EmbeddingModelId::MiniLmMultilingualQ => {
format_document_minilm(title, kind, signature, doc)
}
Expand All @@ -116,6 +118,38 @@ fn try_init_text_embedding(
try_load_bundled_minilm(config.model, config.intra_threads)
.map_err(|e| EmbedderError::Init(format!("{e}. {}", install_hint())))
}
EmbeddingModelId::JinaCodeV2 => {
// Loaded from plain files, not fastembed's hf-hub cache (its
// symlinks fail on Windows without developer mode):
// <models>/jina-code-v2/{model_quantized.onnx, tokenizer*.json, …}
// from huggingface.co/jinaai/jina-embeddings-v2-base-code.
let dir = crate::model_install::default_models_root().join("jina-code-v2");
let read = |name: &str| {
std::fs::read(dir.join(name)).map_err(|e| {
EmbedderError::Init(format!(
"jina_code_v2: {} missing ({e})",
dir.join(name).display()
))
})
};
let user_model = fastembed::UserDefinedEmbeddingModel::new(
read("model_quantized.onnx")?,
fastembed::TokenizerFiles {
tokenizer_file: read("tokenizer.json")?,
config_file: read("config.json")?,
special_tokens_map_file: read("special_tokens_map.json")?,
tokenizer_config_file: read("tokenizer_config.json")?,
},
)
.with_pooling(fastembed::Pooling::Mean)
.with_quantization(fastembed::QuantizationMode::Dynamic);
let mut opts = fastembed::InitOptionsUserDefined::default();
if let Some(n) = config.intra_threads {
opts = opts.with_intra_threads(n);
}
TextEmbedding::try_new_from_user_defined(user_model, opts)
.map_err(|e| EmbedderError::Init(format!("jina_code_v2: {e}")))
}
EmbeddingModelId::Gemma300mQ4 => Err(EmbedderError::Init(format!(
"gemma300m_q4 is not installable yet; use MiniLM ({}). {}",
EmbeddingModelId::MiniLmMultilingualQ.as_str(),
Expand Down
96 changes: 71 additions & 25 deletions crates/neuromesh-embed/src/model_install.rs
Original file line number Diff line number Diff line change
Expand Up @@ -33,7 +33,26 @@ pub const MINILM_MULTILINGUAL_Q: EmbedModelSpec = EmbedModelSpec {
],
};

pub static CATALOG: &[EmbedModelSpec] = &[MINILM_MULTILINGUAL_Q];
/// jinaai/jina-embeddings-v2-base-code, int8 ONNX (768-d): trained on code
/// and its documentation. Fused with the lexical ranking for plain-language
/// questions it lifts the ripgrep holdout from 0.500 to 0.667 recall
/// (`docs/measured.md`); MiniLM, a general paraphrase model, lowered it.
pub const JINA_CODE_V2: EmbedModelSpec = EmbedModelSpec {
id: "jina-code-v2",
dir_name: "jina-code-v2",
label: "Jina embeddings v2 base code, int8 (768-dim, code-aware, ~160 MB)",
aliases: &["jina", "jina-code", "jina_code", "jina_code_v2"],
hf_base: "https://huggingface.co/jinaai/jina-embeddings-v2-base-code/resolve/main",
files: &[
"onnx/model_quantized.onnx",
TOKENIZER_NAME,
"config.json",
"special_tokens_map.json",
"tokenizer_config.json",
],
};

pub static CATALOG: &[EmbedModelSpec] = &[MINILM_MULTILINGUAL_Q, JINA_CODE_V2];

#[derive(Debug, Clone, Copy, Default)]
pub struct InstallOptions {
Expand Down Expand Up @@ -86,19 +105,26 @@ pub fn model_install_dir(spec: &EmbedModelSpec) -> PathBuf {
}

pub fn is_model_installed(spec: &EmbedModelSpec) -> bool {
model_dir_ready(&model_install_dir(spec))
spec_ready(spec, &model_install_dir(spec))
}

fn model_dir_ready(dir: &Path) -> bool {
dir.join(ONNX_NAME).is_file() && dir.join(TOKENIZER_NAME).is_file()
/// Local file name of a spec entry: `onnx/model_quantized.onnx` is saved
/// as `model_quantized.onnx` next to the tokenizer.
fn local_name(name: &str) -> &str {
name.rsplit('/').next().unwrap_or(name)
}

/// Every file of the spec is on disk.
fn spec_ready(spec: &EmbedModelSpec, dir: &Path) -> bool {
spec.files.iter().all(|f| dir.join(local_name(f)).is_file())
}

pub fn list_installed() -> Vec<(EmbedModelSpec, PathBuf)> {
CATALOG
.iter()
.filter_map(|spec| {
let dir = model_install_dir(spec);
if model_dir_ready(&dir) {
if spec_ready(spec, &dir) {
Some((*spec, dir))
} else {
None
Expand Down Expand Up @@ -147,9 +173,9 @@ fn install_model_inner(
let dest = model_install_dir(spec);
std::fs::create_dir_all(&dest)?;

if !opts.force && model_dir_ready(&dest) {
if !opts.force && spec_ready(spec, &dest) {
if !opts.quiet {
eprintln!("MiniLM already installed at {}", dest.display());
eprintln!("{} already installed at {}", spec.id, dest.display());
}
return Ok(dest);
}
Expand All @@ -160,43 +186,63 @@ fn install_model_inner(
.map_err(|e| ModelInstallError::Download(e.to_string()))?;

for name in spec.files {
let out = dest.join(name);
let local = local_name(name);
let out = dest.join(local);
if !opts.force && out.is_file() {
if !opts.quiet {
eprintln!(" skip {name} (exists)");
eprintln!(" skip {local} (exists)");
}
continue;
}
let url = format!("{}/{}", spec.hf_base, name);
if !opts.quiet {
eprintln!(" fetch {name}…");
eprintln!(" fetch {local}…");
}
let response = client
.get(&url)
.send()
.map_err(|e| ModelInstallError::Download(format!("{name}: {e}")))?;
if !response.status().is_success() {
return Err(ModelInstallError::Download(format!(
"{name}: HTTP {}",
response.status()
)));
// A slow or flaky link drops large files mid-body: retry the whole
// file a few times before giving up (the .download temp never
// becomes the real file unless it arrived complete).
let mut last_err = String::new();
let mut bytes = None;
for attempt in 1..=3 {
let result = client
.get(&url)
.send()
.map_err(|e| e.to_string())
.and_then(|r| {
if r.status().is_success() {
r.bytes().map_err(|e| e.to_string())
} else {
Err(format!("HTTP {}", r.status()))
}
});
match result {
Ok(b) => {
bytes = Some(b);
break;
}
Err(e) => {
if !opts.quiet {
eprintln!(" {local}: attempt {attempt} failed ({e})");
}
last_err = e;
}
}
}
let bytes = response
.bytes()
.map_err(|e| ModelInstallError::Download(format!("{name}: {e}")))?;
let tmp = dest.join(format!(".{name}.download"));
let bytes =
bytes.ok_or_else(|| ModelInstallError::Download(format!("{local}: {last_err}")))?;
let tmp = dest.join(format!(".{local}.download"));
std::fs::write(&tmp, &bytes)?;
std::fs::rename(&tmp, &out)?;
}

if !model_dir_ready(&dest) {
if !spec_ready(spec, &dest) {
return Err(ModelInstallError::Download(
"install incomplete after download".into(),
));
}

if !opts.quiet {
eprintln!("MiniLM installed at {}", dest.display());
eprintln!("{} installed at {}", spec.id, dest.display());
}
Ok(dest)
}
Expand Down
Loading
Loading