diff --git a/Cargo.lock b/Cargo.lock index 4051c0f..98c39e9 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -782,7 +782,7 @@ dependencies = [ [[package]] name = "codegraph-harness" -version = "0.20.1" +version = "0.21.0" dependencies = [ "anyhow", "clap", @@ -890,7 +890,7 @@ dependencies = [ [[package]] name = "codegraph-memory" -version = "0.20.1" +version = "0.21.0" dependencies = [ "anyhow", "bincode", @@ -1073,7 +1073,7 @@ dependencies = [ [[package]] name = "codegraph-server" -version = "0.20.1" +version = "0.21.0" dependencies = [ "clap", "codegraph", diff --git a/Cargo.toml b/Cargo.toml index 2c186b3..9b0bcae 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -60,7 +60,7 @@ members = [ ] [workspace.package] -version = "0.20.1" +version = "0.21.0" edition = "2021" license = "Apache-2.0" repository = "https://github.com/codegraph-ai/codegraph" diff --git a/README.md b/README.md index bbcb70b..a402813 100644 --- a/README.md +++ b/README.md @@ -30,7 +30,7 @@ The server indexes the current working directory automatically. Install the VSIX: ```bash -code --install-extension codegraph-0.20.1.vsix +code --install-extension codegraph-0.21.0.vsix ``` One VSIX serves every platform. @@ -85,6 +85,23 @@ one tool and exits without the MCP stdio handshake — ideal for scripting. --- +## Upgrading to 0.21 + +Each project re-embeds once, in the background, the first time 0.21 opens it. +Stored vectors now record the model and settings that built them, and an index from 0.20.1 or earlier records none. +Keyword search keeps working while it runs. + +Restart any `--watch` daemon after upgrading. +A daemon left running keeps writing vectors that 0.21 will not load, and sessions attached to it run without semantic search until it restarts. + +Identifiers whose words run together, like `getUserById`, are now also embedded in word-split form. +This mostly helps natural-language search in camelCase languages such as TypeScript, Java and C#. +Pass `--split-identifiers=false` to keep the old behaviour. + +`--full-body-embedding=false` now takes effect; in 0.20.1 the flag was always on. + +--- + ## Configuration ### MCP Server flags @@ -94,12 +111,19 @@ one tool and exits without the MCP stdio handshake — ideal for scripting. | `--workspace ` | current dir | Directories to index (repeatable for multi-project) | | `--exclude ` | — | Directories to skip (repeatable) | | `--embedding-model ` | `bge-small` | `bge-small` (384d, fast), `jina-code-v2` (768d, 6× slower), `granite-97m` (384d, 32K ctx, ~3× slower), or `static` (model2vec, 256d — ~100× faster indexing, no ONNX; needs a local model dir, see below) | -| `--full-body-embedding` | `true` | Embed full function body (~50 lines) for better semantic search and duplicate detection | +| `--full-body-embedding` | `true` | Embed full function body (~50 lines) for better semantic search and duplicate detection. Takes a value: `--full-body-embedding=false` turns it off | +| `--split-identifiers` | `true` | Also embed the word-split form of identifiers whose words are run together, so `getUserById` embeds as "get user by id" too. Names already separated by `_` or `-` are left alone, since they tokenise into the same words; a leading or trailing delimiter separates nothing, so `_handleClick` is split like `handleClick`. Takes a value: `--split-identifiers=false` embeds raw names, as releases up to 0.20.1 did | | `--max-files ` | 5000 | Maximum files to index | | `--profile ` | `all` | Filter the exposed MCP tool surface to a named subset (see below) | | `--graph-only` | off | Skip embedding generation — build the graph and serve structural tools only. No ONNX model load, 10-50× faster indexing. Semantic search and memory tools unavailable. For CI / one-shot graph queries. | | `--run-tool ` | — | One-shot mode: index, run a single tool, print its result, exit. No MCP handshake. Pair with `--tool-args ''`. | +`--split-identifiers` and `--full-body-embedding` both change the text every symbol is embedded from, and vectors built from different text cannot be ranked against each other. +A project's stored vectors are therefore stamped with the settings that built them - `--embedding-model` included, since models differ in dimension - and are ignored by any process configured differently. +Changing any of the three re-embeds in the background rather than requiring a manual reindex. +Run `--watch` with the same flags as the sessions that read the project, so both sides share one set instead of re-embedding over each other. +For what this means when upgrading from 0.20.1 or earlier, see [Upgrading to 0.21](#upgrading-to-021). + #### `--embedding-model static` — model2vec fast indexing Static (model2vec) embeddings replace the ONNX transformer with a token→vector diff --git a/crates/codegraph-memory/examples/embed_eval.rs b/crates/codegraph-memory/examples/embed_eval.rs index 2215682..cd8b16e 100644 --- a/crates/codegraph-memory/examples/embed_eval.rs +++ b/crates/codegraph-memory/examples/embed_eval.rs @@ -168,7 +168,15 @@ fn metrics(ranks: &[usize]) -> Scores { /// Returns (semantic-only, hybrid 0.4*BM25 + 0.6*cosine) scores. fn evaluate(engine: &VectorEngine, syms: &[Sym]) -> (Scores, Scores) { - let split = std::env::var("CODEGRAPH_SPLIT_IDS").is_ok(); + // Presence is not the question - `CODEGRAPH_SPLIT_IDS=0` must mean off. + // `is_ok()` made setting it to 0 turn the feature ON, which silently turned + // an A/B run into two identical arms. `=1` is the one spelling this harness + // documents, so it is the one spelling it accepts. + // + // Note this forces splitting on every identifier, unlike the engine, which + // applies it only to delimiter-free names: the point of this harness is to + // measure the lever in isolation. + let split = std::env::var("CODEGRAPH_SPLIT_IDS").as_deref() == Ok("1"); let sym_texts: Vec = syms .iter() .map(|s| { diff --git a/crates/codegraph-server/src/ai_query/engine.rs b/crates/codegraph-server/src/ai_query/engine.rs index 1af1ae9..f709506 100644 --- a/crates/codegraph-server/src/ai_query/engine.rs +++ b/crates/codegraph-server/src/ai_query/engine.rs @@ -15,7 +15,7 @@ use super::primitives::{ }; use super::text_index::{TextIndex, TextIndexBuilder}; use crate::domain::node_props; -use codegraph::{CodeGraph, Direction, EdgeType, NodeId, NodeType}; +use codegraph::{CodeGraph, Direction, EdgeType, NamespacedBackend, NodeId, NodeType}; use codegraph_memory::VectorEngine; use std::collections::{HashMap, HashSet, VecDeque}; use std::sync::Arc; @@ -50,8 +50,19 @@ pub struct QueryEngine { /// Prepend split-identifier words to the embed text (helps static /// embedders; off by default to leave the transformer path unchanged). split_identifiers: std::sync::atomic::AtomicBool, + /// Set when the watcher daemon this session attached to holds vectors + /// built with different embed settings. No embed run will follow, so the + /// search status must not say embeddings are building. + daemon_vectors_mismatched: std::sync::atomic::AtomicBool, } +/// Search status when the attached watcher daemon's vectors were built with +/// different embed settings than this session's. +const DAEMON_VECTORS_MISMATCHED_STATUS: &str = "The --watch daemon's stored vectors were built \ + with different embedding settings, so semantic matching is unavailable this session - results \ + are from name/text search only. Restart the daemon with the same --full-body-embedding / \ + --split-identifiers / --embedding-model flags as this session."; + /// Max characters of function body for full-body embedding. /// ~512 tokens ≈ first 40-50 lines of code. const FULL_BODY_MAX_CHARS: usize = 2048; @@ -85,6 +96,289 @@ fn embed_memory_pressured(avail_mb: u64) -> bool { avail_mb > 0 && avail_mb < EMBED_LOW_MEM_MB } +/// Version of the code that builds embedding text, bumped whenever that code +/// changes what it emits. +/// +/// 1 = raw name (through 0.20.1) +/// 2 = delimiter-free identifiers also embedded word-split +/// 3 = leading and trailing `_`/`-` no longer count as word delimiters, so +/// `_handleClick` is split like `handleClick` +const EMBED_TEXT_SCHEMA: u32 = 3; + +/// Identifies the embedding text a set of vectors was built from: the version +/// of the code that builds it, plus the settings that change what it emits. +/// +/// Vectors are only comparable to others built the same way. `getUserById` and +/// `get user by id getUserById` describe the same symbol but land in different +/// places, so ranking one against the other is worse than either scheme alone - +/// and the same holds for a signature-only vector against a full-body one. +/// The model comes first because it decides comparability most bluntly of all: +/// bge-small emits 384 dimensions and jina-code-v2 emits 768, and +/// `cosine_similarity` zips two vectors of different length down to the shorter +/// one while dividing by the longer norm, so mixing them scores every symbol +/// wrong without erroring. +fn embed_text_id(model: &str, full_body: bool, split_identifiers: bool) -> String { + let flag = |on: bool| if on { "on" } else { "off" }; + format!( + "{EMBED_TEXT_SCHEMA}-model={model}-body={}-split={}", + flag(full_body), + flag(split_identifiers) + ) +} + +/// What [`QueryEngine::load_symbol_vectors`] found in the store. +/// +/// The three empty outcomes are not interchangeable. A caller that treats +/// "nothing loaded" as "re-embed the repo" does a whole-corpus ONNX run because +/// a watcher daemon had not reached its first persist yet, or because the store +/// was locked for the moment it took the daemon to write. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum VectorLoad { + /// Vectors built from the same embed text this engine produces. + Loaded(usize), + /// Nothing is stored for this project yet. + Absent, + /// The stored vectors were built from different embed text - another + /// configuration, another model, or 0.20.1, which stamped nothing. + Mismatched, + /// The store could not be read. Says nothing about what it holds. + Unreadable, +} + +impl VectorLoad { + /// Vectors actually loaded; zero for every outcome but [`Self::Loaded`]. + pub fn count(&self) -> usize { + match self { + Self::Loaded(n) => *n, + _ => 0, + } + } +} + +/// Key prefix for a project's persisted symbol vectors. One set per project. +const VECTOR_KEY_PREFIX: &str = "vec:"; + +/// Key holding the [`embed_text_id`] the persisted vectors were built from. +/// +/// It is also the ownership marker. A project belongs to whichever embed text +/// last claimed it, and only a rebuild that is about to write a full set may +/// claim - see [`claim_project`]. +const EMBED_STAMP_KEY: &[u8] = b"embed_text_stamp"; + +/// The node a `vec:` key belongs to, if it is one. +fn vector_key_node(key: &[u8]) -> Option { + std::str::from_utf8(key) + .ok()? + .strip_prefix(VECTOR_KEY_PREFIX)? + .parse::() + .ok() +} + +/// Whether a project's vectors belong to a watcher daemon other than this +/// process. +/// +/// The daemon's own indexing runs before it publishes a heartbeat and matches +/// this pid afterwards, so it is never locked out of its own project; a daemon +/// that died leaves a stale heartbeat that `live_daemon_for` discards, so a +/// crash does not strand the vectors either. +fn owned_by_another_process(daemon: Option<&crate::daemon::DaemonHeartbeat>) -> bool { + daemon.is_some_and(|d| d.pid != std::process::id()) +} + +/// The embed text a project's stored vectors belong to, if any. +fn project_owner(backend: &NamespacedBackend) -> std::result::Result>, String> { + use codegraph::StorageBackend; + + backend + .get(EMBED_STAMP_KEY) + .map_err(|e| format!("Failed to read embed-text stamp: {e}")) +} + +/// Take a project's vector set for `stamp`: drop whatever is stored and stamp +/// the project, in one batch. Returns whether the project was taken. +/// +/// This is the only destructive step in the lifecycle, and the process taking +/// it is about to write a full set. Everything after it - every checkpoint, the +/// final save - only adds to a set this run owns, so a crash mid-rebuild leaves +/// a partial set under its own stamp that the next start loads and the fill +/// pass completes. Losing a marathon first index to an OOM-kill is the failure +/// checkpointing exists for; it only works if the checkpoints are loadable, +/// which means the stamp has to be there from the start. +/// +/// A live watcher daemon owns its workspace's vectors, so no other process may +/// take them - not a session that attached to it, not the reindex tool inside +/// one, not a path added later. The refusal lives here, at the delete, rather +/// than in the callers that would otherwise each have to remember it. +fn claim_project( + backend: &mut NamespacedBackend, + slug: &str, + stamp: &str, +) -> std::result::Result { + use codegraph::storage::BatchOperation; + use codegraph::StorageBackend; + + if owned_by_another_process(crate::daemon::live_daemon_for(slug).as_ref()) { + return Ok(false); + } + + let mut ops: Vec = backend + .scan_prefix_keys(VECTOR_KEY_PREFIX.as_bytes()) + .map_err(|e| format!("Failed to scan stored vector keys: {e}"))? + .into_iter() + .map(|key| BatchOperation::Delete { key }) + .collect(); + ops.push(BatchOperation::Put { + key: EMBED_STAMP_KEY.to_vec(), + value: stamp.as_bytes().to_vec(), + }); + + backend + .write_batch(ops) + .map_err(|e| format!("Failed to claim symbol vector set: {e}"))?; + Ok(true) +} + +/// Vectors per write batch. +/// +/// Each batch is copied into the backend's own buffer, so writing a large set +/// as one batch peaks at several times the set's size. The checkpoint that runs +/// when [`embed_memory_pressured`] fires is exactly the write that must not do +/// that: it exists to survive low memory, not to triple the footprint at the +/// moment memory is already short. +const STORE_BATCH_VECTORS: usize = 2048; + +/// Store `vecs` for a project, returning whether anything was written. +/// +/// Writes only into a set this stamp already owns. A project nobody has claimed +/// is not adopted, and a set belonging to different embed text is left alone: +/// stamping is [`claim_project`]'s job, reserved for a process about to write +/// the whole set. Without that rule an unstamped set - everything a pre-0.21 +/// binary wrote, including a `--watch` daemon still running across an upgrade - +/// would be adopted as current by the first complete save to come along, and +/// then served as if it matched. +/// +/// Only ever adds. [`claim_project`] is the sole path that deletes, so it is +/// the sole place ownership has to be enforced, and a save can never destroy +/// work another writer - a checkpointing rebuild, a watcher daemon, a session +/// indexing a narrower set of paths - has already stored. Vectors for nodes +/// that no longer exist are dropped by the next claim, which clears the set +/// wholesale before the rebuild that follows refills it. +/// +/// Written in chunks rather than one batch. An interrupted write leaves a +/// subset of the set under its own stamp, which is the same state a checkpoint +/// leaves and which the fill pass completes on the next start. +fn store_vectors( + backend: &mut NamespacedBackend, + vecs: &HashMap>, + stamp: &str, +) -> std::result::Result { + use codegraph::storage::BatchOperation; + use codegraph::StorageBackend; + + if project_owner(backend)?.as_deref() != Some(stamp.as_bytes()) { + return Ok(false); + } + + let mut batch: Vec = Vec::with_capacity(STORE_BATCH_VECTORS.min(vecs.len())); + for (&node_id, vec) in vecs.iter() { + batch.push(BatchOperation::Put { + key: format!("{VECTOR_KEY_PREFIX}{node_id}").into_bytes(), + value: vec.iter().flat_map(|f| f.to_le_bytes()).collect(), + }); + if batch.len() == STORE_BATCH_VECTORS { + backend + .write_batch(std::mem::take(&mut batch)) + .map_err(|e| format!("Failed to store symbol vectors: {e}"))?; + batch.reserve(STORE_BATCH_VECTORS); + } + } + if !batch.is_empty() { + backend + .write_batch(batch) + .map_err(|e| format!("Failed to store symbol vectors: {e}"))?; + } + Ok(true) +} + +/// Read a project's persisted vectors, but only when they were built from +/// `stamp`. +/// +/// A set built from other text is left exactly where it is rather than deleted: +/// the process that wrote it may still be using it, and the next complete save +/// replaces it wholesale anyway. `vec:` keys carrying no stamp at all are the +/// 0.20.1 layout and take the same path. +/// +/// The error tells the caller *why* it got nothing, which is not one question +/// but three - see [`VectorLoad`]. +fn read_vectors( + backend: &NamespacedBackend, + stamp: &str, +) -> std::result::Result)>, VectorLoad> { + use codegraph::StorageBackend; + + let stored_stamp = backend.get(EMBED_STAMP_KEY).map_err(|e| { + tracing::warn!("[QueryEngine] Failed to read embed-text stamp: {e}"); + VectorLoad::Unreadable + })?; + + if stored_stamp.as_deref() != Some(stamp.as_bytes()) { + let stored = backend + .scan_prefix_keys(VECTOR_KEY_PREFIX.as_bytes()) + .map_err(|e| { + tracing::warn!("[QueryEngine] Failed to scan vector keys: {e}"); + VectorLoad::Unreadable + })?; + return Err(if stored.is_empty() { + VectorLoad::Absent + } else { + VectorLoad::Mismatched + }); + } + + let entries = backend + .scan_prefix(VECTOR_KEY_PREFIX.as_bytes()) + .map_err(|e| { + tracing::warn!("[QueryEngine] Failed to scan vectors: {e}"); + VectorLoad::Unreadable + })?; + + Ok(entries + .into_iter() + .filter_map(|(key, value)| { + let node_id = vector_key_node(&key)?; + if value.len() % 4 != 0 { + return None; + } + let vec = value + .chunks_exact(4) + .map(|c| f32::from_le_bytes([c[0], c[1], c[2], c[3]])) + .collect(); + Some((node_id, vec)) + }) + .collect()) +} + +/// Whether splitting an identifier into words tells the embedder anything it +/// cannot already see. +/// +/// A `_` or `-` *between* words is a delimiter every tokenizer already splits +/// on, so prepending the split form of `get_user_by_id` just repeats the name. +/// One at the start or the end separates nothing: `_handleClick` reaches the +/// embedder as the same single rare token `handleClick` does, and the +/// private/member conventions that produce it (`_privateField`, `type_`) belong +/// to exactly the camelCase languages splitting was measured to help. +/// +/// Measured on the doc->symbol retrieval eval (pure semantic, R@1, BGE-small): +/// camelCase +79% (compressor, 377 symbols), PascalCase +23% (this repo, 200), +/// snake_case +1.3% (this repo, 723) and -2.5% (SystemVerilog, 696). The Rust +/// snake-vs-Pascal pair is the controlled comparison - same repo, same docs, +/// only the casing differs - so splitting is applied where it pays and skipped +/// where it is a wash or a small loss. +fn needs_word_split(name: &str) -> bool { + let between_words = name.trim_matches(|c| c == '_' || c == '-'); + !between_words.contains('_') && !between_words.contains('-') +} + /// Split an identifier into camelCase/snake_case words (deduped, lowercased), /// reusing the BM25 tokenizer: `getUserById` -> "get user by id". fn split_identifier_words(name: &str) -> String { @@ -109,10 +403,23 @@ impl QueryEngine { symbol_vectors: Arc::new(RwLock::new(HashMap::new())), symbol_texts: Arc::new(RwLock::new(HashMap::new())), full_body_embedding: std::sync::atomic::AtomicBool::new(true), - split_identifiers: std::sync::atomic::AtomicBool::new(false), + // On by default: run-together identifiers are the common case in + // TypeScript, Java, C# and Go, and are exactly where the embedder + // cannot recover the words on its own. needs_word_split() keeps it + // off for snake_case, where it does not help. + split_identifiers: std::sync::atomic::AtomicBool::new(true), + daemon_vectors_mismatched: std::sync::atomic::AtomicBool::new(false), } } + /// Record that the attached watcher daemon's vectors were built with + /// different embed settings, so they will not load and nothing in this + /// session will build replacements. + pub fn set_daemon_vectors_mismatched(&self) { + self.daemon_vectors_mismatched + .store(true, std::sync::atomic::Ordering::Relaxed); + } + /// Enable or disable full-body embedding mode. pub fn set_full_body_embedding(&self, enabled: bool) { self.full_body_embedding @@ -120,12 +427,25 @@ impl QueryEngine { } /// Enable or disable prepending split-identifier words to the embed text. - /// Helps static (lookup-table) embedders; off by default. + /// + /// On by default, and applied only to identifiers whose words are not + /// already separated by `_`/`-` - see `needs_word_split`. Turning it off + /// reverts to embedding the raw name, which releases up to 0.20.1 did. pub fn set_split_identifiers(&self, enabled: bool) { self.split_identifiers .store(enabled, std::sync::atomic::Ordering::Relaxed); } + fn split_identifiers_enabled(&self) -> bool { + self.split_identifiers + .load(std::sync::atomic::Ordering::Relaxed) + } + + fn full_body_enabled(&self) -> bool { + self.full_body_embedding + .load(std::sync::atomic::Ordering::Relaxed) + } + /// Build the embedding text for a symbol node. /// In signature mode: "name: signature — docstring" /// In full-body mode: "name: signature\n" @@ -155,7 +475,7 @@ impl QueryEngine { // Static (lookup-table) embedders can't subword-recover `authenticateUser` // from one rare token; the split words ("authenticate user") are their // strongest signal, front-loaded so they survive truncation. - let base = if split_identifiers { + let base = if split_identifiers && needs_word_split(name) { let words = split_identifier_words(name); if words.is_empty() || words == name.to_lowercase() { base @@ -289,10 +609,10 @@ impl QueryEngine { /// Like [`Self::build_symbol_vectors`], with crash resilience for marathon /// first-index runs (telemetry: linux OOM-kills ~25 min into embedding, /// losing the whole run because the only save happened at the very end): - /// - with a `slug`, persists accumulated vectors every - /// [`EMBED_CHECKPOINT_SYMBOLS`] symbols (status `partial:N`), so a kill - /// loses minutes and the next start resumes via - /// [`Self::embed_missing_symbols`]; + /// - with a `slug`, claims the project's vector set up front and then + /// persists accumulated vectors every [`EMBED_CHECKPOINT_SYMBOLS`] + /// symbols, so a kill loses minutes and the next start loads the partial + /// set and finishes it via [`Self::embed_missing_symbols`]; /// - polls available RAM every [`EMBED_MEM_CHECK_CHUNKS`] chunks; under /// [`EMBED_LOW_MEM_MB`] it halves the ONNX batch and forces a /// checkpoint — degrade to slow instead of being OOM-killed. @@ -306,6 +626,12 @@ impl QueryEngine { }; let start = Instant::now(); + // Read once: every vector this run writes, and the stamp its + // checkpoints are matched against, must describe the same settings. + let full_body = self.full_body_enabled(); + let split_identifiers = self.split_identifiers_enabled(); + let stamp = embed_text_id(engine.model_name(), full_body, split_identifiers); + let graph = self.graph.read().await; // Collect symbol texts for embedding. Texts are stored ONCE here and @@ -334,16 +660,8 @@ impl QueryEngine { } // Build embedding text - let embed_text = Self::build_embed_text( - node, - node_id, - name, - self.full_body_embedding - .load(std::sync::atomic::Ordering::Relaxed), - self.split_identifiers - .load(std::sync::atomic::Ordering::Relaxed), - &graph, - ); + let embed_text = + Self::build_embed_text(node, node_id, name, full_body, split_identifiers, &graph); node_ids.push(node_id); texts.push(embed_text); @@ -355,6 +673,32 @@ impl QueryEngine { return; } + // Claiming replaces the stored set, so it happens only once this run + // is known to have something to write. It used to run first, before a + // single symbol had been collected: a rebuild that found nothing to + // embed - an index whose paths were renamed away, say - cleared the + // project's vectors and then returned having written none. + let slug = match slug { + Some(slug) => match Self::claim_vector_set(slug, &stamp) { + Ok(true) => Some(slug), + Ok(false) => { + tracing::info!( + "[QueryEngine] A watcher daemon owns '{slug}' - embedding in memory only, \ + leaving its vectors alone." + ); + None + } + Err(e) => { + tracing::warn!( + "[QueryEngine] Could not claim the vector set for '{slug}': {e}. \ + Embedding in memory only; an interrupted run will restart." + ); + None + } + }, + None => None, + }; + let model_name = engine.model_name(); tracing::info!( "[QueryEngine] Embedding {} symbols ({})...", @@ -366,10 +710,17 @@ impl QueryEngine { // ONNX Runtime allocates intermediate tensors proportional to batch size × token count. // Signature mode (~20 tokens/item): batch 64 is fine. // Full-body mode (~500 tokens/item): reduce batch to 16 to stay within memory. - let is_full_body = self - .full_body_embedding - .load(std::sync::atomic::Ordering::Relaxed); - let mut chunk_size: usize = if is_full_body { 16 } else { 64 }; + let mut chunk_size: usize = if full_body { 16 } else { 64 }; + + // Accumulated here and published in one swap at the end, never + // incrementally: `symbol_vectors` is also the live search set, and a + // partial one is worse than none. `compute_semantic_scores` succeeds as + // soon as it is non-empty, so every symbol not embedded yet scores 0 on + // the semantic half and is ranked below whatever the first chunks + // happened to cover - for the whole run, on a big repo tens of minutes. + // Leaving it empty keeps search on pure BM25 and keeps + // `are_embeddings_ready` false, which is what tells the client + // embeddings are still building. let mut symbol_vecs = HashMap::with_capacity(texts.len()); let total = texts.len(); @@ -421,13 +772,17 @@ impl QueryEngine { if since_checkpoint >= EMBED_CHECKPOINT_SYMBOLS || (pressured && since_checkpoint > 0) { - match Self::save_vectors_map(slug, &symbol_vecs, false) { - Ok(()) => tracing::info!( + match Self::save_vectors_map(slug, &symbol_vecs, false, &stamp) { + Ok(true) => tracing::info!( "[QueryEngine] Checkpointed {} vectors ({}/{} embedded)", symbol_vecs.len(), pos, total ), + Ok(false) => tracing::debug!( + "[QueryEngine] Checkpoint skipped: the stored set belongs to another \ + embed text and only a complete save may replace it" + ), Err(e) => tracing::warn!("[QueryEngine] Vector checkpoint failed: {e}"), } since_checkpoint = 0; @@ -462,6 +817,11 @@ impl QueryEngine { None => return, }; + // Read once: every vector this run writes, and the stamp its + // checkpoints are matched against, must describe the same settings. + let full_body = self.full_body_enabled(); + let split_identifiers = self.split_identifiers_enabled(); + let stamp = embed_text_id(engine.model_name(), full_body, split_identifiers); let graph = self.graph.read().await; let existing_vecs = self.symbol_vectors.read().await; @@ -487,16 +847,8 @@ impl QueryEngine { if name.is_empty() || name == "arrow_function" || name == "anonymous" { continue; } - let embed_text = Self::build_embed_text( - node, - node_id, - name, - self.full_body_embedding - .load(std::sync::atomic::Ordering::Relaxed), - self.split_identifiers - .load(std::sync::atomic::Ordering::Relaxed), - &graph, - ); + let embed_text = + Self::build_embed_text(node, node_id, name, full_body, split_identifiers, &graph); node_ids.push(node_id); texts.push(embed_text); } @@ -519,10 +871,7 @@ impl QueryEngine { // files, but on the post-crash resume path "missing" can be most of // the corpus, making the single batch its own OOM. Same backpressure // and (with a slug) the same crash-resilience checkpoints. - let is_full_body = self - .full_body_embedding - .load(std::sync::atomic::Ordering::Relaxed); - let mut chunk_size: usize = if is_full_body { 16 } else { 64 }; + let mut chunk_size: usize = if full_body { 16 } else { 64 }; let total = texts.len(); let mut pos = 0usize; let mut chunks_done = 0usize; @@ -570,13 +919,17 @@ impl QueryEngine { || (pressured && since_checkpoint > 0) { let vecs = self.symbol_vectors.read().await; - match Self::save_vectors_map(slug, &vecs, false) { - Ok(()) => tracing::info!( + match Self::save_vectors_map(slug, &vecs, false, &stamp) { + Ok(true) => tracing::info!( "[QueryEngine] Checkpointed {} vectors ({}/{} resumed)", vecs.len(), pos, total ), + Ok(false) => tracing::debug!( + "[QueryEngine] Checkpoint skipped: the stored set belongs to another \ + embed text and only a complete save may replace it" + ), Err(e) => tracing::warn!("[QueryEngine] Vector checkpoint failed: {e}"), } since_checkpoint = 0; @@ -593,28 +946,77 @@ impl QueryEngine { /// Persist symbol vectors to RocksDB alongside the graph. /// - /// Each vector is stored as key `vec:{node_id}` → binary `[f32]` (little-endian). - /// Uses the namespaced backend so vectors are scoped per project. + /// Each vector is stored as key `vec:{node_id}` → binary `[f32]` + /// (little-endian), stamped with the [`embed_text_id`] they were built + /// from. Uses the namespaced backend so vectors are scoped per project. pub async fn save_symbol_vectors(&self, slug: &str) -> std::result::Result<(), String> { + let Some(stamp) = self.embed_stamp().await else { + return Ok(()); + }; let vecs = self.symbol_vectors.read().await; if vecs.is_empty() { return Ok(()); } - Self::save_vectors_map(slug, &vecs, true) + + if !Self::save_vectors_map(slug, &vecs, true, &stamp)? { + tracing::debug!( + "[QueryEngine] Not persisting symbol vectors for '{slug}': the project belongs to \ + different embed text and only a rebuild may take it over" + ); + } + Ok(()) } - /// Write a vector map to RocksDB under `slug`. `complete = false` marks a - /// mid-run checkpoint (`embedding_status = partial:N`) so status readers - /// never mistake a crash-interrupted run for a finished one. + /// Take the project's vector set for `stamp`, so the checkpoints of the + /// rebuild that follows are loadable if it is interrupted. Returns whether + /// the project was taken - see [`claim_project`]. + fn claim_vector_set(slug: &str, stamp: &str) -> std::result::Result { + use codegraph::RocksDBBackend; + + let db_path = crate::memory::shared_graph_db_path().map_err(|e| format!("{e}"))?; + if let Some(parent) = db_path.parent() { + std::fs::create_dir_all(parent) + .map_err(|e| format!("Failed to create ~/.codegraph: {e}"))?; + } + let rocks = + RocksDBBackend::open(&db_path).map_err(|e| format!("Failed to open graph.db: {e}"))?; + let mut namespaced = NamespacedBackend::new(Box::new(rocks), slug); + + claim_project(&mut namespaced, slug, stamp) + } + + /// The embed-text id this engine's vectors carry, or `None` before a vector + /// engine is attached - without a model there is nothing to compare. + async fn embed_stamp(&self) -> Option { + let model = self + .vector_engine + .read() + .await + .as_ref() + .map(|e| e.model_name().to_string())?; + Some(embed_text_id( + &model, + self.full_body_enabled(), + self.split_identifiers_enabled(), + )) + } + + /// Write a vector map to RocksDB under `slug`. `complete` only distinguishes + /// a mid-run checkpoint from a final save in the log; both add to the + /// project's set and neither removes anything from it. + /// + /// `stamp` is the [`embed_text_id`] these vectors were built from. Returns + /// whether anything was written - see [`store_vectors`]. fn save_vectors_map( slug: &str, vecs: &HashMap>, complete: bool, - ) -> std::result::Result<(), String> { - use codegraph::{NamespacedBackend, RocksDBBackend, StorageBackend}; + stamp: &str, + ) -> std::result::Result { + use codegraph::RocksDBBackend; if vecs.is_empty() { - return Ok(()); + return Ok(false); } let db_path = crate::memory::shared_graph_db_path().map_err(|e| format!("{e}"))?; @@ -627,62 +1029,18 @@ impl QueryEngine { RocksDBBackend::open(&db_path).map_err(|e| format!("Failed to open graph.db: {e}"))?; let mut namespaced = NamespacedBackend::new(Box::new(rocks), slug); - // Write each vector as binary f32 array - for (&node_id, vec) in vecs.iter() { - let key = format!("vec:{node_id}"); - let bytes: Vec = vec.iter().flat_map(|f| f.to_le_bytes()).collect(); - namespaced - .put(key.as_bytes(), &bytes) - .map_err(|e| format!("Failed to write vector: {e}"))?; + if !store_vectors(&mut namespaced, vecs, stamp)? { + return Ok(false); } - // Write embedding status - let status = if complete { - format!("complete:{}", vecs.len()) - } else { - format!("partial:{}", vecs.len()) - }; - namespaced - .put(b"embedding_status", status.as_bytes()) - .map_err(|e| format!("Failed to write embedding status: {e}"))?; - tracing::info!( - "[QueryEngine] Saved {} symbol vectors to graph.db (namespace: {}, {})", + "[QueryEngine] Saved {} symbol vectors to graph.db (namespace: {}, embed text: {}, {})", vecs.len(), slug, + stamp, if complete { "complete" } else { "checkpoint" } ); - Ok(()) - } - - /// Check if embeddings are complete (persisted status in RocksDB). - /// Returns (is_complete, count) or (false, 0) if no status found. - pub fn check_embedding_status(slug: &str) -> (bool, usize) { - use codegraph::{NamespacedBackend, RocksDBBackend, StorageBackend}; - - let db_path = match crate::memory::shared_graph_db_path() { - Ok(p) if p.exists() => p, - _ => return (false, 0), - }; - - let rocks = match RocksDBBackend::open(&db_path) { - Ok(r) => r, - Err(_) => return (false, 0), - }; - let namespaced = NamespacedBackend::new(Box::new(rocks), slug); - - match namespaced.get(b"embedding_status") { - Ok(Some(value)) => { - let status = String::from_utf8_lossy(&value); - if let Some(count_str) = status.strip_prefix("complete:") { - let count = count_str.parse::().unwrap_or(0); - (true, count) - } else { - (false, 0) - } - } - _ => (false, 0), - } + Ok(true) } /// Check if embeddings are ready (either loaded from persistence or built in background). @@ -696,74 +1054,63 @@ impl QueryEngine { /// Load persisted symbol vectors from RocksDB. /// - /// Scans for keys prefixed with `vec:` in the project namespace. - /// Returns the number of vectors loaded, or 0 if none found. - pub async fn load_symbol_vectors(&self, slug: &str) -> usize { - use codegraph::{NamespacedBackend, RocksDBBackend, StorageBackend}; + /// Loads nothing unless the stored set is stamped with the embed text this + /// engine is configured to build; a set built from other text is left + /// untouched for whoever wrote it. Returns the number of vectors loaded. + pub async fn load_symbol_vectors(&self, slug: &str) -> VectorLoad { + use codegraph::RocksDBBackend; + + let Some(stamp) = self.embed_stamp().await else { + tracing::warn!("[QueryEngine] No vector engine attached - cannot load symbol vectors"); + return VectorLoad::Unreadable; + }; let db_path = match crate::memory::shared_graph_db_path() { Ok(p) => p, - Err(_) => return 0, + Err(e) => { + tracing::warn!("[QueryEngine] No graph.db path for vectors: {}", e); + return VectorLoad::Unreadable; + } }; if !db_path.exists() { - return 0; + return VectorLoad::Absent; } let rocks = match RocksDBBackend::open(&db_path) { Ok(r) => r, Err(e) => { tracing::warn!("[QueryEngine] Failed to open graph.db for vectors: {}", e); - return 0; + return VectorLoad::Unreadable; } }; let namespaced = NamespacedBackend::new(Box::new(rocks), slug); - let entries = match namespaced.scan_prefix(b"vec:") { + let entries = match read_vectors(&namespaced, &stamp) { + Ok(entries) if entries.is_empty() => return VectorLoad::Absent, Ok(entries) => entries, - Err(e) => { - tracing::warn!("[QueryEngine] Failed to scan vectors: {}", e); - return 0; + Err(outcome) => { + tracing::info!( + "[QueryEngine] No usable symbol vectors for '{}' ({:?}, wanted embed text: {})", + slug, + outcome, + stamp + ); + return outcome; } }; + let loaded = entries.len(); let mut symbol_vecs = self.symbol_vectors.write().await; - let mut loaded = 0; - - for (key, value) in entries { - // Key format: "vec:{node_id}" (already stripped of namespace prefix by NamespacedBackend) - let key_str = match std::str::from_utf8(&key) { - Ok(s) => s, - Err(_) => continue, - }; - let node_id_str = match key_str.strip_prefix("vec:") { - Some(s) => s, - None => continue, - }; - let node_id = match node_id_str.parse::() { - Ok(id) => id, - Err(_) => continue, - }; - - // Decode binary f32 array (little-endian, 4 bytes per float) - if value.len() % 4 != 0 { - continue; - } - let vec: Vec = value - .chunks_exact(4) - .map(|chunk| f32::from_le_bytes([chunk[0], chunk[1], chunk[2], chunk[3]])) - .collect(); - - symbol_vecs.insert(node_id, vec); - loaded += 1; - } + symbol_vecs.extend(entries); tracing::info!( - "[QueryEngine] Loaded {} symbol vectors from graph.db (namespace: {})", + "[QueryEngine] Loaded {} symbol vectors from graph.db (namespace: {}, embed text: {})", loaded, - slug + slug, + stamp ); - loaded + VectorLoad::Loaded(loaded) } /// Remove vectors for symbols from a deleted file. @@ -1078,7 +1425,12 @@ impl QueryEngine { let query_time_ms = start.elapsed().as_millis() as u64; - let embedding_status = if !self.are_embeddings_ready() { + let embedding_status = if self + .daemon_vectors_mismatched + .load(std::sync::atomic::Ordering::Relaxed) + { + Some(DAEMON_VECTORS_MISMATCHED_STATUS.to_string()) + } else if !self.are_embeddings_ready() { Some("Embeddings are building in the background. Semantic matching is temporarily unavailable — results are from name/text search only.".to_string()) } else { None @@ -2739,6 +3091,27 @@ mod tests { assert_eq!(results.results[0].symbol.name, "validateEmail"); } + #[tokio::test] + async fn symbol_search_status_names_mismatched_daemon_vectors_not_a_build() { + let (engine, _) = create_test_engine().await; + let building = engine + .symbol_search("test", &SearchOptions::new()) + .await + .embedding_status + .expect("no vectors yet"); + assert!(building.contains("building")); + + engine.set_daemon_vectors_mismatched(); + let status = engine + .symbol_search("test", &SearchOptions::new()) + .await + .embedding_status + .expect("mismatched daemon vectors leave semantic search unavailable"); + assert!(!status.contains("building"), "{status}"); + assert!(status.contains("different embedding settings"), "{status}"); + assert!(status.contains("Restart the daemon"), "{status}"); + } + #[tokio::test] async fn test_symbol_search_with_type_filter() { let (engine, graph) = create_test_engine().await; @@ -4027,6 +4400,346 @@ mod tests { assert_eq!(split_identifier_words("foo"), "foo"); } + /// An empty project namespace backed by an in-memory store. + fn namespace() -> NamespacedBackend { + use codegraph::MemoryBackend; + NamespacedBackend::new(Box::new(MemoryBackend::new()), "proj") + } + + fn vectors(ids: &[NodeId]) -> HashMap> { + ids.iter().map(|&id| (id, vec![id as f32, 0.5])).collect() + } + + fn sorted(mut got: Vec<(NodeId, Vec)>) -> Vec<(NodeId, Vec)> { + got.sort_by_key(|(id, _)| *id); + got + } + + fn stored_ids(ns: &NamespacedBackend) -> Vec { + use codegraph::StorageBackend; + let mut ids: Vec = ns + .scan_prefix_keys(VECTOR_KEY_PREFIX.as_bytes()) + .unwrap() + .iter() + .filter_map(|k| vector_key_node(k)) + .collect(); + ids.sort_unstable(); + ids + } + + /// A project slug no watcher daemon holds a heartbeat for, so claiming is + /// permitted. `live_daemon_for` reads `~/.codegraph/daemons/.json`. + const UNWATCHED: &str = "no-daemon-owns-this-slug-c0ffee"; + + /// bge-small and jina-code-v2 differ in dimension, which is the sharpest + /// reason two vector sets cannot be ranked against each other. + const MODEL_A: &str = "bge-small"; + const MODEL_B: &str = "jina-code-v2"; + + /// A namespace a rebuild has already taken for `stamp`, which is the only + /// state in which anything may be stored. + fn claimed(stamp: &str) -> NamespacedBackend { + let mut ns = namespace(); + assert!(claim_project(&mut ns, UNWATCHED, stamp).unwrap()); + ns + } + + #[test] + fn vectors_round_trip_under_the_configuration_that_stored_them() { + let stamp = embed_text_id(MODEL_A, true, true); + let mut ns = claimed(&stamp); + let written = vectors(&[1, 2, 3]); + + assert!(store_vectors(&mut ns, &written, &stamp).unwrap()); + + assert_eq!( + sorted(read_vectors(&ns, &stamp).unwrap()), + sorted(written.into_iter().collect()) + ); + } + + #[test] + fn checkpoints_survive_a_crash_on_a_first_index() { + // The whole point of checkpointing: an OOM-kill partway through a + // marathon first index must leave loadable work behind. The rebuild + // claims the project up front, so its checkpoints have a set to land + // in and the next start finds them. + let mut ns = namespace(); + let stamp = embed_text_id(MODEL_A, true, true); + + assert!(claim_project(&mut ns, UNWATCHED, &stamp).unwrap()); + assert!(store_vectors(&mut ns, &vectors(&[1, 2]), &stamp).unwrap()); + assert!(store_vectors(&mut ns, &vectors(&[1, 2, 3, 4]), &stamp).unwrap()); + // ... killed here, before any complete save. + + assert_eq!(sorted(read_vectors(&ns, &stamp).unwrap()).len(), 4); + assert_eq!(stored_ids(&ns), vec![1, 2, 3, 4]); + } + + #[test] + fn checkpoints_survive_a_crash_while_migrating_an_older_index() { + use codegraph::StorageBackend; + + // Same guarantee when the store already holds something incompatible: + // 0.20.1 raw-name vectors under the same prefix with no stamp. + let mut ns = namespace(); + ns.put(b"vec:77", &[0u8; 8]).unwrap(); + let stamp = embed_text_id(MODEL_A, true, true); + + assert!(claim_project(&mut ns, UNWATCHED, &stamp).unwrap()); + assert!(store_vectors(&mut ns, &vectors(&[1, 2]), &stamp).unwrap()); + + assert_eq!(stored_ids(&ns), vec![1, 2], "the claim drops the old set"); + assert_eq!(read_vectors(&ns, &stamp).unwrap().len(), 2); + } + + #[test] + fn switching_the_embedding_model_invalidates_the_stored_set() { + // Regression: before the model was part of the stamp, a 384d set stayed + // loaded after a switch to a 768d model, and every symbol already had a + // vector so nothing re-embedded. Queries then scored 768d against 384d. + let bge = embed_text_id(MODEL_A, true, true); + let jina = embed_text_id(MODEL_B, true, true); + let mut ns = claimed(&bge); + + store_vectors(&mut ns, &vectors(&[1, 2, 3]), &bge).unwrap(); + + assert_eq!(read_vectors(&ns, &jina), Err(VectorLoad::Mismatched)); + assert_eq!(read_vectors(&ns, &bge).unwrap().len(), 3); + } + + #[test] + fn an_empty_store_is_distinguishable_from_a_foreign_one() { + // The caller reacts to these differently: a daemon that has not + // persisted yet must be waited for, a foreign set must be re-embedded. + let ns = namespace(); + assert_eq!( + read_vectors(&ns, &embed_text_id(MODEL_A, true, true)), + Err(VectorLoad::Absent) + ); + + let mut ns = claimed(&embed_text_id(MODEL_A, true, true)); + store_vectors(&mut ns, &vectors(&[1]), &embed_text_id(MODEL_A, true, true)).unwrap(); + assert_eq!( + read_vectors(&ns, &embed_text_id(MODEL_A, false, true)), + Err(VectorLoad::Mismatched) + ); + } + + #[test] + fn unstamped_0_20_1_vectors_are_a_mismatch_not_an_empty_store() { + use codegraph::StorageBackend; + + let mut ns = namespace(); + ns.put(b"vec:1", &[0u8; 8]).unwrap(); + ns.put(b"vec:2", &[0u8; 8]).unwrap(); + + assert_eq!( + read_vectors(&ns, &embed_text_id(MODEL_A, true, true)), + Err(VectorLoad::Mismatched) + ); + } + + #[test] + fn no_save_may_replace_a_set_another_embed_text_owns() { + // Enforced here rather than trusted to callers: a session attached to a + // watcher daemon embeds its own in-memory copy, and its reindex tool + // reaches the same complete-save path the daemon's periodic persist + // does. Taking a project over is a rebuild's job, via claim_project. + use codegraph::StorageBackend; + + let theirs = embed_text_id(MODEL_A, true, true); + let ours = embed_text_id(MODEL_A, true, false); + let mut ns = claimed(&theirs); + + store_vectors(&mut ns, &vectors(&[1, 2, 3]), &theirs).unwrap(); + + assert!(!store_vectors(&mut ns, &vectors(&[9]), &ours).unwrap()); + + assert_eq!(stored_ids(&ns), vec![1, 2, 3]); + assert_eq!(read_vectors(&ns, &theirs).unwrap().len(), 3); + assert_eq!(ns.get(EMBED_STAMP_KEY).unwrap(), Some(theirs.into_bytes())); + } + + #[test] + fn claiming_is_how_a_rebuild_takes_a_project_over() { + let theirs = embed_text_id(MODEL_A, true, true); + let ours = embed_text_id(MODEL_B, false, false); + let mut ns = claimed(&theirs); + + store_vectors(&mut ns, &vectors(&[1, 2, 3]), &theirs).unwrap(); + assert!(claim_project(&mut ns, UNWATCHED, &ours).unwrap()); + + assert!( + stored_ids(&ns).is_empty(), + "storage stays bounded at one set" + ); + assert_eq!(read_vectors(&ns, &theirs), Err(VectorLoad::Absent)); + assert_eq!(read_vectors(&ns, &ours).unwrap(), vec![]); + + assert!(store_vectors(&mut ns, &vectors(&[9]), &ours).unwrap()); + assert_eq!(read_vectors(&ns, &ours).unwrap().len(), 1); + } + + #[test] + fn no_save_writes_into_a_project_nobody_has_claimed() { + // Stamping is a rebuild's act, not a side effect of saving. Otherwise + // the first complete save to come along adopts whatever unstamped set + // is lying there - everything a pre-0.21 binary wrote - and serves it + // as if it matched. + let stamp = embed_text_id(MODEL_A, true, true); + + let mut ns = namespace(); + assert!(!store_vectors(&mut ns, &vectors(&[1]), &stamp).unwrap()); + assert!(stored_ids(&ns).is_empty()); + assert_eq!(read_vectors(&ns, &stamp), Err(VectorLoad::Absent)); + } + + #[test] + fn a_pre_0_21_set_is_never_adopted_by_a_save() { + use codegraph::StorageBackend; + + // A pre-0.21 `--watch` daemon left running across an upgrade keeps + // writing unstamped vectors. If a save could stamp them, later sessions + // would rank split-identifier queries against raw-name vectors and + // report semantic search ready. + let mut ns = namespace(); + ns.put(b"vec:1", &[0u8; 8]).unwrap(); + ns.put(b"vec:2", &[0u8; 8]).unwrap(); + let stamp = embed_text_id(MODEL_A, true, true); + + assert!(!store_vectors(&mut ns, &vectors(&[3]), &stamp).unwrap()); + + assert_eq!(stored_ids(&ns), vec![1, 2]); + assert_eq!(ns.get(EMBED_STAMP_KEY).unwrap(), None); + assert_eq!(read_vectors(&ns, &stamp), Err(VectorLoad::Mismatched)); + } + + #[test] + fn a_save_never_deletes_work_another_writer_stored() { + // Saves only add. A rebuild's checkpoints, a watcher daemon's periodic + // persist and a session indexing a narrower set of paths all reach this + // path with maps covering different parts of the project; any of them + // deleting what its own map does not carry would throw away the others' + // work. Clearing the set is the claim's job. + let stamp = embed_text_id(MODEL_A, true, true); + let mut ns = claimed(&stamp); + + store_vectors(&mut ns, &vectors(&[1, 2, 3]), &stamp).unwrap(); + store_vectors(&mut ns, &vectors(&[1]), &stamp).unwrap(); + store_vectors(&mut ns, &vectors(&[9]), &stamp).unwrap(); + + assert_eq!(stored_ids(&ns), vec![1, 2, 3, 9]); + } + + #[test] + fn a_set_larger_than_one_write_batch_round_trips() { + // Written in chunks so peak memory does not scale with the set: the + // checkpoint forced under memory pressure must not triple the footprint + // at the moment memory is already short. + let stamp = embed_text_id(MODEL_A, true, true); + let mut ns = claimed(&stamp); + let ids: Vec = (1..=(STORE_BATCH_VECTORS as NodeId + 17)).collect(); + + assert!(store_vectors(&mut ns, &vectors(&ids), &stamp).unwrap()); + + assert_eq!(stored_ids(&ns), ids); + assert_eq!(read_vectors(&ns, &stamp).unwrap().len(), ids.len()); + } + + #[test] + fn a_live_daemon_s_project_may_not_be_taken_by_another_process() { + // The reindex tool inside a daemon-attached session reaches the same + // rebuild path a standalone session does, so refusing at the call sites + // that remember to check is not enough - the delete itself refuses. + let mut daemon = crate::daemon::DaemonHeartbeat::new( + std::path::PathBuf::from("/tmp/some-project"), + "some-project".to_string(), + ); + + daemon.pid = u32::MAX; + assert!(owned_by_another_process(Some(&daemon))); + + // The daemon indexes its own workspace, so it must not lock itself out. + daemon.pid = std::process::id(); + assert!(!owned_by_another_process(Some(&daemon))); + + // No live daemon - a stale heartbeat is discarded before this point - + // leaves the project free to claim. + assert!(!owned_by_another_process(None)); + } + + #[test] + fn an_unwatched_project_can_be_claimed_and_rebuilt() { + let mut ns = namespace(); + let stamp = embed_text_id(MODEL_A, true, true); + + assert!(claim_project(&mut ns, UNWATCHED, &stamp).unwrap()); + assert!(store_vectors(&mut ns, &vectors(&[1]), &stamp).unwrap()); + assert_eq!(read_vectors(&ns, &stamp).unwrap().len(), 1); + } + + #[test] + fn needs_word_split_only_when_words_are_not_already_separated() { + // Delimited names already tokenize into words, so prepending the split + // form repeats what the embedder sees. Measured as a wash on Rust + // (+1.3% R@1) and a small loss on SystemVerilog (-2.5%). + assert!(!needs_word_split("authenticate_user")); + assert!(!needs_word_split("axi_lite_slave")); + assert!(!needs_word_split("kebab-case-name")); + + // These arrive as one rare token and are where splitting paid: +79% + // R@1 on camelCase, +23% on PascalCase. + assert!(needs_word_split("getUserById")); + assert!(needs_word_split("ParseRequestBody")); + + // A leading or trailing delimiter separates nothing, so these reach the + // embedder as the same single rare token their undecorated siblings do. + assert!(needs_word_split("_handleClick")); + assert!(needs_word_split("__privateField")); + assert!(needs_word_split("parseType_")); + assert!(needs_word_split("-kebabLeading")); + + // A single lowercase word has no words to separate but splits to itself; + // build_embed_text's own equality guard drops it, so this predicate + // does not need to. + assert!(needs_word_split("foo")); + } + + #[test] + fn build_embed_text_splits_camel_case_but_leaves_snake_case_alone() { + let graph = CodeGraph::in_memory().unwrap(); + let node = codegraph::Node::new( + 0, + codegraph::NodeType::Function, + PropertyMap::new().with("signature", "fn get(id: u64) -> User"), + ); + + // camelCase arrives as one rare token; the split words are front-loaded. + let camel = QueryEngine::build_embed_text(&node, 0, "getUserById", false, true, &graph); + assert!( + camel.starts_with("get user by id"), + "camelCase should be split, got: {camel}" + ); + + // snake_case already tokenizes into the same words, so prepending them + // only repeats the name - measured as a wash to a small loss. + let snake = QueryEngine::build_embed_text(&node, 0, "get_user_by_id", false, true, &graph); + assert!( + snake.starts_with("get_user_by_id"), + "snake_case must be left as-is, got: {snake}" + ); + assert!(!snake.starts_with("get user by id")); + + // A leading underscore is not a word boundary: `_handleClick` is one + // rare token exactly as `handleClick` is, so it gets the same split. + let prefixed = QueryEngine::build_embed_text(&node, 0, "_handleClick", false, true, &graph); + assert!( + prefixed.starts_with("handle click"), + "prefixed camelCase should be split, got: {prefixed}" + ); + } + #[test] fn build_embed_text_prepends_split_name_only_when_enabled() { let graph = CodeGraph::in_memory().unwrap(); @@ -4044,7 +4757,8 @@ mod tests { "got: {with_split}" ); - // Disabled (default): original transformer-path text is unchanged. + // Explicitly disabled: the original raw-name text, as shipped + // through 0.20.1. let without = QueryEngine::build_embed_text(&node, 0, "getUserById", false, false, &graph); assert!(without.starts_with("getUserById"), "got: {without}"); assert!(!without.contains("get user by id")); diff --git a/crates/codegraph-server/src/ai_query/mod.rs b/crates/codegraph-server/src/ai_query/mod.rs index 333f2b4..408ad57 100644 --- a/crates/codegraph-server/src/ai_query/mod.rs +++ b/crates/codegraph-server/src/ai_query/mod.rs @@ -15,6 +15,6 @@ mod engine; mod primitives; mod text_index; -pub use engine::QueryEngine; +pub use engine::{QueryEngine, VectorLoad}; pub use primitives::*; pub use text_index::{Posting, TextIndex, TextIndexBuilder}; diff --git a/crates/codegraph-server/src/backend.rs b/crates/codegraph-server/src/backend.rs index 14a6acf..88c54ce 100644 --- a/crates/codegraph-server/src/backend.rs +++ b/crates/codegraph-server/src/backend.rs @@ -1011,6 +1011,19 @@ impl LanguageServer for CodeGraphBackend { MemoryManager::with_model(extension_path.clone(), embedding_model), ); + // Absent means OFF, which disagrees with the engine, the CLI and the + // daemon, all of which default it on. Deliberate, with a known cost. + // + // full_body is part of the vector stamp, and a project stores one + // vector set. So a bare nvim/emacs/helix client that omits this option + // and a client that sends true cannot share vectors: whichever rebuilds + // claims the project and replaces the other's set. + // + // Left at false because flipping it would silently move every bare LSP + // client to ~3x slower indexing, and VS Code and JetBrains always send + // the option explicitly - so the conflict needs a bare client and an IDE + // client on the same project. Revisit if bare LSP clients become a + // supported path; aligning the default is then the fix. let full_body = init_opts .as_ref() .and_then(|opts| opts.get("fullBodyEmbedding")) @@ -1019,6 +1032,17 @@ impl LanguageServer for CodeGraphBackend { self.query_engine.set_full_body_embedding(full_body); tracing::info!("[LSP::initialize] Full-body embedding: {}", full_body); + // Absent means on, matching the engine default and the CLI flag. A + // client that has never heard of this option should get the behaviour + // the engine ships with, not the opposite of it. + let split_identifiers = init_opts + .as_ref() + .and_then(|opts| opts.get("splitIdentifiers")) + .and_then(|v| v.as_bool()) + .unwrap_or(true); + self.query_engine.set_split_identifiers(split_identifiers); + tracing::info!("[LSP::initialize] Split identifiers: {}", split_identifiers); + // Store workspace folders. // // `workspaceFolders` is optional in LSP: a client may send only @@ -1335,7 +1359,22 @@ impl LanguageServer for CodeGraphBackend { let slug = crate::memory::project_slug(first_folder); // Always try loading persisted vectors first - let loaded = self.query_engine.load_symbol_vectors(&slug).await; + let load = self.query_engine.load_symbol_vectors(&slug).await; + // An unreadable store says nothing about what it holds - the shared + // graph.db is one RocksDB for every project, and a lock held by + // another process reads exactly like this. Rebuilding with a slug + // claims the project, which clears its stored vectors, so a + // transient read failure used to destroy a valid set. Embed in + // memory for this session and leave the store alone. + let persist = !matches!(load, crate::ai_query::VectorLoad::Unreadable); + if !persist { + tracing::warn!( + "Could not read persisted vectors for '{}'; embedding in memory \ + only this session and leaving the store untouched", + slug + ); + } + let loaded = load.count(); if loaded > 0 && files_parsed == 0 { // All files unchanged — persisted vectors are current @@ -1348,8 +1387,8 @@ impl LanguageServer for CodeGraphBackend { .await; // Warm restart: still run a background verify-and- // fill — a crash-interrupted embed run leaves a - // `partial:` checkpoint that loads fine here but is - // missing the tail; embed_missing_symbols no-ops + // partial set that loads fine here but is missing + // the tail; embed_missing_symbols no-ops // (one graph scan, no ONNX work) when complete. The // guard also resets the breadcrumb off `post_onnx` // so warm-restart sessions steady-state at `serving`. @@ -1391,11 +1430,16 @@ impl LanguageServer for CodeGraphBackend { .await; } else { query_engine - .build_symbol_vectors_checkpointed(Some(&slug_bg)) + .build_symbol_vectors_checkpointed( + persist.then_some(slug_bg.as_str()), + ) .await; } - if let Err(e) = query_engine.save_symbol_vectors(&slug_bg).await { - tracing::warn!("Failed to persist symbol vectors: {}", e); + if persist { + if let Err(e) = query_engine.save_symbol_vectors(&slug_bg).await + { + tracing::warn!("Failed to persist symbol vectors: {}", e); + } } tracing::info!("Background embedding generation complete"); }); @@ -1992,7 +2036,13 @@ impl LanguageServer for CodeGraphBackend { // Rebuild AI query engine indexes self.query_engine.build_indexes().await; - self.query_engine.build_symbol_vectors().await; + let reindex_slug = { + let folders = self.workspace_folders.read().await; + folders.first().map(|f| crate::memory::project_slug(f)) + }; + self.query_engine + .build_symbol_vectors_checkpointed(reindex_slug.as_deref()) + .await; // Persist graph and vectors for next session if total_indexed > 0 { diff --git a/crates/codegraph-server/src/daemon.rs b/crates/codegraph-server/src/daemon.rs index 5732dcc..b183a11 100644 --- a/crates/codegraph-server/src/daemon.rs +++ b/crates/codegraph-server/src/daemon.rs @@ -173,6 +173,18 @@ pub struct DaemonConfig { pub extension_path: Option, /// Embedding model id (e.g. `bge-small`). Defaults to the server default. pub embedding_model: Option, + /// Whether to embed full function bodies rather than name + signature. + /// + /// Carried, like [`Self::split_identifiers`], from the same CLI flag the + /// MCP and LSP sessions reading this workspace's vectors get it from. Both + /// decide the embedding text and are stamped on the vectors, so a daemon + /// that defaulted either one independently would persist a set those + /// sessions cannot load - and they would replace it with one the daemon + /// cannot load, indefinitely. + pub full_body_embedding: bool, + /// Whether to split run-together identifiers into words when embedding. + /// Carried for the same reason as [`Self::full_body_embedding`]. + pub split_identifiers: bool, } /// Run the watcher daemon for a workspace until a termination signal arrives. @@ -204,6 +216,8 @@ pub async fn run(config: DaemonConfig) -> Result<(), String> { "indexOnStartup": true, "excludePatterns": config.exclude_patterns, "embeddingModel": config.embedding_model, + "fullBodyEmbedding": config.full_body_embedding, + "splitIdentifiers": config.split_identifiers, "embedOnOpen": true, }); let uri = Url::from_file_path(&workspace) diff --git a/crates/codegraph-server/src/main.rs b/crates/codegraph-server/src/main.rs index 0f25c0b..1418a4e 100644 --- a/crates/codegraph-server/src/main.rs +++ b/crates/codegraph-server/src/main.rs @@ -59,10 +59,39 @@ struct Args { #[arg(long, default_value = "bge-small")] embedding_model: String, - /// Embed full function body instead of just name+signature (captured at parse time, minimal overhead) - #[arg(long, default_value = "true")] + /// Embed full function body instead of just name+signature (captured at parse time, minimal overhead). + /// Takes an optional value: `--full-body-embedding=false` turns it off. + // With a bare `default_value` this parsed as a flag that rejected a value + // and was always true, which made the option impossible to disable. + #[arg( + long, + num_args = 0..=1, + default_value_t = true, + default_missing_value = "true", + action = clap::ArgAction::Set + )] full_body_embedding: bool, + /// Prepend the word-split form of run-together identifiers to their + /// embedding text, so `getUserById` also embeds as "get user by id". + /// + /// Names already separated by `_` or `-` are left alone: they tokenise into + /// the same words, where this measured as a wash to a small loss. A leading + /// or trailing delimiter separates nothing, so `_handleClick` is split. + /// Disable to embed raw names, as releases up to 0.20.1 did. + /// `--split-identifiers` and `--split-identifiers=true` enable it, + /// `--split-identifiers=false` disables it. + // Takes an optional value for the same reason as `--full-body-embedding`: + // a bare `default_value` on a `bool` parses as an always-true flag. + #[arg( + long, + num_args = 0..=1, + default_value_t = true, + default_missing_value = "true", + action = clap::ArgAction::Set + )] + split_identifiers: bool, + /// Scope the MCP tool surface to a named profile. /// /// `all` (default) exposes every tool (community + pro). Narrower profiles @@ -137,10 +166,11 @@ struct Args { /// is shown, and a shared engine would apply the first client's profile to /// every client that connects after it. /// -/// `--full-body-embedding` is not forwarded either: on this release it cannot -/// be set to false (a bare `default_value` on a bool parses as an always-true -/// flag), so the engine's default is already the only value a client can ask -/// for. +/// `--full-body-embedding` and `--split-identifiers` are forwarded with their +/// values, and must be. Both decide the text every symbol is embedded from, +/// so both are part of the vector stamp: an engine started with defaults +/// would write vectors under one stamp while the client that started it +/// expects another, and the client would be served no vectors at all. fn engine_args(args: &Args) -> Vec { let mut out: Vec = vec![ "--embedding-model".into(), @@ -155,6 +185,8 @@ fn engine_args(args: &Args) -> Vec { if args.graph_only { out.push("--graph-only".into()); } + out.push(format!("--full-body-embedding={}", args.full_body_embedding).into()); + out.push(format!("--split-identifiers={}", args.split_identifiers).into()); out } @@ -390,6 +422,8 @@ async fn run() { exclude_patterns: args.exclude.clone(), extension_path: args.extension_path.clone(), embedding_model: Some(args.embedding_model.clone()), + full_body_embedding: args.full_body_embedding, + split_identifiers: args.split_identifiers, }; if let Err(e) = codegraph_server::daemon::run(config).await { eprintln!("daemon error: {e}"); @@ -423,6 +457,7 @@ async fn run() { embedding_model, args.full_body_embedding, ) + .with_split_identifiers(args.split_identifiers) .with_graph_only(args.graph_only); match server.run_single_tool(&tool_name, Some(tool_args)).await { @@ -455,6 +490,7 @@ async fn run() { exclude_dirs: args.exclude.clone(), max_files: args.max_files, full_body_embedding: args.full_body_embedding, + split_identifiers: args.split_identifiers, graph_only: args.graph_only, seeds: args.workspace.clone(), }; @@ -479,6 +515,7 @@ async fn run() { tracing::info!("Workspaces: {:?}", workspaces); tracing::info!("Embedding model: {}", embedding_model.display_name()); tracing::info!("Full-body embedding: {}", args.full_body_embedding); + tracing::info!("Split identifiers: {}", args.split_identifiers); if !args.exclude.is_empty() { tracing::info!("Excluding: {:?}", args.exclude); } @@ -502,6 +539,7 @@ async fn run() { embedding_model, args.full_body_embedding, ) + .with_split_identifiers(args.split_identifiers) .with_tool_profile(tool_profile) .with_graph_only(args.graph_only); codegraph_server::crash_phase::mark("serving"); @@ -673,6 +711,24 @@ mod engine_args_tests { assert_eq!(excludes, ["cache", "generated"]); } + /// Both settings are part of the vector stamp. An engine started without + /// them would stamp its vectors differently from the client that started + /// it, and that client would then load none of them. + #[test] + fn embed_text_settings_reach_the_engine_with_their_values() { + let off = forwarded(&[ + "--connect", + "--full-body-embedding=false", + "--split-identifiers=false", + ]); + assert!(off.contains(&"--full-body-embedding=false".to_string())); + assert!(off.contains(&"--split-identifiers=false".to_string())); + + let defaults = forwarded(&["--connect"]); + assert!(defaults.contains(&"--full-body-embedding=true".to_string())); + assert!(defaults.contains(&"--split-identifiers=true".to_string())); + } + #[test] fn per_client_settings_stay_with_the_client() { // A shared engine would apply one client's profile to every client. diff --git a/crates/codegraph-server/src/mcp/engine.rs b/crates/codegraph-server/src/mcp/engine.rs index 561d0ee..44ab8f8 100644 --- a/crates/codegraph-server/src/mcp/engine.rs +++ b/crates/codegraph-server/src/mcp/engine.rs @@ -29,6 +29,7 @@ pub struct EngineConfig { pub exclude_dirs: Vec, pub max_files: usize, pub full_body_embedding: bool, + pub split_identifiers: bool, /// Build graphs only and never load the embedding model. An auto-spawned /// engine used to drop this along with every other resource setting but the /// model name, so a client that asked for graph-only still caused a model @@ -128,6 +129,7 @@ mod imp { engine.cfg.embedding_model.clone(), engine.cfg.full_body_embedding, ) + .with_split_identifiers(engine.cfg.split_identifiers) .with_graph_only(engine.cfg.graph_only); if let Some(shared) = &engine.shared_engine { server.set_shared_engine(Arc::clone(shared)).await; diff --git a/crates/codegraph-server/src/mcp/server.rs b/crates/codegraph-server/src/mcp/server.rs index 5e4bc79..fa8abc2 100644 --- a/crates/codegraph-server/src/mcp/server.rs +++ b/crates/codegraph-server/src/mcp/server.rs @@ -86,7 +86,7 @@ use super::protocol::*; use super::resources::get_all_resources; use super::tools::{get_all_tools, tool_in_profile, ToolProfile}; use super::transport::AsyncStdioTransport; -use crate::ai_query::QueryEngine; +use crate::ai_query::{QueryEngine, VectorLoad}; use crate::domain::node_props; use crate::index_state::IndexState; use crate::indexer::{IndexConfig, Indexer}; @@ -983,10 +983,24 @@ impl McpBackend { self.query_engine.set_vector_engine(engine).await; // Load persisted vectors synchronously (fast — just reads from RocksDB) - let loaded = self + let load = self .query_engine .load_symbol_vectors(&self.project_slug) .await; + // An unreadable store says nothing about what it holds - graph.db is + // one RocksDB shared by every project, and a lock held elsewhere reads + // exactly like this. Rebuilding with a slug claims the project, which + // clears its stored vectors, so a transient read failure used to + // destroy a valid set. Embed in memory this session; leave the store. + let persist = !matches!(load, crate::ai_query::VectorLoad::Unreadable); + if !persist { + tracing::warn!( + "Could not read persisted vectors for '{}'; embedding in memory \ + only this session and leaving the store untouched", + self.project_slug + ); + } + let loaded = load.count(); if loaded > 0 && result.files_parsed == 0 { tracing::info!( @@ -995,10 +1009,10 @@ impl McpBackend { ); // Steady-state restart: persisted vectors loaded, files // unchanged. Still run a background verify-and-fill — - // a crash-interrupted embed run leaves a `partial:` - // checkpoint that loads fine here but is missing the - // tail; embed_missing_symbols no-ops (one graph scan, no - // ONNX work) when the set is actually complete. + // a crash-interrupted embed run leaves a partial set that + // loads fine here but is missing the tail; + // embed_missing_symbols no-ops (one graph scan, no ONNX + // work) when the set is actually complete. let query_engine = Arc::clone(&self.query_engine); let slug = self.project_slug.clone(); tokio::spawn(async move { @@ -1038,11 +1052,13 @@ impl McpBackend { } else { // No persisted vectors — full build query_engine - .build_symbol_vectors_checkpointed(Some(&slug)) + .build_symbol_vectors_checkpointed(persist.then_some(slug.as_str())) .await; } - if let Err(e) = query_engine.save_symbol_vectors(&slug).await { - tracing::warn!("Failed to persist symbol vectors: {}", e); + if persist { + if let Err(e) = query_engine.save_symbol_vectors(&slug).await { + tracing::warn!("Failed to persist symbol vectors: {}", e); + } } tracing::info!( "Background embedding generation complete — semantic search ready" @@ -1280,6 +1296,15 @@ impl McpServer { self } + /// Split run-together identifiers into words in the embedding text. + /// + /// A builder rather than another positional on `new`: every caller of + /// `new` would otherwise have to be updated to say "yes, the default". + pub fn with_split_identifiers(self, enabled: bool) -> Self { + self.backend.query_engine.set_split_identifiers(enabled); + self + } + /// Skip embedding generation (graph + structural tools only). /// Avoids loading the ONNX model. For CI / one-shot runs. pub fn with_graph_only(mut self, graph_only: bool) -> Self { @@ -1346,15 +1371,52 @@ impl McpServer { if !self.backend.graph_only { if let Some(engine) = self.backend.memory_manager.get_vector_engine().await { self.backend.query_engine.set_vector_engine(engine).await; - let loaded = self + match self .backend .query_engine .load_symbol_vectors(&self.backend.project_slug) - .await; - tracing::info!( - "Loaded {} persisted symbol vectors from daemon-maintained graph", - loaded - ); + .await + { + VectorLoad::Loaded(n) => tracing::info!( + "Loaded {} persisted symbol vectors from daemon-maintained graph", + n + ), + // Built from embed text this session cannot use - a + // daemon left running from before an upgrade (it writes + // unstamped vectors this build will never match), or one + // started with different flags. Serve without semantic + // vectors and say what to do: re-embedding here would + // run the whole corpus through ONNX in every attached + // session, which is the cold-start cost attaching to a + // daemon exists to avoid. The daemon owns the set, so + // the daemon is what should rebuild it. + VectorLoad::Mismatched => { + self.backend.query_engine.set_daemon_vectors_mismatched(); + tracing::warn!( + "Watcher daemon's vectors were built from different embed text - \ + semantic search and similarity tools are unavailable this session. \ + Restart the daemon (after an upgrade), or start it with the same \ + --full-body-embedding / --split-identifiers / --embedding-model \ + flags as this session." + ) + } + // The daemon owns this workspace and simply has not + // finished its first embed run. Re-embedding here would + // duplicate, in every attached session, exactly the + // cold-start work attaching to a daemon exists to + // avoid. Wait for it instead. + VectorLoad::Absent => tracing::warn!( + "Watcher daemon has not persisted any symbol vectors yet - semantic \ + tools are unavailable until it has and this session is restarted." + ), + // Says nothing about what the store holds, so it is not + // grounds for re-embedding the workspace. + VectorLoad::Unreadable => tracing::error!( + "Could not read the vector store - semantic tools are unavailable \ + this session. The daemon may be mid-persist; retry, and check \ + ~/.codegraph/graph.db if it persists." + ), + } } } return; diff --git a/crates/codegraph/src/storage/memory.rs b/crates/codegraph/src/storage/memory.rs index 8730103..be54f25 100644 --- a/crates/codegraph/src/storage/memory.rs +++ b/crates/codegraph/src/storage/memory.rs @@ -86,6 +86,15 @@ impl StorageBackend for MemoryBackend { Ok(results) } + fn scan_prefix_keys(&self, prefix: &[u8]) -> Result>> { + let data = self.data.read().unwrap(); + Ok(data + .range(prefix.to_vec()..) + .take_while(|(k, _)| k.starts_with(prefix)) + .map(|(k, _)| k.clone()) + .collect()) + } + fn write_batch(&mut self, operations: Vec) -> Result<()> { let mut data = self.data.write().unwrap(); for op in operations { diff --git a/crates/codegraph/src/storage/mod.rs b/crates/codegraph/src/storage/mod.rs index a5f170b..e0eef9f 100644 --- a/crates/codegraph/src/storage/mod.rs +++ b/crates/codegraph/src/storage/mod.rs @@ -75,6 +75,28 @@ pub trait StorageBackend: Send + Sync { /// Returns an error if iteration setup fails. fn scan_prefix(&self, prefix: &[u8]) -> Result>; + /// Keys of all pairs whose key starts with the given prefix. + /// + /// The same range as [`Self::scan_prefix`], without retaining the values. + /// Callers that only need to know which keys exist - deciding what to + /// delete, sizing a range - would otherwise hold the whole matching value + /// set in memory at once, which on a vector store is hundreds of megabytes. + /// + /// The default implementation discards the values [`Self::scan_prefix`] + /// returned, which costs the memory this method exists to save; backends + /// that can iterate keys directly should override it. + /// + /// # Errors + /// + /// Returns an error if iteration setup fails. + fn scan_prefix_keys(&self, prefix: &[u8]) -> Result>> { + Ok(self + .scan_prefix(prefix)? + .into_iter() + .map(|(key, _)| key) + .collect()) + } + /// Execute a batch of write operations atomically. /// /// Either all operations succeed or none do. diff --git a/crates/codegraph/src/storage/namespaced.rs b/crates/codegraph/src/storage/namespaced.rs index babd79f..19d095c 100644 --- a/crates/codegraph/src/storage/namespaced.rs +++ b/crates/codegraph/src/storage/namespaced.rs @@ -92,6 +92,16 @@ impl StorageBackend for NamespacedBackend { .collect()) } + fn scan_prefix_keys(&self, prefix: &[u8]) -> Result>> { + let namespaced_prefix = self.prefixed_key(prefix); + Ok(self + .inner + .scan_prefix_keys(&namespaced_prefix)? + .into_iter() + .map(|k| self.strip_prefix(&k).to_vec()) + .collect()) + } + fn write_batch(&mut self, operations: Vec) -> Result<()> { let namespaced_ops = operations .into_iter() @@ -180,6 +190,27 @@ mod tests { assert!(results_b.iter().any(|(k, _)| k == b"node:3")); } + #[test] + fn test_scan_prefix_keys_scoped_and_stripped() { + let inner = MemoryBackend::new(); + let mut backend_a = NamespacedBackend::new(Box::new(inner.clone()), "proj-a"); + let mut backend_b = NamespacedBackend::new(Box::new(inner.clone()), "proj-b"); + + backend_a.put(b"node:1", b"a1").unwrap(); + backend_a.put(b"node:2", b"a2").unwrap(); + backend_a.put(b"edge:1", b"e1").unwrap(); + backend_b.put(b"node:3", b"b3").unwrap(); + + let mut keys = backend_a.scan_prefix_keys(b"node:").unwrap(); + keys.sort(); + assert_eq!(keys, vec![b"node:1".to_vec(), b"node:2".to_vec()]); + + assert_eq!( + backend_b.scan_prefix_keys(b"node:").unwrap(), + vec![b"node:3".to_vec()] + ); + } + #[test] fn test_delete() { let mut backend = create_namespaced(); diff --git a/crates/codegraph/src/storage/rocksdb_backend.rs b/crates/codegraph/src/storage/rocksdb_backend.rs index 2e28451..eede75a 100644 --- a/crates/codegraph/src/storage/rocksdb_backend.rs +++ b/crates/codegraph/src/storage/rocksdb_backend.rs @@ -272,6 +272,26 @@ impl StorageBackend for RocksDBBackend { Ok(results) } + fn scan_prefix_keys(&self, prefix: &[u8]) -> Result>> { + let mut keys = Vec::new(); + let iter = self.db.prefix_iterator(prefix); + + for item in iter { + let (key, _) = + item.map_err(|e| GraphError::storage("Failed to iterate over prefix", Some(e)))?; + + // RocksDB prefix iterator may return keys beyond the prefix + // so we need to check explicitly + if !key.starts_with(prefix) { + break; + } + + keys.push(key.to_vec()); + } + + Ok(keys) + } + fn write_batch(&mut self, operations: Vec) -> Result<()> { let mut batch = WriteBatch::default(); diff --git a/jetbrains/README.md b/jetbrains/README.md index 3d0fd79..fcbd720 100644 --- a/jetbrains/README.md +++ b/jetbrains/README.md @@ -118,7 +118,7 @@ Claude Code, Cursor and the AI Assistant MCP settings all read: "codegraph": { "command": "/path/to/codegraph-server", "args": ["--mcp", "--workspace", "/path/to/project", - "--embedding-model", "bge-small", "--full-body-embedding"] + "--embedding-model", "bge-small"] } } } diff --git a/jetbrains/gradle.properties b/jetbrains/gradle.properties index 90caa7a..3e15109 100644 --- a/jetbrains/gradle.properties +++ b/jetbrains/gradle.properties @@ -5,7 +5,7 @@ # clients. The engine it fetches is pinned separately, in # CodeGraphServerResolver.ENGINE_VERSION, so a plugin-only patch cannot start # asking the release server for a tag that was never published. -pluginVersion=0.20.1 +pluginVersion=0.21.0 # Target platform. 243 = 2024.3, the oldest build LSP4IJ 0.20.x supports that # also has a stable Code Vision API. Bumping this is a compatibility decision, diff --git a/jetbrains/scripts/engine_probe.py b/jetbrains/scripts/engine_probe.py index ddaee78..30e74fe 100644 --- a/jetbrains/scripts/engine_probe.py +++ b/jetbrains/scripts/engine_probe.py @@ -126,6 +126,7 @@ def execute_command(command, arguments): "embeddingModel": "bge-small", "staticModelPath": None, "fullBodyEmbedding": True, + "splitIdentifiers": True, "embedOnOpen": True, } @@ -185,6 +186,7 @@ def execute_command(command, arguments): # while the plugin was still deciding whether to prompt for an index. PARITY_KEYS = { "indexOnStartup": "codegraph.indexOnStartup", + "splitIdentifiers": "codegraph.splitIdentifiers", "maxFileSizeKB": "codegraph.maxFileSizeKB", "embeddingModel": "codegraph.embeddingModel", "fullBodyEmbedding": "codegraph.fullBodyEmbedding", diff --git a/jetbrains/src/main/kotlin/ai/codegraph/jetbrains/mcp/McpRegistration.kt b/jetbrains/src/main/kotlin/ai/codegraph/jetbrains/mcp/McpRegistration.kt index 7e5d7b9..468c0f9 100644 --- a/jetbrains/src/main/kotlin/ai/codegraph/jetbrains/mcp/McpRegistration.kt +++ b/jetbrains/src/main/kotlin/ai/codegraph/jetbrains/mcp/McpRegistration.kt @@ -146,7 +146,14 @@ object McpRegistration { // about what "similar" means. add("--embedding-model") add(settings.embeddingModel) - if (settings.fullBodyEmbedding) add("--full-body-embedding") + // Only passed when off: the engine already defaults + // these on, and both flags take a value precisely so + // the default can be overridden. Passing the bare + // `--full-body-embedding` when the setting was on left + // no way to express "off", so unchecking it in the + // settings had no effect on the MCP-registered engine. + if (!settings.fullBodyEmbedding) add("--full-body-embedding=false") + if (!settings.splitIdentifiers) add("--split-identifiers=false") }, ), ) diff --git a/jetbrains/src/main/kotlin/ai/codegraph/jetbrains/server/CodeGraphConnectionProvider.kt b/jetbrains/src/main/kotlin/ai/codegraph/jetbrains/server/CodeGraphConnectionProvider.kt index 082ea97..d9463f7 100644 --- a/jetbrains/src/main/kotlin/ai/codegraph/jetbrains/server/CodeGraphConnectionProvider.kt +++ b/jetbrains/src/main/kotlin/ai/codegraph/jetbrains/server/CodeGraphConnectionProvider.kt @@ -87,6 +87,7 @@ class CodeGraphConnectionProvider(private val project: Project) : OSProcessStrea "embeddingModel" to settings.embeddingModel, "staticModelPath" to settings.staticModelPath.ifBlank { null }, "fullBodyEmbedding" to settings.fullBodyEmbedding, + "splitIdentifiers" to settings.splitIdentifiers, "embedOnOpen" to settings.embedOnOpen, ) } diff --git a/jetbrains/src/main/kotlin/ai/codegraph/jetbrains/server/CodeGraphServerResolver.kt b/jetbrains/src/main/kotlin/ai/codegraph/jetbrains/server/CodeGraphServerResolver.kt index 1eb31ea..ffc3471 100644 --- a/jetbrains/src/main/kotlin/ai/codegraph/jetbrains/server/CodeGraphServerResolver.kt +++ b/jetbrains/src/main/kotlin/ai/codegraph/jetbrains/server/CodeGraphServerResolver.kt @@ -309,7 +309,7 @@ object CodeGraphServerResolver { * held equal to Cargo.toml by `scripts/publish-release-assets.sh`, which * refuses to publish while they disagree. */ - const val ENGINE_VERSION = "0.20.1" + const val ENGINE_VERSION = "0.21.0" private val ARM64_ARCHES = setOf("aarch64", "arm64") private val X64_ARCHES = setOf("x86_64", "amd64", "x64") diff --git a/jetbrains/src/main/kotlin/ai/codegraph/jetbrains/settings/CodeGraphConfigurable.kt b/jetbrains/src/main/kotlin/ai/codegraph/jetbrains/settings/CodeGraphConfigurable.kt index bb70954..acce587 100644 --- a/jetbrains/src/main/kotlin/ai/codegraph/jetbrains/settings/CodeGraphConfigurable.kt +++ b/jetbrains/src/main/kotlin/ai/codegraph/jetbrains/settings/CodeGraphConfigurable.kt @@ -94,6 +94,17 @@ class CodeGraphConfigurable(private val project: Project) : BoundConfigurable(DI "similarity search. Leave it on unless indexing time is a problem.", ) } + row { + checkBox("Split identifiers into words when embedding") + .bindSelected(state::splitIdentifiers) + .comment( + "Embeds getUserById as \"get user by id\" as well as the " + + "raw name, which helps natural-language search most in camelCase " + + "languages. snake_case names are left alone. Takes effect after the " + + "IDE restarts; the affected embeddings then rebuild automatically in " + + "the background, with no manual reindex.", + ) + } row { checkBox("Embed files as they are opened") .bindSelected(state::embedOnOpen) diff --git a/jetbrains/src/main/kotlin/ai/codegraph/jetbrains/settings/CodeGraphSettings.kt b/jetbrains/src/main/kotlin/ai/codegraph/jetbrains/settings/CodeGraphSettings.kt index 2ce5e26..9fdac58 100644 --- a/jetbrains/src/main/kotlin/ai/codegraph/jetbrains/settings/CodeGraphSettings.kt +++ b/jetbrains/src/main/kotlin/ai/codegraph/jetbrains/settings/CodeGraphSettings.kt @@ -68,6 +68,17 @@ class CodeGraphSettings : PersistentStateComponent { */ @JvmField var fullBodyEmbedding: Boolean = true + /** + * Also embed the word-split form of identifiers whose words are run + * together, so `getUserById` embeds as "get user by id" too. Names + * already separated by `_` or `-` are left alone; a leading or trailing + * delimiter separates nothing, so `_handleClick` is split like + * `handleClick`. Defaults to true, matching the engine and the VS Code + * client - the three are held equal by + * jetbrains/scripts/engine_probe.py. + */ + @JvmField var splitIdentifiers: Boolean = true + @JvmField var embedOnOpen: Boolean = true @JvmField var codeLensEnabled: Boolean = true diff --git a/mcp-package/bin/fetch-engine.js b/mcp-package/bin/fetch-engine.js index d8ba0f1..a630c06 100644 --- a/mcp-package/bin/fetch-engine.js +++ b/mcp-package/bin/fetch-engine.js @@ -75,7 +75,7 @@ const RELEASE_BASE = "https://github.com/codegraph-ai/CodeGraph/releases/downloa * Kept equal to the engine version by `scripts/publish-release-assets.sh`, which * refuses to publish while any channel's pin disagrees with Cargo.toml. */ -const ENGINE_VERSION = "0.20.1"; +const ENGINE_VERSION = "0.21.0"; /** Codes Windows and POSIX use for "something else has this file open". */ const IN_USE_ERROR_CODES = new Set(["EPERM", "EACCES", "EBUSY", "ETXTBSY"]); diff --git a/mcp-package/package.json b/mcp-package/package.json index 1c37e0a..858bdd6 100644 --- a/mcp-package/package.json +++ b/mcp-package/package.json @@ -1,6 +1,6 @@ { "name": "@astudioplus/codegraph-mcp", - "version": "0.20.1", + "version": "0.21.0", "mcpName": "io.github.codegraph-ai/codegraph", "description": "CodeGraph MCP server \u2014 cross-language code intelligence with 42 tools, 38 languages", "author": "Andrey Vasilevsky ", diff --git a/mcp-package/server.json b/mcp-package/server.json index e754580..986b579 100644 --- a/mcp-package/server.json +++ b/mcp-package/server.json @@ -6,12 +6,12 @@ "url": "https://github.com/codegraph-ai/CodeGraph", "source": "github" }, - "version": "0.20.1", + "version": "0.21.0", "packages": [ { "registryType": "npm", "identifier": "@astudioplus/codegraph-mcp", - "version": "0.20.1", + "version": "0.21.0", "transport": { "type": "stdio" }, diff --git a/vscode/README.md b/vscode/README.md index e15e8db..c88bb82 100644 --- a/vscode/README.md +++ b/vscode/README.md @@ -30,7 +30,7 @@ The server indexes the current working directory automatically. Install from the marketplace, or sideload the VSIX: ```bash -code --install-extension codegraph-0.20.1.vsix +code --install-extension codegraph-0.21.0.vsix ``` One VSIX serves every platform. @@ -46,13 +46,7 @@ CodeGraph's Symbols and Memories views live in the CodeGraph activity-bar contai ### MCP Server flags -| Flag | Default | Description | -|------|---------|-------------| -| `--workspace ` | current dir | Directories to index (repeatable for multi-project) | -| `--exclude ` | — | Directories to skip (repeatable) | -| `--embedding-model ` | `bge-small` | `bge-small` (384d, fast), `jina-code-v2` (768d, 6x slower), or `static` (model2vec, 256d — ~100× faster indexing, no ONNX, ~90% of BGE quality in hybrid search; needs a local model directory, see `codegraph.staticModelPath` below) | -| `--full-body-embedding` | `true` | Embed full function body (~50 lines) for better semantic search and duplicate detection | -| `--max-files ` | 5000 | Maximum files to index | +The engine's command-line flags (`--workspace`, `--exclude`, `--embedding-model`, `--full-body-embedding`, `--split-identifiers` and more) are documented in the [main README](https://github.com/codegraph-ai/codegraph#mcp-server-flags). ### VS Code settings diff --git a/vscode/package.json b/vscode/package.json index e32f993..03c8d15 100644 --- a/vscode/package.json +++ b/vscode/package.json @@ -2,7 +2,7 @@ "name": "codegraph", "displayName": "CodeGraph", "description": "Cross-language code intelligence powered by graph analysis", - "version": "0.20.1", + "version": "0.21.0", "publisher": "aStudioPlus", "author": "Andrey Vasilevsky ", "license": "Apache-2.0", @@ -306,7 +306,7 @@ "Jina Code V2 (768d) — 6x slower indexing, no quality advantage with full-body embeddings. ~642MB download.", "Static (model2vec, 256d) — ~100x faster indexing, no ONNX runtime or 1.5GB RAM gate, ~90% of BGE quality in hybrid search. Needs a local model dir: ~/.codegraph/static_models/jina-code-static-256 by default, or codegraph.staticModelPath." ], - "description": "Embedding model for semantic search and code similarity" + "description": "Embedding model for semantic search and code similarity. Takes effect after the window is reloaded; the embeddings are then rebuilt automatically in the background, since vectors from different models cannot be compared." }, "codegraph.staticModelPath": { "type": "string", @@ -318,7 +318,13 @@ "type": "boolean", "default": true, "scope": "resource", - "description": "Embed full function body (first ~50 lines) instead of just name+signature. Better quality for NL search and clone detection, ~3x slower indexing. Requires reindex after changing." + "description": "Embed full function body (first ~50 lines) instead of just name+signature. Better quality for NL search and clone detection, ~3x slower indexing. Takes effect after the window is reloaded; the affected embeddings then rebuild automatically in the background, with no manual reindex." + }, + "codegraph.splitIdentifiers": { + "type": "boolean", + "default": true, + "scope": "resource", + "description": "Also embed the word-split form of identifiers whose words are run together, so getUserById embeds as \"get user by id\" too. Names already separated by _ or - are left alone, since they tokenise into the same words; a leading or trailing delimiter separates nothing, so _handleClick is split like handleClick. Improves natural-language search most in camelCase languages (TypeScript, Java, C#). Takes effect after the window is reloaded; the affected embeddings then rebuild automatically in the background, with no manual reindex." }, "codegraph.embedOnOpen": { "type": "boolean", diff --git a/vscode/src/extension.ts b/vscode/src/extension.ts index 4828d73..bfce45a 100644 --- a/vscode/src/extension.ts +++ b/vscode/src/extension.ts @@ -410,6 +410,7 @@ export async function activate(context: vscode.ExtensionContext): Promise embeddingModel: latestConfig.get('embeddingModel'), staticModelPath: latestConfig.get('staticModelPath'), fullBodyEmbedding: latestConfig.get('fullBodyEmbedding'), + splitIdentifiers: latestConfig.get('splitIdentifiers'), embedOnOpen: latestConfig.get('embedOnOpen'), }; console.log('[CodeGraph] Initialization options:', JSON.stringify(opts)); diff --git a/vscode/src/telemetry/allowlists.ts b/vscode/src/telemetry/allowlists.ts index ec21438..c72cc3b 100644 --- a/vscode/src/telemetry/allowlists.ts +++ b/vscode/src/telemetry/allowlists.ts @@ -410,6 +410,7 @@ export const SETTINGS_SNAPSHOT_KEYS = { 'parallelParsing', 'cache.enabled', 'fullBodyEmbedding', + 'splitIdentifiers', 'memory.enabled', 'memory.autoInvalidate', 'memory.gitMining.enabled',