Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 0 additions & 11 deletions Cargo.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

2 changes: 0 additions & 2 deletions Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -74,8 +74,6 @@ thiserror = "1.0"
rocksdb = "0.22"
# Pin libz-sys to 1.1.25 — v1.1.26 has broken vendored zlib build on macOS/Windows
libz-sys = "=1.1.25"
# Cross-platform advisory file locks (Linux fcntl / Windows LockFileEx)
fs2 = "0.4"
log = "0.4"
uuid = { version = "1.0", features = ["v4", "serde"] }

Expand Down
16 changes: 5 additions & 11 deletions crates/codegraph-server/src/ai_query/engine.rs
Original file line number Diff line number Diff line change
Expand Up @@ -971,15 +971,13 @@ impl QueryEngine {
/// rebuild that follows are loadable if it is interrupted. Returns whether
/// the project was taken - see [`claim_project`].
fn claim_vector_set(slug: &str, stamp: &str) -> std::result::Result<bool, String> {
use codegraph::RocksDBBackend;

let db_path = crate::memory::shared_graph_db_path().map_err(|e| format!("{e}"))?;
if let Some(parent) = db_path.parent() {
std::fs::create_dir_all(parent)
.map_err(|e| format!("Failed to create ~/.codegraph: {e}"))?;
}
let rocks =
RocksDBBackend::open(&db_path).map_err(|e| format!("Failed to open graph.db: {e}"))?;
let rocks = crate::memory::open_shared_graph_db(&db_path)
.map_err(|e| format!("Failed to open graph.db: {e}"))?;
let mut namespaced = NamespacedBackend::new(Box::new(rocks), slug);

claim_project(&mut namespaced, slug, stamp)
Expand Down Expand Up @@ -1013,8 +1011,6 @@ impl QueryEngine {
complete: bool,
stamp: &str,
) -> std::result::Result<bool, String> {
use codegraph::RocksDBBackend;

if vecs.is_empty() {
return Ok(false);
}
Expand All @@ -1025,8 +1021,8 @@ impl QueryEngine {
.map_err(|e| format!("Failed to create ~/.codegraph: {e}"))?;
}

let rocks =
RocksDBBackend::open(&db_path).map_err(|e| format!("Failed to open graph.db: {e}"))?;
let rocks = crate::memory::open_shared_graph_db(&db_path)
.map_err(|e| format!("Failed to open graph.db: {e}"))?;
let mut namespaced = NamespacedBackend::new(Box::new(rocks), slug);

if !store_vectors(&mut namespaced, vecs, stamp)? {
Expand Down Expand Up @@ -1058,8 +1054,6 @@ impl QueryEngine {
/// engine is configured to build; a set built from other text is left
/// untouched for whoever wrote it. Returns the number of vectors loaded.
pub async fn load_symbol_vectors(&self, slug: &str) -> VectorLoad {
use codegraph::RocksDBBackend;

let Some(stamp) = self.embed_stamp().await else {
tracing::warn!("[QueryEngine] No vector engine attached - cannot load symbol vectors");
return VectorLoad::Unreadable;
Expand All @@ -1077,7 +1071,7 @@ impl QueryEngine {
return VectorLoad::Absent;
}

let rocks = match RocksDBBackend::open(&db_path) {
let rocks = match crate::memory::open_shared_graph_db(&db_path) {
Ok(r) => r,
Err(e) => {
tracing::warn!("[QueryEngine] Failed to open graph.db for vectors: {}", e);
Expand Down
20 changes: 13 additions & 7 deletions crates/codegraph-server/src/backend.rs
Original file line number Diff line number Diff line change
Expand Up @@ -515,7 +515,7 @@ impl CodeGraphBackend {
workspace: &std::path::Path,
graph: &codegraph::CodeGraph,
) -> std::result::Result<(), String> {
use codegraph::{NamespacedBackend, RocksDBBackend, StorageBackend};
use codegraph::{NamespacedBackend, StorageBackend};

// Ephemeral workspaces (test harness tempdirs) skip
// persistence to the shared graph.db entirely. The in-memory
Expand All @@ -532,7 +532,7 @@ impl CodeGraphBackend {
.map_err(|e| format!("Failed to create ~/.codegraph: {e}"))?;
}

let mut rocks = RocksDBBackend::open_with_stale_lock_recovery(&db_path)
let mut rocks = crate::memory::open_shared_graph_db(&db_path)
.map_err(|e| format!("Failed to open graph.db: {e}"))?;

let registry_value = serde_json::json!({
Expand Down Expand Up @@ -1206,23 +1206,29 @@ impl LanguageServer for CodeGraphBackend {
}
Err(e) => {
tracing::error!(
"LSP: RocksDB graph.db open failed: {e} — running in-memory only \
this session. Changes will NOT persist across restarts."
"LSP: could not load the persisted graph ({e}) - starting without it."
);
self.client
.log_message(
MessageType::ERROR,
format!(
"CodeGraph: graph database open failed ({e}). \
Index is volatile this session — restart to retry; \
if the error persists, check ~/.codegraph/graph.db for a stale LOCK file."
"CodeGraph: could not load the saved index ({e}), so this session \
starts without it. If another CodeGraph process keeps \
~/.codegraph/graph.db busy, restart once it is idle."
),
)
.await;
}
}
}

// The saved hashes (loaded in `initialize`) vouch that files are already
// in the graph. Without a persisted graph they would make the indexer
// skip every unchanged file and leave the session empty.
if !loaded_from_persistence {
self.index_state.lock().await.clear();
}

// Run incremental indexing: hash-based dedup skips unchanged files.
// - Fresh start (no persisted graph): full index if indexOnStartup=true
// - Loaded from persistence: incremental pass to catch changed files
Expand Down
42 changes: 17 additions & 25 deletions crates/codegraph-server/src/domain/impact.rs
Original file line number Diff line number Diff line change
Expand Up @@ -7,9 +7,7 @@

use crate::ai_query::QueryEngine;
use crate::domain::node_props;
use codegraph::{
CodeGraph, Direction, EdgeType, NamespacedBackend, NodeId, RocksDBBackend, StorageBackend,
};
use codegraph::{CodeGraph, Direction, EdgeType, NamespacedBackend, NodeId, StorageBackend};
use serde::Serialize;
use std::collections::HashSet;
use tokio::sync::RwLock;
Expand Down Expand Up @@ -338,24 +336,22 @@ fn find_cross_project_consumers(
}
};

// Open RocksDB, scan registry, then DROP the connection before per-project
// loading — RocksDB uses exclusive locks, so only one connection at a time.
let entries = {
let rocks = match RocksDBBackend::open(&db_path) {
Ok(r) => r,
Err(e) => {
tracing::warn!("[cross-project] Failed to open graph.db: {}", e);
return Vec::new();
}
};
match StorageBackend::scan_prefix(&rocks, b"_registry:") {
Ok(e) => e,
Err(e) => {
tracing::warn!("[cross-project] Failed to scan registry: {}", e);
return Vec::new();
}
// Open RocksDB once for the registry scan and every per-project load, so
// a busy DB costs this query at most one lock wait. The handle drops, and
// the lock is released, when this function returns.
let rocks = match crate::memory::open_shared_graph_db(&db_path) {
Ok(r) => r,
Err(e) => {
tracing::warn!("[cross-project] Failed to open graph.db: {}", e);
return Vec::new();
}
};
let entries = match StorageBackend::scan_prefix(&rocks, b"_registry:") {
Ok(e) => e,
Err(e) => {
tracing::warn!("[cross-project] Failed to scan registry: {}", e);
return Vec::new();
}
// rocks dropped here — lock released
};

tracing::debug!(
Expand Down Expand Up @@ -401,11 +397,7 @@ fn find_cross_project_consumers(
})
.unwrap_or_else(|| slug.clone());

let other_rocks = match RocksDBBackend::open(&db_path) {
Ok(r) => r,
Err(_) => continue,
};
let namespaced = NamespacedBackend::new(Box::new(other_rocks), &slug);
let namespaced = NamespacedBackend::new(Box::new(rocks.clone()), &slug);
let mut other_graph = match CodeGraph::with_backend(Box::new(namespaced)) {
Ok(g) => g,
Err(_) => continue,
Expand Down
2 changes: 1 addition & 1 deletion crates/codegraph-server/src/main.rs
Original file line number Diff line number Diff line change
Expand Up @@ -276,7 +276,7 @@ fn classify_panic(payload: &str, location: &str) -> (&'static str, &'static str)
/// Strategy: panic / SIGINT / SIGTERM all funnel into `process::exit`.
/// At process exit the kernel releases all fcntl / LockFileEx grants,
/// so the next launch sees only the `LOCK` *file* (no live holder),
/// which `RocksDBBackend::open_with_stale_lock_recovery` clears. WAL
/// which RocksDB reuses without complaint. WAL
/// durability is per-write, so any in-flight batch is either fully
/// applied or fully discarded on next open — `exit` skipping `Drop` is
/// a safe tradeoff here.
Expand Down
97 changes: 58 additions & 39 deletions crates/codegraph-server/src/mcp/server.rs
Original file line number Diff line number Diff line change
Expand Up @@ -92,7 +92,7 @@ use crate::index_state::IndexState;
use crate::indexer::{IndexConfig, Indexer};
use crate::memory::{self, MemoryManager};
use crate::parser_registry::ParserRegistry;
use codegraph::{CodeGraph, NamespacedBackend, RocksDBBackend, StorageBackend};
use codegraph::{CodeGraph, NamespacedBackend, StorageBackend};
use serde_json::Value;
use std::path::PathBuf;
use std::sync::Arc;
Expand Down Expand Up @@ -200,9 +200,9 @@ impl McpBackend {
}
Err(e) => {
tracing::error!(
"RocksDB graph.db open failed: {e} — running in-memory only this session. \
Changes will NOT persist across restarts. Inspect ~/.codegraph/graph.db and \
ensure no other codegraph-server process is running."
"Could not load the persisted graph ({e}) - this session re-indexes the \
workspace instead. If another codegraph process keeps ~/.codegraph/graph.db \
busy, restart this session once it is idle."
);
Arc::new(RwLock::new(
CodeGraph::in_memory().expect("Failed to create in-memory graph"),
Expand Down Expand Up @@ -327,31 +327,22 @@ impl McpBackend {
}
}

// Mark WHERE we are (telemetry) and arm this process's sentinel for
// the load. The phase guard resets to `serving` on return; the
// Mark WHERE we are (telemetry); the load arms this process's
// sentinel. The phase guard resets to `serving` on return; the
// sentinel only clears on a completed load (success or graceful
// error), not on a native AV. The sentinel body carries this
// process's start time so a recycled PID can't impersonate a live
// loader (Windows reuses PIDs aggressively during crash-restart
// churn).
// error) or a lock wait, not on a native AV. The sentinel body
// carries this process's start time so a recycled PID can't
// impersonate a live loader (Windows reuses PIDs aggressively during
// crash-restart churn).
let _phase = crate::crash_phase::enter("graph_load");
let sentinel = db_path
.parent()
.map(|p| p.join(format!("graph.loading.{}", std::process::id())));
if let Some(s) = &sentinel {
let body = Self::own_start_time()
.map(|t| t.to_string())
.unwrap_or_default();
let _ = std::fs::write(s, body);
}

let result = Self::load_persistent_graph_inner(&db_path, slug);
if let Some(s) = &sentinel {
let _ = std::fs::remove_file(s);
}

match result {
Ok(graph) => Ok(graph),
// Another process kept the DB open past the wait. That is
// contention, not damage: redirecting here would abandon every
// project's graph because a sibling session was mid-persist.
Err(e @ codegraph::GraphError::Locked { .. }) => Err(format!("graph.db: {e}")),
Err(e) => {
// RocksDB reported corruption gracefully (didn't AV).
// Redirect to a fresh generation and retry once so the
Expand All @@ -363,6 +354,7 @@ impl McpBackend {
Self::sweep_stale_graph_dbs(parent, &fresh);
}
Self::load_persistent_graph_inner(&fresh, slug)
.map_err(|e| format!("graph.db load failed: {e}"))
}
}
}
Expand Down Expand Up @@ -516,25 +508,43 @@ impl McpBackend {
/// detach storage to release the lock. The fallible core of
/// [`open_persistent_graph`], split out so the poison-pill wrapper can
/// retry it on a fresh DB after quarantining a corrupt one.
///
/// This process's `graph.loading.<pid>` sentinel is armed for every open
/// attempt (open itself can AV on a torn DB) and kept from a successful
/// open through the load, but removed while waiting out another holder:
/// a session killed mid-wait must not leave poison evidence behind.
fn load_persistent_graph_inner(
db_path: &std::path::Path,
slug: &str,
) -> Result<CodeGraph, String> {
// Stale-LOCK recovery: a prior crash can leave LOCK in place. The
// recovery variant only clobbers it after probing for a live holder —
// a healthy concurrent process is still respected.
let rocks = RocksDBBackend::open_with_stale_lock_recovery(db_path)
.map_err(|e| format!("Failed to open graph.db: {e}"))?;
let namespaced = NamespacedBackend::new(Box::new(rocks), slug);
let mut graph = CodeGraph::with_backend(Box::new(namespaced))
.map_err(|e| format!("Failed to load graph: {e}"))?;
) -> Result<CodeGraph, codegraph::GraphError> {
let sentinel = db_path
.parent()
.map(|p| p.join(format!("graph.loading.{}", std::process::id())));
let body = Self::own_start_time()
.map(|t| t.to_string())
.unwrap_or_default();
let arm = || {
if let Some(s) = &sentinel {
let _ = std::fs::write(s, &body);
}
};
let disarm = || {
if let Some(s) = &sentinel {
let _ = std::fs::remove_file(s);
}
};

// Detach to release the RocksDB lock — all data is now in memory
graph
.detach_storage()
.map_err(|e| format!("Failed to detach storage: {e}"))?;
let result = memory::open_shared_graph_db_with(db_path, arm, disarm).and_then(|rocks| {
let namespaced = NamespacedBackend::new(Box::new(rocks), slug);
let mut graph = CodeGraph::with_backend(Box::new(namespaced))?;

// Detach to release the RocksDB lock — all data is now in memory
graph.detach_storage()?;

Ok(graph)
Ok(graph)
});
disarm();
result
}

/// Move a corrupt `graph.db` aside so the next open starts clean. RocksDB
Expand Down Expand Up @@ -580,7 +590,7 @@ impl McpBackend {
.map_err(|e| format!("Failed to create ~/.codegraph: {e}"))?;
}

let mut rocks = RocksDBBackend::open_with_stale_lock_recovery(&db_path)
let mut rocks = memory::open_shared_graph_db(&db_path)
.map_err(|e| format!("Failed to open graph.db for persist: {e}"))?;

// Write project registry entry (un-namespaced, global key)
Expand Down Expand Up @@ -633,7 +643,7 @@ impl McpBackend {
return Ok(vec![]);
}

let rocks = RocksDBBackend::open_with_stale_lock_recovery(&db_path)
let rocks = memory::open_shared_graph_db(&db_path)
.map_err(|e| format!("Failed to open graph.db: {e}"))?;

let entries = rocks
Expand Down Expand Up @@ -903,7 +913,16 @@ impl McpBackend {
}

/// Load saved file hashes from disk. Returns true if state was loaded.
///
/// The hashes vouch that a file's symbols are already in the graph, so
/// they only load alongside a graph that has some. Against an empty graph
/// (a redirected or unloadable DB) they would make the indexer skip every
/// unchanged file and leave the session with no symbols at all.
pub async fn load_index_state(&self) -> bool {
if self.graph.read().await.node_count() == 0 {
tracing::info!("No persisted graph to resume from - indexing every file");
return false;
}
let mut state = self.index_state.lock().await;
let count = state.load();
count > 0
Expand Down
Loading
Loading