refactor: apply 5-pass audit optimizations across mcp-memory codebase

This commit is contained in:
Riz Ashraf committed 2026-10-06 06:05:38 +01:00
1 parent 924b6d09fa
commit 5bd8b1587a
43 files changed
+1866 -1658

No files matched your search

+66 -32
View File
@@ -46,7 +46,7 @@ pub async fn start_background_indexer(state: Arc<MemoryState>) {
.await
.unwrap_or_default();
let idx = state.get_search_index();
let idx = state.get_search_index().await;
for file_path in files_to_process {
if let Ok(content) = std::fs::read_to_string(&file_path) {
@@ -71,36 +71,64 @@ pub async fn start_background_indexer(state: Arc<MemoryState>) {
let mut chunks = Vec::new();
extract_chunks(tree.root_node(), &content, &mut chunks, ext);
for (name, code, desc) in chunks {
// Generate embedding
if let Ok(mut emb) = generate_embeddings_async(vec![code.clone()]).await {
let embedding = emb.pop();
// Gold Standard: Batch generate embeddings in chunks of 16 to eliminate sequential HTTP overhead
for chunk_batch in chunks.chunks(16) {
let texts: Vec<String> = chunk_batch.iter().map(|(_, code, _)| code.clone()).collect();
let embeddings = generate_embeddings_async(texts).await.unwrap_or_default();
let file_name =
file_path.file_name().unwrap_or_default().to_string_lossy();
let mut new_snippets = Vec::with_capacity(chunk_batch.len());
let now = std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.unwrap_or_default()
.as_secs();
for (i, (name, code, desc)) in chunk_batch.iter().enumerate() {
let embedding = embeddings.get(i).cloned();
let file_name = file_path.file_name().unwrap_or_default().to_string_lossy();
let snippet_name = format!("{}:{}", file_name, name);
let snippet = Snippet {
name: snippet_name.to_string(),
name: snippet_name,
language: ext.to_string(),
code: code.clone(),
description: format!("{} in {}", desc, file_path.display()),
updated_at: std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.unwrap_or_default()
.as_secs(),
updated_at: now,
tags: vec![],
embedding,
};
new_snippets.push(snippet);
}
state.code.snippets.modify(|snippets| {
// Prevent duplicates if already indexed
if !snippets.iter().any(|s| s.name == snippet.name) {
snippets.push(snippet.clone());
// Gold Standard: Modify store ONCE per batch with zero-copy HashSet<&str> lookup
let mut snippets_to_index = Vec::new();
state.code.snippets.modify(|snippets| {
let existing_names: std::collections::HashSet<&str> =
snippets.iter().map(|s| s.name.as_str()).collect();
let mut filtered_new = Vec::with_capacity(new_snippets.len());
let mut seen_in_batch = std::collections::HashSet::new();
for snippet in new_snippets {
if !existing_names.contains(snippet.name.as_str())
&& seen_in_batch.insert(snippet.name.clone())
{
filtered_new.push(snippet);
}
});
}
let _ = idx.index_snippet(&snippet).await;
for snippet in filtered_new {
snippets.push(snippet.clone());
snippets_to_index.push(snippet);
}
if snippets.len() > 1000 {
let overflow = snippets.len() - 1000;
snippets.drain(0..overflow);
}
});
for snippet in &snippets_to_index {
let _ = idx.index_snippet(snippet).await;
}
}
}
@@ -111,7 +139,7 @@ pub async fn start_background_indexer(state: Arc<MemoryState>) {
}
fn extract_chunks(node: Node, code: &str, chunks: &mut Vec<(String, String, String)>, ext: &str) {
extract_chunks_with_parent(node, code, chunks, ext, None);
extract_chunks_with_parent(node, code, chunks, ext, None, 0);
}
fn extract_chunks_with_parent(
@@ -120,22 +148,28 @@ fn extract_chunks_with_parent(
chunks: &mut Vec<(String, String, String)>,
ext: &str,
parent_scope: Option<&str>,
depth: usize,
) {
// Stack overflow protection: Cap recursion depth at 100
if depth > 100 {
return;
}
let kind = node.kind();
let is_impl_or_class = matches!(kind, "impl_item" | "class_declaration" | "class_definition");
let current_scope = if is_impl_or_class {
let current_scope: Option<&str> = if is_impl_or_class {
let mut cursor = node.walk();
let mut type_name = None;
for child in node.children(&mut cursor) {
if child.kind() == "type_identifier" || child.kind() == "name" || child.kind() == "identifier" {
type_name = child.utf8_text(code.as_bytes()).ok().map(|s| s.to_string());
type_name = child.utf8_text(code.as_bytes()).ok();
break;
}
}
type_name.or_else(|| parent_scope.map(|s| s.to_string()))
type_name.or(parent_scope)
} else {
parent_scope.map(|s| s.to_string())
parent_scope
};
let is_structural = matches!(
@@ -151,30 +185,30 @@ fn extract_chunks_with_parent(
if is_structural {
let mut raw_text = node.utf8_text(code.as_bytes()).unwrap_or("").to_string();
let mut name = "unknown".to_string();
let mut name = "unknown";
let mut cursor = node.walk();
for child in node.children(&mut cursor) {
let child_kind = child.kind();
if child_kind == "identifier" || child_kind == "name" || child_kind == "type_identifier" {
name = child
.utf8_text(code.as_bytes())
.unwrap_or("unknown")
.to_string();
if let Ok(text) = child.utf8_text(code.as_bytes()) {
name = text;
}
break;
}
}
if let Some(ref scope) = current_scope {
let mut final_name = name.to_string();
if let Some(scope) = current_scope {
raw_text = format!("// Parent Scope: {}\n{}", scope, raw_text);
name = format!("{}::{}", scope, name);
final_name = format!("{}::{}", scope, name);
}
let desc = format!("{} AST node", kind);
chunks.push((name, raw_text, desc));
chunks.push((final_name, raw_text, desc));
} else {
let mut cursor = node.walk();
for child in node.named_children(&mut cursor) {
extract_chunks_with_parent(child, code, chunks, ext, current_scope.as_deref());
extract_chunks_with_parent(child, code, chunks, ext, current_scope, depth + 1);
}
}
}