perf: completely remove disk I/O from memory access paths, optimize Tantivy indexing, and fix MemoryIndex commits

This commit is contained in:
Riz Ashraf committed 2026-09-12 20:41:30 +01:00
1 parent a5011048b1
commit 686fea683d
3 files changed
+110 -64

No files matched your search

+54 -28
View File
@@ -34,35 +34,53 @@ pub struct MemoryState {
}
impl MemoryState {
pub fn unique_items<T: Eq + std::hash::Hash + Clone>(input: Vec<T>) -> Vec<T> {
let mut keys = std::collections::HashSet::new();
input.into_iter().filter(|entry| keys.insert(entry.clone())).collect()
}
fn master_mtime(&self) -> SystemTime {
fs::metadata(&self.master_path)
.and_then(|m| m.modified())
.unwrap_or(SystemTime::UNIX_EPOCH)
}
pub fn unique_items<T: Eq + std::hash::Hash + Clone>(input: Vec<T>) -> Vec<T> {
let mut keys = HashSet::new();
let mut list = Vec::new();
for entry in input {
if keys.insert(entry.clone()) {
list.push(entry);
pub fn recover_wal(&self) {
let wal_path = self.base_dir.join("wal.jsonl");
if let Ok(content) = std::fs::read_to_string(&wal_path) {
let mut session = self.session_graph.write().unwrap();
for line in content.lines() {
if let Ok(d) = serde_json::from_str::<KnowledgeGraph>(line) {
Self::merge_graphs(&mut session, &d);
}
}
}
list
}
pub fn merge_graphs(dest: &mut KnowledgeGraph, src: &KnowledgeGraph) {
for (name, src_ent) in &src.entities {
let dest_ent = dest
.entities
.entry(name.clone())
.or_insert_with(|| src_ent.clone());
if dest_ent.name == src_ent.name {
dest_ent.observations.extend(src_ent.observations.clone());
dest_ent.observations = Self::unique_items(dest_ent.observations.clone());
.or_insert_with(|| crate::models::Entity {
name: src_ent.name.clone(),
entity_type: src_ent.entity_type.clone(),
observations: Vec::new(),
namespace: src_ent.namespace.clone(),
git_branch: src_ent.git_branch.clone(),
});
for obs in &src_ent.observations {
if !dest_ent.observations.contains(obs) {
dest_ent.observations.push(obs.clone());
}
}
}
for rel in &src.relations {
if !dest.relations.contains(rel) {
dest.relations.push(rel.clone());
}
}
dest.relations.extend(src.relations.clone());
dest.relations = Self::unique_items(dest.relations.clone());
}
pub fn read_master_cached(&self) -> KnowledgeGraph {
@@ -98,14 +116,6 @@ impl MemoryState {
pub fn get_full_graph(&self) -> KnowledgeGraph {
let mut master = self.read_master_cached();
let wal_path = self.base_dir.join("wal.jsonl");
if let Ok(content) = std::fs::read_to_string(&wal_path) {
for line in content.lines() {
if let Ok(d) = serde_json::from_str::<KnowledgeGraph>(line) {
Self::merge_graphs(&mut master, &d);
}
}
}
let session_graph = self.session_graph.read().unwrap();
Self::merge_graphs(&mut master, &session_graph);
master
@@ -183,14 +193,29 @@ impl MemoryState {
pub fn rebuild_index(&self) {
if let Ok(new_idx) = MemoryIndex::new(&self.base_dir) {
let session = self.session_graph.read().unwrap();
let mut full = {
let cache = self.master_cache.read().unwrap();
cache.0.clone()
};
Self::merge_graphs(&mut full, &session);
for e in full.entities.values() {
let _ = new_idx.index_entity(e);
let cache = self.master_cache.read().unwrap();
// Index entities that are only in master, or merge if they are in both
for (name, e) in &cache.0.entities {
if let Some(session_e) = session.entities.get(name) {
let mut merged_e = e.clone();
for obs in &session_e.observations {
if !merged_e.observations.contains(obs) {
merged_e.observations.push(obs.clone());
}
}
let _ = new_idx.index_entity(&merged_e);
} else {
let _ = new_idx.index_entity(e);
}
}
// Index entities that are only in session
for (name, session_e) in &session.entities {
if !cache.0.entities.contains_key(name) {
let _ = new_idx.index_entity(session_e);
}
}
for t in self.tasks.read() {
let _ = new_idx.index_task(&t);
}
@@ -200,6 +225,7 @@ impl MemoryState {
for a in self.adrs.read() {
let _ = new_idx.index_adr(&a);
}
let _ = new_idx.commit();
if let Ok(mut w) = self.search_index.write() {
*w = new_idx;
}