From c8dd64331a7800b0720a5c4ab41f2271cce95d55 Mon Sep 17 00:00:00 2001 From: Robert Fleischmann Date: Tue, 3 Feb 2026 19:27:13 -0500 Subject: [PATCH 1/3] =?UTF-8?q?Simplify=20API:=205=20tools=20=E2=86=92=204?= =?UTF-8?q?=20tools,=2018=20params=20=E2=86=92=206=20params?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit BREAKING CHANGES: - Remove `connections` tool (graph is now internal infrastructure) - Simplify `store`: remove title, tags, source, metadata, expires_at, replace, related - Simplify `recall`: remove tags, min_similarity - Simplify `list`: remove tags - Remove fields from Item struct: title, tags, source, metadata, expires_at New minimal API: - store(content, scope?) → id - recall(query, limit?) → results - list(limit?, scope?) → items - forget(id) → success Added: - Auto-migration: existing databases migrated to new schema on startup - Schema versioning (v2) for future migrations All intelligent features (decay scoring, graph relationships, consolidation, project scoping, chunking) continue working invisibly. Co-Authored-By: Claude Opus 4.5 --- CHANGELOG.md | 20 ++ CLAUDE.md | 32 ++-- README.md | 27 ++- src/consolidation.rs | 21 +- src/db.rs | 435 +++++++++++++++--------------------------- src/item.rs | 140 ++------------ src/lib.rs | 2 +- src/main.rs | 20 +- src/mcp/tools.rs | 444 +------------------------------------------ 9 files changed, 238 insertions(+), 903 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 62203f7..928dcca 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,25 @@ # Changelog +## [0.3.0] - 2026-02-03 + +### Changed +- **BREAKING**: Simplified API from 5 tools to 4 tools (removed `connections`) +- **BREAKING**: Simplified `store` from 9 parameters to 2 (`content`, `scope`) +- **BREAKING**: Simplified `recall` from 4 parameters to 2 (`query`, `limit`) +- **BREAKING**: Simplified `list` from 3 parameters to 2 (`limit`, `scope`) +- **BREAKING**: Removed fields from Item: `title`, `tags`, `source`, `metadata`, `expires_at` + +### Added +- Auto-migration: existing databases automatically migrated to new schema on startup +- Schema versioning for future migrations + +### Removed +- `connections` tool (graph is now internal infrastructure) +- Tag-based filtering (semantic search handles categorization) +- Item expiration (simplified lifecycle) +- Replace functionality (use `forget` + `store`) +- Related item linking on store (handled by auto-consolidation) + ## [0.2.3] - 2026-02-03 ### Added diff --git a/CLAUDE.md b/CLAUDE.md index 53f976d..880e0a3 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -52,12 +52,12 @@ Sediment is a semantic memory system for AI agents, running as an MCP (Model Con - **`mod.rs`** - Module exports - **`server.rs`** - stdio JSON-RPC server with shared embedder, graph path, consolidation semaphore -- **`tools.rs`** - 5 MCP tools: `store`, `recall`, `list`, `forget`, `connections` +- **`tools.rs`** - 4 MCP tools: `store`, `recall`, `list`, `forget` - **`protocol.rs`** - MCP protocol types and JSON-RPC handling ### Data Flow -1. **Store**: Content → Embedder (384-dim vector) → LanceDB storage → Graph node creation → Provenance metadata injection → Conflict detection → Consolidation queue → Auto-tag inference +1. **Store**: Content → Embedder (384-dim vector) → LanceDB storage → Graph node creation → Conflict detection → Consolidation queue 2. **Chunking**: Long content (>1000 chars) → Type-aware splitting → Individual chunk embeddings 3. **Recall**: Query → Embedder → Vector similarity search → Project boosting → Decay scoring → Trust-weighted re-ranking → Graph backfill → 1-hop graph expansion → Co-access suggestions → Cross-project flagging → Background consolidation + co-access recording 4. **Consolidation** (background): Queue candidates → >=0.95 similarity: merge (delete old, transfer edges, SUPERSEDES edge) → 0.85-0.95: link (RELATED edge) @@ -74,33 +74,37 @@ Sediment is a semantic memory system for AI agents, running as an MCP (Model Con - **Memory decay scoring**: Recall results re-ranked using freshness (30-day half-life) and access frequency (log-scaled). Tracked in SQLite sidecar since LanceDB is append-oriented. - **Trust-weighted scoring**: `final_score = similarity * freshness * frequency * trust_bonus` where `trust_bonus = 1.0 + 0.05*ln(1+validation_count) + 0.02*edge_count` - **Non-blocking intelligence**: All background tasks (consolidation, co-access tracking, clustering) run as fire-and-forget `tokio::spawn` tasks. Tool responses return immediately. `Semaphore(1)` prevents concurrent consolidation. -- **Auto-provenance**: Every store injects `metadata._provenance` with version, project_path, and supersedes chain - **Lazy graph backfill**: Pre-existing items get graph nodes when they appear in recall results -- **Auto-tagging**: Items stored without tags inherit `auto:` prefixed tags from 2+ similar items sharing the same tag +- **Auto-migration**: Database schema is automatically migrated on startup when upgrading from older versions ## MCP Tools Reference -The 5-tool API is defined in `src/mcp/tools.rs`: +The 4-tool API is defined in `src/mcp/tools.rs`: | Tool | Purpose | |------|---------| -| `store` | Store content with optional title, tags, metadata, expiration, scope, replace, related | -| `recall` | Semantic search with decay scoring, trust weighting, graph expansion, co-access suggestions, cross-project flagging | -| `list` | List items by scope (project/global/all) with tag filtering | +| `store` | Store content with optional scope (project/global) | +| `recall` | Semantic search with decay scoring, trust weighting, graph expansion | +| `list` | List items by scope (project/global/all) | | `forget` | Delete item by ID (removes from LanceDB and graph) | -| `connections` | Show full relationship graph for an item (RELATED, SUPERSEDES, CO_ACCESSED edges with content previews) | ### Store Parameters -- `content` (required), `title`, `tags`, `source`, `metadata`, `expires_at`, `scope` (project/global), `replace` (atomically replace item by ID), `related` (array of item IDs to link in graph) +- `content` (required) — The content to store +- `scope` (optional, default: "project") — Where to store: "project" or "global" + +### Recall Parameters +- `query` (required) — Semantic search query +- `limit` (optional, default: 5) — Maximum number of results + +### List Parameters +- `limit` (optional, default: 10) — Maximum number of results +- `scope` (optional, default: "project") — Which items to list: "project", "global", or "all" ### Recall Response Fields -- `results[]` — standard results with `similarity`, `related_ids`, optional `cross_project` + `project_path` flags +- `results[]` — standard results with `similarity`, `related_ids`, optional `cross_project` flag - `graph_expanded[]` — 1-hop neighbors from graph not in original results (marked `graph_expanded: true`) - `suggested[]` — items frequently co-recalled with top results (co-access count >= 3) -### Connections Response -- `item_id`, `connections[]` — each with `id`, `type` (related/supersedes/co_accessed), `strength`, optional `count`, `content_preview` - ## SQLite Schema (access.db) ```sql diff --git a/README.md b/README.md index 00380b2..79b4267 100644 --- a/README.md +++ b/README.md @@ -12,7 +12,7 @@ Combines vector search, a relationship graph, and access tracking into a unified - **Single binary, zero config** — no Docker, no Postgres, no Qdrant. Just `sediment`. - **Sub-16ms recall** — local embeddings and vector search at 100 items, no network round-trips. -- **5-tool focused API** — `store`, `recall`, `list`, `forget`, `connections`. That's it. +- **4-tool focused API** — `store`, `recall`, `list`, `forget`. That's it. - **Works everywhere** — macOS (Intel + ARM), Linux x86_64. All data stays on your machine. ### Comparison @@ -21,7 +21,7 @@ Combines vector search, a relationship graph, and access tracking into a unified |---|---|---|---| | Install | Single binary | Docker + Postgres + Qdrant | Python + pip | | Dependencies | None | 3 services | Python runtime + deps | -| Tools | 5 | 10+ | 24 | +| Tools | 4 | 10+ | 24 | | Embeddings | Local (all-MiniLM-L6-v2) | API-dependent | API-dependent | | Graph features | Built-in | No | No | | Memory decay | Built-in | No | No | @@ -131,13 +131,12 @@ Go to **Settings > Tools > AI Assistant > MCP Servers**, click **+**, and add: ## Tools -| Tool | Description | -|------|-------------| -| `store` | Save content with optional title, tags, source, metadata, expiration, scope, replace, and related item links | -| `recall` | Search memories by semantic similarity with decay scoring, trust weighting, graph expansion, and co-access suggestions | -| `list` | List stored items by scope (project/global/all) with tag filtering | -| `forget` | Delete an item by ID (removes from vector store and graph) | -| `connections` | Show relationship graph for an item (related, supersedes, co-accessed edges) | +| Tool | Parameters | Description | +|------|------------|-------------| +| `store` | `content`, `scope?` | Save content to memory | +| `recall` | `query`, `limit?` | Search by semantic similarity | +| `list` | `limit?`, `scope?` | List stored items | +| `forget` | `id` | Delete an item by ID | ## CLI @@ -164,19 +163,19 @@ All local, embedded, zero config: - **Project scoping**: Automatic context isolation between projects. Same-project items get a similarity boost. - **Relationship graph**: Items linked via RELATED, SUPERSEDES, and CO_ACCESSED edges. Recall expands results with 1-hop graph neighbors and co-access suggestions. - **Background consolidation**: Near-duplicates (≥0.95 similarity) auto-merged; similar items (0.85–0.95) linked. -- **Auto-tagging**: Items without tags inherit tags from similar existing items. - **Type-aware chunking**: Intelligent splitting for markdown, code, JSON, YAML, and plain text. - **Conflict detection**: Items with ≥0.85 similarity flagged on store. -- **Cross-project recall**: Results from other projects flagged with provenance metadata. +- **Cross-project recall**: Results from other projects flagged. - **Local embeddings**: all-MiniLM-L6-v2 via Candle (384-dim vectors, no API keys). - **Model integrity**: SHA-256 verification of all model files on every load, pinned to a specific revision. +- **Auto-migration**: Database schema automatically migrated when upgrading from older versions. ### Security -- **Input bounds**: Content (1MB), queries (100KB), JSON-RPC lines (10MB), tags (50×200B), metadata (100KB). -- **Rate limiting**: 60 tool calls per minute. +- **Input bounds**: Content (1MB), queries (10KB), JSON-RPC lines (10MB). +- **Rate limiting**: 600 tool calls per minute. - **SQL injection prevention**: Sanitized filter expressions for LanceDB; parameterized queries for SQLite. -- **Cross-project access control**: Replace, forget, and connections enforce project isolation. Cross-project content is redacted in recall results. +- **Cross-project access control**: Forget enforces project isolation. Cross-project content is flagged in recall results. - **Error sanitization**: Internal errors logged to stderr; only generic messages returned to MCP clients. - **Retry with backoff**: Transient failures retried with exponential backoff (3 attempts, 100ms–2s). diff --git a/src/consolidation.rs b/src/consolidation.rs index fa02b98..7344a67 100644 --- a/src/consolidation.rs +++ b/src/consolidation.rs @@ -315,21 +315,12 @@ async fn process_candidate( tracing::warn!("add_related_edge failed: {}", e); } - // Soft-delete: mark item as expired instead of hard-deleting. - // This allows recovery; expired items are excluded from search - // results by default but remain in the database. - let past = chrono::Utc::now() - chrono::Duration::seconds(1); - if let Err(e) = db.expire_item(&remove.id, past).await { - // expire_item is delete-then-insert, so a failure may mean the - // item was already deleted. Only attempt hard delete if the item - // still exists to avoid double-deleting. - warn!("expire_item failed ({}), checking if item still exists", e); - if let Ok(Some(_)) = db.get_item(&remove.id).await { - db.delete_item(&remove.id).await?; - if let Err(e2) = graph.remove_node(&remove.id) { - tracing::warn!("remove_node failed: {}", e2); - } - } + // Delete the duplicate item + if let Err(e) = db.delete_item(&remove.id).await { + warn!("delete_item failed: {}", e); + } + if let Err(e) = graph.remove_node(&remove.id) { + tracing::warn!("remove_node failed: {}", e); } Ok("merged".to_string()) diff --git a/src/db.rs b/src/db.rs index 050530a..dc7ee4b 100644 --- a/src/db.rs +++ b/src/db.rs @@ -80,18 +80,16 @@ pub struct DatabaseStats { pub chunk_count: usize, } +/// Current schema version. Increment when making breaking schema changes. +const SCHEMA_VERSION: i32 = 2; + // Arrow schema builders fn item_schema() -> Schema { Schema::new(vec![ Field::new("id", DataType::Utf8, false), Field::new("content", DataType::Utf8, false), - Field::new("title", DataType::Utf8, true), - Field::new("tags", DataType::Utf8, true), // JSON array as string - Field::new("source", DataType::Utf8, true), - Field::new("metadata", DataType::Utf8, true), // JSON as string Field::new("project_id", DataType::Utf8, true), Field::new("is_chunked", DataType::Boolean, false), - Field::new("expires_at", DataType::Int64, true), // Unix timestamp Field::new("created_at", DataType::Int64, false), // Unix timestamp Field::new( "vector", @@ -194,7 +192,7 @@ impl Database { self.project_id.as_deref() } - /// Ensure all required tables exist + /// Ensure all required tables exist, migrating schema if needed async fn ensure_tables(&mut self) -> Result<()> { // Check for existing tables let table_names = self @@ -204,6 +202,15 @@ impl Database { .await .map_err(|e| SedimentError::Database(format!("Failed to list tables: {}", e)))?; + // Check if migration is needed (items table exists but has old schema) + if table_names.contains(&"items".to_string()) { + let needs_migration = self.check_needs_migration().await?; + if needs_migration { + info!("Migrating database schema to version {}", SCHEMA_VERSION); + self.migrate_schema().await?; + } + } + // Items table if table_names.contains(&"items".to_string()) { self.items_table = @@ -223,6 +230,138 @@ impl Database { Ok(()) } + /// Check if the database needs migration by checking for old schema columns + async fn check_needs_migration(&self) -> Result { + let table = self + .db + .open_table("items") + .execute() + .await + .map_err(|e| SedimentError::Database(format!("Failed to open items for check: {}", e)))?; + + let schema = table.schema().await.map_err(|e| { + SedimentError::Database(format!("Failed to get schema: {}", e)) + })?; + + // Old schema has 'tags' column, new schema doesn't + let has_tags = schema.fields().iter().any(|f| f.name() == "tags"); + Ok(has_tags) + } + + /// Migrate from old schema to new schema + async fn migrate_schema(&mut self) -> Result<()> { + info!("Starting schema migration..."); + + // Open old table + let old_table = self + .db + .open_table("items") + .execute() + .await + .map_err(|e| SedimentError::Database(format!("Failed to open old items: {}", e)))?; + + // Read all items from old table + let results = old_table + .query() + .execute() + .await + .map_err(|e| SedimentError::Database(format!("Migration query failed: {}", e)))? + .try_collect::>() + .await + .map_err(|e| SedimentError::Database(format!("Migration collect failed: {}", e)))?; + + // Convert old items to new format + let mut new_batches = Vec::new(); + for batch in &results { + let converted = self.convert_batch_to_new_schema(batch)?; + new_batches.push(converted); + } + + let item_count: usize = new_batches.iter().map(|b| b.num_rows()).sum(); + info!("Migrating {} items to new schema", item_count); + + // Drop old table + self.db.drop_table("items").await.map_err(|e| { + SedimentError::Database(format!("Failed to drop old items table: {}", e)) + })?; + + // Create new table with new schema + let schema = Arc::new(item_schema()); + let new_table = self + .db + .create_empty_table("items", schema.clone()) + .execute() + .await + .map_err(|e| { + SedimentError::Database(format!("Failed to create new items table: {}", e)) + })?; + + // Insert migrated data + if !new_batches.is_empty() { + let batches = + RecordBatchIterator::new(new_batches.into_iter().map(Ok), schema); + new_table + .add(Box::new(batches)) + .execute() + .await + .map_err(|e| { + SedimentError::Database(format!("Failed to insert migrated items: {}", e)) + })?; + } + + info!("Schema migration completed successfully"); + Ok(()) + } + + /// Convert a batch from old schema to new schema + fn convert_batch_to_new_schema(&self, batch: &RecordBatch) -> Result { + let schema = Arc::new(item_schema()); + + // Extract columns from old batch (handle missing columns gracefully) + let id_col = batch + .column_by_name("id") + .ok_or_else(|| SedimentError::Database("Missing id column".to_string()))? + .clone(); + + let content_col = batch + .column_by_name("content") + .ok_or_else(|| SedimentError::Database("Missing content column".to_string()))? + .clone(); + + let project_id_col = batch + .column_by_name("project_id") + .ok_or_else(|| SedimentError::Database("Missing project_id column".to_string()))? + .clone(); + + let is_chunked_col = batch + .column_by_name("is_chunked") + .ok_or_else(|| SedimentError::Database("Missing is_chunked column".to_string()))? + .clone(); + + let created_at_col = batch + .column_by_name("created_at") + .ok_or_else(|| SedimentError::Database("Missing created_at column".to_string()))? + .clone(); + + let vector_col = batch + .column_by_name("vector") + .ok_or_else(|| SedimentError::Database("Missing vector column".to_string()))? + .clone(); + + RecordBatch::try_new( + schema, + vec![ + id_col, + content_col, + project_id_col, + is_chunked_col, + created_at_col, + vector_col, + ], + ) + .map_err(|e| SedimentError::Database(format!("Failed to create migrated batch: {}", e))) + } + /// Ensure vector indexes exist on tables with enough rows. /// /// LanceDB requires at least 256 rows before creating an index. @@ -425,25 +564,13 @@ impl Database { let mut results_map: std::collections::HashMap = std::collections::HashMap::new(); - // Search items table directly (for non-chunked items and chunked items by title) + // Search items table directly (for non-chunked items and chunked items) if let Some(table) = &self.items_table { - let mut filter_parts = Vec::new(); - - if !filters.include_expired { - let now = Utc::now().timestamp(); - filter_parts.push(format!("(expires_at IS NULL OR expires_at > {})", now)); - } - - let mut query_builder = table + let query_builder = table .vector_search(query_embedding.clone()) .map_err(|e| SedimentError::Database(format!("Failed to build search: {}", e)))? .limit(limit * 2); - if !filter_parts.is_empty() { - let filter_str = filter_parts.join(" AND "); - query_builder = query_builder.only_if(filter_str); - } - let results = query_builder .execute() .await @@ -468,13 +595,6 @@ impl Database { continue; } - // Apply tag filter - if let Some(ref filter_tags) = filters.tags - && !filter_tags.iter().any(|t| item.tags.contains(t)) - { - continue; - } - // Apply project boosting let boosted_similarity = boost_similarity( similarity, @@ -541,13 +661,6 @@ impl Database { // Fetch parent items for chunk matches for (item_id, (excerpt, chunk_similarity)) in chunk_matches { if let Some(item) = self.get_item(&item_id).await? { - // Apply tag filter - if let Some(ref filter_tags) = filters.tags - && !filter_tags.iter().any(|t| item.tags.contains(t)) - { - continue; - } - // Apply project boosting let boosted_similarity = boost_similarity( chunk_similarity, @@ -603,15 +716,10 @@ impl Database { None => return Ok(Vec::new()), }; - // Build filter for non-expired items - let now = Utc::now().timestamp(); - let filter = format!("(expires_at IS NULL OR expires_at > {})", now); - let results = table .vector_search(embedding) .map_err(|e| SedimentError::Database(format!("Failed to build search: {}", e)))? .limit(limit) - .only_if(filter) .execute() .await .map_err(|e| SedimentError::Database(format!("Search failed: {}", e)))? @@ -654,7 +762,7 @@ impl Database { /// List items with optional filters pub async fn list_items( &mut self, - filters: ItemFilters, + _filters: ItemFilters, limit: Option, scope: crate::ListScope, ) -> Result> { @@ -665,11 +773,6 @@ impl Database { let mut filter_parts = Vec::new(); - if !filters.include_expired { - let now = Utc::now().timestamp(); - filter_parts.push(format!("(expires_at IS NULL OR expires_at > {})", now)); - } - // Apply scope filter match scope { crate::ListScope::Project => { @@ -712,11 +815,6 @@ impl Database { items.extend(batch_to_items(&batch)?); } - // Apply tag filter - if let Some(ref filter_tags) = filters.tags { - items.retain(|item| filter_tags.iter().any(|t| item.tags.contains(t))); - } - Ok(items) } @@ -790,76 +888,6 @@ impl Database { Ok(items) } - /// Soft-delete an item by setting its expiration to a past timestamp. - /// The item remains in the database but is excluded from search results. - /// - /// Uses delete-then-insert because both rows share the same ID. If the - /// re-insert fails, retries up to 3 times to avoid data loss. - pub async fn expire_item(&self, id: &str, expires_at: chrono::DateTime) -> Result<()> { - if !is_valid_id(id) { - return Err(SedimentError::Database("Invalid item ID".to_string())); - } - let table = match &self.items_table { - Some(t) => t, - None => return Err(SedimentError::Database("Items table not found".to_string())), - }; - - // Read the item first so we have a full copy for recovery - let original_item = self.get_item(id).await?; - let original_item = match original_item { - Some(i) => i, - None => return Err(SedimentError::Database(format!("Item not found: {}", id))), - }; - - let mut item = original_item.clone(); - item.expires_at = Some(expires_at); - - // Delete first, then re-insert with updated expires_at. - // We must delete-then-insert (not insert-then-delete) because both rows - // share the same id, so a delete filter on id would remove both. - table - .delete(&format!("id = '{}'", sanitize_sql_string(id))) - .await - .map_err(|e| SedimentError::Database(format!("Delete for expire failed: {}", e)))?; - - // Re-insert with retries to prevent data loss if insert fails after delete - let mut last_err = None; - for attempt in 0..3 { - let batch = item_to_batch(&item)?; - let batches = RecordBatchIterator::new(vec![Ok(batch)], Arc::new(item_schema())); - match table.add(Box::new(batches)).execute().await { - Ok(_) => return Ok(()), - Err(e) => { - tracing::warn!( - "Re-insert for expire failed (attempt {}/3): {}", - attempt + 1, - e - ); - last_err = Some(e); - tokio::time::sleep(std::time::Duration::from_millis(100 * (1 << attempt))) - .await; - } - } - } - - // Emergency recovery: re-insert the original item to prevent data loss - tracing::error!("expire_item: re-insert failed after 3 attempts, attempting recovery"); - let batch = item_to_batch(&original_item)?; - let batches = RecordBatchIterator::new(vec![Ok(batch)], Arc::new(item_schema())); - if let Err(recovery_err) = table.add(Box::new(batches)).execute().await { - tracing::error!( - "expire_item: CRITICAL - recovery also failed, item {} may be lost: {}", - id, - recovery_err - ); - } - - Err(SedimentError::Database(format!( - "Re-insert for expire failed after 3 attempts: {}", - last_err.unwrap() - ))) - } - /// Delete an item and its chunks. /// Returns `true` if the item existed, `false` if it was not found. pub async fn delete_item(&self, id: &str) -> Result { @@ -915,86 +943,6 @@ impl Database { Ok(stats) } - /// Delete items whose expires_at timestamp is in the past. - pub async fn cleanup_expired(&self) -> Result { - let table = match &self.items_table { - Some(t) => t, - None => return Ok(0), - }; - - let now = Utc::now().timestamp(); - // now is a system-generated i64 timestamp, no string sanitization needed - let filter = format!("expires_at IS NOT NULL AND expires_at < {}", now); - - // Count how many will be deleted - let count = table.count_rows(Some(filter.clone())).await.unwrap_or(0); - - if count > 0 { - // First, find the IDs of expired items so we can clean up their chunks - if let Ok(expired_ids) = self.get_expired_item_ids(now).await - && let Some(ref chunks_table) = self.chunks_table - { - for item_id in &expired_ids { - let chunk_filter = format!("item_id = '{}'", sanitize_sql_string(item_id)); - if let Err(e) = chunks_table.delete(&chunk_filter).await { - tracing::warn!( - "Failed to delete chunks for expired item {}: {}", - item_id, - e - ); - } - } - } - - table - .delete(&filter) - .await - .map_err(|e| SedimentError::Database(format!("Expired cleanup failed: {}", e)))?; - - info!("Cleaned up {} expired items and their chunks", count); - } - - Ok(count) - } - - /// Get IDs of items that have expired (helper for cleanup) - async fn get_expired_item_ids(&self, now_ts: i64) -> Result> { - let table = match &self.items_table { - Some(t) => t, - None => return Ok(vec![]), - }; - - let filter = format!("expires_at IS NOT NULL AND expires_at < {}", now_ts); - let results = table - .query() - .only_if(filter) - .select(lancedb::query::Select::Columns(vec!["id".to_string()])) - .execute() - .await - .map_err(|e| SedimentError::Database(format!("Query expired IDs failed: {}", e)))?; - - let batches = results - .try_collect::>() - .await - .map_err(|e| SedimentError::Database(format!("Collect expired IDs failed: {}", e)))?; - - let mut ids = Vec::new(); - for batch in &batches { - if let Some(id_col) = batch.column_by_name("id") { - let id_array = match id_col.as_any().downcast_ref::() { - Some(arr) => arr, - None => continue, // Skip batch if column type is unexpected - }; - for i in 0..id_array.len() { - if !id_array.is_null(i) { - ids.push(id_array.value(i).to_string()); - } - } - } - } - - Ok(ids) - } } // ==================== Decay Scoring ==================== @@ -1113,13 +1061,8 @@ fn item_to_batch(item: &Item) -> Result { let id = StringArray::from(vec![item.id.as_str()]); let content = StringArray::from(vec![item.content.as_str()]); - let title = StringArray::from(vec![item.title.as_deref()]); - let tags = StringArray::from(vec![serde_json::to_string(&item.tags).ok()]); - let source = StringArray::from(vec![item.source.as_deref()]); - let metadata = StringArray::from(vec![item.metadata.as_ref().map(|m| m.to_string())]); let project_id = StringArray::from(vec![item.project_id.as_deref()]); let is_chunked = BooleanArray::from(vec![item.is_chunked]); - let expires_at = Int64Array::from(vec![item.expires_at.map(|t| t.timestamp())]); let created_at = Int64Array::from(vec![item.created_at.timestamp()]); let vector = create_embedding_array(&item.embedding)?; @@ -1129,13 +1072,8 @@ fn item_to_batch(item: &Item) -> Result { vec![ Arc::new(id), Arc::new(content), - Arc::new(title), - Arc::new(tags), - Arc::new(source), - Arc::new(metadata), Arc::new(project_id), Arc::new(is_chunked), - Arc::new(expires_at), Arc::new(created_at), Arc::new(vector), ], @@ -1156,22 +1094,6 @@ fn batch_to_items(batch: &RecordBatch) -> Result> { .and_then(|c| c.as_any().downcast_ref::()) .ok_or_else(|| SedimentError::Database("Missing content column".to_string()))?; - let title_col = batch - .column_by_name("title") - .and_then(|c| c.as_any().downcast_ref::()); - - let tags_col = batch - .column_by_name("tags") - .and_then(|c| c.as_any().downcast_ref::()); - - let source_col = batch - .column_by_name("source") - .and_then(|c| c.as_any().downcast_ref::()); - - let metadata_col = batch - .column_by_name("metadata") - .and_then(|c| c.as_any().downcast_ref::()); - let project_id_col = batch .column_by_name("project_id") .and_then(|c| c.as_any().downcast_ref::()); @@ -1180,10 +1102,6 @@ fn batch_to_items(batch: &RecordBatch) -> Result> { .column_by_name("is_chunked") .and_then(|c| c.as_any().downcast_ref::()); - let expires_at_col = batch - .column_by_name("expires_at") - .and_then(|c| c.as_any().downcast_ref::()); - let created_at_col = batch .column_by_name("created_at") .and_then(|c| c.as_any().downcast_ref::()); @@ -1196,40 +1114,6 @@ fn batch_to_items(batch: &RecordBatch) -> Result> { let id = id_col.value(i).to_string(); let content = content_col.value(i).to_string(); - let title = title_col.and_then(|c| { - if c.is_null(i) { - None - } else { - Some(c.value(i).to_string()) - } - }); - - let tags: Vec = tags_col - .and_then(|c| { - if c.is_null(i) { - None - } else { - serde_json::from_str(c.value(i)).ok() - } - }) - .unwrap_or_default(); - - let source = source_col.and_then(|c| { - if c.is_null(i) { - None - } else { - Some(c.value(i).to_string()) - } - }); - - let metadata = metadata_col.and_then(|c| { - if c.is_null(i) { - None - } else { - serde_json::from_str(c.value(i)).ok() - } - }); - let project_id = project_id_col.and_then(|c| { if c.is_null(i) { None @@ -1240,18 +1124,6 @@ fn batch_to_items(batch: &RecordBatch) -> Result> { let is_chunked = is_chunked_col.map(|c| c.value(i)).unwrap_or(false); - let expires_at = expires_at_col.and_then(|c| { - if c.is_null(i) { - None - } else { - Some( - Utc.timestamp_opt(c.value(i), 0) - .single() - .unwrap_or_else(Utc::now), - ) - } - }); - let created_at = created_at_col .map(|c| { Utc.timestamp_opt(c.value(i), 0) @@ -1274,13 +1146,8 @@ fn batch_to_items(batch: &RecordBatch) -> Result> { id, content, embedding, - title, - tags, - source, - metadata, project_id, is_chunked, - expires_at, created_at, }; @@ -1533,4 +1400,10 @@ mod tests { let should_chunk = long_content.chars().count() > CHUNK_THRESHOLD; assert!(should_chunk, "1001 chars should exceed 1000-char threshold"); } + + #[test] + fn test_schema_version() { + // Ensure schema version is set + assert!(SCHEMA_VERSION >= 2, "Schema version should be at least 2"); + } } diff --git a/src/item.rs b/src/item.rs index 34528e4..fa9237c 100644 --- a/src/item.rs +++ b/src/item.rs @@ -4,7 +4,6 @@ use chrono::{DateTime, Utc}; use serde::{Deserialize, Serialize}; -use serde_json::Value; /// A unified item stored in Sediment #[derive(Debug, Clone, Serialize, Deserialize)] @@ -16,26 +15,11 @@ pub struct Item { /// Vector embedding (not serialized to JSON output) #[serde(skip)] pub embedding: Vec, - /// Optional title (recommended for long content) - #[serde(skip_serializing_if = "Option::is_none")] - pub title: Option, - /// Tags for categorization - #[serde(default, skip_serializing_if = "Vec::is_empty")] - pub tags: Vec, - /// Source attribution - #[serde(skip_serializing_if = "Option::is_none")] - pub source: Option, - /// Custom JSON metadata - #[serde(skip_serializing_if = "Option::is_none")] - pub metadata: Option, /// Project ID (None for global items) #[serde(skip_serializing_if = "Option::is_none")] pub project_id: Option, /// Whether this item was chunked (internal) pub is_chunked: bool, - /// When this item expires (optional) - #[serde(skip_serializing_if = "Option::is_none")] - pub expires_at: Option>, /// When this item was created pub created_at: DateTime, } @@ -47,53 +31,18 @@ impl Item { id: uuid::Uuid::new_v4().to_string(), content: content.into(), embedding: Vec::new(), - title: None, - tags: Vec::new(), - source: None, - metadata: None, project_id: None, is_chunked: false, - expires_at: None, created_at: Utc::now(), } } - /// Set the title - pub fn with_title(mut self, title: impl Into) -> Self { - self.title = Some(title.into()); - self - } - - /// Set tags - pub fn with_tags(mut self, tags: Vec) -> Self { - self.tags = tags; - self - } - - /// Set the source - pub fn with_source(mut self, source: impl Into) -> Self { - self.source = Some(source.into()); - self - } - - /// Set custom metadata - pub fn with_metadata(mut self, metadata: Value) -> Self { - self.metadata = Some(metadata); - self - } - /// Set the project ID pub fn with_project_id(mut self, project_id: impl Into) -> Self { self.project_id = Some(project_id.into()); self } - /// Set expiration time - pub fn with_expires_at(mut self, expires_at: DateTime) -> Self { - self.expires_at = Some(expires_at); - self - } - /// Set the embedding pub fn with_embedding(mut self, embedding: Vec) -> Self { self.embedding = embedding; @@ -107,28 +56,15 @@ impl Item { } /// Get the text to embed for this item - /// For chunked items: title + first ~500 chars + /// For chunked items: first ~500 chars /// For non-chunked items: full content pub fn embedding_text(&self) -> String { if self.is_chunked { - let preview: String = self.content.chars().take(500).collect(); - match &self.title { - Some(title) => format!("{} {}", title, preview), - None => preview, - } + self.content.chars().take(500).collect() } else { self.content.clone() } } - - /// Check if this item has expired - pub fn is_expired(&self) -> bool { - if let Some(expires_at) = self.expires_at { - Utc::now() > expires_at - } else { - false - } - } } /// A chunk of an item (internal, not exposed to MCP) @@ -202,27 +138,18 @@ pub struct ConflictInfo { pub struct SearchResult { /// The matching item's id pub id: String, - /// Content (full if short, or title if chunked) + /// Content (full if short, or preview if chunked) pub content: String, /// Most relevant chunk content (if chunked) #[serde(skip_serializing_if = "Option::is_none")] pub relevant_excerpt: Option, /// Similarity score (0.0-1.0, higher is more similar) pub similarity: f32, - /// Tags - #[serde(default, skip_serializing_if = "Vec::is_empty")] - pub tags: Vec, - /// Source attribution - #[serde(skip_serializing_if = "Option::is_none")] - pub source: Option, /// When created pub created_at: DateTime, /// Project ID (not serialized, used internally for cross-project checks and graph backfill) #[serde(skip)] pub project_id: Option, - /// Metadata (not serialized, used internally for cross-project provenance) - #[serde(skip)] - pub metadata: Option, } impl SearchResult { @@ -233,30 +160,22 @@ impl SearchResult { content: item.content.clone(), relevant_excerpt: None, similarity, - tags: item.tags.clone(), - source: item.source.clone(), created_at: item.created_at, project_id: item.project_id.clone(), - metadata: item.metadata.clone(), } } /// Create from an item with chunk excerpt pub fn from_item_with_excerpt(item: &Item, similarity: f32, excerpt: String) -> Self { - let content = item.title.clone().unwrap_or_else(|| { - // For chunked items without title, show a preview - item.content.chars().take(100).collect() - }); + // For chunked items, show a preview of the content + let content: String = item.content.chars().take(100).collect(); Self { id: item.id.clone(), content, relevant_excerpt: Some(excerpt), similarity, - tags: item.tags.clone(), - source: item.source.clone(), created_at: item.created_at, project_id: item.project_id.clone(), - metadata: item.metadata.clone(), } } } @@ -264,12 +183,8 @@ impl SearchResult { /// Filters for search/list queries #[derive(Debug, Default, Clone)] pub struct ItemFilters { - /// Filter by tags (any match) - pub tags: Option>, /// Minimum similarity threshold (0.0-1.0) pub min_similarity: Option, - /// Include expired items - pub include_expired: bool, } impl ItemFilters { @@ -277,20 +192,10 @@ impl ItemFilters { Self::default() } - pub fn with_tags(mut self, tags: Vec) -> Self { - self.tags = Some(tags); - self - } - pub fn with_min_similarity(mut self, min_similarity: f32) -> Self { self.min_similarity = Some(min_similarity); self } - - pub fn include_expired(mut self, include: bool) -> Self { - self.include_expired = include; - self - } } #[cfg(test)] @@ -299,16 +204,9 @@ mod tests { #[test] fn test_item_creation() { - let item = Item::new("Test content") - .with_title("Test Title") - .with_tags(vec!["tag1".to_string(), "tag2".to_string()]) - .with_source("test-source") - .with_project_id("project-123"); + let item = Item::new("Test content").with_project_id("project-123"); assert_eq!(item.content, "Test content"); - assert_eq!(item.title, Some("Test Title".to_string())); - assert_eq!(item.tags, vec!["tag1", "tag2"]); - assert_eq!(item.source, Some("test-source".to_string())); assert_eq!(item.project_id, Some("project-123".to_string())); assert!(!item.is_chunked); } @@ -321,21 +219,9 @@ mod tests { #[test] fn test_embedding_text_chunked() { - let item = Item::new("a".repeat(1000)) - .with_title("My Title") - .with_chunked(true); + let item = Item::new("a".repeat(1000)).with_chunked(true); let text = item.embedding_text(); - assert!(text.starts_with("My Title ")); - assert!(text.len() < 600); - } - - #[test] - fn test_item_expiration() { - let expired = Item::new("Expired").with_expires_at(Utc::now() - chrono::Duration::hours(1)); - assert!(expired.is_expired()); - - let valid = Item::new("Valid").with_expires_at(Utc::now() + chrono::Duration::hours(1)); - assert!(!valid.is_expired()); + assert_eq!(text.len(), 500); } #[test] @@ -350,9 +236,7 @@ mod tests { #[test] fn test_search_result_from_item() { - let item = Item::new("Test content") - .with_tags(vec!["test".to_string()]) - .with_source("test"); + let item = Item::new("Test content"); let result = SearchResult::from_item(&item, 0.95); assert_eq!(result.content, "Test content"); @@ -362,12 +246,10 @@ mod tests { #[test] fn test_search_result_with_excerpt() { - let item = Item::new("Long content here") - .with_title("Document Title") - .with_chunked(true); + let item = Item::new("Long content here").with_chunked(true); let result = SearchResult::from_item_with_excerpt(&item, 0.85, "relevant part".to_string()); - assert_eq!(result.content, "Document Title"); + assert_eq!(result.content, "Long content here"); assert_eq!(result.relevant_excerpt, Some("relevant part".to_string())); } diff --git a/src/lib.rs b/src/lib.rs index dc87006..4d90de8 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -6,7 +6,7 @@ //! //! - **Embedded storage** - LanceDB-powered, directory-based, no server required //! - **Local embeddings** - Uses `all-MiniLM-L6-v2` locally, no API keys needed -//! - **MCP-native** - 5 tools for seamless LLM integration +//! - **MCP-native** - 4 tools for seamless LLM integration //! - **Project-aware** - Scoped memories with automatic project detection //! - **Auto-chunking** - Long content is automatically chunked for better search diff --git a/src/main.rs b/src/main.rs index bc0b397..e276862 100644 --- a/src/main.rs +++ b/src/main.rs @@ -314,20 +314,11 @@ fn run_list(db_override: Option, limit: usize) -> Result<()> { println!("Stored items ({}):\n", items.len()); for item in items { - let title = item.title.as_deref().unwrap_or("(untitled)"); let scope = if item.project_id.is_some() { "project" } else { "global" }; - let tags = if item.tags.is_empty() { - String::new() - } else { - format!(" [{}]", item.tags.join(", ")) - }; - - println!(" {} ({}){}", title, scope, tags); - println!(" ID: {}", item.id); // Show truncated content let content_preview: String = item @@ -341,7 +332,9 @@ fn run_list(db_override: Option, limit: usize) -> Result<()> { } else { "" }; - println!(" Content: {}{}", content_preview, ellipsis); + + println!(" {} ({})", item.id, scope); + println!(" {}{}", content_preview, ellipsis); println!(); } @@ -355,13 +348,12 @@ fn generate_claude_md_instructions() -> String { Use the Sediment MCP tools for persistent memory storage. -## Tools (5 total) +## Tools (4 total) - `mcp__sediment__store` - Store content for later retrieval - `mcp__sediment__recall` - Search by semantic similarity - `mcp__sediment__list` - List stored items - `mcp__sediment__forget` - Delete an item by ID -- `mcp__sediment__connections` - Show relationship graph for an item ## When to Store @@ -374,12 +366,12 @@ Use the Sediment MCP tools for persistent memory storage. Store a preference: ```json -{"content": "User prefers dark mode", "tags": ["preference"]} +{"content": "User prefers dark mode"} ``` Store a document (auto-chunked if long): ```json -{"content": "", "title": "API Reference", "tags": ["docs"]} +{"content": "", "scope": "global"} ``` Search: diff --git a/src/mcp/tools.rs b/src/mcp/tools.rs index 866a4d1..89dd82c 100644 --- a/src/mcp/tools.rs +++ b/src/mcp/tools.rs @@ -1,10 +1,9 @@ //! MCP Tool definitions for Sediment //! -//! 5 tools: store, recall, list, forget, connections +//! 4 tools: store, recall, list, forget use std::sync::Arc; -use chrono::DateTime; use serde::Deserialize; use serde_json::{Value, json}; @@ -30,7 +29,7 @@ fn spawn_logged(name: &'static str, fut: impl std::future::Future + }); } -/// Get all available tools (5 total) +/// Get all available tools (4 total) pub fn get_tools() -> Vec { vec![ Tool { @@ -43,41 +42,11 @@ pub fn get_tools() -> Vec { "type": "string", "description": "The content to store" }, - "title": { - "type": "string", - "description": "Optional title (recommended for long content)" - }, - "tags": { - "type": "array", - "items": { "type": "string" }, - "description": "Tags for categorization" - }, - "source": { - "type": "string", - "description": "Source attribution (e.g., URL, file path, 'conversation')" - }, - "metadata": { - "type": "object", - "description": "Custom JSON metadata" - }, - "expires_at": { - "type": "string", - "description": "ISO datetime when this should expire (optional)" - }, "scope": { "type": "string", "enum": ["project", "global"], "default": "project", "description": "Where to store: 'project' (current project) or 'global' (all projects)" - }, - "replace": { - "type": "string", - "description": "ID of an existing item to replace (stores new item first, then deletes old)" - }, - "related": { - "type": "array", - "items": { "type": "string" }, - "description": "IDs of related items to link in the knowledge graph" } }, "required": ["content"] @@ -97,16 +66,6 @@ pub fn get_tools() -> Vec { "type": "number", "default": 5, "description": "Maximum number of results" - }, - "tags": { - "type": "array", - "items": { "type": "string" }, - "description": "Filter by tags (any match)" - }, - "min_similarity": { - "type": "number", - "default": 0.3, - "description": "Minimum similarity threshold (0.0-1.0). Lower values return more results." } }, "required": ["query"] @@ -114,15 +73,10 @@ pub fn get_tools() -> Vec { }, Tool { name: "list".to_string(), - description: "List stored items with optional filtering.".to_string(), + description: "List stored items.".to_string(), input_schema: json!({ "type": "object", "properties": { - "tags": { - "type": "array", - "items": { "type": "string" }, - "description": "Filter by tags" - }, "limit": { "type": "number", "default": 10, @@ -151,20 +105,6 @@ pub fn get_tools() -> Vec { "required": ["id"] }), }, - Tool { - name: "connections".to_string(), - description: "Show the relationship graph for a stored item. Returns all connections including related items, superseded items, and frequently co-accessed items.".to_string(), - input_schema: json!({ - "type": "object", - "properties": { - "id": { - "type": "string", - "description": "The item ID to show connections for" - } - }, - "required": ["id"] - }), - }, ] } @@ -174,21 +114,7 @@ pub fn get_tools() -> Vec { pub struct StoreParams { pub content: String, #[serde(default)] - pub title: Option, - #[serde(default)] - pub tags: Option>, - #[serde(default)] - pub source: Option, - #[serde(default)] - pub metadata: Option, - #[serde(default)] - pub expires_at: Option, - #[serde(default)] pub scope: Option, - #[serde(default)] - pub replace: Option, - #[serde(default)] - pub related: Option>, } #[derive(Debug, Deserialize)] @@ -196,16 +122,10 @@ pub struct RecallParams { pub query: String, #[serde(default)] pub limit: Option, - #[serde(default)] - pub tags: Option>, - #[serde(default)] - pub min_similarity: Option, } #[derive(Debug, Deserialize)] pub struct ListParams { - #[serde(default)] - pub tags: Option>, #[serde(default)] pub limit: Option, #[serde(default)] @@ -217,11 +137,6 @@ pub struct ForgetParams { pub id: String, } -#[derive(Debug, Deserialize)] -pub struct ConnectionsParams { - pub id: String, -} - // ========== Recall Configuration ========== /// Controls which graph and scoring features are enabled during recall. @@ -289,7 +204,6 @@ pub async fn execute_tool(ctx: &ServerContext, name: &str, args: Option) "recall" => execute_recall(&mut db, &tracker, &graph, ctx_ref, args_clone).await, "list" => execute_list(&mut db, args_clone).await, "forget" => execute_forget(&mut db, &graph, ctx_ref, args_clone).await, - "connections" => execute_connections(&mut db, &graph, ctx_ref, args_clone).await, _ => return Ok(CallToolResult::error(format!("Unknown tool: {}", name_ref))), }; @@ -336,7 +250,7 @@ fn is_retryable_error(error_msg: &str) -> bool { async fn execute_store( db: &mut Database, - tracker: &AccessTracker, + _tracker: &AccessTracker, graph: &GraphStore, ctx: &ServerContext, args: Option, @@ -364,62 +278,6 @@ async fn execute_store( )); } - // Validate field sizes to prevent abuse - const MAX_TITLE_LEN: usize = 1000; - const MAX_SOURCE_LEN: usize = 2000; - const MAX_TAG_LEN: usize = 200; - const MAX_TAG_COUNT: usize = 50; - const MAX_METADATA_BYTES: usize = 100_000; - - if let Some(ref title) = params.title - && title.len() > MAX_TITLE_LEN - { - return CallToolResult::error(format!( - "Title too large: {} bytes (max {})", - title.len(), - MAX_TITLE_LEN - )); - } - - if let Some(ref source) = params.source - && source.len() > MAX_SOURCE_LEN - { - return CallToolResult::error(format!( - "Source too large: {} bytes (max {})", - source.len(), - MAX_SOURCE_LEN - )); - } - - if let Some(ref tags) = params.tags { - if tags.len() > MAX_TAG_COUNT { - return CallToolResult::error(format!( - "Too many tags: {} (max {})", - tags.len(), - MAX_TAG_COUNT - )); - } - for tag in tags { - if tag.len() > MAX_TAG_LEN { - return CallToolResult::error(format!( - "Tag too large: {} bytes (max {})", - tag.len(), - MAX_TAG_LEN - )); - } - } - } - - if let Some(ref metadata) = params.metadata { - let meta_size = metadata.to_string().len(); - if meta_size > MAX_METADATA_BYTES { - return CallToolResult::error(format!( - "Metadata too large: {} bytes (max {})", - meta_size, MAX_METADATA_BYTES - )); - } - } - // Parse scope let scope = params .scope @@ -432,75 +290,8 @@ async fn execute_store( Err(e) => return CallToolResult::error(e), }; - // Parse expires_at if provided - let expires_at = if let Some(ref exp_str) = params.expires_at { - match DateTime::parse_from_rfc3339(exp_str) { - Ok(dt) => Some(dt.with_timezone(&chrono::Utc)), - Err(e) => return CallToolResult::error(format!("Invalid expires_at: {}", e)), - } - } else { - None - }; - - // Validate that the item to replace exists and belongs to the current project - let replaced_id = if let Some(ref replace_id) = params.replace { - match db.get_item(replace_id).await { - Ok(Some(item)) => { - // Access control: prevent replacing items from other projects - if let Some(ref current_pid) = ctx.project_id - && let Some(ref item_pid) = item.project_id - && item_pid != current_pid - { - return CallToolResult::error(format!( - "Cannot replace item {} from a different project", - replace_id - )); - } - Some(replace_id.clone()) - } - Ok(None) => { - return CallToolResult::error(format!( - "Cannot replace: item not found: {}", - replace_id - )); - } - Err(e) => { - return sanitized_error("Failed to look up item for replace", e); - } - } - } else { - None - }; - // Build item - let mut tags = params.tags.unwrap_or_default(); - let mut item = Item::new(¶ms.content).with_tags(tags.clone()); - - if let Some(title) = params.title { - item = item.with_title(title); - } - - if let Some(source) = params.source { - item = item.with_source(source); - } - - // Build metadata with provenance - let mut metadata = params.metadata.unwrap_or(json!({})); - if let Some(obj) = metadata.as_object_mut() { - let mut provenance = json!({ - "v": 1, - "project_path": ctx.cwd.file_name().map(|n| n.to_string_lossy().into_owned()).unwrap_or_else(|| "unknown".to_string()) - }); - if let Some(ref rid) = replaced_id { - provenance["supersedes"] = json!(rid); - } - obj.insert("_provenance".to_string(), provenance); - } - item = item.with_metadata(metadata); - - if let Some(exp) = expires_at { - item = item.with_expires_at(exp); - } + let mut item = Item::new(¶ms.content); // Set project_id based on scope if scope == StoreScope::Project @@ -509,34 +300,6 @@ async fn execute_store( item = item.with_project_id(project_id); } - // Auto-tag inference (Phase 4a): if no user tags, infer from similar items - if tags.is_empty() - && let Ok(similar) = db.find_similar_items(¶ms.content, 0.85, 5).await - { - let mut tag_counts: std::collections::HashMap = - std::collections::HashMap::new(); - for conflict in &similar { - if let Some(similar_item) = db.get_item(&conflict.id).await.ok().flatten() { - for tag in &similar_item.tags { - if !tag.starts_with("auto:") { - *tag_counts.entry(tag.clone()).or_insert(0) += 1; - } - } - } - } - // If 2+ similar items share a tag, auto-apply it - let auto_tags: Vec = tag_counts - .into_iter() - .filter(|(_, count)| *count >= 2) - .map(|(tag, _)| format!("auto:{}", tag)) - .collect(); - if !auto_tags.is_empty() { - tags = item.tags.clone(); - tags.extend(auto_tags); - item = item.with_tags(tags); - } - } - match db.store_item(item).await { Ok(store_result) => { let new_id = store_result.id.clone(); @@ -548,64 +311,6 @@ async fn execute_store( tracing::warn!("graph add_node failed: {}", e); } - // Complete replace: now that the new item is stored, delete the old one. - // Intentionally non-atomic (store-before-delete): if the process crashes - // between store and delete, both items exist (benign duplication, not data - // loss). The consolidation system will detect and merge duplicates. - if let Some(ref old_id) = replaced_id { - // Record validation on the NEW item (the replacement is a "confirmed" version) - let now_ts = chrono::Utc::now().timestamp(); - if let Err(e) = tracker.record_validation(&new_id, now_ts) { - tracing::warn!("record_validation failed: {}", e); - } - // Transfer graph edges from old node to new node before removing old node - // If this fails, abort the replace to avoid inconsistent state - if let Err(e) = graph.transfer_edges(old_id, &new_id) { - tracing::error!("transfer_edges failed, aborting replace: {}", e); - // Compensating action: remove the new item since replace failed - let _ = db.delete_item(&new_id).await; - let _ = graph.remove_node(&new_id); - return CallToolResult::error( - "Replace failed during edge transfer. Original item preserved.", - ); - } - // Create SUPERSEDES edge (non-critical, continue on failure) - if let Err(e) = graph.add_supersedes_edge(&new_id, old_id) { - tracing::warn!("add_supersedes_edge failed: {}", e); - } - // Delete old item from LanceDB - if let Err(e) = db.delete_item(old_id).await { - tracing::error!( - "delete_item failed during replace: {}. Old item may remain as duplicate.", - e - ); - } - // Remove old graph node (and its remaining edges) - if let Err(e) = graph.remove_node(old_id) { - tracing::warn!("remove_node failed: {}", e); - } - } - - // Create RELATED edges if specified (only for IDs that exist in LanceDB) - if let Some(ref related_ids) = params.related { - let rid_refs: Vec<&str> = related_ids.iter().map(|s| s.as_str()).collect(); - let existing_items = db.get_items_batch(&rid_refs).await.unwrap_or_default(); - let valid_ids: std::collections::HashSet<&str> = - existing_items.iter().map(|i| i.id.as_str()).collect(); - - for rid in related_ids { - if !valid_ids.contains(rid.as_str()) { - tracing::warn!("related ID not found, skipping edge: {}", rid); - continue; - } - // Ensure target node exists in graph before creating edge - let _ = graph.ensure_node_exists(rid, None, now); - if let Err(e) = graph.add_related_edge(&new_id, rid, 1.0, "user_linked") { - tracing::warn!("add_related_edge failed: {}", e); - } - } - } - // Enqueue consolidation candidates from conflicts if !store_result.potential_conflicts.is_empty() && let Ok(queue) = ConsolidationQueue::open(&ctx.access_db_path) @@ -861,13 +566,8 @@ async fn execute_recall( } let limit = params.limit.unwrap_or(5).min(100); - let min_similarity = params.min_similarity.unwrap_or(0.3); - - let mut filters = ItemFilters::new().with_min_similarity(min_similarity); - if let Some(tags) = params.tags { - filters = filters.with_tags(tags); - } + let filters = ItemFilters::new(); let config = RecallConfig::default(); @@ -906,28 +606,16 @@ async fn execute_recall( if let Some(ref excerpt) = r.relevant_excerpt { obj["relevant_excerpt"] = json!(excerpt); } - if !r.tags.is_empty() { - obj["tags"] = json!(r.tags); - } - if let Some(ref source) = r.source { - obj["source"] = json!(source); - } - // Cross-project flag (Phase 3c) — uses cached project_id/metadata from SearchResult + // Cross-project flag if let Some(ref current_pid) = ctx.project_id && let Some(ref item_pid) = r.project_id && item_pid != current_pid { obj["cross_project"] = json!(true); - if let Some(ref meta) = r.metadata - && let Some(prov) = meta.get("_provenance") - && let Some(pp) = prov.get("project_path") - { - obj["project_path"] = pp.clone(); - } } - // Related IDs from graph (Phase 1d) + // Related IDs from graph if let Ok(neighbors) = graph.get_neighbors(&[r.id.as_str()], 0.5) { let related: Vec = neighbors.iter().map(|(id, _, _)| id.clone()).collect(); if !related.is_empty() { @@ -1003,21 +691,6 @@ async fn execute_recall( } }); - // Expired item cleanup - let db_path = ctx.db_path.clone(); - let project_id = ctx.project_id.clone(); - let embedder = ctx.embedder.clone(); - spawn_logged("cleanup_expired", async move { - match Database::open_with_embedder(&db_path, project_id, embedder).await { - Ok(db) => { - if let Err(e) = db.cleanup_expired().await { - tracing::warn!("cleanup_expired failed: {}", e); - } - } - Err(e) => tracing::warn!("cleanup_expired: failed to open db: {}", e), - } - }); - // Consolidation queue cleanup: purge old processed entries let access_db_path2 = ctx.access_db_path.clone(); spawn_logged("consolidation_cleanup", async move { @@ -1039,18 +712,13 @@ async fn execute_list(db: &mut Database, args: Option) -> CallToolResult let params: ListParams = args.and_then(|v| serde_json::from_value(v).ok()) .unwrap_or(ListParams { - tags: None, limit: None, scope: None, }); let limit = params.limit.unwrap_or(10).min(100); - let mut filters = ItemFilters::new(); - - if let Some(tags) = params.tags { - filters = filters.with_tags(tags); - } + let filters = ItemFilters::new(); let scope = params .scope @@ -1078,12 +746,6 @@ async fn execute_list(db: &mut Database, args: Option) -> CallToolResult "created": item.created_at.to_rfc3339(), }); - if let Some(ref title) = item.title { - obj["title"] = json!(title); - } - if !item.tags.is_empty() { - obj["tags"] = json!(item.tags); - } if item.is_chunked { obj["chunked"] = json!(true); } @@ -1166,94 +828,6 @@ async fn execute_forget( } } -async fn execute_connections( - db: &mut Database, - graph: &GraphStore, - ctx: &ServerContext, - args: Option, -) -> CallToolResult { - let params: ConnectionsParams = match args { - Some(v) => match serde_json::from_value(v) { - Ok(p) => p, - Err(e) => { - tracing::debug!("Parameter validation failed: {}", e); - return CallToolResult::error("Invalid parameters"); - } - }, - None => return CallToolResult::error("Missing parameters"), - }; - - // Verify item exists and belongs to the current project - match db.get_item(¶ms.id).await { - Ok(Some(item)) => { - if let Some(ref current_pid) = ctx.project_id - && let Some(ref item_pid) = item.project_id - && item_pid != current_pid - { - return CallToolResult::error(format!( - "Cannot view connections for item {} from a different project", - params.id - )); - } - } - Ok(None) => return CallToolResult::error(format!("Item not found: {}", params.id)), - Err(e) => return sanitized_error("Failed to get item", e), - } - - match graph.get_full_connections(¶ms.id) { - Ok(connections) => { - // Batch fetch all connected items - let target_ids: Vec<&str> = connections.iter().map(|c| c.target_id.as_str()).collect(); - let items = db.get_items_batch(&target_ids).await.unwrap_or_default(); - let item_map: std::collections::HashMap<&str, &Item> = - items.iter().map(|item| (item.id.as_str(), item)).collect(); - - let mut conn_json: Vec = Vec::new(); - - for conn in &connections { - let mut obj = json!({ - "id": conn.target_id, - "type": conn.rel_type, - "strength": conn.strength, - }); - - if let Some(count) = conn.count { - obj["count"] = json!(count); - } - - // Add content preview from batch, but only if the connected item - // belongs to the same project (or is global). This prevents - // cross-project content leakage via graph edges. - if let Some(item) = item_map.get(conn.target_id.as_str()) { - let same_project = match (&ctx.project_id, &item.project_id) { - (Some(current), Some(item_pid)) => current == item_pid, - (_, None) => true, // Global items are visible to all - _ => false, - }; - if same_project { - obj["content_preview"] = json!(truncate(&item.content, 80)); - } else { - obj["cross_project"] = json!(true); - } - } - - conn_json.push(obj); - } - - let result = json!({ - "item_id": params.id, - "connections": conn_json - }); - - CallToolResult::success( - serde_json::to_string_pretty(&result) - .unwrap_or_else(|e| format!("{{\"error\": \"serialization failed: {}\"}}", e)), - ) - } - Err(e) => sanitized_error("Failed to get connections", e), - } -} - // ========== Utilities ========== /// Log a detailed internal error and return a sanitized message to the MCP client. From dc4c14fb5c2e1aa8f35646ec03a9b23789fd54c2 Mon Sep 17 00:00:00 2001 From: Robert Fleischmann Date: Tue, 3 Feb 2026 19:27:53 -0500 Subject: [PATCH 2/3] Bump version to 0.3.0 Co-Authored-By: Claude Opus 4.5 --- Cargo.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Cargo.toml b/Cargo.toml index 831b601..233fa3f 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "sediment-mcp" -version = "0.2.4" +version = "0.3.0" edition = "2024" repository = "https://github.com/rendro/sediment" homepage = "https://github.com/rendro/sediment" From 79b438959d48305a364ddb3818dacea829991350 Mon Sep 17 00:00:00 2001 From: Robert Fleischmann Date: Tue, 3 Feb 2026 19:29:34 -0500 Subject: [PATCH 3/3] Fix clippy warnings and update benchmarks for simplified API Co-Authored-By: Claude Opus 4.5 --- benches/recall_bench.rs | 8 +++----- src/db.rs | 23 ++++++++++------------- 2 files changed, 13 insertions(+), 18 deletions(-) diff --git a/benches/recall_bench.rs b/benches/recall_bench.rs index 05c644e..45b998b 100644 --- a/benches/recall_bench.rs +++ b/benches/recall_bench.rs @@ -101,9 +101,7 @@ async fn seed_database(n: usize, embedder: Arc) -> SeededDb { for i in 0..n { let content = generate_content(i); - let item = Item::new(&content) - .with_tags(vec![format!("bench-tag-{}", i % 5)]) - .with_project_id("bench-project"); + let item = Item::new(&content).with_project_id("bench-project"); let result = db.store_item(item).await.expect("store item"); let id = result.id.clone(); @@ -174,7 +172,7 @@ fn recall_benchmarks(c: &mut Criterion) { enable_decay_scoring: false, enable_background_tasks: false, }; - let filters = ItemFilters::new().with_min_similarity(0.3); + let filters = ItemFilters::new(); let _ = recall_pipeline( &mut db, &tracker, &graph, query, 5, filters, &config, ) @@ -210,7 +208,7 @@ fn recall_benchmarks(c: &mut Criterion) { let tracker = AccessTracker::open(&access_path).unwrap(); let graph = GraphStore::open(&access_path).unwrap(); let config = RecallConfig::default(); - let filters = ItemFilters::new().with_min_similarity(0.3); + let filters = ItemFilters::new(); let _ = recall_pipeline( &mut db, &tracker, &graph, query, 5, filters, &config, ) diff --git a/src/db.rs b/src/db.rs index dc7ee4b..3a30be2 100644 --- a/src/db.rs +++ b/src/db.rs @@ -232,17 +232,15 @@ impl Database { /// Check if the database needs migration by checking for old schema columns async fn check_needs_migration(&self) -> Result { - let table = self - .db - .open_table("items") - .execute() - .await - .map_err(|e| SedimentError::Database(format!("Failed to open items for check: {}", e)))?; - - let schema = table.schema().await.map_err(|e| { - SedimentError::Database(format!("Failed to get schema: {}", e)) + let table = self.db.open_table("items").execute().await.map_err(|e| { + SedimentError::Database(format!("Failed to open items for check: {}", e)) })?; + let schema = table + .schema() + .await + .map_err(|e| SedimentError::Database(format!("Failed to get schema: {}", e)))?; + // Old schema has 'tags' column, new schema doesn't let has_tags = schema.fields().iter().any(|f| f.name() == "tags"); Ok(has_tags) @@ -298,8 +296,7 @@ impl Database { // Insert migrated data if !new_batches.is_empty() { - let batches = - RecordBatchIterator::new(new_batches.into_iter().map(Ok), schema); + let batches = RecordBatchIterator::new(new_batches.into_iter().map(Ok), schema); new_table .add(Box::new(batches)) .execute() @@ -942,7 +939,6 @@ impl Database { Ok(stats) } - } // ==================== Decay Scoring ==================== @@ -1404,6 +1400,7 @@ mod tests { #[test] fn test_schema_version() { // Ensure schema version is set - assert!(SCHEMA_VERSION >= 2, "Schema version should be at least 2"); + let version = SCHEMA_VERSION; + assert!(version >= 2, "Schema version should be at least 2"); } }