Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions .env.example
Original file line number Diff line number Diff line change
Expand Up @@ -130,6 +130,9 @@ NEARAI_AUTH_URL=https://private.near.ai
# GEMINI_RESPONSE_JSON_SCHEMA={"type":"object"}
# GEMINI_CACHED_CONTENT=cachedContents/abc123

# Workspace summary backfill on startup.
# SUMMARY_BACKFILL_LIMIT=50 # 0 disables startup backfill

# For full provider setup guide see docs/LLM_PROVIDERS.md

# Channel Configuration
Expand Down
1 change: 1 addition & 0 deletions FEATURE_PARITY.md
Original file line number Diff line number Diff line change
Expand Up @@ -346,6 +346,7 @@ This document tracks feature parity between IronClaw (Rust implementation) and O
| Vector memory | ✅ | ✅ | pgvector |
| Session-based memory | ✅ | ✅ | |
| Hybrid search (BM25 + vector) | ✅ | ✅ | RRF algorithm |
| Tiered context summaries (L0/L1) | ✅ | 🚧 | Workspace docs now store/generate summaries and search defaults to L1 |
| Temporal decay (hybrid search) | ✅ | ❌ | Opt-in time-based scoring factor |
| MMR re-ranking | ✅ | ❌ | Maximal marginal relevance for result diversity |
| LLM-based query expansion | ✅ | ❌ | Expand FTS queries via LLM |
Expand Down
81 changes: 81 additions & 0 deletions migrations/V14__workspace_tiered_summaries.sql
Original file line number Diff line number Diff line change
@@ -0,0 +1,81 @@
ALTER TABLE memory_documents ADD COLUMN summary_l0 TEXT;
ALTER TABLE memory_documents ADD COLUMN summary_l1 TEXT;

CREATE OR REPLACE FUNCTION list_workspace_files(
p_user_id TEXT,
p_agent_id UUID,
p_directory TEXT DEFAULT ''
)
RETURNS TABLE (
path TEXT,
is_directory BOOLEAN,
updated_at TIMESTAMPTZ,
content_preview TEXT
) AS $$
BEGIN
-- Normalize directory path (ensure trailing slash for non-root)
IF p_directory != '' AND NOT p_directory LIKE '%/' THEN
p_directory := p_directory || '/';
END IF;

RETURN QUERY
WITH files AS (
SELECT
d.path,
d.updated_at,
COALESCE(d.summary_l0, LEFT(d.content, 120)) as content_preview,
-- Extract the immediate child name
CASE
WHEN p_directory = '' THEN
CASE
WHEN position('/' in d.path) > 0
THEN substring(d.path from 1 for position('/' in d.path) - 1)
ELSE d.path
END
ELSE
CASE
WHEN position('/' in substring(d.path from length(p_directory) + 1)) > 0
THEN substring(
substring(d.path from length(p_directory) + 1)
from 1
for position('/' in substring(d.path from length(p_directory) + 1)) - 1
)
ELSE substring(d.path from length(p_directory) + 1)
END
END as child_name
FROM memory_documents d
WHERE d.user_id = p_user_id
AND d.agent_id IS NOT DISTINCT FROM p_agent_id
AND (p_directory = '' OR d.path LIKE p_directory || '%')
)
SELECT DISTINCT ON (f.child_name)
CASE
WHEN p_directory = '' THEN f.child_name
ELSE p_directory || f.child_name
END as path,
EXISTS (
SELECT 1 FROM memory_documents d2
WHERE d2.user_id = p_user_id
AND d2.agent_id IS NOT DISTINCT FROM p_agent_id
AND d2.path LIKE
CASE WHEN p_directory = '' THEN f.child_name ELSE p_directory || f.child_name END
|| '/%'
) as is_directory,
MAX(f.updated_at) as updated_at,
CASE
WHEN EXISTS (
SELECT 1 FROM memory_documents d2
WHERE d2.user_id = p_user_id
AND d2.agent_id IS NOT DISTINCT FROM p_agent_id
AND d2.path LIKE
CASE WHEN p_directory = '' THEN f.child_name ELSE p_directory || f.child_name END
|| '/%'
) THEN NULL
ELSE MAX(f.content_preview)
END as content_preview
FROM files f
WHERE f.child_name != '' AND f.child_name IS NOT NULL
GROUP BY f.child_name
ORDER BY f.child_name, is_directory DESC;
END;
$$ LANGUAGE plpgsql;
4 changes: 3 additions & 1 deletion migrations/V1__initial.sql
Original file line number Diff line number Diff line change
Expand Up @@ -172,6 +172,8 @@ CREATE TABLE memory_documents (
-- File path within workspace (e.g., "context/vision.md")
path TEXT NOT NULL,
content TEXT NOT NULL,
summary_l0 TEXT,
summary_l1 TEXT,

created_at TIMESTAMPTZ NOT NULL DEFAULT NOW(),
updated_at TIMESTAMPTZ NOT NULL DEFAULT NOW(),
Expand Down Expand Up @@ -266,7 +268,7 @@ BEGIN
SELECT
d.path,
Comment thread
G7CNF marked this conversation as resolved.
d.updated_at,
LEFT(d.content, 200) as content_preview,
COALESCE(d.summary_l0, LEFT(d.content, 120)) as content_preview,
-- Extract the immediate child name
CASE
WHEN p_directory = '' THEN
Expand Down
8 changes: 4 additions & 4 deletions src/agent/routine_engine.rs
Original file line number Diff line number Diff line change
Expand Up @@ -606,10 +606,12 @@ impl RoutineEngine {
/// Finalize a dispatched routine run: update DB, update routine runtime,
/// persist to conversation thread, and send notification.
async fn complete_dispatched_run(&self, run: &RoutineRun, status: RunStatus, summary: &str) {
let summary = sanitize_summary(summary);

// Complete the run record in DB
if let Err(e) = self
.store
.complete_routine_run(run.id, status, Some(summary), None)
.complete_routine_run(run.id, status, Some(summary.as_str()), None)
.await
{
tracing::error!(
Expand Down Expand Up @@ -739,7 +741,7 @@ impl RoutineEngine {
&routine.user_id,
&routine.name,
status,
Some(summary),
Some(summary.as_str()),
thread_id.as_deref(),
)
.await;
Expand Down Expand Up @@ -2136,7 +2138,6 @@ fn truncate(s: &str, max: usize) -> String {
/// 2. Strip HTML tags to prevent injection in web-rendered notifications
/// 3. Collapse multiple whitespace/newlines to single spaces for cleaner output
/// 4. Truncate to 500 chars to prevent oversized notifications
#[cfg(test)]
fn sanitize_summary(s: &str) -> String {
// Strip control characters (keep newline for now, collapse later)
let no_control: String = s
Expand Down Expand Up @@ -2164,7 +2165,6 @@ fn sanitize_summary(s: &str) -> String {
}

/// Remove HTML/XML tags from a string.
#[cfg(test)]
fn strip_html_tags(s: &str) -> String {
let mut result = String::with_capacity(s.len());
let mut in_tag = false;
Expand Down
17 changes: 16 additions & 1 deletion src/app.rs
Original file line number Diff line number Diff line change
Expand Up @@ -310,6 +310,7 @@ impl AppBuilder {
pub async fn init_tools(
&self,
llm: &Arc<dyn LlmProvider>,
cheap_llm: Option<&Arc<dyn LlmProvider>>,
) -> Result<
(
Arc<SafetyLayer>,
Expand Down Expand Up @@ -374,6 +375,7 @@ impl AppBuilder {
"Workspace configured with multi-scope reads"
);
}
ws = ws.with_llm(cheap_llm.cloned().unwrap_or_else(|| Arc::clone(llm)));
ws = ws.with_memory_layers(self.config.workspace.memory_layers.clone());
let ws = Arc::new(ws);

Expand Down Expand Up @@ -860,7 +862,7 @@ impl AppBuilder {
self.init_llm().await?
};
let (safety, tools, embeddings, workspace, builder, credential_registry) =
self.init_tools(&llm).await?;
self.init_tools(&llm, cheap_llm.as_ref()).await?;

// Create hook registry early so runtime extension activation can register hooks.
let hooks = Arc::new(HookRegistry::new());
Expand Down Expand Up @@ -935,6 +937,19 @@ impl AppBuilder {
}
});
}

let ws_bg = Arc::clone(ws);
tokio::spawn(async move {
match ws_bg.backfill_summaries().await {
Ok(count) if count > 0 => {
tracing::debug!("Backfilled tiered summaries for {} documents", count);
}
Ok(_) => {}
Err(e) => {
tracing::warn!("Failed to backfill tiered summaries: {}", e);
}
}
});
}

// Skills system
Expand Down
8 changes: 5 additions & 3 deletions src/db/libsql/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -410,9 +410,11 @@ pub(crate) fn row_to_memory_document(row: &libsql::Row) -> MemoryDocument {
agent_id: get_opt_text(row, 2).and_then(|s| s.parse().ok()),
path: get_text(row, 3),
content: get_text(row, 4),
created_at: get_ts(row, 5),
updated_at: get_ts(row, 6),
metadata: get_json(row, 7),
summary_l0: get_opt_text(row, 5),
summary_l1: get_opt_text(row, 6),
created_at: get_ts(row, 7),
updated_at: get_ts(row, 8),
metadata: get_json(row, 9),
}
}

Expand Down
74 changes: 61 additions & 13 deletions src/db/libsql/workspace.rs
Original file line number Diff line number Diff line change
Expand Up @@ -259,7 +259,7 @@ impl WorkspaceStore for LibSqlBackend {
.query(
r#"
SELECT id, user_id, agent_id, path, content,
created_at, updated_at, metadata
summary_l0, summary_l1, created_at, updated_at, metadata
FROM memory_documents
WHERE user_id = ?1 AND agent_id IS ?2 AND path = ?3
"#,
Expand Down Expand Up @@ -295,7 +295,7 @@ impl WorkspaceStore for LibSqlBackend {
.query(
r#"
SELECT id, user_id, agent_id, path, content,
created_at, updated_at, metadata
summary_l0, summary_l1, created_at, updated_at, metadata
FROM memory_documents WHERE id = ?1
"#,
params![id.to_string()],
Expand Down Expand Up @@ -366,7 +366,7 @@ impl WorkspaceStore for LibSqlBackend {
})?;
let now = fmt_ts(&Utc::now());
conn.execute(
"UPDATE memory_documents SET content = ?2, updated_at = ?3 WHERE id = ?1",
"UPDATE memory_documents SET content = ?2, summary_l0 = NULL, summary_l1 = NULL, updated_at = ?3 WHERE id = ?1",
params![id.to_string(), content, now],
)
.await
Comment thread
G7CNF marked this conversation as resolved.
Expand All @@ -376,6 +376,29 @@ impl WorkspaceStore for LibSqlBackend {
Ok(())
}

async fn update_document_summaries(
&self,
id: Uuid,
summary_l0: Option<&str>,
summary_l1: Option<&str>,
) -> Result<(), WorkspaceError> {
let conn = self
.connect()
.await
.map_err(|e| WorkspaceError::SearchFailed {
reason: e.to_string(),
})?;
conn.execute(
"UPDATE memory_documents SET summary_l0 = ?2, summary_l1 = ?3 WHERE id = ?1",
params![id.to_string(), summary_l0, summary_l1],
)
.await
.map_err(|e| WorkspaceError::SearchFailed {
reason: format!("Summary update failed: {}", e),
})?;
Ok(())
}

async fn delete_document_by_path(
&self,
user_id: &str,
Expand Down Expand Up @@ -431,7 +454,8 @@ impl WorkspaceStore for LibSqlBackend {
let mut rows = conn
.query(
r#"
SELECT path, updated_at, substr(content, 1, 200) as content_preview
SELECT path, updated_at,
COALESCE(summary_l0, substr(content, 1, 120)) as content_preview
FROM memory_documents
WHERE user_id = ?1 AND agent_id IS ?2
AND (?3 = '%' OR path LIKE ?3)
Expand Down Expand Up @@ -547,6 +571,7 @@ impl WorkspaceStore for LibSqlBackend {
&self,
user_id: &str,
agent_id: Option<Uuid>,
limit: Option<usize>,
) -> Result<Vec<MemoryDocument>, WorkspaceError> {
let conn = self
.connect()
Expand All @@ -555,21 +580,40 @@ impl WorkspaceStore for LibSqlBackend {
reason: e.to_string(),
})?;
let agent_id_str = agent_id.map(|id| id.to_string());
let mut rows = conn
.query(
r#"
let query = if limit.is_some() {
r#"
SELECT id, user_id, agent_id, path, content,
created_at, updated_at, metadata
summary_l0, summary_l1, created_at, updated_at, metadata
FROM memory_documents
WHERE user_id = ?1 AND agent_id IS ?2
ORDER BY updated_at DESC
"#,
params![user_id, agent_id_str.as_deref()],
LIMIT ?3
"#
} else {
r#"
SELECT id, user_id, agent_id, path, content,
summary_l0, summary_l1, created_at, updated_at, metadata
FROM memory_documents
WHERE user_id = ?1 AND agent_id IS ?2
ORDER BY updated_at DESC
"#
};
let mut rows = if let Some(limit) = limit {
conn.query(
query,
params![user_id, agent_id_str.as_deref(), limit as i64],
)
.await
.map_err(|e| WorkspaceError::SearchFailed {
reason: format!("Query failed: {}", e),
})?;
})?
} else {
conn.query(query, params![user_id, agent_id_str.as_deref()])
.await
.map_err(|e| WorkspaceError::SearchFailed {
reason: format!("Query failed: {}", e),
})?
};

let mut docs = Vec::new();
while let Some(row) = rows
Expand Down Expand Up @@ -739,7 +783,7 @@ impl WorkspaceStore for LibSqlBackend {
let mut rows = conn
.query(
r#"
SELECT c.id, c.document_id, d.path, c.content
SELECT c.id, c.document_id, d.path, c.content, d.summary_l0, d.summary_l1
FROM memory_chunks_fts fts
JOIN memory_chunks c ON c._rowid = fts.rowid
JOIN memory_documents d ON d.id = c.document_id
Expand Down Expand Up @@ -768,6 +812,8 @@ impl WorkspaceStore for LibSqlBackend {
document_id: get_text(&row, 1).parse().unwrap_or_default(),
document_path: get_text(&row, 2),
content: get_text(&row, 3),
summary_l0: get_opt_text(&row, 4),
summary_l1: get_opt_text(&row, 5),
rank: results.len() as u32 + 1,
});
}
Expand All @@ -791,7 +837,7 @@ impl WorkspaceStore for LibSqlBackend {
match conn
.query(
r#"
SELECT c.id, c.document_id, d.path, c.content
SELECT c.id, c.document_id, d.path, c.content, d.summary_l0, d.summary_l1
FROM vector_top_k('idx_memory_chunks_embedding', vector(?1), ?2) AS top_k
JOIN memory_chunks c ON c._rowid = top_k.id
JOIN memory_documents d ON d.id = c.document_id
Expand All @@ -815,6 +861,8 @@ impl WorkspaceStore for LibSqlBackend {
document_id: get_text(&row, 1).parse().unwrap_or_default(),
document_path: get_text(&row, 2),
content: get_text(&row, 3),
summary_l0: get_opt_text(&row, 4),
summary_l1: get_opt_text(&row, 5),
rank: results.len() as u32 + 1,
});
}
Expand Down
2 changes: 2 additions & 0 deletions src/db/libsql_migrations.rs
Original file line number Diff line number Diff line change
Expand Up @@ -208,6 +208,8 @@ CREATE TABLE IF NOT EXISTS memory_documents (
agent_id TEXT,
path TEXT NOT NULL,
content TEXT NOT NULL,
summary_l0 TEXT,
summary_l1 TEXT,
created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%fZ', 'now')),
updated_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%fZ', 'now')),
metadata TEXT NOT NULL DEFAULT '{}',
Expand Down
Loading
Loading