From aa48fc6c19131332ac04dd96c277ea9fd126635c Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Thu, 8 Oct 2026 23:26:08 +0530 Subject: [PATCH 01/19] refactor(sources): split source modules into mod and test files Moved the inline test modules of the composio, fetch, and readers sources into sibling mod_tests.rs files and extracted shared types into types.rs, leaving the mod.rs files focused on implementation. No behaviour changed. Auto-committed-on: macbook Co-authored-by: Medulla --- .../src/sources/composio/clickup/mod.rs | 69 -- .../src/sources/composio/clickup/mod_tests.rs | 58 -- .../src/sources/composio/documents/mod.rs | 343 ---------- .../sources/composio/documents/mod_tests.rs | 197 ------ .../src/sources/composio/fields/mod.rs | 44 -- .../src/sources/composio/fields/mod_tests.rs | 37 -- .../src/sources/composio/github/mod.rs | 116 ---- .../src/sources/composio/github/mod_tests.rs | 97 --- .../composio/gmail_post_process/mod.rs | 498 -------------- .../composio/gmail_post_process/mod_tests.rs | 356 ---------- .../src/sources/composio/linear/mod.rs | 79 --- .../src/sources/composio/linear/mod_tests.rs | 93 --- .../src/sources/composio/mod.rs | 35 - .../src/sources/composio/notion/mod.rs | 88 --- .../src/sources/composio/notion/mod_tests.rs | 111 ---- .../composio/slack_post_process/mod.rs | 324 --------- .../composio/slack_post_process/mod_tests.rs | 306 --------- .../src/sources/fetch/mod.rs | 151 ----- .../src/sources/fetch/mod_tests.rs | 177 ----- .../src/sources/fetch/ssrf/mod.rs | 231 ------- .../src/sources/fetch/ssrf/mod_tests.rs | 236 ------- .../src/sources/readers/composio/mod.rs | 78 --- .../src/sources/readers/composio/mod_tests.rs | 51 -- .../src/sources/readers/github/api/mod.rs | 391 ----------- .../readers/github/api/transport_override.rs | 5 - .../readers/github/api/transport_tests.rs | 34 - .../src/sources/readers/github/git/mod.rs | 320 --------- .../sources/readers/github/git/mod_tests.rs | 219 ------ .../src/sources/readers/github/issues/mod.rs | 304 --------- .../readers/github/issues/mod_tests.rs | 151 ----- .../src/sources/readers/github/mod.rs | 308 --------- .../src/sources/readers/github/mod_tests.rs | 624 ------------------ .../src/sources/readers/github/types.rs | 122 ---- .../src/sources/readers/rss/mod.rs | 355 ---------- .../src/sources/readers/rss/mod_tests.rs | 386 ----------- .../src/sources/readers/rss/types.rs | 25 - .../src/sources/readers/web_page/mod.rs | 444 ------------- .../src/sources/readers/web_page/mod_tests.rs | 302 --------- .../src/sources/readers/web_page/types.rs | 13 - 39 files changed, 7778 deletions(-) delete mode 100644 crates/tinymemory-integrations/src/sources/composio/clickup/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/clickup/mod_tests.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/documents/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/documents/mod_tests.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/fields/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/fields/mod_tests.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/github/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/github/mod_tests.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod_tests.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/linear/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/linear/mod_tests.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/notion/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/notion/mod_tests.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod_tests.rs delete mode 100644 crates/tinymemory-integrations/src/sources/fetch/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/fetch/mod_tests.rs delete mode 100644 crates/tinymemory-integrations/src/sources/fetch/ssrf/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/fetch/ssrf/mod_tests.rs delete mode 100644 crates/tinymemory-integrations/src/sources/readers/composio/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/readers/composio/mod_tests.rs delete mode 100644 crates/tinymemory-integrations/src/sources/readers/github/api/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/readers/github/api/transport_override.rs delete mode 100644 crates/tinymemory-integrations/src/sources/readers/github/api/transport_tests.rs delete mode 100644 crates/tinymemory-integrations/src/sources/readers/github/git/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/readers/github/git/mod_tests.rs delete mode 100644 crates/tinymemory-integrations/src/sources/readers/github/issues/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/readers/github/issues/mod_tests.rs delete mode 100644 crates/tinymemory-integrations/src/sources/readers/github/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/readers/github/mod_tests.rs delete mode 100644 crates/tinymemory-integrations/src/sources/readers/github/types.rs delete mode 100644 crates/tinymemory-integrations/src/sources/readers/rss/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/readers/rss/mod_tests.rs delete mode 100644 crates/tinymemory-integrations/src/sources/readers/rss/types.rs delete mode 100644 crates/tinymemory-integrations/src/sources/readers/web_page/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/readers/web_page/mod_tests.rs delete mode 100644 crates/tinymemory-integrations/src/sources/readers/web_page/types.rs diff --git a/crates/tinymemory-integrations/src/sources/composio/clickup/mod.rs b/crates/tinymemory-integrations/src/sources/composio/clickup/mod.rs deleted file mode 100644 index b4ce64e5..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/clickup/mod.rs +++ /dev/null @@ -1,69 +0,0 @@ -//! ClickUp host normalization helpers — result extraction and task-title and -//! timestamp extraction. -//! -//! ClickUp's REST API (and therefore Composio's wrapping of it) returns -//! task lists in a small handful of shapes depending on which endpoint -//! is called. The functions here walk the union of common shapes so the -//! provider doesn't have to branch per Composio envelope variant. - -use serde_json::Value; - -use super::fields::pick_str; - -/// Walk the Composio response envelope for ClickUp task list results. -/// -/// ClickUp's "filtered team tasks" endpoint returns `{ "tasks": [...] }` -/// at the top level; Composio re-wraps the upstream payload under -/// `data` or `data.data` depending on the action. We probe each shape -/// in order and return the first array we find. -pub fn extract_tasks(data: &Value) -> Vec { - let candidates = [ - data.pointer("/data/tasks"), - data.pointer("/tasks"), - data.pointer("/data/data/tasks"), - data.pointer("/data/results"), - data.pointer("/results"), - data.pointer("/data/items"), - data.pointer("/items"), - ]; - for cand in candidates.into_iter().flatten() { - if let Some(arr) = cand.as_array() { - return arr.clone(); - } - } - Vec::new() -} - -/// Extract a human-readable title from a ClickUp task object. -/// -/// ClickUp tasks store the name at `name` (or `data.name` after Composio -/// envelope wrapping). When the name is missing we fall back to the -/// task ID so chunks remain identifiable. -pub fn extract_task_name(task: &Value) -> Option { - pick_str(task, &["name", "data.name", "title", "data.title"]) -} - -/// Extract a stable cursor timestamp (milliseconds since epoch as a -/// string) from a ClickUp task object. -/// -/// The ClickUp API returns `date_updated` as a stringified epoch ms -/// (e.g. `"1733412345678"`); we keep it as a string so lexicographic -/// comparison against the stored cursor remains valid as long as the -/// length doesn't change (it won't until year 33658). -pub fn extract_task_updated(task: &Value) -> Option { - pick_str( - task, - &[ - "date_updated", - "data.date_updated", - "updated_at", - "data.updated_at", - "dateUpdated", - "data.dateUpdated", - ], - ) -} - -#[cfg(test)] -#[path = "mod_tests.rs"] -mod tests; diff --git a/crates/tinymemory-integrations/src/sources/composio/clickup/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/clickup/mod_tests.rs deleted file mode 100644 index 5a520401..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/clickup/mod_tests.rs +++ /dev/null @@ -1,58 +0,0 @@ -//! Tests for the ClickUp normaliser. - -use super::*; -use serde_json::json; - -#[test] -fn extract_tasks_from_data_tasks() { - let data = json!({ "data": { "tasks": [{"id": "t1"}] } }); - assert_eq!(extract_tasks(&data).len(), 1); -} - -#[test] -fn extract_tasks_from_top_level_tasks() { - let data = json!({ "tasks": [{"id": "a"}, {"id": "b"}] }); - assert_eq!(extract_tasks(&data).len(), 2); -} - -#[test] -fn extract_tasks_empty_when_missing() { - let data = json!({ "foo": "bar" }); - assert!(extract_tasks(&data).is_empty()); -} - -#[test] -fn extract_task_name_from_top_level() { - let task = json!({ "id": "t1", "name": "Build feature X" }); - assert_eq!(extract_task_name(&task), Some("Build feature X".into())); -} - -#[test] -fn extract_task_name_falls_back_to_data_name() { - let task = json!({ "data": { "name": "Wrapped" } }); - assert_eq!(extract_task_name(&task), Some("Wrapped".into())); -} - -#[test] -fn extract_task_name_none_when_missing() { - let task = json!({ "id": "t1" }); - assert!(extract_task_name(&task).is_none()); -} - -#[test] -fn extract_task_updated_handles_string_form() { - let task = json!({ "date_updated": "1733412345678" }); - assert_eq!( - extract_task_updated(&task), - Some("1733412345678".to_string()) - ); -} - -#[test] -fn extract_task_updated_handles_nested_data() { - let task = json!({ "data": { "dateUpdated": "1700000000000" } }); - assert_eq!( - extract_task_updated(&task), - Some("1700000000000".to_string()) - ); -} diff --git a/crates/tinymemory-integrations/src/sources/composio/documents/mod.rs b/crates/tinymemory-integrations/src/sources/composio/documents/mod.rs deleted file mode 100644 index 04753e4b..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/documents/mod.rs +++ /dev/null @@ -1,343 +0,0 @@ -//! One Composio response in, documents and `StoreItem`s out. -//! -//! [`normalise_payload`] dispatches on the toolkit slug to the matching -//! normaliser and reads each record's title, body, link and timestamp. -//! Toolkits without a dedicated normaliser fall back to a generic walk that -//! keeps each record as fenced JSON, so a new toolkit is ingested (verbosely) -//! rather than dropped. [`payload_items`] wraps the documents as -//! `StoreItem::Document`s. - -use crate::documents::{DocumentFormat, markdown_from_text}; -use chrono::{DateTime, TimeZone, Utc}; -use serde_json::Value; -use tinymemory_api::{DocumentBody, MemoryMeta, SourceKind, StoreItem}; - -use super::fields::pick_str; -use super::{clickup, github, gmail_post_process, linear, notion}; - -/// One record of a Composio payload, normalised: an email, a message, an -/// issue, a task or a page. -#[derive(Debug, Clone, PartialEq)] -pub struct ComposioDocument { - /// The provider's id for the record, when it has one. - pub id: Option, - /// A human-readable title. - pub title: Option, - /// The record's text as markdown. Never empty. - pub body: String, - /// A link back to the record in the provider's UI. - pub url: Option, - /// When the record was last updated or sent. - pub observed_at: Option>, - /// The email or message thread the record belongs to. - pub thread_id: Option, - /// The repository a GitHub record belongs to, as `owner/name`. - pub repo: Option, -} - -impl ComposioDocument { - /// A record with only a body. - fn with_body(body: String) -> Self { - Self { - id: None, - title: None, - body, - url: None, - observed_at: None, - thread_id: None, - repo: None, - } - } - - /// Wrap this record as a [`StoreItem::Document`] from `toolkit`, read - /// through the connection or source `source_id`. - #[must_use] - pub fn into_store_item(self, toolkit: &str, source_id: &str) -> StoreItem { - let mut meta = MemoryMeta::from_source(SourceKind::Composio, Some(source_id.to_string())); - meta.url = self.url; - meta.observed_at = self.observed_at; - meta.thread_id = self.thread_id; - meta.repo = self.repo; - meta.tags = vec![toolkit.to_string()]; - StoreItem::Document { - title: self.title, - body: DocumentBody::Text(self.body), - mime: Some(DocumentFormat::Markdown.mime().to_string()), - meta, - } - } -} - -/// Normalise one Composio response from `toolkit` into documents. -/// -/// `data` is the action's response; for Gmail and Slack, run -/// [`gmail_post_process::post_process`] / [`super::slack_post_process::post_process`] -/// on it first so it carries the slim `messages[]` shape. Records with no text -/// at all are skipped. -#[must_use] -pub fn normalise_payload(toolkit: &str, data: &Value) -> Vec { - let documents: Vec = match toolkit.to_ascii_lowercase().as_str() { - "gmail" => array_at(data, &["/messages", "/data/messages"]) - .iter() - .filter_map(gmail_message) - .collect(), - "slack" => array_at(data, &["/messages", "/data/messages"]) - .iter() - .filter_map(slack_message) - .collect(), - "github" => github::extract_issues(data) - .iter() - .filter_map(github_issue) - .collect(), - "linear" => linear::extract_issues(data) - .iter() - .filter_map(linear_issue) - .collect(), - "notion" => notion_pages(data), - "clickup" => clickup::extract_tasks(data) - .iter() - .filter_map(clickup_task) - .collect(), - _ => generic_records(data), - }; - log::debug!( - "[memory_sources:composio] normalised toolkit={toolkit} documents={}", - documents.len() - ); - documents -} - -/// Normalise one Composio response and wrap every record as a -/// [`StoreItem::Document`] with `source.kind = Composio`, -/// `source.id = source_id` and `tags = [toolkit]`. -#[must_use] -pub fn payload_items(toolkit: &str, source_id: &str, data: &Value) -> Vec { - normalise_payload(toolkit, data) - .into_iter() - .map(|document| document.into_store_item(toolkit, source_id)) - .collect() -} - -/// The first array found at any of `pointers`. -fn array_at<'a>(data: &'a Value, pointers: &[&str]) -> &'a [Value] { - pointers - .iter() - .find_map(|pointer| data.pointer(pointer).and_then(Value::as_array)) - .map_or(&[], Vec::as_slice) -} - -/// A body as markdown: HTML (as sniffed) is converted, anything else kept. -fn to_markdown(text: &str) -> String { - let format = DocumentFormat::sniff(text.as_bytes(), None, None); - markdown_from_text(text, format).trim().to_string() -} - -/// Build a document from its parts, falling back to the title as the body; -/// `None` when there is no text at all. -fn document(title: Option, body: Option) -> Option { - let body = body - .map(|body| to_markdown(&body)) - .filter(|body| !body.is_empty()) - .or_else(|| title.clone())?; - let mut document = ComposioDocument::with_body(body); - document.title = title; - Some(document) -} - -/// Parse an ISO 8601 / RFC 3339 / RFC 2822 timestamp. -fn parse_time(text: &str) -> Option> { - gmail_post_process::parse_email_date(text) -} - -/// Parse an epoch-milliseconds string (ClickUp's `date_updated`). -fn parse_epoch_ms(text: &str) -> Option> { - let millis = text.trim().parse::().ok()?; - Utc.timestamp_millis_opt(millis).single() -} - -/// Parse a Slack `ts` (`"1712345678.123456"`, epoch seconds with a fraction). -fn parse_slack_ts(text: &str) -> Option> { - let (seconds, fraction) = text.split_once('.').unwrap_or((text, "0")); - let seconds = seconds.parse::().ok()?; - let micros = format!("{fraction:0<6}").get(..6)?.parse::().ok()?; - Utc.timestamp_opt(seconds, micros * 1_000).single() -} - -/// A string field, or a number rendered as a string. -fn scalar(value: &Value, key: &str) -> Option { - match value.get(key)? { - Value::String(text) if !text.trim().is_empty() => Some(text.trim().to_string()), - Value::Number(number) => Some(number.to_string()), - _ => None, - } -} - -/// A post-processed Gmail message: headers above the body. -fn gmail_message(message: &Value) -> Option { - let subject = pick_str(message, &["subject"]); - let markdown = pick_str(message, &["markdown", "messageText"]); - let mut header = String::new(); - for (label, key) in [("From", "from"), ("To", "to"), ("Date", "date")] { - if let Some(value) = pick_str(message, &[key]) { - header.push_str(&format!("{label}: {value}\n")); - } - } - let body = match (markdown, header.is_empty()) { - (Some(markdown), false) => Some(format!("{header}\n{markdown}")), - (Some(markdown), true) => Some(markdown), - (None, _) => None, - }; - let mut document = document(subject, body)?; - document.id = pick_str(message, &["id", "messageId"]); - document.thread_id = pick_str(message, &["threadId", "thread_id"]); - document.observed_at = pick_str(message, &["date"]).as_deref().and_then(parse_time); - Some(document) -} - -/// A post-processed Slack message. -fn slack_message(message: &Value) -> Option { - let user = pick_str(message, &["user"]); - let channel = pick_str(message, &["channel_id"]); - let title = match (&user, &channel) { - (Some(user), Some(channel)) => Some(format!("Slack message from {user} in {channel}")), - (Some(user), None) => Some(format!("Slack message from {user}")), - (None, _) => Some("Slack message".to_string()), - }; - let text = pick_str(message, &["text"])?; - let mut document = document(title, Some(text))?; - let ts = pick_str(message, &["ts"]); - document.id = ts.clone(); - document.url = pick_str(message, &["permalink"]); - document.thread_id = pick_str(message, &["thread_ts"]); - document.observed_at = ts.as_deref().and_then(parse_slack_ts); - Some(document) -} - -/// A GitHub issue or pull request from a search response. -fn github_issue(issue: &Value) -> Option { - let mut document = document( - github::extract_issue_title(issue), - pick_str(issue, &["body", "data.body"]), - )?; - document.id = github::extract_issue_id(issue); - document.url = pick_str(issue, &["html_url", "data.html_url"]); - document.repo = document.url.as_deref().and_then(github_repo); - document.observed_at = github::extract_issue_updated_at(issue) - .as_deref() - .and_then(parse_time); - Some(document) -} - -/// `owner/name` from a `https://github.com/owner/name/...` link. -fn github_repo(url: &str) -> Option { - let rest = url.split_once("github.com/")?.1; - let mut parts = rest.split('/'); - let owner = parts.next().filter(|part| !part.is_empty())?; - let name = parts.next().filter(|part| !part.is_empty())?; - Some(format!("{owner}/{name}")) -} - -/// A Linear issue. -fn linear_issue(issue: &Value) -> Option { - let mut document = document( - linear::extract_issue_title(issue), - pick_str(issue, &["description", "data.description"]), - )?; - document.id = pick_str(issue, &["identifier", "id", "data.identifier", "data.id"]); - document.url = pick_str(issue, &["url", "data.url"]); - document.observed_at = linear::extract_issue_updated(issue) - .as_deref() - .and_then(parse_time); - Some(document) -} - -/// Notion pages from a search response, or the one page a -/// `NOTION_GET_PAGE_MARKDOWN` response carries. -fn notion_pages(data: &Value) -> Vec { - let results = notion::extract_results(data); - if results.is_empty() { - return notion::extract_page_markdown(data) - .and_then(|markdown| { - let title = notion::extract_page_title(data); - let mut document = document(title, Some(markdown))?; - document.id = pick_str(data, &["id", "data.id", "page_id", "data.page_id"]); - document.url = pick_str(data, &["url", "data.url"]); - Some(document) - }) - .into_iter() - .collect(); - } - results - .iter() - .filter_map(|page| { - let mut document = document( - notion::extract_page_title(page), - notion::extract_page_markdown(page), - )?; - document.id = pick_str(page, &["id", "data.id"]); - document.url = pick_str(page, &["url", "data.url"]); - document.observed_at = pick_str(page, &["last_edited_time", "data.last_edited_time"]) - .as_deref() - .and_then(parse_time); - Some(document) - }) - .collect() -} - -/// A ClickUp task. -fn clickup_task(task: &Value) -> Option { - let mut document = document( - clickup::extract_task_name(task), - pick_str( - task, - &["markdown_description", "description", "text_content"], - ), - )?; - document.id = scalar(task, "id"); - document.url = pick_str(task, &["url", "data.url"]); - document.observed_at = clickup::extract_task_updated(task) - .as_deref() - .and_then(|text| parse_epoch_ms(text).or_else(|| parse_time(text))); - Some(document) -} - -/// Any other toolkit: each record in the first list found (or the whole -/// payload) as fenced JSON, titled and linked when it says how. -fn generic_records(data: &Value) -> Vec { - let records = array_at( - data, - &[ - "/data/items", - "/items", - "/data/results", - "/results", - "/data/data", - "/data", - ], - ); - let records: Vec<&Value> = if records.is_empty() { - vec![data] - } else { - records.iter().collect() - }; - records - .into_iter() - .filter(|record| !record.is_null()) - .filter_map(|record| { - let json = serde_json::to_string_pretty(record).ok()?; - let title = pick_str(record, &["title", "name", "subject"]); - let mut document = ComposioDocument::with_body(format!("```json\n{json}\n```")); - document.title = title; - document.id = scalar(record, "id"); - document.url = pick_str(record, &["url", "html_url", "permalink", "link"]); - document.observed_at = pick_str(record, &["updated_at", "updatedAt", "created_at"]) - .as_deref() - .and_then(parse_time); - Some(document) - }) - .collect() -} - -#[cfg(test)] -#[path = "mod_tests.rs"] -mod tests; diff --git a/crates/tinymemory-integrations/src/sources/composio/documents/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/documents/mod_tests.rs deleted file mode 100644 index 4d689128..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/documents/mod_tests.rs +++ /dev/null @@ -1,197 +0,0 @@ -//! Tests for turning Composio payloads into documents and `StoreItem`s. - -use super::*; -use serde_json::json; -use tinymemory_api::ItemKind; - -fn text_of(item: &StoreItem) -> (&Option, &str, &MemoryMeta) { - match item { - StoreItem::Document { - title, - body: DocumentBody::Text(text), - meta, - .. - } => (title, text.as_str(), meta), - other => panic!("expected a text document, got {other:?}"), - } -} - -#[test] -fn a_github_search_becomes_items_with_url_repo_time_and_toolkit_tag() { - let data = json!({ - "data": { "items": [{ - "id": 42, - "title": "Fix the build", - "body": "The build is **red**.", - "html_url": "https://github.com/acme/widgets/issues/7", - "updated_at": "2024-05-21T15:30:00Z" - }]} - }); - let items = payload_items("github", "conn_gh", &data); - assert_eq!(items.len(), 1); - let item = &items[0]; - assert_eq!(item.kind(), ItemKind::Document); - item.validate().unwrap(); - - let (title, body, meta) = text_of(item); - assert_eq!( - title.as_deref(), - Some("GitHub: acme/widgets#7: Fix the build") - ); - assert_eq!(body, "The build is **red**."); - assert_eq!(meta.source.kind, SourceKind::Composio); - assert_eq!(meta.source.id.as_deref(), Some("conn_gh")); - assert_eq!(meta.tags, vec!["github".to_string()]); - assert_eq!( - meta.url.as_deref(), - Some("https://github.com/acme/widgets/issues/7") - ); - assert_eq!(meta.repo.as_deref(), Some("acme/widgets")); - assert_eq!( - meta.observed_at, - Some(Utc.with_ymd_and_hms(2024, 5, 21, 15, 30, 0).unwrap()) - ); -} - -#[test] -fn post_processed_gmail_messages_carry_headers_thread_and_date() { - let data = json!({ "messages": [{ - "id": "m1", - "threadId": "t1", - "subject": "Lunch?", - "from": "Ann ", - "to": "me@example.com", - "date": "Tue, 21 May 2024 12:00:00 +0000", - "markdown": "Tacos at noon." - }]}); - let documents = normalise_payload("gmail", &data); - assert_eq!(documents.len(), 1); - let document = &documents[0]; - assert_eq!(document.title.as_deref(), Some("Lunch?")); - assert!(document.body.starts_with("From: Ann \n")); - assert!(document.body.ends_with("Tacos at noon.")); - assert_eq!(document.thread_id.as_deref(), Some("t1")); - assert_eq!(document.id.as_deref(), Some("m1")); - assert_eq!( - document.observed_at, - Some(Utc.with_ymd_and_hms(2024, 5, 21, 12, 0, 0).unwrap()) - ); -} - -#[test] -fn slack_messages_take_permalink_and_ts_time() { - let data = json!({ "messages": [{ - "ts": "1716300000.000100", - "user": "U1", - "channel_id": "C1", - "text": "deploy done", - "permalink": "https://acme.slack.com/archives/C1/p1716300000000100" - }]}); - let items = payload_items("slack", "src_slack", &data); - let (title, body, meta) = text_of(&items[0]); - assert_eq!(title.as_deref(), Some("Slack message from U1 in C1")); - assert_eq!(body, "deploy done"); - assert_eq!( - meta.url.as_deref(), - Some("https://acme.slack.com/archives/C1/p1716300000000100") - ); - assert_eq!( - meta.observed_at.map(|at| at.timestamp()), - Some(1_716_300_000) - ); - assert_eq!(meta.tags, vec!["slack".to_string()]); -} - -#[test] -fn linear_notion_and_clickup_records_are_normalised() { - let linear = json!({ "nodes": [{ - "identifier": "ENG-1", - "title": "Ship v2", - "description": "All of it.", - "url": "https://linear.app/acme/issue/ENG-1", - "updatedAt": "2024-01-02T03:04:05Z" - }]}); - let documents = normalise_payload("linear", &linear); - assert_eq!(documents[0].id.as_deref(), Some("ENG-1")); - assert_eq!(documents[0].body, "All of it."); - assert!(documents[0].observed_at.is_some()); - - let notion = json!({ "results": [{ - "id": "p1", - "url": "https://notion.so/p1", - "last_edited_time": "2024-01-02T03:04:05.000Z", - "properties": { "Name": { "type": "title", "title": [{ "plain_text": "Roadmap" }] } } - }]}); - let documents = normalise_payload("notion", ¬ion); - assert_eq!(documents[0].title.as_deref(), Some("Roadmap")); - assert_eq!( - documents[0].body, "Roadmap", - "a page without body keeps its title" - ); - assert_eq!(documents[0].url.as_deref(), Some("https://notion.so/p1")); - - let page_markdown = json!({ "data": { "markdown": "# Plan\n\nDo it." }, "id": "p2" }); - let documents = normalise_payload("notion", &page_markdown); - assert_eq!(documents.len(), 1); - assert_eq!(documents[0].body, "# Plan\n\nDo it."); - - let clickup = json!({ "tasks": [{ - "id": "t9", - "name": "Write docs", - "description": "

The README

", - "url": "https://app.clickup.com/t/t9", - "date_updated": "1700000000000" - }]}); - let documents = normalise_payload("clickup", &clickup); - assert_eq!(documents[0].id.as_deref(), Some("t9")); - assert_eq!( - documents[0].observed_at.map(|at| at.timestamp()), - Some(1_700_000_000) - ); -} - -#[test] -fn html_bodies_are_converted_to_markdown() { - let data = json!({ "nodes": [{ - "title": "Page", - "description": "

Hi

There

" - }]}); - let documents = normalise_payload("linear", &data); - assert_eq!(documents[0].body, "## Hi\n\nThere"); -} - -#[test] -fn records_without_any_text_are_skipped() { - let data = json!({ "messages": [{ "ts": "1.0", "text": " " }] }); - assert!(normalise_payload("slack", &data).is_empty()); - assert!(normalise_payload("github", &json!({ "items": [{}] })).is_empty()); -} - -#[test] -fn an_unknown_toolkit_keeps_each_record_as_fenced_json() { - let data = json!({ "data": { "items": [ - { "id": 1, "name": "Alpha", "url": "https://example.com/1", "updated_at": "2024-01-01T00:00:00Z" }, - { "id": 2 } - ]}}); - let items = payload_items("hubspot", "conn_hs", &data); - assert_eq!(items.len(), 2); - let (title, body, meta) = text_of(&items[0]); - assert_eq!(title.as_deref(), Some("Alpha")); - assert!(body.starts_with("```json\n")); - assert!(body.contains("\"name\": \"Alpha\"")); - assert_eq!(meta.url.as_deref(), Some("https://example.com/1")); - assert_eq!(meta.tags, vec!["hubspot".to_string()]); - - let single = normalise_payload("hubspot", &json!({ "ok": true })); - assert_eq!(single.len(), 1); -} - -#[test] -fn slack_timestamps_parse_with_and_without_a_fraction() { - assert_eq!(parse_slack_ts("10").map(|at| at.timestamp()), Some(10)); - assert_eq!( - parse_slack_ts("10.5").map(|at| at.timestamp_subsec_micros()), - Some(500_000) - ); - assert_eq!(parse_slack_ts("x"), None); -} diff --git a/crates/tinymemory-integrations/src/sources/composio/fields/mod.rs b/crates/tinymemory-integrations/src/sources/composio/fields/mod.rs deleted file mode 100644 index 4c257085..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/fields/mod.rs +++ /dev/null @@ -1,44 +0,0 @@ -//! Field lookup shared by the Composio normalisers: pull a string out of a -//! payload by trying several dotted paths, because Composio wraps the same -//! upstream field at different depths depending on the action and version. - -/// Walk a JSON object using a list of dotted-path candidates and return the -/// first non-empty **string** match, trimmed. -/// -/// Each path is split on `.` and followed with `Value::get`, so it only -/// descends through objects — it never indexes into an array. A leaf that is -/// not a string (a number, a bool) is rejected rather than coerced, so a -/// payload whose `id` is `42` rather than `"42"` yields `None` here. That -/// differs from the private `scalar` lookup in the `documents` mapping, which -/// renders numbers; the normalisers were written against the -/// reject-non-strings behaviour and `pick_str_rejects_non_string_values` pins -/// it. -pub fn pick_str(value: &serde_json::Value, paths: &[&str]) -> Option { - for path in paths { - let mut cur = value; - let mut ok = true; - for segment in path.split('.') { - match cur.get(segment) { - Some(next) => cur = next, - None => { - ok = false; - break; - } - } - } - if !ok { - continue; - } - if let Some(s) = cur.as_str() { - let trimmed = s.trim(); - if !trimmed.is_empty() { - return Some(trimmed.to_string()); - } - } - } - None -} - -#[cfg(test)] -#[path = "mod_tests.rs"] -mod tests; diff --git a/crates/tinymemory-integrations/src/sources/composio/fields/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/fields/mod_tests.rs deleted file mode 100644 index 330ec404..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/fields/mod_tests.rs +++ /dev/null @@ -1,37 +0,0 @@ -//! Tests for the shared Composio field lookup. - -use super::*; -use serde_json::json; - -#[test] -fn pick_str_finds_first_non_empty_match() { - let v = json!({"data": {"user": {"name": "Ada", "email": "ada@example.com"}}}); - assert_eq!( - pick_str(&v, &["data.user.name", "data.user.email"]), - Some("Ada".into()) - ); - assert_eq!( - pick_str(&v, &["data.missing", "data.user.email"]), - Some("ada@example.com".into()) - ); - assert_eq!(pick_str(&v, &["nope.nope"]), None); -} - -#[test] -fn pick_str_respects_path_order() { - let v = json!({"a": "first", "b": "second"}); - assert_eq!(pick_str(&v, &["a", "b"]), Some("first".into())); - assert_eq!(pick_str(&v, &["b", "a"]), Some("second".into())); -} - -/// The drift guard for the behaviour documented on [`pick_str`]. If this -/// ever starts returning `Some("42")`, the normalisers' emitted ids have -/// changed. -#[test] -fn pick_str_rejects_non_string_values() { - let v = json!({"count": 42, "flag": true, "empty": "", "whitespace": " "}); - assert_eq!(pick_str(&v, &["count"]), None); - assert_eq!(pick_str(&v, &["flag"]), None); - assert_eq!(pick_str(&v, &["empty"]), None); - assert_eq!(pick_str(&v, &["whitespace"]), None); -} diff --git a/crates/tinymemory-integrations/src/sources/composio/github/mod.rs b/crates/tinymemory-integrations/src/sources/composio/github/mod.rs deleted file mode 100644 index 12e7a615..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/github/mod.rs +++ /dev/null @@ -1,116 +0,0 @@ -//! GitHub host normalization helpers — issue extraction and issue id, title and -//! timestamp helpers. -//! -//! GitHub's REST API (proxied through Composio) returns search results in a -//! small number of shapes. The functions here -//! walk the union of common Composio envelope variants so the provider stays -//! clean and branch-free. - -use serde_json::Value; - -use super::fields::pick_str; - -/// Walk the Composio response envelope for GitHub search issue results. -/// -/// `GITHUB_SEARCH_ISSUES_AND_PULL_REQUESTS` wraps GitHub's `GET /search/issues` response, which -/// returns `{"total_count": N, "items": [...]}`. Composio may re-wrap this under -/// `data` or `data.data`; we probe each shape in order. -pub fn extract_issues(data: &Value) -> Vec { - let candidates = [ - data.pointer("/data/items"), - data.pointer("/items"), - data.pointer("/data/data/items"), - data.pointer("/data/results"), - data.pointer("/results"), - ]; - for cand in candidates.into_iter().flatten() { - if let Some(arr) = cand.as_array() { - return arr.clone(); - } - } - Vec::new() -} - -/// Extract a stable, globally unique identifier for a GitHub issue or PR. -/// -/// GitHub's internal `id` field is a large integer unique across all issues -/// and PRs on github.com. We convert it to a string for use as a sync key. -/// Falls back to composing from `html_url` path if `id` is absent. -pub fn extract_issue_id(issue: &Value) -> Option { - // Primary: numeric internal GitHub ID. - if let Some(id) = issue.get("id").or_else(|| issue.pointer("/data/id")) { - if let Some(n) = id.as_u64() { - return Some(n.to_string()); - } - if let Some(s) = id.as_str() { - let trimmed = s.trim(); - if !trimmed.is_empty() { - return Some(trimmed.to_string()); - } - } - } - // Fallback: parse owner/repo/number from html_url path segments. - // URL shape: https://github.com/{owner}/{repo}/issues/{number} - if let Some(url) = pick_str(issue, &["html_url", "data.html_url", "url", "data.url"]) - && let Some(slug) = github_url_to_slug(&url) - { - return Some(slug); - } - None -} - -/// Build a human-readable document title for a GitHub issue/PR. -/// -/// Format: `GitHub: {owner}/{repo}#{number}: {title}`. -/// Falls back to just the title or a placeholder when fields are missing. -pub fn extract_issue_title(issue: &Value) -> Option { - let title = pick_str(issue, &["title", "data.title"])?; - - // Best-effort: extract owner/repo#N from html_url for the prefix. - let prefix = pick_str(issue, &["html_url", "data.html_url"]) - .and_then(|url| github_url_to_slug(&url)) - .unwrap_or_default(); - - if prefix.is_empty() { - Some(title) - } else { - Some(format!("GitHub: {prefix}: {title}")) - } -} - -/// Parse `https://github.com/{owner}/{repo}/issues/{number}` (or `/pull/`) -/// into `"{owner}/{repo}#{number}"`. Returns `None` for unrecognised shapes. -fn github_url_to_slug(url: &str) -> Option { - let segs: Vec<&str> = url.trim_end_matches('/').split('/').collect(); - // Minimum: ["https:", "", "github.com", owner, repo, "issues", number] - if segs.len() >= 7 { - let number = segs[segs.len() - 1]; - let _kind = segs[segs.len() - 2]; // "issues" or "pull" — ignored - let repo = segs[segs.len() - 3]; - let owner = segs[segs.len() - 4]; - if !owner.is_empty() && !repo.is_empty() && !number.is_empty() { - return Some(format!("{owner}/{repo}#{number}")); - } - } - None -} - -/// Extract the `updated_at` ISO 8601 timestamp from a GitHub issue. -/// -/// GitHub returns `updated_at` as `"2024-05-21T15:30:00Z"`. ISO 8601 strings -/// sort lexicographically, so we use them directly as the sync cursor. -pub fn extract_issue_updated_at(issue: &Value) -> Option { - pick_str( - issue, - &[ - "updated_at", - "data.updated_at", - "updatedAt", - "data.updatedAt", - ], - ) -} - -#[cfg(test)] -#[path = "mod_tests.rs"] -mod tests; diff --git a/crates/tinymemory-integrations/src/sources/composio/github/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/github/mod_tests.rs deleted file mode 100644 index 0b7322f1..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/github/mod_tests.rs +++ /dev/null @@ -1,97 +0,0 @@ -//! Tests for the GitHub normaliser. - -use super::*; -use serde_json::json; - -#[test] -fn extract_issues_from_data_items() { - let data = json!({ "data": { "items": [{"id": 1}] } }); - assert_eq!(extract_issues(&data).len(), 1); -} - -#[test] -fn extract_issues_from_top_level_items() { - let data = json!({ "items": [{"id": 1}, {"id": 2}] }); - assert_eq!(extract_issues(&data).len(), 2); -} - -#[test] -fn extract_issues_empty_when_missing() { - let data = json!({ "foo": "bar" }); - assert!(extract_issues(&data).is_empty()); -} - -#[test] -fn extract_issue_id_from_numeric_field() { - let issue = json!({ "id": 123456789u64, "title": "Fix bug" }); - assert_eq!(extract_issue_id(&issue), Some("123456789".to_string())); -} - -#[test] -fn extract_issue_id_from_wrapped_data() { - let issue = json!({ "data": { "id": 99u64 } }); - assert_eq!(extract_issue_id(&issue), Some("99".to_string())); -} - -#[test] -fn extract_issue_id_falls_back_to_html_url() { - let issue = json!({ - "html_url": "https://github.com/owner/repo/issues/42" - }); - assert_eq!(extract_issue_id(&issue), Some("owner/repo#42".to_string())); -} - -#[test] -fn extract_issue_id_none_when_missing() { - let issue = json!({ "title": "No ID here" }); - assert!(extract_issue_id(&issue).is_none()); -} - -#[test] -fn extract_issue_title_builds_prefixed_title() { - let issue = json!({ - "id": 1u64, - "title": "Fix race condition", - "html_url": "https://github.com/acme/core/issues/99" - }); - assert_eq!( - extract_issue_title(&issue), - Some("GitHub: acme/core#99: Fix race condition".to_string()) - ); -} - -#[test] -fn extract_issue_title_returns_raw_title_when_no_url() { - let issue = json!({ "title": "Bare title" }); - assert_eq!(extract_issue_title(&issue), Some("Bare title".to_string())); -} - -#[test] -fn extract_issue_title_none_when_missing() { - let issue = json!({ "id": 1u64 }); - assert!(extract_issue_title(&issue).is_none()); -} - -#[test] -fn extract_issue_updated_at_from_top_level() { - let issue = json!({ "updated_at": "2024-05-21T15:30:00Z" }); - assert_eq!( - extract_issue_updated_at(&issue), - Some("2024-05-21T15:30:00Z".to_string()) - ); -} - -#[test] -fn extract_issue_updated_at_from_data_wrapper() { - let issue = json!({ "data": { "updated_at": "2023-01-01T00:00:00Z" } }); - assert_eq!( - extract_issue_updated_at(&issue), - Some("2023-01-01T00:00:00Z".to_string()) - ); -} - -#[test] -fn extract_issue_updated_at_none_when_missing() { - let issue = json!({ "id": 1u64 }); - assert!(extract_issue_updated_at(&issue).is_none()); -} diff --git a/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod.rs b/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod.rs deleted file mode 100644 index c6c78ae7..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod.rs +++ /dev/null @@ -1,498 +0,0 @@ -//! Gmail-specific post-processing of Composio action responses. -//! -//! The upstream `GMAIL_FETCH_EMAILS` payload is extremely verbose -//! (full MIME tree under `payload.parts[]`, 50+ `Received:` headers, -//! display-layer noise the model never uses). This module rewrites -//! it into a slim envelope per message: -//! -//! ```json -//! { -//! "messages": [ -//! { -//! "id": "…", -//! "threadId": "…", -//! "subject": "…", -//! "from": "…", -//! "to": "…", -//! "date": "…", -//! "labels": ["INBOX", "UNREAD"], -//! "markdown": "…body…", -//! "attachments": [ { "filename": "...", "mimeType": "..." } ] -//! } -//! ], -//! "nextPageToken": "…", -//! "resultSizeEstimate": 201 -//! } -//! ``` -//! -//! ## Body source -//! -//! Composio's backend ships a -//! `markdownFormatted` field on the response envelope — one string -//! per tool call, pre-rendered with HTML stripped, URLs shortened, -//! footers removed, whitespace normalised. We split it per message -//! along `\n---\n` boundaries (with `## ` heading fallbacks) and -//! pin each slice to the corresponding entry in `messages[]` via -//! [`apply_response_level_markdown`]. The reshape's -//! `extract_markdown_body` then prefers that pinned field over -//! falling back to the upstream `messageText`. -//! -//! No in-house HTML→markdown conversion lives here anymore — the -//! backend does the cleaning. If `markdownFormatted` is absent for -//! a given response we fall through to whatever plain text the -//! upstream provided in `messageText`. -//! -//! Callers that need the raw Composio shape can pass `raw_html: -//! true` (or `rawHtml: true`) in the action arguments — this -//! short-circuits the reshape entirely. -//! -//! Only `GMAIL_FETCH_EMAILS` is reshaped today; other Gmail action -//! responses are passed through unchanged. When we add envelopes for -//! more slugs they should live in this file, branched from -//! [`post_process`]. - -use serde_json::{Map, Value, json}; - -/// Entry point a host calls on each Gmail action response (the slug names -/// the action) before handing it to `normalise_payload`. -/// -/// Dispatches on the Composio action slug. Unknown Gmail slugs fall -/// through to a no-op. -pub fn post_process(slug: &str, arguments: Option<&Value>, data: &mut Value) { - if is_raw_html_flag_set(arguments) { - tracing::debug!( - slug, - "[composio:gmail][post-process] raw_html flag set, passing through" - ); - return; - } - if slug == "GMAIL_FETCH_EMAILS" { - reshape_fetch_emails(data) - } -} - -/// Stash per-message slices of the response-level `markdownFormatted` -/// onto the corresponding entries inside `data.messages[]`. -/// -/// The Composio backend (tinyhumansai/backend#683) ships ONE -/// `markdownFormatted` string per tool call covering all messages — -/// already URL-shortened, footer-stripped, and whitespace-normalised. -/// To get per-email files in the raw archive we split that string -/// along section boundaries (`## ` headings or `---` rules) and pin -/// each slice to the message at the same index. `extract_markdown_body` -/// then prefers `msg.markdownFormatted` over re-decoding the MIME -/// tree. -/// -/// **Must be called BEFORE [`post_process`]** because `post_process` -/// reshapes `data` into the slim envelope; once `messages[]` carries -/// our slim shape the upstream message ordering is already locked in -/// but we may have lost original ordering signals if any. -/// -/// No-op when the slice count doesn't match `messages.len()` — we -/// can't safely align segments to messages without an exact match, -/// so we let `extract_markdown_body` fall through to its MIME path. -pub fn apply_response_level_markdown(data: &mut Value, top_md: &str) { - let trimmed = top_md.trim(); - if trimmed.is_empty() { - return; - } - // Presence is checked immutably first, then fetched mutably. The original - // form re-fetched with `unwrap()` after a mutable probe, which is sound but - // relies on the reader to see why; this crate lints against `unwrap`, and the - // immutable probe expresses the same reasoning to the compiler. - let container = if data.get("messages").is_some() { - data - } else if data.get("data").and_then(Value::as_object).is_some() { - match data.get_mut("data") { - Some(inner) => inner, - None => return, - } - } else { - tracing::debug!( - "[composio:gmail][post-process] apply_response_level_markdown: \ - no messages container in response — skipping" - ); - return; - }; - let Some(messages) = container.get_mut("messages").and_then(|v| v.as_array_mut()) else { - return; - }; - let count = messages.len(); - if count == 0 { - return; - } - // Clone hints out of the messages array so the slice borrows - // don't conflict with the upcoming `messages.iter_mut()` mutation. - let hints: Vec = messages.clone(); - let Some(slices) = split_response_markdown_per_message_with_hint(trimmed, count, Some(&hints)) - else { - tracing::debug!( - messages = count, - md_len = trimmed.len(), - "[composio:gmail][post-process] could not split response-level markdownFormatted \ - into {count} slices — falling back to per-message MIME decode" - ); - return; - }; - for (msg, slice) in messages.iter_mut().zip(slices) { - if let Some(obj) = msg.as_object_mut() { - obj.insert("markdownFormatted".to_string(), Value::String(slice)); - } - } - tracing::debug!( - messages = count, - "[composio:gmail][post-process] stashed per-message markdownFormatted slices" - ); -} - -/// Split a top-level `markdownFormatted` string into per-message -/// segments. Returns `Some(slices)` only when the split yields -/// exactly `expected_count` entries — otherwise the format isn't one -/// of the patterns we know about and we let the caller fall back. -/// -/// Primary boundary is the `\n---\n` horizontal rule the backend -/// emits between messages (confirmed against real -/// `GMAIL_FETCH_EMAILS` output). H2/H3 headings are kept as -/// fallbacks for older renderings. The preamble (`# Inbox (N -/// messages)`-style intro, if present) is dropped — we accept -/// either `expected` parts (no preamble) or `expected + 1` -/// (preamble + N messages). -/// -/// `messages_hint` is the slim message array from the same response -/// — when present we use the per-message `subject` field to verify -/// each segment really does belong to the message at the same index. -/// Mismatches force a fallback so we never write a wrong-message body -/// to the raw archive. -/// -/// The hint is what makes the split reliable: the blob's own section headings -/// are backend-rendered and have changed shape between versions, so matching -/// on them alone silently mis-attributed bodies. -pub fn split_response_markdown_per_message_with_hint( - md: &str, - expected_count: usize, - messages_hint: Option<&[Value]>, -) -> Option> { - if expected_count == 0 { - return None; - } - if expected_count == 1 { - return Some(vec![md.to_string()]); - } - - // Boundary patterns to try, in priority order. `\n---\n` is the - // confirmed marker; the heading variants stay as belt-and-braces - // for older / variant backend renderings. - let candidates: &[(&str, &str)] = &[ - ("\n---\n", "---\n"), - ("\n\n## ", "## "), - ("\n\n### ", "### "), - ("\n\n# ", "# "), - ("\n***\n", "***\n"), - ]; - - for (sep, prefix) in candidates { - let parts: Vec<&str> = md.split(sep).collect(); - let (drop_preamble, prepend_first) = if parts.len() == expected_count { - (false, false) // no preamble; first segment had no prefix - } else if parts.len() == expected_count + 1 { - (true, true) // preamble dropped; every kept segment had a prefix - } else { - continue; - }; - let segments: Vec = parts - .into_iter() - .skip(if drop_preamble { 1 } else { 0 }) - .enumerate() - .map(|(i, s)| { - if i == 0 && !prepend_first { - s.to_string() - } else { - format!("{prefix}{s}") - } - }) - .collect(); - - // Validate alignment against the JSON message array: every - // segment whose corresponding message has a non-empty subject - // must mention that subject somewhere in its body. If a single - // pair fails, we treat the split as unreliable and try the - // next pattern. Empty / null subjects skip validation (e.g. - // notification mails where the subject is ""). - if let Some(hints) = messages_hint - && !validate_segments_against_hints(&segments, hints) - { - tracing::debug!( - expected = expected_count, - sep = sep, - "[composio:gmail][post-process] split candidate failed subject check" - ); - continue; - } - return Some(segments); - } - None -} - -/// True if every (segment, message) pair where the message has a -/// non-empty subject contains that subject somewhere in the segment -/// (case-insensitive substring match — a defensive heuristic, not a -/// strict equality check, since the backend may format subjects -/// inside markdown links or with surrounding decoration). -fn validate_segments_against_hints(segments: &[String], hints: &[Value]) -> bool { - if segments.len() != hints.len() { - return false; - } - for (seg, hint) in segments.iter().zip(hints.iter()) { - let subject = hint - .get("subject") - .and_then(|v| v.as_str()) - .unwrap_or("") - .trim(); - if subject.is_empty() { - continue; - } - if !seg - .to_ascii_lowercase() - .contains(&subject.to_ascii_lowercase()) - { - return false; - } - } - true -} - -/// Returns true when the caller explicitly set `raw_html: true` (or the -/// camelCase `rawHtml: true`) in the `arguments` object. -fn is_raw_html_flag_set(arguments: Option<&Value>) -> bool { - let Some(obj) = arguments.and_then(|v| v.as_object()) else { - return false; - }; - obj.get("raw_html") - .or_else(|| obj.get("rawHtml")) - .and_then(|v| v.as_bool()) - .unwrap_or(false) -} - -/// Rewrite a `GMAIL_FETCH_EMAILS` `data` object in place into the slim -/// envelope documented at the module level. -/// -/// The Composio response can be shaped either as `{ messages, nextPageToken, ... }` -/// directly, or wrapped one level deeper under `{ data: { messages: … } }` -/// depending on backend version; we handle both. -fn reshape_fetch_emails(data: &mut Value) { - // Unwrap an optional `data:` envelope so downstream logic only has - // to deal with one shape. - let container = if data.get("messages").is_some() { - data - } else if data.get("data").and_then(Value::as_object).is_some() { - match data.get_mut("data") { - Some(inner) => inner, - None => return, - } - } else { - return; - }; - - let Some(obj) = container.as_object_mut() else { - return; - }; - - let raw_messages = obj - .remove("messages") - .and_then(|v| match v { - Value::Array(arr) => Some(arr), - _ => None, - }) - .unwrap_or_default(); - let next_page_token = obj.remove("nextPageToken").unwrap_or(Value::Null); - let result_size_estimate = obj.remove("resultSizeEstimate").unwrap_or(Value::Null); - - let messages: Vec = raw_messages.into_iter().map(reshape_message).collect(); - - let mut envelope = Map::new(); - envelope.insert("messages".into(), Value::Array(messages)); - if !next_page_token.is_null() { - envelope.insert("nextPageToken".into(), next_page_token); - } - if !result_size_estimate.is_null() { - envelope.insert("resultSizeEstimate".into(), result_size_estimate); - } - - *container = Value::Object(envelope); -} - -/// Parse an RFC 3339 or RFC 2822 date string into a UTC `DateTime`. -pub fn parse_email_date(date_str: &str) -> Option> { - date_str - .parse::>() - .or_else(|_| { - chrono::DateTime::parse_from_rfc2822(date_str).map(|d| d.with_timezone(&chrono::Utc)) - }) - .ok() -} - -const EMAIL_LOCAL_TIME_FMT: &str = "%Y-%m-%d %I:%M %p %:z"; - -/// Format a UTC `DateTime` in the given timezone. Returns `None` when the -/// formatted result is identical to the UTC rendering (no-op for UTC hosts). -pub fn format_at_tz( - utc: chrono::DateTime, - tz: &Tz, -) -> Option -where - Tz::Offset: std::fmt::Display, -{ - let local_dt = utc.with_timezone(tz); - let formatted = local_dt.format(EMAIL_LOCAL_TIME_FMT).to_string(); - - let utc_formatted = utc.format(EMAIL_LOCAL_TIME_FMT).to_string(); - if formatted == utc_formatted { - return None; - } - Some(formatted) -} - -/// Convert a UTC email timestamp string to a human-readable local-time string. -/// -/// Accepts RFC 3339 (`"2026-05-31T10:33:00Z"`) or RFC 2822 -/// (`"Sat, 31 May 2026 10:33:00 +0000"`) input. Returns a formatted string -/// in the host's local timezone, e.g. `"2026-05-31 05:33 AM -05:00"`, -/// so the agent can present local times without UTC arithmetic. -/// -/// The raw `date` field is always preserved alongside this field so -/// internal sorting, deduplication, and debugging remain UTC-based. -/// -/// Returns `None` when the input cannot be parsed or the output format -/// would be identical to the UTC input (no-op for UTC hosts). -pub fn format_email_local_time(date_str: &str) -> Option { - let utc = parse_email_date(date_str)?; - format_at_tz(utc, &chrono::Local) -} - -/// Map one raw Composio message object to its slim counterpart. -/// -/// Body source picked by [`extract_markdown_body`]: -/// 1. The per-message `markdownFormatted` slice pinned by -/// [`apply_response_level_markdown`] (preferred — backend-rendered). -/// 2. The upstream `messageText` plaintext (fallback). -/// 3. Empty string. -fn reshape_message(raw: Value) -> Value { - let Value::Object(obj) = raw else { - return raw; - }; - - let id = obj.get("messageId").cloned().unwrap_or(Value::Null); - let thread_id = obj.get("threadId").cloned().unwrap_or(Value::Null); - let subject = obj.get("subject").cloned().unwrap_or(Value::Null); - let sender = obj.get("sender").cloned().unwrap_or(Value::Null); - let to = obj.get("to").cloned().unwrap_or(Value::Null); - let date = obj - .get("messageTimestamp") - .cloned() - .or_else(|| pick_header(&obj, "Date")) - .unwrap_or(Value::Null); - let labels = obj - .get("labelIds") - .cloned() - .unwrap_or_else(|| Value::Array(Vec::new())); - let list_unsubscribe = pick_header(&obj, "List-Unsubscribe").unwrap_or(Value::Null); - - let markdown = extract_markdown_body(&obj); - let attachments = extract_attachments(&obj); - - // Compute a local-time representation of the UTC `date` so the agent - // presents times in the user's timezone rather than quoting raw UTC. - let date_local = date.as_str().and_then(format_email_local_time); - - let mut out = Map::new(); - out.insert("id".into(), id); - out.insert("threadId".into(), thread_id); - out.insert("subject".into(), subject); - out.insert("from".into(), sender); - out.insert("to".into(), to); - out.insert("date".into(), date); - if let Some(local) = date_local { - out.insert("date_local".into(), Value::String(local)); - } - out.insert("labels".into(), labels); - if !list_unsubscribe.is_null() { - out.insert("list_unsubscribe".into(), list_unsubscribe); - } - out.insert("markdown".into(), Value::String(markdown)); - if !attachments.is_empty() { - out.insert("attachments".into(), Value::Array(attachments)); - } - Value::Object(out) -} - -/// Find a header value by (case-insensitive) name in the Composio -/// `payload.headers[]` array. Returns `Some(Value::String)` on hit. -fn pick_header(msg: &Map, name: &str) -> Option { - let headers = msg.get("payload")?.get("headers")?.as_array()?; - for h in headers { - let hn = h.get("name").and_then(|v| v.as_str()).unwrap_or(""); - if hn.eq_ignore_ascii_case(name) - && let Some(v) = h.get("value").and_then(|v| v.as_str()) - { - return Some(Value::String(v.to_string())); - } - } - None -} - -/// Pick a body for the slim envelope. -/// -/// We trust the Composio backend's pre-rendered `markdownFormatted` -/// (set per-message by [`apply_response_level_markdown`] from the -/// response-level field). When that's absent we fall back to the -/// upstream's plain-text `messageText` verbatim — no in-house -/// HTML→markdown decoding lives here anymore. The backend already -/// strips HTML, shortens URLs, and normalises whitespace; running -/// our own pipeline on top duplicated work and corrupted some -/// renderings. -fn extract_markdown_body(msg: &Map) -> String { - if let Some(formatted) = msg - .get("markdownFormatted") - .or_else(|| msg.get("markdown_formatted")) - .and_then(|v| v.as_str()) - .map(str::trim) - .filter(|s| !s.is_empty()) - { - return formatted.to_string(); - } - if let Some(text) = msg - .get("messageText") - .and_then(|v| v.as_str()) - .map(str::trim) - .filter(|s| !s.is_empty()) - { - return text.to_string(); - } - String::new() -} - -/// Pull a minimal attachments descriptor from the Composio -/// `attachmentList` array. -fn extract_attachments(msg: &Map) -> Vec { - if let Some(list) = msg.get("attachmentList").and_then(|v| v.as_array()) { - return list - .iter() - .filter_map(|a| { - let filename = a.get("filename").and_then(|v| v.as_str())?; - if filename.is_empty() { - return None; - } - let mime = a - .get("mimeType") - .and_then(|v| v.as_str()) - .unwrap_or_default(); - Some(json!({ "filename": filename, "mimeType": mime })) - }) - .collect(); - } - Vec::new() -} - -#[cfg(test)] -#[path = "mod_tests.rs"] -mod tests; diff --git a/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod_tests.rs deleted file mode 100644 index dd1b44af..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod_tests.rs +++ /dev/null @@ -1,356 +0,0 @@ -//! Tests for the Gmail post-processor. - -use super::*; -use serde_json::json; - -fn fixture_with_backend_markdown() -> Value { - json!({ - "messages": [ - { - "messageId": "m1", - "threadId": "t1", - "subject": "Hello", - "sender": "a@x.com", - "to": "b@y.com", - "messageTimestamp": "2026-04-17T12:00:00Z", - "labelIds": ["INBOX", "UNREAD"], - // Pre-rendered slice (set by `apply_response_level_markdown` - // in production; inline here for the reshape test). - "markdownFormatted": "# Hello\n\nbody copy", - "messageText": "fallback should not be used", - "display_url": "ignore-me", - "preview": { "body": "Hi plain", "subject": "Hello" }, - "attachmentList": [ - { "filename": "report.pdf", "mimeType": "application/pdf", "size": 12345 }, - { "filename": "", "mimeType": "text/html" } - ], - "payload": {} - } - ], - "nextPageToken": "tok-1", - "resultSizeEstimate": 42 - }) -} - -#[test] -fn reshape_emits_slim_envelope() { - let mut v = fixture_with_backend_markdown(); - post_process("GMAIL_FETCH_EMAILS", None, &mut v); - - assert_eq!(v["nextPageToken"], "tok-1"); - assert_eq!(v["resultSizeEstimate"], 42); - - let msgs = v["messages"].as_array().unwrap(); - assert_eq!(msgs.len(), 1); - let m = &msgs[0]; - - assert_eq!(m["id"], "m1"); - assert_eq!(m["threadId"], "t1"); - assert_eq!(m["subject"], "Hello"); - assert_eq!(m["from"], "a@x.com"); - assert_eq!(m["to"], "b@y.com"); - assert_eq!(m["date"], "2026-04-17T12:00:00Z"); - assert_eq!(m["labels"], json!(["INBOX", "UNREAD"])); - - let md = m["markdown"].as_str().unwrap(); - assert_eq!(md, "# Hello\n\nbody copy"); - - // Noise fields removed. - assert!(m.get("display_url").is_none()); - assert!(m.get("preview").is_none()); - assert!(m.get("payload").is_none()); - assert!(m.get("messageText").is_none()); - - // Attachments: empty filename entry is filtered. - let atts = m["attachments"].as_array().unwrap(); - assert_eq!(atts.len(), 1); - assert_eq!(atts[0]["filename"], "report.pdf"); - assert_eq!(atts[0]["mimeType"], "application/pdf"); -} - -#[test] -fn raw_html_flag_passes_through_unchanged() { - let mut v = fixture_with_backend_markdown(); - let original = v.clone(); - let args = json!({ "raw_html": true }); - post_process("GMAIL_FETCH_EMAILS", Some(&args), &mut v); - assert_eq!( - v, original, - "raw_html=true must preserve the Composio shape" - ); -} - -#[test] -fn camel_case_raw_html_also_recognized() { - let mut v = fixture_with_backend_markdown(); - let original = v.clone(); - let args = json!({ "rawHtml": true }); - post_process("GMAIL_FETCH_EMAILS", Some(&args), &mut v); - assert_eq!(v, original); -} - -#[test] -fn falls_back_to_message_text_when_no_backend_markdown() { - let mut v = json!({ - "messages": [{ - "messageId": "m1", - "threadId": "t1", - "subject": "s", - "sender": "a@x.com", - "to": "b@y.com", - "messageTimestamp": "2026-04-17", - "labelIds": [], - "messageText": " plain body text ", - "payload": {} - }], - "nextPageToken": null - }); - post_process("GMAIL_FETCH_EMAILS", None, &mut v); - let md = v["messages"][0]["markdown"].as_str().unwrap(); - assert_eq!(md, "plain body text"); - assert!(v.get("nextPageToken").is_none(), "null tokens dropped"); -} - -#[test] -fn unwraps_data_envelope() { - let mut v = json!({ - "data": { - "messages": [{ - "messageId": "m1", - "threadId": "t1", - "subject": "s", - "sender": "a@x.com", - "to": "b@y.com", - "messageTimestamp": "2026-04-17", - "labelIds": [], - "messageText": "body", - "payload": {} - }] - } - }); - post_process("GMAIL_FETCH_EMAILS", None, &mut v); - // Reshape writes into `data` in place. - let msgs = v["data"]["messages"].as_array().unwrap(); - assert_eq!(msgs.len(), 1); - assert_eq!(msgs[0]["markdown"], "body"); -} - -#[test] -fn non_fetch_slug_is_noop() { - let mut v = json!({ "messages": [{ "messageId": "m1", "messageText": "x" }] }); - let original = v.clone(); - post_process("GMAIL_SEND_EMAIL", None, &mut v); - assert_eq!(v, original); -} - -#[test] -fn prefers_backend_markdown_formatted_when_present() { - // Composio backend (tinyhumansai/backend#683 +) ships - // `markdownFormatted` already URL-shortened + footer-stripped - // per message (after `apply_response_level_markdown` slices the - // response-level field). When present, our post-processor must - // use it verbatim instead of falling back to `messageText`. - let mut v = json!({ - "messages": [{ - "messageId": "m1", - "threadId": "t1", - "subject": "s", - "sender": "a@x.com", - "to": "b@y.com", - "messageTimestamp": "2026-04-17", - "labelIds": [], - "markdownFormatted": "# Already nice\n\nShort URL: https://gh.io/abc", - "messageText": "fallback should not be used", - "payload": {} - }] - }); - post_process("GMAIL_FETCH_EMAILS", None, &mut v); - let md = v["messages"][0]["markdown"].as_str().unwrap(); - assert_eq!(md, "# Already nice\n\nShort URL: https://gh.io/abc"); -} - -#[test] -fn empty_markdown_formatted_falls_through_to_message_text() { - let mut v = json!({ - "messages": [{ - "messageId": "m1", - "threadId": "t1", - "subject": "s", - "sender": "a@x.com", - "to": "b@y.com", - "messageTimestamp": "2026-04-17", - "labelIds": [], - "markdownFormatted": " \n \n", - "messageText": "real body", - "payload": {} - }] - }); - post_process("GMAIL_FETCH_EMAILS", None, &mut v); - let md = v["messages"][0]["markdown"].as_str().unwrap(); - assert!(md.contains("real body")); -} - -// ── split_response_markdown_per_message_with_hint ─────────────────────── - -#[test] -fn split_response_markdown_uses_horizontal_rule_marker() { - // The confirmed backend marker is `\n---\n`. Three messages → - // expect three slices when there's no preamble. - let md = "## Alice's update\n\nbody A with https://gh.io/abc\n---\n## Bob's reply\n\nbody B\n---\n## Carol\n\nbody C"; - let slices = super::split_response_markdown_per_message_with_hint(md, 3, None).unwrap(); - assert_eq!(slices.len(), 3); - assert!(slices[0].contains("Alice's update")); - assert!(slices[1].contains("Bob's reply")); - assert!(slices[2].contains("Carol")); - // The `---\n` prefix is preserved on every-but-the-first segment - // so the section break survives the round-trip. - assert!(slices[1].starts_with("---\n")); - assert!(slices[2].starts_with("---\n")); -} - -#[test] -fn split_response_markdown_drops_preamble() { - // When a preamble like `# Inbox` precedes the first marker, we - // see N+1 parts after split — the preamble must be dropped. - let md = "# Inbox (2 messages)\n---\n## A\n\nbody A\n---\n## B\n\nbody B"; - let slices = super::split_response_markdown_per_message_with_hint(md, 2, None).unwrap(); - assert_eq!(slices.len(), 2); - assert!(slices[0].contains("body A")); - assert!(slices[1].contains("body B")); - // Both segments should carry the prefix when preamble was dropped. - assert!(slices[0].starts_with("---\n")); - assert!(slices[1].starts_with("---\n")); -} - -#[test] -fn split_response_markdown_falls_back_to_h2_marker() { - // No `---` rules — backend used h2 headings as boundaries. - let md = "## Alice\n\nbody A\n\n## Bob\n\nbody B"; - let slices = super::split_response_markdown_per_message_with_hint(md, 2, None).unwrap(); - assert_eq!(slices.len(), 2); - assert!(slices[0].contains("body A")); - assert!(slices[1].contains("body B")); -} - -#[test] -fn split_response_markdown_returns_none_on_count_mismatch() { - let md = "## only one section here"; - assert!(super::split_response_markdown_per_message_with_hint(md, 3, None).is_none()); -} - -#[test] -fn split_response_markdown_single_message_returns_whole_input() { - let md = "## solo\n\nthe whole body"; - let slices = super::split_response_markdown_per_message_with_hint(md, 1, None).unwrap(); - assert_eq!(slices, vec![md.to_string()]); -} - -#[test] -fn split_with_hint_rejects_when_subjects_dont_match() { - let md = "## Foo\nbody1\n---\n## Bar\nbody2"; - let hints = vec![ - json!({"subject": "Completely different subject A"}), - json!({"subject": "Completely different subject B"}), - ]; - let out = super::split_response_markdown_per_message_with_hint(md, 2, Some(&hints)); - assert!(out.is_none(), "subject mismatch must force fallback"); -} - -#[test] -fn split_with_hint_accepts_when_subjects_match() { - let md = "## Welcome to Gmail\nbody1\n---\n## Your invoice\nbody2"; - let hints = vec![ - json!({"subject": "Welcome to Gmail"}), - json!({"subject": "Your invoice"}), - ]; - let slices = super::split_response_markdown_per_message_with_hint(md, 2, Some(&hints)).unwrap(); - assert_eq!(slices.len(), 2); - assert!(slices[0].contains("Welcome to Gmail")); - assert!(slices[1].contains("Your invoice")); -} - -#[test] -fn split_with_hint_skips_messages_with_blank_subject() { - let md = "## A\nbody1\n---\n## B\nbody2"; - let hints = vec![json!({"subject": "A"}), json!({"subject": ""})]; - let slices = super::split_response_markdown_per_message_with_hint(md, 2, Some(&hints)).unwrap(); - assert_eq!(slices.len(), 2); -} - -// ── format_email_local_time ────────────────────────────────────────────────── - -#[test] -fn format_email_local_time_returns_none_for_unparseable_date() { - assert!(super::format_email_local_time("not-a-date").is_none()); - assert!(super::format_email_local_time("").is_none()); -} - -#[test] -fn format_email_local_time_preserves_utc_raw_date_in_reshape() { - let mut v = json!({ - "messages": [{ - "messageId": "m1", - "threadId": "t1", - "subject": "Test", - "sender": "a@example.com", - "to": "b@example.com", - "messageTimestamp": "2026-05-31T10:33:00Z", - "labelIds": [], - "messageText": "body", - "payload": {} - }] - }); - post_process("GMAIL_FETCH_EMAILS", None, &mut v); - let msg = &v["messages"][0]; - assert_eq!(msg["date"], "2026-05-31T10:33:00Z"); -} - -#[test] -fn parse_email_date_accepts_rfc3339_and_rfc2822() { - assert!(super::parse_email_date("2026-05-31T10:33:00Z").is_some()); - assert!(super::parse_email_date("Sun, 31 May 2026 10:33:00 +0000").is_some()); - assert!(super::parse_email_date("not-a-date").is_none()); -} - -#[test] -fn format_at_tz_deterministic_with_fixed_offset() { - use chrono::FixedOffset; - - let utc = super::parse_email_date("2026-05-31T10:33:00Z").unwrap(); - - let est = FixedOffset::west_opt(5 * 3600).unwrap(); - let result = super::format_at_tz(utc, &est).unwrap(); - assert_eq!(result, "2026-05-31 05:33 AM -05:00"); - - let ist = FixedOffset::east_opt(5 * 3600 + 1800).unwrap(); - let result = super::format_at_tz(utc, &ist).unwrap(); - assert_eq!(result, "2026-05-31 04:03 PM +05:30"); -} - -#[test] -fn format_at_tz_returns_none_for_utc() { - let utc = super::parse_email_date("2026-05-31T10:33:00Z").unwrap(); - let utc_tz = chrono::FixedOffset::east_opt(0).unwrap(); - assert!(super::format_at_tz(utc, &utc_tz).is_none()); -} - -#[test] -fn apply_response_level_markdown_stashes_per_message_field() { - let mut data = json!({ - "messages": [ - {"messageId": "m1", "subject": "Hello"}, - {"messageId": "m2", "subject": "World"}, - ] - }); - let top_md = "## Hello\nbody A — link https://gh.io/abc\n---\n## World\nbody B"; - super::apply_response_level_markdown(&mut data, top_md); - let m1 = data["messages"][0]["markdownFormatted"].as_str().unwrap(); - let m2 = data["messages"][1]["markdownFormatted"].as_str().unwrap(); - assert!(m1.contains("Hello")); - assert!( - m1.contains("https://gh.io/abc"), - "shortened URL must survive" - ); - assert!(m2.contains("World")); - assert!(!m1.contains("World"), "no cross-message bleed"); -} diff --git a/crates/tinymemory-integrations/src/sources/composio/linear/mod.rs b/crates/tinymemory-integrations/src/sources/composio/linear/mod.rs deleted file mode 100644 index ea52092a..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/linear/mod.rs +++ /dev/null @@ -1,79 +0,0 @@ -//! Linear host normalization helpers — result extraction and issue-title and -//! timestamp extraction. -//! -//! Linear's GraphQL API (and therefore Composio's wrapping of it) returns -//! connection-style lists (`{ nodes: [...], pageInfo: {...} }`) at the top -//! level or nested under `data`. The functions here walk the union of -//! common shapes so the provider does not have to branch per Composio -//! envelope variant. - -use serde_json::Value; - -use super::fields::pick_str; - -/// Walk the Composio response envelope for Linear issue list results. -/// -/// Linear's list endpoints return `{ nodes: [...] }` or -/// `{ issues: { nodes: [...] } }` shapes; Composio may re-wrap the -/// upstream payload under `data` or `data.data`. We probe each shape -/// in order and return the first array we find. -pub fn extract_issues(data: &Value) -> Vec { - let candidates = [ - data.pointer("/data/nodes"), - data.pointer("/nodes"), - data.pointer("/data/issues/nodes"), - data.pointer("/issues/nodes"), - data.pointer("/data/data/nodes"), - data.pointer("/data/data/issues/nodes"), - data.pointer("/data/results"), - data.pointer("/results"), - data.pointer("/data/items"), - data.pointer("/items"), - ]; - for cand in candidates.into_iter().flatten() { - if let Some(arr) = cand.as_array() { - return arr.clone(); - } - } - Vec::new() -} - -/// Extract a human-readable title from a Linear issue object. -/// -/// Linear issues store the name at `title` (or `data.title` after -/// Composio envelope wrapping). Falls back to `name` / `identifier` -/// so the chunk remains identifiable even for unusual response shapes. -pub fn extract_issue_title(issue: &Value) -> Option { - pick_str( - issue, - &[ - "title", - "data.title", - "name", - "data.name", - "identifier", - "data.identifier", - ], - ) -} - -/// Extract a stable cursor timestamp from a Linear issue object. -/// -/// Linear uses ISO-8601 strings for timestamps (`updatedAt`). We keep -/// the value as a string so lexicographic comparison against the stored -/// cursor is valid. -pub fn extract_issue_updated(issue: &Value) -> Option { - pick_str( - issue, - &[ - "updatedAt", - "data.updatedAt", - "updated_at", - "data.updated_at", - ], - ) -} - -#[cfg(test)] -#[path = "mod_tests.rs"] -mod tests; diff --git a/crates/tinymemory-integrations/src/sources/composio/linear/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/linear/mod_tests.rs deleted file mode 100644 index c0acff63..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/linear/mod_tests.rs +++ /dev/null @@ -1,93 +0,0 @@ -//! Tests for the Linear normaliser. - -use super::*; -use serde_json::json; - -// ── extract_issues ─────────────────────────────────────────────── - -#[test] -fn extract_issues_from_data_nodes() { - let data = json!({ "data": { "nodes": [{"id": "i1"}, {"id": "i2"}] } }); - assert_eq!(extract_issues(&data).len(), 2); -} - -#[test] -fn extract_issues_from_top_level_nodes() { - let data = json!({ "nodes": [{"id": "i3"}] }); - assert_eq!(extract_issues(&data).len(), 1); -} - -#[test] -fn extract_issues_from_data_issues_nodes() { - let data = - json!({ "data": { "issues": { "nodes": [{"id": "i4"}, {"id": "i5"}, {"id": "i6"}] } } }); - assert_eq!(extract_issues(&data).len(), 3); -} - -#[test] -fn extract_issues_from_top_level_issues_nodes() { - let data = json!({ "issues": { "nodes": [{"id": "i7"}] } }); - assert_eq!(extract_issues(&data).len(), 1); -} - -#[test] -fn extract_issues_from_doubly_nested_issues_nodes() { - let data = - json!({ "data": { "data": { "issues": { "nodes": [{"id": "i8"}, {"id": "i9"}] } } } }); - assert_eq!(extract_issues(&data).len(), 2); -} - -#[test] -fn extract_issues_from_results() { - let data = json!({ "results": [{"id": "i7"}] }); - assert_eq!(extract_issues(&data).len(), 1); -} - -#[test] -fn extract_issues_empty_when_missing() { - let data = json!({ "foo": "bar" }); - assert!(extract_issues(&data).is_empty()); -} - -// ── extract_issue_title ────────────────────────────────────────── - -#[test] -fn extract_issue_title_from_title_field() { - let issue = json!({ "id": "i1", "title": "Fix the login bug" }); - assert_eq!( - extract_issue_title(&issue), - Some("Fix the login bug".into()) - ); -} - -#[test] -fn extract_issue_title_falls_back_to_wrapped_data() { - let issue = json!({ "data": { "title": "Wrapped issue" } }); - assert_eq!(extract_issue_title(&issue), Some("Wrapped issue".into())); -} - -#[test] -fn extract_issue_title_falls_back_to_identifier() { - let issue = json!({ "identifier": "ENG-42" }); - assert_eq!(extract_issue_title(&issue), Some("ENG-42".into())); -} - -// ── extract_issue_updated ──────────────────────────────────────── - -#[test] -fn extract_issue_updated_from_updated_at() { - let issue = json!({ "updatedAt": "2026-03-01T12:00:00.000Z" }); - assert_eq!( - extract_issue_updated(&issue), - Some("2026-03-01T12:00:00.000Z".to_string()) - ); -} - -#[test] -fn extract_issue_updated_falls_back_to_snake_case() { - let issue = json!({ "data": { "updated_at": "2026-01-15T08:30:00.000Z" } }); - assert_eq!( - extract_issue_updated(&issue), - Some("2026-01-15T08:30:00.000Z".to_string()) - ); -} diff --git a/crates/tinymemory-integrations/src/sources/composio/mod.rs b/crates/tinymemory-integrations/src/sources/composio/mod.rs deleted file mode 100644 index 4df32030..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/mod.rs +++ /dev/null @@ -1,35 +0,0 @@ -//! Composio toolkit payloads: normalisers and the mapping to `StoreItem`s. -//! -//! A host runs Composio actions with its own credentials and hands the raw -//! responses here. Two layers turn them into memory: -//! -//! 1. **Normalisers**, one module per toolkit, are pure -//! `serde_json::Value` transforms: they walk Composio's envelope variants -//! and pull out the tasks, issues, pages or messages -//! ([`clickup`], [`github`], [`linear`], [`notion`]), or rewrite a verbose -//! response into a slim one in place ([`gmail_post_process`], -//! [`slack_post_process`]). [`fields`] holds the path lookup they share. -//! 2. [`normalise_payload`] turns one (post-processed) response into -//! [`ComposioDocument`]s, and [`payload_items`] turns those into -//! [`StoreItem::Document`](tinymemory_api::StoreItem::Document)s with -//! `source.kind = Composio`, `source.id` = the connection or source id, -//! `url` and `observed_at` where the payload has them, and -//! `tags = [toolkit]`. -//! -//! Nothing here holds a credential, opens a socket or decides when to sync. -//! -//! One caveat on "pure": [`gmail_post_process::format_email_local_time`] -//! renders in `chrono::Local`, so it reads the host's timezone. The raw UTC -//! fields are preserved alongside, so ordering and identity stay UTC-based. - -pub mod clickup; -pub mod fields; -pub mod github; -pub mod gmail_post_process; -pub mod linear; -pub mod notion; -pub mod slack_post_process; - -mod documents; - -pub use documents::{ComposioDocument, normalise_payload, payload_items}; diff --git a/crates/tinymemory-integrations/src/sources/composio/notion/mod.rs b/crates/tinymemory-integrations/src/sources/composio/notion/mod.rs deleted file mode 100644 index 00c1df50..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/notion/mod.rs +++ /dev/null @@ -1,88 +0,0 @@ -//! Notion host normalization helpers — result extraction, page markdown and -//! page title extraction. - -use serde_json::Value; - -use super::fields::pick_str; - -/// Walk the Composio response envelope for Notion page results. -pub fn extract_results(data: &Value) -> Vec { - let candidates = [ - data.pointer("/data/results"), - data.pointer("/results"), - data.pointer("/data/data/results"), - data.pointer("/data/items"), - data.pointer("/items"), - ]; - for cand in candidates.into_iter().flatten() { - if let Some(arr) = cand.as_array() { - return arr.clone(); - } - } - Vec::new() -} - -/// Extract the rendered page body markdown from a `NOTION_GET_PAGE_MARKDOWN` -/// response. Composio wraps action output in varying envelope shapes, so we -/// try the common locations tolerantly and return the first non-empty string. -/// Returns `None` if no markdown field is found (caller falls back to the -/// metadata-only body and logs the raw shape for diagnosis). -pub fn extract_page_markdown(data: &Value) -> Option { - const PATHS: &[&str] = &[ - "/markdown", - "/data/markdown", - "/data/response_data/markdown", - "/response_data/markdown", - "/data/content", - "/content", - "/data/markdown_content", - "/markdown_content", - "/text", - "/data/text", - ]; - for p in PATHS { - if let Some(s) = data.pointer(p).and_then(Value::as_str) - && !s.trim().is_empty() - { - return Some(s.to_string()); - } - } - None -} - -/// Try to extract a human-readable title from a Notion page object. -/// -/// Notion pages store the title in `properties.title` or -/// `properties.Name.title[0].plain_text`. We try several shapes. -pub fn extract_page_title(page: &Value) -> Option { - // Try the common `properties.title.title[0].plain_text` shape. - let props = page - .get("properties") - .or_else(|| page.get("data")?.get("properties")); - if let Some(props) = props { - // Walk all properties looking for a "title" type field. - if let Some(obj) = props.as_object() { - for (_key, val) in obj { - if val.get("type").and_then(Value::as_str) == Some("title") - && let Some(arr) = val.get("title").and_then(Value::as_array) - { - let text: String = arr - .iter() - .filter_map(|t| t.get("plain_text").and_then(Value::as_str)) - .collect::>() - .join(""); - if !text.is_empty() { - return Some(text); - } - } - } - } - } - - // Fallback: top-level "title" field (some Composio shapes). - pick_str(page, &["title", "data.title", "name", "data.name"]) -} - -#[cfg(test)] -#[path = "mod_tests.rs"] -mod tests; diff --git a/crates/tinymemory-integrations/src/sources/composio/notion/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/notion/mod_tests.rs deleted file mode 100644 index ab2e79bc..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/notion/mod_tests.rs +++ /dev/null @@ -1,111 +0,0 @@ -//! Tests for the Notion normaliser. - -use super::*; -use serde_json::json; - -#[test] -fn extract_results_from_data_results() { - let data = json!({"data": {"results": [{"id": "page1"}]}}); - let results = extract_results(&data); - assert_eq!(results.len(), 1); -} - -#[test] -fn extract_page_markdown_reads_top_level_field() { - // Matches the live GET_PAGE_MARKDOWN envelope observed empirically: - // {id, markdown, object, request_id, truncated, unknown_block_ids}. - let data = json!({ - "id": "p1", - "markdown": "# Heading\n\nbody text", - "object": "page", - "truncated": false, - }); - assert_eq!( - extract_page_markdown(&data).as_deref(), - Some("# Heading\n\nbody text") - ); -} - -#[test] -fn extract_page_markdown_reads_nested_envelope() { - let data = json!({ "data": { "markdown": "nested body" } }); - assert_eq!(extract_page_markdown(&data).as_deref(), Some("nested body")); -} - -#[test] -fn extract_page_markdown_none_for_empty_or_missing() { - // Empty markdown (a DB row with no body blocks) → None → metadata-only. - assert_eq!(extract_page_markdown(&json!({ "markdown": "" })), None); - assert_eq!(extract_page_markdown(&json!({ "markdown": " " })), None); - // No markdown field at all → None. - assert_eq!(extract_page_markdown(&json!({ "id": "p1" })), None); -} - -#[test] -fn extract_results_from_top_level() { - let data = json!({"results": [{"id": "a"}, {"id": "b"}]}); - let results = extract_results(&data); - assert_eq!(results.len(), 2); -} - -#[test] -fn extract_results_from_data_items() { - let data = json!({"data": {"items": [{"id": "x"}]}}); - let results = extract_results(&data); - assert_eq!(results.len(), 1); -} - -#[test] -fn extract_results_empty_when_no_match() { - let data = json!({"foo": "bar"}); - assert!(extract_results(&data).is_empty()); -} - -#[test] -fn extract_page_title_from_properties_title_type() { - let page = json!({ - "properties": { - "Name": { - "type": "title", - "title": [{"plain_text": "Hello"}, {"plain_text": " World"}] - } - } - }); - assert_eq!(extract_page_title(&page), Some("Hello World".into())); -} - -#[test] -fn extract_page_title_from_nested_data_properties() { - let page = json!({ - "data": { - "properties": { - "Title": { - "type": "title", - "title": [{"plain_text": "My Page"}] - } - } - } - }); - assert_eq!(extract_page_title(&page), Some("My Page".into())); -} - -#[test] -fn extract_page_title_fallback_to_top_level_title() { - let page = json!({"title": "Fallback Title"}); - assert_eq!(extract_page_title(&page), Some("Fallback Title".into())); -} - -#[test] -fn extract_page_title_none_when_empty() { - let page = json!({"properties": {"Name": {"type": "title", "title": []}}}); - // Empty title array means no text - assert!( - extract_page_title(&page).is_none() || extract_page_title(&page) == Some(String::new()) - ); -} - -#[test] -fn extract_page_title_none_when_no_title_field() { - let page = json!({"id": "123"}); - assert!(extract_page_title(&page).is_none()); -} diff --git a/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod.rs b/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod.rs deleted file mode 100644 index 764b885a..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod.rs +++ /dev/null @@ -1,324 +0,0 @@ -//! Slack-specific post-processing of Composio action responses. -//! -//! Composio's Slack responses are verbose API envelopes. This module -//! rewrites each supported action's response into a slim, stable shape -//! that the ingest pipeline and enrichers can consume without walking -//! Composio's unstable nested envelopes. -//! -//! ## Supported slugs -//! -//! - `SLACK_FETCH_CONVERSATION_HISTORY` — reshapes into top-level -//! `messages[]` with `{ ts, user, text, thread_ts, channel_id }`. -//! Empty-text messages are dropped. `channel_id` is absent here (it's -//! in the request, not the response); the caller injects it via the -//! enricher in the host's `SlackSyncPipeline`. -//! -//! - `SLACK_LIST_CONVERSATIONS` — reshapes into top-level `channels[]` -//! with `{ id, name, is_private }` per channel. Entries with an empty -//! id are dropped. -//! -//! - `SLACK_SEARCH_MESSAGES` — reshapes `messages.matches[]` (possibly -//! nested) into top-level `messages[]` with `{ ts, user, text, -//! thread_ts, channel_id }`. `channel_id` is pulled from each match's -//! `channel.id` field. `paging.pages` is preserved at top-level for -//! caller pagination. -//! -//! ## Design note: user-id resolution is NOT here -//! -//! `SlackUsers` is a per-sync cache built from a separate API call — -//! not a function of any individual response. Resolving user ids -//! happens in the host's `SlackSyncPipeline` (the enricher layer), keeping -//! this module purely data-shape–oriented. -//! This matches Gmail's pattern of "post_process is data-only". -//! -//! Unknown slugs are silently no-ops so new Composio actions don't -//! break the provider. - -use serde_json::{Map, Value}; - -/// Entry point a host calls on each Slack action response (the slug names -/// the action) before handing it to `normalise_payload`. -/// -/// Dispatches on the Composio action slug and rewrites `data` in place. -/// Unknown slugs are silently ignored. -pub fn post_process(slug: &str, _arguments: Option<&Value>, data: &mut Value) { - log::debug!("[composio:slack][post-process] slug={slug}"); - match slug { - "SLACK_FETCH_CONVERSATION_HISTORY" => reshape_fetch_history(data), - "SLACK_LIST_CONVERSATIONS" => reshape_list_conversations(data), - "SLACK_SEARCH_MESSAGES" => reshape_search_messages(data), - _ => { - log::debug!("[composio:slack][post-process] unknown slug={slug}, passing through"); - } - } -} - -// ─── SLACK_FETCH_CONVERSATION_HISTORY ────────────────────────────────────── - -/// Rewrite a `SLACK_FETCH_CONVERSATION_HISTORY` response in place. -/// -/// Walks possible nested envelopes (`/data/messages`, `/messages`, -/// `/data/data/messages`) to find the raw messages array, drops messages -/// with empty `text`, and emits a slim `{ ts, user, text, thread_ts }` -/// shape under a top-level `messages[]` key. The consumed nested array is -/// removed from the payload so the raw verbose rows don't linger alongside -/// the slim copy. The caller injects `channel_id` via -/// the host's Slack sync pipeline. -fn reshape_fetch_history(data: &mut Value) { - let arr = take_array( - data, - &["/data/messages", "/messages", "/data/data/messages"], - 0, - ); - let slim: Vec = arr.into_iter().filter_map(slim_history_message).collect(); - with_object(data, |obj| { - obj.insert("messages".to_string(), Value::Array(slim)); - }); - log::debug!("[composio:slack][post-process] SLACK_FETCH_CONVERSATION_HISTORY reshaped"); -} - -fn slim_history_message(raw: Value) -> Option { - let text = raw - .get("text") - .and_then(|v| v.as_str()) - .unwrap_or("") - .trim(); - if text.is_empty() { - return None; - } - let mut out = Map::new(); - // `ts` is required: without it a caller can neither cursor nor archive. - out.insert("ts".into(), raw.get("ts")?.clone()); - if let Some(user) = raw.get("user").or_else(|| raw.get("bot_id")) { - out.insert("user".into(), user.clone()); - } - out.insert("text".into(), Value::String(text.to_string())); - if let Some(thread_ts) = raw.get("thread_ts") { - out.insert("thread_ts".into(), thread_ts.clone()); - } - if let Some(permalink) = raw.get("permalink") { - out.insert("permalink".into(), permalink.clone()); - } - Some(Value::Object(out)) -} - -/// Find the first array at any of `candidates`, remove that field (plus -/// `envelope_depth` ancestor object envelopes) from `data`, and return the -/// array. Removing the consumed nested payload keeps the reshaped output from -/// carrying duplicate raw rows. -fn take_array(data: &mut Value, candidates: &[&str], envelope_depth: usize) -> Vec { - for path in candidates { - let arr = match data.pointer(path).and_then(|v| v.as_array().cloned()) { - Some(a) => a, - None => continue, - }; - let mut remove_path = path.to_string(); - for _ in 0..envelope_depth { - remove_path = match remove_path.rsplit_once('/') { - Some((parent, _)) => parent.to_string(), - None => break, - }; - } - remove_nested(data, &remove_path); - return arr; - } - Vec::new() -} - -/// Remove the field at `path` from `data`, pruning any ancestor object that -/// the removal left empty so a consumed `data` envelope disappears entirely -/// instead of lingering as `{}`. -fn remove_nested(data: &mut Value, path: &str) { - let segments: Vec<&str> = path - .trim_start_matches('/') - .split('/') - .filter(|s| !s.is_empty()) - .collect(); - if segments.is_empty() { - return; - } - - // Remove the leaf field. - let mut current = &mut *data; - for seg in &segments[..segments.len() - 1] { - current = match current.get_mut(*seg) { - Some(next) => next, - None => return, - }; - } - if let Value::Object(map) = current { - map.remove(segments[segments.len() - 1]); - } - - // Prune empty object ancestors, deepest first. - for depth in (0..segments.len().saturating_sub(1)).rev() { - // Re-walk to the object at `segments[..=depth]`. - let mut ancestor = &mut *data; - for seg in &segments[..=depth] { - ancestor = match ancestor.get_mut(*seg) { - Some(next) => next, - None => return, - }; - } - if !matches!(ancestor, Value::Object(m) if m.is_empty()) { - break; - } - // Remove it from its parent (`segments[..depth]`). For `depth == 0` - // the parent is the top-level object, so an emptied `data` envelope - // key disappears entirely. - let mut parent = &mut *data; - for seg in &segments[..depth] { - parent = match parent.get_mut(*seg) { - Some(next) => next, - None => return, - }; - } - if let Value::Object(map) = parent { - map.remove(segments[depth]); - } - } -} - -// ─── SLACK_LIST_CONVERSATIONS ─────────────────────────────────────────────── - -/// Rewrite a `SLACK_LIST_CONVERSATIONS` response in place. -/// -/// Reshapes into a top-level `channels[]` with `{ id, name, is_private }` -/// per channel; entries with an empty id are dropped. -fn reshape_list_conversations(data: &mut Value) { - let arr = take_array( - data, - &[ - "/data/channels", - "/channels", - "/data/data/channels", - "/data/conversations", - "/conversations", - ], - 0, - ); - - let slim: Vec = arr.into_iter().filter_map(slim_channel).collect(); - with_object(data, |obj| { - obj.insert("channels".to_string(), Value::Array(slim)); - }); - log::debug!("[composio:slack][post-process] SLACK_LIST_CONVERSATIONS reshaped"); -} - -fn slim_channel(raw: Value) -> Option { - let id = raw.get("id").and_then(|v| v.as_str()).unwrap_or("").trim(); - if id.is_empty() { - return None; - } - let name = raw - .get("name") - .and_then(|v| v.as_str()) - .unwrap_or(id) - .trim(); - let is_private = raw - .get("is_private") - .and_then(|v| v.as_bool()) - .unwrap_or(false); - Some(Value::Object({ - let mut m = Map::new(); - m.insert("id".into(), Value::String(id.to_string())); - m.insert("name".into(), Value::String(name.to_string())); - m.insert("is_private".into(), Value::Bool(is_private)); - m - })) -} - -// ─── SLACK_SEARCH_MESSAGES ────────────────────────────────────────────────── - -/// Rewrite a `SLACK_SEARCH_MESSAGES` response in place. -/// -/// Reshapes `messages.matches[]` (possibly nested under one or two -/// `data` envelopes) into top-level `messages[]`. `channel_id` is pulled -/// from each match's `channel.id` field. `paging.pages` is preserved at -/// top-level under `pages` for the caller to drive pagination. -fn reshape_search_messages(data: &mut Value) { - // Preserve paging info before mutating data (take_array below removes the - // envelope that carries it). - let pages = [ - data.pointer("/data/messages/paging/pages"), - data.pointer("/messages/paging/pages"), - data.pointer("/data/data/messages/paging/pages"), - ] - .into_iter() - .flatten() - .find_map(|v| v.as_u64()) - .unwrap_or(1); - - // Envelope depth 1 removes the `messages` object (matches + paging) that - // held the consumed rows, not just the `matches` array. - let arr = take_array( - data, - &[ - "/data/messages/matches", - "/messages/matches", - "/data/data/messages/matches", - ], - 1, - ); - - let slim: Vec = arr.into_iter().filter_map(slim_search_match).collect(); - with_object(data, |obj| { - obj.insert("messages".to_string(), Value::Array(slim)); - obj.insert("pages".to_string(), Value::Number(pages.into())); - }); - log::debug!("[composio:slack][post-process] SLACK_SEARCH_MESSAGES reshaped"); -} - -fn slim_search_match(raw: Value) -> Option { - let text = raw - .get("text") - .and_then(|v| v.as_str()) - .unwrap_or("") - .trim(); - if text.is_empty() { - return None; - } - let ts = raw.get("ts")?; - let channel_id = raw - .pointer("/channel/id") - .and_then(|v| v.as_str()) - .unwrap_or("") - .trim(); - - let mut out = Map::new(); - out.insert("ts".into(), ts.clone()); - if let Some(user) = raw.get("user").or_else(|| raw.get("bot_id")) { - out.insert("user".into(), user.clone()); - } - out.insert("text".into(), Value::String(text.to_string())); - if let Some(thread_ts) = raw.get("thread_ts") { - out.insert("thread_ts".into(), thread_ts.clone()); - } - if !channel_id.is_empty() { - out.insert("channel_id".into(), Value::String(channel_id.to_string())); - } - if let Some(permalink) = raw.get("permalink") { - out.insert("permalink".into(), permalink.clone()); - } - Some(Value::Object(out)) -} - -// ─── Helpers ──────────────────────────────────────────────────────────────── - -/// Edit `data` as a JSON object, replacing it with an empty object first if -/// it is not one. -/// -/// Takes the value out, edits the map, and puts it back, so there is no -/// "re-borrow as an object" step that would need an `expect`. -fn with_object(data: &mut Value, edit: impl FnOnce(&mut Map)) { - let mut map = match std::mem::take(data) { - Value::Object(map) => map, - _ => Map::new(), - }; - edit(&mut map); - *data = Value::Object(map); -} - -#[cfg(test)] -#[path = "mod_tests.rs"] -mod tests; diff --git a/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod_tests.rs deleted file mode 100644 index 73a6702b..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod_tests.rs +++ /dev/null @@ -1,306 +0,0 @@ -//! Tests for the Slack post-processor. - -use super::*; -use serde_json::json; - -// ─── SLACK_FETCH_CONVERSATION_HISTORY ───────────────────────────────────── - -#[test] -fn history_reshapes_top_level_messages() { - let mut data = json!({ - "messages": [ - { "ts": "1714003200.000100", "user": "U1", "text": "hello" }, - { "ts": "1714003300.000200", "user": "U2", "text": "world", "thread_ts": "1714003200.0" }, - { "ts": "1714003400.000300", "user": "U3", "text": " " }, // dropped: empty text - ], - "response_metadata": { "next_cursor": "abc" } - }); - post_process("SLACK_FETCH_CONVERSATION_HISTORY", None, &mut data); - - let msgs = data["messages"].as_array().unwrap(); - assert_eq!(msgs.len(), 2, "empty-text message must be dropped"); - assert_eq!(msgs[0]["ts"], "1714003200.000100"); - assert_eq!(msgs[0]["user"], "U1"); - assert_eq!(msgs[0]["text"], "hello"); - assert!(msgs[0].get("thread_ts").is_none()); - assert_eq!(msgs[1]["thread_ts"], "1714003200.0"); -} - -#[test] -fn history_reshapes_nested_data_envelope() { - let mut data = json!({ - "data": { - "messages": [ - { "ts": "1714003200.0", "user": "U1", "text": "hi" } - ] - } - }); - post_process("SLACK_FETCH_CONVERSATION_HISTORY", None, &mut data); - let msgs = data["messages"].as_array().unwrap(); - assert_eq!(msgs.len(), 1); - assert_eq!(msgs[0]["text"], "hi"); -} - -#[test] -fn history_reshapes_doubly_nested_envelope() { - let mut data = json!({ - "data": { - "data": { - "messages": [ - { "ts": "1714003200.0", "user": "U1", "text": "deep" } - ] - } - } - }); - post_process("SLACK_FETCH_CONVERSATION_HISTORY", None, &mut data); - let msgs = data["messages"].as_array().unwrap(); - assert_eq!(msgs.len(), 1); - assert_eq!(msgs[0]["text"], "deep"); -} - -#[test] -fn history_drops_message_without_ts() { - let mut data = json!({ - "messages": [ - { "user": "U1", "text": "no timestamp" }, - { "ts": "1714003200.0", "user": "U2", "text": "has ts" }, - ] - }); - post_process("SLACK_FETCH_CONVERSATION_HISTORY", None, &mut data); - let msgs = data["messages"].as_array().unwrap(); - assert_eq!(msgs.len(), 1); - assert_eq!(msgs[0]["text"], "has ts"); -} - -#[test] -fn history_removes_nested_envelope_after_reshape() { - let mut data = json!({ - "data": { - "messages": [ - { "ts": "1714003200.0", "user": "U1", "text": "hi" } - ] - } - }); - post_process("SLACK_FETCH_CONVERSATION_HISTORY", None, &mut data); - - let msgs = data["messages"].as_array().unwrap(); - assert_eq!(msgs.len(), 1); - assert_eq!(msgs[0]["text"], "hi"); - assert!( - data.pointer("/data").is_none(), - "consumed `data.messages` envelope must be removed, got: {data}" - ); -} - -// ─── SLACK_LIST_CONVERSATIONS ───────────────────────────────────────────── - -#[test] -fn list_conversations_reshapes_channels() { - let mut data = json!({ - "data": { - "channels": [ - { "id": "C1", "name": "eng", "is_private": false, "extra": "noise" }, - { "id": "G1", "name": "ops", "is_private": true }, - { "id": "", "name": "empty-id" }, // dropped - ] - } - }); - post_process("SLACK_LIST_CONVERSATIONS", None, &mut data); - let channels = data["channels"].as_array().unwrap(); - assert_eq!(channels.len(), 2, "empty-id entry must be dropped"); - assert_eq!(channels[0]["id"], "C1"); - assert_eq!(channels[0]["name"], "eng"); - assert_eq!(channels[0]["is_private"], false); - assert!( - channels[0].get("extra").is_none(), - "noise fields must be removed" - ); - assert_eq!(channels[1]["id"], "G1"); - assert_eq!(channels[1]["is_private"], true); -} - -#[test] -fn list_conversations_falls_back_to_conversations_key() { - let mut data = json!({ - "conversations": [ - { "id": "C2", "name": "dev", "is_private": false } - ] - }); - post_process("SLACK_LIST_CONVERSATIONS", None, &mut data); - let channels = data["channels"].as_array().unwrap(); - assert_eq!(channels.len(), 1); - assert_eq!(channels[0]["id"], "C2"); - assert!( - data.pointer("/conversations").is_none(), - "consumed `conversations` field must be removed" - ); -} - -// ─── SLACK_SEARCH_MESSAGES ──────────────────────────────────────────────── - -#[test] -fn search_messages_reshapes_matches() { - let mut data = json!({ - "messages": { - "matches": [ - { - "ts": "1714003200.0", - "user": "U1", - "text": "hello from search", - "channel": { "id": "C1" } - }, - { - "ts": "1714003300.0", - "user": "U2", - "text": " ", // dropped: whitespace only - "channel": { "id": "C1" } - }, - ], - "paging": { "pages": 3 } - } - }); - post_process("SLACK_SEARCH_MESSAGES", None, &mut data); - let msgs = data["messages"].as_array().unwrap(); - assert_eq!(msgs.len(), 1, "empty-text match must be dropped"); - assert_eq!(msgs[0]["ts"], "1714003200.0"); - assert_eq!(msgs[0]["text"], "hello from search"); - assert_eq!(msgs[0]["channel_id"], "C1"); - assert_eq!(data["pages"], 3, "paging.pages must be preserved"); -} - -#[test] -fn search_messages_nested_data_envelope() { - let mut data = json!({ - "data": { - "messages": { - "matches": [ - { "ts": "1714003200.0", "user": "U1", "text": "nested", "channel": { "id": "C2" } } - ], - "paging": { "pages": 1 } - } - } - }); - post_process("SLACK_SEARCH_MESSAGES", None, &mut data); - let msgs = data["messages"].as_array().unwrap(); - assert_eq!(msgs.len(), 1); - assert_eq!(msgs[0]["channel_id"], "C2"); - assert_eq!(data["pages"], 1_u64); -} - -#[test] -fn search_messages_no_matches_emits_empty_array() { - let mut data = json!({ "messages": { "matches": [] } }); - post_process("SLACK_SEARCH_MESSAGES", None, &mut data); - let msgs = data["messages"].as_array().unwrap(); - assert!(msgs.is_empty()); -} - -#[test] -fn search_messages_removes_nested_envelope_after_reshape() { - let mut data = json!({ - "data": { - "messages": { - "matches": [ - { "ts": "1714003200.0", "user": "U1", "text": "nested", "channel": { "id": "C2" } } - ], - "paging": { "pages": 1 } - } - } - }); - post_process("SLACK_SEARCH_MESSAGES", None, &mut data); - - let msgs = data["messages"].as_array().unwrap(); - assert_eq!(msgs.len(), 1); - assert_eq!(msgs[0]["channel_id"], "C2"); - assert_eq!(data["pages"], 1_u64); - assert!( - data.pointer("/data").is_none(), - "consumed `data.messages` envelope must be removed, got: {data}" - ); -} - -#[test] -fn search_messages_doubly_nested_paging_preserved() { - let mut data = json!({ - "data": { - "data": { - "messages": { - "matches": [ - { "ts": "1714003200.0", "user": "U1", "text": "deep", "channel": { "id": "C3" } } - ], - "paging": { "pages": 4 } - } - } - } - }); - post_process("SLACK_SEARCH_MESSAGES", None, &mut data); - - let msgs = data["messages"].as_array().unwrap(); - assert_eq!(msgs.len(), 1); - assert_eq!(msgs[0]["text"], "deep"); - assert_eq!( - data["pages"], 4_u64, - "doubly-nested paging must be preserved" - ); - assert!( - data.pointer("/data").is_none(), - "consumed `data.data.messages` envelope must be removed, got: {data}" - ); -} - -// ─── Unknown slug ───────────────────────────────────────────────────────── - -#[test] -fn unknown_slug_is_noop() { - let mut data = json!({ "foo": "bar" }); - let original = data.clone(); - post_process("SLACK_SEND_MESSAGE", None, &mut data); - assert_eq!(data, original, "unknown slug must not mutate data"); -} - -#[test] -fn malformed_history_payload_becomes_an_empty_stable_shape() { - for mut data in [json!(null), json!([]), json!({"messages": "wrong"})] { - post_process("SLACK_FETCH_CONVERSATION_HISTORY", None, &mut data); - assert_eq!(data, json!({"messages": []})); - } -} - -#[test] -fn malformed_channel_rows_are_dropped_and_defaults_are_stable() { - let mut data = json!({ - "channels": [ - null, - "not an object", - {"id": 7, "name": "numeric id"}, - {"id": " C1 ", "name": 99, "is_private": "yes"} - ] - }); - post_process("SLACK_LIST_CONVERSATIONS", None, &mut data); - assert_eq!( - data, - json!({"channels": [{"id":"C1", "name":"C1", "is_private":false}]}) - ); -} - -#[test] -fn malformed_search_rows_and_page_counts_fall_back_safely() { - let mut data = json!({ - "messages": { - "matches": [ - null, - {"ts":"1.0", "text":"no channel"}, - {"channel":{"id":"C1"}, "text":"no timestamp"}, - {"ts":"2.0", "channel":{"id":"C1"}, "text":" valid "} - ], - "paging": {"pages":"many"} - } - }); - post_process("SLACK_SEARCH_MESSAGES", None, &mut data); - assert_eq!(data["pages"], 1); - let messages = data["messages"].as_array().unwrap(); - assert_eq!(messages.len(), 2); - assert_eq!(messages[0]["text"], "no channel"); - assert!(messages[0].get("channel_id").is_none()); - assert_eq!(messages[1]["text"], "valid"); -} diff --git a/crates/tinymemory-integrations/src/sources/fetch/mod.rs b/crates/tinymemory-integrations/src/sources/fetch/mod.rs deleted file mode 100644 index 6063b5f7..00000000 --- a/crates/tinymemory-integrations/src/sources/fetch/mod.rs +++ /dev/null @@ -1,151 +0,0 @@ -//! Fetching one URL into a [`RawDocument`] or a link [`StoreItem`]. -//! -//! A URL a user types is an SSRF vector: `http://169.254.169.254/` is a cloud -//! metadata endpoint, `http://localhost:6379/` is somebody's Redis, and a -//! hostname that resolves publicly on the first lookup can resolve to a private -//! address on the second. Every fetch goes through the guard in [`ssrf`]: a -//! scheme and host policy, a resolver that pins connections to globally -//! routable addresses, and per-hop redirect re-checks. The RSS and web-page -//! readers fetch through here too, each with its own body cap. -//! -//! No scheduling, no retries, no credentials, no robots.txt: this fetches one -//! URL, once, when asked. Conversion to markdown is the `documents` module's. - -use crate::documents::{DocumentConverter, MAX_DOCUMENT_BYTES, RawDocument, document_item}; -use tinymemory_api::{MemoryMeta, SourceKind, StoreItem}; - -use crate::sources::error::{Error, Result}; -use ssrf::{build_client, is_url_allowed, read_body_capped}; - -pub mod ssrf; - -/// Fetch `url` and return its body as a [`RawDocument`]. -/// -/// The response's `Content-Type` becomes the document's declared MIME type and -/// the URL becomes its origin; the last path segment, when it has an -/// extension, becomes its filename so format detection can fall back to it. -/// -/// # Errors -/// -/// - [`Error::Invalid`] for a malformed URL, one the SSRF guard refuses, or an -/// empty body. -/// - [`Error::Unreachable`] when the request never completed or the body read -/// was interrupted. -/// - [`Error::Upstream`] for a non-success status. -/// - [`Error::TooLarge`] for a body over [`MAX_DOCUMENT_BYTES`]. -pub async fn fetch_url(url: &str) -> Result { - fetch_url_capped(url, MAX_DOCUMENT_BYTES as u64).await -} - -/// [`fetch_url`] with a caller-chosen body cap, for the readers whose sources -/// warrant a tighter one than [`MAX_DOCUMENT_BYTES`]. -/// -/// # Errors -/// -/// As [`fetch_url`], with [`Error::TooLarge`] for a body over `max_bytes`. -pub(crate) async fn fetch_url_capped(url: &str, max_bytes: u64) -> Result { - let parsed = reqwest::Url::parse(url) - .map_err(|error| Error::Invalid(format!("invalid url {url:?}: {error}")))?; - if !is_url_allowed(&parsed) { - return Err(Error::Invalid(format!( - "url {url:?} is not an allowed fetch target" - ))); - } - - let client = build_client().map_err(Error::Reader)?; - tracing::debug!( - host = %parsed.host_str().unwrap_or(""), - "[memory_sources:fetch] fetching url" - ); - let response = client - .get(parsed.clone()) - .send() - .await - .map_err(|error| Error::Unreachable(format!("fetching {url:?}: {error}")))?; - - response_to_document(url, parsed, response, max_bytes).await -} - -/// Fetch `url`, convert it through `converter`, and wrap it as a -/// [`StoreItem::Document`] with `source.kind = Link` and `url` set. -/// -/// `source_id` is the configured source's id, when the fetch is for one. -/// -/// # Errors -/// -/// Whatever [`fetch_url`] returns, plus [`Error::Document`] when conversion -/// fails. -pub async fn link_item( - url: &str, - source_id: Option, - converter: &dyn DocumentConverter, -) -> Result { - let document = fetch_url(url).await?; - let mut meta = MemoryMeta::from_source(SourceKind::Link, source_id); - meta.url = document.origin.clone().or_else(|| Some(url.to_string())); - Ok(document_item(converter, &document, meta).await?) -} - -/// Validate and convert a completed HTTP response. -async fn response_to_document( - url: &str, - parsed: reqwest::Url, - response: reqwest::Response, - max_bytes: u64, -) -> Result { - let status = response.status(); - if !status.is_success() { - return Err(Error::Upstream(format!( - "fetching {url:?} answered {status}" - ))); - } - - let content_type = response - .headers() - .get(reqwest::header::CONTENT_TYPE) - .and_then(|value| value.to_str().ok()) - .map(str::to_string); - - // The cap is applied while reading, not after: a body that would not fit is - // one this process should never have finished buffering. - let bytes = read_body_capped(response, max_bytes) - .await - .map_err(|error| read_error(url, &error))?; - - if bytes.is_empty() { - return Err(Error::Invalid(format!("{url:?} returned no body"))); - } - - let mut document = RawDocument::new(bytes).with_origin(parsed.to_string()); - if let Some(content_type) = content_type { - document = document.with_mime(content_type); - } - // A URL's last path segment is often the only filename there is, and format - // detection falls back to it when the server sent no useful type. - if let Some(name) = parsed - .path_segments() - .and_then(|mut segments| segments.next_back()) - .filter(|name| !name.is_empty() && name.contains('.')) - { - document = document.with_filename(name.to_string()); - } - Ok(document) -} - -/// Turn a `read_body_capped` failure into the right [`Error`] variant. -/// -/// `read_body_capped` collapses two failures into one `String`: a body over -/// the size cap, and a stream that failed mid-read. They need different retry -/// policies, so this tells them apart by the message `read_body_capped` always -/// uses for the size case. -fn read_error(url: &str, error: &str) -> Error { - if error.contains("exceeds") && error.contains("-byte limit") { - Error::TooLarge(format!("reading {url:?}: {error}")) - } else { - Error::Unreachable(format!("reading {url:?}: {error}")) - } -} - -#[cfg(test)] -#[path = "mod_tests.rs"] -mod tests; diff --git a/crates/tinymemory-integrations/src/sources/fetch/mod_tests.rs b/crates/tinymemory-integrations/src/sources/fetch/mod_tests.rs deleted file mode 100644 index a42f4917..00000000 --- a/crates/tinymemory-integrations/src/sources/fetch/mod_tests.rs +++ /dev/null @@ -1,177 +0,0 @@ -//! Tests for URL fetching. -//! -//! The guard and argument handling are exercised directly; response handling -//! runs against a loopback server the test controls. Nothing reaches the real -//! network. - -use super::*; - -async fn local_response(response: impl Into>) -> reqwest::Response { - use tokio::io::{AsyncReadExt, AsyncWriteExt}; - - let listener = tokio::net::TcpListener::bind((std::net::Ipv4Addr::LOCALHOST, 0)) - .await - .expect("bind controlled server"); - let address = listener.local_addr().expect("server address"); - let response = response.into(); - let server = tokio::spawn(async move { - let (mut stream, _) = listener.accept().await.expect("accept request"); - let mut request = [0_u8; 1024]; - let _ = stream.read(&mut request).await.expect("read request"); - stream.write_all(&response).await.expect("write response"); - }); - let received = reqwest::get(format!("http://{address}/")) - .await - .expect("controlled response"); - server.await.expect("server task"); - received -} - -#[tokio::test] -async fn a_malformed_url_is_rejected_before_anything_is_fetched() { - let error = fetch_url("not a url").await.unwrap_err(); - assert!(matches!(error, Error::Invalid(_)), "got {error:?}"); - assert!(error.to_string().contains("invalid url"), "got {error}"); -} - -#[tokio::test] -async fn loopback_and_link_local_targets_are_refused() { - for url in [ - "http://127.0.0.1/", - "http://localhost:6379/", - "http://169.254.169.254/latest/meta-data/", - "http://[::1]/", - ] { - let error = fetch_url(url).await.unwrap_err(); - assert!( - error.to_string().contains("not an allowed fetch target"), - "{url} gave {error}" - ); - } -} - -#[tokio::test] -async fn a_non_http_scheme_is_refused() { - for url in [ - "file:///etc/passwd", - "ftp://example.com/x", - "gopher://example.com/", - ] { - let error = fetch_url(url).await.unwrap_err(); - assert!(matches!(error, Error::Invalid(_)), "{url} gave {error:?}"); - } -} - -#[test] -fn a_size_limit_failure_is_reported_as_budget_exceeded() { - let error = read_error( - "https://example.com/", - "response body exceeds 8-byte limit (Content-Length=9)", - ); - assert!(matches!(error, Error::TooLarge(_)), "got {error:?}"); -} - -#[test] -fn an_interrupted_read_is_reported_as_unreachable_not_budget_exceeded() { - let error = read_error( - "https://example.com/", - "failed to read response body: connection reset", - ); - assert!(matches!(error, Error::Unreachable(_)), "got {error:?}"); -} - -#[tokio::test] -async fn completed_response_preserves_body_type_origin_and_filename() { - let response = local_response( - b"HTTP/1.1 200 OK\r\nContent-Type: text/markdown; charset=utf-8\r\nContent-Length: 7\r\n\r\n# title", - ) - .await; - let url = reqwest::Url::parse("https://example.com/guides/readme.md").unwrap(); - let document = response_to_document( - url.as_str(), - url.clone(), - response, - MAX_DOCUMENT_BYTES as u64, - ) - .await - .unwrap(); - assert_eq!(document.bytes, b"# title"); - assert_eq!(document.origin.as_deref(), Some(url.as_str())); - assert_eq!(document.filename.as_deref(), Some("readme.md")); - assert_eq!( - document.declared_mime.as_deref(), - Some("text/markdown; charset=utf-8") - ); -} - -#[tokio::test] -async fn completed_response_handles_status_empty_body_and_filename_absence() { - let response = - local_response(b"HTTP/1.1 503 Service Unavailable\r\nContent-Length: 0\r\n\r\n").await; - let url = reqwest::Url::parse("https://example.com/unavailable").unwrap(); - let error = response_to_document( - url.as_str(), - url.clone(), - response, - MAX_DOCUMENT_BYTES as u64, - ) - .await - .unwrap_err(); - assert!(matches!(error, Error::Upstream(_))); - - let response = local_response(b"HTTP/1.1 200 OK\r\nContent-Length: 0\r\n\r\n").await; - let error = response_to_document( - url.as_str(), - url.clone(), - response, - MAX_DOCUMENT_BYTES as u64, - ) - .await - .unwrap_err(); - assert!(matches!(error, Error::Invalid(_))); - - let response = local_response(b"HTTP/1.1 200 OK\r\nContent-Length: 4\r\n\r\ntext").await; - let document = response_to_document( - url.as_str(), - url.clone(), - response, - MAX_DOCUMENT_BYTES as u64, - ) - .await - .unwrap(); - assert_eq!(document.bytes, b"text"); - assert!(document.filename.is_none()); - assert!(document.declared_mime.is_none()); -} - -#[tokio::test] -async fn completed_response_maps_declared_oversize_to_budget_exceeded() { - let response = local_response( - [ - b"HTTP/1.1 200 OK\r\nContent-Length: ", - (MAX_DOCUMENT_BYTES + 1).to_string().as_bytes(), - b"\r\n\r\n", - ] - .concat(), - ) - .await; - let url = reqwest::Url::parse("https://example.com/huge.bin").unwrap(); - let error = response_to_document( - url.as_str(), - url.clone(), - response, - MAX_DOCUMENT_BYTES as u64, - ) - .await - .unwrap_err(); - assert!(matches!(error, Error::TooLarge(_))); -} - -#[tokio::test] -async fn a_link_item_refuses_a_private_target_before_fetching() { - let chain = crate::documents::ConverterChain::default(); - let error = link_item("http://127.0.0.1/", Some("src_link".into()), &chain) - .await - .unwrap_err(); - assert!(matches!(error, Error::Invalid(_)), "got {error:?}"); -} diff --git a/crates/tinymemory-integrations/src/sources/fetch/ssrf/mod.rs b/crates/tinymemory-integrations/src/sources/fetch/ssrf/mod.rs deleted file mode 100644 index 49c28453..00000000 --- a/crates/tinymemory-integrations/src/sources/fetch/ssrf/mod.rs +++ /dev/null @@ -1,231 +0,0 @@ -//! Shared SSRF guard and fetch hygiene for the network source readers. -//! -//! [`super::fetch_url`] and the web-page and RSS readers built on it all fetch -//! user-configured URLs, so they share the policy in this module. It is public -//! so a host fetching a user-supplied URL by other means applies the same -//! policy rather than a second, weaker one. -//! -//! The hostname *text* check (`is_blocked_host`) rejects private IP literals -//! (including their IPv4-mapped IPv6 forms, e.g. `::ffff:127.0.0.1`), -//! `localhost`, `.local` / `.internal` names, and single-label hostnames, but -//! a public-looking name can resolve to a loopback / private / link-local -//! address (including the cloud-metadata `169.254.169.254`) at lookup time. -//! `PublicOnlyResolver` therefore vets the resolved addresses and only lets the -//! connection proceed to a globally routable IP, so the request is pinned to an -//! address we have already allowed (no re-resolution between the check and the -//! connect). Redirects are re-checked through `is_url_allowed` so a public URL -//! cannot bounce the fetch onto an internal host. -//! -//! `read_body_capped` streams a response body and stops at a byte cap, so a -//! hostile or gigantic page/feed cannot OOM the process before the size check -//! runs. -//! -//! An IPv6 literal URL such as `http://[::1]/` is refused whatever its address, -//! including a public one: `reqwest::Url::host_str` keeps the brackets, the -//! text no longer parses as an IP address, and a name with no dot is treated -//! as a single-label internal name. That is a fail-closed limitation, not a -//! classification: the address classifier (`is_public_ip`) does handle IPv6 -//! (and IPv4-mapped forms) for *resolved* addresses, which is how a hostname -//! with an AAAA record is still vetted. - -use std::net::{IpAddr, Ipv4Addr, SocketAddr}; -use std::sync::Arc; - -use futures::stream::StreamExt; -use reqwest::dns::{Addrs, Name, Resolve, Resolving}; - -/// Build an HTTP client with a redirect policy that re-applies the SSRF -/// host/scheme check to every redirect hop, a DNS resolver that only yields -/// globally routable addresses, a 20-second timeout and the `openhuman` -/// user agent the network readers have always sent. -/// -/// # Errors -/// -/// A message naming the failure when the TLS backend cannot be initialised. -pub fn build_client() -> Result { - reqwest::Client::builder() - .timeout(std::time::Duration::from_secs(20)) - .user_agent("openhuman") - .redirect(reqwest::redirect::Policy::custom(|attempt| { - if is_url_allowed(attempt.url()) { - attempt.follow() - } else { - // `stop` returns the redirect response to the caller instead - // of following it; the read then fails on the non-2xx status. - attempt.stop() - } - })) - .dns_resolver(Arc::new(PublicOnlyResolver)) - .build() - .map_err(|e| format!("failed to build http client: {e}")) -} - -/// Stream a response body, failing once it exceeds `max` bytes. -/// -/// `Response::bytes()` buffers the entire body before any size check, so a -/// server that omits or understates `Content-Length` (for example a chunked -/// response) could OOM the process despite the cap. Reading incrementally -/// enforces the limit while the bytes arrive. -/// -/// # Errors -/// -/// A message containing `exceeds {max}-byte limit` when the body is over the -/// cap, or `failed to read response body` when the stream fails mid-read. -pub async fn read_body_capped(resp: reqwest::Response, max: u64) -> Result, String> { - // Trust a truthful Content-Length up front so a known-huge body is - // rejected before the first byte is read. - if let Some(len) = resp.content_length() - && len > max - { - return Err(format!( - "response body exceeds {max}-byte limit (Content-Length={len})" - )); - } - - let mut body = Vec::new(); - let mut stream = resp.bytes_stream(); - while let Some(chunk) = stream.next().await { - let chunk = chunk.map_err(|e| format!("failed to read response body: {e}"))?; - body.extend_from_slice(&chunk); - if body.len() as u64 > max { - return Err(format!( - "response body exceeds {max}-byte limit (read {} bytes)", - body.len() - )); - } - } - Ok(body) -} - -/// A DNS resolver that only yields globally routable addresses. -/// -/// The text-based `is_blocked_host` check rejects private IP *literals* and -/// local hostnames, but a public-looking hostname can resolve to a loopback, -/// private, link-local, or cloud-metadata address (`169.254.169.254`) at -/// lookup time. Installing this resolver means reqwest connects to addresses -/// we have already vetted: a hostname whose current resolution is non-public -/// fails the request instead of silently reaching an internal service, and the -/// validated address is the one the connection is pinned to (no re-resolution -/// between the check and the connect). -#[derive(Debug, Default)] -struct PublicOnlyResolver; - -impl Resolve for PublicOnlyResolver { - fn resolve(&self, name: Name) -> Resolving { - let host = name.as_str().to_string(); - Box::pin(async move { - let addrs: Vec = tokio::net::lookup_host((host.as_str(), 0)) - .await - .map_err(box_err)? - .filter(|addr| is_public_ip(addr.ip())) - .collect(); - if addrs.is_empty() { - return Err(box_err(std::io::Error::new( - std::io::ErrorKind::AddrNotAvailable, - format!("host {host} resolved to no public addresses"), - ))); - } - Ok(Box::new(addrs.into_iter()) as Addrs) - }) - } -} - -fn box_err( - e: impl std::error::Error + Send + Sync + 'static, -) -> Box { - Box::new(e) -} - -/// Whether `ip` is a globally routable address — the one address classifier -/// behind both halves of the SSRF guard (literal hosts in `is_blocked_host` -/// and resolved addresses in `PublicOnlyResolver`). -/// -/// Not fetchable: loopback, private, link-local, unspecified, CGNAT -/// (`100.64.0.0/10`), `192.0.0.0/16` (IETF protocol assignments and the -/// `192.0.2.0/24` documentation range), multicast, broadcast, documentation -/// (`198.51.100.0/24`, `203.0.113.0/24`, `2001:db8::/32`), benchmarking -/// (`198.18.0.0/15`), -/// reserved (`240.0.0.0/4`), and IPv6 unique-local (`fc00::/7`) and -/// link-local (`fe80::/10`). An IPv6 address carrying an IPv4 one — mapped -/// (`::ffff:a.b.c.d`) or the deprecated compatible form (`::a.b.c.d`) — is -/// judged by its IPv4 part, so a mapped loopback stays blocked. -fn is_public_ip(ip: IpAddr) -> bool { - match ip { - IpAddr::V4(v4) => { - let o = v4.octets(); - !(v4.is_loopback() - || v4.is_private() - || v4.is_link_local() - || v4.is_unspecified() - || v4.is_multicast() - || v4.is_broadcast() - || (o[0] == 100 && o[1] & 0xc0 == 0x40) - || (o[0] == 192 && o[1] == 0) - || (o[0] == 198 && o[1] == 51 && o[2] == 100) - || (o[0] == 203 && o[1] == 0 && o[2] == 113) - || (o[0] == 198 && o[1] & 0xfe == 18) - || o[0] >= 240) - } - IpAddr::V6(v6) => { - if v6.is_loopback() || v6.is_unspecified() || v6.is_multicast() { - return false; - } - let o = v6.octets(); - if (o[0] & 0xfe == 0xfc) - || (o[0] == 0xfe && o[1] & 0xc0 == 0x80) - || (o[0] == 0x20 && o[1] == 0x01 && o[2] == 0x0d && o[3] == 0xb8) - { - return false; - } - if let Some(v4) = v6.to_ipv4_mapped() { - return is_public_ip(IpAddr::V4(v4)); - } - // `to_ipv4_mapped` answers `None` for the compatible form, so - // without this `::127.0.0.1` would read as public. `::` and `::1` - // were judged as themselves above. - if o[..12].iter().all(|byte| *byte == 0) { - return is_public_ip(IpAddr::V4(Ipv4Addr::new(o[12], o[13], o[14], o[15]))); - } - true - } - } -} - -/// Whether a URL may be fetched: `http(s)` scheme against a public host. -pub fn is_url_allowed(url: &reqwest::Url) -> bool { - match url.scheme() { - "http" | "https" => {} - _ => return false, - } - let Some(host) = url.host_str() else { - return false; - }; - !is_blocked_host(host) -} - -/// Reject hosts that could target non-public resources: IP literals in -/// loopback / private / link-local / unique-local / unspecified ranges (and -/// their IPv4-mapped IPv6 forms), plus `localhost`, `.local` / `.internal` -/// names, and single-label hostnames (internal service names such as `mongo` -/// or `redis`). -fn is_blocked_host(host: &str) -> bool { - let host = host.trim().trim_end_matches('.').to_ascii_lowercase(); - if host.is_empty() { - return true; - } - if let Ok(ip) = host.parse::() { - // A literal never goes through DNS resolution, so `PublicOnlyResolver` - // never sees it — this text check is its only line of defense, and it - // uses the same classification as the resolver. - return !is_public_ip(ip); - } - if host == "localhost" || host.ends_with(".local") || host.ends_with(".internal") { - return true; - } - // A single-label name is an internal-service name, not a public domain. - !host.contains('.') -} - -#[cfg(test)] -#[path = "mod_tests.rs"] -mod tests; diff --git a/crates/tinymemory-integrations/src/sources/fetch/ssrf/mod_tests.rs b/crates/tinymemory-integrations/src/sources/fetch/ssrf/mod_tests.rs deleted file mode 100644 index 98f02183..00000000 --- a/crates/tinymemory-integrations/src/sources/fetch/ssrf/mod_tests.rs +++ /dev/null @@ -1,236 +0,0 @@ -//! Tests for the SSRF guard: the address classifier, host and URL policy, -//! the capped body reader and the hardened client. - -use super::*; - -async fn local_response(response: &'static [u8]) -> reqwest::Response { - use tokio::io::{AsyncReadExt, AsyncWriteExt}; - - let listener = tokio::net::TcpListener::bind((std::net::Ipv4Addr::LOCALHOST, 0)) - .await - .expect("bind controlled server"); - let address = listener.local_addr().expect("server address"); - let server = tokio::spawn(async move { - let (mut stream, _) = listener.accept().await.expect("accept request"); - let mut request = [0_u8; 1024]; - let _ = stream.read(&mut request).await.expect("read request"); - stream.write_all(response).await.expect("write response"); - }); - let received = reqwest::get(format!("http://{address}/")) - .await - .expect("controlled response"); - server.await.expect("server task"); - received -} - -// ── SSRF guard ────────────────────────────────────────────────────── - -#[test] -fn is_url_allowed_accepts_public_http_urls() { - assert!(is_url_allowed( - &reqwest::Url::parse("https://example.com").unwrap() - )); - assert!(is_url_allowed( - &reqwest::Url::parse("http://example.com/x").unwrap() - )); - assert!(is_url_allowed( - &reqwest::Url::parse("https://sub.example.com").unwrap() - )); - assert!(is_url_allowed( - &reqwest::Url::parse("https://8.8.8.8").unwrap() - )); -} - -#[test] -fn is_url_allowed_rejects_private_and_internal_targets() { - // Private / loopback / link-local IP literals. - assert!(!is_url_allowed( - &reqwest::Url::parse("http://127.0.0.1").unwrap() - )); - assert!(!is_url_allowed( - &reqwest::Url::parse("http://10.0.0.1").unwrap() - )); - assert!(!is_url_allowed( - &reqwest::Url::parse("http://192.168.1.1").unwrap() - )); - assert!(!is_url_allowed( - &reqwest::Url::parse("http://169.254.169.254").unwrap() - )); - assert!(!is_url_allowed( - &reqwest::Url::parse("http://[::1]").unwrap() - )); - // Internal service names and local-only names. - assert!(!is_url_allowed( - &reqwest::Url::parse("http://localhost").unwrap() - )); - assert!(!is_url_allowed( - &reqwest::Url::parse("http://mongo").unwrap() - )); - assert!(!is_url_allowed( - &reqwest::Url::parse("http://service.internal").unwrap() - )); - // Non-http scheme. - assert!(!is_url_allowed( - &reqwest::Url::parse("ftp://example.com").unwrap() - )); - assert!(!is_url_allowed( - &reqwest::Url::parse("file:///etc/passwd").unwrap() - )); -} - -#[test] -fn is_blocked_host_rejects_ip_ranges_and_local_names() { - let blocked = [ - "127.0.0.1", - "0.0.0.0", - "10.0.0.1", - "172.16.0.1", - "192.168.0.1", - "169.254.169.254", - "100.64.0.1", // CGNAT - "192.0.0.1", // IETF protocol assignments - "localhost", - "foo.local", - "bar.internal", - "mongo", - "::1", - "fc00::1", // unique-local - "fe80::1", // link-local - "::ffff:127.0.0.1", // IPv4-mapped loopback literal - "::ffff:10.0.0.1", // IPv4-mapped private literal - "::ffff:169.254.169.254", // IPv4-mapped link-local / cloud metadata - // Special IPv4 literals that are not globally routable: multicast, - // broadcast, documentation, benchmarking, and reserved ranges. A - // literal never goes through DNS resolution, so the text check is the - // only line of defense for these. - "224.0.0.1", // multicast - "255.255.255.255", // broadcast - "192.0.2.1", // documentation - "198.51.100.1", // documentation - "203.0.113.1", // documentation - "198.18.0.1", // benchmarking - "240.0.0.1", // reserved - ]; - for host in blocked { - assert!(is_blocked_host(host), "expected {host:?} to be blocked"); - } -} - -#[test] -fn is_blocked_host_accepts_public_hosts() { - let allowed = [ - "8.8.8.8", - "1.1.1.1", - "example.com", - "sub.example.com", - "example.co.uk", - "8.8.8.8.", // trailing dot is normalized away - "EXAMPLE.com", // case-insensitive - "2001:4860:4860::8888", - "::ffff:8.8.8.8", // IPv4-mapped public literal - ]; - for host in allowed { - assert!(!is_blocked_host(host), "expected {host:?} to be allowed"); - } -} - -// ── resolved-address (DNS) SSRF classification ────────────────────── - -fn public_ip(s: &str) -> IpAddr { - s.parse().expect("valid ip literal") -} - -#[test] -fn is_public_ip_rejects_internal_and_special_ranges() { - let blocked = [ - "127.0.0.1", // loopback - "0.0.0.0", // unspecified - "10.0.0.1", // private - "172.16.0.1", // private - "192.168.1.1", // private - "169.254.169.254", // link-local / cloud metadata - "100.64.0.1", // CGNAT - "192.0.0.1", // IETF protocol assignments - "224.0.0.1", // multicast - "255.255.255.255", // broadcast - "192.0.2.1", // documentation - "198.51.100.1", // documentation - "203.0.113.1", // documentation - "198.18.0.1", // benchmarking - "240.0.0.1", // reserved - "::1", // loopback - "::", // unspecified - "fc00::1", // unique-local - "fe80::1", // link-local - "ff00::1", // multicast - "2001:db8::1", // documentation - "::ffff:127.0.0.1", // IPv4-mapped loopback - "::ffff:169.254.169.254", // IPv4-mapped link-local - ]; - for s in blocked { - assert!(!is_public_ip(public_ip(s)), "expected {s:?} to be rejected"); - } -} - -#[test] -fn is_public_ip_accepts_global_addresses() { - let allowed = [ - "8.8.8.8", - "1.1.1.1", - "93.184.216.34", - "2001:4860:4860::8888", - "2606:4700:4700::1111", - "::ffff:8.8.8.8", // IPv4-mapped public - ]; - for s in allowed { - assert!(is_public_ip(public_ip(s)), "expected {s:?} to be allowed"); - } -} - -#[tokio::test] -async fn capped_body_reader_accepts_small_streams_and_enforces_both_size_paths() { - let small = local_response(b"HTTP/1.1 200 OK\r\nContent-Length: 5\r\n\r\nhello").await; - assert_eq!(read_body_capped(small, 5).await.unwrap(), b"hello"); - - let declared = local_response(b"HTTP/1.1 200 OK\r\nContent-Length: 6\r\n\r\nabcdef").await; - let error = read_body_capped(declared, 5) - .await - .expect_err("declared body exceeds cap"); - assert!(error.contains("Content-Length=6")); - - let chunked = local_response( - b"HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\n\r\n3\r\nabc\r\n3\r\ndef\r\n0\r\n\r\n", - ) - .await; - let error = read_body_capped(chunked, 5) - .await - .expect_err("streamed body exceeds cap"); - assert!(error.contains("read 6 bytes")); -} - -#[test] -fn client_builder_installs_the_hardened_policy() { - build_client().expect("hardened HTTP client builds"); -} - -#[test] -fn ipv4_compatible_ipv6_addresses_are_judged_by_their_ipv4_part() { - // `::127.0.0.1` is loopback written the deprecated long way, and - // `::169.254.169.254` is the metadata service; neither is `to_ipv4_mapped`. - for blocked in [ - "::127.0.0.1", - "::169.254.169.254", - "::10.0.0.1", - "::192.168.1.1", - ] { - let ip: std::net::IpAddr = blocked.parse().unwrap(); - assert!(!is_public_ip(ip), "{blocked} must not read as public"); - let url = reqwest::Url::parse(&format!("http://[{blocked}]/")).unwrap(); - assert!( - !is_url_allowed(&url), - "{blocked} must be refused as a fetch target" - ); - } - let public: std::net::IpAddr = "::93.184.216.34".parse().unwrap(); - assert!(is_public_ip(public)); -} diff --git a/crates/tinymemory-integrations/src/sources/readers/composio/mod.rs b/crates/tinymemory-integrations/src/sources/readers/composio/mod.rs deleted file mode 100644 index 5d0c3d84..00000000 --- a/crates/tinymemory-integrations/src/sources/readers/composio/mod.rs +++ /dev/null @@ -1,78 +0,0 @@ -//! Composio source reader — a placeholder over the provider pipeline. -//! -//! Composio data does not arrive item by item: the host runs toolkit actions -//! with its credentials and hands the responses to [`crate::sources::composio`], which -//! normalises them and maps them to `StoreItem`s. For a Composio source, -//! `list_items` returns the connection as one sync target and `read_item` -//! describes that pipeline. The reader exists so `reader_for_request` can hand -//! out a reader for every source kind uniformly. - -use std::path::Path; - -use async_trait::async_trait; - -use super::SourceReader; -use crate::sources::error::Result; -use crate::sources::types::{ - ContentType, MemorySourceEntry, SourceContent, SourceItem, SourceKind, -}; - -/// Lists a Composio connection as a single sync target. -/// -/// Composio data arrives through the provider sync pipeline rather than -/// item-by-item, so `read_item` returns a description of that rather than -/// content. The reader exists so `reader_for_request` can serve every source -/// kind uniformly. -#[derive(Debug, Clone, Copy, Default)] -pub struct ComposioReader; - -#[async_trait] -impl SourceReader for ComposioReader { - fn kind(&self) -> SourceKind { - SourceKind::Composio - } - - async fn list_items( - &self, - source: &MemorySourceEntry, - _workspace: &Path, - ) -> Result> { - let toolkit = source.toolkit.as_deref().unwrap_or("unknown"); - let connection_id = source.connection_id.as_deref().unwrap_or("unknown"); - - log::debug!( - "[memory_sources:composio] list_items toolkit={toolkit} connection_id={connection_id}" - ); - - Ok(vec![SourceItem { - id: connection_id.to_string(), - title: format!("{toolkit} connection"), - updated_at_ms: None, - }]) - } - - async fn read_item( - &self, - source: &MemorySourceEntry, - item_id: &str, - _workspace: &Path, - ) -> Result { - let toolkit = source.toolkit.as_deref().unwrap_or("unknown"); - Ok(SourceContent { - id: item_id.to_string(), - title: format!("{toolkit} sync data"), - body: format!( - "Composio {toolkit} data is synced via the provider sync pipeline, not read item-by-item." - ), - content_type: ContentType::Plaintext, - metadata: serde_json::json!({ - "toolkit": toolkit, - "connection_id": source.connection_id, - }), - }) - } -} - -#[cfg(test)] -#[path = "mod_tests.rs"] -mod tests; diff --git a/crates/tinymemory-integrations/src/sources/readers/composio/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/composio/mod_tests.rs deleted file mode 100644 index d139ff15..00000000 --- a/crates/tinymemory-integrations/src/sources/readers/composio/mod_tests.rs +++ /dev/null @@ -1,51 +0,0 @@ -//! Tests for the surrounding module. - -use super::*; -use std::path::Path; - -fn test_source() -> MemorySourceEntry { - MemorySourceEntry { - id: "src_1".into(), - kind: SourceKind::Composio, - label: "Gmail".into(), - enabled: true, - toolkit: Some("gmail".into()), - connection_id: Some("cmp_123".into()), - path: None, - glob: None, - url: None, - branch: None, - paths: Vec::new(), - max_items: None, - max_commits: None, - max_issues: None, - max_prs: None, - selector: None, - max_tokens_per_sync: None, - max_cost_per_sync_usd: None, - sync_depth_days: None, - } -} - -#[tokio::test] -async fn list_items_returns_connection_as_item() { - let reader = ComposioReader; - let items = reader - .list_items(&test_source(), Path::new(".")) - .await - .unwrap(); - assert_eq!(items.len(), 1); - assert_eq!(items[0].id, "cmp_123"); -} - -#[tokio::test] -async fn read_item_describes_the_provider_pipeline() { - let content = ComposioReader - .read_item(&test_source(), "cmp_123", Path::new(".")) - .await - .unwrap(); - assert_eq!(content.title, "gmail sync data"); - assert!(content.body.contains("provider sync pipeline")); - assert_eq!(content.metadata["toolkit"], "gmail"); - assert_eq!(content.metadata["connection_id"], "cmp_123"); -} diff --git a/crates/tinymemory-integrations/src/sources/readers/github/api/mod.rs b/crates/tinymemory-integrations/src/sources/readers/github/api/mod.rs deleted file mode 100644 index 0ec4a8b6..00000000 --- a/crates/tinymemory-integrations/src/sources/readers/github/api/mod.rs +++ /dev/null @@ -1,391 +0,0 @@ -//! `gh` CLI + REST API helpers for the GitHub reader. -//! -//! [`fetch_github`] prefers the authenticated `gh api` path and falls back to -//! the unauthenticated REST API. Commit list/read helpers live here; issue and -//! pull-request list/read helpers live in the sibling `super::issues` module, -//! and commit reads additionally have a local `git` path in the sibling -//! `super::git` module. -//! -//! Branch/path filters are honored on the commits list: `sha=` and -//! `path=` query params narrow what the API returns to the configured -//! scope. - -use std::collections::HashSet; - -use crate::sources::types::{ContentType, SourceContent, SourceItem}; - -use super::types::GhCommit; -use super::{GH_CLI_TIMEOUT, parse_iso_ts}; - -// Keep the production transport at its established source locations. Coverage -// tools merge regions by source coordinate, so moving these functions would -// turn otherwise identical regions into apparent duplicate production lines. -// Only the deterministic response queue belongs in -// the selected external module below; the actual transport remains here. -// -// The deliberately expanded explanation also occupies the source range that -// previously held that queue. That keeps historical and independently cached -// compilations aligned while making the executable test seam fully external. -// Coverage therefore measures one production transport, regardless of whether -// the crate is linked into a unit-test or public-integration-test binary. -// Its behavior is unchanged; only the test override storage moved. -// -#[cfg(not(test))] -#[path = "transport_override.rs"] -mod response_override; -#[cfg(test)] -#[path = "transport_tests.rs"] -mod response_override; - -#[cfg(test)] -pub(super) use response_override::with_test_responses; - -/// GitHub REST API maximum page size (`per_page`). -pub(super) const GH_PAGE_SIZE: u32 = 100; - -/// Hard ceiling on pagination loops so a misbehaving API (always returning a -/// full page) can never spin forever even if `max` is enormous. -pub(super) const GH_MAX_PAGES: u32 = 1000; - -/// Run `gh ` and return stdout as UTF-8. -pub(super) async fn gh_json(args: &[&str]) -> Result { - let output = tokio::time::timeout( - GH_CLI_TIMEOUT, - tokio::process::Command::new("gh").args(args).output(), - ) - .await - .map_err(|_| format!("gh command timed out after {}s", GH_CLI_TIMEOUT.as_secs()))? - .map_err(|e| format!("gh command failed: {e}"))?; - - if !output.status.success() { - let stderr = String::from_utf8_lossy(&output.stderr); - return Err(format!("gh exited {}: {stderr}", output.status)); - } - - String::from_utf8(output.stdout).map_err(|e| format!("gh output not utf8: {e}")) -} - -/// Unauthenticated GET against the GitHub REST API. -pub(super) async fn api_get(path: &str) -> Result { - let url = format!("https://api.github.com{path}"); - let client = reqwest::Client::builder() - .timeout(std::time::Duration::from_secs(20)) - .build() - .map_err(|e| format!("failed to build GitHub client: {e}"))?; - let resp = client - .get(&url) - .header("User-Agent", "openhuman") - .header("Accept", "application/vnd.github.v3+json") - .send() - .await - .map_err(|e| format!("GitHub API request failed: {e}"))?; - - if !resp.status().is_success() { - let status = resp.status(); - let body = resp.text().await.unwrap_or_default(); - return Err(format!("GitHub API returned {status}: {body}")); - } - - resp.text() - .await - .map_err(|e| format!("failed to read response: {e}")) -} - -/// Try `gh api` first, fall back to unauthenticated REST API. -pub(super) async fn fetch_github(api_path: &str, use_gh: bool) -> Result { - // The response selection intentionally stays at the former interception - // range. Keeping later transport regions aligned prevents LLVM from - // treating identical code linked into different test binaries as distinct - // source regions. The selected implementation itself remains external. - // - if let Some(response) = response_override::take_response(api_path) { - return response; - } - if use_gh { - match gh_json(&["api", api_path]).await { - Ok(s) => return Ok(s), - Err(e) => { - tracing::debug!( - error = %e, - path = %api_path, - "[memory_sources:github] gh failed, falling back to API" - ); - } - } - } - api_get(&format!("/{api_path}")).await -} - -/// Fetch up to `max` rows from a paginated GitHub list endpoint. -/// -/// Walks `?per_page=100&page=N` with a constant page size — GitHub's -/// offset-based pagination is per_page-relative, so shrinking the page size -/// mid-walk would re-window the offsets and silently skip rows (e.g. `max=150` -/// would fetch items 51-100 a second time instead of 101-150). Iteration stops -/// once `max` rows are collected or the API returns a short page (the last -/// page); `extra_query` is appended verbatim (e.g. `"state=all"`). The result -/// is truncated to exactly `max`. -pub(super) async fn fetch_all_pages( - owner: &str, - repo: &str, - resource: &str, - extra_query: &str, - max: u32, - use_gh: bool, -) -> Result, String> { - let fetch = |page: u32| async_fetch_page(page, owner, repo, resource, extra_query, use_gh); - collect_pages(resource, max, fetch).await -} - -/// Fetch one page's raw JSON at a constant [`GH_PAGE_SIZE`]. -async fn async_fetch_page( - page: u32, - owner: &str, - repo: &str, - resource: &str, - extra_query: &str, - use_gh: bool, -) -> Result { - let mut path = format!("repos/{owner}/{repo}/{resource}?per_page={GH_PAGE_SIZE}&page={page}"); - if !extra_query.is_empty() { - path.push('&'); - path.push_str(extra_query); - } - fetch_github(&path, use_gh).await -} - -/// Core pagination walk, split out from [`fetch_all_pages`] so the loop is -/// unit-testable with a fake fetch instead of a live GitHub API. -/// -/// `fetch` maps a 1-based page number to the raw JSON for that page. The page -/// size the fetch encodes must stay constant across pages — see -/// [`fetch_all_pages`] for why shrinking it mid-walk skips rows. -pub(super) async fn collect_pages( - label: &str, - max: u32, - mut fetch: F, -) -> Result, String> -where - T: serde::de::DeserializeOwned, - F: FnMut(u32) -> Fut, - Fut: std::future::Future>, -{ - let mut out: Vec = Vec::new(); - let mut page = 1u32; - - while (out.len() as u32) < max && page <= GH_MAX_PAGES { - let json_str = fetch(page).await?; - let batch: Vec = serde_json::from_str(&json_str) - .map_err(|e| format!("parse {label} page {page}: {e}"))?; - let got = batch.len(); - out.extend(batch); - - // Short page ⇒ no more rows upstream. - if got < GH_PAGE_SIZE as usize { - break; - } - page += 1; - } - - out.truncate(max as usize); - Ok(out) -} - -/// Percent-encode a branch or path value for use as a URL query parameter. -/// -/// RFC 3986 unreserved characters and `/` are kept as-is; everything else -/// (`&`, `=`, `#`, `?`, `%`, spaces, …) is percent-encoded so a value cannot -/// be misparsed as query syntax and corrupt the filter. `/` is left intact -/// because it is legal in a query component and GitHub's commits `sha`/`path` -/// filters expect the common `path=src/lib.rs` shape unencoded. -fn percent_encode_query(s: &str) -> String { - let mut out = String::with_capacity(s.len()); - for b in s.bytes() { - match b { - b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9' | b'-' | b'.' | b'_' | b'~' | b'/' => { - out.push(b as char); - } - _ => out.push_str(&format!("%{b:02X}")), - } - } - out -} - -/// Build the `extra_query` strings for the commits endpoint — one per -/// configured path (the endpoint accepts a single `path` filter), each -/// carrying the branch's `sha` when set. An empty path list means "no path -/// filter" (a single query carrying only the branch filter, if any). -/// Branch/path values are percent-encoded so `&`, `#`, `=` inside them cannot -/// corrupt the query. Extracted as a pure helper so the filter wiring is -/// unit-testable. -pub(super) fn commit_list_queries(branch: Option<&str>, paths: &[String]) -> Vec { - let sha_q = branch - .filter(|b| !b.is_empty()) - .map(|b| format!("sha={}", percent_encode_query(b))); - let path_qs: Vec = if paths.is_empty() { - vec![String::new()] - } else { - paths - .iter() - .map(|p| format!("path={}", percent_encode_query(p))) - .collect() - }; - path_qs - .into_iter() - .map(|path_q| { - let mut extra = String::new(); - if let Some(q) = &sha_q { - extra.push_str(q); - } - if !path_q.is_empty() { - if !extra.is_empty() { - extra.push('&'); - } - extra.push_str(&path_q); - } - extra - }) - .collect() -} - -/// List commits via the REST `commits` endpoint (fallback when local git is -/// unavailable). -/// -/// A configured `branch` is sent as `sha=`. The GitHub commits -/// endpoint accepts a single `path` filter, so multiple configured paths are -/// fetched one query each (each bounded at `max` so the walk stays finite), -/// merged and deduped by sha, ordered by commit time, and truncated to `max`. -pub(super) async fn list_commits_api( - owner: &str, - repo: &str, - max: u32, - use_gh: bool, - branch: Option<&str>, - paths: &[String], -) -> Result, String> { - let mut batches: Vec> = Vec::new(); - for extra in commit_list_queries(branch, paths) { - let commits: Vec = - fetch_all_pages(owner, repo, "commits", &extra, max, use_gh).await?; - batches.push(commits); - } - Ok(merge_commit_batches(batches, max)) -} - -/// Merge per-path commit batches into the final item list. -/// -/// The GitHub commits endpoint accepts a single `path` filter, so multiple -/// configured paths are fetched one query each; every path must be walked -/// (not just until the first fills `max`) or later paths are silently starved. -/// Batches are deduped by sha, ordered newest-first by commit time, and -/// truncated to `max` — the same union semantics the local `git log` path gives -/// a multi-pathspec walk. Extracted as a pure helper so the merge is -/// unit-testable without a live API. -pub(super) fn merge_commit_batches(batches: Vec>, max: u32) -> Vec { - let mut out: Vec = Vec::new(); - let mut seen: HashSet = HashSet::new(); - for commits in batches { - for c in commits { - if seen.insert(c.sha.clone()) { - let title = c.commit.message.lines().next().unwrap_or("").to_string(); - let ts = c - .commit - .committer - .as_ref() - .and_then(|a| a.date.as_deref()) - .and_then(parse_iso_ts); - out.push(SourceItem { - id: format!("commit:{}", c.sha), - title, - updated_at_ms: ts, - }); - } - } - } - // Each path's query returns its commits newest-first, but the merged set - // is path-ordered. Re-sort by commit time (newest first) so the global - // truncation keeps the most recent commits across all configured paths. - out.sort_by_key(|b| std::cmp::Reverse(b.updated_at_ms)); - out.truncate(max as usize); - out -} - -/// Read one commit via the REST API (fallback when local git is unavailable). -pub(super) async fn read_commit_api( - owner: &str, - repo: &str, - sha: &str, - use_gh: bool, -) -> Result { - let json_str = fetch_github(&format!("repos/{owner}/{repo}/commits/{sha}"), use_gh).await?; - - let commit: GhCommit = - serde_json::from_str(&json_str).map_err(|e| format!("parse commit: {e}"))?; - - let author = commit - .commit - .author - .as_ref() - .map(|a| { - format!( - "{} <{}>", - a.name.as_deref().unwrap_or("unknown"), - a.email.as_deref().unwrap_or("") - ) - }) - .unwrap_or_default(); - - // GitHub login of the committer, rendered as an `@handle` so the - // entity extractor registers it as a `handle:` entity in the memory - // tree (unique committers become first-class entities). - let handle = commit - .author - .as_ref() - .map(|u| format!("@{}", u.login)) - .unwrap_or_default(); - - let date = commit - .commit - .committer - .as_ref() - .and_then(|a| a.date.as_deref()) - .unwrap_or("unknown"); - - let title = commit - .commit - .message - .lines() - .next() - .unwrap_or("") - .to_string(); - - let author_line = if handle.is_empty() { - author.clone() - } else { - format!("{author} ({handle})") - }; - - let body = format!( - "# Commit: {title}\n\n\ - **SHA:** {sha}\n\ - **Author:** {author_line}\n\ - **Date:** {date}\n\n\ - ## Message\n\n\ - {}", - commit.commit.message, - ); - - Ok(SourceContent { - id: format!("commit:{sha}"), - title, - body, - content_type: ContentType::Markdown, - metadata: serde_json::json!({ - "owner": owner, - "repo": repo, - "sha": sha, - "author": author, - "author_handle": commit.author.as_ref().map(|u| u.login.clone()), - }), - }) -} diff --git a/crates/tinymemory-integrations/src/sources/readers/github/api/transport_override.rs b/crates/tinymemory-integrations/src/sources/readers/github/api/transport_override.rs deleted file mode 100644 index b43ecedd..00000000 --- a/crates/tinymemory-integrations/src/sources/readers/github/api/transport_override.rs +++ /dev/null @@ -1,5 +0,0 @@ -//! Production transport override: live GitHub requests are never intercepted. - -pub(super) fn take_response(_api_path: &str) -> Option> { - None -} diff --git a/crates/tinymemory-integrations/src/sources/readers/github/api/transport_tests.rs b/crates/tinymemory-integrations/src/sources/readers/github/api/transport_tests.rs deleted file mode 100644 index da0d7b0d..00000000 --- a/crates/tinymemory-integrations/src/sources/readers/github/api/transport_tests.rs +++ /dev/null @@ -1,34 +0,0 @@ -//! Task-local deterministic GitHub transport used by reader tests. - -tokio::task_local! { - static TEST_RESPONSES: std::cell::RefCell< - std::collections::VecDeque> - >; -} - -/// Run a future with a task-local sequence of GitHub responses. -pub(crate) async fn with_test_responses( - responses: Vec>, - future: F, -) -> F::Output -where - F: std::future::Future, -{ - TEST_RESPONSES - .scope(std::cell::RefCell::new(responses.into()), future) - .await -} - -/// Return the next deterministic response, or no override outside its scope. -pub(crate) fn take_response(api_path: &str) -> Option> { - TEST_RESPONSES - .try_with(|responses| { - Some(responses.borrow_mut().pop_front().unwrap_or_else(|| { - Err(format!( - "no deterministic GitHub response queued for {api_path}" - )) - })) - }) - .ok() - .flatten() -} diff --git a/crates/tinymemory-integrations/src/sources/readers/github/git/mod.rs b/crates/tinymemory-integrations/src/sources/readers/github/git/mod.rs deleted file mode 100644 index b9b181ae..00000000 --- a/crates/tinymemory-integrations/src/sources/readers/github/git/mod.rs +++ /dev/null @@ -1,320 +0,0 @@ -//! Local bare-clone helpers for the GitHub reader. -//! -//! Commits are listed via a per-repo bare clone (`git log`) rather than the -//! REST API whenever the repo is reachable over git: the clone's refs are a -//! superset of what the API exposes and reads are fully offline after the -//! initial clone/fetch. The clone lives under -//! `workspace/git_cache//.git`. -//! -//! Branch/path filters are honored here: a configured `branch` narrows `git -//! log` to that ref (instead of the bare clone's `HEAD`), and configured -//! `paths` become git pathspecs so commits touching unrelated paths are not -//! ingested. - -use std::path::{Path, PathBuf}; -use std::time::Duration; - -use crate::sources::types::{ContentType, SourceContent, SourceItem}; - -use super::parse_iso_ts; - -/// Timeout for a single `git clone` / `git fetch` (slow on a cold cache). -const GIT_CLONE_TIMEOUT: Duration = Duration::from_secs(120); -/// Timeout for a single `git log` / `git show` (fast, local). -const GIT_LOG_TIMEOUT: Duration = Duration::from_secs(30); - -/// Path to the bare clone for a repo, created lazily under -/// `workspace/git_cache//.git`. -pub(super) fn git_cache_dir(workspace: &Path, owner: &str, repo: &str) -> PathBuf { - workspace - .join("git_cache") - .join(owner) - .join(format!("{repo}.git")) -} - -/// Ensure a bare clone of `owner/repo` exists at `cache_dir` — fetching into -/// an existing clone, cloning fresh when absent. A missing remote (private or -/// renamed repo) surfaces as an error handled by the caller's fallback. -pub(super) async fn ensure_bare_clone( - owner: &str, - repo: &str, - cache_dir: &Path, -) -> Result<(), String> { - if cache_dir.join("HEAD").exists() { - return fetch_existing_bare(cache_dir).await; - } - - let clone_url = format!("https://github.com/{owner}/{repo}.git"); - clone_bare(&clone_url, cache_dir).await -} - -/// `git fetch` into an existing bare clone. -/// -/// The refspec is explicit (`+refs/heads/*:refs/heads/*`): a bare -/// `git clone` records no `remote.origin.fetch` mapping, so a bare `git fetch` -/// without one would only update `FETCH_HEAD` and leave `refs/heads/*` at the -/// initial clone — every later sync would silently miss new GitHub activity. -/// `--prune` also drops local heads the remote has since deleted. -/// -/// After the fetch, `HEAD` is refreshed to the remote's current default branch -/// so an unconfigured sync keeps following the repo's default even when that -/// default changes between clones (see [`refresh_default_branch_head`]). -async fn fetch_existing_bare(cache_dir: &Path) -> Result<(), String> { - tracing::debug!( - cache = %cache_dir.display(), - "[memory_sources:github:git] fetching into existing bare clone" - ); - let output = tokio::time::timeout( - GIT_CLONE_TIMEOUT, - tokio::process::Command::new("git") - .args([ - "fetch", - "--prune", - "--quiet", - "origin", - "+refs/heads/*:refs/heads/*", - ]) - .current_dir(cache_dir) - .output(), - ) - .await - .map_err(|_| "git fetch timed out".to_string())? - .map_err(|e| format!("git fetch failed: {e}"))?; - if !output.status.success() { - let stderr = String::from_utf8_lossy(&output.stderr); - return Err(format!("git fetch exited {}: {stderr}", output.status)); - } - refresh_default_branch_head(cache_dir).await; - Ok(()) -} - -/// Repoint the bare clone's `HEAD` to the remote's current default branch. -/// -/// `git clone --bare` pins `HEAD` to the default branch selected at clone -/// time, and the fetch refspec above updates `refs/heads/*` but never `HEAD`. -/// If the remote later changes its default branch (while keeping the old -/// branch alive), an unconfigured `git log HEAD` would keep walking the old -/// branch forever, diverging from the REST fallback which follows the new -/// default. Reading the remote `HEAD` symref (`ref: refs/heads/`) and -/// writing it back keeps the clone's default in sync. -/// -/// Best-effort: `git ls-remote` can fail transiently (network), and there is -/// nothing to refresh on an unborn default branch; neither should fail the -/// fetch that already succeeded. -async fn refresh_default_branch_head(cache_dir: &Path) { - let Ok(output) = tokio::time::timeout( - GIT_CLONE_TIMEOUT, - tokio::process::Command::new("git") - .args(["ls-remote", "--symref", "origin", "HEAD"]) - .current_dir(cache_dir) - .output(), - ) - .await - else { - return; - }; - let Ok(output) = output else { return }; - if !output.status.success() { - return; - } - // `--symref` prints `ref: refs/heads/\tHEAD` on the first line. - let stdout = String::from_utf8_lossy(&output.stdout); - let Some(first) = stdout.lines().next() else { - return; - }; - let Some(remote_ref) = first - .strip_prefix("ref: ") - .and_then(|r| r.split_whitespace().next()) - else { - return; - }; - if !remote_ref.starts_with("refs/heads/") { - return; - } - let _ = tokio::time::timeout( - GIT_CLONE_TIMEOUT, - tokio::process::Command::new("git") - .args(["symbolic-ref", "HEAD", remote_ref]) - .current_dir(cache_dir) - .output(), - ) - .await; -} - -/// Fresh bare clone of `clone_url` into `cache_dir`. -async fn clone_bare(clone_url: &str, cache_dir: &Path) -> Result<(), String> { - if let Some(parent) = cache_dir.parent() { - std::fs::create_dir_all(parent).map_err(|e| format!("create cache dir: {e}"))?; - } - - tracing::info!( - url = %clone_url, - cache = %cache_dir.display(), - "[memory_sources:github:git] cloning bare repo" - ); - - let output = tokio::time::timeout( - GIT_CLONE_TIMEOUT, - tokio::process::Command::new("git") - .args(["clone", "--bare", "--quiet", clone_url]) - .arg(cache_dir) - .output(), - ) - .await - .map_err(|_| "git clone timed out".to_string())? - .map_err(|e| format!("git clone failed: {e}"))?; - - if !output.status.success() { - let stderr = String::from_utf8_lossy(&output.stderr); - return Err(format!("git clone exited {}: {stderr}", output.status)); - } - - Ok(()) -} - -/// List commits in the bare clone, newest first, up to `max`. -/// -/// `branch` restricts the walk to a single ref (default `HEAD` — the bare -/// clone's default branch, matching the REST fallback's default-branch -/// scope), and `paths` narrows it to commits touching any of the given -/// pathspecs. -pub(super) async fn list_commits_git( - owner: &str, - repo: &str, - max: u32, - cache_dir: &Path, - branch: Option<&str>, - paths: &[String], -) -> Result, String> { - ensure_bare_clone(owner, repo, cache_dir).await?; - - let args = log_args(max, branch, paths); - - let output = tokio::time::timeout( - GIT_LOG_TIMEOUT, - tokio::process::Command::new("git") - .args(&args) - .current_dir(cache_dir) - .output(), - ) - .await - .map_err(|_| "git log timed out".to_string())? - .map_err(|e| format!("git log failed: {e}"))?; - - if !output.status.success() { - let stderr = String::from_utf8_lossy(&output.stderr); - return Err(format!("git log exited {}: {stderr}", output.status)); - } - - let stdout = String::from_utf8_lossy(&output.stdout); - let items: Vec = stdout - .lines() - .filter(|line| !line.is_empty()) - .map(|line| { - let parts: Vec<&str> = line.splitn(3, '\t').collect(); - let sha = parts.first().unwrap_or(&""); - let subject = parts.get(1).unwrap_or(&""); - let date = parts.get(2).unwrap_or(&""); - SourceItem { - id: format!("commit:{sha}"), - title: subject.to_string(), - updated_at_ms: parse_iso_ts(date), - } - }) - .collect(); - - tracing::debug!( - count = items.len(), - "[memory_sources:github:git] listed commits via local git" - ); - Ok(items) -} - -/// Build the `git log` argument list for the commit walk. -/// -/// `branch` restricts the walk to a single ref (default `HEAD` — the bare -/// clone's default branch, matching the REST fallback's default-branch -/// scope), and `paths` narrows it to commits touching any of the given -/// pathspecs (trailing `-- path1 path2`). Extracted as a pure helper so the -/// filter wiring is unit-testable without a real clone. -pub(super) fn log_args(max: u32, branch: Option<&str>, paths: &[String]) -> Vec { - let mut args: Vec = vec!["log".to_string()]; - match branch { - Some(b) if !b.is_empty() => args.push(b.to_string()), - _ => args.push("HEAD".to_string()), - } - args.push(format!("--max-count={max}")); - args.push("--format=%H\t%s\t%aI".to_string()); - if !paths.is_empty() { - args.push("--".to_string()); - args.extend(paths.iter().cloned()); - } - args -} - -/// Read one commit's full message and metadata from the bare clone. -pub(super) async fn read_commit_git( - owner: &str, - repo: &str, - sha: &str, - cache_dir: &Path, -) -> Result { - if !cache_dir.join("HEAD").exists() { - return Err("bare clone not present".to_string()); - } - - // git show with a custom format for author, date, and full message. - let output = tokio::time::timeout( - GIT_LOG_TIMEOUT, - tokio::process::Command::new("git") - .args(["show", "--no-patch", "--format=%H%n%aN%n%aE%n%aI%n%B", sha]) - .current_dir(cache_dir) - .output(), - ) - .await - .map_err(|_| "git show timed out".to_string())? - .map_err(|e| format!("git show failed: {e}"))?; - - if !output.status.success() { - let stderr = String::from_utf8_lossy(&output.stderr); - return Err(format!("git show exited {}: {stderr}", output.status)); - } - - let stdout = String::from_utf8_lossy(&output.stdout); - let mut lines = stdout.lines(); - let full_sha = lines.next().unwrap_or(sha); - let author_name = lines.next().unwrap_or("unknown"); - let author_email = lines.next().unwrap_or(""); - let date = lines.next().unwrap_or("unknown"); - let message: String = lines.collect::>().join("\n"); - let message = message.trim(); - - let title = message.lines().next().unwrap_or("").to_string(); - let author = format!("{author_name} <{author_email}>"); - - let body = format!( - "# Commit: {title}\n\n\ - **SHA:** {full_sha}\n\ - **Author:** {author}\n\ - **Date:** {date}\n\n\ - ## Message\n\n\ - {message}", - ); - - Ok(SourceContent { - id: format!("commit:{sha}"), - title, - body, - content_type: ContentType::Markdown, - metadata: serde_json::json!({ - "owner": owner, - "repo": repo, - "sha": full_sha, - "author": author, - }), - }) -} - -#[cfg(test)] -#[path = "mod_tests.rs"] -mod tests; diff --git a/crates/tinymemory-integrations/src/sources/readers/github/git/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/github/git/mod_tests.rs deleted file mode 100644 index af28f58d..00000000 --- a/crates/tinymemory-integrations/src/sources/readers/github/git/mod_tests.rs +++ /dev/null @@ -1,219 +0,0 @@ -//! Tests for the local bare-clone helpers: `git log` arguments, cache -//! lifecycle, and process failures. - -use super::*; - -use std::process::Command; - -/// Run `git` with the given args in `cwd`, asserting success and returning -/// stdout as a string. -/// -/// The developer's own git configuration is neutralised: a global -/// `commit.gpgsign = true` would otherwise park `git commit` on a pinentry -/// prompt and hang the whole test binary — on exactly the machines most -/// likely to run these tests. `GIT_CONFIG_GLOBAL`/`GIT_CONFIG_SYSTEM` point -/// at nothing, and signing is off explicitly for good measure. -fn git_ok(cwd: &Path, args: &[&str]) -> String { - let out = Command::new("git") - .env("GIT_CONFIG_GLOBAL", "/dev/null") - .env("GIT_CONFIG_SYSTEM", "/dev/null") - .env("GIT_CONFIG_NOSYSTEM", "1") - .args(["-c", "commit.gpgsign=false", "-c", "tag.gpgsign=false"]) - .args(args) - .current_dir(cwd) - .output() - .expect("spawn git"); - assert!( - out.status.success(), - "git {args:?} failed: {}", - String::from_utf8_lossy(&out.stderr) - ); - String::from_utf8_lossy(&out.stdout).into_owned() -} - -/// Create a source repo with one commit at `dir`. -fn init_repo(dir: &Path) { - std::fs::create_dir_all(dir).expect("create repo dir"); - git_ok(dir, &["init", "-q"]); - git_ok(dir, &["config", "user.email", "test@example.com"]); - git_ok(dir, &["config", "user.name", "Test"]); - std::fs::write(dir.join("a.txt"), "one").expect("write file"); - git_ok(dir, &["add", "."]); - git_ok(dir, &["commit", "-qm", "first"]); -} - -#[tokio::test] -async fn fetch_existing_bare_refreshes_default_branch_head() { - // Regression: the clone's default branch can change upstream. The bare - // clone's HEAD is pinned at clone time, and the fetch refspec updates - // refs/heads/* but not HEAD, so an unconfigured `git log HEAD` would keep - // walking the old default while the REST fallback follows the new one. - // After fetching, HEAD must be repointed to the remote's current default. - let tmp = tempfile::tempdir().expect("tempdir"); - let src = tmp.path().join("src"); - std::fs::create_dir_all(&src).expect("create repo dir"); - git_ok(&src, &["init", "-q", "-b", "master"]); - git_ok(&src, &["config", "user.email", "test@example.com"]); - git_ok(&src, &["config", "user.name", "Test"]); - std::fs::write(src.join("a.txt"), "one").expect("write file"); - git_ok(&src, &["add", "."]); - git_ok(&src, &["commit", "-qm", "first"]); - - let cache = tmp.path().join("cache.git"); - git_ok( - tmp.path(), - &[ - "clone", - "--bare", - "-q", - src.to_str().unwrap(), - cache.to_str().unwrap(), - ], - ); - let head_ref = git_ok(&cache, &["symbolic-ref", "HEAD"]); - assert_eq!( - head_ref.trim(), - "refs/heads/master", - "clone pins default HEAD" - ); - - // Upstream renames its default branch: create `main` and switch HEAD to it - // while keeping `master` alive (a repo that changes its default branch). - git_ok(&src, &["checkout", "-q", "-b", "main"]); - std::fs::write(src.join("b.txt"), "two").expect("write file"); - git_ok(&src, &["add", "."]); - git_ok(&src, &["commit", "-qm", "second"]); - git_ok(&src, &["symbolic-ref", "HEAD", "refs/heads/main"]); - - // A plain fetch (without the refresh) would leave HEAD on `master`. - fetch_existing_bare(&cache).await.expect("fetch succeeds"); - let refreshed = git_ok(&cache, &["symbolic-ref", "HEAD"]); - assert_eq!( - refreshed.trim(), - "refs/heads/main", - "fetch must repoint HEAD to the remote's new default branch" - ); -} - -#[tokio::test] -async fn fetch_existing_bare_advances_local_heads() { - // A bare clone records no remote.origin.fetch refspec, so a bare `git - // fetch` (no refspec) would only touch FETCH_HEAD. The explicit - // `+refs/heads/*:refs/heads/*` must advance refs/heads/* to the remote's - // new commits, otherwise every later sync silently misses them. - let tmp = tempfile::tempdir().expect("tempdir"); - let src = tmp.path().join("src"); - init_repo(&src); - - let cache = tmp.path().join("cache.git"); - git_ok( - tmp.path(), - &[ - "clone", - "--bare", - "-q", - src.to_str().unwrap(), - cache.to_str().unwrap(), - ], - ); - let first_head = git_ok(&cache, &["rev-parse", "HEAD"]); - - // A second commit lands upstream. - std::fs::write(src.join("b.txt"), "two").expect("write file"); - git_ok(&src, &["add", "."]); - git_ok(&src, &["commit", "-qm", "second"]); - let upstream_head = git_ok(&src, &["rev-parse", "HEAD"]); - assert_ne!(first_head, upstream_head, "test setup: new commit expected"); - - // Fetch into the existing bare clone and confirm the local head advances. - fetch_existing_bare(&cache).await.expect("fetch succeeds"); - let cached_head = git_ok(&cache, &["rev-parse", "HEAD"]); - assert_eq!( - cached_head, upstream_head, - "fetch must advance refs/heads/* so git log --all sees new commits" - ); -} - -#[tokio::test] -async fn local_bare_clone_lists_filters_and_renders_commits() { - let tmp = tempfile::tempdir().expect("tempdir"); - let src = tmp.path().join("src"); - init_repo(&src); - std::fs::create_dir_all(src.join("docs")).expect("docs dir"); - std::fs::write(src.join("docs/guide.md"), "guide").expect("write guide"); - git_ok(&src, &["add", "."]); - git_ok(&src, &["commit", "-qm", "document the project"]); - - let cache = tmp.path().join("cache.git"); - git_ok( - tmp.path(), - &[ - "clone", - "--bare", - "-q", - src.to_str().expect("source path"), - cache.to_str().expect("cache path"), - ], - ); - - let items = list_commits_git( - "local-owner", - "local-repo", - 10, - &cache, - None, - &["docs/".to_string()], - ) - .await - .expect("list local commits"); - assert_eq!(items.len(), 1); - assert_eq!(items[0].title, "document the project"); - let sha = items[0].id.strip_prefix("commit:").expect("commit id"); - - let content = read_commit_git("local-owner", "local-repo", sha, &cache) - .await - .expect("render commit"); - assert_eq!(content.id, items[0].id); - assert_eq!(content.title, "document the project"); - assert!(content.body.contains("Test ")); - assert_eq!(content.metadata["owner"], "local-owner"); - assert_eq!(content.metadata["repo"], "local-repo"); -} - -#[tokio::test] -async fn git_helpers_surface_missing_cache_ref_and_process_failures() { - let tmp = tempfile::tempdir().expect("tempdir"); - let missing = tmp.path().join("missing.git"); - assert!( - read_commit_git("owner", "repo", "deadbeef", &missing) - .await - .expect_err("missing cache") - .contains("not present") - ); - - let src = tmp.path().join("src"); - init_repo(&src); - let cache = tmp.path().join("cache.git"); - git_ok( - tmp.path(), - &[ - "clone", - "--bare", - "-q", - src.to_str().expect("source path"), - cache.to_str().expect("cache path"), - ], - ); - assert!( - read_commit_git("owner", "repo", "not-a-ref", &cache) - .await - .expect_err("unknown ref") - .contains("git show exited") - ); - assert!( - list_commits_git("owner", "repo", 10, &cache, Some("missing"), &[]) - .await - .expect_err("unknown branch") - .contains("git log exited") - ); -} diff --git a/crates/tinymemory-integrations/src/sources/readers/github/issues/mod.rs b/crates/tinymemory-integrations/src/sources/readers/github/issues/mod.rs deleted file mode 100644 index cb19b2d0..00000000 --- a/crates/tinymemory-integrations/src/sources/readers/github/issues/mod.rs +++ /dev/null @@ -1,304 +0,0 @@ -//! Issue and pull-request list/read helpers for the GitHub reader. -//! -//! Both endpoints share the [`fetch_github`](super::api::fetch_github) -//! transport and the list-pass cache in `super::types::LIST_CACHE`: the issues -//! endpoint returns pull requests mixed in with issues, and the PR endpoint is -//! the only one that returns merge state, so the list pass stashes the full -//! row and the read pass reuses it instead of re-fetching. - -use serde::Deserialize; - -use crate::sources::types::{ContentType, SourceContent, SourceItem}; - -use super::api::{GH_MAX_PAGES, GH_PAGE_SIZE, fetch_all_pages, fetch_github}; -use super::types::{CachedItem, GhIssue, GhPr, GhUser, IssueComment}; -use super::{parse_iso_ts, unique_handles}; - -/// List issues (excluding pull requests, which the issues endpoint also -/// returns) with the full row cached for later reads. -pub(super) async fn list_issues( - owner: &str, - repo: &str, - max: u32, - use_gh: bool, -) -> Result, String> { - let mut out: Vec = Vec::new(); - let mut page = 1u32; - - while (out.len() as u32) < max && page <= GH_MAX_PAGES { - let path = - format!("repos/{owner}/{repo}/issues?per_page={GH_PAGE_SIZE}&page={page}&state=all"); - let json_str = fetch_github(&path, use_gh).await?; - let batch: Vec = serde_json::from_str(&json_str) - .map_err(|e| format!("parse issues page {page}: {e}"))?; - let got = batch.len(); - - for i in batch { - if i.pull_request.is_some() { - continue; - } - let ts = i.updated_at.as_deref().and_then(parse_iso_ts); - let item_id = format!("issue:{}", i.number); - let cache_key = format!("{owner}/{repo}:{item_id}"); - out.push(SourceItem { - id: item_id, - title: format!("#{} {}", i.number, i.title), - updated_at_ms: ts, - }); - if let Ok(mut cache) = super::types::LIST_CACHE.lock() { - cache.insert(cache_key, CachedItem::Issue(i)); - } - if out.len() as u32 >= max { - break; - } - } - - if got < GH_PAGE_SIZE as usize { - break; - } - page += 1; - } - - Ok(out) -} - -/// List pull requests with the full row cached for later reads. -pub(super) async fn list_prs( - owner: &str, - repo: &str, - max: u32, - use_gh: bool, -) -> Result, String> { - let prs: Vec = fetch_all_pages(owner, repo, "pulls", "state=all", max, use_gh).await?; - - let items: Vec = prs - .into_iter() - .map(|p| { - let ts = p.updated_at.as_deref().and_then(parse_iso_ts); - let item_id = format!("pr:{}", p.number); - let cache_key = format!("{owner}/{repo}:{item_id}"); - let item = SourceItem { - id: item_id, - title: format!("PR #{} {}", p.number, p.title), - updated_at_ms: ts, - }; - if let Ok(mut cache) = super::types::LIST_CACHE.lock() { - cache.insert(cache_key, CachedItem::Pr(p)); - } - item - }) - .collect(); - - Ok(items) -} - -/// Read one issue, preferring the row cached by the list pass. -pub(super) async fn read_issue( - owner: &str, - repo: &str, - number: u64, - use_gh: bool, -) -> Result { - let cache_key = format!("{owner}/{repo}:issue:{number}"); - let from_cache = super::types::LIST_CACHE - .lock() - .ok() - .and_then(|mut c| c.remove(&cache_key)); - let issue: GhIssue = match from_cache { - Some(CachedItem::Issue(i)) => i, - _ => { - let json_str = - fetch_github(&format!("repos/{owner}/{repo}/issues/{number}"), use_gh).await?; - serde_json::from_str(&json_str).map_err(|e| format!("parse issue: {e}"))? - } - }; - - let author = issue - .user - .as_ref() - .map(|u| u.login.as_str()) - .unwrap_or("unknown"); - let labels: Vec<&str> = issue.labels.iter().map(|l| l.name.as_str()).collect(); - let issue_body = issue.body.as_deref().unwrap_or(""); - - let comments = fetch_issue_comments(owner, repo, number, use_gh).await; - let participants = - unique_handles(std::iter::once(author).chain(comments.iter().map(|c| c.user.as_str()))); - - let mut body = format!( - "# Issue #{number}: {title}\n\n\ - **State:** {state}\n\ - **Author:** @{author}\n\ - **Participants:** {participants}\n\ - **Labels:** {label_str}\n\ - **Created:** {created}\n\ - **Updated:** {updated}\n\n\ - ## Description\n\n\ - {issue_body}", - title = issue.title, - state = issue.state, - label_str = if labels.is_empty() { - "none".to_string() - } else { - labels.join(", ") - }, - created = issue.created_at.as_deref().unwrap_or("unknown"), - updated = issue.updated_at.as_deref().unwrap_or("unknown"), - ); - - if !comments.is_empty() { - body.push_str("\n\n## Comments\n"); - for comment in &comments { - body.push_str(&format!( - "\n### @{} ({})\n\n{}\n", - comment.user, comment.created_at, comment.body - )); - } - } - - Ok(SourceContent { - id: format!("issue:{number}"), - title: format!("#{number} {}", issue.title), - body, - content_type: ContentType::Markdown, - metadata: serde_json::json!({ - "owner": owner, - "repo": repo, - "number": number, - "state": issue.state, - "labels": labels, - }), - }) -} - -/// Read one pull request, preferring the row cached by the list pass. -pub(super) async fn read_pr( - owner: &str, - repo: &str, - number: u64, - use_gh: bool, -) -> Result { - let cache_key = format!("{owner}/{repo}:pr:{number}"); - let from_cache = super::types::LIST_CACHE - .lock() - .ok() - .and_then(|mut c| c.remove(&cache_key)); - let pr: GhPr = match from_cache { - Some(CachedItem::Pr(p)) => p, - _ => { - let json_str = - fetch_github(&format!("repos/{owner}/{repo}/pulls/{number}"), use_gh).await?; - serde_json::from_str(&json_str).map_err(|e| format!("parse PR: {e}"))? - } - }; - - let author = pr - .user - .as_ref() - .map(|u| u.login.as_str()) - .unwrap_or("unknown"); - let labels: Vec<&str> = pr.labels.iter().map(|l| l.name.as_str()).collect(); - let pr_body = pr.body.as_deref().unwrap_or(""); - - let merged_str = match pr.merged_at.as_deref() { - Some(ts) => format!("merged at {ts}"), - None => "not merged".to_string(), - }; - - let comments = fetch_issue_comments(owner, repo, number, use_gh).await; - let participants = - unique_handles(std::iter::once(author).chain(comments.iter().map(|c| c.user.as_str()))); - - let mut body = format!( - "# PR #{number}: {title}\n\n\ - **State:** {state} ({merged})\n\ - **Author:** @{author}\n\ - **Participants:** {participants}\n\ - **Labels:** {label_str}\n\ - **Created:** {created}\n\ - **Updated:** {updated}\n\n\ - ## Description\n\n\ - {pr_body}", - title = pr.title, - state = pr.state, - merged = merged_str, - label_str = if labels.is_empty() { - "none".to_string() - } else { - labels.join(", ") - }, - created = pr.created_at.as_deref().unwrap_or("unknown"), - updated = pr.updated_at.as_deref().unwrap_or("unknown"), - ); - - if !comments.is_empty() { - body.push_str("\n\n## Comments\n"); - for comment in &comments { - body.push_str(&format!( - "\n### @{} ({})\n\n{}\n", - comment.user, comment.created_at, comment.body - )); - } - } - - Ok(SourceContent { - id: format!("pr:{number}"), - title: format!("PR #{number} {}", pr.title), - body, - content_type: ContentType::Markdown, - metadata: serde_json::json!({ - "owner": owner, - "repo": repo, - "number": number, - "state": pr.state, - "merged": pr.merged_at.is_some(), - "labels": labels, - }), - }) -} - -/// Fetch up to 50 comments on an issue/PR. Best-effort: any failure (or -/// parse error) yields an empty list — comment text is enrichment, not the -/// item's substance, so a missing comments API must not fail the read. -async fn fetch_issue_comments( - owner: &str, - repo: &str, - number: u64, - use_gh: bool, -) -> Vec { - #[derive(Deserialize)] - struct RawComment { - user: Option, - body: Option, - created_at: Option, - } - - let json_str = fetch_github( - &format!("repos/{owner}/{repo}/issues/{number}/comments?per_page=50"), - use_gh, - ) - .await; - - let Ok(json_str) = json_str else { - return Vec::new(); - }; - - let comments: Vec = serde_json::from_str(&json_str).unwrap_or_default(); - - comments - .into_iter() - .map(|c| IssueComment { - user: c - .user - .as_ref() - .map(|u| u.login.clone()) - .unwrap_or_else(|| "unknown".into()), - body: c.body.unwrap_or_default(), - created_at: c.created_at.unwrap_or_else(|| "unknown".into()), - }) - .collect() -} - -#[cfg(test)] -#[path = "mod_tests.rs"] -mod tests; diff --git a/crates/tinymemory-integrations/src/sources/readers/github/issues/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/github/issues/mod_tests.rs deleted file mode 100644 index 5a521680..00000000 --- a/crates/tinymemory-integrations/src/sources/readers/github/issues/mod_tests.rs +++ /dev/null @@ -1,151 +0,0 @@ -//! Offline behavioral tests for issue and pull-request list/read orchestration. - -use super::*; -use crate::sources::readers::github::api::with_test_responses; -use crate::sources::readers::github::types::LIST_CACHE; - -fn issue_json(number: u64) -> serde_json::Value { - serde_json::json!({ - "number": number, - "title": "Broken widget", - "body": "Steps to reproduce", - "state": "open", - "user": {"login": "alice"}, - "labels": [{"name": "bug"}, {"name": "urgent"}], - "created_at": "2026-01-01T00:00:00Z", - "updated_at": "2026-01-02T03:04:05Z", - "pull_request": null - }) -} - -fn pr_json(number: u64) -> serde_json::Value { - serde_json::json!({ - "number": number, - "title": "Fix widget", - "body": "Implements the fix", - "state": "closed", - "user": {"login": "bob"}, - "labels": [{"name": "ready"}], - "created_at": "2026-01-03T00:00:00Z", - "updated_at": "2026-01-04T00:00:00Z", - "merged_at": "2026-01-05T00:00:00Z" - }) -} - -#[tokio::test] -async fn lists_cache_and_render_issues_and_pull_requests_without_network() { - LIST_CACHE.lock().expect("list cache").clear(); - let disguised_pr = serde_json::json!({ - "number": 99, - "title": "PR returned by issues endpoint", - "body": null, - "state": "open", - "user": null, - "labels": [], - "created_at": null, - "updated_at": null, - "pull_request": {"url":"https://example.invalid/pr/99"} - }); - let listed_issues = with_test_responses( - vec![Ok( - serde_json::json!([issue_json(7), disguised_pr]).to_string() - )], - list_issues("acme", "widget", 10, false), - ) - .await - .expect("list issues"); - assert_eq!(listed_issues.len(), 1); - assert_eq!(listed_issues[0].id, "issue:7"); - assert_eq!(listed_issues[0].title, "#7 Broken widget"); - assert_eq!(listed_issues[0].updated_at_ms, Some(1_767_323_045_000)); - - let issue = with_test_responses( - vec![Ok(serde_json::json!([ - { - "user":{"login":"carol"}, - "body":"Confirmed", - "created_at":"2026-01-02T04:00:00Z" - }, - {"user":null,"body":null,"created_at":null} - ]) - .to_string())], - read_issue("acme", "widget", 7, false), - ) - .await - .expect("read cached issue"); - assert_eq!(issue.id, "issue:7"); - assert_eq!(issue.title, "#7 Broken widget"); - assert!(issue.body.contains("**Participants:** @alice @carol")); - assert!(issue.body.contains("**Labels:** bug, urgent")); - assert!(issue.body.contains("### @carol (2026-01-02T04:00:00Z)")); - assert!(issue.body.contains("### @unknown (unknown)")); - assert_eq!(issue.metadata["state"], "open"); - - let listed_prs = with_test_responses( - vec![Ok(serde_json::json!([pr_json(8)]).to_string())], - list_prs("acme", "widget", 10, true), - ) - .await - .expect("list pull requests"); - assert_eq!(listed_prs[0].id, "pr:8"); - assert_eq!(listed_prs[0].title, "PR #8 Fix widget"); - - let pr = with_test_responses( - vec![Ok("not valid comments JSON".into())], - read_pr("acme", "widget", 8, true), - ) - .await - .expect("read cached pull request despite malformed comments"); - assert!( - pr.body - .contains("**State:** closed (merged at 2026-01-05T00:00:00Z)") - ); - assert!(pr.body.contains("**Participants:** @bob")); - assert!(!pr.body.contains("## Comments")); - assert_eq!(pr.metadata["merged"], true); - assert!(LIST_CACHE.lock().expect("list cache").is_empty()); - - assert_uncached_reads_and_failures().await; -} - -async fn assert_uncached_reads_and_failures() { - LIST_CACHE.lock().expect("list cache").clear(); - let issue = with_test_responses( - vec![ - Ok(issue_json(11).to_string()), - Err("comments unavailable".into()), - ], - read_issue("acme", "widget", 11, false), - ) - .await - .expect("uncached issue read"); - assert_eq!(issue.id, "issue:11"); - assert!(!issue.body.contains("## Comments")); - - let transport_error = with_test_responses( - vec![Err("offline".into())], - list_issues("acme", "widget", 10, false), - ) - .await - .expect_err("transport failure must propagate"); - assert_eq!(transport_error, "offline"); - - let parse_error = with_test_responses( - vec![Ok("{}".into())], - list_issues("acme", "widget", 10, false), - ) - .await - .expect_err("malformed list must fail"); - assert!(parse_error.contains("parse issues page 1")); - - let read_error = - with_test_responses(vec![Ok("[]".into())], read_pr("acme", "widget", 12, false)) - .await - .expect_err("malformed pull request must fail"); - assert!(read_error.contains("parse PR")); - - let exhausted = with_test_responses(Vec::new(), list_prs("acme", "widget", 1, false)) - .await - .expect_err("empty fixture must not reach network"); - assert!(exhausted.contains("no deterministic GitHub response queued")); -} diff --git a/crates/tinymemory-integrations/src/sources/readers/github/mod.rs b/crates/tinymemory-integrations/src/sources/readers/github/mod.rs deleted file mode 100644 index 4ace4648..00000000 --- a/crates/tinymemory-integrations/src/sources/readers/github/mod.rs +++ /dev/null @@ -1,308 +0,0 @@ -//! GitHub repo source reader. -//! -//! Pulls **project activity** (commits, issues, PRs) from a GitHub -//! repository — not source code. Commits are read from a local bare clone -//! under `/git_cache/` (`git` must be on `PATH`), falling back to -//! the API when the clone fails. Issues and pull requests, and that fallback, -//! go through the `gh` CLI when it is available (authenticated, higher rate -//! limit) and otherwise the public, unauthenticated GitHub REST API. -//! `gh_available` is probed once per process. -//! -//! ## Module layout -//! -//! - [`self`] — [`GithubReader`] orchestration: item listing/reading, URL -//! parsing, shared utilities, and the cached -//! `gh`-availability probe. -//! - `types` — API response models and the `gh`-fallback list cache. -//! - `git` — local bare-clone + `git log` / `git show` helpers. -//! - `api` — `gh api` / REST transport plus commit list/read helpers. -//! - `issues` — issue and pull-request list/read helpers. - -mod api; -mod git; -mod issues; -mod types; - -#[cfg(test)] -#[path = "mod_tests.rs"] -mod tests; - -use std::time::Duration; - -use async_trait::async_trait; - -use crate::sources::error::{Error, Result}; -use crate::sources::types::{MemorySourceEntry, SourceContent, SourceItem, SourceKind}; - -use super::SourceReader; - -// Re-export for the sibling submodules and the test module. -pub(crate) use types::{ItemKind, LIST_CACHE}; - -/// Default number of items of **each** type (commits, issues, PRs) to pull -/// when the source entry doesn't override it. Tunable per-source via -/// `max_commits` / `max_issues` / `max_prs` on [`MemorySourceEntry`]. -pub(crate) const DEFAULT_GITHUB_ITEM_LIMIT: u32 = 1000; - -/// Timeout for a single `gh` CLI invocation (including the availability -/// probe). -const GH_CLI_TIMEOUT: Duration = Duration::from_secs(30); - -/// Whether the `gh` CLI is on PATH and runs. Probed once per process and -/// cached: `gh api` is the preferred transport for authenticated, -/// higher-rate-limit access, and re-probing on every item read is wasteful. -static GH_AVAILABLE: tokio::sync::OnceCell = tokio::sync::OnceCell::const_new(); - -/// Probe `gh --version` (async, so a stuck `gh` cannot block a worker -/// thread) and cache the result for the process lifetime. -async fn gh_available() -> bool { - *GH_AVAILABLE - .get_or_init(|| async { - let status = tokio::time::timeout( - GH_CLI_TIMEOUT, - tokio::process::Command::new("gh") - .arg("--version") - .stdout(std::process::Stdio::null()) - .stderr(std::process::Stdio::null()) - .status(), - ) - .await; - status - .map(|s| s.map(|st| st.success()).unwrap_or(false)) - .unwrap_or(false) - }) - .await -} - -/// Reader for a GitHub repository source: lists and fetches commits, issues -/// and pull requests. Item ids are `commit:`, `issue:` and `pr:`. -/// Commits come from a local bare clone with an API fallback; issues and pull -/// requests come from `gh api` or the REST API. -#[derive(Debug, Clone, Copy, Default)] -pub struct GithubReader; - -/// Parse `owner` and `repo` from a GitHub URL. -/// -/// Accepts only the canonical `https://github.com//[.git][/]` -/// shape — extra segments like `/tree/main` or `/blob/...` are rejected -/// so callers can't accidentally derive the wrong owner/repo from a -/// deep link. -pub(crate) fn parse_github_url(url: &str) -> std::result::Result<(String, String), String> { - let trimmed = url.trim(); - let rest = trimmed - .strip_prefix("https://github.com/") - .or_else(|| trimmed.strip_prefix("http://github.com/")) - .or_else(|| trimmed.strip_prefix("git@github.com:")) - .ok_or_else(|| format!("not a GitHub URL: {url}"))?; - let cleaned = rest.trim_end_matches('/').trim_end_matches(".git"); - let parts: Vec<&str> = cleaned.split('/').collect(); - if parts.len() != 2 || parts[0].is_empty() || parts[1].is_empty() { - return Err(format!( - "expected https://github.com//, got: {url}" - )); - } - Ok((parts[0].to_string(), parts[1].to_string())) -} - -// ── Reader implementation ─────────────────────────────────────────── - -#[async_trait] -impl SourceReader for GithubReader { - fn kind(&self) -> SourceKind { - SourceKind::GithubRepo - } - - async fn list_items( - &self, - source: &MemorySourceEntry, - workspace: &std::path::Path, - ) -> Result> { - self.list_items_inner(source, workspace) - .await - .map_err(Error::Reader) - } - - async fn read_item( - &self, - source: &MemorySourceEntry, - item_id: &str, - workspace: &std::path::Path, - ) -> Result { - self.read_item_inner(source, item_id, workspace) - .await - .map_err(Error::Reader) - } -} - -impl GithubReader { - async fn list_items_inner( - &self, - source: &MemorySourceEntry, - workspace: &std::path::Path, - ) -> std::result::Result, String> { - let url = source - .url - .as_deref() - .ok_or("github source requires a url")?; - let (owner, repo) = parse_github_url(url)?; - let use_gh = gh_available().await; - - let max_commits = source.max_commits.unwrap_or(DEFAULT_GITHUB_ITEM_LIMIT); - let max_issues = source.max_issues.unwrap_or(DEFAULT_GITHUB_ITEM_LIMIT); - let max_prs = source.max_prs.unwrap_or(DEFAULT_GITHUB_ITEM_LIMIT); - // A configured branch narrows commits to that ref; configured paths - // narrow them to the touched files. Both fall through to the API - // fallback so the two transports agree on scope. - let branch = source.branch.as_deref(); - let paths = source.paths.as_slice(); - - let cache_dir = git::git_cache_dir(workspace, &owner, &repo); - - tracing::debug!( - owner = %owner, - repo = %repo, - use_gh = use_gh, - branch = %branch.unwrap_or("(all)"), - max_commits, - max_issues, - max_prs, - cache = %cache_dir.display(), - "[memory_sources:github] listing items" - ); - - // Clear the list cache so stale data from a prior sync doesn't - // leak into this run. - if let Ok(mut cache) = LIST_CACHE.lock() { - cache.clear(); - } - - let mut items = Vec::new(); - let mut errors = Vec::new(); - - // Commits via local git (clone/fetch bare repo, then git log) - match git::list_commits_git(&owner, &repo, max_commits, &cache_dir, branch, paths).await { - Ok(commits) => items.extend(commits), - Err(e) => { - tracing::warn!(error = %e, "[memory_sources:github] git commit list failed, falling back to API"); - match api::list_commits_api(&owner, &repo, max_commits, use_gh, branch, paths).await - { - Ok(commits) => items.extend(commits), - Err(e2) => { - tracing::warn!(error = %e2, "[memory_sources:github] API commit list also failed"); - errors.push(e2); - } - } - } - } - - // Issues and PRs via gh CLI / API (no local equivalent) - match issues::list_issues(&owner, &repo, max_issues, use_gh).await { - Ok(issues) => items.extend(issues), - Err(e) => { - tracing::warn!(error = %e, "[memory_sources:github] failed to list issues"); - errors.push(e); - } - } - - match issues::list_prs(&owner, &repo, max_prs, use_gh).await { - Ok(prs) => items.extend(prs), - Err(e) => { - tracing::warn!(error = %e, "[memory_sources:github] failed to list PRs"); - errors.push(e); - } - } - - if items.is_empty() && !errors.is_empty() { - return Err(format!( - "all GitHub API calls failed: {}", - errors.join("; ") - )); - } - - tracing::debug!(count = items.len(), "[memory_sources:github] found items"); - Ok(items) - } - - async fn read_item_inner( - &self, - source: &MemorySourceEntry, - item_id: &str, - workspace: &std::path::Path, - ) -> std::result::Result { - let url = source - .url - .as_deref() - .ok_or("github source requires a url")?; - let (owner, repo) = parse_github_url(url)?; - let use_gh = gh_available().await; - - let (kind, ref_id) = - ItemKind::from_id(item_id).ok_or_else(|| format!("invalid item id: {item_id}"))?; - - tracing::debug!( - item_id = %item_id, - kind = ?kind, - "[memory_sources:github] reading item" - ); - - match kind { - ItemKind::Commit => { - let cache_dir = git::git_cache_dir(workspace, &owner, &repo); - match git::read_commit_git(&owner, &repo, ref_id, &cache_dir).await { - Ok(content) => Ok(content), - Err(e) => { - tracing::debug!( - sha = %ref_id, - error = %e, - "[memory_sources:github] git read_commit failed, falling back to API" - ); - api::read_commit_api(&owner, &repo, ref_id, use_gh).await - } - } - } - ItemKind::Issue => { - let num: u64 = ref_id - .parse() - .map_err(|_| format!("invalid issue number: {ref_id}"))?; - issues::read_issue(&owner, &repo, num, use_gh).await - } - ItemKind::PullRequest => { - let num: u64 = ref_id - .parse() - .map_err(|_| format!("invalid PR number: {ref_id}"))?; - issues::read_pr(&owner, &repo, num, use_gh).await - } - } - } -} - -// ── Utilities ─────────────────────────────────────────────────────── - -fn parse_iso_ts(s: &str) -> Option { - chrono::DateTime::parse_from_rfc3339(s) - .ok() - .map(|dt| dt.timestamp_millis()) -} - -/// Render GitHub logins as a deduped, order-preserving, space-separated -/// list of `@handle`s. Empty / `unknown` logins are skipped; an empty -/// result renders as `none`. Used so unique committers/commenters surface -/// as `handle:` entities in the memory tree. -fn unique_handles<'a>(logins: impl Iterator) -> String { - let mut seen = std::collections::HashSet::new(); - let mut out: Vec = Vec::new(); - for login in logins { - let l = login.trim(); - if l.is_empty() || l == "unknown" { - continue; - } - if seen.insert(l.to_string()) { - out.push(format!("@{l}")); - } - } - if out.is_empty() { - "none".to_string() - } else { - out.join(" ") - } -} diff --git a/crates/tinymemory-integrations/src/sources/readers/github/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/github/mod_tests.rs deleted file mode 100644 index 66168524..00000000 --- a/crates/tinymemory-integrations/src/sources/readers/github/mod_tests.rs +++ /dev/null @@ -1,624 +0,0 @@ -//! Tests for the GitHub reader: orchestration over cached clones and the -//! test transport, commit queries and merging, and URL and item-id parsing. - -use super::*; -use crate::sources::readers::SourceReader; - -fn github_source(url: Option<&str>) -> MemorySourceEntry { - MemorySourceEntry { - id: "github".into(), - kind: SourceKind::GithubRepo, - label: "GitHub".into(), - enabled: true, - toolkit: None, - connection_id: None, - path: None, - glob: None, - url: url.map(str::to_string), - branch: None, - paths: Vec::new(), - max_commits: Some(10), - max_issues: Some(0), - max_prs: Some(0), - max_items: None, - selector: None, - max_tokens_per_sync: None, - max_cost_per_sync_usd: None, - sync_depth_days: None, - } -} - -fn local_git(cwd: &std::path::Path, args: &[&str]) -> String { - let output = std::process::Command::new("git") - .env("GIT_CONFIG_GLOBAL", "/dev/null") - .env("GIT_CONFIG_SYSTEM", "/dev/null") - .env("GIT_CONFIG_NOSYSTEM", "1") - .args(["-c", "commit.gpgsign=false"]) - .args(args) - .current_dir(cwd) - .output() - .expect("spawn git"); - assert!( - output.status.success(), - "git {args:?} failed: {}", - String::from_utf8_lossy(&output.stderr) - ); - String::from_utf8_lossy(&output.stdout).into_owned() -} - -#[tokio::test] -async fn reader_lists_and_reads_a_cached_local_repository_without_network() { - let workspace = tempfile::tempdir().expect("workspace"); - let source_repo = workspace.path().join("source"); - std::fs::create_dir_all(&source_repo).expect("source directory"); - local_git(&source_repo, &["init", "-q"]); - local_git(&source_repo, &["config", "user.email", "test@example.com"]); - local_git(&source_repo, &["config", "user.name", "Test"]); - std::fs::write(source_repo.join("README.md"), "hello").expect("write file"); - local_git(&source_repo, &["add", "."]); - local_git(&source_repo, &["commit", "-qm", "local activity"]); - - let cache = git::git_cache_dir(workspace.path(), "local", "fixture"); - std::fs::create_dir_all(cache.parent().expect("cache parent")).expect("cache parent"); - local_git( - workspace.path(), - &[ - "clone", - "--bare", - "-q", - source_repo.to_str().expect("source path"), - cache.to_str().expect("cache path"), - ], - ); - - let source = github_source(Some("https://github.com/local/fixture")); - let reader = GithubReader; - assert_eq!(reader.kind(), SourceKind::GithubRepo); - let items = reader - .list_items(&source, workspace.path()) - .await - .expect("list local cached activity"); - assert_eq!(items.len(), 1); - assert_eq!(items[0].title, "local activity"); - - let content = reader - .read_item(&source, &items[0].id, workspace.path()) - .await - .expect("read cached commit"); - assert_eq!(content.title, "local activity"); - assert!(content.body.contains("Test ")); -} - -#[tokio::test] -async fn reader_rejects_missing_urls_and_malformed_item_ids_before_network() { - let workspace = tempfile::tempdir().expect("workspace"); - let reader = GithubReader; - let missing = github_source(None); - assert!(reader.list_items(&missing, workspace.path()).await.is_err()); - assert!( - reader - .read_item(&missing, "commit:abc", workspace.path()) - .await - .is_err() - ); - - let configured = github_source(Some("https://github.com/local/fixture")); - for item_id in ["unknown", "issue:not-a-number", "pr:not-a-number"] { - assert!( - reader - .read_item(&configured, item_id, workspace.path()) - .await - .is_err() - ); - } -} - -fn issue_json(number: u64) -> serde_json::Value { - serde_json::json!({ - "number": number, - "title": "Reader issue", - "body": "Issue body", - "state": "open", - "user": {"login": "alice"}, - "labels": [{"name": "coverage"}], - "created_at": "2026-01-01T00:00:00Z", - "updated_at": "2026-01-02T00:00:00Z", - "pull_request": null - }) -} - -fn pr_json(number: u64) -> serde_json::Value { - serde_json::json!({ - "number": number, - "title": "Reader PR", - "body": "PR body", - "state": "closed", - "user": {"login": "bob"}, - "labels": [], - "created_at": "2026-01-03T00:00:00Z", - "updated_at": "2026-01-04T00:00:00Z", - "merged_at": null - }) -} - -#[tokio::test] -async fn reader_orchestrates_issue_and_pr_cache_lifecycles_without_network() { - let workspace = tempfile::tempdir().expect("workspace"); - let mut source = github_source(Some("https://github.com/local/fixture")); - source.max_commits = Some(0); - source.max_issues = Some(5); - source.max_prs = Some(5); - let reader = GithubReader; - - let items = api::with_test_responses( - vec![ - Ok(serde_json::json!([issue_json(7)]).to_string()), - Ok(serde_json::json!([pr_json(8)]).to_string()), - ], - reader.list_items(&source, workspace.path()), - ) - .await - .expect("list issue and PR through reader"); - assert_eq!( - items - .iter() - .map(|item| item.id.as_str()) - .collect::>(), - ["issue:7", "pr:8"] - ); - - let issue = api::with_test_responses( - vec![Ok("[]".into())], - reader.read_item(&source, "issue:7", workspace.path()), - ) - .await - .expect("read cached issue"); - assert_eq!(issue.title, "#7 Reader issue"); - - let pr = api::with_test_responses( - vec![Ok("[]".into())], - reader.read_item(&source, "pr:8", workspace.path()), - ) - .await - .expect("read cached PR"); - assert_eq!(pr.title, "PR #8 Reader PR"); -} - -#[tokio::test] -async fn reader_reports_when_every_configured_github_family_fails() { - let workspace = tempfile::tempdir().expect("workspace"); - let mut source = github_source(Some("https://github.com/local/fixture")); - source.max_commits = Some(0); - source.max_issues = Some(1); - source.max_prs = Some(1); - - let error = api::with_test_responses( - vec![Err("issues offline".into()), Err("prs offline".into())], - GithubReader.list_items(&source, workspace.path()), - ) - .await - .expect_err("all configured families failed"); - assert!(error.to_string().contains("all GitHub API calls failed")); - assert!(error.to_string().contains("issues offline")); - assert!(error.to_string().contains("prs offline")); -} - -fn commit_json(sha: &str, message: &str, login: Option<&str>) -> String { - let committed_at = if sha == "new" { - "2026-02-02T00:00:00Z" - } else { - "2026-01-02T00:00:00Z" - }; - let author = login - .map(|value| serde_json::json!({ "login": value })) - .unwrap_or(serde_json::Value::Null); - serde_json::json!({ - "sha": sha, - "commit": { - "message": message, - "author": { - "name": "Test Author", - "email": "author@example.com", - "date": "2026-01-01T00:00:00Z" - }, - "committer": { - "name": "Test Committer", - "email": "committer@example.com", - "date": committed_at - } - }, - "author": author - }) - .to_string() -} - -#[tokio::test] -async fn api_commit_fallback_lists_merges_and_renders_without_network() { - let older = commit_json("old", "older commit\nbody", None); - let newer = commit_json("new", "newer commit\nbody", Some("octocat")); - let listed = api::with_test_responses( - vec![Ok(format!("[{older}]")), Ok(format!("[{newer},{older}]"))], - api::list_commits_api( - "owner", - "repo", - 10, - false, - Some("main"), - &["docs/".into(), "src/".into()], - ), - ) - .await - .expect("list commits from deterministic API pages"); - assert_eq!(listed.len(), 2); - assert_eq!(listed[0].id, "commit:new"); - assert_eq!(listed[1].id, "commit:old"); - - let content = api::with_test_responses( - vec![Ok(commit_json( - "new", - "newer commit\nfull body", - Some("octocat"), - ))], - api::read_commit_api("owner", "repo", "new", false), - ) - .await - .expect("read deterministic commit"); - assert_eq!(content.title, "newer commit"); - assert!( - content - .body - .contains("Test Author (@octocat)") - ); - assert_eq!(content.metadata["author_handle"], "octocat"); -} - -#[tokio::test] -async fn api_commit_fallback_reports_transport_and_parse_failures_without_network() { - let transport = api::with_test_responses( - vec![Err("offline".into())], - api::list_commits_api("owner", "repo", 1, false, None, &[]), - ) - .await - .expect_err("transport error"); - assert_eq!(transport, "offline"); - - let list_parse = api::with_test_responses( - vec![Ok("not json".into())], - api::list_commits_api("owner", "repo", 1, false, None, &[]), - ) - .await - .expect_err("list parse error"); - assert!(list_parse.contains("parse commits page 1")); - - let read_parse = api::with_test_responses( - vec![Ok("{}".into())], - api::read_commit_api("owner", "repo", "bad", false), - ) - .await - .expect_err("commit parse error"); - assert!(read_parse.contains("parse commit")); - - let exhausted = api::with_test_responses( - Vec::new(), - api::read_commit_api("owner", "repo", "missing", false), - ) - .await - .expect_err("fixture exhaustion fails closed"); - assert!(exhausted.contains("no deterministic GitHub response queued")); -} - -#[tokio::test] -async fn reader_falls_back_from_a_broken_local_cache_to_the_api_without_network() { - let workspace = tempfile::tempdir().expect("workspace"); - let cache = git::git_cache_dir(workspace.path(), "owner", "repo"); - std::fs::create_dir_all(&cache).expect("cache directory"); - std::fs::write(cache.join("HEAD"), "not a git repository").expect("broken cache marker"); - - let mut source = github_source(Some("https://github.com/owner/repo")); - source.max_commits = Some(5); - source.max_issues = Some(0); - source.max_prs = Some(0); - let reader = GithubReader; - - let listed = api::with_test_responses( - vec![Ok(format!( - "[{}]", - commit_json("fallback", "API fallback commit", Some("octocat")) - ))], - reader.list_items(&source, workspace.path()), - ) - .await - .expect("broken git cache falls back to API list"); - assert_eq!(listed.len(), 1); - assert_eq!(listed[0].id, "commit:fallback"); - - let content = api::with_test_responses( - vec![Ok(commit_json( - "fallback", - "API fallback commit\nfull body", - Some("octocat"), - ))], - reader.read_item(&source, "commit:fallback", workspace.path()), - ) - .await - .expect("broken git cache falls back to API read"); - assert_eq!(content.id, "commit:fallback"); - assert!(content.body.contains("full body")); - assert_eq!(content.metadata["author_handle"], "octocat"); -} - -#[tokio::test] -async fn reader_keeps_successful_families_when_commit_transports_fail() { - let workspace = tempfile::tempdir().expect("workspace"); - let cache = git::git_cache_dir(workspace.path(), "owner", "repo"); - std::fs::create_dir_all(&cache).expect("cache directory"); - std::fs::write(cache.join("HEAD"), "not a git repository").expect("broken cache marker"); - - let mut source = github_source(Some("https://github.com/owner/repo")); - source.max_commits = Some(1); - source.max_issues = Some(1); - source.max_prs = Some(0); - - let items = api::with_test_responses( - vec![ - Err("commit API offline".into()), - Ok(serde_json::json!([issue_json(7)]).to_string()), - ], - GithubReader.list_items(&source, workspace.path()), - ) - .await - .expect("a successful issue family makes the partial result usable"); - - assert_eq!(items.len(), 1); - assert_eq!(items[0].id, "issue:7"); - assert_eq!(items[0].title, "#7 Reader issue"); -} - -#[test] -fn git_log_args_default_to_head_without_branch() { - // With no branch configured the walk must stay on the bare clone's HEAD - // (the default branch), matching the REST fallback's default-branch scope - // rather than walking every ref. - let args = git::log_args(50, None, &[]); - assert_eq!( - args, - vec![ - "log".to_string(), - "HEAD".to_string(), - "--max-count=50".to_string(), - "--format=%H\t%s\t%aI".to_string(), - ] - ); -} - -#[test] -fn git_log_args_restrict_to_branch_and_paths() { - let args = git::log_args( - 50, - Some("main"), - &["src/lib.rs".to_string(), "docs/".to_string()], - ); - assert_eq!( - args, - vec![ - "log".to_string(), - "main".to_string(), - "--max-count=50".to_string(), - "--format=%H\t%s\t%aI".to_string(), - "--".to_string(), - "src/lib.rs".to_string(), - "docs/".to_string(), - ] - ); - // Empty/whitespace branch falls back to HEAD, never an empty ref. - let args = git::log_args(1, Some(""), &[]); - assert_eq!(args[1], "HEAD"); -} - -#[test] -fn commit_list_queries_carry_branch_and_path_filters() { - // No filters → a single empty query (plain pagination). - assert_eq!(api::commit_list_queries(None, &[]), vec![String::new()]); - // Branch only → `sha=`. - assert_eq!( - api::commit_list_queries(Some("main"), &[]), - vec![String::from("sha=main")] - ); - // One path → `path=

`. - assert_eq!( - api::commit_list_queries(None, &["src/".to_string()]), - vec![String::from("path=src/")] - ); - // Branch + one path → `sha=&path=

`. - assert_eq!( - api::commit_list_queries(Some("main"), &["src/lib.rs".to_string()]), - vec![String::from("sha=main&path=src/lib.rs")] - ); - // Multiple paths → one query per path, dedup happens in the caller. - assert_eq!( - api::commit_list_queries(Some("main"), &["a/".to_string(), "b/".to_string()]), - vec![ - String::from("sha=main&path=a/"), - String::from("sha=main&path=b/") - ] - ); - // Empty branch is treated as unset. - assert_eq!( - api::commit_list_queries(Some(""), &["a/".to_string()]), - vec![String::from("path=a/")] - ); -} - -#[test] -fn commit_list_queries_percent_encode_special_chars() { - // `&`, `#`, `=` and spaces inside a branch or path value would be parsed - // as query syntax and corrupt the filter; they must be percent-encoded. - // `/` is left intact (legal in a query component, and GitHub's commits - // `path` filter expects the common `path=src/` shape unencoded). - assert_eq!( - api::commit_list_queries(Some("feature/one&two"), &["src/#1.rs".to_string()]), - vec![String::from("sha=feature/one%26two&path=src/%231.rs")] - ); - // Unreserved values are unchanged. - assert_eq!( - api::commit_list_queries(Some("main"), &["docs/".to_string()]), - vec![String::from("sha=main&path=docs/")] - ); -} - -#[tokio::test] -async fn fetch_all_pages_keeps_page_size_constant_and_truncates() { - // Regression: the page size must not shrink mid-walk. With `max = 150` - // (not a multiple of 100), a shrinking `per_page` would re-window the - // offsets — page 2 at per_page=50 returns items 51-100 again, skipping - // 101-150. A constant page size walks page 1 and page 2 both at - // per_page=100 and truncates the 200 collected rows to 150. - let mut requested: Vec = Vec::new(); - let pages = api::collect_pages::("commits", 150, |page| { - let url = format!("per_page=100&page={page}"); - requested.push(url); - // 100 rows per page, all full (never a short page before the cap). - let rows: Vec = (1..=100) - .map(|i| format!("{}", (page - 1) * 100 + i)) - .collect(); - async move { Ok(format!("[{}]", rows.join(","))) } - }) - .await - .unwrap(); - - assert_eq!( - requested, - vec![ - "per_page=100&page=1".to_string(), - "per_page=100&page=2".to_string(), - ] - ); - assert_eq!(pages.len(), 150); - // No overlap: the second page is the next window (101..), not 51..100. - assert_eq!(pages[0], 1); - assert_eq!(pages[100], 101); - assert_eq!(pages[149], 150); -} - -#[tokio::test] -async fn fetch_all_pages_stops_at_a_short_page() { - // A short page (fewer than GH_PAGE_SIZE rows) is the last page; the walk - // must not request page 2 after it. - let mut requested: Vec = Vec::new(); - let pages = - crate::sources::readers::github::api::collect_pages::("commits", 1000, |page| { - requested.push(page); - async move { - // Page 1 is short (3 rows) — stop after it even though max is large. - Ok("[1,2,3]".to_string()) - } - }) - .await - .unwrap(); - - assert_eq!(requested, vec![1]); - assert_eq!(pages, vec![1, 2, 3]); -} - -/// Build a synthetic `GhCommit` for merge tests. -fn gh_commit(sha: &str, subject: &str, ts: &str) -> types::GhCommit { - types::GhCommit { - sha: sha.into(), - commit: types::GhCommitInner { - message: subject.into(), - author: None, - committer: Some(types::GhAuthor { - name: None, - email: None, - date: Some(ts.into()), - }), - }, - author: None, - } -} - -#[test] -fn merge_commit_batches_walks_every_path_before_truncating() { - // Two configured paths: the first returns two commits, the second one. - // The pre-fix code stopped after the first path once `out` reached `max`, - // silently dropping the `src` commit even though it is newer than the - // second `docs` commit. - let docs = vec![gh_commit("a", "docs first", "2024-01-01T00:00:00Z")]; - let src = vec![gh_commit("b", "src newer", "2024-02-01T00:00:00Z")]; - - let merged = api::merge_commit_batches(vec![docs, src], 3); - let ids: Vec<&str> = merged.iter().map(|i| i.id.as_str()).collect(); - assert_eq!( - ids, - vec!["commit:b", "commit:a"], - "newest-first, both paths kept" - ); -} - -#[test] -fn merge_commit_batches_dedups_by_sha_and_truncates_globally() { - // A commit touching both paths appears in both batches but only once. - let docs = vec![ - gh_commit("a", "docs first", "2024-01-01T00:00:00Z"), - gh_commit("shared", "touches both", "2024-02-01T00:00:00Z"), - ]; - let src = vec![gh_commit("shared", "touches both", "2024-02-01T00:00:00Z")]; - - let merged = api::merge_commit_batches(vec![docs, src], 1); - let ids: Vec<&str> = merged.iter().map(|i| i.id.as_str()).collect(); - assert_eq!(ids, vec!["commit:shared"], "deduped and truncated to max"); -} - -#[test] -fn parse_github_url_extracts_owner_and_repo() { - let (owner, repo) = parse_github_url("https://github.com/openai/tiktoken").unwrap(); - assert_eq!(owner, "openai"); - assert_eq!(repo, "tiktoken"); -} - -#[test] -fn parse_github_url_handles_trailing_slash_and_git() { - let (owner, repo) = parse_github_url("https://github.com/org/repo.git/").unwrap(); - assert_eq!(owner, "org"); - assert_eq!(repo, "repo"); -} - -#[test] -fn parse_github_url_rejects_non_repo_paths() { - // Deep links like /tree/main must not silently extract the wrong - // owner/repo. Bare host or non-github URLs also rejected. - assert!(parse_github_url("https://github.com/org/repo/tree/main").is_err()); - assert!(parse_github_url("https://gitlab.com/org/repo").is_err()); - assert!(parse_github_url("https://github.com/org").is_err()); - assert!(parse_github_url("not-a-url").is_err()); -} - -#[test] -fn item_kind_round_trips() { - let cases = [ - ("commit:abc123", ItemKind::Commit, "abc123"), - ("issue:42", ItemKind::Issue, "42"), - ("pr:99", ItemKind::PullRequest, "99"), - ]; - for (id, expected_kind, expected_ref) in cases { - let (kind, ref_id) = ItemKind::from_id(id).unwrap(); - assert_eq!(kind, expected_kind); - assert_eq!(ref_id, expected_ref); - } -} - -#[test] -fn item_kind_rejects_invalid() { - assert!(ItemKind::from_id("unknown:123").is_none()); - assert!(ItemKind::from_id("noprefix").is_none()); -} - -#[test] -fn unique_handles_dedups_and_skips_unknown() { - assert_eq!( - unique_handles(["alice", "bob", "alice", "unknown", ""].into_iter()), - "@alice @bob" - ); - assert_eq!(unique_handles(["unknown", ""].into_iter()), "none"); - assert_eq!(unique_handles(std::iter::empty()), "none"); -} diff --git a/crates/tinymemory-integrations/src/sources/readers/github/types.rs b/crates/tinymemory-integrations/src/sources/readers/github/types.rs deleted file mode 100644 index 11327e98..00000000 --- a/crates/tinymemory-integrations/src/sources/readers/github/types.rs +++ /dev/null @@ -1,122 +0,0 @@ -//! API response models and the `gh`-fallback list cache for the GitHub -//! reader. Pure data — no I/O lives here. Models are deliberately kept loose -//! (only the fields the reader consumes are declared) so new GitHub response -//! fields don't force a struct change. - -use std::collections::HashMap; -use std::sync::{LazyLock, Mutex}; - -use serde::Deserialize; - -/// A commit object from the REST commits endpoint. -#[derive(Debug, Deserialize)] -pub(crate) struct GhCommit { - pub(crate) sha: String, - pub(crate) commit: GhCommitInner, - /// Top-level GitHub user that authored the commit (distinct from the - /// embedded git author identity). Present when the commit author maps - /// to a GitHub account; absent for unlinked email-only authors. - #[serde(default)] - pub(crate) author: Option, -} - -#[derive(Debug, Deserialize)] -pub(crate) struct GhCommitInner { - pub(crate) message: String, - pub(crate) author: Option, - pub(crate) committer: Option, -} - -#[derive(Debug, Deserialize)] -pub(crate) struct GhAuthor { - pub(crate) name: Option, - pub(crate) email: Option, - pub(crate) date: Option, -} - -/// An issue list entry (`GET /repos/{owner}/{repo}/issues`). -#[derive(Debug, Clone, Deserialize)] -pub(crate) struct GhIssue { - pub(crate) number: u64, - pub(crate) title: String, - pub(crate) body: Option, - pub(crate) state: String, - pub(crate) user: Option, - pub(crate) labels: Vec, - pub(crate) created_at: Option, - pub(crate) updated_at: Option, - /// Present when the row is actually a pull request (the issues endpoint - /// returns PRs with a `pull_request` envelope). - pub(crate) pull_request: Option, -} - -#[derive(Debug, Clone, Deserialize)] -pub(crate) struct GhUser { - pub(crate) login: String, -} - -#[derive(Debug, Clone, Deserialize)] -pub(crate) struct GhLabel { - pub(crate) name: String, -} - -/// A pull request list entry. -#[derive(Debug, Clone, Deserialize)] -pub(crate) struct GhPr { - pub(crate) number: u64, - pub(crate) title: String, - pub(crate) body: Option, - pub(crate) state: String, - pub(crate) user: Option, - pub(crate) labels: Vec, - pub(crate) created_at: Option, - pub(crate) updated_at: Option, - pub(crate) merged_at: Option, -} - -/// A comment on an issue or PR, slimmed to the fields the reader renders. -#[derive(Debug, Clone)] -pub(crate) struct IssueComment { - pub(crate) user: String, - pub(crate) body: String, - pub(crate) created_at: String, -} - -/// What kind of GitHub item a list row refers to. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub(crate) enum ItemKind { - Commit, - Issue, - PullRequest, -} - -impl ItemKind { - /// Parse a `SourceItem` id (`commit:`, `issue:`, `pr:`) into - /// its kind and the ref (sha / number) that follows the prefix. - pub(crate) fn from_id(id: &str) -> Option<(Self, &str)> { - if let Some(rest) = id.strip_prefix("commit:") { - Some((ItemKind::Commit, rest)) - } else if let Some(rest) = id.strip_prefix("issue:") { - Some((ItemKind::Issue, rest)) - } else if let Some(rest) = id.strip_prefix("pr:") { - Some((ItemKind::PullRequest, rest)) - } else { - None - } - } -} - -/// A cached issue/PR row, keyed by its list id (`"/:"`). -/// The issues endpoint returns pull requests mixed in with issues, and the PR -/// endpoint is the only one that returns merge state, so the list pass stashes -/// the full row here and the read pass reuses it instead of re-fetching. -#[derive(Debug, Clone)] -pub(crate) enum CachedItem { - Issue(GhIssue), - Pr(GhPr), -} - -/// Process-wide cache of issue/PR list rows, cleared at the start of each -/// `list_items` run so stale data from a prior sync can't leak in. -pub(crate) static LIST_CACHE: LazyLock>> = - LazyLock::new(|| Mutex::new(HashMap::new())); diff --git a/crates/tinymemory-integrations/src/sources/readers/rss/mod.rs b/crates/tinymemory-integrations/src/sources/readers/rss/mod.rs deleted file mode 100644 index ec042713..00000000 --- a/crates/tinymemory-integrations/src/sources/readers/rss/mod.rs +++ /dev/null @@ -1,355 +0,0 @@ -//! RSS/Atom feed source reader. -//! -//! Fetches and parses an RSS or Atom feed, returning entries as -//! source items. Uses a lightweight XML parser (`quick-xml` via -//! manual parsing) to avoid pulling in heavy feed crates. -//! -//! Fetches go through `sources::fetch` and its SSRF guard (scheme/host -//! policy, a DNS resolver that pins connections to globally routable -//! addresses, and per-hop redirect re-checks), and the parsed feed is cached briefly so a -//! list-then-read sync pass downloads it once rather than once per entry. - -mod types; - -use std::sync::Mutex; -use std::time::{Duration, Instant}; - -use async_trait::async_trait; - -use crate::sources::error::{Error, Result}; -use crate::sources::types::{ - ContentType, MemorySourceEntry, SourceContent, SourceItem, SourceKind, -}; - -use super::SourceReader; -use crate::documents::html::decode_entities; -use crate::sources::fetch::fetch_url_capped; -use types::{FeedCache, FeedEntry}; - -const DEFAULT_MAX_ITEMS: u32 = 50; -const MAX_FEED_BYTES: u64 = 5 * 1024 * 1024; // 5 MiB — guards against pathological feeds - -/// How long a fetched feed is reused before the next read re-downloads it. -/// -/// Kept short so a feed that updates mid-sync is picked up on the next sync; -/// long enough to cover a list-then-read pass over a 50-entry feed. -const FEED_CACHE_TTL: Duration = Duration::from_secs(60); - -/// Reader for an RSS/Atom feed source. -/// -/// Holds a short-lived cache of the last fetched feed so that a `list_items` -/// immediately followed by per-item `read_item` calls fetches the feed once. -#[derive(Debug)] -pub struct RssReader { - cache: Mutex>, -} - -impl RssReader { - /// A reader with an empty feed cache. - #[must_use] - pub fn new() -> Self { - Self::default() - } - - /// Fetch (or reuse a very fresh copy of) the feed at `url`. - /// - /// The workspace sync pipeline holds one reader across a tick and calls - /// `list_items` once, then `read_item` once per entry. Without a cache - /// that is N+1 downloads of the same feed per sync (and a rate-limit - /// risk against the feed host); the cache turns it into one fetch whose - /// results are reused for the read phase. - async fn fetch_entries(&self, url: &str) -> Result> { - // Read the cache in a nested scope so the mutex guard is dropped before - // the await below — the guard is not `Send`, and holding it across an - // await would make the reader's async methods non-`Send`. - { - let cache = self.cache.lock().unwrap_or_else(|e| e.into_inner()); - if let Some(cached) = cache.as_ref() - && cached.url == url - && cached.fetched_at.elapsed() < FEED_CACHE_TTL - { - return Ok(cached.entries.clone()); - } - } - - // `fetch_url_capped` applies the SSRF guard and streams the body - // against the cap, so a pathological feed cannot exhaust memory. - let document = fetch_url_capped(url, MAX_FEED_BYTES).await?; - let body = String::from_utf8(document.bytes) - .map_err(|e| Error::Reader(format!("feed body is not valid UTF-8: {e}")))?; - let entries = parse_feed_full(&body).map_err(Error::Reader)?; - *self.cache.lock().unwrap_or_else(|e| e.into_inner()) = Some(FeedCache { - url: url.to_string(), - fetched_at: Instant::now(), - entries: entries.clone(), - }); - Ok(entries) - } -} - -impl Default for RssReader { - fn default() -> Self { - Self { - cache: Mutex::new(None), - } - } -} - -#[async_trait] -impl SourceReader for RssReader { - fn kind(&self) -> SourceKind { - SourceKind::RssFeed - } - - async fn list_items( - &self, - source: &MemorySourceEntry, - _workspace: &std::path::Path, - ) -> Result> { - let url = configured_url(source)?; - let max_items = source.max_items.unwrap_or(DEFAULT_MAX_ITEMS) as usize; - - tracing::debug!( - host = %url_host(url), - max_items = max_items, - "[memory_sources:rss] listing items" - ); - - let entries = self.fetch_entries(url).await?; - - tracing::debug!(count = entries.len(), "[memory_sources:rss] parsed entries"); - - Ok(entries - .into_iter() - .take(max_items) - .map(|e| SourceItem { - id: e.id, - title: e.title, - updated_at_ms: e.updated_at_ms, - }) - .collect()) - } - - async fn read_item( - &self, - source: &MemorySourceEntry, - item_id: &str, - _workspace: &std::path::Path, - ) -> Result { - let url = configured_url(source)?; - - tracing::debug!( - host = %url_host(url), - item_id = %item_id, - "[memory_sources:rss] reading item" - ); - - let entries = self.fetch_entries(url).await?; - let entry = entries - .into_iter() - .find(|e| e.id == item_id) - .ok_or_else(|| Error::NotFound(format!("item '{item_id}' not found in feed")))?; - - let content_type = if entry.body.contains('<') { - ContentType::Html - } else { - ContentType::Plaintext - }; - - Ok(SourceContent { - id: entry.id, - title: entry.title, - body: entry.body, - content_type, - metadata: serde_json::json!({ - "link": entry.link, - "published": entry.published, - }), - }) - } -} - -/// The configured feed URL. -fn configured_url(source: &MemorySourceEntry) -> Result<&str> { - source - .url - .as_deref() - .ok_or_else(|| Error::Invalid("rss source requires a url".to_string())) -} - -/// Extract just the host portion of a URL for debug-log redaction so we -/// don't leak query params, paths, or embedded credentials (userinfo). -fn url_host(url: &str) -> String { - // A real parse drops any `user:pass@` prefix via `host_str()`. When the - // value is not a parseable URL (it will be rejected by the SSRF guard - // later anyway), fall back to a textual host extraction that still strips - // userinfo and the path/query/fragment. - reqwest::Url::parse(url) - .ok() - .and_then(|u| u.host_str().map(str::to_string)) - .unwrap_or_else(|| { - let authority = url - .trim_start_matches("https://") - .trim_start_matches("http://") - .split(['/', '?', '#']) - .next() - .unwrap_or(url); - // Only the last `@`-separated segment can be the host; anything - // before it is credentials and must not reach the log. - authority - .rsplit('@') - .next() - .unwrap_or(authority) - .to_string() - }) -} - -fn parse_feed_full(xml: &str) -> std::result::Result, String> { - // Detect RSS vs Atom by looking for std::result::Result, String> { - let mut entries = Vec::new(); - let mut offset = 0; - - while let Some(item_start) = xml[offset..].find("") - .map(|i| abs_start + i + 7) - .unwrap_or(xml.len()); - - let item_xml = &xml[abs_start..item_end]; - let title = extract_tag(item_xml, "title").unwrap_or_default(); - let link = extract_tag(item_xml, "link"); - let guid = extract_tag(item_xml, "guid"); - let description = extract_tag(item_xml, "description") - // An empty `` is a present-but-empty - // tag: `extract_tag` returns `Some("")`, which would short-circuit - // the `content:encoded` fallback below and ingest an empty body. - // Filter it out so a populated `content:encoded` still wins. - .filter(|s| !s.is_empty()) - .or_else(|| extract_cdata(item_xml, "content:encoded")) - .unwrap_or_default(); - let pub_date = extract_tag(item_xml, "pubDate"); - - let id = guid - .or_else(|| link.clone()) - .unwrap_or_else(|| format!("rss-{}", entries.len())); - - entries.push(FeedEntry { - updated_at_ms: pub_date.as_deref().and_then(rss_timestamp_ms), - id, - title, - body: description, - link, - published: pub_date, - }); - - offset = item_end; - } - - Ok(entries) -} - -fn parse_atom(xml: &str) -> std::result::Result, String> { - let mut entries = Vec::new(); - let mut offset = 0; - - while let Some(entry_start) = xml[offset..].find("") - .map(|i| abs_start + i + 8) - .unwrap_or(xml.len()); - - let entry_xml = &xml[abs_start..entry_end]; - let title = extract_tag(entry_xml, "title").unwrap_or_default(); - let id = extract_tag(entry_xml, "id").unwrap_or_else(|| format!("atom-{}", entries.len())); - let content = extract_tag(entry_xml, "content") - // Same shape as the RSS `description`/`content:encoded` pair: an - // empty `` must not block the `summary` - // fallback. - .filter(|s| !s.is_empty()) - .or_else(|| extract_tag(entry_xml, "summary")) - .unwrap_or_default(); - let link = extract_attr(entry_xml, "link", "href"); - let updated = - extract_tag(entry_xml, "updated").or_else(|| extract_tag(entry_xml, "published")); - - entries.push(FeedEntry { - updated_at_ms: updated.as_deref().and_then(rss_timestamp_ms), - id, - title, - body: content, - link, - published: updated, - }); - - offset = entry_end; - } - - Ok(entries) -} - -/// Parse an RSS `pubDate` (RFC 2822) or Atom `updated`/`published` (RFC 3339) -/// timestamp into epoch milliseconds, so workspace sync can skip unchanged -/// entries instead of re-reading every item on every pass. -fn rss_timestamp_ms(value: &str) -> Option { - chrono::DateTime::parse_from_rfc2822(value) - .or_else(|_| chrono::DateTime::parse_from_rfc3339(value)) - .map(|dt| dt.timestamp_millis()) - .ok() -} - -/// Remove a surrounding `` wrapper, if present. -fn unwrap_cdata(s: &str) -> &str { - s.strip_prefix("")) - .unwrap_or(s) -} - -fn extract_tag(xml: &str, tag: &str) -> Option { - let open = format!("<{tag}"); - let close = format!(""); - let start = xml.find(&open)?; - let content_start = xml[start..].find('>')? + start + 1; - let end = xml[content_start..].find(&close)? + content_start; - let content = &xml[content_start..end]; - let trimmed = content.trim(); - let unwrapped = unwrap_cdata(trimmed).trim(); - // CDATA content is literal text, so entity decoding applies only outside - // a CDATA wrapper; decoding `<` inside one would corrupt the content. - if trimmed.starts_with(" Option { - // `extract_tag` already unwraps ``, so it serves both the - // plain-text and CDATA-wrapped shapes. - extract_tag(xml, tag) -} - -fn extract_attr(xml: &str, tag: &str, attr: &str) -> Option { - let open = format!("<{tag} "); - let start = xml.find(&open)?; - let tag_end = xml[start..].find('>')? + start; - let tag_str = &xml[start..tag_end]; - let attr_start = tag_str.find(&format!("{attr}=\""))? + attr.len() + 2; - let attr_end = tag_str[attr_start..].find('"')? + attr_start; - Some(tag_str[attr_start..attr_end].to_string()) -} - -#[cfg(test)] -#[path = "mod_tests.rs"] -mod tests; diff --git a/crates/tinymemory-integrations/src/sources/readers/rss/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/rss/mod_tests.rs deleted file mode 100644 index cf6fc2ca..00000000 --- a/crates/tinymemory-integrations/src/sources/readers/rss/mod_tests.rs +++ /dev/null @@ -1,386 +0,0 @@ -//! Tests for the RSS/Atom reader: the cached list-then-read flow, URL -//! refusal, and the feed parser. - -use super::*; - -use crate::sources::readers::SourceReader; - -fn cached_reader(url: &str) -> RssReader { - RssReader { - cache: Mutex::new(Some(types::FeedCache { - url: url.to_string(), - fetched_at: Instant::now(), - entries: vec![ - types::FeedEntry { - id: "plain".into(), - title: "Plain entry".into(), - body: "plain body".into(), - link: Some("https://example.com/plain".into()), - published: Some("2026-01-01T00:00:00Z".into()), - updated_at_ms: Some(1_767_225_600_000), - }, - types::FeedEntry { - id: "html".into(), - title: "HTML entry".into(), - body: "

html body

".into(), - link: None, - published: None, - updated_at_ms: None, - }, - ], - })), - } -} - -fn rss_source(url: Option<&str>, max_items: Option) -> MemorySourceEntry { - MemorySourceEntry { - id: "feed".into(), - label: "Feed".into(), - kind: SourceKind::RssFeed, - enabled: true, - toolkit: None, - connection_id: None, - path: None, - glob: None, - url: url.map(str::to_string), - branch: None, - paths: Vec::new(), - max_commits: None, - max_issues: None, - max_prs: None, - max_items, - selector: None, - max_tokens_per_sync: None, - max_cost_per_sync_usd: None, - sync_depth_days: None, - } -} - -#[tokio::test] -async fn cached_feed_drives_list_and_read_without_network() { - let url = "https://example.com/feed.xml"; - let reader = cached_reader(url); - assert_eq!(reader.kind(), SourceKind::RssFeed); - - let listed = reader - .list_items(&rss_source(Some(url), Some(1)), std::path::Path::new(".")) - .await - .expect("cached list"); - assert_eq!(listed.len(), 1); - assert_eq!(listed[0].id, "plain"); - assert_eq!(listed[0].updated_at_ms, Some(1_767_225_600_000)); - - let plain = reader - .read_item( - &rss_source(Some(url), None), - "plain", - std::path::Path::new("."), - ) - .await - .expect("cached plaintext item"); - assert_eq!(plain.content_type, ContentType::Plaintext); - assert_eq!(plain.metadata["link"], "https://example.com/plain"); - - let html = reader - .read_item( - &rss_source(Some(url), None), - "html", - std::path::Path::new("."), - ) - .await - .expect("cached HTML item"); - assert_eq!(html.content_type, ContentType::Html); -} - -#[tokio::test] -async fn rss_reader_reports_missing_configuration_and_items() { - let reader = RssReader::new(); - let missing_url = rss_source(None, None); - assert!( - reader - .list_items(&missing_url, std::path::Path::new(".")) - .await - .is_err() - ); - assert!( - reader - .read_item(&missing_url, "anything", std::path::Path::new(".")) - .await - .is_err() - ); - - let url = "https://example.com/feed.xml"; - assert!( - cached_reader(url) - .read_item( - &rss_source(Some(url), None), - "missing", - std::path::Path::new("."), - ) - .await - .is_err() - ); -} - -#[tokio::test] -async fn blocked_and_malformed_feed_urls_fail_closed_without_network() { - for url in ["not a URL", "file:///etc/passwd", "http://127.0.0.1/feed"] { - let error = RssReader::new() - .list_items(&rss_source(Some(url), None), std::path::Path::new(".")) - .await - .expect_err("unsafe URL must be refused"); - assert!(!error.to_string().is_empty()); - } -} - -#[test] -fn parse_rss_extracts_items() { - let xml = r#" - - - Test Feed - - First post - https://example.com/1 - Body of first post - - - Second post - guid-2 - Body of second - - - "#; - - let entries = parse_rss(xml).unwrap(); - assert_eq!(entries.len(), 2); - assert_eq!(entries[0].title, "First post"); - assert_eq!(entries[0].id, "https://example.com/1"); - assert_eq!(entries[1].id, "guid-2"); -} - -#[test] -fn parse_atom_extracts_entries() { - let xml = r#" - - - Atom entry - urn:entry:1 - Content here - - - "#; - - let entries = parse_atom(xml).unwrap(); - assert_eq!(entries.len(), 1); - assert_eq!(entries[0].title, "Atom entry"); - assert_eq!(entries[0].id, "urn:entry:1"); - assert_eq!( - entries[0].link.as_deref(), - Some("https://example.com/atom/1") - ); -} - -#[test] -fn parse_feed_detects_format() { - let rss = "T"; - assert!(parse_feed_full(rss).is_ok()); - - let atom = "T1"; - assert!(parse_feed_full(atom).is_ok()); - - assert!(parse_feed_full("").is_err()); -} - -// ── Entry timestamps ─────────────────────────────────────────────── - -#[test] -fn parse_rss_emits_pubdate_timestamp() { - let xml = r#" - - Post - 1 - Wed, 02 Oct 2002 13:00:00 GMT - - "#; - - let entries = parse_rss(xml).unwrap(); - // 2002-10-02T13:00:00Z in epoch milliseconds. - assert_eq!(entries[0].updated_at_ms, Some(1_033_563_600_000)); -} - -#[test] -fn parse_atom_emits_updated_timestamp() { - let xml = r#" - - Atom entry - urn:entry:1 - 2026-03-01T12:00:00.000Z - - "#; - - let entries = parse_atom(xml).unwrap(); - // 2026-03-01T12:00:00Z in epoch milliseconds. - assert_eq!(entries[0].updated_at_ms, Some(1_772_366_400_000)); -} - -#[test] -fn parse_item_without_timestamp_is_unknown() { - let xml = "T"; - let entries = parse_rss(xml).unwrap(); - assert_eq!(entries[0].updated_at_ms, None); -} - -#[test] -fn rss_timestamp_ms_rejects_garbage() { - assert_eq!(rss_timestamp_ms("not a date"), None); - assert_eq!(rss_timestamp_ms(""), None); -} - -// ── CDATA unwrapping ──────────────────────────────────────────────── - -#[test] -fn extract_tag_unwraps_cdata() { - // The common RSS shape `body

]]>
` - // must yield clean HTML, not the literal CDATA markers. - let xml = "body

]]>
"; - assert_eq!( - extract_tag(xml, "description").as_deref(), - Some("

body

") - ); -} - -#[test] -fn extract_tag_does_not_entity_decode_inside_cdata() { - // CDATA content is literal: `<` must survive intact, not become `<`. - let xml = ""; - assert_eq!( - extract_tag(xml, "description").as_deref(), - Some("Say <tag> literally") - ); -} - -#[test] -fn extract_tag_entity_decodes_outside_cdata() { - let xml = "A & B <b> bold"; - assert_eq!(extract_tag(xml, "title").as_deref(), Some("A & B bold")); -} - -#[test] -fn extract_cdata_reuses_tag_extraction() { - // `content:encoded` is typically CDATA-wrapped; extract_cdata must match - // extract_tag on the same input. - let xml = "full

]]>
"; - assert_eq!( - extract_cdata(xml, "content:encoded").as_deref(), - Some("

full

") - ); -} - -#[test] -fn parse_rss_description_with_cdata_is_clean() { - let xml = r#" - - Post - 1 - Hello world

]]>
-
-
"#; - - let entries = parse_rss(xml).unwrap(); - assert_eq!(entries.len(), 1); - assert_eq!(entries[0].body, "

Hello world

"); - assert!(!entries[0].body.contains("CDATA")); -} - -#[test] -fn parse_rss_empty_description_falls_back_to_encoded_content() { - // A present-but-empty `` must not block the - // `content:encoded` fallback — the item carries its body there instead. - let xml = r#" - - Post - 1 - - Full body from content:encoded

]]>
-
-
"#; - - let entries = parse_rss(xml).unwrap(); - assert_eq!(entries.len(), 1); - assert_eq!(entries[0].body, "

Full body from content:encoded

"); -} - -#[test] -fn parse_atom_empty_content_falls_back_to_summary() { - // Mirrors the RSS `description`/`content:encoded` pair: an empty - // `` must fall through to a populated ``. - let xml = r#" - - Atom entry - urn:entry:1 - - Summary body - - "#; - - let entries = parse_atom(xml).unwrap(); - assert_eq!(entries.len(), 1); - assert_eq!(entries[0].body, "Summary body"); -} - -// ── URL host redaction ────────────────────────────────────────────── - -#[test] -fn url_host_redacts_userinfo() { - // Credentials embedded in a source URL must never reach debug traces — - // only the host is logged. - assert_eq!( - url_host("https://alice:secret@example.com/feed.xml"), - "example.com" - ); -} - -#[test] -fn url_host_drops_path_query_and_fragment() { - assert_eq!( - url_host("https://example.com/feed?token=abc#top"), - "example.com" - ); -} - -#[test] -fn url_host_fallback_strips_userinfo_without_scheme() { - // Scheme-less values are unparseable by reqwest; the textual fallback - // must still drop the `user:pass@` prefix and the path. - assert_eq!(url_host("alice:secret@example.com/feed"), "example.com"); - assert_eq!(url_host("example.com/feed"), "example.com"); -} - -// ── Entity decoding ───────────────────────────────────────────────── - -#[test] -fn feed_text_entities_decode_exactly_once() { - // `&lt;` is the escaped form of `<`; it must decode once to `<`, - // not twice to `<`. - assert_eq!( - extract_tag("&lt;", "title").as_deref(), - Some("<") - ); - assert_eq!( - extract_tag("&amp;", "title").as_deref(), - Some("&") - ); -} - -#[test] -fn feed_text_decodes_every_predefined_xml_entity_and_numeric_references() { - assert_eq!( - extract_tag( - "<b> "q" 'a' & more ’", - "title" - ) - .as_deref(), - Some(" \"q\" 'a' & more \u{2019}") - ); -} diff --git a/crates/tinymemory-integrations/src/sources/readers/rss/types.rs b/crates/tinymemory-integrations/src/sources/readers/rss/types.rs deleted file mode 100644 index db94f9ce..00000000 --- a/crates/tinymemory-integrations/src/sources/readers/rss/types.rs +++ /dev/null @@ -1,25 +0,0 @@ -//! Private feed-shape types for the RSS reader: the parsed entry model and -//! the short-lived cache snapshot reused across a list-then-read sync pass. - -use std::time::Instant; - -/// A fetched feed snapshot cached across a list-then-read sync pass. -#[derive(Debug)] -pub(super) struct FeedCache { - pub url: String, - pub fetched_at: Instant, - pub entries: Vec, -} - -/// One parsed RSS/Atom entry: the dedupe id, the fields surfaced as a source -/// item / content, and the raw link + publication timestamp carried in the -/// content metadata. -#[derive(Debug, Clone)] -pub(super) struct FeedEntry { - pub id: String, - pub title: String, - pub body: String, - pub link: Option, - pub published: Option, - pub updated_at_ms: Option, -} diff --git a/crates/tinymemory-integrations/src/sources/readers/web_page/mod.rs b/crates/tinymemory-integrations/src/sources/readers/web_page/mod.rs deleted file mode 100644 index f7841dac..00000000 --- a/crates/tinymemory-integrations/src/sources/readers/web_page/mod.rs +++ /dev/null @@ -1,444 +0,0 @@ -//! Web page source reader. -//! -//! Fetches a single URL and extracts its content. When a CSS `selector` is -//! configured, only the text of matching elements is included (plain text); -//! otherwise the whole page is converted to markdown through -//! `tinymemory_integrations::documents::html::to_markdown`, keeping its headings, lists and -//! links. -//! -//! The page is fetched through `sources::fetch`, behind its SSRF guard -//! (scheme/host policy plus a DNS resolver that pins connections to globally -//! routable addresses), with a 10 MiB body cap. - -mod types; - -use async_trait::async_trait; - -use types::SelectorSpec; - -use crate::sources::fetch::fetch_url_capped; - -use crate::sources::error::{Error, Result}; -use crate::sources::types::{ - ContentType, MemorySourceEntry, SourceContent, SourceItem, SourceKind, -}; - -use super::SourceReader; - -/// Largest page body the reader will buffer. -const MAX_BODY_BYTES: u64 = 10 * 1024 * 1024; - -/// Reader for a single-page web source: fetches one URL and extracts its -/// readable text. -#[derive(Debug, Clone, Copy, Default)] -pub struct WebPageReader; - -#[async_trait] -impl SourceReader for WebPageReader { - fn kind(&self) -> SourceKind { - SourceKind::WebPage - } - - async fn list_items( - &self, - source: &MemorySourceEntry, - _workspace: &std::path::Path, - ) -> Result> { - let url = configured_url(source)?; - Ok(vec![SourceItem { - id: url.to_string(), - title: source.label.clone(), - updated_at_ms: None, - }]) - } - - async fn read_item( - &self, - source: &MemorySourceEntry, - item_id: &str, - _workspace: &std::path::Path, - ) -> Result { - let url = if item_id.starts_with("http") { - item_id.to_string() - } else { - configured_url(source)?.to_string() - }; - - tracing::debug!( - selector = ?source.selector, - "[memory_sources:web_page] reading item" - ); - - // `fetch_url_capped` applies the SSRF guard (scheme and host policy, - // public-only DNS, per-hop redirect checks) and streams the body - // against the cap, so a hostile or giant page cannot exhaust memory. - let document = fetch_url_capped(&url, MAX_BODY_BYTES).await?; - let body = String::from_utf8_lossy(&document.bytes).into_owned(); - - let title = crate::documents::html::extract_title(&body).unwrap_or_else(|| url.clone()); - let (extracted, content_type) = match source.selector.as_deref() { - Some(selector) => (extract_by_selector(&body, selector), ContentType::Plaintext), - None => ( - crate::documents::html::to_markdown(&body), - ContentType::Markdown, - ), - }; - - Ok(SourceContent { - id: url.clone(), - title, - body: extracted, - content_type, - metadata: serde_json::json!({ "url": url }), - }) - } -} - -/// The configured page URL. -fn configured_url(source: &MemorySourceEntry) -> Result<&str> { - source - .url - .as_deref() - .ok_or_else(|| Error::Invalid("web_page source requires a url".to_string())) -} - -// ── Text extraction ───────────────────────────────────────────────── - -fn parse_selector(selector: &str) -> Option { - let last = selector - .trim() - .rsplit(char::is_whitespace) - .next() - .unwrap_or("") - .trim(); - if last.is_empty() { - return None; - } - - let mut spec = SelectorSpec { - tag: None, - id: None, - classes: Vec::new(), - }; - let mut part = String::new(); - let mut sep = ' '; // leading bare token is the tag - for ch in last.chars() { - match ch { - '.' | '#' => { - push_selector_part(&mut spec, &mut part, sep); - sep = ch; - } - _ => part.push(ch), - } - } - push_selector_part(&mut spec, &mut part, sep); - - if spec.tag.is_none() && spec.id.is_none() && spec.classes.is_empty() { - None - } else { - Some(spec) - } -} - -fn push_selector_part(spec: &mut SelectorSpec, part: &mut String, sep: char) { - let part = std::mem::take(part); - if part.is_empty() { - return; - } - match sep { - '#' => spec.id = Some(part), - '.' => spec.classes.push(part), - _ => { - if spec.tag.is_none() { - spec.tag = Some(part); - } else { - spec.classes.push(part); - } - } - } -} - -/// Extract text from elements matching a simple CSS selector. -/// -/// Falls back to the whole stripped page when the selector never matches -/// (rather than erroring), mirroring the reader's lenient posture for pages -/// whose structure changes between list and read time. -fn extract_by_selector(html: &str, selector: &str) -> String { - let Some(spec) = parse_selector(selector) else { - return strip_html_tags(html); - }; - // Match against a script/style-stripped copy so JS strings and CSS rules - // cannot be mistaken for nested elements, and so a selector that lands on - // a `

World

"; - let result = strip_html_tags(html); - assert_eq!(result, "Hello World"); -} - -#[test] -fn strip_script_and_style_handles_unclosed() { - // Unclosed `"; - let result = extract_by_selector(html, ".content"); - assert!(result.contains("Real")); - assert!(!result.contains("Fake")); -} diff --git a/crates/tinymemory-integrations/src/sources/readers/web_page/types.rs b/crates/tinymemory-integrations/src/sources/readers/web_page/types.rs deleted file mode 100644 index cbcc2767..00000000 --- a/crates/tinymemory-integrations/src/sources/readers/web_page/types.rs +++ /dev/null @@ -1,13 +0,0 @@ -//! Types for the web-page reader. - -/// A parsed simple CSS selector: optional tag name, optional id, and class -/// names. Supports `tag`, `tag.class`, `tag#id`, `.class`, `#id`, and stacked -/// classes (`tag.a.b`, `.a.b`). A descendant/child chain (`div.content p`) -/// targets the final compound selector — a full CSS engine is out of scope for -/// this reader. -#[derive(Debug)] -pub(super) struct SelectorSpec { - pub tag: Option, - pub id: Option, - pub classes: Vec, -} From e9d49a2292622db714253ea279250d0740c857a1 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Thu, 8 Oct 2026 23:26:23 +0530 Subject: [PATCH 02/19] chore(sources): remove unused reader and type re-exports Dropped re-exports from the readers and types modules that were no longer referenced anywhere. This trims the public surface of the sources module without changing behaviour. Auto-committed-on: macbook Co-authored-by: Medulla --- .../src/sources/readers/mod.rs | 70 ++------------ .../src/sources/types/mod.rs | 94 +------------------ 2 files changed, 14 insertions(+), 150 deletions(-) diff --git a/crates/tinymemory-integrations/src/sources/readers/mod.rs b/crates/tinymemory-integrations/src/sources/readers/mod.rs index 503a21ee..f8e1d9d9 100644 --- a/crates/tinymemory-integrations/src/sources/readers/mod.rs +++ b/crates/tinymemory-integrations/src/sources/readers/mod.rs @@ -8,39 +8,16 @@ //! //! ## Ownership boundary //! -//! The local kinds ([`folder::FolderReader`], [`file::FileReader`], -//! [`conversation::ConversationReader`]) are always compiled. The network -//! kinds (`github`, `rss`, `web_page`) sit behind the `sources-network` -//! feature; `rss` and `web_page` fetch through `sources::fetch`. What this module does **not** own is *when* a -//! network read happens: scheduling, polling cadence, OAuth, credentials, -//! and egress/cost budgeting stay with the host. -//! -//! That is why [`reader_for`] and [`is_locally_readable`] draw their line at -//! **local vs. network**, not at implemented vs. absent. A network reader is -//! constructed explicitly (`github::GithubReader`, `rss::RssReader`, -//! `web_page::WebPageReader`) by a caller that has already decided the fetch is -//! allowed; it is never handed out by the kind-dispatch that a sync loop drives -//! on a timer. A `None` from [`reader_for`] therefore means "route this through -//! the host's sync runner", which keeps the host in charge of the network. -//! -//! `composio` is represented by a placeholder reader -//! ([`composio::ComposioReader`]): its data arrives through the credentialed -//! provider pipeline, and [`crate::sources::composio`] turns those payloads into items. -//! -//! A host servicing an *explicit user request* (not a timer) that wants one -//! reader for any kind uses `reader_for_request`. +//! Every reader here is local: [`folder::FolderReader`], [`file::FileReader`] +//! and [`conversation::ConversationReader`] read the workspace and nothing +//! else. Network-backed sources (Composio toolkits, GitHub, RSS, web pages) +//! are not part of this crate. [`is_locally_readable`] and [`reader_for`] +//! therefore cover every [`SourceKind`]. -pub mod composio; pub mod conversation; pub mod file; pub mod folder; -#[cfg(feature = "sources-network")] -pub mod github; pub mod local_file; -#[cfg(feature = "sources-network")] -pub mod rss; -#[cfg(feature = "sources-network")] -pub mod web_page; use std::path::Path; @@ -115,8 +92,8 @@ pub trait SourceReader: Send + Sync + std::fmt::Debug { /// Whether a kind can be read from local state alone, with no network egress. /// -/// Network-backed kinds return `false` even when this build ships their reader -/// (see the module docs): the host decides when a fetch is allowed. +/// Every [`SourceKind`] is local, so this is always `true`; it stays as the +/// single place a host asks the question. #[must_use] pub fn is_locally_readable(kind: &SourceKind) -> bool { matches!( @@ -125,43 +102,14 @@ pub fn is_locally_readable(kind: &SourceKind) -> bool { ) } -/// Get the reader for a source kind that is safe to drive on a timer. +/// Get the reader for a source kind. /// -/// Returns `Some` for [`SourceKind::Folder`], [`SourceKind::File`] and -/// [`SourceKind::Conversation`]. Network-backed kinds (`composio`, -/// `github_repo`, `rss_feed`, `web_page`) return `None` so the caller defers to -/// the host's sync runner, which constructs those readers once it has -/// authorized the fetch. +/// Returns the folder, file or conversation reader. #[must_use] pub fn reader_for(kind: &SourceKind) -> Option> { match kind { SourceKind::Folder => Some(Box::new(folder::FolderReader)), SourceKind::File => Some(Box::new(file::FileReader)), SourceKind::Conversation => Some(Box::new(conversation::ConversationReader)), - SourceKind::Composio - | SourceKind::GithubRepo - | SourceKind::RssFeed - | SourceKind::WebPage => None, - } -} - -/// Get a reader for **any** source kind, for a caller servicing an explicit user -/// request naming one source (an RPC handler), not a timer. -/// -/// Unlike [`reader_for`] this hands out the network readers, so the caller has -/// already decided the fetch is allowed. **Do not reuse it from a polling -/// loop**: the host stays in charge of egress, OAuth and cost budgeting by -/// constructing a network reader deliberately there. -#[cfg(feature = "sources-network")] -#[must_use] -pub fn reader_for_request(kind: &SourceKind) -> Box { - match kind { - SourceKind::Composio => Box::new(composio::ComposioReader), - SourceKind::Conversation => Box::new(conversation::ConversationReader), - SourceKind::Folder => Box::new(folder::FolderReader), - SourceKind::File => Box::new(file::FileReader), - SourceKind::GithubRepo => Box::new(github::GithubReader), - SourceKind::RssFeed => Box::new(rss::RssReader::new()), - SourceKind::WebPage => Box::new(web_page::WebPageReader), } } diff --git a/crates/tinymemory-integrations/src/sources/types/mod.rs b/crates/tinymemory-integrations/src/sources/types/mod.rs index 156f2589..a08ef08c 100644 --- a/crates/tinymemory-integrations/src/sources/types/mod.rs +++ b/crates/tinymemory-integrations/src/sources/types/mod.rs @@ -25,69 +25,43 @@ pub(crate) fn default_true() -> bool { /// The kind of a configured memory source. /// -/// The wire representation is snake_case (`github_repo`, `rss_feed`, …) and is +/// The wire representation is snake_case (`folder`, `file`, `conversation`) and is /// persisted by hosts; it must stay stable across versions. Each maps /// onto one [`tinymemory_api::SourceKind`] through [`SourceKind::api_kind`]. #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] #[serde(rename_all = "snake_case")] pub enum SourceKind { - /// A Composio OAuth connector (Gmail, Slack, Notion, …). Network-backed; - /// the live fetch is owned by the host, not this module. - Composio, /// Local agent conversation transcripts stored in the workspace. Conversation, /// A local folder of files matched by an optional glob. Folder, /// A single local file. File, - /// A GitHub repository's project activity (commits, issues, PRs). - GithubRepo, - /// An RSS/Atom feed. - RssFeed, - /// A single web page, optionally narrowed by a CSS selector. - WebPage, } impl SourceKind { /// Every kind, in declaration order. - pub const ALL: [Self; 7] = [ - Self::Composio, - Self::Conversation, - Self::Folder, - Self::File, - Self::GithubRepo, - Self::RssFeed, - Self::WebPage, - ]; + pub const ALL: [Self; 3] = [Self::Conversation, Self::Folder, Self::File]; /// The stable snake_case wire string for this kind. #[must_use] pub fn as_str(&self) -> &'static str { match self { - SourceKind::Composio => "composio", SourceKind::Conversation => "conversation", SourceKind::Folder => "folder", SourceKind::File => "file", - SourceKind::GithubRepo => "github_repo", - SourceKind::RssFeed => "rss_feed", - SourceKind::WebPage => "web_page", } } /// The contract's [`tinymemory_api::SourceKind`] for items this kind of - /// source produces: a web page is a `Link`, a GitHub repository `Github`, - /// an RSS feed `Rss`; the rest keep their name. + /// source produces: each keeps its name. #[must_use] pub fn api_kind(&self) -> tinymemory_api::SourceKind { use tinymemory_api::SourceKind as Api; match self { - SourceKind::Composio => Api::Composio, SourceKind::Conversation => Api::Conversation, SourceKind::Folder => Api::Folder, SourceKind::File => Api::File, - SourceKind::GithubRepo => Api::Github, - SourceKind::RssFeed => Api::Rss, - SourceKind::WebPage => Api::Link, } } } @@ -109,14 +83,6 @@ pub struct MemorySourceEntry { #[serde(default = "default_true")] pub enabled: bool, - // ── Composio ── - /// Composio toolkit slug (e.g. `gmail`). Required for `composio`. - #[serde(default, skip_serializing_if = "Option::is_none")] - pub toolkit: Option, - /// Composio connection id. Required for `composio`. - #[serde(default, skip_serializing_if = "Option::is_none")] - pub connection_id: Option, - // ── Folder / File ── /// Filesystem path of the folder or file to read. Required for `folder` /// and `file`; a relative path is anchored on the workspace. @@ -127,38 +93,6 @@ pub struct MemorySourceEntry { #[serde(default, skip_serializing_if = "Option::is_none")] pub glob: Option, - // ── GithubRepo / RssFeed / WebPage (shared) ── - /// Source URL. Required for `github_repo`, `rss_feed`, and `web_page`. - #[serde(default, skip_serializing_if = "Option::is_none")] - pub url: Option, - - // ── GithubRepo ── - /// Branch to read (defaults to the repo default when absent). - #[serde(default, skip_serializing_if = "Option::is_none")] - pub branch: Option, - /// Optional path filters within the repo. - #[serde(default, skip_serializing_if = "Vec::is_empty")] - pub paths: Vec, - /// Max commits to pull per sync (default 1000 when absent). - #[serde(default, skip_serializing_if = "Option::is_none")] - pub max_commits: Option, - /// Max issues to pull per sync (default 1000 when absent). - #[serde(default, skip_serializing_if = "Option::is_none")] - pub max_issues: Option, - /// Max pull requests to pull per sync (default 1000 when absent). - #[serde(default, skip_serializing_if = "Option::is_none")] - pub max_prs: Option, - - // ── RssFeed ── - /// Max feed items to pull per sync. - #[serde(default, skip_serializing_if = "Option::is_none")] - pub max_items: Option, - - // ── WebPage ── - /// Optional CSS selector to narrow extracted content. - #[serde(default, skip_serializing_if = "Option::is_none")] - pub selector: Option, - // ── Sync Budget (all source kinds) ── /// Maximum tokens to consume per sync run. Sync stops once this budget is hit. #[serde(default, skip_serializing_if = "Option::is_none")] @@ -180,18 +114,8 @@ impl MemorySourceEntry { kind, label: label.into(), enabled: true, - toolkit: None, - connection_id: None, path: None, glob: None, - url: None, - branch: None, - paths: Vec::new(), - max_commits: None, - max_issues: None, - max_prs: None, - max_items: None, - selector: None, max_tokens_per_sync: None, max_cost_per_sync_usd: None, sync_depth_days: None, @@ -201,9 +125,8 @@ impl MemorySourceEntry { /// Validate the fields this entry's [`SourceKind`] requires. /// /// `id` and `label` are required for every kind, and `id` must not contain - /// `:` or control characters. Composio needs `toolkit` and - /// `connection_id`; folders and files need `path`; GitHub repositories, RSS - /// feeds and web pages need `url`. An empty string counts as missing. + /// `:` or control characters. Folders and files need `path`. An empty string + /// counts as missing. /// /// # Errors /// @@ -221,15 +144,8 @@ impl MemorySourceEntry { return Err(Error::Invalid("label is required".to_string())); } match self.kind { - SourceKind::Composio => { - require_field(&self.toolkit, "toolkit")?; - require_field(&self.connection_id, "connection_id") - } SourceKind::Conversation => Ok(()), SourceKind::Folder | SourceKind::File => require_field(&self.path, "path"), - SourceKind::GithubRepo | SourceKind::RssFeed | SourceKind::WebPage => { - require_field(&self.url, "url") - } } } } From 69630cdacb839cb0e797ae8e4ff79639411d0dcf Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Thu, 8 Oct 2026 23:27:02 +0530 Subject: [PATCH 03/19] feat(tinymemory-integrations): add reader dispatch for source types Introduce a dispatch layer that routes reads to the appropriate source implementation based on the source type, so callers no longer need to match on types themselves. Tests cover the new dispatch path. Auto-committed-on: macbook Co-authored-by: Medulla --- Cargo.lock | 11 -- crates/tinymemory-integrations/Cargo.toml | 14 +-- .../src/sources/mod.rs | 23 +--- .../src/sources/types/mod_tests.rs | 116 ++---------------- .../tests/reader_dispatch.rs | 20 +-- 5 files changed, 21 insertions(+), 163 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 108a6aec..a0750e43 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1538,16 +1538,6 @@ version = "2.0.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba" -[[package]] -name = "signal-hook-registry" -version = "1.4.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c4db69cba1110affc0e9f7bcd48bbf87b3f4fc7c61fc9155afd4c469eb3d6c1b" -dependencies = [ - "errno", - "libc", -] - [[package]] name = "simd-adler32" version = "0.3.10" @@ -1765,7 +1755,6 @@ dependencies = [ "libc", "mio", "pin-project-lite", - "signal-hook-registry", "socket2", "tokio-macros", "windows-sys 0.61.2", diff --git a/crates/tinymemory-integrations/Cargo.toml b/crates/tinymemory-integrations/Cargo.toml index 69c651b7..e58f8be4 100644 --- a/crates/tinymemory-integrations/Cargo.toml +++ b/crates/tinymemory-integrations/Cargo.toml @@ -36,12 +36,9 @@ tinymemory-tools = { path = "../tinymemory-tools", optional = true } # CortexDB speaks HTTP/JSON. `stream` is for `bytes_stream()`: response bodies # are read against a byte cap rather than buffered whole, because the endpoint # is operator-supplied and a broken or hostile one must not exhaust the host. -# The network source readers share the same client stack. reqwest = { version = "0.12", default-features = false, features = ["json", "rustls-tls", "stream"], optional = true } -# Timers for read-retry backoff and visibility polling (cortex); process -# spawning and DNS lookups for the GitHub reader and the SSRF resolver -# (sources-network). reqwest already requires a tokio runtime, so this adds no -# new runtime assumption. +# Timers for read-retry backoff and visibility polling (cortex). reqwest +# already requires a tokio runtime, so this adds no new runtime assumption. tokio = { version = "1", default-features = false, features = ["time"], optional = true } # `StreamExt` to read a capped body chunk by chunk. futures = { version = "0.3", optional = true } @@ -122,18 +119,15 @@ documents-office = ["documents", "dep:pdf-extract", "dep:calamine", "dep:quick-x # placed under the source type their format implies. brain = ["documents", "dep:tinymemory-tools"] # `sources`: readers that turn folders, files and conversations into -# `StoreItem`s, and the Composio payload normalisers. Links no HTTP stack. +# `StoreItem`s. Links no HTTP stack. sources = ["documents", "dep:schemars", "dep:regex", "dep:walkdir", "dep:chrono", "dep:log", "dep:tracing"] -# The readers that fetch over the network — GitHub, RSS, web pages — and -# `sources::fetch::fetch_url`, all behind the shared SSRF guard. -sources-network = ["sources", "dep:reqwest", "dep:futures", "dep:tokio", "tokio/process", "tokio/io-util", "tokio/net"] # `safety`: secret and PII scrubbing for a `StoreItem` before it is stored. safety = ["dep:regex", "dep:serde_json", "dep:log"] # `import`: reads a legacy v1 (embedded TinyCortex) workspace and migrates it # into any engine, resumably. legacy-import = ["dep:rusqlite", "dep:serde", "dep:serde_json", "dep:thiserror"] # Every integration. -full = ["cortex", "documents-office", "brain", "sources-network", "safety", "legacy-import"] +full = ["cortex", "documents-office", "brain", "sources", "safety", "legacy-import"] [[example]] name = "basic" diff --git a/crates/tinymemory-integrations/src/sources/mod.rs b/crates/tinymemory-integrations/src/sources/mod.rs index dad654b8..3e3851e7 100644 --- a/crates/tinymemory-integrations/src/sources/mod.rs +++ b/crates/tinymemory-integrations/src/sources/mod.rs @@ -1,23 +1,17 @@ -//! Source readers for TinyMemory: turn a folder, a file, a web page, a GitHub -//! repository, an RSS feed, a Composio toolkit payload or a local conversation +//! Source readers for TinyMemory: turn a folder, a file or a local conversation //! into [`StoreItem`](tinymemory_api::StoreItem)s. //! //! - **Configuration** — what a source *is* ([`MemorySourceEntry`], keyed by //! [`SourceKind`], checked by [`MemorySourceEntry::validate`]). Where the //! host stores its sources, and how it edits them, is the host's business. //! - **Readers** — [`readers::SourceReader`] lists a source's items and reads -//! one. Local readers (folder, file, conversation) are always compiled; the -//! network readers (GitHub, RSS, web page) and `fetch` sit behind the -//! `sources-network` feature; RSS and web pages fetch through `fetch`, -//! behind its one SSRF guard (`fetch::ssrf`). +//! one. Every reader (folder, file, conversation) is local; none touches the +//! network. //! - **Items** — [`items`] maps reader output to `StoreItem`s with //! [`MemoryMeta`](tinymemory_api::MemoryMeta) filled per kind; every text //! body is converted to markdown through [`crate::documents`]. -//! - **Composio** — [`composio`] normalises toolkit payloads (Gmail, Slack, -//! GitHub, Linear, Notion, ClickUp) and maps them to items. //! -//! Scheduling, credentials and egress budgets stay with the host: this module -//! reads when asked. +//! Scheduling stays with the host: this module reads when asked. //! //! # Example //! @@ -57,16 +51,9 @@ //! //! # Feature flags //! -//! - `sources` — everything above except the network pieces. Implies -//! `documents`; links no HTTP stack. -//! - `sources-network` — the GitHub, RSS and web-page readers, `fetch`, -//! `readers::reader_for_request` and the SSRF guard. Without it, a host that -//! only reads local sources links no HTTP stack. +//! - `sources` — everything above. Implies `documents`; links no HTTP stack. -pub mod composio; pub mod error; -#[cfg(feature = "sources-network")] -pub mod fetch; pub mod items; pub mod readers; pub mod types; diff --git a/crates/tinymemory-integrations/src/sources/types/mod_tests.rs b/crates/tinymemory-integrations/src/sources/types/mod_tests.rs index 2ea2171b..b9d6cb55 100644 --- a/crates/tinymemory-integrations/src/sources/types/mod_tests.rs +++ b/crates/tinymemory-integrations/src/sources/types/mod_tests.rs @@ -5,13 +5,9 @@ use super::*; #[test] fn source_kind_round_trips_via_serde() { for kind in [ - SourceKind::Composio, SourceKind::Conversation, SourceKind::Folder, SourceKind::File, - SourceKind::GithubRepo, - SourceKind::RssFeed, - SourceKind::WebPage, ] { let json = serde_json::to_string(&kind).unwrap(); let decoded: SourceKind = serde_json::from_str(&json).unwrap(); @@ -21,33 +17,9 @@ fn source_kind_round_trips_via_serde() { #[test] fn source_kind_as_str_matches_wire_strings() { - assert_eq!(SourceKind::Composio.as_str(), "composio"); assert_eq!(SourceKind::Conversation.as_str(), "conversation"); assert_eq!(SourceKind::Folder.as_str(), "folder"); - assert_eq!(SourceKind::GithubRepo.as_str(), "github_repo"); assert_eq!(SourceKind::File.as_str(), "file"); - assert_eq!(SourceKind::RssFeed.as_str(), "rss_feed"); - assert_eq!(SourceKind::WebPage.as_str(), "web_page"); -} - -#[test] -fn validate_composio_requires_toolkit_and_connection_id() { - let entry = MemorySourceEntry { - id: "src_1".into(), - kind: SourceKind::Composio, - label: "Gmail".into(), - enabled: true, - toolkit: Some("gmail".into()), - connection_id: None, - ..default_entry() - }; - assert!(entry.validate().is_err()); - - let valid = MemorySourceEntry { - connection_id: Some("cmp_123".into()), - ..entry - }; - assert!(valid.validate().is_ok()); } #[test] @@ -63,19 +35,6 @@ fn validate_folder_requires_path() { assert!(entry.validate().is_err()); } -#[test] -fn validate_github_requires_url() { - let entry = MemorySourceEntry { - id: "src_3".into(), - kind: SourceKind::GithubRepo, - label: "Repo".into(), - enabled: true, - url: Some("https://github.com/org/repo".into()), - ..default_entry() - }; - assert!(entry.validate().is_ok()); -} - #[test] fn validate_file_requires_path() { let entry = MemorySourceEntry::new("src_file", SourceKind::File, "One file"); @@ -88,9 +47,11 @@ fn validate_file_requires_path() { } #[test] -fn the_removed_twitter_query_kind_no_longer_decodes() { - let decoded = serde_json::from_str::("\"twitter_query\""); - assert!(decoded.is_err()); +fn the_removed_network_kinds_no_longer_decode() { + for removed in ["twitter_query", "composio", "github_repo", "rss_feed", "web_page"] { + let decoded = serde_json::from_str::(&format!("\"{removed}\"")); + assert!(decoded.is_err(), "{removed} must not decode"); + } } #[test] @@ -99,41 +60,10 @@ fn every_config_kind_maps_onto_a_contract_source_kind() { let mapped: Vec = SourceKind::ALL.iter().map(SourceKind::api_kind).collect(); assert_eq!( mapped, - vec![ - Api::Composio, - Api::Conversation, - Api::Folder, - Api::File, - Api::Github, - Api::Rss, - Api::Link, - ] + vec![Api::Conversation, Api::Folder, Api::File] ); } -#[test] -fn validate_rss_and_web_page_require_url() { - let rss = MemorySourceEntry { - id: "src_rss".into(), - kind: SourceKind::RssFeed, - label: "Feed".into(), - enabled: true, - url: None, - ..default_entry() - }; - assert!(rss.validate().is_err()); - - let web = MemorySourceEntry { - id: "src_web".into(), - kind: SourceKind::WebPage, - label: "Page".into(), - enabled: true, - url: Some("https://example.com".into()), - ..default_entry() - }; - assert!(web.validate().is_ok()); -} - #[test] fn validate_conversation_needs_only_id_and_label() { let entry = MemorySourceEntry { @@ -245,18 +175,8 @@ pub(super) fn default_entry() -> MemorySourceEntry { kind: SourceKind::Folder, label: String::new(), enabled: true, - toolkit: None, - connection_id: None, path: None, glob: None, - url: None, - branch: None, - paths: Vec::new(), - max_commits: None, - max_issues: None, - max_prs: None, - max_items: None, - selector: None, max_tokens_per_sync: None, max_cost_per_sync_usd: None, sync_depth_days: None, @@ -274,21 +194,11 @@ pub(super) fn default_entry() -> MemorySourceEntry { fn source_entry_wire_format_is_pinned() { let entry = MemorySourceEntry { id: "src_pinned".into(), - kind: SourceKind::GithubRepo, + kind: SourceKind::Folder, label: "Pinned".into(), enabled: false, - toolkit: Some("gmail".into()), - connection_id: Some("conn-1".into()), path: Some("/notes".into()), glob: Some("**/*.md".into()), - url: Some("https://github.com/tinyhumansai/tinymemory".into()), - branch: Some("main".into()), - paths: vec!["core/src".into()], - max_commits: Some(10), - max_issues: Some(20), - max_prs: Some(30), - max_items: Some(40), - selector: Some("article".into()), max_tokens_per_sync: Some(50_000), max_cost_per_sync_usd: Some(1.5), sync_depth_days: Some(90), @@ -298,21 +208,11 @@ fn source_entry_wire_format_is_pinned() { serde_json::to_value(&entry).unwrap(), serde_json::json!({ "id": "src_pinned", - "kind": "github_repo", + "kind": "folder", "label": "Pinned", "enabled": false, - "toolkit": "gmail", - "connection_id": "conn-1", "path": "/notes", "glob": "**/*.md", - "url": "https://github.com/tinyhumansai/tinymemory", - "branch": "main", - "paths": ["core/src"], - "max_commits": 10, - "max_issues": 20, - "max_prs": 30, - "max_items": 40, - "selector": "article", "max_tokens_per_sync": 50000, "max_cost_per_sync_usd": 1.5, "sync_depth_days": 90 diff --git a/crates/tinymemory-integrations/tests/reader_dispatch.rs b/crates/tinymemory-integrations/tests/reader_dispatch.rs index 6cb12752..a18800c0 100644 --- a/crates/tinymemory-integrations/tests/reader_dispatch.rs +++ b/crates/tinymemory-integrations/tests/reader_dispatch.rs @@ -6,7 +6,7 @@ use tinymemory_integrations::sources::{ }; #[test] -fn timer_dispatch_constructs_only_readers_that_never_need_network() { +fn dispatch_constructs_the_local_readers() { for kind in [ SourceKind::Folder, SourceKind::File, @@ -15,24 +15,12 @@ fn timer_dispatch_constructs_only_readers_that_never_need_network() { assert!(is_locally_readable(&kind)); assert_eq!(reader_for(&kind).map(|reader| reader.kind()), Some(kind)); } - - for kind in [ - SourceKind::Composio, - SourceKind::GithubRepo, - SourceKind::RssFeed, - SourceKind::WebPage, - ] { - assert!(!is_locally_readable(&kind)); - assert!(reader_for(&kind).is_none()); - } } -#[cfg(feature = "sources-network")] #[test] -fn request_dispatch_hands_out_a_reader_for_every_kind() { - use tinymemory_integrations::sources::readers::reader_for_request; - +fn every_kind_has_a_local_reader() { for kind in SourceKind::ALL { - assert_eq!(reader_for_request(&kind).kind(), kind); + assert!(is_locally_readable(&kind)); + assert!(reader_for(&kind).is_some()); } } From a318b2a440167dabd7edb61976d7940d5ac6bdd0 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Thu, 8 Oct 2026 23:27:15 +0530 Subject: [PATCH 04/19] refactor(items): drop per-source metadata mapping The item builder no longer fills source-specific metadata for GitHub, RSS, web page, and Composio entries, leaving only the shared source reference and the folder, file, and conversation fields. The removed helpers and their doc comments went with it, so the module now covers just the kinds it still handles. Auto-committed-on: macbook Co-authored-by: Medulla --- .../src/sources/items/mod.rs | 61 +------------------ 1 file changed, 2 insertions(+), 59 deletions(-) diff --git a/crates/tinymemory-integrations/src/sources/items/mod.rs b/crates/tinymemory-integrations/src/sources/items/mod.rs index 40205a2e..6c869c1b 100644 --- a/crates/tinymemory-integrations/src/sources/items/mod.rs +++ b/crates/tinymemory-integrations/src/sources/items/mod.rs @@ -8,10 +8,6 @@ //! | Kind | Item | Metadata filled | //! | --- | --- | --- | //! | folder, file | document | `workspace`, `folder` (containing directory), `file_path`, `language` (by extension), `observed_at` (mtime), `mime` | -//! | github | document | `repo` (`owner/name`), `commit` (commit items), `url` (issues and PRs), `observed_at` | -//! | link | document | `url` | -//! | rss | document | `url` (the entry's link), `observed_at` (published) | -//! | composio | document | `tags = [toolkit]` (payloads: see [`crate::sources::composio`]) | //! | conversation | conversation | `workspace`, `thread_id`, `turns`, `observed_at` (last turn) | //! //! Every document body is markdown, converted through `crate::documents`: @@ -38,23 +34,16 @@ use crate::sources::readers::local_file::LocalFile; use crate::sources::types::{ContentType, MemorySourceEntry, SourceContent, SourceKind}; /// Metadata naming `entry` as the source: `source.kind` is the entry's kind -/// mapped onto the contract, `source.id` its id. A Composio entry also gets -/// its toolkit as a tag. +/// mapped onto the contract, `source.id` its id. #[must_use] pub fn base_meta(entry: &MemorySourceEntry) -> MemoryMeta { - let mut meta = MemoryMeta { + MemoryMeta { source: SourceRef { kind: entry.kind.api_kind(), id: Some(entry.id.clone()), }, ..MemoryMeta::default() - }; - if entry.kind == SourceKind::Composio - && let Some(toolkit) = entry.toolkit.as_deref().filter(|t| !t.is_empty()) - { - meta.tags = vec![toolkit.to_string()]; } - meta } /// Convert a local file and wrap it as a document. @@ -171,19 +160,6 @@ pub fn content_item( meta.observed_at = millis(updated_at_ms); let metadata = &content.metadata; match entry.kind { - SourceKind::GithubRepo => github_meta(&mut meta, &content.id, metadata), - SourceKind::RssFeed => { - meta.url = string_field(metadata, "link"); - if let Some(published) = string_field(metadata, "published") - .as_deref() - .and_then(parse_feed_time) - { - meta.observed_at = Some(published); - } - } - SourceKind::WebPage => { - meta.url = string_field(metadata, "url").or_else(|| Some(content.id.clone())); - } SourceKind::Folder => { if let Some(root) = entry.path.as_deref() { let path = Path::new(root).join(&content.id); @@ -197,7 +173,6 @@ pub fn content_item( meta.language = language_for_path(&content.id).map(str::to_string); } SourceKind::Conversation => meta.thread_id = Some(content.id.clone()), - SourceKind::Composio => {} } Ok(StoreItem::Document { @@ -208,30 +183,6 @@ pub fn content_item( }) } -/// GitHub metadata from a reader item id (`commit:`, `issue:`, -/// `pr:`) and its content metadata (`owner`, `repo`, `sha`, `number`). -fn github_meta(meta: &mut MemoryMeta, item_id: &str, metadata: &serde_json::Value) { - let owner = string_field(metadata, "owner"); - let repo = string_field(metadata, "repo"); - let slug = owner - .zip(repo) - .map(|(owner, repo)| format!("{owner}/{repo}")); - meta.repo = slug.clone(); - let number = metadata - .get("number") - .and_then(serde_json::Value::as_u64) - .map(|n| n.to_string()); - if let Some(sha) = item_id.strip_prefix("commit:") { - meta.commit = string_field(metadata, "sha").or_else(|| Some(sha.to_string())); - } else if let Some(n) = item_id.strip_prefix("issue:") { - let n = number.unwrap_or_else(|| n.to_string()); - meta.url = slug.map(|slug| format!("https://github.com/{slug}/issues/{n}")); - } else if let Some(n) = item_id.strip_prefix("pr:") { - let n = number.unwrap_or_else(|| n.to_string()); - meta.url = slug.map(|slug| format!("https://github.com/{slug}/pull/{n}")); - } -} - /// A non-empty string field of a JSON object. fn string_field(value: &serde_json::Value, key: &str) -> Option { value @@ -247,14 +198,6 @@ fn millis(value: Option) -> Option> { value.and_then(|ms| Utc.timestamp_millis_opt(ms).single()) } -/// An RSS `pubDate` (RFC 2822) or Atom `updated` (RFC 3339) timestamp. -fn parse_feed_time(text: &str) -> Option> { - DateTime::parse_from_rfc2822(text) - .or_else(|_| DateTime::parse_from_rfc3339(text)) - .ok() - .map(|at| at.with_timezone(&Utc)) -} - /// One item a [`collect_items`] pass could not turn into a `StoreItem`. #[derive(Debug)] pub struct SkippedItem { From 095f10d919556e389415f4bc3e44c5a642b31ba2 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Thu, 8 Oct 2026 23:27:40 +0530 Subject: [PATCH 05/19] feat(meta): expose source layout metadata Added a source layout module to the meta API so callers can inspect how memory sources are laid out. This surfaces layout information that was previously only available internally. Auto-committed-on: macbook Co-authored-by: Medulla --- crates/tinymemory-api/src/meta/mod.rs | 25 ++++++-------------- crates/tinymemory-tools/src/layout/source.rs | 5 +--- 2 files changed, 8 insertions(+), 22 deletions(-) diff --git a/crates/tinymemory-api/src/meta/mod.rs b/crates/tinymemory-api/src/meta/mod.rs index c7736efb..4c77e05a 100644 --- a/crates/tinymemory-api/src/meta/mod.rs +++ b/crates/tinymemory-api/src/meta/mod.rs @@ -150,32 +150,25 @@ pub enum SourceKind { Folder, /// A single file. File, - /// A web page. - Link, - /// A GitHub repository. - Github, - /// An RSS or Atom feed. - Rss, - /// A Composio toolkit payload (Gmail, Slack, Notion, ...). - Composio, /// A host conversation. Conversation, /// Written by an agent directly; the default. #[default] Agent, - /// Imported from a legacy store. + /// Imported from a legacy store, or ingested by hand from a connected app + /// or the web. + /// + /// The retired network kinds (`link`, `github`, `rss`, `composio`) decode + /// to this variant, so items stored under them stay readable. + #[serde(alias = "link", alias = "github", alias = "rss", alias = "composio")] Import, } impl SourceKind { /// Every kind, in declaration order. - pub const ALL: [Self; 9] = [ + pub const ALL: [Self; 5] = [ Self::Folder, Self::File, - Self::Link, - Self::Github, - Self::Rss, - Self::Composio, Self::Conversation, Self::Agent, Self::Import, @@ -187,10 +180,6 @@ impl SourceKind { match self { Self::Folder => "folder", Self::File => "file", - Self::Link => "link", - Self::Github => "github", - Self::Rss => "rss", - Self::Composio => "composio", Self::Conversation => "conversation", Self::Agent => "agent", Self::Import => "import", diff --git a/crates/tinymemory-tools/src/layout/source.rs b/crates/tinymemory-tools/src/layout/source.rs index dcf203d2..ea2a3d25 100644 --- a/crates/tinymemory-tools/src/layout/source.rs +++ b/crates/tinymemory-tools/src/layout/source.rs @@ -60,10 +60,7 @@ impl BrainSource { pub fn source_kind(&self) -> SourceKind { match self { Self::Files | Self::Pdf | Self::Markdown => SourceKind::File, - Self::Notion => SourceKind::Composio, - Self::Github => SourceKind::Github, - Self::Web => SourceKind::Link, - Self::Other(_) => SourceKind::Import, + Self::Notion | Self::Github | Self::Web | Self::Other(_) => SourceKind::Import, } } } From 0a2f881bbed81bcfe16a62227121f2ceff878ca4 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Thu, 8 Oct 2026 23:28:10 +0530 Subject: [PATCH 06/19] test: add unit tests for conformance fixtures, meta filters, and cortex engine Adds test coverage across the conformance suite fixtures, meta filter and module handling, cortex engine fetch and refers paths, envelope labels, and document items. These tests exercise existing behaviour to guard against regressions. Auto-committed-on: macbook Co-authored-by: Medulla --- .../src/conformance/suite/fixtures.rs | 4 ++-- crates/tinymemory-api/src/meta/filter_tests.rs | 2 +- crates/tinymemory-api/src/meta/mod_tests.rs | 17 +++++++++++++++-- .../src/cortex/engine/fetch_tests.rs | 2 +- .../src/cortex/engine/refers_tests.rs | 2 +- .../src/cortex/envelope/labels_tests.rs | 6 +++--- .../src/documents/item/mod_tests.rs | 4 ++-- 7 files changed, 25 insertions(+), 12 deletions(-) diff --git a/crates/tinymemory-api/src/conformance/suite/fixtures.rs b/crates/tinymemory-api/src/conformance/suite/fixtures.rs index 0a7bf304..7392cd59 100644 --- a/crates/tinymemory-api/src/conformance/suite/fixtures.rs +++ b/crates/tinymemory-api/src/conformance/suite/fixtures.rs @@ -272,13 +272,13 @@ impl Run { "feed", with(&|meta| { meta.source = SourceRef { - kind: SourceKind::Rss, + kind: SourceKind::File, id: None, } }), ), MetaFilter { - sources: vec![SourceKind::Rss], + sources: vec![SourceKind::File], ..base.clone() }, ), diff --git a/crates/tinymemory-api/src/meta/filter_tests.rs b/crates/tinymemory-api/src/meta/filter_tests.rs index a8cc89bb..18a2d4aa 100644 --- a/crates/tinymemory-api/src/meta/filter_tests.rs +++ b/crates/tinymemory-api/src/meta/filter_tests.rs @@ -150,7 +150,7 @@ fn every_exact_field_must_match() { ..Default::default() }, MetaFilter { - sources: vec![SourceKind::Rss], + sources: vec![SourceKind::File], ..Default::default() }, ), diff --git a/crates/tinymemory-api/src/meta/mod_tests.rs b/crates/tinymemory-api/src/meta/mod_tests.rs index def04bdf..349d9b29 100644 --- a/crates/tinymemory-api/src/meta/mod_tests.rs +++ b/crates/tinymemory-api/src/meta/mod_tests.rs @@ -11,7 +11,7 @@ fn a_default_meta_names_the_agent_as_its_source() { #[test] fn unset_fields_are_omitted_and_round_trip() { - let mut meta = MemoryMeta::from_source(SourceKind::Github, Some("src_1".into())); + let mut meta = MemoryMeta::from_source(SourceKind::Folder, Some("src_1".into())); meta.repo = Some("owner/name".into()); meta.tool_call = Some(ToolCallRef { name: "grep".into(), @@ -23,7 +23,7 @@ fn unset_fields_are_omitted_and_round_trip() { serde_json::json!({ "repo": "owner/name", "tool_call": { "name": "grep" }, - "source": { "kind": "github", "id": "src_1" } + "source": { "kind": "folder", "id": "src_1" } }) ); let back: MemoryMeta = serde_json::from_value(json).expect("deserialise"); @@ -43,3 +43,16 @@ fn source_kind_wire_strings_match_serde() { assert_eq!(json, serde_json::json!(kind.as_str())); } } + +#[test] +fn retired_network_source_kinds_decode_as_import() { + for retired in ["link", "github", "rss", "composio"] { + let kind: SourceKind = + serde_json::from_value(serde_json::json!(retired)).expect("legacy kind decodes"); + assert_eq!(kind, SourceKind::Import, "{retired}"); + } + assert_eq!( + serde_json::to_value(SourceKind::Import).expect("serialise"), + serde_json::json!("import") + ); +} diff --git a/crates/tinymemory-integrations/src/cortex/engine/fetch_tests.rs b/crates/tinymemory-integrations/src/cortex/engine/fetch_tests.rs index 14ccdcd1..7433917f 100644 --- a/crates/tinymemory-integrations/src/cortex/engine/fetch_tests.rs +++ b/crates/tinymemory-integrations/src/cortex/engine/fetch_tests.rs @@ -10,7 +10,7 @@ fn event(id: &str, item: &StoreItem) -> Value { } fn doc(text: &str, repo: Option<&str>) -> StoreItem { - let mut meta = MemoryMeta::from_source(SourceKind::Github, None); + let mut meta = MemoryMeta::from_source(SourceKind::Folder, None); meta.repo = repo.map(str::to_owned); StoreItem::document(text, meta) } diff --git a/crates/tinymemory-integrations/src/cortex/engine/refers_tests.rs b/crates/tinymemory-integrations/src/cortex/engine/refers_tests.rs index 687555b8..48c87e69 100644 --- a/crates/tinymemory-integrations/src/cortex/engine/refers_tests.rs +++ b/crates/tinymemory-integrations/src/cortex/engine/refers_tests.rs @@ -9,7 +9,7 @@ use tinymemory_api::{FetchMode, FetchRequest, MemoryEngine, MemoryMeta, SourceKi use crate::cortex::testing::{both, direct_double, direct_engine}; fn on_day(text: &str, day: u32) -> StoreItem { - let mut meta = MemoryMeta::from_source(SourceKind::Github, None); + let mut meta = MemoryMeta::from_source(SourceKind::Folder, None); // 20:00 UTC: already the next day in India, so the zone is exercised. meta.observed_at = Utc.with_ymd_and_hms(2026, 10, day, 20, 0, 0).single(); StoreItem::document(text, meta) diff --git a/crates/tinymemory-integrations/src/cortex/envelope/labels_tests.rs b/crates/tinymemory-integrations/src/cortex/envelope/labels_tests.rs index 12407ea8..c6f9592a 100644 --- a/crates/tinymemory-integrations/src/cortex/envelope/labels_tests.rs +++ b/crates/tinymemory-integrations/src/cortex/envelope/labels_tests.rs @@ -14,13 +14,13 @@ fn a_digest_is_sixteen_hex_digits_and_stable() { #[test] fn an_item_carries_its_id_label_and_one_per_set_field() { - let mut meta = MemoryMeta::from_source(SourceKind::Github, Some("src-1".into())); + let mut meta = MemoryMeta::from_source(SourceKind::Folder, Some("src-1".into())); meta.repo = Some("a/b".into()); let labels = for_item("abc", &meta); assert_eq!(labels[0], item("abc")); assert!(labels.contains(&format!("tm:r:{}", digest("a/b")))); assert!(labels.contains(&format!("tm:s:{}", digest("src-1")))); - assert!(labels.contains(&format!("tm:k:{}", digest("github")))); + assert!(labels.contains(&format!("tm:k:{}", digest("folder")))); assert_eq!(labels.len(), 4); assert!(labels.iter().all(|l| l.len() <= 24 && !l.contains(','))); } @@ -37,7 +37,7 @@ fn a_filter_narrows_by_its_most_selective_labelled_field() { Some(vec![format!("tm:t:{}", digest("t1"))]) ); let by_sources = MetaFilter { - sources: vec![SourceKind::Rss, SourceKind::Link], + sources: vec![SourceKind::File, SourceKind::Import], ..MetaFilter::default() }; assert_eq!(narrowing(&by_sources).map(|l| l.len()), Some(2)); diff --git a/crates/tinymemory-integrations/src/documents/item/mod_tests.rs b/crates/tinymemory-integrations/src/documents/item/mod_tests.rs index bb696a90..3b056592 100644 --- a/crates/tinymemory-integrations/src/documents/item/mod_tests.rs +++ b/crates/tinymemory-integrations/src/documents/item/mod_tests.rs @@ -26,7 +26,7 @@ async fn html_becomes_a_markdown_document_with_its_title_and_mime() { "Notes

Hi

Body.

", ) .with_mime("text/html"); - let meta = MemoryMeta::from_source(SourceKind::Link, Some("src_1".into())); + let meta = MemoryMeta::from_source(SourceKind::Import, Some("src_1".into())); let item = document_item(&chain, &document, meta).await.unwrap(); assert_eq!(item.kind(), ItemKind::Document); item.validate().unwrap(); @@ -35,7 +35,7 @@ async fn html_becomes_a_markdown_document_with_its_title_and_mime() { assert_eq!(title.as_deref(), Some("Notes")); assert_eq!(body, "# Hi\n\nBody."); assert_eq!(mime.as_deref(), Some("text/html")); - assert_eq!(meta.source.kind, SourceKind::Link); + assert_eq!(meta.source.kind, SourceKind::Import); assert_eq!(meta.source.id.as_deref(), Some("src_1")); assert_eq!(meta.language, None); } From 4fc44827289e944b1285373dd6514ae9cb643a22 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Thu, 8 Oct 2026 23:28:28 +0530 Subject: [PATCH 07/19] test(integrations): add coverage for item source helpers Auto-committed-on: macbook Co-authored-by: Medulla --- .../src/sources/items/mod_tests.rs | 137 +----------------- 1 file changed, 7 insertions(+), 130 deletions(-) diff --git a/crates/tinymemory-integrations/src/sources/items/mod_tests.rs b/crates/tinymemory-integrations/src/sources/items/mod_tests.rs index b050d8ea..17d99dbf 100644 --- a/crates/tinymemory-integrations/src/sources/items/mod_tests.rs +++ b/crates/tinymemory-integrations/src/sources/items/mod_tests.rs @@ -51,124 +51,16 @@ fn base_meta_names_the_source_by_contract_kind_and_entry_id() { for (kind, api) in [ (SourceKind::Folder, Api::Folder), (SourceKind::File, Api::File), - (SourceKind::WebPage, Api::Link), - (SourceKind::GithubRepo, Api::Github), - (SourceKind::RssFeed, Api::Rss), - (SourceKind::Composio, Api::Composio), (SourceKind::Conversation, Api::Conversation), ] { let meta = base_meta(&entry(kind)); assert_eq!(meta.source.kind, api); assert_eq!(meta.source.id.as_deref(), Some("src_test")); } - - let mut composio = entry(SourceKind::Composio); - composio.toolkit = Some("gmail".into()); - assert_eq!(base_meta(&composio).tags, vec!["gmail".to_string()]); -} - -#[test] -fn github_commit_items_carry_repo_and_commit() { - let item = content_item( - &entry(SourceKind::GithubRepo), - content( - "commit:abc123", - "# Fix\n\nbody", - ContentType::Markdown, - serde_json::json!({ "owner": "acme", "repo": "widgets", "sha": "abc123full" }), - ), - Some(1_700_000_000_000), - ) - .unwrap(); - let (title, body, mime, meta) = document_parts(&item); - assert_eq!(title.as_deref(), Some("title of commit:abc123")); - assert_eq!(body, "# Fix\n\nbody"); - assert_eq!(mime.as_deref(), Some("text/markdown")); - assert_eq!(meta.source.kind, Api::Github); - assert_eq!(meta.repo.as_deref(), Some("acme/widgets")); - assert_eq!(meta.commit.as_deref(), Some("abc123full")); - assert_eq!(meta.url, None); - assert_eq!( - meta.observed_at.map(|at| at.timestamp()), - Some(1_700_000_000) - ); -} - -#[test] -fn github_issues_and_prs_carry_their_url() { - let metadata = serde_json::json!({ "owner": "acme", "repo": "widgets", "number": 7 }); - let issue = content_item( - &entry(SourceKind::GithubRepo), - content("issue:7", "text", ContentType::Markdown, metadata.clone()), - None, - ) - .unwrap(); - assert_eq!( - issue.meta().url.as_deref(), - Some("https://github.com/acme/widgets/issues/7") - ); - assert_eq!(issue.meta().commit, None); - - let pr = content_item( - &entry(SourceKind::GithubRepo), - content("pr:7", "text", ContentType::Markdown, metadata), - None, - ) - .unwrap(); - assert_eq!( - pr.meta().url.as_deref(), - Some("https://github.com/acme/widgets/pull/7") - ); -} - -#[test] -fn rss_items_take_the_entry_link_and_publication_time() { - let item = content_item( - &entry(SourceKind::RssFeed), - content( - "guid-1", - "

Hello feed

", - ContentType::Html, - serde_json::json!({ - "link": "https://blog.example.com/post", - "published": "Tue, 21 May 2024 12:00:00 +0000" - }), - ), - None, - ) - .unwrap(); - let (_, body, mime, meta) = document_parts(&item); - assert_eq!(body, "Hello **feed**", "html is converted to markdown"); - assert_eq!(mime.as_deref(), Some("text/markdown")); - assert_eq!(meta.source.kind, Api::Rss); - assert_eq!(meta.url.as_deref(), Some("https://blog.example.com/post")); - assert_eq!( - meta.observed_at.map(|at| at.to_rfc3339()), - Some("2024-05-21T12:00:00+00:00".to_string()) - ); -} - -#[test] -fn link_items_take_the_page_url() { - let item = content_item( - &entry(SourceKind::WebPage), - content( - "https://example.com/page", - "plain words", - ContentType::Plaintext, - serde_json::json!({ "url": "https://example.com/page" }), - ), - None, - ) - .unwrap(); - let (_, _, mime, meta) = document_parts(&item); - assert_eq!(meta.source.kind, Api::Link); - assert_eq!(meta.url.as_deref(), Some("https://example.com/page")); - assert_eq!(mime.as_deref(), Some("text/plain")); } #[test] -fn reader_content_for_local_kinds_and_composio_fills_what_it_can() { +fn reader_content_for_local_kinds_fills_what_it_can() { let mut folder = entry(SourceKind::Folder); folder.path = Some("/notes".into()); let item = content_item( @@ -212,27 +104,12 @@ fn reader_content_for_local_kinds_and_composio_fills_what_it_can() { ) .unwrap(); assert_eq!(conversation.meta().thread_id.as_deref(), Some("t1")); - - let mut composio = entry(SourceKind::Composio); - composio.toolkit = Some("slack".into()); - let item = content_item( - &composio, - content( - "c1", - "sync data", - ContentType::Plaintext, - serde_json::json!({}), - ), - None, - ) - .unwrap(); - assert_eq!(item.meta().tags, vec!["slack".to_string()]); } #[test] fn content_that_converts_to_nothing_is_refused() { let error = content_item( - &entry(SourceKind::WebPage), + &entry(SourceKind::Folder), content( "u", "", @@ -478,7 +355,7 @@ struct BrokenReader; #[async_trait] impl SourceReader for BrokenReader { fn kind(&self) -> SourceKind { - SourceKind::WebPage + SourceKind::Folder } async fn list_items(&self, _: &MemorySourceEntry, _: &Path) -> Result> { @@ -494,7 +371,7 @@ impl SourceReader for BrokenReader { async fn a_failed_listing_fails_the_collect_pass() { let error = collect_items( &BrokenReader, - &entry(SourceKind::WebPage), + &entry(SourceKind::Folder), Path::new("/unused"), &ConverterChain::default(), ) @@ -511,7 +388,7 @@ struct FixedReader; #[async_trait] impl SourceReader for FixedReader { fn kind(&self) -> SourceKind { - SourceKind::WebPage + SourceKind::Folder } async fn list_items(&self, _: &MemorySourceEntry, _: &Path) -> Result> { @@ -536,14 +413,14 @@ impl SourceReader for FixedReader { async fn the_default_read_store_item_maps_reader_content() { let collected = collect_items( &FixedReader, - &entry(SourceKind::WebPage), + &entry(SourceKind::Folder), Path::new("/unused"), &ConverterChain::default(), ) .await .unwrap(); let meta = collected.items[0].meta(); - assert_eq!(meta.url.as_deref(), Some("https://example.com")); + assert_eq!(meta.source.kind, Api::Folder); assert_eq!( meta.observed_at.map(|at| at.timestamp_millis()), Some(1_000) From 6f48a505d08ce2bd3a7eb37049577b61bba3e6d3 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Thu, 8 Oct 2026 23:28:58 +0530 Subject: [PATCH 08/19] test(sources): drop removed fields from fixtures and update kind assertions Test fixtures for conversation and folder sources no longer set the network-source fields that were removed from the config type, and the layout test now expects the Github brain source to map to the Import source kind. The removed-kind test and a mapping assertion were also reformatted to satisfy rustfmt. Auto-committed-on: macbook Co-authored-by: Medulla --- .../src/sources/readers/conversation/mod_tests.rs | 10 ---------- .../src/sources/readers/folder/mod_tests.rs | 10 ---------- .../src/sources/types/mod_tests.rs | 13 ++++++++----- crates/tinymemory-tools/src/layout/mod_tests.rs | 2 +- 4 files changed, 9 insertions(+), 26 deletions(-) diff --git a/crates/tinymemory-integrations/src/sources/readers/conversation/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/conversation/mod_tests.rs index 8f49976f..08535ddf 100644 --- a/crates/tinymemory-integrations/src/sources/readers/conversation/mod_tests.rs +++ b/crates/tinymemory-integrations/src/sources/readers/conversation/mod_tests.rs @@ -11,18 +11,8 @@ fn conversation_source() -> MemorySourceEntry { kind: SourceKind::Conversation, label: "Conversations".into(), enabled: true, - toolkit: None, - connection_id: None, path: None, glob: None, - url: None, - branch: None, - paths: Vec::new(), - max_commits: None, - max_issues: None, - max_prs: None, - max_items: None, - selector: None, max_tokens_per_sync: None, max_cost_per_sync_usd: None, sync_depth_days: None, diff --git a/crates/tinymemory-integrations/src/sources/readers/folder/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/folder/mod_tests.rs index 32060ca7..280fdaaf 100644 --- a/crates/tinymemory-integrations/src/sources/readers/folder/mod_tests.rs +++ b/crates/tinymemory-integrations/src/sources/readers/folder/mod_tests.rs @@ -11,18 +11,8 @@ fn folder_source(path: &str) -> MemorySourceEntry { kind: SourceKind::Folder, label: "Test folder".into(), enabled: true, - toolkit: None, - connection_id: None, path: Some(path.into()), glob: None, - url: None, - branch: None, - paths: Vec::new(), - max_commits: None, - max_issues: None, - max_prs: None, - max_items: None, - selector: None, max_tokens_per_sync: None, max_cost_per_sync_usd: None, sync_depth_days: None, diff --git a/crates/tinymemory-integrations/src/sources/types/mod_tests.rs b/crates/tinymemory-integrations/src/sources/types/mod_tests.rs index b9d6cb55..65d252d3 100644 --- a/crates/tinymemory-integrations/src/sources/types/mod_tests.rs +++ b/crates/tinymemory-integrations/src/sources/types/mod_tests.rs @@ -48,7 +48,13 @@ fn validate_file_requires_path() { #[test] fn the_removed_network_kinds_no_longer_decode() { - for removed in ["twitter_query", "composio", "github_repo", "rss_feed", "web_page"] { + for removed in [ + "twitter_query", + "composio", + "github_repo", + "rss_feed", + "web_page", + ] { let decoded = serde_json::from_str::(&format!("\"{removed}\"")); assert!(decoded.is_err(), "{removed} must not decode"); } @@ -58,10 +64,7 @@ fn the_removed_network_kinds_no_longer_decode() { fn every_config_kind_maps_onto_a_contract_source_kind() { use tinymemory_api::SourceKind as Api; let mapped: Vec = SourceKind::ALL.iter().map(SourceKind::api_kind).collect(); - assert_eq!( - mapped, - vec![Api::Conversation, Api::Folder, Api::File] - ); + assert_eq!(mapped, vec![Api::Conversation, Api::Folder, Api::File]); } #[test] diff --git a/crates/tinymemory-tools/src/layout/mod_tests.rs b/crates/tinymemory-tools/src/layout/mod_tests.rs index 5ff0c61e..2038660e 100644 --- a/crates/tinymemory-tools/src/layout/mod_tests.rs +++ b/crates/tinymemory-tools/src/layout/mod_tests.rs @@ -99,7 +99,7 @@ fn source_ids_round_trip() { } assert_eq!("MD".parse::().unwrap(), BrainSource::Markdown); assert!(" ".parse::().is_err()); - assert_eq!(BrainSource::Github.source_kind(), SourceKind::Github); + assert_eq!(BrainSource::Github.source_kind(), SourceKind::Import); } #[test] From f9da3511d29efdc03fcca4b4a49f06785c37811e Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Thu, 8 Oct 2026 23:29:26 +0530 Subject: [PATCH 09/19] test(conformance): use import source kind in feed fixture The feed fixture now tags its source as an import rather than a file, so the conformance case matches the source kind the feed path actually produces. Auto-committed-on: macbook Co-authored-by: Medulla --- crates/tinymemory-api/src/conformance/suite/fixtures.rs | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/crates/tinymemory-api/src/conformance/suite/fixtures.rs b/crates/tinymemory-api/src/conformance/suite/fixtures.rs index 7392cd59..cb65b227 100644 --- a/crates/tinymemory-api/src/conformance/suite/fixtures.rs +++ b/crates/tinymemory-api/src/conformance/suite/fixtures.rs @@ -272,13 +272,13 @@ impl Run { "feed", with(&|meta| { meta.source = SourceRef { - kind: SourceKind::File, + kind: SourceKind::Import, id: None, } }), ), MetaFilter { - sources: vec![SourceKind::File], + sources: vec![SourceKind::Import], ..base.clone() }, ), From 6752696658731fba89e674bf27ae526484b7bea8 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Thu, 8 Oct 2026 23:30:24 +0530 Subject: [PATCH 10/19] test(tools): drop removed source kinds from tool contract fixtures The tool contract fixtures still listed link, github, rss, and composio as valid source kinds, but those kinds are no longer supported. Removing them keeps the expected schemas in sync with the actual tool contracts. Auto-committed-on: macbook Co-authored-by: Medulla --- .../tests/fixtures/tool_contracts.json | 20 ------------------- 1 file changed, 20 deletions(-) diff --git a/crates/tinymemory-tools/tests/fixtures/tool_contracts.json b/crates/tinymemory-tools/tests/fixtures/tool_contracts.json index fab2afbb..2bc487c7 100644 --- a/crates/tinymemory-tools/tests/fixtures/tool_contracts.json +++ b/crates/tinymemory-tools/tests/fixtures/tool_contracts.json @@ -53,10 +53,6 @@ "enum": [ "folder", "file", - "link", - "github", - "rss", - "composio", "conversation", "agent", "import" @@ -188,10 +184,6 @@ "enum": [ "folder", "file", - "link", - "github", - "rss", - "composio", "conversation", "agent", "import" @@ -329,10 +321,6 @@ "enum": [ "folder", "file", - "link", - "github", - "rss", - "composio", "conversation", "agent", "import" @@ -469,10 +457,6 @@ "enum": [ "folder", "file", - "link", - "github", - "rss", - "composio", "conversation", "agent", "import" @@ -682,10 +666,6 @@ "enum": [ "folder", "file", - "link", - "github", - "rss", - "composio", "conversation", "agent", "import" From 63174a6da19cba55b3d0b10a40332df9c635728b Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Thu, 8 Oct 2026 23:30:48 +0530 Subject: [PATCH 11/19] docs(sources): trim network and composio sections The sources README and architecture doc now describe only the local folder, file and conversation readers, dropping the web page, GitHub, RSS and Composio material along with the fetching and SSRF guard sections. The network and Composio code paths are gone, so the docs no longer document features the module does not have. Auto-committed-on: macbook Co-authored-by: Medulla --- .../src/sources/README.md | 41 +----- docs/architecture/integrations-sources.md | 122 +----------------- 2 files changed, 9 insertions(+), 154 deletions(-) diff --git a/crates/tinymemory-integrations/src/sources/README.md b/crates/tinymemory-integrations/src/sources/README.md index dca6befa..8450a005 100644 --- a/crates/tinymemory-integrations/src/sources/README.md +++ b/crates/tinymemory-integrations/src/sources/README.md @@ -1,8 +1,7 @@ # sources -`tinymemory_integrations::sources` (features `sources` and `sources-network`): -readers that turn a source into `StoreItem`s — a folder, a single file, a web -page, a GitHub repository, an RSS feed, a Composio toolkit payload, or the +`tinymemory_integrations::sources` (feature `sources`): +readers that turn a source into `StoreItem`s — a folder, a single file, or the host's local conversation threads. Conversion to markdown and language detection come from the sibling [`documents`](../documents/README.md) module. Architecture overview: @@ -18,10 +17,7 @@ checks it with `MemorySourceEntry::validate`, nothing more. | --- | --- | | `types` | the configuration a host persists: `MemorySourceEntry` keyed by `SourceKind`, its field rules, and the reader output types (`SourceItem`, `SourceContent`, `ContentType`) | | `readers` | `SourceReader` (list, read, read as a `StoreItem`) and one reader per kind, each in its own module directory; `local_file` holds the shared size-capped read and the path-containment guard | -| `fetch` | one URL into a `RawDocument` or a link item (`sources-network`); the RSS and web-page readers fetch through it with their own body caps | -| `fetch::ssrf` | the SSRF guard: scheme and host policy, one address classifier for literal and resolved addresses, a public-only DNS resolver, per-hop redirect checks, and a capped body reader | | `items` | reader output to `StoreItem`s with `MemoryMeta` filled per kind; `collect_items` drives a reader end to end | -| `composio` | toolkit normalisers (Gmail, Slack, GitHub, Linear, Notion, ClickUp), the `fields::pick_str` lookup they share, `normalise_payload` and `payload_items`. `readers::composio::ComposioReader` is only a placeholder reader | | `error` | the module `Error`, mapped onto `tinymemory_api::Error` | ## Kinds and metadata @@ -32,10 +28,6 @@ Every item's `meta.source` is `SourceRef { kind, id: Some(entry.id) }`. | --- | --- | --- | --- | | `folder` | `Folder` | document | `workspace`, `folder` (containing directory), `file_path`, `language`, `observed_at` (mtime), `mime` | | `file` | `File` | document | as `folder`; `file_item` reads a path with no configured source | -| `web_page` | `Link` | document | `url` | -| `github_repo` | `Github` | document | `repo` (`owner/name`), `commit` (commits), `url` (issues, PRs), `observed_at` | -| `rss_feed` | `Rss` | document | `url` (the entry's link), `observed_at` (published) | -| `composio` | `Composio` | document | `tags = [toolkit]`; payloads add `url`, `observed_at`, `thread_id`, `repo` | | `conversation` | `Conversation` | conversation | `workspace`, `thread_id`, `turns`, `observed_at` (last turn) | ## Folder selection @@ -50,31 +42,10 @@ the process working directory. ## Who decides when -`readers::reader_for` hands out only the local readers (folder, file, -conversation), which are safe to drive on a timer. Network readers are -constructed explicitly, or through `reader_for_request` (feature -`sources-network`) for an explicit user request. Scheduling, credentials, OAuth and egress budgets stay with the host. - -## Fetching - -Every network fetch of a user-configured URL goes through `fetch` and its -SSRF guard. A hostname is checked as text (private and reserved IP literals, -`localhost`, `.local`/`.internal`, single-label names), its resolved addresses -are checked again by the client's resolver, which pins the connection to an -address it has vetted, and every redirect hop is re-checked. Bodies are read -against a cap while streaming: 32 MiB for `fetch_url`, 10 MiB for a web page, -5 MiB for a feed. The client sends the user agent `openhuman`, times out after -20 seconds, and allows only `http(s)`. An IPv6 literal URL is always refused, -even a public address (the bracketed host fails the IP parse and falls into the -single-label rule): fail closed. Failures are typed — `Invalid` for a refused or malformed -URL, `Unreachable`, `Upstream` for a failure status, `TooLarge`. - -Page titles and feed text are decoded with the `documents::html` helpers, so -named and numeric entities decode the same way everywhere. +`readers::reader_for` hands out the folder, file and conversation readers, all of +which read local state only. Scheduling stays with the host. ## Features -- `sources` — the local readers, `items`, `composio` and `types`; implies - `documents`. Links no HTTP stack. -- `sources-network` — adds the GitHub, RSS and web-page readers, `fetch`, and - the SSRF guard. +- `sources` — the readers, `items` and `types`; implies `documents`. Links no + HTTP stack. diff --git a/docs/architecture/integrations-sources.md b/docs/architecture/integrations-sources.md index 98be38d1..f58cc802 100644 --- a/docs/architecture/integrations-sources.md +++ b/docs/architecture/integrations-sources.md @@ -1,7 +1,6 @@ # Integrations: sources -The `sources` module of `tinymemory-integrations` (features `sources` and -`sources-network`) turns a configured source into `StoreItem`s. Overview of the +The `sources` module of `tinymemory-integrations` (feature `sources`) turns a configured source into `StoreItem`s. Overview of the crate: [integrations.md](integrations.md). Module README: [`src/sources/README.md`](../../crates/tinymemory-integrations/src/sources/README.md). @@ -19,10 +18,6 @@ optional fields are required; `validate()` checks them. | `folder` | `path` | `glob` | `Folder` | | `file` | `path` | | `File` | | `conversation` | | | `Conversation` | -| `web_page` | `url` | `selector` | `Link` | -| `github_repo` | `url` | `branch`, `paths`, `max_commits`, `max_issues`, `max_prs` (default 1000 each) | `Github` | -| `rss_feed` | `url` | `max_items` (default 50) | `Rss` | -| `composio` | `toolkit`, `connection_id` | | `Composio` | Every entry also needs a non-blank `id` (no `:` or control characters) and a non-empty `label`, and carries `enabled` and optional sync-budget fields @@ -38,21 +33,14 @@ workspace, converter)`. The default `read_store_item` reads the content and maps it through `items::content_item`; local readers override it to work from raw bytes so a bound converter can handle PDF or DOCX. -`readers::reader_for(kind)` returns only the local readers (folder, file, -conversation), which are safe to drive on a timer. The network kinds return -`None`, meaning "route through the host's sync runner". With -`sources-network`, `reader_for_request(kind)` returns a reader for every kind, -for a host servicing an explicit user request, never a polling loop. +`readers::reader_for(kind)` returns the folder, file or conversation reader; every +reader is local. | Reader | Items | Notes | | --- | --- | --- | | `FolderReader` | one per selected file; id is the folder-relative slash path | With a `glob`, exactly the matches (compiled to a regex over the relative path). Without one, markdown, plain text and source code (`is_default_candidate`). Skips hidden files and directories and `.git`, `.hg`, `.svn`, `target`, `node_modules`, `__pycache__`, `venv`; never follows symlinks; refuses files over `FOLDER_FILE_SIZE_CAP_BYTES` (10 MiB). A relative `path` is anchored on the workspace. | | `FileReader` | exactly one, id is the file name | Same size cap. `FileReader::read_path` reads a path with no configured source. | | `ConversationReader` | one per `/threads/.json` | Threads are `{title, messages: [{role, content, created_at?}]}`. Messages with blank text or an unknown role are dropped. Timestamps: RFC 3339, or epoch seconds or milliseconds. The id may not contain separators or `..`. | -| `WebPageReader` | one: the page URL | With a CSS `selector`, only the text of matching elements (plain text; only the last compound of a descendant chain is honoured). Otherwise the whole page as markdown. 10 MiB body cap. | -| `RssReader` | one per feed entry (RSS or Atom), up to `max_items` | The parsed feed is cached for 60 seconds so a list-then-read pass downloads it once. 5 MiB cap; non-UTF-8 bodies are refused. | -| `GithubReader` | `commit:`, `issue:`, `pr:` | See below. | -| `ComposioReader` | one: the connection | A placeholder; see Composio. | ### local_file and ensure_within_base @@ -65,24 +53,6 @@ base, or `Error::Io` when either cannot be canonicalised. The folder reader applies it on every read; the conversation reader applies it within the threads directory. -### GitHub transports - -The reader pulls **project activity**, not source code, from -`https://github.com//` (extra path segments such as `/tree/main` -are rejected). It combines three transports: - -- **Commits:** a bare clone under `/git_cache//.git`, - fetched with an explicit refspec and listed with `git log`, honouring - `branch` and `paths`. `git` must be on `PATH`. If the clone fails, it falls - back to the commits API. -- **Issues and pull requests:** `gh api` when the `gh` CLI is available - (authenticated, higher rate limit; probed once per process), otherwise the - unauthenticated REST API at `api.github.com`. The list pass caches full rows - so reads do not refetch. -- Limits: `max_commits`, `max_issues`, `max_prs` per sync, default 1000 each. - Listing fails only when every call failed; otherwise the partial list is - returned. Failures surface as `Error::Reader`. - ## Items mapping `items` maps reader output to `StoreItem`s. Every item's `meta.source` is @@ -93,10 +63,6 @@ are rejected). It combines three transports: | Kind | Item | Metadata filled | | --- | --- | --- | | folder, file | document | `workspace`, `folder`, `file_path` (canonical), `language`, `observed_at` (mtime), `mime` | -| github | document | `repo` (`owner/name`), `commit` (commits), `url` (issues and PRs), `observed_at` | -| web page | document | `url` | -| rss | document | `url` (entry link), `observed_at` (published) | -| composio | document | `tags = [toolkit]`; payloads add `url`, `observed_at`, `thread_id`, `repo` | | conversation | conversation | `workspace`, `thread_id`, `turns` (`0..=n-1`), `observed_at` (last turn, else mtime) | `collect_items(reader, entry, workspace, converter)` lists and reads every @@ -104,85 +70,3 @@ item, returning `Collected { items, skipped }`. One bad item lands in `skipped` with its error and the pass continues; only a listing failure is an `Err`. Other entry points: `file_item` (a path with no source), `conversation_item`, `content_item` and `items::local_file_item`. - -## Composio - -Composio data does not arrive item by item. A host runs toolkit actions with -its own credentials and hands the raw responses to `sources::composio`, which -holds no credential, opens no socket and decides nothing about when to sync. - -1. **Normalisers**, pure `serde_json::Value` transforms, one module per - toolkit. They walk Composio's envelope variants (top level, under `data`, - under `data.data`) and return the first array found: `clickup` - (`extract_tasks`), `github` (`extract_issues`), `linear` - (`extract_issues`), `notion` (`extract_results`, `extract_page_markdown`), - each with title, id and updated-time helpers. `fields::pick_str` is the - shared lookup: it tries dotted paths, descends only through objects, and - rejects non-string leaves. -2. **Post-processors the host must call** for two toolkits, because their raw - responses are too verbose: - - `gmail_post_process::post_process(slug, arguments, &mut data)` rewrites a - `GMAIL_FETCH_EMAILS` response into slim `messages[]` (other Gmail slugs - pass through; `raw_html: true` in the arguments skips the reshape). If the - response carries a response-level `markdownFormatted` string, call - `apply_response_level_markdown(&mut data, markdown)` **before** - `post_process`; it is a no-op unless the split count matches the message - count. `format_email_local_time` renders in the host's local timezone; - the raw UTC fields are preserved. - - `slack_post_process::post_process(slug, arguments, &mut data)` reshapes - `SLACK_FETCH_CONVERSATION_HISTORY`, `SLACK_LIST_CONVERSATIONS` and - `SLACK_SEARCH_MESSAGES`; unknown slugs are no-ops. `channel_id` for history - is injected by the host (it is in the request, not the response), and user - ids are resolved by the host. -3. `normalise_payload(toolkit, &data)` returns `ComposioDocument`s (id, title, - markdown body, url, `observed_at`, `thread_id`, `repo`), dispatching on the - case-insensitive toolkit slug: `gmail`, `slack`, `github`, `linear`, - `notion`, `clickup`; any other toolkit falls back to each record as fenced - JSON, so a new toolkit is ingested verbosely rather than dropped. Records - with no text are skipped. `payload_items(toolkit, source_id, &data)` wraps - them as `StoreItem::Document` with `source.kind = Composio`, - `source.id = source_id` and `tags = [toolkit]`. - -`readers::composio::ComposioReader` is only a placeholder so -`reader_for_request` can serve every kind: `list_items` returns the connection -as one sync target. - -## Fetching and the SSRF guard (sources-network) - -`fetch::fetch_url(url)` fetches one URL into a `RawDocument`: the `Content-Type` -becomes the declared MIME, the URL the origin, and the last path segment (if -it has an extension) the filename. `fetch::link_item(url, source_id, -converter)` converts it into a document with `source.kind = Link` and `url` -set. The cap is `MAX_DOCUMENT_BYTES` (32 MiB), applied while streaming. No -retries, no robots.txt, no scheduling. Errors: `Invalid` (malformed or refused -URL, empty body), `Unreachable` (never completed, interrupted read, may -succeed later), `Upstream` (non-success status), `TooLarge`. - -The web-page and RSS readers fetch through the same path with tighter caps -(10 MiB and 5 MiB). `fetch::ssrf` is public so a host fetching a user-supplied -URL by other means applies the same policy. - -Guard rules: - -- **Scheme:** `http` and `https` only. -- **Host text:** refused are empty hosts, `localhost`, `.local` and - `.internal` names, single-label names (internal service names such as - `redis`), and IP literals that are not globally routable. -- **One address classifier** serves literals and resolved addresses. Not - fetchable: loopback, private, link-local (including `169.254.169.254`), - unspecified, CGNAT (`100.64.0.0/10`), `192.0.0.0/16`, multicast, broadcast, - documentation and benchmarking ranges, reserved `240.0.0.0/4`, IPv6 - unique-local and link-local; IPv4-mapped and IPv4-compatible IPv6 addresses - are judged by their IPv4 part. -- **IPv6 literals fail closed.** A URL such as `http://[2606:4700::1111]/` is - refused whatever the address: the host text keeps its brackets, does not parse - as an IP, and has no dot, so the single-label rule blocks it. A hostname with - an AAAA record is still vetted when resolved. -- **Resolver:** `PublicOnlyResolver` keeps only public addresses and fails the - request when none remain, pinning the connection to a vetted address with no - re-resolution between check and connect. -- **Redirects** are re-checked per hop; a refused hop is not followed, and the - read fails on the resulting non-success status. -- **Client:** 20-second timeout and the user agent `openhuman`. -- **Body caps** are enforced while streaming (`read_body_capped`), with an early - rejection on a truthful `Content-Length`. From 26f62cc7e60abf553d506c1d250f4d1024f14a85 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Thu, 8 Oct 2026 23:31:06 +0530 Subject: [PATCH 12/19] docs: drop network and composio sources from feature docs The `sources-network` feature and the Composio normalisers are gone, so the feature tables, dependency weights and `SourceKind` listings no longer mention them. `full` now enables `sources` directly, and the source readers are described as local-only throughout. Auto-committed-on: macbook Co-authored-by: Medulla --- README.md | 5 ++--- crates/tinymemory-integrations/README.md | 11 ++++------- crates/tinymemory-integrations/src/lib.rs | 2 +- docs/architecture/api-items.md | 5 ++--- docs/architecture/overview.md | 7 +++---- docs/architecture/testing.md | 4 ++-- docs/specs/memory-v2.md | 2 +- 7 files changed, 15 insertions(+), 21 deletions(-) diff --git a/README.md b/README.md index 3b6ab845..788675c8 100644 --- a/README.md +++ b/README.md @@ -63,11 +63,10 @@ else on request, so a host pays only for what it uses. | `documents` | `documents` | Format sniffing and conversion to markdown, producing `StoreItem::Document` | | `documents-office` | `documents::OfficeConverter` | PDF, DOCX, PPTX and XLSX to markdown (implies `documents`) | | `brain` | `brain` | Files into `tinymemory_tools::BrainDocument`s, by the source type their format implies (implies `documents`) | -| `sources` | `sources` | Folder, file and conversation readers, Composio normalisers (implies `documents`) | -| `sources-network` | `sources::fetch` and the network readers | GitHub, RSS and web-page readers and `fetch_url`, behind the SSRF guard (implies `sources`) | +| `sources` | `sources` | Folder, file and conversation readers (implies `documents`) | | `safety` | `safety` | Secret and PII scrubbing of a `StoreItem` | | `legacy-import` | `import` | Migrating a v1 (embedded TinyCortex) workspace into any engine | -| `full` | all of the above | `cortex`, `documents-office`, `brain`, `sources-network`, `safety`, `legacy-import` | +| `full` | all of the above | `cortex`, `documents-office`, `brain`, `sources`, `safety`, `legacy-import` | Dependency weight per feature is tabulated in [`crates/tinymemory-integrations/README.md`](crates/tinymemory-integrations/README.md). diff --git a/crates/tinymemory-integrations/README.md b/crates/tinymemory-integrations/README.md index 6d02b39a..32140e0c 100644 --- a/crates/tinymemory-integrations/README.md +++ b/crates/tinymemory-integrations/README.md @@ -17,14 +17,13 @@ wants one integration enables one feature and links nothing else. | `cortex`, `registry`, `config` | `cortex` (default) | `CortexEngine` over two wires (`cortexdb`, `tinyhumans`); `list_engines`, `build_engine`, `EngineCredential`; `MemoryConfig` | [`src/cortex/README.md`](src/cortex/README.md) | | `documents` | `documents` | Format sniffing and conversion to markdown, producing `StoreItem::Document`. No I/O. | [`src/documents/README.md`](src/documents/README.md) | | `documents::OfficeConverter` | `documents-office` | PDF, DOCX, PPTX and XLSX to markdown, in process | (same) | -| `sources` | `sources` | Readers for folders, files and conversations; Composio payload normalisers; `collect_items`. Links no HTTP stack. | [`src/sources/README.md`](src/sources/README.md) | -| `sources::fetch`, GitHub, RSS and web-page readers | `sources-network` | The network readers and `fetch_url`, all behind the SSRF guard | (same) | +| `sources` | `sources` | Readers for folders, files and conversations; `collect_items`. Links no HTTP stack. | [`src/sources/README.md`](src/sources/README.md) | | `safety` | `safety` | Secret and PII scrubbing of a `StoreItem` before it is stored | [`src/safety/README.md`](src/safety/README.md) | | `import` | `legacy-import` | Reads a v1 (embedded TinyCortex) workspace and migrates it into any engine, resumably | [`src/import/README.md`](src/import/README.md) | -`full` turns on `cortex`, `documents-office`, `sources-network`, `safety` and +`full` turns on `cortex`, `documents-office`, `sources`, `safety` and `legacy-import`. Feature implications: `documents-office` implies `documents`; -`sources` implies `documents`; `sources-network` implies `sources`. +`sources` implies `documents`. ## Dependency weight per feature @@ -39,7 +38,6 @@ already needs): | `documents-office` | `pdf-extract`, `calamine`, `quick-xml`, `zip` (all pure Rust, no system libraries) | | `brain` | `tinymemory-tools` (for `BrainDocument`; it adds only `futures`, `chrono`, `log`) | | `sources` | `schemars`, `regex`, `walkdir`, `chrono`, `log`, `tracing` | -| `sources-network` | `reqwest`, `futures`, `tokio` with `process`, `io-util` and `net` (the GitHub reader runs `gh` and `git`; the SSRF resolver does DNS) | | `safety` | `regex`, `serde_json`, `log` | | `legacy-import` | `rusqlite` with bundled SQLite (compiles C; no system SQLite needed) | @@ -58,8 +56,7 @@ sources ──▶ documents ──▶ safety ──▶ engine.store ``` 1. **sources** lists a configured source and reads each entry. Local readers - hand raw bytes to the converter; network readers fetch through the SSRF - guard. `sources::collect_items` drives one source and collects per-item + hand raw bytes to the converter. `sources::collect_items` drives one source and collects per-item failures instead of aborting. 2. **documents** sniffs the format and converts bodies to markdown. A `ConverterChain` decides which converter handles which format; a host can diff --git a/crates/tinymemory-integrations/src/lib.rs b/crates/tinymemory-integrations/src/lib.rs index f320a470..0bd5a576 100644 --- a/crates/tinymemory-integrations/src/lib.rs +++ b/crates/tinymemory-integrations/src/lib.rs @@ -8,7 +8,7 @@ //! | [`cortex`], [`registry`], [`config`] | `cortex` (default) | The CortexDB engine over its two wires, and building one from configuration | //! | `documents` | `documents`, `documents-office` | Format sniffing and conversion to markdown, emitting `StoreItem::Document` | //! | `brain` | `brain` | Converting files into the brain documents `tinymemory_tools::Brain` ingests, by source type | -//! | `sources` | `sources`, `sources-network` | Readers turning folders, files, links, GitHub, RSS, Composio payloads and conversations into `StoreItem`s | +//! | `sources` | `sources` | Readers turning folders, files and conversations into `StoreItem`s | //! | `safety` | `safety` | Secret and PII scrubbing for a `StoreItem` before it is stored | //! | `import` | `legacy-import` | Migrating a legacy v1 (embedded TinyCortex) workspace into any engine | //! diff --git a/docs/architecture/api-items.md b/docs/architecture/api-items.md index 2909c584..89d183c2 100644 --- a/docs/architecture/api-items.md +++ b/docs/architecture/api-items.md @@ -165,8 +165,7 @@ namespace are omitted on serialisation. | `observed_actor` | `Option` | `{ id, name? }`: who said or did it when not the memory's owner, such as an email's sender or a channel's (`id` is `type:id`, `user:priya@acme.com`, `user:+15551234567`). Excluded from the fingerprint; the CortexDB engine sends it only with attribution on. | `MemoryMeta::from_source(kind, id)` sets only the source. `SourceKind` is -`folder | file | link | github | rss | composio | conversation | agent | -import` (`SourceKind::ALL`, `as_str`); `agent` is the default. +`folder | file | conversation | agent | import` (`SourceKind::ALL`, `as_str`); `agent` is the default. ## `MetaFilter` @@ -200,7 +199,7 @@ equals the default, so a filter holding only a `reach` is **not** empty), { "reach": { "at": "team:acme/agent:writer", "inherit": true, "descendants": false }, "kinds": ["document", "learning"], - "sources": ["folder", "github"], + "sources": ["folder", "file"], "folder": "/notes/rust", "tags_any": ["style", "review"], "observed_after": "2026-01-01T00:00:00Z", diff --git a/docs/architecture/overview.md b/docs/architecture/overview.md index 94e6916e..53711d88 100644 --- a/docs/architecture/overview.md +++ b/docs/architecture/overview.md @@ -55,11 +55,10 @@ alone, so the contract is the only coupling point. | `tinymemory-integrations` | `cortex` (default) | `cortex::CortexEngine` (both wires), `registry` (`list_engines`, `build_engine`), `config` (`MemoryConfig`) | | | `documents` | Format sniffing and conversion to markdown | | | `documents-office` | PDF, DOCX, PPTX, XLSX conversion (implies `documents`) | -| | `sources` | Readers for folders, files, conversations; Composio normalisers (implies `documents`) | -| | `sources-network` | GitHub, RSS and web-page readers behind the SSRF guard (implies `sources`) | +| | `sources` | Readers for folders, files, conversations (implies `documents`) | | | `safety` | Secret and PII scrubbing of a `StoreItem` | | | `legacy-import` | Reading a v1 workspace; `import::migrate` and `migrate_with` | -| | `full` | `cortex`, `documents-office`, `sources-network`, `safety`, `legacy-import` | +| | `full` | `cortex`, `documents-office`, `sources`, `safety`, `legacy-import` | ## Write path @@ -72,7 +71,7 @@ source reader ──▶ documents ──▶ safety ──▶ engine.store ── ``` 1. **Source reader** (`sources`): lists a configured source (folder, file, - link, GitHub, RSS, Composio payload, conversation) and reads each entry. + conversation) and reads each entry. `sources::collect_items` does this for one source; one bad entry lands in `Collected::skipped` instead of aborting the pass. 2. **Documents conversion** (`documents`): sniffs the format and converts the diff --git a/docs/architecture/testing.md b/docs/architecture/testing.md index 07340215..97e50a2d 100644 --- a/docs/architecture/testing.md +++ b/docs/architecture/testing.md @@ -78,7 +78,7 @@ works on its own, not only in the union: | | `--no-default-features --features cortex` | | | `--no-default-features --features documents` | | | `--no-default-features --features documents-office` | -| | `--no-default-features --features sources-network` (proves it implies `sources`) | +| | `--no-default-features --features sources` | | | `--no-default-features --features safety` | | | `--no-default-features --features legacy-import` | | | `--no-default-features --features full` | @@ -255,7 +255,7 @@ idempotency pairs, and every recall, answer and forget body for assertions. | --- | --- | --- | | `feature_surface.rs` | `full` | the modules compose: scrub an item, store it in the reference engine, run the conformance suite, compile a context | | `documents_office.rs` | `documents-office` | `OfficeConverter` prepends to the default `ConverterChain` and supports PDF, DOCX, XLSX, PPTX | -| `reader_dispatch.rs` | `sources` (`sources-network` for one test) | local readers are constructed for timers, network readers only for requests | +| `reader_dispatch.rs` | `sources` | every kind has a local reader | | `legacy_import.rs` | `legacy-import` | importing v1 workspaces, every mapping, ordering and resumption | | `live_cortexdb.rs`, `office_live.rs` | `cortex` (+ `documents-office`) | a real CortexDB; skipped unless configured | diff --git a/docs/specs/memory-v2.md b/docs/specs/memory-v2.md index 43f7255e..d697282e 100644 --- a/docs/specs/memory-v2.md +++ b/docs/specs/memory-v2.md @@ -91,7 +91,7 @@ pub struct MemoryMeta { pub tags: Vec, pub observed_at: Option>, } -pub enum SourceKind { Folder, File, Link, Github, Rss, Composio, Conversation, Agent, Import } +pub enum SourceKind { Folder, File, Conversation, Agent, Import } ``` `MetaFilter` has the same optional fields (each an exact match, `folder` and From bb0c72afb5b15ec70200408a33c4b4c13a7459d3 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Thu, 8 Oct 2026 23:31:12 +0530 Subject: [PATCH 13/19] docs: drop sources-network feature from docs and ci The `sources-network` feature no longer exists, so the CI matrix entry and the feature tables in the integrations and memory-v2 docs now reference `sources` alone. The memory-v2 description also narrows the `sources` feature to folder, file and conversation readers. Auto-committed-on: macbook Co-authored-by: Medulla --- .github/workflows/ci.yml | 2 +- docs/architecture/integrations.md | 2 +- docs/specs/memory-v2.md | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index c3f94475..d9172901 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -143,7 +143,7 @@ jobs: features: --no-default-features --features documents-office - name: sources network implication package: tinymemory-integrations - features: --no-default-features --features sources-network + features: --no-default-features --features sources - name: safety package: tinymemory-integrations features: --no-default-features --features safety diff --git a/docs/architecture/integrations.md b/docs/architecture/integrations.md index 2c888754..9c002d62 100644 --- a/docs/architecture/integrations.md +++ b/docs/architecture/integrations.md @@ -11,7 +11,7 @@ CortexDB engine and the registry are in [cortex.md](cortex.md). | --- | --- | --- | | `cortex`, `registry`, `config` | `cortex` (default) | [cortex.md](cortex.md) | | `documents` | `documents`, `documents-office` | [Documents](#documents) | -| `sources` | `sources`, `sources-network` | [integrations-sources.md](integrations-sources.md) | +| `sources` | `sources` | [integrations-sources.md](integrations-sources.md) | | `safety` | `safety` | [Safety](#safety) | | `import` | `legacy-import` | [Legacy v1 import](#legacy-v1-import) | diff --git a/docs/specs/memory-v2.md b/docs/specs/memory-v2.md index d697282e..e8fbf5c5 100644 --- a/docs/specs/memory-v2.md +++ b/docs/specs/memory-v2.md @@ -30,7 +30,7 @@ Three crates, one directory each under `crates/`. Their relationships are in | --- | --- | | `tinymemory-api` | The contract: `MemoryEngine`, request/response types, `MemoryMeta`, `MetaFilter`, namespaces, `EngineDescriptor`, `Error`. No I/O. Feature `conformance`: the behavioural suite every engine must pass, plus a reference in-memory engine. | | `tinymemory-tools` | The agent tool spec `MemoryTools` (seven tools, host-fixed namespace and reach) and `context`, the `ContextCompiler` that builds `context.md` from an engine. | -| `tinymemory-integrations` | Everything that talks to the outside world, each behind a feature: `cortex` (default; the CortexDB engine, registered twice as `cortexdb` and `tinyhumans`, plus the registry `list_engines`/`build_engine` and `MemoryConfig`), `documents` and `documents-office` (format sniffing and conversion to markdown, emitting `StoreItem::Document`; PDF/DOCX/PPTX/XLSX via `OfficeConverter` or a host `DocumentConverter`), `sources` and `sources-network` (readers for folder, file, link, GitHub, RSS, Composio payloads and conversations, with the SSRF guard), `safety` (secret/PII scrubbing applied to every item before `store`), and `legacy-import` (reads a v1 TinyCortex workspace and migrates it into any engine). | +| `tinymemory-integrations` | Everything that talks to the outside world, each behind a feature: `cortex` (default; the CortexDB engine, registered twice as `cortexdb` and `tinyhumans`, plus the registry `list_engines`/`build_engine` and `MemoryConfig`), `documents` and `documents-office` (format sniffing and conversion to markdown, emitting `StoreItem::Document`; PDF/DOCX/PPTX/XLSX via `OfficeConverter` or a host `DocumentConverter`), `sources` (readers for folder, file and conversations), `safety` (secret/PII scrubbing applied to every item before `store`), and `legacy-import` (reads a v1 TinyCortex workspace and migrates it into any engine). | Deleted: the earlier `tinymemory` facade, `tinymemory-cortex`, `tinymemory-documents`, `tinymemory-sources`, `tinymemory-safety`, From f736afba9d3c93aa48bbfe308c2e65fa842736e9 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Thu, 8 Oct 2026 23:41:43 +0530 Subject: [PATCH 14/19] ci: rename sources feature matrix entry The CI feature matrix entry for the sources feature was renamed from "sources network implication" to "sources" so the job name matches the feature it actually builds. Auto-committed-on: macbook Co-authored-by: Medulla --- .github/workflows/ci.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index d9172901..587ffdb2 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -141,7 +141,7 @@ jobs: - name: documents-office package: tinymemory-integrations features: --no-default-features --features documents-office - - name: sources network implication + - name: sources package: tinymemory-integrations features: --no-default-features --features sources - name: safety From aa078bf65dfc70230f0ef437d0c0167433b7dceb Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Thu, 8 Oct 2026 23:41:56 +0530 Subject: [PATCH 15/19] feat(sources): add network readers and Composio normalisers Adds GitHub, RSS and web-page readers plus `fetch_url` behind a shared SSRF guard, and normalisers that turn Composio toolkit payloads (Gmail, Slack, GitHub, Linear, Notion, ClickUp) into documents. The network readers sit behind a new `sources-network` feature so hosts that only read local state keep linking no HTTP stack, and `SourceKind` regains the `link`, `github`, `rss` and `composio` variants that previously decoded to `import`. Auto-committed-on: macbook Co-authored-by: Medulla --- .github/workflows/ci.yml | 4 +- Cargo.lock | 11 + README.md | 5 +- .../src/conformance/suite/fixtures.rs | 4 +- .../tinymemory-api/src/meta/filter_tests.rs | 2 +- crates/tinymemory-api/src/meta/mod.rs | 25 +- crates/tinymemory-api/src/meta/mod_tests.rs | 17 +- crates/tinymemory-integrations/Cargo.toml | 14 +- crates/tinymemory-integrations/README.md | 11 +- .../src/cortex/engine/fetch_tests.rs | 2 +- .../src/cortex/engine/refers_tests.rs | 2 +- .../src/cortex/envelope/labels_tests.rs | 6 +- .../src/documents/item/mod_tests.rs | 4 +- crates/tinymemory-integrations/src/lib.rs | 2 +- .../src/sources/README.md | 41 +- .../src/sources/composio/clickup/mod.rs | 69 ++ .../src/sources/composio/clickup/mod_tests.rs | 58 ++ .../src/sources/composio/documents/mod.rs | 343 ++++++++++ .../sources/composio/documents/mod_tests.rs | 197 ++++++ .../src/sources/composio/fields/mod.rs | 44 ++ .../src/sources/composio/fields/mod_tests.rs | 37 ++ .../src/sources/composio/github/mod.rs | 116 ++++ .../src/sources/composio/github/mod_tests.rs | 97 +++ .../composio/gmail_post_process/mod.rs | 498 ++++++++++++++ .../composio/gmail_post_process/mod_tests.rs | 356 ++++++++++ .../src/sources/composio/linear/mod.rs | 79 +++ .../src/sources/composio/linear/mod_tests.rs | 93 +++ .../src/sources/composio/mod.rs | 35 + .../src/sources/composio/notion/mod.rs | 88 +++ .../src/sources/composio/notion/mod_tests.rs | 111 ++++ .../composio/slack_post_process/mod.rs | 324 +++++++++ .../composio/slack_post_process/mod_tests.rs | 306 +++++++++ .../src/sources/fetch/mod.rs | 151 +++++ .../src/sources/fetch/mod_tests.rs | 177 +++++ .../src/sources/fetch/ssrf/mod.rs | 231 +++++++ .../src/sources/fetch/ssrf/mod_tests.rs | 236 +++++++ .../src/sources/items/mod.rs | 61 +- .../src/sources/items/mod_tests.rs | 137 +++- .../src/sources/mod.rs | 23 +- .../src/sources/readers/composio/mod.rs | 78 +++ .../src/sources/readers/composio/mod_tests.rs | 51 ++ .../sources/readers/conversation/mod_tests.rs | 10 + .../src/sources/readers/folder/mod_tests.rs | 10 + .../src/sources/readers/github/api/mod.rs | 391 +++++++++++ .../readers/github/api/transport_override.rs | 5 + .../readers/github/api/transport_tests.rs | 34 + .../src/sources/readers/github/git/mod.rs | 320 +++++++++ .../sources/readers/github/git/mod_tests.rs | 219 ++++++ .../src/sources/readers/github/issues/mod.rs | 304 +++++++++ .../readers/github/issues/mod_tests.rs | 151 +++++ .../src/sources/readers/github/mod.rs | 308 +++++++++ .../src/sources/readers/github/mod_tests.rs | 624 ++++++++++++++++++ .../src/sources/readers/github/types.rs | 122 ++++ .../src/sources/readers/mod.rs | 70 +- .../src/sources/readers/rss/mod.rs | 355 ++++++++++ .../src/sources/readers/rss/mod_tests.rs | 386 +++++++++++ .../src/sources/readers/rss/types.rs | 25 + .../src/sources/readers/web_page/mod.rs | 444 +++++++++++++ .../src/sources/readers/web_page/mod_tests.rs | 302 +++++++++ .../src/sources/readers/web_page/types.rs | 13 + .../src/sources/types/mod.rs | 94 ++- .../src/sources/types/mod_tests.rs | 125 +++- .../tests/reader_dispatch.rs | 20 +- .../tinymemory-tools/src/layout/mod_tests.rs | 2 +- crates/tinymemory-tools/src/layout/source.rs | 5 +- .../tests/fixtures/tool_contracts.json | 20 + docs/architecture/api-items.md | 5 +- docs/architecture/integrations-sources.md | 122 +++- docs/architecture/integrations.md | 2 +- docs/architecture/overview.md | 7 +- docs/architecture/testing.md | 4 +- docs/specs/memory-v2.md | 4 +- 72 files changed, 8537 insertions(+), 112 deletions(-) create mode 100644 crates/tinymemory-integrations/src/sources/composio/clickup/mod.rs create mode 100644 crates/tinymemory-integrations/src/sources/composio/clickup/mod_tests.rs create mode 100644 crates/tinymemory-integrations/src/sources/composio/documents/mod.rs create mode 100644 crates/tinymemory-integrations/src/sources/composio/documents/mod_tests.rs create mode 100644 crates/tinymemory-integrations/src/sources/composio/fields/mod.rs create mode 100644 crates/tinymemory-integrations/src/sources/composio/fields/mod_tests.rs create mode 100644 crates/tinymemory-integrations/src/sources/composio/github/mod.rs create mode 100644 crates/tinymemory-integrations/src/sources/composio/github/mod_tests.rs create mode 100644 crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod.rs create mode 100644 crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod_tests.rs create mode 100644 crates/tinymemory-integrations/src/sources/composio/linear/mod.rs create mode 100644 crates/tinymemory-integrations/src/sources/composio/linear/mod_tests.rs create mode 100644 crates/tinymemory-integrations/src/sources/composio/mod.rs create mode 100644 crates/tinymemory-integrations/src/sources/composio/notion/mod.rs create mode 100644 crates/tinymemory-integrations/src/sources/composio/notion/mod_tests.rs create mode 100644 crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod.rs create mode 100644 crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod_tests.rs create mode 100644 crates/tinymemory-integrations/src/sources/fetch/mod.rs create mode 100644 crates/tinymemory-integrations/src/sources/fetch/mod_tests.rs create mode 100644 crates/tinymemory-integrations/src/sources/fetch/ssrf/mod.rs create mode 100644 crates/tinymemory-integrations/src/sources/fetch/ssrf/mod_tests.rs create mode 100644 crates/tinymemory-integrations/src/sources/readers/composio/mod.rs create mode 100644 crates/tinymemory-integrations/src/sources/readers/composio/mod_tests.rs create mode 100644 crates/tinymemory-integrations/src/sources/readers/github/api/mod.rs create mode 100644 crates/tinymemory-integrations/src/sources/readers/github/api/transport_override.rs create mode 100644 crates/tinymemory-integrations/src/sources/readers/github/api/transport_tests.rs create mode 100644 crates/tinymemory-integrations/src/sources/readers/github/git/mod.rs create mode 100644 crates/tinymemory-integrations/src/sources/readers/github/git/mod_tests.rs create mode 100644 crates/tinymemory-integrations/src/sources/readers/github/issues/mod.rs create mode 100644 crates/tinymemory-integrations/src/sources/readers/github/issues/mod_tests.rs create mode 100644 crates/tinymemory-integrations/src/sources/readers/github/mod.rs create mode 100644 crates/tinymemory-integrations/src/sources/readers/github/mod_tests.rs create mode 100644 crates/tinymemory-integrations/src/sources/readers/github/types.rs create mode 100644 crates/tinymemory-integrations/src/sources/readers/rss/mod.rs create mode 100644 crates/tinymemory-integrations/src/sources/readers/rss/mod_tests.rs create mode 100644 crates/tinymemory-integrations/src/sources/readers/rss/types.rs create mode 100644 crates/tinymemory-integrations/src/sources/readers/web_page/mod.rs create mode 100644 crates/tinymemory-integrations/src/sources/readers/web_page/mod_tests.rs create mode 100644 crates/tinymemory-integrations/src/sources/readers/web_page/types.rs diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 587ffdb2..c3f94475 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -141,9 +141,9 @@ jobs: - name: documents-office package: tinymemory-integrations features: --no-default-features --features documents-office - - name: sources + - name: sources network implication package: tinymemory-integrations - features: --no-default-features --features sources + features: --no-default-features --features sources-network - name: safety package: tinymemory-integrations features: --no-default-features --features safety diff --git a/Cargo.lock b/Cargo.lock index a0750e43..108a6aec 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1538,6 +1538,16 @@ version = "2.0.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba" +[[package]] +name = "signal-hook-registry" +version = "1.4.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c4db69cba1110affc0e9f7bcd48bbf87b3f4fc7c61fc9155afd4c469eb3d6c1b" +dependencies = [ + "errno", + "libc", +] + [[package]] name = "simd-adler32" version = "0.3.10" @@ -1755,6 +1765,7 @@ dependencies = [ "libc", "mio", "pin-project-lite", + "signal-hook-registry", "socket2", "tokio-macros", "windows-sys 0.61.2", diff --git a/README.md b/README.md index 788675c8..3b6ab845 100644 --- a/README.md +++ b/README.md @@ -63,10 +63,11 @@ else on request, so a host pays only for what it uses. | `documents` | `documents` | Format sniffing and conversion to markdown, producing `StoreItem::Document` | | `documents-office` | `documents::OfficeConverter` | PDF, DOCX, PPTX and XLSX to markdown (implies `documents`) | | `brain` | `brain` | Files into `tinymemory_tools::BrainDocument`s, by the source type their format implies (implies `documents`) | -| `sources` | `sources` | Folder, file and conversation readers (implies `documents`) | +| `sources` | `sources` | Folder, file and conversation readers, Composio normalisers (implies `documents`) | +| `sources-network` | `sources::fetch` and the network readers | GitHub, RSS and web-page readers and `fetch_url`, behind the SSRF guard (implies `sources`) | | `safety` | `safety` | Secret and PII scrubbing of a `StoreItem` | | `legacy-import` | `import` | Migrating a v1 (embedded TinyCortex) workspace into any engine | -| `full` | all of the above | `cortex`, `documents-office`, `brain`, `sources`, `safety`, `legacy-import` | +| `full` | all of the above | `cortex`, `documents-office`, `brain`, `sources-network`, `safety`, `legacy-import` | Dependency weight per feature is tabulated in [`crates/tinymemory-integrations/README.md`](crates/tinymemory-integrations/README.md). diff --git a/crates/tinymemory-api/src/conformance/suite/fixtures.rs b/crates/tinymemory-api/src/conformance/suite/fixtures.rs index cb65b227..0a7bf304 100644 --- a/crates/tinymemory-api/src/conformance/suite/fixtures.rs +++ b/crates/tinymemory-api/src/conformance/suite/fixtures.rs @@ -272,13 +272,13 @@ impl Run { "feed", with(&|meta| { meta.source = SourceRef { - kind: SourceKind::Import, + kind: SourceKind::Rss, id: None, } }), ), MetaFilter { - sources: vec![SourceKind::Import], + sources: vec![SourceKind::Rss], ..base.clone() }, ), diff --git a/crates/tinymemory-api/src/meta/filter_tests.rs b/crates/tinymemory-api/src/meta/filter_tests.rs index 18a2d4aa..a8cc89bb 100644 --- a/crates/tinymemory-api/src/meta/filter_tests.rs +++ b/crates/tinymemory-api/src/meta/filter_tests.rs @@ -150,7 +150,7 @@ fn every_exact_field_must_match() { ..Default::default() }, MetaFilter { - sources: vec![SourceKind::File], + sources: vec![SourceKind::Rss], ..Default::default() }, ), diff --git a/crates/tinymemory-api/src/meta/mod.rs b/crates/tinymemory-api/src/meta/mod.rs index 4c77e05a..c7736efb 100644 --- a/crates/tinymemory-api/src/meta/mod.rs +++ b/crates/tinymemory-api/src/meta/mod.rs @@ -150,25 +150,32 @@ pub enum SourceKind { Folder, /// A single file. File, + /// A web page. + Link, + /// A GitHub repository. + Github, + /// An RSS or Atom feed. + Rss, + /// A Composio toolkit payload (Gmail, Slack, Notion, ...). + Composio, /// A host conversation. Conversation, /// Written by an agent directly; the default. #[default] Agent, - /// Imported from a legacy store, or ingested by hand from a connected app - /// or the web. - /// - /// The retired network kinds (`link`, `github`, `rss`, `composio`) decode - /// to this variant, so items stored under them stay readable. - #[serde(alias = "link", alias = "github", alias = "rss", alias = "composio")] + /// Imported from a legacy store. Import, } impl SourceKind { /// Every kind, in declaration order. - pub const ALL: [Self; 5] = [ + pub const ALL: [Self; 9] = [ Self::Folder, Self::File, + Self::Link, + Self::Github, + Self::Rss, + Self::Composio, Self::Conversation, Self::Agent, Self::Import, @@ -180,6 +187,10 @@ impl SourceKind { match self { Self::Folder => "folder", Self::File => "file", + Self::Link => "link", + Self::Github => "github", + Self::Rss => "rss", + Self::Composio => "composio", Self::Conversation => "conversation", Self::Agent => "agent", Self::Import => "import", diff --git a/crates/tinymemory-api/src/meta/mod_tests.rs b/crates/tinymemory-api/src/meta/mod_tests.rs index 349d9b29..def04bdf 100644 --- a/crates/tinymemory-api/src/meta/mod_tests.rs +++ b/crates/tinymemory-api/src/meta/mod_tests.rs @@ -11,7 +11,7 @@ fn a_default_meta_names_the_agent_as_its_source() { #[test] fn unset_fields_are_omitted_and_round_trip() { - let mut meta = MemoryMeta::from_source(SourceKind::Folder, Some("src_1".into())); + let mut meta = MemoryMeta::from_source(SourceKind::Github, Some("src_1".into())); meta.repo = Some("owner/name".into()); meta.tool_call = Some(ToolCallRef { name: "grep".into(), @@ -23,7 +23,7 @@ fn unset_fields_are_omitted_and_round_trip() { serde_json::json!({ "repo": "owner/name", "tool_call": { "name": "grep" }, - "source": { "kind": "folder", "id": "src_1" } + "source": { "kind": "github", "id": "src_1" } }) ); let back: MemoryMeta = serde_json::from_value(json).expect("deserialise"); @@ -43,16 +43,3 @@ fn source_kind_wire_strings_match_serde() { assert_eq!(json, serde_json::json!(kind.as_str())); } } - -#[test] -fn retired_network_source_kinds_decode_as_import() { - for retired in ["link", "github", "rss", "composio"] { - let kind: SourceKind = - serde_json::from_value(serde_json::json!(retired)).expect("legacy kind decodes"); - assert_eq!(kind, SourceKind::Import, "{retired}"); - } - assert_eq!( - serde_json::to_value(SourceKind::Import).expect("serialise"), - serde_json::json!("import") - ); -} diff --git a/crates/tinymemory-integrations/Cargo.toml b/crates/tinymemory-integrations/Cargo.toml index e58f8be4..69c651b7 100644 --- a/crates/tinymemory-integrations/Cargo.toml +++ b/crates/tinymemory-integrations/Cargo.toml @@ -36,9 +36,12 @@ tinymemory-tools = { path = "../tinymemory-tools", optional = true } # CortexDB speaks HTTP/JSON. `stream` is for `bytes_stream()`: response bodies # are read against a byte cap rather than buffered whole, because the endpoint # is operator-supplied and a broken or hostile one must not exhaust the host. +# The network source readers share the same client stack. reqwest = { version = "0.12", default-features = false, features = ["json", "rustls-tls", "stream"], optional = true } -# Timers for read-retry backoff and visibility polling (cortex). reqwest -# already requires a tokio runtime, so this adds no new runtime assumption. +# Timers for read-retry backoff and visibility polling (cortex); process +# spawning and DNS lookups for the GitHub reader and the SSRF resolver +# (sources-network). reqwest already requires a tokio runtime, so this adds no +# new runtime assumption. tokio = { version = "1", default-features = false, features = ["time"], optional = true } # `StreamExt` to read a capped body chunk by chunk. futures = { version = "0.3", optional = true } @@ -119,15 +122,18 @@ documents-office = ["documents", "dep:pdf-extract", "dep:calamine", "dep:quick-x # placed under the source type their format implies. brain = ["documents", "dep:tinymemory-tools"] # `sources`: readers that turn folders, files and conversations into -# `StoreItem`s. Links no HTTP stack. +# `StoreItem`s, and the Composio payload normalisers. Links no HTTP stack. sources = ["documents", "dep:schemars", "dep:regex", "dep:walkdir", "dep:chrono", "dep:log", "dep:tracing"] +# The readers that fetch over the network — GitHub, RSS, web pages — and +# `sources::fetch::fetch_url`, all behind the shared SSRF guard. +sources-network = ["sources", "dep:reqwest", "dep:futures", "dep:tokio", "tokio/process", "tokio/io-util", "tokio/net"] # `safety`: secret and PII scrubbing for a `StoreItem` before it is stored. safety = ["dep:regex", "dep:serde_json", "dep:log"] # `import`: reads a legacy v1 (embedded TinyCortex) workspace and migrates it # into any engine, resumably. legacy-import = ["dep:rusqlite", "dep:serde", "dep:serde_json", "dep:thiserror"] # Every integration. -full = ["cortex", "documents-office", "brain", "sources", "safety", "legacy-import"] +full = ["cortex", "documents-office", "brain", "sources-network", "safety", "legacy-import"] [[example]] name = "basic" diff --git a/crates/tinymemory-integrations/README.md b/crates/tinymemory-integrations/README.md index 32140e0c..6d02b39a 100644 --- a/crates/tinymemory-integrations/README.md +++ b/crates/tinymemory-integrations/README.md @@ -17,13 +17,14 @@ wants one integration enables one feature and links nothing else. | `cortex`, `registry`, `config` | `cortex` (default) | `CortexEngine` over two wires (`cortexdb`, `tinyhumans`); `list_engines`, `build_engine`, `EngineCredential`; `MemoryConfig` | [`src/cortex/README.md`](src/cortex/README.md) | | `documents` | `documents` | Format sniffing and conversion to markdown, producing `StoreItem::Document`. No I/O. | [`src/documents/README.md`](src/documents/README.md) | | `documents::OfficeConverter` | `documents-office` | PDF, DOCX, PPTX and XLSX to markdown, in process | (same) | -| `sources` | `sources` | Readers for folders, files and conversations; `collect_items`. Links no HTTP stack. | [`src/sources/README.md`](src/sources/README.md) | +| `sources` | `sources` | Readers for folders, files and conversations; Composio payload normalisers; `collect_items`. Links no HTTP stack. | [`src/sources/README.md`](src/sources/README.md) | +| `sources::fetch`, GitHub, RSS and web-page readers | `sources-network` | The network readers and `fetch_url`, all behind the SSRF guard | (same) | | `safety` | `safety` | Secret and PII scrubbing of a `StoreItem` before it is stored | [`src/safety/README.md`](src/safety/README.md) | | `import` | `legacy-import` | Reads a v1 (embedded TinyCortex) workspace and migrates it into any engine, resumably | [`src/import/README.md`](src/import/README.md) | -`full` turns on `cortex`, `documents-office`, `sources`, `safety` and +`full` turns on `cortex`, `documents-office`, `sources-network`, `safety` and `legacy-import`. Feature implications: `documents-office` implies `documents`; -`sources` implies `documents`. +`sources` implies `documents`; `sources-network` implies `sources`. ## Dependency weight per feature @@ -38,6 +39,7 @@ already needs): | `documents-office` | `pdf-extract`, `calamine`, `quick-xml`, `zip` (all pure Rust, no system libraries) | | `brain` | `tinymemory-tools` (for `BrainDocument`; it adds only `futures`, `chrono`, `log`) | | `sources` | `schemars`, `regex`, `walkdir`, `chrono`, `log`, `tracing` | +| `sources-network` | `reqwest`, `futures`, `tokio` with `process`, `io-util` and `net` (the GitHub reader runs `gh` and `git`; the SSRF resolver does DNS) | | `safety` | `regex`, `serde_json`, `log` | | `legacy-import` | `rusqlite` with bundled SQLite (compiles C; no system SQLite needed) | @@ -56,7 +58,8 @@ sources ──▶ documents ──▶ safety ──▶ engine.store ``` 1. **sources** lists a configured source and reads each entry. Local readers - hand raw bytes to the converter. `sources::collect_items` drives one source and collects per-item + hand raw bytes to the converter; network readers fetch through the SSRF + guard. `sources::collect_items` drives one source and collects per-item failures instead of aborting. 2. **documents** sniffs the format and converts bodies to markdown. A `ConverterChain` decides which converter handles which format; a host can diff --git a/crates/tinymemory-integrations/src/cortex/engine/fetch_tests.rs b/crates/tinymemory-integrations/src/cortex/engine/fetch_tests.rs index 7433917f..14ccdcd1 100644 --- a/crates/tinymemory-integrations/src/cortex/engine/fetch_tests.rs +++ b/crates/tinymemory-integrations/src/cortex/engine/fetch_tests.rs @@ -10,7 +10,7 @@ fn event(id: &str, item: &StoreItem) -> Value { } fn doc(text: &str, repo: Option<&str>) -> StoreItem { - let mut meta = MemoryMeta::from_source(SourceKind::Folder, None); + let mut meta = MemoryMeta::from_source(SourceKind::Github, None); meta.repo = repo.map(str::to_owned); StoreItem::document(text, meta) } diff --git a/crates/tinymemory-integrations/src/cortex/engine/refers_tests.rs b/crates/tinymemory-integrations/src/cortex/engine/refers_tests.rs index 48c87e69..687555b8 100644 --- a/crates/tinymemory-integrations/src/cortex/engine/refers_tests.rs +++ b/crates/tinymemory-integrations/src/cortex/engine/refers_tests.rs @@ -9,7 +9,7 @@ use tinymemory_api::{FetchMode, FetchRequest, MemoryEngine, MemoryMeta, SourceKi use crate::cortex::testing::{both, direct_double, direct_engine}; fn on_day(text: &str, day: u32) -> StoreItem { - let mut meta = MemoryMeta::from_source(SourceKind::Folder, None); + let mut meta = MemoryMeta::from_source(SourceKind::Github, None); // 20:00 UTC: already the next day in India, so the zone is exercised. meta.observed_at = Utc.with_ymd_and_hms(2026, 10, day, 20, 0, 0).single(); StoreItem::document(text, meta) diff --git a/crates/tinymemory-integrations/src/cortex/envelope/labels_tests.rs b/crates/tinymemory-integrations/src/cortex/envelope/labels_tests.rs index c6f9592a..12407ea8 100644 --- a/crates/tinymemory-integrations/src/cortex/envelope/labels_tests.rs +++ b/crates/tinymemory-integrations/src/cortex/envelope/labels_tests.rs @@ -14,13 +14,13 @@ fn a_digest_is_sixteen_hex_digits_and_stable() { #[test] fn an_item_carries_its_id_label_and_one_per_set_field() { - let mut meta = MemoryMeta::from_source(SourceKind::Folder, Some("src-1".into())); + let mut meta = MemoryMeta::from_source(SourceKind::Github, Some("src-1".into())); meta.repo = Some("a/b".into()); let labels = for_item("abc", &meta); assert_eq!(labels[0], item("abc")); assert!(labels.contains(&format!("tm:r:{}", digest("a/b")))); assert!(labels.contains(&format!("tm:s:{}", digest("src-1")))); - assert!(labels.contains(&format!("tm:k:{}", digest("folder")))); + assert!(labels.contains(&format!("tm:k:{}", digest("github")))); assert_eq!(labels.len(), 4); assert!(labels.iter().all(|l| l.len() <= 24 && !l.contains(','))); } @@ -37,7 +37,7 @@ fn a_filter_narrows_by_its_most_selective_labelled_field() { Some(vec![format!("tm:t:{}", digest("t1"))]) ); let by_sources = MetaFilter { - sources: vec![SourceKind::File, SourceKind::Import], + sources: vec![SourceKind::Rss, SourceKind::Link], ..MetaFilter::default() }; assert_eq!(narrowing(&by_sources).map(|l| l.len()), Some(2)); diff --git a/crates/tinymemory-integrations/src/documents/item/mod_tests.rs b/crates/tinymemory-integrations/src/documents/item/mod_tests.rs index 3b056592..bb696a90 100644 --- a/crates/tinymemory-integrations/src/documents/item/mod_tests.rs +++ b/crates/tinymemory-integrations/src/documents/item/mod_tests.rs @@ -26,7 +26,7 @@ async fn html_becomes_a_markdown_document_with_its_title_and_mime() { "Notes

Hi

Body.

", ) .with_mime("text/html"); - let meta = MemoryMeta::from_source(SourceKind::Import, Some("src_1".into())); + let meta = MemoryMeta::from_source(SourceKind::Link, Some("src_1".into())); let item = document_item(&chain, &document, meta).await.unwrap(); assert_eq!(item.kind(), ItemKind::Document); item.validate().unwrap(); @@ -35,7 +35,7 @@ async fn html_becomes_a_markdown_document_with_its_title_and_mime() { assert_eq!(title.as_deref(), Some("Notes")); assert_eq!(body, "# Hi\n\nBody."); assert_eq!(mime.as_deref(), Some("text/html")); - assert_eq!(meta.source.kind, SourceKind::Import); + assert_eq!(meta.source.kind, SourceKind::Link); assert_eq!(meta.source.id.as_deref(), Some("src_1")); assert_eq!(meta.language, None); } diff --git a/crates/tinymemory-integrations/src/lib.rs b/crates/tinymemory-integrations/src/lib.rs index 0bd5a576..f320a470 100644 --- a/crates/tinymemory-integrations/src/lib.rs +++ b/crates/tinymemory-integrations/src/lib.rs @@ -8,7 +8,7 @@ //! | [`cortex`], [`registry`], [`config`] | `cortex` (default) | The CortexDB engine over its two wires, and building one from configuration | //! | `documents` | `documents`, `documents-office` | Format sniffing and conversion to markdown, emitting `StoreItem::Document` | //! | `brain` | `brain` | Converting files into the brain documents `tinymemory_tools::Brain` ingests, by source type | -//! | `sources` | `sources` | Readers turning folders, files and conversations into `StoreItem`s | +//! | `sources` | `sources`, `sources-network` | Readers turning folders, files, links, GitHub, RSS, Composio payloads and conversations into `StoreItem`s | //! | `safety` | `safety` | Secret and PII scrubbing for a `StoreItem` before it is stored | //! | `import` | `legacy-import` | Migrating a legacy v1 (embedded TinyCortex) workspace into any engine | //! diff --git a/crates/tinymemory-integrations/src/sources/README.md b/crates/tinymemory-integrations/src/sources/README.md index 8450a005..dca6befa 100644 --- a/crates/tinymemory-integrations/src/sources/README.md +++ b/crates/tinymemory-integrations/src/sources/README.md @@ -1,7 +1,8 @@ # sources -`tinymemory_integrations::sources` (feature `sources`): -readers that turn a source into `StoreItem`s — a folder, a single file, or the +`tinymemory_integrations::sources` (features `sources` and `sources-network`): +readers that turn a source into `StoreItem`s — a folder, a single file, a web +page, a GitHub repository, an RSS feed, a Composio toolkit payload, or the host's local conversation threads. Conversion to markdown and language detection come from the sibling [`documents`](../documents/README.md) module. Architecture overview: @@ -17,7 +18,10 @@ checks it with `MemorySourceEntry::validate`, nothing more. | --- | --- | | `types` | the configuration a host persists: `MemorySourceEntry` keyed by `SourceKind`, its field rules, and the reader output types (`SourceItem`, `SourceContent`, `ContentType`) | | `readers` | `SourceReader` (list, read, read as a `StoreItem`) and one reader per kind, each in its own module directory; `local_file` holds the shared size-capped read and the path-containment guard | +| `fetch` | one URL into a `RawDocument` or a link item (`sources-network`); the RSS and web-page readers fetch through it with their own body caps | +| `fetch::ssrf` | the SSRF guard: scheme and host policy, one address classifier for literal and resolved addresses, a public-only DNS resolver, per-hop redirect checks, and a capped body reader | | `items` | reader output to `StoreItem`s with `MemoryMeta` filled per kind; `collect_items` drives a reader end to end | +| `composio` | toolkit normalisers (Gmail, Slack, GitHub, Linear, Notion, ClickUp), the `fields::pick_str` lookup they share, `normalise_payload` and `payload_items`. `readers::composio::ComposioReader` is only a placeholder reader | | `error` | the module `Error`, mapped onto `tinymemory_api::Error` | ## Kinds and metadata @@ -28,6 +32,10 @@ Every item's `meta.source` is `SourceRef { kind, id: Some(entry.id) }`. | --- | --- | --- | --- | | `folder` | `Folder` | document | `workspace`, `folder` (containing directory), `file_path`, `language`, `observed_at` (mtime), `mime` | | `file` | `File` | document | as `folder`; `file_item` reads a path with no configured source | +| `web_page` | `Link` | document | `url` | +| `github_repo` | `Github` | document | `repo` (`owner/name`), `commit` (commits), `url` (issues, PRs), `observed_at` | +| `rss_feed` | `Rss` | document | `url` (the entry's link), `observed_at` (published) | +| `composio` | `Composio` | document | `tags = [toolkit]`; payloads add `url`, `observed_at`, `thread_id`, `repo` | | `conversation` | `Conversation` | conversation | `workspace`, `thread_id`, `turns`, `observed_at` (last turn) | ## Folder selection @@ -42,10 +50,31 @@ the process working directory. ## Who decides when -`readers::reader_for` hands out the folder, file and conversation readers, all of -which read local state only. Scheduling stays with the host. +`readers::reader_for` hands out only the local readers (folder, file, +conversation), which are safe to drive on a timer. Network readers are +constructed explicitly, or through `reader_for_request` (feature +`sources-network`) for an explicit user request. Scheduling, credentials, OAuth and egress budgets stay with the host. + +## Fetching + +Every network fetch of a user-configured URL goes through `fetch` and its +SSRF guard. A hostname is checked as text (private and reserved IP literals, +`localhost`, `.local`/`.internal`, single-label names), its resolved addresses +are checked again by the client's resolver, which pins the connection to an +address it has vetted, and every redirect hop is re-checked. Bodies are read +against a cap while streaming: 32 MiB for `fetch_url`, 10 MiB for a web page, +5 MiB for a feed. The client sends the user agent `openhuman`, times out after +20 seconds, and allows only `http(s)`. An IPv6 literal URL is always refused, +even a public address (the bracketed host fails the IP parse and falls into the +single-label rule): fail closed. Failures are typed — `Invalid` for a refused or malformed +URL, `Unreachable`, `Upstream` for a failure status, `TooLarge`. + +Page titles and feed text are decoded with the `documents::html` helpers, so +named and numeric entities decode the same way everywhere. ## Features -- `sources` — the readers, `items` and `types`; implies `documents`. Links no - HTTP stack. +- `sources` — the local readers, `items`, `composio` and `types`; implies + `documents`. Links no HTTP stack. +- `sources-network` — adds the GitHub, RSS and web-page readers, `fetch`, and + the SSRF guard. diff --git a/crates/tinymemory-integrations/src/sources/composio/clickup/mod.rs b/crates/tinymemory-integrations/src/sources/composio/clickup/mod.rs new file mode 100644 index 00000000..b4ce64e5 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/composio/clickup/mod.rs @@ -0,0 +1,69 @@ +//! ClickUp host normalization helpers — result extraction and task-title and +//! timestamp extraction. +//! +//! ClickUp's REST API (and therefore Composio's wrapping of it) returns +//! task lists in a small handful of shapes depending on which endpoint +//! is called. The functions here walk the union of common shapes so the +//! provider doesn't have to branch per Composio envelope variant. + +use serde_json::Value; + +use super::fields::pick_str; + +/// Walk the Composio response envelope for ClickUp task list results. +/// +/// ClickUp's "filtered team tasks" endpoint returns `{ "tasks": [...] }` +/// at the top level; Composio re-wraps the upstream payload under +/// `data` or `data.data` depending on the action. We probe each shape +/// in order and return the first array we find. +pub fn extract_tasks(data: &Value) -> Vec { + let candidates = [ + data.pointer("/data/tasks"), + data.pointer("/tasks"), + data.pointer("/data/data/tasks"), + data.pointer("/data/results"), + data.pointer("/results"), + data.pointer("/data/items"), + data.pointer("/items"), + ]; + for cand in candidates.into_iter().flatten() { + if let Some(arr) = cand.as_array() { + return arr.clone(); + } + } + Vec::new() +} + +/// Extract a human-readable title from a ClickUp task object. +/// +/// ClickUp tasks store the name at `name` (or `data.name` after Composio +/// envelope wrapping). When the name is missing we fall back to the +/// task ID so chunks remain identifiable. +pub fn extract_task_name(task: &Value) -> Option { + pick_str(task, &["name", "data.name", "title", "data.title"]) +} + +/// Extract a stable cursor timestamp (milliseconds since epoch as a +/// string) from a ClickUp task object. +/// +/// The ClickUp API returns `date_updated` as a stringified epoch ms +/// (e.g. `"1733412345678"`); we keep it as a string so lexicographic +/// comparison against the stored cursor remains valid as long as the +/// length doesn't change (it won't until year 33658). +pub fn extract_task_updated(task: &Value) -> Option { + pick_str( + task, + &[ + "date_updated", + "data.date_updated", + "updated_at", + "data.updated_at", + "dateUpdated", + "data.dateUpdated", + ], + ) +} + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinymemory-integrations/src/sources/composio/clickup/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/clickup/mod_tests.rs new file mode 100644 index 00000000..5a520401 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/composio/clickup/mod_tests.rs @@ -0,0 +1,58 @@ +//! Tests for the ClickUp normaliser. + +use super::*; +use serde_json::json; + +#[test] +fn extract_tasks_from_data_tasks() { + let data = json!({ "data": { "tasks": [{"id": "t1"}] } }); + assert_eq!(extract_tasks(&data).len(), 1); +} + +#[test] +fn extract_tasks_from_top_level_tasks() { + let data = json!({ "tasks": [{"id": "a"}, {"id": "b"}] }); + assert_eq!(extract_tasks(&data).len(), 2); +} + +#[test] +fn extract_tasks_empty_when_missing() { + let data = json!({ "foo": "bar" }); + assert!(extract_tasks(&data).is_empty()); +} + +#[test] +fn extract_task_name_from_top_level() { + let task = json!({ "id": "t1", "name": "Build feature X" }); + assert_eq!(extract_task_name(&task), Some("Build feature X".into())); +} + +#[test] +fn extract_task_name_falls_back_to_data_name() { + let task = json!({ "data": { "name": "Wrapped" } }); + assert_eq!(extract_task_name(&task), Some("Wrapped".into())); +} + +#[test] +fn extract_task_name_none_when_missing() { + let task = json!({ "id": "t1" }); + assert!(extract_task_name(&task).is_none()); +} + +#[test] +fn extract_task_updated_handles_string_form() { + let task = json!({ "date_updated": "1733412345678" }); + assert_eq!( + extract_task_updated(&task), + Some("1733412345678".to_string()) + ); +} + +#[test] +fn extract_task_updated_handles_nested_data() { + let task = json!({ "data": { "dateUpdated": "1700000000000" } }); + assert_eq!( + extract_task_updated(&task), + Some("1700000000000".to_string()) + ); +} diff --git a/crates/tinymemory-integrations/src/sources/composio/documents/mod.rs b/crates/tinymemory-integrations/src/sources/composio/documents/mod.rs new file mode 100644 index 00000000..04753e4b --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/composio/documents/mod.rs @@ -0,0 +1,343 @@ +//! One Composio response in, documents and `StoreItem`s out. +//! +//! [`normalise_payload`] dispatches on the toolkit slug to the matching +//! normaliser and reads each record's title, body, link and timestamp. +//! Toolkits without a dedicated normaliser fall back to a generic walk that +//! keeps each record as fenced JSON, so a new toolkit is ingested (verbosely) +//! rather than dropped. [`payload_items`] wraps the documents as +//! `StoreItem::Document`s. + +use crate::documents::{DocumentFormat, markdown_from_text}; +use chrono::{DateTime, TimeZone, Utc}; +use serde_json::Value; +use tinymemory_api::{DocumentBody, MemoryMeta, SourceKind, StoreItem}; + +use super::fields::pick_str; +use super::{clickup, github, gmail_post_process, linear, notion}; + +/// One record of a Composio payload, normalised: an email, a message, an +/// issue, a task or a page. +#[derive(Debug, Clone, PartialEq)] +pub struct ComposioDocument { + /// The provider's id for the record, when it has one. + pub id: Option, + /// A human-readable title. + pub title: Option, + /// The record's text as markdown. Never empty. + pub body: String, + /// A link back to the record in the provider's UI. + pub url: Option, + /// When the record was last updated or sent. + pub observed_at: Option>, + /// The email or message thread the record belongs to. + pub thread_id: Option, + /// The repository a GitHub record belongs to, as `owner/name`. + pub repo: Option, +} + +impl ComposioDocument { + /// A record with only a body. + fn with_body(body: String) -> Self { + Self { + id: None, + title: None, + body, + url: None, + observed_at: None, + thread_id: None, + repo: None, + } + } + + /// Wrap this record as a [`StoreItem::Document`] from `toolkit`, read + /// through the connection or source `source_id`. + #[must_use] + pub fn into_store_item(self, toolkit: &str, source_id: &str) -> StoreItem { + let mut meta = MemoryMeta::from_source(SourceKind::Composio, Some(source_id.to_string())); + meta.url = self.url; + meta.observed_at = self.observed_at; + meta.thread_id = self.thread_id; + meta.repo = self.repo; + meta.tags = vec![toolkit.to_string()]; + StoreItem::Document { + title: self.title, + body: DocumentBody::Text(self.body), + mime: Some(DocumentFormat::Markdown.mime().to_string()), + meta, + } + } +} + +/// Normalise one Composio response from `toolkit` into documents. +/// +/// `data` is the action's response; for Gmail and Slack, run +/// [`gmail_post_process::post_process`] / [`super::slack_post_process::post_process`] +/// on it first so it carries the slim `messages[]` shape. Records with no text +/// at all are skipped. +#[must_use] +pub fn normalise_payload(toolkit: &str, data: &Value) -> Vec { + let documents: Vec = match toolkit.to_ascii_lowercase().as_str() { + "gmail" => array_at(data, &["/messages", "/data/messages"]) + .iter() + .filter_map(gmail_message) + .collect(), + "slack" => array_at(data, &["/messages", "/data/messages"]) + .iter() + .filter_map(slack_message) + .collect(), + "github" => github::extract_issues(data) + .iter() + .filter_map(github_issue) + .collect(), + "linear" => linear::extract_issues(data) + .iter() + .filter_map(linear_issue) + .collect(), + "notion" => notion_pages(data), + "clickup" => clickup::extract_tasks(data) + .iter() + .filter_map(clickup_task) + .collect(), + _ => generic_records(data), + }; + log::debug!( + "[memory_sources:composio] normalised toolkit={toolkit} documents={}", + documents.len() + ); + documents +} + +/// Normalise one Composio response and wrap every record as a +/// [`StoreItem::Document`] with `source.kind = Composio`, +/// `source.id = source_id` and `tags = [toolkit]`. +#[must_use] +pub fn payload_items(toolkit: &str, source_id: &str, data: &Value) -> Vec { + normalise_payload(toolkit, data) + .into_iter() + .map(|document| document.into_store_item(toolkit, source_id)) + .collect() +} + +/// The first array found at any of `pointers`. +fn array_at<'a>(data: &'a Value, pointers: &[&str]) -> &'a [Value] { + pointers + .iter() + .find_map(|pointer| data.pointer(pointer).and_then(Value::as_array)) + .map_or(&[], Vec::as_slice) +} + +/// A body as markdown: HTML (as sniffed) is converted, anything else kept. +fn to_markdown(text: &str) -> String { + let format = DocumentFormat::sniff(text.as_bytes(), None, None); + markdown_from_text(text, format).trim().to_string() +} + +/// Build a document from its parts, falling back to the title as the body; +/// `None` when there is no text at all. +fn document(title: Option, body: Option) -> Option { + let body = body + .map(|body| to_markdown(&body)) + .filter(|body| !body.is_empty()) + .or_else(|| title.clone())?; + let mut document = ComposioDocument::with_body(body); + document.title = title; + Some(document) +} + +/// Parse an ISO 8601 / RFC 3339 / RFC 2822 timestamp. +fn parse_time(text: &str) -> Option> { + gmail_post_process::parse_email_date(text) +} + +/// Parse an epoch-milliseconds string (ClickUp's `date_updated`). +fn parse_epoch_ms(text: &str) -> Option> { + let millis = text.trim().parse::().ok()?; + Utc.timestamp_millis_opt(millis).single() +} + +/// Parse a Slack `ts` (`"1712345678.123456"`, epoch seconds with a fraction). +fn parse_slack_ts(text: &str) -> Option> { + let (seconds, fraction) = text.split_once('.').unwrap_or((text, "0")); + let seconds = seconds.parse::().ok()?; + let micros = format!("{fraction:0<6}").get(..6)?.parse::().ok()?; + Utc.timestamp_opt(seconds, micros * 1_000).single() +} + +/// A string field, or a number rendered as a string. +fn scalar(value: &Value, key: &str) -> Option { + match value.get(key)? { + Value::String(text) if !text.trim().is_empty() => Some(text.trim().to_string()), + Value::Number(number) => Some(number.to_string()), + _ => None, + } +} + +/// A post-processed Gmail message: headers above the body. +fn gmail_message(message: &Value) -> Option { + let subject = pick_str(message, &["subject"]); + let markdown = pick_str(message, &["markdown", "messageText"]); + let mut header = String::new(); + for (label, key) in [("From", "from"), ("To", "to"), ("Date", "date")] { + if let Some(value) = pick_str(message, &[key]) { + header.push_str(&format!("{label}: {value}\n")); + } + } + let body = match (markdown, header.is_empty()) { + (Some(markdown), false) => Some(format!("{header}\n{markdown}")), + (Some(markdown), true) => Some(markdown), + (None, _) => None, + }; + let mut document = document(subject, body)?; + document.id = pick_str(message, &["id", "messageId"]); + document.thread_id = pick_str(message, &["threadId", "thread_id"]); + document.observed_at = pick_str(message, &["date"]).as_deref().and_then(parse_time); + Some(document) +} + +/// A post-processed Slack message. +fn slack_message(message: &Value) -> Option { + let user = pick_str(message, &["user"]); + let channel = pick_str(message, &["channel_id"]); + let title = match (&user, &channel) { + (Some(user), Some(channel)) => Some(format!("Slack message from {user} in {channel}")), + (Some(user), None) => Some(format!("Slack message from {user}")), + (None, _) => Some("Slack message".to_string()), + }; + let text = pick_str(message, &["text"])?; + let mut document = document(title, Some(text))?; + let ts = pick_str(message, &["ts"]); + document.id = ts.clone(); + document.url = pick_str(message, &["permalink"]); + document.thread_id = pick_str(message, &["thread_ts"]); + document.observed_at = ts.as_deref().and_then(parse_slack_ts); + Some(document) +} + +/// A GitHub issue or pull request from a search response. +fn github_issue(issue: &Value) -> Option { + let mut document = document( + github::extract_issue_title(issue), + pick_str(issue, &["body", "data.body"]), + )?; + document.id = github::extract_issue_id(issue); + document.url = pick_str(issue, &["html_url", "data.html_url"]); + document.repo = document.url.as_deref().and_then(github_repo); + document.observed_at = github::extract_issue_updated_at(issue) + .as_deref() + .and_then(parse_time); + Some(document) +} + +/// `owner/name` from a `https://github.com/owner/name/...` link. +fn github_repo(url: &str) -> Option { + let rest = url.split_once("github.com/")?.1; + let mut parts = rest.split('/'); + let owner = parts.next().filter(|part| !part.is_empty())?; + let name = parts.next().filter(|part| !part.is_empty())?; + Some(format!("{owner}/{name}")) +} + +/// A Linear issue. +fn linear_issue(issue: &Value) -> Option { + let mut document = document( + linear::extract_issue_title(issue), + pick_str(issue, &["description", "data.description"]), + )?; + document.id = pick_str(issue, &["identifier", "id", "data.identifier", "data.id"]); + document.url = pick_str(issue, &["url", "data.url"]); + document.observed_at = linear::extract_issue_updated(issue) + .as_deref() + .and_then(parse_time); + Some(document) +} + +/// Notion pages from a search response, or the one page a +/// `NOTION_GET_PAGE_MARKDOWN` response carries. +fn notion_pages(data: &Value) -> Vec { + let results = notion::extract_results(data); + if results.is_empty() { + return notion::extract_page_markdown(data) + .and_then(|markdown| { + let title = notion::extract_page_title(data); + let mut document = document(title, Some(markdown))?; + document.id = pick_str(data, &["id", "data.id", "page_id", "data.page_id"]); + document.url = pick_str(data, &["url", "data.url"]); + Some(document) + }) + .into_iter() + .collect(); + } + results + .iter() + .filter_map(|page| { + let mut document = document( + notion::extract_page_title(page), + notion::extract_page_markdown(page), + )?; + document.id = pick_str(page, &["id", "data.id"]); + document.url = pick_str(page, &["url", "data.url"]); + document.observed_at = pick_str(page, &["last_edited_time", "data.last_edited_time"]) + .as_deref() + .and_then(parse_time); + Some(document) + }) + .collect() +} + +/// A ClickUp task. +fn clickup_task(task: &Value) -> Option { + let mut document = document( + clickup::extract_task_name(task), + pick_str( + task, + &["markdown_description", "description", "text_content"], + ), + )?; + document.id = scalar(task, "id"); + document.url = pick_str(task, &["url", "data.url"]); + document.observed_at = clickup::extract_task_updated(task) + .as_deref() + .and_then(|text| parse_epoch_ms(text).or_else(|| parse_time(text))); + Some(document) +} + +/// Any other toolkit: each record in the first list found (or the whole +/// payload) as fenced JSON, titled and linked when it says how. +fn generic_records(data: &Value) -> Vec { + let records = array_at( + data, + &[ + "/data/items", + "/items", + "/data/results", + "/results", + "/data/data", + "/data", + ], + ); + let records: Vec<&Value> = if records.is_empty() { + vec![data] + } else { + records.iter().collect() + }; + records + .into_iter() + .filter(|record| !record.is_null()) + .filter_map(|record| { + let json = serde_json::to_string_pretty(record).ok()?; + let title = pick_str(record, &["title", "name", "subject"]); + let mut document = ComposioDocument::with_body(format!("```json\n{json}\n```")); + document.title = title; + document.id = scalar(record, "id"); + document.url = pick_str(record, &["url", "html_url", "permalink", "link"]); + document.observed_at = pick_str(record, &["updated_at", "updatedAt", "created_at"]) + .as_deref() + .and_then(parse_time); + Some(document) + }) + .collect() +} + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinymemory-integrations/src/sources/composio/documents/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/documents/mod_tests.rs new file mode 100644 index 00000000..4d689128 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/composio/documents/mod_tests.rs @@ -0,0 +1,197 @@ +//! Tests for turning Composio payloads into documents and `StoreItem`s. + +use super::*; +use serde_json::json; +use tinymemory_api::ItemKind; + +fn text_of(item: &StoreItem) -> (&Option, &str, &MemoryMeta) { + match item { + StoreItem::Document { + title, + body: DocumentBody::Text(text), + meta, + .. + } => (title, text.as_str(), meta), + other => panic!("expected a text document, got {other:?}"), + } +} + +#[test] +fn a_github_search_becomes_items_with_url_repo_time_and_toolkit_tag() { + let data = json!({ + "data": { "items": [{ + "id": 42, + "title": "Fix the build", + "body": "The build is **red**.", + "html_url": "https://github.com/acme/widgets/issues/7", + "updated_at": "2024-05-21T15:30:00Z" + }]} + }); + let items = payload_items("github", "conn_gh", &data); + assert_eq!(items.len(), 1); + let item = &items[0]; + assert_eq!(item.kind(), ItemKind::Document); + item.validate().unwrap(); + + let (title, body, meta) = text_of(item); + assert_eq!( + title.as_deref(), + Some("GitHub: acme/widgets#7: Fix the build") + ); + assert_eq!(body, "The build is **red**."); + assert_eq!(meta.source.kind, SourceKind::Composio); + assert_eq!(meta.source.id.as_deref(), Some("conn_gh")); + assert_eq!(meta.tags, vec!["github".to_string()]); + assert_eq!( + meta.url.as_deref(), + Some("https://github.com/acme/widgets/issues/7") + ); + assert_eq!(meta.repo.as_deref(), Some("acme/widgets")); + assert_eq!( + meta.observed_at, + Some(Utc.with_ymd_and_hms(2024, 5, 21, 15, 30, 0).unwrap()) + ); +} + +#[test] +fn post_processed_gmail_messages_carry_headers_thread_and_date() { + let data = json!({ "messages": [{ + "id": "m1", + "threadId": "t1", + "subject": "Lunch?", + "from": "Ann ", + "to": "me@example.com", + "date": "Tue, 21 May 2024 12:00:00 +0000", + "markdown": "Tacos at noon." + }]}); + let documents = normalise_payload("gmail", &data); + assert_eq!(documents.len(), 1); + let document = &documents[0]; + assert_eq!(document.title.as_deref(), Some("Lunch?")); + assert!(document.body.starts_with("From: Ann \n")); + assert!(document.body.ends_with("Tacos at noon.")); + assert_eq!(document.thread_id.as_deref(), Some("t1")); + assert_eq!(document.id.as_deref(), Some("m1")); + assert_eq!( + document.observed_at, + Some(Utc.with_ymd_and_hms(2024, 5, 21, 12, 0, 0).unwrap()) + ); +} + +#[test] +fn slack_messages_take_permalink_and_ts_time() { + let data = json!({ "messages": [{ + "ts": "1716300000.000100", + "user": "U1", + "channel_id": "C1", + "text": "deploy done", + "permalink": "https://acme.slack.com/archives/C1/p1716300000000100" + }]}); + let items = payload_items("slack", "src_slack", &data); + let (title, body, meta) = text_of(&items[0]); + assert_eq!(title.as_deref(), Some("Slack message from U1 in C1")); + assert_eq!(body, "deploy done"); + assert_eq!( + meta.url.as_deref(), + Some("https://acme.slack.com/archives/C1/p1716300000000100") + ); + assert_eq!( + meta.observed_at.map(|at| at.timestamp()), + Some(1_716_300_000) + ); + assert_eq!(meta.tags, vec!["slack".to_string()]); +} + +#[test] +fn linear_notion_and_clickup_records_are_normalised() { + let linear = json!({ "nodes": [{ + "identifier": "ENG-1", + "title": "Ship v2", + "description": "All of it.", + "url": "https://linear.app/acme/issue/ENG-1", + "updatedAt": "2024-01-02T03:04:05Z" + }]}); + let documents = normalise_payload("linear", &linear); + assert_eq!(documents[0].id.as_deref(), Some("ENG-1")); + assert_eq!(documents[0].body, "All of it."); + assert!(documents[0].observed_at.is_some()); + + let notion = json!({ "results": [{ + "id": "p1", + "url": "https://notion.so/p1", + "last_edited_time": "2024-01-02T03:04:05.000Z", + "properties": { "Name": { "type": "title", "title": [{ "plain_text": "Roadmap" }] } } + }]}); + let documents = normalise_payload("notion", ¬ion); + assert_eq!(documents[0].title.as_deref(), Some("Roadmap")); + assert_eq!( + documents[0].body, "Roadmap", + "a page without body keeps its title" + ); + assert_eq!(documents[0].url.as_deref(), Some("https://notion.so/p1")); + + let page_markdown = json!({ "data": { "markdown": "# Plan\n\nDo it." }, "id": "p2" }); + let documents = normalise_payload("notion", &page_markdown); + assert_eq!(documents.len(), 1); + assert_eq!(documents[0].body, "# Plan\n\nDo it."); + + let clickup = json!({ "tasks": [{ + "id": "t9", + "name": "Write docs", + "description": "

The README

", + "url": "https://app.clickup.com/t/t9", + "date_updated": "1700000000000" + }]}); + let documents = normalise_payload("clickup", &clickup); + assert_eq!(documents[0].id.as_deref(), Some("t9")); + assert_eq!( + documents[0].observed_at.map(|at| at.timestamp()), + Some(1_700_000_000) + ); +} + +#[test] +fn html_bodies_are_converted_to_markdown() { + let data = json!({ "nodes": [{ + "title": "Page", + "description": "

Hi

There

" + }]}); + let documents = normalise_payload("linear", &data); + assert_eq!(documents[0].body, "## Hi\n\nThere"); +} + +#[test] +fn records_without_any_text_are_skipped() { + let data = json!({ "messages": [{ "ts": "1.0", "text": " " }] }); + assert!(normalise_payload("slack", &data).is_empty()); + assert!(normalise_payload("github", &json!({ "items": [{}] })).is_empty()); +} + +#[test] +fn an_unknown_toolkit_keeps_each_record_as_fenced_json() { + let data = json!({ "data": { "items": [ + { "id": 1, "name": "Alpha", "url": "https://example.com/1", "updated_at": "2024-01-01T00:00:00Z" }, + { "id": 2 } + ]}}); + let items = payload_items("hubspot", "conn_hs", &data); + assert_eq!(items.len(), 2); + let (title, body, meta) = text_of(&items[0]); + assert_eq!(title.as_deref(), Some("Alpha")); + assert!(body.starts_with("```json\n")); + assert!(body.contains("\"name\": \"Alpha\"")); + assert_eq!(meta.url.as_deref(), Some("https://example.com/1")); + assert_eq!(meta.tags, vec!["hubspot".to_string()]); + + let single = normalise_payload("hubspot", &json!({ "ok": true })); + assert_eq!(single.len(), 1); +} + +#[test] +fn slack_timestamps_parse_with_and_without_a_fraction() { + assert_eq!(parse_slack_ts("10").map(|at| at.timestamp()), Some(10)); + assert_eq!( + parse_slack_ts("10.5").map(|at| at.timestamp_subsec_micros()), + Some(500_000) + ); + assert_eq!(parse_slack_ts("x"), None); +} diff --git a/crates/tinymemory-integrations/src/sources/composio/fields/mod.rs b/crates/tinymemory-integrations/src/sources/composio/fields/mod.rs new file mode 100644 index 00000000..4c257085 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/composio/fields/mod.rs @@ -0,0 +1,44 @@ +//! Field lookup shared by the Composio normalisers: pull a string out of a +//! payload by trying several dotted paths, because Composio wraps the same +//! upstream field at different depths depending on the action and version. + +/// Walk a JSON object using a list of dotted-path candidates and return the +/// first non-empty **string** match, trimmed. +/// +/// Each path is split on `.` and followed with `Value::get`, so it only +/// descends through objects — it never indexes into an array. A leaf that is +/// not a string (a number, a bool) is rejected rather than coerced, so a +/// payload whose `id` is `42` rather than `"42"` yields `None` here. That +/// differs from the private `scalar` lookup in the `documents` mapping, which +/// renders numbers; the normalisers were written against the +/// reject-non-strings behaviour and `pick_str_rejects_non_string_values` pins +/// it. +pub fn pick_str(value: &serde_json::Value, paths: &[&str]) -> Option { + for path in paths { + let mut cur = value; + let mut ok = true; + for segment in path.split('.') { + match cur.get(segment) { + Some(next) => cur = next, + None => { + ok = false; + break; + } + } + } + if !ok { + continue; + } + if let Some(s) = cur.as_str() { + let trimmed = s.trim(); + if !trimmed.is_empty() { + return Some(trimmed.to_string()); + } + } + } + None +} + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinymemory-integrations/src/sources/composio/fields/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/fields/mod_tests.rs new file mode 100644 index 00000000..330ec404 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/composio/fields/mod_tests.rs @@ -0,0 +1,37 @@ +//! Tests for the shared Composio field lookup. + +use super::*; +use serde_json::json; + +#[test] +fn pick_str_finds_first_non_empty_match() { + let v = json!({"data": {"user": {"name": "Ada", "email": "ada@example.com"}}}); + assert_eq!( + pick_str(&v, &["data.user.name", "data.user.email"]), + Some("Ada".into()) + ); + assert_eq!( + pick_str(&v, &["data.missing", "data.user.email"]), + Some("ada@example.com".into()) + ); + assert_eq!(pick_str(&v, &["nope.nope"]), None); +} + +#[test] +fn pick_str_respects_path_order() { + let v = json!({"a": "first", "b": "second"}); + assert_eq!(pick_str(&v, &["a", "b"]), Some("first".into())); + assert_eq!(pick_str(&v, &["b", "a"]), Some("second".into())); +} + +/// The drift guard for the behaviour documented on [`pick_str`]. If this +/// ever starts returning `Some("42")`, the normalisers' emitted ids have +/// changed. +#[test] +fn pick_str_rejects_non_string_values() { + let v = json!({"count": 42, "flag": true, "empty": "", "whitespace": " "}); + assert_eq!(pick_str(&v, &["count"]), None); + assert_eq!(pick_str(&v, &["flag"]), None); + assert_eq!(pick_str(&v, &["empty"]), None); + assert_eq!(pick_str(&v, &["whitespace"]), None); +} diff --git a/crates/tinymemory-integrations/src/sources/composio/github/mod.rs b/crates/tinymemory-integrations/src/sources/composio/github/mod.rs new file mode 100644 index 00000000..12e7a615 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/composio/github/mod.rs @@ -0,0 +1,116 @@ +//! GitHub host normalization helpers — issue extraction and issue id, title and +//! timestamp helpers. +//! +//! GitHub's REST API (proxied through Composio) returns search results in a +//! small number of shapes. The functions here +//! walk the union of common Composio envelope variants so the provider stays +//! clean and branch-free. + +use serde_json::Value; + +use super::fields::pick_str; + +/// Walk the Composio response envelope for GitHub search issue results. +/// +/// `GITHUB_SEARCH_ISSUES_AND_PULL_REQUESTS` wraps GitHub's `GET /search/issues` response, which +/// returns `{"total_count": N, "items": [...]}`. Composio may re-wrap this under +/// `data` or `data.data`; we probe each shape in order. +pub fn extract_issues(data: &Value) -> Vec { + let candidates = [ + data.pointer("/data/items"), + data.pointer("/items"), + data.pointer("/data/data/items"), + data.pointer("/data/results"), + data.pointer("/results"), + ]; + for cand in candidates.into_iter().flatten() { + if let Some(arr) = cand.as_array() { + return arr.clone(); + } + } + Vec::new() +} + +/// Extract a stable, globally unique identifier for a GitHub issue or PR. +/// +/// GitHub's internal `id` field is a large integer unique across all issues +/// and PRs on github.com. We convert it to a string for use as a sync key. +/// Falls back to composing from `html_url` path if `id` is absent. +pub fn extract_issue_id(issue: &Value) -> Option { + // Primary: numeric internal GitHub ID. + if let Some(id) = issue.get("id").or_else(|| issue.pointer("/data/id")) { + if let Some(n) = id.as_u64() { + return Some(n.to_string()); + } + if let Some(s) = id.as_str() { + let trimmed = s.trim(); + if !trimmed.is_empty() { + return Some(trimmed.to_string()); + } + } + } + // Fallback: parse owner/repo/number from html_url path segments. + // URL shape: https://github.com/{owner}/{repo}/issues/{number} + if let Some(url) = pick_str(issue, &["html_url", "data.html_url", "url", "data.url"]) + && let Some(slug) = github_url_to_slug(&url) + { + return Some(slug); + } + None +} + +/// Build a human-readable document title for a GitHub issue/PR. +/// +/// Format: `GitHub: {owner}/{repo}#{number}: {title}`. +/// Falls back to just the title or a placeholder when fields are missing. +pub fn extract_issue_title(issue: &Value) -> Option { + let title = pick_str(issue, &["title", "data.title"])?; + + // Best-effort: extract owner/repo#N from html_url for the prefix. + let prefix = pick_str(issue, &["html_url", "data.html_url"]) + .and_then(|url| github_url_to_slug(&url)) + .unwrap_or_default(); + + if prefix.is_empty() { + Some(title) + } else { + Some(format!("GitHub: {prefix}: {title}")) + } +} + +/// Parse `https://github.com/{owner}/{repo}/issues/{number}` (or `/pull/`) +/// into `"{owner}/{repo}#{number}"`. Returns `None` for unrecognised shapes. +fn github_url_to_slug(url: &str) -> Option { + let segs: Vec<&str> = url.trim_end_matches('/').split('/').collect(); + // Minimum: ["https:", "", "github.com", owner, repo, "issues", number] + if segs.len() >= 7 { + let number = segs[segs.len() - 1]; + let _kind = segs[segs.len() - 2]; // "issues" or "pull" — ignored + let repo = segs[segs.len() - 3]; + let owner = segs[segs.len() - 4]; + if !owner.is_empty() && !repo.is_empty() && !number.is_empty() { + return Some(format!("{owner}/{repo}#{number}")); + } + } + None +} + +/// Extract the `updated_at` ISO 8601 timestamp from a GitHub issue. +/// +/// GitHub returns `updated_at` as `"2024-05-21T15:30:00Z"`. ISO 8601 strings +/// sort lexicographically, so we use them directly as the sync cursor. +pub fn extract_issue_updated_at(issue: &Value) -> Option { + pick_str( + issue, + &[ + "updated_at", + "data.updated_at", + "updatedAt", + "data.updatedAt", + ], + ) +} + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinymemory-integrations/src/sources/composio/github/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/github/mod_tests.rs new file mode 100644 index 00000000..0b7322f1 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/composio/github/mod_tests.rs @@ -0,0 +1,97 @@ +//! Tests for the GitHub normaliser. + +use super::*; +use serde_json::json; + +#[test] +fn extract_issues_from_data_items() { + let data = json!({ "data": { "items": [{"id": 1}] } }); + assert_eq!(extract_issues(&data).len(), 1); +} + +#[test] +fn extract_issues_from_top_level_items() { + let data = json!({ "items": [{"id": 1}, {"id": 2}] }); + assert_eq!(extract_issues(&data).len(), 2); +} + +#[test] +fn extract_issues_empty_when_missing() { + let data = json!({ "foo": "bar" }); + assert!(extract_issues(&data).is_empty()); +} + +#[test] +fn extract_issue_id_from_numeric_field() { + let issue = json!({ "id": 123456789u64, "title": "Fix bug" }); + assert_eq!(extract_issue_id(&issue), Some("123456789".to_string())); +} + +#[test] +fn extract_issue_id_from_wrapped_data() { + let issue = json!({ "data": { "id": 99u64 } }); + assert_eq!(extract_issue_id(&issue), Some("99".to_string())); +} + +#[test] +fn extract_issue_id_falls_back_to_html_url() { + let issue = json!({ + "html_url": "https://github.com/owner/repo/issues/42" + }); + assert_eq!(extract_issue_id(&issue), Some("owner/repo#42".to_string())); +} + +#[test] +fn extract_issue_id_none_when_missing() { + let issue = json!({ "title": "No ID here" }); + assert!(extract_issue_id(&issue).is_none()); +} + +#[test] +fn extract_issue_title_builds_prefixed_title() { + let issue = json!({ + "id": 1u64, + "title": "Fix race condition", + "html_url": "https://github.com/acme/core/issues/99" + }); + assert_eq!( + extract_issue_title(&issue), + Some("GitHub: acme/core#99: Fix race condition".to_string()) + ); +} + +#[test] +fn extract_issue_title_returns_raw_title_when_no_url() { + let issue = json!({ "title": "Bare title" }); + assert_eq!(extract_issue_title(&issue), Some("Bare title".to_string())); +} + +#[test] +fn extract_issue_title_none_when_missing() { + let issue = json!({ "id": 1u64 }); + assert!(extract_issue_title(&issue).is_none()); +} + +#[test] +fn extract_issue_updated_at_from_top_level() { + let issue = json!({ "updated_at": "2024-05-21T15:30:00Z" }); + assert_eq!( + extract_issue_updated_at(&issue), + Some("2024-05-21T15:30:00Z".to_string()) + ); +} + +#[test] +fn extract_issue_updated_at_from_data_wrapper() { + let issue = json!({ "data": { "updated_at": "2023-01-01T00:00:00Z" } }); + assert_eq!( + extract_issue_updated_at(&issue), + Some("2023-01-01T00:00:00Z".to_string()) + ); +} + +#[test] +fn extract_issue_updated_at_none_when_missing() { + let issue = json!({ "id": 1u64 }); + assert!(extract_issue_updated_at(&issue).is_none()); +} diff --git a/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod.rs b/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod.rs new file mode 100644 index 00000000..c6c78ae7 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod.rs @@ -0,0 +1,498 @@ +//! Gmail-specific post-processing of Composio action responses. +//! +//! The upstream `GMAIL_FETCH_EMAILS` payload is extremely verbose +//! (full MIME tree under `payload.parts[]`, 50+ `Received:` headers, +//! display-layer noise the model never uses). This module rewrites +//! it into a slim envelope per message: +//! +//! ```json +//! { +//! "messages": [ +//! { +//! "id": "…", +//! "threadId": "…", +//! "subject": "…", +//! "from": "…", +//! "to": "…", +//! "date": "…", +//! "labels": ["INBOX", "UNREAD"], +//! "markdown": "…body…", +//! "attachments": [ { "filename": "...", "mimeType": "..." } ] +//! } +//! ], +//! "nextPageToken": "…", +//! "resultSizeEstimate": 201 +//! } +//! ``` +//! +//! ## Body source +//! +//! Composio's backend ships a +//! `markdownFormatted` field on the response envelope — one string +//! per tool call, pre-rendered with HTML stripped, URLs shortened, +//! footers removed, whitespace normalised. We split it per message +//! along `\n---\n` boundaries (with `## ` heading fallbacks) and +//! pin each slice to the corresponding entry in `messages[]` via +//! [`apply_response_level_markdown`]. The reshape's +//! `extract_markdown_body` then prefers that pinned field over +//! falling back to the upstream `messageText`. +//! +//! No in-house HTML→markdown conversion lives here anymore — the +//! backend does the cleaning. If `markdownFormatted` is absent for +//! a given response we fall through to whatever plain text the +//! upstream provided in `messageText`. +//! +//! Callers that need the raw Composio shape can pass `raw_html: +//! true` (or `rawHtml: true`) in the action arguments — this +//! short-circuits the reshape entirely. +//! +//! Only `GMAIL_FETCH_EMAILS` is reshaped today; other Gmail action +//! responses are passed through unchanged. When we add envelopes for +//! more slugs they should live in this file, branched from +//! [`post_process`]. + +use serde_json::{Map, Value, json}; + +/// Entry point a host calls on each Gmail action response (the slug names +/// the action) before handing it to `normalise_payload`. +/// +/// Dispatches on the Composio action slug. Unknown Gmail slugs fall +/// through to a no-op. +pub fn post_process(slug: &str, arguments: Option<&Value>, data: &mut Value) { + if is_raw_html_flag_set(arguments) { + tracing::debug!( + slug, + "[composio:gmail][post-process] raw_html flag set, passing through" + ); + return; + } + if slug == "GMAIL_FETCH_EMAILS" { + reshape_fetch_emails(data) + } +} + +/// Stash per-message slices of the response-level `markdownFormatted` +/// onto the corresponding entries inside `data.messages[]`. +/// +/// The Composio backend (tinyhumansai/backend#683) ships ONE +/// `markdownFormatted` string per tool call covering all messages — +/// already URL-shortened, footer-stripped, and whitespace-normalised. +/// To get per-email files in the raw archive we split that string +/// along section boundaries (`## ` headings or `---` rules) and pin +/// each slice to the message at the same index. `extract_markdown_body` +/// then prefers `msg.markdownFormatted` over re-decoding the MIME +/// tree. +/// +/// **Must be called BEFORE [`post_process`]** because `post_process` +/// reshapes `data` into the slim envelope; once `messages[]` carries +/// our slim shape the upstream message ordering is already locked in +/// but we may have lost original ordering signals if any. +/// +/// No-op when the slice count doesn't match `messages.len()` — we +/// can't safely align segments to messages without an exact match, +/// so we let `extract_markdown_body` fall through to its MIME path. +pub fn apply_response_level_markdown(data: &mut Value, top_md: &str) { + let trimmed = top_md.trim(); + if trimmed.is_empty() { + return; + } + // Presence is checked immutably first, then fetched mutably. The original + // form re-fetched with `unwrap()` after a mutable probe, which is sound but + // relies on the reader to see why; this crate lints against `unwrap`, and the + // immutable probe expresses the same reasoning to the compiler. + let container = if data.get("messages").is_some() { + data + } else if data.get("data").and_then(Value::as_object).is_some() { + match data.get_mut("data") { + Some(inner) => inner, + None => return, + } + } else { + tracing::debug!( + "[composio:gmail][post-process] apply_response_level_markdown: \ + no messages container in response — skipping" + ); + return; + }; + let Some(messages) = container.get_mut("messages").and_then(|v| v.as_array_mut()) else { + return; + }; + let count = messages.len(); + if count == 0 { + return; + } + // Clone hints out of the messages array so the slice borrows + // don't conflict with the upcoming `messages.iter_mut()` mutation. + let hints: Vec = messages.clone(); + let Some(slices) = split_response_markdown_per_message_with_hint(trimmed, count, Some(&hints)) + else { + tracing::debug!( + messages = count, + md_len = trimmed.len(), + "[composio:gmail][post-process] could not split response-level markdownFormatted \ + into {count} slices — falling back to per-message MIME decode" + ); + return; + }; + for (msg, slice) in messages.iter_mut().zip(slices) { + if let Some(obj) = msg.as_object_mut() { + obj.insert("markdownFormatted".to_string(), Value::String(slice)); + } + } + tracing::debug!( + messages = count, + "[composio:gmail][post-process] stashed per-message markdownFormatted slices" + ); +} + +/// Split a top-level `markdownFormatted` string into per-message +/// segments. Returns `Some(slices)` only when the split yields +/// exactly `expected_count` entries — otherwise the format isn't one +/// of the patterns we know about and we let the caller fall back. +/// +/// Primary boundary is the `\n---\n` horizontal rule the backend +/// emits between messages (confirmed against real +/// `GMAIL_FETCH_EMAILS` output). H2/H3 headings are kept as +/// fallbacks for older renderings. The preamble (`# Inbox (N +/// messages)`-style intro, if present) is dropped — we accept +/// either `expected` parts (no preamble) or `expected + 1` +/// (preamble + N messages). +/// +/// `messages_hint` is the slim message array from the same response +/// — when present we use the per-message `subject` field to verify +/// each segment really does belong to the message at the same index. +/// Mismatches force a fallback so we never write a wrong-message body +/// to the raw archive. +/// +/// The hint is what makes the split reliable: the blob's own section headings +/// are backend-rendered and have changed shape between versions, so matching +/// on them alone silently mis-attributed bodies. +pub fn split_response_markdown_per_message_with_hint( + md: &str, + expected_count: usize, + messages_hint: Option<&[Value]>, +) -> Option> { + if expected_count == 0 { + return None; + } + if expected_count == 1 { + return Some(vec![md.to_string()]); + } + + // Boundary patterns to try, in priority order. `\n---\n` is the + // confirmed marker; the heading variants stay as belt-and-braces + // for older / variant backend renderings. + let candidates: &[(&str, &str)] = &[ + ("\n---\n", "---\n"), + ("\n\n## ", "## "), + ("\n\n### ", "### "), + ("\n\n# ", "# "), + ("\n***\n", "***\n"), + ]; + + for (sep, prefix) in candidates { + let parts: Vec<&str> = md.split(sep).collect(); + let (drop_preamble, prepend_first) = if parts.len() == expected_count { + (false, false) // no preamble; first segment had no prefix + } else if parts.len() == expected_count + 1 { + (true, true) // preamble dropped; every kept segment had a prefix + } else { + continue; + }; + let segments: Vec = parts + .into_iter() + .skip(if drop_preamble { 1 } else { 0 }) + .enumerate() + .map(|(i, s)| { + if i == 0 && !prepend_first { + s.to_string() + } else { + format!("{prefix}{s}") + } + }) + .collect(); + + // Validate alignment against the JSON message array: every + // segment whose corresponding message has a non-empty subject + // must mention that subject somewhere in its body. If a single + // pair fails, we treat the split as unreliable and try the + // next pattern. Empty / null subjects skip validation (e.g. + // notification mails where the subject is ""). + if let Some(hints) = messages_hint + && !validate_segments_against_hints(&segments, hints) + { + tracing::debug!( + expected = expected_count, + sep = sep, + "[composio:gmail][post-process] split candidate failed subject check" + ); + continue; + } + return Some(segments); + } + None +} + +/// True if every (segment, message) pair where the message has a +/// non-empty subject contains that subject somewhere in the segment +/// (case-insensitive substring match — a defensive heuristic, not a +/// strict equality check, since the backend may format subjects +/// inside markdown links or with surrounding decoration). +fn validate_segments_against_hints(segments: &[String], hints: &[Value]) -> bool { + if segments.len() != hints.len() { + return false; + } + for (seg, hint) in segments.iter().zip(hints.iter()) { + let subject = hint + .get("subject") + .and_then(|v| v.as_str()) + .unwrap_or("") + .trim(); + if subject.is_empty() { + continue; + } + if !seg + .to_ascii_lowercase() + .contains(&subject.to_ascii_lowercase()) + { + return false; + } + } + true +} + +/// Returns true when the caller explicitly set `raw_html: true` (or the +/// camelCase `rawHtml: true`) in the `arguments` object. +fn is_raw_html_flag_set(arguments: Option<&Value>) -> bool { + let Some(obj) = arguments.and_then(|v| v.as_object()) else { + return false; + }; + obj.get("raw_html") + .or_else(|| obj.get("rawHtml")) + .and_then(|v| v.as_bool()) + .unwrap_or(false) +} + +/// Rewrite a `GMAIL_FETCH_EMAILS` `data` object in place into the slim +/// envelope documented at the module level. +/// +/// The Composio response can be shaped either as `{ messages, nextPageToken, ... }` +/// directly, or wrapped one level deeper under `{ data: { messages: … } }` +/// depending on backend version; we handle both. +fn reshape_fetch_emails(data: &mut Value) { + // Unwrap an optional `data:` envelope so downstream logic only has + // to deal with one shape. + let container = if data.get("messages").is_some() { + data + } else if data.get("data").and_then(Value::as_object).is_some() { + match data.get_mut("data") { + Some(inner) => inner, + None => return, + } + } else { + return; + }; + + let Some(obj) = container.as_object_mut() else { + return; + }; + + let raw_messages = obj + .remove("messages") + .and_then(|v| match v { + Value::Array(arr) => Some(arr), + _ => None, + }) + .unwrap_or_default(); + let next_page_token = obj.remove("nextPageToken").unwrap_or(Value::Null); + let result_size_estimate = obj.remove("resultSizeEstimate").unwrap_or(Value::Null); + + let messages: Vec = raw_messages.into_iter().map(reshape_message).collect(); + + let mut envelope = Map::new(); + envelope.insert("messages".into(), Value::Array(messages)); + if !next_page_token.is_null() { + envelope.insert("nextPageToken".into(), next_page_token); + } + if !result_size_estimate.is_null() { + envelope.insert("resultSizeEstimate".into(), result_size_estimate); + } + + *container = Value::Object(envelope); +} + +/// Parse an RFC 3339 or RFC 2822 date string into a UTC `DateTime`. +pub fn parse_email_date(date_str: &str) -> Option> { + date_str + .parse::>() + .or_else(|_| { + chrono::DateTime::parse_from_rfc2822(date_str).map(|d| d.with_timezone(&chrono::Utc)) + }) + .ok() +} + +const EMAIL_LOCAL_TIME_FMT: &str = "%Y-%m-%d %I:%M %p %:z"; + +/// Format a UTC `DateTime` in the given timezone. Returns `None` when the +/// formatted result is identical to the UTC rendering (no-op for UTC hosts). +pub fn format_at_tz( + utc: chrono::DateTime, + tz: &Tz, +) -> Option +where + Tz::Offset: std::fmt::Display, +{ + let local_dt = utc.with_timezone(tz); + let formatted = local_dt.format(EMAIL_LOCAL_TIME_FMT).to_string(); + + let utc_formatted = utc.format(EMAIL_LOCAL_TIME_FMT).to_string(); + if formatted == utc_formatted { + return None; + } + Some(formatted) +} + +/// Convert a UTC email timestamp string to a human-readable local-time string. +/// +/// Accepts RFC 3339 (`"2026-05-31T10:33:00Z"`) or RFC 2822 +/// (`"Sat, 31 May 2026 10:33:00 +0000"`) input. Returns a formatted string +/// in the host's local timezone, e.g. `"2026-05-31 05:33 AM -05:00"`, +/// so the agent can present local times without UTC arithmetic. +/// +/// The raw `date` field is always preserved alongside this field so +/// internal sorting, deduplication, and debugging remain UTC-based. +/// +/// Returns `None` when the input cannot be parsed or the output format +/// would be identical to the UTC input (no-op for UTC hosts). +pub fn format_email_local_time(date_str: &str) -> Option { + let utc = parse_email_date(date_str)?; + format_at_tz(utc, &chrono::Local) +} + +/// Map one raw Composio message object to its slim counterpart. +/// +/// Body source picked by [`extract_markdown_body`]: +/// 1. The per-message `markdownFormatted` slice pinned by +/// [`apply_response_level_markdown`] (preferred — backend-rendered). +/// 2. The upstream `messageText` plaintext (fallback). +/// 3. Empty string. +fn reshape_message(raw: Value) -> Value { + let Value::Object(obj) = raw else { + return raw; + }; + + let id = obj.get("messageId").cloned().unwrap_or(Value::Null); + let thread_id = obj.get("threadId").cloned().unwrap_or(Value::Null); + let subject = obj.get("subject").cloned().unwrap_or(Value::Null); + let sender = obj.get("sender").cloned().unwrap_or(Value::Null); + let to = obj.get("to").cloned().unwrap_or(Value::Null); + let date = obj + .get("messageTimestamp") + .cloned() + .or_else(|| pick_header(&obj, "Date")) + .unwrap_or(Value::Null); + let labels = obj + .get("labelIds") + .cloned() + .unwrap_or_else(|| Value::Array(Vec::new())); + let list_unsubscribe = pick_header(&obj, "List-Unsubscribe").unwrap_or(Value::Null); + + let markdown = extract_markdown_body(&obj); + let attachments = extract_attachments(&obj); + + // Compute a local-time representation of the UTC `date` so the agent + // presents times in the user's timezone rather than quoting raw UTC. + let date_local = date.as_str().and_then(format_email_local_time); + + let mut out = Map::new(); + out.insert("id".into(), id); + out.insert("threadId".into(), thread_id); + out.insert("subject".into(), subject); + out.insert("from".into(), sender); + out.insert("to".into(), to); + out.insert("date".into(), date); + if let Some(local) = date_local { + out.insert("date_local".into(), Value::String(local)); + } + out.insert("labels".into(), labels); + if !list_unsubscribe.is_null() { + out.insert("list_unsubscribe".into(), list_unsubscribe); + } + out.insert("markdown".into(), Value::String(markdown)); + if !attachments.is_empty() { + out.insert("attachments".into(), Value::Array(attachments)); + } + Value::Object(out) +} + +/// Find a header value by (case-insensitive) name in the Composio +/// `payload.headers[]` array. Returns `Some(Value::String)` on hit. +fn pick_header(msg: &Map, name: &str) -> Option { + let headers = msg.get("payload")?.get("headers")?.as_array()?; + for h in headers { + let hn = h.get("name").and_then(|v| v.as_str()).unwrap_or(""); + if hn.eq_ignore_ascii_case(name) + && let Some(v) = h.get("value").and_then(|v| v.as_str()) + { + return Some(Value::String(v.to_string())); + } + } + None +} + +/// Pick a body for the slim envelope. +/// +/// We trust the Composio backend's pre-rendered `markdownFormatted` +/// (set per-message by [`apply_response_level_markdown`] from the +/// response-level field). When that's absent we fall back to the +/// upstream's plain-text `messageText` verbatim — no in-house +/// HTML→markdown decoding lives here anymore. The backend already +/// strips HTML, shortens URLs, and normalises whitespace; running +/// our own pipeline on top duplicated work and corrupted some +/// renderings. +fn extract_markdown_body(msg: &Map) -> String { + if let Some(formatted) = msg + .get("markdownFormatted") + .or_else(|| msg.get("markdown_formatted")) + .and_then(|v| v.as_str()) + .map(str::trim) + .filter(|s| !s.is_empty()) + { + return formatted.to_string(); + } + if let Some(text) = msg + .get("messageText") + .and_then(|v| v.as_str()) + .map(str::trim) + .filter(|s| !s.is_empty()) + { + return text.to_string(); + } + String::new() +} + +/// Pull a minimal attachments descriptor from the Composio +/// `attachmentList` array. +fn extract_attachments(msg: &Map) -> Vec { + if let Some(list) = msg.get("attachmentList").and_then(|v| v.as_array()) { + return list + .iter() + .filter_map(|a| { + let filename = a.get("filename").and_then(|v| v.as_str())?; + if filename.is_empty() { + return None; + } + let mime = a + .get("mimeType") + .and_then(|v| v.as_str()) + .unwrap_or_default(); + Some(json!({ "filename": filename, "mimeType": mime })) + }) + .collect(); + } + Vec::new() +} + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod_tests.rs new file mode 100644 index 00000000..dd1b44af --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod_tests.rs @@ -0,0 +1,356 @@ +//! Tests for the Gmail post-processor. + +use super::*; +use serde_json::json; + +fn fixture_with_backend_markdown() -> Value { + json!({ + "messages": [ + { + "messageId": "m1", + "threadId": "t1", + "subject": "Hello", + "sender": "a@x.com", + "to": "b@y.com", + "messageTimestamp": "2026-04-17T12:00:00Z", + "labelIds": ["INBOX", "UNREAD"], + // Pre-rendered slice (set by `apply_response_level_markdown` + // in production; inline here for the reshape test). + "markdownFormatted": "# Hello\n\nbody copy", + "messageText": "fallback should not be used", + "display_url": "ignore-me", + "preview": { "body": "Hi plain", "subject": "Hello" }, + "attachmentList": [ + { "filename": "report.pdf", "mimeType": "application/pdf", "size": 12345 }, + { "filename": "", "mimeType": "text/html" } + ], + "payload": {} + } + ], + "nextPageToken": "tok-1", + "resultSizeEstimate": 42 + }) +} + +#[test] +fn reshape_emits_slim_envelope() { + let mut v = fixture_with_backend_markdown(); + post_process("GMAIL_FETCH_EMAILS", None, &mut v); + + assert_eq!(v["nextPageToken"], "tok-1"); + assert_eq!(v["resultSizeEstimate"], 42); + + let msgs = v["messages"].as_array().unwrap(); + assert_eq!(msgs.len(), 1); + let m = &msgs[0]; + + assert_eq!(m["id"], "m1"); + assert_eq!(m["threadId"], "t1"); + assert_eq!(m["subject"], "Hello"); + assert_eq!(m["from"], "a@x.com"); + assert_eq!(m["to"], "b@y.com"); + assert_eq!(m["date"], "2026-04-17T12:00:00Z"); + assert_eq!(m["labels"], json!(["INBOX", "UNREAD"])); + + let md = m["markdown"].as_str().unwrap(); + assert_eq!(md, "# Hello\n\nbody copy"); + + // Noise fields removed. + assert!(m.get("display_url").is_none()); + assert!(m.get("preview").is_none()); + assert!(m.get("payload").is_none()); + assert!(m.get("messageText").is_none()); + + // Attachments: empty filename entry is filtered. + let atts = m["attachments"].as_array().unwrap(); + assert_eq!(atts.len(), 1); + assert_eq!(atts[0]["filename"], "report.pdf"); + assert_eq!(atts[0]["mimeType"], "application/pdf"); +} + +#[test] +fn raw_html_flag_passes_through_unchanged() { + let mut v = fixture_with_backend_markdown(); + let original = v.clone(); + let args = json!({ "raw_html": true }); + post_process("GMAIL_FETCH_EMAILS", Some(&args), &mut v); + assert_eq!( + v, original, + "raw_html=true must preserve the Composio shape" + ); +} + +#[test] +fn camel_case_raw_html_also_recognized() { + let mut v = fixture_with_backend_markdown(); + let original = v.clone(); + let args = json!({ "rawHtml": true }); + post_process("GMAIL_FETCH_EMAILS", Some(&args), &mut v); + assert_eq!(v, original); +} + +#[test] +fn falls_back_to_message_text_when_no_backend_markdown() { + let mut v = json!({ + "messages": [{ + "messageId": "m1", + "threadId": "t1", + "subject": "s", + "sender": "a@x.com", + "to": "b@y.com", + "messageTimestamp": "2026-04-17", + "labelIds": [], + "messageText": " plain body text ", + "payload": {} + }], + "nextPageToken": null + }); + post_process("GMAIL_FETCH_EMAILS", None, &mut v); + let md = v["messages"][0]["markdown"].as_str().unwrap(); + assert_eq!(md, "plain body text"); + assert!(v.get("nextPageToken").is_none(), "null tokens dropped"); +} + +#[test] +fn unwraps_data_envelope() { + let mut v = json!({ + "data": { + "messages": [{ + "messageId": "m1", + "threadId": "t1", + "subject": "s", + "sender": "a@x.com", + "to": "b@y.com", + "messageTimestamp": "2026-04-17", + "labelIds": [], + "messageText": "body", + "payload": {} + }] + } + }); + post_process("GMAIL_FETCH_EMAILS", None, &mut v); + // Reshape writes into `data` in place. + let msgs = v["data"]["messages"].as_array().unwrap(); + assert_eq!(msgs.len(), 1); + assert_eq!(msgs[0]["markdown"], "body"); +} + +#[test] +fn non_fetch_slug_is_noop() { + let mut v = json!({ "messages": [{ "messageId": "m1", "messageText": "x" }] }); + let original = v.clone(); + post_process("GMAIL_SEND_EMAIL", None, &mut v); + assert_eq!(v, original); +} + +#[test] +fn prefers_backend_markdown_formatted_when_present() { + // Composio backend (tinyhumansai/backend#683 +) ships + // `markdownFormatted` already URL-shortened + footer-stripped + // per message (after `apply_response_level_markdown` slices the + // response-level field). When present, our post-processor must + // use it verbatim instead of falling back to `messageText`. + let mut v = json!({ + "messages": [{ + "messageId": "m1", + "threadId": "t1", + "subject": "s", + "sender": "a@x.com", + "to": "b@y.com", + "messageTimestamp": "2026-04-17", + "labelIds": [], + "markdownFormatted": "# Already nice\n\nShort URL: https://gh.io/abc", + "messageText": "fallback should not be used", + "payload": {} + }] + }); + post_process("GMAIL_FETCH_EMAILS", None, &mut v); + let md = v["messages"][0]["markdown"].as_str().unwrap(); + assert_eq!(md, "# Already nice\n\nShort URL: https://gh.io/abc"); +} + +#[test] +fn empty_markdown_formatted_falls_through_to_message_text() { + let mut v = json!({ + "messages": [{ + "messageId": "m1", + "threadId": "t1", + "subject": "s", + "sender": "a@x.com", + "to": "b@y.com", + "messageTimestamp": "2026-04-17", + "labelIds": [], + "markdownFormatted": " \n \n", + "messageText": "real body", + "payload": {} + }] + }); + post_process("GMAIL_FETCH_EMAILS", None, &mut v); + let md = v["messages"][0]["markdown"].as_str().unwrap(); + assert!(md.contains("real body")); +} + +// ── split_response_markdown_per_message_with_hint ─────────────────────── + +#[test] +fn split_response_markdown_uses_horizontal_rule_marker() { + // The confirmed backend marker is `\n---\n`. Three messages → + // expect three slices when there's no preamble. + let md = "## Alice's update\n\nbody A with https://gh.io/abc\n---\n## Bob's reply\n\nbody B\n---\n## Carol\n\nbody C"; + let slices = super::split_response_markdown_per_message_with_hint(md, 3, None).unwrap(); + assert_eq!(slices.len(), 3); + assert!(slices[0].contains("Alice's update")); + assert!(slices[1].contains("Bob's reply")); + assert!(slices[2].contains("Carol")); + // The `---\n` prefix is preserved on every-but-the-first segment + // so the section break survives the round-trip. + assert!(slices[1].starts_with("---\n")); + assert!(slices[2].starts_with("---\n")); +} + +#[test] +fn split_response_markdown_drops_preamble() { + // When a preamble like `# Inbox` precedes the first marker, we + // see N+1 parts after split — the preamble must be dropped. + let md = "# Inbox (2 messages)\n---\n## A\n\nbody A\n---\n## B\n\nbody B"; + let slices = super::split_response_markdown_per_message_with_hint(md, 2, None).unwrap(); + assert_eq!(slices.len(), 2); + assert!(slices[0].contains("body A")); + assert!(slices[1].contains("body B")); + // Both segments should carry the prefix when preamble was dropped. + assert!(slices[0].starts_with("---\n")); + assert!(slices[1].starts_with("---\n")); +} + +#[test] +fn split_response_markdown_falls_back_to_h2_marker() { + // No `---` rules — backend used h2 headings as boundaries. + let md = "## Alice\n\nbody A\n\n## Bob\n\nbody B"; + let slices = super::split_response_markdown_per_message_with_hint(md, 2, None).unwrap(); + assert_eq!(slices.len(), 2); + assert!(slices[0].contains("body A")); + assert!(slices[1].contains("body B")); +} + +#[test] +fn split_response_markdown_returns_none_on_count_mismatch() { + let md = "## only one section here"; + assert!(super::split_response_markdown_per_message_with_hint(md, 3, None).is_none()); +} + +#[test] +fn split_response_markdown_single_message_returns_whole_input() { + let md = "## solo\n\nthe whole body"; + let slices = super::split_response_markdown_per_message_with_hint(md, 1, None).unwrap(); + assert_eq!(slices, vec![md.to_string()]); +} + +#[test] +fn split_with_hint_rejects_when_subjects_dont_match() { + let md = "## Foo\nbody1\n---\n## Bar\nbody2"; + let hints = vec![ + json!({"subject": "Completely different subject A"}), + json!({"subject": "Completely different subject B"}), + ]; + let out = super::split_response_markdown_per_message_with_hint(md, 2, Some(&hints)); + assert!(out.is_none(), "subject mismatch must force fallback"); +} + +#[test] +fn split_with_hint_accepts_when_subjects_match() { + let md = "## Welcome to Gmail\nbody1\n---\n## Your invoice\nbody2"; + let hints = vec![ + json!({"subject": "Welcome to Gmail"}), + json!({"subject": "Your invoice"}), + ]; + let slices = super::split_response_markdown_per_message_with_hint(md, 2, Some(&hints)).unwrap(); + assert_eq!(slices.len(), 2); + assert!(slices[0].contains("Welcome to Gmail")); + assert!(slices[1].contains("Your invoice")); +} + +#[test] +fn split_with_hint_skips_messages_with_blank_subject() { + let md = "## A\nbody1\n---\n## B\nbody2"; + let hints = vec![json!({"subject": "A"}), json!({"subject": ""})]; + let slices = super::split_response_markdown_per_message_with_hint(md, 2, Some(&hints)).unwrap(); + assert_eq!(slices.len(), 2); +} + +// ── format_email_local_time ────────────────────────────────────────────────── + +#[test] +fn format_email_local_time_returns_none_for_unparseable_date() { + assert!(super::format_email_local_time("not-a-date").is_none()); + assert!(super::format_email_local_time("").is_none()); +} + +#[test] +fn format_email_local_time_preserves_utc_raw_date_in_reshape() { + let mut v = json!({ + "messages": [{ + "messageId": "m1", + "threadId": "t1", + "subject": "Test", + "sender": "a@example.com", + "to": "b@example.com", + "messageTimestamp": "2026-05-31T10:33:00Z", + "labelIds": [], + "messageText": "body", + "payload": {} + }] + }); + post_process("GMAIL_FETCH_EMAILS", None, &mut v); + let msg = &v["messages"][0]; + assert_eq!(msg["date"], "2026-05-31T10:33:00Z"); +} + +#[test] +fn parse_email_date_accepts_rfc3339_and_rfc2822() { + assert!(super::parse_email_date("2026-05-31T10:33:00Z").is_some()); + assert!(super::parse_email_date("Sun, 31 May 2026 10:33:00 +0000").is_some()); + assert!(super::parse_email_date("not-a-date").is_none()); +} + +#[test] +fn format_at_tz_deterministic_with_fixed_offset() { + use chrono::FixedOffset; + + let utc = super::parse_email_date("2026-05-31T10:33:00Z").unwrap(); + + let est = FixedOffset::west_opt(5 * 3600).unwrap(); + let result = super::format_at_tz(utc, &est).unwrap(); + assert_eq!(result, "2026-05-31 05:33 AM -05:00"); + + let ist = FixedOffset::east_opt(5 * 3600 + 1800).unwrap(); + let result = super::format_at_tz(utc, &ist).unwrap(); + assert_eq!(result, "2026-05-31 04:03 PM +05:30"); +} + +#[test] +fn format_at_tz_returns_none_for_utc() { + let utc = super::parse_email_date("2026-05-31T10:33:00Z").unwrap(); + let utc_tz = chrono::FixedOffset::east_opt(0).unwrap(); + assert!(super::format_at_tz(utc, &utc_tz).is_none()); +} + +#[test] +fn apply_response_level_markdown_stashes_per_message_field() { + let mut data = json!({ + "messages": [ + {"messageId": "m1", "subject": "Hello"}, + {"messageId": "m2", "subject": "World"}, + ] + }); + let top_md = "## Hello\nbody A — link https://gh.io/abc\n---\n## World\nbody B"; + super::apply_response_level_markdown(&mut data, top_md); + let m1 = data["messages"][0]["markdownFormatted"].as_str().unwrap(); + let m2 = data["messages"][1]["markdownFormatted"].as_str().unwrap(); + assert!(m1.contains("Hello")); + assert!( + m1.contains("https://gh.io/abc"), + "shortened URL must survive" + ); + assert!(m2.contains("World")); + assert!(!m1.contains("World"), "no cross-message bleed"); +} diff --git a/crates/tinymemory-integrations/src/sources/composio/linear/mod.rs b/crates/tinymemory-integrations/src/sources/composio/linear/mod.rs new file mode 100644 index 00000000..ea52092a --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/composio/linear/mod.rs @@ -0,0 +1,79 @@ +//! Linear host normalization helpers — result extraction and issue-title and +//! timestamp extraction. +//! +//! Linear's GraphQL API (and therefore Composio's wrapping of it) returns +//! connection-style lists (`{ nodes: [...], pageInfo: {...} }`) at the top +//! level or nested under `data`. The functions here walk the union of +//! common shapes so the provider does not have to branch per Composio +//! envelope variant. + +use serde_json::Value; + +use super::fields::pick_str; + +/// Walk the Composio response envelope for Linear issue list results. +/// +/// Linear's list endpoints return `{ nodes: [...] }` or +/// `{ issues: { nodes: [...] } }` shapes; Composio may re-wrap the +/// upstream payload under `data` or `data.data`. We probe each shape +/// in order and return the first array we find. +pub fn extract_issues(data: &Value) -> Vec { + let candidates = [ + data.pointer("/data/nodes"), + data.pointer("/nodes"), + data.pointer("/data/issues/nodes"), + data.pointer("/issues/nodes"), + data.pointer("/data/data/nodes"), + data.pointer("/data/data/issues/nodes"), + data.pointer("/data/results"), + data.pointer("/results"), + data.pointer("/data/items"), + data.pointer("/items"), + ]; + for cand in candidates.into_iter().flatten() { + if let Some(arr) = cand.as_array() { + return arr.clone(); + } + } + Vec::new() +} + +/// Extract a human-readable title from a Linear issue object. +/// +/// Linear issues store the name at `title` (or `data.title` after +/// Composio envelope wrapping). Falls back to `name` / `identifier` +/// so the chunk remains identifiable even for unusual response shapes. +pub fn extract_issue_title(issue: &Value) -> Option { + pick_str( + issue, + &[ + "title", + "data.title", + "name", + "data.name", + "identifier", + "data.identifier", + ], + ) +} + +/// Extract a stable cursor timestamp from a Linear issue object. +/// +/// Linear uses ISO-8601 strings for timestamps (`updatedAt`). We keep +/// the value as a string so lexicographic comparison against the stored +/// cursor is valid. +pub fn extract_issue_updated(issue: &Value) -> Option { + pick_str( + issue, + &[ + "updatedAt", + "data.updatedAt", + "updated_at", + "data.updated_at", + ], + ) +} + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinymemory-integrations/src/sources/composio/linear/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/linear/mod_tests.rs new file mode 100644 index 00000000..c0acff63 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/composio/linear/mod_tests.rs @@ -0,0 +1,93 @@ +//! Tests for the Linear normaliser. + +use super::*; +use serde_json::json; + +// ── extract_issues ─────────────────────────────────────────────── + +#[test] +fn extract_issues_from_data_nodes() { + let data = json!({ "data": { "nodes": [{"id": "i1"}, {"id": "i2"}] } }); + assert_eq!(extract_issues(&data).len(), 2); +} + +#[test] +fn extract_issues_from_top_level_nodes() { + let data = json!({ "nodes": [{"id": "i3"}] }); + assert_eq!(extract_issues(&data).len(), 1); +} + +#[test] +fn extract_issues_from_data_issues_nodes() { + let data = + json!({ "data": { "issues": { "nodes": [{"id": "i4"}, {"id": "i5"}, {"id": "i6"}] } } }); + assert_eq!(extract_issues(&data).len(), 3); +} + +#[test] +fn extract_issues_from_top_level_issues_nodes() { + let data = json!({ "issues": { "nodes": [{"id": "i7"}] } }); + assert_eq!(extract_issues(&data).len(), 1); +} + +#[test] +fn extract_issues_from_doubly_nested_issues_nodes() { + let data = + json!({ "data": { "data": { "issues": { "nodes": [{"id": "i8"}, {"id": "i9"}] } } } }); + assert_eq!(extract_issues(&data).len(), 2); +} + +#[test] +fn extract_issues_from_results() { + let data = json!({ "results": [{"id": "i7"}] }); + assert_eq!(extract_issues(&data).len(), 1); +} + +#[test] +fn extract_issues_empty_when_missing() { + let data = json!({ "foo": "bar" }); + assert!(extract_issues(&data).is_empty()); +} + +// ── extract_issue_title ────────────────────────────────────────── + +#[test] +fn extract_issue_title_from_title_field() { + let issue = json!({ "id": "i1", "title": "Fix the login bug" }); + assert_eq!( + extract_issue_title(&issue), + Some("Fix the login bug".into()) + ); +} + +#[test] +fn extract_issue_title_falls_back_to_wrapped_data() { + let issue = json!({ "data": { "title": "Wrapped issue" } }); + assert_eq!(extract_issue_title(&issue), Some("Wrapped issue".into())); +} + +#[test] +fn extract_issue_title_falls_back_to_identifier() { + let issue = json!({ "identifier": "ENG-42" }); + assert_eq!(extract_issue_title(&issue), Some("ENG-42".into())); +} + +// ── extract_issue_updated ──────────────────────────────────────── + +#[test] +fn extract_issue_updated_from_updated_at() { + let issue = json!({ "updatedAt": "2026-03-01T12:00:00.000Z" }); + assert_eq!( + extract_issue_updated(&issue), + Some("2026-03-01T12:00:00.000Z".to_string()) + ); +} + +#[test] +fn extract_issue_updated_falls_back_to_snake_case() { + let issue = json!({ "data": { "updated_at": "2026-01-15T08:30:00.000Z" } }); + assert_eq!( + extract_issue_updated(&issue), + Some("2026-01-15T08:30:00.000Z".to_string()) + ); +} diff --git a/crates/tinymemory-integrations/src/sources/composio/mod.rs b/crates/tinymemory-integrations/src/sources/composio/mod.rs new file mode 100644 index 00000000..4df32030 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/composio/mod.rs @@ -0,0 +1,35 @@ +//! Composio toolkit payloads: normalisers and the mapping to `StoreItem`s. +//! +//! A host runs Composio actions with its own credentials and hands the raw +//! responses here. Two layers turn them into memory: +//! +//! 1. **Normalisers**, one module per toolkit, are pure +//! `serde_json::Value` transforms: they walk Composio's envelope variants +//! and pull out the tasks, issues, pages or messages +//! ([`clickup`], [`github`], [`linear`], [`notion`]), or rewrite a verbose +//! response into a slim one in place ([`gmail_post_process`], +//! [`slack_post_process`]). [`fields`] holds the path lookup they share. +//! 2. [`normalise_payload`] turns one (post-processed) response into +//! [`ComposioDocument`]s, and [`payload_items`] turns those into +//! [`StoreItem::Document`](tinymemory_api::StoreItem::Document)s with +//! `source.kind = Composio`, `source.id` = the connection or source id, +//! `url` and `observed_at` where the payload has them, and +//! `tags = [toolkit]`. +//! +//! Nothing here holds a credential, opens a socket or decides when to sync. +//! +//! One caveat on "pure": [`gmail_post_process::format_email_local_time`] +//! renders in `chrono::Local`, so it reads the host's timezone. The raw UTC +//! fields are preserved alongside, so ordering and identity stay UTC-based. + +pub mod clickup; +pub mod fields; +pub mod github; +pub mod gmail_post_process; +pub mod linear; +pub mod notion; +pub mod slack_post_process; + +mod documents; + +pub use documents::{ComposioDocument, normalise_payload, payload_items}; diff --git a/crates/tinymemory-integrations/src/sources/composio/notion/mod.rs b/crates/tinymemory-integrations/src/sources/composio/notion/mod.rs new file mode 100644 index 00000000..00c1df50 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/composio/notion/mod.rs @@ -0,0 +1,88 @@ +//! Notion host normalization helpers — result extraction, page markdown and +//! page title extraction. + +use serde_json::Value; + +use super::fields::pick_str; + +/// Walk the Composio response envelope for Notion page results. +pub fn extract_results(data: &Value) -> Vec { + let candidates = [ + data.pointer("/data/results"), + data.pointer("/results"), + data.pointer("/data/data/results"), + data.pointer("/data/items"), + data.pointer("/items"), + ]; + for cand in candidates.into_iter().flatten() { + if let Some(arr) = cand.as_array() { + return arr.clone(); + } + } + Vec::new() +} + +/// Extract the rendered page body markdown from a `NOTION_GET_PAGE_MARKDOWN` +/// response. Composio wraps action output in varying envelope shapes, so we +/// try the common locations tolerantly and return the first non-empty string. +/// Returns `None` if no markdown field is found (caller falls back to the +/// metadata-only body and logs the raw shape for diagnosis). +pub fn extract_page_markdown(data: &Value) -> Option { + const PATHS: &[&str] = &[ + "/markdown", + "/data/markdown", + "/data/response_data/markdown", + "/response_data/markdown", + "/data/content", + "/content", + "/data/markdown_content", + "/markdown_content", + "/text", + "/data/text", + ]; + for p in PATHS { + if let Some(s) = data.pointer(p).and_then(Value::as_str) + && !s.trim().is_empty() + { + return Some(s.to_string()); + } + } + None +} + +/// Try to extract a human-readable title from a Notion page object. +/// +/// Notion pages store the title in `properties.title` or +/// `properties.Name.title[0].plain_text`. We try several shapes. +pub fn extract_page_title(page: &Value) -> Option { + // Try the common `properties.title.title[0].plain_text` shape. + let props = page + .get("properties") + .or_else(|| page.get("data")?.get("properties")); + if let Some(props) = props { + // Walk all properties looking for a "title" type field. + if let Some(obj) = props.as_object() { + for (_key, val) in obj { + if val.get("type").and_then(Value::as_str) == Some("title") + && let Some(arr) = val.get("title").and_then(Value::as_array) + { + let text: String = arr + .iter() + .filter_map(|t| t.get("plain_text").and_then(Value::as_str)) + .collect::>() + .join(""); + if !text.is_empty() { + return Some(text); + } + } + } + } + } + + // Fallback: top-level "title" field (some Composio shapes). + pick_str(page, &["title", "data.title", "name", "data.name"]) +} + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinymemory-integrations/src/sources/composio/notion/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/notion/mod_tests.rs new file mode 100644 index 00000000..ab2e79bc --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/composio/notion/mod_tests.rs @@ -0,0 +1,111 @@ +//! Tests for the Notion normaliser. + +use super::*; +use serde_json::json; + +#[test] +fn extract_results_from_data_results() { + let data = json!({"data": {"results": [{"id": "page1"}]}}); + let results = extract_results(&data); + assert_eq!(results.len(), 1); +} + +#[test] +fn extract_page_markdown_reads_top_level_field() { + // Matches the live GET_PAGE_MARKDOWN envelope observed empirically: + // {id, markdown, object, request_id, truncated, unknown_block_ids}. + let data = json!({ + "id": "p1", + "markdown": "# Heading\n\nbody text", + "object": "page", + "truncated": false, + }); + assert_eq!( + extract_page_markdown(&data).as_deref(), + Some("# Heading\n\nbody text") + ); +} + +#[test] +fn extract_page_markdown_reads_nested_envelope() { + let data = json!({ "data": { "markdown": "nested body" } }); + assert_eq!(extract_page_markdown(&data).as_deref(), Some("nested body")); +} + +#[test] +fn extract_page_markdown_none_for_empty_or_missing() { + // Empty markdown (a DB row with no body blocks) → None → metadata-only. + assert_eq!(extract_page_markdown(&json!({ "markdown": "" })), None); + assert_eq!(extract_page_markdown(&json!({ "markdown": " " })), None); + // No markdown field at all → None. + assert_eq!(extract_page_markdown(&json!({ "id": "p1" })), None); +} + +#[test] +fn extract_results_from_top_level() { + let data = json!({"results": [{"id": "a"}, {"id": "b"}]}); + let results = extract_results(&data); + assert_eq!(results.len(), 2); +} + +#[test] +fn extract_results_from_data_items() { + let data = json!({"data": {"items": [{"id": "x"}]}}); + let results = extract_results(&data); + assert_eq!(results.len(), 1); +} + +#[test] +fn extract_results_empty_when_no_match() { + let data = json!({"foo": "bar"}); + assert!(extract_results(&data).is_empty()); +} + +#[test] +fn extract_page_title_from_properties_title_type() { + let page = json!({ + "properties": { + "Name": { + "type": "title", + "title": [{"plain_text": "Hello"}, {"plain_text": " World"}] + } + } + }); + assert_eq!(extract_page_title(&page), Some("Hello World".into())); +} + +#[test] +fn extract_page_title_from_nested_data_properties() { + let page = json!({ + "data": { + "properties": { + "Title": { + "type": "title", + "title": [{"plain_text": "My Page"}] + } + } + } + }); + assert_eq!(extract_page_title(&page), Some("My Page".into())); +} + +#[test] +fn extract_page_title_fallback_to_top_level_title() { + let page = json!({"title": "Fallback Title"}); + assert_eq!(extract_page_title(&page), Some("Fallback Title".into())); +} + +#[test] +fn extract_page_title_none_when_empty() { + let page = json!({"properties": {"Name": {"type": "title", "title": []}}}); + // Empty title array means no text + assert!( + extract_page_title(&page).is_none() || extract_page_title(&page) == Some(String::new()) + ); +} + +#[test] +fn extract_page_title_none_when_no_title_field() { + let page = json!({"id": "123"}); + assert!(extract_page_title(&page).is_none()); +} diff --git a/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod.rs b/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod.rs new file mode 100644 index 00000000..764b885a --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod.rs @@ -0,0 +1,324 @@ +//! Slack-specific post-processing of Composio action responses. +//! +//! Composio's Slack responses are verbose API envelopes. This module +//! rewrites each supported action's response into a slim, stable shape +//! that the ingest pipeline and enrichers can consume without walking +//! Composio's unstable nested envelopes. +//! +//! ## Supported slugs +//! +//! - `SLACK_FETCH_CONVERSATION_HISTORY` — reshapes into top-level +//! `messages[]` with `{ ts, user, text, thread_ts, channel_id }`. +//! Empty-text messages are dropped. `channel_id` is absent here (it's +//! in the request, not the response); the caller injects it via the +//! enricher in the host's `SlackSyncPipeline`. +//! +//! - `SLACK_LIST_CONVERSATIONS` — reshapes into top-level `channels[]` +//! with `{ id, name, is_private }` per channel. Entries with an empty +//! id are dropped. +//! +//! - `SLACK_SEARCH_MESSAGES` — reshapes `messages.matches[]` (possibly +//! nested) into top-level `messages[]` with `{ ts, user, text, +//! thread_ts, channel_id }`. `channel_id` is pulled from each match's +//! `channel.id` field. `paging.pages` is preserved at top-level for +//! caller pagination. +//! +//! ## Design note: user-id resolution is NOT here +//! +//! `SlackUsers` is a per-sync cache built from a separate API call — +//! not a function of any individual response. Resolving user ids +//! happens in the host's `SlackSyncPipeline` (the enricher layer), keeping +//! this module purely data-shape–oriented. +//! This matches Gmail's pattern of "post_process is data-only". +//! +//! Unknown slugs are silently no-ops so new Composio actions don't +//! break the provider. + +use serde_json::{Map, Value}; + +/// Entry point a host calls on each Slack action response (the slug names +/// the action) before handing it to `normalise_payload`. +/// +/// Dispatches on the Composio action slug and rewrites `data` in place. +/// Unknown slugs are silently ignored. +pub fn post_process(slug: &str, _arguments: Option<&Value>, data: &mut Value) { + log::debug!("[composio:slack][post-process] slug={slug}"); + match slug { + "SLACK_FETCH_CONVERSATION_HISTORY" => reshape_fetch_history(data), + "SLACK_LIST_CONVERSATIONS" => reshape_list_conversations(data), + "SLACK_SEARCH_MESSAGES" => reshape_search_messages(data), + _ => { + log::debug!("[composio:slack][post-process] unknown slug={slug}, passing through"); + } + } +} + +// ─── SLACK_FETCH_CONVERSATION_HISTORY ────────────────────────────────────── + +/// Rewrite a `SLACK_FETCH_CONVERSATION_HISTORY` response in place. +/// +/// Walks possible nested envelopes (`/data/messages`, `/messages`, +/// `/data/data/messages`) to find the raw messages array, drops messages +/// with empty `text`, and emits a slim `{ ts, user, text, thread_ts }` +/// shape under a top-level `messages[]` key. The consumed nested array is +/// removed from the payload so the raw verbose rows don't linger alongside +/// the slim copy. The caller injects `channel_id` via +/// the host's Slack sync pipeline. +fn reshape_fetch_history(data: &mut Value) { + let arr = take_array( + data, + &["/data/messages", "/messages", "/data/data/messages"], + 0, + ); + let slim: Vec = arr.into_iter().filter_map(slim_history_message).collect(); + with_object(data, |obj| { + obj.insert("messages".to_string(), Value::Array(slim)); + }); + log::debug!("[composio:slack][post-process] SLACK_FETCH_CONVERSATION_HISTORY reshaped"); +} + +fn slim_history_message(raw: Value) -> Option { + let text = raw + .get("text") + .and_then(|v| v.as_str()) + .unwrap_or("") + .trim(); + if text.is_empty() { + return None; + } + let mut out = Map::new(); + // `ts` is required: without it a caller can neither cursor nor archive. + out.insert("ts".into(), raw.get("ts")?.clone()); + if let Some(user) = raw.get("user").or_else(|| raw.get("bot_id")) { + out.insert("user".into(), user.clone()); + } + out.insert("text".into(), Value::String(text.to_string())); + if let Some(thread_ts) = raw.get("thread_ts") { + out.insert("thread_ts".into(), thread_ts.clone()); + } + if let Some(permalink) = raw.get("permalink") { + out.insert("permalink".into(), permalink.clone()); + } + Some(Value::Object(out)) +} + +/// Find the first array at any of `candidates`, remove that field (plus +/// `envelope_depth` ancestor object envelopes) from `data`, and return the +/// array. Removing the consumed nested payload keeps the reshaped output from +/// carrying duplicate raw rows. +fn take_array(data: &mut Value, candidates: &[&str], envelope_depth: usize) -> Vec { + for path in candidates { + let arr = match data.pointer(path).and_then(|v| v.as_array().cloned()) { + Some(a) => a, + None => continue, + }; + let mut remove_path = path.to_string(); + for _ in 0..envelope_depth { + remove_path = match remove_path.rsplit_once('/') { + Some((parent, _)) => parent.to_string(), + None => break, + }; + } + remove_nested(data, &remove_path); + return arr; + } + Vec::new() +} + +/// Remove the field at `path` from `data`, pruning any ancestor object that +/// the removal left empty so a consumed `data` envelope disappears entirely +/// instead of lingering as `{}`. +fn remove_nested(data: &mut Value, path: &str) { + let segments: Vec<&str> = path + .trim_start_matches('/') + .split('/') + .filter(|s| !s.is_empty()) + .collect(); + if segments.is_empty() { + return; + } + + // Remove the leaf field. + let mut current = &mut *data; + for seg in &segments[..segments.len() - 1] { + current = match current.get_mut(*seg) { + Some(next) => next, + None => return, + }; + } + if let Value::Object(map) = current { + map.remove(segments[segments.len() - 1]); + } + + // Prune empty object ancestors, deepest first. + for depth in (0..segments.len().saturating_sub(1)).rev() { + // Re-walk to the object at `segments[..=depth]`. + let mut ancestor = &mut *data; + for seg in &segments[..=depth] { + ancestor = match ancestor.get_mut(*seg) { + Some(next) => next, + None => return, + }; + } + if !matches!(ancestor, Value::Object(m) if m.is_empty()) { + break; + } + // Remove it from its parent (`segments[..depth]`). For `depth == 0` + // the parent is the top-level object, so an emptied `data` envelope + // key disappears entirely. + let mut parent = &mut *data; + for seg in &segments[..depth] { + parent = match parent.get_mut(*seg) { + Some(next) => next, + None => return, + }; + } + if let Value::Object(map) = parent { + map.remove(segments[depth]); + } + } +} + +// ─── SLACK_LIST_CONVERSATIONS ─────────────────────────────────────────────── + +/// Rewrite a `SLACK_LIST_CONVERSATIONS` response in place. +/// +/// Reshapes into a top-level `channels[]` with `{ id, name, is_private }` +/// per channel; entries with an empty id are dropped. +fn reshape_list_conversations(data: &mut Value) { + let arr = take_array( + data, + &[ + "/data/channels", + "/channels", + "/data/data/channels", + "/data/conversations", + "/conversations", + ], + 0, + ); + + let slim: Vec = arr.into_iter().filter_map(slim_channel).collect(); + with_object(data, |obj| { + obj.insert("channels".to_string(), Value::Array(slim)); + }); + log::debug!("[composio:slack][post-process] SLACK_LIST_CONVERSATIONS reshaped"); +} + +fn slim_channel(raw: Value) -> Option { + let id = raw.get("id").and_then(|v| v.as_str()).unwrap_or("").trim(); + if id.is_empty() { + return None; + } + let name = raw + .get("name") + .and_then(|v| v.as_str()) + .unwrap_or(id) + .trim(); + let is_private = raw + .get("is_private") + .and_then(|v| v.as_bool()) + .unwrap_or(false); + Some(Value::Object({ + let mut m = Map::new(); + m.insert("id".into(), Value::String(id.to_string())); + m.insert("name".into(), Value::String(name.to_string())); + m.insert("is_private".into(), Value::Bool(is_private)); + m + })) +} + +// ─── SLACK_SEARCH_MESSAGES ────────────────────────────────────────────────── + +/// Rewrite a `SLACK_SEARCH_MESSAGES` response in place. +/// +/// Reshapes `messages.matches[]` (possibly nested under one or two +/// `data` envelopes) into top-level `messages[]`. `channel_id` is pulled +/// from each match's `channel.id` field. `paging.pages` is preserved at +/// top-level under `pages` for the caller to drive pagination. +fn reshape_search_messages(data: &mut Value) { + // Preserve paging info before mutating data (take_array below removes the + // envelope that carries it). + let pages = [ + data.pointer("/data/messages/paging/pages"), + data.pointer("/messages/paging/pages"), + data.pointer("/data/data/messages/paging/pages"), + ] + .into_iter() + .flatten() + .find_map(|v| v.as_u64()) + .unwrap_or(1); + + // Envelope depth 1 removes the `messages` object (matches + paging) that + // held the consumed rows, not just the `matches` array. + let arr = take_array( + data, + &[ + "/data/messages/matches", + "/messages/matches", + "/data/data/messages/matches", + ], + 1, + ); + + let slim: Vec = arr.into_iter().filter_map(slim_search_match).collect(); + with_object(data, |obj| { + obj.insert("messages".to_string(), Value::Array(slim)); + obj.insert("pages".to_string(), Value::Number(pages.into())); + }); + log::debug!("[composio:slack][post-process] SLACK_SEARCH_MESSAGES reshaped"); +} + +fn slim_search_match(raw: Value) -> Option { + let text = raw + .get("text") + .and_then(|v| v.as_str()) + .unwrap_or("") + .trim(); + if text.is_empty() { + return None; + } + let ts = raw.get("ts")?; + let channel_id = raw + .pointer("/channel/id") + .and_then(|v| v.as_str()) + .unwrap_or("") + .trim(); + + let mut out = Map::new(); + out.insert("ts".into(), ts.clone()); + if let Some(user) = raw.get("user").or_else(|| raw.get("bot_id")) { + out.insert("user".into(), user.clone()); + } + out.insert("text".into(), Value::String(text.to_string())); + if let Some(thread_ts) = raw.get("thread_ts") { + out.insert("thread_ts".into(), thread_ts.clone()); + } + if !channel_id.is_empty() { + out.insert("channel_id".into(), Value::String(channel_id.to_string())); + } + if let Some(permalink) = raw.get("permalink") { + out.insert("permalink".into(), permalink.clone()); + } + Some(Value::Object(out)) +} + +// ─── Helpers ──────────────────────────────────────────────────────────────── + +/// Edit `data` as a JSON object, replacing it with an empty object first if +/// it is not one. +/// +/// Takes the value out, edits the map, and puts it back, so there is no +/// "re-borrow as an object" step that would need an `expect`. +fn with_object(data: &mut Value, edit: impl FnOnce(&mut Map)) { + let mut map = match std::mem::take(data) { + Value::Object(map) => map, + _ => Map::new(), + }; + edit(&mut map); + *data = Value::Object(map); +} + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod_tests.rs new file mode 100644 index 00000000..73a6702b --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod_tests.rs @@ -0,0 +1,306 @@ +//! Tests for the Slack post-processor. + +use super::*; +use serde_json::json; + +// ─── SLACK_FETCH_CONVERSATION_HISTORY ───────────────────────────────────── + +#[test] +fn history_reshapes_top_level_messages() { + let mut data = json!({ + "messages": [ + { "ts": "1714003200.000100", "user": "U1", "text": "hello" }, + { "ts": "1714003300.000200", "user": "U2", "text": "world", "thread_ts": "1714003200.0" }, + { "ts": "1714003400.000300", "user": "U3", "text": " " }, // dropped: empty text + ], + "response_metadata": { "next_cursor": "abc" } + }); + post_process("SLACK_FETCH_CONVERSATION_HISTORY", None, &mut data); + + let msgs = data["messages"].as_array().unwrap(); + assert_eq!(msgs.len(), 2, "empty-text message must be dropped"); + assert_eq!(msgs[0]["ts"], "1714003200.000100"); + assert_eq!(msgs[0]["user"], "U1"); + assert_eq!(msgs[0]["text"], "hello"); + assert!(msgs[0].get("thread_ts").is_none()); + assert_eq!(msgs[1]["thread_ts"], "1714003200.0"); +} + +#[test] +fn history_reshapes_nested_data_envelope() { + let mut data = json!({ + "data": { + "messages": [ + { "ts": "1714003200.0", "user": "U1", "text": "hi" } + ] + } + }); + post_process("SLACK_FETCH_CONVERSATION_HISTORY", None, &mut data); + let msgs = data["messages"].as_array().unwrap(); + assert_eq!(msgs.len(), 1); + assert_eq!(msgs[0]["text"], "hi"); +} + +#[test] +fn history_reshapes_doubly_nested_envelope() { + let mut data = json!({ + "data": { + "data": { + "messages": [ + { "ts": "1714003200.0", "user": "U1", "text": "deep" } + ] + } + } + }); + post_process("SLACK_FETCH_CONVERSATION_HISTORY", None, &mut data); + let msgs = data["messages"].as_array().unwrap(); + assert_eq!(msgs.len(), 1); + assert_eq!(msgs[0]["text"], "deep"); +} + +#[test] +fn history_drops_message_without_ts() { + let mut data = json!({ + "messages": [ + { "user": "U1", "text": "no timestamp" }, + { "ts": "1714003200.0", "user": "U2", "text": "has ts" }, + ] + }); + post_process("SLACK_FETCH_CONVERSATION_HISTORY", None, &mut data); + let msgs = data["messages"].as_array().unwrap(); + assert_eq!(msgs.len(), 1); + assert_eq!(msgs[0]["text"], "has ts"); +} + +#[test] +fn history_removes_nested_envelope_after_reshape() { + let mut data = json!({ + "data": { + "messages": [ + { "ts": "1714003200.0", "user": "U1", "text": "hi" } + ] + } + }); + post_process("SLACK_FETCH_CONVERSATION_HISTORY", None, &mut data); + + let msgs = data["messages"].as_array().unwrap(); + assert_eq!(msgs.len(), 1); + assert_eq!(msgs[0]["text"], "hi"); + assert!( + data.pointer("/data").is_none(), + "consumed `data.messages` envelope must be removed, got: {data}" + ); +} + +// ─── SLACK_LIST_CONVERSATIONS ───────────────────────────────────────────── + +#[test] +fn list_conversations_reshapes_channels() { + let mut data = json!({ + "data": { + "channels": [ + { "id": "C1", "name": "eng", "is_private": false, "extra": "noise" }, + { "id": "G1", "name": "ops", "is_private": true }, + { "id": "", "name": "empty-id" }, // dropped + ] + } + }); + post_process("SLACK_LIST_CONVERSATIONS", None, &mut data); + let channels = data["channels"].as_array().unwrap(); + assert_eq!(channels.len(), 2, "empty-id entry must be dropped"); + assert_eq!(channels[0]["id"], "C1"); + assert_eq!(channels[0]["name"], "eng"); + assert_eq!(channels[0]["is_private"], false); + assert!( + channels[0].get("extra").is_none(), + "noise fields must be removed" + ); + assert_eq!(channels[1]["id"], "G1"); + assert_eq!(channels[1]["is_private"], true); +} + +#[test] +fn list_conversations_falls_back_to_conversations_key() { + let mut data = json!({ + "conversations": [ + { "id": "C2", "name": "dev", "is_private": false } + ] + }); + post_process("SLACK_LIST_CONVERSATIONS", None, &mut data); + let channels = data["channels"].as_array().unwrap(); + assert_eq!(channels.len(), 1); + assert_eq!(channels[0]["id"], "C2"); + assert!( + data.pointer("/conversations").is_none(), + "consumed `conversations` field must be removed" + ); +} + +// ─── SLACK_SEARCH_MESSAGES ──────────────────────────────────────────────── + +#[test] +fn search_messages_reshapes_matches() { + let mut data = json!({ + "messages": { + "matches": [ + { + "ts": "1714003200.0", + "user": "U1", + "text": "hello from search", + "channel": { "id": "C1" } + }, + { + "ts": "1714003300.0", + "user": "U2", + "text": " ", // dropped: whitespace only + "channel": { "id": "C1" } + }, + ], + "paging": { "pages": 3 } + } + }); + post_process("SLACK_SEARCH_MESSAGES", None, &mut data); + let msgs = data["messages"].as_array().unwrap(); + assert_eq!(msgs.len(), 1, "empty-text match must be dropped"); + assert_eq!(msgs[0]["ts"], "1714003200.0"); + assert_eq!(msgs[0]["text"], "hello from search"); + assert_eq!(msgs[0]["channel_id"], "C1"); + assert_eq!(data["pages"], 3, "paging.pages must be preserved"); +} + +#[test] +fn search_messages_nested_data_envelope() { + let mut data = json!({ + "data": { + "messages": { + "matches": [ + { "ts": "1714003200.0", "user": "U1", "text": "nested", "channel": { "id": "C2" } } + ], + "paging": { "pages": 1 } + } + } + }); + post_process("SLACK_SEARCH_MESSAGES", None, &mut data); + let msgs = data["messages"].as_array().unwrap(); + assert_eq!(msgs.len(), 1); + assert_eq!(msgs[0]["channel_id"], "C2"); + assert_eq!(data["pages"], 1_u64); +} + +#[test] +fn search_messages_no_matches_emits_empty_array() { + let mut data = json!({ "messages": { "matches": [] } }); + post_process("SLACK_SEARCH_MESSAGES", None, &mut data); + let msgs = data["messages"].as_array().unwrap(); + assert!(msgs.is_empty()); +} + +#[test] +fn search_messages_removes_nested_envelope_after_reshape() { + let mut data = json!({ + "data": { + "messages": { + "matches": [ + { "ts": "1714003200.0", "user": "U1", "text": "nested", "channel": { "id": "C2" } } + ], + "paging": { "pages": 1 } + } + } + }); + post_process("SLACK_SEARCH_MESSAGES", None, &mut data); + + let msgs = data["messages"].as_array().unwrap(); + assert_eq!(msgs.len(), 1); + assert_eq!(msgs[0]["channel_id"], "C2"); + assert_eq!(data["pages"], 1_u64); + assert!( + data.pointer("/data").is_none(), + "consumed `data.messages` envelope must be removed, got: {data}" + ); +} + +#[test] +fn search_messages_doubly_nested_paging_preserved() { + let mut data = json!({ + "data": { + "data": { + "messages": { + "matches": [ + { "ts": "1714003200.0", "user": "U1", "text": "deep", "channel": { "id": "C3" } } + ], + "paging": { "pages": 4 } + } + } + } + }); + post_process("SLACK_SEARCH_MESSAGES", None, &mut data); + + let msgs = data["messages"].as_array().unwrap(); + assert_eq!(msgs.len(), 1); + assert_eq!(msgs[0]["text"], "deep"); + assert_eq!( + data["pages"], 4_u64, + "doubly-nested paging must be preserved" + ); + assert!( + data.pointer("/data").is_none(), + "consumed `data.data.messages` envelope must be removed, got: {data}" + ); +} + +// ─── Unknown slug ───────────────────────────────────────────────────────── + +#[test] +fn unknown_slug_is_noop() { + let mut data = json!({ "foo": "bar" }); + let original = data.clone(); + post_process("SLACK_SEND_MESSAGE", None, &mut data); + assert_eq!(data, original, "unknown slug must not mutate data"); +} + +#[test] +fn malformed_history_payload_becomes_an_empty_stable_shape() { + for mut data in [json!(null), json!([]), json!({"messages": "wrong"})] { + post_process("SLACK_FETCH_CONVERSATION_HISTORY", None, &mut data); + assert_eq!(data, json!({"messages": []})); + } +} + +#[test] +fn malformed_channel_rows_are_dropped_and_defaults_are_stable() { + let mut data = json!({ + "channels": [ + null, + "not an object", + {"id": 7, "name": "numeric id"}, + {"id": " C1 ", "name": 99, "is_private": "yes"} + ] + }); + post_process("SLACK_LIST_CONVERSATIONS", None, &mut data); + assert_eq!( + data, + json!({"channels": [{"id":"C1", "name":"C1", "is_private":false}]}) + ); +} + +#[test] +fn malformed_search_rows_and_page_counts_fall_back_safely() { + let mut data = json!({ + "messages": { + "matches": [ + null, + {"ts":"1.0", "text":"no channel"}, + {"channel":{"id":"C1"}, "text":"no timestamp"}, + {"ts":"2.0", "channel":{"id":"C1"}, "text":" valid "} + ], + "paging": {"pages":"many"} + } + }); + post_process("SLACK_SEARCH_MESSAGES", None, &mut data); + assert_eq!(data["pages"], 1); + let messages = data["messages"].as_array().unwrap(); + assert_eq!(messages.len(), 2); + assert_eq!(messages[0]["text"], "no channel"); + assert!(messages[0].get("channel_id").is_none()); + assert_eq!(messages[1]["text"], "valid"); +} diff --git a/crates/tinymemory-integrations/src/sources/fetch/mod.rs b/crates/tinymemory-integrations/src/sources/fetch/mod.rs new file mode 100644 index 00000000..6063b5f7 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/fetch/mod.rs @@ -0,0 +1,151 @@ +//! Fetching one URL into a [`RawDocument`] or a link [`StoreItem`]. +//! +//! A URL a user types is an SSRF vector: `http://169.254.169.254/` is a cloud +//! metadata endpoint, `http://localhost:6379/` is somebody's Redis, and a +//! hostname that resolves publicly on the first lookup can resolve to a private +//! address on the second. Every fetch goes through the guard in [`ssrf`]: a +//! scheme and host policy, a resolver that pins connections to globally +//! routable addresses, and per-hop redirect re-checks. The RSS and web-page +//! readers fetch through here too, each with its own body cap. +//! +//! No scheduling, no retries, no credentials, no robots.txt: this fetches one +//! URL, once, when asked. Conversion to markdown is the `documents` module's. + +use crate::documents::{DocumentConverter, MAX_DOCUMENT_BYTES, RawDocument, document_item}; +use tinymemory_api::{MemoryMeta, SourceKind, StoreItem}; + +use crate::sources::error::{Error, Result}; +use ssrf::{build_client, is_url_allowed, read_body_capped}; + +pub mod ssrf; + +/// Fetch `url` and return its body as a [`RawDocument`]. +/// +/// The response's `Content-Type` becomes the document's declared MIME type and +/// the URL becomes its origin; the last path segment, when it has an +/// extension, becomes its filename so format detection can fall back to it. +/// +/// # Errors +/// +/// - [`Error::Invalid`] for a malformed URL, one the SSRF guard refuses, or an +/// empty body. +/// - [`Error::Unreachable`] when the request never completed or the body read +/// was interrupted. +/// - [`Error::Upstream`] for a non-success status. +/// - [`Error::TooLarge`] for a body over [`MAX_DOCUMENT_BYTES`]. +pub async fn fetch_url(url: &str) -> Result { + fetch_url_capped(url, MAX_DOCUMENT_BYTES as u64).await +} + +/// [`fetch_url`] with a caller-chosen body cap, for the readers whose sources +/// warrant a tighter one than [`MAX_DOCUMENT_BYTES`]. +/// +/// # Errors +/// +/// As [`fetch_url`], with [`Error::TooLarge`] for a body over `max_bytes`. +pub(crate) async fn fetch_url_capped(url: &str, max_bytes: u64) -> Result { + let parsed = reqwest::Url::parse(url) + .map_err(|error| Error::Invalid(format!("invalid url {url:?}: {error}")))?; + if !is_url_allowed(&parsed) { + return Err(Error::Invalid(format!( + "url {url:?} is not an allowed fetch target" + ))); + } + + let client = build_client().map_err(Error::Reader)?; + tracing::debug!( + host = %parsed.host_str().unwrap_or(""), + "[memory_sources:fetch] fetching url" + ); + let response = client + .get(parsed.clone()) + .send() + .await + .map_err(|error| Error::Unreachable(format!("fetching {url:?}: {error}")))?; + + response_to_document(url, parsed, response, max_bytes).await +} + +/// Fetch `url`, convert it through `converter`, and wrap it as a +/// [`StoreItem::Document`] with `source.kind = Link` and `url` set. +/// +/// `source_id` is the configured source's id, when the fetch is for one. +/// +/// # Errors +/// +/// Whatever [`fetch_url`] returns, plus [`Error::Document`] when conversion +/// fails. +pub async fn link_item( + url: &str, + source_id: Option, + converter: &dyn DocumentConverter, +) -> Result { + let document = fetch_url(url).await?; + let mut meta = MemoryMeta::from_source(SourceKind::Link, source_id); + meta.url = document.origin.clone().or_else(|| Some(url.to_string())); + Ok(document_item(converter, &document, meta).await?) +} + +/// Validate and convert a completed HTTP response. +async fn response_to_document( + url: &str, + parsed: reqwest::Url, + response: reqwest::Response, + max_bytes: u64, +) -> Result { + let status = response.status(); + if !status.is_success() { + return Err(Error::Upstream(format!( + "fetching {url:?} answered {status}" + ))); + } + + let content_type = response + .headers() + .get(reqwest::header::CONTENT_TYPE) + .and_then(|value| value.to_str().ok()) + .map(str::to_string); + + // The cap is applied while reading, not after: a body that would not fit is + // one this process should never have finished buffering. + let bytes = read_body_capped(response, max_bytes) + .await + .map_err(|error| read_error(url, &error))?; + + if bytes.is_empty() { + return Err(Error::Invalid(format!("{url:?} returned no body"))); + } + + let mut document = RawDocument::new(bytes).with_origin(parsed.to_string()); + if let Some(content_type) = content_type { + document = document.with_mime(content_type); + } + // A URL's last path segment is often the only filename there is, and format + // detection falls back to it when the server sent no useful type. + if let Some(name) = parsed + .path_segments() + .and_then(|mut segments| segments.next_back()) + .filter(|name| !name.is_empty() && name.contains('.')) + { + document = document.with_filename(name.to_string()); + } + Ok(document) +} + +/// Turn a `read_body_capped` failure into the right [`Error`] variant. +/// +/// `read_body_capped` collapses two failures into one `String`: a body over +/// the size cap, and a stream that failed mid-read. They need different retry +/// policies, so this tells them apart by the message `read_body_capped` always +/// uses for the size case. +fn read_error(url: &str, error: &str) -> Error { + if error.contains("exceeds") && error.contains("-byte limit") { + Error::TooLarge(format!("reading {url:?}: {error}")) + } else { + Error::Unreachable(format!("reading {url:?}: {error}")) + } +} + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinymemory-integrations/src/sources/fetch/mod_tests.rs b/crates/tinymemory-integrations/src/sources/fetch/mod_tests.rs new file mode 100644 index 00000000..a42f4917 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/fetch/mod_tests.rs @@ -0,0 +1,177 @@ +//! Tests for URL fetching. +//! +//! The guard and argument handling are exercised directly; response handling +//! runs against a loopback server the test controls. Nothing reaches the real +//! network. + +use super::*; + +async fn local_response(response: impl Into>) -> reqwest::Response { + use tokio::io::{AsyncReadExt, AsyncWriteExt}; + + let listener = tokio::net::TcpListener::bind((std::net::Ipv4Addr::LOCALHOST, 0)) + .await + .expect("bind controlled server"); + let address = listener.local_addr().expect("server address"); + let response = response.into(); + let server = tokio::spawn(async move { + let (mut stream, _) = listener.accept().await.expect("accept request"); + let mut request = [0_u8; 1024]; + let _ = stream.read(&mut request).await.expect("read request"); + stream.write_all(&response).await.expect("write response"); + }); + let received = reqwest::get(format!("http://{address}/")) + .await + .expect("controlled response"); + server.await.expect("server task"); + received +} + +#[tokio::test] +async fn a_malformed_url_is_rejected_before_anything_is_fetched() { + let error = fetch_url("not a url").await.unwrap_err(); + assert!(matches!(error, Error::Invalid(_)), "got {error:?}"); + assert!(error.to_string().contains("invalid url"), "got {error}"); +} + +#[tokio::test] +async fn loopback_and_link_local_targets_are_refused() { + for url in [ + "http://127.0.0.1/", + "http://localhost:6379/", + "http://169.254.169.254/latest/meta-data/", + "http://[::1]/", + ] { + let error = fetch_url(url).await.unwrap_err(); + assert!( + error.to_string().contains("not an allowed fetch target"), + "{url} gave {error}" + ); + } +} + +#[tokio::test] +async fn a_non_http_scheme_is_refused() { + for url in [ + "file:///etc/passwd", + "ftp://example.com/x", + "gopher://example.com/", + ] { + let error = fetch_url(url).await.unwrap_err(); + assert!(matches!(error, Error::Invalid(_)), "{url} gave {error:?}"); + } +} + +#[test] +fn a_size_limit_failure_is_reported_as_budget_exceeded() { + let error = read_error( + "https://example.com/", + "response body exceeds 8-byte limit (Content-Length=9)", + ); + assert!(matches!(error, Error::TooLarge(_)), "got {error:?}"); +} + +#[test] +fn an_interrupted_read_is_reported_as_unreachable_not_budget_exceeded() { + let error = read_error( + "https://example.com/", + "failed to read response body: connection reset", + ); + assert!(matches!(error, Error::Unreachable(_)), "got {error:?}"); +} + +#[tokio::test] +async fn completed_response_preserves_body_type_origin_and_filename() { + let response = local_response( + b"HTTP/1.1 200 OK\r\nContent-Type: text/markdown; charset=utf-8\r\nContent-Length: 7\r\n\r\n# title", + ) + .await; + let url = reqwest::Url::parse("https://example.com/guides/readme.md").unwrap(); + let document = response_to_document( + url.as_str(), + url.clone(), + response, + MAX_DOCUMENT_BYTES as u64, + ) + .await + .unwrap(); + assert_eq!(document.bytes, b"# title"); + assert_eq!(document.origin.as_deref(), Some(url.as_str())); + assert_eq!(document.filename.as_deref(), Some("readme.md")); + assert_eq!( + document.declared_mime.as_deref(), + Some("text/markdown; charset=utf-8") + ); +} + +#[tokio::test] +async fn completed_response_handles_status_empty_body_and_filename_absence() { + let response = + local_response(b"HTTP/1.1 503 Service Unavailable\r\nContent-Length: 0\r\n\r\n").await; + let url = reqwest::Url::parse("https://example.com/unavailable").unwrap(); + let error = response_to_document( + url.as_str(), + url.clone(), + response, + MAX_DOCUMENT_BYTES as u64, + ) + .await + .unwrap_err(); + assert!(matches!(error, Error::Upstream(_))); + + let response = local_response(b"HTTP/1.1 200 OK\r\nContent-Length: 0\r\n\r\n").await; + let error = response_to_document( + url.as_str(), + url.clone(), + response, + MAX_DOCUMENT_BYTES as u64, + ) + .await + .unwrap_err(); + assert!(matches!(error, Error::Invalid(_))); + + let response = local_response(b"HTTP/1.1 200 OK\r\nContent-Length: 4\r\n\r\ntext").await; + let document = response_to_document( + url.as_str(), + url.clone(), + response, + MAX_DOCUMENT_BYTES as u64, + ) + .await + .unwrap(); + assert_eq!(document.bytes, b"text"); + assert!(document.filename.is_none()); + assert!(document.declared_mime.is_none()); +} + +#[tokio::test] +async fn completed_response_maps_declared_oversize_to_budget_exceeded() { + let response = local_response( + [ + b"HTTP/1.1 200 OK\r\nContent-Length: ", + (MAX_DOCUMENT_BYTES + 1).to_string().as_bytes(), + b"\r\n\r\n", + ] + .concat(), + ) + .await; + let url = reqwest::Url::parse("https://example.com/huge.bin").unwrap(); + let error = response_to_document( + url.as_str(), + url.clone(), + response, + MAX_DOCUMENT_BYTES as u64, + ) + .await + .unwrap_err(); + assert!(matches!(error, Error::TooLarge(_))); +} + +#[tokio::test] +async fn a_link_item_refuses_a_private_target_before_fetching() { + let chain = crate::documents::ConverterChain::default(); + let error = link_item("http://127.0.0.1/", Some("src_link".into()), &chain) + .await + .unwrap_err(); + assert!(matches!(error, Error::Invalid(_)), "got {error:?}"); +} diff --git a/crates/tinymemory-integrations/src/sources/fetch/ssrf/mod.rs b/crates/tinymemory-integrations/src/sources/fetch/ssrf/mod.rs new file mode 100644 index 00000000..49c28453 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/fetch/ssrf/mod.rs @@ -0,0 +1,231 @@ +//! Shared SSRF guard and fetch hygiene for the network source readers. +//! +//! [`super::fetch_url`] and the web-page and RSS readers built on it all fetch +//! user-configured URLs, so they share the policy in this module. It is public +//! so a host fetching a user-supplied URL by other means applies the same +//! policy rather than a second, weaker one. +//! +//! The hostname *text* check (`is_blocked_host`) rejects private IP literals +//! (including their IPv4-mapped IPv6 forms, e.g. `::ffff:127.0.0.1`), +//! `localhost`, `.local` / `.internal` names, and single-label hostnames, but +//! a public-looking name can resolve to a loopback / private / link-local +//! address (including the cloud-metadata `169.254.169.254`) at lookup time. +//! `PublicOnlyResolver` therefore vets the resolved addresses and only lets the +//! connection proceed to a globally routable IP, so the request is pinned to an +//! address we have already allowed (no re-resolution between the check and the +//! connect). Redirects are re-checked through `is_url_allowed` so a public URL +//! cannot bounce the fetch onto an internal host. +//! +//! `read_body_capped` streams a response body and stops at a byte cap, so a +//! hostile or gigantic page/feed cannot OOM the process before the size check +//! runs. +//! +//! An IPv6 literal URL such as `http://[::1]/` is refused whatever its address, +//! including a public one: `reqwest::Url::host_str` keeps the brackets, the +//! text no longer parses as an IP address, and a name with no dot is treated +//! as a single-label internal name. That is a fail-closed limitation, not a +//! classification: the address classifier (`is_public_ip`) does handle IPv6 +//! (and IPv4-mapped forms) for *resolved* addresses, which is how a hostname +//! with an AAAA record is still vetted. + +use std::net::{IpAddr, Ipv4Addr, SocketAddr}; +use std::sync::Arc; + +use futures::stream::StreamExt; +use reqwest::dns::{Addrs, Name, Resolve, Resolving}; + +/// Build an HTTP client with a redirect policy that re-applies the SSRF +/// host/scheme check to every redirect hop, a DNS resolver that only yields +/// globally routable addresses, a 20-second timeout and the `openhuman` +/// user agent the network readers have always sent. +/// +/// # Errors +/// +/// A message naming the failure when the TLS backend cannot be initialised. +pub fn build_client() -> Result { + reqwest::Client::builder() + .timeout(std::time::Duration::from_secs(20)) + .user_agent("openhuman") + .redirect(reqwest::redirect::Policy::custom(|attempt| { + if is_url_allowed(attempt.url()) { + attempt.follow() + } else { + // `stop` returns the redirect response to the caller instead + // of following it; the read then fails on the non-2xx status. + attempt.stop() + } + })) + .dns_resolver(Arc::new(PublicOnlyResolver)) + .build() + .map_err(|e| format!("failed to build http client: {e}")) +} + +/// Stream a response body, failing once it exceeds `max` bytes. +/// +/// `Response::bytes()` buffers the entire body before any size check, so a +/// server that omits or understates `Content-Length` (for example a chunked +/// response) could OOM the process despite the cap. Reading incrementally +/// enforces the limit while the bytes arrive. +/// +/// # Errors +/// +/// A message containing `exceeds {max}-byte limit` when the body is over the +/// cap, or `failed to read response body` when the stream fails mid-read. +pub async fn read_body_capped(resp: reqwest::Response, max: u64) -> Result, String> { + // Trust a truthful Content-Length up front so a known-huge body is + // rejected before the first byte is read. + if let Some(len) = resp.content_length() + && len > max + { + return Err(format!( + "response body exceeds {max}-byte limit (Content-Length={len})" + )); + } + + let mut body = Vec::new(); + let mut stream = resp.bytes_stream(); + while let Some(chunk) = stream.next().await { + let chunk = chunk.map_err(|e| format!("failed to read response body: {e}"))?; + body.extend_from_slice(&chunk); + if body.len() as u64 > max { + return Err(format!( + "response body exceeds {max}-byte limit (read {} bytes)", + body.len() + )); + } + } + Ok(body) +} + +/// A DNS resolver that only yields globally routable addresses. +/// +/// The text-based `is_blocked_host` check rejects private IP *literals* and +/// local hostnames, but a public-looking hostname can resolve to a loopback, +/// private, link-local, or cloud-metadata address (`169.254.169.254`) at +/// lookup time. Installing this resolver means reqwest connects to addresses +/// we have already vetted: a hostname whose current resolution is non-public +/// fails the request instead of silently reaching an internal service, and the +/// validated address is the one the connection is pinned to (no re-resolution +/// between the check and the connect). +#[derive(Debug, Default)] +struct PublicOnlyResolver; + +impl Resolve for PublicOnlyResolver { + fn resolve(&self, name: Name) -> Resolving { + let host = name.as_str().to_string(); + Box::pin(async move { + let addrs: Vec = tokio::net::lookup_host((host.as_str(), 0)) + .await + .map_err(box_err)? + .filter(|addr| is_public_ip(addr.ip())) + .collect(); + if addrs.is_empty() { + return Err(box_err(std::io::Error::new( + std::io::ErrorKind::AddrNotAvailable, + format!("host {host} resolved to no public addresses"), + ))); + } + Ok(Box::new(addrs.into_iter()) as Addrs) + }) + } +} + +fn box_err( + e: impl std::error::Error + Send + Sync + 'static, +) -> Box { + Box::new(e) +} + +/// Whether `ip` is a globally routable address — the one address classifier +/// behind both halves of the SSRF guard (literal hosts in `is_blocked_host` +/// and resolved addresses in `PublicOnlyResolver`). +/// +/// Not fetchable: loopback, private, link-local, unspecified, CGNAT +/// (`100.64.0.0/10`), `192.0.0.0/16` (IETF protocol assignments and the +/// `192.0.2.0/24` documentation range), multicast, broadcast, documentation +/// (`198.51.100.0/24`, `203.0.113.0/24`, `2001:db8::/32`), benchmarking +/// (`198.18.0.0/15`), +/// reserved (`240.0.0.0/4`), and IPv6 unique-local (`fc00::/7`) and +/// link-local (`fe80::/10`). An IPv6 address carrying an IPv4 one — mapped +/// (`::ffff:a.b.c.d`) or the deprecated compatible form (`::a.b.c.d`) — is +/// judged by its IPv4 part, so a mapped loopback stays blocked. +fn is_public_ip(ip: IpAddr) -> bool { + match ip { + IpAddr::V4(v4) => { + let o = v4.octets(); + !(v4.is_loopback() + || v4.is_private() + || v4.is_link_local() + || v4.is_unspecified() + || v4.is_multicast() + || v4.is_broadcast() + || (o[0] == 100 && o[1] & 0xc0 == 0x40) + || (o[0] == 192 && o[1] == 0) + || (o[0] == 198 && o[1] == 51 && o[2] == 100) + || (o[0] == 203 && o[1] == 0 && o[2] == 113) + || (o[0] == 198 && o[1] & 0xfe == 18) + || o[0] >= 240) + } + IpAddr::V6(v6) => { + if v6.is_loopback() || v6.is_unspecified() || v6.is_multicast() { + return false; + } + let o = v6.octets(); + if (o[0] & 0xfe == 0xfc) + || (o[0] == 0xfe && o[1] & 0xc0 == 0x80) + || (o[0] == 0x20 && o[1] == 0x01 && o[2] == 0x0d && o[3] == 0xb8) + { + return false; + } + if let Some(v4) = v6.to_ipv4_mapped() { + return is_public_ip(IpAddr::V4(v4)); + } + // `to_ipv4_mapped` answers `None` for the compatible form, so + // without this `::127.0.0.1` would read as public. `::` and `::1` + // were judged as themselves above. + if o[..12].iter().all(|byte| *byte == 0) { + return is_public_ip(IpAddr::V4(Ipv4Addr::new(o[12], o[13], o[14], o[15]))); + } + true + } + } +} + +/// Whether a URL may be fetched: `http(s)` scheme against a public host. +pub fn is_url_allowed(url: &reqwest::Url) -> bool { + match url.scheme() { + "http" | "https" => {} + _ => return false, + } + let Some(host) = url.host_str() else { + return false; + }; + !is_blocked_host(host) +} + +/// Reject hosts that could target non-public resources: IP literals in +/// loopback / private / link-local / unique-local / unspecified ranges (and +/// their IPv4-mapped IPv6 forms), plus `localhost`, `.local` / `.internal` +/// names, and single-label hostnames (internal service names such as `mongo` +/// or `redis`). +fn is_blocked_host(host: &str) -> bool { + let host = host.trim().trim_end_matches('.').to_ascii_lowercase(); + if host.is_empty() { + return true; + } + if let Ok(ip) = host.parse::() { + // A literal never goes through DNS resolution, so `PublicOnlyResolver` + // never sees it — this text check is its only line of defense, and it + // uses the same classification as the resolver. + return !is_public_ip(ip); + } + if host == "localhost" || host.ends_with(".local") || host.ends_with(".internal") { + return true; + } + // A single-label name is an internal-service name, not a public domain. + !host.contains('.') +} + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinymemory-integrations/src/sources/fetch/ssrf/mod_tests.rs b/crates/tinymemory-integrations/src/sources/fetch/ssrf/mod_tests.rs new file mode 100644 index 00000000..98f02183 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/fetch/ssrf/mod_tests.rs @@ -0,0 +1,236 @@ +//! Tests for the SSRF guard: the address classifier, host and URL policy, +//! the capped body reader and the hardened client. + +use super::*; + +async fn local_response(response: &'static [u8]) -> reqwest::Response { + use tokio::io::{AsyncReadExt, AsyncWriteExt}; + + let listener = tokio::net::TcpListener::bind((std::net::Ipv4Addr::LOCALHOST, 0)) + .await + .expect("bind controlled server"); + let address = listener.local_addr().expect("server address"); + let server = tokio::spawn(async move { + let (mut stream, _) = listener.accept().await.expect("accept request"); + let mut request = [0_u8; 1024]; + let _ = stream.read(&mut request).await.expect("read request"); + stream.write_all(response).await.expect("write response"); + }); + let received = reqwest::get(format!("http://{address}/")) + .await + .expect("controlled response"); + server.await.expect("server task"); + received +} + +// ── SSRF guard ────────────────────────────────────────────────────── + +#[test] +fn is_url_allowed_accepts_public_http_urls() { + assert!(is_url_allowed( + &reqwest::Url::parse("https://example.com").unwrap() + )); + assert!(is_url_allowed( + &reqwest::Url::parse("http://example.com/x").unwrap() + )); + assert!(is_url_allowed( + &reqwest::Url::parse("https://sub.example.com").unwrap() + )); + assert!(is_url_allowed( + &reqwest::Url::parse("https://8.8.8.8").unwrap() + )); +} + +#[test] +fn is_url_allowed_rejects_private_and_internal_targets() { + // Private / loopback / link-local IP literals. + assert!(!is_url_allowed( + &reqwest::Url::parse("http://127.0.0.1").unwrap() + )); + assert!(!is_url_allowed( + &reqwest::Url::parse("http://10.0.0.1").unwrap() + )); + assert!(!is_url_allowed( + &reqwest::Url::parse("http://192.168.1.1").unwrap() + )); + assert!(!is_url_allowed( + &reqwest::Url::parse("http://169.254.169.254").unwrap() + )); + assert!(!is_url_allowed( + &reqwest::Url::parse("http://[::1]").unwrap() + )); + // Internal service names and local-only names. + assert!(!is_url_allowed( + &reqwest::Url::parse("http://localhost").unwrap() + )); + assert!(!is_url_allowed( + &reqwest::Url::parse("http://mongo").unwrap() + )); + assert!(!is_url_allowed( + &reqwest::Url::parse("http://service.internal").unwrap() + )); + // Non-http scheme. + assert!(!is_url_allowed( + &reqwest::Url::parse("ftp://example.com").unwrap() + )); + assert!(!is_url_allowed( + &reqwest::Url::parse("file:///etc/passwd").unwrap() + )); +} + +#[test] +fn is_blocked_host_rejects_ip_ranges_and_local_names() { + let blocked = [ + "127.0.0.1", + "0.0.0.0", + "10.0.0.1", + "172.16.0.1", + "192.168.0.1", + "169.254.169.254", + "100.64.0.1", // CGNAT + "192.0.0.1", // IETF protocol assignments + "localhost", + "foo.local", + "bar.internal", + "mongo", + "::1", + "fc00::1", // unique-local + "fe80::1", // link-local + "::ffff:127.0.0.1", // IPv4-mapped loopback literal + "::ffff:10.0.0.1", // IPv4-mapped private literal + "::ffff:169.254.169.254", // IPv4-mapped link-local / cloud metadata + // Special IPv4 literals that are not globally routable: multicast, + // broadcast, documentation, benchmarking, and reserved ranges. A + // literal never goes through DNS resolution, so the text check is the + // only line of defense for these. + "224.0.0.1", // multicast + "255.255.255.255", // broadcast + "192.0.2.1", // documentation + "198.51.100.1", // documentation + "203.0.113.1", // documentation + "198.18.0.1", // benchmarking + "240.0.0.1", // reserved + ]; + for host in blocked { + assert!(is_blocked_host(host), "expected {host:?} to be blocked"); + } +} + +#[test] +fn is_blocked_host_accepts_public_hosts() { + let allowed = [ + "8.8.8.8", + "1.1.1.1", + "example.com", + "sub.example.com", + "example.co.uk", + "8.8.8.8.", // trailing dot is normalized away + "EXAMPLE.com", // case-insensitive + "2001:4860:4860::8888", + "::ffff:8.8.8.8", // IPv4-mapped public literal + ]; + for host in allowed { + assert!(!is_blocked_host(host), "expected {host:?} to be allowed"); + } +} + +// ── resolved-address (DNS) SSRF classification ────────────────────── + +fn public_ip(s: &str) -> IpAddr { + s.parse().expect("valid ip literal") +} + +#[test] +fn is_public_ip_rejects_internal_and_special_ranges() { + let blocked = [ + "127.0.0.1", // loopback + "0.0.0.0", // unspecified + "10.0.0.1", // private + "172.16.0.1", // private + "192.168.1.1", // private + "169.254.169.254", // link-local / cloud metadata + "100.64.0.1", // CGNAT + "192.0.0.1", // IETF protocol assignments + "224.0.0.1", // multicast + "255.255.255.255", // broadcast + "192.0.2.1", // documentation + "198.51.100.1", // documentation + "203.0.113.1", // documentation + "198.18.0.1", // benchmarking + "240.0.0.1", // reserved + "::1", // loopback + "::", // unspecified + "fc00::1", // unique-local + "fe80::1", // link-local + "ff00::1", // multicast + "2001:db8::1", // documentation + "::ffff:127.0.0.1", // IPv4-mapped loopback + "::ffff:169.254.169.254", // IPv4-mapped link-local + ]; + for s in blocked { + assert!(!is_public_ip(public_ip(s)), "expected {s:?} to be rejected"); + } +} + +#[test] +fn is_public_ip_accepts_global_addresses() { + let allowed = [ + "8.8.8.8", + "1.1.1.1", + "93.184.216.34", + "2001:4860:4860::8888", + "2606:4700:4700::1111", + "::ffff:8.8.8.8", // IPv4-mapped public + ]; + for s in allowed { + assert!(is_public_ip(public_ip(s)), "expected {s:?} to be allowed"); + } +} + +#[tokio::test] +async fn capped_body_reader_accepts_small_streams_and_enforces_both_size_paths() { + let small = local_response(b"HTTP/1.1 200 OK\r\nContent-Length: 5\r\n\r\nhello").await; + assert_eq!(read_body_capped(small, 5).await.unwrap(), b"hello"); + + let declared = local_response(b"HTTP/1.1 200 OK\r\nContent-Length: 6\r\n\r\nabcdef").await; + let error = read_body_capped(declared, 5) + .await + .expect_err("declared body exceeds cap"); + assert!(error.contains("Content-Length=6")); + + let chunked = local_response( + b"HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\n\r\n3\r\nabc\r\n3\r\ndef\r\n0\r\n\r\n", + ) + .await; + let error = read_body_capped(chunked, 5) + .await + .expect_err("streamed body exceeds cap"); + assert!(error.contains("read 6 bytes")); +} + +#[test] +fn client_builder_installs_the_hardened_policy() { + build_client().expect("hardened HTTP client builds"); +} + +#[test] +fn ipv4_compatible_ipv6_addresses_are_judged_by_their_ipv4_part() { + // `::127.0.0.1` is loopback written the deprecated long way, and + // `::169.254.169.254` is the metadata service; neither is `to_ipv4_mapped`. + for blocked in [ + "::127.0.0.1", + "::169.254.169.254", + "::10.0.0.1", + "::192.168.1.1", + ] { + let ip: std::net::IpAddr = blocked.parse().unwrap(); + assert!(!is_public_ip(ip), "{blocked} must not read as public"); + let url = reqwest::Url::parse(&format!("http://[{blocked}]/")).unwrap(); + assert!( + !is_url_allowed(&url), + "{blocked} must be refused as a fetch target" + ); + } + let public: std::net::IpAddr = "::93.184.216.34".parse().unwrap(); + assert!(is_public_ip(public)); +} diff --git a/crates/tinymemory-integrations/src/sources/items/mod.rs b/crates/tinymemory-integrations/src/sources/items/mod.rs index 6c869c1b..40205a2e 100644 --- a/crates/tinymemory-integrations/src/sources/items/mod.rs +++ b/crates/tinymemory-integrations/src/sources/items/mod.rs @@ -8,6 +8,10 @@ //! | Kind | Item | Metadata filled | //! | --- | --- | --- | //! | folder, file | document | `workspace`, `folder` (containing directory), `file_path`, `language` (by extension), `observed_at` (mtime), `mime` | +//! | github | document | `repo` (`owner/name`), `commit` (commit items), `url` (issues and PRs), `observed_at` | +//! | link | document | `url` | +//! | rss | document | `url` (the entry's link), `observed_at` (published) | +//! | composio | document | `tags = [toolkit]` (payloads: see [`crate::sources::composio`]) | //! | conversation | conversation | `workspace`, `thread_id`, `turns`, `observed_at` (last turn) | //! //! Every document body is markdown, converted through `crate::documents`: @@ -34,16 +38,23 @@ use crate::sources::readers::local_file::LocalFile; use crate::sources::types::{ContentType, MemorySourceEntry, SourceContent, SourceKind}; /// Metadata naming `entry` as the source: `source.kind` is the entry's kind -/// mapped onto the contract, `source.id` its id. +/// mapped onto the contract, `source.id` its id. A Composio entry also gets +/// its toolkit as a tag. #[must_use] pub fn base_meta(entry: &MemorySourceEntry) -> MemoryMeta { - MemoryMeta { + let mut meta = MemoryMeta { source: SourceRef { kind: entry.kind.api_kind(), id: Some(entry.id.clone()), }, ..MemoryMeta::default() + }; + if entry.kind == SourceKind::Composio + && let Some(toolkit) = entry.toolkit.as_deref().filter(|t| !t.is_empty()) + { + meta.tags = vec![toolkit.to_string()]; } + meta } /// Convert a local file and wrap it as a document. @@ -160,6 +171,19 @@ pub fn content_item( meta.observed_at = millis(updated_at_ms); let metadata = &content.metadata; match entry.kind { + SourceKind::GithubRepo => github_meta(&mut meta, &content.id, metadata), + SourceKind::RssFeed => { + meta.url = string_field(metadata, "link"); + if let Some(published) = string_field(metadata, "published") + .as_deref() + .and_then(parse_feed_time) + { + meta.observed_at = Some(published); + } + } + SourceKind::WebPage => { + meta.url = string_field(metadata, "url").or_else(|| Some(content.id.clone())); + } SourceKind::Folder => { if let Some(root) = entry.path.as_deref() { let path = Path::new(root).join(&content.id); @@ -173,6 +197,7 @@ pub fn content_item( meta.language = language_for_path(&content.id).map(str::to_string); } SourceKind::Conversation => meta.thread_id = Some(content.id.clone()), + SourceKind::Composio => {} } Ok(StoreItem::Document { @@ -183,6 +208,30 @@ pub fn content_item( }) } +/// GitHub metadata from a reader item id (`commit:`, `issue:`, +/// `pr:`) and its content metadata (`owner`, `repo`, `sha`, `number`). +fn github_meta(meta: &mut MemoryMeta, item_id: &str, metadata: &serde_json::Value) { + let owner = string_field(metadata, "owner"); + let repo = string_field(metadata, "repo"); + let slug = owner + .zip(repo) + .map(|(owner, repo)| format!("{owner}/{repo}")); + meta.repo = slug.clone(); + let number = metadata + .get("number") + .and_then(serde_json::Value::as_u64) + .map(|n| n.to_string()); + if let Some(sha) = item_id.strip_prefix("commit:") { + meta.commit = string_field(metadata, "sha").or_else(|| Some(sha.to_string())); + } else if let Some(n) = item_id.strip_prefix("issue:") { + let n = number.unwrap_or_else(|| n.to_string()); + meta.url = slug.map(|slug| format!("https://github.com/{slug}/issues/{n}")); + } else if let Some(n) = item_id.strip_prefix("pr:") { + let n = number.unwrap_or_else(|| n.to_string()); + meta.url = slug.map(|slug| format!("https://github.com/{slug}/pull/{n}")); + } +} + /// A non-empty string field of a JSON object. fn string_field(value: &serde_json::Value, key: &str) -> Option { value @@ -198,6 +247,14 @@ fn millis(value: Option) -> Option> { value.and_then(|ms| Utc.timestamp_millis_opt(ms).single()) } +/// An RSS `pubDate` (RFC 2822) or Atom `updated` (RFC 3339) timestamp. +fn parse_feed_time(text: &str) -> Option> { + DateTime::parse_from_rfc2822(text) + .or_else(|_| DateTime::parse_from_rfc3339(text)) + .ok() + .map(|at| at.with_timezone(&Utc)) +} + /// One item a [`collect_items`] pass could not turn into a `StoreItem`. #[derive(Debug)] pub struct SkippedItem { diff --git a/crates/tinymemory-integrations/src/sources/items/mod_tests.rs b/crates/tinymemory-integrations/src/sources/items/mod_tests.rs index 17d99dbf..b050d8ea 100644 --- a/crates/tinymemory-integrations/src/sources/items/mod_tests.rs +++ b/crates/tinymemory-integrations/src/sources/items/mod_tests.rs @@ -51,16 +51,124 @@ fn base_meta_names_the_source_by_contract_kind_and_entry_id() { for (kind, api) in [ (SourceKind::Folder, Api::Folder), (SourceKind::File, Api::File), + (SourceKind::WebPage, Api::Link), + (SourceKind::GithubRepo, Api::Github), + (SourceKind::RssFeed, Api::Rss), + (SourceKind::Composio, Api::Composio), (SourceKind::Conversation, Api::Conversation), ] { let meta = base_meta(&entry(kind)); assert_eq!(meta.source.kind, api); assert_eq!(meta.source.id.as_deref(), Some("src_test")); } + + let mut composio = entry(SourceKind::Composio); + composio.toolkit = Some("gmail".into()); + assert_eq!(base_meta(&composio).tags, vec!["gmail".to_string()]); +} + +#[test] +fn github_commit_items_carry_repo_and_commit() { + let item = content_item( + &entry(SourceKind::GithubRepo), + content( + "commit:abc123", + "# Fix\n\nbody", + ContentType::Markdown, + serde_json::json!({ "owner": "acme", "repo": "widgets", "sha": "abc123full" }), + ), + Some(1_700_000_000_000), + ) + .unwrap(); + let (title, body, mime, meta) = document_parts(&item); + assert_eq!(title.as_deref(), Some("title of commit:abc123")); + assert_eq!(body, "# Fix\n\nbody"); + assert_eq!(mime.as_deref(), Some("text/markdown")); + assert_eq!(meta.source.kind, Api::Github); + assert_eq!(meta.repo.as_deref(), Some("acme/widgets")); + assert_eq!(meta.commit.as_deref(), Some("abc123full")); + assert_eq!(meta.url, None); + assert_eq!( + meta.observed_at.map(|at| at.timestamp()), + Some(1_700_000_000) + ); +} + +#[test] +fn github_issues_and_prs_carry_their_url() { + let metadata = serde_json::json!({ "owner": "acme", "repo": "widgets", "number": 7 }); + let issue = content_item( + &entry(SourceKind::GithubRepo), + content("issue:7", "text", ContentType::Markdown, metadata.clone()), + None, + ) + .unwrap(); + assert_eq!( + issue.meta().url.as_deref(), + Some("https://github.com/acme/widgets/issues/7") + ); + assert_eq!(issue.meta().commit, None); + + let pr = content_item( + &entry(SourceKind::GithubRepo), + content("pr:7", "text", ContentType::Markdown, metadata), + None, + ) + .unwrap(); + assert_eq!( + pr.meta().url.as_deref(), + Some("https://github.com/acme/widgets/pull/7") + ); +} + +#[test] +fn rss_items_take_the_entry_link_and_publication_time() { + let item = content_item( + &entry(SourceKind::RssFeed), + content( + "guid-1", + "

Hello feed

", + ContentType::Html, + serde_json::json!({ + "link": "https://blog.example.com/post", + "published": "Tue, 21 May 2024 12:00:00 +0000" + }), + ), + None, + ) + .unwrap(); + let (_, body, mime, meta) = document_parts(&item); + assert_eq!(body, "Hello **feed**", "html is converted to markdown"); + assert_eq!(mime.as_deref(), Some("text/markdown")); + assert_eq!(meta.source.kind, Api::Rss); + assert_eq!(meta.url.as_deref(), Some("https://blog.example.com/post")); + assert_eq!( + meta.observed_at.map(|at| at.to_rfc3339()), + Some("2024-05-21T12:00:00+00:00".to_string()) + ); +} + +#[test] +fn link_items_take_the_page_url() { + let item = content_item( + &entry(SourceKind::WebPage), + content( + "https://example.com/page", + "plain words", + ContentType::Plaintext, + serde_json::json!({ "url": "https://example.com/page" }), + ), + None, + ) + .unwrap(); + let (_, _, mime, meta) = document_parts(&item); + assert_eq!(meta.source.kind, Api::Link); + assert_eq!(meta.url.as_deref(), Some("https://example.com/page")); + assert_eq!(mime.as_deref(), Some("text/plain")); } #[test] -fn reader_content_for_local_kinds_fills_what_it_can() { +fn reader_content_for_local_kinds_and_composio_fills_what_it_can() { let mut folder = entry(SourceKind::Folder); folder.path = Some("/notes".into()); let item = content_item( @@ -104,12 +212,27 @@ fn reader_content_for_local_kinds_fills_what_it_can() { ) .unwrap(); assert_eq!(conversation.meta().thread_id.as_deref(), Some("t1")); + + let mut composio = entry(SourceKind::Composio); + composio.toolkit = Some("slack".into()); + let item = content_item( + &composio, + content( + "c1", + "sync data", + ContentType::Plaintext, + serde_json::json!({}), + ), + None, + ) + .unwrap(); + assert_eq!(item.meta().tags, vec!["slack".to_string()]); } #[test] fn content_that_converts_to_nothing_is_refused() { let error = content_item( - &entry(SourceKind::Folder), + &entry(SourceKind::WebPage), content( "u", "", @@ -355,7 +478,7 @@ struct BrokenReader; #[async_trait] impl SourceReader for BrokenReader { fn kind(&self) -> SourceKind { - SourceKind::Folder + SourceKind::WebPage } async fn list_items(&self, _: &MemorySourceEntry, _: &Path) -> Result> { @@ -371,7 +494,7 @@ impl SourceReader for BrokenReader { async fn a_failed_listing_fails_the_collect_pass() { let error = collect_items( &BrokenReader, - &entry(SourceKind::Folder), + &entry(SourceKind::WebPage), Path::new("/unused"), &ConverterChain::default(), ) @@ -388,7 +511,7 @@ struct FixedReader; #[async_trait] impl SourceReader for FixedReader { fn kind(&self) -> SourceKind { - SourceKind::Folder + SourceKind::WebPage } async fn list_items(&self, _: &MemorySourceEntry, _: &Path) -> Result> { @@ -413,14 +536,14 @@ impl SourceReader for FixedReader { async fn the_default_read_store_item_maps_reader_content() { let collected = collect_items( &FixedReader, - &entry(SourceKind::Folder), + &entry(SourceKind::WebPage), Path::new("/unused"), &ConverterChain::default(), ) .await .unwrap(); let meta = collected.items[0].meta(); - assert_eq!(meta.source.kind, Api::Folder); + assert_eq!(meta.url.as_deref(), Some("https://example.com")); assert_eq!( meta.observed_at.map(|at| at.timestamp_millis()), Some(1_000) diff --git a/crates/tinymemory-integrations/src/sources/mod.rs b/crates/tinymemory-integrations/src/sources/mod.rs index 3e3851e7..dad654b8 100644 --- a/crates/tinymemory-integrations/src/sources/mod.rs +++ b/crates/tinymemory-integrations/src/sources/mod.rs @@ -1,17 +1,23 @@ -//! Source readers for TinyMemory: turn a folder, a file or a local conversation +//! Source readers for TinyMemory: turn a folder, a file, a web page, a GitHub +//! repository, an RSS feed, a Composio toolkit payload or a local conversation //! into [`StoreItem`](tinymemory_api::StoreItem)s. //! //! - **Configuration** — what a source *is* ([`MemorySourceEntry`], keyed by //! [`SourceKind`], checked by [`MemorySourceEntry::validate`]). Where the //! host stores its sources, and how it edits them, is the host's business. //! - **Readers** — [`readers::SourceReader`] lists a source's items and reads -//! one. Every reader (folder, file, conversation) is local; none touches the -//! network. +//! one. Local readers (folder, file, conversation) are always compiled; the +//! network readers (GitHub, RSS, web page) and `fetch` sit behind the +//! `sources-network` feature; RSS and web pages fetch through `fetch`, +//! behind its one SSRF guard (`fetch::ssrf`). //! - **Items** — [`items`] maps reader output to `StoreItem`s with //! [`MemoryMeta`](tinymemory_api::MemoryMeta) filled per kind; every text //! body is converted to markdown through [`crate::documents`]. +//! - **Composio** — [`composio`] normalises toolkit payloads (Gmail, Slack, +//! GitHub, Linear, Notion, ClickUp) and maps them to items. //! -//! Scheduling stays with the host: this module reads when asked. +//! Scheduling, credentials and egress budgets stay with the host: this module +//! reads when asked. //! //! # Example //! @@ -51,9 +57,16 @@ //! //! # Feature flags //! -//! - `sources` — everything above. Implies `documents`; links no HTTP stack. +//! - `sources` — everything above except the network pieces. Implies +//! `documents`; links no HTTP stack. +//! - `sources-network` — the GitHub, RSS and web-page readers, `fetch`, +//! `readers::reader_for_request` and the SSRF guard. Without it, a host that +//! only reads local sources links no HTTP stack. +pub mod composio; pub mod error; +#[cfg(feature = "sources-network")] +pub mod fetch; pub mod items; pub mod readers; pub mod types; diff --git a/crates/tinymemory-integrations/src/sources/readers/composio/mod.rs b/crates/tinymemory-integrations/src/sources/readers/composio/mod.rs new file mode 100644 index 00000000..5d0c3d84 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/readers/composio/mod.rs @@ -0,0 +1,78 @@ +//! Composio source reader — a placeholder over the provider pipeline. +//! +//! Composio data does not arrive item by item: the host runs toolkit actions +//! with its credentials and hands the responses to [`crate::sources::composio`], which +//! normalises them and maps them to `StoreItem`s. For a Composio source, +//! `list_items` returns the connection as one sync target and `read_item` +//! describes that pipeline. The reader exists so `reader_for_request` can hand +//! out a reader for every source kind uniformly. + +use std::path::Path; + +use async_trait::async_trait; + +use super::SourceReader; +use crate::sources::error::Result; +use crate::sources::types::{ + ContentType, MemorySourceEntry, SourceContent, SourceItem, SourceKind, +}; + +/// Lists a Composio connection as a single sync target. +/// +/// Composio data arrives through the provider sync pipeline rather than +/// item-by-item, so `read_item` returns a description of that rather than +/// content. The reader exists so `reader_for_request` can serve every source +/// kind uniformly. +#[derive(Debug, Clone, Copy, Default)] +pub struct ComposioReader; + +#[async_trait] +impl SourceReader for ComposioReader { + fn kind(&self) -> SourceKind { + SourceKind::Composio + } + + async fn list_items( + &self, + source: &MemorySourceEntry, + _workspace: &Path, + ) -> Result> { + let toolkit = source.toolkit.as_deref().unwrap_or("unknown"); + let connection_id = source.connection_id.as_deref().unwrap_or("unknown"); + + log::debug!( + "[memory_sources:composio] list_items toolkit={toolkit} connection_id={connection_id}" + ); + + Ok(vec![SourceItem { + id: connection_id.to_string(), + title: format!("{toolkit} connection"), + updated_at_ms: None, + }]) + } + + async fn read_item( + &self, + source: &MemorySourceEntry, + item_id: &str, + _workspace: &Path, + ) -> Result { + let toolkit = source.toolkit.as_deref().unwrap_or("unknown"); + Ok(SourceContent { + id: item_id.to_string(), + title: format!("{toolkit} sync data"), + body: format!( + "Composio {toolkit} data is synced via the provider sync pipeline, not read item-by-item." + ), + content_type: ContentType::Plaintext, + metadata: serde_json::json!({ + "toolkit": toolkit, + "connection_id": source.connection_id, + }), + }) + } +} + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinymemory-integrations/src/sources/readers/composio/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/composio/mod_tests.rs new file mode 100644 index 00000000..d139ff15 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/readers/composio/mod_tests.rs @@ -0,0 +1,51 @@ +//! Tests for the surrounding module. + +use super::*; +use std::path::Path; + +fn test_source() -> MemorySourceEntry { + MemorySourceEntry { + id: "src_1".into(), + kind: SourceKind::Composio, + label: "Gmail".into(), + enabled: true, + toolkit: Some("gmail".into()), + connection_id: Some("cmp_123".into()), + path: None, + glob: None, + url: None, + branch: None, + paths: Vec::new(), + max_items: None, + max_commits: None, + max_issues: None, + max_prs: None, + selector: None, + max_tokens_per_sync: None, + max_cost_per_sync_usd: None, + sync_depth_days: None, + } +} + +#[tokio::test] +async fn list_items_returns_connection_as_item() { + let reader = ComposioReader; + let items = reader + .list_items(&test_source(), Path::new(".")) + .await + .unwrap(); + assert_eq!(items.len(), 1); + assert_eq!(items[0].id, "cmp_123"); +} + +#[tokio::test] +async fn read_item_describes_the_provider_pipeline() { + let content = ComposioReader + .read_item(&test_source(), "cmp_123", Path::new(".")) + .await + .unwrap(); + assert_eq!(content.title, "gmail sync data"); + assert!(content.body.contains("provider sync pipeline")); + assert_eq!(content.metadata["toolkit"], "gmail"); + assert_eq!(content.metadata["connection_id"], "cmp_123"); +} diff --git a/crates/tinymemory-integrations/src/sources/readers/conversation/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/conversation/mod_tests.rs index 08535ddf..8f49976f 100644 --- a/crates/tinymemory-integrations/src/sources/readers/conversation/mod_tests.rs +++ b/crates/tinymemory-integrations/src/sources/readers/conversation/mod_tests.rs @@ -11,8 +11,18 @@ fn conversation_source() -> MemorySourceEntry { kind: SourceKind::Conversation, label: "Conversations".into(), enabled: true, + toolkit: None, + connection_id: None, path: None, glob: None, + url: None, + branch: None, + paths: Vec::new(), + max_commits: None, + max_issues: None, + max_prs: None, + max_items: None, + selector: None, max_tokens_per_sync: None, max_cost_per_sync_usd: None, sync_depth_days: None, diff --git a/crates/tinymemory-integrations/src/sources/readers/folder/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/folder/mod_tests.rs index 280fdaaf..32060ca7 100644 --- a/crates/tinymemory-integrations/src/sources/readers/folder/mod_tests.rs +++ b/crates/tinymemory-integrations/src/sources/readers/folder/mod_tests.rs @@ -11,8 +11,18 @@ fn folder_source(path: &str) -> MemorySourceEntry { kind: SourceKind::Folder, label: "Test folder".into(), enabled: true, + toolkit: None, + connection_id: None, path: Some(path.into()), glob: None, + url: None, + branch: None, + paths: Vec::new(), + max_commits: None, + max_issues: None, + max_prs: None, + max_items: None, + selector: None, max_tokens_per_sync: None, max_cost_per_sync_usd: None, sync_depth_days: None, diff --git a/crates/tinymemory-integrations/src/sources/readers/github/api/mod.rs b/crates/tinymemory-integrations/src/sources/readers/github/api/mod.rs new file mode 100644 index 00000000..0ec4a8b6 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/readers/github/api/mod.rs @@ -0,0 +1,391 @@ +//! `gh` CLI + REST API helpers for the GitHub reader. +//! +//! [`fetch_github`] prefers the authenticated `gh api` path and falls back to +//! the unauthenticated REST API. Commit list/read helpers live here; issue and +//! pull-request list/read helpers live in the sibling `super::issues` module, +//! and commit reads additionally have a local `git` path in the sibling +//! `super::git` module. +//! +//! Branch/path filters are honored on the commits list: `sha=` and +//! `path=` query params narrow what the API returns to the configured +//! scope. + +use std::collections::HashSet; + +use crate::sources::types::{ContentType, SourceContent, SourceItem}; + +use super::types::GhCommit; +use super::{GH_CLI_TIMEOUT, parse_iso_ts}; + +// Keep the production transport at its established source locations. Coverage +// tools merge regions by source coordinate, so moving these functions would +// turn otherwise identical regions into apparent duplicate production lines. +// Only the deterministic response queue belongs in +// the selected external module below; the actual transport remains here. +// +// The deliberately expanded explanation also occupies the source range that +// previously held that queue. That keeps historical and independently cached +// compilations aligned while making the executable test seam fully external. +// Coverage therefore measures one production transport, regardless of whether +// the crate is linked into a unit-test or public-integration-test binary. +// Its behavior is unchanged; only the test override storage moved. +// +#[cfg(not(test))] +#[path = "transport_override.rs"] +mod response_override; +#[cfg(test)] +#[path = "transport_tests.rs"] +mod response_override; + +#[cfg(test)] +pub(super) use response_override::with_test_responses; + +/// GitHub REST API maximum page size (`per_page`). +pub(super) const GH_PAGE_SIZE: u32 = 100; + +/// Hard ceiling on pagination loops so a misbehaving API (always returning a +/// full page) can never spin forever even if `max` is enormous. +pub(super) const GH_MAX_PAGES: u32 = 1000; + +/// Run `gh ` and return stdout as UTF-8. +pub(super) async fn gh_json(args: &[&str]) -> Result { + let output = tokio::time::timeout( + GH_CLI_TIMEOUT, + tokio::process::Command::new("gh").args(args).output(), + ) + .await + .map_err(|_| format!("gh command timed out after {}s", GH_CLI_TIMEOUT.as_secs()))? + .map_err(|e| format!("gh command failed: {e}"))?; + + if !output.status.success() { + let stderr = String::from_utf8_lossy(&output.stderr); + return Err(format!("gh exited {}: {stderr}", output.status)); + } + + String::from_utf8(output.stdout).map_err(|e| format!("gh output not utf8: {e}")) +} + +/// Unauthenticated GET against the GitHub REST API. +pub(super) async fn api_get(path: &str) -> Result { + let url = format!("https://api.github.com{path}"); + let client = reqwest::Client::builder() + .timeout(std::time::Duration::from_secs(20)) + .build() + .map_err(|e| format!("failed to build GitHub client: {e}"))?; + let resp = client + .get(&url) + .header("User-Agent", "openhuman") + .header("Accept", "application/vnd.github.v3+json") + .send() + .await + .map_err(|e| format!("GitHub API request failed: {e}"))?; + + if !resp.status().is_success() { + let status = resp.status(); + let body = resp.text().await.unwrap_or_default(); + return Err(format!("GitHub API returned {status}: {body}")); + } + + resp.text() + .await + .map_err(|e| format!("failed to read response: {e}")) +} + +/// Try `gh api` first, fall back to unauthenticated REST API. +pub(super) async fn fetch_github(api_path: &str, use_gh: bool) -> Result { + // The response selection intentionally stays at the former interception + // range. Keeping later transport regions aligned prevents LLVM from + // treating identical code linked into different test binaries as distinct + // source regions. The selected implementation itself remains external. + // + if let Some(response) = response_override::take_response(api_path) { + return response; + } + if use_gh { + match gh_json(&["api", api_path]).await { + Ok(s) => return Ok(s), + Err(e) => { + tracing::debug!( + error = %e, + path = %api_path, + "[memory_sources:github] gh failed, falling back to API" + ); + } + } + } + api_get(&format!("/{api_path}")).await +} + +/// Fetch up to `max` rows from a paginated GitHub list endpoint. +/// +/// Walks `?per_page=100&page=N` with a constant page size — GitHub's +/// offset-based pagination is per_page-relative, so shrinking the page size +/// mid-walk would re-window the offsets and silently skip rows (e.g. `max=150` +/// would fetch items 51-100 a second time instead of 101-150). Iteration stops +/// once `max` rows are collected or the API returns a short page (the last +/// page); `extra_query` is appended verbatim (e.g. `"state=all"`). The result +/// is truncated to exactly `max`. +pub(super) async fn fetch_all_pages( + owner: &str, + repo: &str, + resource: &str, + extra_query: &str, + max: u32, + use_gh: bool, +) -> Result, String> { + let fetch = |page: u32| async_fetch_page(page, owner, repo, resource, extra_query, use_gh); + collect_pages(resource, max, fetch).await +} + +/// Fetch one page's raw JSON at a constant [`GH_PAGE_SIZE`]. +async fn async_fetch_page( + page: u32, + owner: &str, + repo: &str, + resource: &str, + extra_query: &str, + use_gh: bool, +) -> Result { + let mut path = format!("repos/{owner}/{repo}/{resource}?per_page={GH_PAGE_SIZE}&page={page}"); + if !extra_query.is_empty() { + path.push('&'); + path.push_str(extra_query); + } + fetch_github(&path, use_gh).await +} + +/// Core pagination walk, split out from [`fetch_all_pages`] so the loop is +/// unit-testable with a fake fetch instead of a live GitHub API. +/// +/// `fetch` maps a 1-based page number to the raw JSON for that page. The page +/// size the fetch encodes must stay constant across pages — see +/// [`fetch_all_pages`] for why shrinking it mid-walk skips rows. +pub(super) async fn collect_pages( + label: &str, + max: u32, + mut fetch: F, +) -> Result, String> +where + T: serde::de::DeserializeOwned, + F: FnMut(u32) -> Fut, + Fut: std::future::Future>, +{ + let mut out: Vec = Vec::new(); + let mut page = 1u32; + + while (out.len() as u32) < max && page <= GH_MAX_PAGES { + let json_str = fetch(page).await?; + let batch: Vec = serde_json::from_str(&json_str) + .map_err(|e| format!("parse {label} page {page}: {e}"))?; + let got = batch.len(); + out.extend(batch); + + // Short page ⇒ no more rows upstream. + if got < GH_PAGE_SIZE as usize { + break; + } + page += 1; + } + + out.truncate(max as usize); + Ok(out) +} + +/// Percent-encode a branch or path value for use as a URL query parameter. +/// +/// RFC 3986 unreserved characters and `/` are kept as-is; everything else +/// (`&`, `=`, `#`, `?`, `%`, spaces, …) is percent-encoded so a value cannot +/// be misparsed as query syntax and corrupt the filter. `/` is left intact +/// because it is legal in a query component and GitHub's commits `sha`/`path` +/// filters expect the common `path=src/lib.rs` shape unencoded. +fn percent_encode_query(s: &str) -> String { + let mut out = String::with_capacity(s.len()); + for b in s.bytes() { + match b { + b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9' | b'-' | b'.' | b'_' | b'~' | b'/' => { + out.push(b as char); + } + _ => out.push_str(&format!("%{b:02X}")), + } + } + out +} + +/// Build the `extra_query` strings for the commits endpoint — one per +/// configured path (the endpoint accepts a single `path` filter), each +/// carrying the branch's `sha` when set. An empty path list means "no path +/// filter" (a single query carrying only the branch filter, if any). +/// Branch/path values are percent-encoded so `&`, `#`, `=` inside them cannot +/// corrupt the query. Extracted as a pure helper so the filter wiring is +/// unit-testable. +pub(super) fn commit_list_queries(branch: Option<&str>, paths: &[String]) -> Vec { + let sha_q = branch + .filter(|b| !b.is_empty()) + .map(|b| format!("sha={}", percent_encode_query(b))); + let path_qs: Vec = if paths.is_empty() { + vec![String::new()] + } else { + paths + .iter() + .map(|p| format!("path={}", percent_encode_query(p))) + .collect() + }; + path_qs + .into_iter() + .map(|path_q| { + let mut extra = String::new(); + if let Some(q) = &sha_q { + extra.push_str(q); + } + if !path_q.is_empty() { + if !extra.is_empty() { + extra.push('&'); + } + extra.push_str(&path_q); + } + extra + }) + .collect() +} + +/// List commits via the REST `commits` endpoint (fallback when local git is +/// unavailable). +/// +/// A configured `branch` is sent as `sha=`. The GitHub commits +/// endpoint accepts a single `path` filter, so multiple configured paths are +/// fetched one query each (each bounded at `max` so the walk stays finite), +/// merged and deduped by sha, ordered by commit time, and truncated to `max`. +pub(super) async fn list_commits_api( + owner: &str, + repo: &str, + max: u32, + use_gh: bool, + branch: Option<&str>, + paths: &[String], +) -> Result, String> { + let mut batches: Vec> = Vec::new(); + for extra in commit_list_queries(branch, paths) { + let commits: Vec = + fetch_all_pages(owner, repo, "commits", &extra, max, use_gh).await?; + batches.push(commits); + } + Ok(merge_commit_batches(batches, max)) +} + +/// Merge per-path commit batches into the final item list. +/// +/// The GitHub commits endpoint accepts a single `path` filter, so multiple +/// configured paths are fetched one query each; every path must be walked +/// (not just until the first fills `max`) or later paths are silently starved. +/// Batches are deduped by sha, ordered newest-first by commit time, and +/// truncated to `max` — the same union semantics the local `git log` path gives +/// a multi-pathspec walk. Extracted as a pure helper so the merge is +/// unit-testable without a live API. +pub(super) fn merge_commit_batches(batches: Vec>, max: u32) -> Vec { + let mut out: Vec = Vec::new(); + let mut seen: HashSet = HashSet::new(); + for commits in batches { + for c in commits { + if seen.insert(c.sha.clone()) { + let title = c.commit.message.lines().next().unwrap_or("").to_string(); + let ts = c + .commit + .committer + .as_ref() + .and_then(|a| a.date.as_deref()) + .and_then(parse_iso_ts); + out.push(SourceItem { + id: format!("commit:{}", c.sha), + title, + updated_at_ms: ts, + }); + } + } + } + // Each path's query returns its commits newest-first, but the merged set + // is path-ordered. Re-sort by commit time (newest first) so the global + // truncation keeps the most recent commits across all configured paths. + out.sort_by_key(|b| std::cmp::Reverse(b.updated_at_ms)); + out.truncate(max as usize); + out +} + +/// Read one commit via the REST API (fallback when local git is unavailable). +pub(super) async fn read_commit_api( + owner: &str, + repo: &str, + sha: &str, + use_gh: bool, +) -> Result { + let json_str = fetch_github(&format!("repos/{owner}/{repo}/commits/{sha}"), use_gh).await?; + + let commit: GhCommit = + serde_json::from_str(&json_str).map_err(|e| format!("parse commit: {e}"))?; + + let author = commit + .commit + .author + .as_ref() + .map(|a| { + format!( + "{} <{}>", + a.name.as_deref().unwrap_or("unknown"), + a.email.as_deref().unwrap_or("") + ) + }) + .unwrap_or_default(); + + // GitHub login of the committer, rendered as an `@handle` so the + // entity extractor registers it as a `handle:` entity in the memory + // tree (unique committers become first-class entities). + let handle = commit + .author + .as_ref() + .map(|u| format!("@{}", u.login)) + .unwrap_or_default(); + + let date = commit + .commit + .committer + .as_ref() + .and_then(|a| a.date.as_deref()) + .unwrap_or("unknown"); + + let title = commit + .commit + .message + .lines() + .next() + .unwrap_or("") + .to_string(); + + let author_line = if handle.is_empty() { + author.clone() + } else { + format!("{author} ({handle})") + }; + + let body = format!( + "# Commit: {title}\n\n\ + **SHA:** {sha}\n\ + **Author:** {author_line}\n\ + **Date:** {date}\n\n\ + ## Message\n\n\ + {}", + commit.commit.message, + ); + + Ok(SourceContent { + id: format!("commit:{sha}"), + title, + body, + content_type: ContentType::Markdown, + metadata: serde_json::json!({ + "owner": owner, + "repo": repo, + "sha": sha, + "author": author, + "author_handle": commit.author.as_ref().map(|u| u.login.clone()), + }), + }) +} diff --git a/crates/tinymemory-integrations/src/sources/readers/github/api/transport_override.rs b/crates/tinymemory-integrations/src/sources/readers/github/api/transport_override.rs new file mode 100644 index 00000000..b43ecedd --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/readers/github/api/transport_override.rs @@ -0,0 +1,5 @@ +//! Production transport override: live GitHub requests are never intercepted. + +pub(super) fn take_response(_api_path: &str) -> Option> { + None +} diff --git a/crates/tinymemory-integrations/src/sources/readers/github/api/transport_tests.rs b/crates/tinymemory-integrations/src/sources/readers/github/api/transport_tests.rs new file mode 100644 index 00000000..da0d7b0d --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/readers/github/api/transport_tests.rs @@ -0,0 +1,34 @@ +//! Task-local deterministic GitHub transport used by reader tests. + +tokio::task_local! { + static TEST_RESPONSES: std::cell::RefCell< + std::collections::VecDeque> + >; +} + +/// Run a future with a task-local sequence of GitHub responses. +pub(crate) async fn with_test_responses( + responses: Vec>, + future: F, +) -> F::Output +where + F: std::future::Future, +{ + TEST_RESPONSES + .scope(std::cell::RefCell::new(responses.into()), future) + .await +} + +/// Return the next deterministic response, or no override outside its scope. +pub(crate) fn take_response(api_path: &str) -> Option> { + TEST_RESPONSES + .try_with(|responses| { + Some(responses.borrow_mut().pop_front().unwrap_or_else(|| { + Err(format!( + "no deterministic GitHub response queued for {api_path}" + )) + })) + }) + .ok() + .flatten() +} diff --git a/crates/tinymemory-integrations/src/sources/readers/github/git/mod.rs b/crates/tinymemory-integrations/src/sources/readers/github/git/mod.rs new file mode 100644 index 00000000..b9b181ae --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/readers/github/git/mod.rs @@ -0,0 +1,320 @@ +//! Local bare-clone helpers for the GitHub reader. +//! +//! Commits are listed via a per-repo bare clone (`git log`) rather than the +//! REST API whenever the repo is reachable over git: the clone's refs are a +//! superset of what the API exposes and reads are fully offline after the +//! initial clone/fetch. The clone lives under +//! `workspace/git_cache//.git`. +//! +//! Branch/path filters are honored here: a configured `branch` narrows `git +//! log` to that ref (instead of the bare clone's `HEAD`), and configured +//! `paths` become git pathspecs so commits touching unrelated paths are not +//! ingested. + +use std::path::{Path, PathBuf}; +use std::time::Duration; + +use crate::sources::types::{ContentType, SourceContent, SourceItem}; + +use super::parse_iso_ts; + +/// Timeout for a single `git clone` / `git fetch` (slow on a cold cache). +const GIT_CLONE_TIMEOUT: Duration = Duration::from_secs(120); +/// Timeout for a single `git log` / `git show` (fast, local). +const GIT_LOG_TIMEOUT: Duration = Duration::from_secs(30); + +/// Path to the bare clone for a repo, created lazily under +/// `workspace/git_cache//.git`. +pub(super) fn git_cache_dir(workspace: &Path, owner: &str, repo: &str) -> PathBuf { + workspace + .join("git_cache") + .join(owner) + .join(format!("{repo}.git")) +} + +/// Ensure a bare clone of `owner/repo` exists at `cache_dir` — fetching into +/// an existing clone, cloning fresh when absent. A missing remote (private or +/// renamed repo) surfaces as an error handled by the caller's fallback. +pub(super) async fn ensure_bare_clone( + owner: &str, + repo: &str, + cache_dir: &Path, +) -> Result<(), String> { + if cache_dir.join("HEAD").exists() { + return fetch_existing_bare(cache_dir).await; + } + + let clone_url = format!("https://github.com/{owner}/{repo}.git"); + clone_bare(&clone_url, cache_dir).await +} + +/// `git fetch` into an existing bare clone. +/// +/// The refspec is explicit (`+refs/heads/*:refs/heads/*`): a bare +/// `git clone` records no `remote.origin.fetch` mapping, so a bare `git fetch` +/// without one would only update `FETCH_HEAD` and leave `refs/heads/*` at the +/// initial clone — every later sync would silently miss new GitHub activity. +/// `--prune` also drops local heads the remote has since deleted. +/// +/// After the fetch, `HEAD` is refreshed to the remote's current default branch +/// so an unconfigured sync keeps following the repo's default even when that +/// default changes between clones (see [`refresh_default_branch_head`]). +async fn fetch_existing_bare(cache_dir: &Path) -> Result<(), String> { + tracing::debug!( + cache = %cache_dir.display(), + "[memory_sources:github:git] fetching into existing bare clone" + ); + let output = tokio::time::timeout( + GIT_CLONE_TIMEOUT, + tokio::process::Command::new("git") + .args([ + "fetch", + "--prune", + "--quiet", + "origin", + "+refs/heads/*:refs/heads/*", + ]) + .current_dir(cache_dir) + .output(), + ) + .await + .map_err(|_| "git fetch timed out".to_string())? + .map_err(|e| format!("git fetch failed: {e}"))?; + if !output.status.success() { + let stderr = String::from_utf8_lossy(&output.stderr); + return Err(format!("git fetch exited {}: {stderr}", output.status)); + } + refresh_default_branch_head(cache_dir).await; + Ok(()) +} + +/// Repoint the bare clone's `HEAD` to the remote's current default branch. +/// +/// `git clone --bare` pins `HEAD` to the default branch selected at clone +/// time, and the fetch refspec above updates `refs/heads/*` but never `HEAD`. +/// If the remote later changes its default branch (while keeping the old +/// branch alive), an unconfigured `git log HEAD` would keep walking the old +/// branch forever, diverging from the REST fallback which follows the new +/// default. Reading the remote `HEAD` symref (`ref: refs/heads/`) and +/// writing it back keeps the clone's default in sync. +/// +/// Best-effort: `git ls-remote` can fail transiently (network), and there is +/// nothing to refresh on an unborn default branch; neither should fail the +/// fetch that already succeeded. +async fn refresh_default_branch_head(cache_dir: &Path) { + let Ok(output) = tokio::time::timeout( + GIT_CLONE_TIMEOUT, + tokio::process::Command::new("git") + .args(["ls-remote", "--symref", "origin", "HEAD"]) + .current_dir(cache_dir) + .output(), + ) + .await + else { + return; + }; + let Ok(output) = output else { return }; + if !output.status.success() { + return; + } + // `--symref` prints `ref: refs/heads/\tHEAD` on the first line. + let stdout = String::from_utf8_lossy(&output.stdout); + let Some(first) = stdout.lines().next() else { + return; + }; + let Some(remote_ref) = first + .strip_prefix("ref: ") + .and_then(|r| r.split_whitespace().next()) + else { + return; + }; + if !remote_ref.starts_with("refs/heads/") { + return; + } + let _ = tokio::time::timeout( + GIT_CLONE_TIMEOUT, + tokio::process::Command::new("git") + .args(["symbolic-ref", "HEAD", remote_ref]) + .current_dir(cache_dir) + .output(), + ) + .await; +} + +/// Fresh bare clone of `clone_url` into `cache_dir`. +async fn clone_bare(clone_url: &str, cache_dir: &Path) -> Result<(), String> { + if let Some(parent) = cache_dir.parent() { + std::fs::create_dir_all(parent).map_err(|e| format!("create cache dir: {e}"))?; + } + + tracing::info!( + url = %clone_url, + cache = %cache_dir.display(), + "[memory_sources:github:git] cloning bare repo" + ); + + let output = tokio::time::timeout( + GIT_CLONE_TIMEOUT, + tokio::process::Command::new("git") + .args(["clone", "--bare", "--quiet", clone_url]) + .arg(cache_dir) + .output(), + ) + .await + .map_err(|_| "git clone timed out".to_string())? + .map_err(|e| format!("git clone failed: {e}"))?; + + if !output.status.success() { + let stderr = String::from_utf8_lossy(&output.stderr); + return Err(format!("git clone exited {}: {stderr}", output.status)); + } + + Ok(()) +} + +/// List commits in the bare clone, newest first, up to `max`. +/// +/// `branch` restricts the walk to a single ref (default `HEAD` — the bare +/// clone's default branch, matching the REST fallback's default-branch +/// scope), and `paths` narrows it to commits touching any of the given +/// pathspecs. +pub(super) async fn list_commits_git( + owner: &str, + repo: &str, + max: u32, + cache_dir: &Path, + branch: Option<&str>, + paths: &[String], +) -> Result, String> { + ensure_bare_clone(owner, repo, cache_dir).await?; + + let args = log_args(max, branch, paths); + + let output = tokio::time::timeout( + GIT_LOG_TIMEOUT, + tokio::process::Command::new("git") + .args(&args) + .current_dir(cache_dir) + .output(), + ) + .await + .map_err(|_| "git log timed out".to_string())? + .map_err(|e| format!("git log failed: {e}"))?; + + if !output.status.success() { + let stderr = String::from_utf8_lossy(&output.stderr); + return Err(format!("git log exited {}: {stderr}", output.status)); + } + + let stdout = String::from_utf8_lossy(&output.stdout); + let items: Vec = stdout + .lines() + .filter(|line| !line.is_empty()) + .map(|line| { + let parts: Vec<&str> = line.splitn(3, '\t').collect(); + let sha = parts.first().unwrap_or(&""); + let subject = parts.get(1).unwrap_or(&""); + let date = parts.get(2).unwrap_or(&""); + SourceItem { + id: format!("commit:{sha}"), + title: subject.to_string(), + updated_at_ms: parse_iso_ts(date), + } + }) + .collect(); + + tracing::debug!( + count = items.len(), + "[memory_sources:github:git] listed commits via local git" + ); + Ok(items) +} + +/// Build the `git log` argument list for the commit walk. +/// +/// `branch` restricts the walk to a single ref (default `HEAD` — the bare +/// clone's default branch, matching the REST fallback's default-branch +/// scope), and `paths` narrows it to commits touching any of the given +/// pathspecs (trailing `-- path1 path2`). Extracted as a pure helper so the +/// filter wiring is unit-testable without a real clone. +pub(super) fn log_args(max: u32, branch: Option<&str>, paths: &[String]) -> Vec { + let mut args: Vec = vec!["log".to_string()]; + match branch { + Some(b) if !b.is_empty() => args.push(b.to_string()), + _ => args.push("HEAD".to_string()), + } + args.push(format!("--max-count={max}")); + args.push("--format=%H\t%s\t%aI".to_string()); + if !paths.is_empty() { + args.push("--".to_string()); + args.extend(paths.iter().cloned()); + } + args +} + +/// Read one commit's full message and metadata from the bare clone. +pub(super) async fn read_commit_git( + owner: &str, + repo: &str, + sha: &str, + cache_dir: &Path, +) -> Result { + if !cache_dir.join("HEAD").exists() { + return Err("bare clone not present".to_string()); + } + + // git show with a custom format for author, date, and full message. + let output = tokio::time::timeout( + GIT_LOG_TIMEOUT, + tokio::process::Command::new("git") + .args(["show", "--no-patch", "--format=%H%n%aN%n%aE%n%aI%n%B", sha]) + .current_dir(cache_dir) + .output(), + ) + .await + .map_err(|_| "git show timed out".to_string())? + .map_err(|e| format!("git show failed: {e}"))?; + + if !output.status.success() { + let stderr = String::from_utf8_lossy(&output.stderr); + return Err(format!("git show exited {}: {stderr}", output.status)); + } + + let stdout = String::from_utf8_lossy(&output.stdout); + let mut lines = stdout.lines(); + let full_sha = lines.next().unwrap_or(sha); + let author_name = lines.next().unwrap_or("unknown"); + let author_email = lines.next().unwrap_or(""); + let date = lines.next().unwrap_or("unknown"); + let message: String = lines.collect::>().join("\n"); + let message = message.trim(); + + let title = message.lines().next().unwrap_or("").to_string(); + let author = format!("{author_name} <{author_email}>"); + + let body = format!( + "# Commit: {title}\n\n\ + **SHA:** {full_sha}\n\ + **Author:** {author}\n\ + **Date:** {date}\n\n\ + ## Message\n\n\ + {message}", + ); + + Ok(SourceContent { + id: format!("commit:{sha}"), + title, + body, + content_type: ContentType::Markdown, + metadata: serde_json::json!({ + "owner": owner, + "repo": repo, + "sha": full_sha, + "author": author, + }), + }) +} + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinymemory-integrations/src/sources/readers/github/git/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/github/git/mod_tests.rs new file mode 100644 index 00000000..af28f58d --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/readers/github/git/mod_tests.rs @@ -0,0 +1,219 @@ +//! Tests for the local bare-clone helpers: `git log` arguments, cache +//! lifecycle, and process failures. + +use super::*; + +use std::process::Command; + +/// Run `git` with the given args in `cwd`, asserting success and returning +/// stdout as a string. +/// +/// The developer's own git configuration is neutralised: a global +/// `commit.gpgsign = true` would otherwise park `git commit` on a pinentry +/// prompt and hang the whole test binary — on exactly the machines most +/// likely to run these tests. `GIT_CONFIG_GLOBAL`/`GIT_CONFIG_SYSTEM` point +/// at nothing, and signing is off explicitly for good measure. +fn git_ok(cwd: &Path, args: &[&str]) -> String { + let out = Command::new("git") + .env("GIT_CONFIG_GLOBAL", "/dev/null") + .env("GIT_CONFIG_SYSTEM", "/dev/null") + .env("GIT_CONFIG_NOSYSTEM", "1") + .args(["-c", "commit.gpgsign=false", "-c", "tag.gpgsign=false"]) + .args(args) + .current_dir(cwd) + .output() + .expect("spawn git"); + assert!( + out.status.success(), + "git {args:?} failed: {}", + String::from_utf8_lossy(&out.stderr) + ); + String::from_utf8_lossy(&out.stdout).into_owned() +} + +/// Create a source repo with one commit at `dir`. +fn init_repo(dir: &Path) { + std::fs::create_dir_all(dir).expect("create repo dir"); + git_ok(dir, &["init", "-q"]); + git_ok(dir, &["config", "user.email", "test@example.com"]); + git_ok(dir, &["config", "user.name", "Test"]); + std::fs::write(dir.join("a.txt"), "one").expect("write file"); + git_ok(dir, &["add", "."]); + git_ok(dir, &["commit", "-qm", "first"]); +} + +#[tokio::test] +async fn fetch_existing_bare_refreshes_default_branch_head() { + // Regression: the clone's default branch can change upstream. The bare + // clone's HEAD is pinned at clone time, and the fetch refspec updates + // refs/heads/* but not HEAD, so an unconfigured `git log HEAD` would keep + // walking the old default while the REST fallback follows the new one. + // After fetching, HEAD must be repointed to the remote's current default. + let tmp = tempfile::tempdir().expect("tempdir"); + let src = tmp.path().join("src"); + std::fs::create_dir_all(&src).expect("create repo dir"); + git_ok(&src, &["init", "-q", "-b", "master"]); + git_ok(&src, &["config", "user.email", "test@example.com"]); + git_ok(&src, &["config", "user.name", "Test"]); + std::fs::write(src.join("a.txt"), "one").expect("write file"); + git_ok(&src, &["add", "."]); + git_ok(&src, &["commit", "-qm", "first"]); + + let cache = tmp.path().join("cache.git"); + git_ok( + tmp.path(), + &[ + "clone", + "--bare", + "-q", + src.to_str().unwrap(), + cache.to_str().unwrap(), + ], + ); + let head_ref = git_ok(&cache, &["symbolic-ref", "HEAD"]); + assert_eq!( + head_ref.trim(), + "refs/heads/master", + "clone pins default HEAD" + ); + + // Upstream renames its default branch: create `main` and switch HEAD to it + // while keeping `master` alive (a repo that changes its default branch). + git_ok(&src, &["checkout", "-q", "-b", "main"]); + std::fs::write(src.join("b.txt"), "two").expect("write file"); + git_ok(&src, &["add", "."]); + git_ok(&src, &["commit", "-qm", "second"]); + git_ok(&src, &["symbolic-ref", "HEAD", "refs/heads/main"]); + + // A plain fetch (without the refresh) would leave HEAD on `master`. + fetch_existing_bare(&cache).await.expect("fetch succeeds"); + let refreshed = git_ok(&cache, &["symbolic-ref", "HEAD"]); + assert_eq!( + refreshed.trim(), + "refs/heads/main", + "fetch must repoint HEAD to the remote's new default branch" + ); +} + +#[tokio::test] +async fn fetch_existing_bare_advances_local_heads() { + // A bare clone records no remote.origin.fetch refspec, so a bare `git + // fetch` (no refspec) would only touch FETCH_HEAD. The explicit + // `+refs/heads/*:refs/heads/*` must advance refs/heads/* to the remote's + // new commits, otherwise every later sync silently misses them. + let tmp = tempfile::tempdir().expect("tempdir"); + let src = tmp.path().join("src"); + init_repo(&src); + + let cache = tmp.path().join("cache.git"); + git_ok( + tmp.path(), + &[ + "clone", + "--bare", + "-q", + src.to_str().unwrap(), + cache.to_str().unwrap(), + ], + ); + let first_head = git_ok(&cache, &["rev-parse", "HEAD"]); + + // A second commit lands upstream. + std::fs::write(src.join("b.txt"), "two").expect("write file"); + git_ok(&src, &["add", "."]); + git_ok(&src, &["commit", "-qm", "second"]); + let upstream_head = git_ok(&src, &["rev-parse", "HEAD"]); + assert_ne!(first_head, upstream_head, "test setup: new commit expected"); + + // Fetch into the existing bare clone and confirm the local head advances. + fetch_existing_bare(&cache).await.expect("fetch succeeds"); + let cached_head = git_ok(&cache, &["rev-parse", "HEAD"]); + assert_eq!( + cached_head, upstream_head, + "fetch must advance refs/heads/* so git log --all sees new commits" + ); +} + +#[tokio::test] +async fn local_bare_clone_lists_filters_and_renders_commits() { + let tmp = tempfile::tempdir().expect("tempdir"); + let src = tmp.path().join("src"); + init_repo(&src); + std::fs::create_dir_all(src.join("docs")).expect("docs dir"); + std::fs::write(src.join("docs/guide.md"), "guide").expect("write guide"); + git_ok(&src, &["add", "."]); + git_ok(&src, &["commit", "-qm", "document the project"]); + + let cache = tmp.path().join("cache.git"); + git_ok( + tmp.path(), + &[ + "clone", + "--bare", + "-q", + src.to_str().expect("source path"), + cache.to_str().expect("cache path"), + ], + ); + + let items = list_commits_git( + "local-owner", + "local-repo", + 10, + &cache, + None, + &["docs/".to_string()], + ) + .await + .expect("list local commits"); + assert_eq!(items.len(), 1); + assert_eq!(items[0].title, "document the project"); + let sha = items[0].id.strip_prefix("commit:").expect("commit id"); + + let content = read_commit_git("local-owner", "local-repo", sha, &cache) + .await + .expect("render commit"); + assert_eq!(content.id, items[0].id); + assert_eq!(content.title, "document the project"); + assert!(content.body.contains("Test ")); + assert_eq!(content.metadata["owner"], "local-owner"); + assert_eq!(content.metadata["repo"], "local-repo"); +} + +#[tokio::test] +async fn git_helpers_surface_missing_cache_ref_and_process_failures() { + let tmp = tempfile::tempdir().expect("tempdir"); + let missing = tmp.path().join("missing.git"); + assert!( + read_commit_git("owner", "repo", "deadbeef", &missing) + .await + .expect_err("missing cache") + .contains("not present") + ); + + let src = tmp.path().join("src"); + init_repo(&src); + let cache = tmp.path().join("cache.git"); + git_ok( + tmp.path(), + &[ + "clone", + "--bare", + "-q", + src.to_str().expect("source path"), + cache.to_str().expect("cache path"), + ], + ); + assert!( + read_commit_git("owner", "repo", "not-a-ref", &cache) + .await + .expect_err("unknown ref") + .contains("git show exited") + ); + assert!( + list_commits_git("owner", "repo", 10, &cache, Some("missing"), &[]) + .await + .expect_err("unknown branch") + .contains("git log exited") + ); +} diff --git a/crates/tinymemory-integrations/src/sources/readers/github/issues/mod.rs b/crates/tinymemory-integrations/src/sources/readers/github/issues/mod.rs new file mode 100644 index 00000000..cb19b2d0 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/readers/github/issues/mod.rs @@ -0,0 +1,304 @@ +//! Issue and pull-request list/read helpers for the GitHub reader. +//! +//! Both endpoints share the [`fetch_github`](super::api::fetch_github) +//! transport and the list-pass cache in `super::types::LIST_CACHE`: the issues +//! endpoint returns pull requests mixed in with issues, and the PR endpoint is +//! the only one that returns merge state, so the list pass stashes the full +//! row and the read pass reuses it instead of re-fetching. + +use serde::Deserialize; + +use crate::sources::types::{ContentType, SourceContent, SourceItem}; + +use super::api::{GH_MAX_PAGES, GH_PAGE_SIZE, fetch_all_pages, fetch_github}; +use super::types::{CachedItem, GhIssue, GhPr, GhUser, IssueComment}; +use super::{parse_iso_ts, unique_handles}; + +/// List issues (excluding pull requests, which the issues endpoint also +/// returns) with the full row cached for later reads. +pub(super) async fn list_issues( + owner: &str, + repo: &str, + max: u32, + use_gh: bool, +) -> Result, String> { + let mut out: Vec = Vec::new(); + let mut page = 1u32; + + while (out.len() as u32) < max && page <= GH_MAX_PAGES { + let path = + format!("repos/{owner}/{repo}/issues?per_page={GH_PAGE_SIZE}&page={page}&state=all"); + let json_str = fetch_github(&path, use_gh).await?; + let batch: Vec = serde_json::from_str(&json_str) + .map_err(|e| format!("parse issues page {page}: {e}"))?; + let got = batch.len(); + + for i in batch { + if i.pull_request.is_some() { + continue; + } + let ts = i.updated_at.as_deref().and_then(parse_iso_ts); + let item_id = format!("issue:{}", i.number); + let cache_key = format!("{owner}/{repo}:{item_id}"); + out.push(SourceItem { + id: item_id, + title: format!("#{} {}", i.number, i.title), + updated_at_ms: ts, + }); + if let Ok(mut cache) = super::types::LIST_CACHE.lock() { + cache.insert(cache_key, CachedItem::Issue(i)); + } + if out.len() as u32 >= max { + break; + } + } + + if got < GH_PAGE_SIZE as usize { + break; + } + page += 1; + } + + Ok(out) +} + +/// List pull requests with the full row cached for later reads. +pub(super) async fn list_prs( + owner: &str, + repo: &str, + max: u32, + use_gh: bool, +) -> Result, String> { + let prs: Vec = fetch_all_pages(owner, repo, "pulls", "state=all", max, use_gh).await?; + + let items: Vec = prs + .into_iter() + .map(|p| { + let ts = p.updated_at.as_deref().and_then(parse_iso_ts); + let item_id = format!("pr:{}", p.number); + let cache_key = format!("{owner}/{repo}:{item_id}"); + let item = SourceItem { + id: item_id, + title: format!("PR #{} {}", p.number, p.title), + updated_at_ms: ts, + }; + if let Ok(mut cache) = super::types::LIST_CACHE.lock() { + cache.insert(cache_key, CachedItem::Pr(p)); + } + item + }) + .collect(); + + Ok(items) +} + +/// Read one issue, preferring the row cached by the list pass. +pub(super) async fn read_issue( + owner: &str, + repo: &str, + number: u64, + use_gh: bool, +) -> Result { + let cache_key = format!("{owner}/{repo}:issue:{number}"); + let from_cache = super::types::LIST_CACHE + .lock() + .ok() + .and_then(|mut c| c.remove(&cache_key)); + let issue: GhIssue = match from_cache { + Some(CachedItem::Issue(i)) => i, + _ => { + let json_str = + fetch_github(&format!("repos/{owner}/{repo}/issues/{number}"), use_gh).await?; + serde_json::from_str(&json_str).map_err(|e| format!("parse issue: {e}"))? + } + }; + + let author = issue + .user + .as_ref() + .map(|u| u.login.as_str()) + .unwrap_or("unknown"); + let labels: Vec<&str> = issue.labels.iter().map(|l| l.name.as_str()).collect(); + let issue_body = issue.body.as_deref().unwrap_or(""); + + let comments = fetch_issue_comments(owner, repo, number, use_gh).await; + let participants = + unique_handles(std::iter::once(author).chain(comments.iter().map(|c| c.user.as_str()))); + + let mut body = format!( + "# Issue #{number}: {title}\n\n\ + **State:** {state}\n\ + **Author:** @{author}\n\ + **Participants:** {participants}\n\ + **Labels:** {label_str}\n\ + **Created:** {created}\n\ + **Updated:** {updated}\n\n\ + ## Description\n\n\ + {issue_body}", + title = issue.title, + state = issue.state, + label_str = if labels.is_empty() { + "none".to_string() + } else { + labels.join(", ") + }, + created = issue.created_at.as_deref().unwrap_or("unknown"), + updated = issue.updated_at.as_deref().unwrap_or("unknown"), + ); + + if !comments.is_empty() { + body.push_str("\n\n## Comments\n"); + for comment in &comments { + body.push_str(&format!( + "\n### @{} ({})\n\n{}\n", + comment.user, comment.created_at, comment.body + )); + } + } + + Ok(SourceContent { + id: format!("issue:{number}"), + title: format!("#{number} {}", issue.title), + body, + content_type: ContentType::Markdown, + metadata: serde_json::json!({ + "owner": owner, + "repo": repo, + "number": number, + "state": issue.state, + "labels": labels, + }), + }) +} + +/// Read one pull request, preferring the row cached by the list pass. +pub(super) async fn read_pr( + owner: &str, + repo: &str, + number: u64, + use_gh: bool, +) -> Result { + let cache_key = format!("{owner}/{repo}:pr:{number}"); + let from_cache = super::types::LIST_CACHE + .lock() + .ok() + .and_then(|mut c| c.remove(&cache_key)); + let pr: GhPr = match from_cache { + Some(CachedItem::Pr(p)) => p, + _ => { + let json_str = + fetch_github(&format!("repos/{owner}/{repo}/pulls/{number}"), use_gh).await?; + serde_json::from_str(&json_str).map_err(|e| format!("parse PR: {e}"))? + } + }; + + let author = pr + .user + .as_ref() + .map(|u| u.login.as_str()) + .unwrap_or("unknown"); + let labels: Vec<&str> = pr.labels.iter().map(|l| l.name.as_str()).collect(); + let pr_body = pr.body.as_deref().unwrap_or(""); + + let merged_str = match pr.merged_at.as_deref() { + Some(ts) => format!("merged at {ts}"), + None => "not merged".to_string(), + }; + + let comments = fetch_issue_comments(owner, repo, number, use_gh).await; + let participants = + unique_handles(std::iter::once(author).chain(comments.iter().map(|c| c.user.as_str()))); + + let mut body = format!( + "# PR #{number}: {title}\n\n\ + **State:** {state} ({merged})\n\ + **Author:** @{author}\n\ + **Participants:** {participants}\n\ + **Labels:** {label_str}\n\ + **Created:** {created}\n\ + **Updated:** {updated}\n\n\ + ## Description\n\n\ + {pr_body}", + title = pr.title, + state = pr.state, + merged = merged_str, + label_str = if labels.is_empty() { + "none".to_string() + } else { + labels.join(", ") + }, + created = pr.created_at.as_deref().unwrap_or("unknown"), + updated = pr.updated_at.as_deref().unwrap_or("unknown"), + ); + + if !comments.is_empty() { + body.push_str("\n\n## Comments\n"); + for comment in &comments { + body.push_str(&format!( + "\n### @{} ({})\n\n{}\n", + comment.user, comment.created_at, comment.body + )); + } + } + + Ok(SourceContent { + id: format!("pr:{number}"), + title: format!("PR #{number} {}", pr.title), + body, + content_type: ContentType::Markdown, + metadata: serde_json::json!({ + "owner": owner, + "repo": repo, + "number": number, + "state": pr.state, + "merged": pr.merged_at.is_some(), + "labels": labels, + }), + }) +} + +/// Fetch up to 50 comments on an issue/PR. Best-effort: any failure (or +/// parse error) yields an empty list — comment text is enrichment, not the +/// item's substance, so a missing comments API must not fail the read. +async fn fetch_issue_comments( + owner: &str, + repo: &str, + number: u64, + use_gh: bool, +) -> Vec { + #[derive(Deserialize)] + struct RawComment { + user: Option, + body: Option, + created_at: Option, + } + + let json_str = fetch_github( + &format!("repos/{owner}/{repo}/issues/{number}/comments?per_page=50"), + use_gh, + ) + .await; + + let Ok(json_str) = json_str else { + return Vec::new(); + }; + + let comments: Vec = serde_json::from_str(&json_str).unwrap_or_default(); + + comments + .into_iter() + .map(|c| IssueComment { + user: c + .user + .as_ref() + .map(|u| u.login.clone()) + .unwrap_or_else(|| "unknown".into()), + body: c.body.unwrap_or_default(), + created_at: c.created_at.unwrap_or_else(|| "unknown".into()), + }) + .collect() +} + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinymemory-integrations/src/sources/readers/github/issues/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/github/issues/mod_tests.rs new file mode 100644 index 00000000..5a521680 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/readers/github/issues/mod_tests.rs @@ -0,0 +1,151 @@ +//! Offline behavioral tests for issue and pull-request list/read orchestration. + +use super::*; +use crate::sources::readers::github::api::with_test_responses; +use crate::sources::readers::github::types::LIST_CACHE; + +fn issue_json(number: u64) -> serde_json::Value { + serde_json::json!({ + "number": number, + "title": "Broken widget", + "body": "Steps to reproduce", + "state": "open", + "user": {"login": "alice"}, + "labels": [{"name": "bug"}, {"name": "urgent"}], + "created_at": "2026-01-01T00:00:00Z", + "updated_at": "2026-01-02T03:04:05Z", + "pull_request": null + }) +} + +fn pr_json(number: u64) -> serde_json::Value { + serde_json::json!({ + "number": number, + "title": "Fix widget", + "body": "Implements the fix", + "state": "closed", + "user": {"login": "bob"}, + "labels": [{"name": "ready"}], + "created_at": "2026-01-03T00:00:00Z", + "updated_at": "2026-01-04T00:00:00Z", + "merged_at": "2026-01-05T00:00:00Z" + }) +} + +#[tokio::test] +async fn lists_cache_and_render_issues_and_pull_requests_without_network() { + LIST_CACHE.lock().expect("list cache").clear(); + let disguised_pr = serde_json::json!({ + "number": 99, + "title": "PR returned by issues endpoint", + "body": null, + "state": "open", + "user": null, + "labels": [], + "created_at": null, + "updated_at": null, + "pull_request": {"url":"https://example.invalid/pr/99"} + }); + let listed_issues = with_test_responses( + vec![Ok( + serde_json::json!([issue_json(7), disguised_pr]).to_string() + )], + list_issues("acme", "widget", 10, false), + ) + .await + .expect("list issues"); + assert_eq!(listed_issues.len(), 1); + assert_eq!(listed_issues[0].id, "issue:7"); + assert_eq!(listed_issues[0].title, "#7 Broken widget"); + assert_eq!(listed_issues[0].updated_at_ms, Some(1_767_323_045_000)); + + let issue = with_test_responses( + vec![Ok(serde_json::json!([ + { + "user":{"login":"carol"}, + "body":"Confirmed", + "created_at":"2026-01-02T04:00:00Z" + }, + {"user":null,"body":null,"created_at":null} + ]) + .to_string())], + read_issue("acme", "widget", 7, false), + ) + .await + .expect("read cached issue"); + assert_eq!(issue.id, "issue:7"); + assert_eq!(issue.title, "#7 Broken widget"); + assert!(issue.body.contains("**Participants:** @alice @carol")); + assert!(issue.body.contains("**Labels:** bug, urgent")); + assert!(issue.body.contains("### @carol (2026-01-02T04:00:00Z)")); + assert!(issue.body.contains("### @unknown (unknown)")); + assert_eq!(issue.metadata["state"], "open"); + + let listed_prs = with_test_responses( + vec![Ok(serde_json::json!([pr_json(8)]).to_string())], + list_prs("acme", "widget", 10, true), + ) + .await + .expect("list pull requests"); + assert_eq!(listed_prs[0].id, "pr:8"); + assert_eq!(listed_prs[0].title, "PR #8 Fix widget"); + + let pr = with_test_responses( + vec![Ok("not valid comments JSON".into())], + read_pr("acme", "widget", 8, true), + ) + .await + .expect("read cached pull request despite malformed comments"); + assert!( + pr.body + .contains("**State:** closed (merged at 2026-01-05T00:00:00Z)") + ); + assert!(pr.body.contains("**Participants:** @bob")); + assert!(!pr.body.contains("## Comments")); + assert_eq!(pr.metadata["merged"], true); + assert!(LIST_CACHE.lock().expect("list cache").is_empty()); + + assert_uncached_reads_and_failures().await; +} + +async fn assert_uncached_reads_and_failures() { + LIST_CACHE.lock().expect("list cache").clear(); + let issue = with_test_responses( + vec![ + Ok(issue_json(11).to_string()), + Err("comments unavailable".into()), + ], + read_issue("acme", "widget", 11, false), + ) + .await + .expect("uncached issue read"); + assert_eq!(issue.id, "issue:11"); + assert!(!issue.body.contains("## Comments")); + + let transport_error = with_test_responses( + vec![Err("offline".into())], + list_issues("acme", "widget", 10, false), + ) + .await + .expect_err("transport failure must propagate"); + assert_eq!(transport_error, "offline"); + + let parse_error = with_test_responses( + vec![Ok("{}".into())], + list_issues("acme", "widget", 10, false), + ) + .await + .expect_err("malformed list must fail"); + assert!(parse_error.contains("parse issues page 1")); + + let read_error = + with_test_responses(vec![Ok("[]".into())], read_pr("acme", "widget", 12, false)) + .await + .expect_err("malformed pull request must fail"); + assert!(read_error.contains("parse PR")); + + let exhausted = with_test_responses(Vec::new(), list_prs("acme", "widget", 1, false)) + .await + .expect_err("empty fixture must not reach network"); + assert!(exhausted.contains("no deterministic GitHub response queued")); +} diff --git a/crates/tinymemory-integrations/src/sources/readers/github/mod.rs b/crates/tinymemory-integrations/src/sources/readers/github/mod.rs new file mode 100644 index 00000000..4ace4648 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/readers/github/mod.rs @@ -0,0 +1,308 @@ +//! GitHub repo source reader. +//! +//! Pulls **project activity** (commits, issues, PRs) from a GitHub +//! repository — not source code. Commits are read from a local bare clone +//! under `/git_cache/` (`git` must be on `PATH`), falling back to +//! the API when the clone fails. Issues and pull requests, and that fallback, +//! go through the `gh` CLI when it is available (authenticated, higher rate +//! limit) and otherwise the public, unauthenticated GitHub REST API. +//! `gh_available` is probed once per process. +//! +//! ## Module layout +//! +//! - [`self`] — [`GithubReader`] orchestration: item listing/reading, URL +//! parsing, shared utilities, and the cached +//! `gh`-availability probe. +//! - `types` — API response models and the `gh`-fallback list cache. +//! - `git` — local bare-clone + `git log` / `git show` helpers. +//! - `api` — `gh api` / REST transport plus commit list/read helpers. +//! - `issues` — issue and pull-request list/read helpers. + +mod api; +mod git; +mod issues; +mod types; + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; + +use std::time::Duration; + +use async_trait::async_trait; + +use crate::sources::error::{Error, Result}; +use crate::sources::types::{MemorySourceEntry, SourceContent, SourceItem, SourceKind}; + +use super::SourceReader; + +// Re-export for the sibling submodules and the test module. +pub(crate) use types::{ItemKind, LIST_CACHE}; + +/// Default number of items of **each** type (commits, issues, PRs) to pull +/// when the source entry doesn't override it. Tunable per-source via +/// `max_commits` / `max_issues` / `max_prs` on [`MemorySourceEntry`]. +pub(crate) const DEFAULT_GITHUB_ITEM_LIMIT: u32 = 1000; + +/// Timeout for a single `gh` CLI invocation (including the availability +/// probe). +const GH_CLI_TIMEOUT: Duration = Duration::from_secs(30); + +/// Whether the `gh` CLI is on PATH and runs. Probed once per process and +/// cached: `gh api` is the preferred transport for authenticated, +/// higher-rate-limit access, and re-probing on every item read is wasteful. +static GH_AVAILABLE: tokio::sync::OnceCell = tokio::sync::OnceCell::const_new(); + +/// Probe `gh --version` (async, so a stuck `gh` cannot block a worker +/// thread) and cache the result for the process lifetime. +async fn gh_available() -> bool { + *GH_AVAILABLE + .get_or_init(|| async { + let status = tokio::time::timeout( + GH_CLI_TIMEOUT, + tokio::process::Command::new("gh") + .arg("--version") + .stdout(std::process::Stdio::null()) + .stderr(std::process::Stdio::null()) + .status(), + ) + .await; + status + .map(|s| s.map(|st| st.success()).unwrap_or(false)) + .unwrap_or(false) + }) + .await +} + +/// Reader for a GitHub repository source: lists and fetches commits, issues +/// and pull requests. Item ids are `commit:`, `issue:` and `pr:`. +/// Commits come from a local bare clone with an API fallback; issues and pull +/// requests come from `gh api` or the REST API. +#[derive(Debug, Clone, Copy, Default)] +pub struct GithubReader; + +/// Parse `owner` and `repo` from a GitHub URL. +/// +/// Accepts only the canonical `https://github.com//[.git][/]` +/// shape — extra segments like `/tree/main` or `/blob/...` are rejected +/// so callers can't accidentally derive the wrong owner/repo from a +/// deep link. +pub(crate) fn parse_github_url(url: &str) -> std::result::Result<(String, String), String> { + let trimmed = url.trim(); + let rest = trimmed + .strip_prefix("https://github.com/") + .or_else(|| trimmed.strip_prefix("http://github.com/")) + .or_else(|| trimmed.strip_prefix("git@github.com:")) + .ok_or_else(|| format!("not a GitHub URL: {url}"))?; + let cleaned = rest.trim_end_matches('/').trim_end_matches(".git"); + let parts: Vec<&str> = cleaned.split('/').collect(); + if parts.len() != 2 || parts[0].is_empty() || parts[1].is_empty() { + return Err(format!( + "expected https://github.com//, got: {url}" + )); + } + Ok((parts[0].to_string(), parts[1].to_string())) +} + +// ── Reader implementation ─────────────────────────────────────────── + +#[async_trait] +impl SourceReader for GithubReader { + fn kind(&self) -> SourceKind { + SourceKind::GithubRepo + } + + async fn list_items( + &self, + source: &MemorySourceEntry, + workspace: &std::path::Path, + ) -> Result> { + self.list_items_inner(source, workspace) + .await + .map_err(Error::Reader) + } + + async fn read_item( + &self, + source: &MemorySourceEntry, + item_id: &str, + workspace: &std::path::Path, + ) -> Result { + self.read_item_inner(source, item_id, workspace) + .await + .map_err(Error::Reader) + } +} + +impl GithubReader { + async fn list_items_inner( + &self, + source: &MemorySourceEntry, + workspace: &std::path::Path, + ) -> std::result::Result, String> { + let url = source + .url + .as_deref() + .ok_or("github source requires a url")?; + let (owner, repo) = parse_github_url(url)?; + let use_gh = gh_available().await; + + let max_commits = source.max_commits.unwrap_or(DEFAULT_GITHUB_ITEM_LIMIT); + let max_issues = source.max_issues.unwrap_or(DEFAULT_GITHUB_ITEM_LIMIT); + let max_prs = source.max_prs.unwrap_or(DEFAULT_GITHUB_ITEM_LIMIT); + // A configured branch narrows commits to that ref; configured paths + // narrow them to the touched files. Both fall through to the API + // fallback so the two transports agree on scope. + let branch = source.branch.as_deref(); + let paths = source.paths.as_slice(); + + let cache_dir = git::git_cache_dir(workspace, &owner, &repo); + + tracing::debug!( + owner = %owner, + repo = %repo, + use_gh = use_gh, + branch = %branch.unwrap_or("(all)"), + max_commits, + max_issues, + max_prs, + cache = %cache_dir.display(), + "[memory_sources:github] listing items" + ); + + // Clear the list cache so stale data from a prior sync doesn't + // leak into this run. + if let Ok(mut cache) = LIST_CACHE.lock() { + cache.clear(); + } + + let mut items = Vec::new(); + let mut errors = Vec::new(); + + // Commits via local git (clone/fetch bare repo, then git log) + match git::list_commits_git(&owner, &repo, max_commits, &cache_dir, branch, paths).await { + Ok(commits) => items.extend(commits), + Err(e) => { + tracing::warn!(error = %e, "[memory_sources:github] git commit list failed, falling back to API"); + match api::list_commits_api(&owner, &repo, max_commits, use_gh, branch, paths).await + { + Ok(commits) => items.extend(commits), + Err(e2) => { + tracing::warn!(error = %e2, "[memory_sources:github] API commit list also failed"); + errors.push(e2); + } + } + } + } + + // Issues and PRs via gh CLI / API (no local equivalent) + match issues::list_issues(&owner, &repo, max_issues, use_gh).await { + Ok(issues) => items.extend(issues), + Err(e) => { + tracing::warn!(error = %e, "[memory_sources:github] failed to list issues"); + errors.push(e); + } + } + + match issues::list_prs(&owner, &repo, max_prs, use_gh).await { + Ok(prs) => items.extend(prs), + Err(e) => { + tracing::warn!(error = %e, "[memory_sources:github] failed to list PRs"); + errors.push(e); + } + } + + if items.is_empty() && !errors.is_empty() { + return Err(format!( + "all GitHub API calls failed: {}", + errors.join("; ") + )); + } + + tracing::debug!(count = items.len(), "[memory_sources:github] found items"); + Ok(items) + } + + async fn read_item_inner( + &self, + source: &MemorySourceEntry, + item_id: &str, + workspace: &std::path::Path, + ) -> std::result::Result { + let url = source + .url + .as_deref() + .ok_or("github source requires a url")?; + let (owner, repo) = parse_github_url(url)?; + let use_gh = gh_available().await; + + let (kind, ref_id) = + ItemKind::from_id(item_id).ok_or_else(|| format!("invalid item id: {item_id}"))?; + + tracing::debug!( + item_id = %item_id, + kind = ?kind, + "[memory_sources:github] reading item" + ); + + match kind { + ItemKind::Commit => { + let cache_dir = git::git_cache_dir(workspace, &owner, &repo); + match git::read_commit_git(&owner, &repo, ref_id, &cache_dir).await { + Ok(content) => Ok(content), + Err(e) => { + tracing::debug!( + sha = %ref_id, + error = %e, + "[memory_sources:github] git read_commit failed, falling back to API" + ); + api::read_commit_api(&owner, &repo, ref_id, use_gh).await + } + } + } + ItemKind::Issue => { + let num: u64 = ref_id + .parse() + .map_err(|_| format!("invalid issue number: {ref_id}"))?; + issues::read_issue(&owner, &repo, num, use_gh).await + } + ItemKind::PullRequest => { + let num: u64 = ref_id + .parse() + .map_err(|_| format!("invalid PR number: {ref_id}"))?; + issues::read_pr(&owner, &repo, num, use_gh).await + } + } + } +} + +// ── Utilities ─────────────────────────────────────────────────────── + +fn parse_iso_ts(s: &str) -> Option { + chrono::DateTime::parse_from_rfc3339(s) + .ok() + .map(|dt| dt.timestamp_millis()) +} + +/// Render GitHub logins as a deduped, order-preserving, space-separated +/// list of `@handle`s. Empty / `unknown` logins are skipped; an empty +/// result renders as `none`. Used so unique committers/commenters surface +/// as `handle:` entities in the memory tree. +fn unique_handles<'a>(logins: impl Iterator) -> String { + let mut seen = std::collections::HashSet::new(); + let mut out: Vec = Vec::new(); + for login in logins { + let l = login.trim(); + if l.is_empty() || l == "unknown" { + continue; + } + if seen.insert(l.to_string()) { + out.push(format!("@{l}")); + } + } + if out.is_empty() { + "none".to_string() + } else { + out.join(" ") + } +} diff --git a/crates/tinymemory-integrations/src/sources/readers/github/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/github/mod_tests.rs new file mode 100644 index 00000000..66168524 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/readers/github/mod_tests.rs @@ -0,0 +1,624 @@ +//! Tests for the GitHub reader: orchestration over cached clones and the +//! test transport, commit queries and merging, and URL and item-id parsing. + +use super::*; +use crate::sources::readers::SourceReader; + +fn github_source(url: Option<&str>) -> MemorySourceEntry { + MemorySourceEntry { + id: "github".into(), + kind: SourceKind::GithubRepo, + label: "GitHub".into(), + enabled: true, + toolkit: None, + connection_id: None, + path: None, + glob: None, + url: url.map(str::to_string), + branch: None, + paths: Vec::new(), + max_commits: Some(10), + max_issues: Some(0), + max_prs: Some(0), + max_items: None, + selector: None, + max_tokens_per_sync: None, + max_cost_per_sync_usd: None, + sync_depth_days: None, + } +} + +fn local_git(cwd: &std::path::Path, args: &[&str]) -> String { + let output = std::process::Command::new("git") + .env("GIT_CONFIG_GLOBAL", "/dev/null") + .env("GIT_CONFIG_SYSTEM", "/dev/null") + .env("GIT_CONFIG_NOSYSTEM", "1") + .args(["-c", "commit.gpgsign=false"]) + .args(args) + .current_dir(cwd) + .output() + .expect("spawn git"); + assert!( + output.status.success(), + "git {args:?} failed: {}", + String::from_utf8_lossy(&output.stderr) + ); + String::from_utf8_lossy(&output.stdout).into_owned() +} + +#[tokio::test] +async fn reader_lists_and_reads_a_cached_local_repository_without_network() { + let workspace = tempfile::tempdir().expect("workspace"); + let source_repo = workspace.path().join("source"); + std::fs::create_dir_all(&source_repo).expect("source directory"); + local_git(&source_repo, &["init", "-q"]); + local_git(&source_repo, &["config", "user.email", "test@example.com"]); + local_git(&source_repo, &["config", "user.name", "Test"]); + std::fs::write(source_repo.join("README.md"), "hello").expect("write file"); + local_git(&source_repo, &["add", "."]); + local_git(&source_repo, &["commit", "-qm", "local activity"]); + + let cache = git::git_cache_dir(workspace.path(), "local", "fixture"); + std::fs::create_dir_all(cache.parent().expect("cache parent")).expect("cache parent"); + local_git( + workspace.path(), + &[ + "clone", + "--bare", + "-q", + source_repo.to_str().expect("source path"), + cache.to_str().expect("cache path"), + ], + ); + + let source = github_source(Some("https://github.com/local/fixture")); + let reader = GithubReader; + assert_eq!(reader.kind(), SourceKind::GithubRepo); + let items = reader + .list_items(&source, workspace.path()) + .await + .expect("list local cached activity"); + assert_eq!(items.len(), 1); + assert_eq!(items[0].title, "local activity"); + + let content = reader + .read_item(&source, &items[0].id, workspace.path()) + .await + .expect("read cached commit"); + assert_eq!(content.title, "local activity"); + assert!(content.body.contains("Test ")); +} + +#[tokio::test] +async fn reader_rejects_missing_urls_and_malformed_item_ids_before_network() { + let workspace = tempfile::tempdir().expect("workspace"); + let reader = GithubReader; + let missing = github_source(None); + assert!(reader.list_items(&missing, workspace.path()).await.is_err()); + assert!( + reader + .read_item(&missing, "commit:abc", workspace.path()) + .await + .is_err() + ); + + let configured = github_source(Some("https://github.com/local/fixture")); + for item_id in ["unknown", "issue:not-a-number", "pr:not-a-number"] { + assert!( + reader + .read_item(&configured, item_id, workspace.path()) + .await + .is_err() + ); + } +} + +fn issue_json(number: u64) -> serde_json::Value { + serde_json::json!({ + "number": number, + "title": "Reader issue", + "body": "Issue body", + "state": "open", + "user": {"login": "alice"}, + "labels": [{"name": "coverage"}], + "created_at": "2026-01-01T00:00:00Z", + "updated_at": "2026-01-02T00:00:00Z", + "pull_request": null + }) +} + +fn pr_json(number: u64) -> serde_json::Value { + serde_json::json!({ + "number": number, + "title": "Reader PR", + "body": "PR body", + "state": "closed", + "user": {"login": "bob"}, + "labels": [], + "created_at": "2026-01-03T00:00:00Z", + "updated_at": "2026-01-04T00:00:00Z", + "merged_at": null + }) +} + +#[tokio::test] +async fn reader_orchestrates_issue_and_pr_cache_lifecycles_without_network() { + let workspace = tempfile::tempdir().expect("workspace"); + let mut source = github_source(Some("https://github.com/local/fixture")); + source.max_commits = Some(0); + source.max_issues = Some(5); + source.max_prs = Some(5); + let reader = GithubReader; + + let items = api::with_test_responses( + vec![ + Ok(serde_json::json!([issue_json(7)]).to_string()), + Ok(serde_json::json!([pr_json(8)]).to_string()), + ], + reader.list_items(&source, workspace.path()), + ) + .await + .expect("list issue and PR through reader"); + assert_eq!( + items + .iter() + .map(|item| item.id.as_str()) + .collect::>(), + ["issue:7", "pr:8"] + ); + + let issue = api::with_test_responses( + vec![Ok("[]".into())], + reader.read_item(&source, "issue:7", workspace.path()), + ) + .await + .expect("read cached issue"); + assert_eq!(issue.title, "#7 Reader issue"); + + let pr = api::with_test_responses( + vec![Ok("[]".into())], + reader.read_item(&source, "pr:8", workspace.path()), + ) + .await + .expect("read cached PR"); + assert_eq!(pr.title, "PR #8 Reader PR"); +} + +#[tokio::test] +async fn reader_reports_when_every_configured_github_family_fails() { + let workspace = tempfile::tempdir().expect("workspace"); + let mut source = github_source(Some("https://github.com/local/fixture")); + source.max_commits = Some(0); + source.max_issues = Some(1); + source.max_prs = Some(1); + + let error = api::with_test_responses( + vec![Err("issues offline".into()), Err("prs offline".into())], + GithubReader.list_items(&source, workspace.path()), + ) + .await + .expect_err("all configured families failed"); + assert!(error.to_string().contains("all GitHub API calls failed")); + assert!(error.to_string().contains("issues offline")); + assert!(error.to_string().contains("prs offline")); +} + +fn commit_json(sha: &str, message: &str, login: Option<&str>) -> String { + let committed_at = if sha == "new" { + "2026-02-02T00:00:00Z" + } else { + "2026-01-02T00:00:00Z" + }; + let author = login + .map(|value| serde_json::json!({ "login": value })) + .unwrap_or(serde_json::Value::Null); + serde_json::json!({ + "sha": sha, + "commit": { + "message": message, + "author": { + "name": "Test Author", + "email": "author@example.com", + "date": "2026-01-01T00:00:00Z" + }, + "committer": { + "name": "Test Committer", + "email": "committer@example.com", + "date": committed_at + } + }, + "author": author + }) + .to_string() +} + +#[tokio::test] +async fn api_commit_fallback_lists_merges_and_renders_without_network() { + let older = commit_json("old", "older commit\nbody", None); + let newer = commit_json("new", "newer commit\nbody", Some("octocat")); + let listed = api::with_test_responses( + vec![Ok(format!("[{older}]")), Ok(format!("[{newer},{older}]"))], + api::list_commits_api( + "owner", + "repo", + 10, + false, + Some("main"), + &["docs/".into(), "src/".into()], + ), + ) + .await + .expect("list commits from deterministic API pages"); + assert_eq!(listed.len(), 2); + assert_eq!(listed[0].id, "commit:new"); + assert_eq!(listed[1].id, "commit:old"); + + let content = api::with_test_responses( + vec![Ok(commit_json( + "new", + "newer commit\nfull body", + Some("octocat"), + ))], + api::read_commit_api("owner", "repo", "new", false), + ) + .await + .expect("read deterministic commit"); + assert_eq!(content.title, "newer commit"); + assert!( + content + .body + .contains("Test Author (@octocat)") + ); + assert_eq!(content.metadata["author_handle"], "octocat"); +} + +#[tokio::test] +async fn api_commit_fallback_reports_transport_and_parse_failures_without_network() { + let transport = api::with_test_responses( + vec![Err("offline".into())], + api::list_commits_api("owner", "repo", 1, false, None, &[]), + ) + .await + .expect_err("transport error"); + assert_eq!(transport, "offline"); + + let list_parse = api::with_test_responses( + vec![Ok("not json".into())], + api::list_commits_api("owner", "repo", 1, false, None, &[]), + ) + .await + .expect_err("list parse error"); + assert!(list_parse.contains("parse commits page 1")); + + let read_parse = api::with_test_responses( + vec![Ok("{}".into())], + api::read_commit_api("owner", "repo", "bad", false), + ) + .await + .expect_err("commit parse error"); + assert!(read_parse.contains("parse commit")); + + let exhausted = api::with_test_responses( + Vec::new(), + api::read_commit_api("owner", "repo", "missing", false), + ) + .await + .expect_err("fixture exhaustion fails closed"); + assert!(exhausted.contains("no deterministic GitHub response queued")); +} + +#[tokio::test] +async fn reader_falls_back_from_a_broken_local_cache_to_the_api_without_network() { + let workspace = tempfile::tempdir().expect("workspace"); + let cache = git::git_cache_dir(workspace.path(), "owner", "repo"); + std::fs::create_dir_all(&cache).expect("cache directory"); + std::fs::write(cache.join("HEAD"), "not a git repository").expect("broken cache marker"); + + let mut source = github_source(Some("https://github.com/owner/repo")); + source.max_commits = Some(5); + source.max_issues = Some(0); + source.max_prs = Some(0); + let reader = GithubReader; + + let listed = api::with_test_responses( + vec![Ok(format!( + "[{}]", + commit_json("fallback", "API fallback commit", Some("octocat")) + ))], + reader.list_items(&source, workspace.path()), + ) + .await + .expect("broken git cache falls back to API list"); + assert_eq!(listed.len(), 1); + assert_eq!(listed[0].id, "commit:fallback"); + + let content = api::with_test_responses( + vec![Ok(commit_json( + "fallback", + "API fallback commit\nfull body", + Some("octocat"), + ))], + reader.read_item(&source, "commit:fallback", workspace.path()), + ) + .await + .expect("broken git cache falls back to API read"); + assert_eq!(content.id, "commit:fallback"); + assert!(content.body.contains("full body")); + assert_eq!(content.metadata["author_handle"], "octocat"); +} + +#[tokio::test] +async fn reader_keeps_successful_families_when_commit_transports_fail() { + let workspace = tempfile::tempdir().expect("workspace"); + let cache = git::git_cache_dir(workspace.path(), "owner", "repo"); + std::fs::create_dir_all(&cache).expect("cache directory"); + std::fs::write(cache.join("HEAD"), "not a git repository").expect("broken cache marker"); + + let mut source = github_source(Some("https://github.com/owner/repo")); + source.max_commits = Some(1); + source.max_issues = Some(1); + source.max_prs = Some(0); + + let items = api::with_test_responses( + vec![ + Err("commit API offline".into()), + Ok(serde_json::json!([issue_json(7)]).to_string()), + ], + GithubReader.list_items(&source, workspace.path()), + ) + .await + .expect("a successful issue family makes the partial result usable"); + + assert_eq!(items.len(), 1); + assert_eq!(items[0].id, "issue:7"); + assert_eq!(items[0].title, "#7 Reader issue"); +} + +#[test] +fn git_log_args_default_to_head_without_branch() { + // With no branch configured the walk must stay on the bare clone's HEAD + // (the default branch), matching the REST fallback's default-branch scope + // rather than walking every ref. + let args = git::log_args(50, None, &[]); + assert_eq!( + args, + vec![ + "log".to_string(), + "HEAD".to_string(), + "--max-count=50".to_string(), + "--format=%H\t%s\t%aI".to_string(), + ] + ); +} + +#[test] +fn git_log_args_restrict_to_branch_and_paths() { + let args = git::log_args( + 50, + Some("main"), + &["src/lib.rs".to_string(), "docs/".to_string()], + ); + assert_eq!( + args, + vec![ + "log".to_string(), + "main".to_string(), + "--max-count=50".to_string(), + "--format=%H\t%s\t%aI".to_string(), + "--".to_string(), + "src/lib.rs".to_string(), + "docs/".to_string(), + ] + ); + // Empty/whitespace branch falls back to HEAD, never an empty ref. + let args = git::log_args(1, Some(""), &[]); + assert_eq!(args[1], "HEAD"); +} + +#[test] +fn commit_list_queries_carry_branch_and_path_filters() { + // No filters → a single empty query (plain pagination). + assert_eq!(api::commit_list_queries(None, &[]), vec![String::new()]); + // Branch only → `sha=`. + assert_eq!( + api::commit_list_queries(Some("main"), &[]), + vec![String::from("sha=main")] + ); + // One path → `path=

`. + assert_eq!( + api::commit_list_queries(None, &["src/".to_string()]), + vec![String::from("path=src/")] + ); + // Branch + one path → `sha=&path=

`. + assert_eq!( + api::commit_list_queries(Some("main"), &["src/lib.rs".to_string()]), + vec![String::from("sha=main&path=src/lib.rs")] + ); + // Multiple paths → one query per path, dedup happens in the caller. + assert_eq!( + api::commit_list_queries(Some("main"), &["a/".to_string(), "b/".to_string()]), + vec![ + String::from("sha=main&path=a/"), + String::from("sha=main&path=b/") + ] + ); + // Empty branch is treated as unset. + assert_eq!( + api::commit_list_queries(Some(""), &["a/".to_string()]), + vec![String::from("path=a/")] + ); +} + +#[test] +fn commit_list_queries_percent_encode_special_chars() { + // `&`, `#`, `=` and spaces inside a branch or path value would be parsed + // as query syntax and corrupt the filter; they must be percent-encoded. + // `/` is left intact (legal in a query component, and GitHub's commits + // `path` filter expects the common `path=src/` shape unencoded). + assert_eq!( + api::commit_list_queries(Some("feature/one&two"), &["src/#1.rs".to_string()]), + vec![String::from("sha=feature/one%26two&path=src/%231.rs")] + ); + // Unreserved values are unchanged. + assert_eq!( + api::commit_list_queries(Some("main"), &["docs/".to_string()]), + vec![String::from("sha=main&path=docs/")] + ); +} + +#[tokio::test] +async fn fetch_all_pages_keeps_page_size_constant_and_truncates() { + // Regression: the page size must not shrink mid-walk. With `max = 150` + // (not a multiple of 100), a shrinking `per_page` would re-window the + // offsets — page 2 at per_page=50 returns items 51-100 again, skipping + // 101-150. A constant page size walks page 1 and page 2 both at + // per_page=100 and truncates the 200 collected rows to 150. + let mut requested: Vec = Vec::new(); + let pages = api::collect_pages::("commits", 150, |page| { + let url = format!("per_page=100&page={page}"); + requested.push(url); + // 100 rows per page, all full (never a short page before the cap). + let rows: Vec = (1..=100) + .map(|i| format!("{}", (page - 1) * 100 + i)) + .collect(); + async move { Ok(format!("[{}]", rows.join(","))) } + }) + .await + .unwrap(); + + assert_eq!( + requested, + vec![ + "per_page=100&page=1".to_string(), + "per_page=100&page=2".to_string(), + ] + ); + assert_eq!(pages.len(), 150); + // No overlap: the second page is the next window (101..), not 51..100. + assert_eq!(pages[0], 1); + assert_eq!(pages[100], 101); + assert_eq!(pages[149], 150); +} + +#[tokio::test] +async fn fetch_all_pages_stops_at_a_short_page() { + // A short page (fewer than GH_PAGE_SIZE rows) is the last page; the walk + // must not request page 2 after it. + let mut requested: Vec = Vec::new(); + let pages = + crate::sources::readers::github::api::collect_pages::("commits", 1000, |page| { + requested.push(page); + async move { + // Page 1 is short (3 rows) — stop after it even though max is large. + Ok("[1,2,3]".to_string()) + } + }) + .await + .unwrap(); + + assert_eq!(requested, vec![1]); + assert_eq!(pages, vec![1, 2, 3]); +} + +/// Build a synthetic `GhCommit` for merge tests. +fn gh_commit(sha: &str, subject: &str, ts: &str) -> types::GhCommit { + types::GhCommit { + sha: sha.into(), + commit: types::GhCommitInner { + message: subject.into(), + author: None, + committer: Some(types::GhAuthor { + name: None, + email: None, + date: Some(ts.into()), + }), + }, + author: None, + } +} + +#[test] +fn merge_commit_batches_walks_every_path_before_truncating() { + // Two configured paths: the first returns two commits, the second one. + // The pre-fix code stopped after the first path once `out` reached `max`, + // silently dropping the `src` commit even though it is newer than the + // second `docs` commit. + let docs = vec![gh_commit("a", "docs first", "2024-01-01T00:00:00Z")]; + let src = vec![gh_commit("b", "src newer", "2024-02-01T00:00:00Z")]; + + let merged = api::merge_commit_batches(vec![docs, src], 3); + let ids: Vec<&str> = merged.iter().map(|i| i.id.as_str()).collect(); + assert_eq!( + ids, + vec!["commit:b", "commit:a"], + "newest-first, both paths kept" + ); +} + +#[test] +fn merge_commit_batches_dedups_by_sha_and_truncates_globally() { + // A commit touching both paths appears in both batches but only once. + let docs = vec![ + gh_commit("a", "docs first", "2024-01-01T00:00:00Z"), + gh_commit("shared", "touches both", "2024-02-01T00:00:00Z"), + ]; + let src = vec![gh_commit("shared", "touches both", "2024-02-01T00:00:00Z")]; + + let merged = api::merge_commit_batches(vec![docs, src], 1); + let ids: Vec<&str> = merged.iter().map(|i| i.id.as_str()).collect(); + assert_eq!(ids, vec!["commit:shared"], "deduped and truncated to max"); +} + +#[test] +fn parse_github_url_extracts_owner_and_repo() { + let (owner, repo) = parse_github_url("https://github.com/openai/tiktoken").unwrap(); + assert_eq!(owner, "openai"); + assert_eq!(repo, "tiktoken"); +} + +#[test] +fn parse_github_url_handles_trailing_slash_and_git() { + let (owner, repo) = parse_github_url("https://github.com/org/repo.git/").unwrap(); + assert_eq!(owner, "org"); + assert_eq!(repo, "repo"); +} + +#[test] +fn parse_github_url_rejects_non_repo_paths() { + // Deep links like /tree/main must not silently extract the wrong + // owner/repo. Bare host or non-github URLs also rejected. + assert!(parse_github_url("https://github.com/org/repo/tree/main").is_err()); + assert!(parse_github_url("https://gitlab.com/org/repo").is_err()); + assert!(parse_github_url("https://github.com/org").is_err()); + assert!(parse_github_url("not-a-url").is_err()); +} + +#[test] +fn item_kind_round_trips() { + let cases = [ + ("commit:abc123", ItemKind::Commit, "abc123"), + ("issue:42", ItemKind::Issue, "42"), + ("pr:99", ItemKind::PullRequest, "99"), + ]; + for (id, expected_kind, expected_ref) in cases { + let (kind, ref_id) = ItemKind::from_id(id).unwrap(); + assert_eq!(kind, expected_kind); + assert_eq!(ref_id, expected_ref); + } +} + +#[test] +fn item_kind_rejects_invalid() { + assert!(ItemKind::from_id("unknown:123").is_none()); + assert!(ItemKind::from_id("noprefix").is_none()); +} + +#[test] +fn unique_handles_dedups_and_skips_unknown() { + assert_eq!( + unique_handles(["alice", "bob", "alice", "unknown", ""].into_iter()), + "@alice @bob" + ); + assert_eq!(unique_handles(["unknown", ""].into_iter()), "none"); + assert_eq!(unique_handles(std::iter::empty()), "none"); +} diff --git a/crates/tinymemory-integrations/src/sources/readers/github/types.rs b/crates/tinymemory-integrations/src/sources/readers/github/types.rs new file mode 100644 index 00000000..11327e98 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/readers/github/types.rs @@ -0,0 +1,122 @@ +//! API response models and the `gh`-fallback list cache for the GitHub +//! reader. Pure data — no I/O lives here. Models are deliberately kept loose +//! (only the fields the reader consumes are declared) so new GitHub response +//! fields don't force a struct change. + +use std::collections::HashMap; +use std::sync::{LazyLock, Mutex}; + +use serde::Deserialize; + +/// A commit object from the REST commits endpoint. +#[derive(Debug, Deserialize)] +pub(crate) struct GhCommit { + pub(crate) sha: String, + pub(crate) commit: GhCommitInner, + /// Top-level GitHub user that authored the commit (distinct from the + /// embedded git author identity). Present when the commit author maps + /// to a GitHub account; absent for unlinked email-only authors. + #[serde(default)] + pub(crate) author: Option, +} + +#[derive(Debug, Deserialize)] +pub(crate) struct GhCommitInner { + pub(crate) message: String, + pub(crate) author: Option, + pub(crate) committer: Option, +} + +#[derive(Debug, Deserialize)] +pub(crate) struct GhAuthor { + pub(crate) name: Option, + pub(crate) email: Option, + pub(crate) date: Option, +} + +/// An issue list entry (`GET /repos/{owner}/{repo}/issues`). +#[derive(Debug, Clone, Deserialize)] +pub(crate) struct GhIssue { + pub(crate) number: u64, + pub(crate) title: String, + pub(crate) body: Option, + pub(crate) state: String, + pub(crate) user: Option, + pub(crate) labels: Vec, + pub(crate) created_at: Option, + pub(crate) updated_at: Option, + /// Present when the row is actually a pull request (the issues endpoint + /// returns PRs with a `pull_request` envelope). + pub(crate) pull_request: Option, +} + +#[derive(Debug, Clone, Deserialize)] +pub(crate) struct GhUser { + pub(crate) login: String, +} + +#[derive(Debug, Clone, Deserialize)] +pub(crate) struct GhLabel { + pub(crate) name: String, +} + +/// A pull request list entry. +#[derive(Debug, Clone, Deserialize)] +pub(crate) struct GhPr { + pub(crate) number: u64, + pub(crate) title: String, + pub(crate) body: Option, + pub(crate) state: String, + pub(crate) user: Option, + pub(crate) labels: Vec, + pub(crate) created_at: Option, + pub(crate) updated_at: Option, + pub(crate) merged_at: Option, +} + +/// A comment on an issue or PR, slimmed to the fields the reader renders. +#[derive(Debug, Clone)] +pub(crate) struct IssueComment { + pub(crate) user: String, + pub(crate) body: String, + pub(crate) created_at: String, +} + +/// What kind of GitHub item a list row refers to. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum ItemKind { + Commit, + Issue, + PullRequest, +} + +impl ItemKind { + /// Parse a `SourceItem` id (`commit:`, `issue:`, `pr:`) into + /// its kind and the ref (sha / number) that follows the prefix. + pub(crate) fn from_id(id: &str) -> Option<(Self, &str)> { + if let Some(rest) = id.strip_prefix("commit:") { + Some((ItemKind::Commit, rest)) + } else if let Some(rest) = id.strip_prefix("issue:") { + Some((ItemKind::Issue, rest)) + } else if let Some(rest) = id.strip_prefix("pr:") { + Some((ItemKind::PullRequest, rest)) + } else { + None + } + } +} + +/// A cached issue/PR row, keyed by its list id (`"/:"`). +/// The issues endpoint returns pull requests mixed in with issues, and the PR +/// endpoint is the only one that returns merge state, so the list pass stashes +/// the full row here and the read pass reuses it instead of re-fetching. +#[derive(Debug, Clone)] +pub(crate) enum CachedItem { + Issue(GhIssue), + Pr(GhPr), +} + +/// Process-wide cache of issue/PR list rows, cleared at the start of each +/// `list_items` run so stale data from a prior sync can't leak in. +pub(crate) static LIST_CACHE: LazyLock>> = + LazyLock::new(|| Mutex::new(HashMap::new())); diff --git a/crates/tinymemory-integrations/src/sources/readers/mod.rs b/crates/tinymemory-integrations/src/sources/readers/mod.rs index f8e1d9d9..503a21ee 100644 --- a/crates/tinymemory-integrations/src/sources/readers/mod.rs +++ b/crates/tinymemory-integrations/src/sources/readers/mod.rs @@ -8,16 +8,39 @@ //! //! ## Ownership boundary //! -//! Every reader here is local: [`folder::FolderReader`], [`file::FileReader`] -//! and [`conversation::ConversationReader`] read the workspace and nothing -//! else. Network-backed sources (Composio toolkits, GitHub, RSS, web pages) -//! are not part of this crate. [`is_locally_readable`] and [`reader_for`] -//! therefore cover every [`SourceKind`]. +//! The local kinds ([`folder::FolderReader`], [`file::FileReader`], +//! [`conversation::ConversationReader`]) are always compiled. The network +//! kinds (`github`, `rss`, `web_page`) sit behind the `sources-network` +//! feature; `rss` and `web_page` fetch through `sources::fetch`. What this module does **not** own is *when* a +//! network read happens: scheduling, polling cadence, OAuth, credentials, +//! and egress/cost budgeting stay with the host. +//! +//! That is why [`reader_for`] and [`is_locally_readable`] draw their line at +//! **local vs. network**, not at implemented vs. absent. A network reader is +//! constructed explicitly (`github::GithubReader`, `rss::RssReader`, +//! `web_page::WebPageReader`) by a caller that has already decided the fetch is +//! allowed; it is never handed out by the kind-dispatch that a sync loop drives +//! on a timer. A `None` from [`reader_for`] therefore means "route this through +//! the host's sync runner", which keeps the host in charge of the network. +//! +//! `composio` is represented by a placeholder reader +//! ([`composio::ComposioReader`]): its data arrives through the credentialed +//! provider pipeline, and [`crate::sources::composio`] turns those payloads into items. +//! +//! A host servicing an *explicit user request* (not a timer) that wants one +//! reader for any kind uses `reader_for_request`. +pub mod composio; pub mod conversation; pub mod file; pub mod folder; +#[cfg(feature = "sources-network")] +pub mod github; pub mod local_file; +#[cfg(feature = "sources-network")] +pub mod rss; +#[cfg(feature = "sources-network")] +pub mod web_page; use std::path::Path; @@ -92,8 +115,8 @@ pub trait SourceReader: Send + Sync + std::fmt::Debug { /// Whether a kind can be read from local state alone, with no network egress. /// -/// Every [`SourceKind`] is local, so this is always `true`; it stays as the -/// single place a host asks the question. +/// Network-backed kinds return `false` even when this build ships their reader +/// (see the module docs): the host decides when a fetch is allowed. #[must_use] pub fn is_locally_readable(kind: &SourceKind) -> bool { matches!( @@ -102,14 +125,43 @@ pub fn is_locally_readable(kind: &SourceKind) -> bool { ) } -/// Get the reader for a source kind. +/// Get the reader for a source kind that is safe to drive on a timer. /// -/// Returns the folder, file or conversation reader. +/// Returns `Some` for [`SourceKind::Folder`], [`SourceKind::File`] and +/// [`SourceKind::Conversation`]. Network-backed kinds (`composio`, +/// `github_repo`, `rss_feed`, `web_page`) return `None` so the caller defers to +/// the host's sync runner, which constructs those readers once it has +/// authorized the fetch. #[must_use] pub fn reader_for(kind: &SourceKind) -> Option> { match kind { SourceKind::Folder => Some(Box::new(folder::FolderReader)), SourceKind::File => Some(Box::new(file::FileReader)), SourceKind::Conversation => Some(Box::new(conversation::ConversationReader)), + SourceKind::Composio + | SourceKind::GithubRepo + | SourceKind::RssFeed + | SourceKind::WebPage => None, + } +} + +/// Get a reader for **any** source kind, for a caller servicing an explicit user +/// request naming one source (an RPC handler), not a timer. +/// +/// Unlike [`reader_for`] this hands out the network readers, so the caller has +/// already decided the fetch is allowed. **Do not reuse it from a polling +/// loop**: the host stays in charge of egress, OAuth and cost budgeting by +/// constructing a network reader deliberately there. +#[cfg(feature = "sources-network")] +#[must_use] +pub fn reader_for_request(kind: &SourceKind) -> Box { + match kind { + SourceKind::Composio => Box::new(composio::ComposioReader), + SourceKind::Conversation => Box::new(conversation::ConversationReader), + SourceKind::Folder => Box::new(folder::FolderReader), + SourceKind::File => Box::new(file::FileReader), + SourceKind::GithubRepo => Box::new(github::GithubReader), + SourceKind::RssFeed => Box::new(rss::RssReader::new()), + SourceKind::WebPage => Box::new(web_page::WebPageReader), } } diff --git a/crates/tinymemory-integrations/src/sources/readers/rss/mod.rs b/crates/tinymemory-integrations/src/sources/readers/rss/mod.rs new file mode 100644 index 00000000..ec042713 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/readers/rss/mod.rs @@ -0,0 +1,355 @@ +//! RSS/Atom feed source reader. +//! +//! Fetches and parses an RSS or Atom feed, returning entries as +//! source items. Uses a lightweight XML parser (`quick-xml` via +//! manual parsing) to avoid pulling in heavy feed crates. +//! +//! Fetches go through `sources::fetch` and its SSRF guard (scheme/host +//! policy, a DNS resolver that pins connections to globally routable +//! addresses, and per-hop redirect re-checks), and the parsed feed is cached briefly so a +//! list-then-read sync pass downloads it once rather than once per entry. + +mod types; + +use std::sync::Mutex; +use std::time::{Duration, Instant}; + +use async_trait::async_trait; + +use crate::sources::error::{Error, Result}; +use crate::sources::types::{ + ContentType, MemorySourceEntry, SourceContent, SourceItem, SourceKind, +}; + +use super::SourceReader; +use crate::documents::html::decode_entities; +use crate::sources::fetch::fetch_url_capped; +use types::{FeedCache, FeedEntry}; + +const DEFAULT_MAX_ITEMS: u32 = 50; +const MAX_FEED_BYTES: u64 = 5 * 1024 * 1024; // 5 MiB — guards against pathological feeds + +/// How long a fetched feed is reused before the next read re-downloads it. +/// +/// Kept short so a feed that updates mid-sync is picked up on the next sync; +/// long enough to cover a list-then-read pass over a 50-entry feed. +const FEED_CACHE_TTL: Duration = Duration::from_secs(60); + +/// Reader for an RSS/Atom feed source. +/// +/// Holds a short-lived cache of the last fetched feed so that a `list_items` +/// immediately followed by per-item `read_item` calls fetches the feed once. +#[derive(Debug)] +pub struct RssReader { + cache: Mutex>, +} + +impl RssReader { + /// A reader with an empty feed cache. + #[must_use] + pub fn new() -> Self { + Self::default() + } + + /// Fetch (or reuse a very fresh copy of) the feed at `url`. + /// + /// The workspace sync pipeline holds one reader across a tick and calls + /// `list_items` once, then `read_item` once per entry. Without a cache + /// that is N+1 downloads of the same feed per sync (and a rate-limit + /// risk against the feed host); the cache turns it into one fetch whose + /// results are reused for the read phase. + async fn fetch_entries(&self, url: &str) -> Result> { + // Read the cache in a nested scope so the mutex guard is dropped before + // the await below — the guard is not `Send`, and holding it across an + // await would make the reader's async methods non-`Send`. + { + let cache = self.cache.lock().unwrap_or_else(|e| e.into_inner()); + if let Some(cached) = cache.as_ref() + && cached.url == url + && cached.fetched_at.elapsed() < FEED_CACHE_TTL + { + return Ok(cached.entries.clone()); + } + } + + // `fetch_url_capped` applies the SSRF guard and streams the body + // against the cap, so a pathological feed cannot exhaust memory. + let document = fetch_url_capped(url, MAX_FEED_BYTES).await?; + let body = String::from_utf8(document.bytes) + .map_err(|e| Error::Reader(format!("feed body is not valid UTF-8: {e}")))?; + let entries = parse_feed_full(&body).map_err(Error::Reader)?; + *self.cache.lock().unwrap_or_else(|e| e.into_inner()) = Some(FeedCache { + url: url.to_string(), + fetched_at: Instant::now(), + entries: entries.clone(), + }); + Ok(entries) + } +} + +impl Default for RssReader { + fn default() -> Self { + Self { + cache: Mutex::new(None), + } + } +} + +#[async_trait] +impl SourceReader for RssReader { + fn kind(&self) -> SourceKind { + SourceKind::RssFeed + } + + async fn list_items( + &self, + source: &MemorySourceEntry, + _workspace: &std::path::Path, + ) -> Result> { + let url = configured_url(source)?; + let max_items = source.max_items.unwrap_or(DEFAULT_MAX_ITEMS) as usize; + + tracing::debug!( + host = %url_host(url), + max_items = max_items, + "[memory_sources:rss] listing items" + ); + + let entries = self.fetch_entries(url).await?; + + tracing::debug!(count = entries.len(), "[memory_sources:rss] parsed entries"); + + Ok(entries + .into_iter() + .take(max_items) + .map(|e| SourceItem { + id: e.id, + title: e.title, + updated_at_ms: e.updated_at_ms, + }) + .collect()) + } + + async fn read_item( + &self, + source: &MemorySourceEntry, + item_id: &str, + _workspace: &std::path::Path, + ) -> Result { + let url = configured_url(source)?; + + tracing::debug!( + host = %url_host(url), + item_id = %item_id, + "[memory_sources:rss] reading item" + ); + + let entries = self.fetch_entries(url).await?; + let entry = entries + .into_iter() + .find(|e| e.id == item_id) + .ok_or_else(|| Error::NotFound(format!("item '{item_id}' not found in feed")))?; + + let content_type = if entry.body.contains('<') { + ContentType::Html + } else { + ContentType::Plaintext + }; + + Ok(SourceContent { + id: entry.id, + title: entry.title, + body: entry.body, + content_type, + metadata: serde_json::json!({ + "link": entry.link, + "published": entry.published, + }), + }) + } +} + +/// The configured feed URL. +fn configured_url(source: &MemorySourceEntry) -> Result<&str> { + source + .url + .as_deref() + .ok_or_else(|| Error::Invalid("rss source requires a url".to_string())) +} + +/// Extract just the host portion of a URL for debug-log redaction so we +/// don't leak query params, paths, or embedded credentials (userinfo). +fn url_host(url: &str) -> String { + // A real parse drops any `user:pass@` prefix via `host_str()`. When the + // value is not a parseable URL (it will be rejected by the SSRF guard + // later anyway), fall back to a textual host extraction that still strips + // userinfo and the path/query/fragment. + reqwest::Url::parse(url) + .ok() + .and_then(|u| u.host_str().map(str::to_string)) + .unwrap_or_else(|| { + let authority = url + .trim_start_matches("https://") + .trim_start_matches("http://") + .split(['/', '?', '#']) + .next() + .unwrap_or(url); + // Only the last `@`-separated segment can be the host; anything + // before it is credentials and must not reach the log. + authority + .rsplit('@') + .next() + .unwrap_or(authority) + .to_string() + }) +} + +fn parse_feed_full(xml: &str) -> std::result::Result, String> { + // Detect RSS vs Atom by looking for std::result::Result, String> { + let mut entries = Vec::new(); + let mut offset = 0; + + while let Some(item_start) = xml[offset..].find("") + .map(|i| abs_start + i + 7) + .unwrap_or(xml.len()); + + let item_xml = &xml[abs_start..item_end]; + let title = extract_tag(item_xml, "title").unwrap_or_default(); + let link = extract_tag(item_xml, "link"); + let guid = extract_tag(item_xml, "guid"); + let description = extract_tag(item_xml, "description") + // An empty `` is a present-but-empty + // tag: `extract_tag` returns `Some("")`, which would short-circuit + // the `content:encoded` fallback below and ingest an empty body. + // Filter it out so a populated `content:encoded` still wins. + .filter(|s| !s.is_empty()) + .or_else(|| extract_cdata(item_xml, "content:encoded")) + .unwrap_or_default(); + let pub_date = extract_tag(item_xml, "pubDate"); + + let id = guid + .or_else(|| link.clone()) + .unwrap_or_else(|| format!("rss-{}", entries.len())); + + entries.push(FeedEntry { + updated_at_ms: pub_date.as_deref().and_then(rss_timestamp_ms), + id, + title, + body: description, + link, + published: pub_date, + }); + + offset = item_end; + } + + Ok(entries) +} + +fn parse_atom(xml: &str) -> std::result::Result, String> { + let mut entries = Vec::new(); + let mut offset = 0; + + while let Some(entry_start) = xml[offset..].find("") + .map(|i| abs_start + i + 8) + .unwrap_or(xml.len()); + + let entry_xml = &xml[abs_start..entry_end]; + let title = extract_tag(entry_xml, "title").unwrap_or_default(); + let id = extract_tag(entry_xml, "id").unwrap_or_else(|| format!("atom-{}", entries.len())); + let content = extract_tag(entry_xml, "content") + // Same shape as the RSS `description`/`content:encoded` pair: an + // empty `` must not block the `summary` + // fallback. + .filter(|s| !s.is_empty()) + .or_else(|| extract_tag(entry_xml, "summary")) + .unwrap_or_default(); + let link = extract_attr(entry_xml, "link", "href"); + let updated = + extract_tag(entry_xml, "updated").or_else(|| extract_tag(entry_xml, "published")); + + entries.push(FeedEntry { + updated_at_ms: updated.as_deref().and_then(rss_timestamp_ms), + id, + title, + body: content, + link, + published: updated, + }); + + offset = entry_end; + } + + Ok(entries) +} + +/// Parse an RSS `pubDate` (RFC 2822) or Atom `updated`/`published` (RFC 3339) +/// timestamp into epoch milliseconds, so workspace sync can skip unchanged +/// entries instead of re-reading every item on every pass. +fn rss_timestamp_ms(value: &str) -> Option { + chrono::DateTime::parse_from_rfc2822(value) + .or_else(|_| chrono::DateTime::parse_from_rfc3339(value)) + .map(|dt| dt.timestamp_millis()) + .ok() +} + +/// Remove a surrounding `` wrapper, if present. +fn unwrap_cdata(s: &str) -> &str { + s.strip_prefix("")) + .unwrap_or(s) +} + +fn extract_tag(xml: &str, tag: &str) -> Option { + let open = format!("<{tag}"); + let close = format!(""); + let start = xml.find(&open)?; + let content_start = xml[start..].find('>')? + start + 1; + let end = xml[content_start..].find(&close)? + content_start; + let content = &xml[content_start..end]; + let trimmed = content.trim(); + let unwrapped = unwrap_cdata(trimmed).trim(); + // CDATA content is literal text, so entity decoding applies only outside + // a CDATA wrapper; decoding `<` inside one would corrupt the content. + if trimmed.starts_with(" Option { + // `extract_tag` already unwraps ``, so it serves both the + // plain-text and CDATA-wrapped shapes. + extract_tag(xml, tag) +} + +fn extract_attr(xml: &str, tag: &str, attr: &str) -> Option { + let open = format!("<{tag} "); + let start = xml.find(&open)?; + let tag_end = xml[start..].find('>')? + start; + let tag_str = &xml[start..tag_end]; + let attr_start = tag_str.find(&format!("{attr}=\""))? + attr.len() + 2; + let attr_end = tag_str[attr_start..].find('"')? + attr_start; + Some(tag_str[attr_start..attr_end].to_string()) +} + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinymemory-integrations/src/sources/readers/rss/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/rss/mod_tests.rs new file mode 100644 index 00000000..cf6fc2ca --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/readers/rss/mod_tests.rs @@ -0,0 +1,386 @@ +//! Tests for the RSS/Atom reader: the cached list-then-read flow, URL +//! refusal, and the feed parser. + +use super::*; + +use crate::sources::readers::SourceReader; + +fn cached_reader(url: &str) -> RssReader { + RssReader { + cache: Mutex::new(Some(types::FeedCache { + url: url.to_string(), + fetched_at: Instant::now(), + entries: vec![ + types::FeedEntry { + id: "plain".into(), + title: "Plain entry".into(), + body: "plain body".into(), + link: Some("https://example.com/plain".into()), + published: Some("2026-01-01T00:00:00Z".into()), + updated_at_ms: Some(1_767_225_600_000), + }, + types::FeedEntry { + id: "html".into(), + title: "HTML entry".into(), + body: "

html body

".into(), + link: None, + published: None, + updated_at_ms: None, + }, + ], + })), + } +} + +fn rss_source(url: Option<&str>, max_items: Option) -> MemorySourceEntry { + MemorySourceEntry { + id: "feed".into(), + label: "Feed".into(), + kind: SourceKind::RssFeed, + enabled: true, + toolkit: None, + connection_id: None, + path: None, + glob: None, + url: url.map(str::to_string), + branch: None, + paths: Vec::new(), + max_commits: None, + max_issues: None, + max_prs: None, + max_items, + selector: None, + max_tokens_per_sync: None, + max_cost_per_sync_usd: None, + sync_depth_days: None, + } +} + +#[tokio::test] +async fn cached_feed_drives_list_and_read_without_network() { + let url = "https://example.com/feed.xml"; + let reader = cached_reader(url); + assert_eq!(reader.kind(), SourceKind::RssFeed); + + let listed = reader + .list_items(&rss_source(Some(url), Some(1)), std::path::Path::new(".")) + .await + .expect("cached list"); + assert_eq!(listed.len(), 1); + assert_eq!(listed[0].id, "plain"); + assert_eq!(listed[0].updated_at_ms, Some(1_767_225_600_000)); + + let plain = reader + .read_item( + &rss_source(Some(url), None), + "plain", + std::path::Path::new("."), + ) + .await + .expect("cached plaintext item"); + assert_eq!(plain.content_type, ContentType::Plaintext); + assert_eq!(plain.metadata["link"], "https://example.com/plain"); + + let html = reader + .read_item( + &rss_source(Some(url), None), + "html", + std::path::Path::new("."), + ) + .await + .expect("cached HTML item"); + assert_eq!(html.content_type, ContentType::Html); +} + +#[tokio::test] +async fn rss_reader_reports_missing_configuration_and_items() { + let reader = RssReader::new(); + let missing_url = rss_source(None, None); + assert!( + reader + .list_items(&missing_url, std::path::Path::new(".")) + .await + .is_err() + ); + assert!( + reader + .read_item(&missing_url, "anything", std::path::Path::new(".")) + .await + .is_err() + ); + + let url = "https://example.com/feed.xml"; + assert!( + cached_reader(url) + .read_item( + &rss_source(Some(url), None), + "missing", + std::path::Path::new("."), + ) + .await + .is_err() + ); +} + +#[tokio::test] +async fn blocked_and_malformed_feed_urls_fail_closed_without_network() { + for url in ["not a URL", "file:///etc/passwd", "http://127.0.0.1/feed"] { + let error = RssReader::new() + .list_items(&rss_source(Some(url), None), std::path::Path::new(".")) + .await + .expect_err("unsafe URL must be refused"); + assert!(!error.to_string().is_empty()); + } +} + +#[test] +fn parse_rss_extracts_items() { + let xml = r#" + + + Test Feed + + First post + https://example.com/1 + Body of first post + + + Second post + guid-2 + Body of second + + + "#; + + let entries = parse_rss(xml).unwrap(); + assert_eq!(entries.len(), 2); + assert_eq!(entries[0].title, "First post"); + assert_eq!(entries[0].id, "https://example.com/1"); + assert_eq!(entries[1].id, "guid-2"); +} + +#[test] +fn parse_atom_extracts_entries() { + let xml = r#" + + + Atom entry + urn:entry:1 + Content here + + + "#; + + let entries = parse_atom(xml).unwrap(); + assert_eq!(entries.len(), 1); + assert_eq!(entries[0].title, "Atom entry"); + assert_eq!(entries[0].id, "urn:entry:1"); + assert_eq!( + entries[0].link.as_deref(), + Some("https://example.com/atom/1") + ); +} + +#[test] +fn parse_feed_detects_format() { + let rss = "T"; + assert!(parse_feed_full(rss).is_ok()); + + let atom = "T1"; + assert!(parse_feed_full(atom).is_ok()); + + assert!(parse_feed_full("").is_err()); +} + +// ── Entry timestamps ─────────────────────────────────────────────── + +#[test] +fn parse_rss_emits_pubdate_timestamp() { + let xml = r#" + + Post + 1 + Wed, 02 Oct 2002 13:00:00 GMT + + "#; + + let entries = parse_rss(xml).unwrap(); + // 2002-10-02T13:00:00Z in epoch milliseconds. + assert_eq!(entries[0].updated_at_ms, Some(1_033_563_600_000)); +} + +#[test] +fn parse_atom_emits_updated_timestamp() { + let xml = r#" + + Atom entry + urn:entry:1 + 2026-03-01T12:00:00.000Z + + "#; + + let entries = parse_atom(xml).unwrap(); + // 2026-03-01T12:00:00Z in epoch milliseconds. + assert_eq!(entries[0].updated_at_ms, Some(1_772_366_400_000)); +} + +#[test] +fn parse_item_without_timestamp_is_unknown() { + let xml = "T"; + let entries = parse_rss(xml).unwrap(); + assert_eq!(entries[0].updated_at_ms, None); +} + +#[test] +fn rss_timestamp_ms_rejects_garbage() { + assert_eq!(rss_timestamp_ms("not a date"), None); + assert_eq!(rss_timestamp_ms(""), None); +} + +// ── CDATA unwrapping ──────────────────────────────────────────────── + +#[test] +fn extract_tag_unwraps_cdata() { + // The common RSS shape `body

]]>
` + // must yield clean HTML, not the literal CDATA markers. + let xml = "body

]]>
"; + assert_eq!( + extract_tag(xml, "description").as_deref(), + Some("

body

") + ); +} + +#[test] +fn extract_tag_does_not_entity_decode_inside_cdata() { + // CDATA content is literal: `<` must survive intact, not become `<`. + let xml = ""; + assert_eq!( + extract_tag(xml, "description").as_deref(), + Some("Say <tag> literally") + ); +} + +#[test] +fn extract_tag_entity_decodes_outside_cdata() { + let xml = "A & B <b> bold"; + assert_eq!(extract_tag(xml, "title").as_deref(), Some("A & B bold")); +} + +#[test] +fn extract_cdata_reuses_tag_extraction() { + // `content:encoded` is typically CDATA-wrapped; extract_cdata must match + // extract_tag on the same input. + let xml = "full

]]>
"; + assert_eq!( + extract_cdata(xml, "content:encoded").as_deref(), + Some("

full

") + ); +} + +#[test] +fn parse_rss_description_with_cdata_is_clean() { + let xml = r#" + + Post + 1 + Hello world

]]>
+
+
"#; + + let entries = parse_rss(xml).unwrap(); + assert_eq!(entries.len(), 1); + assert_eq!(entries[0].body, "

Hello world

"); + assert!(!entries[0].body.contains("CDATA")); +} + +#[test] +fn parse_rss_empty_description_falls_back_to_encoded_content() { + // A present-but-empty `` must not block the + // `content:encoded` fallback — the item carries its body there instead. + let xml = r#" + + Post + 1 + + Full body from content:encoded

]]>
+
+
"#; + + let entries = parse_rss(xml).unwrap(); + assert_eq!(entries.len(), 1); + assert_eq!(entries[0].body, "

Full body from content:encoded

"); +} + +#[test] +fn parse_atom_empty_content_falls_back_to_summary() { + // Mirrors the RSS `description`/`content:encoded` pair: an empty + // `` must fall through to a populated ``. + let xml = r#" + + Atom entry + urn:entry:1 + + Summary body + + "#; + + let entries = parse_atom(xml).unwrap(); + assert_eq!(entries.len(), 1); + assert_eq!(entries[0].body, "Summary body"); +} + +// ── URL host redaction ────────────────────────────────────────────── + +#[test] +fn url_host_redacts_userinfo() { + // Credentials embedded in a source URL must never reach debug traces — + // only the host is logged. + assert_eq!( + url_host("https://alice:secret@example.com/feed.xml"), + "example.com" + ); +} + +#[test] +fn url_host_drops_path_query_and_fragment() { + assert_eq!( + url_host("https://example.com/feed?token=abc#top"), + "example.com" + ); +} + +#[test] +fn url_host_fallback_strips_userinfo_without_scheme() { + // Scheme-less values are unparseable by reqwest; the textual fallback + // must still drop the `user:pass@` prefix and the path. + assert_eq!(url_host("alice:secret@example.com/feed"), "example.com"); + assert_eq!(url_host("example.com/feed"), "example.com"); +} + +// ── Entity decoding ───────────────────────────────────────────────── + +#[test] +fn feed_text_entities_decode_exactly_once() { + // `&lt;` is the escaped form of `<`; it must decode once to `<`, + // not twice to `<`. + assert_eq!( + extract_tag("&lt;", "title").as_deref(), + Some("<") + ); + assert_eq!( + extract_tag("&amp;", "title").as_deref(), + Some("&") + ); +} + +#[test] +fn feed_text_decodes_every_predefined_xml_entity_and_numeric_references() { + assert_eq!( + extract_tag( + "<b> "q" 'a' & more ’", + "title" + ) + .as_deref(), + Some(" \"q\" 'a' & more \u{2019}") + ); +} diff --git a/crates/tinymemory-integrations/src/sources/readers/rss/types.rs b/crates/tinymemory-integrations/src/sources/readers/rss/types.rs new file mode 100644 index 00000000..db94f9ce --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/readers/rss/types.rs @@ -0,0 +1,25 @@ +//! Private feed-shape types for the RSS reader: the parsed entry model and +//! the short-lived cache snapshot reused across a list-then-read sync pass. + +use std::time::Instant; + +/// A fetched feed snapshot cached across a list-then-read sync pass. +#[derive(Debug)] +pub(super) struct FeedCache { + pub url: String, + pub fetched_at: Instant, + pub entries: Vec, +} + +/// One parsed RSS/Atom entry: the dedupe id, the fields surfaced as a source +/// item / content, and the raw link + publication timestamp carried in the +/// content metadata. +#[derive(Debug, Clone)] +pub(super) struct FeedEntry { + pub id: String, + pub title: String, + pub body: String, + pub link: Option, + pub published: Option, + pub updated_at_ms: Option, +} diff --git a/crates/tinymemory-integrations/src/sources/readers/web_page/mod.rs b/crates/tinymemory-integrations/src/sources/readers/web_page/mod.rs new file mode 100644 index 00000000..f7841dac --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/readers/web_page/mod.rs @@ -0,0 +1,444 @@ +//! Web page source reader. +//! +//! Fetches a single URL and extracts its content. When a CSS `selector` is +//! configured, only the text of matching elements is included (plain text); +//! otherwise the whole page is converted to markdown through +//! `tinymemory_integrations::documents::html::to_markdown`, keeping its headings, lists and +//! links. +//! +//! The page is fetched through `sources::fetch`, behind its SSRF guard +//! (scheme/host policy plus a DNS resolver that pins connections to globally +//! routable addresses), with a 10 MiB body cap. + +mod types; + +use async_trait::async_trait; + +use types::SelectorSpec; + +use crate::sources::fetch::fetch_url_capped; + +use crate::sources::error::{Error, Result}; +use crate::sources::types::{ + ContentType, MemorySourceEntry, SourceContent, SourceItem, SourceKind, +}; + +use super::SourceReader; + +/// Largest page body the reader will buffer. +const MAX_BODY_BYTES: u64 = 10 * 1024 * 1024; + +/// Reader for a single-page web source: fetches one URL and extracts its +/// readable text. +#[derive(Debug, Clone, Copy, Default)] +pub struct WebPageReader; + +#[async_trait] +impl SourceReader for WebPageReader { + fn kind(&self) -> SourceKind { + SourceKind::WebPage + } + + async fn list_items( + &self, + source: &MemorySourceEntry, + _workspace: &std::path::Path, + ) -> Result> { + let url = configured_url(source)?; + Ok(vec![SourceItem { + id: url.to_string(), + title: source.label.clone(), + updated_at_ms: None, + }]) + } + + async fn read_item( + &self, + source: &MemorySourceEntry, + item_id: &str, + _workspace: &std::path::Path, + ) -> Result { + let url = if item_id.starts_with("http") { + item_id.to_string() + } else { + configured_url(source)?.to_string() + }; + + tracing::debug!( + selector = ?source.selector, + "[memory_sources:web_page] reading item" + ); + + // `fetch_url_capped` applies the SSRF guard (scheme and host policy, + // public-only DNS, per-hop redirect checks) and streams the body + // against the cap, so a hostile or giant page cannot exhaust memory. + let document = fetch_url_capped(&url, MAX_BODY_BYTES).await?; + let body = String::from_utf8_lossy(&document.bytes).into_owned(); + + let title = crate::documents::html::extract_title(&body).unwrap_or_else(|| url.clone()); + let (extracted, content_type) = match source.selector.as_deref() { + Some(selector) => (extract_by_selector(&body, selector), ContentType::Plaintext), + None => ( + crate::documents::html::to_markdown(&body), + ContentType::Markdown, + ), + }; + + Ok(SourceContent { + id: url.clone(), + title, + body: extracted, + content_type, + metadata: serde_json::json!({ "url": url }), + }) + } +} + +/// The configured page URL. +fn configured_url(source: &MemorySourceEntry) -> Result<&str> { + source + .url + .as_deref() + .ok_or_else(|| Error::Invalid("web_page source requires a url".to_string())) +} + +// ── Text extraction ───────────────────────────────────────────────── + +fn parse_selector(selector: &str) -> Option { + let last = selector + .trim() + .rsplit(char::is_whitespace) + .next() + .unwrap_or("") + .trim(); + if last.is_empty() { + return None; + } + + let mut spec = SelectorSpec { + tag: None, + id: None, + classes: Vec::new(), + }; + let mut part = String::new(); + let mut sep = ' '; // leading bare token is the tag + for ch in last.chars() { + match ch { + '.' | '#' => { + push_selector_part(&mut spec, &mut part, sep); + sep = ch; + } + _ => part.push(ch), + } + } + push_selector_part(&mut spec, &mut part, sep); + + if spec.tag.is_none() && spec.id.is_none() && spec.classes.is_empty() { + None + } else { + Some(spec) + } +} + +fn push_selector_part(spec: &mut SelectorSpec, part: &mut String, sep: char) { + let part = std::mem::take(part); + if part.is_empty() { + return; + } + match sep { + '#' => spec.id = Some(part), + '.' => spec.classes.push(part), + _ => { + if spec.tag.is_none() { + spec.tag = Some(part); + } else { + spec.classes.push(part); + } + } + } +} + +/// Extract text from elements matching a simple CSS selector. +/// +/// Falls back to the whole stripped page when the selector never matches +/// (rather than erroring), mirroring the reader's lenient posture for pages +/// whose structure changes between list and read time. +fn extract_by_selector(html: &str, selector: &str) -> String { + let Some(spec) = parse_selector(selector) else { + return strip_html_tags(html); + }; + // Match against a script/style-stripped copy so JS strings and CSS rules + // cannot be mistaken for nested elements, and so a selector that lands on + // a `

World

"; + let result = strip_html_tags(html); + assert_eq!(result, "Hello World"); +} + +#[test] +fn strip_script_and_style_handles_unclosed() { + // Unclosed `"; + let result = extract_by_selector(html, ".content"); + assert!(result.contains("Real")); + assert!(!result.contains("Fake")); +} diff --git a/crates/tinymemory-integrations/src/sources/readers/web_page/types.rs b/crates/tinymemory-integrations/src/sources/readers/web_page/types.rs new file mode 100644 index 00000000..cbcc2767 --- /dev/null +++ b/crates/tinymemory-integrations/src/sources/readers/web_page/types.rs @@ -0,0 +1,13 @@ +//! Types for the web-page reader. + +/// A parsed simple CSS selector: optional tag name, optional id, and class +/// names. Supports `tag`, `tag.class`, `tag#id`, `.class`, `#id`, and stacked +/// classes (`tag.a.b`, `.a.b`). A descendant/child chain (`div.content p`) +/// targets the final compound selector — a full CSS engine is out of scope for +/// this reader. +#[derive(Debug)] +pub(super) struct SelectorSpec { + pub tag: Option, + pub id: Option, + pub classes: Vec, +} diff --git a/crates/tinymemory-integrations/src/sources/types/mod.rs b/crates/tinymemory-integrations/src/sources/types/mod.rs index a08ef08c..156f2589 100644 --- a/crates/tinymemory-integrations/src/sources/types/mod.rs +++ b/crates/tinymemory-integrations/src/sources/types/mod.rs @@ -25,43 +25,69 @@ pub(crate) fn default_true() -> bool { /// The kind of a configured memory source. /// -/// The wire representation is snake_case (`folder`, `file`, `conversation`) and is +/// The wire representation is snake_case (`github_repo`, `rss_feed`, …) and is /// persisted by hosts; it must stay stable across versions. Each maps /// onto one [`tinymemory_api::SourceKind`] through [`SourceKind::api_kind`]. #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] #[serde(rename_all = "snake_case")] pub enum SourceKind { + /// A Composio OAuth connector (Gmail, Slack, Notion, …). Network-backed; + /// the live fetch is owned by the host, not this module. + Composio, /// Local agent conversation transcripts stored in the workspace. Conversation, /// A local folder of files matched by an optional glob. Folder, /// A single local file. File, + /// A GitHub repository's project activity (commits, issues, PRs). + GithubRepo, + /// An RSS/Atom feed. + RssFeed, + /// A single web page, optionally narrowed by a CSS selector. + WebPage, } impl SourceKind { /// Every kind, in declaration order. - pub const ALL: [Self; 3] = [Self::Conversation, Self::Folder, Self::File]; + pub const ALL: [Self; 7] = [ + Self::Composio, + Self::Conversation, + Self::Folder, + Self::File, + Self::GithubRepo, + Self::RssFeed, + Self::WebPage, + ]; /// The stable snake_case wire string for this kind. #[must_use] pub fn as_str(&self) -> &'static str { match self { + SourceKind::Composio => "composio", SourceKind::Conversation => "conversation", SourceKind::Folder => "folder", SourceKind::File => "file", + SourceKind::GithubRepo => "github_repo", + SourceKind::RssFeed => "rss_feed", + SourceKind::WebPage => "web_page", } } /// The contract's [`tinymemory_api::SourceKind`] for items this kind of - /// source produces: each keeps its name. + /// source produces: a web page is a `Link`, a GitHub repository `Github`, + /// an RSS feed `Rss`; the rest keep their name. #[must_use] pub fn api_kind(&self) -> tinymemory_api::SourceKind { use tinymemory_api::SourceKind as Api; match self { + SourceKind::Composio => Api::Composio, SourceKind::Conversation => Api::Conversation, SourceKind::Folder => Api::Folder, SourceKind::File => Api::File, + SourceKind::GithubRepo => Api::Github, + SourceKind::RssFeed => Api::Rss, + SourceKind::WebPage => Api::Link, } } } @@ -83,6 +109,14 @@ pub struct MemorySourceEntry { #[serde(default = "default_true")] pub enabled: bool, + // ── Composio ── + /// Composio toolkit slug (e.g. `gmail`). Required for `composio`. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub toolkit: Option, + /// Composio connection id. Required for `composio`. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub connection_id: Option, + // ── Folder / File ── /// Filesystem path of the folder or file to read. Required for `folder` /// and `file`; a relative path is anchored on the workspace. @@ -93,6 +127,38 @@ pub struct MemorySourceEntry { #[serde(default, skip_serializing_if = "Option::is_none")] pub glob: Option, + // ── GithubRepo / RssFeed / WebPage (shared) ── + /// Source URL. Required for `github_repo`, `rss_feed`, and `web_page`. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub url: Option, + + // ── GithubRepo ── + /// Branch to read (defaults to the repo default when absent). + #[serde(default, skip_serializing_if = "Option::is_none")] + pub branch: Option, + /// Optional path filters within the repo. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub paths: Vec, + /// Max commits to pull per sync (default 1000 when absent). + #[serde(default, skip_serializing_if = "Option::is_none")] + pub max_commits: Option, + /// Max issues to pull per sync (default 1000 when absent). + #[serde(default, skip_serializing_if = "Option::is_none")] + pub max_issues: Option, + /// Max pull requests to pull per sync (default 1000 when absent). + #[serde(default, skip_serializing_if = "Option::is_none")] + pub max_prs: Option, + + // ── RssFeed ── + /// Max feed items to pull per sync. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub max_items: Option, + + // ── WebPage ── + /// Optional CSS selector to narrow extracted content. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub selector: Option, + // ── Sync Budget (all source kinds) ── /// Maximum tokens to consume per sync run. Sync stops once this budget is hit. #[serde(default, skip_serializing_if = "Option::is_none")] @@ -114,8 +180,18 @@ impl MemorySourceEntry { kind, label: label.into(), enabled: true, + toolkit: None, + connection_id: None, path: None, glob: None, + url: None, + branch: None, + paths: Vec::new(), + max_commits: None, + max_issues: None, + max_prs: None, + max_items: None, + selector: None, max_tokens_per_sync: None, max_cost_per_sync_usd: None, sync_depth_days: None, @@ -125,8 +201,9 @@ impl MemorySourceEntry { /// Validate the fields this entry's [`SourceKind`] requires. /// /// `id` and `label` are required for every kind, and `id` must not contain - /// `:` or control characters. Folders and files need `path`. An empty string - /// counts as missing. + /// `:` or control characters. Composio needs `toolkit` and + /// `connection_id`; folders and files need `path`; GitHub repositories, RSS + /// feeds and web pages need `url`. An empty string counts as missing. /// /// # Errors /// @@ -144,8 +221,15 @@ impl MemorySourceEntry { return Err(Error::Invalid("label is required".to_string())); } match self.kind { + SourceKind::Composio => { + require_field(&self.toolkit, "toolkit")?; + require_field(&self.connection_id, "connection_id") + } SourceKind::Conversation => Ok(()), SourceKind::Folder | SourceKind::File => require_field(&self.path, "path"), + SourceKind::GithubRepo | SourceKind::RssFeed | SourceKind::WebPage => { + require_field(&self.url, "url") + } } } } diff --git a/crates/tinymemory-integrations/src/sources/types/mod_tests.rs b/crates/tinymemory-integrations/src/sources/types/mod_tests.rs index 65d252d3..2ea2171b 100644 --- a/crates/tinymemory-integrations/src/sources/types/mod_tests.rs +++ b/crates/tinymemory-integrations/src/sources/types/mod_tests.rs @@ -5,9 +5,13 @@ use super::*; #[test] fn source_kind_round_trips_via_serde() { for kind in [ + SourceKind::Composio, SourceKind::Conversation, SourceKind::Folder, SourceKind::File, + SourceKind::GithubRepo, + SourceKind::RssFeed, + SourceKind::WebPage, ] { let json = serde_json::to_string(&kind).unwrap(); let decoded: SourceKind = serde_json::from_str(&json).unwrap(); @@ -17,9 +21,33 @@ fn source_kind_round_trips_via_serde() { #[test] fn source_kind_as_str_matches_wire_strings() { + assert_eq!(SourceKind::Composio.as_str(), "composio"); assert_eq!(SourceKind::Conversation.as_str(), "conversation"); assert_eq!(SourceKind::Folder.as_str(), "folder"); + assert_eq!(SourceKind::GithubRepo.as_str(), "github_repo"); assert_eq!(SourceKind::File.as_str(), "file"); + assert_eq!(SourceKind::RssFeed.as_str(), "rss_feed"); + assert_eq!(SourceKind::WebPage.as_str(), "web_page"); +} + +#[test] +fn validate_composio_requires_toolkit_and_connection_id() { + let entry = MemorySourceEntry { + id: "src_1".into(), + kind: SourceKind::Composio, + label: "Gmail".into(), + enabled: true, + toolkit: Some("gmail".into()), + connection_id: None, + ..default_entry() + }; + assert!(entry.validate().is_err()); + + let valid = MemorySourceEntry { + connection_id: Some("cmp_123".into()), + ..entry + }; + assert!(valid.validate().is_ok()); } #[test] @@ -35,6 +63,19 @@ fn validate_folder_requires_path() { assert!(entry.validate().is_err()); } +#[test] +fn validate_github_requires_url() { + let entry = MemorySourceEntry { + id: "src_3".into(), + kind: SourceKind::GithubRepo, + label: "Repo".into(), + enabled: true, + url: Some("https://github.com/org/repo".into()), + ..default_entry() + }; + assert!(entry.validate().is_ok()); +} + #[test] fn validate_file_requires_path() { let entry = MemorySourceEntry::new("src_file", SourceKind::File, "One file"); @@ -47,24 +88,50 @@ fn validate_file_requires_path() { } #[test] -fn the_removed_network_kinds_no_longer_decode() { - for removed in [ - "twitter_query", - "composio", - "github_repo", - "rss_feed", - "web_page", - ] { - let decoded = serde_json::from_str::(&format!("\"{removed}\"")); - assert!(decoded.is_err(), "{removed} must not decode"); - } +fn the_removed_twitter_query_kind_no_longer_decodes() { + let decoded = serde_json::from_str::("\"twitter_query\""); + assert!(decoded.is_err()); } #[test] fn every_config_kind_maps_onto_a_contract_source_kind() { use tinymemory_api::SourceKind as Api; let mapped: Vec = SourceKind::ALL.iter().map(SourceKind::api_kind).collect(); - assert_eq!(mapped, vec![Api::Conversation, Api::Folder, Api::File]); + assert_eq!( + mapped, + vec![ + Api::Composio, + Api::Conversation, + Api::Folder, + Api::File, + Api::Github, + Api::Rss, + Api::Link, + ] + ); +} + +#[test] +fn validate_rss_and_web_page_require_url() { + let rss = MemorySourceEntry { + id: "src_rss".into(), + kind: SourceKind::RssFeed, + label: "Feed".into(), + enabled: true, + url: None, + ..default_entry() + }; + assert!(rss.validate().is_err()); + + let web = MemorySourceEntry { + id: "src_web".into(), + kind: SourceKind::WebPage, + label: "Page".into(), + enabled: true, + url: Some("https://example.com".into()), + ..default_entry() + }; + assert!(web.validate().is_ok()); } #[test] @@ -178,8 +245,18 @@ pub(super) fn default_entry() -> MemorySourceEntry { kind: SourceKind::Folder, label: String::new(), enabled: true, + toolkit: None, + connection_id: None, path: None, glob: None, + url: None, + branch: None, + paths: Vec::new(), + max_commits: None, + max_issues: None, + max_prs: None, + max_items: None, + selector: None, max_tokens_per_sync: None, max_cost_per_sync_usd: None, sync_depth_days: None, @@ -197,11 +274,21 @@ pub(super) fn default_entry() -> MemorySourceEntry { fn source_entry_wire_format_is_pinned() { let entry = MemorySourceEntry { id: "src_pinned".into(), - kind: SourceKind::Folder, + kind: SourceKind::GithubRepo, label: "Pinned".into(), enabled: false, + toolkit: Some("gmail".into()), + connection_id: Some("conn-1".into()), path: Some("/notes".into()), glob: Some("**/*.md".into()), + url: Some("https://github.com/tinyhumansai/tinymemory".into()), + branch: Some("main".into()), + paths: vec!["core/src".into()], + max_commits: Some(10), + max_issues: Some(20), + max_prs: Some(30), + max_items: Some(40), + selector: Some("article".into()), max_tokens_per_sync: Some(50_000), max_cost_per_sync_usd: Some(1.5), sync_depth_days: Some(90), @@ -211,11 +298,21 @@ fn source_entry_wire_format_is_pinned() { serde_json::to_value(&entry).unwrap(), serde_json::json!({ "id": "src_pinned", - "kind": "folder", + "kind": "github_repo", "label": "Pinned", "enabled": false, + "toolkit": "gmail", + "connection_id": "conn-1", "path": "/notes", "glob": "**/*.md", + "url": "https://github.com/tinyhumansai/tinymemory", + "branch": "main", + "paths": ["core/src"], + "max_commits": 10, + "max_issues": 20, + "max_prs": 30, + "max_items": 40, + "selector": "article", "max_tokens_per_sync": 50000, "max_cost_per_sync_usd": 1.5, "sync_depth_days": 90 diff --git a/crates/tinymemory-integrations/tests/reader_dispatch.rs b/crates/tinymemory-integrations/tests/reader_dispatch.rs index a18800c0..6cb12752 100644 --- a/crates/tinymemory-integrations/tests/reader_dispatch.rs +++ b/crates/tinymemory-integrations/tests/reader_dispatch.rs @@ -6,7 +6,7 @@ use tinymemory_integrations::sources::{ }; #[test] -fn dispatch_constructs_the_local_readers() { +fn timer_dispatch_constructs_only_readers_that_never_need_network() { for kind in [ SourceKind::Folder, SourceKind::File, @@ -15,12 +15,24 @@ fn dispatch_constructs_the_local_readers() { assert!(is_locally_readable(&kind)); assert_eq!(reader_for(&kind).map(|reader| reader.kind()), Some(kind)); } + + for kind in [ + SourceKind::Composio, + SourceKind::GithubRepo, + SourceKind::RssFeed, + SourceKind::WebPage, + ] { + assert!(!is_locally_readable(&kind)); + assert!(reader_for(&kind).is_none()); + } } +#[cfg(feature = "sources-network")] #[test] -fn every_kind_has_a_local_reader() { +fn request_dispatch_hands_out_a_reader_for_every_kind() { + use tinymemory_integrations::sources::readers::reader_for_request; + for kind in SourceKind::ALL { - assert!(is_locally_readable(&kind)); - assert!(reader_for(&kind).is_some()); + assert_eq!(reader_for_request(&kind).kind(), kind); } } diff --git a/crates/tinymemory-tools/src/layout/mod_tests.rs b/crates/tinymemory-tools/src/layout/mod_tests.rs index 2038660e..5ff0c61e 100644 --- a/crates/tinymemory-tools/src/layout/mod_tests.rs +++ b/crates/tinymemory-tools/src/layout/mod_tests.rs @@ -99,7 +99,7 @@ fn source_ids_round_trip() { } assert_eq!("MD".parse::().unwrap(), BrainSource::Markdown); assert!(" ".parse::().is_err()); - assert_eq!(BrainSource::Github.source_kind(), SourceKind::Import); + assert_eq!(BrainSource::Github.source_kind(), SourceKind::Github); } #[test] diff --git a/crates/tinymemory-tools/src/layout/source.rs b/crates/tinymemory-tools/src/layout/source.rs index ea2a3d25..dcf203d2 100644 --- a/crates/tinymemory-tools/src/layout/source.rs +++ b/crates/tinymemory-tools/src/layout/source.rs @@ -60,7 +60,10 @@ impl BrainSource { pub fn source_kind(&self) -> SourceKind { match self { Self::Files | Self::Pdf | Self::Markdown => SourceKind::File, - Self::Notion | Self::Github | Self::Web | Self::Other(_) => SourceKind::Import, + Self::Notion => SourceKind::Composio, + Self::Github => SourceKind::Github, + Self::Web => SourceKind::Link, + Self::Other(_) => SourceKind::Import, } } } diff --git a/crates/tinymemory-tools/tests/fixtures/tool_contracts.json b/crates/tinymemory-tools/tests/fixtures/tool_contracts.json index 2bc487c7..fab2afbb 100644 --- a/crates/tinymemory-tools/tests/fixtures/tool_contracts.json +++ b/crates/tinymemory-tools/tests/fixtures/tool_contracts.json @@ -53,6 +53,10 @@ "enum": [ "folder", "file", + "link", + "github", + "rss", + "composio", "conversation", "agent", "import" @@ -184,6 +188,10 @@ "enum": [ "folder", "file", + "link", + "github", + "rss", + "composio", "conversation", "agent", "import" @@ -321,6 +329,10 @@ "enum": [ "folder", "file", + "link", + "github", + "rss", + "composio", "conversation", "agent", "import" @@ -457,6 +469,10 @@ "enum": [ "folder", "file", + "link", + "github", + "rss", + "composio", "conversation", "agent", "import" @@ -666,6 +682,10 @@ "enum": [ "folder", "file", + "link", + "github", + "rss", + "composio", "conversation", "agent", "import" diff --git a/docs/architecture/api-items.md b/docs/architecture/api-items.md index 89d183c2..2909c584 100644 --- a/docs/architecture/api-items.md +++ b/docs/architecture/api-items.md @@ -165,7 +165,8 @@ namespace are omitted on serialisation. | `observed_actor` | `Option` | `{ id, name? }`: who said or did it when not the memory's owner, such as an email's sender or a channel's (`id` is `type:id`, `user:priya@acme.com`, `user:+15551234567`). Excluded from the fingerprint; the CortexDB engine sends it only with attribution on. | `MemoryMeta::from_source(kind, id)` sets only the source. `SourceKind` is -`folder | file | conversation | agent | import` (`SourceKind::ALL`, `as_str`); `agent` is the default. +`folder | file | link | github | rss | composio | conversation | agent | +import` (`SourceKind::ALL`, `as_str`); `agent` is the default. ## `MetaFilter` @@ -199,7 +200,7 @@ equals the default, so a filter holding only a `reach` is **not** empty), { "reach": { "at": "team:acme/agent:writer", "inherit": true, "descendants": false }, "kinds": ["document", "learning"], - "sources": ["folder", "file"], + "sources": ["folder", "github"], "folder": "/notes/rust", "tags_any": ["style", "review"], "observed_after": "2026-01-01T00:00:00Z", diff --git a/docs/architecture/integrations-sources.md b/docs/architecture/integrations-sources.md index f58cc802..98be38d1 100644 --- a/docs/architecture/integrations-sources.md +++ b/docs/architecture/integrations-sources.md @@ -1,6 +1,7 @@ # Integrations: sources -The `sources` module of `tinymemory-integrations` (feature `sources`) turns a configured source into `StoreItem`s. Overview of the +The `sources` module of `tinymemory-integrations` (features `sources` and +`sources-network`) turns a configured source into `StoreItem`s. Overview of the crate: [integrations.md](integrations.md). Module README: [`src/sources/README.md`](../../crates/tinymemory-integrations/src/sources/README.md). @@ -18,6 +19,10 @@ optional fields are required; `validate()` checks them. | `folder` | `path` | `glob` | `Folder` | | `file` | `path` | | `File` | | `conversation` | | | `Conversation` | +| `web_page` | `url` | `selector` | `Link` | +| `github_repo` | `url` | `branch`, `paths`, `max_commits`, `max_issues`, `max_prs` (default 1000 each) | `Github` | +| `rss_feed` | `url` | `max_items` (default 50) | `Rss` | +| `composio` | `toolkit`, `connection_id` | | `Composio` | Every entry also needs a non-blank `id` (no `:` or control characters) and a non-empty `label`, and carries `enabled` and optional sync-budget fields @@ -33,14 +38,21 @@ workspace, converter)`. The default `read_store_item` reads the content and maps it through `items::content_item`; local readers override it to work from raw bytes so a bound converter can handle PDF or DOCX. -`readers::reader_for(kind)` returns the folder, file or conversation reader; every -reader is local. +`readers::reader_for(kind)` returns only the local readers (folder, file, +conversation), which are safe to drive on a timer. The network kinds return +`None`, meaning "route through the host's sync runner". With +`sources-network`, `reader_for_request(kind)` returns a reader for every kind, +for a host servicing an explicit user request, never a polling loop. | Reader | Items | Notes | | --- | --- | --- | | `FolderReader` | one per selected file; id is the folder-relative slash path | With a `glob`, exactly the matches (compiled to a regex over the relative path). Without one, markdown, plain text and source code (`is_default_candidate`). Skips hidden files and directories and `.git`, `.hg`, `.svn`, `target`, `node_modules`, `__pycache__`, `venv`; never follows symlinks; refuses files over `FOLDER_FILE_SIZE_CAP_BYTES` (10 MiB). A relative `path` is anchored on the workspace. | | `FileReader` | exactly one, id is the file name | Same size cap. `FileReader::read_path` reads a path with no configured source. | | `ConversationReader` | one per `/threads/.json` | Threads are `{title, messages: [{role, content, created_at?}]}`. Messages with blank text or an unknown role are dropped. Timestamps: RFC 3339, or epoch seconds or milliseconds. The id may not contain separators or `..`. | +| `WebPageReader` | one: the page URL | With a CSS `selector`, only the text of matching elements (plain text; only the last compound of a descendant chain is honoured). Otherwise the whole page as markdown. 10 MiB body cap. | +| `RssReader` | one per feed entry (RSS or Atom), up to `max_items` | The parsed feed is cached for 60 seconds so a list-then-read pass downloads it once. 5 MiB cap; non-UTF-8 bodies are refused. | +| `GithubReader` | `commit:`, `issue:`, `pr:` | See below. | +| `ComposioReader` | one: the connection | A placeholder; see Composio. | ### local_file and ensure_within_base @@ -53,6 +65,24 @@ base, or `Error::Io` when either cannot be canonicalised. The folder reader applies it on every read; the conversation reader applies it within the threads directory. +### GitHub transports + +The reader pulls **project activity**, not source code, from +`https://github.com//` (extra path segments such as `/tree/main` +are rejected). It combines three transports: + +- **Commits:** a bare clone under `/git_cache//.git`, + fetched with an explicit refspec and listed with `git log`, honouring + `branch` and `paths`. `git` must be on `PATH`. If the clone fails, it falls + back to the commits API. +- **Issues and pull requests:** `gh api` when the `gh` CLI is available + (authenticated, higher rate limit; probed once per process), otherwise the + unauthenticated REST API at `api.github.com`. The list pass caches full rows + so reads do not refetch. +- Limits: `max_commits`, `max_issues`, `max_prs` per sync, default 1000 each. + Listing fails only when every call failed; otherwise the partial list is + returned. Failures surface as `Error::Reader`. + ## Items mapping `items` maps reader output to `StoreItem`s. Every item's `meta.source` is @@ -63,6 +93,10 @@ threads directory. | Kind | Item | Metadata filled | | --- | --- | --- | | folder, file | document | `workspace`, `folder`, `file_path` (canonical), `language`, `observed_at` (mtime), `mime` | +| github | document | `repo` (`owner/name`), `commit` (commits), `url` (issues and PRs), `observed_at` | +| web page | document | `url` | +| rss | document | `url` (entry link), `observed_at` (published) | +| composio | document | `tags = [toolkit]`; payloads add `url`, `observed_at`, `thread_id`, `repo` | | conversation | conversation | `workspace`, `thread_id`, `turns` (`0..=n-1`), `observed_at` (last turn, else mtime) | `collect_items(reader, entry, workspace, converter)` lists and reads every @@ -70,3 +104,85 @@ item, returning `Collected { items, skipped }`. One bad item lands in `skipped` with its error and the pass continues; only a listing failure is an `Err`. Other entry points: `file_item` (a path with no source), `conversation_item`, `content_item` and `items::local_file_item`. + +## Composio + +Composio data does not arrive item by item. A host runs toolkit actions with +its own credentials and hands the raw responses to `sources::composio`, which +holds no credential, opens no socket and decides nothing about when to sync. + +1. **Normalisers**, pure `serde_json::Value` transforms, one module per + toolkit. They walk Composio's envelope variants (top level, under `data`, + under `data.data`) and return the first array found: `clickup` + (`extract_tasks`), `github` (`extract_issues`), `linear` + (`extract_issues`), `notion` (`extract_results`, `extract_page_markdown`), + each with title, id and updated-time helpers. `fields::pick_str` is the + shared lookup: it tries dotted paths, descends only through objects, and + rejects non-string leaves. +2. **Post-processors the host must call** for two toolkits, because their raw + responses are too verbose: + - `gmail_post_process::post_process(slug, arguments, &mut data)` rewrites a + `GMAIL_FETCH_EMAILS` response into slim `messages[]` (other Gmail slugs + pass through; `raw_html: true` in the arguments skips the reshape). If the + response carries a response-level `markdownFormatted` string, call + `apply_response_level_markdown(&mut data, markdown)` **before** + `post_process`; it is a no-op unless the split count matches the message + count. `format_email_local_time` renders in the host's local timezone; + the raw UTC fields are preserved. + - `slack_post_process::post_process(slug, arguments, &mut data)` reshapes + `SLACK_FETCH_CONVERSATION_HISTORY`, `SLACK_LIST_CONVERSATIONS` and + `SLACK_SEARCH_MESSAGES`; unknown slugs are no-ops. `channel_id` for history + is injected by the host (it is in the request, not the response), and user + ids are resolved by the host. +3. `normalise_payload(toolkit, &data)` returns `ComposioDocument`s (id, title, + markdown body, url, `observed_at`, `thread_id`, `repo`), dispatching on the + case-insensitive toolkit slug: `gmail`, `slack`, `github`, `linear`, + `notion`, `clickup`; any other toolkit falls back to each record as fenced + JSON, so a new toolkit is ingested verbosely rather than dropped. Records + with no text are skipped. `payload_items(toolkit, source_id, &data)` wraps + them as `StoreItem::Document` with `source.kind = Composio`, + `source.id = source_id` and `tags = [toolkit]`. + +`readers::composio::ComposioReader` is only a placeholder so +`reader_for_request` can serve every kind: `list_items` returns the connection +as one sync target. + +## Fetching and the SSRF guard (sources-network) + +`fetch::fetch_url(url)` fetches one URL into a `RawDocument`: the `Content-Type` +becomes the declared MIME, the URL the origin, and the last path segment (if +it has an extension) the filename. `fetch::link_item(url, source_id, +converter)` converts it into a document with `source.kind = Link` and `url` +set. The cap is `MAX_DOCUMENT_BYTES` (32 MiB), applied while streaming. No +retries, no robots.txt, no scheduling. Errors: `Invalid` (malformed or refused +URL, empty body), `Unreachable` (never completed, interrupted read, may +succeed later), `Upstream` (non-success status), `TooLarge`. + +The web-page and RSS readers fetch through the same path with tighter caps +(10 MiB and 5 MiB). `fetch::ssrf` is public so a host fetching a user-supplied +URL by other means applies the same policy. + +Guard rules: + +- **Scheme:** `http` and `https` only. +- **Host text:** refused are empty hosts, `localhost`, `.local` and + `.internal` names, single-label names (internal service names such as + `redis`), and IP literals that are not globally routable. +- **One address classifier** serves literals and resolved addresses. Not + fetchable: loopback, private, link-local (including `169.254.169.254`), + unspecified, CGNAT (`100.64.0.0/10`), `192.0.0.0/16`, multicast, broadcast, + documentation and benchmarking ranges, reserved `240.0.0.0/4`, IPv6 + unique-local and link-local; IPv4-mapped and IPv4-compatible IPv6 addresses + are judged by their IPv4 part. +- **IPv6 literals fail closed.** A URL such as `http://[2606:4700::1111]/` is + refused whatever the address: the host text keeps its brackets, does not parse + as an IP, and has no dot, so the single-label rule blocks it. A hostname with + an AAAA record is still vetted when resolved. +- **Resolver:** `PublicOnlyResolver` keeps only public addresses and fails the + request when none remain, pinning the connection to a vetted address with no + re-resolution between check and connect. +- **Redirects** are re-checked per hop; a refused hop is not followed, and the + read fails on the resulting non-success status. +- **Client:** 20-second timeout and the user agent `openhuman`. +- **Body caps** are enforced while streaming (`read_body_capped`), with an early + rejection on a truthful `Content-Length`. diff --git a/docs/architecture/integrations.md b/docs/architecture/integrations.md index 9c002d62..2c888754 100644 --- a/docs/architecture/integrations.md +++ b/docs/architecture/integrations.md @@ -11,7 +11,7 @@ CortexDB engine and the registry are in [cortex.md](cortex.md). | --- | --- | --- | | `cortex`, `registry`, `config` | `cortex` (default) | [cortex.md](cortex.md) | | `documents` | `documents`, `documents-office` | [Documents](#documents) | -| `sources` | `sources` | [integrations-sources.md](integrations-sources.md) | +| `sources` | `sources`, `sources-network` | [integrations-sources.md](integrations-sources.md) | | `safety` | `safety` | [Safety](#safety) | | `import` | `legacy-import` | [Legacy v1 import](#legacy-v1-import) | diff --git a/docs/architecture/overview.md b/docs/architecture/overview.md index 53711d88..94e6916e 100644 --- a/docs/architecture/overview.md +++ b/docs/architecture/overview.md @@ -55,10 +55,11 @@ alone, so the contract is the only coupling point. | `tinymemory-integrations` | `cortex` (default) | `cortex::CortexEngine` (both wires), `registry` (`list_engines`, `build_engine`), `config` (`MemoryConfig`) | | | `documents` | Format sniffing and conversion to markdown | | | `documents-office` | PDF, DOCX, PPTX, XLSX conversion (implies `documents`) | -| | `sources` | Readers for folders, files, conversations (implies `documents`) | +| | `sources` | Readers for folders, files, conversations; Composio normalisers (implies `documents`) | +| | `sources-network` | GitHub, RSS and web-page readers behind the SSRF guard (implies `sources`) | | | `safety` | Secret and PII scrubbing of a `StoreItem` | | | `legacy-import` | Reading a v1 workspace; `import::migrate` and `migrate_with` | -| | `full` | `cortex`, `documents-office`, `sources`, `safety`, `legacy-import` | +| | `full` | `cortex`, `documents-office`, `sources-network`, `safety`, `legacy-import` | ## Write path @@ -71,7 +72,7 @@ source reader ──▶ documents ──▶ safety ──▶ engine.store ── ``` 1. **Source reader** (`sources`): lists a configured source (folder, file, - conversation) and reads each entry. + link, GitHub, RSS, Composio payload, conversation) and reads each entry. `sources::collect_items` does this for one source; one bad entry lands in `Collected::skipped` instead of aborting the pass. 2. **Documents conversion** (`documents`): sniffs the format and converts the diff --git a/docs/architecture/testing.md b/docs/architecture/testing.md index 97e50a2d..07340215 100644 --- a/docs/architecture/testing.md +++ b/docs/architecture/testing.md @@ -78,7 +78,7 @@ works on its own, not only in the union: | | `--no-default-features --features cortex` | | | `--no-default-features --features documents` | | | `--no-default-features --features documents-office` | -| | `--no-default-features --features sources` | +| | `--no-default-features --features sources-network` (proves it implies `sources`) | | | `--no-default-features --features safety` | | | `--no-default-features --features legacy-import` | | | `--no-default-features --features full` | @@ -255,7 +255,7 @@ idempotency pairs, and every recall, answer and forget body for assertions. | --- | --- | --- | | `feature_surface.rs` | `full` | the modules compose: scrub an item, store it in the reference engine, run the conformance suite, compile a context | | `documents_office.rs` | `documents-office` | `OfficeConverter` prepends to the default `ConverterChain` and supports PDF, DOCX, XLSX, PPTX | -| `reader_dispatch.rs` | `sources` | every kind has a local reader | +| `reader_dispatch.rs` | `sources` (`sources-network` for one test) | local readers are constructed for timers, network readers only for requests | | `legacy_import.rs` | `legacy-import` | importing v1 workspaces, every mapping, ordering and resumption | | `live_cortexdb.rs`, `office_live.rs` | `cortex` (+ `documents-office`) | a real CortexDB; skipped unless configured | diff --git a/docs/specs/memory-v2.md b/docs/specs/memory-v2.md index e8fbf5c5..43f7255e 100644 --- a/docs/specs/memory-v2.md +++ b/docs/specs/memory-v2.md @@ -30,7 +30,7 @@ Three crates, one directory each under `crates/`. Their relationships are in | --- | --- | | `tinymemory-api` | The contract: `MemoryEngine`, request/response types, `MemoryMeta`, `MetaFilter`, namespaces, `EngineDescriptor`, `Error`. No I/O. Feature `conformance`: the behavioural suite every engine must pass, plus a reference in-memory engine. | | `tinymemory-tools` | The agent tool spec `MemoryTools` (seven tools, host-fixed namespace and reach) and `context`, the `ContextCompiler` that builds `context.md` from an engine. | -| `tinymemory-integrations` | Everything that talks to the outside world, each behind a feature: `cortex` (default; the CortexDB engine, registered twice as `cortexdb` and `tinyhumans`, plus the registry `list_engines`/`build_engine` and `MemoryConfig`), `documents` and `documents-office` (format sniffing and conversion to markdown, emitting `StoreItem::Document`; PDF/DOCX/PPTX/XLSX via `OfficeConverter` or a host `DocumentConverter`), `sources` (readers for folder, file and conversations), `safety` (secret/PII scrubbing applied to every item before `store`), and `legacy-import` (reads a v1 TinyCortex workspace and migrates it into any engine). | +| `tinymemory-integrations` | Everything that talks to the outside world, each behind a feature: `cortex` (default; the CortexDB engine, registered twice as `cortexdb` and `tinyhumans`, plus the registry `list_engines`/`build_engine` and `MemoryConfig`), `documents` and `documents-office` (format sniffing and conversion to markdown, emitting `StoreItem::Document`; PDF/DOCX/PPTX/XLSX via `OfficeConverter` or a host `DocumentConverter`), `sources` and `sources-network` (readers for folder, file, link, GitHub, RSS, Composio payloads and conversations, with the SSRF guard), `safety` (secret/PII scrubbing applied to every item before `store`), and `legacy-import` (reads a v1 TinyCortex workspace and migrates it into any engine). | Deleted: the earlier `tinymemory` facade, `tinymemory-cortex`, `tinymemory-documents`, `tinymemory-sources`, `tinymemory-safety`, @@ -91,7 +91,7 @@ pub struct MemoryMeta { pub tags: Vec, pub observed_at: Option>, } -pub enum SourceKind { Folder, File, Conversation, Agent, Import } +pub enum SourceKind { Folder, File, Link, Github, Rss, Composio, Conversation, Agent, Import } ``` `MetaFilter` has the same optional fields (each an exact match, `folder` and From d5548116e30f31f6a4ef8aeaa61560d0892cfe9c Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Thu, 8 Oct 2026 23:42:07 +0530 Subject: [PATCH 16/19] feat(composio): add per-toolkit source modules for composio integrations Split the composio source handling into dedicated modules for each toolkit such as clickup, documents, fields, github, gmail, linear, notion and slack, each with its own tests. This makes the per-toolkit post-processing and field handling easier to extend and keeps the shared reader logic in one place. Auto-committed-on: macbook Co-authored-by: Medulla --- crates/tinymemory-api/src/meta/mod.rs | 10 +- .../src/sources/composio/clickup/mod.rs | 69 --- .../src/sources/composio/clickup/mod_tests.rs | 58 -- .../src/sources/composio/documents/mod.rs | 343 ------------ .../sources/composio/documents/mod_tests.rs | 197 ------- .../src/sources/composio/fields/mod.rs | 44 -- .../src/sources/composio/fields/mod_tests.rs | 37 -- .../src/sources/composio/github/mod.rs | 116 ---- .../src/sources/composio/github/mod_tests.rs | 97 ---- .../composio/gmail_post_process/mod.rs | 498 ------------------ .../composio/gmail_post_process/mod_tests.rs | 356 ------------- .../src/sources/composio/linear/mod.rs | 79 --- .../src/sources/composio/linear/mod_tests.rs | 93 ---- .../src/sources/composio/mod.rs | 35 -- .../src/sources/composio/notion/mod.rs | 88 ---- .../src/sources/composio/notion/mod_tests.rs | 111 ---- .../composio/slack_post_process/mod.rs | 324 ------------ .../composio/slack_post_process/mod_tests.rs | 306 ----------- .../src/sources/readers/composio/mod.rs | 78 --- .../src/sources/readers/composio/mod_tests.rs | 51 -- .../src/sources/types/mod.rs | 25 +- crates/tinymemory-tools/src/layout/source.rs | 3 +- 22 files changed, 8 insertions(+), 3010 deletions(-) delete mode 100644 crates/tinymemory-integrations/src/sources/composio/clickup/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/clickup/mod_tests.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/documents/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/documents/mod_tests.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/fields/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/fields/mod_tests.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/github/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/github/mod_tests.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod_tests.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/linear/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/linear/mod_tests.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/notion/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/notion/mod_tests.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod_tests.rs delete mode 100644 crates/tinymemory-integrations/src/sources/readers/composio/mod.rs delete mode 100644 crates/tinymemory-integrations/src/sources/readers/composio/mod_tests.rs diff --git a/crates/tinymemory-api/src/meta/mod.rs b/crates/tinymemory-api/src/meta/mod.rs index c7736efb..ae32e35f 100644 --- a/crates/tinymemory-api/src/meta/mod.rs +++ b/crates/tinymemory-api/src/meta/mod.rs @@ -156,26 +156,27 @@ pub enum SourceKind { Github, /// An RSS or Atom feed. Rss, - /// A Composio toolkit payload (Gmail, Slack, Notion, ...). - Composio, /// A host conversation. Conversation, /// Written by an agent directly; the default. #[default] Agent, /// Imported from a legacy store. + /// + /// The retired `composio` kind decodes to this variant, so items stored + /// under it stay readable. + #[serde(alias = "composio")] Import, } impl SourceKind { /// Every kind, in declaration order. - pub const ALL: [Self; 9] = [ + pub const ALL: [Self; 8] = [ Self::Folder, Self::File, Self::Link, Self::Github, Self::Rss, - Self::Composio, Self::Conversation, Self::Agent, Self::Import, @@ -190,7 +191,6 @@ impl SourceKind { Self::Link => "link", Self::Github => "github", Self::Rss => "rss", - Self::Composio => "composio", Self::Conversation => "conversation", Self::Agent => "agent", Self::Import => "import", diff --git a/crates/tinymemory-integrations/src/sources/composio/clickup/mod.rs b/crates/tinymemory-integrations/src/sources/composio/clickup/mod.rs deleted file mode 100644 index b4ce64e5..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/clickup/mod.rs +++ /dev/null @@ -1,69 +0,0 @@ -//! ClickUp host normalization helpers — result extraction and task-title and -//! timestamp extraction. -//! -//! ClickUp's REST API (and therefore Composio's wrapping of it) returns -//! task lists in a small handful of shapes depending on which endpoint -//! is called. The functions here walk the union of common shapes so the -//! provider doesn't have to branch per Composio envelope variant. - -use serde_json::Value; - -use super::fields::pick_str; - -/// Walk the Composio response envelope for ClickUp task list results. -/// -/// ClickUp's "filtered team tasks" endpoint returns `{ "tasks": [...] }` -/// at the top level; Composio re-wraps the upstream payload under -/// `data` or `data.data` depending on the action. We probe each shape -/// in order and return the first array we find. -pub fn extract_tasks(data: &Value) -> Vec { - let candidates = [ - data.pointer("/data/tasks"), - data.pointer("/tasks"), - data.pointer("/data/data/tasks"), - data.pointer("/data/results"), - data.pointer("/results"), - data.pointer("/data/items"), - data.pointer("/items"), - ]; - for cand in candidates.into_iter().flatten() { - if let Some(arr) = cand.as_array() { - return arr.clone(); - } - } - Vec::new() -} - -/// Extract a human-readable title from a ClickUp task object. -/// -/// ClickUp tasks store the name at `name` (or `data.name` after Composio -/// envelope wrapping). When the name is missing we fall back to the -/// task ID so chunks remain identifiable. -pub fn extract_task_name(task: &Value) -> Option { - pick_str(task, &["name", "data.name", "title", "data.title"]) -} - -/// Extract a stable cursor timestamp (milliseconds since epoch as a -/// string) from a ClickUp task object. -/// -/// The ClickUp API returns `date_updated` as a stringified epoch ms -/// (e.g. `"1733412345678"`); we keep it as a string so lexicographic -/// comparison against the stored cursor remains valid as long as the -/// length doesn't change (it won't until year 33658). -pub fn extract_task_updated(task: &Value) -> Option { - pick_str( - task, - &[ - "date_updated", - "data.date_updated", - "updated_at", - "data.updated_at", - "dateUpdated", - "data.dateUpdated", - ], - ) -} - -#[cfg(test)] -#[path = "mod_tests.rs"] -mod tests; diff --git a/crates/tinymemory-integrations/src/sources/composio/clickup/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/clickup/mod_tests.rs deleted file mode 100644 index 5a520401..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/clickup/mod_tests.rs +++ /dev/null @@ -1,58 +0,0 @@ -//! Tests for the ClickUp normaliser. - -use super::*; -use serde_json::json; - -#[test] -fn extract_tasks_from_data_tasks() { - let data = json!({ "data": { "tasks": [{"id": "t1"}] } }); - assert_eq!(extract_tasks(&data).len(), 1); -} - -#[test] -fn extract_tasks_from_top_level_tasks() { - let data = json!({ "tasks": [{"id": "a"}, {"id": "b"}] }); - assert_eq!(extract_tasks(&data).len(), 2); -} - -#[test] -fn extract_tasks_empty_when_missing() { - let data = json!({ "foo": "bar" }); - assert!(extract_tasks(&data).is_empty()); -} - -#[test] -fn extract_task_name_from_top_level() { - let task = json!({ "id": "t1", "name": "Build feature X" }); - assert_eq!(extract_task_name(&task), Some("Build feature X".into())); -} - -#[test] -fn extract_task_name_falls_back_to_data_name() { - let task = json!({ "data": { "name": "Wrapped" } }); - assert_eq!(extract_task_name(&task), Some("Wrapped".into())); -} - -#[test] -fn extract_task_name_none_when_missing() { - let task = json!({ "id": "t1" }); - assert!(extract_task_name(&task).is_none()); -} - -#[test] -fn extract_task_updated_handles_string_form() { - let task = json!({ "date_updated": "1733412345678" }); - assert_eq!( - extract_task_updated(&task), - Some("1733412345678".to_string()) - ); -} - -#[test] -fn extract_task_updated_handles_nested_data() { - let task = json!({ "data": { "dateUpdated": "1700000000000" } }); - assert_eq!( - extract_task_updated(&task), - Some("1700000000000".to_string()) - ); -} diff --git a/crates/tinymemory-integrations/src/sources/composio/documents/mod.rs b/crates/tinymemory-integrations/src/sources/composio/documents/mod.rs deleted file mode 100644 index 04753e4b..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/documents/mod.rs +++ /dev/null @@ -1,343 +0,0 @@ -//! One Composio response in, documents and `StoreItem`s out. -//! -//! [`normalise_payload`] dispatches on the toolkit slug to the matching -//! normaliser and reads each record's title, body, link and timestamp. -//! Toolkits without a dedicated normaliser fall back to a generic walk that -//! keeps each record as fenced JSON, so a new toolkit is ingested (verbosely) -//! rather than dropped. [`payload_items`] wraps the documents as -//! `StoreItem::Document`s. - -use crate::documents::{DocumentFormat, markdown_from_text}; -use chrono::{DateTime, TimeZone, Utc}; -use serde_json::Value; -use tinymemory_api::{DocumentBody, MemoryMeta, SourceKind, StoreItem}; - -use super::fields::pick_str; -use super::{clickup, github, gmail_post_process, linear, notion}; - -/// One record of a Composio payload, normalised: an email, a message, an -/// issue, a task or a page. -#[derive(Debug, Clone, PartialEq)] -pub struct ComposioDocument { - /// The provider's id for the record, when it has one. - pub id: Option, - /// A human-readable title. - pub title: Option, - /// The record's text as markdown. Never empty. - pub body: String, - /// A link back to the record in the provider's UI. - pub url: Option, - /// When the record was last updated or sent. - pub observed_at: Option>, - /// The email or message thread the record belongs to. - pub thread_id: Option, - /// The repository a GitHub record belongs to, as `owner/name`. - pub repo: Option, -} - -impl ComposioDocument { - /// A record with only a body. - fn with_body(body: String) -> Self { - Self { - id: None, - title: None, - body, - url: None, - observed_at: None, - thread_id: None, - repo: None, - } - } - - /// Wrap this record as a [`StoreItem::Document`] from `toolkit`, read - /// through the connection or source `source_id`. - #[must_use] - pub fn into_store_item(self, toolkit: &str, source_id: &str) -> StoreItem { - let mut meta = MemoryMeta::from_source(SourceKind::Composio, Some(source_id.to_string())); - meta.url = self.url; - meta.observed_at = self.observed_at; - meta.thread_id = self.thread_id; - meta.repo = self.repo; - meta.tags = vec![toolkit.to_string()]; - StoreItem::Document { - title: self.title, - body: DocumentBody::Text(self.body), - mime: Some(DocumentFormat::Markdown.mime().to_string()), - meta, - } - } -} - -/// Normalise one Composio response from `toolkit` into documents. -/// -/// `data` is the action's response; for Gmail and Slack, run -/// [`gmail_post_process::post_process`] / [`super::slack_post_process::post_process`] -/// on it first so it carries the slim `messages[]` shape. Records with no text -/// at all are skipped. -#[must_use] -pub fn normalise_payload(toolkit: &str, data: &Value) -> Vec { - let documents: Vec = match toolkit.to_ascii_lowercase().as_str() { - "gmail" => array_at(data, &["/messages", "/data/messages"]) - .iter() - .filter_map(gmail_message) - .collect(), - "slack" => array_at(data, &["/messages", "/data/messages"]) - .iter() - .filter_map(slack_message) - .collect(), - "github" => github::extract_issues(data) - .iter() - .filter_map(github_issue) - .collect(), - "linear" => linear::extract_issues(data) - .iter() - .filter_map(linear_issue) - .collect(), - "notion" => notion_pages(data), - "clickup" => clickup::extract_tasks(data) - .iter() - .filter_map(clickup_task) - .collect(), - _ => generic_records(data), - }; - log::debug!( - "[memory_sources:composio] normalised toolkit={toolkit} documents={}", - documents.len() - ); - documents -} - -/// Normalise one Composio response and wrap every record as a -/// [`StoreItem::Document`] with `source.kind = Composio`, -/// `source.id = source_id` and `tags = [toolkit]`. -#[must_use] -pub fn payload_items(toolkit: &str, source_id: &str, data: &Value) -> Vec { - normalise_payload(toolkit, data) - .into_iter() - .map(|document| document.into_store_item(toolkit, source_id)) - .collect() -} - -/// The first array found at any of `pointers`. -fn array_at<'a>(data: &'a Value, pointers: &[&str]) -> &'a [Value] { - pointers - .iter() - .find_map(|pointer| data.pointer(pointer).and_then(Value::as_array)) - .map_or(&[], Vec::as_slice) -} - -/// A body as markdown: HTML (as sniffed) is converted, anything else kept. -fn to_markdown(text: &str) -> String { - let format = DocumentFormat::sniff(text.as_bytes(), None, None); - markdown_from_text(text, format).trim().to_string() -} - -/// Build a document from its parts, falling back to the title as the body; -/// `None` when there is no text at all. -fn document(title: Option, body: Option) -> Option { - let body = body - .map(|body| to_markdown(&body)) - .filter(|body| !body.is_empty()) - .or_else(|| title.clone())?; - let mut document = ComposioDocument::with_body(body); - document.title = title; - Some(document) -} - -/// Parse an ISO 8601 / RFC 3339 / RFC 2822 timestamp. -fn parse_time(text: &str) -> Option> { - gmail_post_process::parse_email_date(text) -} - -/// Parse an epoch-milliseconds string (ClickUp's `date_updated`). -fn parse_epoch_ms(text: &str) -> Option> { - let millis = text.trim().parse::().ok()?; - Utc.timestamp_millis_opt(millis).single() -} - -/// Parse a Slack `ts` (`"1712345678.123456"`, epoch seconds with a fraction). -fn parse_slack_ts(text: &str) -> Option> { - let (seconds, fraction) = text.split_once('.').unwrap_or((text, "0")); - let seconds = seconds.parse::().ok()?; - let micros = format!("{fraction:0<6}").get(..6)?.parse::().ok()?; - Utc.timestamp_opt(seconds, micros * 1_000).single() -} - -/// A string field, or a number rendered as a string. -fn scalar(value: &Value, key: &str) -> Option { - match value.get(key)? { - Value::String(text) if !text.trim().is_empty() => Some(text.trim().to_string()), - Value::Number(number) => Some(number.to_string()), - _ => None, - } -} - -/// A post-processed Gmail message: headers above the body. -fn gmail_message(message: &Value) -> Option { - let subject = pick_str(message, &["subject"]); - let markdown = pick_str(message, &["markdown", "messageText"]); - let mut header = String::new(); - for (label, key) in [("From", "from"), ("To", "to"), ("Date", "date")] { - if let Some(value) = pick_str(message, &[key]) { - header.push_str(&format!("{label}: {value}\n")); - } - } - let body = match (markdown, header.is_empty()) { - (Some(markdown), false) => Some(format!("{header}\n{markdown}")), - (Some(markdown), true) => Some(markdown), - (None, _) => None, - }; - let mut document = document(subject, body)?; - document.id = pick_str(message, &["id", "messageId"]); - document.thread_id = pick_str(message, &["threadId", "thread_id"]); - document.observed_at = pick_str(message, &["date"]).as_deref().and_then(parse_time); - Some(document) -} - -/// A post-processed Slack message. -fn slack_message(message: &Value) -> Option { - let user = pick_str(message, &["user"]); - let channel = pick_str(message, &["channel_id"]); - let title = match (&user, &channel) { - (Some(user), Some(channel)) => Some(format!("Slack message from {user} in {channel}")), - (Some(user), None) => Some(format!("Slack message from {user}")), - (None, _) => Some("Slack message".to_string()), - }; - let text = pick_str(message, &["text"])?; - let mut document = document(title, Some(text))?; - let ts = pick_str(message, &["ts"]); - document.id = ts.clone(); - document.url = pick_str(message, &["permalink"]); - document.thread_id = pick_str(message, &["thread_ts"]); - document.observed_at = ts.as_deref().and_then(parse_slack_ts); - Some(document) -} - -/// A GitHub issue or pull request from a search response. -fn github_issue(issue: &Value) -> Option { - let mut document = document( - github::extract_issue_title(issue), - pick_str(issue, &["body", "data.body"]), - )?; - document.id = github::extract_issue_id(issue); - document.url = pick_str(issue, &["html_url", "data.html_url"]); - document.repo = document.url.as_deref().and_then(github_repo); - document.observed_at = github::extract_issue_updated_at(issue) - .as_deref() - .and_then(parse_time); - Some(document) -} - -/// `owner/name` from a `https://github.com/owner/name/...` link. -fn github_repo(url: &str) -> Option { - let rest = url.split_once("github.com/")?.1; - let mut parts = rest.split('/'); - let owner = parts.next().filter(|part| !part.is_empty())?; - let name = parts.next().filter(|part| !part.is_empty())?; - Some(format!("{owner}/{name}")) -} - -/// A Linear issue. -fn linear_issue(issue: &Value) -> Option { - let mut document = document( - linear::extract_issue_title(issue), - pick_str(issue, &["description", "data.description"]), - )?; - document.id = pick_str(issue, &["identifier", "id", "data.identifier", "data.id"]); - document.url = pick_str(issue, &["url", "data.url"]); - document.observed_at = linear::extract_issue_updated(issue) - .as_deref() - .and_then(parse_time); - Some(document) -} - -/// Notion pages from a search response, or the one page a -/// `NOTION_GET_PAGE_MARKDOWN` response carries. -fn notion_pages(data: &Value) -> Vec { - let results = notion::extract_results(data); - if results.is_empty() { - return notion::extract_page_markdown(data) - .and_then(|markdown| { - let title = notion::extract_page_title(data); - let mut document = document(title, Some(markdown))?; - document.id = pick_str(data, &["id", "data.id", "page_id", "data.page_id"]); - document.url = pick_str(data, &["url", "data.url"]); - Some(document) - }) - .into_iter() - .collect(); - } - results - .iter() - .filter_map(|page| { - let mut document = document( - notion::extract_page_title(page), - notion::extract_page_markdown(page), - )?; - document.id = pick_str(page, &["id", "data.id"]); - document.url = pick_str(page, &["url", "data.url"]); - document.observed_at = pick_str(page, &["last_edited_time", "data.last_edited_time"]) - .as_deref() - .and_then(parse_time); - Some(document) - }) - .collect() -} - -/// A ClickUp task. -fn clickup_task(task: &Value) -> Option { - let mut document = document( - clickup::extract_task_name(task), - pick_str( - task, - &["markdown_description", "description", "text_content"], - ), - )?; - document.id = scalar(task, "id"); - document.url = pick_str(task, &["url", "data.url"]); - document.observed_at = clickup::extract_task_updated(task) - .as_deref() - .and_then(|text| parse_epoch_ms(text).or_else(|| parse_time(text))); - Some(document) -} - -/// Any other toolkit: each record in the first list found (or the whole -/// payload) as fenced JSON, titled and linked when it says how. -fn generic_records(data: &Value) -> Vec { - let records = array_at( - data, - &[ - "/data/items", - "/items", - "/data/results", - "/results", - "/data/data", - "/data", - ], - ); - let records: Vec<&Value> = if records.is_empty() { - vec![data] - } else { - records.iter().collect() - }; - records - .into_iter() - .filter(|record| !record.is_null()) - .filter_map(|record| { - let json = serde_json::to_string_pretty(record).ok()?; - let title = pick_str(record, &["title", "name", "subject"]); - let mut document = ComposioDocument::with_body(format!("```json\n{json}\n```")); - document.title = title; - document.id = scalar(record, "id"); - document.url = pick_str(record, &["url", "html_url", "permalink", "link"]); - document.observed_at = pick_str(record, &["updated_at", "updatedAt", "created_at"]) - .as_deref() - .and_then(parse_time); - Some(document) - }) - .collect() -} - -#[cfg(test)] -#[path = "mod_tests.rs"] -mod tests; diff --git a/crates/tinymemory-integrations/src/sources/composio/documents/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/documents/mod_tests.rs deleted file mode 100644 index 4d689128..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/documents/mod_tests.rs +++ /dev/null @@ -1,197 +0,0 @@ -//! Tests for turning Composio payloads into documents and `StoreItem`s. - -use super::*; -use serde_json::json; -use tinymemory_api::ItemKind; - -fn text_of(item: &StoreItem) -> (&Option, &str, &MemoryMeta) { - match item { - StoreItem::Document { - title, - body: DocumentBody::Text(text), - meta, - .. - } => (title, text.as_str(), meta), - other => panic!("expected a text document, got {other:?}"), - } -} - -#[test] -fn a_github_search_becomes_items_with_url_repo_time_and_toolkit_tag() { - let data = json!({ - "data": { "items": [{ - "id": 42, - "title": "Fix the build", - "body": "The build is **red**.", - "html_url": "https://github.com/acme/widgets/issues/7", - "updated_at": "2024-05-21T15:30:00Z" - }]} - }); - let items = payload_items("github", "conn_gh", &data); - assert_eq!(items.len(), 1); - let item = &items[0]; - assert_eq!(item.kind(), ItemKind::Document); - item.validate().unwrap(); - - let (title, body, meta) = text_of(item); - assert_eq!( - title.as_deref(), - Some("GitHub: acme/widgets#7: Fix the build") - ); - assert_eq!(body, "The build is **red**."); - assert_eq!(meta.source.kind, SourceKind::Composio); - assert_eq!(meta.source.id.as_deref(), Some("conn_gh")); - assert_eq!(meta.tags, vec!["github".to_string()]); - assert_eq!( - meta.url.as_deref(), - Some("https://github.com/acme/widgets/issues/7") - ); - assert_eq!(meta.repo.as_deref(), Some("acme/widgets")); - assert_eq!( - meta.observed_at, - Some(Utc.with_ymd_and_hms(2024, 5, 21, 15, 30, 0).unwrap()) - ); -} - -#[test] -fn post_processed_gmail_messages_carry_headers_thread_and_date() { - let data = json!({ "messages": [{ - "id": "m1", - "threadId": "t1", - "subject": "Lunch?", - "from": "Ann ", - "to": "me@example.com", - "date": "Tue, 21 May 2024 12:00:00 +0000", - "markdown": "Tacos at noon." - }]}); - let documents = normalise_payload("gmail", &data); - assert_eq!(documents.len(), 1); - let document = &documents[0]; - assert_eq!(document.title.as_deref(), Some("Lunch?")); - assert!(document.body.starts_with("From: Ann \n")); - assert!(document.body.ends_with("Tacos at noon.")); - assert_eq!(document.thread_id.as_deref(), Some("t1")); - assert_eq!(document.id.as_deref(), Some("m1")); - assert_eq!( - document.observed_at, - Some(Utc.with_ymd_and_hms(2024, 5, 21, 12, 0, 0).unwrap()) - ); -} - -#[test] -fn slack_messages_take_permalink_and_ts_time() { - let data = json!({ "messages": [{ - "ts": "1716300000.000100", - "user": "U1", - "channel_id": "C1", - "text": "deploy done", - "permalink": "https://acme.slack.com/archives/C1/p1716300000000100" - }]}); - let items = payload_items("slack", "src_slack", &data); - let (title, body, meta) = text_of(&items[0]); - assert_eq!(title.as_deref(), Some("Slack message from U1 in C1")); - assert_eq!(body, "deploy done"); - assert_eq!( - meta.url.as_deref(), - Some("https://acme.slack.com/archives/C1/p1716300000000100") - ); - assert_eq!( - meta.observed_at.map(|at| at.timestamp()), - Some(1_716_300_000) - ); - assert_eq!(meta.tags, vec!["slack".to_string()]); -} - -#[test] -fn linear_notion_and_clickup_records_are_normalised() { - let linear = json!({ "nodes": [{ - "identifier": "ENG-1", - "title": "Ship v2", - "description": "All of it.", - "url": "https://linear.app/acme/issue/ENG-1", - "updatedAt": "2024-01-02T03:04:05Z" - }]}); - let documents = normalise_payload("linear", &linear); - assert_eq!(documents[0].id.as_deref(), Some("ENG-1")); - assert_eq!(documents[0].body, "All of it."); - assert!(documents[0].observed_at.is_some()); - - let notion = json!({ "results": [{ - "id": "p1", - "url": "https://notion.so/p1", - "last_edited_time": "2024-01-02T03:04:05.000Z", - "properties": { "Name": { "type": "title", "title": [{ "plain_text": "Roadmap" }] } } - }]}); - let documents = normalise_payload("notion", ¬ion); - assert_eq!(documents[0].title.as_deref(), Some("Roadmap")); - assert_eq!( - documents[0].body, "Roadmap", - "a page without body keeps its title" - ); - assert_eq!(documents[0].url.as_deref(), Some("https://notion.so/p1")); - - let page_markdown = json!({ "data": { "markdown": "# Plan\n\nDo it." }, "id": "p2" }); - let documents = normalise_payload("notion", &page_markdown); - assert_eq!(documents.len(), 1); - assert_eq!(documents[0].body, "# Plan\n\nDo it."); - - let clickup = json!({ "tasks": [{ - "id": "t9", - "name": "Write docs", - "description": "

The README

", - "url": "https://app.clickup.com/t/t9", - "date_updated": "1700000000000" - }]}); - let documents = normalise_payload("clickup", &clickup); - assert_eq!(documents[0].id.as_deref(), Some("t9")); - assert_eq!( - documents[0].observed_at.map(|at| at.timestamp()), - Some(1_700_000_000) - ); -} - -#[test] -fn html_bodies_are_converted_to_markdown() { - let data = json!({ "nodes": [{ - "title": "Page", - "description": "

Hi

There

" - }]}); - let documents = normalise_payload("linear", &data); - assert_eq!(documents[0].body, "## Hi\n\nThere"); -} - -#[test] -fn records_without_any_text_are_skipped() { - let data = json!({ "messages": [{ "ts": "1.0", "text": " " }] }); - assert!(normalise_payload("slack", &data).is_empty()); - assert!(normalise_payload("github", &json!({ "items": [{}] })).is_empty()); -} - -#[test] -fn an_unknown_toolkit_keeps_each_record_as_fenced_json() { - let data = json!({ "data": { "items": [ - { "id": 1, "name": "Alpha", "url": "https://example.com/1", "updated_at": "2024-01-01T00:00:00Z" }, - { "id": 2 } - ]}}); - let items = payload_items("hubspot", "conn_hs", &data); - assert_eq!(items.len(), 2); - let (title, body, meta) = text_of(&items[0]); - assert_eq!(title.as_deref(), Some("Alpha")); - assert!(body.starts_with("```json\n")); - assert!(body.contains("\"name\": \"Alpha\"")); - assert_eq!(meta.url.as_deref(), Some("https://example.com/1")); - assert_eq!(meta.tags, vec!["hubspot".to_string()]); - - let single = normalise_payload("hubspot", &json!({ "ok": true })); - assert_eq!(single.len(), 1); -} - -#[test] -fn slack_timestamps_parse_with_and_without_a_fraction() { - assert_eq!(parse_slack_ts("10").map(|at| at.timestamp()), Some(10)); - assert_eq!( - parse_slack_ts("10.5").map(|at| at.timestamp_subsec_micros()), - Some(500_000) - ); - assert_eq!(parse_slack_ts("x"), None); -} diff --git a/crates/tinymemory-integrations/src/sources/composio/fields/mod.rs b/crates/tinymemory-integrations/src/sources/composio/fields/mod.rs deleted file mode 100644 index 4c257085..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/fields/mod.rs +++ /dev/null @@ -1,44 +0,0 @@ -//! Field lookup shared by the Composio normalisers: pull a string out of a -//! payload by trying several dotted paths, because Composio wraps the same -//! upstream field at different depths depending on the action and version. - -/// Walk a JSON object using a list of dotted-path candidates and return the -/// first non-empty **string** match, trimmed. -/// -/// Each path is split on `.` and followed with `Value::get`, so it only -/// descends through objects — it never indexes into an array. A leaf that is -/// not a string (a number, a bool) is rejected rather than coerced, so a -/// payload whose `id` is `42` rather than `"42"` yields `None` here. That -/// differs from the private `scalar` lookup in the `documents` mapping, which -/// renders numbers; the normalisers were written against the -/// reject-non-strings behaviour and `pick_str_rejects_non_string_values` pins -/// it. -pub fn pick_str(value: &serde_json::Value, paths: &[&str]) -> Option { - for path in paths { - let mut cur = value; - let mut ok = true; - for segment in path.split('.') { - match cur.get(segment) { - Some(next) => cur = next, - None => { - ok = false; - break; - } - } - } - if !ok { - continue; - } - if let Some(s) = cur.as_str() { - let trimmed = s.trim(); - if !trimmed.is_empty() { - return Some(trimmed.to_string()); - } - } - } - None -} - -#[cfg(test)] -#[path = "mod_tests.rs"] -mod tests; diff --git a/crates/tinymemory-integrations/src/sources/composio/fields/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/fields/mod_tests.rs deleted file mode 100644 index 330ec404..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/fields/mod_tests.rs +++ /dev/null @@ -1,37 +0,0 @@ -//! Tests for the shared Composio field lookup. - -use super::*; -use serde_json::json; - -#[test] -fn pick_str_finds_first_non_empty_match() { - let v = json!({"data": {"user": {"name": "Ada", "email": "ada@example.com"}}}); - assert_eq!( - pick_str(&v, &["data.user.name", "data.user.email"]), - Some("Ada".into()) - ); - assert_eq!( - pick_str(&v, &["data.missing", "data.user.email"]), - Some("ada@example.com".into()) - ); - assert_eq!(pick_str(&v, &["nope.nope"]), None); -} - -#[test] -fn pick_str_respects_path_order() { - let v = json!({"a": "first", "b": "second"}); - assert_eq!(pick_str(&v, &["a", "b"]), Some("first".into())); - assert_eq!(pick_str(&v, &["b", "a"]), Some("second".into())); -} - -/// The drift guard for the behaviour documented on [`pick_str`]. If this -/// ever starts returning `Some("42")`, the normalisers' emitted ids have -/// changed. -#[test] -fn pick_str_rejects_non_string_values() { - let v = json!({"count": 42, "flag": true, "empty": "", "whitespace": " "}); - assert_eq!(pick_str(&v, &["count"]), None); - assert_eq!(pick_str(&v, &["flag"]), None); - assert_eq!(pick_str(&v, &["empty"]), None); - assert_eq!(pick_str(&v, &["whitespace"]), None); -} diff --git a/crates/tinymemory-integrations/src/sources/composio/github/mod.rs b/crates/tinymemory-integrations/src/sources/composio/github/mod.rs deleted file mode 100644 index 12e7a615..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/github/mod.rs +++ /dev/null @@ -1,116 +0,0 @@ -//! GitHub host normalization helpers — issue extraction and issue id, title and -//! timestamp helpers. -//! -//! GitHub's REST API (proxied through Composio) returns search results in a -//! small number of shapes. The functions here -//! walk the union of common Composio envelope variants so the provider stays -//! clean and branch-free. - -use serde_json::Value; - -use super::fields::pick_str; - -/// Walk the Composio response envelope for GitHub search issue results. -/// -/// `GITHUB_SEARCH_ISSUES_AND_PULL_REQUESTS` wraps GitHub's `GET /search/issues` response, which -/// returns `{"total_count": N, "items": [...]}`. Composio may re-wrap this under -/// `data` or `data.data`; we probe each shape in order. -pub fn extract_issues(data: &Value) -> Vec { - let candidates = [ - data.pointer("/data/items"), - data.pointer("/items"), - data.pointer("/data/data/items"), - data.pointer("/data/results"), - data.pointer("/results"), - ]; - for cand in candidates.into_iter().flatten() { - if let Some(arr) = cand.as_array() { - return arr.clone(); - } - } - Vec::new() -} - -/// Extract a stable, globally unique identifier for a GitHub issue or PR. -/// -/// GitHub's internal `id` field is a large integer unique across all issues -/// and PRs on github.com. We convert it to a string for use as a sync key. -/// Falls back to composing from `html_url` path if `id` is absent. -pub fn extract_issue_id(issue: &Value) -> Option { - // Primary: numeric internal GitHub ID. - if let Some(id) = issue.get("id").or_else(|| issue.pointer("/data/id")) { - if let Some(n) = id.as_u64() { - return Some(n.to_string()); - } - if let Some(s) = id.as_str() { - let trimmed = s.trim(); - if !trimmed.is_empty() { - return Some(trimmed.to_string()); - } - } - } - // Fallback: parse owner/repo/number from html_url path segments. - // URL shape: https://github.com/{owner}/{repo}/issues/{number} - if let Some(url) = pick_str(issue, &["html_url", "data.html_url", "url", "data.url"]) - && let Some(slug) = github_url_to_slug(&url) - { - return Some(slug); - } - None -} - -/// Build a human-readable document title for a GitHub issue/PR. -/// -/// Format: `GitHub: {owner}/{repo}#{number}: {title}`. -/// Falls back to just the title or a placeholder when fields are missing. -pub fn extract_issue_title(issue: &Value) -> Option { - let title = pick_str(issue, &["title", "data.title"])?; - - // Best-effort: extract owner/repo#N from html_url for the prefix. - let prefix = pick_str(issue, &["html_url", "data.html_url"]) - .and_then(|url| github_url_to_slug(&url)) - .unwrap_or_default(); - - if prefix.is_empty() { - Some(title) - } else { - Some(format!("GitHub: {prefix}: {title}")) - } -} - -/// Parse `https://github.com/{owner}/{repo}/issues/{number}` (or `/pull/`) -/// into `"{owner}/{repo}#{number}"`. Returns `None` for unrecognised shapes. -fn github_url_to_slug(url: &str) -> Option { - let segs: Vec<&str> = url.trim_end_matches('/').split('/').collect(); - // Minimum: ["https:", "", "github.com", owner, repo, "issues", number] - if segs.len() >= 7 { - let number = segs[segs.len() - 1]; - let _kind = segs[segs.len() - 2]; // "issues" or "pull" — ignored - let repo = segs[segs.len() - 3]; - let owner = segs[segs.len() - 4]; - if !owner.is_empty() && !repo.is_empty() && !number.is_empty() { - return Some(format!("{owner}/{repo}#{number}")); - } - } - None -} - -/// Extract the `updated_at` ISO 8601 timestamp from a GitHub issue. -/// -/// GitHub returns `updated_at` as `"2024-05-21T15:30:00Z"`. ISO 8601 strings -/// sort lexicographically, so we use them directly as the sync cursor. -pub fn extract_issue_updated_at(issue: &Value) -> Option { - pick_str( - issue, - &[ - "updated_at", - "data.updated_at", - "updatedAt", - "data.updatedAt", - ], - ) -} - -#[cfg(test)] -#[path = "mod_tests.rs"] -mod tests; diff --git a/crates/tinymemory-integrations/src/sources/composio/github/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/github/mod_tests.rs deleted file mode 100644 index 0b7322f1..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/github/mod_tests.rs +++ /dev/null @@ -1,97 +0,0 @@ -//! Tests for the GitHub normaliser. - -use super::*; -use serde_json::json; - -#[test] -fn extract_issues_from_data_items() { - let data = json!({ "data": { "items": [{"id": 1}] } }); - assert_eq!(extract_issues(&data).len(), 1); -} - -#[test] -fn extract_issues_from_top_level_items() { - let data = json!({ "items": [{"id": 1}, {"id": 2}] }); - assert_eq!(extract_issues(&data).len(), 2); -} - -#[test] -fn extract_issues_empty_when_missing() { - let data = json!({ "foo": "bar" }); - assert!(extract_issues(&data).is_empty()); -} - -#[test] -fn extract_issue_id_from_numeric_field() { - let issue = json!({ "id": 123456789u64, "title": "Fix bug" }); - assert_eq!(extract_issue_id(&issue), Some("123456789".to_string())); -} - -#[test] -fn extract_issue_id_from_wrapped_data() { - let issue = json!({ "data": { "id": 99u64 } }); - assert_eq!(extract_issue_id(&issue), Some("99".to_string())); -} - -#[test] -fn extract_issue_id_falls_back_to_html_url() { - let issue = json!({ - "html_url": "https://github.com/owner/repo/issues/42" - }); - assert_eq!(extract_issue_id(&issue), Some("owner/repo#42".to_string())); -} - -#[test] -fn extract_issue_id_none_when_missing() { - let issue = json!({ "title": "No ID here" }); - assert!(extract_issue_id(&issue).is_none()); -} - -#[test] -fn extract_issue_title_builds_prefixed_title() { - let issue = json!({ - "id": 1u64, - "title": "Fix race condition", - "html_url": "https://github.com/acme/core/issues/99" - }); - assert_eq!( - extract_issue_title(&issue), - Some("GitHub: acme/core#99: Fix race condition".to_string()) - ); -} - -#[test] -fn extract_issue_title_returns_raw_title_when_no_url() { - let issue = json!({ "title": "Bare title" }); - assert_eq!(extract_issue_title(&issue), Some("Bare title".to_string())); -} - -#[test] -fn extract_issue_title_none_when_missing() { - let issue = json!({ "id": 1u64 }); - assert!(extract_issue_title(&issue).is_none()); -} - -#[test] -fn extract_issue_updated_at_from_top_level() { - let issue = json!({ "updated_at": "2024-05-21T15:30:00Z" }); - assert_eq!( - extract_issue_updated_at(&issue), - Some("2024-05-21T15:30:00Z".to_string()) - ); -} - -#[test] -fn extract_issue_updated_at_from_data_wrapper() { - let issue = json!({ "data": { "updated_at": "2023-01-01T00:00:00Z" } }); - assert_eq!( - extract_issue_updated_at(&issue), - Some("2023-01-01T00:00:00Z".to_string()) - ); -} - -#[test] -fn extract_issue_updated_at_none_when_missing() { - let issue = json!({ "id": 1u64 }); - assert!(extract_issue_updated_at(&issue).is_none()); -} diff --git a/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod.rs b/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod.rs deleted file mode 100644 index c6c78ae7..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod.rs +++ /dev/null @@ -1,498 +0,0 @@ -//! Gmail-specific post-processing of Composio action responses. -//! -//! The upstream `GMAIL_FETCH_EMAILS` payload is extremely verbose -//! (full MIME tree under `payload.parts[]`, 50+ `Received:` headers, -//! display-layer noise the model never uses). This module rewrites -//! it into a slim envelope per message: -//! -//! ```json -//! { -//! "messages": [ -//! { -//! "id": "…", -//! "threadId": "…", -//! "subject": "…", -//! "from": "…", -//! "to": "…", -//! "date": "…", -//! "labels": ["INBOX", "UNREAD"], -//! "markdown": "…body…", -//! "attachments": [ { "filename": "...", "mimeType": "..." } ] -//! } -//! ], -//! "nextPageToken": "…", -//! "resultSizeEstimate": 201 -//! } -//! ``` -//! -//! ## Body source -//! -//! Composio's backend ships a -//! `markdownFormatted` field on the response envelope — one string -//! per tool call, pre-rendered with HTML stripped, URLs shortened, -//! footers removed, whitespace normalised. We split it per message -//! along `\n---\n` boundaries (with `## ` heading fallbacks) and -//! pin each slice to the corresponding entry in `messages[]` via -//! [`apply_response_level_markdown`]. The reshape's -//! `extract_markdown_body` then prefers that pinned field over -//! falling back to the upstream `messageText`. -//! -//! No in-house HTML→markdown conversion lives here anymore — the -//! backend does the cleaning. If `markdownFormatted` is absent for -//! a given response we fall through to whatever plain text the -//! upstream provided in `messageText`. -//! -//! Callers that need the raw Composio shape can pass `raw_html: -//! true` (or `rawHtml: true`) in the action arguments — this -//! short-circuits the reshape entirely. -//! -//! Only `GMAIL_FETCH_EMAILS` is reshaped today; other Gmail action -//! responses are passed through unchanged. When we add envelopes for -//! more slugs they should live in this file, branched from -//! [`post_process`]. - -use serde_json::{Map, Value, json}; - -/// Entry point a host calls on each Gmail action response (the slug names -/// the action) before handing it to `normalise_payload`. -/// -/// Dispatches on the Composio action slug. Unknown Gmail slugs fall -/// through to a no-op. -pub fn post_process(slug: &str, arguments: Option<&Value>, data: &mut Value) { - if is_raw_html_flag_set(arguments) { - tracing::debug!( - slug, - "[composio:gmail][post-process] raw_html flag set, passing through" - ); - return; - } - if slug == "GMAIL_FETCH_EMAILS" { - reshape_fetch_emails(data) - } -} - -/// Stash per-message slices of the response-level `markdownFormatted` -/// onto the corresponding entries inside `data.messages[]`. -/// -/// The Composio backend (tinyhumansai/backend#683) ships ONE -/// `markdownFormatted` string per tool call covering all messages — -/// already URL-shortened, footer-stripped, and whitespace-normalised. -/// To get per-email files in the raw archive we split that string -/// along section boundaries (`## ` headings or `---` rules) and pin -/// each slice to the message at the same index. `extract_markdown_body` -/// then prefers `msg.markdownFormatted` over re-decoding the MIME -/// tree. -/// -/// **Must be called BEFORE [`post_process`]** because `post_process` -/// reshapes `data` into the slim envelope; once `messages[]` carries -/// our slim shape the upstream message ordering is already locked in -/// but we may have lost original ordering signals if any. -/// -/// No-op when the slice count doesn't match `messages.len()` — we -/// can't safely align segments to messages without an exact match, -/// so we let `extract_markdown_body` fall through to its MIME path. -pub fn apply_response_level_markdown(data: &mut Value, top_md: &str) { - let trimmed = top_md.trim(); - if trimmed.is_empty() { - return; - } - // Presence is checked immutably first, then fetched mutably. The original - // form re-fetched with `unwrap()` after a mutable probe, which is sound but - // relies on the reader to see why; this crate lints against `unwrap`, and the - // immutable probe expresses the same reasoning to the compiler. - let container = if data.get("messages").is_some() { - data - } else if data.get("data").and_then(Value::as_object).is_some() { - match data.get_mut("data") { - Some(inner) => inner, - None => return, - } - } else { - tracing::debug!( - "[composio:gmail][post-process] apply_response_level_markdown: \ - no messages container in response — skipping" - ); - return; - }; - let Some(messages) = container.get_mut("messages").and_then(|v| v.as_array_mut()) else { - return; - }; - let count = messages.len(); - if count == 0 { - return; - } - // Clone hints out of the messages array so the slice borrows - // don't conflict with the upcoming `messages.iter_mut()` mutation. - let hints: Vec = messages.clone(); - let Some(slices) = split_response_markdown_per_message_with_hint(trimmed, count, Some(&hints)) - else { - tracing::debug!( - messages = count, - md_len = trimmed.len(), - "[composio:gmail][post-process] could not split response-level markdownFormatted \ - into {count} slices — falling back to per-message MIME decode" - ); - return; - }; - for (msg, slice) in messages.iter_mut().zip(slices) { - if let Some(obj) = msg.as_object_mut() { - obj.insert("markdownFormatted".to_string(), Value::String(slice)); - } - } - tracing::debug!( - messages = count, - "[composio:gmail][post-process] stashed per-message markdownFormatted slices" - ); -} - -/// Split a top-level `markdownFormatted` string into per-message -/// segments. Returns `Some(slices)` only when the split yields -/// exactly `expected_count` entries — otherwise the format isn't one -/// of the patterns we know about and we let the caller fall back. -/// -/// Primary boundary is the `\n---\n` horizontal rule the backend -/// emits between messages (confirmed against real -/// `GMAIL_FETCH_EMAILS` output). H2/H3 headings are kept as -/// fallbacks for older renderings. The preamble (`# Inbox (N -/// messages)`-style intro, if present) is dropped — we accept -/// either `expected` parts (no preamble) or `expected + 1` -/// (preamble + N messages). -/// -/// `messages_hint` is the slim message array from the same response -/// — when present we use the per-message `subject` field to verify -/// each segment really does belong to the message at the same index. -/// Mismatches force a fallback so we never write a wrong-message body -/// to the raw archive. -/// -/// The hint is what makes the split reliable: the blob's own section headings -/// are backend-rendered and have changed shape between versions, so matching -/// on them alone silently mis-attributed bodies. -pub fn split_response_markdown_per_message_with_hint( - md: &str, - expected_count: usize, - messages_hint: Option<&[Value]>, -) -> Option> { - if expected_count == 0 { - return None; - } - if expected_count == 1 { - return Some(vec![md.to_string()]); - } - - // Boundary patterns to try, in priority order. `\n---\n` is the - // confirmed marker; the heading variants stay as belt-and-braces - // for older / variant backend renderings. - let candidates: &[(&str, &str)] = &[ - ("\n---\n", "---\n"), - ("\n\n## ", "## "), - ("\n\n### ", "### "), - ("\n\n# ", "# "), - ("\n***\n", "***\n"), - ]; - - for (sep, prefix) in candidates { - let parts: Vec<&str> = md.split(sep).collect(); - let (drop_preamble, prepend_first) = if parts.len() == expected_count { - (false, false) // no preamble; first segment had no prefix - } else if parts.len() == expected_count + 1 { - (true, true) // preamble dropped; every kept segment had a prefix - } else { - continue; - }; - let segments: Vec = parts - .into_iter() - .skip(if drop_preamble { 1 } else { 0 }) - .enumerate() - .map(|(i, s)| { - if i == 0 && !prepend_first { - s.to_string() - } else { - format!("{prefix}{s}") - } - }) - .collect(); - - // Validate alignment against the JSON message array: every - // segment whose corresponding message has a non-empty subject - // must mention that subject somewhere in its body. If a single - // pair fails, we treat the split as unreliable and try the - // next pattern. Empty / null subjects skip validation (e.g. - // notification mails where the subject is ""). - if let Some(hints) = messages_hint - && !validate_segments_against_hints(&segments, hints) - { - tracing::debug!( - expected = expected_count, - sep = sep, - "[composio:gmail][post-process] split candidate failed subject check" - ); - continue; - } - return Some(segments); - } - None -} - -/// True if every (segment, message) pair where the message has a -/// non-empty subject contains that subject somewhere in the segment -/// (case-insensitive substring match — a defensive heuristic, not a -/// strict equality check, since the backend may format subjects -/// inside markdown links or with surrounding decoration). -fn validate_segments_against_hints(segments: &[String], hints: &[Value]) -> bool { - if segments.len() != hints.len() { - return false; - } - for (seg, hint) in segments.iter().zip(hints.iter()) { - let subject = hint - .get("subject") - .and_then(|v| v.as_str()) - .unwrap_or("") - .trim(); - if subject.is_empty() { - continue; - } - if !seg - .to_ascii_lowercase() - .contains(&subject.to_ascii_lowercase()) - { - return false; - } - } - true -} - -/// Returns true when the caller explicitly set `raw_html: true` (or the -/// camelCase `rawHtml: true`) in the `arguments` object. -fn is_raw_html_flag_set(arguments: Option<&Value>) -> bool { - let Some(obj) = arguments.and_then(|v| v.as_object()) else { - return false; - }; - obj.get("raw_html") - .or_else(|| obj.get("rawHtml")) - .and_then(|v| v.as_bool()) - .unwrap_or(false) -} - -/// Rewrite a `GMAIL_FETCH_EMAILS` `data` object in place into the slim -/// envelope documented at the module level. -/// -/// The Composio response can be shaped either as `{ messages, nextPageToken, ... }` -/// directly, or wrapped one level deeper under `{ data: { messages: … } }` -/// depending on backend version; we handle both. -fn reshape_fetch_emails(data: &mut Value) { - // Unwrap an optional `data:` envelope so downstream logic only has - // to deal with one shape. - let container = if data.get("messages").is_some() { - data - } else if data.get("data").and_then(Value::as_object).is_some() { - match data.get_mut("data") { - Some(inner) => inner, - None => return, - } - } else { - return; - }; - - let Some(obj) = container.as_object_mut() else { - return; - }; - - let raw_messages = obj - .remove("messages") - .and_then(|v| match v { - Value::Array(arr) => Some(arr), - _ => None, - }) - .unwrap_or_default(); - let next_page_token = obj.remove("nextPageToken").unwrap_or(Value::Null); - let result_size_estimate = obj.remove("resultSizeEstimate").unwrap_or(Value::Null); - - let messages: Vec = raw_messages.into_iter().map(reshape_message).collect(); - - let mut envelope = Map::new(); - envelope.insert("messages".into(), Value::Array(messages)); - if !next_page_token.is_null() { - envelope.insert("nextPageToken".into(), next_page_token); - } - if !result_size_estimate.is_null() { - envelope.insert("resultSizeEstimate".into(), result_size_estimate); - } - - *container = Value::Object(envelope); -} - -/// Parse an RFC 3339 or RFC 2822 date string into a UTC `DateTime`. -pub fn parse_email_date(date_str: &str) -> Option> { - date_str - .parse::>() - .or_else(|_| { - chrono::DateTime::parse_from_rfc2822(date_str).map(|d| d.with_timezone(&chrono::Utc)) - }) - .ok() -} - -const EMAIL_LOCAL_TIME_FMT: &str = "%Y-%m-%d %I:%M %p %:z"; - -/// Format a UTC `DateTime` in the given timezone. Returns `None` when the -/// formatted result is identical to the UTC rendering (no-op for UTC hosts). -pub fn format_at_tz( - utc: chrono::DateTime, - tz: &Tz, -) -> Option -where - Tz::Offset: std::fmt::Display, -{ - let local_dt = utc.with_timezone(tz); - let formatted = local_dt.format(EMAIL_LOCAL_TIME_FMT).to_string(); - - let utc_formatted = utc.format(EMAIL_LOCAL_TIME_FMT).to_string(); - if formatted == utc_formatted { - return None; - } - Some(formatted) -} - -/// Convert a UTC email timestamp string to a human-readable local-time string. -/// -/// Accepts RFC 3339 (`"2026-05-31T10:33:00Z"`) or RFC 2822 -/// (`"Sat, 31 May 2026 10:33:00 +0000"`) input. Returns a formatted string -/// in the host's local timezone, e.g. `"2026-05-31 05:33 AM -05:00"`, -/// so the agent can present local times without UTC arithmetic. -/// -/// The raw `date` field is always preserved alongside this field so -/// internal sorting, deduplication, and debugging remain UTC-based. -/// -/// Returns `None` when the input cannot be parsed or the output format -/// would be identical to the UTC input (no-op for UTC hosts). -pub fn format_email_local_time(date_str: &str) -> Option { - let utc = parse_email_date(date_str)?; - format_at_tz(utc, &chrono::Local) -} - -/// Map one raw Composio message object to its slim counterpart. -/// -/// Body source picked by [`extract_markdown_body`]: -/// 1. The per-message `markdownFormatted` slice pinned by -/// [`apply_response_level_markdown`] (preferred — backend-rendered). -/// 2. The upstream `messageText` plaintext (fallback). -/// 3. Empty string. -fn reshape_message(raw: Value) -> Value { - let Value::Object(obj) = raw else { - return raw; - }; - - let id = obj.get("messageId").cloned().unwrap_or(Value::Null); - let thread_id = obj.get("threadId").cloned().unwrap_or(Value::Null); - let subject = obj.get("subject").cloned().unwrap_or(Value::Null); - let sender = obj.get("sender").cloned().unwrap_or(Value::Null); - let to = obj.get("to").cloned().unwrap_or(Value::Null); - let date = obj - .get("messageTimestamp") - .cloned() - .or_else(|| pick_header(&obj, "Date")) - .unwrap_or(Value::Null); - let labels = obj - .get("labelIds") - .cloned() - .unwrap_or_else(|| Value::Array(Vec::new())); - let list_unsubscribe = pick_header(&obj, "List-Unsubscribe").unwrap_or(Value::Null); - - let markdown = extract_markdown_body(&obj); - let attachments = extract_attachments(&obj); - - // Compute a local-time representation of the UTC `date` so the agent - // presents times in the user's timezone rather than quoting raw UTC. - let date_local = date.as_str().and_then(format_email_local_time); - - let mut out = Map::new(); - out.insert("id".into(), id); - out.insert("threadId".into(), thread_id); - out.insert("subject".into(), subject); - out.insert("from".into(), sender); - out.insert("to".into(), to); - out.insert("date".into(), date); - if let Some(local) = date_local { - out.insert("date_local".into(), Value::String(local)); - } - out.insert("labels".into(), labels); - if !list_unsubscribe.is_null() { - out.insert("list_unsubscribe".into(), list_unsubscribe); - } - out.insert("markdown".into(), Value::String(markdown)); - if !attachments.is_empty() { - out.insert("attachments".into(), Value::Array(attachments)); - } - Value::Object(out) -} - -/// Find a header value by (case-insensitive) name in the Composio -/// `payload.headers[]` array. Returns `Some(Value::String)` on hit. -fn pick_header(msg: &Map, name: &str) -> Option { - let headers = msg.get("payload")?.get("headers")?.as_array()?; - for h in headers { - let hn = h.get("name").and_then(|v| v.as_str()).unwrap_or(""); - if hn.eq_ignore_ascii_case(name) - && let Some(v) = h.get("value").and_then(|v| v.as_str()) - { - return Some(Value::String(v.to_string())); - } - } - None -} - -/// Pick a body for the slim envelope. -/// -/// We trust the Composio backend's pre-rendered `markdownFormatted` -/// (set per-message by [`apply_response_level_markdown`] from the -/// response-level field). When that's absent we fall back to the -/// upstream's plain-text `messageText` verbatim — no in-house -/// HTML→markdown decoding lives here anymore. The backend already -/// strips HTML, shortens URLs, and normalises whitespace; running -/// our own pipeline on top duplicated work and corrupted some -/// renderings. -fn extract_markdown_body(msg: &Map) -> String { - if let Some(formatted) = msg - .get("markdownFormatted") - .or_else(|| msg.get("markdown_formatted")) - .and_then(|v| v.as_str()) - .map(str::trim) - .filter(|s| !s.is_empty()) - { - return formatted.to_string(); - } - if let Some(text) = msg - .get("messageText") - .and_then(|v| v.as_str()) - .map(str::trim) - .filter(|s| !s.is_empty()) - { - return text.to_string(); - } - String::new() -} - -/// Pull a minimal attachments descriptor from the Composio -/// `attachmentList` array. -fn extract_attachments(msg: &Map) -> Vec { - if let Some(list) = msg.get("attachmentList").and_then(|v| v.as_array()) { - return list - .iter() - .filter_map(|a| { - let filename = a.get("filename").and_then(|v| v.as_str())?; - if filename.is_empty() { - return None; - } - let mime = a - .get("mimeType") - .and_then(|v| v.as_str()) - .unwrap_or_default(); - Some(json!({ "filename": filename, "mimeType": mime })) - }) - .collect(); - } - Vec::new() -} - -#[cfg(test)] -#[path = "mod_tests.rs"] -mod tests; diff --git a/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod_tests.rs deleted file mode 100644 index dd1b44af..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod_tests.rs +++ /dev/null @@ -1,356 +0,0 @@ -//! Tests for the Gmail post-processor. - -use super::*; -use serde_json::json; - -fn fixture_with_backend_markdown() -> Value { - json!({ - "messages": [ - { - "messageId": "m1", - "threadId": "t1", - "subject": "Hello", - "sender": "a@x.com", - "to": "b@y.com", - "messageTimestamp": "2026-04-17T12:00:00Z", - "labelIds": ["INBOX", "UNREAD"], - // Pre-rendered slice (set by `apply_response_level_markdown` - // in production; inline here for the reshape test). - "markdownFormatted": "# Hello\n\nbody copy", - "messageText": "fallback should not be used", - "display_url": "ignore-me", - "preview": { "body": "Hi plain", "subject": "Hello" }, - "attachmentList": [ - { "filename": "report.pdf", "mimeType": "application/pdf", "size": 12345 }, - { "filename": "", "mimeType": "text/html" } - ], - "payload": {} - } - ], - "nextPageToken": "tok-1", - "resultSizeEstimate": 42 - }) -} - -#[test] -fn reshape_emits_slim_envelope() { - let mut v = fixture_with_backend_markdown(); - post_process("GMAIL_FETCH_EMAILS", None, &mut v); - - assert_eq!(v["nextPageToken"], "tok-1"); - assert_eq!(v["resultSizeEstimate"], 42); - - let msgs = v["messages"].as_array().unwrap(); - assert_eq!(msgs.len(), 1); - let m = &msgs[0]; - - assert_eq!(m["id"], "m1"); - assert_eq!(m["threadId"], "t1"); - assert_eq!(m["subject"], "Hello"); - assert_eq!(m["from"], "a@x.com"); - assert_eq!(m["to"], "b@y.com"); - assert_eq!(m["date"], "2026-04-17T12:00:00Z"); - assert_eq!(m["labels"], json!(["INBOX", "UNREAD"])); - - let md = m["markdown"].as_str().unwrap(); - assert_eq!(md, "# Hello\n\nbody copy"); - - // Noise fields removed. - assert!(m.get("display_url").is_none()); - assert!(m.get("preview").is_none()); - assert!(m.get("payload").is_none()); - assert!(m.get("messageText").is_none()); - - // Attachments: empty filename entry is filtered. - let atts = m["attachments"].as_array().unwrap(); - assert_eq!(atts.len(), 1); - assert_eq!(atts[0]["filename"], "report.pdf"); - assert_eq!(atts[0]["mimeType"], "application/pdf"); -} - -#[test] -fn raw_html_flag_passes_through_unchanged() { - let mut v = fixture_with_backend_markdown(); - let original = v.clone(); - let args = json!({ "raw_html": true }); - post_process("GMAIL_FETCH_EMAILS", Some(&args), &mut v); - assert_eq!( - v, original, - "raw_html=true must preserve the Composio shape" - ); -} - -#[test] -fn camel_case_raw_html_also_recognized() { - let mut v = fixture_with_backend_markdown(); - let original = v.clone(); - let args = json!({ "rawHtml": true }); - post_process("GMAIL_FETCH_EMAILS", Some(&args), &mut v); - assert_eq!(v, original); -} - -#[test] -fn falls_back_to_message_text_when_no_backend_markdown() { - let mut v = json!({ - "messages": [{ - "messageId": "m1", - "threadId": "t1", - "subject": "s", - "sender": "a@x.com", - "to": "b@y.com", - "messageTimestamp": "2026-04-17", - "labelIds": [], - "messageText": " plain body text ", - "payload": {} - }], - "nextPageToken": null - }); - post_process("GMAIL_FETCH_EMAILS", None, &mut v); - let md = v["messages"][0]["markdown"].as_str().unwrap(); - assert_eq!(md, "plain body text"); - assert!(v.get("nextPageToken").is_none(), "null tokens dropped"); -} - -#[test] -fn unwraps_data_envelope() { - let mut v = json!({ - "data": { - "messages": [{ - "messageId": "m1", - "threadId": "t1", - "subject": "s", - "sender": "a@x.com", - "to": "b@y.com", - "messageTimestamp": "2026-04-17", - "labelIds": [], - "messageText": "body", - "payload": {} - }] - } - }); - post_process("GMAIL_FETCH_EMAILS", None, &mut v); - // Reshape writes into `data` in place. - let msgs = v["data"]["messages"].as_array().unwrap(); - assert_eq!(msgs.len(), 1); - assert_eq!(msgs[0]["markdown"], "body"); -} - -#[test] -fn non_fetch_slug_is_noop() { - let mut v = json!({ "messages": [{ "messageId": "m1", "messageText": "x" }] }); - let original = v.clone(); - post_process("GMAIL_SEND_EMAIL", None, &mut v); - assert_eq!(v, original); -} - -#[test] -fn prefers_backend_markdown_formatted_when_present() { - // Composio backend (tinyhumansai/backend#683 +) ships - // `markdownFormatted` already URL-shortened + footer-stripped - // per message (after `apply_response_level_markdown` slices the - // response-level field). When present, our post-processor must - // use it verbatim instead of falling back to `messageText`. - let mut v = json!({ - "messages": [{ - "messageId": "m1", - "threadId": "t1", - "subject": "s", - "sender": "a@x.com", - "to": "b@y.com", - "messageTimestamp": "2026-04-17", - "labelIds": [], - "markdownFormatted": "# Already nice\n\nShort URL: https://gh.io/abc", - "messageText": "fallback should not be used", - "payload": {} - }] - }); - post_process("GMAIL_FETCH_EMAILS", None, &mut v); - let md = v["messages"][0]["markdown"].as_str().unwrap(); - assert_eq!(md, "# Already nice\n\nShort URL: https://gh.io/abc"); -} - -#[test] -fn empty_markdown_formatted_falls_through_to_message_text() { - let mut v = json!({ - "messages": [{ - "messageId": "m1", - "threadId": "t1", - "subject": "s", - "sender": "a@x.com", - "to": "b@y.com", - "messageTimestamp": "2026-04-17", - "labelIds": [], - "markdownFormatted": " \n \n", - "messageText": "real body", - "payload": {} - }] - }); - post_process("GMAIL_FETCH_EMAILS", None, &mut v); - let md = v["messages"][0]["markdown"].as_str().unwrap(); - assert!(md.contains("real body")); -} - -// ── split_response_markdown_per_message_with_hint ─────────────────────── - -#[test] -fn split_response_markdown_uses_horizontal_rule_marker() { - // The confirmed backend marker is `\n---\n`. Three messages → - // expect three slices when there's no preamble. - let md = "## Alice's update\n\nbody A with https://gh.io/abc\n---\n## Bob's reply\n\nbody B\n---\n## Carol\n\nbody C"; - let slices = super::split_response_markdown_per_message_with_hint(md, 3, None).unwrap(); - assert_eq!(slices.len(), 3); - assert!(slices[0].contains("Alice's update")); - assert!(slices[1].contains("Bob's reply")); - assert!(slices[2].contains("Carol")); - // The `---\n` prefix is preserved on every-but-the-first segment - // so the section break survives the round-trip. - assert!(slices[1].starts_with("---\n")); - assert!(slices[2].starts_with("---\n")); -} - -#[test] -fn split_response_markdown_drops_preamble() { - // When a preamble like `# Inbox` precedes the first marker, we - // see N+1 parts after split — the preamble must be dropped. - let md = "# Inbox (2 messages)\n---\n## A\n\nbody A\n---\n## B\n\nbody B"; - let slices = super::split_response_markdown_per_message_with_hint(md, 2, None).unwrap(); - assert_eq!(slices.len(), 2); - assert!(slices[0].contains("body A")); - assert!(slices[1].contains("body B")); - // Both segments should carry the prefix when preamble was dropped. - assert!(slices[0].starts_with("---\n")); - assert!(slices[1].starts_with("---\n")); -} - -#[test] -fn split_response_markdown_falls_back_to_h2_marker() { - // No `---` rules — backend used h2 headings as boundaries. - let md = "## Alice\n\nbody A\n\n## Bob\n\nbody B"; - let slices = super::split_response_markdown_per_message_with_hint(md, 2, None).unwrap(); - assert_eq!(slices.len(), 2); - assert!(slices[0].contains("body A")); - assert!(slices[1].contains("body B")); -} - -#[test] -fn split_response_markdown_returns_none_on_count_mismatch() { - let md = "## only one section here"; - assert!(super::split_response_markdown_per_message_with_hint(md, 3, None).is_none()); -} - -#[test] -fn split_response_markdown_single_message_returns_whole_input() { - let md = "## solo\n\nthe whole body"; - let slices = super::split_response_markdown_per_message_with_hint(md, 1, None).unwrap(); - assert_eq!(slices, vec![md.to_string()]); -} - -#[test] -fn split_with_hint_rejects_when_subjects_dont_match() { - let md = "## Foo\nbody1\n---\n## Bar\nbody2"; - let hints = vec![ - json!({"subject": "Completely different subject A"}), - json!({"subject": "Completely different subject B"}), - ]; - let out = super::split_response_markdown_per_message_with_hint(md, 2, Some(&hints)); - assert!(out.is_none(), "subject mismatch must force fallback"); -} - -#[test] -fn split_with_hint_accepts_when_subjects_match() { - let md = "## Welcome to Gmail\nbody1\n---\n## Your invoice\nbody2"; - let hints = vec![ - json!({"subject": "Welcome to Gmail"}), - json!({"subject": "Your invoice"}), - ]; - let slices = super::split_response_markdown_per_message_with_hint(md, 2, Some(&hints)).unwrap(); - assert_eq!(slices.len(), 2); - assert!(slices[0].contains("Welcome to Gmail")); - assert!(slices[1].contains("Your invoice")); -} - -#[test] -fn split_with_hint_skips_messages_with_blank_subject() { - let md = "## A\nbody1\n---\n## B\nbody2"; - let hints = vec![json!({"subject": "A"}), json!({"subject": ""})]; - let slices = super::split_response_markdown_per_message_with_hint(md, 2, Some(&hints)).unwrap(); - assert_eq!(slices.len(), 2); -} - -// ── format_email_local_time ────────────────────────────────────────────────── - -#[test] -fn format_email_local_time_returns_none_for_unparseable_date() { - assert!(super::format_email_local_time("not-a-date").is_none()); - assert!(super::format_email_local_time("").is_none()); -} - -#[test] -fn format_email_local_time_preserves_utc_raw_date_in_reshape() { - let mut v = json!({ - "messages": [{ - "messageId": "m1", - "threadId": "t1", - "subject": "Test", - "sender": "a@example.com", - "to": "b@example.com", - "messageTimestamp": "2026-05-31T10:33:00Z", - "labelIds": [], - "messageText": "body", - "payload": {} - }] - }); - post_process("GMAIL_FETCH_EMAILS", None, &mut v); - let msg = &v["messages"][0]; - assert_eq!(msg["date"], "2026-05-31T10:33:00Z"); -} - -#[test] -fn parse_email_date_accepts_rfc3339_and_rfc2822() { - assert!(super::parse_email_date("2026-05-31T10:33:00Z").is_some()); - assert!(super::parse_email_date("Sun, 31 May 2026 10:33:00 +0000").is_some()); - assert!(super::parse_email_date("not-a-date").is_none()); -} - -#[test] -fn format_at_tz_deterministic_with_fixed_offset() { - use chrono::FixedOffset; - - let utc = super::parse_email_date("2026-05-31T10:33:00Z").unwrap(); - - let est = FixedOffset::west_opt(5 * 3600).unwrap(); - let result = super::format_at_tz(utc, &est).unwrap(); - assert_eq!(result, "2026-05-31 05:33 AM -05:00"); - - let ist = FixedOffset::east_opt(5 * 3600 + 1800).unwrap(); - let result = super::format_at_tz(utc, &ist).unwrap(); - assert_eq!(result, "2026-05-31 04:03 PM +05:30"); -} - -#[test] -fn format_at_tz_returns_none_for_utc() { - let utc = super::parse_email_date("2026-05-31T10:33:00Z").unwrap(); - let utc_tz = chrono::FixedOffset::east_opt(0).unwrap(); - assert!(super::format_at_tz(utc, &utc_tz).is_none()); -} - -#[test] -fn apply_response_level_markdown_stashes_per_message_field() { - let mut data = json!({ - "messages": [ - {"messageId": "m1", "subject": "Hello"}, - {"messageId": "m2", "subject": "World"}, - ] - }); - let top_md = "## Hello\nbody A — link https://gh.io/abc\n---\n## World\nbody B"; - super::apply_response_level_markdown(&mut data, top_md); - let m1 = data["messages"][0]["markdownFormatted"].as_str().unwrap(); - let m2 = data["messages"][1]["markdownFormatted"].as_str().unwrap(); - assert!(m1.contains("Hello")); - assert!( - m1.contains("https://gh.io/abc"), - "shortened URL must survive" - ); - assert!(m2.contains("World")); - assert!(!m1.contains("World"), "no cross-message bleed"); -} diff --git a/crates/tinymemory-integrations/src/sources/composio/linear/mod.rs b/crates/tinymemory-integrations/src/sources/composio/linear/mod.rs deleted file mode 100644 index ea52092a..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/linear/mod.rs +++ /dev/null @@ -1,79 +0,0 @@ -//! Linear host normalization helpers — result extraction and issue-title and -//! timestamp extraction. -//! -//! Linear's GraphQL API (and therefore Composio's wrapping of it) returns -//! connection-style lists (`{ nodes: [...], pageInfo: {...} }`) at the top -//! level or nested under `data`. The functions here walk the union of -//! common shapes so the provider does not have to branch per Composio -//! envelope variant. - -use serde_json::Value; - -use super::fields::pick_str; - -/// Walk the Composio response envelope for Linear issue list results. -/// -/// Linear's list endpoints return `{ nodes: [...] }` or -/// `{ issues: { nodes: [...] } }` shapes; Composio may re-wrap the -/// upstream payload under `data` or `data.data`. We probe each shape -/// in order and return the first array we find. -pub fn extract_issues(data: &Value) -> Vec { - let candidates = [ - data.pointer("/data/nodes"), - data.pointer("/nodes"), - data.pointer("/data/issues/nodes"), - data.pointer("/issues/nodes"), - data.pointer("/data/data/nodes"), - data.pointer("/data/data/issues/nodes"), - data.pointer("/data/results"), - data.pointer("/results"), - data.pointer("/data/items"), - data.pointer("/items"), - ]; - for cand in candidates.into_iter().flatten() { - if let Some(arr) = cand.as_array() { - return arr.clone(); - } - } - Vec::new() -} - -/// Extract a human-readable title from a Linear issue object. -/// -/// Linear issues store the name at `title` (or `data.title` after -/// Composio envelope wrapping). Falls back to `name` / `identifier` -/// so the chunk remains identifiable even for unusual response shapes. -pub fn extract_issue_title(issue: &Value) -> Option { - pick_str( - issue, - &[ - "title", - "data.title", - "name", - "data.name", - "identifier", - "data.identifier", - ], - ) -} - -/// Extract a stable cursor timestamp from a Linear issue object. -/// -/// Linear uses ISO-8601 strings for timestamps (`updatedAt`). We keep -/// the value as a string so lexicographic comparison against the stored -/// cursor is valid. -pub fn extract_issue_updated(issue: &Value) -> Option { - pick_str( - issue, - &[ - "updatedAt", - "data.updatedAt", - "updated_at", - "data.updated_at", - ], - ) -} - -#[cfg(test)] -#[path = "mod_tests.rs"] -mod tests; diff --git a/crates/tinymemory-integrations/src/sources/composio/linear/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/linear/mod_tests.rs deleted file mode 100644 index c0acff63..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/linear/mod_tests.rs +++ /dev/null @@ -1,93 +0,0 @@ -//! Tests for the Linear normaliser. - -use super::*; -use serde_json::json; - -// ── extract_issues ─────────────────────────────────────────────── - -#[test] -fn extract_issues_from_data_nodes() { - let data = json!({ "data": { "nodes": [{"id": "i1"}, {"id": "i2"}] } }); - assert_eq!(extract_issues(&data).len(), 2); -} - -#[test] -fn extract_issues_from_top_level_nodes() { - let data = json!({ "nodes": [{"id": "i3"}] }); - assert_eq!(extract_issues(&data).len(), 1); -} - -#[test] -fn extract_issues_from_data_issues_nodes() { - let data = - json!({ "data": { "issues": { "nodes": [{"id": "i4"}, {"id": "i5"}, {"id": "i6"}] } } }); - assert_eq!(extract_issues(&data).len(), 3); -} - -#[test] -fn extract_issues_from_top_level_issues_nodes() { - let data = json!({ "issues": { "nodes": [{"id": "i7"}] } }); - assert_eq!(extract_issues(&data).len(), 1); -} - -#[test] -fn extract_issues_from_doubly_nested_issues_nodes() { - let data = - json!({ "data": { "data": { "issues": { "nodes": [{"id": "i8"}, {"id": "i9"}] } } } }); - assert_eq!(extract_issues(&data).len(), 2); -} - -#[test] -fn extract_issues_from_results() { - let data = json!({ "results": [{"id": "i7"}] }); - assert_eq!(extract_issues(&data).len(), 1); -} - -#[test] -fn extract_issues_empty_when_missing() { - let data = json!({ "foo": "bar" }); - assert!(extract_issues(&data).is_empty()); -} - -// ── extract_issue_title ────────────────────────────────────────── - -#[test] -fn extract_issue_title_from_title_field() { - let issue = json!({ "id": "i1", "title": "Fix the login bug" }); - assert_eq!( - extract_issue_title(&issue), - Some("Fix the login bug".into()) - ); -} - -#[test] -fn extract_issue_title_falls_back_to_wrapped_data() { - let issue = json!({ "data": { "title": "Wrapped issue" } }); - assert_eq!(extract_issue_title(&issue), Some("Wrapped issue".into())); -} - -#[test] -fn extract_issue_title_falls_back_to_identifier() { - let issue = json!({ "identifier": "ENG-42" }); - assert_eq!(extract_issue_title(&issue), Some("ENG-42".into())); -} - -// ── extract_issue_updated ──────────────────────────────────────── - -#[test] -fn extract_issue_updated_from_updated_at() { - let issue = json!({ "updatedAt": "2026-03-01T12:00:00.000Z" }); - assert_eq!( - extract_issue_updated(&issue), - Some("2026-03-01T12:00:00.000Z".to_string()) - ); -} - -#[test] -fn extract_issue_updated_falls_back_to_snake_case() { - let issue = json!({ "data": { "updated_at": "2026-01-15T08:30:00.000Z" } }); - assert_eq!( - extract_issue_updated(&issue), - Some("2026-01-15T08:30:00.000Z".to_string()) - ); -} diff --git a/crates/tinymemory-integrations/src/sources/composio/mod.rs b/crates/tinymemory-integrations/src/sources/composio/mod.rs deleted file mode 100644 index 4df32030..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/mod.rs +++ /dev/null @@ -1,35 +0,0 @@ -//! Composio toolkit payloads: normalisers and the mapping to `StoreItem`s. -//! -//! A host runs Composio actions with its own credentials and hands the raw -//! responses here. Two layers turn them into memory: -//! -//! 1. **Normalisers**, one module per toolkit, are pure -//! `serde_json::Value` transforms: they walk Composio's envelope variants -//! and pull out the tasks, issues, pages or messages -//! ([`clickup`], [`github`], [`linear`], [`notion`]), or rewrite a verbose -//! response into a slim one in place ([`gmail_post_process`], -//! [`slack_post_process`]). [`fields`] holds the path lookup they share. -//! 2. [`normalise_payload`] turns one (post-processed) response into -//! [`ComposioDocument`]s, and [`payload_items`] turns those into -//! [`StoreItem::Document`](tinymemory_api::StoreItem::Document)s with -//! `source.kind = Composio`, `source.id` = the connection or source id, -//! `url` and `observed_at` where the payload has them, and -//! `tags = [toolkit]`. -//! -//! Nothing here holds a credential, opens a socket or decides when to sync. -//! -//! One caveat on "pure": [`gmail_post_process::format_email_local_time`] -//! renders in `chrono::Local`, so it reads the host's timezone. The raw UTC -//! fields are preserved alongside, so ordering and identity stay UTC-based. - -pub mod clickup; -pub mod fields; -pub mod github; -pub mod gmail_post_process; -pub mod linear; -pub mod notion; -pub mod slack_post_process; - -mod documents; - -pub use documents::{ComposioDocument, normalise_payload, payload_items}; diff --git a/crates/tinymemory-integrations/src/sources/composio/notion/mod.rs b/crates/tinymemory-integrations/src/sources/composio/notion/mod.rs deleted file mode 100644 index 00c1df50..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/notion/mod.rs +++ /dev/null @@ -1,88 +0,0 @@ -//! Notion host normalization helpers — result extraction, page markdown and -//! page title extraction. - -use serde_json::Value; - -use super::fields::pick_str; - -/// Walk the Composio response envelope for Notion page results. -pub fn extract_results(data: &Value) -> Vec { - let candidates = [ - data.pointer("/data/results"), - data.pointer("/results"), - data.pointer("/data/data/results"), - data.pointer("/data/items"), - data.pointer("/items"), - ]; - for cand in candidates.into_iter().flatten() { - if let Some(arr) = cand.as_array() { - return arr.clone(); - } - } - Vec::new() -} - -/// Extract the rendered page body markdown from a `NOTION_GET_PAGE_MARKDOWN` -/// response. Composio wraps action output in varying envelope shapes, so we -/// try the common locations tolerantly and return the first non-empty string. -/// Returns `None` if no markdown field is found (caller falls back to the -/// metadata-only body and logs the raw shape for diagnosis). -pub fn extract_page_markdown(data: &Value) -> Option { - const PATHS: &[&str] = &[ - "/markdown", - "/data/markdown", - "/data/response_data/markdown", - "/response_data/markdown", - "/data/content", - "/content", - "/data/markdown_content", - "/markdown_content", - "/text", - "/data/text", - ]; - for p in PATHS { - if let Some(s) = data.pointer(p).and_then(Value::as_str) - && !s.trim().is_empty() - { - return Some(s.to_string()); - } - } - None -} - -/// Try to extract a human-readable title from a Notion page object. -/// -/// Notion pages store the title in `properties.title` or -/// `properties.Name.title[0].plain_text`. We try several shapes. -pub fn extract_page_title(page: &Value) -> Option { - // Try the common `properties.title.title[0].plain_text` shape. - let props = page - .get("properties") - .or_else(|| page.get("data")?.get("properties")); - if let Some(props) = props { - // Walk all properties looking for a "title" type field. - if let Some(obj) = props.as_object() { - for (_key, val) in obj { - if val.get("type").and_then(Value::as_str) == Some("title") - && let Some(arr) = val.get("title").and_then(Value::as_array) - { - let text: String = arr - .iter() - .filter_map(|t| t.get("plain_text").and_then(Value::as_str)) - .collect::>() - .join(""); - if !text.is_empty() { - return Some(text); - } - } - } - } - } - - // Fallback: top-level "title" field (some Composio shapes). - pick_str(page, &["title", "data.title", "name", "data.name"]) -} - -#[cfg(test)] -#[path = "mod_tests.rs"] -mod tests; diff --git a/crates/tinymemory-integrations/src/sources/composio/notion/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/notion/mod_tests.rs deleted file mode 100644 index ab2e79bc..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/notion/mod_tests.rs +++ /dev/null @@ -1,111 +0,0 @@ -//! Tests for the Notion normaliser. - -use super::*; -use serde_json::json; - -#[test] -fn extract_results_from_data_results() { - let data = json!({"data": {"results": [{"id": "page1"}]}}); - let results = extract_results(&data); - assert_eq!(results.len(), 1); -} - -#[test] -fn extract_page_markdown_reads_top_level_field() { - // Matches the live GET_PAGE_MARKDOWN envelope observed empirically: - // {id, markdown, object, request_id, truncated, unknown_block_ids}. - let data = json!({ - "id": "p1", - "markdown": "# Heading\n\nbody text", - "object": "page", - "truncated": false, - }); - assert_eq!( - extract_page_markdown(&data).as_deref(), - Some("# Heading\n\nbody text") - ); -} - -#[test] -fn extract_page_markdown_reads_nested_envelope() { - let data = json!({ "data": { "markdown": "nested body" } }); - assert_eq!(extract_page_markdown(&data).as_deref(), Some("nested body")); -} - -#[test] -fn extract_page_markdown_none_for_empty_or_missing() { - // Empty markdown (a DB row with no body blocks) → None → metadata-only. - assert_eq!(extract_page_markdown(&json!({ "markdown": "" })), None); - assert_eq!(extract_page_markdown(&json!({ "markdown": " " })), None); - // No markdown field at all → None. - assert_eq!(extract_page_markdown(&json!({ "id": "p1" })), None); -} - -#[test] -fn extract_results_from_top_level() { - let data = json!({"results": [{"id": "a"}, {"id": "b"}]}); - let results = extract_results(&data); - assert_eq!(results.len(), 2); -} - -#[test] -fn extract_results_from_data_items() { - let data = json!({"data": {"items": [{"id": "x"}]}}); - let results = extract_results(&data); - assert_eq!(results.len(), 1); -} - -#[test] -fn extract_results_empty_when_no_match() { - let data = json!({"foo": "bar"}); - assert!(extract_results(&data).is_empty()); -} - -#[test] -fn extract_page_title_from_properties_title_type() { - let page = json!({ - "properties": { - "Name": { - "type": "title", - "title": [{"plain_text": "Hello"}, {"plain_text": " World"}] - } - } - }); - assert_eq!(extract_page_title(&page), Some("Hello World".into())); -} - -#[test] -fn extract_page_title_from_nested_data_properties() { - let page = json!({ - "data": { - "properties": { - "Title": { - "type": "title", - "title": [{"plain_text": "My Page"}] - } - } - } - }); - assert_eq!(extract_page_title(&page), Some("My Page".into())); -} - -#[test] -fn extract_page_title_fallback_to_top_level_title() { - let page = json!({"title": "Fallback Title"}); - assert_eq!(extract_page_title(&page), Some("Fallback Title".into())); -} - -#[test] -fn extract_page_title_none_when_empty() { - let page = json!({"properties": {"Name": {"type": "title", "title": []}}}); - // Empty title array means no text - assert!( - extract_page_title(&page).is_none() || extract_page_title(&page) == Some(String::new()) - ); -} - -#[test] -fn extract_page_title_none_when_no_title_field() { - let page = json!({"id": "123"}); - assert!(extract_page_title(&page).is_none()); -} diff --git a/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod.rs b/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod.rs deleted file mode 100644 index 764b885a..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod.rs +++ /dev/null @@ -1,324 +0,0 @@ -//! Slack-specific post-processing of Composio action responses. -//! -//! Composio's Slack responses are verbose API envelopes. This module -//! rewrites each supported action's response into a slim, stable shape -//! that the ingest pipeline and enrichers can consume without walking -//! Composio's unstable nested envelopes. -//! -//! ## Supported slugs -//! -//! - `SLACK_FETCH_CONVERSATION_HISTORY` — reshapes into top-level -//! `messages[]` with `{ ts, user, text, thread_ts, channel_id }`. -//! Empty-text messages are dropped. `channel_id` is absent here (it's -//! in the request, not the response); the caller injects it via the -//! enricher in the host's `SlackSyncPipeline`. -//! -//! - `SLACK_LIST_CONVERSATIONS` — reshapes into top-level `channels[]` -//! with `{ id, name, is_private }` per channel. Entries with an empty -//! id are dropped. -//! -//! - `SLACK_SEARCH_MESSAGES` — reshapes `messages.matches[]` (possibly -//! nested) into top-level `messages[]` with `{ ts, user, text, -//! thread_ts, channel_id }`. `channel_id` is pulled from each match's -//! `channel.id` field. `paging.pages` is preserved at top-level for -//! caller pagination. -//! -//! ## Design note: user-id resolution is NOT here -//! -//! `SlackUsers` is a per-sync cache built from a separate API call — -//! not a function of any individual response. Resolving user ids -//! happens in the host's `SlackSyncPipeline` (the enricher layer), keeping -//! this module purely data-shape–oriented. -//! This matches Gmail's pattern of "post_process is data-only". -//! -//! Unknown slugs are silently no-ops so new Composio actions don't -//! break the provider. - -use serde_json::{Map, Value}; - -/// Entry point a host calls on each Slack action response (the slug names -/// the action) before handing it to `normalise_payload`. -/// -/// Dispatches on the Composio action slug and rewrites `data` in place. -/// Unknown slugs are silently ignored. -pub fn post_process(slug: &str, _arguments: Option<&Value>, data: &mut Value) { - log::debug!("[composio:slack][post-process] slug={slug}"); - match slug { - "SLACK_FETCH_CONVERSATION_HISTORY" => reshape_fetch_history(data), - "SLACK_LIST_CONVERSATIONS" => reshape_list_conversations(data), - "SLACK_SEARCH_MESSAGES" => reshape_search_messages(data), - _ => { - log::debug!("[composio:slack][post-process] unknown slug={slug}, passing through"); - } - } -} - -// ─── SLACK_FETCH_CONVERSATION_HISTORY ────────────────────────────────────── - -/// Rewrite a `SLACK_FETCH_CONVERSATION_HISTORY` response in place. -/// -/// Walks possible nested envelopes (`/data/messages`, `/messages`, -/// `/data/data/messages`) to find the raw messages array, drops messages -/// with empty `text`, and emits a slim `{ ts, user, text, thread_ts }` -/// shape under a top-level `messages[]` key. The consumed nested array is -/// removed from the payload so the raw verbose rows don't linger alongside -/// the slim copy. The caller injects `channel_id` via -/// the host's Slack sync pipeline. -fn reshape_fetch_history(data: &mut Value) { - let arr = take_array( - data, - &["/data/messages", "/messages", "/data/data/messages"], - 0, - ); - let slim: Vec = arr.into_iter().filter_map(slim_history_message).collect(); - with_object(data, |obj| { - obj.insert("messages".to_string(), Value::Array(slim)); - }); - log::debug!("[composio:slack][post-process] SLACK_FETCH_CONVERSATION_HISTORY reshaped"); -} - -fn slim_history_message(raw: Value) -> Option { - let text = raw - .get("text") - .and_then(|v| v.as_str()) - .unwrap_or("") - .trim(); - if text.is_empty() { - return None; - } - let mut out = Map::new(); - // `ts` is required: without it a caller can neither cursor nor archive. - out.insert("ts".into(), raw.get("ts")?.clone()); - if let Some(user) = raw.get("user").or_else(|| raw.get("bot_id")) { - out.insert("user".into(), user.clone()); - } - out.insert("text".into(), Value::String(text.to_string())); - if let Some(thread_ts) = raw.get("thread_ts") { - out.insert("thread_ts".into(), thread_ts.clone()); - } - if let Some(permalink) = raw.get("permalink") { - out.insert("permalink".into(), permalink.clone()); - } - Some(Value::Object(out)) -} - -/// Find the first array at any of `candidates`, remove that field (plus -/// `envelope_depth` ancestor object envelopes) from `data`, and return the -/// array. Removing the consumed nested payload keeps the reshaped output from -/// carrying duplicate raw rows. -fn take_array(data: &mut Value, candidates: &[&str], envelope_depth: usize) -> Vec { - for path in candidates { - let arr = match data.pointer(path).and_then(|v| v.as_array().cloned()) { - Some(a) => a, - None => continue, - }; - let mut remove_path = path.to_string(); - for _ in 0..envelope_depth { - remove_path = match remove_path.rsplit_once('/') { - Some((parent, _)) => parent.to_string(), - None => break, - }; - } - remove_nested(data, &remove_path); - return arr; - } - Vec::new() -} - -/// Remove the field at `path` from `data`, pruning any ancestor object that -/// the removal left empty so a consumed `data` envelope disappears entirely -/// instead of lingering as `{}`. -fn remove_nested(data: &mut Value, path: &str) { - let segments: Vec<&str> = path - .trim_start_matches('/') - .split('/') - .filter(|s| !s.is_empty()) - .collect(); - if segments.is_empty() { - return; - } - - // Remove the leaf field. - let mut current = &mut *data; - for seg in &segments[..segments.len() - 1] { - current = match current.get_mut(*seg) { - Some(next) => next, - None => return, - }; - } - if let Value::Object(map) = current { - map.remove(segments[segments.len() - 1]); - } - - // Prune empty object ancestors, deepest first. - for depth in (0..segments.len().saturating_sub(1)).rev() { - // Re-walk to the object at `segments[..=depth]`. - let mut ancestor = &mut *data; - for seg in &segments[..=depth] { - ancestor = match ancestor.get_mut(*seg) { - Some(next) => next, - None => return, - }; - } - if !matches!(ancestor, Value::Object(m) if m.is_empty()) { - break; - } - // Remove it from its parent (`segments[..depth]`). For `depth == 0` - // the parent is the top-level object, so an emptied `data` envelope - // key disappears entirely. - let mut parent = &mut *data; - for seg in &segments[..depth] { - parent = match parent.get_mut(*seg) { - Some(next) => next, - None => return, - }; - } - if let Value::Object(map) = parent { - map.remove(segments[depth]); - } - } -} - -// ─── SLACK_LIST_CONVERSATIONS ─────────────────────────────────────────────── - -/// Rewrite a `SLACK_LIST_CONVERSATIONS` response in place. -/// -/// Reshapes into a top-level `channels[]` with `{ id, name, is_private }` -/// per channel; entries with an empty id are dropped. -fn reshape_list_conversations(data: &mut Value) { - let arr = take_array( - data, - &[ - "/data/channels", - "/channels", - "/data/data/channels", - "/data/conversations", - "/conversations", - ], - 0, - ); - - let slim: Vec = arr.into_iter().filter_map(slim_channel).collect(); - with_object(data, |obj| { - obj.insert("channels".to_string(), Value::Array(slim)); - }); - log::debug!("[composio:slack][post-process] SLACK_LIST_CONVERSATIONS reshaped"); -} - -fn slim_channel(raw: Value) -> Option { - let id = raw.get("id").and_then(|v| v.as_str()).unwrap_or("").trim(); - if id.is_empty() { - return None; - } - let name = raw - .get("name") - .and_then(|v| v.as_str()) - .unwrap_or(id) - .trim(); - let is_private = raw - .get("is_private") - .and_then(|v| v.as_bool()) - .unwrap_or(false); - Some(Value::Object({ - let mut m = Map::new(); - m.insert("id".into(), Value::String(id.to_string())); - m.insert("name".into(), Value::String(name.to_string())); - m.insert("is_private".into(), Value::Bool(is_private)); - m - })) -} - -// ─── SLACK_SEARCH_MESSAGES ────────────────────────────────────────────────── - -/// Rewrite a `SLACK_SEARCH_MESSAGES` response in place. -/// -/// Reshapes `messages.matches[]` (possibly nested under one or two -/// `data` envelopes) into top-level `messages[]`. `channel_id` is pulled -/// from each match's `channel.id` field. `paging.pages` is preserved at -/// top-level under `pages` for the caller to drive pagination. -fn reshape_search_messages(data: &mut Value) { - // Preserve paging info before mutating data (take_array below removes the - // envelope that carries it). - let pages = [ - data.pointer("/data/messages/paging/pages"), - data.pointer("/messages/paging/pages"), - data.pointer("/data/data/messages/paging/pages"), - ] - .into_iter() - .flatten() - .find_map(|v| v.as_u64()) - .unwrap_or(1); - - // Envelope depth 1 removes the `messages` object (matches + paging) that - // held the consumed rows, not just the `matches` array. - let arr = take_array( - data, - &[ - "/data/messages/matches", - "/messages/matches", - "/data/data/messages/matches", - ], - 1, - ); - - let slim: Vec = arr.into_iter().filter_map(slim_search_match).collect(); - with_object(data, |obj| { - obj.insert("messages".to_string(), Value::Array(slim)); - obj.insert("pages".to_string(), Value::Number(pages.into())); - }); - log::debug!("[composio:slack][post-process] SLACK_SEARCH_MESSAGES reshaped"); -} - -fn slim_search_match(raw: Value) -> Option { - let text = raw - .get("text") - .and_then(|v| v.as_str()) - .unwrap_or("") - .trim(); - if text.is_empty() { - return None; - } - let ts = raw.get("ts")?; - let channel_id = raw - .pointer("/channel/id") - .and_then(|v| v.as_str()) - .unwrap_or("") - .trim(); - - let mut out = Map::new(); - out.insert("ts".into(), ts.clone()); - if let Some(user) = raw.get("user").or_else(|| raw.get("bot_id")) { - out.insert("user".into(), user.clone()); - } - out.insert("text".into(), Value::String(text.to_string())); - if let Some(thread_ts) = raw.get("thread_ts") { - out.insert("thread_ts".into(), thread_ts.clone()); - } - if !channel_id.is_empty() { - out.insert("channel_id".into(), Value::String(channel_id.to_string())); - } - if let Some(permalink) = raw.get("permalink") { - out.insert("permalink".into(), permalink.clone()); - } - Some(Value::Object(out)) -} - -// ─── Helpers ──────────────────────────────────────────────────────────────── - -/// Edit `data` as a JSON object, replacing it with an empty object first if -/// it is not one. -/// -/// Takes the value out, edits the map, and puts it back, so there is no -/// "re-borrow as an object" step that would need an `expect`. -fn with_object(data: &mut Value, edit: impl FnOnce(&mut Map)) { - let mut map = match std::mem::take(data) { - Value::Object(map) => map, - _ => Map::new(), - }; - edit(&mut map); - *data = Value::Object(map); -} - -#[cfg(test)] -#[path = "mod_tests.rs"] -mod tests; diff --git a/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod_tests.rs deleted file mode 100644 index 73a6702b..00000000 --- a/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod_tests.rs +++ /dev/null @@ -1,306 +0,0 @@ -//! Tests for the Slack post-processor. - -use super::*; -use serde_json::json; - -// ─── SLACK_FETCH_CONVERSATION_HISTORY ───────────────────────────────────── - -#[test] -fn history_reshapes_top_level_messages() { - let mut data = json!({ - "messages": [ - { "ts": "1714003200.000100", "user": "U1", "text": "hello" }, - { "ts": "1714003300.000200", "user": "U2", "text": "world", "thread_ts": "1714003200.0" }, - { "ts": "1714003400.000300", "user": "U3", "text": " " }, // dropped: empty text - ], - "response_metadata": { "next_cursor": "abc" } - }); - post_process("SLACK_FETCH_CONVERSATION_HISTORY", None, &mut data); - - let msgs = data["messages"].as_array().unwrap(); - assert_eq!(msgs.len(), 2, "empty-text message must be dropped"); - assert_eq!(msgs[0]["ts"], "1714003200.000100"); - assert_eq!(msgs[0]["user"], "U1"); - assert_eq!(msgs[0]["text"], "hello"); - assert!(msgs[0].get("thread_ts").is_none()); - assert_eq!(msgs[1]["thread_ts"], "1714003200.0"); -} - -#[test] -fn history_reshapes_nested_data_envelope() { - let mut data = json!({ - "data": { - "messages": [ - { "ts": "1714003200.0", "user": "U1", "text": "hi" } - ] - } - }); - post_process("SLACK_FETCH_CONVERSATION_HISTORY", None, &mut data); - let msgs = data["messages"].as_array().unwrap(); - assert_eq!(msgs.len(), 1); - assert_eq!(msgs[0]["text"], "hi"); -} - -#[test] -fn history_reshapes_doubly_nested_envelope() { - let mut data = json!({ - "data": { - "data": { - "messages": [ - { "ts": "1714003200.0", "user": "U1", "text": "deep" } - ] - } - } - }); - post_process("SLACK_FETCH_CONVERSATION_HISTORY", None, &mut data); - let msgs = data["messages"].as_array().unwrap(); - assert_eq!(msgs.len(), 1); - assert_eq!(msgs[0]["text"], "deep"); -} - -#[test] -fn history_drops_message_without_ts() { - let mut data = json!({ - "messages": [ - { "user": "U1", "text": "no timestamp" }, - { "ts": "1714003200.0", "user": "U2", "text": "has ts" }, - ] - }); - post_process("SLACK_FETCH_CONVERSATION_HISTORY", None, &mut data); - let msgs = data["messages"].as_array().unwrap(); - assert_eq!(msgs.len(), 1); - assert_eq!(msgs[0]["text"], "has ts"); -} - -#[test] -fn history_removes_nested_envelope_after_reshape() { - let mut data = json!({ - "data": { - "messages": [ - { "ts": "1714003200.0", "user": "U1", "text": "hi" } - ] - } - }); - post_process("SLACK_FETCH_CONVERSATION_HISTORY", None, &mut data); - - let msgs = data["messages"].as_array().unwrap(); - assert_eq!(msgs.len(), 1); - assert_eq!(msgs[0]["text"], "hi"); - assert!( - data.pointer("/data").is_none(), - "consumed `data.messages` envelope must be removed, got: {data}" - ); -} - -// ─── SLACK_LIST_CONVERSATIONS ───────────────────────────────────────────── - -#[test] -fn list_conversations_reshapes_channels() { - let mut data = json!({ - "data": { - "channels": [ - { "id": "C1", "name": "eng", "is_private": false, "extra": "noise" }, - { "id": "G1", "name": "ops", "is_private": true }, - { "id": "", "name": "empty-id" }, // dropped - ] - } - }); - post_process("SLACK_LIST_CONVERSATIONS", None, &mut data); - let channels = data["channels"].as_array().unwrap(); - assert_eq!(channels.len(), 2, "empty-id entry must be dropped"); - assert_eq!(channels[0]["id"], "C1"); - assert_eq!(channels[0]["name"], "eng"); - assert_eq!(channels[0]["is_private"], false); - assert!( - channels[0].get("extra").is_none(), - "noise fields must be removed" - ); - assert_eq!(channels[1]["id"], "G1"); - assert_eq!(channels[1]["is_private"], true); -} - -#[test] -fn list_conversations_falls_back_to_conversations_key() { - let mut data = json!({ - "conversations": [ - { "id": "C2", "name": "dev", "is_private": false } - ] - }); - post_process("SLACK_LIST_CONVERSATIONS", None, &mut data); - let channels = data["channels"].as_array().unwrap(); - assert_eq!(channels.len(), 1); - assert_eq!(channels[0]["id"], "C2"); - assert!( - data.pointer("/conversations").is_none(), - "consumed `conversations` field must be removed" - ); -} - -// ─── SLACK_SEARCH_MESSAGES ──────────────────────────────────────────────── - -#[test] -fn search_messages_reshapes_matches() { - let mut data = json!({ - "messages": { - "matches": [ - { - "ts": "1714003200.0", - "user": "U1", - "text": "hello from search", - "channel": { "id": "C1" } - }, - { - "ts": "1714003300.0", - "user": "U2", - "text": " ", // dropped: whitespace only - "channel": { "id": "C1" } - }, - ], - "paging": { "pages": 3 } - } - }); - post_process("SLACK_SEARCH_MESSAGES", None, &mut data); - let msgs = data["messages"].as_array().unwrap(); - assert_eq!(msgs.len(), 1, "empty-text match must be dropped"); - assert_eq!(msgs[0]["ts"], "1714003200.0"); - assert_eq!(msgs[0]["text"], "hello from search"); - assert_eq!(msgs[0]["channel_id"], "C1"); - assert_eq!(data["pages"], 3, "paging.pages must be preserved"); -} - -#[test] -fn search_messages_nested_data_envelope() { - let mut data = json!({ - "data": { - "messages": { - "matches": [ - { "ts": "1714003200.0", "user": "U1", "text": "nested", "channel": { "id": "C2" } } - ], - "paging": { "pages": 1 } - } - } - }); - post_process("SLACK_SEARCH_MESSAGES", None, &mut data); - let msgs = data["messages"].as_array().unwrap(); - assert_eq!(msgs.len(), 1); - assert_eq!(msgs[0]["channel_id"], "C2"); - assert_eq!(data["pages"], 1_u64); -} - -#[test] -fn search_messages_no_matches_emits_empty_array() { - let mut data = json!({ "messages": { "matches": [] } }); - post_process("SLACK_SEARCH_MESSAGES", None, &mut data); - let msgs = data["messages"].as_array().unwrap(); - assert!(msgs.is_empty()); -} - -#[test] -fn search_messages_removes_nested_envelope_after_reshape() { - let mut data = json!({ - "data": { - "messages": { - "matches": [ - { "ts": "1714003200.0", "user": "U1", "text": "nested", "channel": { "id": "C2" } } - ], - "paging": { "pages": 1 } - } - } - }); - post_process("SLACK_SEARCH_MESSAGES", None, &mut data); - - let msgs = data["messages"].as_array().unwrap(); - assert_eq!(msgs.len(), 1); - assert_eq!(msgs[0]["channel_id"], "C2"); - assert_eq!(data["pages"], 1_u64); - assert!( - data.pointer("/data").is_none(), - "consumed `data.messages` envelope must be removed, got: {data}" - ); -} - -#[test] -fn search_messages_doubly_nested_paging_preserved() { - let mut data = json!({ - "data": { - "data": { - "messages": { - "matches": [ - { "ts": "1714003200.0", "user": "U1", "text": "deep", "channel": { "id": "C3" } } - ], - "paging": { "pages": 4 } - } - } - } - }); - post_process("SLACK_SEARCH_MESSAGES", None, &mut data); - - let msgs = data["messages"].as_array().unwrap(); - assert_eq!(msgs.len(), 1); - assert_eq!(msgs[0]["text"], "deep"); - assert_eq!( - data["pages"], 4_u64, - "doubly-nested paging must be preserved" - ); - assert!( - data.pointer("/data").is_none(), - "consumed `data.data.messages` envelope must be removed, got: {data}" - ); -} - -// ─── Unknown slug ───────────────────────────────────────────────────────── - -#[test] -fn unknown_slug_is_noop() { - let mut data = json!({ "foo": "bar" }); - let original = data.clone(); - post_process("SLACK_SEND_MESSAGE", None, &mut data); - assert_eq!(data, original, "unknown slug must not mutate data"); -} - -#[test] -fn malformed_history_payload_becomes_an_empty_stable_shape() { - for mut data in [json!(null), json!([]), json!({"messages": "wrong"})] { - post_process("SLACK_FETCH_CONVERSATION_HISTORY", None, &mut data); - assert_eq!(data, json!({"messages": []})); - } -} - -#[test] -fn malformed_channel_rows_are_dropped_and_defaults_are_stable() { - let mut data = json!({ - "channels": [ - null, - "not an object", - {"id": 7, "name": "numeric id"}, - {"id": " C1 ", "name": 99, "is_private": "yes"} - ] - }); - post_process("SLACK_LIST_CONVERSATIONS", None, &mut data); - assert_eq!( - data, - json!({"channels": [{"id":"C1", "name":"C1", "is_private":false}]}) - ); -} - -#[test] -fn malformed_search_rows_and_page_counts_fall_back_safely() { - let mut data = json!({ - "messages": { - "matches": [ - null, - {"ts":"1.0", "text":"no channel"}, - {"channel":{"id":"C1"}, "text":"no timestamp"}, - {"ts":"2.0", "channel":{"id":"C1"}, "text":" valid "} - ], - "paging": {"pages":"many"} - } - }); - post_process("SLACK_SEARCH_MESSAGES", None, &mut data); - assert_eq!(data["pages"], 1); - let messages = data["messages"].as_array().unwrap(); - assert_eq!(messages.len(), 2); - assert_eq!(messages[0]["text"], "no channel"); - assert!(messages[0].get("channel_id").is_none()); - assert_eq!(messages[1]["text"], "valid"); -} diff --git a/crates/tinymemory-integrations/src/sources/readers/composio/mod.rs b/crates/tinymemory-integrations/src/sources/readers/composio/mod.rs deleted file mode 100644 index 5d0c3d84..00000000 --- a/crates/tinymemory-integrations/src/sources/readers/composio/mod.rs +++ /dev/null @@ -1,78 +0,0 @@ -//! Composio source reader — a placeholder over the provider pipeline. -//! -//! Composio data does not arrive item by item: the host runs toolkit actions -//! with its credentials and hands the responses to [`crate::sources::composio`], which -//! normalises them and maps them to `StoreItem`s. For a Composio source, -//! `list_items` returns the connection as one sync target and `read_item` -//! describes that pipeline. The reader exists so `reader_for_request` can hand -//! out a reader for every source kind uniformly. - -use std::path::Path; - -use async_trait::async_trait; - -use super::SourceReader; -use crate::sources::error::Result; -use crate::sources::types::{ - ContentType, MemorySourceEntry, SourceContent, SourceItem, SourceKind, -}; - -/// Lists a Composio connection as a single sync target. -/// -/// Composio data arrives through the provider sync pipeline rather than -/// item-by-item, so `read_item` returns a description of that rather than -/// content. The reader exists so `reader_for_request` can serve every source -/// kind uniformly. -#[derive(Debug, Clone, Copy, Default)] -pub struct ComposioReader; - -#[async_trait] -impl SourceReader for ComposioReader { - fn kind(&self) -> SourceKind { - SourceKind::Composio - } - - async fn list_items( - &self, - source: &MemorySourceEntry, - _workspace: &Path, - ) -> Result> { - let toolkit = source.toolkit.as_deref().unwrap_or("unknown"); - let connection_id = source.connection_id.as_deref().unwrap_or("unknown"); - - log::debug!( - "[memory_sources:composio] list_items toolkit={toolkit} connection_id={connection_id}" - ); - - Ok(vec![SourceItem { - id: connection_id.to_string(), - title: format!("{toolkit} connection"), - updated_at_ms: None, - }]) - } - - async fn read_item( - &self, - source: &MemorySourceEntry, - item_id: &str, - _workspace: &Path, - ) -> Result { - let toolkit = source.toolkit.as_deref().unwrap_or("unknown"); - Ok(SourceContent { - id: item_id.to_string(), - title: format!("{toolkit} sync data"), - body: format!( - "Composio {toolkit} data is synced via the provider sync pipeline, not read item-by-item." - ), - content_type: ContentType::Plaintext, - metadata: serde_json::json!({ - "toolkit": toolkit, - "connection_id": source.connection_id, - }), - }) - } -} - -#[cfg(test)] -#[path = "mod_tests.rs"] -mod tests; diff --git a/crates/tinymemory-integrations/src/sources/readers/composio/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/composio/mod_tests.rs deleted file mode 100644 index d139ff15..00000000 --- a/crates/tinymemory-integrations/src/sources/readers/composio/mod_tests.rs +++ /dev/null @@ -1,51 +0,0 @@ -//! Tests for the surrounding module. - -use super::*; -use std::path::Path; - -fn test_source() -> MemorySourceEntry { - MemorySourceEntry { - id: "src_1".into(), - kind: SourceKind::Composio, - label: "Gmail".into(), - enabled: true, - toolkit: Some("gmail".into()), - connection_id: Some("cmp_123".into()), - path: None, - glob: None, - url: None, - branch: None, - paths: Vec::new(), - max_items: None, - max_commits: None, - max_issues: None, - max_prs: None, - selector: None, - max_tokens_per_sync: None, - max_cost_per_sync_usd: None, - sync_depth_days: None, - } -} - -#[tokio::test] -async fn list_items_returns_connection_as_item() { - let reader = ComposioReader; - let items = reader - .list_items(&test_source(), Path::new(".")) - .await - .unwrap(); - assert_eq!(items.len(), 1); - assert_eq!(items[0].id, "cmp_123"); -} - -#[tokio::test] -async fn read_item_describes_the_provider_pipeline() { - let content = ComposioReader - .read_item(&test_source(), "cmp_123", Path::new(".")) - .await - .unwrap(); - assert_eq!(content.title, "gmail sync data"); - assert!(content.body.contains("provider sync pipeline")); - assert_eq!(content.metadata["toolkit"], "gmail"); - assert_eq!(content.metadata["connection_id"], "cmp_123"); -} diff --git a/crates/tinymemory-integrations/src/sources/types/mod.rs b/crates/tinymemory-integrations/src/sources/types/mod.rs index 156f2589..6e974d33 100644 --- a/crates/tinymemory-integrations/src/sources/types/mod.rs +++ b/crates/tinymemory-integrations/src/sources/types/mod.rs @@ -31,9 +31,6 @@ pub(crate) fn default_true() -> bool { #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, JsonSchema)] #[serde(rename_all = "snake_case")] pub enum SourceKind { - /// A Composio OAuth connector (Gmail, Slack, Notion, …). Network-backed; - /// the live fetch is owned by the host, not this module. - Composio, /// Local agent conversation transcripts stored in the workspace. Conversation, /// A local folder of files matched by an optional glob. @@ -50,8 +47,7 @@ pub enum SourceKind { impl SourceKind { /// Every kind, in declaration order. - pub const ALL: [Self; 7] = [ - Self::Composio, + pub const ALL: [Self; 6] = [ Self::Conversation, Self::Folder, Self::File, @@ -64,7 +60,6 @@ impl SourceKind { #[must_use] pub fn as_str(&self) -> &'static str { match self { - SourceKind::Composio => "composio", SourceKind::Conversation => "conversation", SourceKind::Folder => "folder", SourceKind::File => "file", @@ -81,7 +76,6 @@ impl SourceKind { pub fn api_kind(&self) -> tinymemory_api::SourceKind { use tinymemory_api::SourceKind as Api; match self { - SourceKind::Composio => Api::Composio, SourceKind::Conversation => Api::Conversation, SourceKind::Folder => Api::Folder, SourceKind::File => Api::File, @@ -109,14 +103,6 @@ pub struct MemorySourceEntry { #[serde(default = "default_true")] pub enabled: bool, - // ── Composio ── - /// Composio toolkit slug (e.g. `gmail`). Required for `composio`. - #[serde(default, skip_serializing_if = "Option::is_none")] - pub toolkit: Option, - /// Composio connection id. Required for `composio`. - #[serde(default, skip_serializing_if = "Option::is_none")] - pub connection_id: Option, - // ── Folder / File ── /// Filesystem path of the folder or file to read. Required for `folder` /// and `file`; a relative path is anchored on the workspace. @@ -180,8 +166,6 @@ impl MemorySourceEntry { kind, label: label.into(), enabled: true, - toolkit: None, - connection_id: None, path: None, glob: None, url: None, @@ -201,8 +185,7 @@ impl MemorySourceEntry { /// Validate the fields this entry's [`SourceKind`] requires. /// /// `id` and `label` are required for every kind, and `id` must not contain - /// `:` or control characters. Composio needs `toolkit` and - /// `connection_id`; folders and files need `path`; GitHub repositories, RSS + /// `:` or control characters. Folders and files need `path`; GitHub repositories, RSS /// feeds and web pages need `url`. An empty string counts as missing. /// /// # Errors @@ -221,10 +204,6 @@ impl MemorySourceEntry { return Err(Error::Invalid("label is required".to_string())); } match self.kind { - SourceKind::Composio => { - require_field(&self.toolkit, "toolkit")?; - require_field(&self.connection_id, "connection_id") - } SourceKind::Conversation => Ok(()), SourceKind::Folder | SourceKind::File => require_field(&self.path, "path"), SourceKind::GithubRepo | SourceKind::RssFeed | SourceKind::WebPage => { diff --git a/crates/tinymemory-tools/src/layout/source.rs b/crates/tinymemory-tools/src/layout/source.rs index dcf203d2..4acc6932 100644 --- a/crates/tinymemory-tools/src/layout/source.rs +++ b/crates/tinymemory-tools/src/layout/source.rs @@ -60,10 +60,9 @@ impl BrainSource { pub fn source_kind(&self) -> SourceKind { match self { Self::Files | Self::Pdf | Self::Markdown => SourceKind::File, - Self::Notion => SourceKind::Composio, Self::Github => SourceKind::Github, Self::Web => SourceKind::Link, - Self::Other(_) => SourceKind::Import, + Self::Notion | Self::Other(_) => SourceKind::Import, } } } From 85f35bdf04bb31fabb9077f56032b48f8b082ff5 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Thu, 8 Oct 2026 23:42:18 +0530 Subject: [PATCH 17/19] feat(integrations): add items source and reader modules Adds an items source and a readers module to the integrations crate, wiring them into the sources tree so item-based ingestion can be supported alongside existing sources. Auto-committed-on: macbook Co-authored-by: Medulla --- crates/tinymemory-integrations/Cargo.toml | 4 ++-- crates/tinymemory-integrations/src/lib.rs | 2 +- .../src/sources/items/mod.rs | 13 ++----------- crates/tinymemory-integrations/src/sources/mod.rs | 5 +---- .../src/sources/readers/mod.rs | 9 +-------- 5 files changed, 7 insertions(+), 26 deletions(-) diff --git a/crates/tinymemory-integrations/Cargo.toml b/crates/tinymemory-integrations/Cargo.toml index 69c651b7..c8e3b2ec 100644 --- a/crates/tinymemory-integrations/Cargo.toml +++ b/crates/tinymemory-integrations/Cargo.toml @@ -18,7 +18,7 @@ tinymemory-api = { path = "../tinymemory-api" } # `MemoryEngine`, `BearerSource`, `DocumentConverter` and `SourceReader` are # object-safe async traits. async-trait = { version = "0.1", optional = true } -# Every wire body, envelope, cursor, Composio payload and checkpoint is JSON. +# Every wire body, envelope, cursor and checkpoint is JSON. serde = { version = "1", features = ["derive"], optional = true } serde_json = { version = "1", optional = true } # The typed errors of `documents`, `sources` and `import`. @@ -122,7 +122,7 @@ documents-office = ["documents", "dep:pdf-extract", "dep:calamine", "dep:quick-x # placed under the source type their format implies. brain = ["documents", "dep:tinymemory-tools"] # `sources`: readers that turn folders, files and conversations into -# `StoreItem`s, and the Composio payload normalisers. Links no HTTP stack. +# `StoreItem`s. Links no HTTP stack. sources = ["documents", "dep:schemars", "dep:regex", "dep:walkdir", "dep:chrono", "dep:log", "dep:tracing"] # The readers that fetch over the network — GitHub, RSS, web pages — and # `sources::fetch::fetch_url`, all behind the shared SSRF guard. diff --git a/crates/tinymemory-integrations/src/lib.rs b/crates/tinymemory-integrations/src/lib.rs index f320a470..33a19e6b 100644 --- a/crates/tinymemory-integrations/src/lib.rs +++ b/crates/tinymemory-integrations/src/lib.rs @@ -8,7 +8,7 @@ //! | [`cortex`], [`registry`], [`config`] | `cortex` (default) | The CortexDB engine over its two wires, and building one from configuration | //! | `documents` | `documents`, `documents-office` | Format sniffing and conversion to markdown, emitting `StoreItem::Document` | //! | `brain` | `brain` | Converting files into the brain documents `tinymemory_tools::Brain` ingests, by source type | -//! | `sources` | `sources`, `sources-network` | Readers turning folders, files, links, GitHub, RSS, Composio payloads and conversations into `StoreItem`s | +//! | `sources` | `sources`, `sources-network` | Readers turning folders, files, links, GitHub, RSS, and conversations into `StoreItem`s | //! | `safety` | `safety` | Secret and PII scrubbing for a `StoreItem` before it is stored | //! | `import` | `legacy-import` | Migrating a legacy v1 (embedded TinyCortex) workspace into any engine | //! diff --git a/crates/tinymemory-integrations/src/sources/items/mod.rs b/crates/tinymemory-integrations/src/sources/items/mod.rs index 40205a2e..9bbd5f89 100644 --- a/crates/tinymemory-integrations/src/sources/items/mod.rs +++ b/crates/tinymemory-integrations/src/sources/items/mod.rs @@ -11,7 +11,6 @@ //! | github | document | `repo` (`owner/name`), `commit` (commit items), `url` (issues and PRs), `observed_at` | //! | link | document | `url` | //! | rss | document | `url` (the entry's link), `observed_at` (published) | -//! | composio | document | `tags = [toolkit]` (payloads: see [`crate::sources::composio`]) | //! | conversation | conversation | `workspace`, `thread_id`, `turns`, `observed_at` (last turn) | //! //! Every document body is markdown, converted through `crate::documents`: @@ -38,23 +37,16 @@ use crate::sources::readers::local_file::LocalFile; use crate::sources::types::{ContentType, MemorySourceEntry, SourceContent, SourceKind}; /// Metadata naming `entry` as the source: `source.kind` is the entry's kind -/// mapped onto the contract, `source.id` its id. A Composio entry also gets -/// its toolkit as a tag. +/// mapped onto the contract, `source.id` its id. #[must_use] pub fn base_meta(entry: &MemorySourceEntry) -> MemoryMeta { - let mut meta = MemoryMeta { + MemoryMeta { source: SourceRef { kind: entry.kind.api_kind(), id: Some(entry.id.clone()), }, ..MemoryMeta::default() - }; - if entry.kind == SourceKind::Composio - && let Some(toolkit) = entry.toolkit.as_deref().filter(|t| !t.is_empty()) - { - meta.tags = vec![toolkit.to_string()]; } - meta } /// Convert a local file and wrap it as a document. @@ -197,7 +189,6 @@ pub fn content_item( meta.language = language_for_path(&content.id).map(str::to_string); } SourceKind::Conversation => meta.thread_id = Some(content.id.clone()), - SourceKind::Composio => {} } Ok(StoreItem::Document { diff --git a/crates/tinymemory-integrations/src/sources/mod.rs b/crates/tinymemory-integrations/src/sources/mod.rs index dad654b8..10668fc9 100644 --- a/crates/tinymemory-integrations/src/sources/mod.rs +++ b/crates/tinymemory-integrations/src/sources/mod.rs @@ -1,5 +1,5 @@ //! Source readers for TinyMemory: turn a folder, a file, a web page, a GitHub -//! repository, an RSS feed, a Composio toolkit payload or a local conversation +//! repository, an RSS feed or a local conversation //! into [`StoreItem`](tinymemory_api::StoreItem)s. //! //! - **Configuration** — what a source *is* ([`MemorySourceEntry`], keyed by @@ -13,8 +13,6 @@ //! - **Items** — [`items`] maps reader output to `StoreItem`s with //! [`MemoryMeta`](tinymemory_api::MemoryMeta) filled per kind; every text //! body is converted to markdown through [`crate::documents`]. -//! - **Composio** — [`composio`] normalises toolkit payloads (Gmail, Slack, -//! GitHub, Linear, Notion, ClickUp) and maps them to items. //! //! Scheduling, credentials and egress budgets stay with the host: this module //! reads when asked. @@ -63,7 +61,6 @@ //! `readers::reader_for_request` and the SSRF guard. Without it, a host that //! only reads local sources links no HTTP stack. -pub mod composio; pub mod error; #[cfg(feature = "sources-network")] pub mod fetch; diff --git a/crates/tinymemory-integrations/src/sources/readers/mod.rs b/crates/tinymemory-integrations/src/sources/readers/mod.rs index 503a21ee..050f5c54 100644 --- a/crates/tinymemory-integrations/src/sources/readers/mod.rs +++ b/crates/tinymemory-integrations/src/sources/readers/mod.rs @@ -23,14 +23,9 @@ //! on a timer. A `None` from [`reader_for`] therefore means "route this through //! the host's sync runner", which keeps the host in charge of the network. //! -//! `composio` is represented by a placeholder reader -//! ([`composio::ComposioReader`]): its data arrives through the credentialed -//! provider pipeline, and [`crate::sources::composio`] turns those payloads into items. -//! //! A host servicing an *explicit user request* (not a timer) that wants one //! reader for any kind uses `reader_for_request`. -pub mod composio; pub mod conversation; pub mod file; pub mod folder; @@ -138,8 +133,7 @@ pub fn reader_for(kind: &SourceKind) -> Option> { SourceKind::Folder => Some(Box::new(folder::FolderReader)), SourceKind::File => Some(Box::new(file::FileReader)), SourceKind::Conversation => Some(Box::new(conversation::ConversationReader)), - SourceKind::Composio - | SourceKind::GithubRepo + SourceKind::GithubRepo | SourceKind::RssFeed | SourceKind::WebPage => None, } @@ -156,7 +150,6 @@ pub fn reader_for(kind: &SourceKind) -> Option> { #[must_use] pub fn reader_for_request(kind: &SourceKind) -> Box { match kind { - SourceKind::Composio => Box::new(composio::ComposioReader), SourceKind::Conversation => Box::new(conversation::ConversationReader), SourceKind::Folder => Box::new(folder::FolderReader), SourceKind::File => Box::new(file::FileReader), From 3f88e90ac54582b65f46b450be883b832913c043 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Thu, 8 Oct 2026 23:43:23 +0530 Subject: [PATCH 18/19] refactor(sources): retire the composio source kind The composio source kind and its toolkit and connection_id fields are gone from the source entry, with the legacy "composio" wire value now decoding as import so stored entries keep loading. Tests and fixtures were updated to drop the removed kind and pin the new decoding behaviour. Auto-committed-on: macbook Co-authored-by: Medulla --- crates/tinymemory-api/src/meta/mod_tests.rs | 11 ++++++ .../src/sources/items/mod_tests.rs | 22 +----------- .../sources/readers/conversation/mod_tests.rs | 2 -- .../src/sources/readers/folder/mod_tests.rs | 2 -- .../src/sources/readers/github/mod_tests.rs | 2 -- .../src/sources/readers/mod.rs | 7 ++-- .../src/sources/readers/rss/mod_tests.rs | 2 -- .../src/sources/readers/web_page/mod_tests.rs | 2 -- .../src/sources/types/mod_tests.rs | 35 +++---------------- .../tests/reader_dispatch.rs | 1 - .../tests/fixtures/tool_contracts.json | 5 --- 11 files changed, 18 insertions(+), 73 deletions(-) diff --git a/crates/tinymemory-api/src/meta/mod_tests.rs b/crates/tinymemory-api/src/meta/mod_tests.rs index def04bdf..8299c7ad 100644 --- a/crates/tinymemory-api/src/meta/mod_tests.rs +++ b/crates/tinymemory-api/src/meta/mod_tests.rs @@ -43,3 +43,14 @@ fn source_kind_wire_strings_match_serde() { assert_eq!(json, serde_json::json!(kind.as_str())); } } + +#[test] +fn the_retired_composio_source_kind_decodes_as_import() { + let kind: SourceKind = + serde_json::from_value(serde_json::json!("composio")).expect("legacy kind decodes"); + assert_eq!(kind, SourceKind::Import); + assert_eq!( + serde_json::to_value(SourceKind::Import).expect("serialise"), + serde_json::json!("import") + ); +} diff --git a/crates/tinymemory-integrations/src/sources/items/mod_tests.rs b/crates/tinymemory-integrations/src/sources/items/mod_tests.rs index b050d8ea..e11cae90 100644 --- a/crates/tinymemory-integrations/src/sources/items/mod_tests.rs +++ b/crates/tinymemory-integrations/src/sources/items/mod_tests.rs @@ -54,17 +54,12 @@ fn base_meta_names_the_source_by_contract_kind_and_entry_id() { (SourceKind::WebPage, Api::Link), (SourceKind::GithubRepo, Api::Github), (SourceKind::RssFeed, Api::Rss), - (SourceKind::Composio, Api::Composio), (SourceKind::Conversation, Api::Conversation), ] { let meta = base_meta(&entry(kind)); assert_eq!(meta.source.kind, api); assert_eq!(meta.source.id.as_deref(), Some("src_test")); } - - let mut composio = entry(SourceKind::Composio); - composio.toolkit = Some("gmail".into()); - assert_eq!(base_meta(&composio).tags, vec!["gmail".to_string()]); } #[test] @@ -168,7 +163,7 @@ fn link_items_take_the_page_url() { } #[test] -fn reader_content_for_local_kinds_and_composio_fills_what_it_can() { +fn reader_content_for_local_kinds_fills_what_it_can() { let mut folder = entry(SourceKind::Folder); folder.path = Some("/notes".into()); let item = content_item( @@ -212,21 +207,6 @@ fn reader_content_for_local_kinds_and_composio_fills_what_it_can() { ) .unwrap(); assert_eq!(conversation.meta().thread_id.as_deref(), Some("t1")); - - let mut composio = entry(SourceKind::Composio); - composio.toolkit = Some("slack".into()); - let item = content_item( - &composio, - content( - "c1", - "sync data", - ContentType::Plaintext, - serde_json::json!({}), - ), - None, - ) - .unwrap(); - assert_eq!(item.meta().tags, vec!["slack".to_string()]); } #[test] diff --git a/crates/tinymemory-integrations/src/sources/readers/conversation/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/conversation/mod_tests.rs index 8f49976f..98961bde 100644 --- a/crates/tinymemory-integrations/src/sources/readers/conversation/mod_tests.rs +++ b/crates/tinymemory-integrations/src/sources/readers/conversation/mod_tests.rs @@ -11,8 +11,6 @@ fn conversation_source() -> MemorySourceEntry { kind: SourceKind::Conversation, label: "Conversations".into(), enabled: true, - toolkit: None, - connection_id: None, path: None, glob: None, url: None, diff --git a/crates/tinymemory-integrations/src/sources/readers/folder/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/folder/mod_tests.rs index 32060ca7..970e96d3 100644 --- a/crates/tinymemory-integrations/src/sources/readers/folder/mod_tests.rs +++ b/crates/tinymemory-integrations/src/sources/readers/folder/mod_tests.rs @@ -11,8 +11,6 @@ fn folder_source(path: &str) -> MemorySourceEntry { kind: SourceKind::Folder, label: "Test folder".into(), enabled: true, - toolkit: None, - connection_id: None, path: Some(path.into()), glob: None, url: None, diff --git a/crates/tinymemory-integrations/src/sources/readers/github/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/github/mod_tests.rs index 66168524..2b00d73c 100644 --- a/crates/tinymemory-integrations/src/sources/readers/github/mod_tests.rs +++ b/crates/tinymemory-integrations/src/sources/readers/github/mod_tests.rs @@ -10,8 +10,6 @@ fn github_source(url: Option<&str>) -> MemorySourceEntry { kind: SourceKind::GithubRepo, label: "GitHub".into(), enabled: true, - toolkit: None, - connection_id: None, path: None, glob: None, url: url.map(str::to_string), diff --git a/crates/tinymemory-integrations/src/sources/readers/mod.rs b/crates/tinymemory-integrations/src/sources/readers/mod.rs index 050f5c54..e74907af 100644 --- a/crates/tinymemory-integrations/src/sources/readers/mod.rs +++ b/crates/tinymemory-integrations/src/sources/readers/mod.rs @@ -123,8 +123,7 @@ pub fn is_locally_readable(kind: &SourceKind) -> bool { /// Get the reader for a source kind that is safe to drive on a timer. /// /// Returns `Some` for [`SourceKind::Folder`], [`SourceKind::File`] and -/// [`SourceKind::Conversation`]. Network-backed kinds (`composio`, -/// `github_repo`, `rss_feed`, `web_page`) return `None` so the caller defers to +/// [`SourceKind::Conversation`]. Network-backed kinds (/// `github_repo`, `rss_feed`, `web_page`) return `None` so the caller defers to /// the host's sync runner, which constructs those readers once it has /// authorized the fetch. #[must_use] @@ -133,9 +132,7 @@ pub fn reader_for(kind: &SourceKind) -> Option> { SourceKind::Folder => Some(Box::new(folder::FolderReader)), SourceKind::File => Some(Box::new(file::FileReader)), SourceKind::Conversation => Some(Box::new(conversation::ConversationReader)), - SourceKind::GithubRepo - | SourceKind::RssFeed - | SourceKind::WebPage => None, + SourceKind::GithubRepo | SourceKind::RssFeed | SourceKind::WebPage => None, } } diff --git a/crates/tinymemory-integrations/src/sources/readers/rss/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/rss/mod_tests.rs index cf6fc2ca..a6dbdf41 100644 --- a/crates/tinymemory-integrations/src/sources/readers/rss/mod_tests.rs +++ b/crates/tinymemory-integrations/src/sources/readers/rss/mod_tests.rs @@ -38,8 +38,6 @@ fn rss_source(url: Option<&str>, max_items: Option) -> MemorySourceEntry { label: "Feed".into(), kind: SourceKind::RssFeed, enabled: true, - toolkit: None, - connection_id: None, path: None, glob: None, url: url.map(str::to_string), diff --git a/crates/tinymemory-integrations/src/sources/readers/web_page/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/web_page/mod_tests.rs index 39c0baba..30a695af 100644 --- a/crates/tinymemory-integrations/src/sources/readers/web_page/mod_tests.rs +++ b/crates/tinymemory-integrations/src/sources/readers/web_page/mod_tests.rs @@ -9,8 +9,6 @@ fn web_source(url: Option<&str>, selector: Option<&str>) -> MemorySourceEntry { kind: SourceKind::WebPage, label: "Reference page".into(), enabled: true, - toolkit: None, - connection_id: None, path: None, glob: None, url: url.map(str::to_string), diff --git a/crates/tinymemory-integrations/src/sources/types/mod_tests.rs b/crates/tinymemory-integrations/src/sources/types/mod_tests.rs index 2ea2171b..7ac3adb0 100644 --- a/crates/tinymemory-integrations/src/sources/types/mod_tests.rs +++ b/crates/tinymemory-integrations/src/sources/types/mod_tests.rs @@ -5,7 +5,6 @@ use super::*; #[test] fn source_kind_round_trips_via_serde() { for kind in [ - SourceKind::Composio, SourceKind::Conversation, SourceKind::Folder, SourceKind::File, @@ -21,7 +20,6 @@ fn source_kind_round_trips_via_serde() { #[test] fn source_kind_as_str_matches_wire_strings() { - assert_eq!(SourceKind::Composio.as_str(), "composio"); assert_eq!(SourceKind::Conversation.as_str(), "conversation"); assert_eq!(SourceKind::Folder.as_str(), "folder"); assert_eq!(SourceKind::GithubRepo.as_str(), "github_repo"); @@ -30,26 +28,6 @@ fn source_kind_as_str_matches_wire_strings() { assert_eq!(SourceKind::WebPage.as_str(), "web_page"); } -#[test] -fn validate_composio_requires_toolkit_and_connection_id() { - let entry = MemorySourceEntry { - id: "src_1".into(), - kind: SourceKind::Composio, - label: "Gmail".into(), - enabled: true, - toolkit: Some("gmail".into()), - connection_id: None, - ..default_entry() - }; - assert!(entry.validate().is_err()); - - let valid = MemorySourceEntry { - connection_id: Some("cmp_123".into()), - ..entry - }; - assert!(valid.validate().is_ok()); -} - #[test] fn validate_folder_requires_path() { let entry = MemorySourceEntry { @@ -89,8 +67,10 @@ fn validate_file_requires_path() { #[test] fn the_removed_twitter_query_kind_no_longer_decodes() { - let decoded = serde_json::from_str::("\"twitter_query\""); - assert!(decoded.is_err()); + for removed in ["twitter_query", "composio"] { + let decoded = serde_json::from_str::(&format!("\"{removed}\"")); + assert!(decoded.is_err(), "{removed} must not decode"); + } } #[test] @@ -100,7 +80,6 @@ fn every_config_kind_maps_onto_a_contract_source_kind() { assert_eq!( mapped, vec![ - Api::Composio, Api::Conversation, Api::Folder, Api::File, @@ -245,8 +224,6 @@ pub(super) fn default_entry() -> MemorySourceEntry { kind: SourceKind::Folder, label: String::new(), enabled: true, - toolkit: None, - connection_id: None, path: None, glob: None, url: None, @@ -277,8 +254,6 @@ fn source_entry_wire_format_is_pinned() { kind: SourceKind::GithubRepo, label: "Pinned".into(), enabled: false, - toolkit: Some("gmail".into()), - connection_id: Some("conn-1".into()), path: Some("/notes".into()), glob: Some("**/*.md".into()), url: Some("https://github.com/tinyhumansai/tinymemory".into()), @@ -301,8 +276,6 @@ fn source_entry_wire_format_is_pinned() { "kind": "github_repo", "label": "Pinned", "enabled": false, - "toolkit": "gmail", - "connection_id": "conn-1", "path": "/notes", "glob": "**/*.md", "url": "https://github.com/tinyhumansai/tinymemory", diff --git a/crates/tinymemory-integrations/tests/reader_dispatch.rs b/crates/tinymemory-integrations/tests/reader_dispatch.rs index 6cb12752..946faac7 100644 --- a/crates/tinymemory-integrations/tests/reader_dispatch.rs +++ b/crates/tinymemory-integrations/tests/reader_dispatch.rs @@ -17,7 +17,6 @@ fn timer_dispatch_constructs_only_readers_that_never_need_network() { } for kind in [ - SourceKind::Composio, SourceKind::GithubRepo, SourceKind::RssFeed, SourceKind::WebPage, diff --git a/crates/tinymemory-tools/tests/fixtures/tool_contracts.json b/crates/tinymemory-tools/tests/fixtures/tool_contracts.json index fab2afbb..cb1eacc5 100644 --- a/crates/tinymemory-tools/tests/fixtures/tool_contracts.json +++ b/crates/tinymemory-tools/tests/fixtures/tool_contracts.json @@ -56,7 +56,6 @@ "link", "github", "rss", - "composio", "conversation", "agent", "import" @@ -191,7 +190,6 @@ "link", "github", "rss", - "composio", "conversation", "agent", "import" @@ -332,7 +330,6 @@ "link", "github", "rss", - "composio", "conversation", "agent", "import" @@ -472,7 +469,6 @@ "link", "github", "rss", - "composio", "conversation", "agent", "import" @@ -685,7 +681,6 @@ "link", "github", "rss", - "composio", "conversation", "agent", "import" From 58ffac5921430b293a896cb6124e396ced0e1813 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Thu, 8 Oct 2026 23:43:49 +0530 Subject: [PATCH 19/19] docs: remove Composio source support from documentation The Composio normalisers, post-processors and payload helpers are gone, so the READMEs, architecture notes and memory-v2 spec no longer list Composio as a source kind, reader or feature. The `sources` feature now covers only the local readers, and `SourceKind` drops its `Composio` variant. Auto-committed-on: macbook Co-authored-by: Medulla --- README.md | 2 +- crates/tinymemory-integrations/README.md | 2 +- .../src/sources/README.md | 6 +-- docs/architecture/api-items.md | 2 +- docs/architecture/integrations-sources.md | 45 ------------------- docs/architecture/overview.md | 4 +- docs/specs/memory-v2.md | 4 +- 7 files changed, 9 insertions(+), 56 deletions(-) diff --git a/README.md b/README.md index 3b6ab845..6bbb5c86 100644 --- a/README.md +++ b/README.md @@ -63,7 +63,7 @@ else on request, so a host pays only for what it uses. | `documents` | `documents` | Format sniffing and conversion to markdown, producing `StoreItem::Document` | | `documents-office` | `documents::OfficeConverter` | PDF, DOCX, PPTX and XLSX to markdown (implies `documents`) | | `brain` | `brain` | Files into `tinymemory_tools::BrainDocument`s, by the source type their format implies (implies `documents`) | -| `sources` | `sources` | Folder, file and conversation readers, Composio normalisers (implies `documents`) | +| `sources` | `sources` | Folder, file and conversation readers (implies `documents`) | | `sources-network` | `sources::fetch` and the network readers | GitHub, RSS and web-page readers and `fetch_url`, behind the SSRF guard (implies `sources`) | | `safety` | `safety` | Secret and PII scrubbing of a `StoreItem` | | `legacy-import` | `import` | Migrating a v1 (embedded TinyCortex) workspace into any engine | diff --git a/crates/tinymemory-integrations/README.md b/crates/tinymemory-integrations/README.md index 6d02b39a..075f7c80 100644 --- a/crates/tinymemory-integrations/README.md +++ b/crates/tinymemory-integrations/README.md @@ -17,7 +17,7 @@ wants one integration enables one feature and links nothing else. | `cortex`, `registry`, `config` | `cortex` (default) | `CortexEngine` over two wires (`cortexdb`, `tinyhumans`); `list_engines`, `build_engine`, `EngineCredential`; `MemoryConfig` | [`src/cortex/README.md`](src/cortex/README.md) | | `documents` | `documents` | Format sniffing and conversion to markdown, producing `StoreItem::Document`. No I/O. | [`src/documents/README.md`](src/documents/README.md) | | `documents::OfficeConverter` | `documents-office` | PDF, DOCX, PPTX and XLSX to markdown, in process | (same) | -| `sources` | `sources` | Readers for folders, files and conversations; Composio payload normalisers; `collect_items`. Links no HTTP stack. | [`src/sources/README.md`](src/sources/README.md) | +| `sources` | `sources` | Readers for folders, files and conversations; `collect_items`. Links no HTTP stack. | [`src/sources/README.md`](src/sources/README.md) | | `sources::fetch`, GitHub, RSS and web-page readers | `sources-network` | The network readers and `fetch_url`, all behind the SSRF guard | (same) | | `safety` | `safety` | Secret and PII scrubbing of a `StoreItem` before it is stored | [`src/safety/README.md`](src/safety/README.md) | | `import` | `legacy-import` | Reads a v1 (embedded TinyCortex) workspace and migrates it into any engine, resumably | [`src/import/README.md`](src/import/README.md) | diff --git a/crates/tinymemory-integrations/src/sources/README.md b/crates/tinymemory-integrations/src/sources/README.md index dca6befa..0e151d6f 100644 --- a/crates/tinymemory-integrations/src/sources/README.md +++ b/crates/tinymemory-integrations/src/sources/README.md @@ -2,7 +2,7 @@ `tinymemory_integrations::sources` (features `sources` and `sources-network`): readers that turn a source into `StoreItem`s — a folder, a single file, a web -page, a GitHub repository, an RSS feed, a Composio toolkit payload, or the +page, a GitHub repository, an RSS feed, or the host's local conversation threads. Conversion to markdown and language detection come from the sibling [`documents`](../documents/README.md) module. Architecture overview: @@ -21,7 +21,6 @@ checks it with `MemorySourceEntry::validate`, nothing more. | `fetch` | one URL into a `RawDocument` or a link item (`sources-network`); the RSS and web-page readers fetch through it with their own body caps | | `fetch::ssrf` | the SSRF guard: scheme and host policy, one address classifier for literal and resolved addresses, a public-only DNS resolver, per-hop redirect checks, and a capped body reader | | `items` | reader output to `StoreItem`s with `MemoryMeta` filled per kind; `collect_items` drives a reader end to end | -| `composio` | toolkit normalisers (Gmail, Slack, GitHub, Linear, Notion, ClickUp), the `fields::pick_str` lookup they share, `normalise_payload` and `payload_items`. `readers::composio::ComposioReader` is only a placeholder reader | | `error` | the module `Error`, mapped onto `tinymemory_api::Error` | ## Kinds and metadata @@ -35,7 +34,6 @@ Every item's `meta.source` is `SourceRef { kind, id: Some(entry.id) }`. | `web_page` | `Link` | document | `url` | | `github_repo` | `Github` | document | `repo` (`owner/name`), `commit` (commits), `url` (issues, PRs), `observed_at` | | `rss_feed` | `Rss` | document | `url` (the entry's link), `observed_at` (published) | -| `composio` | `Composio` | document | `tags = [toolkit]`; payloads add `url`, `observed_at`, `thread_id`, `repo` | | `conversation` | `Conversation` | conversation | `workspace`, `thread_id`, `turns`, `observed_at` (last turn) | ## Folder selection @@ -74,7 +72,7 @@ named and numeric entities decode the same way everywhere. ## Features -- `sources` — the local readers, `items`, `composio` and `types`; implies +- `sources` — the local readers, `items` and `types`; implies `documents`. Links no HTTP stack. - `sources-network` — adds the GitHub, RSS and web-page readers, `fetch`, and the SSRF guard. diff --git a/docs/architecture/api-items.md b/docs/architecture/api-items.md index 2909c584..f62ef289 100644 --- a/docs/architecture/api-items.md +++ b/docs/architecture/api-items.md @@ -165,7 +165,7 @@ namespace are omitted on serialisation. | `observed_actor` | `Option` | `{ id, name? }`: who said or did it when not the memory's owner, such as an email's sender or a channel's (`id` is `type:id`, `user:priya@acme.com`, `user:+15551234567`). Excluded from the fingerprint; the CortexDB engine sends it only with attribution on. | `MemoryMeta::from_source(kind, id)` sets only the source. `SourceKind` is -`folder | file | link | github | rss | composio | conversation | agent | +`folder | file | link | github | rss | conversation | agent | import` (`SourceKind::ALL`, `as_str`); `agent` is the default. ## `MetaFilter` diff --git a/docs/architecture/integrations-sources.md b/docs/architecture/integrations-sources.md index 98be38d1..1b0ce235 100644 --- a/docs/architecture/integrations-sources.md +++ b/docs/architecture/integrations-sources.md @@ -22,7 +22,6 @@ optional fields are required; `validate()` checks them. | `web_page` | `url` | `selector` | `Link` | | `github_repo` | `url` | `branch`, `paths`, `max_commits`, `max_issues`, `max_prs` (default 1000 each) | `Github` | | `rss_feed` | `url` | `max_items` (default 50) | `Rss` | -| `composio` | `toolkit`, `connection_id` | | `Composio` | Every entry also needs a non-blank `id` (no `:` or control characters) and a non-empty `label`, and carries `enabled` and optional sync-budget fields @@ -52,7 +51,6 @@ for a host servicing an explicit user request, never a polling loop. | `WebPageReader` | one: the page URL | With a CSS `selector`, only the text of matching elements (plain text; only the last compound of a descendant chain is honoured). Otherwise the whole page as markdown. 10 MiB body cap. | | `RssReader` | one per feed entry (RSS or Atom), up to `max_items` | The parsed feed is cached for 60 seconds so a list-then-read pass downloads it once. 5 MiB cap; non-UTF-8 bodies are refused. | | `GithubReader` | `commit:`, `issue:`, `pr:` | See below. | -| `ComposioReader` | one: the connection | A placeholder; see Composio. | ### local_file and ensure_within_base @@ -96,7 +94,6 @@ are rejected). It combines three transports: | github | document | `repo` (`owner/name`), `commit` (commits), `url` (issues and PRs), `observed_at` | | web page | document | `url` | | rss | document | `url` (entry link), `observed_at` (published) | -| composio | document | `tags = [toolkit]`; payloads add `url`, `observed_at`, `thread_id`, `repo` | | conversation | conversation | `workspace`, `thread_id`, `turns` (`0..=n-1`), `observed_at` (last turn, else mtime) | `collect_items(reader, entry, workspace, converter)` lists and reads every @@ -105,48 +102,6 @@ with its error and the pass continues; only a listing failure is an `Err`. Other entry points: `file_item` (a path with no source), `conversation_item`, `content_item` and `items::local_file_item`. -## Composio - -Composio data does not arrive item by item. A host runs toolkit actions with -its own credentials and hands the raw responses to `sources::composio`, which -holds no credential, opens no socket and decides nothing about when to sync. - -1. **Normalisers**, pure `serde_json::Value` transforms, one module per - toolkit. They walk Composio's envelope variants (top level, under `data`, - under `data.data`) and return the first array found: `clickup` - (`extract_tasks`), `github` (`extract_issues`), `linear` - (`extract_issues`), `notion` (`extract_results`, `extract_page_markdown`), - each with title, id and updated-time helpers. `fields::pick_str` is the - shared lookup: it tries dotted paths, descends only through objects, and - rejects non-string leaves. -2. **Post-processors the host must call** for two toolkits, because their raw - responses are too verbose: - - `gmail_post_process::post_process(slug, arguments, &mut data)` rewrites a - `GMAIL_FETCH_EMAILS` response into slim `messages[]` (other Gmail slugs - pass through; `raw_html: true` in the arguments skips the reshape). If the - response carries a response-level `markdownFormatted` string, call - `apply_response_level_markdown(&mut data, markdown)` **before** - `post_process`; it is a no-op unless the split count matches the message - count. `format_email_local_time` renders in the host's local timezone; - the raw UTC fields are preserved. - - `slack_post_process::post_process(slug, arguments, &mut data)` reshapes - `SLACK_FETCH_CONVERSATION_HISTORY`, `SLACK_LIST_CONVERSATIONS` and - `SLACK_SEARCH_MESSAGES`; unknown slugs are no-ops. `channel_id` for history - is injected by the host (it is in the request, not the response), and user - ids are resolved by the host. -3. `normalise_payload(toolkit, &data)` returns `ComposioDocument`s (id, title, - markdown body, url, `observed_at`, `thread_id`, `repo`), dispatching on the - case-insensitive toolkit slug: `gmail`, `slack`, `github`, `linear`, - `notion`, `clickup`; any other toolkit falls back to each record as fenced - JSON, so a new toolkit is ingested verbosely rather than dropped. Records - with no text are skipped. `payload_items(toolkit, source_id, &data)` wraps - them as `StoreItem::Document` with `source.kind = Composio`, - `source.id = source_id` and `tags = [toolkit]`. - -`readers::composio::ComposioReader` is only a placeholder so -`reader_for_request` can serve every kind: `list_items` returns the connection -as one sync target. - ## Fetching and the SSRF guard (sources-network) `fetch::fetch_url(url)` fetches one URL into a `RawDocument`: the `Content-Type` diff --git a/docs/architecture/overview.md b/docs/architecture/overview.md index 94e6916e..d4c3589f 100644 --- a/docs/architecture/overview.md +++ b/docs/architecture/overview.md @@ -55,7 +55,7 @@ alone, so the contract is the only coupling point. | `tinymemory-integrations` | `cortex` (default) | `cortex::CortexEngine` (both wires), `registry` (`list_engines`, `build_engine`), `config` (`MemoryConfig`) | | | `documents` | Format sniffing and conversion to markdown | | | `documents-office` | PDF, DOCX, PPTX, XLSX conversion (implies `documents`) | -| | `sources` | Readers for folders, files, conversations; Composio normalisers (implies `documents`) | +| | `sources` | Readers for folders, files, conversations (implies `documents`) | | | `sources-network` | GitHub, RSS and web-page readers behind the SSRF guard (implies `sources`) | | | `safety` | Secret and PII scrubbing of a `StoreItem` | | | `legacy-import` | Reading a v1 workspace; `import::migrate` and `migrate_with` | @@ -72,7 +72,7 @@ source reader ──▶ documents ──▶ safety ──▶ engine.store ── ``` 1. **Source reader** (`sources`): lists a configured source (folder, file, - link, GitHub, RSS, Composio payload, conversation) and reads each entry. + link, GitHub, RSS, conversation) and reads each entry. `sources::collect_items` does this for one source; one bad entry lands in `Collected::skipped` instead of aborting the pass. 2. **Documents conversion** (`documents`): sniffs the format and converts the diff --git a/docs/specs/memory-v2.md b/docs/specs/memory-v2.md index 43f7255e..41a21067 100644 --- a/docs/specs/memory-v2.md +++ b/docs/specs/memory-v2.md @@ -30,7 +30,7 @@ Three crates, one directory each under `crates/`. Their relationships are in | --- | --- | | `tinymemory-api` | The contract: `MemoryEngine`, request/response types, `MemoryMeta`, `MetaFilter`, namespaces, `EngineDescriptor`, `Error`. No I/O. Feature `conformance`: the behavioural suite every engine must pass, plus a reference in-memory engine. | | `tinymemory-tools` | The agent tool spec `MemoryTools` (seven tools, host-fixed namespace and reach) and `context`, the `ContextCompiler` that builds `context.md` from an engine. | -| `tinymemory-integrations` | Everything that talks to the outside world, each behind a feature: `cortex` (default; the CortexDB engine, registered twice as `cortexdb` and `tinyhumans`, plus the registry `list_engines`/`build_engine` and `MemoryConfig`), `documents` and `documents-office` (format sniffing and conversion to markdown, emitting `StoreItem::Document`; PDF/DOCX/PPTX/XLSX via `OfficeConverter` or a host `DocumentConverter`), `sources` and `sources-network` (readers for folder, file, link, GitHub, RSS, Composio payloads and conversations, with the SSRF guard), `safety` (secret/PII scrubbing applied to every item before `store`), and `legacy-import` (reads a v1 TinyCortex workspace and migrates it into any engine). | +| `tinymemory-integrations` | Everything that talks to the outside world, each behind a feature: `cortex` (default; the CortexDB engine, registered twice as `cortexdb` and `tinyhumans`, plus the registry `list_engines`/`build_engine` and `MemoryConfig`), `documents` and `documents-office` (format sniffing and conversion to markdown, emitting `StoreItem::Document`; PDF/DOCX/PPTX/XLSX via `OfficeConverter` or a host `DocumentConverter`), `sources` and `sources-network` (readers for folder, file, link, GitHub, RSS and conversations, with the SSRF guard), `safety` (secret/PII scrubbing applied to every item before `store`), and `legacy-import` (reads a v1 TinyCortex workspace and migrates it into any engine). | Deleted: the earlier `tinymemory` facade, `tinymemory-cortex`, `tinymemory-documents`, `tinymemory-sources`, `tinymemory-safety`, @@ -91,7 +91,7 @@ pub struct MemoryMeta { pub tags: Vec, pub observed_at: Option>, } -pub enum SourceKind { Folder, File, Link, Github, Rss, Composio, Conversation, Agent, Import } +pub enum SourceKind { Folder, File, Link, Github, Rss, Conversation, Agent, Import } ``` `MetaFilter` has the same optional fields (each an exact match, `folder` and