From aa48fc6c19131332ac04dd96c277ea9fd126635c Mon Sep 17 00:00:00 2001
From: Steven Enamakel
Date: Thu, 8 Oct 2026 23:26:08 +0530
Subject: [PATCH 01/19] refactor(sources): split source modules into mod and
test files
Moved the inline test modules of the composio, fetch, and readers sources into
sibling mod_tests.rs files and extracted shared types into types.rs, leaving the
mod.rs files focused on implementation. No behaviour changed.
Auto-committed-on: macbook
Co-authored-by: Medulla
---
.../src/sources/composio/clickup/mod.rs | 69 --
.../src/sources/composio/clickup/mod_tests.rs | 58 --
.../src/sources/composio/documents/mod.rs | 343 ----------
.../sources/composio/documents/mod_tests.rs | 197 ------
.../src/sources/composio/fields/mod.rs | 44 --
.../src/sources/composio/fields/mod_tests.rs | 37 --
.../src/sources/composio/github/mod.rs | 116 ----
.../src/sources/composio/github/mod_tests.rs | 97 ---
.../composio/gmail_post_process/mod.rs | 498 --------------
.../composio/gmail_post_process/mod_tests.rs | 356 ----------
.../src/sources/composio/linear/mod.rs | 79 ---
.../src/sources/composio/linear/mod_tests.rs | 93 ---
.../src/sources/composio/mod.rs | 35 -
.../src/sources/composio/notion/mod.rs | 88 ---
.../src/sources/composio/notion/mod_tests.rs | 111 ----
.../composio/slack_post_process/mod.rs | 324 ---------
.../composio/slack_post_process/mod_tests.rs | 306 ---------
.../src/sources/fetch/mod.rs | 151 -----
.../src/sources/fetch/mod_tests.rs | 177 -----
.../src/sources/fetch/ssrf/mod.rs | 231 -------
.../src/sources/fetch/ssrf/mod_tests.rs | 236 -------
.../src/sources/readers/composio/mod.rs | 78 ---
.../src/sources/readers/composio/mod_tests.rs | 51 --
.../src/sources/readers/github/api/mod.rs | 391 -----------
.../readers/github/api/transport_override.rs | 5 -
.../readers/github/api/transport_tests.rs | 34 -
.../src/sources/readers/github/git/mod.rs | 320 ---------
.../sources/readers/github/git/mod_tests.rs | 219 ------
.../src/sources/readers/github/issues/mod.rs | 304 ---------
.../readers/github/issues/mod_tests.rs | 151 -----
.../src/sources/readers/github/mod.rs | 308 ---------
.../src/sources/readers/github/mod_tests.rs | 624 ------------------
.../src/sources/readers/github/types.rs | 122 ----
.../src/sources/readers/rss/mod.rs | 355 ----------
.../src/sources/readers/rss/mod_tests.rs | 386 -----------
.../src/sources/readers/rss/types.rs | 25 -
.../src/sources/readers/web_page/mod.rs | 444 -------------
.../src/sources/readers/web_page/mod_tests.rs | 302 ---------
.../src/sources/readers/web_page/types.rs | 13 -
39 files changed, 7778 deletions(-)
delete mode 100644 crates/tinymemory-integrations/src/sources/composio/clickup/mod.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/composio/clickup/mod_tests.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/composio/documents/mod.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/composio/documents/mod_tests.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/composio/fields/mod.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/composio/fields/mod_tests.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/composio/github/mod.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/composio/github/mod_tests.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod_tests.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/composio/linear/mod.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/composio/linear/mod_tests.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/composio/mod.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/composio/notion/mod.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/composio/notion/mod_tests.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod_tests.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/fetch/mod.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/fetch/mod_tests.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/fetch/ssrf/mod.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/fetch/ssrf/mod_tests.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/readers/composio/mod.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/readers/composio/mod_tests.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/readers/github/api/mod.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/readers/github/api/transport_override.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/readers/github/api/transport_tests.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/readers/github/git/mod.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/readers/github/git/mod_tests.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/readers/github/issues/mod.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/readers/github/issues/mod_tests.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/readers/github/mod.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/readers/github/mod_tests.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/readers/github/types.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/readers/rss/mod.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/readers/rss/mod_tests.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/readers/rss/types.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/readers/web_page/mod.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/readers/web_page/mod_tests.rs
delete mode 100644 crates/tinymemory-integrations/src/sources/readers/web_page/types.rs
diff --git a/crates/tinymemory-integrations/src/sources/composio/clickup/mod.rs b/crates/tinymemory-integrations/src/sources/composio/clickup/mod.rs
deleted file mode 100644
index b4ce64e5..00000000
--- a/crates/tinymemory-integrations/src/sources/composio/clickup/mod.rs
+++ /dev/null
@@ -1,69 +0,0 @@
-//! ClickUp host normalization helpers — result extraction and task-title and
-//! timestamp extraction.
-//!
-//! ClickUp's REST API (and therefore Composio's wrapping of it) returns
-//! task lists in a small handful of shapes depending on which endpoint
-//! is called. The functions here walk the union of common shapes so the
-//! provider doesn't have to branch per Composio envelope variant.
-
-use serde_json::Value;
-
-use super::fields::pick_str;
-
-/// Walk the Composio response envelope for ClickUp task list results.
-///
-/// ClickUp's "filtered team tasks" endpoint returns `{ "tasks": [...] }`
-/// at the top level; Composio re-wraps the upstream payload under
-/// `data` or `data.data` depending on the action. We probe each shape
-/// in order and return the first array we find.
-pub fn extract_tasks(data: &Value) -> Vec {
- let candidates = [
- data.pointer("/data/tasks"),
- data.pointer("/tasks"),
- data.pointer("/data/data/tasks"),
- data.pointer("/data/results"),
- data.pointer("/results"),
- data.pointer("/data/items"),
- data.pointer("/items"),
- ];
- for cand in candidates.into_iter().flatten() {
- if let Some(arr) = cand.as_array() {
- return arr.clone();
- }
- }
- Vec::new()
-}
-
-/// Extract a human-readable title from a ClickUp task object.
-///
-/// ClickUp tasks store the name at `name` (or `data.name` after Composio
-/// envelope wrapping). When the name is missing we fall back to the
-/// task ID so chunks remain identifiable.
-pub fn extract_task_name(task: &Value) -> Option {
- pick_str(task, &["name", "data.name", "title", "data.title"])
-}
-
-/// Extract a stable cursor timestamp (milliseconds since epoch as a
-/// string) from a ClickUp task object.
-///
-/// The ClickUp API returns `date_updated` as a stringified epoch ms
-/// (e.g. `"1733412345678"`); we keep it as a string so lexicographic
-/// comparison against the stored cursor remains valid as long as the
-/// length doesn't change (it won't until year 33658).
-pub fn extract_task_updated(task: &Value) -> Option {
- pick_str(
- task,
- &[
- "date_updated",
- "data.date_updated",
- "updated_at",
- "data.updated_at",
- "dateUpdated",
- "data.dateUpdated",
- ],
- )
-}
-
-#[cfg(test)]
-#[path = "mod_tests.rs"]
-mod tests;
diff --git a/crates/tinymemory-integrations/src/sources/composio/clickup/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/clickup/mod_tests.rs
deleted file mode 100644
index 5a520401..00000000
--- a/crates/tinymemory-integrations/src/sources/composio/clickup/mod_tests.rs
+++ /dev/null
@@ -1,58 +0,0 @@
-//! Tests for the ClickUp normaliser.
-
-use super::*;
-use serde_json::json;
-
-#[test]
-fn extract_tasks_from_data_tasks() {
- let data = json!({ "data": { "tasks": [{"id": "t1"}] } });
- assert_eq!(extract_tasks(&data).len(), 1);
-}
-
-#[test]
-fn extract_tasks_from_top_level_tasks() {
- let data = json!({ "tasks": [{"id": "a"}, {"id": "b"}] });
- assert_eq!(extract_tasks(&data).len(), 2);
-}
-
-#[test]
-fn extract_tasks_empty_when_missing() {
- let data = json!({ "foo": "bar" });
- assert!(extract_tasks(&data).is_empty());
-}
-
-#[test]
-fn extract_task_name_from_top_level() {
- let task = json!({ "id": "t1", "name": "Build feature X" });
- assert_eq!(extract_task_name(&task), Some("Build feature X".into()));
-}
-
-#[test]
-fn extract_task_name_falls_back_to_data_name() {
- let task = json!({ "data": { "name": "Wrapped" } });
- assert_eq!(extract_task_name(&task), Some("Wrapped".into()));
-}
-
-#[test]
-fn extract_task_name_none_when_missing() {
- let task = json!({ "id": "t1" });
- assert!(extract_task_name(&task).is_none());
-}
-
-#[test]
-fn extract_task_updated_handles_string_form() {
- let task = json!({ "date_updated": "1733412345678" });
- assert_eq!(
- extract_task_updated(&task),
- Some("1733412345678".to_string())
- );
-}
-
-#[test]
-fn extract_task_updated_handles_nested_data() {
- let task = json!({ "data": { "dateUpdated": "1700000000000" } });
- assert_eq!(
- extract_task_updated(&task),
- Some("1700000000000".to_string())
- );
-}
diff --git a/crates/tinymemory-integrations/src/sources/composio/documents/mod.rs b/crates/tinymemory-integrations/src/sources/composio/documents/mod.rs
deleted file mode 100644
index 04753e4b..00000000
--- a/crates/tinymemory-integrations/src/sources/composio/documents/mod.rs
+++ /dev/null
@@ -1,343 +0,0 @@
-//! One Composio response in, documents and `StoreItem`s out.
-//!
-//! [`normalise_payload`] dispatches on the toolkit slug to the matching
-//! normaliser and reads each record's title, body, link and timestamp.
-//! Toolkits without a dedicated normaliser fall back to a generic walk that
-//! keeps each record as fenced JSON, so a new toolkit is ingested (verbosely)
-//! rather than dropped. [`payload_items`] wraps the documents as
-//! `StoreItem::Document`s.
-
-use crate::documents::{DocumentFormat, markdown_from_text};
-use chrono::{DateTime, TimeZone, Utc};
-use serde_json::Value;
-use tinymemory_api::{DocumentBody, MemoryMeta, SourceKind, StoreItem};
-
-use super::fields::pick_str;
-use super::{clickup, github, gmail_post_process, linear, notion};
-
-/// One record of a Composio payload, normalised: an email, a message, an
-/// issue, a task or a page.
-#[derive(Debug, Clone, PartialEq)]
-pub struct ComposioDocument {
- /// The provider's id for the record, when it has one.
- pub id: Option,
- /// A human-readable title.
- pub title: Option,
- /// The record's text as markdown. Never empty.
- pub body: String,
- /// A link back to the record in the provider's UI.
- pub url: Option,
- /// When the record was last updated or sent.
- pub observed_at: Option>,
- /// The email or message thread the record belongs to.
- pub thread_id: Option,
- /// The repository a GitHub record belongs to, as `owner/name`.
- pub repo: Option,
-}
-
-impl ComposioDocument {
- /// A record with only a body.
- fn with_body(body: String) -> Self {
- Self {
- id: None,
- title: None,
- body,
- url: None,
- observed_at: None,
- thread_id: None,
- repo: None,
- }
- }
-
- /// Wrap this record as a [`StoreItem::Document`] from `toolkit`, read
- /// through the connection or source `source_id`.
- #[must_use]
- pub fn into_store_item(self, toolkit: &str, source_id: &str) -> StoreItem {
- let mut meta = MemoryMeta::from_source(SourceKind::Composio, Some(source_id.to_string()));
- meta.url = self.url;
- meta.observed_at = self.observed_at;
- meta.thread_id = self.thread_id;
- meta.repo = self.repo;
- meta.tags = vec![toolkit.to_string()];
- StoreItem::Document {
- title: self.title,
- body: DocumentBody::Text(self.body),
- mime: Some(DocumentFormat::Markdown.mime().to_string()),
- meta,
- }
- }
-}
-
-/// Normalise one Composio response from `toolkit` into documents.
-///
-/// `data` is the action's response; for Gmail and Slack, run
-/// [`gmail_post_process::post_process`] / [`super::slack_post_process::post_process`]
-/// on it first so it carries the slim `messages[]` shape. Records with no text
-/// at all are skipped.
-#[must_use]
-pub fn normalise_payload(toolkit: &str, data: &Value) -> Vec {
- let documents: Vec = match toolkit.to_ascii_lowercase().as_str() {
- "gmail" => array_at(data, &["/messages", "/data/messages"])
- .iter()
- .filter_map(gmail_message)
- .collect(),
- "slack" => array_at(data, &["/messages", "/data/messages"])
- .iter()
- .filter_map(slack_message)
- .collect(),
- "github" => github::extract_issues(data)
- .iter()
- .filter_map(github_issue)
- .collect(),
- "linear" => linear::extract_issues(data)
- .iter()
- .filter_map(linear_issue)
- .collect(),
- "notion" => notion_pages(data),
- "clickup" => clickup::extract_tasks(data)
- .iter()
- .filter_map(clickup_task)
- .collect(),
- _ => generic_records(data),
- };
- log::debug!(
- "[memory_sources:composio] normalised toolkit={toolkit} documents={}",
- documents.len()
- );
- documents
-}
-
-/// Normalise one Composio response and wrap every record as a
-/// [`StoreItem::Document`] with `source.kind = Composio`,
-/// `source.id = source_id` and `tags = [toolkit]`.
-#[must_use]
-pub fn payload_items(toolkit: &str, source_id: &str, data: &Value) -> Vec {
- normalise_payload(toolkit, data)
- .into_iter()
- .map(|document| document.into_store_item(toolkit, source_id))
- .collect()
-}
-
-/// The first array found at any of `pointers`.
-fn array_at<'a>(data: &'a Value, pointers: &[&str]) -> &'a [Value] {
- pointers
- .iter()
- .find_map(|pointer| data.pointer(pointer).and_then(Value::as_array))
- .map_or(&[], Vec::as_slice)
-}
-
-/// A body as markdown: HTML (as sniffed) is converted, anything else kept.
-fn to_markdown(text: &str) -> String {
- let format = DocumentFormat::sniff(text.as_bytes(), None, None);
- markdown_from_text(text, format).trim().to_string()
-}
-
-/// Build a document from its parts, falling back to the title as the body;
-/// `None` when there is no text at all.
-fn document(title: Option, body: Option) -> Option {
- let body = body
- .map(|body| to_markdown(&body))
- .filter(|body| !body.is_empty())
- .or_else(|| title.clone())?;
- let mut document = ComposioDocument::with_body(body);
- document.title = title;
- Some(document)
-}
-
-/// Parse an ISO 8601 / RFC 3339 / RFC 2822 timestamp.
-fn parse_time(text: &str) -> Option> {
- gmail_post_process::parse_email_date(text)
-}
-
-/// Parse an epoch-milliseconds string (ClickUp's `date_updated`).
-fn parse_epoch_ms(text: &str) -> Option> {
- let millis = text.trim().parse::().ok()?;
- Utc.timestamp_millis_opt(millis).single()
-}
-
-/// Parse a Slack `ts` (`"1712345678.123456"`, epoch seconds with a fraction).
-fn parse_slack_ts(text: &str) -> Option> {
- let (seconds, fraction) = text.split_once('.').unwrap_or((text, "0"));
- let seconds = seconds.parse::().ok()?;
- let micros = format!("{fraction:0<6}").get(..6)?.parse::().ok()?;
- Utc.timestamp_opt(seconds, micros * 1_000).single()
-}
-
-/// A string field, or a number rendered as a string.
-fn scalar(value: &Value, key: &str) -> Option {
- match value.get(key)? {
- Value::String(text) if !text.trim().is_empty() => Some(text.trim().to_string()),
- Value::Number(number) => Some(number.to_string()),
- _ => None,
- }
-}
-
-/// A post-processed Gmail message: headers above the body.
-fn gmail_message(message: &Value) -> Option {
- let subject = pick_str(message, &["subject"]);
- let markdown = pick_str(message, &["markdown", "messageText"]);
- let mut header = String::new();
- for (label, key) in [("From", "from"), ("To", "to"), ("Date", "date")] {
- if let Some(value) = pick_str(message, &[key]) {
- header.push_str(&format!("{label}: {value}\n"));
- }
- }
- let body = match (markdown, header.is_empty()) {
- (Some(markdown), false) => Some(format!("{header}\n{markdown}")),
- (Some(markdown), true) => Some(markdown),
- (None, _) => None,
- };
- let mut document = document(subject, body)?;
- document.id = pick_str(message, &["id", "messageId"]);
- document.thread_id = pick_str(message, &["threadId", "thread_id"]);
- document.observed_at = pick_str(message, &["date"]).as_deref().and_then(parse_time);
- Some(document)
-}
-
-/// A post-processed Slack message.
-fn slack_message(message: &Value) -> Option {
- let user = pick_str(message, &["user"]);
- let channel = pick_str(message, &["channel_id"]);
- let title = match (&user, &channel) {
- (Some(user), Some(channel)) => Some(format!("Slack message from {user} in {channel}")),
- (Some(user), None) => Some(format!("Slack message from {user}")),
- (None, _) => Some("Slack message".to_string()),
- };
- let text = pick_str(message, &["text"])?;
- let mut document = document(title, Some(text))?;
- let ts = pick_str(message, &["ts"]);
- document.id = ts.clone();
- document.url = pick_str(message, &["permalink"]);
- document.thread_id = pick_str(message, &["thread_ts"]);
- document.observed_at = ts.as_deref().and_then(parse_slack_ts);
- Some(document)
-}
-
-/// A GitHub issue or pull request from a search response.
-fn github_issue(issue: &Value) -> Option {
- let mut document = document(
- github::extract_issue_title(issue),
- pick_str(issue, &["body", "data.body"]),
- )?;
- document.id = github::extract_issue_id(issue);
- document.url = pick_str(issue, &["html_url", "data.html_url"]);
- document.repo = document.url.as_deref().and_then(github_repo);
- document.observed_at = github::extract_issue_updated_at(issue)
- .as_deref()
- .and_then(parse_time);
- Some(document)
-}
-
-/// `owner/name` from a `https://github.com/owner/name/...` link.
-fn github_repo(url: &str) -> Option {
- let rest = url.split_once("github.com/")?.1;
- let mut parts = rest.split('/');
- let owner = parts.next().filter(|part| !part.is_empty())?;
- let name = parts.next().filter(|part| !part.is_empty())?;
- Some(format!("{owner}/{name}"))
-}
-
-/// A Linear issue.
-fn linear_issue(issue: &Value) -> Option {
- let mut document = document(
- linear::extract_issue_title(issue),
- pick_str(issue, &["description", "data.description"]),
- )?;
- document.id = pick_str(issue, &["identifier", "id", "data.identifier", "data.id"]);
- document.url = pick_str(issue, &["url", "data.url"]);
- document.observed_at = linear::extract_issue_updated(issue)
- .as_deref()
- .and_then(parse_time);
- Some(document)
-}
-
-/// Notion pages from a search response, or the one page a
-/// `NOTION_GET_PAGE_MARKDOWN` response carries.
-fn notion_pages(data: &Value) -> Vec {
- let results = notion::extract_results(data);
- if results.is_empty() {
- return notion::extract_page_markdown(data)
- .and_then(|markdown| {
- let title = notion::extract_page_title(data);
- let mut document = document(title, Some(markdown))?;
- document.id = pick_str(data, &["id", "data.id", "page_id", "data.page_id"]);
- document.url = pick_str(data, &["url", "data.url"]);
- Some(document)
- })
- .into_iter()
- .collect();
- }
- results
- .iter()
- .filter_map(|page| {
- let mut document = document(
- notion::extract_page_title(page),
- notion::extract_page_markdown(page),
- )?;
- document.id = pick_str(page, &["id", "data.id"]);
- document.url = pick_str(page, &["url", "data.url"]);
- document.observed_at = pick_str(page, &["last_edited_time", "data.last_edited_time"])
- .as_deref()
- .and_then(parse_time);
- Some(document)
- })
- .collect()
-}
-
-/// A ClickUp task.
-fn clickup_task(task: &Value) -> Option {
- let mut document = document(
- clickup::extract_task_name(task),
- pick_str(
- task,
- &["markdown_description", "description", "text_content"],
- ),
- )?;
- document.id = scalar(task, "id");
- document.url = pick_str(task, &["url", "data.url"]);
- document.observed_at = clickup::extract_task_updated(task)
- .as_deref()
- .and_then(|text| parse_epoch_ms(text).or_else(|| parse_time(text)));
- Some(document)
-}
-
-/// Any other toolkit: each record in the first list found (or the whole
-/// payload) as fenced JSON, titled and linked when it says how.
-fn generic_records(data: &Value) -> Vec {
- let records = array_at(
- data,
- &[
- "/data/items",
- "/items",
- "/data/results",
- "/results",
- "/data/data",
- "/data",
- ],
- );
- let records: Vec<&Value> = if records.is_empty() {
- vec![data]
- } else {
- records.iter().collect()
- };
- records
- .into_iter()
- .filter(|record| !record.is_null())
- .filter_map(|record| {
- let json = serde_json::to_string_pretty(record).ok()?;
- let title = pick_str(record, &["title", "name", "subject"]);
- let mut document = ComposioDocument::with_body(format!("```json\n{json}\n```"));
- document.title = title;
- document.id = scalar(record, "id");
- document.url = pick_str(record, &["url", "html_url", "permalink", "link"]);
- document.observed_at = pick_str(record, &["updated_at", "updatedAt", "created_at"])
- .as_deref()
- .and_then(parse_time);
- Some(document)
- })
- .collect()
-}
-
-#[cfg(test)]
-#[path = "mod_tests.rs"]
-mod tests;
diff --git a/crates/tinymemory-integrations/src/sources/composio/documents/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/documents/mod_tests.rs
deleted file mode 100644
index 4d689128..00000000
--- a/crates/tinymemory-integrations/src/sources/composio/documents/mod_tests.rs
+++ /dev/null
@@ -1,197 +0,0 @@
-//! Tests for turning Composio payloads into documents and `StoreItem`s.
-
-use super::*;
-use serde_json::json;
-use tinymemory_api::ItemKind;
-
-fn text_of(item: &StoreItem) -> (&Option, &str, &MemoryMeta) {
- match item {
- StoreItem::Document {
- title,
- body: DocumentBody::Text(text),
- meta,
- ..
- } => (title, text.as_str(), meta),
- other => panic!("expected a text document, got {other:?}"),
- }
-}
-
-#[test]
-fn a_github_search_becomes_items_with_url_repo_time_and_toolkit_tag() {
- let data = json!({
- "data": { "items": [{
- "id": 42,
- "title": "Fix the build",
- "body": "The build is **red**.",
- "html_url": "https://github.com/acme/widgets/issues/7",
- "updated_at": "2024-05-21T15:30:00Z"
- }]}
- });
- let items = payload_items("github", "conn_gh", &data);
- assert_eq!(items.len(), 1);
- let item = &items[0];
- assert_eq!(item.kind(), ItemKind::Document);
- item.validate().unwrap();
-
- let (title, body, meta) = text_of(item);
- assert_eq!(
- title.as_deref(),
- Some("GitHub: acme/widgets#7: Fix the build")
- );
- assert_eq!(body, "The build is **red**.");
- assert_eq!(meta.source.kind, SourceKind::Composio);
- assert_eq!(meta.source.id.as_deref(), Some("conn_gh"));
- assert_eq!(meta.tags, vec!["github".to_string()]);
- assert_eq!(
- meta.url.as_deref(),
- Some("https://github.com/acme/widgets/issues/7")
- );
- assert_eq!(meta.repo.as_deref(), Some("acme/widgets"));
- assert_eq!(
- meta.observed_at,
- Some(Utc.with_ymd_and_hms(2024, 5, 21, 15, 30, 0).unwrap())
- );
-}
-
-#[test]
-fn post_processed_gmail_messages_carry_headers_thread_and_date() {
- let data = json!({ "messages": [{
- "id": "m1",
- "threadId": "t1",
- "subject": "Lunch?",
- "from": "Ann ",
- "to": "me@example.com",
- "date": "Tue, 21 May 2024 12:00:00 +0000",
- "markdown": "Tacos at noon."
- }]});
- let documents = normalise_payload("gmail", &data);
- assert_eq!(documents.len(), 1);
- let document = &documents[0];
- assert_eq!(document.title.as_deref(), Some("Lunch?"));
- assert!(document.body.starts_with("From: Ann \n"));
- assert!(document.body.ends_with("Tacos at noon."));
- assert_eq!(document.thread_id.as_deref(), Some("t1"));
- assert_eq!(document.id.as_deref(), Some("m1"));
- assert_eq!(
- document.observed_at,
- Some(Utc.with_ymd_and_hms(2024, 5, 21, 12, 0, 0).unwrap())
- );
-}
-
-#[test]
-fn slack_messages_take_permalink_and_ts_time() {
- let data = json!({ "messages": [{
- "ts": "1716300000.000100",
- "user": "U1",
- "channel_id": "C1",
- "text": "deploy done",
- "permalink": "https://acme.slack.com/archives/C1/p1716300000000100"
- }]});
- let items = payload_items("slack", "src_slack", &data);
- let (title, body, meta) = text_of(&items[0]);
- assert_eq!(title.as_deref(), Some("Slack message from U1 in C1"));
- assert_eq!(body, "deploy done");
- assert_eq!(
- meta.url.as_deref(),
- Some("https://acme.slack.com/archives/C1/p1716300000000100")
- );
- assert_eq!(
- meta.observed_at.map(|at| at.timestamp()),
- Some(1_716_300_000)
- );
- assert_eq!(meta.tags, vec!["slack".to_string()]);
-}
-
-#[test]
-fn linear_notion_and_clickup_records_are_normalised() {
- let linear = json!({ "nodes": [{
- "identifier": "ENG-1",
- "title": "Ship v2",
- "description": "All of it.",
- "url": "https://linear.app/acme/issue/ENG-1",
- "updatedAt": "2024-01-02T03:04:05Z"
- }]});
- let documents = normalise_payload("linear", &linear);
- assert_eq!(documents[0].id.as_deref(), Some("ENG-1"));
- assert_eq!(documents[0].body, "All of it.");
- assert!(documents[0].observed_at.is_some());
-
- let notion = json!({ "results": [{
- "id": "p1",
- "url": "https://notion.so/p1",
- "last_edited_time": "2024-01-02T03:04:05.000Z",
- "properties": { "Name": { "type": "title", "title": [{ "plain_text": "Roadmap" }] } }
- }]});
- let documents = normalise_payload("notion", ¬ion);
- assert_eq!(documents[0].title.as_deref(), Some("Roadmap"));
- assert_eq!(
- documents[0].body, "Roadmap",
- "a page without body keeps its title"
- );
- assert_eq!(documents[0].url.as_deref(), Some("https://notion.so/p1"));
-
- let page_markdown = json!({ "data": { "markdown": "# Plan\n\nDo it." }, "id": "p2" });
- let documents = normalise_payload("notion", &page_markdown);
- assert_eq!(documents.len(), 1);
- assert_eq!(documents[0].body, "# Plan\n\nDo it.");
-
- let clickup = json!({ "tasks": [{
- "id": "t9",
- "name": "Write docs",
- "description": "The README
",
- "url": "https://app.clickup.com/t/t9",
- "date_updated": "1700000000000"
- }]});
- let documents = normalise_payload("clickup", &clickup);
- assert_eq!(documents[0].id.as_deref(), Some("t9"));
- assert_eq!(
- documents[0].observed_at.map(|at| at.timestamp()),
- Some(1_700_000_000)
- );
-}
-
-#[test]
-fn html_bodies_are_converted_to_markdown() {
- let data = json!({ "nodes": [{
- "title": "Page",
- "description": "Hi There
"
- }]});
- let documents = normalise_payload("linear", &data);
- assert_eq!(documents[0].body, "## Hi\n\nThere");
-}
-
-#[test]
-fn records_without_any_text_are_skipped() {
- let data = json!({ "messages": [{ "ts": "1.0", "text": " " }] });
- assert!(normalise_payload("slack", &data).is_empty());
- assert!(normalise_payload("github", &json!({ "items": [{}] })).is_empty());
-}
-
-#[test]
-fn an_unknown_toolkit_keeps_each_record_as_fenced_json() {
- let data = json!({ "data": { "items": [
- { "id": 1, "name": "Alpha", "url": "https://example.com/1", "updated_at": "2024-01-01T00:00:00Z" },
- { "id": 2 }
- ]}});
- let items = payload_items("hubspot", "conn_hs", &data);
- assert_eq!(items.len(), 2);
- let (title, body, meta) = text_of(&items[0]);
- assert_eq!(title.as_deref(), Some("Alpha"));
- assert!(body.starts_with("```json\n"));
- assert!(body.contains("\"name\": \"Alpha\""));
- assert_eq!(meta.url.as_deref(), Some("https://example.com/1"));
- assert_eq!(meta.tags, vec!["hubspot".to_string()]);
-
- let single = normalise_payload("hubspot", &json!({ "ok": true }));
- assert_eq!(single.len(), 1);
-}
-
-#[test]
-fn slack_timestamps_parse_with_and_without_a_fraction() {
- assert_eq!(parse_slack_ts("10").map(|at| at.timestamp()), Some(10));
- assert_eq!(
- parse_slack_ts("10.5").map(|at| at.timestamp_subsec_micros()),
- Some(500_000)
- );
- assert_eq!(parse_slack_ts("x"), None);
-}
diff --git a/crates/tinymemory-integrations/src/sources/composio/fields/mod.rs b/crates/tinymemory-integrations/src/sources/composio/fields/mod.rs
deleted file mode 100644
index 4c257085..00000000
--- a/crates/tinymemory-integrations/src/sources/composio/fields/mod.rs
+++ /dev/null
@@ -1,44 +0,0 @@
-//! Field lookup shared by the Composio normalisers: pull a string out of a
-//! payload by trying several dotted paths, because Composio wraps the same
-//! upstream field at different depths depending on the action and version.
-
-/// Walk a JSON object using a list of dotted-path candidates and return the
-/// first non-empty **string** match, trimmed.
-///
-/// Each path is split on `.` and followed with `Value::get`, so it only
-/// descends through objects — it never indexes into an array. A leaf that is
-/// not a string (a number, a bool) is rejected rather than coerced, so a
-/// payload whose `id` is `42` rather than `"42"` yields `None` here. That
-/// differs from the private `scalar` lookup in the `documents` mapping, which
-/// renders numbers; the normalisers were written against the
-/// reject-non-strings behaviour and `pick_str_rejects_non_string_values` pins
-/// it.
-pub fn pick_str(value: &serde_json::Value, paths: &[&str]) -> Option {
- for path in paths {
- let mut cur = value;
- let mut ok = true;
- for segment in path.split('.') {
- match cur.get(segment) {
- Some(next) => cur = next,
- None => {
- ok = false;
- break;
- }
- }
- }
- if !ok {
- continue;
- }
- if let Some(s) = cur.as_str() {
- let trimmed = s.trim();
- if !trimmed.is_empty() {
- return Some(trimmed.to_string());
- }
- }
- }
- None
-}
-
-#[cfg(test)]
-#[path = "mod_tests.rs"]
-mod tests;
diff --git a/crates/tinymemory-integrations/src/sources/composio/fields/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/fields/mod_tests.rs
deleted file mode 100644
index 330ec404..00000000
--- a/crates/tinymemory-integrations/src/sources/composio/fields/mod_tests.rs
+++ /dev/null
@@ -1,37 +0,0 @@
-//! Tests for the shared Composio field lookup.
-
-use super::*;
-use serde_json::json;
-
-#[test]
-fn pick_str_finds_first_non_empty_match() {
- let v = json!({"data": {"user": {"name": "Ada", "email": "ada@example.com"}}});
- assert_eq!(
- pick_str(&v, &["data.user.name", "data.user.email"]),
- Some("Ada".into())
- );
- assert_eq!(
- pick_str(&v, &["data.missing", "data.user.email"]),
- Some("ada@example.com".into())
- );
- assert_eq!(pick_str(&v, &["nope.nope"]), None);
-}
-
-#[test]
-fn pick_str_respects_path_order() {
- let v = json!({"a": "first", "b": "second"});
- assert_eq!(pick_str(&v, &["a", "b"]), Some("first".into()));
- assert_eq!(pick_str(&v, &["b", "a"]), Some("second".into()));
-}
-
-/// The drift guard for the behaviour documented on [`pick_str`]. If this
-/// ever starts returning `Some("42")`, the normalisers' emitted ids have
-/// changed.
-#[test]
-fn pick_str_rejects_non_string_values() {
- let v = json!({"count": 42, "flag": true, "empty": "", "whitespace": " "});
- assert_eq!(pick_str(&v, &["count"]), None);
- assert_eq!(pick_str(&v, &["flag"]), None);
- assert_eq!(pick_str(&v, &["empty"]), None);
- assert_eq!(pick_str(&v, &["whitespace"]), None);
-}
diff --git a/crates/tinymemory-integrations/src/sources/composio/github/mod.rs b/crates/tinymemory-integrations/src/sources/composio/github/mod.rs
deleted file mode 100644
index 12e7a615..00000000
--- a/crates/tinymemory-integrations/src/sources/composio/github/mod.rs
+++ /dev/null
@@ -1,116 +0,0 @@
-//! GitHub host normalization helpers — issue extraction and issue id, title and
-//! timestamp helpers.
-//!
-//! GitHub's REST API (proxied through Composio) returns search results in a
-//! small number of shapes. The functions here
-//! walk the union of common Composio envelope variants so the provider stays
-//! clean and branch-free.
-
-use serde_json::Value;
-
-use super::fields::pick_str;
-
-/// Walk the Composio response envelope for GitHub search issue results.
-///
-/// `GITHUB_SEARCH_ISSUES_AND_PULL_REQUESTS` wraps GitHub's `GET /search/issues` response, which
-/// returns `{"total_count": N, "items": [...]}`. Composio may re-wrap this under
-/// `data` or `data.data`; we probe each shape in order.
-pub fn extract_issues(data: &Value) -> Vec {
- let candidates = [
- data.pointer("/data/items"),
- data.pointer("/items"),
- data.pointer("/data/data/items"),
- data.pointer("/data/results"),
- data.pointer("/results"),
- ];
- for cand in candidates.into_iter().flatten() {
- if let Some(arr) = cand.as_array() {
- return arr.clone();
- }
- }
- Vec::new()
-}
-
-/// Extract a stable, globally unique identifier for a GitHub issue or PR.
-///
-/// GitHub's internal `id` field is a large integer unique across all issues
-/// and PRs on github.com. We convert it to a string for use as a sync key.
-/// Falls back to composing from `html_url` path if `id` is absent.
-pub fn extract_issue_id(issue: &Value) -> Option {
- // Primary: numeric internal GitHub ID.
- if let Some(id) = issue.get("id").or_else(|| issue.pointer("/data/id")) {
- if let Some(n) = id.as_u64() {
- return Some(n.to_string());
- }
- if let Some(s) = id.as_str() {
- let trimmed = s.trim();
- if !trimmed.is_empty() {
- return Some(trimmed.to_string());
- }
- }
- }
- // Fallback: parse owner/repo/number from html_url path segments.
- // URL shape: https://github.com/{owner}/{repo}/issues/{number}
- if let Some(url) = pick_str(issue, &["html_url", "data.html_url", "url", "data.url"])
- && let Some(slug) = github_url_to_slug(&url)
- {
- return Some(slug);
- }
- None
-}
-
-/// Build a human-readable document title for a GitHub issue/PR.
-///
-/// Format: `GitHub: {owner}/{repo}#{number}: {title}`.
-/// Falls back to just the title or a placeholder when fields are missing.
-pub fn extract_issue_title(issue: &Value) -> Option {
- let title = pick_str(issue, &["title", "data.title"])?;
-
- // Best-effort: extract owner/repo#N from html_url for the prefix.
- let prefix = pick_str(issue, &["html_url", "data.html_url"])
- .and_then(|url| github_url_to_slug(&url))
- .unwrap_or_default();
-
- if prefix.is_empty() {
- Some(title)
- } else {
- Some(format!("GitHub: {prefix}: {title}"))
- }
-}
-
-/// Parse `https://github.com/{owner}/{repo}/issues/{number}` (or `/pull/`)
-/// into `"{owner}/{repo}#{number}"`. Returns `None` for unrecognised shapes.
-fn github_url_to_slug(url: &str) -> Option {
- let segs: Vec<&str> = url.trim_end_matches('/').split('/').collect();
- // Minimum: ["https:", "", "github.com", owner, repo, "issues", number]
- if segs.len() >= 7 {
- let number = segs[segs.len() - 1];
- let _kind = segs[segs.len() - 2]; // "issues" or "pull" — ignored
- let repo = segs[segs.len() - 3];
- let owner = segs[segs.len() - 4];
- if !owner.is_empty() && !repo.is_empty() && !number.is_empty() {
- return Some(format!("{owner}/{repo}#{number}"));
- }
- }
- None
-}
-
-/// Extract the `updated_at` ISO 8601 timestamp from a GitHub issue.
-///
-/// GitHub returns `updated_at` as `"2024-05-21T15:30:00Z"`. ISO 8601 strings
-/// sort lexicographically, so we use them directly as the sync cursor.
-pub fn extract_issue_updated_at(issue: &Value) -> Option {
- pick_str(
- issue,
- &[
- "updated_at",
- "data.updated_at",
- "updatedAt",
- "data.updatedAt",
- ],
- )
-}
-
-#[cfg(test)]
-#[path = "mod_tests.rs"]
-mod tests;
diff --git a/crates/tinymemory-integrations/src/sources/composio/github/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/github/mod_tests.rs
deleted file mode 100644
index 0b7322f1..00000000
--- a/crates/tinymemory-integrations/src/sources/composio/github/mod_tests.rs
+++ /dev/null
@@ -1,97 +0,0 @@
-//! Tests for the GitHub normaliser.
-
-use super::*;
-use serde_json::json;
-
-#[test]
-fn extract_issues_from_data_items() {
- let data = json!({ "data": { "items": [{"id": 1}] } });
- assert_eq!(extract_issues(&data).len(), 1);
-}
-
-#[test]
-fn extract_issues_from_top_level_items() {
- let data = json!({ "items": [{"id": 1}, {"id": 2}] });
- assert_eq!(extract_issues(&data).len(), 2);
-}
-
-#[test]
-fn extract_issues_empty_when_missing() {
- let data = json!({ "foo": "bar" });
- assert!(extract_issues(&data).is_empty());
-}
-
-#[test]
-fn extract_issue_id_from_numeric_field() {
- let issue = json!({ "id": 123456789u64, "title": "Fix bug" });
- assert_eq!(extract_issue_id(&issue), Some("123456789".to_string()));
-}
-
-#[test]
-fn extract_issue_id_from_wrapped_data() {
- let issue = json!({ "data": { "id": 99u64 } });
- assert_eq!(extract_issue_id(&issue), Some("99".to_string()));
-}
-
-#[test]
-fn extract_issue_id_falls_back_to_html_url() {
- let issue = json!({
- "html_url": "https://github.com/owner/repo/issues/42"
- });
- assert_eq!(extract_issue_id(&issue), Some("owner/repo#42".to_string()));
-}
-
-#[test]
-fn extract_issue_id_none_when_missing() {
- let issue = json!({ "title": "No ID here" });
- assert!(extract_issue_id(&issue).is_none());
-}
-
-#[test]
-fn extract_issue_title_builds_prefixed_title() {
- let issue = json!({
- "id": 1u64,
- "title": "Fix race condition",
- "html_url": "https://github.com/acme/core/issues/99"
- });
- assert_eq!(
- extract_issue_title(&issue),
- Some("GitHub: acme/core#99: Fix race condition".to_string())
- );
-}
-
-#[test]
-fn extract_issue_title_returns_raw_title_when_no_url() {
- let issue = json!({ "title": "Bare title" });
- assert_eq!(extract_issue_title(&issue), Some("Bare title".to_string()));
-}
-
-#[test]
-fn extract_issue_title_none_when_missing() {
- let issue = json!({ "id": 1u64 });
- assert!(extract_issue_title(&issue).is_none());
-}
-
-#[test]
-fn extract_issue_updated_at_from_top_level() {
- let issue = json!({ "updated_at": "2024-05-21T15:30:00Z" });
- assert_eq!(
- extract_issue_updated_at(&issue),
- Some("2024-05-21T15:30:00Z".to_string())
- );
-}
-
-#[test]
-fn extract_issue_updated_at_from_data_wrapper() {
- let issue = json!({ "data": { "updated_at": "2023-01-01T00:00:00Z" } });
- assert_eq!(
- extract_issue_updated_at(&issue),
- Some("2023-01-01T00:00:00Z".to_string())
- );
-}
-
-#[test]
-fn extract_issue_updated_at_none_when_missing() {
- let issue = json!({ "id": 1u64 });
- assert!(extract_issue_updated_at(&issue).is_none());
-}
diff --git a/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod.rs b/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod.rs
deleted file mode 100644
index c6c78ae7..00000000
--- a/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod.rs
+++ /dev/null
@@ -1,498 +0,0 @@
-//! Gmail-specific post-processing of Composio action responses.
-//!
-//! The upstream `GMAIL_FETCH_EMAILS` payload is extremely verbose
-//! (full MIME tree under `payload.parts[]`, 50+ `Received:` headers,
-//! display-layer noise the model never uses). This module rewrites
-//! it into a slim envelope per message:
-//!
-//! ```json
-//! {
-//! "messages": [
-//! {
-//! "id": "…",
-//! "threadId": "…",
-//! "subject": "…",
-//! "from": "…",
-//! "to": "…",
-//! "date": "…",
-//! "labels": ["INBOX", "UNREAD"],
-//! "markdown": "…body…",
-//! "attachments": [ { "filename": "...", "mimeType": "..." } ]
-//! }
-//! ],
-//! "nextPageToken": "…",
-//! "resultSizeEstimate": 201
-//! }
-//! ```
-//!
-//! ## Body source
-//!
-//! Composio's backend ships a
-//! `markdownFormatted` field on the response envelope — one string
-//! per tool call, pre-rendered with HTML stripped, URLs shortened,
-//! footers removed, whitespace normalised. We split it per message
-//! along `\n---\n` boundaries (with `## ` heading fallbacks) and
-//! pin each slice to the corresponding entry in `messages[]` via
-//! [`apply_response_level_markdown`]. The reshape's
-//! `extract_markdown_body` then prefers that pinned field over
-//! falling back to the upstream `messageText`.
-//!
-//! No in-house HTML→markdown conversion lives here anymore — the
-//! backend does the cleaning. If `markdownFormatted` is absent for
-//! a given response we fall through to whatever plain text the
-//! upstream provided in `messageText`.
-//!
-//! Callers that need the raw Composio shape can pass `raw_html:
-//! true` (or `rawHtml: true`) in the action arguments — this
-//! short-circuits the reshape entirely.
-//!
-//! Only `GMAIL_FETCH_EMAILS` is reshaped today; other Gmail action
-//! responses are passed through unchanged. When we add envelopes for
-//! more slugs they should live in this file, branched from
-//! [`post_process`].
-
-use serde_json::{Map, Value, json};
-
-/// Entry point a host calls on each Gmail action response (the slug names
-/// the action) before handing it to `normalise_payload`.
-///
-/// Dispatches on the Composio action slug. Unknown Gmail slugs fall
-/// through to a no-op.
-pub fn post_process(slug: &str, arguments: Option<&Value>, data: &mut Value) {
- if is_raw_html_flag_set(arguments) {
- tracing::debug!(
- slug,
- "[composio:gmail][post-process] raw_html flag set, passing through"
- );
- return;
- }
- if slug == "GMAIL_FETCH_EMAILS" {
- reshape_fetch_emails(data)
- }
-}
-
-/// Stash per-message slices of the response-level `markdownFormatted`
-/// onto the corresponding entries inside `data.messages[]`.
-///
-/// The Composio backend (tinyhumansai/backend#683) ships ONE
-/// `markdownFormatted` string per tool call covering all messages —
-/// already URL-shortened, footer-stripped, and whitespace-normalised.
-/// To get per-email files in the raw archive we split that string
-/// along section boundaries (`## ` headings or `---` rules) and pin
-/// each slice to the message at the same index. `extract_markdown_body`
-/// then prefers `msg.markdownFormatted` over re-decoding the MIME
-/// tree.
-///
-/// **Must be called BEFORE [`post_process`]** because `post_process`
-/// reshapes `data` into the slim envelope; once `messages[]` carries
-/// our slim shape the upstream message ordering is already locked in
-/// but we may have lost original ordering signals if any.
-///
-/// No-op when the slice count doesn't match `messages.len()` — we
-/// can't safely align segments to messages without an exact match,
-/// so we let `extract_markdown_body` fall through to its MIME path.
-pub fn apply_response_level_markdown(data: &mut Value, top_md: &str) {
- let trimmed = top_md.trim();
- if trimmed.is_empty() {
- return;
- }
- // Presence is checked immutably first, then fetched mutably. The original
- // form re-fetched with `unwrap()` after a mutable probe, which is sound but
- // relies on the reader to see why; this crate lints against `unwrap`, and the
- // immutable probe expresses the same reasoning to the compiler.
- let container = if data.get("messages").is_some() {
- data
- } else if data.get("data").and_then(Value::as_object).is_some() {
- match data.get_mut("data") {
- Some(inner) => inner,
- None => return,
- }
- } else {
- tracing::debug!(
- "[composio:gmail][post-process] apply_response_level_markdown: \
- no messages container in response — skipping"
- );
- return;
- };
- let Some(messages) = container.get_mut("messages").and_then(|v| v.as_array_mut()) else {
- return;
- };
- let count = messages.len();
- if count == 0 {
- return;
- }
- // Clone hints out of the messages array so the slice borrows
- // don't conflict with the upcoming `messages.iter_mut()` mutation.
- let hints: Vec = messages.clone();
- let Some(slices) = split_response_markdown_per_message_with_hint(trimmed, count, Some(&hints))
- else {
- tracing::debug!(
- messages = count,
- md_len = trimmed.len(),
- "[composio:gmail][post-process] could not split response-level markdownFormatted \
- into {count} slices — falling back to per-message MIME decode"
- );
- return;
- };
- for (msg, slice) in messages.iter_mut().zip(slices) {
- if let Some(obj) = msg.as_object_mut() {
- obj.insert("markdownFormatted".to_string(), Value::String(slice));
- }
- }
- tracing::debug!(
- messages = count,
- "[composio:gmail][post-process] stashed per-message markdownFormatted slices"
- );
-}
-
-/// Split a top-level `markdownFormatted` string into per-message
-/// segments. Returns `Some(slices)` only when the split yields
-/// exactly `expected_count` entries — otherwise the format isn't one
-/// of the patterns we know about and we let the caller fall back.
-///
-/// Primary boundary is the `\n---\n` horizontal rule the backend
-/// emits between messages (confirmed against real
-/// `GMAIL_FETCH_EMAILS` output). H2/H3 headings are kept as
-/// fallbacks for older renderings. The preamble (`# Inbox (N
-/// messages)`-style intro, if present) is dropped — we accept
-/// either `expected` parts (no preamble) or `expected + 1`
-/// (preamble + N messages).
-///
-/// `messages_hint` is the slim message array from the same response
-/// — when present we use the per-message `subject` field to verify
-/// each segment really does belong to the message at the same index.
-/// Mismatches force a fallback so we never write a wrong-message body
-/// to the raw archive.
-///
-/// The hint is what makes the split reliable: the blob's own section headings
-/// are backend-rendered and have changed shape between versions, so matching
-/// on them alone silently mis-attributed bodies.
-pub fn split_response_markdown_per_message_with_hint(
- md: &str,
- expected_count: usize,
- messages_hint: Option<&[Value]>,
-) -> Option> {
- if expected_count == 0 {
- return None;
- }
- if expected_count == 1 {
- return Some(vec![md.to_string()]);
- }
-
- // Boundary patterns to try, in priority order. `\n---\n` is the
- // confirmed marker; the heading variants stay as belt-and-braces
- // for older / variant backend renderings.
- let candidates: &[(&str, &str)] = &[
- ("\n---\n", "---\n"),
- ("\n\n## ", "## "),
- ("\n\n### ", "### "),
- ("\n\n# ", "# "),
- ("\n***\n", "***\n"),
- ];
-
- for (sep, prefix) in candidates {
- let parts: Vec<&str> = md.split(sep).collect();
- let (drop_preamble, prepend_first) = if parts.len() == expected_count {
- (false, false) // no preamble; first segment had no prefix
- } else if parts.len() == expected_count + 1 {
- (true, true) // preamble dropped; every kept segment had a prefix
- } else {
- continue;
- };
- let segments: Vec = parts
- .into_iter()
- .skip(if drop_preamble { 1 } else { 0 })
- .enumerate()
- .map(|(i, s)| {
- if i == 0 && !prepend_first {
- s.to_string()
- } else {
- format!("{prefix}{s}")
- }
- })
- .collect();
-
- // Validate alignment against the JSON message array: every
- // segment whose corresponding message has a non-empty subject
- // must mention that subject somewhere in its body. If a single
- // pair fails, we treat the split as unreliable and try the
- // next pattern. Empty / null subjects skip validation (e.g.
- // notification mails where the subject is "").
- if let Some(hints) = messages_hint
- && !validate_segments_against_hints(&segments, hints)
- {
- tracing::debug!(
- expected = expected_count,
- sep = sep,
- "[composio:gmail][post-process] split candidate failed subject check"
- );
- continue;
- }
- return Some(segments);
- }
- None
-}
-
-/// True if every (segment, message) pair where the message has a
-/// non-empty subject contains that subject somewhere in the segment
-/// (case-insensitive substring match — a defensive heuristic, not a
-/// strict equality check, since the backend may format subjects
-/// inside markdown links or with surrounding decoration).
-fn validate_segments_against_hints(segments: &[String], hints: &[Value]) -> bool {
- if segments.len() != hints.len() {
- return false;
- }
- for (seg, hint) in segments.iter().zip(hints.iter()) {
- let subject = hint
- .get("subject")
- .and_then(|v| v.as_str())
- .unwrap_or("")
- .trim();
- if subject.is_empty() {
- continue;
- }
- if !seg
- .to_ascii_lowercase()
- .contains(&subject.to_ascii_lowercase())
- {
- return false;
- }
- }
- true
-}
-
-/// Returns true when the caller explicitly set `raw_html: true` (or the
-/// camelCase `rawHtml: true`) in the `arguments` object.
-fn is_raw_html_flag_set(arguments: Option<&Value>) -> bool {
- let Some(obj) = arguments.and_then(|v| v.as_object()) else {
- return false;
- };
- obj.get("raw_html")
- .or_else(|| obj.get("rawHtml"))
- .and_then(|v| v.as_bool())
- .unwrap_or(false)
-}
-
-/// Rewrite a `GMAIL_FETCH_EMAILS` `data` object in place into the slim
-/// envelope documented at the module level.
-///
-/// The Composio response can be shaped either as `{ messages, nextPageToken, ... }`
-/// directly, or wrapped one level deeper under `{ data: { messages: … } }`
-/// depending on backend version; we handle both.
-fn reshape_fetch_emails(data: &mut Value) {
- // Unwrap an optional `data:` envelope so downstream logic only has
- // to deal with one shape.
- let container = if data.get("messages").is_some() {
- data
- } else if data.get("data").and_then(Value::as_object).is_some() {
- match data.get_mut("data") {
- Some(inner) => inner,
- None => return,
- }
- } else {
- return;
- };
-
- let Some(obj) = container.as_object_mut() else {
- return;
- };
-
- let raw_messages = obj
- .remove("messages")
- .and_then(|v| match v {
- Value::Array(arr) => Some(arr),
- _ => None,
- })
- .unwrap_or_default();
- let next_page_token = obj.remove("nextPageToken").unwrap_or(Value::Null);
- let result_size_estimate = obj.remove("resultSizeEstimate").unwrap_or(Value::Null);
-
- let messages: Vec = raw_messages.into_iter().map(reshape_message).collect();
-
- let mut envelope = Map::new();
- envelope.insert("messages".into(), Value::Array(messages));
- if !next_page_token.is_null() {
- envelope.insert("nextPageToken".into(), next_page_token);
- }
- if !result_size_estimate.is_null() {
- envelope.insert("resultSizeEstimate".into(), result_size_estimate);
- }
-
- *container = Value::Object(envelope);
-}
-
-/// Parse an RFC 3339 or RFC 2822 date string into a UTC `DateTime`.
-pub fn parse_email_date(date_str: &str) -> Option> {
- date_str
- .parse::>()
- .or_else(|_| {
- chrono::DateTime::parse_from_rfc2822(date_str).map(|d| d.with_timezone(&chrono::Utc))
- })
- .ok()
-}
-
-const EMAIL_LOCAL_TIME_FMT: &str = "%Y-%m-%d %I:%M %p %:z";
-
-/// Format a UTC `DateTime` in the given timezone. Returns `None` when the
-/// formatted result is identical to the UTC rendering (no-op for UTC hosts).
-pub fn format_at_tz(
- utc: chrono::DateTime,
- tz: &Tz,
-) -> Option
-where
- Tz::Offset: std::fmt::Display,
-{
- let local_dt = utc.with_timezone(tz);
- let formatted = local_dt.format(EMAIL_LOCAL_TIME_FMT).to_string();
-
- let utc_formatted = utc.format(EMAIL_LOCAL_TIME_FMT).to_string();
- if formatted == utc_formatted {
- return None;
- }
- Some(formatted)
-}
-
-/// Convert a UTC email timestamp string to a human-readable local-time string.
-///
-/// Accepts RFC 3339 (`"2026-05-31T10:33:00Z"`) or RFC 2822
-/// (`"Sat, 31 May 2026 10:33:00 +0000"`) input. Returns a formatted string
-/// in the host's local timezone, e.g. `"2026-05-31 05:33 AM -05:00"`,
-/// so the agent can present local times without UTC arithmetic.
-///
-/// The raw `date` field is always preserved alongside this field so
-/// internal sorting, deduplication, and debugging remain UTC-based.
-///
-/// Returns `None` when the input cannot be parsed or the output format
-/// would be identical to the UTC input (no-op for UTC hosts).
-pub fn format_email_local_time(date_str: &str) -> Option {
- let utc = parse_email_date(date_str)?;
- format_at_tz(utc, &chrono::Local)
-}
-
-/// Map one raw Composio message object to its slim counterpart.
-///
-/// Body source picked by [`extract_markdown_body`]:
-/// 1. The per-message `markdownFormatted` slice pinned by
-/// [`apply_response_level_markdown`] (preferred — backend-rendered).
-/// 2. The upstream `messageText` plaintext (fallback).
-/// 3. Empty string.
-fn reshape_message(raw: Value) -> Value {
- let Value::Object(obj) = raw else {
- return raw;
- };
-
- let id = obj.get("messageId").cloned().unwrap_or(Value::Null);
- let thread_id = obj.get("threadId").cloned().unwrap_or(Value::Null);
- let subject = obj.get("subject").cloned().unwrap_or(Value::Null);
- let sender = obj.get("sender").cloned().unwrap_or(Value::Null);
- let to = obj.get("to").cloned().unwrap_or(Value::Null);
- let date = obj
- .get("messageTimestamp")
- .cloned()
- .or_else(|| pick_header(&obj, "Date"))
- .unwrap_or(Value::Null);
- let labels = obj
- .get("labelIds")
- .cloned()
- .unwrap_or_else(|| Value::Array(Vec::new()));
- let list_unsubscribe = pick_header(&obj, "List-Unsubscribe").unwrap_or(Value::Null);
-
- let markdown = extract_markdown_body(&obj);
- let attachments = extract_attachments(&obj);
-
- // Compute a local-time representation of the UTC `date` so the agent
- // presents times in the user's timezone rather than quoting raw UTC.
- let date_local = date.as_str().and_then(format_email_local_time);
-
- let mut out = Map::new();
- out.insert("id".into(), id);
- out.insert("threadId".into(), thread_id);
- out.insert("subject".into(), subject);
- out.insert("from".into(), sender);
- out.insert("to".into(), to);
- out.insert("date".into(), date);
- if let Some(local) = date_local {
- out.insert("date_local".into(), Value::String(local));
- }
- out.insert("labels".into(), labels);
- if !list_unsubscribe.is_null() {
- out.insert("list_unsubscribe".into(), list_unsubscribe);
- }
- out.insert("markdown".into(), Value::String(markdown));
- if !attachments.is_empty() {
- out.insert("attachments".into(), Value::Array(attachments));
- }
- Value::Object(out)
-}
-
-/// Find a header value by (case-insensitive) name in the Composio
-/// `payload.headers[]` array. Returns `Some(Value::String)` on hit.
-fn pick_header(msg: &Map, name: &str) -> Option {
- let headers = msg.get("payload")?.get("headers")?.as_array()?;
- for h in headers {
- let hn = h.get("name").and_then(|v| v.as_str()).unwrap_or("");
- if hn.eq_ignore_ascii_case(name)
- && let Some(v) = h.get("value").and_then(|v| v.as_str())
- {
- return Some(Value::String(v.to_string()));
- }
- }
- None
-}
-
-/// Pick a body for the slim envelope.
-///
-/// We trust the Composio backend's pre-rendered `markdownFormatted`
-/// (set per-message by [`apply_response_level_markdown`] from the
-/// response-level field). When that's absent we fall back to the
-/// upstream's plain-text `messageText` verbatim — no in-house
-/// HTML→markdown decoding lives here anymore. The backend already
-/// strips HTML, shortens URLs, and normalises whitespace; running
-/// our own pipeline on top duplicated work and corrupted some
-/// renderings.
-fn extract_markdown_body(msg: &Map) -> String {
- if let Some(formatted) = msg
- .get("markdownFormatted")
- .or_else(|| msg.get("markdown_formatted"))
- .and_then(|v| v.as_str())
- .map(str::trim)
- .filter(|s| !s.is_empty())
- {
- return formatted.to_string();
- }
- if let Some(text) = msg
- .get("messageText")
- .and_then(|v| v.as_str())
- .map(str::trim)
- .filter(|s| !s.is_empty())
- {
- return text.to_string();
- }
- String::new()
-}
-
-/// Pull a minimal attachments descriptor from the Composio
-/// `attachmentList` array.
-fn extract_attachments(msg: &Map) -> Vec {
- if let Some(list) = msg.get("attachmentList").and_then(|v| v.as_array()) {
- return list
- .iter()
- .filter_map(|a| {
- let filename = a.get("filename").and_then(|v| v.as_str())?;
- if filename.is_empty() {
- return None;
- }
- let mime = a
- .get("mimeType")
- .and_then(|v| v.as_str())
- .unwrap_or_default();
- Some(json!({ "filename": filename, "mimeType": mime }))
- })
- .collect();
- }
- Vec::new()
-}
-
-#[cfg(test)]
-#[path = "mod_tests.rs"]
-mod tests;
diff --git a/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod_tests.rs
deleted file mode 100644
index dd1b44af..00000000
--- a/crates/tinymemory-integrations/src/sources/composio/gmail_post_process/mod_tests.rs
+++ /dev/null
@@ -1,356 +0,0 @@
-//! Tests for the Gmail post-processor.
-
-use super::*;
-use serde_json::json;
-
-fn fixture_with_backend_markdown() -> Value {
- json!({
- "messages": [
- {
- "messageId": "m1",
- "threadId": "t1",
- "subject": "Hello",
- "sender": "a@x.com",
- "to": "b@y.com",
- "messageTimestamp": "2026-04-17T12:00:00Z",
- "labelIds": ["INBOX", "UNREAD"],
- // Pre-rendered slice (set by `apply_response_level_markdown`
- // in production; inline here for the reshape test).
- "markdownFormatted": "# Hello\n\nbody copy",
- "messageText": "fallback should not be used",
- "display_url": "ignore-me",
- "preview": { "body": "Hi plain", "subject": "Hello" },
- "attachmentList": [
- { "filename": "report.pdf", "mimeType": "application/pdf", "size": 12345 },
- { "filename": "", "mimeType": "text/html" }
- ],
- "payload": {}
- }
- ],
- "nextPageToken": "tok-1",
- "resultSizeEstimate": 42
- })
-}
-
-#[test]
-fn reshape_emits_slim_envelope() {
- let mut v = fixture_with_backend_markdown();
- post_process("GMAIL_FETCH_EMAILS", None, &mut v);
-
- assert_eq!(v["nextPageToken"], "tok-1");
- assert_eq!(v["resultSizeEstimate"], 42);
-
- let msgs = v["messages"].as_array().unwrap();
- assert_eq!(msgs.len(), 1);
- let m = &msgs[0];
-
- assert_eq!(m["id"], "m1");
- assert_eq!(m["threadId"], "t1");
- assert_eq!(m["subject"], "Hello");
- assert_eq!(m["from"], "a@x.com");
- assert_eq!(m["to"], "b@y.com");
- assert_eq!(m["date"], "2026-04-17T12:00:00Z");
- assert_eq!(m["labels"], json!(["INBOX", "UNREAD"]));
-
- let md = m["markdown"].as_str().unwrap();
- assert_eq!(md, "# Hello\n\nbody copy");
-
- // Noise fields removed.
- assert!(m.get("display_url").is_none());
- assert!(m.get("preview").is_none());
- assert!(m.get("payload").is_none());
- assert!(m.get("messageText").is_none());
-
- // Attachments: empty filename entry is filtered.
- let atts = m["attachments"].as_array().unwrap();
- assert_eq!(atts.len(), 1);
- assert_eq!(atts[0]["filename"], "report.pdf");
- assert_eq!(atts[0]["mimeType"], "application/pdf");
-}
-
-#[test]
-fn raw_html_flag_passes_through_unchanged() {
- let mut v = fixture_with_backend_markdown();
- let original = v.clone();
- let args = json!({ "raw_html": true });
- post_process("GMAIL_FETCH_EMAILS", Some(&args), &mut v);
- assert_eq!(
- v, original,
- "raw_html=true must preserve the Composio shape"
- );
-}
-
-#[test]
-fn camel_case_raw_html_also_recognized() {
- let mut v = fixture_with_backend_markdown();
- let original = v.clone();
- let args = json!({ "rawHtml": true });
- post_process("GMAIL_FETCH_EMAILS", Some(&args), &mut v);
- assert_eq!(v, original);
-}
-
-#[test]
-fn falls_back_to_message_text_when_no_backend_markdown() {
- let mut v = json!({
- "messages": [{
- "messageId": "m1",
- "threadId": "t1",
- "subject": "s",
- "sender": "a@x.com",
- "to": "b@y.com",
- "messageTimestamp": "2026-04-17",
- "labelIds": [],
- "messageText": " plain body text ",
- "payload": {}
- }],
- "nextPageToken": null
- });
- post_process("GMAIL_FETCH_EMAILS", None, &mut v);
- let md = v["messages"][0]["markdown"].as_str().unwrap();
- assert_eq!(md, "plain body text");
- assert!(v.get("nextPageToken").is_none(), "null tokens dropped");
-}
-
-#[test]
-fn unwraps_data_envelope() {
- let mut v = json!({
- "data": {
- "messages": [{
- "messageId": "m1",
- "threadId": "t1",
- "subject": "s",
- "sender": "a@x.com",
- "to": "b@y.com",
- "messageTimestamp": "2026-04-17",
- "labelIds": [],
- "messageText": "body",
- "payload": {}
- }]
- }
- });
- post_process("GMAIL_FETCH_EMAILS", None, &mut v);
- // Reshape writes into `data` in place.
- let msgs = v["data"]["messages"].as_array().unwrap();
- assert_eq!(msgs.len(), 1);
- assert_eq!(msgs[0]["markdown"], "body");
-}
-
-#[test]
-fn non_fetch_slug_is_noop() {
- let mut v = json!({ "messages": [{ "messageId": "m1", "messageText": "x" }] });
- let original = v.clone();
- post_process("GMAIL_SEND_EMAIL", None, &mut v);
- assert_eq!(v, original);
-}
-
-#[test]
-fn prefers_backend_markdown_formatted_when_present() {
- // Composio backend (tinyhumansai/backend#683 +) ships
- // `markdownFormatted` already URL-shortened + footer-stripped
- // per message (after `apply_response_level_markdown` slices the
- // response-level field). When present, our post-processor must
- // use it verbatim instead of falling back to `messageText`.
- let mut v = json!({
- "messages": [{
- "messageId": "m1",
- "threadId": "t1",
- "subject": "s",
- "sender": "a@x.com",
- "to": "b@y.com",
- "messageTimestamp": "2026-04-17",
- "labelIds": [],
- "markdownFormatted": "# Already nice\n\nShort URL: https://gh.io/abc",
- "messageText": "fallback should not be used",
- "payload": {}
- }]
- });
- post_process("GMAIL_FETCH_EMAILS", None, &mut v);
- let md = v["messages"][0]["markdown"].as_str().unwrap();
- assert_eq!(md, "# Already nice\n\nShort URL: https://gh.io/abc");
-}
-
-#[test]
-fn empty_markdown_formatted_falls_through_to_message_text() {
- let mut v = json!({
- "messages": [{
- "messageId": "m1",
- "threadId": "t1",
- "subject": "s",
- "sender": "a@x.com",
- "to": "b@y.com",
- "messageTimestamp": "2026-04-17",
- "labelIds": [],
- "markdownFormatted": " \n \n",
- "messageText": "real body",
- "payload": {}
- }]
- });
- post_process("GMAIL_FETCH_EMAILS", None, &mut v);
- let md = v["messages"][0]["markdown"].as_str().unwrap();
- assert!(md.contains("real body"));
-}
-
-// ── split_response_markdown_per_message_with_hint ───────────────────────
-
-#[test]
-fn split_response_markdown_uses_horizontal_rule_marker() {
- // The confirmed backend marker is `\n---\n`. Three messages →
- // expect three slices when there's no preamble.
- let md = "## Alice's update\n\nbody A with https://gh.io/abc\n---\n## Bob's reply\n\nbody B\n---\n## Carol\n\nbody C";
- let slices = super::split_response_markdown_per_message_with_hint(md, 3, None).unwrap();
- assert_eq!(slices.len(), 3);
- assert!(slices[0].contains("Alice's update"));
- assert!(slices[1].contains("Bob's reply"));
- assert!(slices[2].contains("Carol"));
- // The `---\n` prefix is preserved on every-but-the-first segment
- // so the section break survives the round-trip.
- assert!(slices[1].starts_with("---\n"));
- assert!(slices[2].starts_with("---\n"));
-}
-
-#[test]
-fn split_response_markdown_drops_preamble() {
- // When a preamble like `# Inbox` precedes the first marker, we
- // see N+1 parts after split — the preamble must be dropped.
- let md = "# Inbox (2 messages)\n---\n## A\n\nbody A\n---\n## B\n\nbody B";
- let slices = super::split_response_markdown_per_message_with_hint(md, 2, None).unwrap();
- assert_eq!(slices.len(), 2);
- assert!(slices[0].contains("body A"));
- assert!(slices[1].contains("body B"));
- // Both segments should carry the prefix when preamble was dropped.
- assert!(slices[0].starts_with("---\n"));
- assert!(slices[1].starts_with("---\n"));
-}
-
-#[test]
-fn split_response_markdown_falls_back_to_h2_marker() {
- // No `---` rules — backend used h2 headings as boundaries.
- let md = "## Alice\n\nbody A\n\n## Bob\n\nbody B";
- let slices = super::split_response_markdown_per_message_with_hint(md, 2, None).unwrap();
- assert_eq!(slices.len(), 2);
- assert!(slices[0].contains("body A"));
- assert!(slices[1].contains("body B"));
-}
-
-#[test]
-fn split_response_markdown_returns_none_on_count_mismatch() {
- let md = "## only one section here";
- assert!(super::split_response_markdown_per_message_with_hint(md, 3, None).is_none());
-}
-
-#[test]
-fn split_response_markdown_single_message_returns_whole_input() {
- let md = "## solo\n\nthe whole body";
- let slices = super::split_response_markdown_per_message_with_hint(md, 1, None).unwrap();
- assert_eq!(slices, vec![md.to_string()]);
-}
-
-#[test]
-fn split_with_hint_rejects_when_subjects_dont_match() {
- let md = "## Foo\nbody1\n---\n## Bar\nbody2";
- let hints = vec![
- json!({"subject": "Completely different subject A"}),
- json!({"subject": "Completely different subject B"}),
- ];
- let out = super::split_response_markdown_per_message_with_hint(md, 2, Some(&hints));
- assert!(out.is_none(), "subject mismatch must force fallback");
-}
-
-#[test]
-fn split_with_hint_accepts_when_subjects_match() {
- let md = "## Welcome to Gmail\nbody1\n---\n## Your invoice\nbody2";
- let hints = vec![
- json!({"subject": "Welcome to Gmail"}),
- json!({"subject": "Your invoice"}),
- ];
- let slices = super::split_response_markdown_per_message_with_hint(md, 2, Some(&hints)).unwrap();
- assert_eq!(slices.len(), 2);
- assert!(slices[0].contains("Welcome to Gmail"));
- assert!(slices[1].contains("Your invoice"));
-}
-
-#[test]
-fn split_with_hint_skips_messages_with_blank_subject() {
- let md = "## A\nbody1\n---\n## B\nbody2";
- let hints = vec![json!({"subject": "A"}), json!({"subject": ""})];
- let slices = super::split_response_markdown_per_message_with_hint(md, 2, Some(&hints)).unwrap();
- assert_eq!(slices.len(), 2);
-}
-
-// ── format_email_local_time ──────────────────────────────────────────────────
-
-#[test]
-fn format_email_local_time_returns_none_for_unparseable_date() {
- assert!(super::format_email_local_time("not-a-date").is_none());
- assert!(super::format_email_local_time("").is_none());
-}
-
-#[test]
-fn format_email_local_time_preserves_utc_raw_date_in_reshape() {
- let mut v = json!({
- "messages": [{
- "messageId": "m1",
- "threadId": "t1",
- "subject": "Test",
- "sender": "a@example.com",
- "to": "b@example.com",
- "messageTimestamp": "2026-05-31T10:33:00Z",
- "labelIds": [],
- "messageText": "body",
- "payload": {}
- }]
- });
- post_process("GMAIL_FETCH_EMAILS", None, &mut v);
- let msg = &v["messages"][0];
- assert_eq!(msg["date"], "2026-05-31T10:33:00Z");
-}
-
-#[test]
-fn parse_email_date_accepts_rfc3339_and_rfc2822() {
- assert!(super::parse_email_date("2026-05-31T10:33:00Z").is_some());
- assert!(super::parse_email_date("Sun, 31 May 2026 10:33:00 +0000").is_some());
- assert!(super::parse_email_date("not-a-date").is_none());
-}
-
-#[test]
-fn format_at_tz_deterministic_with_fixed_offset() {
- use chrono::FixedOffset;
-
- let utc = super::parse_email_date("2026-05-31T10:33:00Z").unwrap();
-
- let est = FixedOffset::west_opt(5 * 3600).unwrap();
- let result = super::format_at_tz(utc, &est).unwrap();
- assert_eq!(result, "2026-05-31 05:33 AM -05:00");
-
- let ist = FixedOffset::east_opt(5 * 3600 + 1800).unwrap();
- let result = super::format_at_tz(utc, &ist).unwrap();
- assert_eq!(result, "2026-05-31 04:03 PM +05:30");
-}
-
-#[test]
-fn format_at_tz_returns_none_for_utc() {
- let utc = super::parse_email_date("2026-05-31T10:33:00Z").unwrap();
- let utc_tz = chrono::FixedOffset::east_opt(0).unwrap();
- assert!(super::format_at_tz(utc, &utc_tz).is_none());
-}
-
-#[test]
-fn apply_response_level_markdown_stashes_per_message_field() {
- let mut data = json!({
- "messages": [
- {"messageId": "m1", "subject": "Hello"},
- {"messageId": "m2", "subject": "World"},
- ]
- });
- let top_md = "## Hello\nbody A — link https://gh.io/abc\n---\n## World\nbody B";
- super::apply_response_level_markdown(&mut data, top_md);
- let m1 = data["messages"][0]["markdownFormatted"].as_str().unwrap();
- let m2 = data["messages"][1]["markdownFormatted"].as_str().unwrap();
- assert!(m1.contains("Hello"));
- assert!(
- m1.contains("https://gh.io/abc"),
- "shortened URL must survive"
- );
- assert!(m2.contains("World"));
- assert!(!m1.contains("World"), "no cross-message bleed");
-}
diff --git a/crates/tinymemory-integrations/src/sources/composio/linear/mod.rs b/crates/tinymemory-integrations/src/sources/composio/linear/mod.rs
deleted file mode 100644
index ea52092a..00000000
--- a/crates/tinymemory-integrations/src/sources/composio/linear/mod.rs
+++ /dev/null
@@ -1,79 +0,0 @@
-//! Linear host normalization helpers — result extraction and issue-title and
-//! timestamp extraction.
-//!
-//! Linear's GraphQL API (and therefore Composio's wrapping of it) returns
-//! connection-style lists (`{ nodes: [...], pageInfo: {...} }`) at the top
-//! level or nested under `data`. The functions here walk the union of
-//! common shapes so the provider does not have to branch per Composio
-//! envelope variant.
-
-use serde_json::Value;
-
-use super::fields::pick_str;
-
-/// Walk the Composio response envelope for Linear issue list results.
-///
-/// Linear's list endpoints return `{ nodes: [...] }` or
-/// `{ issues: { nodes: [...] } }` shapes; Composio may re-wrap the
-/// upstream payload under `data` or `data.data`. We probe each shape
-/// in order and return the first array we find.
-pub fn extract_issues(data: &Value) -> Vec {
- let candidates = [
- data.pointer("/data/nodes"),
- data.pointer("/nodes"),
- data.pointer("/data/issues/nodes"),
- data.pointer("/issues/nodes"),
- data.pointer("/data/data/nodes"),
- data.pointer("/data/data/issues/nodes"),
- data.pointer("/data/results"),
- data.pointer("/results"),
- data.pointer("/data/items"),
- data.pointer("/items"),
- ];
- for cand in candidates.into_iter().flatten() {
- if let Some(arr) = cand.as_array() {
- return arr.clone();
- }
- }
- Vec::new()
-}
-
-/// Extract a human-readable title from a Linear issue object.
-///
-/// Linear issues store the name at `title` (or `data.title` after
-/// Composio envelope wrapping). Falls back to `name` / `identifier`
-/// so the chunk remains identifiable even for unusual response shapes.
-pub fn extract_issue_title(issue: &Value) -> Option {
- pick_str(
- issue,
- &[
- "title",
- "data.title",
- "name",
- "data.name",
- "identifier",
- "data.identifier",
- ],
- )
-}
-
-/// Extract a stable cursor timestamp from a Linear issue object.
-///
-/// Linear uses ISO-8601 strings for timestamps (`updatedAt`). We keep
-/// the value as a string so lexicographic comparison against the stored
-/// cursor is valid.
-pub fn extract_issue_updated(issue: &Value) -> Option {
- pick_str(
- issue,
- &[
- "updatedAt",
- "data.updatedAt",
- "updated_at",
- "data.updated_at",
- ],
- )
-}
-
-#[cfg(test)]
-#[path = "mod_tests.rs"]
-mod tests;
diff --git a/crates/tinymemory-integrations/src/sources/composio/linear/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/linear/mod_tests.rs
deleted file mode 100644
index c0acff63..00000000
--- a/crates/tinymemory-integrations/src/sources/composio/linear/mod_tests.rs
+++ /dev/null
@@ -1,93 +0,0 @@
-//! Tests for the Linear normaliser.
-
-use super::*;
-use serde_json::json;
-
-// ── extract_issues ───────────────────────────────────────────────
-
-#[test]
-fn extract_issues_from_data_nodes() {
- let data = json!({ "data": { "nodes": [{"id": "i1"}, {"id": "i2"}] } });
- assert_eq!(extract_issues(&data).len(), 2);
-}
-
-#[test]
-fn extract_issues_from_top_level_nodes() {
- let data = json!({ "nodes": [{"id": "i3"}] });
- assert_eq!(extract_issues(&data).len(), 1);
-}
-
-#[test]
-fn extract_issues_from_data_issues_nodes() {
- let data =
- json!({ "data": { "issues": { "nodes": [{"id": "i4"}, {"id": "i5"}, {"id": "i6"}] } } });
- assert_eq!(extract_issues(&data).len(), 3);
-}
-
-#[test]
-fn extract_issues_from_top_level_issues_nodes() {
- let data = json!({ "issues": { "nodes": [{"id": "i7"}] } });
- assert_eq!(extract_issues(&data).len(), 1);
-}
-
-#[test]
-fn extract_issues_from_doubly_nested_issues_nodes() {
- let data =
- json!({ "data": { "data": { "issues": { "nodes": [{"id": "i8"}, {"id": "i9"}] } } } });
- assert_eq!(extract_issues(&data).len(), 2);
-}
-
-#[test]
-fn extract_issues_from_results() {
- let data = json!({ "results": [{"id": "i7"}] });
- assert_eq!(extract_issues(&data).len(), 1);
-}
-
-#[test]
-fn extract_issues_empty_when_missing() {
- let data = json!({ "foo": "bar" });
- assert!(extract_issues(&data).is_empty());
-}
-
-// ── extract_issue_title ──────────────────────────────────────────
-
-#[test]
-fn extract_issue_title_from_title_field() {
- let issue = json!({ "id": "i1", "title": "Fix the login bug" });
- assert_eq!(
- extract_issue_title(&issue),
- Some("Fix the login bug".into())
- );
-}
-
-#[test]
-fn extract_issue_title_falls_back_to_wrapped_data() {
- let issue = json!({ "data": { "title": "Wrapped issue" } });
- assert_eq!(extract_issue_title(&issue), Some("Wrapped issue".into()));
-}
-
-#[test]
-fn extract_issue_title_falls_back_to_identifier() {
- let issue = json!({ "identifier": "ENG-42" });
- assert_eq!(extract_issue_title(&issue), Some("ENG-42".into()));
-}
-
-// ── extract_issue_updated ────────────────────────────────────────
-
-#[test]
-fn extract_issue_updated_from_updated_at() {
- let issue = json!({ "updatedAt": "2026-03-01T12:00:00.000Z" });
- assert_eq!(
- extract_issue_updated(&issue),
- Some("2026-03-01T12:00:00.000Z".to_string())
- );
-}
-
-#[test]
-fn extract_issue_updated_falls_back_to_snake_case() {
- let issue = json!({ "data": { "updated_at": "2026-01-15T08:30:00.000Z" } });
- assert_eq!(
- extract_issue_updated(&issue),
- Some("2026-01-15T08:30:00.000Z".to_string())
- );
-}
diff --git a/crates/tinymemory-integrations/src/sources/composio/mod.rs b/crates/tinymemory-integrations/src/sources/composio/mod.rs
deleted file mode 100644
index 4df32030..00000000
--- a/crates/tinymemory-integrations/src/sources/composio/mod.rs
+++ /dev/null
@@ -1,35 +0,0 @@
-//! Composio toolkit payloads: normalisers and the mapping to `StoreItem`s.
-//!
-//! A host runs Composio actions with its own credentials and hands the raw
-//! responses here. Two layers turn them into memory:
-//!
-//! 1. **Normalisers**, one module per toolkit, are pure
-//! `serde_json::Value` transforms: they walk Composio's envelope variants
-//! and pull out the tasks, issues, pages or messages
-//! ([`clickup`], [`github`], [`linear`], [`notion`]), or rewrite a verbose
-//! response into a slim one in place ([`gmail_post_process`],
-//! [`slack_post_process`]). [`fields`] holds the path lookup they share.
-//! 2. [`normalise_payload`] turns one (post-processed) response into
-//! [`ComposioDocument`]s, and [`payload_items`] turns those into
-//! [`StoreItem::Document`](tinymemory_api::StoreItem::Document)s with
-//! `source.kind = Composio`, `source.id` = the connection or source id,
-//! `url` and `observed_at` where the payload has them, and
-//! `tags = [toolkit]`.
-//!
-//! Nothing here holds a credential, opens a socket or decides when to sync.
-//!
-//! One caveat on "pure": [`gmail_post_process::format_email_local_time`]
-//! renders in `chrono::Local`, so it reads the host's timezone. The raw UTC
-//! fields are preserved alongside, so ordering and identity stay UTC-based.
-
-pub mod clickup;
-pub mod fields;
-pub mod github;
-pub mod gmail_post_process;
-pub mod linear;
-pub mod notion;
-pub mod slack_post_process;
-
-mod documents;
-
-pub use documents::{ComposioDocument, normalise_payload, payload_items};
diff --git a/crates/tinymemory-integrations/src/sources/composio/notion/mod.rs b/crates/tinymemory-integrations/src/sources/composio/notion/mod.rs
deleted file mode 100644
index 00c1df50..00000000
--- a/crates/tinymemory-integrations/src/sources/composio/notion/mod.rs
+++ /dev/null
@@ -1,88 +0,0 @@
-//! Notion host normalization helpers — result extraction, page markdown and
-//! page title extraction.
-
-use serde_json::Value;
-
-use super::fields::pick_str;
-
-/// Walk the Composio response envelope for Notion page results.
-pub fn extract_results(data: &Value) -> Vec {
- let candidates = [
- data.pointer("/data/results"),
- data.pointer("/results"),
- data.pointer("/data/data/results"),
- data.pointer("/data/items"),
- data.pointer("/items"),
- ];
- for cand in candidates.into_iter().flatten() {
- if let Some(arr) = cand.as_array() {
- return arr.clone();
- }
- }
- Vec::new()
-}
-
-/// Extract the rendered page body markdown from a `NOTION_GET_PAGE_MARKDOWN`
-/// response. Composio wraps action output in varying envelope shapes, so we
-/// try the common locations tolerantly and return the first non-empty string.
-/// Returns `None` if no markdown field is found (caller falls back to the
-/// metadata-only body and logs the raw shape for diagnosis).
-pub fn extract_page_markdown(data: &Value) -> Option {
- const PATHS: &[&str] = &[
- "/markdown",
- "/data/markdown",
- "/data/response_data/markdown",
- "/response_data/markdown",
- "/data/content",
- "/content",
- "/data/markdown_content",
- "/markdown_content",
- "/text",
- "/data/text",
- ];
- for p in PATHS {
- if let Some(s) = data.pointer(p).and_then(Value::as_str)
- && !s.trim().is_empty()
- {
- return Some(s.to_string());
- }
- }
- None
-}
-
-/// Try to extract a human-readable title from a Notion page object.
-///
-/// Notion pages store the title in `properties.title` or
-/// `properties.Name.title[0].plain_text`. We try several shapes.
-pub fn extract_page_title(page: &Value) -> Option {
- // Try the common `properties.title.title[0].plain_text` shape.
- let props = page
- .get("properties")
- .or_else(|| page.get("data")?.get("properties"));
- if let Some(props) = props {
- // Walk all properties looking for a "title" type field.
- if let Some(obj) = props.as_object() {
- for (_key, val) in obj {
- if val.get("type").and_then(Value::as_str) == Some("title")
- && let Some(arr) = val.get("title").and_then(Value::as_array)
- {
- let text: String = arr
- .iter()
- .filter_map(|t| t.get("plain_text").and_then(Value::as_str))
- .collect::>()
- .join("");
- if !text.is_empty() {
- return Some(text);
- }
- }
- }
- }
- }
-
- // Fallback: top-level "title" field (some Composio shapes).
- pick_str(page, &["title", "data.title", "name", "data.name"])
-}
-
-#[cfg(test)]
-#[path = "mod_tests.rs"]
-mod tests;
diff --git a/crates/tinymemory-integrations/src/sources/composio/notion/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/notion/mod_tests.rs
deleted file mode 100644
index ab2e79bc..00000000
--- a/crates/tinymemory-integrations/src/sources/composio/notion/mod_tests.rs
+++ /dev/null
@@ -1,111 +0,0 @@
-//! Tests for the Notion normaliser.
-
-use super::*;
-use serde_json::json;
-
-#[test]
-fn extract_results_from_data_results() {
- let data = json!({"data": {"results": [{"id": "page1"}]}});
- let results = extract_results(&data);
- assert_eq!(results.len(), 1);
-}
-
-#[test]
-fn extract_page_markdown_reads_top_level_field() {
- // Matches the live GET_PAGE_MARKDOWN envelope observed empirically:
- // {id, markdown, object, request_id, truncated, unknown_block_ids}.
- let data = json!({
- "id": "p1",
- "markdown": "# Heading\n\nbody text",
- "object": "page",
- "truncated": false,
- });
- assert_eq!(
- extract_page_markdown(&data).as_deref(),
- Some("# Heading\n\nbody text")
- );
-}
-
-#[test]
-fn extract_page_markdown_reads_nested_envelope() {
- let data = json!({ "data": { "markdown": "nested body" } });
- assert_eq!(extract_page_markdown(&data).as_deref(), Some("nested body"));
-}
-
-#[test]
-fn extract_page_markdown_none_for_empty_or_missing() {
- // Empty markdown (a DB row with no body blocks) → None → metadata-only.
- assert_eq!(extract_page_markdown(&json!({ "markdown": "" })), None);
- assert_eq!(extract_page_markdown(&json!({ "markdown": " " })), None);
- // No markdown field at all → None.
- assert_eq!(extract_page_markdown(&json!({ "id": "p1" })), None);
-}
-
-#[test]
-fn extract_results_from_top_level() {
- let data = json!({"results": [{"id": "a"}, {"id": "b"}]});
- let results = extract_results(&data);
- assert_eq!(results.len(), 2);
-}
-
-#[test]
-fn extract_results_from_data_items() {
- let data = json!({"data": {"items": [{"id": "x"}]}});
- let results = extract_results(&data);
- assert_eq!(results.len(), 1);
-}
-
-#[test]
-fn extract_results_empty_when_no_match() {
- let data = json!({"foo": "bar"});
- assert!(extract_results(&data).is_empty());
-}
-
-#[test]
-fn extract_page_title_from_properties_title_type() {
- let page = json!({
- "properties": {
- "Name": {
- "type": "title",
- "title": [{"plain_text": "Hello"}, {"plain_text": " World"}]
- }
- }
- });
- assert_eq!(extract_page_title(&page), Some("Hello World".into()));
-}
-
-#[test]
-fn extract_page_title_from_nested_data_properties() {
- let page = json!({
- "data": {
- "properties": {
- "Title": {
- "type": "title",
- "title": [{"plain_text": "My Page"}]
- }
- }
- }
- });
- assert_eq!(extract_page_title(&page), Some("My Page".into()));
-}
-
-#[test]
-fn extract_page_title_fallback_to_top_level_title() {
- let page = json!({"title": "Fallback Title"});
- assert_eq!(extract_page_title(&page), Some("Fallback Title".into()));
-}
-
-#[test]
-fn extract_page_title_none_when_empty() {
- let page = json!({"properties": {"Name": {"type": "title", "title": []}}});
- // Empty title array means no text
- assert!(
- extract_page_title(&page).is_none() || extract_page_title(&page) == Some(String::new())
- );
-}
-
-#[test]
-fn extract_page_title_none_when_no_title_field() {
- let page = json!({"id": "123"});
- assert!(extract_page_title(&page).is_none());
-}
diff --git a/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod.rs b/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod.rs
deleted file mode 100644
index 764b885a..00000000
--- a/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod.rs
+++ /dev/null
@@ -1,324 +0,0 @@
-//! Slack-specific post-processing of Composio action responses.
-//!
-//! Composio's Slack responses are verbose API envelopes. This module
-//! rewrites each supported action's response into a slim, stable shape
-//! that the ingest pipeline and enrichers can consume without walking
-//! Composio's unstable nested envelopes.
-//!
-//! ## Supported slugs
-//!
-//! - `SLACK_FETCH_CONVERSATION_HISTORY` — reshapes into top-level
-//! `messages[]` with `{ ts, user, text, thread_ts, channel_id }`.
-//! Empty-text messages are dropped. `channel_id` is absent here (it's
-//! in the request, not the response); the caller injects it via the
-//! enricher in the host's `SlackSyncPipeline`.
-//!
-//! - `SLACK_LIST_CONVERSATIONS` — reshapes into top-level `channels[]`
-//! with `{ id, name, is_private }` per channel. Entries with an empty
-//! id are dropped.
-//!
-//! - `SLACK_SEARCH_MESSAGES` — reshapes `messages.matches[]` (possibly
-//! nested) into top-level `messages[]` with `{ ts, user, text,
-//! thread_ts, channel_id }`. `channel_id` is pulled from each match's
-//! `channel.id` field. `paging.pages` is preserved at top-level for
-//! caller pagination.
-//!
-//! ## Design note: user-id resolution is NOT here
-//!
-//! `SlackUsers` is a per-sync cache built from a separate API call —
-//! not a function of any individual response. Resolving user ids
-//! happens in the host's `SlackSyncPipeline` (the enricher layer), keeping
-//! this module purely data-shape–oriented.
-//! This matches Gmail's pattern of "post_process is data-only".
-//!
-//! Unknown slugs are silently no-ops so new Composio actions don't
-//! break the provider.
-
-use serde_json::{Map, Value};
-
-/// Entry point a host calls on each Slack action response (the slug names
-/// the action) before handing it to `normalise_payload`.
-///
-/// Dispatches on the Composio action slug and rewrites `data` in place.
-/// Unknown slugs are silently ignored.
-pub fn post_process(slug: &str, _arguments: Option<&Value>, data: &mut Value) {
- log::debug!("[composio:slack][post-process] slug={slug}");
- match slug {
- "SLACK_FETCH_CONVERSATION_HISTORY" => reshape_fetch_history(data),
- "SLACK_LIST_CONVERSATIONS" => reshape_list_conversations(data),
- "SLACK_SEARCH_MESSAGES" => reshape_search_messages(data),
- _ => {
- log::debug!("[composio:slack][post-process] unknown slug={slug}, passing through");
- }
- }
-}
-
-// ─── SLACK_FETCH_CONVERSATION_HISTORY ──────────────────────────────────────
-
-/// Rewrite a `SLACK_FETCH_CONVERSATION_HISTORY` response in place.
-///
-/// Walks possible nested envelopes (`/data/messages`, `/messages`,
-/// `/data/data/messages`) to find the raw messages array, drops messages
-/// with empty `text`, and emits a slim `{ ts, user, text, thread_ts }`
-/// shape under a top-level `messages[]` key. The consumed nested array is
-/// removed from the payload so the raw verbose rows don't linger alongside
-/// the slim copy. The caller injects `channel_id` via
-/// the host's Slack sync pipeline.
-fn reshape_fetch_history(data: &mut Value) {
- let arr = take_array(
- data,
- &["/data/messages", "/messages", "/data/data/messages"],
- 0,
- );
- let slim: Vec = arr.into_iter().filter_map(slim_history_message).collect();
- with_object(data, |obj| {
- obj.insert("messages".to_string(), Value::Array(slim));
- });
- log::debug!("[composio:slack][post-process] SLACK_FETCH_CONVERSATION_HISTORY reshaped");
-}
-
-fn slim_history_message(raw: Value) -> Option {
- let text = raw
- .get("text")
- .and_then(|v| v.as_str())
- .unwrap_or("")
- .trim();
- if text.is_empty() {
- return None;
- }
- let mut out = Map::new();
- // `ts` is required: without it a caller can neither cursor nor archive.
- out.insert("ts".into(), raw.get("ts")?.clone());
- if let Some(user) = raw.get("user").or_else(|| raw.get("bot_id")) {
- out.insert("user".into(), user.clone());
- }
- out.insert("text".into(), Value::String(text.to_string()));
- if let Some(thread_ts) = raw.get("thread_ts") {
- out.insert("thread_ts".into(), thread_ts.clone());
- }
- if let Some(permalink) = raw.get("permalink") {
- out.insert("permalink".into(), permalink.clone());
- }
- Some(Value::Object(out))
-}
-
-/// Find the first array at any of `candidates`, remove that field (plus
-/// `envelope_depth` ancestor object envelopes) from `data`, and return the
-/// array. Removing the consumed nested payload keeps the reshaped output from
-/// carrying duplicate raw rows.
-fn take_array(data: &mut Value, candidates: &[&str], envelope_depth: usize) -> Vec {
- for path in candidates {
- let arr = match data.pointer(path).and_then(|v| v.as_array().cloned()) {
- Some(a) => a,
- None => continue,
- };
- let mut remove_path = path.to_string();
- for _ in 0..envelope_depth {
- remove_path = match remove_path.rsplit_once('/') {
- Some((parent, _)) => parent.to_string(),
- None => break,
- };
- }
- remove_nested(data, &remove_path);
- return arr;
- }
- Vec::new()
-}
-
-/// Remove the field at `path` from `data`, pruning any ancestor object that
-/// the removal left empty so a consumed `data` envelope disappears entirely
-/// instead of lingering as `{}`.
-fn remove_nested(data: &mut Value, path: &str) {
- let segments: Vec<&str> = path
- .trim_start_matches('/')
- .split('/')
- .filter(|s| !s.is_empty())
- .collect();
- if segments.is_empty() {
- return;
- }
-
- // Remove the leaf field.
- let mut current = &mut *data;
- for seg in &segments[..segments.len() - 1] {
- current = match current.get_mut(*seg) {
- Some(next) => next,
- None => return,
- };
- }
- if let Value::Object(map) = current {
- map.remove(segments[segments.len() - 1]);
- }
-
- // Prune empty object ancestors, deepest first.
- for depth in (0..segments.len().saturating_sub(1)).rev() {
- // Re-walk to the object at `segments[..=depth]`.
- let mut ancestor = &mut *data;
- for seg in &segments[..=depth] {
- ancestor = match ancestor.get_mut(*seg) {
- Some(next) => next,
- None => return,
- };
- }
- if !matches!(ancestor, Value::Object(m) if m.is_empty()) {
- break;
- }
- // Remove it from its parent (`segments[..depth]`). For `depth == 0`
- // the parent is the top-level object, so an emptied `data` envelope
- // key disappears entirely.
- let mut parent = &mut *data;
- for seg in &segments[..depth] {
- parent = match parent.get_mut(*seg) {
- Some(next) => next,
- None => return,
- };
- }
- if let Value::Object(map) = parent {
- map.remove(segments[depth]);
- }
- }
-}
-
-// ─── SLACK_LIST_CONVERSATIONS ───────────────────────────────────────────────
-
-/// Rewrite a `SLACK_LIST_CONVERSATIONS` response in place.
-///
-/// Reshapes into a top-level `channels[]` with `{ id, name, is_private }`
-/// per channel; entries with an empty id are dropped.
-fn reshape_list_conversations(data: &mut Value) {
- let arr = take_array(
- data,
- &[
- "/data/channels",
- "/channels",
- "/data/data/channels",
- "/data/conversations",
- "/conversations",
- ],
- 0,
- );
-
- let slim: Vec = arr.into_iter().filter_map(slim_channel).collect();
- with_object(data, |obj| {
- obj.insert("channels".to_string(), Value::Array(slim));
- });
- log::debug!("[composio:slack][post-process] SLACK_LIST_CONVERSATIONS reshaped");
-}
-
-fn slim_channel(raw: Value) -> Option {
- let id = raw.get("id").and_then(|v| v.as_str()).unwrap_or("").trim();
- if id.is_empty() {
- return None;
- }
- let name = raw
- .get("name")
- .and_then(|v| v.as_str())
- .unwrap_or(id)
- .trim();
- let is_private = raw
- .get("is_private")
- .and_then(|v| v.as_bool())
- .unwrap_or(false);
- Some(Value::Object({
- let mut m = Map::new();
- m.insert("id".into(), Value::String(id.to_string()));
- m.insert("name".into(), Value::String(name.to_string()));
- m.insert("is_private".into(), Value::Bool(is_private));
- m
- }))
-}
-
-// ─── SLACK_SEARCH_MESSAGES ──────────────────────────────────────────────────
-
-/// Rewrite a `SLACK_SEARCH_MESSAGES` response in place.
-///
-/// Reshapes `messages.matches[]` (possibly nested under one or two
-/// `data` envelopes) into top-level `messages[]`. `channel_id` is pulled
-/// from each match's `channel.id` field. `paging.pages` is preserved at
-/// top-level under `pages` for the caller to drive pagination.
-fn reshape_search_messages(data: &mut Value) {
- // Preserve paging info before mutating data (take_array below removes the
- // envelope that carries it).
- let pages = [
- data.pointer("/data/messages/paging/pages"),
- data.pointer("/messages/paging/pages"),
- data.pointer("/data/data/messages/paging/pages"),
- ]
- .into_iter()
- .flatten()
- .find_map(|v| v.as_u64())
- .unwrap_or(1);
-
- // Envelope depth 1 removes the `messages` object (matches + paging) that
- // held the consumed rows, not just the `matches` array.
- let arr = take_array(
- data,
- &[
- "/data/messages/matches",
- "/messages/matches",
- "/data/data/messages/matches",
- ],
- 1,
- );
-
- let slim: Vec = arr.into_iter().filter_map(slim_search_match).collect();
- with_object(data, |obj| {
- obj.insert("messages".to_string(), Value::Array(slim));
- obj.insert("pages".to_string(), Value::Number(pages.into()));
- });
- log::debug!("[composio:slack][post-process] SLACK_SEARCH_MESSAGES reshaped");
-}
-
-fn slim_search_match(raw: Value) -> Option {
- let text = raw
- .get("text")
- .and_then(|v| v.as_str())
- .unwrap_or("")
- .trim();
- if text.is_empty() {
- return None;
- }
- let ts = raw.get("ts")?;
- let channel_id = raw
- .pointer("/channel/id")
- .and_then(|v| v.as_str())
- .unwrap_or("")
- .trim();
-
- let mut out = Map::new();
- out.insert("ts".into(), ts.clone());
- if let Some(user) = raw.get("user").or_else(|| raw.get("bot_id")) {
- out.insert("user".into(), user.clone());
- }
- out.insert("text".into(), Value::String(text.to_string()));
- if let Some(thread_ts) = raw.get("thread_ts") {
- out.insert("thread_ts".into(), thread_ts.clone());
- }
- if !channel_id.is_empty() {
- out.insert("channel_id".into(), Value::String(channel_id.to_string()));
- }
- if let Some(permalink) = raw.get("permalink") {
- out.insert("permalink".into(), permalink.clone());
- }
- Some(Value::Object(out))
-}
-
-// ─── Helpers ────────────────────────────────────────────────────────────────
-
-/// Edit `data` as a JSON object, replacing it with an empty object first if
-/// it is not one.
-///
-/// Takes the value out, edits the map, and puts it back, so there is no
-/// "re-borrow as an object" step that would need an `expect`.
-fn with_object(data: &mut Value, edit: impl FnOnce(&mut Map)) {
- let mut map = match std::mem::take(data) {
- Value::Object(map) => map,
- _ => Map::new(),
- };
- edit(&mut map);
- *data = Value::Object(map);
-}
-
-#[cfg(test)]
-#[path = "mod_tests.rs"]
-mod tests;
diff --git a/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod_tests.rs b/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod_tests.rs
deleted file mode 100644
index 73a6702b..00000000
--- a/crates/tinymemory-integrations/src/sources/composio/slack_post_process/mod_tests.rs
+++ /dev/null
@@ -1,306 +0,0 @@
-//! Tests for the Slack post-processor.
-
-use super::*;
-use serde_json::json;
-
-// ─── SLACK_FETCH_CONVERSATION_HISTORY ─────────────────────────────────────
-
-#[test]
-fn history_reshapes_top_level_messages() {
- let mut data = json!({
- "messages": [
- { "ts": "1714003200.000100", "user": "U1", "text": "hello" },
- { "ts": "1714003300.000200", "user": "U2", "text": "world", "thread_ts": "1714003200.0" },
- { "ts": "1714003400.000300", "user": "U3", "text": " " }, // dropped: empty text
- ],
- "response_metadata": { "next_cursor": "abc" }
- });
- post_process("SLACK_FETCH_CONVERSATION_HISTORY", None, &mut data);
-
- let msgs = data["messages"].as_array().unwrap();
- assert_eq!(msgs.len(), 2, "empty-text message must be dropped");
- assert_eq!(msgs[0]["ts"], "1714003200.000100");
- assert_eq!(msgs[0]["user"], "U1");
- assert_eq!(msgs[0]["text"], "hello");
- assert!(msgs[0].get("thread_ts").is_none());
- assert_eq!(msgs[1]["thread_ts"], "1714003200.0");
-}
-
-#[test]
-fn history_reshapes_nested_data_envelope() {
- let mut data = json!({
- "data": {
- "messages": [
- { "ts": "1714003200.0", "user": "U1", "text": "hi" }
- ]
- }
- });
- post_process("SLACK_FETCH_CONVERSATION_HISTORY", None, &mut data);
- let msgs = data["messages"].as_array().unwrap();
- assert_eq!(msgs.len(), 1);
- assert_eq!(msgs[0]["text"], "hi");
-}
-
-#[test]
-fn history_reshapes_doubly_nested_envelope() {
- let mut data = json!({
- "data": {
- "data": {
- "messages": [
- { "ts": "1714003200.0", "user": "U1", "text": "deep" }
- ]
- }
- }
- });
- post_process("SLACK_FETCH_CONVERSATION_HISTORY", None, &mut data);
- let msgs = data["messages"].as_array().unwrap();
- assert_eq!(msgs.len(), 1);
- assert_eq!(msgs[0]["text"], "deep");
-}
-
-#[test]
-fn history_drops_message_without_ts() {
- let mut data = json!({
- "messages": [
- { "user": "U1", "text": "no timestamp" },
- { "ts": "1714003200.0", "user": "U2", "text": "has ts" },
- ]
- });
- post_process("SLACK_FETCH_CONVERSATION_HISTORY", None, &mut data);
- let msgs = data["messages"].as_array().unwrap();
- assert_eq!(msgs.len(), 1);
- assert_eq!(msgs[0]["text"], "has ts");
-}
-
-#[test]
-fn history_removes_nested_envelope_after_reshape() {
- let mut data = json!({
- "data": {
- "messages": [
- { "ts": "1714003200.0", "user": "U1", "text": "hi" }
- ]
- }
- });
- post_process("SLACK_FETCH_CONVERSATION_HISTORY", None, &mut data);
-
- let msgs = data["messages"].as_array().unwrap();
- assert_eq!(msgs.len(), 1);
- assert_eq!(msgs[0]["text"], "hi");
- assert!(
- data.pointer("/data").is_none(),
- "consumed `data.messages` envelope must be removed, got: {data}"
- );
-}
-
-// ─── SLACK_LIST_CONVERSATIONS ─────────────────────────────────────────────
-
-#[test]
-fn list_conversations_reshapes_channels() {
- let mut data = json!({
- "data": {
- "channels": [
- { "id": "C1", "name": "eng", "is_private": false, "extra": "noise" },
- { "id": "G1", "name": "ops", "is_private": true },
- { "id": "", "name": "empty-id" }, // dropped
- ]
- }
- });
- post_process("SLACK_LIST_CONVERSATIONS", None, &mut data);
- let channels = data["channels"].as_array().unwrap();
- assert_eq!(channels.len(), 2, "empty-id entry must be dropped");
- assert_eq!(channels[0]["id"], "C1");
- assert_eq!(channels[0]["name"], "eng");
- assert_eq!(channels[0]["is_private"], false);
- assert!(
- channels[0].get("extra").is_none(),
- "noise fields must be removed"
- );
- assert_eq!(channels[1]["id"], "G1");
- assert_eq!(channels[1]["is_private"], true);
-}
-
-#[test]
-fn list_conversations_falls_back_to_conversations_key() {
- let mut data = json!({
- "conversations": [
- { "id": "C2", "name": "dev", "is_private": false }
- ]
- });
- post_process("SLACK_LIST_CONVERSATIONS", None, &mut data);
- let channels = data["channels"].as_array().unwrap();
- assert_eq!(channels.len(), 1);
- assert_eq!(channels[0]["id"], "C2");
- assert!(
- data.pointer("/conversations").is_none(),
- "consumed `conversations` field must be removed"
- );
-}
-
-// ─── SLACK_SEARCH_MESSAGES ────────────────────────────────────────────────
-
-#[test]
-fn search_messages_reshapes_matches() {
- let mut data = json!({
- "messages": {
- "matches": [
- {
- "ts": "1714003200.0",
- "user": "U1",
- "text": "hello from search",
- "channel": { "id": "C1" }
- },
- {
- "ts": "1714003300.0",
- "user": "U2",
- "text": " ", // dropped: whitespace only
- "channel": { "id": "C1" }
- },
- ],
- "paging": { "pages": 3 }
- }
- });
- post_process("SLACK_SEARCH_MESSAGES", None, &mut data);
- let msgs = data["messages"].as_array().unwrap();
- assert_eq!(msgs.len(), 1, "empty-text match must be dropped");
- assert_eq!(msgs[0]["ts"], "1714003200.0");
- assert_eq!(msgs[0]["text"], "hello from search");
- assert_eq!(msgs[0]["channel_id"], "C1");
- assert_eq!(data["pages"], 3, "paging.pages must be preserved");
-}
-
-#[test]
-fn search_messages_nested_data_envelope() {
- let mut data = json!({
- "data": {
- "messages": {
- "matches": [
- { "ts": "1714003200.0", "user": "U1", "text": "nested", "channel": { "id": "C2" } }
- ],
- "paging": { "pages": 1 }
- }
- }
- });
- post_process("SLACK_SEARCH_MESSAGES", None, &mut data);
- let msgs = data["messages"].as_array().unwrap();
- assert_eq!(msgs.len(), 1);
- assert_eq!(msgs[0]["channel_id"], "C2");
- assert_eq!(data["pages"], 1_u64);
-}
-
-#[test]
-fn search_messages_no_matches_emits_empty_array() {
- let mut data = json!({ "messages": { "matches": [] } });
- post_process("SLACK_SEARCH_MESSAGES", None, &mut data);
- let msgs = data["messages"].as_array().unwrap();
- assert!(msgs.is_empty());
-}
-
-#[test]
-fn search_messages_removes_nested_envelope_after_reshape() {
- let mut data = json!({
- "data": {
- "messages": {
- "matches": [
- { "ts": "1714003200.0", "user": "U1", "text": "nested", "channel": { "id": "C2" } }
- ],
- "paging": { "pages": 1 }
- }
- }
- });
- post_process("SLACK_SEARCH_MESSAGES", None, &mut data);
-
- let msgs = data["messages"].as_array().unwrap();
- assert_eq!(msgs.len(), 1);
- assert_eq!(msgs[0]["channel_id"], "C2");
- assert_eq!(data["pages"], 1_u64);
- assert!(
- data.pointer("/data").is_none(),
- "consumed `data.messages` envelope must be removed, got: {data}"
- );
-}
-
-#[test]
-fn search_messages_doubly_nested_paging_preserved() {
- let mut data = json!({
- "data": {
- "data": {
- "messages": {
- "matches": [
- { "ts": "1714003200.0", "user": "U1", "text": "deep", "channel": { "id": "C3" } }
- ],
- "paging": { "pages": 4 }
- }
- }
- }
- });
- post_process("SLACK_SEARCH_MESSAGES", None, &mut data);
-
- let msgs = data["messages"].as_array().unwrap();
- assert_eq!(msgs.len(), 1);
- assert_eq!(msgs[0]["text"], "deep");
- assert_eq!(
- data["pages"], 4_u64,
- "doubly-nested paging must be preserved"
- );
- assert!(
- data.pointer("/data").is_none(),
- "consumed `data.data.messages` envelope must be removed, got: {data}"
- );
-}
-
-// ─── Unknown slug ─────────────────────────────────────────────────────────
-
-#[test]
-fn unknown_slug_is_noop() {
- let mut data = json!({ "foo": "bar" });
- let original = data.clone();
- post_process("SLACK_SEND_MESSAGE", None, &mut data);
- assert_eq!(data, original, "unknown slug must not mutate data");
-}
-
-#[test]
-fn malformed_history_payload_becomes_an_empty_stable_shape() {
- for mut data in [json!(null), json!([]), json!({"messages": "wrong"})] {
- post_process("SLACK_FETCH_CONVERSATION_HISTORY", None, &mut data);
- assert_eq!(data, json!({"messages": []}));
- }
-}
-
-#[test]
-fn malformed_channel_rows_are_dropped_and_defaults_are_stable() {
- let mut data = json!({
- "channels": [
- null,
- "not an object",
- {"id": 7, "name": "numeric id"},
- {"id": " C1 ", "name": 99, "is_private": "yes"}
- ]
- });
- post_process("SLACK_LIST_CONVERSATIONS", None, &mut data);
- assert_eq!(
- data,
- json!({"channels": [{"id":"C1", "name":"C1", "is_private":false}]})
- );
-}
-
-#[test]
-fn malformed_search_rows_and_page_counts_fall_back_safely() {
- let mut data = json!({
- "messages": {
- "matches": [
- null,
- {"ts":"1.0", "text":"no channel"},
- {"channel":{"id":"C1"}, "text":"no timestamp"},
- {"ts":"2.0", "channel":{"id":"C1"}, "text":" valid "}
- ],
- "paging": {"pages":"many"}
- }
- });
- post_process("SLACK_SEARCH_MESSAGES", None, &mut data);
- assert_eq!(data["pages"], 1);
- let messages = data["messages"].as_array().unwrap();
- assert_eq!(messages.len(), 2);
- assert_eq!(messages[0]["text"], "no channel");
- assert!(messages[0].get("channel_id").is_none());
- assert_eq!(messages[1]["text"], "valid");
-}
diff --git a/crates/tinymemory-integrations/src/sources/fetch/mod.rs b/crates/tinymemory-integrations/src/sources/fetch/mod.rs
deleted file mode 100644
index 6063b5f7..00000000
--- a/crates/tinymemory-integrations/src/sources/fetch/mod.rs
+++ /dev/null
@@ -1,151 +0,0 @@
-//! Fetching one URL into a [`RawDocument`] or a link [`StoreItem`].
-//!
-//! A URL a user types is an SSRF vector: `http://169.254.169.254/` is a cloud
-//! metadata endpoint, `http://localhost:6379/` is somebody's Redis, and a
-//! hostname that resolves publicly on the first lookup can resolve to a private
-//! address on the second. Every fetch goes through the guard in [`ssrf`]: a
-//! scheme and host policy, a resolver that pins connections to globally
-//! routable addresses, and per-hop redirect re-checks. The RSS and web-page
-//! readers fetch through here too, each with its own body cap.
-//!
-//! No scheduling, no retries, no credentials, no robots.txt: this fetches one
-//! URL, once, when asked. Conversion to markdown is the `documents` module's.
-
-use crate::documents::{DocumentConverter, MAX_DOCUMENT_BYTES, RawDocument, document_item};
-use tinymemory_api::{MemoryMeta, SourceKind, StoreItem};
-
-use crate::sources::error::{Error, Result};
-use ssrf::{build_client, is_url_allowed, read_body_capped};
-
-pub mod ssrf;
-
-/// Fetch `url` and return its body as a [`RawDocument`].
-///
-/// The response's `Content-Type` becomes the document's declared MIME type and
-/// the URL becomes its origin; the last path segment, when it has an
-/// extension, becomes its filename so format detection can fall back to it.
-///
-/// # Errors
-///
-/// - [`Error::Invalid`] for a malformed URL, one the SSRF guard refuses, or an
-/// empty body.
-/// - [`Error::Unreachable`] when the request never completed or the body read
-/// was interrupted.
-/// - [`Error::Upstream`] for a non-success status.
-/// - [`Error::TooLarge`] for a body over [`MAX_DOCUMENT_BYTES`].
-pub async fn fetch_url(url: &str) -> Result {
- fetch_url_capped(url, MAX_DOCUMENT_BYTES as u64).await
-}
-
-/// [`fetch_url`] with a caller-chosen body cap, for the readers whose sources
-/// warrant a tighter one than [`MAX_DOCUMENT_BYTES`].
-///
-/// # Errors
-///
-/// As [`fetch_url`], with [`Error::TooLarge`] for a body over `max_bytes`.
-pub(crate) async fn fetch_url_capped(url: &str, max_bytes: u64) -> Result {
- let parsed = reqwest::Url::parse(url)
- .map_err(|error| Error::Invalid(format!("invalid url {url:?}: {error}")))?;
- if !is_url_allowed(&parsed) {
- return Err(Error::Invalid(format!(
- "url {url:?} is not an allowed fetch target"
- )));
- }
-
- let client = build_client().map_err(Error::Reader)?;
- tracing::debug!(
- host = %parsed.host_str().unwrap_or(""),
- "[memory_sources:fetch] fetching url"
- );
- let response = client
- .get(parsed.clone())
- .send()
- .await
- .map_err(|error| Error::Unreachable(format!("fetching {url:?}: {error}")))?;
-
- response_to_document(url, parsed, response, max_bytes).await
-}
-
-/// Fetch `url`, convert it through `converter`, and wrap it as a
-/// [`StoreItem::Document`] with `source.kind = Link` and `url` set.
-///
-/// `source_id` is the configured source's id, when the fetch is for one.
-///
-/// # Errors
-///
-/// Whatever [`fetch_url`] returns, plus [`Error::Document`] when conversion
-/// fails.
-pub async fn link_item(
- url: &str,
- source_id: Option,
- converter: &dyn DocumentConverter,
-) -> Result {
- let document = fetch_url(url).await?;
- let mut meta = MemoryMeta::from_source(SourceKind::Link, source_id);
- meta.url = document.origin.clone().or_else(|| Some(url.to_string()));
- Ok(document_item(converter, &document, meta).await?)
-}
-
-/// Validate and convert a completed HTTP response.
-async fn response_to_document(
- url: &str,
- parsed: reqwest::Url,
- response: reqwest::Response,
- max_bytes: u64,
-) -> Result {
- let status = response.status();
- if !status.is_success() {
- return Err(Error::Upstream(format!(
- "fetching {url:?} answered {status}"
- )));
- }
-
- let content_type = response
- .headers()
- .get(reqwest::header::CONTENT_TYPE)
- .and_then(|value| value.to_str().ok())
- .map(str::to_string);
-
- // The cap is applied while reading, not after: a body that would not fit is
- // one this process should never have finished buffering.
- let bytes = read_body_capped(response, max_bytes)
- .await
- .map_err(|error| read_error(url, &error))?;
-
- if bytes.is_empty() {
- return Err(Error::Invalid(format!("{url:?} returned no body")));
- }
-
- let mut document = RawDocument::new(bytes).with_origin(parsed.to_string());
- if let Some(content_type) = content_type {
- document = document.with_mime(content_type);
- }
- // A URL's last path segment is often the only filename there is, and format
- // detection falls back to it when the server sent no useful type.
- if let Some(name) = parsed
- .path_segments()
- .and_then(|mut segments| segments.next_back())
- .filter(|name| !name.is_empty() && name.contains('.'))
- {
- document = document.with_filename(name.to_string());
- }
- Ok(document)
-}
-
-/// Turn a `read_body_capped` failure into the right [`Error`] variant.
-///
-/// `read_body_capped` collapses two failures into one `String`: a body over
-/// the size cap, and a stream that failed mid-read. They need different retry
-/// policies, so this tells them apart by the message `read_body_capped` always
-/// uses for the size case.
-fn read_error(url: &str, error: &str) -> Error {
- if error.contains("exceeds") && error.contains("-byte limit") {
- Error::TooLarge(format!("reading {url:?}: {error}"))
- } else {
- Error::Unreachable(format!("reading {url:?}: {error}"))
- }
-}
-
-#[cfg(test)]
-#[path = "mod_tests.rs"]
-mod tests;
diff --git a/crates/tinymemory-integrations/src/sources/fetch/mod_tests.rs b/crates/tinymemory-integrations/src/sources/fetch/mod_tests.rs
deleted file mode 100644
index a42f4917..00000000
--- a/crates/tinymemory-integrations/src/sources/fetch/mod_tests.rs
+++ /dev/null
@@ -1,177 +0,0 @@
-//! Tests for URL fetching.
-//!
-//! The guard and argument handling are exercised directly; response handling
-//! runs against a loopback server the test controls. Nothing reaches the real
-//! network.
-
-use super::*;
-
-async fn local_response(response: impl Into>) -> reqwest::Response {
- use tokio::io::{AsyncReadExt, AsyncWriteExt};
-
- let listener = tokio::net::TcpListener::bind((std::net::Ipv4Addr::LOCALHOST, 0))
- .await
- .expect("bind controlled server");
- let address = listener.local_addr().expect("server address");
- let response = response.into();
- let server = tokio::spawn(async move {
- let (mut stream, _) = listener.accept().await.expect("accept request");
- let mut request = [0_u8; 1024];
- let _ = stream.read(&mut request).await.expect("read request");
- stream.write_all(&response).await.expect("write response");
- });
- let received = reqwest::get(format!("http://{address}/"))
- .await
- .expect("controlled response");
- server.await.expect("server task");
- received
-}
-
-#[tokio::test]
-async fn a_malformed_url_is_rejected_before_anything_is_fetched() {
- let error = fetch_url("not a url").await.unwrap_err();
- assert!(matches!(error, Error::Invalid(_)), "got {error:?}");
- assert!(error.to_string().contains("invalid url"), "got {error}");
-}
-
-#[tokio::test]
-async fn loopback_and_link_local_targets_are_refused() {
- for url in [
- "http://127.0.0.1/",
- "http://localhost:6379/",
- "http://169.254.169.254/latest/meta-data/",
- "http://[::1]/",
- ] {
- let error = fetch_url(url).await.unwrap_err();
- assert!(
- error.to_string().contains("not an allowed fetch target"),
- "{url} gave {error}"
- );
- }
-}
-
-#[tokio::test]
-async fn a_non_http_scheme_is_refused() {
- for url in [
- "file:///etc/passwd",
- "ftp://example.com/x",
- "gopher://example.com/",
- ] {
- let error = fetch_url(url).await.unwrap_err();
- assert!(matches!(error, Error::Invalid(_)), "{url} gave {error:?}");
- }
-}
-
-#[test]
-fn a_size_limit_failure_is_reported_as_budget_exceeded() {
- let error = read_error(
- "https://example.com/",
- "response body exceeds 8-byte limit (Content-Length=9)",
- );
- assert!(matches!(error, Error::TooLarge(_)), "got {error:?}");
-}
-
-#[test]
-fn an_interrupted_read_is_reported_as_unreachable_not_budget_exceeded() {
- let error = read_error(
- "https://example.com/",
- "failed to read response body: connection reset",
- );
- assert!(matches!(error, Error::Unreachable(_)), "got {error:?}");
-}
-
-#[tokio::test]
-async fn completed_response_preserves_body_type_origin_and_filename() {
- let response = local_response(
- b"HTTP/1.1 200 OK\r\nContent-Type: text/markdown; charset=utf-8\r\nContent-Length: 7\r\n\r\n# title",
- )
- .await;
- let url = reqwest::Url::parse("https://example.com/guides/readme.md").unwrap();
- let document = response_to_document(
- url.as_str(),
- url.clone(),
- response,
- MAX_DOCUMENT_BYTES as u64,
- )
- .await
- .unwrap();
- assert_eq!(document.bytes, b"# title");
- assert_eq!(document.origin.as_deref(), Some(url.as_str()));
- assert_eq!(document.filename.as_deref(), Some("readme.md"));
- assert_eq!(
- document.declared_mime.as_deref(),
- Some("text/markdown; charset=utf-8")
- );
-}
-
-#[tokio::test]
-async fn completed_response_handles_status_empty_body_and_filename_absence() {
- let response =
- local_response(b"HTTP/1.1 503 Service Unavailable\r\nContent-Length: 0\r\n\r\n").await;
- let url = reqwest::Url::parse("https://example.com/unavailable").unwrap();
- let error = response_to_document(
- url.as_str(),
- url.clone(),
- response,
- MAX_DOCUMENT_BYTES as u64,
- )
- .await
- .unwrap_err();
- assert!(matches!(error, Error::Upstream(_)));
-
- let response = local_response(b"HTTP/1.1 200 OK\r\nContent-Length: 0\r\n\r\n").await;
- let error = response_to_document(
- url.as_str(),
- url.clone(),
- response,
- MAX_DOCUMENT_BYTES as u64,
- )
- .await
- .unwrap_err();
- assert!(matches!(error, Error::Invalid(_)));
-
- let response = local_response(b"HTTP/1.1 200 OK\r\nContent-Length: 4\r\n\r\ntext").await;
- let document = response_to_document(
- url.as_str(),
- url.clone(),
- response,
- MAX_DOCUMENT_BYTES as u64,
- )
- .await
- .unwrap();
- assert_eq!(document.bytes, b"text");
- assert!(document.filename.is_none());
- assert!(document.declared_mime.is_none());
-}
-
-#[tokio::test]
-async fn completed_response_maps_declared_oversize_to_budget_exceeded() {
- let response = local_response(
- [
- b"HTTP/1.1 200 OK\r\nContent-Length: ",
- (MAX_DOCUMENT_BYTES + 1).to_string().as_bytes(),
- b"\r\n\r\n",
- ]
- .concat(),
- )
- .await;
- let url = reqwest::Url::parse("https://example.com/huge.bin").unwrap();
- let error = response_to_document(
- url.as_str(),
- url.clone(),
- response,
- MAX_DOCUMENT_BYTES as u64,
- )
- .await
- .unwrap_err();
- assert!(matches!(error, Error::TooLarge(_)));
-}
-
-#[tokio::test]
-async fn a_link_item_refuses_a_private_target_before_fetching() {
- let chain = crate::documents::ConverterChain::default();
- let error = link_item("http://127.0.0.1/", Some("src_link".into()), &chain)
- .await
- .unwrap_err();
- assert!(matches!(error, Error::Invalid(_)), "got {error:?}");
-}
diff --git a/crates/tinymemory-integrations/src/sources/fetch/ssrf/mod.rs b/crates/tinymemory-integrations/src/sources/fetch/ssrf/mod.rs
deleted file mode 100644
index 49c28453..00000000
--- a/crates/tinymemory-integrations/src/sources/fetch/ssrf/mod.rs
+++ /dev/null
@@ -1,231 +0,0 @@
-//! Shared SSRF guard and fetch hygiene for the network source readers.
-//!
-//! [`super::fetch_url`] and the web-page and RSS readers built on it all fetch
-//! user-configured URLs, so they share the policy in this module. It is public
-//! so a host fetching a user-supplied URL by other means applies the same
-//! policy rather than a second, weaker one.
-//!
-//! The hostname *text* check (`is_blocked_host`) rejects private IP literals
-//! (including their IPv4-mapped IPv6 forms, e.g. `::ffff:127.0.0.1`),
-//! `localhost`, `.local` / `.internal` names, and single-label hostnames, but
-//! a public-looking name can resolve to a loopback / private / link-local
-//! address (including the cloud-metadata `169.254.169.254`) at lookup time.
-//! `PublicOnlyResolver` therefore vets the resolved addresses and only lets the
-//! connection proceed to a globally routable IP, so the request is pinned to an
-//! address we have already allowed (no re-resolution between the check and the
-//! connect). Redirects are re-checked through `is_url_allowed` so a public URL
-//! cannot bounce the fetch onto an internal host.
-//!
-//! `read_body_capped` streams a response body and stops at a byte cap, so a
-//! hostile or gigantic page/feed cannot OOM the process before the size check
-//! runs.
-//!
-//! An IPv6 literal URL such as `http://[::1]/` is refused whatever its address,
-//! including a public one: `reqwest::Url::host_str` keeps the brackets, the
-//! text no longer parses as an IP address, and a name with no dot is treated
-//! as a single-label internal name. That is a fail-closed limitation, not a
-//! classification: the address classifier (`is_public_ip`) does handle IPv6
-//! (and IPv4-mapped forms) for *resolved* addresses, which is how a hostname
-//! with an AAAA record is still vetted.
-
-use std::net::{IpAddr, Ipv4Addr, SocketAddr};
-use std::sync::Arc;
-
-use futures::stream::StreamExt;
-use reqwest::dns::{Addrs, Name, Resolve, Resolving};
-
-/// Build an HTTP client with a redirect policy that re-applies the SSRF
-/// host/scheme check to every redirect hop, a DNS resolver that only yields
-/// globally routable addresses, a 20-second timeout and the `openhuman`
-/// user agent the network readers have always sent.
-///
-/// # Errors
-///
-/// A message naming the failure when the TLS backend cannot be initialised.
-pub fn build_client() -> Result {
- reqwest::Client::builder()
- .timeout(std::time::Duration::from_secs(20))
- .user_agent("openhuman")
- .redirect(reqwest::redirect::Policy::custom(|attempt| {
- if is_url_allowed(attempt.url()) {
- attempt.follow()
- } else {
- // `stop` returns the redirect response to the caller instead
- // of following it; the read then fails on the non-2xx status.
- attempt.stop()
- }
- }))
- .dns_resolver(Arc::new(PublicOnlyResolver))
- .build()
- .map_err(|e| format!("failed to build http client: {e}"))
-}
-
-/// Stream a response body, failing once it exceeds `max` bytes.
-///
-/// `Response::bytes()` buffers the entire body before any size check, so a
-/// server that omits or understates `Content-Length` (for example a chunked
-/// response) could OOM the process despite the cap. Reading incrementally
-/// enforces the limit while the bytes arrive.
-///
-/// # Errors
-///
-/// A message containing `exceeds {max}-byte limit` when the body is over the
-/// cap, or `failed to read response body` when the stream fails mid-read.
-pub async fn read_body_capped(resp: reqwest::Response, max: u64) -> Result, String> {
- // Trust a truthful Content-Length up front so a known-huge body is
- // rejected before the first byte is read.
- if let Some(len) = resp.content_length()
- && len > max
- {
- return Err(format!(
- "response body exceeds {max}-byte limit (Content-Length={len})"
- ));
- }
-
- let mut body = Vec::new();
- let mut stream = resp.bytes_stream();
- while let Some(chunk) = stream.next().await {
- let chunk = chunk.map_err(|e| format!("failed to read response body: {e}"))?;
- body.extend_from_slice(&chunk);
- if body.len() as u64 > max {
- return Err(format!(
- "response body exceeds {max}-byte limit (read {} bytes)",
- body.len()
- ));
- }
- }
- Ok(body)
-}
-
-/// A DNS resolver that only yields globally routable addresses.
-///
-/// The text-based `is_blocked_host` check rejects private IP *literals* and
-/// local hostnames, but a public-looking hostname can resolve to a loopback,
-/// private, link-local, or cloud-metadata address (`169.254.169.254`) at
-/// lookup time. Installing this resolver means reqwest connects to addresses
-/// we have already vetted: a hostname whose current resolution is non-public
-/// fails the request instead of silently reaching an internal service, and the
-/// validated address is the one the connection is pinned to (no re-resolution
-/// between the check and the connect).
-#[derive(Debug, Default)]
-struct PublicOnlyResolver;
-
-impl Resolve for PublicOnlyResolver {
- fn resolve(&self, name: Name) -> Resolving {
- let host = name.as_str().to_string();
- Box::pin(async move {
- let addrs: Vec = tokio::net::lookup_host((host.as_str(), 0))
- .await
- .map_err(box_err)?
- .filter(|addr| is_public_ip(addr.ip()))
- .collect();
- if addrs.is_empty() {
- return Err(box_err(std::io::Error::new(
- std::io::ErrorKind::AddrNotAvailable,
- format!("host {host} resolved to no public addresses"),
- )));
- }
- Ok(Box::new(addrs.into_iter()) as Addrs)
- })
- }
-}
-
-fn box_err(
- e: impl std::error::Error + Send + Sync + 'static,
-) -> Box {
- Box::new(e)
-}
-
-/// Whether `ip` is a globally routable address — the one address classifier
-/// behind both halves of the SSRF guard (literal hosts in `is_blocked_host`
-/// and resolved addresses in `PublicOnlyResolver`).
-///
-/// Not fetchable: loopback, private, link-local, unspecified, CGNAT
-/// (`100.64.0.0/10`), `192.0.0.0/16` (IETF protocol assignments and the
-/// `192.0.2.0/24` documentation range), multicast, broadcast, documentation
-/// (`198.51.100.0/24`, `203.0.113.0/24`, `2001:db8::/32`), benchmarking
-/// (`198.18.0.0/15`),
-/// reserved (`240.0.0.0/4`), and IPv6 unique-local (`fc00::/7`) and
-/// link-local (`fe80::/10`). An IPv6 address carrying an IPv4 one — mapped
-/// (`::ffff:a.b.c.d`) or the deprecated compatible form (`::a.b.c.d`) — is
-/// judged by its IPv4 part, so a mapped loopback stays blocked.
-fn is_public_ip(ip: IpAddr) -> bool {
- match ip {
- IpAddr::V4(v4) => {
- let o = v4.octets();
- !(v4.is_loopback()
- || v4.is_private()
- || v4.is_link_local()
- || v4.is_unspecified()
- || v4.is_multicast()
- || v4.is_broadcast()
- || (o[0] == 100 && o[1] & 0xc0 == 0x40)
- || (o[0] == 192 && o[1] == 0)
- || (o[0] == 198 && o[1] == 51 && o[2] == 100)
- || (o[0] == 203 && o[1] == 0 && o[2] == 113)
- || (o[0] == 198 && o[1] & 0xfe == 18)
- || o[0] >= 240)
- }
- IpAddr::V6(v6) => {
- if v6.is_loopback() || v6.is_unspecified() || v6.is_multicast() {
- return false;
- }
- let o = v6.octets();
- if (o[0] & 0xfe == 0xfc)
- || (o[0] == 0xfe && o[1] & 0xc0 == 0x80)
- || (o[0] == 0x20 && o[1] == 0x01 && o[2] == 0x0d && o[3] == 0xb8)
- {
- return false;
- }
- if let Some(v4) = v6.to_ipv4_mapped() {
- return is_public_ip(IpAddr::V4(v4));
- }
- // `to_ipv4_mapped` answers `None` for the compatible form, so
- // without this `::127.0.0.1` would read as public. `::` and `::1`
- // were judged as themselves above.
- if o[..12].iter().all(|byte| *byte == 0) {
- return is_public_ip(IpAddr::V4(Ipv4Addr::new(o[12], o[13], o[14], o[15])));
- }
- true
- }
- }
-}
-
-/// Whether a URL may be fetched: `http(s)` scheme against a public host.
-pub fn is_url_allowed(url: &reqwest::Url) -> bool {
- match url.scheme() {
- "http" | "https" => {}
- _ => return false,
- }
- let Some(host) = url.host_str() else {
- return false;
- };
- !is_blocked_host(host)
-}
-
-/// Reject hosts that could target non-public resources: IP literals in
-/// loopback / private / link-local / unique-local / unspecified ranges (and
-/// their IPv4-mapped IPv6 forms), plus `localhost`, `.local` / `.internal`
-/// names, and single-label hostnames (internal service names such as `mongo`
-/// or `redis`).
-fn is_blocked_host(host: &str) -> bool {
- let host = host.trim().trim_end_matches('.').to_ascii_lowercase();
- if host.is_empty() {
- return true;
- }
- if let Ok(ip) = host.parse::() {
- // A literal never goes through DNS resolution, so `PublicOnlyResolver`
- // never sees it — this text check is its only line of defense, and it
- // uses the same classification as the resolver.
- return !is_public_ip(ip);
- }
- if host == "localhost" || host.ends_with(".local") || host.ends_with(".internal") {
- return true;
- }
- // A single-label name is an internal-service name, not a public domain.
- !host.contains('.')
-}
-
-#[cfg(test)]
-#[path = "mod_tests.rs"]
-mod tests;
diff --git a/crates/tinymemory-integrations/src/sources/fetch/ssrf/mod_tests.rs b/crates/tinymemory-integrations/src/sources/fetch/ssrf/mod_tests.rs
deleted file mode 100644
index 98f02183..00000000
--- a/crates/tinymemory-integrations/src/sources/fetch/ssrf/mod_tests.rs
+++ /dev/null
@@ -1,236 +0,0 @@
-//! Tests for the SSRF guard: the address classifier, host and URL policy,
-//! the capped body reader and the hardened client.
-
-use super::*;
-
-async fn local_response(response: &'static [u8]) -> reqwest::Response {
- use tokio::io::{AsyncReadExt, AsyncWriteExt};
-
- let listener = tokio::net::TcpListener::bind((std::net::Ipv4Addr::LOCALHOST, 0))
- .await
- .expect("bind controlled server");
- let address = listener.local_addr().expect("server address");
- let server = tokio::spawn(async move {
- let (mut stream, _) = listener.accept().await.expect("accept request");
- let mut request = [0_u8; 1024];
- let _ = stream.read(&mut request).await.expect("read request");
- stream.write_all(response).await.expect("write response");
- });
- let received = reqwest::get(format!("http://{address}/"))
- .await
- .expect("controlled response");
- server.await.expect("server task");
- received
-}
-
-// ── SSRF guard ──────────────────────────────────────────────────────
-
-#[test]
-fn is_url_allowed_accepts_public_http_urls() {
- assert!(is_url_allowed(
- &reqwest::Url::parse("https://example.com").unwrap()
- ));
- assert!(is_url_allowed(
- &reqwest::Url::parse("http://example.com/x").unwrap()
- ));
- assert!(is_url_allowed(
- &reqwest::Url::parse("https://sub.example.com").unwrap()
- ));
- assert!(is_url_allowed(
- &reqwest::Url::parse("https://8.8.8.8").unwrap()
- ));
-}
-
-#[test]
-fn is_url_allowed_rejects_private_and_internal_targets() {
- // Private / loopback / link-local IP literals.
- assert!(!is_url_allowed(
- &reqwest::Url::parse("http://127.0.0.1").unwrap()
- ));
- assert!(!is_url_allowed(
- &reqwest::Url::parse("http://10.0.0.1").unwrap()
- ));
- assert!(!is_url_allowed(
- &reqwest::Url::parse("http://192.168.1.1").unwrap()
- ));
- assert!(!is_url_allowed(
- &reqwest::Url::parse("http://169.254.169.254").unwrap()
- ));
- assert!(!is_url_allowed(
- &reqwest::Url::parse("http://[::1]").unwrap()
- ));
- // Internal service names and local-only names.
- assert!(!is_url_allowed(
- &reqwest::Url::parse("http://localhost").unwrap()
- ));
- assert!(!is_url_allowed(
- &reqwest::Url::parse("http://mongo").unwrap()
- ));
- assert!(!is_url_allowed(
- &reqwest::Url::parse("http://service.internal").unwrap()
- ));
- // Non-http scheme.
- assert!(!is_url_allowed(
- &reqwest::Url::parse("ftp://example.com").unwrap()
- ));
- assert!(!is_url_allowed(
- &reqwest::Url::parse("file:///etc/passwd").unwrap()
- ));
-}
-
-#[test]
-fn is_blocked_host_rejects_ip_ranges_and_local_names() {
- let blocked = [
- "127.0.0.1",
- "0.0.0.0",
- "10.0.0.1",
- "172.16.0.1",
- "192.168.0.1",
- "169.254.169.254",
- "100.64.0.1", // CGNAT
- "192.0.0.1", // IETF protocol assignments
- "localhost",
- "foo.local",
- "bar.internal",
- "mongo",
- "::1",
- "fc00::1", // unique-local
- "fe80::1", // link-local
- "::ffff:127.0.0.1", // IPv4-mapped loopback literal
- "::ffff:10.0.0.1", // IPv4-mapped private literal
- "::ffff:169.254.169.254", // IPv4-mapped link-local / cloud metadata
- // Special IPv4 literals that are not globally routable: multicast,
- // broadcast, documentation, benchmarking, and reserved ranges. A
- // literal never goes through DNS resolution, so the text check is the
- // only line of defense for these.
- "224.0.0.1", // multicast
- "255.255.255.255", // broadcast
- "192.0.2.1", // documentation
- "198.51.100.1", // documentation
- "203.0.113.1", // documentation
- "198.18.0.1", // benchmarking
- "240.0.0.1", // reserved
- ];
- for host in blocked {
- assert!(is_blocked_host(host), "expected {host:?} to be blocked");
- }
-}
-
-#[test]
-fn is_blocked_host_accepts_public_hosts() {
- let allowed = [
- "8.8.8.8",
- "1.1.1.1",
- "example.com",
- "sub.example.com",
- "example.co.uk",
- "8.8.8.8.", // trailing dot is normalized away
- "EXAMPLE.com", // case-insensitive
- "2001:4860:4860::8888",
- "::ffff:8.8.8.8", // IPv4-mapped public literal
- ];
- for host in allowed {
- assert!(!is_blocked_host(host), "expected {host:?} to be allowed");
- }
-}
-
-// ── resolved-address (DNS) SSRF classification ──────────────────────
-
-fn public_ip(s: &str) -> IpAddr {
- s.parse().expect("valid ip literal")
-}
-
-#[test]
-fn is_public_ip_rejects_internal_and_special_ranges() {
- let blocked = [
- "127.0.0.1", // loopback
- "0.0.0.0", // unspecified
- "10.0.0.1", // private
- "172.16.0.1", // private
- "192.168.1.1", // private
- "169.254.169.254", // link-local / cloud metadata
- "100.64.0.1", // CGNAT
- "192.0.0.1", // IETF protocol assignments
- "224.0.0.1", // multicast
- "255.255.255.255", // broadcast
- "192.0.2.1", // documentation
- "198.51.100.1", // documentation
- "203.0.113.1", // documentation
- "198.18.0.1", // benchmarking
- "240.0.0.1", // reserved
- "::1", // loopback
- "::", // unspecified
- "fc00::1", // unique-local
- "fe80::1", // link-local
- "ff00::1", // multicast
- "2001:db8::1", // documentation
- "::ffff:127.0.0.1", // IPv4-mapped loopback
- "::ffff:169.254.169.254", // IPv4-mapped link-local
- ];
- for s in blocked {
- assert!(!is_public_ip(public_ip(s)), "expected {s:?} to be rejected");
- }
-}
-
-#[test]
-fn is_public_ip_accepts_global_addresses() {
- let allowed = [
- "8.8.8.8",
- "1.1.1.1",
- "93.184.216.34",
- "2001:4860:4860::8888",
- "2606:4700:4700::1111",
- "::ffff:8.8.8.8", // IPv4-mapped public
- ];
- for s in allowed {
- assert!(is_public_ip(public_ip(s)), "expected {s:?} to be allowed");
- }
-}
-
-#[tokio::test]
-async fn capped_body_reader_accepts_small_streams_and_enforces_both_size_paths() {
- let small = local_response(b"HTTP/1.1 200 OK\r\nContent-Length: 5\r\n\r\nhello").await;
- assert_eq!(read_body_capped(small, 5).await.unwrap(), b"hello");
-
- let declared = local_response(b"HTTP/1.1 200 OK\r\nContent-Length: 6\r\n\r\nabcdef").await;
- let error = read_body_capped(declared, 5)
- .await
- .expect_err("declared body exceeds cap");
- assert!(error.contains("Content-Length=6"));
-
- let chunked = local_response(
- b"HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\n\r\n3\r\nabc\r\n3\r\ndef\r\n0\r\n\r\n",
- )
- .await;
- let error = read_body_capped(chunked, 5)
- .await
- .expect_err("streamed body exceeds cap");
- assert!(error.contains("read 6 bytes"));
-}
-
-#[test]
-fn client_builder_installs_the_hardened_policy() {
- build_client().expect("hardened HTTP client builds");
-}
-
-#[test]
-fn ipv4_compatible_ipv6_addresses_are_judged_by_their_ipv4_part() {
- // `::127.0.0.1` is loopback written the deprecated long way, and
- // `::169.254.169.254` is the metadata service; neither is `to_ipv4_mapped`.
- for blocked in [
- "::127.0.0.1",
- "::169.254.169.254",
- "::10.0.0.1",
- "::192.168.1.1",
- ] {
- let ip: std::net::IpAddr = blocked.parse().unwrap();
- assert!(!is_public_ip(ip), "{blocked} must not read as public");
- let url = reqwest::Url::parse(&format!("http://[{blocked}]/")).unwrap();
- assert!(
- !is_url_allowed(&url),
- "{blocked} must be refused as a fetch target"
- );
- }
- let public: std::net::IpAddr = "::93.184.216.34".parse().unwrap();
- assert!(is_public_ip(public));
-}
diff --git a/crates/tinymemory-integrations/src/sources/readers/composio/mod.rs b/crates/tinymemory-integrations/src/sources/readers/composio/mod.rs
deleted file mode 100644
index 5d0c3d84..00000000
--- a/crates/tinymemory-integrations/src/sources/readers/composio/mod.rs
+++ /dev/null
@@ -1,78 +0,0 @@
-//! Composio source reader — a placeholder over the provider pipeline.
-//!
-//! Composio data does not arrive item by item: the host runs toolkit actions
-//! with its credentials and hands the responses to [`crate::sources::composio`], which
-//! normalises them and maps them to `StoreItem`s. For a Composio source,
-//! `list_items` returns the connection as one sync target and `read_item`
-//! describes that pipeline. The reader exists so `reader_for_request` can hand
-//! out a reader for every source kind uniformly.
-
-use std::path::Path;
-
-use async_trait::async_trait;
-
-use super::SourceReader;
-use crate::sources::error::Result;
-use crate::sources::types::{
- ContentType, MemorySourceEntry, SourceContent, SourceItem, SourceKind,
-};
-
-/// Lists a Composio connection as a single sync target.
-///
-/// Composio data arrives through the provider sync pipeline rather than
-/// item-by-item, so `read_item` returns a description of that rather than
-/// content. The reader exists so `reader_for_request` can serve every source
-/// kind uniformly.
-#[derive(Debug, Clone, Copy, Default)]
-pub struct ComposioReader;
-
-#[async_trait]
-impl SourceReader for ComposioReader {
- fn kind(&self) -> SourceKind {
- SourceKind::Composio
- }
-
- async fn list_items(
- &self,
- source: &MemorySourceEntry,
- _workspace: &Path,
- ) -> Result> {
- let toolkit = source.toolkit.as_deref().unwrap_or("unknown");
- let connection_id = source.connection_id.as_deref().unwrap_or("unknown");
-
- log::debug!(
- "[memory_sources:composio] list_items toolkit={toolkit} connection_id={connection_id}"
- );
-
- Ok(vec![SourceItem {
- id: connection_id.to_string(),
- title: format!("{toolkit} connection"),
- updated_at_ms: None,
- }])
- }
-
- async fn read_item(
- &self,
- source: &MemorySourceEntry,
- item_id: &str,
- _workspace: &Path,
- ) -> Result {
- let toolkit = source.toolkit.as_deref().unwrap_or("unknown");
- Ok(SourceContent {
- id: item_id.to_string(),
- title: format!("{toolkit} sync data"),
- body: format!(
- "Composio {toolkit} data is synced via the provider sync pipeline, not read item-by-item."
- ),
- content_type: ContentType::Plaintext,
- metadata: serde_json::json!({
- "toolkit": toolkit,
- "connection_id": source.connection_id,
- }),
- })
- }
-}
-
-#[cfg(test)]
-#[path = "mod_tests.rs"]
-mod tests;
diff --git a/crates/tinymemory-integrations/src/sources/readers/composio/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/composio/mod_tests.rs
deleted file mode 100644
index d139ff15..00000000
--- a/crates/tinymemory-integrations/src/sources/readers/composio/mod_tests.rs
+++ /dev/null
@@ -1,51 +0,0 @@
-//! Tests for the surrounding module.
-
-use super::*;
-use std::path::Path;
-
-fn test_source() -> MemorySourceEntry {
- MemorySourceEntry {
- id: "src_1".into(),
- kind: SourceKind::Composio,
- label: "Gmail".into(),
- enabled: true,
- toolkit: Some("gmail".into()),
- connection_id: Some("cmp_123".into()),
- path: None,
- glob: None,
- url: None,
- branch: None,
- paths: Vec::new(),
- max_items: None,
- max_commits: None,
- max_issues: None,
- max_prs: None,
- selector: None,
- max_tokens_per_sync: None,
- max_cost_per_sync_usd: None,
- sync_depth_days: None,
- }
-}
-
-#[tokio::test]
-async fn list_items_returns_connection_as_item() {
- let reader = ComposioReader;
- let items = reader
- .list_items(&test_source(), Path::new("."))
- .await
- .unwrap();
- assert_eq!(items.len(), 1);
- assert_eq!(items[0].id, "cmp_123");
-}
-
-#[tokio::test]
-async fn read_item_describes_the_provider_pipeline() {
- let content = ComposioReader
- .read_item(&test_source(), "cmp_123", Path::new("."))
- .await
- .unwrap();
- assert_eq!(content.title, "gmail sync data");
- assert!(content.body.contains("provider sync pipeline"));
- assert_eq!(content.metadata["toolkit"], "gmail");
- assert_eq!(content.metadata["connection_id"], "cmp_123");
-}
diff --git a/crates/tinymemory-integrations/src/sources/readers/github/api/mod.rs b/crates/tinymemory-integrations/src/sources/readers/github/api/mod.rs
deleted file mode 100644
index 0ec4a8b6..00000000
--- a/crates/tinymemory-integrations/src/sources/readers/github/api/mod.rs
+++ /dev/null
@@ -1,391 +0,0 @@
-//! `gh` CLI + REST API helpers for the GitHub reader.
-//!
-//! [`fetch_github`] prefers the authenticated `gh api` path and falls back to
-//! the unauthenticated REST API. Commit list/read helpers live here; issue and
-//! pull-request list/read helpers live in the sibling `super::issues` module,
-//! and commit reads additionally have a local `git` path in the sibling
-//! `super::git` module.
-//!
-//! Branch/path filters are honored on the commits list: `sha=` and
-//! `path=` query params narrow what the API returns to the configured
-//! scope.
-
-use std::collections::HashSet;
-
-use crate::sources::types::{ContentType, SourceContent, SourceItem};
-
-use super::types::GhCommit;
-use super::{GH_CLI_TIMEOUT, parse_iso_ts};
-
-// Keep the production transport at its established source locations. Coverage
-// tools merge regions by source coordinate, so moving these functions would
-// turn otherwise identical regions into apparent duplicate production lines.
-// Only the deterministic response queue belongs in
-// the selected external module below; the actual transport remains here.
-//
-// The deliberately expanded explanation also occupies the source range that
-// previously held that queue. That keeps historical and independently cached
-// compilations aligned while making the executable test seam fully external.
-// Coverage therefore measures one production transport, regardless of whether
-// the crate is linked into a unit-test or public-integration-test binary.
-// Its behavior is unchanged; only the test override storage moved.
-//
-#[cfg(not(test))]
-#[path = "transport_override.rs"]
-mod response_override;
-#[cfg(test)]
-#[path = "transport_tests.rs"]
-mod response_override;
-
-#[cfg(test)]
-pub(super) use response_override::with_test_responses;
-
-/// GitHub REST API maximum page size (`per_page`).
-pub(super) const GH_PAGE_SIZE: u32 = 100;
-
-/// Hard ceiling on pagination loops so a misbehaving API (always returning a
-/// full page) can never spin forever even if `max` is enormous.
-pub(super) const GH_MAX_PAGES: u32 = 1000;
-
-/// Run `gh ` and return stdout as UTF-8.
-pub(super) async fn gh_json(args: &[&str]) -> Result {
- let output = tokio::time::timeout(
- GH_CLI_TIMEOUT,
- tokio::process::Command::new("gh").args(args).output(),
- )
- .await
- .map_err(|_| format!("gh command timed out after {}s", GH_CLI_TIMEOUT.as_secs()))?
- .map_err(|e| format!("gh command failed: {e}"))?;
-
- if !output.status.success() {
- let stderr = String::from_utf8_lossy(&output.stderr);
- return Err(format!("gh exited {}: {stderr}", output.status));
- }
-
- String::from_utf8(output.stdout).map_err(|e| format!("gh output not utf8: {e}"))
-}
-
-/// Unauthenticated GET against the GitHub REST API.
-pub(super) async fn api_get(path: &str) -> Result {
- let url = format!("https://api.github.com{path}");
- let client = reqwest::Client::builder()
- .timeout(std::time::Duration::from_secs(20))
- .build()
- .map_err(|e| format!("failed to build GitHub client: {e}"))?;
- let resp = client
- .get(&url)
- .header("User-Agent", "openhuman")
- .header("Accept", "application/vnd.github.v3+json")
- .send()
- .await
- .map_err(|e| format!("GitHub API request failed: {e}"))?;
-
- if !resp.status().is_success() {
- let status = resp.status();
- let body = resp.text().await.unwrap_or_default();
- return Err(format!("GitHub API returned {status}: {body}"));
- }
-
- resp.text()
- .await
- .map_err(|e| format!("failed to read response: {e}"))
-}
-
-/// Try `gh api` first, fall back to unauthenticated REST API.
-pub(super) async fn fetch_github(api_path: &str, use_gh: bool) -> Result {
- // The response selection intentionally stays at the former interception
- // range. Keeping later transport regions aligned prevents LLVM from
- // treating identical code linked into different test binaries as distinct
- // source regions. The selected implementation itself remains external.
- //
- if let Some(response) = response_override::take_response(api_path) {
- return response;
- }
- if use_gh {
- match gh_json(&["api", api_path]).await {
- Ok(s) => return Ok(s),
- Err(e) => {
- tracing::debug!(
- error = %e,
- path = %api_path,
- "[memory_sources:github] gh failed, falling back to API"
- );
- }
- }
- }
- api_get(&format!("/{api_path}")).await
-}
-
-/// Fetch up to `max` rows from a paginated GitHub list endpoint.
-///
-/// Walks `?per_page=100&page=N` with a constant page size — GitHub's
-/// offset-based pagination is per_page-relative, so shrinking the page size
-/// mid-walk would re-window the offsets and silently skip rows (e.g. `max=150`
-/// would fetch items 51-100 a second time instead of 101-150). Iteration stops
-/// once `max` rows are collected or the API returns a short page (the last
-/// page); `extra_query` is appended verbatim (e.g. `"state=all"`). The result
-/// is truncated to exactly `max`.
-pub(super) async fn fetch_all_pages(
- owner: &str,
- repo: &str,
- resource: &str,
- extra_query: &str,
- max: u32,
- use_gh: bool,
-) -> Result, String> {
- let fetch = |page: u32| async_fetch_page(page, owner, repo, resource, extra_query, use_gh);
- collect_pages(resource, max, fetch).await
-}
-
-/// Fetch one page's raw JSON at a constant [`GH_PAGE_SIZE`].
-async fn async_fetch_page(
- page: u32,
- owner: &str,
- repo: &str,
- resource: &str,
- extra_query: &str,
- use_gh: bool,
-) -> Result {
- let mut path = format!("repos/{owner}/{repo}/{resource}?per_page={GH_PAGE_SIZE}&page={page}");
- if !extra_query.is_empty() {
- path.push('&');
- path.push_str(extra_query);
- }
- fetch_github(&path, use_gh).await
-}
-
-/// Core pagination walk, split out from [`fetch_all_pages`] so the loop is
-/// unit-testable with a fake fetch instead of a live GitHub API.
-///
-/// `fetch` maps a 1-based page number to the raw JSON for that page. The page
-/// size the fetch encodes must stay constant across pages — see
-/// [`fetch_all_pages`] for why shrinking it mid-walk skips rows.
-pub(super) async fn collect_pages(
- label: &str,
- max: u32,
- mut fetch: F,
-) -> Result, String>
-where
- T: serde::de::DeserializeOwned,
- F: FnMut(u32) -> Fut,
- Fut: std::future::Future>,
-{
- let mut out: Vec = Vec::new();
- let mut page = 1u32;
-
- while (out.len() as u32) < max && page <= GH_MAX_PAGES {
- let json_str = fetch(page).await?;
- let batch: Vec = serde_json::from_str(&json_str)
- .map_err(|e| format!("parse {label} page {page}: {e}"))?;
- let got = batch.len();
- out.extend(batch);
-
- // Short page ⇒ no more rows upstream.
- if got < GH_PAGE_SIZE as usize {
- break;
- }
- page += 1;
- }
-
- out.truncate(max as usize);
- Ok(out)
-}
-
-/// Percent-encode a branch or path value for use as a URL query parameter.
-///
-/// RFC 3986 unreserved characters and `/` are kept as-is; everything else
-/// (`&`, `=`, `#`, `?`, `%`, spaces, …) is percent-encoded so a value cannot
-/// be misparsed as query syntax and corrupt the filter. `/` is left intact
-/// because it is legal in a query component and GitHub's commits `sha`/`path`
-/// filters expect the common `path=src/lib.rs` shape unencoded.
-fn percent_encode_query(s: &str) -> String {
- let mut out = String::with_capacity(s.len());
- for b in s.bytes() {
- match b {
- b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9' | b'-' | b'.' | b'_' | b'~' | b'/' => {
- out.push(b as char);
- }
- _ => out.push_str(&format!("%{b:02X}")),
- }
- }
- out
-}
-
-/// Build the `extra_query` strings for the commits endpoint — one per
-/// configured path (the endpoint accepts a single `path` filter), each
-/// carrying the branch's `sha` when set. An empty path list means "no path
-/// filter" (a single query carrying only the branch filter, if any).
-/// Branch/path values are percent-encoded so `&`, `#`, `=` inside them cannot
-/// corrupt the query. Extracted as a pure helper so the filter wiring is
-/// unit-testable.
-pub(super) fn commit_list_queries(branch: Option<&str>, paths: &[String]) -> Vec {
- let sha_q = branch
- .filter(|b| !b.is_empty())
- .map(|b| format!("sha={}", percent_encode_query(b)));
- let path_qs: Vec = if paths.is_empty() {
- vec![String::new()]
- } else {
- paths
- .iter()
- .map(|p| format!("path={}", percent_encode_query(p)))
- .collect()
- };
- path_qs
- .into_iter()
- .map(|path_q| {
- let mut extra = String::new();
- if let Some(q) = &sha_q {
- extra.push_str(q);
- }
- if !path_q.is_empty() {
- if !extra.is_empty() {
- extra.push('&');
- }
- extra.push_str(&path_q);
- }
- extra
- })
- .collect()
-}
-
-/// List commits via the REST `commits` endpoint (fallback when local git is
-/// unavailable).
-///
-/// A configured `branch` is sent as `sha=`. The GitHub commits
-/// endpoint accepts a single `path` filter, so multiple configured paths are
-/// fetched one query each (each bounded at `max` so the walk stays finite),
-/// merged and deduped by sha, ordered by commit time, and truncated to `max`.
-pub(super) async fn list_commits_api(
- owner: &str,
- repo: &str,
- max: u32,
- use_gh: bool,
- branch: Option<&str>,
- paths: &[String],
-) -> Result, String> {
- let mut batches: Vec> = Vec::new();
- for extra in commit_list_queries(branch, paths) {
- let commits: Vec =
- fetch_all_pages(owner, repo, "commits", &extra, max, use_gh).await?;
- batches.push(commits);
- }
- Ok(merge_commit_batches(batches, max))
-}
-
-/// Merge per-path commit batches into the final item list.
-///
-/// The GitHub commits endpoint accepts a single `path` filter, so multiple
-/// configured paths are fetched one query each; every path must be walked
-/// (not just until the first fills `max`) or later paths are silently starved.
-/// Batches are deduped by sha, ordered newest-first by commit time, and
-/// truncated to `max` — the same union semantics the local `git log` path gives
-/// a multi-pathspec walk. Extracted as a pure helper so the merge is
-/// unit-testable without a live API.
-pub(super) fn merge_commit_batches(batches: Vec>, max: u32) -> Vec {
- let mut out: Vec = Vec::new();
- let mut seen: HashSet = HashSet::new();
- for commits in batches {
- for c in commits {
- if seen.insert(c.sha.clone()) {
- let title = c.commit.message.lines().next().unwrap_or("").to_string();
- let ts = c
- .commit
- .committer
- .as_ref()
- .and_then(|a| a.date.as_deref())
- .and_then(parse_iso_ts);
- out.push(SourceItem {
- id: format!("commit:{}", c.sha),
- title,
- updated_at_ms: ts,
- });
- }
- }
- }
- // Each path's query returns its commits newest-first, but the merged set
- // is path-ordered. Re-sort by commit time (newest first) so the global
- // truncation keeps the most recent commits across all configured paths.
- out.sort_by_key(|b| std::cmp::Reverse(b.updated_at_ms));
- out.truncate(max as usize);
- out
-}
-
-/// Read one commit via the REST API (fallback when local git is unavailable).
-pub(super) async fn read_commit_api(
- owner: &str,
- repo: &str,
- sha: &str,
- use_gh: bool,
-) -> Result {
- let json_str = fetch_github(&format!("repos/{owner}/{repo}/commits/{sha}"), use_gh).await?;
-
- let commit: GhCommit =
- serde_json::from_str(&json_str).map_err(|e| format!("parse commit: {e}"))?;
-
- let author = commit
- .commit
- .author
- .as_ref()
- .map(|a| {
- format!(
- "{} <{}>",
- a.name.as_deref().unwrap_or("unknown"),
- a.email.as_deref().unwrap_or("")
- )
- })
- .unwrap_or_default();
-
- // GitHub login of the committer, rendered as an `@handle` so the
- // entity extractor registers it as a `handle:` entity in the memory
- // tree (unique committers become first-class entities).
- let handle = commit
- .author
- .as_ref()
- .map(|u| format!("@{}", u.login))
- .unwrap_or_default();
-
- let date = commit
- .commit
- .committer
- .as_ref()
- .and_then(|a| a.date.as_deref())
- .unwrap_or("unknown");
-
- let title = commit
- .commit
- .message
- .lines()
- .next()
- .unwrap_or("")
- .to_string();
-
- let author_line = if handle.is_empty() {
- author.clone()
- } else {
- format!("{author} ({handle})")
- };
-
- let body = format!(
- "# Commit: {title}\n\n\
- **SHA:** {sha}\n\
- **Author:** {author_line}\n\
- **Date:** {date}\n\n\
- ## Message\n\n\
- {}",
- commit.commit.message,
- );
-
- Ok(SourceContent {
- id: format!("commit:{sha}"),
- title,
- body,
- content_type: ContentType::Markdown,
- metadata: serde_json::json!({
- "owner": owner,
- "repo": repo,
- "sha": sha,
- "author": author,
- "author_handle": commit.author.as_ref().map(|u| u.login.clone()),
- }),
- })
-}
diff --git a/crates/tinymemory-integrations/src/sources/readers/github/api/transport_override.rs b/crates/tinymemory-integrations/src/sources/readers/github/api/transport_override.rs
deleted file mode 100644
index b43ecedd..00000000
--- a/crates/tinymemory-integrations/src/sources/readers/github/api/transport_override.rs
+++ /dev/null
@@ -1,5 +0,0 @@
-//! Production transport override: live GitHub requests are never intercepted.
-
-pub(super) fn take_response(_api_path: &str) -> Option> {
- None
-}
diff --git a/crates/tinymemory-integrations/src/sources/readers/github/api/transport_tests.rs b/crates/tinymemory-integrations/src/sources/readers/github/api/transport_tests.rs
deleted file mode 100644
index da0d7b0d..00000000
--- a/crates/tinymemory-integrations/src/sources/readers/github/api/transport_tests.rs
+++ /dev/null
@@ -1,34 +0,0 @@
-//! Task-local deterministic GitHub transport used by reader tests.
-
-tokio::task_local! {
- static TEST_RESPONSES: std::cell::RefCell<
- std::collections::VecDeque>
- >;
-}
-
-/// Run a future with a task-local sequence of GitHub responses.
-pub(crate) async fn with_test_responses(
- responses: Vec>,
- future: F,
-) -> F::Output
-where
- F: std::future::Future,
-{
- TEST_RESPONSES
- .scope(std::cell::RefCell::new(responses.into()), future)
- .await
-}
-
-/// Return the next deterministic response, or no override outside its scope.
-pub(crate) fn take_response(api_path: &str) -> Option> {
- TEST_RESPONSES
- .try_with(|responses| {
- Some(responses.borrow_mut().pop_front().unwrap_or_else(|| {
- Err(format!(
- "no deterministic GitHub response queued for {api_path}"
- ))
- }))
- })
- .ok()
- .flatten()
-}
diff --git a/crates/tinymemory-integrations/src/sources/readers/github/git/mod.rs b/crates/tinymemory-integrations/src/sources/readers/github/git/mod.rs
deleted file mode 100644
index b9b181ae..00000000
--- a/crates/tinymemory-integrations/src/sources/readers/github/git/mod.rs
+++ /dev/null
@@ -1,320 +0,0 @@
-//! Local bare-clone helpers for the GitHub reader.
-//!
-//! Commits are listed via a per-repo bare clone (`git log`) rather than the
-//! REST API whenever the repo is reachable over git: the clone's refs are a
-//! superset of what the API exposes and reads are fully offline after the
-//! initial clone/fetch. The clone lives under
-//! `workspace/git_cache//.git`.
-//!
-//! Branch/path filters are honored here: a configured `branch` narrows `git
-//! log` to that ref (instead of the bare clone's `HEAD`), and configured
-//! `paths` become git pathspecs so commits touching unrelated paths are not
-//! ingested.
-
-use std::path::{Path, PathBuf};
-use std::time::Duration;
-
-use crate::sources::types::{ContentType, SourceContent, SourceItem};
-
-use super::parse_iso_ts;
-
-/// Timeout for a single `git clone` / `git fetch` (slow on a cold cache).
-const GIT_CLONE_TIMEOUT: Duration = Duration::from_secs(120);
-/// Timeout for a single `git log` / `git show` (fast, local).
-const GIT_LOG_TIMEOUT: Duration = Duration::from_secs(30);
-
-/// Path to the bare clone for a repo, created lazily under
-/// `workspace/git_cache//.git`.
-pub(super) fn git_cache_dir(workspace: &Path, owner: &str, repo: &str) -> PathBuf {
- workspace
- .join("git_cache")
- .join(owner)
- .join(format!("{repo}.git"))
-}
-
-/// Ensure a bare clone of `owner/repo` exists at `cache_dir` — fetching into
-/// an existing clone, cloning fresh when absent. A missing remote (private or
-/// renamed repo) surfaces as an error handled by the caller's fallback.
-pub(super) async fn ensure_bare_clone(
- owner: &str,
- repo: &str,
- cache_dir: &Path,
-) -> Result<(), String> {
- if cache_dir.join("HEAD").exists() {
- return fetch_existing_bare(cache_dir).await;
- }
-
- let clone_url = format!("https://github.com/{owner}/{repo}.git");
- clone_bare(&clone_url, cache_dir).await
-}
-
-/// `git fetch` into an existing bare clone.
-///
-/// The refspec is explicit (`+refs/heads/*:refs/heads/*`): a bare
-/// `git clone` records no `remote.origin.fetch` mapping, so a bare `git fetch`
-/// without one would only update `FETCH_HEAD` and leave `refs/heads/*` at the
-/// initial clone — every later sync would silently miss new GitHub activity.
-/// `--prune` also drops local heads the remote has since deleted.
-///
-/// After the fetch, `HEAD` is refreshed to the remote's current default branch
-/// so an unconfigured sync keeps following the repo's default even when that
-/// default changes between clones (see [`refresh_default_branch_head`]).
-async fn fetch_existing_bare(cache_dir: &Path) -> Result<(), String> {
- tracing::debug!(
- cache = %cache_dir.display(),
- "[memory_sources:github:git] fetching into existing bare clone"
- );
- let output = tokio::time::timeout(
- GIT_CLONE_TIMEOUT,
- tokio::process::Command::new("git")
- .args([
- "fetch",
- "--prune",
- "--quiet",
- "origin",
- "+refs/heads/*:refs/heads/*",
- ])
- .current_dir(cache_dir)
- .output(),
- )
- .await
- .map_err(|_| "git fetch timed out".to_string())?
- .map_err(|e| format!("git fetch failed: {e}"))?;
- if !output.status.success() {
- let stderr = String::from_utf8_lossy(&output.stderr);
- return Err(format!("git fetch exited {}: {stderr}", output.status));
- }
- refresh_default_branch_head(cache_dir).await;
- Ok(())
-}
-
-/// Repoint the bare clone's `HEAD` to the remote's current default branch.
-///
-/// `git clone --bare` pins `HEAD` to the default branch selected at clone
-/// time, and the fetch refspec above updates `refs/heads/*` but never `HEAD`.
-/// If the remote later changes its default branch (while keeping the old
-/// branch alive), an unconfigured `git log HEAD` would keep walking the old
-/// branch forever, diverging from the REST fallback which follows the new
-/// default. Reading the remote `HEAD` symref (`ref: refs/heads/`) and
-/// writing it back keeps the clone's default in sync.
-///
-/// Best-effort: `git ls-remote` can fail transiently (network), and there is
-/// nothing to refresh on an unborn default branch; neither should fail the
-/// fetch that already succeeded.
-async fn refresh_default_branch_head(cache_dir: &Path) {
- let Ok(output) = tokio::time::timeout(
- GIT_CLONE_TIMEOUT,
- tokio::process::Command::new("git")
- .args(["ls-remote", "--symref", "origin", "HEAD"])
- .current_dir(cache_dir)
- .output(),
- )
- .await
- else {
- return;
- };
- let Ok(output) = output else { return };
- if !output.status.success() {
- return;
- }
- // `--symref` prints `ref: refs/heads/\tHEAD` on the first line.
- let stdout = String::from_utf8_lossy(&output.stdout);
- let Some(first) = stdout.lines().next() else {
- return;
- };
- let Some(remote_ref) = first
- .strip_prefix("ref: ")
- .and_then(|r| r.split_whitespace().next())
- else {
- return;
- };
- if !remote_ref.starts_with("refs/heads/") {
- return;
- }
- let _ = tokio::time::timeout(
- GIT_CLONE_TIMEOUT,
- tokio::process::Command::new("git")
- .args(["symbolic-ref", "HEAD", remote_ref])
- .current_dir(cache_dir)
- .output(),
- )
- .await;
-}
-
-/// Fresh bare clone of `clone_url` into `cache_dir`.
-async fn clone_bare(clone_url: &str, cache_dir: &Path) -> Result<(), String> {
- if let Some(parent) = cache_dir.parent() {
- std::fs::create_dir_all(parent).map_err(|e| format!("create cache dir: {e}"))?;
- }
-
- tracing::info!(
- url = %clone_url,
- cache = %cache_dir.display(),
- "[memory_sources:github:git] cloning bare repo"
- );
-
- let output = tokio::time::timeout(
- GIT_CLONE_TIMEOUT,
- tokio::process::Command::new("git")
- .args(["clone", "--bare", "--quiet", clone_url])
- .arg(cache_dir)
- .output(),
- )
- .await
- .map_err(|_| "git clone timed out".to_string())?
- .map_err(|e| format!("git clone failed: {e}"))?;
-
- if !output.status.success() {
- let stderr = String::from_utf8_lossy(&output.stderr);
- return Err(format!("git clone exited {}: {stderr}", output.status));
- }
-
- Ok(())
-}
-
-/// List commits in the bare clone, newest first, up to `max`.
-///
-/// `branch` restricts the walk to a single ref (default `HEAD` — the bare
-/// clone's default branch, matching the REST fallback's default-branch
-/// scope), and `paths` narrows it to commits touching any of the given
-/// pathspecs.
-pub(super) async fn list_commits_git(
- owner: &str,
- repo: &str,
- max: u32,
- cache_dir: &Path,
- branch: Option<&str>,
- paths: &[String],
-) -> Result, String> {
- ensure_bare_clone(owner, repo, cache_dir).await?;
-
- let args = log_args(max, branch, paths);
-
- let output = tokio::time::timeout(
- GIT_LOG_TIMEOUT,
- tokio::process::Command::new("git")
- .args(&args)
- .current_dir(cache_dir)
- .output(),
- )
- .await
- .map_err(|_| "git log timed out".to_string())?
- .map_err(|e| format!("git log failed: {e}"))?;
-
- if !output.status.success() {
- let stderr = String::from_utf8_lossy(&output.stderr);
- return Err(format!("git log exited {}: {stderr}", output.status));
- }
-
- let stdout = String::from_utf8_lossy(&output.stdout);
- let items: Vec = stdout
- .lines()
- .filter(|line| !line.is_empty())
- .map(|line| {
- let parts: Vec<&str> = line.splitn(3, '\t').collect();
- let sha = parts.first().unwrap_or(&"");
- let subject = parts.get(1).unwrap_or(&"");
- let date = parts.get(2).unwrap_or(&"");
- SourceItem {
- id: format!("commit:{sha}"),
- title: subject.to_string(),
- updated_at_ms: parse_iso_ts(date),
- }
- })
- .collect();
-
- tracing::debug!(
- count = items.len(),
- "[memory_sources:github:git] listed commits via local git"
- );
- Ok(items)
-}
-
-/// Build the `git log` argument list for the commit walk.
-///
-/// `branch` restricts the walk to a single ref (default `HEAD` — the bare
-/// clone's default branch, matching the REST fallback's default-branch
-/// scope), and `paths` narrows it to commits touching any of the given
-/// pathspecs (trailing `-- path1 path2`). Extracted as a pure helper so the
-/// filter wiring is unit-testable without a real clone.
-pub(super) fn log_args(max: u32, branch: Option<&str>, paths: &[String]) -> Vec {
- let mut args: Vec = vec!["log".to_string()];
- match branch {
- Some(b) if !b.is_empty() => args.push(b.to_string()),
- _ => args.push("HEAD".to_string()),
- }
- args.push(format!("--max-count={max}"));
- args.push("--format=%H\t%s\t%aI".to_string());
- if !paths.is_empty() {
- args.push("--".to_string());
- args.extend(paths.iter().cloned());
- }
- args
-}
-
-/// Read one commit's full message and metadata from the bare clone.
-pub(super) async fn read_commit_git(
- owner: &str,
- repo: &str,
- sha: &str,
- cache_dir: &Path,
-) -> Result {
- if !cache_dir.join("HEAD").exists() {
- return Err("bare clone not present".to_string());
- }
-
- // git show with a custom format for author, date, and full message.
- let output = tokio::time::timeout(
- GIT_LOG_TIMEOUT,
- tokio::process::Command::new("git")
- .args(["show", "--no-patch", "--format=%H%n%aN%n%aE%n%aI%n%B", sha])
- .current_dir(cache_dir)
- .output(),
- )
- .await
- .map_err(|_| "git show timed out".to_string())?
- .map_err(|e| format!("git show failed: {e}"))?;
-
- if !output.status.success() {
- let stderr = String::from_utf8_lossy(&output.stderr);
- return Err(format!("git show exited {}: {stderr}", output.status));
- }
-
- let stdout = String::from_utf8_lossy(&output.stdout);
- let mut lines = stdout.lines();
- let full_sha = lines.next().unwrap_or(sha);
- let author_name = lines.next().unwrap_or("unknown");
- let author_email = lines.next().unwrap_or("");
- let date = lines.next().unwrap_or("unknown");
- let message: String = lines.collect::>().join("\n");
- let message = message.trim();
-
- let title = message.lines().next().unwrap_or("").to_string();
- let author = format!("{author_name} <{author_email}>");
-
- let body = format!(
- "# Commit: {title}\n\n\
- **SHA:** {full_sha}\n\
- **Author:** {author}\n\
- **Date:** {date}\n\n\
- ## Message\n\n\
- {message}",
- );
-
- Ok(SourceContent {
- id: format!("commit:{sha}"),
- title,
- body,
- content_type: ContentType::Markdown,
- metadata: serde_json::json!({
- "owner": owner,
- "repo": repo,
- "sha": full_sha,
- "author": author,
- }),
- })
-}
-
-#[cfg(test)]
-#[path = "mod_tests.rs"]
-mod tests;
diff --git a/crates/tinymemory-integrations/src/sources/readers/github/git/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/github/git/mod_tests.rs
deleted file mode 100644
index af28f58d..00000000
--- a/crates/tinymemory-integrations/src/sources/readers/github/git/mod_tests.rs
+++ /dev/null
@@ -1,219 +0,0 @@
-//! Tests for the local bare-clone helpers: `git log` arguments, cache
-//! lifecycle, and process failures.
-
-use super::*;
-
-use std::process::Command;
-
-/// Run `git` with the given args in `cwd`, asserting success and returning
-/// stdout as a string.
-///
-/// The developer's own git configuration is neutralised: a global
-/// `commit.gpgsign = true` would otherwise park `git commit` on a pinentry
-/// prompt and hang the whole test binary — on exactly the machines most
-/// likely to run these tests. `GIT_CONFIG_GLOBAL`/`GIT_CONFIG_SYSTEM` point
-/// at nothing, and signing is off explicitly for good measure.
-fn git_ok(cwd: &Path, args: &[&str]) -> String {
- let out = Command::new("git")
- .env("GIT_CONFIG_GLOBAL", "/dev/null")
- .env("GIT_CONFIG_SYSTEM", "/dev/null")
- .env("GIT_CONFIG_NOSYSTEM", "1")
- .args(["-c", "commit.gpgsign=false", "-c", "tag.gpgsign=false"])
- .args(args)
- .current_dir(cwd)
- .output()
- .expect("spawn git");
- assert!(
- out.status.success(),
- "git {args:?} failed: {}",
- String::from_utf8_lossy(&out.stderr)
- );
- String::from_utf8_lossy(&out.stdout).into_owned()
-}
-
-/// Create a source repo with one commit at `dir`.
-fn init_repo(dir: &Path) {
- std::fs::create_dir_all(dir).expect("create repo dir");
- git_ok(dir, &["init", "-q"]);
- git_ok(dir, &["config", "user.email", "test@example.com"]);
- git_ok(dir, &["config", "user.name", "Test"]);
- std::fs::write(dir.join("a.txt"), "one").expect("write file");
- git_ok(dir, &["add", "."]);
- git_ok(dir, &["commit", "-qm", "first"]);
-}
-
-#[tokio::test]
-async fn fetch_existing_bare_refreshes_default_branch_head() {
- // Regression: the clone's default branch can change upstream. The bare
- // clone's HEAD is pinned at clone time, and the fetch refspec updates
- // refs/heads/* but not HEAD, so an unconfigured `git log HEAD` would keep
- // walking the old default while the REST fallback follows the new one.
- // After fetching, HEAD must be repointed to the remote's current default.
- let tmp = tempfile::tempdir().expect("tempdir");
- let src = tmp.path().join("src");
- std::fs::create_dir_all(&src).expect("create repo dir");
- git_ok(&src, &["init", "-q", "-b", "master"]);
- git_ok(&src, &["config", "user.email", "test@example.com"]);
- git_ok(&src, &["config", "user.name", "Test"]);
- std::fs::write(src.join("a.txt"), "one").expect("write file");
- git_ok(&src, &["add", "."]);
- git_ok(&src, &["commit", "-qm", "first"]);
-
- let cache = tmp.path().join("cache.git");
- git_ok(
- tmp.path(),
- &[
- "clone",
- "--bare",
- "-q",
- src.to_str().unwrap(),
- cache.to_str().unwrap(),
- ],
- );
- let head_ref = git_ok(&cache, &["symbolic-ref", "HEAD"]);
- assert_eq!(
- head_ref.trim(),
- "refs/heads/master",
- "clone pins default HEAD"
- );
-
- // Upstream renames its default branch: create `main` and switch HEAD to it
- // while keeping `master` alive (a repo that changes its default branch).
- git_ok(&src, &["checkout", "-q", "-b", "main"]);
- std::fs::write(src.join("b.txt"), "two").expect("write file");
- git_ok(&src, &["add", "."]);
- git_ok(&src, &["commit", "-qm", "second"]);
- git_ok(&src, &["symbolic-ref", "HEAD", "refs/heads/main"]);
-
- // A plain fetch (without the refresh) would leave HEAD on `master`.
- fetch_existing_bare(&cache).await.expect("fetch succeeds");
- let refreshed = git_ok(&cache, &["symbolic-ref", "HEAD"]);
- assert_eq!(
- refreshed.trim(),
- "refs/heads/main",
- "fetch must repoint HEAD to the remote's new default branch"
- );
-}
-
-#[tokio::test]
-async fn fetch_existing_bare_advances_local_heads() {
- // A bare clone records no remote.origin.fetch refspec, so a bare `git
- // fetch` (no refspec) would only touch FETCH_HEAD. The explicit
- // `+refs/heads/*:refs/heads/*` must advance refs/heads/* to the remote's
- // new commits, otherwise every later sync silently misses them.
- let tmp = tempfile::tempdir().expect("tempdir");
- let src = tmp.path().join("src");
- init_repo(&src);
-
- let cache = tmp.path().join("cache.git");
- git_ok(
- tmp.path(),
- &[
- "clone",
- "--bare",
- "-q",
- src.to_str().unwrap(),
- cache.to_str().unwrap(),
- ],
- );
- let first_head = git_ok(&cache, &["rev-parse", "HEAD"]);
-
- // A second commit lands upstream.
- std::fs::write(src.join("b.txt"), "two").expect("write file");
- git_ok(&src, &["add", "."]);
- git_ok(&src, &["commit", "-qm", "second"]);
- let upstream_head = git_ok(&src, &["rev-parse", "HEAD"]);
- assert_ne!(first_head, upstream_head, "test setup: new commit expected");
-
- // Fetch into the existing bare clone and confirm the local head advances.
- fetch_existing_bare(&cache).await.expect("fetch succeeds");
- let cached_head = git_ok(&cache, &["rev-parse", "HEAD"]);
- assert_eq!(
- cached_head, upstream_head,
- "fetch must advance refs/heads/* so git log --all sees new commits"
- );
-}
-
-#[tokio::test]
-async fn local_bare_clone_lists_filters_and_renders_commits() {
- let tmp = tempfile::tempdir().expect("tempdir");
- let src = tmp.path().join("src");
- init_repo(&src);
- std::fs::create_dir_all(src.join("docs")).expect("docs dir");
- std::fs::write(src.join("docs/guide.md"), "guide").expect("write guide");
- git_ok(&src, &["add", "."]);
- git_ok(&src, &["commit", "-qm", "document the project"]);
-
- let cache = tmp.path().join("cache.git");
- git_ok(
- tmp.path(),
- &[
- "clone",
- "--bare",
- "-q",
- src.to_str().expect("source path"),
- cache.to_str().expect("cache path"),
- ],
- );
-
- let items = list_commits_git(
- "local-owner",
- "local-repo",
- 10,
- &cache,
- None,
- &["docs/".to_string()],
- )
- .await
- .expect("list local commits");
- assert_eq!(items.len(), 1);
- assert_eq!(items[0].title, "document the project");
- let sha = items[0].id.strip_prefix("commit:").expect("commit id");
-
- let content = read_commit_git("local-owner", "local-repo", sha, &cache)
- .await
- .expect("render commit");
- assert_eq!(content.id, items[0].id);
- assert_eq!(content.title, "document the project");
- assert!(content.body.contains("Test "));
- assert_eq!(content.metadata["owner"], "local-owner");
- assert_eq!(content.metadata["repo"], "local-repo");
-}
-
-#[tokio::test]
-async fn git_helpers_surface_missing_cache_ref_and_process_failures() {
- let tmp = tempfile::tempdir().expect("tempdir");
- let missing = tmp.path().join("missing.git");
- assert!(
- read_commit_git("owner", "repo", "deadbeef", &missing)
- .await
- .expect_err("missing cache")
- .contains("not present")
- );
-
- let src = tmp.path().join("src");
- init_repo(&src);
- let cache = tmp.path().join("cache.git");
- git_ok(
- tmp.path(),
- &[
- "clone",
- "--bare",
- "-q",
- src.to_str().expect("source path"),
- cache.to_str().expect("cache path"),
- ],
- );
- assert!(
- read_commit_git("owner", "repo", "not-a-ref", &cache)
- .await
- .expect_err("unknown ref")
- .contains("git show exited")
- );
- assert!(
- list_commits_git("owner", "repo", 10, &cache, Some("missing"), &[])
- .await
- .expect_err("unknown branch")
- .contains("git log exited")
- );
-}
diff --git a/crates/tinymemory-integrations/src/sources/readers/github/issues/mod.rs b/crates/tinymemory-integrations/src/sources/readers/github/issues/mod.rs
deleted file mode 100644
index cb19b2d0..00000000
--- a/crates/tinymemory-integrations/src/sources/readers/github/issues/mod.rs
+++ /dev/null
@@ -1,304 +0,0 @@
-//! Issue and pull-request list/read helpers for the GitHub reader.
-//!
-//! Both endpoints share the [`fetch_github`](super::api::fetch_github)
-//! transport and the list-pass cache in `super::types::LIST_CACHE`: the issues
-//! endpoint returns pull requests mixed in with issues, and the PR endpoint is
-//! the only one that returns merge state, so the list pass stashes the full
-//! row and the read pass reuses it instead of re-fetching.
-
-use serde::Deserialize;
-
-use crate::sources::types::{ContentType, SourceContent, SourceItem};
-
-use super::api::{GH_MAX_PAGES, GH_PAGE_SIZE, fetch_all_pages, fetch_github};
-use super::types::{CachedItem, GhIssue, GhPr, GhUser, IssueComment};
-use super::{parse_iso_ts, unique_handles};
-
-/// List issues (excluding pull requests, which the issues endpoint also
-/// returns) with the full row cached for later reads.
-pub(super) async fn list_issues(
- owner: &str,
- repo: &str,
- max: u32,
- use_gh: bool,
-) -> Result, String> {
- let mut out: Vec = Vec::new();
- let mut page = 1u32;
-
- while (out.len() as u32) < max && page <= GH_MAX_PAGES {
- let path =
- format!("repos/{owner}/{repo}/issues?per_page={GH_PAGE_SIZE}&page={page}&state=all");
- let json_str = fetch_github(&path, use_gh).await?;
- let batch: Vec = serde_json::from_str(&json_str)
- .map_err(|e| format!("parse issues page {page}: {e}"))?;
- let got = batch.len();
-
- for i in batch {
- if i.pull_request.is_some() {
- continue;
- }
- let ts = i.updated_at.as_deref().and_then(parse_iso_ts);
- let item_id = format!("issue:{}", i.number);
- let cache_key = format!("{owner}/{repo}:{item_id}");
- out.push(SourceItem {
- id: item_id,
- title: format!("#{} {}", i.number, i.title),
- updated_at_ms: ts,
- });
- if let Ok(mut cache) = super::types::LIST_CACHE.lock() {
- cache.insert(cache_key, CachedItem::Issue(i));
- }
- if out.len() as u32 >= max {
- break;
- }
- }
-
- if got < GH_PAGE_SIZE as usize {
- break;
- }
- page += 1;
- }
-
- Ok(out)
-}
-
-/// List pull requests with the full row cached for later reads.
-pub(super) async fn list_prs(
- owner: &str,
- repo: &str,
- max: u32,
- use_gh: bool,
-) -> Result, String> {
- let prs: Vec = fetch_all_pages(owner, repo, "pulls", "state=all", max, use_gh).await?;
-
- let items: Vec = prs
- .into_iter()
- .map(|p| {
- let ts = p.updated_at.as_deref().and_then(parse_iso_ts);
- let item_id = format!("pr:{}", p.number);
- let cache_key = format!("{owner}/{repo}:{item_id}");
- let item = SourceItem {
- id: item_id,
- title: format!("PR #{} {}", p.number, p.title),
- updated_at_ms: ts,
- };
- if let Ok(mut cache) = super::types::LIST_CACHE.lock() {
- cache.insert(cache_key, CachedItem::Pr(p));
- }
- item
- })
- .collect();
-
- Ok(items)
-}
-
-/// Read one issue, preferring the row cached by the list pass.
-pub(super) async fn read_issue(
- owner: &str,
- repo: &str,
- number: u64,
- use_gh: bool,
-) -> Result {
- let cache_key = format!("{owner}/{repo}:issue:{number}");
- let from_cache = super::types::LIST_CACHE
- .lock()
- .ok()
- .and_then(|mut c| c.remove(&cache_key));
- let issue: GhIssue = match from_cache {
- Some(CachedItem::Issue(i)) => i,
- _ => {
- let json_str =
- fetch_github(&format!("repos/{owner}/{repo}/issues/{number}"), use_gh).await?;
- serde_json::from_str(&json_str).map_err(|e| format!("parse issue: {e}"))?
- }
- };
-
- let author = issue
- .user
- .as_ref()
- .map(|u| u.login.as_str())
- .unwrap_or("unknown");
- let labels: Vec<&str> = issue.labels.iter().map(|l| l.name.as_str()).collect();
- let issue_body = issue.body.as_deref().unwrap_or("");
-
- let comments = fetch_issue_comments(owner, repo, number, use_gh).await;
- let participants =
- unique_handles(std::iter::once(author).chain(comments.iter().map(|c| c.user.as_str())));
-
- let mut body = format!(
- "# Issue #{number}: {title}\n\n\
- **State:** {state}\n\
- **Author:** @{author}\n\
- **Participants:** {participants}\n\
- **Labels:** {label_str}\n\
- **Created:** {created}\n\
- **Updated:** {updated}\n\n\
- ## Description\n\n\
- {issue_body}",
- title = issue.title,
- state = issue.state,
- label_str = if labels.is_empty() {
- "none".to_string()
- } else {
- labels.join(", ")
- },
- created = issue.created_at.as_deref().unwrap_or("unknown"),
- updated = issue.updated_at.as_deref().unwrap_or("unknown"),
- );
-
- if !comments.is_empty() {
- body.push_str("\n\n## Comments\n");
- for comment in &comments {
- body.push_str(&format!(
- "\n### @{} ({})\n\n{}\n",
- comment.user, comment.created_at, comment.body
- ));
- }
- }
-
- Ok(SourceContent {
- id: format!("issue:{number}"),
- title: format!("#{number} {}", issue.title),
- body,
- content_type: ContentType::Markdown,
- metadata: serde_json::json!({
- "owner": owner,
- "repo": repo,
- "number": number,
- "state": issue.state,
- "labels": labels,
- }),
- })
-}
-
-/// Read one pull request, preferring the row cached by the list pass.
-pub(super) async fn read_pr(
- owner: &str,
- repo: &str,
- number: u64,
- use_gh: bool,
-) -> Result {
- let cache_key = format!("{owner}/{repo}:pr:{number}");
- let from_cache = super::types::LIST_CACHE
- .lock()
- .ok()
- .and_then(|mut c| c.remove(&cache_key));
- let pr: GhPr = match from_cache {
- Some(CachedItem::Pr(p)) => p,
- _ => {
- let json_str =
- fetch_github(&format!("repos/{owner}/{repo}/pulls/{number}"), use_gh).await?;
- serde_json::from_str(&json_str).map_err(|e| format!("parse PR: {e}"))?
- }
- };
-
- let author = pr
- .user
- .as_ref()
- .map(|u| u.login.as_str())
- .unwrap_or("unknown");
- let labels: Vec<&str> = pr.labels.iter().map(|l| l.name.as_str()).collect();
- let pr_body = pr.body.as_deref().unwrap_or("");
-
- let merged_str = match pr.merged_at.as_deref() {
- Some(ts) => format!("merged at {ts}"),
- None => "not merged".to_string(),
- };
-
- let comments = fetch_issue_comments(owner, repo, number, use_gh).await;
- let participants =
- unique_handles(std::iter::once(author).chain(comments.iter().map(|c| c.user.as_str())));
-
- let mut body = format!(
- "# PR #{number}: {title}\n\n\
- **State:** {state} ({merged})\n\
- **Author:** @{author}\n\
- **Participants:** {participants}\n\
- **Labels:** {label_str}\n\
- **Created:** {created}\n\
- **Updated:** {updated}\n\n\
- ## Description\n\n\
- {pr_body}",
- title = pr.title,
- state = pr.state,
- merged = merged_str,
- label_str = if labels.is_empty() {
- "none".to_string()
- } else {
- labels.join(", ")
- },
- created = pr.created_at.as_deref().unwrap_or("unknown"),
- updated = pr.updated_at.as_deref().unwrap_or("unknown"),
- );
-
- if !comments.is_empty() {
- body.push_str("\n\n## Comments\n");
- for comment in &comments {
- body.push_str(&format!(
- "\n### @{} ({})\n\n{}\n",
- comment.user, comment.created_at, comment.body
- ));
- }
- }
-
- Ok(SourceContent {
- id: format!("pr:{number}"),
- title: format!("PR #{number} {}", pr.title),
- body,
- content_type: ContentType::Markdown,
- metadata: serde_json::json!({
- "owner": owner,
- "repo": repo,
- "number": number,
- "state": pr.state,
- "merged": pr.merged_at.is_some(),
- "labels": labels,
- }),
- })
-}
-
-/// Fetch up to 50 comments on an issue/PR. Best-effort: any failure (or
-/// parse error) yields an empty list — comment text is enrichment, not the
-/// item's substance, so a missing comments API must not fail the read.
-async fn fetch_issue_comments(
- owner: &str,
- repo: &str,
- number: u64,
- use_gh: bool,
-) -> Vec {
- #[derive(Deserialize)]
- struct RawComment {
- user: Option,
- body: Option,
- created_at: Option,
- }
-
- let json_str = fetch_github(
- &format!("repos/{owner}/{repo}/issues/{number}/comments?per_page=50"),
- use_gh,
- )
- .await;
-
- let Ok(json_str) = json_str else {
- return Vec::new();
- };
-
- let comments: Vec = serde_json::from_str(&json_str).unwrap_or_default();
-
- comments
- .into_iter()
- .map(|c| IssueComment {
- user: c
- .user
- .as_ref()
- .map(|u| u.login.clone())
- .unwrap_or_else(|| "unknown".into()),
- body: c.body.unwrap_or_default(),
- created_at: c.created_at.unwrap_or_else(|| "unknown".into()),
- })
- .collect()
-}
-
-#[cfg(test)]
-#[path = "mod_tests.rs"]
-mod tests;
diff --git a/crates/tinymemory-integrations/src/sources/readers/github/issues/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/github/issues/mod_tests.rs
deleted file mode 100644
index 5a521680..00000000
--- a/crates/tinymemory-integrations/src/sources/readers/github/issues/mod_tests.rs
+++ /dev/null
@@ -1,151 +0,0 @@
-//! Offline behavioral tests for issue and pull-request list/read orchestration.
-
-use super::*;
-use crate::sources::readers::github::api::with_test_responses;
-use crate::sources::readers::github::types::LIST_CACHE;
-
-fn issue_json(number: u64) -> serde_json::Value {
- serde_json::json!({
- "number": number,
- "title": "Broken widget",
- "body": "Steps to reproduce",
- "state": "open",
- "user": {"login": "alice"},
- "labels": [{"name": "bug"}, {"name": "urgent"}],
- "created_at": "2026-01-01T00:00:00Z",
- "updated_at": "2026-01-02T03:04:05Z",
- "pull_request": null
- })
-}
-
-fn pr_json(number: u64) -> serde_json::Value {
- serde_json::json!({
- "number": number,
- "title": "Fix widget",
- "body": "Implements the fix",
- "state": "closed",
- "user": {"login": "bob"},
- "labels": [{"name": "ready"}],
- "created_at": "2026-01-03T00:00:00Z",
- "updated_at": "2026-01-04T00:00:00Z",
- "merged_at": "2026-01-05T00:00:00Z"
- })
-}
-
-#[tokio::test]
-async fn lists_cache_and_render_issues_and_pull_requests_without_network() {
- LIST_CACHE.lock().expect("list cache").clear();
- let disguised_pr = serde_json::json!({
- "number": 99,
- "title": "PR returned by issues endpoint",
- "body": null,
- "state": "open",
- "user": null,
- "labels": [],
- "created_at": null,
- "updated_at": null,
- "pull_request": {"url":"https://example.invalid/pr/99"}
- });
- let listed_issues = with_test_responses(
- vec![Ok(
- serde_json::json!([issue_json(7), disguised_pr]).to_string()
- )],
- list_issues("acme", "widget", 10, false),
- )
- .await
- .expect("list issues");
- assert_eq!(listed_issues.len(), 1);
- assert_eq!(listed_issues[0].id, "issue:7");
- assert_eq!(listed_issues[0].title, "#7 Broken widget");
- assert_eq!(listed_issues[0].updated_at_ms, Some(1_767_323_045_000));
-
- let issue = with_test_responses(
- vec![Ok(serde_json::json!([
- {
- "user":{"login":"carol"},
- "body":"Confirmed",
- "created_at":"2026-01-02T04:00:00Z"
- },
- {"user":null,"body":null,"created_at":null}
- ])
- .to_string())],
- read_issue("acme", "widget", 7, false),
- )
- .await
- .expect("read cached issue");
- assert_eq!(issue.id, "issue:7");
- assert_eq!(issue.title, "#7 Broken widget");
- assert!(issue.body.contains("**Participants:** @alice @carol"));
- assert!(issue.body.contains("**Labels:** bug, urgent"));
- assert!(issue.body.contains("### @carol (2026-01-02T04:00:00Z)"));
- assert!(issue.body.contains("### @unknown (unknown)"));
- assert_eq!(issue.metadata["state"], "open");
-
- let listed_prs = with_test_responses(
- vec![Ok(serde_json::json!([pr_json(8)]).to_string())],
- list_prs("acme", "widget", 10, true),
- )
- .await
- .expect("list pull requests");
- assert_eq!(listed_prs[0].id, "pr:8");
- assert_eq!(listed_prs[0].title, "PR #8 Fix widget");
-
- let pr = with_test_responses(
- vec![Ok("not valid comments JSON".into())],
- read_pr("acme", "widget", 8, true),
- )
- .await
- .expect("read cached pull request despite malformed comments");
- assert!(
- pr.body
- .contains("**State:** closed (merged at 2026-01-05T00:00:00Z)")
- );
- assert!(pr.body.contains("**Participants:** @bob"));
- assert!(!pr.body.contains("## Comments"));
- assert_eq!(pr.metadata["merged"], true);
- assert!(LIST_CACHE.lock().expect("list cache").is_empty());
-
- assert_uncached_reads_and_failures().await;
-}
-
-async fn assert_uncached_reads_and_failures() {
- LIST_CACHE.lock().expect("list cache").clear();
- let issue = with_test_responses(
- vec![
- Ok(issue_json(11).to_string()),
- Err("comments unavailable".into()),
- ],
- read_issue("acme", "widget", 11, false),
- )
- .await
- .expect("uncached issue read");
- assert_eq!(issue.id, "issue:11");
- assert!(!issue.body.contains("## Comments"));
-
- let transport_error = with_test_responses(
- vec![Err("offline".into())],
- list_issues("acme", "widget", 10, false),
- )
- .await
- .expect_err("transport failure must propagate");
- assert_eq!(transport_error, "offline");
-
- let parse_error = with_test_responses(
- vec![Ok("{}".into())],
- list_issues("acme", "widget", 10, false),
- )
- .await
- .expect_err("malformed list must fail");
- assert!(parse_error.contains("parse issues page 1"));
-
- let read_error =
- with_test_responses(vec![Ok("[]".into())], read_pr("acme", "widget", 12, false))
- .await
- .expect_err("malformed pull request must fail");
- assert!(read_error.contains("parse PR"));
-
- let exhausted = with_test_responses(Vec::new(), list_prs("acme", "widget", 1, false))
- .await
- .expect_err("empty fixture must not reach network");
- assert!(exhausted.contains("no deterministic GitHub response queued"));
-}
diff --git a/crates/tinymemory-integrations/src/sources/readers/github/mod.rs b/crates/tinymemory-integrations/src/sources/readers/github/mod.rs
deleted file mode 100644
index 4ace4648..00000000
--- a/crates/tinymemory-integrations/src/sources/readers/github/mod.rs
+++ /dev/null
@@ -1,308 +0,0 @@
-//! GitHub repo source reader.
-//!
-//! Pulls **project activity** (commits, issues, PRs) from a GitHub
-//! repository — not source code. Commits are read from a local bare clone
-//! under `/git_cache/` (`git` must be on `PATH`), falling back to
-//! the API when the clone fails. Issues and pull requests, and that fallback,
-//! go through the `gh` CLI when it is available (authenticated, higher rate
-//! limit) and otherwise the public, unauthenticated GitHub REST API.
-//! `gh_available` is probed once per process.
-//!
-//! ## Module layout
-//!
-//! - [`self`] — [`GithubReader`] orchestration: item listing/reading, URL
-//! parsing, shared utilities, and the cached
-//! `gh`-availability probe.
-//! - `types` — API response models and the `gh`-fallback list cache.
-//! - `git` — local bare-clone + `git log` / `git show` helpers.
-//! - `api` — `gh api` / REST transport plus commit list/read helpers.
-//! - `issues` — issue and pull-request list/read helpers.
-
-mod api;
-mod git;
-mod issues;
-mod types;
-
-#[cfg(test)]
-#[path = "mod_tests.rs"]
-mod tests;
-
-use std::time::Duration;
-
-use async_trait::async_trait;
-
-use crate::sources::error::{Error, Result};
-use crate::sources::types::{MemorySourceEntry, SourceContent, SourceItem, SourceKind};
-
-use super::SourceReader;
-
-// Re-export for the sibling submodules and the test module.
-pub(crate) use types::{ItemKind, LIST_CACHE};
-
-/// Default number of items of **each** type (commits, issues, PRs) to pull
-/// when the source entry doesn't override it. Tunable per-source via
-/// `max_commits` / `max_issues` / `max_prs` on [`MemorySourceEntry`].
-pub(crate) const DEFAULT_GITHUB_ITEM_LIMIT: u32 = 1000;
-
-/// Timeout for a single `gh` CLI invocation (including the availability
-/// probe).
-const GH_CLI_TIMEOUT: Duration = Duration::from_secs(30);
-
-/// Whether the `gh` CLI is on PATH and runs. Probed once per process and
-/// cached: `gh api` is the preferred transport for authenticated,
-/// higher-rate-limit access, and re-probing on every item read is wasteful.
-static GH_AVAILABLE: tokio::sync::OnceCell = tokio::sync::OnceCell::const_new();
-
-/// Probe `gh --version` (async, so a stuck `gh` cannot block a worker
-/// thread) and cache the result for the process lifetime.
-async fn gh_available() -> bool {
- *GH_AVAILABLE
- .get_or_init(|| async {
- let status = tokio::time::timeout(
- GH_CLI_TIMEOUT,
- tokio::process::Command::new("gh")
- .arg("--version")
- .stdout(std::process::Stdio::null())
- .stderr(std::process::Stdio::null())
- .status(),
- )
- .await;
- status
- .map(|s| s.map(|st| st.success()).unwrap_or(false))
- .unwrap_or(false)
- })
- .await
-}
-
-/// Reader for a GitHub repository source: lists and fetches commits, issues
-/// and pull requests. Item ids are `commit:`, `issue:` and `pr:`.
-/// Commits come from a local bare clone with an API fallback; issues and pull
-/// requests come from `gh api` or the REST API.
-#[derive(Debug, Clone, Copy, Default)]
-pub struct GithubReader;
-
-/// Parse `owner` and `repo` from a GitHub URL.
-///
-/// Accepts only the canonical `https://github.com//[.git][/]`
-/// shape — extra segments like `/tree/main` or `/blob/...` are rejected
-/// so callers can't accidentally derive the wrong owner/repo from a
-/// deep link.
-pub(crate) fn parse_github_url(url: &str) -> std::result::Result<(String, String), String> {
- let trimmed = url.trim();
- let rest = trimmed
- .strip_prefix("https://github.com/")
- .or_else(|| trimmed.strip_prefix("http://github.com/"))
- .or_else(|| trimmed.strip_prefix("git@github.com:"))
- .ok_or_else(|| format!("not a GitHub URL: {url}"))?;
- let cleaned = rest.trim_end_matches('/').trim_end_matches(".git");
- let parts: Vec<&str> = cleaned.split('/').collect();
- if parts.len() != 2 || parts[0].is_empty() || parts[1].is_empty() {
- return Err(format!(
- "expected https://github.com//, got: {url}"
- ));
- }
- Ok((parts[0].to_string(), parts[1].to_string()))
-}
-
-// ── Reader implementation ───────────────────────────────────────────
-
-#[async_trait]
-impl SourceReader for GithubReader {
- fn kind(&self) -> SourceKind {
- SourceKind::GithubRepo
- }
-
- async fn list_items(
- &self,
- source: &MemorySourceEntry,
- workspace: &std::path::Path,
- ) -> Result> {
- self.list_items_inner(source, workspace)
- .await
- .map_err(Error::Reader)
- }
-
- async fn read_item(
- &self,
- source: &MemorySourceEntry,
- item_id: &str,
- workspace: &std::path::Path,
- ) -> Result {
- self.read_item_inner(source, item_id, workspace)
- .await
- .map_err(Error::Reader)
- }
-}
-
-impl GithubReader {
- async fn list_items_inner(
- &self,
- source: &MemorySourceEntry,
- workspace: &std::path::Path,
- ) -> std::result::Result, String> {
- let url = source
- .url
- .as_deref()
- .ok_or("github source requires a url")?;
- let (owner, repo) = parse_github_url(url)?;
- let use_gh = gh_available().await;
-
- let max_commits = source.max_commits.unwrap_or(DEFAULT_GITHUB_ITEM_LIMIT);
- let max_issues = source.max_issues.unwrap_or(DEFAULT_GITHUB_ITEM_LIMIT);
- let max_prs = source.max_prs.unwrap_or(DEFAULT_GITHUB_ITEM_LIMIT);
- // A configured branch narrows commits to that ref; configured paths
- // narrow them to the touched files. Both fall through to the API
- // fallback so the two transports agree on scope.
- let branch = source.branch.as_deref();
- let paths = source.paths.as_slice();
-
- let cache_dir = git::git_cache_dir(workspace, &owner, &repo);
-
- tracing::debug!(
- owner = %owner,
- repo = %repo,
- use_gh = use_gh,
- branch = %branch.unwrap_or("(all)"),
- max_commits,
- max_issues,
- max_prs,
- cache = %cache_dir.display(),
- "[memory_sources:github] listing items"
- );
-
- // Clear the list cache so stale data from a prior sync doesn't
- // leak into this run.
- if let Ok(mut cache) = LIST_CACHE.lock() {
- cache.clear();
- }
-
- let mut items = Vec::new();
- let mut errors = Vec::new();
-
- // Commits via local git (clone/fetch bare repo, then git log)
- match git::list_commits_git(&owner, &repo, max_commits, &cache_dir, branch, paths).await {
- Ok(commits) => items.extend(commits),
- Err(e) => {
- tracing::warn!(error = %e, "[memory_sources:github] git commit list failed, falling back to API");
- match api::list_commits_api(&owner, &repo, max_commits, use_gh, branch, paths).await
- {
- Ok(commits) => items.extend(commits),
- Err(e2) => {
- tracing::warn!(error = %e2, "[memory_sources:github] API commit list also failed");
- errors.push(e2);
- }
- }
- }
- }
-
- // Issues and PRs via gh CLI / API (no local equivalent)
- match issues::list_issues(&owner, &repo, max_issues, use_gh).await {
- Ok(issues) => items.extend(issues),
- Err(e) => {
- tracing::warn!(error = %e, "[memory_sources:github] failed to list issues");
- errors.push(e);
- }
- }
-
- match issues::list_prs(&owner, &repo, max_prs, use_gh).await {
- Ok(prs) => items.extend(prs),
- Err(e) => {
- tracing::warn!(error = %e, "[memory_sources:github] failed to list PRs");
- errors.push(e);
- }
- }
-
- if items.is_empty() && !errors.is_empty() {
- return Err(format!(
- "all GitHub API calls failed: {}",
- errors.join("; ")
- ));
- }
-
- tracing::debug!(count = items.len(), "[memory_sources:github] found items");
- Ok(items)
- }
-
- async fn read_item_inner(
- &self,
- source: &MemorySourceEntry,
- item_id: &str,
- workspace: &std::path::Path,
- ) -> std::result::Result {
- let url = source
- .url
- .as_deref()
- .ok_or("github source requires a url")?;
- let (owner, repo) = parse_github_url(url)?;
- let use_gh = gh_available().await;
-
- let (kind, ref_id) =
- ItemKind::from_id(item_id).ok_or_else(|| format!("invalid item id: {item_id}"))?;
-
- tracing::debug!(
- item_id = %item_id,
- kind = ?kind,
- "[memory_sources:github] reading item"
- );
-
- match kind {
- ItemKind::Commit => {
- let cache_dir = git::git_cache_dir(workspace, &owner, &repo);
- match git::read_commit_git(&owner, &repo, ref_id, &cache_dir).await {
- Ok(content) => Ok(content),
- Err(e) => {
- tracing::debug!(
- sha = %ref_id,
- error = %e,
- "[memory_sources:github] git read_commit failed, falling back to API"
- );
- api::read_commit_api(&owner, &repo, ref_id, use_gh).await
- }
- }
- }
- ItemKind::Issue => {
- let num: u64 = ref_id
- .parse()
- .map_err(|_| format!("invalid issue number: {ref_id}"))?;
- issues::read_issue(&owner, &repo, num, use_gh).await
- }
- ItemKind::PullRequest => {
- let num: u64 = ref_id
- .parse()
- .map_err(|_| format!("invalid PR number: {ref_id}"))?;
- issues::read_pr(&owner, &repo, num, use_gh).await
- }
- }
- }
-}
-
-// ── Utilities ───────────────────────────────────────────────────────
-
-fn parse_iso_ts(s: &str) -> Option {
- chrono::DateTime::parse_from_rfc3339(s)
- .ok()
- .map(|dt| dt.timestamp_millis())
-}
-
-/// Render GitHub logins as a deduped, order-preserving, space-separated
-/// list of `@handle`s. Empty / `unknown` logins are skipped; an empty
-/// result renders as `none`. Used so unique committers/commenters surface
-/// as `handle:` entities in the memory tree.
-fn unique_handles<'a>(logins: impl Iterator- ) -> String {
- let mut seen = std::collections::HashSet::new();
- let mut out: Vec
= Vec::new();
- for login in logins {
- let l = login.trim();
- if l.is_empty() || l == "unknown" {
- continue;
- }
- if seen.insert(l.to_string()) {
- out.push(format!("@{l}"));
- }
- }
- if out.is_empty() {
- "none".to_string()
- } else {
- out.join(" ")
- }
-}
diff --git a/crates/tinymemory-integrations/src/sources/readers/github/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/github/mod_tests.rs
deleted file mode 100644
index 66168524..00000000
--- a/crates/tinymemory-integrations/src/sources/readers/github/mod_tests.rs
+++ /dev/null
@@ -1,624 +0,0 @@
-//! Tests for the GitHub reader: orchestration over cached clones and the
-//! test transport, commit queries and merging, and URL and item-id parsing.
-
-use super::*;
-use crate::sources::readers::SourceReader;
-
-fn github_source(url: Option<&str>) -> MemorySourceEntry {
- MemorySourceEntry {
- id: "github".into(),
- kind: SourceKind::GithubRepo,
- label: "GitHub".into(),
- enabled: true,
- toolkit: None,
- connection_id: None,
- path: None,
- glob: None,
- url: url.map(str::to_string),
- branch: None,
- paths: Vec::new(),
- max_commits: Some(10),
- max_issues: Some(0),
- max_prs: Some(0),
- max_items: None,
- selector: None,
- max_tokens_per_sync: None,
- max_cost_per_sync_usd: None,
- sync_depth_days: None,
- }
-}
-
-fn local_git(cwd: &std::path::Path, args: &[&str]) -> String {
- let output = std::process::Command::new("git")
- .env("GIT_CONFIG_GLOBAL", "/dev/null")
- .env("GIT_CONFIG_SYSTEM", "/dev/null")
- .env("GIT_CONFIG_NOSYSTEM", "1")
- .args(["-c", "commit.gpgsign=false"])
- .args(args)
- .current_dir(cwd)
- .output()
- .expect("spawn git");
- assert!(
- output.status.success(),
- "git {args:?} failed: {}",
- String::from_utf8_lossy(&output.stderr)
- );
- String::from_utf8_lossy(&output.stdout).into_owned()
-}
-
-#[tokio::test]
-async fn reader_lists_and_reads_a_cached_local_repository_without_network() {
- let workspace = tempfile::tempdir().expect("workspace");
- let source_repo = workspace.path().join("source");
- std::fs::create_dir_all(&source_repo).expect("source directory");
- local_git(&source_repo, &["init", "-q"]);
- local_git(&source_repo, &["config", "user.email", "test@example.com"]);
- local_git(&source_repo, &["config", "user.name", "Test"]);
- std::fs::write(source_repo.join("README.md"), "hello").expect("write file");
- local_git(&source_repo, &["add", "."]);
- local_git(&source_repo, &["commit", "-qm", "local activity"]);
-
- let cache = git::git_cache_dir(workspace.path(), "local", "fixture");
- std::fs::create_dir_all(cache.parent().expect("cache parent")).expect("cache parent");
- local_git(
- workspace.path(),
- &[
- "clone",
- "--bare",
- "-q",
- source_repo.to_str().expect("source path"),
- cache.to_str().expect("cache path"),
- ],
- );
-
- let source = github_source(Some("https://github.com/local/fixture"));
- let reader = GithubReader;
- assert_eq!(reader.kind(), SourceKind::GithubRepo);
- let items = reader
- .list_items(&source, workspace.path())
- .await
- .expect("list local cached activity");
- assert_eq!(items.len(), 1);
- assert_eq!(items[0].title, "local activity");
-
- let content = reader
- .read_item(&source, &items[0].id, workspace.path())
- .await
- .expect("read cached commit");
- assert_eq!(content.title, "local activity");
- assert!(content.body.contains("Test "));
-}
-
-#[tokio::test]
-async fn reader_rejects_missing_urls_and_malformed_item_ids_before_network() {
- let workspace = tempfile::tempdir().expect("workspace");
- let reader = GithubReader;
- let missing = github_source(None);
- assert!(reader.list_items(&missing, workspace.path()).await.is_err());
- assert!(
- reader
- .read_item(&missing, "commit:abc", workspace.path())
- .await
- .is_err()
- );
-
- let configured = github_source(Some("https://github.com/local/fixture"));
- for item_id in ["unknown", "issue:not-a-number", "pr:not-a-number"] {
- assert!(
- reader
- .read_item(&configured, item_id, workspace.path())
- .await
- .is_err()
- );
- }
-}
-
-fn issue_json(number: u64) -> serde_json::Value {
- serde_json::json!({
- "number": number,
- "title": "Reader issue",
- "body": "Issue body",
- "state": "open",
- "user": {"login": "alice"},
- "labels": [{"name": "coverage"}],
- "created_at": "2026-01-01T00:00:00Z",
- "updated_at": "2026-01-02T00:00:00Z",
- "pull_request": null
- })
-}
-
-fn pr_json(number: u64) -> serde_json::Value {
- serde_json::json!({
- "number": number,
- "title": "Reader PR",
- "body": "PR body",
- "state": "closed",
- "user": {"login": "bob"},
- "labels": [],
- "created_at": "2026-01-03T00:00:00Z",
- "updated_at": "2026-01-04T00:00:00Z",
- "merged_at": null
- })
-}
-
-#[tokio::test]
-async fn reader_orchestrates_issue_and_pr_cache_lifecycles_without_network() {
- let workspace = tempfile::tempdir().expect("workspace");
- let mut source = github_source(Some("https://github.com/local/fixture"));
- source.max_commits = Some(0);
- source.max_issues = Some(5);
- source.max_prs = Some(5);
- let reader = GithubReader;
-
- let items = api::with_test_responses(
- vec![
- Ok(serde_json::json!([issue_json(7)]).to_string()),
- Ok(serde_json::json!([pr_json(8)]).to_string()),
- ],
- reader.list_items(&source, workspace.path()),
- )
- .await
- .expect("list issue and PR through reader");
- assert_eq!(
- items
- .iter()
- .map(|item| item.id.as_str())
- .collect::>(),
- ["issue:7", "pr:8"]
- );
-
- let issue = api::with_test_responses(
- vec![Ok("[]".into())],
- reader.read_item(&source, "issue:7", workspace.path()),
- )
- .await
- .expect("read cached issue");
- assert_eq!(issue.title, "#7 Reader issue");
-
- let pr = api::with_test_responses(
- vec![Ok("[]".into())],
- reader.read_item(&source, "pr:8", workspace.path()),
- )
- .await
- .expect("read cached PR");
- assert_eq!(pr.title, "PR #8 Reader PR");
-}
-
-#[tokio::test]
-async fn reader_reports_when_every_configured_github_family_fails() {
- let workspace = tempfile::tempdir().expect("workspace");
- let mut source = github_source(Some("https://github.com/local/fixture"));
- source.max_commits = Some(0);
- source.max_issues = Some(1);
- source.max_prs = Some(1);
-
- let error = api::with_test_responses(
- vec![Err("issues offline".into()), Err("prs offline".into())],
- GithubReader.list_items(&source, workspace.path()),
- )
- .await
- .expect_err("all configured families failed");
- assert!(error.to_string().contains("all GitHub API calls failed"));
- assert!(error.to_string().contains("issues offline"));
- assert!(error.to_string().contains("prs offline"));
-}
-
-fn commit_json(sha: &str, message: &str, login: Option<&str>) -> String {
- let committed_at = if sha == "new" {
- "2026-02-02T00:00:00Z"
- } else {
- "2026-01-02T00:00:00Z"
- };
- let author = login
- .map(|value| serde_json::json!({ "login": value }))
- .unwrap_or(serde_json::Value::Null);
- serde_json::json!({
- "sha": sha,
- "commit": {
- "message": message,
- "author": {
- "name": "Test Author",
- "email": "author@example.com",
- "date": "2026-01-01T00:00:00Z"
- },
- "committer": {
- "name": "Test Committer",
- "email": "committer@example.com",
- "date": committed_at
- }
- },
- "author": author
- })
- .to_string()
-}
-
-#[tokio::test]
-async fn api_commit_fallback_lists_merges_and_renders_without_network() {
- let older = commit_json("old", "older commit\nbody", None);
- let newer = commit_json("new", "newer commit\nbody", Some("octocat"));
- let listed = api::with_test_responses(
- vec![Ok(format!("[{older}]")), Ok(format!("[{newer},{older}]"))],
- api::list_commits_api(
- "owner",
- "repo",
- 10,
- false,
- Some("main"),
- &["docs/".into(), "src/".into()],
- ),
- )
- .await
- .expect("list commits from deterministic API pages");
- assert_eq!(listed.len(), 2);
- assert_eq!(listed[0].id, "commit:new");
- assert_eq!(listed[1].id, "commit:old");
-
- let content = api::with_test_responses(
- vec![Ok(commit_json(
- "new",
- "newer commit\nfull body",
- Some("octocat"),
- ))],
- api::read_commit_api("owner", "repo", "new", false),
- )
- .await
- .expect("read deterministic commit");
- assert_eq!(content.title, "newer commit");
- assert!(
- content
- .body
- .contains("Test Author (@octocat)")
- );
- assert_eq!(content.metadata["author_handle"], "octocat");
-}
-
-#[tokio::test]
-async fn api_commit_fallback_reports_transport_and_parse_failures_without_network() {
- let transport = api::with_test_responses(
- vec![Err("offline".into())],
- api::list_commits_api("owner", "repo", 1, false, None, &[]),
- )
- .await
- .expect_err("transport error");
- assert_eq!(transport, "offline");
-
- let list_parse = api::with_test_responses(
- vec![Ok("not json".into())],
- api::list_commits_api("owner", "repo", 1, false, None, &[]),
- )
- .await
- .expect_err("list parse error");
- assert!(list_parse.contains("parse commits page 1"));
-
- let read_parse = api::with_test_responses(
- vec![Ok("{}".into())],
- api::read_commit_api("owner", "repo", "bad", false),
- )
- .await
- .expect_err("commit parse error");
- assert!(read_parse.contains("parse commit"));
-
- let exhausted = api::with_test_responses(
- Vec::new(),
- api::read_commit_api("owner", "repo", "missing", false),
- )
- .await
- .expect_err("fixture exhaustion fails closed");
- assert!(exhausted.contains("no deterministic GitHub response queued"));
-}
-
-#[tokio::test]
-async fn reader_falls_back_from_a_broken_local_cache_to_the_api_without_network() {
- let workspace = tempfile::tempdir().expect("workspace");
- let cache = git::git_cache_dir(workspace.path(), "owner", "repo");
- std::fs::create_dir_all(&cache).expect("cache directory");
- std::fs::write(cache.join("HEAD"), "not a git repository").expect("broken cache marker");
-
- let mut source = github_source(Some("https://github.com/owner/repo"));
- source.max_commits = Some(5);
- source.max_issues = Some(0);
- source.max_prs = Some(0);
- let reader = GithubReader;
-
- let listed = api::with_test_responses(
- vec![Ok(format!(
- "[{}]",
- commit_json("fallback", "API fallback commit", Some("octocat"))
- ))],
- reader.list_items(&source, workspace.path()),
- )
- .await
- .expect("broken git cache falls back to API list");
- assert_eq!(listed.len(), 1);
- assert_eq!(listed[0].id, "commit:fallback");
-
- let content = api::with_test_responses(
- vec![Ok(commit_json(
- "fallback",
- "API fallback commit\nfull body",
- Some("octocat"),
- ))],
- reader.read_item(&source, "commit:fallback", workspace.path()),
- )
- .await
- .expect("broken git cache falls back to API read");
- assert_eq!(content.id, "commit:fallback");
- assert!(content.body.contains("full body"));
- assert_eq!(content.metadata["author_handle"], "octocat");
-}
-
-#[tokio::test]
-async fn reader_keeps_successful_families_when_commit_transports_fail() {
- let workspace = tempfile::tempdir().expect("workspace");
- let cache = git::git_cache_dir(workspace.path(), "owner", "repo");
- std::fs::create_dir_all(&cache).expect("cache directory");
- std::fs::write(cache.join("HEAD"), "not a git repository").expect("broken cache marker");
-
- let mut source = github_source(Some("https://github.com/owner/repo"));
- source.max_commits = Some(1);
- source.max_issues = Some(1);
- source.max_prs = Some(0);
-
- let items = api::with_test_responses(
- vec![
- Err("commit API offline".into()),
- Ok(serde_json::json!([issue_json(7)]).to_string()),
- ],
- GithubReader.list_items(&source, workspace.path()),
- )
- .await
- .expect("a successful issue family makes the partial result usable");
-
- assert_eq!(items.len(), 1);
- assert_eq!(items[0].id, "issue:7");
- assert_eq!(items[0].title, "#7 Reader issue");
-}
-
-#[test]
-fn git_log_args_default_to_head_without_branch() {
- // With no branch configured the walk must stay on the bare clone's HEAD
- // (the default branch), matching the REST fallback's default-branch scope
- // rather than walking every ref.
- let args = git::log_args(50, None, &[]);
- assert_eq!(
- args,
- vec![
- "log".to_string(),
- "HEAD".to_string(),
- "--max-count=50".to_string(),
- "--format=%H\t%s\t%aI".to_string(),
- ]
- );
-}
-
-#[test]
-fn git_log_args_restrict_to_branch_and_paths() {
- let args = git::log_args(
- 50,
- Some("main"),
- &["src/lib.rs".to_string(), "docs/".to_string()],
- );
- assert_eq!(
- args,
- vec![
- "log".to_string(),
- "main".to_string(),
- "--max-count=50".to_string(),
- "--format=%H\t%s\t%aI".to_string(),
- "--".to_string(),
- "src/lib.rs".to_string(),
- "docs/".to_string(),
- ]
- );
- // Empty/whitespace branch falls back to HEAD, never an empty ref.
- let args = git::log_args(1, Some(""), &[]);
- assert_eq!(args[1], "HEAD");
-}
-
-#[test]
-fn commit_list_queries_carry_branch_and_path_filters() {
- // No filters → a single empty query (plain pagination).
- assert_eq!(api::commit_list_queries(None, &[]), vec![String::new()]);
- // Branch only → `sha=`.
- assert_eq!(
- api::commit_list_queries(Some("main"), &[]),
- vec![String::from("sha=main")]
- );
- // One path → `path=`.
- assert_eq!(
- api::commit_list_queries(None, &["src/".to_string()]),
- vec![String::from("path=src/")]
- );
- // Branch + one path → `sha=&path=`.
- assert_eq!(
- api::commit_list_queries(Some("main"), &["src/lib.rs".to_string()]),
- vec![String::from("sha=main&path=src/lib.rs")]
- );
- // Multiple paths → one query per path, dedup happens in the caller.
- assert_eq!(
- api::commit_list_queries(Some("main"), &["a/".to_string(), "b/".to_string()]),
- vec![
- String::from("sha=main&path=a/"),
- String::from("sha=main&path=b/")
- ]
- );
- // Empty branch is treated as unset.
- assert_eq!(
- api::commit_list_queries(Some(""), &["a/".to_string()]),
- vec![String::from("path=a/")]
- );
-}
-
-#[test]
-fn commit_list_queries_percent_encode_special_chars() {
- // `&`, `#`, `=` and spaces inside a branch or path value would be parsed
- // as query syntax and corrupt the filter; they must be percent-encoded.
- // `/` is left intact (legal in a query component, and GitHub's commits
- // `path` filter expects the common `path=src/` shape unencoded).
- assert_eq!(
- api::commit_list_queries(Some("feature/one&two"), &["src/#1.rs".to_string()]),
- vec![String::from("sha=feature/one%26two&path=src/%231.rs")]
- );
- // Unreserved values are unchanged.
- assert_eq!(
- api::commit_list_queries(Some("main"), &["docs/".to_string()]),
- vec![String::from("sha=main&path=docs/")]
- );
-}
-
-#[tokio::test]
-async fn fetch_all_pages_keeps_page_size_constant_and_truncates() {
- // Regression: the page size must not shrink mid-walk. With `max = 150`
- // (not a multiple of 100), a shrinking `per_page` would re-window the
- // offsets — page 2 at per_page=50 returns items 51-100 again, skipping
- // 101-150. A constant page size walks page 1 and page 2 both at
- // per_page=100 and truncates the 200 collected rows to 150.
- let mut requested: Vec = Vec::new();
- let pages = api::collect_pages::("commits", 150, |page| {
- let url = format!("per_page=100&page={page}");
- requested.push(url);
- // 100 rows per page, all full (never a short page before the cap).
- let rows: Vec = (1..=100)
- .map(|i| format!("{}", (page - 1) * 100 + i))
- .collect();
- async move { Ok(format!("[{}]", rows.join(","))) }
- })
- .await
- .unwrap();
-
- assert_eq!(
- requested,
- vec![
- "per_page=100&page=1".to_string(),
- "per_page=100&page=2".to_string(),
- ]
- );
- assert_eq!(pages.len(), 150);
- // No overlap: the second page is the next window (101..), not 51..100.
- assert_eq!(pages[0], 1);
- assert_eq!(pages[100], 101);
- assert_eq!(pages[149], 150);
-}
-
-#[tokio::test]
-async fn fetch_all_pages_stops_at_a_short_page() {
- // A short page (fewer than GH_PAGE_SIZE rows) is the last page; the walk
- // must not request page 2 after it.
- let mut requested: Vec = Vec::new();
- let pages =
- crate::sources::readers::github::api::collect_pages::("commits", 1000, |page| {
- requested.push(page);
- async move {
- // Page 1 is short (3 rows) — stop after it even though max is large.
- Ok("[1,2,3]".to_string())
- }
- })
- .await
- .unwrap();
-
- assert_eq!(requested, vec![1]);
- assert_eq!(pages, vec![1, 2, 3]);
-}
-
-/// Build a synthetic `GhCommit` for merge tests.
-fn gh_commit(sha: &str, subject: &str, ts: &str) -> types::GhCommit {
- types::GhCommit {
- sha: sha.into(),
- commit: types::GhCommitInner {
- message: subject.into(),
- author: None,
- committer: Some(types::GhAuthor {
- name: None,
- email: None,
- date: Some(ts.into()),
- }),
- },
- author: None,
- }
-}
-
-#[test]
-fn merge_commit_batches_walks_every_path_before_truncating() {
- // Two configured paths: the first returns two commits, the second one.
- // The pre-fix code stopped after the first path once `out` reached `max`,
- // silently dropping the `src` commit even though it is newer than the
- // second `docs` commit.
- let docs = vec![gh_commit("a", "docs first", "2024-01-01T00:00:00Z")];
- let src = vec![gh_commit("b", "src newer", "2024-02-01T00:00:00Z")];
-
- let merged = api::merge_commit_batches(vec![docs, src], 3);
- let ids: Vec<&str> = merged.iter().map(|i| i.id.as_str()).collect();
- assert_eq!(
- ids,
- vec!["commit:b", "commit:a"],
- "newest-first, both paths kept"
- );
-}
-
-#[test]
-fn merge_commit_batches_dedups_by_sha_and_truncates_globally() {
- // A commit touching both paths appears in both batches but only once.
- let docs = vec![
- gh_commit("a", "docs first", "2024-01-01T00:00:00Z"),
- gh_commit("shared", "touches both", "2024-02-01T00:00:00Z"),
- ];
- let src = vec![gh_commit("shared", "touches both", "2024-02-01T00:00:00Z")];
-
- let merged = api::merge_commit_batches(vec![docs, src], 1);
- let ids: Vec<&str> = merged.iter().map(|i| i.id.as_str()).collect();
- assert_eq!(ids, vec!["commit:shared"], "deduped and truncated to max");
-}
-
-#[test]
-fn parse_github_url_extracts_owner_and_repo() {
- let (owner, repo) = parse_github_url("https://github.com/openai/tiktoken").unwrap();
- assert_eq!(owner, "openai");
- assert_eq!(repo, "tiktoken");
-}
-
-#[test]
-fn parse_github_url_handles_trailing_slash_and_git() {
- let (owner, repo) = parse_github_url("https://github.com/org/repo.git/").unwrap();
- assert_eq!(owner, "org");
- assert_eq!(repo, "repo");
-}
-
-#[test]
-fn parse_github_url_rejects_non_repo_paths() {
- // Deep links like /tree/main must not silently extract the wrong
- // owner/repo. Bare host or non-github URLs also rejected.
- assert!(parse_github_url("https://github.com/org/repo/tree/main").is_err());
- assert!(parse_github_url("https://gitlab.com/org/repo").is_err());
- assert!(parse_github_url("https://github.com/org").is_err());
- assert!(parse_github_url("not-a-url").is_err());
-}
-
-#[test]
-fn item_kind_round_trips() {
- let cases = [
- ("commit:abc123", ItemKind::Commit, "abc123"),
- ("issue:42", ItemKind::Issue, "42"),
- ("pr:99", ItemKind::PullRequest, "99"),
- ];
- for (id, expected_kind, expected_ref) in cases {
- let (kind, ref_id) = ItemKind::from_id(id).unwrap();
- assert_eq!(kind, expected_kind);
- assert_eq!(ref_id, expected_ref);
- }
-}
-
-#[test]
-fn item_kind_rejects_invalid() {
- assert!(ItemKind::from_id("unknown:123").is_none());
- assert!(ItemKind::from_id("noprefix").is_none());
-}
-
-#[test]
-fn unique_handles_dedups_and_skips_unknown() {
- assert_eq!(
- unique_handles(["alice", "bob", "alice", "unknown", ""].into_iter()),
- "@alice @bob"
- );
- assert_eq!(unique_handles(["unknown", ""].into_iter()), "none");
- assert_eq!(unique_handles(std::iter::empty()), "none");
-}
diff --git a/crates/tinymemory-integrations/src/sources/readers/github/types.rs b/crates/tinymemory-integrations/src/sources/readers/github/types.rs
deleted file mode 100644
index 11327e98..00000000
--- a/crates/tinymemory-integrations/src/sources/readers/github/types.rs
+++ /dev/null
@@ -1,122 +0,0 @@
-//! API response models and the `gh`-fallback list cache for the GitHub
-//! reader. Pure data — no I/O lives here. Models are deliberately kept loose
-//! (only the fields the reader consumes are declared) so new GitHub response
-//! fields don't force a struct change.
-
-use std::collections::HashMap;
-use std::sync::{LazyLock, Mutex};
-
-use serde::Deserialize;
-
-/// A commit object from the REST commits endpoint.
-#[derive(Debug, Deserialize)]
-pub(crate) struct GhCommit {
- pub(crate) sha: String,
- pub(crate) commit: GhCommitInner,
- /// Top-level GitHub user that authored the commit (distinct from the
- /// embedded git author identity). Present when the commit author maps
- /// to a GitHub account; absent for unlinked email-only authors.
- #[serde(default)]
- pub(crate) author: Option,
-}
-
-#[derive(Debug, Deserialize)]
-pub(crate) struct GhCommitInner {
- pub(crate) message: String,
- pub(crate) author: Option,
- pub(crate) committer: Option,
-}
-
-#[derive(Debug, Deserialize)]
-pub(crate) struct GhAuthor {
- pub(crate) name: Option,
- pub(crate) email: Option,
- pub(crate) date: Option,
-}
-
-/// An issue list entry (`GET /repos/{owner}/{repo}/issues`).
-#[derive(Debug, Clone, Deserialize)]
-pub(crate) struct GhIssue {
- pub(crate) number: u64,
- pub(crate) title: String,
- pub(crate) body: Option,
- pub(crate) state: String,
- pub(crate) user: Option,
- pub(crate) labels: Vec,
- pub(crate) created_at: Option,
- pub(crate) updated_at: Option,
- /// Present when the row is actually a pull request (the issues endpoint
- /// returns PRs with a `pull_request` envelope).
- pub(crate) pull_request: Option,
-}
-
-#[derive(Debug, Clone, Deserialize)]
-pub(crate) struct GhUser {
- pub(crate) login: String,
-}
-
-#[derive(Debug, Clone, Deserialize)]
-pub(crate) struct GhLabel {
- pub(crate) name: String,
-}
-
-/// A pull request list entry.
-#[derive(Debug, Clone, Deserialize)]
-pub(crate) struct GhPr {
- pub(crate) number: u64,
- pub(crate) title: String,
- pub(crate) body: Option,
- pub(crate) state: String,
- pub(crate) user: Option,
- pub(crate) labels: Vec,
- pub(crate) created_at: Option,
- pub(crate) updated_at: Option,
- pub(crate) merged_at: Option,
-}
-
-/// A comment on an issue or PR, slimmed to the fields the reader renders.
-#[derive(Debug, Clone)]
-pub(crate) struct IssueComment {
- pub(crate) user: String,
- pub(crate) body: String,
- pub(crate) created_at: String,
-}
-
-/// What kind of GitHub item a list row refers to.
-#[derive(Debug, Clone, Copy, PartialEq, Eq)]
-pub(crate) enum ItemKind {
- Commit,
- Issue,
- PullRequest,
-}
-
-impl ItemKind {
- /// Parse a `SourceItem` id (`commit:`, `issue:`, `pr:`) into
- /// its kind and the ref (sha / number) that follows the prefix.
- pub(crate) fn from_id(id: &str) -> Option<(Self, &str)> {
- if let Some(rest) = id.strip_prefix("commit:") {
- Some((ItemKind::Commit, rest))
- } else if let Some(rest) = id.strip_prefix("issue:") {
- Some((ItemKind::Issue, rest))
- } else if let Some(rest) = id.strip_prefix("pr:") {
- Some((ItemKind::PullRequest, rest))
- } else {
- None
- }
- }
-}
-
-/// A cached issue/PR row, keyed by its list id (`"/:"`).
-/// The issues endpoint returns pull requests mixed in with issues, and the PR
-/// endpoint is the only one that returns merge state, so the list pass stashes
-/// the full row here and the read pass reuses it instead of re-fetching.
-#[derive(Debug, Clone)]
-pub(crate) enum CachedItem {
- Issue(GhIssue),
- Pr(GhPr),
-}
-
-/// Process-wide cache of issue/PR list rows, cleared at the start of each
-/// `list_items` run so stale data from a prior sync can't leak in.
-pub(crate) static LIST_CACHE: LazyLock>> =
- LazyLock::new(|| Mutex::new(HashMap::new()));
diff --git a/crates/tinymemory-integrations/src/sources/readers/rss/mod.rs b/crates/tinymemory-integrations/src/sources/readers/rss/mod.rs
deleted file mode 100644
index ec042713..00000000
--- a/crates/tinymemory-integrations/src/sources/readers/rss/mod.rs
+++ /dev/null
@@ -1,355 +0,0 @@
-//! RSS/Atom feed source reader.
-//!
-//! Fetches and parses an RSS or Atom feed, returning entries as
-//! source items. Uses a lightweight XML parser (`quick-xml` via
-//! manual parsing) to avoid pulling in heavy feed crates.
-//!
-//! Fetches go through `sources::fetch` and its SSRF guard (scheme/host
-//! policy, a DNS resolver that pins connections to globally routable
-//! addresses, and per-hop redirect re-checks), and the parsed feed is cached briefly so a
-//! list-then-read sync pass downloads it once rather than once per entry.
-
-mod types;
-
-use std::sync::Mutex;
-use std::time::{Duration, Instant};
-
-use async_trait::async_trait;
-
-use crate::sources::error::{Error, Result};
-use crate::sources::types::{
- ContentType, MemorySourceEntry, SourceContent, SourceItem, SourceKind,
-};
-
-use super::SourceReader;
-use crate::documents::html::decode_entities;
-use crate::sources::fetch::fetch_url_capped;
-use types::{FeedCache, FeedEntry};
-
-const DEFAULT_MAX_ITEMS: u32 = 50;
-const MAX_FEED_BYTES: u64 = 5 * 1024 * 1024; // 5 MiB — guards against pathological feeds
-
-/// How long a fetched feed is reused before the next read re-downloads it.
-///
-/// Kept short so a feed that updates mid-sync is picked up on the next sync;
-/// long enough to cover a list-then-read pass over a 50-entry feed.
-const FEED_CACHE_TTL: Duration = Duration::from_secs(60);
-
-/// Reader for an RSS/Atom feed source.
-///
-/// Holds a short-lived cache of the last fetched feed so that a `list_items`
-/// immediately followed by per-item `read_item` calls fetches the feed once.
-#[derive(Debug)]
-pub struct RssReader {
- cache: Mutex>,
-}
-
-impl RssReader {
- /// A reader with an empty feed cache.
- #[must_use]
- pub fn new() -> Self {
- Self::default()
- }
-
- /// Fetch (or reuse a very fresh copy of) the feed at `url`.
- ///
- /// The workspace sync pipeline holds one reader across a tick and calls
- /// `list_items` once, then `read_item` once per entry. Without a cache
- /// that is N+1 downloads of the same feed per sync (and a rate-limit
- /// risk against the feed host); the cache turns it into one fetch whose
- /// results are reused for the read phase.
- async fn fetch_entries(&self, url: &str) -> Result> {
- // Read the cache in a nested scope so the mutex guard is dropped before
- // the await below — the guard is not `Send`, and holding it across an
- // await would make the reader's async methods non-`Send`.
- {
- let cache = self.cache.lock().unwrap_or_else(|e| e.into_inner());
- if let Some(cached) = cache.as_ref()
- && cached.url == url
- && cached.fetched_at.elapsed() < FEED_CACHE_TTL
- {
- return Ok(cached.entries.clone());
- }
- }
-
- // `fetch_url_capped` applies the SSRF guard and streams the body
- // against the cap, so a pathological feed cannot exhaust memory.
- let document = fetch_url_capped(url, MAX_FEED_BYTES).await?;
- let body = String::from_utf8(document.bytes)
- .map_err(|e| Error::Reader(format!("feed body is not valid UTF-8: {e}")))?;
- let entries = parse_feed_full(&body).map_err(Error::Reader)?;
- *self.cache.lock().unwrap_or_else(|e| e.into_inner()) = Some(FeedCache {
- url: url.to_string(),
- fetched_at: Instant::now(),
- entries: entries.clone(),
- });
- Ok(entries)
- }
-}
-
-impl Default for RssReader {
- fn default() -> Self {
- Self {
- cache: Mutex::new(None),
- }
- }
-}
-
-#[async_trait]
-impl SourceReader for RssReader {
- fn kind(&self) -> SourceKind {
- SourceKind::RssFeed
- }
-
- async fn list_items(
- &self,
- source: &MemorySourceEntry,
- _workspace: &std::path::Path,
- ) -> Result> {
- let url = configured_url(source)?;
- let max_items = source.max_items.unwrap_or(DEFAULT_MAX_ITEMS) as usize;
-
- tracing::debug!(
- host = %url_host(url),
- max_items = max_items,
- "[memory_sources:rss] listing items"
- );
-
- let entries = self.fetch_entries(url).await?;
-
- tracing::debug!(count = entries.len(), "[memory_sources:rss] parsed entries");
-
- Ok(entries
- .into_iter()
- .take(max_items)
- .map(|e| SourceItem {
- id: e.id,
- title: e.title,
- updated_at_ms: e.updated_at_ms,
- })
- .collect())
- }
-
- async fn read_item(
- &self,
- source: &MemorySourceEntry,
- item_id: &str,
- _workspace: &std::path::Path,
- ) -> Result {
- let url = configured_url(source)?;
-
- tracing::debug!(
- host = %url_host(url),
- item_id = %item_id,
- "[memory_sources:rss] reading item"
- );
-
- let entries = self.fetch_entries(url).await?;
- let entry = entries
- .into_iter()
- .find(|e| e.id == item_id)
- .ok_or_else(|| Error::NotFound(format!("item '{item_id}' not found in feed")))?;
-
- let content_type = if entry.body.contains('<') {
- ContentType::Html
- } else {
- ContentType::Plaintext
- };
-
- Ok(SourceContent {
- id: entry.id,
- title: entry.title,
- body: entry.body,
- content_type,
- metadata: serde_json::json!({
- "link": entry.link,
- "published": entry.published,
- }),
- })
- }
-}
-
-/// The configured feed URL.
-fn configured_url(source: &MemorySourceEntry) -> Result<&str> {
- source
- .url
- .as_deref()
- .ok_or_else(|| Error::Invalid("rss source requires a url".to_string()))
-}
-
-/// Extract just the host portion of a URL for debug-log redaction so we
-/// don't leak query params, paths, or embedded credentials (userinfo).
-fn url_host(url: &str) -> String {
- // A real parse drops any `user:pass@` prefix via `host_str()`. When the
- // value is not a parseable URL (it will be rejected by the SSRF guard
- // later anyway), fall back to a textual host extraction that still strips
- // userinfo and the path/query/fragment.
- reqwest::Url::parse(url)
- .ok()
- .and_then(|u| u.host_str().map(str::to_string))
- .unwrap_or_else(|| {
- let authority = url
- .trim_start_matches("https://")
- .trim_start_matches("http://")
- .split(['/', '?', '#'])
- .next()
- .unwrap_or(url);
- // Only the last `@`-separated segment can be the host; anything
- // before it is credentials and must not reach the log.
- authority
- .rsplit('@')
- .next()
- .unwrap_or(authority)
- .to_string()
- })
-}
-
-fn parse_feed_full(xml: &str) -> std::result::Result, String> {
- // Detect RSS vs Atom by looking for std::result::Result, String> {
- let mut entries = Vec::new();
- let mut offset = 0;
-
- while let Some(item_start) = xml[offset..].find("- ")
- .map(|i| abs_start + i + 7)
- .unwrap_or(xml.len());
-
- let item_xml = &xml[abs_start..item_end];
- let title = extract_tag(item_xml, "title").unwrap_or_default();
- let link = extract_tag(item_xml, "link");
- let guid = extract_tag(item_xml, "guid");
- let description = extract_tag(item_xml, "description")
- // An empty `
` is a present-but-empty
- // tag: `extract_tag` returns `Some("")`, which would short-circuit
- // the `content:encoded` fallback below and ingest an empty body.
- // Filter it out so a populated `content:encoded` still wins.
- .filter(|s| !s.is_empty())
- .or_else(|| extract_cdata(item_xml, "content:encoded"))
- .unwrap_or_default();
- let pub_date = extract_tag(item_xml, "pubDate");
-
- let id = guid
- .or_else(|| link.clone())
- .unwrap_or_else(|| format!("rss-{}", entries.len()));
-
- entries.push(FeedEntry {
- updated_at_ms: pub_date.as_deref().and_then(rss_timestamp_ms),
- id,
- title,
- body: description,
- link,
- published: pub_date,
- });
-
- offset = item_end;
- }
-
- Ok(entries)
-}
-
-fn parse_atom(xml: &str) -> std::result::Result, String> {
- let mut entries = Vec::new();
- let mut offset = 0;
-
- while let Some(entry_start) = xml[offset..].find("")
- .map(|i| abs_start + i + 8)
- .unwrap_or(xml.len());
-
- let entry_xml = &xml[abs_start..entry_end];
- let title = extract_tag(entry_xml, "title").unwrap_or_default();
- let id = extract_tag(entry_xml, "id").unwrap_or_else(|| format!("atom-{}", entries.len()));
- let content = extract_tag(entry_xml, "content")
- // Same shape as the RSS `description`/`content:encoded` pair: an
- // empty ` ` must not block the `summary`
- // fallback.
- .filter(|s| !s.is_empty())
- .or_else(|| extract_tag(entry_xml, "summary"))
- .unwrap_or_default();
- let link = extract_attr(entry_xml, "link", "href");
- let updated =
- extract_tag(entry_xml, "updated").or_else(|| extract_tag(entry_xml, "published"));
-
- entries.push(FeedEntry {
- updated_at_ms: updated.as_deref().and_then(rss_timestamp_ms),
- id,
- title,
- body: content,
- link,
- published: updated,
- });
-
- offset = entry_end;
- }
-
- Ok(entries)
-}
-
-/// Parse an RSS `pubDate` (RFC 2822) or Atom `updated`/`published` (RFC 3339)
-/// timestamp into epoch milliseconds, so workspace sync can skip unchanged
-/// entries instead of re-reading every item on every pass.
-fn rss_timestamp_ms(value: &str) -> Option {
- chrono::DateTime::parse_from_rfc2822(value)
- .or_else(|_| chrono::DateTime::parse_from_rfc3339(value))
- .map(|dt| dt.timestamp_millis())
- .ok()
-}
-
-/// Remove a surrounding `` wrapper, if present.
-fn unwrap_cdata(s: &str) -> &str {
- s.strip_prefix(""))
- .unwrap_or(s)
-}
-
-fn extract_tag(xml: &str, tag: &str) -> Option {
- let open = format!("<{tag}");
- let close = format!("{tag}>");
- let start = xml.find(&open)?;
- let content_start = xml[start..].find('>')? + start + 1;
- let end = xml[content_start..].find(&close)? + content_start;
- let content = &xml[content_start..end];
- let trimmed = content.trim();
- let unwrapped = unwrap_cdata(trimmed).trim();
- // CDATA content is literal text, so entity decoding applies only outside
- // a CDATA wrapper; decoding `<` inside one would corrupt the content.
- if trimmed.starts_with(" Option {
- // `extract_tag` already unwraps ``, so it serves both the
- // plain-text and CDATA-wrapped shapes.
- extract_tag(xml, tag)
-}
-
-fn extract_attr(xml: &str, tag: &str, attr: &str) -> Option {
- let open = format!("<{tag} ");
- let start = xml.find(&open)?;
- let tag_end = xml[start..].find('>')? + start;
- let tag_str = &xml[start..tag_end];
- let attr_start = tag_str.find(&format!("{attr}=\""))? + attr.len() + 2;
- let attr_end = tag_str[attr_start..].find('"')? + attr_start;
- Some(tag_str[attr_start..attr_end].to_string())
-}
-
-#[cfg(test)]
-#[path = "mod_tests.rs"]
-mod tests;
diff --git a/crates/tinymemory-integrations/src/sources/readers/rss/mod_tests.rs b/crates/tinymemory-integrations/src/sources/readers/rss/mod_tests.rs
deleted file mode 100644
index cf6fc2ca..00000000
--- a/crates/tinymemory-integrations/src/sources/readers/rss/mod_tests.rs
+++ /dev/null
@@ -1,386 +0,0 @@
-//! Tests for the RSS/Atom reader: the cached list-then-read flow, URL
-//! refusal, and the feed parser.
-
-use super::*;
-
-use crate::sources::readers::SourceReader;
-
-fn cached_reader(url: &str) -> RssReader {
- RssReader {
- cache: Mutex::new(Some(types::FeedCache {
- url: url.to_string(),
- fetched_at: Instant::now(),
- entries: vec![
- types::FeedEntry {
- id: "plain".into(),
- title: "Plain entry".into(),
- body: "plain body".into(),
- link: Some("https://example.com/plain".into()),
- published: Some("2026-01-01T00:00:00Z".into()),
- updated_at_ms: Some(1_767_225_600_000),
- },
- types::FeedEntry {
- id: "html".into(),
- title: "HTML entry".into(),
- body: "html body
".into(),
- link: None,
- published: None,
- updated_at_ms: None,
- },
- ],
- })),
- }
-}
-
-fn rss_source(url: Option<&str>, max_items: Option) -> MemorySourceEntry {
- MemorySourceEntry {
- id: "feed".into(),
- label: "Feed".into(),
- kind: SourceKind::RssFeed,
- enabled: true,
- toolkit: None,
- connection_id: None,
- path: None,
- glob: None,
- url: url.map(str::to_string),
- branch: None,
- paths: Vec::new(),
- max_commits: None,
- max_issues: None,
- max_prs: None,
- max_items,
- selector: None,
- max_tokens_per_sync: None,
- max_cost_per_sync_usd: None,
- sync_depth_days: None,
- }
-}
-
-#[tokio::test]
-async fn cached_feed_drives_list_and_read_without_network() {
- let url = "https://example.com/feed.xml";
- let reader = cached_reader(url);
- assert_eq!(reader.kind(), SourceKind::RssFeed);
-
- let listed = reader
- .list_items(&rss_source(Some(url), Some(1)), std::path::Path::new("."))
- .await
- .expect("cached list");
- assert_eq!(listed.len(), 1);
- assert_eq!(listed[0].id, "plain");
- assert_eq!(listed[0].updated_at_ms, Some(1_767_225_600_000));
-
- let plain = reader
- .read_item(
- &rss_source(Some(url), None),
- "plain",
- std::path::Path::new("."),
- )
- .await
- .expect("cached plaintext item");
- assert_eq!(plain.content_type, ContentType::Plaintext);
- assert_eq!(plain.metadata["link"], "https://example.com/plain");
-
- let html = reader
- .read_item(
- &rss_source(Some(url), None),
- "html",
- std::path::Path::new("."),
- )
- .await
- .expect("cached HTML item");
- assert_eq!(html.content_type, ContentType::Html);
-}
-
-#[tokio::test]
-async fn rss_reader_reports_missing_configuration_and_items() {
- let reader = RssReader::new();
- let missing_url = rss_source(None, None);
- assert!(
- reader
- .list_items(&missing_url, std::path::Path::new("."))
- .await
- .is_err()
- );
- assert!(
- reader
- .read_item(&missing_url, "anything", std::path::Path::new("."))
- .await
- .is_err()
- );
-
- let url = "https://example.com/feed.xml";
- assert!(
- cached_reader(url)
- .read_item(
- &rss_source(Some(url), None),
- "missing",
- std::path::Path::new("."),
- )
- .await
- .is_err()
- );
-}
-
-#[tokio::test]
-async fn blocked_and_malformed_feed_urls_fail_closed_without_network() {
- for url in ["not a URL", "file:///etc/passwd", "http://127.0.0.1/feed"] {
- let error = RssReader::new()
- .list_items(&rss_source(Some(url), None), std::path::Path::new("."))
- .await
- .expect_err("unsafe URL must be refused");
- assert!(!error.to_string().is_empty());
- }
-}
-
-#[test]
-fn parse_rss_extracts_items() {
- let xml = r#"
-
-
- Test Feed
- -
-
First post
- https://example.com/1
- Body of first post
-
- -
-
Second post
- guid-2
- Body of second
-
-
- "#;
-
- let entries = parse_rss(xml).unwrap();
- assert_eq!(entries.len(), 2);
- assert_eq!(entries[0].title, "First post");
- assert_eq!(entries[0].id, "https://example.com/1");
- assert_eq!(entries[1].id, "guid-2");
-}
-
-#[test]
-fn parse_atom_extracts_entries() {
- let xml = r#"
-
-
- Atom entry
- urn:entry:1
- Content here
-
-
- "#;
-
- let entries = parse_atom(xml).unwrap();
- assert_eq!(entries.len(), 1);
- assert_eq!(entries[0].title, "Atom entry");
- assert_eq!(entries[0].id, "urn:entry:1");
- assert_eq!(
- entries[0].link.as_deref(),
- Some("https://example.com/atom/1")
- );
-}
-
-#[test]
-fn parse_feed_detects_format() {
- let rss = "T ";
- assert!(parse_feed_full(rss).is_ok());
-
- let atom = "T 1 ";
- assert!(parse_feed_full(atom).is_ok());
-
- assert!(parse_feed_full("").is_err());
-}
-
-// ── Entry timestamps ───────────────────────────────────────────────
-
-#[test]
-fn parse_rss_emits_pubdate_timestamp() {
- let xml = r#"
- -
-
Post
- 1
- Wed, 02 Oct 2002 13:00:00 GMT
-
- "#;
-
- let entries = parse_rss(xml).unwrap();
- // 2002-10-02T13:00:00Z in epoch milliseconds.
- assert_eq!(entries[0].updated_at_ms, Some(1_033_563_600_000));
-}
-
-#[test]
-fn parse_atom_emits_updated_timestamp() {
- let xml = r#"
-
- Atom entry
- urn:entry:1
- 2026-03-01T12:00:00.000Z
-
- "#;
-
- let entries = parse_atom(xml).unwrap();
- // 2026-03-01T12:00:00Z in epoch milliseconds.
- assert_eq!(entries[0].updated_at_ms, Some(1_772_366_400_000));
-}
-
-#[test]
-fn parse_item_without_timestamp_is_unknown() {
- let xml = "T ";
- let entries = parse_rss(xml).unwrap();
- assert_eq!(entries[0].updated_at_ms, None);
-}
-
-#[test]
-fn rss_timestamp_ms_rejects_garbage() {
- assert_eq!(rss_timestamp_ms("not a date"), None);
- assert_eq!(rss_timestamp_ms(""), None);
-}
-
-// ── CDATA unwrapping ────────────────────────────────────────────────
-
-#[test]
-fn extract_tag_unwraps_cdata() {
- // The common RSS shape `body
]]>`
- // must yield clean HTML, not the literal CDATA markers.
- let xml = "body
]]>";
- assert_eq!(
- extract_tag(xml, "description").as_deref(),
- Some("body
")
- );
-}
-
-#[test]
-fn extract_tag_does_not_entity_decode_inside_cdata() {
- // CDATA content is literal: `<` must survive intact, not become `<`.
- let xml = " ";
- assert_eq!(
- extract_tag(xml, "description").as_deref(),
- Some("Say <tag> literally")
- );
-}
-
-#[test]
-fn extract_tag_entity_decodes_outside_cdata() {
- let xml = "A & B <b> bold ";
- assert_eq!(extract_tag(xml, "title").as_deref(), Some("A & B bold"));
-}
-
-#[test]
-fn extract_cdata_reuses_tag_extraction() {
- // `content:encoded` is typically CDATA-wrapped; extract_cdata must match
- // extract_tag on the same input.
- let xml = "full
]]>";
- assert_eq!(
- extract_cdata(xml, "content:encoded").as_deref(),
- Some("full
")
- );
-}
-
-#[test]
-fn parse_rss_description_with_cdata_is_clean() {
- let xml = r#"
- -
-
Post
- 1
- Hello world ]]>
-
- "#;
-
- let entries = parse_rss(xml).unwrap();
- assert_eq!(entries.len(), 1);
- assert_eq!(entries[0].body, "Hello world
");
- assert!(!entries[0].body.contains("CDATA"));
-}
-
-#[test]
-fn parse_rss_empty_description_falls_back_to_encoded_content() {
- // A present-but-empty ` ` must not block the
- // `content:encoded` fallback — the item carries its body there instead.
- let xml = r#"
- -
-
Post
- 1
-
- Full body from content:encoded]]>
-
- "#;
-
- let entries = parse_rss(xml).unwrap();
- assert_eq!(entries.len(), 1);
- assert_eq!(entries[0].body, "Full body from content:encoded
");
-}
-
-#[test]
-fn parse_atom_empty_content_falls_back_to_summary() {
- // Mirrors the RSS `description`/`content:encoded` pair: an empty
- // ` ` must fall through to a populated ``.
- let xml = r#"
-
- Atom entry
- urn:entry:1
-
- Summary body
-
- "#;
-
- let entries = parse_atom(xml).unwrap();
- assert_eq!(entries.len(), 1);
- assert_eq!(entries[0].body, "Summary body");
-}
-
-// ── URL host redaction ──────────────────────────────────────────────
-
-#[test]
-fn url_host_redacts_userinfo() {
- // Credentials embedded in a source URL must never reach debug traces —
- // only the host is logged.
- assert_eq!(
- url_host("https://alice:secret@example.com/feed.xml"),
- "example.com"
- );
-}
-
-#[test]
-fn url_host_drops_path_query_and_fragment() {
- assert_eq!(
- url_host("https://example.com/feed?token=abc#top"),
- "example.com"
- );
-}
-
-#[test]
-fn url_host_fallback_strips_userinfo_without_scheme() {
- // Scheme-less values are unparseable by reqwest; the textual fallback
- // must still drop the `user:pass@` prefix and the path.
- assert_eq!(url_host("alice:secret@example.com/feed"), "example.com");
- assert_eq!(url_host("example.com/feed"), "example.com");
-}
-
-// ── Entity decoding ─────────────────────────────────────────────────
-
-#[test]
-fn feed_text_entities_decode_exactly_once() {
- // `<` is the escaped form of `<`; it must decode once to `<`,
- // not twice to `<`.
- assert_eq!(
- extract_tag("< ", "title").as_deref(),
- Some("<")
- );
- assert_eq!(
- extract_tag("& ", "title").as_deref(),
- Some("&")
- );
-}
-
-#[test]
-fn feed_text_decodes_every_predefined_xml_entity_and_numeric_references() {
- assert_eq!(
- extract_tag(
- "<b> "q" 'a' & more ’ ",
- "title"
- )
- .as_deref(),
- Some(" \"q\" 'a' & more \u{2019}")
- );
-}
diff --git a/crates/tinymemory-integrations/src/sources/readers/rss/types.rs b/crates/tinymemory-integrations/src/sources/readers/rss/types.rs
deleted file mode 100644
index db94f9ce..00000000
--- a/crates/tinymemory-integrations/src/sources/readers/rss/types.rs
+++ /dev/null
@@ -1,25 +0,0 @@
-//! Private feed-shape types for the RSS reader: the parsed entry model and
-//! the short-lived cache snapshot reused across a list-then-read sync pass.
-
-use std::time::Instant;
-
-/// A fetched feed snapshot cached across a list-then-read sync pass.
-#[derive(Debug)]
-pub(super) struct FeedCache {
- pub url: String,
- pub fetched_at: Instant,
- pub entries: Vec,
-}
-
-/// One parsed RSS/Atom entry: the dedupe id, the fields surfaced as a source
-/// item / content, and the raw link + publication timestamp carried in the
-/// content metadata.
-#[derive(Debug, Clone)]
-pub(super) struct FeedEntry {
- pub id: String,
- pub title: String,
- pub body: String,
- pub link: Option,
- pub published: Option,
- pub updated_at_ms: Option,
-}
diff --git a/crates/tinymemory-integrations/src/sources/readers/web_page/mod.rs b/crates/tinymemory-integrations/src/sources/readers/web_page/mod.rs
deleted file mode 100644
index f7841dac..00000000
--- a/crates/tinymemory-integrations/src/sources/readers/web_page/mod.rs
+++ /dev/null
@@ -1,444 +0,0 @@
-//! Web page source reader.
-//!
-//! Fetches a single URL and extracts its content. When a CSS `selector` is
-//! configured, only the text of matching elements is included (plain text);
-//! otherwise the whole page is converted to markdown through
-//! `tinymemory_integrations::documents::html::to_markdown`, keeping its headings, lists and
-//! links.
-//!
-//! The page is fetched through `sources::fetch`, behind its SSRF guard
-//! (scheme/host policy plus a DNS resolver that pins connections to globally
-//! routable addresses), with a 10 MiB body cap.
-
-mod types;
-
-use async_trait::async_trait;
-
-use types::SelectorSpec;
-
-use crate::sources::fetch::fetch_url_capped;
-
-use crate::sources::error::{Error, Result};
-use crate::sources::types::{
- ContentType, MemorySourceEntry, SourceContent, SourceItem, SourceKind,
-};
-
-use super::SourceReader;
-
-/// Largest page body the reader will buffer.
-const MAX_BODY_BYTES: u64 = 10 * 1024 * 1024;
-
-/// Reader for a single-page web source: fetches one URL and extracts its
-/// readable text.
-#[derive(Debug, Clone, Copy, Default)]
-pub struct WebPageReader;
-
-#[async_trait]
-impl SourceReader for WebPageReader {
- fn kind(&self) -> SourceKind {
- SourceKind::WebPage
- }
-
- async fn list_items(
- &self,
- source: &MemorySourceEntry,
- _workspace: &std::path::Path,
- ) -> Result> {
- let url = configured_url(source)?;
- Ok(vec![SourceItem {
- id: url.to_string(),
- title: source.label.clone(),
- updated_at_ms: None,
- }])
- }
-
- async fn read_item(
- &self,
- source: &MemorySourceEntry,
- item_id: &str,
- _workspace: &std::path::Path,
- ) -> Result {
- let url = if item_id.starts_with("http") {
- item_id.to_string()
- } else {
- configured_url(source)?.to_string()
- };
-
- tracing::debug!(
- selector = ?source.selector,
- "[memory_sources:web_page] reading item"
- );
-
- // `fetch_url_capped` applies the SSRF guard (scheme and host policy,
- // public-only DNS, per-hop redirect checks) and streams the body
- // against the cap, so a hostile or giant page cannot exhaust memory.
- let document = fetch_url_capped(&url, MAX_BODY_BYTES).await?;
- let body = String::from_utf8_lossy(&document.bytes).into_owned();
-
- let title = crate::documents::html::extract_title(&body).unwrap_or_else(|| url.clone());
- let (extracted, content_type) = match source.selector.as_deref() {
- Some(selector) => (extract_by_selector(&body, selector), ContentType::Plaintext),
- None => (
- crate::documents::html::to_markdown(&body),
- ContentType::Markdown,
- ),
- };
-
- Ok(SourceContent {
- id: url.clone(),
- title,
- body: extracted,
- content_type,
- metadata: serde_json::json!({ "url": url }),
- })
- }
-}
-
-/// The configured page URL.
-fn configured_url(source: &MemorySourceEntry) -> Result<&str> {
- source
- .url
- .as_deref()
- .ok_or_else(|| Error::Invalid("web_page source requires a url".to_string()))
-}
-
-// ── Text extraction ─────────────────────────────────────────────────
-
-fn parse_selector(selector: &str) -> Option {
- let last = selector
- .trim()
- .rsplit(char::is_whitespace)
- .next()
- .unwrap_or("")
- .trim();
- if last.is_empty() {
- return None;
- }
-
- let mut spec = SelectorSpec {
- tag: None,
- id: None,
- classes: Vec::new(),
- };
- let mut part = String::new();
- let mut sep = ' '; // leading bare token is the tag
- for ch in last.chars() {
- match ch {
- '.' | '#' => {
- push_selector_part(&mut spec, &mut part, sep);
- sep = ch;
- }
- _ => part.push(ch),
- }
- }
- push_selector_part(&mut spec, &mut part, sep);
-
- if spec.tag.is_none() && spec.id.is_none() && spec.classes.is_empty() {
- None
- } else {
- Some(spec)
- }
-}
-
-fn push_selector_part(spec: &mut SelectorSpec, part: &mut String, sep: char) {
- let part = std::mem::take(part);
- if part.is_empty() {
- return;
- }
- match sep {
- '#' => spec.id = Some(part),
- '.' => spec.classes.push(part),
- _ => {
- if spec.tag.is_none() {
- spec.tag = Some(part);
- } else {
- spec.classes.push(part);
- }
- }
- }
-}
-
-/// Extract text from elements matching a simple CSS selector.
-///
-/// Falls back to the whole stripped page when the selector never matches
-/// (rather than erroring), mirroring the reader's lenient posture for pages
-/// whose structure changes between list and read time.
-fn extract_by_selector(html: &str, selector: &str) -> String {
- let Some(spec) = parse_selector(selector) else {
- return strip_html_tags(html);
- };
- // Match against a script/style-stripped copy so JS strings and CSS rules
- // cannot be mistaken for nested elements, and so a selector that lands on
- // a `World
";
- let result = strip_html_tags(html);
- assert_eq!(result, "Hello World");
-}
-
-#[test]
-fn strip_script_and_style_handles_unclosed() {
- // Unclosed `