diff --git a/.env.example b/.env.example index 5bcfa8639..b8aee7430 100644 --- a/.env.example +++ b/.env.example @@ -50,9 +50,11 @@ TEMPORAL_TASK_QUEUE=investigations # API Configuration # ----------------------------------------------------------------------------- -# LLM model to use for investigations -# Options: claude-sonnet-4-20250514, claude-opus-4-20250514 -LLM_MODEL=claude-sonnet-4-20250514 +# Claude models +# Model overrides. Unset, the defaults in python-packages/dataing/src/dataing/config.py +# apply (claude-sonnet-5-5 for investigations and issue chat). +# LLM_MODEL=claude-sonnet-5-5 +# CHAT_AGENT_MODEL=claude-sonnet-5-5 # ----------------------------------------------------------------------------- # Optional: Local Data Sources diff --git a/.flow/tasks/fn-69.12.json b/.flow/tasks/fn-69.12.json index a990daef8..a0c371082 100644 --- a/.flow/tasks/fn-69.12.json +++ b/.flow/tasks/fn-69.12.json @@ -13,5 +13,5 @@ "spec_path": ".flow/tasks/fn-69.12.md", "status": "todo", "title": "[M1] Frontend: per-user datasource credentials page", - "updated_at": "2026-09-27T15:11:28.435439Z" + "updated_at": "2026-09-29T01:05:48.953573Z" } diff --git a/.flow/tasks/fn-69.12.md b/.flow/tasks/fn-69.12.md index 38673dfcd..a5b9a59a1 100644 --- a/.flow/tasks/fn-69.12.md +++ b/.flow/tasks/fn-69.12.md @@ -2,11 +2,18 @@ ## Description ### Goal -Build the page that the gateway's 403 links to (`/settings/datasources/{id}/credentials`). It doesn't exist yet (design §5.2). +Extend the per-user credentials page that the gateway's 403 and the issue agent's `credentials_missing` link go to (`/settings/datasources/{id}/credentials`, design §5.2). + +A first cut already exists from the issue hub (spec 0001 §8.5): `features/settings/datasource-credentials-page.tsx`, with hooks in `lib/api/credentials.ts`. It has: +- username and password, plus role and warehouse where the source type's config takes them +- a test of the login before it is saved +- Remove +- a message for source types with no login +- a return to the page that linked to it ### Implementation -- **Page:** a route and page under `features/settings/`, with forms generated from fn-69.31's schema for each adapter: username/password, key pair, or service-account JSON. - - Test and Delete buttons. +- **Page:** replace the fixed fields with forms generated from fn-69.31's schema for each adapter: username/password, key pair, or service-account JSON. + - Keep test-before-save and Remove. - Never echo secrets back. - **403 handling:** catch 403 `credentials_not_configured` globally in `lib/api/client.ts`, next to the existing `feature_not_available` handling at `:58-65`. Show a toast that links to the page. ## Acceptance diff --git a/.flow/tasks/fn-70.17.json b/.flow/tasks/fn-70.17.json index 7d82faa90..940ae7702 100644 --- a/.flow/tasks/fn-70.17.json +++ b/.flow/tasks/fn-70.17.json @@ -1,14 +1,26 @@ { - "assignee": null, + "assignee": "bordumbb@gmail.com", "claim_note": "", - "claimed_at": null, + "claimed_at": "2026-09-28T17:52:29.670737Z", "created_at": "2026-09-28T15:16:44.759472Z", "depends_on": [], "epic": "fn-70", + "evidence": { + "commits": [ + "1a31082c" + ], + "prs": [ + "https://github.com/bordumb/dataing/pull/211" + ], + "tests": [ + "tests/unit/test_config.py::test_chat_agent_defaults_to_claude_opus_5_5", + "tests/unit/agents/test_chat_agent.py::TestBriefDrafting::test_draft_is_parsed_from_a_json_reply_without_forcing_a_tool" + ] + }, "id": "fn-70.17", "priority": null, "spec_path": ".flow/tasks/fn-70.17.md", - "status": "todo", + "status": "done", "title": "Run the issue chat agent on Claude Opus 5.5", - "updated_at": "2026-09-28T15:16:44.762271Z" + "updated_at": "2026-09-28T17:52:29.952602Z" } diff --git a/.flow/tasks/fn-70.17.md b/.flow/tasks/fn-70.17.md index f03a5a15f..b4a120ecd 100644 --- a/.flow/tasks/fn-70.17.md +++ b/.flow/tasks/fn-70.17.md @@ -11,9 +11,8 @@ TBD ## Done summary -TBD - +CHAT_AGENT_MODEL defaults to claude-opus-5-5; brief drafting returns JSON text (PromptedOutput) instead of a forced output tool, which Opus 5.5 rejects. Merged as PR #211 (1a31082c). ## Evidence -- Commits: -- Tests: -- PRs: +- Commits: 1a31082c +- Tests: tests/unit/test_config.py::test_chat_agent_defaults_to_claude_opus_5_5, tests/unit/agents/test_chat_agent.py::TestBriefDrafting::test_draft_is_parsed_from_a_json_reply_without_forcing_a_tool +- PRs: https://github.com/bordumb/dataing/pull/211 diff --git a/.flow/tasks/fn-70.18.json b/.flow/tasks/fn-70.18.json new file mode 100644 index 000000000..1acd0c8e6 --- /dev/null +++ b/.flow/tasks/fn-70.18.json @@ -0,0 +1,21 @@ +{ + "assignee": "bordumbb@gmail.com", + "claim_note": "", + "claimed_at": "2026-09-28T17:54:28.759022Z", + "created_at": "2026-09-28T17:51:58.959592Z", + "depends_on": [], + "epic": "fn-70", + "evidence": { + "commits": [ + "f292971a" + ], + "prs": [], + "tests": [] + }, + "id": "fn-70.18", + "priority": null, + "spec_path": ".flow/tasks/fn-70.18.md", + "status": "done", + "title": "M5: spec revision: every run in the hub, one-step start, details page, LLM failures", + "updated_at": "2026-09-28T17:54:29.035901Z" +} diff --git a/.flow/tasks/fn-70.18.md b/.flow/tasks/fn-70.18.md new file mode 100644 index 000000000..ad8941afc --- /dev/null +++ b/.flow/tasks/fn-70.18.md @@ -0,0 +1,16 @@ +# fn-70.18 M5: spec revision: every run in the hub, one-step start, details page, LLM failures + +## Description +TBD + +## Acceptance +- docs/specs/0001_issue_chat.md records D12–D15, the §2 "found after M1–M4" table, §7.11 (one starter, open_issue, every path), §7.12 (LLM error table, failing the run, key check), §8.1–8.4 (mockup fidelity, Investigate…, details page, banner), M5 in §10 and its tests in §11 +- Owner decisions (2026-09-28): one-click Investigate…, details page off the card, always open an issue, fail the run + key check + + +## Done summary +Spec revision committed: D12-D15, §7.11, §7.12, §8.1-8.4, M5. +## Evidence +- Commits: f292971a +- Tests: +- PRs: diff --git a/.flow/tasks/fn-70.19.json b/.flow/tasks/fn-70.19.json new file mode 100644 index 000000000..5de08f7b6 --- /dev/null +++ b/.flow/tasks/fn-70.19.json @@ -0,0 +1,30 @@ +{ + "assignee": "bordumbb@gmail.com", + "claim_note": "", + "claimed_at": "2026-09-28T17:55:54.224028Z", + "created_at": "2026-09-28T17:51:59.221401Z", + "depends_on": [ + "fn-70.18" + ], + "epic": "fn-70", + "evidence": { + "commits": [ + "b54b8725" + ], + "prs": [], + "tests": [ + "tests/unit/agents/test_llm_errors.py", + "tests/unit/temporal/test_llm_activity_errors.py", + "tests/unit/temporal/test_investigation_failures.py", + "tests/unit/temporal/test_investigation_replay.py", + "tests/integration/test_publish_outcome.py", + "CE unit suite: 3169 passed" + ] + }, + "id": "fn-70.19", + "priority": null, + "spec_path": ".flow/tasks/fn-70.19.md", + "status": "done", + "title": "M5: LLM failures fail the run (classify, activities raise, workflow fails)", + "updated_at": "2026-09-28T18:22:06.119415Z" +} diff --git a/.flow/tasks/fn-70.19.md b/.flow/tasks/fn-70.19.md new file mode 100644 index 000000000..4e88b4de1 --- /dev/null +++ b/.flow/tasks/fn-70.19.md @@ -0,0 +1,20 @@ +# fn-70.19 M5: LLM failures fail the run (classify, activities raise, workflow fails) + +## Description +TBD + +## Acceptance +- `agents/errors.py` `classify_llm_error(exc)` maps the real exception chain (LLMError → ModelHTTPError/ModelAPIError → anthropic errors, UserError for a missing key) to the §7.12 codes, retryability and messages; unit-tested per row +- generate_hypotheses, generate_query, interpret_evidence, synthesize and counter_analyze raise ApplicationError `LLMRejected` (non-retryable) or `LLMUnavailable` with `{code, message}` details; AgentClient.interpret_evidence no longer swallows errors +- Every LLM activity call has an explicit RetryPolicy (4 attempts, 5 s, ×2, max 60 s, LLMRejected non-retryable) +- Behind `workflow.patched("llm-failures-v1")`, the workflow fails the run on: generation failure or no hypotheses, any subagent LLM error (others cancelled), all hypotheses untested from errors, synthesis failure. Counter-analysis failure keeps the synthesis and records counter_analysis.error +- A failed run publishes `{"status":"failed","error":{code,message,step}}` via publish_investigation_outcome (outcome, run row completed_at, thread card), then raises a non-retryable ApplicationError +- Workflow tests with fake activities cover each case; Replayer tests pass on the recorded histories + + +## Done summary +classify_llm_error (agents/errors.py) maps anthropic → pydantic-ai → LLMError chains to the §7.12 codes. LLM activities raise LLMRejected/LLMUnavailable with an explicit retry policy; AgentClient.interpret_evidence no longer swallows errors. Behind llm-failures-v1 the workflow fails the run (generation failure/no hypotheses, subagent LLM error cancelling the others, nothing testable, synthesis failure), publishes {"status":"failed","error":{code,message,step}} and fails the Temporal execution as InvestigationFailed. Counter-analysis failure keeps the conclusion. +## Evidence +- Commits: b54b8725 +- Tests: tests/unit/agents/test_llm_errors.py, tests/unit/temporal/test_llm_activity_errors.py, tests/unit/temporal/test_investigation_failures.py, tests/unit/temporal/test_investigation_replay.py, tests/integration/test_publish_outcome.py, CE unit suite: 3169 passed +- PRs: diff --git a/.flow/tasks/fn-70.20.json b/.flow/tasks/fn-70.20.json new file mode 100644 index 000000000..28b8cc8d1 --- /dev/null +++ b/.flow/tasks/fn-70.20.json @@ -0,0 +1,29 @@ +{ + "assignee": "bordumbb@gmail.com", + "claim_note": "", + "claimed_at": "2026-09-28T19:34:45.513313Z", + "created_at": "2026-09-28T17:51:59.475477Z", + "depends_on": [ + "fn-70.19" + ], + "epic": "fn-70", + "evidence": { + "commits": [ + "784d88d6" + ], + "prs": [], + "tests": [ + "tests/unit/services/test_llm_status.py", + "tests/unit/entrypoints/api/routes/test_system.py", + "tests/unit/temporal/test_issue_thread_workflow.py", + "tests/integration/test_agent_turn_activity.py", + "CE unit 3186, CE integration 145, EE unit 714" + ] + }, + "id": "fn-70.20", + "priority": null, + "spec_path": ".flow/tasks/fn-70.20.md", + "status": "done", + "title": "M5: LLM key check at startup, GET /system/llm, readable chat errors", + "updated_at": "2026-09-28T20:10:32.733100Z" +} diff --git a/.flow/tasks/fn-70.20.md b/.flow/tasks/fn-70.20.md new file mode 100644 index 000000000..6ca918b6e --- /dev/null +++ b/.flow/tasks/fn-70.20.md @@ -0,0 +1,18 @@ +# fn-70.20 M5: LLM key check at startup, GET /system/llm, readable chat errors + +## Description +TBD + +## Acceptance +- API startup checks `GET /v1/models/{id}` for LLM_MODEL and CHAT_AGENT_MODEL in the background (10 s timeout, no SDK retries); an empty key is reported without a request +- `GET /api/v1/system/llm` (ANY_USER, POLICY entry) returns {state, message, models, checked_at}; stale `unreachable` re-checks on read +- run_agent_turn classifies model errors: the reply/brief shows the §7.12 message, not the raw ModelHTTPError; non-retryable errors aren't retried +- Tests fake the Anthropic client for missing, rejected (401), unknown model (404) and ok + + +## Done summary +LLMStatusChecker checks GET /v1/models/{id} for LLM_MODEL and CHAT_AGENT_MODEL in the background at API startup (10 s timeout, no SDK retries; empty key reported without a request). GET /api/v1/system/llm returns {state, message, models, checked_at}; transient problems re-check after 60 s. Chat turns and brief drafts raise the classified LLMRejected/LLMUnavailable error, so replies show what to fix and rejected keys aren't retried. Verified live: a bogus key → invalid_key. +## Evidence +- Commits: 784d88d6 +- Tests: tests/unit/services/test_llm_status.py, tests/unit/entrypoints/api/routes/test_system.py, tests/unit/temporal/test_issue_thread_workflow.py, tests/integration/test_agent_turn_activity.py, CE unit 3186, CE integration 145, EE unit 714 +- PRs: diff --git a/.flow/tasks/fn-70.21.json b/.flow/tasks/fn-70.21.json new file mode 100644 index 000000000..1803c1fd7 --- /dev/null +++ b/.flow/tasks/fn-70.21.json @@ -0,0 +1,29 @@ +{ + "assignee": "bordumbb@gmail.com", + "claim_note": "", + "claimed_at": "2026-09-28T18:22:06.382549Z", + "created_at": "2026-09-28T17:51:59.729817Z", + "depends_on": [ + "fn-70.18" + ], + "epic": "fn-70", + "evidence": { + "commits": [ + "d7522281" + ], + "prs": [], + "tests": [ + "tests/integration/test_open_issue.py", + "tests/integration/api/test_start_investigation.py", + "tests/integration/api/test_issue_investigation_runs.py", + "dataing-ee tests/integration/core/automation/test_spawn_investigation.py", + "CE unit 3173, EE unit 714, CE integration 143, EE integration 16, SDK+CLI 272" + ] + }, + "id": "fn-70.21", + "priority": null, + "spec_path": ".flow/tasks/fn-70.21.md", + "status": "done", + "title": "M5: one starter and open_issue() for every start path", + "updated_at": "2026-09-28T19:34:45.173067Z" +} diff --git a/.flow/tasks/fn-70.21.md b/.flow/tasks/fn-70.21.md new file mode 100644 index 000000000..32441cb34 --- /dev/null +++ b/.flow/tasks/fn-70.21.md @@ -0,0 +1,22 @@ +# fn-70.21 M5: one starter and open_issue() for every start path + +## Description +TBD + +## Acceptance +- `adapters/db/issues.py` `open_issue()` is the only issue insert (POST /issues, CE + EE webhooks, the starter) and posts the thread's first entry via the `created` event +- `InvestigationStarterService.start()` resolves or opens the issue, builds the missing brief/alert, writes the investigation row, run row (trigger_type human/api/webhook/rule) and start card, starts the workflow with alert.issue_id, and records a failed outcome if the start fails +- POST /investigations takes exactly one of brief/alert (+ datasource_id, execution_profile, issue_id); bad alert → 422; response adds run_id, issue_id, issue_number +- POST /issues/{id}/investigation-runs, CE webhook-generic AUTO, EE provider webhook AUTO and the EE rule action all use the starter +- InvestigationRunResponse gains number, status, error; InvestigationStateResponse gains issue_id, issue_number, issue_title, run_number, brief, execution_profile, error +- ToolCallRecord stores duration_ms and row_count for run_query +- SDK Investigation gains issue_id/issue_number; `dataing run start` prints the issue URL +- Integration tests on migrated_db for each path in the §7.11 table + + +## Done summary +open_issue() is the only issue insert (API, CE + EE webhooks, starter) and posts the thread's opening event. InvestigationStarterService.start() resolves/opens the issue, builds the missing brief/alert, writes investigation + run row + card, starts the workflow with alert.issue_id, and records a failed start on the card. POST /investigations takes brief|alert (+issue_id, profile) and returns issue_id/issue_number/run_id; spawn route, webhooks and EE rule action use the starter. Runs report number/status/error; GET /investigations/{id} adds issue, run number, brief, error, hypotheses. Tool calls record duration_ms/row_count. SDK Investigation gains issue fields; CLI links the issue. +## Evidence +- Commits: d7522281 +- Tests: tests/integration/test_open_issue.py, tests/integration/api/test_start_investigation.py, tests/integration/api/test_issue_investigation_runs.py, dataing-ee tests/integration/core/automation/test_spawn_investigation.py, CE unit 3173, EE unit 714, CE integration 143, EE integration 16, SDK+CLI 272 +- PRs: diff --git a/.flow/tasks/fn-70.22.json b/.flow/tasks/fn-70.22.json new file mode 100644 index 000000000..b218d9c67 --- /dev/null +++ b/.flow/tasks/fn-70.22.json @@ -0,0 +1,26 @@ +{ + "assignee": "bordumbb@gmail.com", + "claim_note": "", + "claimed_at": "2026-09-28T20:41:22.721926Z", + "created_at": "2026-09-28T17:51:59.979116Z", + "depends_on": [ + "fn-70.21" + ], + "epic": "fn-70", + "evidence": { + "commits": [ + "3a71fb9b" + ], + "prs": [], + "tests": [ + "StartInvestigation.test.tsx", + "frontend vitest 172 passed" + ] + }, + "id": "fn-70.22", + "priority": null, + "spec_path": ".flow/tasks/fn-70.22.md", + "status": "done", + "title": "M5: frontend: Investigate\u2026 everywhere, remove /investigations/new", + "updated_at": "2026-09-28T20:41:23.797636Z" +} diff --git a/.flow/tasks/fn-70.22.md b/.flow/tasks/fn-70.22.md new file mode 100644 index 000000000..7d73aaf78 --- /dev/null +++ b/.flow/tasks/fn-70.22.md @@ -0,0 +1,18 @@ +# fn-70.22 M5: frontend: Investigate… everywhere, remove /investigations/new + +## Description +TBD + +## Acceptance +- /investigations/new route, NewInvestigation.tsx and the dead components/Layout.tsx are gone; no link points at /investigations/new +- Investigate… (sidebar quick action, dashboard header + empty state, investigations list header + empty state, dataset page header "Investigate this dataset") opens the brief editor in new mode, pre-filled from the page +- New mode: "Start an investigation", symptom + ≥1 scope table required, datasource required only with >1 datasource; Start → POST /investigations → navigate to /issues/{issue_id} +- vitest covers each entry point and the new-mode submit + navigation + + +## Done summary +Investigate… everywhere: StartInvestigationDialog/InvestigateButton (brief editor in new mode) on the sidebar, dashboard, investigations list and dataset page; POST /investigations → /issues/{id}; /investigations/new, NewInvestigation.tsx and Layout.tsx removed. +## Evidence +- Commits: 3a71fb9b +- Tests: StartInvestigation.test.tsx, frontend vitest 172 passed +- PRs: diff --git a/.flow/tasks/fn-70.23.json b/.flow/tasks/fn-70.23.json new file mode 100644 index 000000000..d14dabeb6 --- /dev/null +++ b/.flow/tasks/fn-70.23.json @@ -0,0 +1,28 @@ +{ + "assignee": "bordumbb@gmail.com", + "claim_note": "", + "claimed_at": "2026-09-28T20:41:23.059873Z", + "created_at": "2026-09-28T17:52:00.239444Z", + "depends_on": [ + "fn-70.21" + ], + "epic": "fn-70", + "evidence": { + "commits": [ + "8192ad97" + ], + "prs": [], + "tests": [ + "IssueWorkspace.test.tsx", + "IssueSidebar.test.tsx", + "IssueThread.test.tsx", + "InvestigationCards.test.tsx" + ] + }, + "id": "fn-70.23", + "priority": null, + "spec_path": ".flow/tasks/fn-70.23.md", + "status": "done", + "title": "M5: frontend: issue page matches the mockup", + "updated_at": "2026-09-28T20:41:24.133203Z" +} diff --git a/.flow/tasks/fn-70.23.md b/.flow/tasks/fn-70.23.md new file mode 100644 index 000000000..45c9270f4 --- /dev/null +++ b/.flow/tasks/fn-70.23.md @@ -0,0 +1,19 @@ +# fn-70.23 M5: frontend: issue page matches the mockup + +## Description +TBD + +## Acceptance +- Issue page matches 0001_issue_chat_mockup.html: one-row top bar; description as the thread's first entry; tabs Shared thread / My scratch chats (N) + "N watching · live"; one sidebar panel (Status with note, Details, Dataset + open dataset page, Investigations, Watchers, Your scratch chats); mockup pill colours +- Tool calls collapse to "Ran N queries · X ms · Y rows"; footer "Snapshot saved with this message · copy SQL" +- Investigation card: "Investigation #N", details → link, failed pill + reason + Retry (reopens the editor with the same brief); headless runs attributed to dataing +- Sidebar runs show number, depth and status (failed no longer "running") +- vitest updated/added; screenshots compared against the mockup + + +## Done summary +Issue page follows the mockup: one-row top bar, IssueOpened first entry (description, dataing-opened line), Shared thread / My scratch chats tabs with watching · live, one sidebar panel (status note, details, dataset link, runs #N with status, watchers, scratch chats), mockup pills, tool-call totals, Investigation #N card with details → and failed state + Retry. +## Evidence +- Commits: 8192ad97 +- Tests: IssueWorkspace.test.tsx, IssueSidebar.test.tsx, IssueThread.test.tsx, InvestigationCards.test.tsx +- PRs: diff --git a/.flow/tasks/fn-70.24.json b/.flow/tasks/fn-70.24.json new file mode 100644 index 000000000..f88eb66e6 --- /dev/null +++ b/.flow/tasks/fn-70.24.json @@ -0,0 +1,27 @@ +{ + "assignee": "bordumbb@gmail.com", + "claim_note": "", + "claimed_at": "2026-09-28T20:41:23.396318Z", + "created_at": "2026-09-28T17:52:00.495921Z", + "depends_on": [ + "fn-70.20", + "fn-70.21" + ], + "epic": "fn-70", + "evidence": { + "commits": [ + "049049dc" + ], + "prs": [], + "tests": [ + "InvestigationDetail.test.tsx", + "llm-status-banner.test.tsx" + ] + }, + "id": "fn-70.24", + "priority": null, + "spec_path": ".flow/tasks/fn-70.24.md", + "status": "done", + "title": "M5: frontend: run details page and LLM banner", + "updated_at": "2026-09-28T20:41:24.467832Z" +} diff --git a/.flow/tasks/fn-70.24.md b/.flow/tasks/fn-70.24.md new file mode 100644 index 000000000..c36b2be27 --- /dev/null +++ b/.flow/tasks/fn-70.24.md @@ -0,0 +1,17 @@ +# fn-70.24 M5: frontend: run details page and LLM banner + +## Description +TBD + +## Acceptance +- /investigations/:id is the run's details page: back link to the issue thread, "Investigation #N" + status/depth pills, Share copies the link (mock removed), Export snapshot downloads GET /investigations/{id}/snapshot, Add as check (renamed from Codify Test), Cancel while running; brief; hypotheses with status and evidence (queries collapse like the thread's); outcome card; failed runs show the reason +- A banner under the header on every page shows GET /system/llm problems and what to fix +- vitest covers the back link, export, failed state and the banner + + +## Done summary +Run details page: back link to the issue thread, Investigation #N with status/depth, Share copies the link, Export snapshot (blob GET), Add as check, Cancel; brief, hypotheses with grouped evidence, outcome card, failed reason. LLM status banner in AppLayout polling GET /system/llm. +## Evidence +- Commits: 049049dc +- Tests: InvestigationDetail.test.tsx, llm-status-banner.test.tsx +- PRs: diff --git a/.flow/tasks/fn-70.25.json b/.flow/tasks/fn-70.25.json new file mode 100644 index 000000000..abf3a1385 --- /dev/null +++ b/.flow/tasks/fn-70.25.json @@ -0,0 +1,21 @@ +{ + "assignee": null, + "claim_note": "", + "claimed_at": null, + "created_at": "2026-09-28T17:52:00.750051Z", + "depends_on": [ + "fn-70.19", + "fn-70.20", + "fn-70.21", + "fn-70.22", + "fn-70.23", + "fn-70.24" + ], + "epic": "fn-70", + "id": "fn-70.25", + "priority": null, + "spec_path": ".flow/tasks/fn-70.25.md", + "status": "todo", + "title": "M5: OpenAPI client refresh, full checks, demo walk-through", + "updated_at": "2026-09-28T17:52:00.750371Z" +} diff --git a/.flow/tasks/fn-70.25.md b/.flow/tasks/fn-70.25.md new file mode 100644 index 000000000..834320aed --- /dev/null +++ b/.flow/tasks/fn-70.25.md @@ -0,0 +1,18 @@ +# fn-70.25 M5: OpenAPI client refresh, full checks, demo walk-through + +## Description +TBD + +## Acceptance +- python-packages/dataing/openapi.json and the orval client carry the new/changed operations only (surgical: no unrelated drift) +- CE + EE pytest, ruff, ruff format --check, mypy, frontend vitest + typecheck + lint all green +- Demo stack walk-through: Investigate… from the dashboard lands in the issue thread; a bad key shows the banner and a failed card with the reason + + +## Done summary +TBD + +## Evidence +- Commits: +- Tests: +- PRs: diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index d4eea318a..e714a710d 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -488,7 +488,7 @@ Environment variables (`.env` or system): # Application ANTHROPIC_API_KEY=sk-... # Required for LLM DATABASE_URL=postgresql://... # App database -LLM_MODEL=claude-sonnet-4-20250514 # Model selection +LLM_MODEL=claude-sonnet-5-5 # Investigation model; defaults live in dataing/config.py # Security SECRET_KEY=... # JWT signing diff --git a/CLAUDE.md b/CLAUDE.md index 0f5655f72..d47ae0db9 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -82,7 +82,7 @@ The repo is open-core: ## Backend Architecture (CE) Core domain: `python-packages/dataing/src/dataing/core/` -- `investigation/` - Domain entities, repository, collaboration service +- `investigation/` - The investigation brief (what an issue thread hands to the manager) - `auth/`, `rbac/`, `entitlements/` - Identity and feature gating - `quality/` - LLM-as-judge quality validation - `state.py`, `domain_types.py`, `interfaces.py` - Event-sourced state + protocols diff --git a/docker-compose.yml b/docker-compose.yml index 823af2f6c..f4b93fc76 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -138,7 +138,9 @@ services: JWT_SECRET_KEY: ${JWT_SECRET_KEY:-} REDIS_HOST: redis REDIS_PORT: 6379 - LLM_MODEL: ${LLM_MODEL:-claude-sonnet-4-20250514} + # Model overrides; unset, dataing.config's defaults apply + LLM_MODEL: ${LLM_MODEL:-} + CHAT_AGENT_MODEL: ${CHAT_AGENT_MODEL:-} ports: - "8000:8000" depends_on: @@ -176,7 +178,9 @@ services: ANTHROPIC_API_KEY: ${ANTHROPIC_API_KEY:-} REDIS_HOST: redis REDIS_PORT: 6379 - LLM_MODEL: ${LLM_MODEL:-claude-sonnet-4-20250514} + # Model overrides; unset, dataing.config's defaults apply + LLM_MODEL: ${LLM_MODEL:-} + CHAT_AGENT_MODEL: ${CHAT_AGENT_MODEL:-} depends_on: db-migrate: condition: service_completed_successfully diff --git a/docs/docs/architecture.md b/docs/docs/architecture.md index 5baed77a3..229386bc8 100644 --- a/docs/docs/architecture.md +++ b/docs/docs/architecture.md @@ -231,7 +231,7 @@ class GenerateHypothesesStep(BondStep[Context, None, list[Hypothesis], str]): def create_agent(self, context): return BondAgent( name="hypothesis_generator", - model="anthropic:claude-sonnet-4-20250514", + model="anthropic:claude-sonnet-5-5", ) def build_prompt(self, context, input_data): diff --git a/docs/docs/integrations/warehouses/snowflake.md b/docs/docs/integrations/warehouses/snowflake.md index 19b3aaf37..779927f45 100644 --- a/docs/docs/integrations/warehouses/snowflake.md +++ b/docs/docs/integrations/warehouses/snowflake.md @@ -101,26 +101,25 @@ GRANT ROLE dataing_role TO USER dataing_user; ## Example Investigation -```python -import asyncio -from dataing.core.investigation.service import InvestigationService -from dataing.core.domain_types import AnomalyAlert +Start an investigation with the Python SDK. Every run opens an issue, and its +thread is where the team follows and steers it. -async def investigate_null_spike(): - alert = AnomalyAlert( - table="orders", - column="user_id", - metric="null_rate", - anomaly_type="spike", - description="NULL rate increased from 1% to 15%" - ) +```python +from dataing_sdk import DataingClient - service = InvestigationService() - result = await service.investigate(alert) +client = DataingClient(base_url="https://dataing.example.com", api_key="dd_...") - print(f"Root cause: {result.synthesis.root_cause}") +investigation = client.start_investigation( + dataset="analytics.public.orders", + anomaly_type="null_rate", + goal="NULL rate of user_id rose from 1% to 15%", + column="user_id", + expected_value=0.01, + actual_value=0.15, + datasource_id="", +) -asyncio.run(investigate_null_spike()) +print(f"Follow it in issue #{investigation.issue_number}") ``` --- diff --git a/docs/specs/0001_issue_chat.md b/docs/specs/0001_issue_chat.md index 27b893f4f..b73cca822 100644 --- a/docs/specs/0001_issue_chat.md +++ b/docs/specs/0001_issue_chat.md @@ -1,6 +1,6 @@ # 0001: Issue hub: shared agent chat, handoff and steering -**Status:** Draft, 2026-09-27 +**Status:** M1–M4 shipped in PR #209. Revised 2026-09-28 with D12–D15 and M5: every run reaches the hub, one-step start, the run's details page, LLM failures, and the mockup as the acceptance reference. **Edition:** CE. Nothing here is EE-only. @@ -23,6 +23,7 @@ The issue page becomes the place where a team works a data problem: - People can **steer a running investigation** (add context, rule out or add a hypothesis, stop and conclude) without restarting it. - Each person can keep **private scratch chats** and publish from them. - Results, confirmation and the resolution land back in the thread. +- **Every investigation lives in an issue**, however it started, and starting one is a single step from any page (D12, D13). "Manager" is `InvestigationWorkflow`. "Subagents" are the `EvaluateHypothesisWorkflow` children it starts, one per hypothesis (`python-packages/dataing/src/dataing/temporal/workflows/`). @@ -42,6 +43,17 @@ Verified on main at `6e8812c8`. Paths are under `python-packages/dataing/src/dat | Results | Nothing writes `issue_investigation_runs.synthesis_summary`. The summary card never renders, and "resolved via a linked investigation" can never pass. | `migrations/018_issues.sql`; `routes/issues.py` | | Ad-hoc questions | No agent can answer them. The investigation agent is built with no tools. | `agents/client.py` | +**Found after M1–M4 shipped** (main at `245c9d2e`, 2026-09-28): + +| Area | Today | Where | +|---|---|---| +| Starting a run | Every "New investigation" button opens `/investigations/new`. That page starts a run with no issue and no thread, then lands on the old run page, so nobody who starts there sees the hub. | `features/investigation/NewInvestigation.tsx` and 7 links to it | +| Runs outside the UI | `POST /investigations` (SDK, CLI, notebook) and both webhooks insert the run inline, with no run row and no start card. The EE rule action never writes its outcome back. Only one of 8 creation paths links fully to an issue. | `routes/investigations.py`, `routes/integrations.py`, EE `routes/integrations.py`, EE `core/automation/executor.py` | +| LLM failures | With a rejected key, every LLM step fails. The activities return empty results and the workflow logs warnings. The run "completes" at 0% confidence with "Unable to determine a definitive root cause". An interpretation failure reads as *refuted* evidence. The chat shows the raw `ModelHTTPError`. | `temporal/activities/*.py`, `agents/client.py`, `temporal/workflows/investigation.py` | +| Key check | Nothing checks the key. The first sign of a bad key is a run that did nothing. | `entrypoints/api/deps.py` | +| The issue page | It is close to the mockup but not the same: a separate description card, no Shared thread / scratch tabs, the sidebar split into three cards, tool calls without totals, and a card link labelled "Open". | `features/issues/` | +| The run page | Its Share menu is mocked. It has no hypotheses list, no snapshot export and no link back to the issue. | `features/investigation/InvestigationDetail.tsx` | + --- ## 3. Goals and non-goals @@ -55,6 +67,9 @@ Verified on main at `6e8812c8`. Paths are under `python-packages/dataing/src/dat 5. Private scratch chats with explicit publishing. 6. Results, confirmation and resolution flow back into the thread. 7. A sidebar that only offers moves that will succeed, with inline editing. +8. Every investigation reaches the hub: one step to start from any page, and runs started by the API, SDK, webhooks or checks open an issue. +9. A run that can't reach the model fails with the reason, and a broken key is visible before anyone starts a run. +10. The issue page follows the mockup (`0001_issue_chat_mockup.html`), which is the acceptance reference for §8. **Non-goals for v1** @@ -95,7 +110,11 @@ At any point, Maya can explore in a private scratch chat, then publish the usefu | D8 | Agent turns in a thread run **one at a time, in order**, through one Temporal workflow per thread. | A shared thread stays coherent when two people ask at once. | | D9 | Comments, agent replies and system events share **one timeline table**. Issue events also append a system entry. | One cursor for streaming, and no merging of sources in the UI. | | D10 | Every result the agent saw is **snapshotted with its message**, and every query is in the gateway's audit log. | Data changes; the thread must show what the agent actually saw. | -| D11 | Model: `claude-opus-5-5` (Claude Opus 5.5) with adaptive thinking, which it can't turn off. Effort is `low` for chat turns and `medium` for brief drafting, configurable per route. Brief drafting returns JSON text instead of calling an output tool, because Claude Opus 5.5 rejects a forced `tool_choice`. | Speed comes from effort, not from a smaller model. The owner chose Claude Opus 5.5 over Claude Opus 5 for cost: $4 / $20 per MTok against $5 / $25 (§12). | +| D11 | Model: `claude-sonnet-5-5` (Claude Sonnet 5.5), the same model as investigations; `dataing/config.py` is the one place that names models. Effort is `low` for chat turns and `medium` for brief drafting, configurable per route. Brief drafting returns JSON text instead of calling an output tool, because recent models (Claude Opus 5.5) reject a forced `tool_choice`. | Speed comes from effort. The owner moved chat from Claude Opus 5 to Opus 5.5 for cost, then on 2026-09-29 to Sonnet 5.5 with investigations, after Sonnet 4 was retired (§12). | +| D12 | **Every investigation belongs to an issue.** A run started without one (API, SDK, webhook, check) opens an issue, or reuses the open one for the same alert, and reports into its shared thread. There is no issue-less run. | One place to follow every run. The hub (thread, card, steering, outcome review) works for every run, not only the ones started from an issue. | +| D13 | **One step from intent to a running investigation.** Every start button opens the brief editor where the person already is, pre-filled with what that page knows (the dataset, the alert). Start opens the issue and the run together and lands on the issue's thread. `/investigations/new` is removed. | The old page was a middle layer that started runs outside the hub. Opening the issue on Start, not on click, means a cancelled editor leaves no empty issue, and the issue's title is the symptom the person typed. | +| D14 | **The thread's card is the main view of a run.** `/investigations/:id` becomes the run's details page, linked from the card and the sidebar. It keeps what the card has no room for: the evidence, every query with its result, share, snapshot export and Add as check. | The team works in the thread; the details page is for digging in. Nothing the old page did is lost. | +| D15 | **LLM failures fail the run, loudly.** Errors a retry can't fix (a missing or rejected key, an unknown model, a rejected request) fail the run at once with the reason on its card. Rate limits, overload and server errors retry with backoff first. A run never synthesizes a conclusion from steps that all failed. The API checks the key and models at startup, and every page shows a banner while they don't work. | A run that "completes" at 0% confidence after doing nothing hides the real problem. The fix is usually one environment variable, so say which. | --- @@ -212,6 +231,8 @@ All paths are under `/api/v1`. Gates use the names from the route authorization | `POST /investigations/{investigation_id}/steers` | SCOPE_WRITE | Body: `kind`, `text`, `hypothesis_id` | | `GET /investigations/{investigation_id}/steers` | ANY_USER | Includes each steer's status and outcome | | `POST /investigations/{investigation_id}/outcome-review` | SCOPE_WRITE | Body: `verdict` (`confirmed` or `rejected`), `note` | +| `POST /investigations` | SCOPE_WRITE | Starts a run from a `brief` or an `alert`, opening an issue unless `issue_id` is given (§7.11) | +| `GET /system/llm` | ANY_USER | Whether the key and models work (§7.12) | **Removed:** `GET/POST /issues/{issue_id}/comments`, `POST /investigations/{investigation_id}/messages` and `POST /investigations/{investigation_id}/input`. @@ -246,7 +267,7 @@ If polling load becomes a problem, switch to Postgres `LISTEN/NOTIFY` behind the 1. Builds the prompt (§7.6) and a `BondAgent` with the chat tools (§7.5). 2. Creates the reply message with `status = 'streaming'` and `requested_by_user_id` set to the asker. 3. Streams text into `body_md`, flushing every 250 ms or 200 characters. -4. Records each tool call in `payload.tool_calls` as `{id, tool, input, status, summary, query_result_id}`. +4. Records each tool call in `payload.tool_calls` as `{id, tool, input, status, summary, query_result_id, duration_ms, row_count}`. The last two are set for `run_query` only. 5. Ends in `complete`, `error` (with the message) or `cancelled`, and records token usage, including cache reads, in `payload.usage`. **Limits** (configuration now, per tenant later): @@ -277,7 +298,7 @@ How `run_query` works: Tools from the other specs join when they land: `get_lineage`, `get_recent_changes`, `get_code_links` and `get_dataset_knowledge` (0002), and `search_knowledge` (0003). -No tool writes anything. If the asker has no credentials for the datasource, `run_query` returns a typed `credentials_missing` error. The agent explains, and the UI links the asker to their credentials page (checks as code §5.2 adds that page). +No tool writes anything. If the asker has no credentials for the datasource, `run_query` returns a typed `credentials_missing` error (`credentials_invalid` for a login the database refuses). The agent explains and links the error's `action_url` as Markdown. Links to app pages open in place in the thread, and the page is §8.5's. ### 7.6 Prompt assembly and caching @@ -388,9 +409,126 @@ The brief schema is `InvestigationBrief` (Pydantic, versioned): - **Confirm or Reject** (`outcome-review`, SCOPE_WRITE) records investigation feedback and posts to the thread. - Confirm pre-fills the resolution note. - Reject asks why and offers Continue investigating. -- **Add as check** works as described in checks as code §7.6. +- **Add as check** works as described in checks as code §7.6. A person's review outranks the model's confidence: + - a confirmed root cause can become a check at any confidence + - a rejected one never can, so the card hides the button + - an unreviewed one needs a confidence of at least 60%. Below that, the button is disabled and the card says why: "The root cause's confidence (35%) is below 60%. Confirm it to add it as a check." + - `core/codify.codify_refusal` holds the rule for the codify endpoint, and `codifyBlocker` mirrors it in the UI. `GET /investigations/{id}` returns `outcome_verdict`, so the run's details page applies it too. - **Downstream:** resolving an issue with a confirmed outcome emits `issue.resolved_with_cause`. 0002 lists these on the dataset, and 0003 can export them. +### 7.11 Starting investigations (D12, D13) + +**One starter.** `InvestigationStarterService.start()` (`services/investigation.py`) is the only code that starts an investigation. It: +1. **Resolves the issue.** + - Given an `issue_id`, it uses that issue, which must belong to the tenant. + - Otherwise it opens one with `open_issue()`. The title is the brief's symptom, the dataset is the first scope table, and the severity comes from the alert. `created_by` is the person, or NULL for API keys and webhooks. +2. **Builds the missing half** of the input: + - Given a brief, it builds the workflow's alert from it, as the spawn route does today. + - Given an alert (SDK, webhooks), it builds the brief: + - the symptom from the alert's display name and values + - the scope from its dataset ids and datasource + - the time window as the anomaly date ± 1 day +3. **Writes the run.** It inserts the `investigations` row (with `created_by`), the `issue_investigation_runs` row and the thread's `investigation` card. + - The run row's `trigger_type` is `human`, `api`, `webhook` or `rule`. + - A run without a person is attributed to dataing ("dataing started an investigation"). +4. **Starts the workflow** with `alert.issue_id` set, so the outcome is always written back. + - If the start fails, the run gets a failed outcome ("Couldn't start the investigation: …") and the route returns 503. +5. **Returns** `investigation_id`, `run_id`, `issue_id` and `issue_number`. + +**Opening issues.** `open_issue()` (`adapters/db/issues.py`) is the one way to insert an issue: `POST /issues`, both webhooks and the starter all use it. +- It records the `created` event, which also posts the thread's first entry: "Maya opened the issue", or "Issue opened by dataing from Monte Carlo". +- Today, webhook-opened issues get no thread entry. + +**Every path, after this change:** + +| Path | Today | After | +|---|---|---| +| `POST /issues/{id}/investigation-runs` | Starter, then the run row and card inline | Starter, with the issue | +| `POST /investigations` (UI, SDK, CLI, notebook) | Inline insert; no issue, run row or card | Starter; opens an issue unless `issue_id` is given | +| CE `webhook-generic` and EE provider webhooks, policy AUTO | Inline insert and a direct Temporal start. Only `alert.issue_id` links it: no run row, no start card, and outcome review returns 404. | Starter, with the issue the webhook opened, `trigger_type = webhook` | +| EE rule action `spawn_investigation` | Starter, but no `alert.issue_id`, so the outcome is never written back | Starter, with the issue, `trigger_type = rule` | +| `POST /investigations/import` | Inserts a finished replay record | Unchanged. It is a record, not a run, so it has no issue; its details page has no back link. | + +Two dormant creation paths had no callers and are removed: the Redis queue worker (`adapters/queue/`) and `InvestigationService` with its branch/collaboration domain (`core/investigation/service.py` and friends, `adapters/db/investigation_repository.py`). The starter is the only code that creates an investigation. + +**`POST /investigations`** (SCOPE_WRITE): + +```json +{ + "brief": {"symptom": "…", "scope": {"tables": ["public.orders"]}}, + "alert": null, + "datasource_id": null, + "execution_profile": "standard", + "issue_id": null +} +``` + +- **Body:** + - Exactly one of `brief` and `alert`. A bad alert is a 422, not today's 500. + - The datasource resolves as it does today; an ambiguous one is a 409 with `ambiguous_datasource`. +- **Response:** `investigation_id`, `run_id`, `issue_id`, `issue_number`, `status: "queued"`, and `main_branch_id` for the SDK. +- **SDK and CLI:** the SDK's `Investigation` gains `issue_id` and `issue_number`. `dataing run start` prints the issue's URL. + +**Run numbers and status:** +- `InvestigationRunResponse` gains `number` (the run's position among the issue's runs, by start time), `status` (`running`, `completed` or `failed`, from the outcome) and `error`. Today a failed run shows "running" forever. +- `InvestigationStateResponse` (`GET /investigations/{id}`) gains `issue_id`, `issue_number`, `issue_title`, `run_number`, `brief`, `execution_profile` and `error`. + +### 7.12 LLM failures and the key check (D15) + +**Today:** a rejected key never fails a run. +1. Every LLM activity catches the exception and returns empty results. +2. `AgentClient.interpret_evidence` turns an LLM error into evidence, so the hypothesis reads as *refuted*. +3. The workflow logs each error as a warning, synthesizes from nothing, and publishes `status: "completed"` at confidence 0. + +The chat agent shows the raw `ModelHTTPError` text. + +**Classifying errors.** `classify_llm_error(exc)` (`agents/errors.py`) follows the exception chain (`LLMError` → pydantic-ai `ModelHTTPError`/`ModelAPIError` → `anthropic.*Error`) and returns a code, whether a retry can help, and a message that says what to fix: + +| Code | Cause | Retry? | Message | +|---|---|---|---| +| `missing_key` | `ANTHROPIC_API_KEY` is empty | no | "ANTHROPIC_API_KEY isn't set. Set it and restart the API and the worker." | +| `invalid_key` | 401 | no | "Anthropic rejected the API key (401). Set a valid ANTHROPIC_API_KEY and restart the API and the worker." | +| `forbidden` | 403 | no | "The API key isn't allowed to use {model} (403)." | +| `unknown_model` | 404 | no | "Anthropic doesn't know the model {model} (404). Check LLM_MODEL and CHAT_AGENT_MODEL." | +| `bad_request` | 400, 413, 422 | no | "Anthropic rejected the request ({status}): {detail}" | +| `rate_limited` | 429 | yes | "Anthropic rate-limited the request (429)." | +| `overloaded` | 529 | yes | "Anthropic is overloaded (529)." | +| `server_error` | other 5xx | yes | "Anthropic returned an error ({status})." | +| `unreachable` | connection error, timeout | yes | "Couldn't reach Anthropic: {detail}" | + +**Activities raise instead of returning an error.** +- On an LLM error, `generate_hypotheses`, `generate_query`, `interpret_evidence`, `synthesize` and `counter_analyze` raise a Temporal `ApplicationError`: + - type `LLMRejected` and `non_retryable=True` when a retry can't help + - type `LLMUnavailable` otherwise + - details `{code, message}` in both cases +- Every LLM activity call gets an explicit retry policy: 4 attempts, 5 s initial backoff, ×2, capped at 60 s. `LLMRejected` is non-retryable. The Anthropic SDK already retries 429/5xx twice inside each attempt. +- `AgentClient` stops swallowing errors in `interpret_evidence`. + +**The workflow fails the run.** New code paths are guarded with `workflow.patched("llm-failures-v1")`. The run fails when: +- hypothesis generation fails, or proposes nothing and no person added a hypothesis; +- any hypothesis evaluation fails with an LLM error. The key or model is broken for every subagent, so the others are cancelled. +- every hypothesis ended untested because of errors. Hypotheses a person ruled out or stopped don't count. +- synthesis fails. + +Counter-analysis failing doesn't fail a run whose synthesis succeeded. The outcome records `counter_analysis.error` and the card says the check didn't run. + +**What a failed run does:** +1. It publishes a failed outcome through `publish_investigation_outcome`: `{"status": "failed", "error": {"code", "message", "step"}}`. This writes `investigations.outcome`, the run row (`completed_at`) and the thread card. +2. It raises a non-retryable `ApplicationError`, so Temporal shows the execution as failed too. + +The card shows a red "failed" pill, the message and **Retry**. + +**Chat turns and brief drafts.** `run_agent_turn` classifies the error the same way. The reply shows the message, not the raw exception, and a turn that can't succeed on retry isn't retried. + +**The key check.** +- At startup the API calls `GET /v1/models/{id}` for each configured model (`LLM_MODEL`, `CHAT_AGENT_MODEL`). This costs no tokens. + - It runs in the background with a 10-second timeout and no SDK retries, so a slow or missing network never delays startup. + - An empty key is reported without a request. +- `GET /system/llm` (ANY_USER) returns `{state, message, models, checked_at}`. + - `state` is `ok`, `checking` or one of the codes above. + - An `unreachable` result older than 60 seconds is checked again on read. +- The key is read from the environment, so a fixed key takes effect when the API and the worker restart; the check reruns at startup. + --- ## 8. Frontend @@ -412,6 +550,79 @@ The brief schema is `InvestigationBrief` (Pydantic, versioned): - **Removed:** "Ask a question" and "Collaborate → Create Branch" on the investigation page, plus `BranchTree` and `MergeIndicator`. - **API client:** regenerate the orval client for the new routes. The committed `openapi.json` has drifted from the app, so either regenerate only these operations or do the full refresh as a separate PR. +### 8.1 Following the mockup + +`0001_issue_chat_mockup.html` is the acceptance reference for the issue page. Where the page and the mockup differ, the mockup wins: + +- **Top bar:** the `#N` pill, the title, the status and priority pills, and the dataset on the right, in one row. +- **No separate description card.** The description is the thread's first entry: "Maya opened the issue", with the description as its body and Edit for its author. An issue opened by dataing starts with the event line instead ("Check … failed · issue opened by dataing"). +- **Thread head:** two tabs, **Shared thread** and **My scratch chats (N)**, and "N watching · live" on the right. The scratch tab opens the scratch drawer. +- **One sidebar panel,** in this order: + 1. Status, with **Change ▾**. The menu starts with the note "Only moves that will succeed are shown", and Resolved says it asks for a note pre-filled from the confirmed cause. + 2. Details: assignee, priority, severity, labels, observed, column. + 3. Dataset, with "open dataset page →". + 4. Investigations: one row per run ("#2 · standard" and a status pill), linking to the run's details page. + 5. Watchers, by name. + 6. Your scratch chats, and **+ New scratch chat**. + + The Timeline section goes; the thread already records when things happened. +- **Tool calls** collapse to one line with the totals: "▸ Ran 1 query · 212 ms · 4 rows" or "▸ Ran 2 queries · 480 ms". Expanded, the footer reads "Snapshot saved with this message · copy SQL". The tool-call record stores `duration_ms` and `row_count` so the line needs no extra request. +- **Pills** use the mockup's colours: purple for the agent and running work, green for supported and done, amber for in progress and untested, red for refuted and failed, blue for people's additions. +- **Investigation card:** "Investigation #N", where N counts the issue's runs in start order. It has a **details →** link to the run's page. A failed run shows a red "failed" pill, the reason and **Retry**, which reopens the brief editor with the same brief. + +### 8.2 Starting an investigation (D13) + +- **Removed:** the `investigations/new` route, `NewInvestigation.tsx`, and the unused `components/Layout.tsx`. +- **Every start button becomes Investigate…** and opens the brief editor in *new* mode where the person is: + - the sidebar's quick action + - the dashboard header and its empty recent-investigations card + - the investigations list header and its empty state + - the dataset page, as **Investigate this dataset** in the header, shown whether or not the dataset has runs +- **New mode** is the same editor, titled "Start an investigation": + - It is pre-filled from the page. The dataset page supplies the table's `native_path` as the scope table and its `datasource_id`; the other pages supply nothing. + - Symptom and at least one scope table are required. + - Findings and ruled out start empty, and there are no leads; there is no thread to draft from. + - **Start investigation** calls `POST /investigations` (§7.11) and navigates to `/issues/{issue_id}`, where the card is already live. +- **Both modes** pick scope with the removed page's components: + - **Table(s)** is a list of `DatasetEntry` rows: a datasource select and a table field that looks tables up in that datasource's schema as the person types. **Add another table** adds a row. + - A run investigates one datasource, so every row shows the same one and changing it in any row changes it for all. It starts as the brief's datasource, else the tenant's default, else its first; a datasource the tenant no longer has is replaced the same way. There is no "The issue's datasource" option: the server never used the issue for this (it falls back to the tenant's only datasource and answers 409 when there are several), so the page shows its pick instead. + - **Time window (optional)** is the `DatePicker`: one day or a range, with quick picks. Days map to whole UTC days, `[first day 00:00Z, day after the last 00:00Z)`. A window that ends mid-day opens as the day it ends on. +- **Hand off** mode (from a thread) keeps its leads, tested first. + +### 8.3 The run's details page (D14) + +`/investigations/:id` keeps its URL and becomes the run's details page: + +- **Header:** + - a back link to the issue's thread ("← #42 Completed orders dropped") + - "Investigation #N" with its status and depth pills + - **Share**, which copies the link (the mocked user picker goes) + - **Export snapshot**, which downloads `GET /investigations/{id}/snapshot` + - **Add as check** (renamed from "Codify Test", same gating) + - Cancel, while the run is running +- **Brief:** the brief the run was given. +- **Hypotheses:** each with its status pill (including "ruled out by a person" and "untested") and its evidence. Queries collapse like the thread's tool calls and expand to the SQL, the result summary and the interpretation. +- **Outcome:** the root cause card in the thread's style, with the causal chain, onset, affected scope and recommendations, plus the existing feedback buttons. +- **A failed run** shows the failure reason and what to fix in place of the outcome. +- **Backend:** `InvestigationStateResponse` gains `issue_id`, `issue_number` and `issue_title`. + +### 8.4 LLM status banner (D15) + +- The app polls `GET /system/llm` (§7.12) once a minute. +- While it reports a problem, a banner under the header on every page names it and says what to fix, for example "Anthropic rejected the API key. Investigations and the agent can't run until ANTHROPIC_API_KEY is fixed and the API and worker are restarted." +- It can't be dismissed while the problem lasts. + +### 8.5 Your login for a datasource (D3) + +`/settings/datasources/{id}/credentials` is where a person saves their own login for a datasource, and where the agent's `credentials_missing` link goes. It was planned in checks as code (fn-69.12). This is its first cut, built because chat can't query anything without it. + +- **Layout:** the sign-in page's: a centered card with an icon, "Connect to prod", and inputs with icons. +- **Fields:** username and password. A source type whose config takes `role` or `warehouse` (Snowflake) also gets those, both optional. +- **Connect** tests the login against the database (`POST …/credentials/test`) and saves it only if it connects. If it doesn't, the database's reason shows in the form and nothing is saved. On success, the page goes back to where the person came from, usually the issue thread, so they can ask again. A page opened directly goes to `/datasources`. +- **A saved login** shows as "You're connected as demo", with the username filled in and the password empty; the API never returns it. **Remove my login** deletes it. +- **A source type with no `username` in its config** has no login to add, so the page says so instead of showing the form (#205). +- **Later (fn-69.12, fn-69.31):** key pairs and service-account JSON, and the 403 toast for checks' "Run now". + --- ## 9. Security and privacy @@ -436,6 +647,7 @@ The brief schema is `InvestigationBrief` (Pydantic, versioned): | M2 | Handoff and results: | About 1.5 weeks | | M3 | Steering: | About 1.5 weeks | | M4 | Scratch chats, publishing, investigate from a scratch chat | About 1 week | +| M5 | Every run in the hub: | About 1 week | `run_query` in M1 needs `UserPrincipal` through `QueryGateway` (checks as code §5.2). If that hasn't landed yet, M1 builds the user path as specified there. Tools from 0002 and 0003 plug in as those specs land. @@ -473,12 +685,33 @@ The brief schema is `InvestigationBrief` (Pydantic, versioned): - the brief editor - the steer controls - **End to end,** on the `null_spike` demo fixture: open an issue, ask the agent, hand off with a brief, rule out a hypothesis, get the outcome, confirm, resolve. +- **M5:** + - `classify_llm_error` for each row of the §7.12 table, through the real exception chain + - each LLM activity raises `LLMRejected` or `LLMUnavailable` and never returns empty results + - workflow tests with fake activities: + - a rejected key fails the run with a failed outcome and a failed Temporal execution + - one subagent's LLM error cancels the others and fails the run + - all hypotheses untested from errors fails the run + - a counter-analysis failure keeps the synthesis + - Replayer tests still pass on the recorded histories + - the starter: + - each path in the §7.11 table writes the run row and the start card, and sets `alert.issue_id` + - `POST /investigations` without `issue_id` opens exactly one issue + - a failed workflow start leaves a failed outcome + - `GET /system/llm` for missing, rejected and valid keys and an unknown model, with the Anthropic client faked + - frontend: + - Investigate… on each page opens the editor pre-filled and navigates to the new issue + - the banner + - the failed card with Retry + - the tool-call totals + - the tabs + - the details page's back link and export --- ## 12. Open questions -1. **Model cost.** Resolved: the owner moved chat and brief drafting from `claude-opus-5` to `claude-opus-5-5`, which costs less per token. Effort stays `low` for chat turns and `medium` for briefs. +1. **Model cost.** Resolved: chat and brief drafting run on `claude-sonnet-5-5`, like investigations (owner, 2026-09-29), after a stop at `claude-opus-5-5`. Effort stays `low` for chat turns and `medium` for briefs; both settings were checked against the API. 2. **Raw tables in the shared thread.** Should result tables be hidden from viewers who have no credentials for that datasource, showing them only the agent's text? 3. **PII redaction.** Should redacting tool results before they reach the model be on by default? 4. **Steering automated runs.** Can people steer investigations that checks started, or only start follow-ups? @@ -501,4 +734,15 @@ The brief schema is `InvestigationBrief` (Pydantic, versioned): - `frontend/app/src/features/issues/IssueWorkspace.tsx`, `IssueCreate.tsx`, `IssueList.tsx` - `frontend/app/src/features/investigation/InvestigationDetail.tsx`, `components/index.ts` (`BranchTree`, `MergeIndicator`) - `python-packages/dataing/tests/unit/entrypoints/api/routes/test_route_authorization.py`: `POLICY` +- M5: + - `python-packages/dataing/src/dataing/services/investigation.py`: `InvestigationStarterService` + - `python-packages/dataing/src/dataing/entrypoints/api/routes/investigations.py`: `start_investigation`, `InvestigationStateResponse` + - `python-packages/dataing/src/dataing/entrypoints/api/routes/integrations.py`: `_start_auto_investigation` + - `python-packages/dataing-ee/src/dataing_ee/entrypoints/api/routes/integrations.py`: `_evaluate_and_start_investigation` + - `python-packages/dataing-ee/src/dataing_ee/core/automation/executor.py`: `spawn_investigation` + - `python-packages/dataing/src/dataing/temporal/activities/`: `generate_hypotheses.py`, `generate_query.py`, `interpret_evidence.py`, `synthesize.py`, `counter_analyze.py`, `publish_outcome.py` + - `python-packages/dataing/src/dataing/agents/client.py`: `LLMError` wrapping; `interpret_evidence` swallows errors today + - `frontend/app/src/features/investigation/NewInvestigation.tsx`, `InvestigationDetail.tsx` + - `frontend/app/src/features/issues/brief/BriefEditor.tsx`, `thread/ToolCalls.tsx`, `thread/InvestigationCard.tsx`, `IssueSidebar.tsx`, `IssueWorkspace.tsx` + - `0001_issue_chat_mockup.html`: the acceptance reference for §8 - [Checks as code](../plans/2026-09-26-checks-as-code-design.md): §5.2 principals, §7 failure loop, §7.6 codify diff --git a/frontend/app/src/App.tsx b/frontend/app/src/App.tsx index aa55546b7..769114972 100644 --- a/frontend/app/src/App.tsx +++ b/frontend/app/src/App.tsx @@ -9,6 +9,7 @@ import { import { AppSidebar } from "@/components/layout/app-sidebar"; import { Separator } from "@/components/ui/separator"; import { ModeToggle } from "@/components/mode-toggle"; +import { LlmStatusBanner } from "@/components/llm-status-banner"; import { ErrorBoundary, FeatureErrorBoundary, @@ -26,10 +27,10 @@ import { import { DashboardPage } from "@/features/dashboard/dashboard-page"; import { InvestigationList } from "@/features/investigation/InvestigationList"; import { InvestigationDetail } from "@/features/investigation/InvestigationDetail"; -import { NewInvestigation } from "@/features/investigation/NewInvestigation"; import { DataSourcePage } from "@/features/datasources/datasource-page"; import { DatasetListPage, DatasetDetailPage } from "@/features/datasets"; import { SettingsPage } from "@/features/settings/settings-page"; +import { DatasourceCredentialsPage } from "@/features/settings/datasource-credentials-page"; import { UsagePage } from "@/features/usage/usage-page"; import { NotificationsPage } from "@/features/notifications"; import { AdminRoute } from "@/features/admin"; @@ -85,6 +86,8 @@ function AppLayout({ children }: { children: React.ReactNode }) { + {/* A broken LLM key or model is visible on every page. */} +
{children}
@@ -145,19 +148,6 @@ function AppWithEntitlements() { } /> - - - - - - } - /> } /> + {/* The issue agent links here when a question needs your login */} + + + + } + /> -
-
- - - dataing - - -
-
-
- -
- - ); -} diff --git a/frontend/app/src/components/layout/app-sidebar.test.tsx b/frontend/app/src/components/layout/app-sidebar.test.tsx index 45d03fbc7..73ded7159 100644 --- a/frontend/app/src/components/layout/app-sidebar.test.tsx +++ b/frontend/app/src/components/layout/app-sidebar.test.tsx @@ -1,8 +1,10 @@ -import { afterEach, describe, expect, it, vi } from "vitest"; +import { afterEach, beforeAll, describe, expect, it, vi } from "vitest"; import { screen } from "@testing-library/react"; +import userEvent from "@testing-library/user-event"; import { SidebarProvider } from "@/components/ui/sidebar"; import type { OrgRole } from "@/lib/auth/types"; +import { stubApi, stubRadixDom } from "@/test/api"; import { renderAsRole } from "@/test/auth"; import { AppSidebar } from "./app-sidebar"; @@ -24,20 +26,36 @@ function renderSidebar(role: OrgRole) { ); } -afterEach(() => localStorage.clear()); +beforeAll(() => stubRadixDom()); + +afterEach(() => { + vi.unstubAllGlobals(); + localStorage.clear(); +}); describe("AppSidebar", () => { - it("does not offer viewers the new investigation shortcut", async () => { + it("does not offer viewers the Investigate… shortcut", async () => { renderSidebar("viewer"); expect(await screen.findByText("Platform")).toBeInTheDocument(); - expect(screen.queryByText("New Investigation")).not.toBeInTheDocument(); + expect(screen.queryByText("Investigate…")).not.toBeInTheDocument(); }); - it("offers members the new investigation shortcut", async () => { + it("opens the brief editor from the Investigate… shortcut", async () => { + stubApi({ "GET /api/v1/datasources": { body: { items: [], total: 0 } } }); renderSidebar("member"); - expect(await screen.findByText("New Investigation")).toBeInTheDocument(); + await userEvent.click( + await screen.findByRole("button", { name: "Investigate…" }), + ); + + expect( + await screen.findByRole("dialog", { name: "Start an investigation" }), + ).toBeInTheDocument(); + // No page links to the removed /investigations/new. + expect( + document.querySelector('a[href="/investigations/new"]'), + ).not.toBeInTheDocument(); }); it.each(["viewer", "member"] as const)( diff --git a/frontend/app/src/components/layout/app-sidebar.tsx b/frontend/app/src/components/layout/app-sidebar.tsx index 6a494e794..eeae00f79 100644 --- a/frontend/app/src/components/layout/app-sidebar.tsx +++ b/frontend/app/src/components/layout/app-sidebar.tsx @@ -1,3 +1,4 @@ +import { useState } from "react"; import { Link, useLocation } from "react-router-dom"; import { Search, @@ -35,6 +36,7 @@ import { } from "@/components/ui/dropdown-menu"; import { Avatar, AvatarFallback } from "@/components/ui/avatar"; import { Badge } from "@/components/ui/Badge"; +import { StartInvestigationDialog } from "@/features/issues/brief/StartInvestigation"; import { useJwtAuth } from "@/lib/auth/jwt-context"; import { useRole } from "@/lib/auth"; import { useNotifications } from "@/lib/notifications"; @@ -80,6 +82,7 @@ export function AppSidebar() { const { logout, org } = useJwtAuth(); const { isAdmin, isMember } = useRole(); const { unreadCount } = useNotifications(); + const [startOpen, setStartOpen] = useState(false); // Build settings nav items based on role // Admin link only visible to admin/owner roles @@ -133,19 +136,24 @@ export function AppSidebar() { - {/* Quick Action */} + {/* Quick action: the brief editor in new mode, over this page */} {isMember && ( - - - - New Investigation - + setStartOpen(true)} + > + + Investigate… + )} diff --git a/frontend/app/src/components/llm-status-banner.test.tsx b/frontend/app/src/components/llm-status-banner.test.tsx new file mode 100644 index 000000000..cd86f3bb4 --- /dev/null +++ b/frontend/app/src/components/llm-status-banner.test.tsx @@ -0,0 +1,119 @@ +import { afterEach, describe, expect, it, vi } from "vitest"; +import { act, screen, waitFor } from "@testing-library/react"; + +import { stubApi, type StubResponse } from "@/test/api"; +import { renderAsRole } from "@/test/auth"; + +import { LlmStatusBanner } from "./llm-status-banner"; + +const LLM = "/api/v1/system/llm"; + +function llm(state: string, message = "") { + return { + body: { + state, + message, + models: ["claude-opus-5-5"], + checked_at: "2026-09-28T08:00:00Z", + }, + }; +} + +const INVALID_KEY = llm( + "invalid_key", + "Anthropic rejected the API key (401). Set a valid ANTHROPIC_API_KEY and restart the API and the worker.", +); + +function renderBanner(response: StubResponse | (() => StubResponse)) { + const api = stubApi({ [`GET ${LLM}`]: response }); + renderAsRole( + <> +

Page

+ + , + "viewer", + ); + return api; +} + +/** React Query hands results over on a zero-delay timer. */ +async function flush() { + await act(() => new Promise((resolve) => setTimeout(resolve, 0))); +} + +afterEach(() => { + vi.useRealTimers(); + vi.unstubAllGlobals(); + localStorage.clear(); +}); + +describe("LlmStatusBanner", () => { + it("says what to fix when the key is rejected", async () => { + renderBanner(INVALID_KEY); + + const banner = await screen.findByRole("alert"); + expect(banner).toHaveTextContent( + "Anthropic rejected the API key (401). Set a valid ANTHROPIC_API_KEY and restart the API and the worker.", + ); + expect(banner).toHaveTextContent( + "Investigations and the agent can't run until this is fixed.", + ); + expect(banner.className).toContain("destructive"); + // It stays until the problem does: there is nothing to dismiss it with. + expect(screen.queryByRole("button")).not.toBeInTheDocument(); + }); + + it.each(["ok", "checking"])( + "stays out of the way while %s", + async (state) => { + const api = renderBanner(llm(state)); + + expect(await screen.findByText("Page")).toBeInTheDocument(); + await waitFor(() => expect(api.find("GET", LLM)).toHaveLength(1)); + await flush(); + expect(screen.queryByRole("alert")).not.toBeInTheDocument(); + }, + ); + + it("warns, rather than alarms, when Anthropic is unreachable", async () => { + renderBanner( + llm("unreachable", "Couldn't reach Anthropic: connection timed out."), + ); + + const banner = await screen.findByRole("alert"); + expect(banner).toHaveTextContent( + "Couldn't reach Anthropic: connection timed out.", + ); + expect(banner.className).toContain("amber"); + expect(banner.className).not.toContain("destructive"); + }); + + it("shows nothing when the status can't be read", async () => { + const api = renderBanner({ status: 500, body: { detail: "boom" } }); + + await waitFor(() => expect(api.find("GET", LLM)).toHaveLength(1)); + await flush(); + expect(screen.queryByRole("alert")).not.toBeInTheDocument(); + }); + + it("checks again once a minute and clears when the key works", async () => { + vi.useFakeTimers({ shouldAdvanceTime: true }); + let response = INVALID_KEY; + const api = renderBanner(() => response); + + expect(await screen.findByRole("alert")).toBeInTheDocument(); + response = llm("ok"); + + await act(async () => { + await vi.advanceTimersByTimeAsync(59_000); + }); + expect(api.find("GET", LLM)).toHaveLength(1); + await act(async () => { + await vi.advanceTimersByTimeAsync(1_500); + }); + expect(api.find("GET", LLM)).toHaveLength(2); + await waitFor(() => + expect(screen.queryByRole("alert")).not.toBeInTheDocument(), + ); + }); +}); diff --git a/frontend/app/src/components/llm-status-banner.tsx b/frontend/app/src/components/llm-status-banner.tsx new file mode 100644 index 000000000..e12782c86 --- /dev/null +++ b/frontend/app/src/components/llm-status-banner.tsx @@ -0,0 +1,49 @@ +/** + * The LLM status banner (spec 0001 §8.4): under the header on every page + * while the API's key check reports a problem, naming it and what to fix. + * It can't be dismissed; it goes away when the problem does. + */ + +import { AlertTriangle } from "lucide-react"; + +import { useLlmStatus } from "@/lib/api/system"; +import { cn } from "@/lib/utils"; + +/** Nothing to report. */ +const QUIET = new Set(["ok", "checking"]); + +/** Problems a retry can outlast; everything else needs someone to fix it. */ +const TRANSIENT = new Set([ + "unreachable", + "rate_limited", + "overloaded", + "server_error", +]); + +export function LlmStatusBanner() { + const { data } = useLlmStatus(); + if (!data || QUIET.has(data.state)) return null; + const transient = TRANSIENT.has(data.state); + + return ( +
+ +

+ + {data.message || "The LLM isn't working."} + {" "} + {transient + ? "Investigations and agent answers may be slow or fail until it recovers." + : "Investigations and the agent can't run until this is fixed."} +

+
+ ); +} diff --git a/frontend/app/src/components/markdown.test.tsx b/frontend/app/src/components/markdown.test.tsx index 0bd12956d..3fbe4f75b 100644 --- a/frontend/app/src/components/markdown.test.tsx +++ b/frontend/app/src/components/markdown.test.tsx @@ -1,5 +1,6 @@ import { describe, expect, it } from "vitest"; import { render, screen } from "@testing-library/react"; +import { MemoryRouter } from "react-router-dom"; import { Markdown } from "./markdown"; @@ -27,4 +28,27 @@ describe("Markdown", () => { const link = screen.getByText("click"); expect(link.getAttribute("href") ?? "").not.toContain("javascript:"); }); + + it("opens app pages in place and other links in a new tab", () => { + render( + + + { + "[Add your login](/settings/datasources/ds-1/credentials) or read [the docs](https://example.com)" + } + + , + ); + + const page = screen.getByRole("link", { name: "Add your login" }); + expect(page).toHaveAttribute( + "href", + "/settings/datasources/ds-1/credentials", + ); + expect(page).not.toHaveAttribute("target"); + expect(screen.getByRole("link", { name: "the docs" })).toHaveAttribute( + "target", + "_blank", + ); + }); }); diff --git a/frontend/app/src/components/markdown.tsx b/frontend/app/src/components/markdown.tsx index 1d4cfc772..5fedc234f 100644 --- a/frontend/app/src/components/markdown.tsx +++ b/frontend/app/src/components/markdown.tsx @@ -7,24 +7,48 @@ * lists, strikethrough and autolinks. */ -import ReactMarkdown, { type Components } from "react-markdown"; +import type { ComponentProps } from "react"; +import ReactMarkdown, { + type Components, + type ExtraProps, +} from "react-markdown"; +import { Link, useInRouterContext } from "react-router-dom"; import remarkGfm from "remark-gfm"; import rehypeSanitize from "rehype-sanitize"; import { cn } from "@/lib/utils"; -const components: Components = { - p: ({ node: _node, ...props }) => ( -

- ), - a: ({ node: _node, ...props }) => ( +const LINK_CLASS = "text-primary underline underline-offset-2"; + +/** + * A link to a page of the app, like the credentials page the agent points to, + * opens in place. Anything else opens in a new tab. + */ +function MarkdownLink({ + node: _node, + href, + ...props +}: ComponentProps<"a"> & ExtraProps) { + const inRouter = useInRouterContext(); + if (inRouter && href?.startsWith("/") && !href.startsWith("//")) { + return ; + } + return ( + ); +} + +const components: Components = { + p: ({ node: _node, ...props }) => ( +

), + a: MarkdownLink, ul: ({ node: _node, ...props }) => (