From 5b2766bbbc277fad420473d3e870951fc0c8d9b5 Mon Sep 17 00:00:00 2001 From: Nicholas Redd Date: Thu, 17 Sep 2026 11:27:39 -0700 Subject: [PATCH 1/2] Generalize token tracking to Claude Code, Codex, and pi `claude-metrics` only ever read one harness off one hardcoded schema, and the live sessions table depended on Claude Code's own PID registry -- a mechanism Codex and pi don't have. Rather than bolt Codex/pi tracking on as a second special case, drop the Claude-only live path entirely and read every harness, Claude included, the same lagging, tailed-transcript way. Changes: - Rename `crates/claude-metrics` to `crates/harness-metrics`. Add a `Harness` enum (Claude/Codex/Pi) and a generic `Ledger` engine (mtime-gated discovery, byte-offset tailing, windowed history, unbounded running cumulative totals) shared by all three backends via a small `HarnessRecord` trait - Add the Codex backend: parses `~/.codex/sessions/**/rollout-*.jsonl` (+ `archived_sessions`), prefers `token_count`'s cumulative `total_token_usage` and diffs it against the previous snapshot per file, folds reasoning tokens into output, and has no cache-write bucket since Codex doesn't expose one - Add the pi backend: parses `~/.pi/agent/sessions/**/*.jsonl`, sums each assistant message's own `usage` object directly (no running-total trick needed, unlike Claude), and folds series by *provider* rather than model -- pi is multi-provider by design, so a per-model series would balloon - Delete the Claude live-session/PID-registry/statusline machinery outright: `session.rs`'s registry reader, `statusline.rs`, and the `claude` sessions table widget. No cost/context/rate-limit columns survive -- that data only ever came from the statusline tee this removes - Replace the `claude`/`claude_graph`/`claude_stats` widgets with two harness-agnostic ones, `agent_graph`/`agent_stats`, each taking a per-instance `source = "claude" | "codex" | "pi" | "all"` on its layout entry. `source = "all"` merges every harness into one series per harness instead of one per model family. A layout can mix instances with different `source`s; time-series keys are namespaced `"{source}::{label}"` so two backends' `Other` buckets can't collide - Rename `[claude]`/`[styles.claude]` config to `[agent]`/`[styles.agent]` - Add `sample_configs/{codex,pi,all_harnesses}_config.toml` alongside a regenerated `claude_config.toml`, each drawing a stats graph and a rate graph and nothing else - Add a `cargo run` Quickstart section to `README.md` covering every fork feature, with `all_harnesses_config.toml` as the flagship "track everything" example. Replace `docs/content/{usage/widgets,configuration/config-file}/ claude.md` with `agent.md`, documenting all three harnesses' counting rules - Regenerate `schema/nightly/bottom.json` This is a breaking rename with no back-compat alias for the old widget names -- low risk given this fork's small, known user base. Verified: `./scripts/smoke`'s four checks (`cargo fmt --all -- --check`, `cargo clippy --workspace --all-targets --all-features -- -D warnings`, `cargo test --workspace --all-features` [`nextest` unavailable in this environment, `cargo test` covers the same 407 tests], `cargo doc --no-deps --all-features --workspace`) are all clean, plus `harness-metrics`'s own pedantic clippy pass. Ran all four sample configs through a real pty and confirmed each renders without error. Co-Authored-By: pi --- Cargo.lock | 19 +- Cargo.toml | 4 +- README.md | 71 +- crates/claude-metrics/examples/claude_dump.rs | 151 ---- crates/claude-metrics/src/history.rs | 575 ------------- crates/claude-metrics/src/lib.rs | 406 --------- crates/claude-metrics/src/session.rs | 337 -------- crates/claude-metrics/src/statusline.rs | 259 ------ crates/claude-metrics/src/transcript.rs | 604 -------------- .../Cargo.toml | 8 +- crates/harness-metrics/src/claude/discover.rs | 153 ++++ .../src/claude/family.rs} | 59 +- crates/harness-metrics/src/claude/mod.rs | 11 + .../harness-metrics/src/claude/transcript.rs | 317 +++++++ crates/harness-metrics/src/codex/discover.rs | 138 ++++ crates/harness-metrics/src/codex/family.rs | 142 ++++ crates/harness-metrics/src/codex/mod.rs | 11 + .../harness-metrics/src/codex/transcript.rs | 398 +++++++++ crates/harness-metrics/src/iso8601.rs | 170 ++++ crates/harness-metrics/src/ledger.rs | 774 ++++++++++++++++++ crates/harness-metrics/src/lib.rs | 554 +++++++++++++ crates/harness-metrics/src/pi/discover.rs | 107 +++ crates/harness-metrics/src/pi/family.rs | 88 ++ crates/harness-metrics/src/pi/mod.rs | 10 + crates/harness-metrics/src/pi/transcript.rs | 293 +++++++ .../src/tailer.rs | 0 .../configuration/config-file/agent.md | 64 ++ .../configuration/config-file/claude.md | 36 - .../configuration/config-file/layout.md | 13 +- .../configuration/config-file/styling.md | 6 +- docs/content/usage/widgets/agent.md | 150 ++++ docs/content/usage/widgets/claude.md | 146 ---- docs/mkdocs.yml | 4 +- examples/history_probe.rs | 19 +- sample_configs/all_harnesses_config.toml | 53 ++ sample_configs/claude_config.toml | 46 +- sample_configs/codex_config.toml | 53 ++ sample_configs/pi_config.toml | 52 ++ schema/nightly/bottom.json | 102 ++- scripts/schema_gen/Cargo.lock | 19 +- scripts/smoke | 2 +- src/app.rs | 14 +- src/app/data/store.rs | 65 +- src/app/data/time_series.rs | 48 +- src/app/layout_manager.rs | 84 +- src/app/states.rs | 52 +- src/canvas.rs | 30 +- .../{claude_graph.rs => agent_graph.rs} | 69 +- .../{claude_stats.rs => agent_stats.rs} | 118 +-- src/canvas/widgets/claude_table.rs | 39 - src/canvas/widgets/mod.rs | 5 +- src/collection.rs | 26 +- src/collection/agent.rs | 243 ++++++ src/collection/claude.rs | 197 ----- src/lib.rs | 6 - src/options.rs | 91 +- src/options/config.rs | 6 +- src/options/config/{claude.rs => agent.rs} | 11 +- src/options/config/layout.rs | 20 +- src/options/config/style.rs | 14 +- src/options/config/style/agent.rs | 19 + src/options/config/style/claude.rs | 14 - src/options/config/style/themes/default.rs | 4 +- src/options/config/style/themes/gruvbox.rs | 4 +- src/options/config/style/themes/nord.rs | 4 +- src/widgets/agent_graph.rs | 40 + src/widgets/agent_stats.rs | 48 ++ src/widgets/claude_graph.rs | 25 - src/widgets/claude_stats.rs | 33 - src/widgets/claude_table.rs | 279 ------- src/widgets/mod.rs | 10 +- tests/integration/valid_config_tests.rs | 40 +- tests/valid_configs/widget/agent.toml | 21 + tests/valid_configs/widget/claude.toml | 19 - 74 files changed, 4459 insertions(+), 3663 deletions(-) delete mode 100644 crates/claude-metrics/examples/claude_dump.rs delete mode 100644 crates/claude-metrics/src/history.rs delete mode 100644 crates/claude-metrics/src/lib.rs delete mode 100644 crates/claude-metrics/src/session.rs delete mode 100644 crates/claude-metrics/src/statusline.rs delete mode 100644 crates/claude-metrics/src/transcript.rs rename crates/{claude-metrics => harness-metrics}/Cargo.toml (71%) create mode 100644 crates/harness-metrics/src/claude/discover.rs rename crates/{claude-metrics/src/model.rs => harness-metrics/src/claude/family.rs} (67%) create mode 100644 crates/harness-metrics/src/claude/mod.rs create mode 100644 crates/harness-metrics/src/claude/transcript.rs create mode 100644 crates/harness-metrics/src/codex/discover.rs create mode 100644 crates/harness-metrics/src/codex/family.rs create mode 100644 crates/harness-metrics/src/codex/mod.rs create mode 100644 crates/harness-metrics/src/codex/transcript.rs create mode 100644 crates/harness-metrics/src/iso8601.rs create mode 100644 crates/harness-metrics/src/ledger.rs create mode 100644 crates/harness-metrics/src/lib.rs create mode 100644 crates/harness-metrics/src/pi/discover.rs create mode 100644 crates/harness-metrics/src/pi/family.rs create mode 100644 crates/harness-metrics/src/pi/mod.rs create mode 100644 crates/harness-metrics/src/pi/transcript.rs rename crates/{claude-metrics => harness-metrics}/src/tailer.rs (100%) create mode 100644 docs/content/configuration/config-file/agent.md delete mode 100644 docs/content/configuration/config-file/claude.md create mode 100644 docs/content/usage/widgets/agent.md delete mode 100644 docs/content/usage/widgets/claude.md create mode 100644 sample_configs/all_harnesses_config.toml create mode 100644 sample_configs/codex_config.toml create mode 100644 sample_configs/pi_config.toml rename src/canvas/widgets/{claude_graph.rs => agent_graph.rs} (82%) rename src/canvas/widgets/{claude_stats.rs => agent_stats.rs} (78%) delete mode 100644 src/canvas/widgets/claude_table.rs create mode 100644 src/collection/agent.rs delete mode 100644 src/collection/claude.rs rename src/options/config/{claude.rs => agent.rs} (73%) create mode 100644 src/options/config/style/agent.rs delete mode 100644 src/options/config/style/claude.rs create mode 100644 src/widgets/agent_graph.rs create mode 100644 src/widgets/agent_stats.rs delete mode 100644 src/widgets/claude_graph.rs delete mode 100644 src/widgets/claude_stats.rs delete mode 100644 src/widgets/claude_table.rs create mode 100644 tests/valid_configs/widget/agent.toml delete mode 100644 tests/valid_configs/widget/claude.toml diff --git a/Cargo.lock b/Cargo.lock index 0312a128..935a93b4 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -277,13 +277,13 @@ dependencies = [ "clap_complete_fig", "clap_complete_nushell", "clap_mangen", - "claude-metrics", "concat-string", "core-foundation", "crossterm", "ctrlc", "dirs", "fern", + "harness-metrics", "humantime", "image", "indexmap", @@ -499,15 +499,6 @@ dependencies = [ "roff", ] -[[package]] -name = "claude-metrics" -version = "0.1.0" -dependencies = [ - "libc", - "serde", - "serde_json", -] - [[package]] name = "color_quant" version = "1.1.0" @@ -1049,6 +1040,14 @@ dependencies = [ "zerocopy", ] +[[package]] +name = "harness-metrics" +version = "0.1.0" +dependencies = [ + "serde", + "serde_json", +] + [[package]] name = "hashbrown" version = "0.16.1" diff --git a/Cargo.toml b/Cargo.toml index d39c9c59..2f316545 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -2,7 +2,7 @@ # and the `mon` binary name are the only additions to upstream's manifest header -- keeping # `[package] name = "bottom"` intact is deliberate so rebases onto upstream stay clean. [workspace] -members = ["crates/claude-metrics"] +members = ["crates/harness-metrics"] # `scripts/schema_gen` is a standalone tool with its own `Cargo.lock`. Adding the # workspace above swept it in and broke `scripts/schema/nightly.sh`; it stays out. exclude = ["scripts/schema_gen"] @@ -75,7 +75,7 @@ logging = ["fern", "log"] generate_schema = ["schemars", "strum"] [dependencies] -claude-metrics = { path = "crates/claude-metrics" } +harness-metrics = { path = "crates/harness-metrics" } # `default-features = false` is mandatory, not tidiness: the default `chafa-dyn` feature # runs a `build.rs` that `.expect()`s a pkg-config probe for libchafa and panics the whole # build when it is absent. diff --git a/README.md b/README.md index da6b1832..0d5e5304 100644 --- a/README.md +++ b/README.md @@ -26,6 +26,30 @@ > additions, described below. Everything past that section is upstream's documentation and > still applies -- the binary is just named `mon` instead of `btm`. See `NOTICE`. +## Quickstart + +One `cargo run` example per addition, run from a source checkout rather than assuming an +installed `mon`. Drop `--release` for a faster compile / slower runtime while iterating. + +```bash +# Apple Silicon power widget -- macOS + Apple Silicon only. +cargo run --release -- --pixel_graphs kitty + +# Agent token metrics for one harness. +cargo run --release -- -C sample_configs/claude_config.toml --pixel_graphs kitty +cargo run --release -- -C sample_configs/codex_config.toml --pixel_graphs kitty +cargo run --release -- -C sample_configs/pi_config.toml --pixel_graphs kitty + +# Track every harness's token usage in one view -- the flagship example. +cargo run --release -- -C sample_configs/all_harnesses_config.toml --pixel_graphs kitty + +# Custom graph marker. +cargo run --release -- --marker sextant + +# Kitty pixel-graph rendering, with any config/layout. +cargo run --release -- --pixel_graphs kitty +``` + ## What this fork adds Nothing here changes the default layout or default behaviour. Every addition is opt-in. @@ -43,37 +67,34 @@ drawing flat lines that read as "idle". Cluster labels come from `SocInfo` rather than being hardcoded: they are `E`/`P` on M1-M4 but `P`/`S` on M5+. -### Claude Code metrics +### Agent token metrics -Three widgets reading live [Claude Code](https://claude.com/claude-code) activity off -`~/.claude`: +Two widgets, `agent_graph` and `agent_stats`, tracking live token usage for a coding-agent +harness off its own local transcript files. Harness-agnostic: which harness (or harnesses) a +given instance reads is set per-instance with `source = "claude" | "codex" | "pi" | "all"`. -- `claude` -- a sortable table of live sessions: name, directory, model family, state, - tokens, cost, context-window occupancy, subagent count -- `claude_graph` -- token throughput by model family over time, on a log axis by default -- `claude_stats` -- the equivalent of Claude Code's own `/status` stats screen, as stacked - rounded-staircase bands of token spend by model family. That screen bars by day; this buckets by minute over - the last hour, so the shape of a working session is visible rather than collapsed into a - single bar. Built by walking `~/.claude/projects` and attributing each record to a bucket - from its own timestamp, so the window is complete the moment the widget appears rather - than having to be accumulated live -- and it keeps the tokens of sessions that have since - exited, which the live-session view cannot +- `agent_graph` -- token throughput by series over time, on a log axis by default +- `agent_stats` -- the equivalent of Claude Code's own `/status` stats screen, as stacked + rounded-staircase bands of token spend by series. That screen bars by day; this buckets by + minute over the last hour, so the shape of a working session is visible rather than + collapsed into a single bar -Backed by the `claude-metrics` workspace crate, which has no dependency on bottom. Counting -is the fiddly part and the rules are documented in that crate: dedupe on -`requestId` + `message.id`, take `cache_creation_input_tokens` without the ephemeral buckets -it already sums, ignore `usage.iterations[]`, and treat `output_tokens` as a running total -across a message's per-content-block records. +A series is a model family for a single-harness `source` (e.g. Claude: Opus/Sonnet/Haiku/ +Fable/Other), or a harness (Claude/Codex/Pi) for `source = "all"`. -Cost, context, and rate limits need a small tee in your statusline -- see -[the docs](https://github.com/nredd/mon/blob/main/docs/content/usage/widgets/claude.md). -Without it those columns read `N/A` and everything else still works. +There is no live-session table. Codex and pi write no PID registry the way Claude Code does, +so rather than give one harness a live view the others can't have, every harness -- Claude +included -- is read the same lagging way: tailing whatever transcripts it has written to +disk, a refresh tick behind the actual model call. -`sample_configs/claude_config.toml` is a ready-made layout with all three and nothing else: +Backed by the `harness-metrics` workspace crate, which has no dependency on bottom and knows +nothing about the other two. Counting is the fiddly part and the rules -- and each harness's +transcript format -- are documented in that crate and in +[the docs](https://github.com/nredd/mon/blob/main/docs/content/usage/widgets/agent.md). -```console -$ mon -C sample_configs/claude_config.toml --pixel_graphs kitty -``` +Four ready-made layouts, each drawing a stats graph and a rate graph and nothing else: +`sample_configs/claude_config.toml`, `codex_config.toml`, `pi_config.toml`, and +`all_harnesses_config.toml` (every harness combined -- see the Quickstart above). ### Configurable graph markers diff --git a/crates/claude-metrics/examples/claude_dump.rs b/crates/claude-metrics/examples/claude_dump.rs deleted file mode 100644 index b6b67cba..00000000 --- a/crates/claude-metrics/examples/claude_dump.rs +++ /dev/null @@ -1,151 +0,0 @@ -//! Dumps what `claude-metrics` reads out of the real `~/.claude` tree. -//! -//! Used to cross-check totals against `~/.claude.json`'s `projects[cwd].lastModelUsage`, -//! which records real per-model counts and cost -- but only for the *last* session in each -//! project, written at shutdown. It is a calibration reference, not a live source. -//! -//! ```console -//! $ cargo run -p claude-metrics --example claude_dump -//! ``` - -fn main() { - // Calibration mode: run the accumulator over one transcript so its totals can be - // compared against `~/.claude.json`'s `lastModelUsage` for that session. - let mut args = std::env::args().skip(1); - if args.next().as_deref() == Some("--transcript") { - // Every remaining argument is folded into one accumulator. A session's totals live - // across its main transcript *and* its `subagents/agent-*.jsonl` files, so pass all - // of them to compare against `lastModelUsage`, which aggregates the lot. - let paths: Vec = args.collect(); - if paths.is_empty() { - eprintln!("--transcript wants one or more paths"); - std::process::exit(2); - } - dump_transcript(&paths); - return; - } - - let Some(mut metrics) = claude_metrics::ClaudeMetrics::with_default_root() else { - eprintln!("No $HOME, so no ~/.claude to read."); - std::process::exit(1); - }; - - println!("root: {}", metrics.root().display()); - metrics.refresh(); - - let sessions = metrics.sessions(); - println!("\n{} live session(s):", sessions.len()); - - for session in sessions { - let id = session.session_id.as_deref().unwrap_or("?"); - println!( - "\n pid {pid:<8} {name:<12} {status:<6} {cwd}\n id {id}\n tmux {tmux} pane {pane}", - pid = session.pid, - name = session.name.as_deref().unwrap_or("-"), - status = session.status.as_deref().unwrap_or("-"), - cwd = session.cwd.as_deref().unwrap_or("-"), - tmux = session.tmux.as_deref().unwrap_or("-"), - pane = session.tmux_pane().unwrap_or("-"), - ); - - let totals = metrics.totals_for(id); - if totals.is_empty() { - println!(" (no usage read yet)"); - continue; - } - - println!( - " {:<8} {:>12} {:>12} {:>14} {:>14} {:>14}", - "model", "input", "output", "cache read", "cache write", "total" - ); - for (family, t) in totals { - println!( - " {:<8} {:>12} {:>12} {:>14} {:>14} {:>14}", - family.label(), - t.input, - t.output, - t.cache_read, - t.cache_creation, - t.total() - ); - } - - if let Some(sl) = metrics.statusline_for(id) { - println!( - " cost ${:.4} ctx {:.1}% 5h {:.1}% 7d {:.1}% [{} {}]", - sl.cost.total_cost_usd, - sl.context_window.used_percentage, - sl.rate_limits.five_hour.used_percentage, - sl.rate_limits.seven_day.used_percentage, - sl.model_display_name().unwrap_or("?"), - sl.effort_level().unwrap_or("?"), - ); - } else { - println!(" (no statusline cache -- is the tee installed?)"); - } - - println!(" subagent messages: {}", metrics.subagent_messages(id)); - if let Some(ms) = metrics.last_turn_duration_ms(id) { - // Integer division: a turn duration in whole seconds is plenty, and it keeps - // the lint about u64 -> f64 precision honest rather than silenced. - println!(" last turn: {}.{}s", ms / 1000, (ms % 1000) / 100); - } - } - - let all = metrics.totals_by_model(); - if !all.is_empty() { - println!("\nacross all live sessions:"); - for (family, t) in all { - println!( - " {:<8} in {:>10} out {:>10} cache_r {:>12} cache_w {:>12}", - family.label(), - t.input, - t.output, - t.cache_read, - t.cache_creation - ); - } - } -} - -/// Accumulate one or more transcript files into a single set of totals. -fn dump_transcript(paths: &[String]) { - use claude_metrics::{Record, UsageAccumulator}; - - let mut acc = UsageAccumulator::default(); - let mut lines = 0u64; - - for path in paths { - let Ok(contents) = std::fs::read_to_string(path) else { - eprintln!("Could not read: '{path}'"); - std::process::exit(1); - }; - - for line in contents.lines() { - lines += 1; - if let Some(record) = Record::parse(line) { - acc.ingest(&record); - } - } - } - - println!("{} file(s), {lines} lines\n", paths.len()); - println!( - "{:<8} {:>10} {:>10} {:>12} {:>12}", - "model", "input", "output", "cache read", "cache write" - ); - for (family, t) in acc.totals() { - println!( - "{:<8} {:>10} {:>10} {:>12} {:>12}", - family.label(), - t.input, - t.output, - t.cache_read, - t.cache_creation - ); - } - println!( - "\nmain messages: {} subagent messages: {}", - acc.main_messages, acc.sidechain_messages - ); -} diff --git a/crates/claude-metrics/src/history.rs b/crates/claude-metrics/src/history.rs deleted file mode 100644 index 3e4a61de..00000000 --- a/crates/claude-metrics/src/history.rs +++ /dev/null @@ -1,575 +0,0 @@ -//! Token usage bucketed by wall-clock time, across every transcript in the tree. -//! -//! # Why this is not built from the live session state -//! -//! [`crate::ClaudeMetrics::refresh`] tails only the sessions currently in the registry, and -//! drops a session's state as soon as it leaves. That is right for "what is running now", -//! and wrong for "what happened in the last hour" -- a session that exited five minutes ago -//! contributed real tokens that would silently vanish from the graph the moment it closed. -//! -//! So this walks `/projects` itself. Files are filtered by modification time before -//! being opened, which is what keeps a full-tree scan cheap: a tree with a year of -//! transcripts in it still only opens the handful touched inside the window. -//! -//! # Why the records carry their own time -//! -//! Every billable record has an ISO-8601 `timestamp`, so a bucket is attributed from the -//! record rather than from when this happened to read it. That means history is correct on -//! the very first refresh -- the past hour is already on disk -- instead of having to be -//! accumulated live over an hour before the graph says anything. - -use std::{ - collections::{BTreeMap, HashMap}, - path::{Path, PathBuf}, - time::{Duration, SystemTime, UNIX_EPOCH}, -}; - -use crate::{ - model::ModelFamily, - tailer::{ReadKind, Tailer}, - transcript::{Record, TokenTotals}, -}; - -/// One bucket's worth of usage, by model family. -#[derive(Clone, Debug, Default, PartialEq, Eq)] -pub struct Bucket { - /// Start of the bucket, in Unix epoch milliseconds. - pub start_ms: u64, - /// Families that contributed, in first-seen order. - pub totals: Vec<(ModelFamily, TokenTotals)>, -} - -impl Bucket { - /// Every token in this bucket, across all families. - #[must_use] - pub fn total(&self) -> u64 { - self.totals - .iter() - .fold(0u64, |sum, (_, totals)| sum.saturating_add(totals.total())) - } - - /// This family's tokens in this bucket, or zero if it did not contribute. - #[must_use] - pub fn total_for(&self, family: ModelFamily) -> u64 { - self.totals - .iter() - .find(|(candidate, _)| *candidate == family) - .map_or(0, |(_, totals)| totals.total()) - } -} - -/// What a message contributed, remembered so replays are not counted twice. -#[derive(Clone, Copy, Debug)] -struct Counted { - family: ModelFamily, - /// Highest `output_tokens` seen for this message so far. - output: u64, - /// Which bucket it landed in, so later blocks of the same message land there too. - bucket: u64, -} - -/// Token usage over a rolling window, bucketed by time. -/// -/// Refreshing is incremental: each transcript keeps a checkpoint, so a refresh only parses -/// what has been appended since the last one. -#[derive(Debug)] -pub struct TokenHistory { - root: PathBuf, - window: Duration, - bucket: Duration, - buckets: BTreeMap>, - tailers: HashMap, - /// Dedupe key mapped to what it contributed. Retries and resumed sessions replay - /// identical messages, and the same message appears in both a session transcript and - /// its subagent files. - seen: HashMap, -} - -impl TokenHistory { - /// A history over `window`, split into buckets of `bucket`. - /// - /// A zero or absurd `bucket` is clamped to something drawable rather than rejected -- - /// this sits in a draw path and a config typo should not take the app down. - #[must_use] - pub fn new(root: impl Into, window: Duration, bucket: Duration) -> Self { - let bucket = bucket.clamp(Duration::from_secs(1), Duration::from_hours(24)); - let window = window.clamp(bucket, Duration::from_hours(90 * 24)); - - Self { - root: root.into(), - window, - bucket, - buckets: BTreeMap::new(), - tailers: HashMap::new(), - seen: HashMap::new(), - } - } - - /// The bucket width, so a caller can label an axis without restating it. - #[must_use] - pub fn bucket(&self) -> Duration { - self.bucket - } - - /// The window covered, oldest bucket to now. - #[must_use] - pub fn window(&self) -> Duration { - self.window - } - - /// Re-read whatever the transcripts have appended and drop anything now out of window. - /// - /// `now_ms` is passed in rather than read from the clock so this is testable without - /// waiting for real time to pass. - pub fn refresh_at(&mut self, now_ms: u64) { - let cutoff = now_ms.saturating_sub(millis(self.window)); - - for path in transcripts_modified_since(&self.root, cutoff) { - let tailer = self - .tailers - .entry(path.clone()) - .or_insert_with(|| Tailer::new(path)); - - let (lines, kind) = tailer.read_new(); - - // A replaced file means the checkpoint described a file that is no longer - // there. The buckets already built from it stay -- they describe real tokens - // that were really spent -- but the dedupe state cannot be trusted across the - // swap, so the replacement is read from the top. - if kind == ReadKind::Restarted { - self.seen.clear(); - } - - for line in lines { - let Some(record) = Record::parse(&line) else { - continue; - }; - - self.ingest(&record, cutoff); - } - } - - self.evict(cutoff); - } - - /// Refresh against the system clock. - pub fn refresh(&mut self) { - self.refresh_at(now_ms()); - } - - /// Fold one record into its bucket. - fn ingest(&mut self, record: &Record, cutoff: u64) { - let Some(usage) = record.billable_usage() else { - return; - }; - - let Some(key) = record.dedupe_key() else { - return; - }; - - // A message already counted contributes only its *new* output tokens, and they go - // to the bucket the message started in. Output is a running total that grows with - // each content block, so the difference is the only new information; splitting the - // later blocks into a neighbouring bucket would smear one message across two. - if let Some(counted) = self.seen.get_mut(&key) { - if usage.output <= counted.output { - return; - } - - let delta = usage.output - counted.output; - let (family, bucket) = (counted.family, counted.bucket); - counted.output = usage.output; - - self.totals_for_mut(bucket, family).output = self - .totals_for_mut(bucket, family) - .output - .saturating_add(delta); - return; - } - - // Without a timestamp there is no bucket to attribute it to. Dropping it is right: - // guessing "now" would pile every undated record onto the newest bucket and draw a - // spike that never happened. - let Some(stamp) = record.timestamp_ms() else { - return; - }; - - if stamp < cutoff { - return; - } - - let bucket = stamp - (stamp % millis(self.bucket)); - let family = ModelFamily::from_id(record.model_id().unwrap_or_default()); - - self.seen.insert( - key, - Counted { - family, - output: usage.output, - bucket, - }, - ); - - let totals = self.totals_for_mut(bucket, family); - totals.input = totals.input.saturating_add(usage.input); - totals.cache_read = totals.cache_read.saturating_add(usage.cache_read); - totals.cache_creation = totals.cache_creation.saturating_add(usage.cache_creation); - totals.output = totals.output.saturating_add(usage.output); - } - - fn totals_for_mut(&mut self, bucket: u64, family: ModelFamily) -> &mut TokenTotals { - let entry = self.buckets.entry(bucket).or_default(); - - if let Some(index) = entry.iter().position(|(candidate, _)| *candidate == family) { - return &mut entry[index].1; - } - - entry.push((family, TokenTotals::default())); - - // The push above guarantees a last element; an `expect` here would be a panic path - // in a refresh loop for a case the compiler simply cannot see. - match entry.last_mut() { - Some((_, totals)) => totals, - None => unreachable!("an element was just pushed"), - } - } - - /// Drop buckets, and the dedupe keys pointing at them, that have aged out. - fn evict(&mut self, cutoff: u64) { - let stale = cutoff - (cutoff % millis(self.bucket)); - self.buckets.retain(|start, _| *start >= stale); - // Otherwise a long-running process grows this map forever. - self.seen.retain(|_, counted| counted.bucket >= stale); - } - - /// Buckets in the window, oldest first, with empty ones filled in. - /// - /// The gaps matter: a graph drawn from only the buckets that saw traffic would join a - /// point at 10:00 straight to one at 10:40 and draw a plateau across half an hour of - /// silence. `now_ms` fixes the right-hand edge. - #[must_use] - pub fn buckets_at(&self, now_ms: u64) -> Vec { - let step = millis(self.bucket); - let newest = now_ms - (now_ms % step); - let count = (millis(self.window) / step).max(1); - let oldest = newest.saturating_sub(step.saturating_mul(count - 1)); - - (0..count) - .map(|index| { - let start_ms = oldest + index * step; - - Bucket { - start_ms, - totals: self.buckets.get(&start_ms).cloned().unwrap_or_default(), - } - }) - .collect() - } - - /// Buckets in the window against the system clock. - #[must_use] - pub fn buckets(&self) -> Vec { - self.buckets_at(now_ms()) - } - - /// Families that contributed anything in the window, in a stable draw order. - /// - /// Ordered by [`ModelFamily::ALL`] rather than by first appearance or by volume, so a - /// family going quiet cannot repaint the ones that remain. - #[must_use] - pub fn families(&self) -> Vec { - ModelFamily::ALL - .into_iter() - .filter(|family| { - self.buckets - .values() - .flatten() - .any(|(candidate, totals)| candidate == family && totals.total() > 0) - }) - .collect() - } -} - -/// Unix epoch milliseconds, now. -fn now_ms() -> u64 { - SystemTime::now() - .duration_since(UNIX_EPOCH) - .map_or(0, millis) -} - -fn millis(duration: Duration) -> u64 { - u64::try_from(duration.as_millis()).unwrap_or(u64::MAX) -} - -/// Every `*.jsonl` under `/projects` touched at or after `cutoff`. -/// -/// The modification-time filter is what makes a full-tree walk affordable. A file last -/// written before the window opened cannot contain a record inside it, so it is never -/// opened at all -- only its directory entry is read. -fn transcripts_modified_since(root: &Path, cutoff: u64) -> Vec { - let projects = root.join("projects"); - let mut found = Vec::new(); - - let Ok(entries) = std::fs::read_dir(&projects) else { - return found; - }; - - for project in entries.flatten() { - let Ok(files) = std::fs::read_dir(project.path()) else { - continue; - }; - - for file in files.flatten() { - let path = file.path(); - - if path.extension().is_none_or(|ext| ext != "jsonl") { - continue; - } - - let modified = file - .metadata() - .and_then(|meta| meta.modified()) - .ok() - .and_then(|time| time.duration_since(UNIX_EPOCH).ok()) - .map(millis); - - // A file with no readable mtime is read rather than skipped. Being wrong in - // the cheap direction costs one parse; being wrong the other way loses data. - if modified.is_none_or(|stamp| stamp >= cutoff) { - found.push(path); - } - } - } - - found -} - -#[cfg(test)] -mod tests { - use super::*; - - /// `2026-08-24T20:00:00.000Z` in epoch millis, a round bucket boundary. - const BASE: u64 = 1_787_601_600_000; - - fn assistant(stamp: &str, model: &str, id: &str, output: u64) -> String { - format!( - r#"{{"type":"assistant","timestamp":"{stamp}","requestId":"req-{id}","message":{{"id":"msg-{id}","model":"{model}","usage":{{"input_tokens":10,"output_tokens":{output},"cache_read_input_tokens":100,"cache_creation_input_tokens":5}}}}}}"# - ) - } - - fn history() -> TokenHistory { - TokenHistory::new( - "/nonexistent", - Duration::from_secs(3600), - Duration::from_secs(60), - ) - } - - fn ingest(history: &mut TokenHistory, line: &str, cutoff: u64) { - let Some(record) = Record::parse(line) else { - panic!("fixture must parse: {line}"); - }; - history.ingest(&record, cutoff); - } - - #[test] - fn a_record_lands_in_the_bucket_its_timestamp_names() { - // Attribution comes from the record, not from when this happened to read it, which - // is what makes the first refresh already correct about the past hour. - let mut history = history(); - - ingest( - &mut history, - &assistant("2026-08-24T20:03:30.000Z", "claude-opus-5", "a", 20), - BASE, - ); - - let buckets = history.buckets_at(BASE + 600_000); - let hit: Vec<&Bucket> = buckets.iter().filter(|b| b.total() > 0).collect(); - - assert_eq!(hit.len(), 1, "exactly one bucket should have traffic"); - assert_eq!( - hit[0].start_ms, - BASE + 180_000, - "20:03:30 belongs to the 20:03 bucket" - ); - assert_eq!(hit[0].total_for(ModelFamily::Opus), 135); - } - - #[test] - fn a_replayed_message_is_not_counted_twice() { - // Retries and resumed sessions replay identical messages, and the same message - // appears in both a session transcript and its subagent files. - let mut history = history(); - let line = assistant("2026-08-24T20:03:30.000Z", "claude-opus-5", "a", 20); - - ingest(&mut history, &line, BASE); - ingest(&mut history, &line, BASE); - - let buckets = history.buckets_at(BASE + 600_000); - assert_eq!( - buckets.iter().map(Bucket::total).sum::(), - 135, - "the replay must contribute nothing" - ); - } - - #[test] - fn later_blocks_of_a_message_stay_in_the_bucket_it_started_in() { - // `output_tokens` is a running total across a message's content-block records, so - // only the delta is new. Letting a late block open its own bucket would smear one - // message across a boundary and draw a step that never happened. - let mut history = history(); - - ingest( - &mut history, - &assistant("2026-08-24T20:03:59.000Z", "claude-opus-5", "a", 20), - BASE, - ); - ingest( - &mut history, - &assistant("2026-08-24T20:04:01.000Z", "claude-opus-5", "a", 50), - BASE, - ); - - let buckets = history.buckets_at(BASE + 600_000); - let hit: Vec<&Bucket> = buckets.iter().filter(|b| b.total() > 0).collect(); - - assert_eq!(hit.len(), 1, "the message must not straddle two buckets"); - assert_eq!( - hit[0].total_for(ModelFamily::Opus), - 165, - "the second record contributes its 30 new output tokens only" - ); - } - - #[test] - fn families_are_kept_apart_within_a_bucket() { - let mut history = history(); - - ingest( - &mut history, - &assistant("2026-08-24T20:03:10.000Z", "claude-opus-5", "a", 20), - BASE, - ); - ingest( - &mut history, - &assistant("2026-08-24T20:03:20.000Z", "claude-sonnet-5", "b", 40), - BASE, - ); - - let buckets = history.buckets_at(BASE + 600_000); - let Some(hit) = buckets.iter().find(|b| b.total() > 0) else { - panic!("one bucket should have traffic"); - }; - - assert_eq!(hit.total_for(ModelFamily::Opus), 135); - assert_eq!(hit.total_for(ModelFamily::Sonnet), 155); - assert_eq!( - history.families(), - vec![ModelFamily::Opus, ModelFamily::Sonnet] - ); - } - - #[test] - fn quiet_buckets_are_filled_in_rather_than_skipped() { - // A graph drawn from only the buckets that saw traffic would join 20:03 straight to - // 20:40 and draw a plateau across half an hour of silence. - let mut history = history(); - - ingest( - &mut history, - &assistant("2026-08-24T20:03:30.000Z", "claude-opus-5", "a", 20), - BASE, - ); - - let buckets = history.buckets_at(BASE + 600_000); - - assert_eq!(buckets.len(), 60, "an hour of one-minute buckets"); - assert!( - buckets.windows(2).all(|w| w[1].start_ms > w[0].start_ms), - "buckets must come back oldest first" - ); - assert_eq!( - buckets.iter().filter(|b| b.total() == 0).count(), - 59, - "every bucket but the busy one is present and empty" - ); - } - - #[test] - fn a_record_older_than_the_window_is_dropped() { - let mut history = history(); - - ingest( - &mut history, - &assistant("2026-08-24T18:00:00.000Z", "claude-opus-5", "a", 20), - BASE, - ); - - assert_eq!( - history - .buckets_at(BASE + 600_000) - .iter() - .map(Bucket::total) - .sum::(), - 0 - ); - } - - #[test] - fn an_undated_record_is_dropped_rather_than_dated_now() { - // Guessing "now" would pile every undated record onto the newest bucket and draw a - // spike that never happened. - let mut history = history(); - let line = r#"{"type":"assistant","requestId":"r","message":{"id":"m","model":"claude-opus-5","usage":{"input_tokens":10,"output_tokens":20}}}"#; - - ingest(&mut history, line, BASE); - - assert_eq!( - history - .buckets_at(BASE + 600_000) - .iter() - .map(Bucket::total) - .sum::(), - 0 - ); - } - - #[test] - fn eviction_drops_the_dedupe_keys_along_with_the_buckets() { - // Otherwise a long-running process grows `seen` forever. - let mut history = history(); - - ingest( - &mut history, - &assistant("2026-08-24T20:03:30.000Z", "claude-opus-5", "a", 20), - BASE, - ); - assert_eq!(history.seen.len(), 1); - - // Two hours later, everything in the window has aged out. - history.evict(BASE + 2 * 3_600_000); - - assert!(history.buckets.is_empty(), "stale buckets must go"); - assert!(history.seen.is_empty(), "and so must their dedupe keys"); - } - - #[test] - fn a_degenerate_bucket_is_clamped_rather_than_dividing_by_zero() { - // This sits in a draw path; a config typo must not take the app down. - let history = TokenHistory::new("/nonexistent", Duration::from_secs(60), Duration::ZERO); - - assert!(history.bucket() >= Duration::from_secs(1)); - assert!(!history.buckets_at(BASE).is_empty()); - } - - #[test] - fn a_missing_tree_yields_an_empty_window_rather_than_failing() { - let mut history = history(); - history.refresh_at(BASE); - - assert!(history.families().is_empty()); - assert_eq!(history.buckets_at(BASE).len(), 60); - } -} diff --git a/crates/claude-metrics/src/lib.rs b/crates/claude-metrics/src/lib.rs deleted file mode 100644 index e21960ca..00000000 --- a/crates/claude-metrics/src/lib.rs +++ /dev/null @@ -1,406 +0,0 @@ -//! Reads live Claude Code metrics off the local `~/.claude` tree. -//! -//! This crate has no dependency on `bottom`. It hands back plain data; rendering lives in -//! `src/canvas/widgets/`. -//! -//! Everything here parses defensively. The `~/.claude` schema is undocumented, it drifts -//! between releases, and a schema surprise must never take a widget down -- every field is -//! optional, unknown fields are ignored, and an unreadable file is treated as absent rather -//! than as an error. -//! -//! # Layout it reads -//! -//! - `~/.claude/sessions/.json` -- the live session registry, pruned with `kill(pid, 0)` -//! - `~/.claude/projects//.jsonl` -- the transcript -//! - `~/.claude/projects///subagents/agent-*.jsonl` -- subagents -//! -//! # Counting rules -//! -//! These are not obvious and getting any of them wrong inflates every number: -//! -//! - Dedupe on `requestId` + `message.id`. Retries and resumed sessions replay identical -//! messages, so counting lines double-counts. -//! - A message with several content blocks carries one `usage` object and counts **once**. -//! - Take `cache_creation_input_tokens` alone, never plus the `cache_creation.ephemeral_*` -//! buckets -- the former is exactly the sum of the latter. -//! - Ignore `usage.iterations[]`; it restates the message-level counts. -//! - Skip `` models and `isApiErrorMessage` records. -//! - `isSidechain: true` marks a subagent, which carries the **parent** session's id. -//! - A message is written as one record **per content block**. The per-request fields -//! (`input_tokens`, `cache_read_input_tokens`, `cache_creation_input_tokens`) repeat -//! identically across those records and are counted once; `output_tokens` is a **running -//! total** and is tracked as a high-water mark. -//! -//! # Known limits -//! -//! Totals were calibrated against `~/.claude.json`'s `projects[cwd].lastModelUsage`, which -//! records real per-model counts for the last session in each project. Two gaps remain, and -//! both are properties of the data source rather than of this crate: -//! -//! - **Background Haiku calls never reach a transcript.** Session titles and similar -//! internal calls are billed but not written to `~/.claude/projects`, so Haiku totals read -//! as zero even when `lastModelUsage` shows a small amount. Nothing here can recover them. -//! - **A few percent of a long session's tokens can be missing.** On one 1199-line session -//! the Opus input, cache-read, and cache-write figures matched `lastModelUsage` exactly -//! while output landed at 97.7%; a short session matched on all four fields exactly. -//! -//! Treat the numbers as a close live estimate, not as billing truth. -//! -//! # Example -//! -//! ```no_run -//! use claude_metrics::ClaudeMetrics; -//! -//! let mut metrics = ClaudeMetrics::with_default_root().expect("no home directory"); -//! metrics.refresh(); -//! -//! for session in metrics.sessions() { -//! println!("{:?} in {:?}", session.name, session.cwd); -//! } -//! ``` - -pub mod history; -pub mod model; -pub mod session; -pub mod statusline; -pub mod tailer; -pub mod transcript; - -use std::{ - collections::HashMap, - path::{Path, PathBuf}, -}; - -pub use history::{Bucket, TokenHistory}; -pub use model::ModelFamily; -pub use session::Session; -pub use statusline::Statusline; -pub use tailer::Tailer; -pub use transcript::{Record, TokenTotals, UsageAccumulator}; - -/// Per-session reading state, carried between refreshes. -#[derive(Debug)] -struct SessionState { - main: Tailer, - subagents: Vec, - usage: UsageAccumulator, - /// Set once the transcript has been located, so a miss is not retried every tick. - located: bool, -} - -/// Reads and accumulates Claude Code metrics from a `~/.claude` tree. -#[derive(Debug)] -pub struct ClaudeMetrics { - root: PathBuf, - sessions: Vec, - states: HashMap, -} - -impl ClaudeMetrics { - /// Read from an explicit `~/.claude` root. Useful for tests and fixtures. - pub fn new(root: impl Into) -> Self { - Self { - root: root.into(), - sessions: Vec::new(), - states: HashMap::new(), - } - } - - /// Read from `$HOME/.claude`. - /// - /// Returns `None` only when there is no home directory to derive a path from. - #[must_use] - pub fn with_default_root() -> Option { - let home = std::env::var_os("HOME")?; - Some(Self::new(Path::new(&home).join(".claude"))) - } - - /// The `~/.claude` root being read. - #[must_use] - pub fn root(&self) -> &Path { - &self.root - } - - /// Re-read the registry and consume whatever the transcripts have appended. - /// - /// Cheap enough to call on every collection tick: the registry is a handful of small - /// files, and transcripts are read from a checkpoint rather than re-parsed. - pub fn refresh(&mut self) { - self.sessions = session::read_registry(&self.root.join("sessions")); - - // Drop state for sessions that have gone away, so a long-running process does not - // accumulate tailers for every session it has ever seen. - let live: Vec = self - .sessions - .iter() - .filter_map(|s| s.session_id.clone()) - .collect(); - self.states.retain(|id, _| live.contains(id)); - - for session in &self.sessions { - let Some(session_id) = session.session_id.clone() else { - continue; - }; - let cwd = session.cwd.clone(); - - let located = self.locate_transcript(&session_id, cwd.as_deref()); - - let state = self - .states - .entry(session_id.clone()) - .or_insert_with(|| SessionState { - main: Tailer::new(PathBuf::new()), - subagents: Vec::new(), - usage: UsageAccumulator::default(), - located: false, - }); - - // A brand-new session's transcript may not exist yet when it first registers. - if !state.located { - let Some(path) = located else { - continue; - }; - - if state.main.path() != path { - state.main = Tailer::new(path); - } - state.located = true; - } - - let (lines, kind) = state.main.read_new(); - - // A replaced transcript means the accumulated totals describe a file that is - // no longer there, so start the count over rather than mixing two files. - if kind == tailer::ReadKind::Restarted { - state.usage = UsageAccumulator::default(); - } - - for line in lines { - if let Some(record) = Record::parse(&line) { - state.usage.ingest(&record); - } - } - - // Subagent files come and go during a session, so rescan rather than caching - // the list. Existing tailers keep their offsets. - let found = session::find_subagent_transcripts(&self.root, &session_id, cwd.as_deref()); - for path in found { - if !state.subagents.iter().any(|t| t.path() == path) { - state.subagents.push(Tailer::new(path)); - } - } - - for tailer in &mut state.subagents { - let (lines, _) = tailer.read_new(); - for line in lines { - if let Some(record) = Record::parse(&line) { - state.usage.ingest(&record); - } - } - } - } - } - - /// Live sessions, newest first. - #[must_use] - pub fn sessions(&self) -> &[Session] { - &self.sessions - } - - /// Token totals for one session, by model family. - #[must_use] - pub fn totals_for(&self, session_id: &str) -> Vec<(ModelFamily, TokenTotals)> { - self.states - .get(session_id) - .map(|state| state.usage.totals()) - .unwrap_or_default() - } - - /// Token totals across every live session, by model family. - #[must_use] - pub fn totals_by_model(&self) -> Vec<(ModelFamily, TokenTotals)> { - let mut merged: HashMap = HashMap::new(); - - for state in self.states.values() { - for (family, totals) in state.usage.totals() { - let entry = merged.entry(family).or_default(); - entry.input = entry.input.saturating_add(totals.input); - entry.output = entry.output.saturating_add(totals.output); - entry.cache_read = entry.cache_read.saturating_add(totals.cache_read); - entry.cache_creation = entry.cache_creation.saturating_add(totals.cache_creation); - } - } - - let mut totals: Vec<(ModelFamily, TokenTotals)> = merged.into_iter().collect(); - totals.sort_unstable_by_key(|(family, _)| *family); - totals - } - - /// Find a session's transcript. - /// - /// Prefers `transcript_path` out of the cached statusline payload, which is exact. - /// Falls back to deriving the path from `cwd` and then to scanning, both of which rest - /// on a slug encoding that is inferred rather than documented. - fn locate_transcript(&self, session_id: &str, cwd: Option<&str>) -> Option { - if let Some(path) = statusline::read(&self.root, Some(session_id), cwd) - .and_then(|s| s.transcript_path) - .map(PathBuf::from) - && path.is_file() - { - return Some(path); - } - - session::find_transcript(&self.root, session_id, cwd) - } - - /// The cached statusline payload for a session, if the tee has written one. - /// - /// This is the only place cost, context-window occupancy, and the 5h/7d rate limits are - /// available. `None` means the tee is not installed or has not run yet, which is a - /// normal state rather than an error. - #[must_use] - pub fn statusline_for(&self, session_id: &str) -> Option { - let cwd = self - .sessions - .iter() - .find(|s| s.session_id.as_deref() == Some(session_id)) - .and_then(|s| s.cwd.as_deref()); - - statusline::read(&self.root, Some(session_id), cwd) - } - - /// How many subagent messages one session has produced. - #[must_use] - pub fn subagent_messages(&self, session_id: &str) -> u64 { - self.states - .get(session_id) - .map_or(0, |state| state.usage.sidechain_messages) - } - - /// The most recent turn duration for one session, in milliseconds. - #[must_use] - pub fn last_turn_duration_ms(&self, session_id: &str) -> Option { - self.states.get(session_id)?.usage.last_turn_duration_ms - } -} - -#[cfg(test)] -mod tests { - // Panicking on a bad fixture is the point in a test -- a fixture that will not - // parse is a broken test, not a runtime condition to handle. - #![allow(clippy::unwrap_used, clippy::expect_used)] - - use std::{fs, io::Write, path::PathBuf}; - - use super::*; - - fn fixture_root(tag: &str) -> PathBuf { - let base = std::env::temp_dir().join(format!("claude-metrics-lib-{tag}")); - let _ = fs::remove_dir_all(&base); - fs::create_dir_all(&base).unwrap(); - base - } - - fn write(path: &Path, contents: &str) { - fs::create_dir_all(path.parent().unwrap()).unwrap(); - fs::File::create(path) - .unwrap() - .write_all(contents.as_bytes()) - .unwrap(); - } - - fn assistant(request: &str, message: &str, model: &str, output: u64) -> String { - format!( - r#"{{"type":"assistant","requestId":"{request}","sessionId":"sess-1","isSidechain":false,"message":{{"id":"{message}","model":"{model}","content":[{{"type":"text"}},{{"type":"tool_use"}}],"usage":{{"input_tokens":1,"output_tokens":{output},"cache_read_input_tokens":2,"cache_creation_input_tokens":3,"cache_creation":{{"ephemeral_1h_input_tokens":3,"ephemeral_5m_input_tokens":0}}}}}}}}"# - ) - } - - #[test] - fn a_live_session_is_read_end_to_end() { - let root = fixture_root("e2e"); - let pid = std::process::id(); - - write( - &root.join(format!("sessions/{pid}.json")), - &format!( - r#"{{"pid":{pid},"sessionId":"sess-1","cwd":"/Users/redd/code","status":"busy","startedAt":1}}"# - ), - ); - - let transcript = root.join("projects/-Users-redd-code/sess-1.jsonl"); - write( - &transcript, - &format!( - "{}\n{}\n", - assistant("req-1", "msg-1", "claude-sonnet-5", 10), - assistant("req-2", "msg-2", "claude-haiku-4-5-20251001", 5), - ), - ); - - let mut metrics = ClaudeMetrics::new(&root); - metrics.refresh(); - - assert_eq!(metrics.sessions().len(), 1); - assert_eq!(metrics.sessions()[0].session_id.as_deref(), Some("sess-1")); - - let totals = metrics.totals_for("sess-1"); - assert_eq!(totals.len(), 2, "two model families"); - - let grand: u64 = totals.iter().map(|(_, t)| t.output).sum(); - assert_eq!(grand, 15); - - fs::remove_dir_all(&root).unwrap(); - } - - #[test] - fn refreshing_twice_does_not_double_count() { - let root = fixture_root("idempotent"); - let pid = std::process::id(); - - write( - &root.join(format!("sessions/{pid}.json")), - &format!( - r#"{{"pid":{pid},"sessionId":"sess-1","cwd":"/Users/redd/code","startedAt":1}}"# - ), - ); - let transcript = root.join("projects/-Users-redd-code/sess-1.jsonl"); - write( - &transcript, - &format!("{}\n", assistant("req-1", "msg-1", "claude-opus-5", 10)), - ); - - let mut metrics = ClaudeMetrics::new(&root); - metrics.refresh(); - metrics.refresh(); - metrics.refresh(); - - let totals = metrics.totals_for("sess-1"); - assert_eq!(totals.len(), 1); - assert_eq!( - totals[0].1.output, 10, - "the checkpointed tailer must not re-read what it already consumed" - ); - - // And an append is picked up. - let mut file = fs::OpenOptions::new() - .append(true) - .open(&transcript) - .unwrap(); - writeln!(file, "{}", assistant("req-2", "msg-2", "claude-opus-5", 7)).unwrap(); - drop(file); - - metrics.refresh(); - assert_eq!(metrics.totals_for("sess-1")[0].1.output, 17); - - fs::remove_dir_all(&root).unwrap(); - } - - #[test] - fn a_missing_claude_tree_yields_nothing_rather_than_failing() { - let mut metrics = ClaudeMetrics::new("/definitely/not/a/real/claude/root"); - metrics.refresh(); - - assert!(metrics.sessions().is_empty()); - assert!(metrics.totals_by_model().is_empty()); - } -} diff --git a/crates/claude-metrics/src/session.rs b/crates/claude-metrics/src/session.rs deleted file mode 100644 index 15ab996f..00000000 --- a/crates/claude-metrics/src/session.rs +++ /dev/null @@ -1,337 +0,0 @@ -//! The live session registry at `~/.claude/sessions/.json`. - -use std::{ - fs, - path::{Path, PathBuf}, -}; - -use serde::Deserialize; - -/// One live Claude Code session, as recorded by its own process. -/// -/// Every field past `pid` is optional: the registry file is written by a running process -/// and can be read mid-write, and its schema is undocumented. -#[derive(Clone, Debug, Default, Deserialize)] -#[serde(default, rename_all = "camelCase")] -pub struct Session { - /// OS process id. Also the registry filename. - pub pid: i32, - /// The session's UUID, which names its transcript file. - pub session_id: Option, - /// Working directory the session was started in. - pub cwd: Option, - /// Unix epoch milliseconds. - pub started_at: Option, - /// Claude Code version string. - pub version: Option, - /// `interactive`, and others. - pub kind: Option, - /// How it was launched, e.g. `cli`. - pub entrypoint: Option, - /// tmux location as `session:@window.%pane`, when running under tmux. - pub tmux: Option, - /// Human-facing session name. - pub name: Option, - /// `busy`, `idle`, and others. - pub status: Option, - /// Unix epoch milliseconds of the last update. - pub updated_at: Option, -} - -impl Session { - /// Whether the owning process is still alive. - /// - /// Sessions are not always cleaned up on exit -- a killed process leaves its registry - /// file behind -- so liveness has to be checked rather than assumed. - #[must_use] - pub fn is_alive(&self) -> bool { - process_is_alive(self.pid) - } - - /// Whether the session reports itself as actively working. - #[must_use] - pub fn is_busy(&self) -> bool { - self.status.as_deref() == Some("busy") - } - - /// The tmux pane id (`%28`) this session runs in, if any. - #[must_use] - pub fn tmux_pane(&self) -> Option<&str> { - self.tmux.as_ref()?.rsplit('.').next() - } -} - -/// `kill(pid, 0)`: succeeds if the process exists and we may signal it, and fails with -/// `EPERM` if it exists but we may not. Both mean alive. -#[cfg(unix)] -fn process_is_alive(pid: i32) -> bool { - if pid <= 0 { - return false; - } - - // SAFETY: `kill` with signal 0 performs the permission and existence checks without - // delivering a signal. It cannot affect the target process. - let result = unsafe { libc::kill(pid, 0) }; - - if result == 0 { - return true; - } - - // EPERM means the process exists but belongs to someone else. - std::io::Error::last_os_error().raw_os_error() == Some(libc::EPERM) -} - -#[cfg(not(unix))] -fn process_is_alive(_pid: i32) -> bool { - // No portable equivalent; assume alive rather than hiding live sessions. - true -} - -/// Read every live session out of a registry directory. -/// -/// Unreadable or unparseable entries are skipped. A registry that does not exist yet is -/// simply empty -- that is the normal state on a machine with no sessions running. -#[must_use] -pub fn read_registry(dir: &Path) -> Vec { - let Ok(entries) = fs::read_dir(dir) else { - return Vec::new(); - }; - - let mut sessions: Vec = entries - .flatten() - .filter_map(|entry| { - let path = entry.path(); - - // The directory also holds `..key` files, which are not sessions. - if path.extension()?.to_str()? != "json" { - return None; - } - - let contents = fs::read_to_string(&path).ok()?; - let session: Session = serde_json::from_str(&contents).ok()?; - - session.is_alive().then_some(session) - }) - .collect(); - - // Newest first, with a stable tiebreak so the table does not shuffle between frames. - sessions.sort_unstable_by(|a, b| { - b.started_at - .cmp(&a.started_at) - .then_with(|| a.pid.cmp(&b.pid)) - }); - - sessions -} - -/// Encode a working directory the way `~/.claude/projects` names its subdirectories. -/// -/// The encoding is inferred from observed directory names, not documented, so callers -/// should treat a miss as ordinary and fall back to searching. -#[must_use] -pub fn project_slug(cwd: &str) -> String { - cwd.chars() - .map(|c| { - if c == '/' || c == '.' || c == '_' { - '-' - } else { - c - } - }) - .collect() -} - -/// Locate a session's transcript under `/projects`. -/// -/// Tries the slug derived from `cwd` first, then falls back to scanning every project -/// directory. The fallback matters: the slug encoding is inferred, and a session whose -/// `cwd` moved still has its transcript under the original slug. -#[must_use] -pub fn find_transcript(root: &Path, session_id: &str, cwd: Option<&str>) -> Option { - let projects = root.join("projects"); - let filename = format!("{session_id}.jsonl"); - - if let Some(cwd) = cwd { - let direct = projects.join(project_slug(cwd)).join(&filename); - if direct.is_file() { - return Some(direct); - } - } - - fs::read_dir(&projects).ok()?.flatten().find_map(|entry| { - let candidate = entry.path().join(&filename); - candidate.is_file().then_some(candidate) - }) -} - -/// Locate a session's subagent transcripts. -#[must_use] -pub fn find_subagent_transcripts(root: &Path, session_id: &str, cwd: Option<&str>) -> Vec { - let projects = root.join("projects"); - - let dirs = cwd - .map(|cwd| vec![projects.join(project_slug(cwd)).join(session_id)]) - .unwrap_or_default() - .into_iter() - .chain( - fs::read_dir(&projects) - .into_iter() - .flatten() - .flatten() - .map(|entry| entry.path().join(session_id)), - ); - - for dir in dirs { - let subagents = dir.join("subagents"); - let Ok(entries) = fs::read_dir(&subagents) else { - continue; - }; - - let mut found: Vec = entries - .flatten() - .map(|entry| entry.path()) - .filter(|path| { - path.extension().and_then(|e| e.to_str()) == Some("jsonl") - && path - .file_name() - .and_then(|n| n.to_str()) - .is_some_and(|n| n.starts_with("agent-")) - }) - .collect(); - - if !found.is_empty() { - found.sort_unstable(); - return found; - } - } - - Vec::new() -} - -#[cfg(test)] -mod tests { - // Panicking on a bad fixture is the point in a test -- a fixture that will not - // parse is a broken test, not a runtime condition to handle. - #![allow(clippy::unwrap_used, clippy::expect_used)] - - use std::io::Write; - - use super::*; - - fn tempdir() -> PathBuf { - let base = std::env::temp_dir().join(format!( - "claude-metrics-test-{}-{:?}", - std::process::id(), - std::thread::current().id() - )); - let _ = fs::remove_dir_all(&base); - fs::create_dir_all(&base).unwrap(); - base - } - - fn write(path: &Path, contents: &str) { - if let Some(parent) = path.parent() { - fs::create_dir_all(parent).unwrap(); - } - let mut file = fs::File::create(path).unwrap(); - file.write_all(contents.as_bytes()).unwrap(); - } - - #[test] - fn the_current_process_reads_as_alive_and_pid_one_does_not_read_as_dead() { - let me = Session { - pid: i32::try_from(std::process::id()).unwrap_or(-1), - ..Default::default() - }; - assert!(me.is_alive()); - - // pid 1 exists but belongs to root: the EPERM branch must still say "alive". - let init = Session { - pid: 1, - ..Default::default() - }; - assert!(init.is_alive(), "EPERM means alive, not dead"); - - let bogus = Session { - pid: -1, - ..Default::default() - }; - assert!(!bogus.is_alive()); - } - - #[test] - fn dead_sessions_and_non_json_entries_are_pruned() { - let dir = tempdir(); - let live = std::process::id(); - - write( - &dir.join(format!("{live}.json")), - &format!(r#"{{"pid":{live},"sessionId":"live","status":"busy","startedAt":2}}"#), - ); - // A pid that cannot be running: far above any real pid_max. - write( - &dir.join("2147483646.json"), - r#"{"pid":2147483646,"sessionId":"dead","startedAt":1}"#, - ); - write(&dir.join("12345.abc.key"), "not json"); - write(&dir.join("garbage.json"), "{{{{"); - - let sessions = read_registry(&dir); - assert_eq!(sessions.len(), 1, "only the live session survives"); - assert_eq!(sessions[0].session_id.as_deref(), Some("live")); - assert!(sessions[0].is_busy()); - - fs::remove_dir_all(&dir).unwrap(); - } - - #[test] - fn transcripts_are_found_by_slug_and_by_fallback_scan() { - let root = tempdir(); - write( - &root.join("projects/-Users-redd-code/abc.jsonl"), - "{\"type\":\"user\"}\n", - ); - write( - &root.join("projects/-some-other-place/moved.jsonl"), - "{\"type\":\"user\"}\n", - ); - - let by_slug = find_transcript(&root, "abc", Some("/Users/redd/code")); - assert!(by_slug.is_some(), "the derived slug must resolve"); - - // The slug encoding is inferred, so a wrong or stale `cwd` has to fall back. - let by_scan = find_transcript(&root, "moved", Some("/Users/redd/code")); - assert!( - by_scan.is_some(), - "a transcript under an unexpected slug must still be found" - ); - - assert!(find_transcript(&root, "nope", None).is_none()); - - fs::remove_dir_all(&root).unwrap(); - } - - #[test] - fn subagent_transcripts_are_found_and_sorted() { - let root = tempdir(); - let base = root.join("projects/-Users-redd-code/sess/subagents"); - write(&base.join("agent-bbb.jsonl"), "{}\n"); - write(&base.join("agent-aaa.jsonl"), "{}\n"); - write(&base.join("agent-aaa.meta.json"), "{}"); - - let found = find_subagent_transcripts(&root, "sess", Some("/Users/redd/code")); - assert_eq!(found.len(), 2, "only the .jsonl files, not the .meta.json"); - assert!(found[0].ends_with("agent-aaa.jsonl"), "must be sorted"); - - fs::remove_dir_all(&root).unwrap(); - } - - #[test] - fn project_slug_matches_the_observed_encoding() { - assert_eq!(project_slug("/Users/redd/code"), "-Users-redd-code"); - assert_eq!( - project_slug("/Users/redd/.local/share/chezmoi"), - "-Users-redd--local-share-chezmoi" - ); - } -} diff --git a/crates/claude-metrics/src/statusline.rs b/crates/claude-metrics/src/statusline.rs deleted file mode 100644 index 0a4b920e..00000000 --- a/crates/claude-metrics/src/statusline.rs +++ /dev/null @@ -1,259 +0,0 @@ -//! Reading the statusline payload cache. -//! -//! Cost, context-window occupancy, and the 5h/7d rate limits are handed to the statusline -//! command on stdin and written nowhere else on disk. A small tee in `~/.claude/statusline.sh` -//! keeps the most recent payload per session at -//! `~/.claude/statusline-cache/.json`; this module reads it back. -//! -//! Verified against a real payload, which carries: `context_window`, `cost`, `cwd`, -//! `effort`, `exceeds_200k_tokens`, `fast_mode`, `model`, `output_style`, `prompt_id`, -//! `rate_limits`, `session_id`, `session_name`, `thinking`, `transcript_path`, `version`, -//! `vim`, and `workspace`. -//! -//! So `session_id` **is** present and is the cache key. The tee still falls back to the -//! working directory slugified the way [`crate::session::project_slug`] does it, and this -//! module still tries that second, in case the key ever goes away. -//! -//! `transcript_path` is the useful surprise: it is the exact path to the session's -//! transcript, which removes the slug-guessing that [`crate::session::find_transcript`] -//! otherwise has to do. -//! -//! Without the tee installed there is simply no cache, and every read here returns `None`. -//! That is a normal state, not an error. - -use std::path::{Path, PathBuf}; - -use serde::Deserialize; - -use crate::session::project_slug; - -/// Spend and edit counters for a session. -#[derive(Clone, Debug, Default, Deserialize)] -#[serde(default)] -pub struct Cost { - /// Total cost so far, in USD. - pub total_cost_usd: f64, - /// Wall-clock duration of the session, in milliseconds. - pub total_duration_ms: u64, - /// Lines added across the session. - pub total_lines_added: u64, - /// Lines removed across the session. - pub total_lines_removed: u64, -} - -/// How full the context window is. -#[derive(Clone, Debug, Default, Deserialize)] -#[serde(default)] -pub struct ContextWindow { - /// Occupancy as a percentage, `0.0..=100.0`. - pub used_percentage: f64, - /// Total window size in tokens. - pub context_window_size: u64, - /// Tokens currently in the window. - pub total_input_tokens: u64, -} - -/// One rate-limit bucket. -#[derive(Clone, Debug, Default, Deserialize)] -#[serde(default)] -pub struct RateLimit { - /// Consumption as a percentage, `0.0..=100.0`. - pub used_percentage: f64, - /// Unix epoch seconds at which the bucket resets. - pub resets_at: u64, -} - -/// The 5-hour and 7-day rate-limit buckets. -#[derive(Clone, Debug, Default, Deserialize)] -#[serde(default)] -pub struct RateLimits { - /// The rolling 5-hour bucket. - pub five_hour: RateLimit, - /// The rolling 7-day bucket. - pub seven_day: RateLimit, -} - -#[derive(Clone, Debug, Default, Deserialize)] -#[serde(default)] -struct ModelInfo { - id: String, - display_name: String, -} - -#[derive(Clone, Debug, Default, Deserialize)] -#[serde(default)] -struct Effort { - level: String, -} - -/// A cached statusline payload. -/// -/// Every field is optional and defaulted: this is an undocumented schema that will drift, -/// and a missing key must never take the widget down. -#[derive(Clone, Debug, Default, Deserialize)] -#[serde(default)] -pub struct Statusline { - /// Spend and edit counters. - pub cost: Cost, - /// Context-window occupancy. - pub context_window: ContextWindow, - /// Rate-limit buckets. The only source for these anywhere on disk. - pub rate_limits: RateLimits, - /// Whether the session has exceeded the 200k-token tier. - pub exceeds_200k_tokens: bool, - /// Whether fast mode is on. - pub fast_mode: bool, - /// The session's id. Present on real payloads; the cache key is derived from it. - pub session_id: Option, - /// The session's human-facing name. - pub session_name: Option, - /// Exact path to the session's transcript. More reliable than deriving it from `cwd`. - pub transcript_path: Option, - /// The Claude Code version that wrote this payload. - pub version: Option, - model: ModelInfo, - effort: Effort, -} - -impl Statusline { - /// The model's display name, e.g. `Opus 5`. - #[must_use] - pub fn model_display_name(&self) -> Option<&str> { - (!self.model.display_name.is_empty()).then_some(self.model.display_name.as_str()) - } - - /// The raw model id, e.g. `claude-opus-5[1m]`. - /// - /// Feed this to [`crate::ModelFamily::from_id`] rather than parsing the display name. - #[must_use] - pub fn model_id(&self) -> Option<&str> { - (!self.model.id.is_empty()).then_some(self.model.id.as_str()) - } - - /// The configured effort level, e.g. `high`. - #[must_use] - pub fn effort_level(&self) -> Option<&str> { - (!self.effort.level.is_empty()).then_some(self.effort.level.as_str()) - } -} - -/// The directory the statusline tee writes to. -#[must_use] -pub fn cache_dir(root: &Path) -> PathBuf { - root.join("statusline-cache") -} - -/// Read the cached payload for a session. -/// -/// Tries the session id first, then the slugified working directory, matching the key the -/// tee writes. Returns `None` when the tee is not installed, has not run yet, or wrote -/// something unparseable. -#[must_use] -pub fn read(root: &Path, session_id: Option<&str>, cwd: Option<&str>) -> Option { - let dir = cache_dir(root); - - let candidates = session_id - .map(|id| dir.join(format!("{id}.json"))) - .into_iter() - .chain(cwd.map(|cwd| dir.join(format!("{}.json", project_slug(cwd))))); - - for path in candidates { - let Ok(contents) = std::fs::read_to_string(&path) else { - continue; - }; - - if let Ok(statusline) = serde_json::from_str::(&contents) { - return Some(statusline); - } - } - - None -} - -#[cfg(test)] -mod tests { - // Panicking on a bad fixture is the point in a test -- a fixture that will not - // parse is a broken test, not a runtime condition to handle. - #![allow(clippy::unwrap_used, clippy::expect_used)] - - use std::fs; - - use super::*; - - /// A payload in the older shape, carrying no session id, to exercise the cwd fallback. - const PAYLOAD: &str = r#"{"model":{"display_name":"Opus 5"},"effort":{"level":"high"}, - "fast_mode":false,"workspace":{"current_dir":"/Users/redd/code"},"cwd":"/Users/redd/code", - "cost":{"total_cost_usd":12.3456,"total_duration_ms":3600000,"total_lines_added":420,"total_lines_removed":69}, - "context_window":{"used_percentage":41.5,"context_window_size":1000000,"total_input_tokens":415000}, - "rate_limits":{"five_hour":{"used_percentage":22.5,"resets_at":9999999999}, - "seven_day":{"used_percentage":61.0,"resets_at":8888888888}}, - "output_style":{"name":"default"},"exceeds_200k_tokens":true,"vim":{"mode":""}}"#; - - fn root(tag: &str) -> PathBuf { - let base = std::env::temp_dir().join(format!("claude-metrics-statusline-{tag}")); - let _ = fs::remove_dir_all(&base); - fs::create_dir_all(base.join("statusline-cache")).unwrap(); - base - } - - #[test] - fn a_payload_keyed_by_cwd_slug_is_found() { - // The fallback path: a payload with no session id, keyed by slugified cwd. - let base = root("cwd"); - fs::write(base.join("statusline-cache/-Users-redd-code.json"), PAYLOAD).unwrap(); - - let found = read(&base, Some("no-such-session"), Some("/Users/redd/code")) - .expect("the cwd-slug fallback must resolve"); - - assert!((found.cost.total_cost_usd - 12.3456).abs() < f64::EPSILON); - assert!((found.rate_limits.five_hour.used_percentage - 22.5).abs() < f64::EPSILON); - assert_eq!(found.rate_limits.seven_day.resets_at, 8_888_888_888); - assert!((found.context_window.used_percentage - 41.5).abs() < f64::EPSILON); - assert!(found.exceeds_200k_tokens); - assert_eq!(found.model_display_name(), Some("Opus 5")); - assert_eq!(found.effort_level(), Some("high")); - - fs::remove_dir_all(&base).unwrap(); - } - - #[test] - fn a_session_keyed_payload_wins_over_the_cwd_fallback() { - let base = root("session"); - fs::write(base.join("statusline-cache/sess-1.json"), PAYLOAD).unwrap(); - fs::write( - base.join("statusline-cache/-Users-redd-code.json"), - r#"{"cost":{"total_cost_usd":999.0}}"#, - ) - .unwrap(); - - let found = read(&base, Some("sess-1"), Some("/Users/redd/code")).unwrap(); - assert!( - (found.cost.total_cost_usd - 12.3456).abs() < f64::EPSILON, - "the session-keyed file must be preferred" - ); - - fs::remove_dir_all(&base).unwrap(); - } - - #[test] - fn a_missing_or_broken_cache_is_not_an_error() { - let base = root("missing"); - - assert!(read(&base, Some("nope"), Some("/nowhere")).is_none()); - - // A half-written or garbage file must be skipped, not panicked on. - fs::write(base.join("statusline-cache/-nowhere.json"), "{not json").unwrap(); - assert!(read(&base, None, Some("/nowhere")).is_none()); - - // And an unknown future key set must still parse, with defaults for what is absent. - fs::write( - base.join("statusline-cache/-nowhere.json"), - r#"{"brand_new_thing":{"a":1}}"#, - ) - .unwrap(); - let found = read(&base, None, Some("/nowhere")).expect("unknown fields must be ignored"); - assert!((found.cost.total_cost_usd - 0.0).abs() < f64::EPSILON); - - fs::remove_dir_all(&base).unwrap(); - } -} diff --git a/crates/claude-metrics/src/transcript.rs b/crates/claude-metrics/src/transcript.rs deleted file mode 100644 index a884e211..00000000 --- a/crates/claude-metrics/src/transcript.rs +++ /dev/null @@ -1,604 +0,0 @@ -//! Parsing transcript records and folding them into token totals. -//! -//! Everything here is deliberately permissive. The `~/.claude` schema is undocumented and -//! drifts between releases, so every field is optional, unknown fields are ignored, and a -//! line that will not parse is skipped rather than failing the whole read. - -use std::collections::HashMap; - -use serde::Deserialize; - -use crate::model::ModelFamily; - -/// Token counts for one model family. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -pub struct TokenTotals { - /// Fresh input tokens. - pub input: u64, - /// Generated tokens. - pub output: u64, - /// Tokens read from the prompt cache. - pub cache_read: u64, - /// Tokens written to the prompt cache. - pub cache_creation: u64, -} - -impl TokenTotals { - /// Every token this family was billed for, cache included. - #[must_use] - pub fn total(&self) -> u64 { - self.input - .saturating_add(self.output) - .saturating_add(self.cache_read) - .saturating_add(self.cache_creation) - } - - /// Add the per-request fields. These repeat identically across a message's records, so - /// they are contributed exactly once, the first time the message is seen. - fn add_request_fields(&mut self, usage: &Usage) { - self.input = self.input.saturating_add(usage.input); - self.cache_read = self.cache_read.saturating_add(usage.cache_read); - // NOTE(redd): the message-level `cache_creation_input_tokens` alone, never plus the - // `cache_creation.ephemeral_*` buckets. Verified against a real record: the former - // is exactly the sum of the latter, so adding both double-counts every cache write. - self.cache_creation = self.cache_creation.saturating_add(usage.cache_creation); - } - - /// Add newly-generated output tokens. - fn add_output(&mut self, delta: u64) { - self.output = self.output.saturating_add(delta); - } -} - -/// The `message.usage` object. -/// -/// `iterations` is deliberately absent: it restates the same counts as the message-level -/// fields, so reading it would double-count. -#[derive(Clone, Debug, Default, Deserialize)] -#[serde(default)] -pub(crate) struct Usage { - #[serde(rename = "input_tokens")] - pub(crate) input: u64, - #[serde(rename = "output_tokens")] - pub(crate) output: u64, - #[serde(rename = "cache_read_input_tokens")] - pub(crate) cache_read: u64, - #[serde(rename = "cache_creation_input_tokens")] - pub(crate) cache_creation: u64, -} - -/// The `message` object on an assistant record. -#[derive(Clone, Debug, Default, Deserialize)] -#[serde(default)] -struct Message { - id: Option, - model: Option, - usage: Option, -} - -/// One transcript line. -/// -/// A single struct covers every record type: irrelevant ones simply leave `message` and -/// `usage` empty and get skipped. That is cheaper and far more drift-tolerant than an -/// enum over `type`, which would have to grow an arm for every new record kind. -#[derive(Clone, Debug, Default, Deserialize)] -#[serde(default, rename_all = "camelCase")] -pub struct Record { - /// Record kind: `assistant`, `user`, `system`, `attachment`, and others. - #[serde(rename = "type")] - pub kind: Option, - /// Present on `system` records, e.g. `turn_duration`. - pub subtype: Option, - /// Wall-clock duration of a turn, on `system` / `turn_duration` records. - pub duration_ms: Option, - /// The API request this record came from. Half of the dedupe key. - pub request_id: Option, - /// The session this record belongs to. - pub session_id: Option, - /// True on subagent records. Subagents carry the *parent* session's id. - pub is_sidechain: bool, - /// Set when the record is an API error rather than a real response. - pub is_api_error_message: bool, - /// ISO-8601 UTC instant the record was written, e.g. `2026-08-24T20:26:13.919Z`. - /// - /// Present on every billable record in practice, but optional like everything else - /// here -- the schema is undocumented and drifts between releases. - pub timestamp: Option, - message: Option, -} - -impl Record { - /// Parse one JSONL line, returning `None` if it is not usable. - #[must_use] - pub fn parse(line: &str) -> Option { - let line = line.trim(); - if line.is_empty() { - return None; - } - - // A malformed or truncated line is skipped, never fatal -- a tailer can legitimately - // read a half-written line at the end of a live file. - serde_json::from_str(line).ok() - } - - /// Whether this record's usage should count toward totals. - fn is_billable(&self) -> bool { - if self.is_api_error_message { - return false; - } - - if self.kind.as_deref() != Some("assistant") { - return false; - } - - // `` is the placeholder model on locally-generated messages that never - // hit the API. - !matches!(self.model_id(), Some("") | None) - } - - pub(crate) fn model_id(&self) -> Option<&str> { - self.message.as_ref()?.model.as_deref() - } - - /// This record's usage, but only if it should count toward totals at all. - /// - /// Bundling the billability check with the lookup keeps the two from drifting apart: - /// every caller that reads usage needs exactly the same filter applied first. - pub(crate) fn billable_usage(&self) -> Option<&Usage> { - if !self.is_billable() { - return None; - } - - self.message.as_ref()?.usage.as_ref() - } - - /// The record's instant, in Unix epoch milliseconds. - /// - /// Hand-parsed rather than pulled in through a date library. The format is fixed and - /// always UTC -- `YYYY-MM-DDTHH:MM:SS.sssZ` -- so this is a few field reads and a - /// days-since-epoch calculation, and it keeps a crate whose whole point is having no - /// dependency on bottom from growing one on `chrono` for a single field. - #[must_use] - pub fn timestamp_ms(&self) -> Option { - parse_iso8601_ms(self.timestamp.as_deref()?) - } - - /// The dedupe key: request id plus message id. - /// - /// Retries and resumed sessions replay identical messages, so counting by line would - /// inflate every total. - pub(crate) fn dedupe_key(&self) -> Option { - let message = self.message.as_ref()?; - let request = self.request_id.as_deref().unwrap_or(""); - let id = message.id.as_deref()?; - Some(format!("{request}\u{0}{id}")) - } -} - -/// Parse `YYYY-MM-DDTHH:MM:SS[.sss]Z` into Unix epoch milliseconds. -/// -/// Only the shape Claude Code actually writes is accepted: fixed-width fields, always UTC. -/// Anything else returns `None` and the caller drops the record rather than guessing a -/// time for it. Offsets other than `Z` are deliberately unsupported -- silently treating -/// `+02:00` as UTC would put records in the wrong bucket, which is worse than losing them. -fn parse_iso8601_ms(stamp: &str) -> Option { - let bytes = stamp.as_bytes(); - - // The shortest accepted form is `YYYY-MM-DDTHH:MM:SSZ`. - if bytes.len() < 20 || bytes[4] != b'-' || bytes[7] != b'-' || bytes[10] != b'T' { - return None; - } - - if bytes[13] != b':' || bytes[16] != b':' || !stamp.ends_with('Z') { - return None; - } - - let field = |from: usize, to: usize| stamp.get(from..to)?.parse::().ok(); - - let (year, month, day) = (field(0, 4)?, field(5, 7)?, field(8, 10)?); - let (hour, minute, second) = (field(11, 13)?, field(14, 16)?, field(17, 19)?); - - // The fractional part is optional and of unspecified length; take up to three digits - // and pad, so both `.9` and `.919` mean what they say. - let millis = match bytes.get(19) { - Some(b'.') => { - let digits: String = stamp[20..] - .chars() - .take_while(char::is_ascii_digit) - .take(3) - .collect(); - - // At most three digits were taken, so this cannot underflow. - let missing = 3 - u32::try_from(digits.len()).ok()?; - digits.parse::().ok()? * 10u64.pow(missing) - } - _ => 0, - }; - - if !(1..=12).contains(&month) || !(1..=31).contains(&day) { - return None; - } - - if hour > 23 || minute > 59 || second > 60 { - return None; - } - - let days = days_from_civil(year, month, day)?; - - Some( - days.saturating_mul(86_400_000) - .saturating_add(hour * 3_600_000) - .saturating_add(minute * 60_000) - .saturating_add(second * 1_000) - .saturating_add(millis), - ) -} - -/// Days from 1970-01-01 to a civil date, by Howard Hinnant's `days_from_civil`. -/// -/// The shift-the-year-to-March trick is what makes leap days fall out of the arithmetic -/// instead of needing a special case: with March as month zero, the leap day lands at the -/// end of the year where it cannot disturb the month-length series. -fn days_from_civil(year: u64, month: u64, day: u64) -> Option { - // Before the epoch there is nothing to attribute, and the unsigned arithmetic below - // would wrap rather than go negative. - if year < 1970 { - return None; - } - - let year = if month <= 2 { year - 1 } else { year }; - let era = year / 400; - let year_of_era = year - era * 400; - - let month_shift = if month > 2 { month - 3 } else { month + 9 }; - let day_of_year = (153 * month_shift + 2) / 5 + day - 1; - let day_of_era = year_of_era * 365 + year_of_era / 4 - year_of_era / 100 + day_of_year; - - // 719468 is the day count from 0000-03-01 to 1970-01-01. - Some(era * 146_097 + day_of_era - 719_468) -} - -/// Accumulates token totals across records, skipping duplicates. -#[derive(Clone, Debug, Default)] -pub struct UsageAccumulator { - /// Dedupe key, mapped to the model family plus the highest `output_tokens` counted - /// for that message so far. - seen: HashMap, - totals: Vec<(ModelFamily, TokenTotals)>, - /// Number of subagent (`isSidechain`) messages counted. - pub sidechain_messages: u64, - /// Number of non-subagent messages counted. - pub main_messages: u64, - /// Most recent turn duration seen, in milliseconds. - pub last_turn_duration_ms: Option, -} - -impl UsageAccumulator { - /// Fold one record in. Returns true if it contributed new usage. - pub fn ingest(&mut self, record: &Record) -> bool { - if record.kind.as_deref() == Some("system") - && record.subtype.as_deref() == Some("turn_duration") - && let Some(duration) = record.duration_ms - { - self.last_turn_duration_ms = Some(duration); - } - - if !record.is_billable() { - return false; - } - - let Some(key) = record.dedupe_key() else { - return false; - }; - - let Some(message) = record.message.as_ref() else { - return false; - }; - let Some(usage) = message.usage.as_ref() else { - return false; - }; - - let family = ModelFamily::from_id(message.model.as_deref().unwrap_or_default()); - - // A message is written as one record per content block. Verified against real - // transcripts: `input_tokens`, `cache_read_input_tokens`, and - // `cache_creation_input_tokens` are per-request and repeat identically across those - // records, but `output_tokens` is a **running total** that grows with each block. - // So the request-level fields are taken once and output is tracked as a high-water - // mark. - // - // Taking the first record's output instead undercounts badly -- on one real session - // that turned 18365 Opus output tokens into 1090. - if let Some((seen_family, counted_output)) = self.seen.get_mut(&key) { - if usage.output <= *counted_output { - return false; - } - - let delta = usage.output - *counted_output; - *counted_output = usage.output; - let seen_family = *seen_family; - - self.totals_for_mut(seen_family).add_output(delta); - return true; - } - - self.seen.insert(key, (family, usage.output)); - - let totals = self.totals_for_mut(family); - totals.add_request_fields(usage); - totals.add_output(usage.output); - - if record.is_sidechain { - self.sidechain_messages = self.sidechain_messages.saturating_add(1); - } else { - self.main_messages = self.main_messages.saturating_add(1); - } - - true - } - - /// Mutable totals for one family, inserting an empty entry if it is new. - fn totals_for_mut(&mut self, family: ModelFamily) -> &mut TokenTotals { - if let Some(index) = self.totals.iter().position(|(f, _)| *f == family) { - return &mut self.totals[index].1; - } - - self.totals.push((family, TokenTotals::default())); - let last = self.totals.len() - 1; - &mut self.totals[last].1 - } - - /// Per-family totals, ordered by family. - #[must_use] - pub fn totals(&self) -> Vec<(ModelFamily, TokenTotals)> { - let mut totals = self.totals.clone(); - totals.sort_unstable_by_key(|(family, _)| *family); - totals - } - - /// Totals summed across every family. - #[must_use] - pub fn grand_total(&self) -> TokenTotals { - self.totals - .iter() - .fold(TokenTotals::default(), |mut acc, (_, t)| { - acc.input = acc.input.saturating_add(t.input); - acc.output = acc.output.saturating_add(t.output); - acc.cache_read = acc.cache_read.saturating_add(t.cache_read); - acc.cache_creation = acc.cache_creation.saturating_add(t.cache_creation); - acc - }) - } -} - -#[cfg(test)] -mod tests { - #![allow(clippy::unwrap_used, clippy::expect_used)] - - use super::*; - - #[test] - fn a_real_timestamp_parses_to_epoch_millis() { - // Taken verbatim from a live transcript. - let record = Record::parse( - r#"{"type":"assistant","timestamp":"2026-08-24T20:26:13.919Z","message":{"id":"m","model":"claude-opus-5"}}"#, - ) - .expect("fixture parses"); - - assert_eq!(record.timestamp_ms(), Some(1_787_603_173_919)); - } - - #[test] - fn the_epoch_itself_is_zero() { - assert_eq!(parse_iso8601_ms("1970-01-01T00:00:00.000Z"), Some(0)); - } - - #[test] - fn leap_days_are_counted() { - // The whole reason for the shift-the-year-to-March trick. A day either side of the - // 2024 leap day must be exactly 86400000ms apart across it. - let before = parse_iso8601_ms("2024-02-28T00:00:00Z").expect("valid"); - let leap = parse_iso8601_ms("2024-02-29T00:00:00Z").expect("valid"); - let after = parse_iso8601_ms("2024-03-01T00:00:00Z").expect("valid"); - - assert_eq!(leap - before, 86_400_000); - assert_eq!(after - leap, 86_400_000); - } - - #[test] - fn a_century_that_is_not_a_leap_year_is_handled() { - // 1900 is not a leap year but 2000 is, which is the case a naive `% 4` gets wrong. - let before = parse_iso8601_ms("2000-02-28T00:00:00Z").expect("valid"); - let after = parse_iso8601_ms("2000-03-01T00:00:00Z").expect("valid"); - - assert_eq!(after - before, 2 * 86_400_000, "2000 had a leap day"); - } - - #[test] - fn a_short_fraction_is_padded_rather_than_misread() { - // `.9` is nine hundred milliseconds, not nine. - assert_eq!( - parse_iso8601_ms("2026-08-24T20:26:13.9Z"), - parse_iso8601_ms("2026-08-24T20:26:13.900Z") - ); - } - - #[test] - fn a_missing_fraction_is_fine() { - assert_eq!( - parse_iso8601_ms("2026-08-24T20:26:13Z"), - parse_iso8601_ms("2026-08-24T20:26:13.000Z") - ); - } - - #[test] - fn a_non_utc_offset_is_refused_rather_than_silently_misread() { - // Treating `+02:00` as UTC would put records two hours into the wrong bucket, which - // is worse than losing them. - assert_eq!(parse_iso8601_ms("2026-08-24T20:26:13.919+02:00"), None); - } - - #[test] - fn junk_timestamps_yield_nothing_rather_than_panicking() { - for stamp in [ - "", - "not a date", - "2026-08-24", - "2026-13-01T00:00:00Z", - "2026-08-24T25:00:00Z", - "1969-12-31T23:59:59Z", - ] { - assert_eq!(parse_iso8601_ms(stamp), None, "{stamp:?} must not parse"); - } - } - // Panicking on a bad fixture is the point in a test -- a fixture that will not - // parse is a broken test, not a runtime condition to handle. - - /// Build one record of a streaming assistant message. - /// - /// This mirrors what real transcripts contain: a message is written as one record per - /// content block, all sharing `requestId` + `message.id`. The per-request fields repeat - /// identically; `output_tokens` is a **running total** that grows with each block. - fn block( - request: &str, message: &str, model: &str, block: &str, running_output: u64, - ) -> String { - format!( - r#"{{"type":"assistant","requestId":"{request}","sessionId":"s1","isSidechain":false,"message":{{"id":"{message}","model":"{model}","content":[{{"type":"{block}"}}],"usage":{{"input_tokens":10,"output_tokens":{running_output},"cache_read_input_tokens":30,"cache_creation_input_tokens":40,"cache_creation":{{"ephemeral_1h_input_tokens":40,"ephemeral_5m_input_tokens":0}},"iterations":[{{"input_tokens":10,"output_tokens":{running_output},"cache_read_input_tokens":30,"cache_creation_input_tokens":40}}]}}}}}}"# - ) - } - - /// The three records of one streamed message: thinking, then text, then a tool call. - fn multi_block() -> Vec { - vec![ - block("req_1", "msg_1", "claude-sonnet-5", "thinking", 3), - block("req_1", "msg_1", "claude-sonnet-5", "text", 12), - block("req_1", "msg_1", "claude-sonnet-5", "tool_use", 20), - ] - } - - #[test] - fn a_multi_content_block_message_counts_once() { - // The load-bearing assertion. One message, three content blocks, three records. - let mut acc = UsageAccumulator::default(); - for line in multi_block() { - acc.ingest(&Record::parse(&line).expect("record must parse")); - } - - let totals = acc.grand_total(); - assert_eq!( - totals.input, 10, - "per-request fields repeat across blocks and must be counted once" - ); - assert_eq!(totals.cache_read, 30); - assert_eq!( - totals.cache_creation, 40, - "cache_creation must come from the message-level field alone -- adding the \ - ephemeral_* buckets on top double-counts, they sum to the same number" - ); - assert_eq!(acc.main_messages, 1, "three records, but one message"); - } - - #[test] - fn output_tokens_are_a_running_total_not_a_per_block_amount() { - // Verified against real transcripts: `output_tokens` grows with each content block - // and the last record carries the final figure. Counting the first record alone - // undercounts badly -- on one real session that turned 18365 tokens into 1090. - // Summing every record instead over-counts, here 3 + 12 + 20 = 35. - let mut acc = UsageAccumulator::default(); - for line in multi_block() { - acc.ingest(&Record::parse(&line).unwrap()); - } - - assert_eq!(acc.grand_total().output, 20); - } - - #[test] - fn a_replayed_message_is_not_counted_twice() { - // Retries and resumed sessions replay identical lines. - let lines = multi_block(); - let mut acc = UsageAccumulator::default(); - - for line in &lines { - acc.ingest(&Record::parse(line).unwrap()); - } - for line in &lines { - assert!( - !acc.ingest(&Record::parse(line).unwrap()), - "a replay adds no new output, so nothing must be contributed" - ); - } - - let totals = acc.grand_total(); - assert_eq!(totals.output, 20); - assert_eq!(totals.input, 10); - assert_eq!(acc.main_messages, 1); - } - - #[test] - fn iterations_are_ignored_rather_than_summed() { - // `iterations[]` restates the message-level counts. If it were read, output doubles. - let mut acc = UsageAccumulator::default(); - for line in multi_block() { - acc.ingest(&Record::parse(&line).unwrap()); - } - assert_eq!(acc.grand_total().output, 20); - } - - #[test] - fn synthetic_and_api_error_records_are_filtered() { - let synthetic = r#"{"type":"assistant","requestId":"r","message":{"id":"m1","model":"","usage":{"output_tokens":99}}}"#; - let api_error = r#"{"type":"assistant","isApiErrorMessage":true,"requestId":"r","message":{"id":"m2","model":"claude-opus-5","usage":{"output_tokens":99}}}"#; - let user = r#"{"type":"user","message":{"id":"m3","usage":{"output_tokens":99}}}"#; - - let mut acc = UsageAccumulator::default(); - for line in [synthetic, api_error, user] { - acc.ingest(&Record::parse(line).unwrap()); - } - - assert_eq!(acc.grand_total().total(), 0); - } - - #[test] - fn sidechain_records_are_counted_separately() { - let sub = r#"{"type":"assistant","requestId":"r2","sessionId":"s1","isSidechain":true,"message":{"id":"m9","model":"claude-haiku-4-5-20251001","usage":{"output_tokens":5}}}"#; - - let mut acc = UsageAccumulator::default(); - for line in multi_block() { - acc.ingest(&Record::parse(&line).unwrap()); - } - acc.ingest(&Record::parse(sub).unwrap()); - - assert_eq!(acc.main_messages, 1); - assert_eq!(acc.sidechain_messages, 1); - - let totals = acc.totals(); - assert_eq!(totals.len(), 2, "each family gets its own bucket"); - } - - #[test] - fn turn_duration_is_picked_up_from_system_records() { - let line = - r#"{"type":"system","subtype":"turn_duration","durationMs":5783500,"sessionId":"s1"}"#; - let mut acc = UsageAccumulator::default(); - acc.ingest(&Record::parse(line).unwrap()); - assert_eq!(acc.last_turn_duration_ms, Some(5_783_500)); - } - - #[test] - fn unknown_fields_and_junk_lines_do_not_fail() { - // The whole point: `~/.claude` drifts, and a widget must never die on a surprise. - let future = r#"{"type":"assistant","requestId":"r","brandNewField":{"nested":true},"message":{"id":"m","model":"claude-opus-6","usage":{"output_tokens":7,"someNewCounter":123}}}"#; - let record = Record::parse(future).expect("unknown fields must be ignored"); - - let mut acc = UsageAccumulator::default(); - assert!(acc.ingest(&record)); - assert_eq!(acc.grand_total().output, 7); - - assert!(Record::parse("{not json").is_none()); - assert!(Record::parse("").is_none()); - assert!( - Record::parse(r#"{"type":"assistant","message":{"id":"x"#).is_none(), - "a half-written trailing line must be skipped, not fatal" - ); - } -} diff --git a/crates/claude-metrics/Cargo.toml b/crates/harness-metrics/Cargo.toml similarity index 71% rename from crates/claude-metrics/Cargo.toml rename to crates/harness-metrics/Cargo.toml index 0c6bdf48..b0e6ff67 100644 --- a/crates/claude-metrics/Cargo.toml +++ b/crates/harness-metrics/Cargo.toml @@ -1,20 +1,16 @@ [package] -name = "claude-metrics" +name = "harness-metrics" version = "0.1.0" edition = "2024" rust-version = "1.95.0" license = "MIT" -description = "Reads live Claude Code session, token, agent, and cost metrics off the local `~/.claude` tree." +description = "Reads live token metrics off the local transcript trees of Claude Code, Codex CLI, and pi." publish = false [dependencies] serde = { version = "1.0", features = ["derive"] } serde_json = "1.0" -# `kill(pid, 0)` for pruning the session registry. No portable std equivalent. -[target.'cfg(unix)'.dependencies] -libc = "0.2" - [dev-dependencies] [lints.rust] diff --git a/crates/harness-metrics/src/claude/discover.rs b/crates/harness-metrics/src/claude/discover.rs new file mode 100644 index 00000000..ebfcc1c5 --- /dev/null +++ b/crates/harness-metrics/src/claude/discover.rs @@ -0,0 +1,153 @@ +//! Finding Claude Code transcripts under `~/.claude/projects`. +//! +//! # Layout +//! +//! - `/projects//.jsonl` -- a session's main transcript. +//! - `/projects///subagents/agent-*.jsonl` -- its subagents. +//! +//! The `` encoding is inferred from observed directory names, not documented, so +//! discovery never tries to derive it -- it walks every project directory instead. +//! +//! The modification-time filter is what makes a full-tree walk affordable. A file last +//! written before the window opened cannot contain a record inside it, so it is never +//! opened at all -- only its directory entry is read. + +use std::{ + path::{Path, PathBuf}, + time::UNIX_EPOCH, +}; + +/// Every transcript -- main or subagent -- touched at or after `cutoff` (Unix epoch +/// milliseconds). +#[must_use] +pub(crate) fn discover(root: &Path, cutoff: u64) -> Vec { + let projects = root.join("projects"); + let mut found = Vec::new(); + + let Ok(entries) = std::fs::read_dir(&projects) else { + return found; + }; + + for project in entries.flatten() { + let Ok(children) = std::fs::read_dir(project.path()) else { + continue; + }; + + for child in children.flatten() { + let path = child.path(); + + if path.is_dir() { + // A session-id directory holding subagent transcripts. + collect_modified_since(&path.join("subagents"), "jsonl", cutoff, &mut found); + continue; + } + + if is_fresh_jsonl(&child, cutoff) { + found.push(path); + } + } + } + + found +} + +/// Every `*.` directly inside `dir` touched at or after `cutoff`. +fn collect_modified_since(dir: &Path, ext: &str, cutoff: u64, found: &mut Vec) { + let Ok(entries) = std::fs::read_dir(dir) else { + return; + }; + + for entry in entries.flatten() { + if entry.path().extension().is_some_and(|e| e == ext) && is_fresh(&entry, cutoff) { + found.push(entry.path()); + } + } +} + +fn is_fresh_jsonl(entry: &std::fs::DirEntry, cutoff: u64) -> bool { + entry.path().extension().is_some_and(|e| e == "jsonl") && is_fresh(entry, cutoff) +} + +/// A file with no readable mtime is treated as fresh rather than skipped. Being wrong in +/// the cheap direction costs one parse; being wrong the other way loses data. +fn is_fresh(entry: &std::fs::DirEntry, cutoff: u64) -> bool { + let modified = entry + .metadata() + .and_then(|meta| meta.modified()) + .ok() + .and_then(|time| time.duration_since(UNIX_EPOCH).ok()) + .map(|d| u64::try_from(d.as_millis()).unwrap_or(u64::MAX)); + + modified.is_none_or(|stamp| stamp >= cutoff) +} + +#[cfg(test)] +mod tests { + // Panicking on a bad fixture is the point in a test -- a fixture that will not + // parse is a broken test, not a runtime condition to handle. + #![allow(clippy::unwrap_used, clippy::expect_used)] + + use std::fs; + + use super::*; + + fn tempdir(tag: &str) -> PathBuf { + let base = std::env::temp_dir().join(format!("harness-metrics-claude-discover-{tag}")); + let _ = fs::remove_dir_all(&base); + fs::create_dir_all(&base).unwrap(); + base + } + + fn write(path: &Path, contents: &str) { + fs::create_dir_all(path.parent().unwrap()).unwrap(); + fs::write(path, contents).unwrap(); + } + + #[test] + fn main_transcripts_and_subagent_transcripts_are_both_found() { + let root = tempdir("both"); + write(&root.join("projects/-Users-redd-code/sess-1.jsonl"), "{}\n"); + write( + &root.join("projects/-Users-redd-code/sess-1/subagents/agent-a.jsonl"), + "{}\n", + ); + write( + &root.join("projects/-Users-redd-code/sess-1/subagents/agent-a.meta.json"), + "{}", + ); + + let found = discover(&root, 0); + + assert_eq!( + found.len(), + 2, + "the main transcript and the one subagent transcript" + ); + assert!(found.iter().any(|p| p.ends_with("sess-1.jsonl"))); + assert!(found.iter().any(|p| p.ends_with("agent-a.jsonl"))); + assert!( + !found.iter().any(|p| p.ends_with("agent-a.meta.json")), + "only .jsonl files" + ); + + fs::remove_dir_all(&root).unwrap(); + } + + #[test] + fn files_older_than_the_cutoff_are_skipped() { + let root = tempdir("cutoff"); + write(&root.join("projects/-Users-redd-code/old.jsonl"), "{}\n"); + + // A cutoff far in the future: the fixture file was just written, so it is older. + let far_future = u64::MAX - 1; + assert!(discover(&root, far_future).is_empty()); + assert_eq!(discover(&root, 0).len(), 1); + + fs::remove_dir_all(&root).unwrap(); + } + + #[test] + fn a_missing_tree_yields_nothing() { + assert!(discover(Path::new("/definitely/not/a/real/claude/root"), 0).is_empty()); + } +} diff --git a/crates/claude-metrics/src/model.rs b/crates/harness-metrics/src/claude/family.rs similarity index 67% rename from crates/claude-metrics/src/model.rs rename to crates/harness-metrics/src/claude/family.rs index ba22601b..3f1fb448 100644 --- a/crates/claude-metrics/src/model.rs +++ b/crates/harness-metrics/src/claude/family.rs @@ -1,10 +1,10 @@ -//! Folding raw model IDs into families. +//! Folding raw Claude model IDs into families. use std::fmt; -/// A model family, folded from a raw model ID. +/// A Claude model family, folded from a raw model ID. #[derive(Copy, Clone, Debug, PartialEq, Eq, Hash, PartialOrd, Ord)] -pub enum ModelFamily { +pub enum ClaudeFamily { /// Claude Opus. Opus, /// Claude Sonnet. @@ -17,10 +17,10 @@ pub enum ModelFamily { Other, } -impl ModelFamily { +impl ClaudeFamily { /// Fold a raw model ID such as `claude-opus-5[1m]` or `claude-haiku-4-5-20251001`. /// - /// This is a **prefix** match on purpose. Model IDs are not stable in shape: real + /// This is a **substring** match on purpose. Model IDs are not stable in shape: real /// transcripts on this machine carry both `claude-sonnet-5` (undated) and /// `claude-haiku-4-5-20251001` (dated), and new IDs appear without warning. An /// exact-match table silently drops every future ID into `Other`, which is exactly the @@ -34,15 +34,15 @@ impl ModelFamily { // Match on the family segment anywhere in the ID rather than anchoring at the // start, so vendor-prefixed IDs work without a separate table. if id.contains("opus") { - ModelFamily::Opus + ClaudeFamily::Opus } else if id.contains("sonnet") { - ModelFamily::Sonnet + ClaudeFamily::Sonnet } else if id.contains("haiku") { - ModelFamily::Haiku + ClaudeFamily::Haiku } else if id.contains("fable") { - ModelFamily::Fable + ClaudeFamily::Fable } else { - ModelFamily::Other + ClaudeFamily::Other } } @@ -63,16 +63,16 @@ impl ModelFamily { #[must_use] pub fn label(self) -> &'static str { match self { - ModelFamily::Opus => "Opus", - ModelFamily::Sonnet => "Sonnet", - ModelFamily::Haiku => "Haiku", - ModelFamily::Fable => "Fable", - ModelFamily::Other => "Other", + ClaudeFamily::Opus => "Opus", + ClaudeFamily::Sonnet => "Sonnet", + ClaudeFamily::Haiku => "Haiku", + ClaudeFamily::Fable => "Fable", + ClaudeFamily::Other => "Other", } } } -impl fmt::Display for ModelFamily { +impl fmt::Display for ClaudeFamily { fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { f.write_str(self.label()) } @@ -84,40 +84,43 @@ mod tests { // parse is a broken test, not a runtime condition to handle. #![allow(clippy::unwrap_used, clippy::expect_used)] - use super::ModelFamily; + use super::ClaudeFamily; #[test] fn undated_and_dated_ids_fold_to_the_same_family() { // Both of these are real IDs seen in transcripts on this machine. An exact-match // table would put one of them in `Other`. - assert_eq!(ModelFamily::from_id("claude-sonnet-5"), ModelFamily::Sonnet); assert_eq!( - ModelFamily::from_id("claude-haiku-4-5-20251001"), - ModelFamily::Haiku + ClaudeFamily::from_id("claude-sonnet-5"), + ClaudeFamily::Sonnet + ); + assert_eq!( + ClaudeFamily::from_id("claude-haiku-4-5-20251001"), + ClaudeFamily::Haiku ); } #[test] fn suffixed_and_vendor_prefixed_ids_still_fold() { assert_eq!( - ModelFamily::from_id("claude-opus-5[1m]"), - ModelFamily::Opus, + ClaudeFamily::from_id("claude-opus-5[1m]"), + ClaudeFamily::Opus, "a context-window suffix must not change the family" ); assert_eq!( - ModelFamily::from_id("us.anthropic.claude-opus-5-v1:0"), - ModelFamily::Opus, + ClaudeFamily::from_id("us.anthropic.claude-opus-5-v1:0"), + ClaudeFamily::Opus, "a Bedrock-style vendor prefix must not change the family" ); - assert_eq!(ModelFamily::from_id("claude-fable-5"), ModelFamily::Fable); + assert_eq!(ClaudeFamily::from_id("claude-fable-5"), ClaudeFamily::Fable); } #[test] fn unknown_ids_fold_to_other_rather_than_failing() { assert_eq!( - ModelFamily::from_id("some-future-model"), - ModelFamily::Other + ClaudeFamily::from_id("some-future-model"), + ClaudeFamily::Other ); - assert_eq!(ModelFamily::from_id(""), ModelFamily::Other); + assert_eq!(ClaudeFamily::from_id(""), ClaudeFamily::Other); } } diff --git a/crates/harness-metrics/src/claude/mod.rs b/crates/harness-metrics/src/claude/mod.rs new file mode 100644 index 00000000..71d0ad5f --- /dev/null +++ b/crates/harness-metrics/src/claude/mod.rs @@ -0,0 +1,11 @@ +//! The Claude Code backend: reads `~/.claude/projects/**/*.jsonl` transcripts, main +//! sessions and their subagents alike. + +mod discover; +mod family; +mod transcript; + +pub use family::ClaudeFamily; +pub(crate) use transcript::Record; + +pub(crate) use discover::discover; diff --git a/crates/harness-metrics/src/claude/transcript.rs b/crates/harness-metrics/src/claude/transcript.rs new file mode 100644 index 00000000..de39013c --- /dev/null +++ b/crates/harness-metrics/src/claude/transcript.rs @@ -0,0 +1,317 @@ +//! Parsing Claude Code transcript records. +//! +//! Everything here is deliberately permissive. The `~/.claude` schema is undocumented and +//! drifts between releases, so every field is optional, unknown fields are ignored, and a +//! line that will not parse is skipped rather than failing the whole read. +//! +//! # Counting rules +//! +//! These are not obvious and getting any of them wrong inflates every number: +//! +//! - Dedupe on `requestId` + `message.id`. Retries and resumed sessions replay identical +//! messages, so counting lines double-counts. The same message can also appear in both a +//! session's main transcript and its subagent transcript, so the dedupe key is treated as +//! global by [`crate::ledger::Ledger`] rather than scoped to one file. +//! - A message with several content blocks carries one `usage` object and counts **once**. +//! - Take `cache_creation_input_tokens` alone, never plus the `cache_creation.ephemeral_*` +//! buckets -- the former is exactly the sum of the latter. +//! - Ignore `usage.iterations[]`; it restates the message-level counts. +//! - Skip `` models and `isApiErrorMessage` records. +//! - `isSidechain: true` marks a subagent, which carries the **parent** session's id. +//! - A message is written as one record **per content block**. The per-request fields +//! (`input_tokens`, `cache_read_input_tokens`, `cache_creation_input_tokens`) repeat +//! identically across those records and are counted once; `output_tokens` is a **running +//! total**. [`billable_usage`](Record::billable_usage) hands back the running snapshot +//! as-is on every record, and the ledger's generic keyed-delta logic (first occurrence of +//! a key contributes the full snapshot, later occurrences contribute the field-by-field +//! growth) turns that into the right per-block increment without this module having to +//! track a high-water mark itself. +//! +//! # Known limits +//! +//! Totals were calibrated against `~/.claude.json`'s `projects[cwd].lastModelUsage`, which +//! records real per-model counts for the last session in each project. Two gaps remain, and +//! both are properties of the data source rather than of this crate: +//! +//! - **Background Haiku calls never reach a transcript.** Session titles and similar +//! internal calls are billed but not written to `~/.claude/projects`, so Haiku totals read +//! as zero even when `lastModelUsage` shows a small amount. Nothing here can recover them. +//! - **A few percent of a long session's tokens can be missing.** On one 1199-line session +//! the Opus input, cache-read, and cache-write figures matched `lastModelUsage` exactly +//! while output landed at 97.7%; a short session matched on all four fields exactly. +//! +//! Treat the numbers as a close live estimate, not as billing truth. + +use serde::Deserialize; + +use crate::{ + iso8601, + ledger::{HarnessRecord, TokenTotals}, +}; + +use super::family::ClaudeFamily; + +/// The `message.usage` object. +/// +/// `iterations` is deliberately absent: it restates the same counts as the message-level +/// fields, so reading it would double-count. +#[derive(Clone, Debug, Default, Deserialize)] +#[serde(default)] +struct Usage { + #[serde(rename = "input_tokens")] + input: u64, + #[serde(rename = "output_tokens")] + output: u64, + #[serde(rename = "cache_read_input_tokens")] + cache_read: u64, + #[serde(rename = "cache_creation_input_tokens")] + cache_creation: u64, +} + +impl From<&Usage> for TokenTotals { + fn from(usage: &Usage) -> Self { + TokenTotals { + input: usage.input, + output: usage.output, + cache_read: usage.cache_read, + cache_creation: usage.cache_creation, + } + } +} + +/// The `message` object on an assistant record. +#[derive(Clone, Debug, Default, Deserialize)] +#[serde(default)] +struct Message { + id: Option, + model: Option, + usage: Option, +} + +/// One transcript line. +/// +/// A single struct covers every record type: irrelevant ones simply leave `message` and +/// `usage` empty and get skipped. That is cheaper and far more drift-tolerant than an enum +/// over `type`, which would have to grow an arm for every new record kind. +#[derive(Clone, Debug, Default, Deserialize)] +#[serde(default, rename_all = "camelCase")] +pub struct Record { + /// Record kind: `assistant`, `user`, `system`, `attachment`, and others. + #[serde(rename = "type")] + kind: Option, + /// The API request this record came from. Half of the dedupe key. + request_id: Option, + /// Set when the record is an API error rather than a real response. + is_api_error_message: bool, + /// ISO-8601 UTC instant the record was written, e.g. `2026-08-24T20:26:13.919Z`. + /// + /// Present on every billable record in practice, but optional like everything else + /// here -- the schema is undocumented and drifts between releases. + timestamp: Option, + message: Option, +} + +impl Record { + /// Whether this record's usage should count toward totals. + fn is_billable(&self) -> bool { + if self.is_api_error_message { + return false; + } + + if self.kind.as_deref() != Some("assistant") { + return false; + } + + // `` is the placeholder model on locally-generated messages that never + // hit the API. + !matches!(self.model_id(), Some("") | None) + } + + fn model_id(&self) -> Option<&str> { + self.message.as_ref()?.model.as_deref() + } +} + +impl HarnessRecord for Record { + type State = (); + + fn parse(line: &str, (): &mut Self::State) -> Option { + let line = line.trim(); + if line.is_empty() { + return None; + } + + // A malformed or truncated line is skipped, never fatal -- a tailer can legitimately + // read a half-written line at the end of a live file. + serde_json::from_str(line).ok() + } + + /// The record's instant, in Unix epoch milliseconds. + fn timestamp_ms(&self) -> Option { + iso8601::parse_ms(self.timestamp.as_deref()?) + } + + /// The dedupe key: request id plus message id. + fn dedupe_key(&self) -> Option { + let message = self.message.as_ref()?; + let request = self.request_id.as_deref().unwrap_or(""); + let id = message.id.as_deref()?; + Some(format!("{request}\u{0}{id}")) + } + + fn family_label(&self) -> &'static str { + ClaudeFamily::from_id(self.model_id().unwrap_or_default()).label() + } + + fn billable_usage(&self) -> Option { + if !self.is_billable() { + return None; + } + + Some(TokenTotals::from(self.message.as_ref()?.usage.as_ref()?)) + } +} + +#[cfg(test)] +mod tests { + #![allow(clippy::unwrap_used, clippy::expect_used)] + + use super::*; + + fn parse(line: &str) -> Record { + Record::parse(line, &mut ()).expect("fixture parses") + } + + #[test] + fn a_real_timestamp_parses_to_epoch_millis() { + // Taken verbatim from a live transcript. + let record = parse( + r#"{"type":"assistant","timestamp":"2026-08-24T20:26:13.919Z","message":{"id":"m","model":"claude-opus-5"}}"#, + ); + + assert_eq!(record.timestamp_ms(), Some(1_787_603_173_919)); + } + + /// Build one record of a streaming assistant message. + /// + /// The ISO-8601 parser itself is tested in `crate::iso8601`; only this backend's use of + /// it (feeding `Record::timestamp_ms`) is covered here. + /// + /// This mirrors what real transcripts contain: a message is written as one record per + /// content block, all sharing `requestId` + `message.id`. The per-request fields repeat + /// identically; `output_tokens` is a **running total** that grows with each block. + fn block( + stamp: &str, request: &str, message: &str, model: &str, block: &str, running_output: u64, + ) -> String { + format!( + r#"{{"type":"assistant","timestamp":"{stamp}","requestId":"{request}","sessionId":"s1","isSidechain":false,"message":{{"id":"{message}","model":"{model}","content":[{{"type":"{block}"}}],"usage":{{"input_tokens":10,"output_tokens":{running_output},"cache_read_input_tokens":30,"cache_creation_input_tokens":40,"cache_creation":{{"ephemeral_1h_input_tokens":40,"ephemeral_5m_input_tokens":0}},"iterations":[{{"input_tokens":10,"output_tokens":{running_output},"cache_read_input_tokens":30,"cache_creation_input_tokens":40}}]}}}}}}"# + ) + } + + /// The three records of one streamed message: thinking, then text, then a tool call. + fn multi_block() -> Vec { + vec![ + block( + "2026-08-24T20:03:30.000Z", + "req_1", + "msg_1", + "claude-sonnet-5", + "thinking", + 3, + ), + block( + "2026-08-24T20:03:30.500Z", + "req_1", + "msg_1", + "claude-sonnet-5", + "text", + 12, + ), + block( + "2026-08-24T20:03:31.000Z", + "req_1", + "msg_1", + "claude-sonnet-5", + "tool_use", + 20, + ), + ] + } + + #[test] + fn a_multi_content_block_message_reports_the_same_snapshot_every_block() { + // The per-request fields repeat identically across blocks; the ledger's keyed-delta + // logic is what turns that into "counted once" -- this module just hands back the + // running snapshot as-is on every record. + for line in multi_block() { + let record = parse(&line); + let usage = record.billable_usage().expect("billable"); + assert_eq!(usage.input, 10); + assert_eq!(usage.cache_read, 30); + assert_eq!( + usage.cache_creation, 40, + "cache_creation must come from the message-level field alone -- adding the \ + ephemeral_* buckets on top double-counts, they sum to the same number" + ); + } + } + + #[test] + fn output_tokens_are_a_running_total_not_a_per_block_amount() { + // Verified against real transcripts: `output_tokens` grows with each content block + // and the last record carries the final figure. + let outputs: Vec = multi_block() + .iter() + .map(|line| parse(line).billable_usage().unwrap().output) + .collect(); + assert_eq!(outputs, vec![3, 12, 20]); + } + + #[test] + fn every_block_shares_one_dedupe_key() { + let keys: Vec> = multi_block() + .iter() + .map(|line| parse(line).dedupe_key()) + .collect(); + assert_eq!(keys[0], keys[1]); + assert_eq!(keys[1], keys[2]); + assert!(keys[0].is_some()); + } + + #[test] + fn synthetic_and_api_error_records_are_filtered() { + let synthetic = r#"{"type":"assistant","requestId":"r","message":{"id":"m1","model":"","usage":{"output_tokens":99}}}"#; + let api_error = r#"{"type":"assistant","isApiErrorMessage":true,"requestId":"r","message":{"id":"m2","model":"claude-opus-5","usage":{"output_tokens":99}}}"#; + let user = r#"{"type":"user","message":{"id":"m3","usage":{"output_tokens":99}}}"#; + + for line in [synthetic, api_error, user] { + assert!(parse(line).billable_usage().is_none(), "{line}"); + } + } + + #[test] + fn a_sidechain_record_is_still_billable_and_folds_to_its_own_family() { + // `isSidechain` and `sessionId` are unknown fields to this struct now -- subagent + // records are counted the same as main-transcript ones, they are just discovered + // from a different file (see `claude::discover`). + let line = r#"{"type":"assistant","requestId":"r2","sessionId":"parent-1","isSidechain":true,"message":{"id":"m9","model":"claude-haiku-4-5-20251001","usage":{"output_tokens":5}}}"#; + let record = parse(line); + assert!(record.billable_usage().is_some()); + assert_eq!(record.family_label(), "Haiku"); + } + + #[test] + fn unknown_fields_and_junk_lines_do_not_fail() { + // The whole point: `~/.claude` drifts, and a widget must never die on a surprise. + let future = r#"{"type":"assistant","requestId":"r","brandNewField":{"nested":true},"message":{"id":"m","model":"claude-opus-6","usage":{"output_tokens":7,"someNewCounter":123}}}"#; + let record = Record::parse(future, &mut ()).expect("unknown fields must be ignored"); + assert_eq!(record.billable_usage().unwrap().output, 7); + + assert!(Record::parse("{not json", &mut ()).is_none()); + assert!(Record::parse("", &mut ()).is_none()); + assert!( + Record::parse(r#"{"type":"assistant","message":{"id":"x"#, &mut ()).is_none(), + "a half-written trailing line must be skipped, not fatal" + ); + } +} diff --git a/crates/harness-metrics/src/codex/discover.rs b/crates/harness-metrics/src/codex/discover.rs new file mode 100644 index 00000000..2ec3131d --- /dev/null +++ b/crates/harness-metrics/src/codex/discover.rs @@ -0,0 +1,138 @@ +//! Finding Codex CLI rollout files under `/sessions` and `/archived_sessions`. +//! +//! # Layout +//! +//! - `/sessions///
/rollout--.jsonl` +//! - `/archived_sessions/rollout--.jsonl` +//! +//! The modification-time filter is what makes a full-tree walk affordable. A file last +//! written before the window opened cannot contain a record inside it, so it is never +//! opened at all -- only its directory entry is read. + +use std::{ + path::{Path, PathBuf}, + time::UNIX_EPOCH, +}; + +/// Every `rollout-*.jsonl` file touched at or after `cutoff` (Unix epoch milliseconds), +/// under both the dated `sessions` tree and the flat `archived_sessions` directory. +#[must_use] +pub(crate) fn discover(root: &Path, cutoff: u64) -> Vec { + let mut found = Vec::new(); + walk(&root.join("sessions"), cutoff, &mut found); + walk(&root.join("archived_sessions"), cutoff, &mut found); + found +} + +/// Recurse into `dir`, collecting every fresh `rollout-*.jsonl` file at any depth. The +/// `sessions` tree is three levels deep (`YYYY/MM/DD`) and `archived_sessions` is flat; +/// walking generically covers both without hardcoding either shape. +fn walk(dir: &Path, cutoff: u64, found: &mut Vec) { + let Ok(entries) = std::fs::read_dir(dir) else { + return; + }; + + for entry in entries.flatten() { + let path = entry.path(); + + if path.is_dir() { + walk(&path, cutoff, found); + continue; + } + + let is_rollout = path + .file_stem() + .and_then(|n| n.to_str()) + .is_some_and(|stem| stem.starts_with("rollout-")) + && path + .extension() + .is_some_and(|ext| ext.eq_ignore_ascii_case("jsonl")); + + if !is_rollout { + continue; + } + + let modified = entry + .metadata() + .and_then(|meta| meta.modified()) + .ok() + .and_then(|time| time.duration_since(UNIX_EPOCH).ok()) + .map(|d| u64::try_from(d.as_millis()).unwrap_or(u64::MAX)); + + // A file with no readable mtime is read rather than skipped. Being wrong in the + // cheap direction costs one parse; being wrong the other way loses data. + if modified.is_none_or(|stamp| stamp >= cutoff) { + found.push(path); + } + } +} + +#[cfg(test)] +mod tests { + // Panicking on a bad fixture is the point in a test -- a fixture that will not + // parse is a broken test, not a runtime condition to handle. + #![allow(clippy::unwrap_used, clippy::expect_used)] + + use std::fs; + + use super::*; + + fn tempdir(tag: &str) -> PathBuf { + let base = std::env::temp_dir().join(format!("harness-metrics-codex-discover-{tag}")); + let _ = fs::remove_dir_all(&base); + fs::create_dir_all(&base).unwrap(); + base + } + + fn write(path: &Path, contents: &str) { + fs::create_dir_all(path.parent().unwrap()).unwrap(); + fs::write(path, contents).unwrap(); + } + + #[test] + fn rollouts_are_found_in_the_dated_tree_and_the_archive() { + let root = tempdir("both"); + write( + &root.join("sessions/2026/01/15/rollout-2026-01-15T00-00-00-abc.jsonl"), + "{}\n", + ); + write( + &root.join("archived_sessions/rollout-old-def.jsonl"), + "{}\n", + ); + write( + &root.join("sessions/2026/01/15/not-a-rollout.jsonl"), + "{}\n", + ); + + let found = discover(&root, 0); + + assert_eq!(found.len(), 2); + assert!( + found + .iter() + .any(|p| p.ends_with("rollout-2026-01-15T00-00-00-abc.jsonl")) + ); + assert!(found.iter().any(|p| p.ends_with("rollout-old-def.jsonl"))); + assert!(!found.iter().any(|p| p.ends_with("not-a-rollout.jsonl"))); + + fs::remove_dir_all(&root).unwrap(); + } + + #[test] + fn files_older_than_the_cutoff_are_skipped() { + let root = tempdir("cutoff"); + write(&root.join("sessions/2026/01/15/rollout-a.jsonl"), "{}\n"); + + let far_future = u64::MAX - 1; + assert!(discover(&root, far_future).is_empty()); + assert_eq!(discover(&root, 0).len(), 1); + + fs::remove_dir_all(&root).unwrap(); + } + + #[test] + fn a_missing_tree_yields_nothing() { + assert!(discover(Path::new("/definitely/not/a/real/codex/root"), 0).is_empty()); + } +} diff --git a/crates/harness-metrics/src/codex/family.rs b/crates/harness-metrics/src/codex/family.rs new file mode 100644 index 00000000..74e3a3bb --- /dev/null +++ b/crates/harness-metrics/src/codex/family.rs @@ -0,0 +1,142 @@ +//! Folding raw Codex CLI model IDs into families. + +use std::fmt; + +/// A Codex model family, folded from a raw model ID. +#[derive(Copy, Clone, Debug, PartialEq, Eq, Hash, PartialOrd, Ord)] +pub enum CodexFamily { + /// The GPT-5 line. + Gpt5, + /// The GPT-4 line. + Gpt4, + /// The reasoning `o1`/`o3`/`o4` line. + OSeries, + /// `codex-mini` and similar Codex-branded models. + CodexMini, + /// Anything that did not match a known family. + Other, +} + +impl CodexFamily { + /// Fold a raw model ID such as `gpt-5.1-codex` or `o3-mini`. + /// + /// A substring match, same reasoning as Claude's `ClaudeFamily::from_id`: model IDs + /// drift, and an exact-match table would silently drop every future ID into `Other`. + /// + /// The `o1`/`o3`/`o4` check is the one exception -- it matches a **standalone token** + /// (the ID split on non-alphanumeric characters) rather than a substring, because a + /// substring match on `"o1"` would false-positive inside unrelated IDs. + #[must_use] + pub fn from_id(id: &str) -> Self { + let id = id.to_ascii_lowercase(); + + if id.contains("gpt-5") { + CodexFamily::Gpt5 + } else if id.contains("gpt-4") { + CodexFamily::Gpt4 + } else if has_standalone_o_series_token(&id) { + CodexFamily::OSeries + } else if id.contains("codex") { + CodexFamily::CodexMini + } else { + CodexFamily::Other + } + } + + /// Every family, in a fixed order. + /// + /// Callers use this as a draw order and as a colour index. Fixed rather than sorted by + /// volume or by first appearance on purpose: a family that goes quiet and drops out + /// must not repaint the ones that remain. + pub const ALL: [Self; 5] = [ + Self::Gpt5, + Self::Gpt4, + Self::OSeries, + Self::CodexMini, + Self::Other, + ]; + + /// A short display label. + #[must_use] + pub fn label(self) -> &'static str { + match self { + CodexFamily::Gpt5 => "GPT-5", + CodexFamily::Gpt4 => "GPT-4", + CodexFamily::OSeries => "o-series", + CodexFamily::CodexMini => "Codex Mini", + CodexFamily::Other => "Other", + } + } +} + +impl fmt::Display for CodexFamily { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str(self.label()) + } +} + +/// Whether `id` contains `o1`, `o3`, or `o4` as a whole token, split on any character that +/// is not ASCII alphanumeric. `"foo1bar"` and `"promo1"` must not match; `"o1-mini"` and +/// `"gpt-o3"` must. +fn has_standalone_o_series_token(id: &str) -> bool { + id.split(|c: char| !c.is_ascii_alphanumeric()) + .any(|token| matches!(token, "o1" | "o3" | "o4")) +} + +#[cfg(test)] +mod tests { + use super::CodexFamily; + + #[test] + fn gpt5_and_gpt4_fold_by_substring() { + assert_eq!(CodexFamily::from_id("gpt-5.1-codex"), CodexFamily::Gpt5); + assert_eq!(CodexFamily::from_id("gpt-4o"), CodexFamily::Gpt4); + assert_eq!( + CodexFamily::from_id("GPT-5"), + CodexFamily::Gpt5, + "case-insensitive" + ); + } + + #[test] + fn gpt5_wins_over_codex_when_both_present() { + // `gpt-5.1-codex` contains both "gpt-5" and "codex"; the more specific model line + // must win, checked before the generic "codex" fallback. + assert_eq!(CodexFamily::from_id("gpt-5.1-codex"), CodexFamily::Gpt5); + } + + #[test] + fn o_series_matches_a_standalone_token_only() { + assert_eq!(CodexFamily::from_id("o1-mini"), CodexFamily::OSeries); + assert_eq!(CodexFamily::from_id("o3"), CodexFamily::OSeries); + assert_eq!(CodexFamily::from_id("gpt-o4"), CodexFamily::OSeries); + + assert_eq!( + CodexFamily::from_id("foo1bar"), + CodexFamily::Other, + "o1 embedded inside another token must not match" + ); + assert_eq!( + CodexFamily::from_id("promo1"), + CodexFamily::Other, + "o1 as a suffix of another word must not match" + ); + } + + #[test] + fn codex_branded_models_fold_to_codex_mini() { + assert_eq!( + CodexFamily::from_id("codex-mini-latest"), + CodexFamily::CodexMini + ); + } + + #[test] + fn unknown_ids_fold_to_other_rather_than_failing() { + assert_eq!( + CodexFamily::from_id("some-future-model"), + CodexFamily::Other + ); + assert_eq!(CodexFamily::from_id(""), CodexFamily::Other); + } +} diff --git a/crates/harness-metrics/src/codex/mod.rs b/crates/harness-metrics/src/codex/mod.rs new file mode 100644 index 00000000..f792ae58 --- /dev/null +++ b/crates/harness-metrics/src/codex/mod.rs @@ -0,0 +1,11 @@ +//! The Codex CLI backend: reads `~/.codex/sessions/**/rollout-*.jsonl` and +//! `~/.codex/archived_sessions/rollout-*.jsonl`. + +mod discover; +mod family; +mod transcript; + +pub use family::CodexFamily; +pub(crate) use transcript::Record; + +pub(crate) use discover::discover; diff --git a/crates/harness-metrics/src/codex/transcript.rs b/crates/harness-metrics/src/codex/transcript.rs new file mode 100644 index 00000000..4757b1b2 --- /dev/null +++ b/crates/harness-metrics/src/codex/transcript.rs @@ -0,0 +1,398 @@ +//! Parsing Codex CLI rollout records. +//! +//! Verified against upstream `openai/codex` and community documentation of the rollout +//! format, trusted over any other assumption. A rollout file is one JSON object per line: +//! `{"timestamp": "...", "type": "...", "payload": {...}}`. +//! +//! # Counting rules +//! +//! - A file is validated as a Codex rollout by checking that its **first line** has +//! `type == "session_meta"`. Every later line in a file that fails this check is dropped. +//! - Only `event_msg` records with `payload.type == "token_count"` carry usage. +//! `payload.info` is `null` on session-start pings and aborted turns; those are skipped. +//! - When `info` is present, `info.total_token_usage` is preferred: it is **cumulative for +//! the whole file**. The previous cumulative snapshot seen in this file is tracked in +//! [`Record`]'s parsing [`State`], and only the field-by-field growth since that snapshot +//! is reported -- otherwise a repeated identical cumulative value would double-count, and +//! every later event would restate everything that came before it. A cumulative value +//! that *decreases* (the file was truncated and replaced out from under an in-flight +//! read) is treated as "start over from here": the whole new value becomes the delta, +//! never a negative one. +//! - If `total_token_usage` is absent but `info.last_token_usage` is present, that value is +//! used directly as the delta -- it already describes one turn's usage, no running state +//! needed. +//! - Field mapping into [`TokenTotals`]: `input = input_tokens - cached_input_tokens`, +//! `output = output_tokens + reasoning_output_tokens`, `cache_read = cached_input_tokens`, +//! `cache_creation = 0`. `OpenAI`'s usage payload exposes no cache-*write* count, unlike +//! `Anthropic`'s -- there is nothing to put there. +//! - Model attribution: the most recently seen `payload.model` on a `turn_context` record, +//! falling back to `session_meta.payload.model` if no `turn_context` has appeared yet, +//! and finally to a hardcoded `"gpt-5"` guess if genuinely nothing has been seen. This +//! mirrors upstream Codex tooling's own fallback for an unknown model. +//! - Every usage record already carries its own computed delta, so +//! [`Record::dedupe_key`] always returns `None`: there is no cross-record repeat for the +//! ledger to detect, because the repeat detection already happened here, in `State`. + +use serde::Deserialize; + +use crate::{ + iso8601, + ledger::{HarnessRecord, TokenTotals}, +}; + +use super::family::CodexFamily; + +/// The raw four-field usage object nested under `info.last_token_usage` / +/// `info.total_token_usage`. +#[derive(Clone, Copy, Debug, Default, Deserialize)] +#[serde(default)] +// The shared `_tokens` suffix names exactly what these fields are (a token count) and +// matches the wire format field names; stripping it would make them less legible, not more. +#[allow(clippy::struct_field_names)] +struct RawUsage { + input_tokens: u64, + cached_input_tokens: u64, + output_tokens: u64, + reasoning_output_tokens: u64, +} + +impl From for TokenTotals { + fn from(raw: RawUsage) -> Self { + TokenTotals { + input: raw.input_tokens.saturating_sub(raw.cached_input_tokens), + output: raw + .output_tokens + .saturating_add(raw.reasoning_output_tokens), + cache_read: raw.cached_input_tokens, + // OpenAI exposes no cache-write count in this payload. + cache_creation: 0, + } + } +} + +#[derive(Clone, Debug, Default, Deserialize)] +#[serde(default)] +struct TokenCountInfo { + last_token_usage: Option, + total_token_usage: Option, +} + +/// The union of every `payload` shape this module reads from. Irrelevant fields for a +/// given record kind are simply absent and ignored. +#[derive(Clone, Debug, Default, Deserialize)] +#[serde(default)] +struct Payload { + /// `session_meta`'s model, and `turn_context`'s model. + model: Option, + /// `event_msg`'s own kind, e.g. `token_count`. + #[serde(rename = "type")] + kind: Option, + /// `event_msg` / `token_count`'s usage, `null` on a ping or an aborted turn. + info: Option, +} + +#[derive(Clone, Debug, Default, Deserialize)] +#[serde(default)] +struct Line { + timestamp: Option, + #[serde(rename = "type")] + kind: Option, + payload: Option, +} + +/// Parsing state threaded across every line of one rollout file. +#[derive(Debug, Default)] +pub(crate) struct State { + /// Set on the first line; `Some(false)` means the file failed the `session_meta` check + /// and every later line is dropped without being parsed. + valid: Option, + /// The most recently attributed model, updated by `session_meta` (once) and by every + /// `turn_context`. + model: Option, + /// The last cumulative `total_token_usage` snapshot seen in this file, used to compute + /// the next event's delta. + prev_cumulative: Option, +} + +/// One billable Codex token-count event, already reduced to the exact delta to add. +#[derive(Clone, Copy, Debug)] +pub struct Record { + timestamp_ms: Option, + family: &'static str, + usage: TokenTotals, +} + +impl HarnessRecord for Record { + type State = State; + + fn parse(line: &str, state: &mut Self::State) -> Option { + let line = line.trim(); + if line.is_empty() { + return None; + } + + // A malformed or truncated line is skipped, never fatal -- a tailer can legitimately + // read a half-written line at the end of a live file. + let parsed: Line = serde_json::from_str(line).ok()?; + + let valid = *state + .valid + .get_or_insert_with(|| parsed.kind.as_deref() == Some("session_meta")); + if !valid { + return None; + } + + match parsed.kind.as_deref() { + Some("session_meta") => { + // Only a fallback: `turn_context` overrides this the moment one appears. + if state.model.is_none() + && let Some(model) = parsed.payload.and_then(|p| p.model) + { + state.model = Some(model); + } + None + } + Some("turn_context") => { + if let Some(model) = parsed.payload.and_then(|p| p.model) { + state.model = Some(model); + } + None + } + Some("event_msg") => { + let payload = parsed.payload?; + if payload.kind.as_deref() != Some("token_count") { + return None; + } + + // `null` on a session-start ping or an aborted turn: nothing billable here. + let info = payload.info?; + + let usage = if let Some(total) = info.total_token_usage { + let mapped = TokenTotals::from(total); + let previous = state.prev_cumulative.unwrap_or_default(); + state.prev_cumulative = Some(mapped); + TokenTotals::delta_from(mapped, previous) + } else { + TokenTotals::from(info.last_token_usage?) + }; + + let family = + CodexFamily::from_id(state.model.as_deref().unwrap_or("gpt-5")).label(); + + Some(Record { + timestamp_ms: parsed.timestamp.as_deref().and_then(iso8601::parse_ms), + family, + usage, + }) + } + _ => None, + } + } + + fn timestamp_ms(&self) -> Option { + self.timestamp_ms + } + + /// Always `None`: the delta was already computed in [`State`] while parsing, so there + /// is nothing left for the ledger to deduplicate. + fn dedupe_key(&self) -> Option { + None + } + + fn family_label(&self) -> &'static str { + self.family + } + + fn billable_usage(&self) -> Option { + Some(self.usage) + } +} + +#[cfg(test)] +mod tests { + #![allow(clippy::unwrap_used, clippy::expect_used)] + + use super::*; + + const SESSION_META: &str = r#"{"timestamp":"2026-01-01T00:00:00.000Z","type":"session_meta","payload":{"id":"sess-1","cwd":"/repo","originator":"cli","model_provider":"openai","model":"gpt-5"}}"#; + + fn turn_context(model: &str) -> String { + format!( + r#"{{"timestamp":"2026-01-01T00:00:01.000Z","type":"turn_context","payload":{{"model":"{model}"}}}}"# + ) + } + + fn token_count_total( + stamp: &str, input: u64, cached: u64, output: u64, reasoning: u64, + ) -> String { + format!( + r#"{{"timestamp":"{stamp}","type":"event_msg","payload":{{"type":"token_count","info":{{"total_token_usage":{{"input_tokens":{input},"cached_input_tokens":{cached},"output_tokens":{output},"reasoning_output_tokens":{reasoning},"total_tokens":{}}}}}}}}}"#, + input + output + ) + } + + fn token_count_last( + stamp: &str, input: u64, cached: u64, output: u64, reasoning: u64, + ) -> String { + format!( + r#"{{"timestamp":"{stamp}","type":"event_msg","payload":{{"type":"token_count","info":{{"last_token_usage":{{"input_tokens":{input},"cached_input_tokens":{cached},"output_tokens":{output},"reasoning_output_tokens":{reasoning},"total_tokens":{}}}}}}}}}"#, + input + output + ) + } + + fn token_count_null_info(stamp: &str) -> String { + format!( + r#"{{"timestamp":"{stamp}","type":"event_msg","payload":{{"type":"token_count","info":null}}}}"# + ) + } + + fn parse_all(lines: &[String]) -> Vec { + let mut state = State::default(); + lines + .iter() + .filter_map(|line| Record::parse(line, &mut state)) + .collect() + } + + #[test] + fn a_file_not_starting_with_session_meta_is_rejected_entirely() { + let lines = vec![ + turn_context("gpt-5"), + token_count_total("2026-01-01T00:00:02.000Z", 100, 0, 20, 0), + ]; + + assert!( + parse_all(&lines).is_empty(), + "no session_meta first line, nothing counts" + ); + } + + #[test] + fn cumulative_snapshots_are_reduced_to_their_growth() { + let lines = vec![ + SESSION_META.to_owned(), + token_count_total("2026-01-01T00:00:02.000Z", 100, 0, 20, 0), + token_count_total("2026-01-01T00:00:03.000Z", 250, 0, 55, 0), + ]; + + let records = parse_all(&lines); + assert_eq!(records.len(), 2); + assert_eq!(records[0].usage.input, 100); + assert_eq!(records[0].usage.output, 20); + assert_eq!(records[1].usage.input, 150, "250 - 100"); + assert_eq!(records[1].usage.output, 35, "55 - 20"); + } + + #[test] + fn a_repeated_identical_cumulative_snapshot_does_not_double_count() { + let lines = vec![ + SESSION_META.to_owned(), + token_count_total("2026-01-01T00:00:02.000Z", 100, 0, 20, 0), + token_count_total("2026-01-01T00:00:03.000Z", 100, 0, 20, 0), + ]; + + let records = parse_all(&lines); + assert_eq!(records[1].usage, TokenTotals::default()); + } + + #[test] + fn a_decreasing_cumulative_snapshot_starts_over_rather_than_underflowing() { + let lines = vec![ + SESSION_META.to_owned(), + token_count_total("2026-01-01T00:00:02.000Z", 100, 0, 20, 0), + // Simulates the file having been truncated and replaced. + token_count_total("2026-01-01T00:00:03.000Z", 10, 0, 5, 0), + ]; + + let records = parse_all(&lines); + assert_eq!( + records[1].usage.input, 10, + "the lower value is taken fresh, not as -90" + ); + assert_eq!(records[1].usage.output, 5); + } + + #[test] + fn null_info_events_are_skipped() { + let lines = vec![ + SESSION_META.to_owned(), + token_count_null_info("2026-01-01T00:00:02.000Z"), + ]; + + assert!(parse_all(&lines).is_empty()); + } + + #[test] + fn reasoning_tokens_fold_into_output() { + let lines = vec![ + SESSION_META.to_owned(), + token_count_total("2026-01-01T00:00:02.000Z", 100, 0, 20, 15), + ]; + + assert_eq!(parse_all(&lines)[0].usage.output, 35); + } + + #[test] + fn cached_input_is_split_out_of_input_and_reported_as_cache_read() { + let lines = vec![ + SESSION_META.to_owned(), + token_count_total("2026-01-01T00:00:02.000Z", 100, 40, 20, 0), + ]; + + let usage = parse_all(&lines)[0].usage; + assert_eq!(usage.input, 60, "100 total input minus the 40 cached"); + assert_eq!(usage.cache_read, 40); + assert_eq!( + usage.cache_creation, 0, + "OpenAI exposes no cache-write count" + ); + } + + #[test] + fn last_token_usage_is_used_directly_when_total_is_absent() { + let lines = vec![ + SESSION_META.to_owned(), + token_count_last("2026-01-01T00:00:02.000Z", 30, 0, 10, 0), + ]; + + let usage = parse_all(&lines)[0].usage; + assert_eq!(usage.input, 30); + assert_eq!(usage.output, 10); + assert!( + parse_all(&lines)[0].family == "GPT-5", + "falls back to the session_meta model" + ); + } + + #[test] + fn model_tracking_follows_the_most_recent_turn_context() { + let lines = vec![ + SESSION_META.to_owned(), + turn_context("o3"), + token_count_total("2026-01-01T00:00:02.000Z", 10, 0, 5, 0), + turn_context("gpt-4o"), + token_count_total("2026-01-01T00:00:03.000Z", 20, 0, 10, 0), + ]; + + let records = parse_all(&lines); + assert_eq!(records[0].family, "o-series"); + assert_eq!(records[1].family, "GPT-4"); + } + + #[test] + fn dedupe_key_is_always_none() { + let lines = vec![ + SESSION_META.to_owned(), + token_count_total("2026-01-01T00:00:02.000Z", 10, 0, 5, 0), + ]; + + assert_eq!(parse_all(&lines)[0].dedupe_key(), None); + } + + #[test] + fn junk_and_empty_lines_are_skipped_rather_than_panicking() { + let mut state = State::default(); + assert!(Record::parse("", &mut state).is_none()); + assert!(Record::parse("{not json", &mut state).is_none()); + } +} diff --git a/crates/harness-metrics/src/iso8601.rs b/crates/harness-metrics/src/iso8601.rs new file mode 100644 index 00000000..b2b4bf4b --- /dev/null +++ b/crates/harness-metrics/src/iso8601.rs @@ -0,0 +1,170 @@ +//! A minimal `YYYY-MM-DDTHH:MM:SS[.sss]Z` parser, shared by every backend. +//! +//! Every harness this crate reads timestamps every billable record with a fixed-width, +//! always-UTC ISO-8601 instant. Hand-parsed rather than pulled in through a date library: +//! it is a few field reads and a days-since-epoch calculation, and it keeps a crate whose +//! whole point is having no dependency on `mon` from growing one on `chrono` for one field. + +/// Parse `YYYY-MM-DDTHH:MM:SS[.sss]Z` into Unix epoch milliseconds. +/// +/// Only this exact shape is accepted. Anything else returns `None` and the caller drops +/// the record rather than guessing a time for it. Offsets other than `Z` are deliberately +/// unsupported -- silently treating `+02:00` as UTC would put records in the wrong bucket, +/// which is worse than losing them. +#[must_use] +pub(crate) fn parse_ms(stamp: &str) -> Option { + let bytes = stamp.as_bytes(); + + // The shortest accepted form is `YYYY-MM-DDTHH:MM:SSZ`. + if bytes.len() < 20 || bytes[4] != b'-' || bytes[7] != b'-' || bytes[10] != b'T' { + return None; + } + + if bytes[13] != b':' || bytes[16] != b':' || !stamp.ends_with('Z') { + return None; + } + + let field = |from: usize, to: usize| stamp.get(from..to)?.parse::().ok(); + + let (year, month, day) = (field(0, 4)?, field(5, 7)?, field(8, 10)?); + let (hour, minute, second) = (field(11, 13)?, field(14, 16)?, field(17, 19)?); + + // The fractional part is optional and of unspecified length; take up to three digits + // and pad, so both `.9` and `.919` mean what they say. + let millis = match bytes.get(19) { + Some(b'.') => { + let digits: String = stamp[20..] + .chars() + .take_while(char::is_ascii_digit) + .take(3) + .collect(); + + // At most three digits were taken, so this cannot underflow. + let missing = 3 - u32::try_from(digits.len()).ok()?; + digits.parse::().ok()? * 10u64.pow(missing) + } + _ => 0, + }; + + if !(1..=12).contains(&month) || !(1..=31).contains(&day) { + return None; + } + + if hour > 23 || minute > 59 || second > 60 { + return None; + } + + let days = days_from_civil(year, month, day)?; + + Some( + days.saturating_mul(86_400_000) + .saturating_add(hour * 3_600_000) + .saturating_add(minute * 60_000) + .saturating_add(second * 1_000) + .saturating_add(millis), + ) +} + +/// Days from 1970-01-01 to a civil date, by Howard Hinnant's `days_from_civil`. +/// +/// The shift-the-year-to-March trick is what makes leap days fall out of the arithmetic +/// instead of needing a special case: with March as month zero, the leap day lands at the +/// end of the year where it cannot disturb the month-length series. +fn days_from_civil(year: u64, month: u64, day: u64) -> Option { + // Before the epoch there is nothing to attribute, and the unsigned arithmetic below + // would wrap rather than go negative. + if year < 1970 { + return None; + } + + let year = if month <= 2 { year - 1 } else { year }; + let era = year / 400; + let year_of_era = year - era * 400; + + let month_shift = if month > 2 { month - 3 } else { month + 9 }; + let day_of_year = (153 * month_shift + 2) / 5 + day - 1; + let day_of_era = year_of_era * 365 + year_of_era / 4 - year_of_era / 100 + day_of_year; + + // 719468 is the day count from 0000-03-01 to 1970-01-01. + Some(era * 146_097 + day_of_era - 719_468) +} + +#[cfg(test)] +mod tests { + #![allow(clippy::unwrap_used, clippy::expect_used)] + + use super::*; + + #[test] + fn a_real_timestamp_parses_to_epoch_millis() { + // Taken verbatim from a live transcript. + assert_eq!( + parse_ms("2026-08-24T20:26:13.919Z"), + Some(1_787_603_173_919) + ); + } + + #[test] + fn the_epoch_itself_is_zero() { + assert_eq!(parse_ms("1970-01-01T00:00:00.000Z"), Some(0)); + } + + #[test] + fn leap_days_are_counted() { + // The whole reason for the shift-the-year-to-March trick. A day either side of the + // 2024 leap day must be exactly 86400000ms apart across it. + let before = parse_ms("2024-02-28T00:00:00Z").expect("valid"); + let leap = parse_ms("2024-02-29T00:00:00Z").expect("valid"); + let after = parse_ms("2024-03-01T00:00:00Z").expect("valid"); + + assert_eq!(leap - before, 86_400_000); + assert_eq!(after - leap, 86_400_000); + } + + #[test] + fn a_century_that_is_not_a_leap_year_is_handled() { + // 1900 is not a leap year but 2000 is, which is the case a naive `% 4` gets wrong. + let before = parse_ms("2000-02-28T00:00:00Z").expect("valid"); + let after = parse_ms("2000-03-01T00:00:00Z").expect("valid"); + + assert_eq!(after - before, 2 * 86_400_000, "2000 had a leap day"); + } + + #[test] + fn a_short_fraction_is_padded_rather_than_misread() { + // `.9` is nine hundred milliseconds, not nine. + assert_eq!( + parse_ms("2026-08-24T20:26:13.9Z"), + parse_ms("2026-08-24T20:26:13.900Z") + ); + } + + #[test] + fn a_missing_fraction_is_fine() { + assert_eq!( + parse_ms("2026-08-24T20:26:13Z"), + parse_ms("2026-08-24T20:26:13.000Z") + ); + } + + #[test] + fn a_non_utc_offset_is_refused_rather_than_silently_misread() { + // Treating `+02:00` as UTC would put records two hours into the wrong bucket, which + // is worse than losing them. + assert_eq!(parse_ms("2026-08-24T20:26:13.919+02:00"), None); + } + + #[test] + fn junk_timestamps_yield_nothing_rather_than_panicking() { + for stamp in [ + "", + "not a date", + "2026-08-24", + "2026-13-01T00:00:00Z", + "2026-08-24T25:00:00Z", + "1969-12-31T23:59:59Z", + ] { + assert_eq!(parse_ms(stamp), None, "{stamp:?} must not parse"); + } + } +} diff --git a/crates/harness-metrics/src/ledger.rs b/crates/harness-metrics/src/ledger.rs new file mode 100644 index 00000000..0ddeb102 --- /dev/null +++ b/crates/harness-metrics/src/ledger.rs @@ -0,0 +1,774 @@ +//! The harness-agnostic engine: tails a set of transcript files and turns them into a +//! time-bucketed history plus an unbounded running cumulative total, per model family. +//! +//! # Why this drives every harness identically +//! +//! Claude Code, Codex CLI, and pi each write their own token usage to disk in their own +//! shape, on their own schedule, with their own idea of "how much has this session used so +//! far". Rather than teach this module three sets of rules, each backend (`claude`, +//! `codex`, `pi`) implements [`HarnessRecord`] once and does all of its own counting-rule +//! work internally -- dedup, running-total high-water-marks, cumulative-snapshot deltas, +//! whatever its schema requires. This module never inspects which harness it is reading; +//! it only calls the trait. +//! +//! That uniformity has a real cost: every harness's transcript is read a tick behind the +//! model call that produced it, never synchronously. A caller wanting "tokens per second" +//! differences the cumulative snapshot between two refreshes rather than hooking the model +//! call directly. This lag has been accepted deliberately in exchange for treating every +//! harness the same way, with no special-cased harness sitting closer to real time than the +//! others. +//! +//! # Why bucket assignment is not pinned like a "message" +//! +//! [`Ledger::ingest`] attributes every delta to the bucket named by *that record's own* +//! timestamp, not to the bucket the dedupe key first appeared in. A backend whose dedupe +//! key spans an entire file (a Codex rollout's cumulative counter, say) can otherwise emit +//! events minutes or hours apart; pinning all of them to the file's first event would smear +//! a whole session's usage into one bucket. A backend whose dedupe key spans one message's +//! content blocks (Claude) still lands correctly, because those blocks are written within +//! the same turn and, in practice, the same bucket. + +use std::{ + collections::{BTreeMap, HashMap}, + marker::PhantomData, + path::{Path, PathBuf}, + time::{Duration, SystemTime, UNIX_EPOCH}, +}; + +use crate::tailer::{ReadKind, Tailer}; + +/// Token counts for one model family. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] +pub struct TokenTotals { + /// Fresh input tokens. + pub input: u64, + /// Generated tokens. + pub output: u64, + /// Tokens read from the prompt cache. + pub cache_read: u64, + /// Tokens written to the prompt cache. + pub cache_creation: u64, +} + +impl TokenTotals { + /// Every token this family was billed for, cache included. + #[must_use] + pub fn total(&self) -> u64 { + self.input + .saturating_add(self.output) + .saturating_add(self.cache_read) + .saturating_add(self.cache_creation) + } + + /// Add another totals into this one, field by field, saturating rather than wrapping. + pub fn saturating_add_assign(&mut self, other: TokenTotals) { + self.input = self.input.saturating_add(other.input); + self.output = self.output.saturating_add(other.output); + self.cache_read = self.cache_read.saturating_add(other.cache_read); + self.cache_creation = self.cache_creation.saturating_add(other.cache_creation); + } + + /// The field-by-field difference from `old` to `new`. + /// + /// Used to turn a running or cumulative snapshot into an incremental delta. A field + /// that has *decreased* -- the file was truncated or replaced out from under a counter + /// that is not otherwise tracked as restarted -- is treated as "start over from here" + /// rather than underflowing: the whole new value is taken as the delta, never a + /// negative one. This never panics. + /// + /// `pub(crate)` rather than private: a backend that computes its own cumulative delta + /// internally (Codex's per-file running counter, tracked in its own [`HarnessRecord::State`] + /// rather than through [`Ledger`]'s keyed-snapshot path) reuses this exact rule rather + /// than reimplementing it. + #[must_use] + pub(crate) fn delta_from(new: TokenTotals, old: TokenTotals) -> TokenTotals { + fn field(new: u64, old: u64) -> u64 { + if new >= old { new - old } else { new } + } + + TokenTotals { + input: field(new.input, old.input), + output: field(new.output, old.output), + cache_read: field(new.cache_read, old.cache_read), + cache_creation: field(new.cache_creation, old.cache_creation), + } + } +} + +/// One backend's parsed line, plus everything the ledger needs to fold it in. +/// +/// Every harness-specific counting rule lives behind this trait, implemented once per +/// backend. [`Ledger`] never matches on which harness it is driving; it only calls these +/// methods. +pub(crate) trait HarnessRecord: Sized { + /// Parsing state threaded across every line read from one file, reset whenever that + /// file's tailer restarts (replaced or truncated). Claude and pi have no use for this + /// and set it to `()`; Codex uses it to track the most recently seen model, which is + /// reported on a separate record type than the one that carries token usage. + type State: Default; + + /// Parse one line, given the running state for the file it came from. May update + /// `state` in place. Returns `None` for a line that does not parse or carries nothing + /// this backend cares about. + fn parse(line: &str, state: &mut Self::State) -> Option; + + /// This record's instant, in Unix epoch milliseconds. `None` records are dropped: with + /// no timestamp there is no bucket to attribute the usage to, and guessing "now" would + /// put every undated record in the newest bucket and draw a spike that never happened. + fn timestamp_ms(&self) -> Option; + + /// A key identifying a value that may repeat or supersede an earlier one for the same + /// logical unit of work, so the ledger can take a delta instead of double-counting. + /// + /// `None` means this record's [`billable_usage`](Self::billable_usage) is already the + /// exact amount to add -- no further tracking needed. `Some(key)` means it is a + /// snapshot (a running total or a cumulative counter): the first time a key is seen its + /// full value is added, and every later record sharing the key contributes only the + /// field-by-field growth since the previous snapshot under that key. + fn dedupe_key(&self) -> Option; + + /// The model family this record belongs to, already folded by the backend. + fn family_label(&self) -> &'static str; + + /// This record's usage, or `None` if it carries none. + fn billable_usage(&self) -> Option; +} + +/// One bucket's worth of usage, by model family label. +#[derive(Clone, Debug, Default, PartialEq, Eq)] +pub struct Bucket { + /// Start of the bucket, in Unix epoch milliseconds. + pub start_ms: u64, + /// Families that contributed, in first-seen order. + pub totals: Vec<(&'static str, TokenTotals)>, +} + +impl Bucket { + /// Every token in this bucket, across all families. + #[must_use] + pub fn total(&self) -> u64 { + self.totals + .iter() + .fold(0u64, |sum, (_, totals)| sum.saturating_add(totals.total())) + } + + /// This family's tokens in this bucket, or zero if it did not contribute. + #[must_use] + pub fn total_for(&self, label: &str) -> u64 { + self.totals + .iter() + .find(|(candidate, _)| *candidate == label) + .map_or(0, |(_, totals)| totals.total()) + } +} + +/// What a dedupe key last contributed, remembered so a later snapshot can be turned into a +/// delta instead of double-counted. +#[derive(Clone, Copy, Debug)] +struct Counted { + /// The most recent full snapshot seen under this key. + prev: TokenTotals, + /// The bucket this key last contributed to, used only to decide when its entry has + /// aged out of the window -- it does not pin where future deltas land. + last_bucket: u64, +} + +/// Per-file tailing state: the byte-offset tailer plus the backend's own parsing state. +struct FileState { + tailer: Tailer, + parser: S, +} + +/// Tails a set of discovered files for one harness backend and turns what they contain +/// into a time-bucketed history and an unbounded running cumulative total, both keyed by +/// model family label. +/// +/// Refreshing is incremental: each file keeps a byte-offset checkpoint, so a refresh only +/// parses what has been appended since the last one. +pub(crate) struct Ledger { + root: PathBuf, + window: Duration, + bucket: Duration, + discover: fn(&Path, u64) -> Vec, + buckets: BTreeMap>, + cumulative: HashMap<&'static str, TokenTotals>, + files: HashMap>, + seen: HashMap, + _record: PhantomData, +} + +impl Ledger { + /// A ledger over `window`, split into buckets of `bucket`, reading `root` through + /// `discover` -- a backend-supplied function mapping a root and a cutoff (Unix epoch + /// milliseconds) to every file that might contain a record at or after that cutoff. + /// + /// A zero or absurd `bucket` is clamped to something drawable rather than rejected -- + /// this sits in a draw path and a config typo should not take the app down. + pub(crate) fn new( + root: impl Into, window: Duration, bucket: Duration, + discover: fn(&Path, u64) -> Vec, + ) -> Self { + let bucket = bucket.clamp(Duration::from_secs(1), Duration::from_hours(24)); + let window = window.clamp(bucket, Duration::from_hours(90 * 24)); + + Self { + root: root.into(), + window, + bucket, + discover, + buckets: BTreeMap::new(), + cumulative: HashMap::new(), + files: HashMap::new(), + seen: HashMap::new(), + _record: PhantomData, + } + } + + /// The root being read. + #[must_use] + pub(crate) fn root(&self) -> &Path { + &self.root + } + + /// The bucket width. + #[must_use] + pub(crate) fn bucket(&self) -> Duration { + self.bucket + } + + /// The window covered, oldest bucket to now. + #[must_use] + pub(crate) fn window(&self) -> Duration { + self.window + } + + /// Re-read whatever the discovered files have appended and drop anything now out of + /// window. `now_ms` is passed in rather than read from the clock so this is testable + /// without waiting for real time to pass. + pub(crate) fn refresh_at(&mut self, now_ms: u64) { + let cutoff = now_ms.saturating_sub(millis(self.window)); + + for path in (self.discover)(&self.root, cutoff) { + let state = self.files.entry(path.clone()).or_insert_with(|| FileState { + tailer: Tailer::new(path), + parser: R::State::default(), + }); + + let (lines, kind) = state.tailer.read_new(); + + // A replaced file means any in-flight parsing state, and any dedupe keys + // tracking it, describe a file that is no longer there. The buckets already + // built stay -- they describe real tokens that were really spent -- but + // continuing to track deltas against the old snapshot would corrupt every + // record after the swap. + if kind == ReadKind::Restarted { + state.parser = R::State::default(); + self.seen.clear(); + } + + // Parsed while `state` still borrows `self.files`, then folded in afterwards -- + // `ingest` needs `self` back to update `self.buckets`/`self.cumulative`/`self.seen`. + let records: Vec = lines + .into_iter() + .filter_map(|line| R::parse(&line, &mut state.parser)) + .collect(); + + for record in &records { + self.ingest(record, cutoff); + } + } + + self.evict(cutoff); + } + + /// Refresh against the system clock. + pub(crate) fn refresh(&mut self) { + self.refresh_at(now_ms()); + } + + /// Fold one record in. + fn ingest(&mut self, record: &R, cutoff: u64) { + let Some(usage) = record.billable_usage() else { + return; + }; + + // Without a timestamp there is no bucket to attribute it to. + let Some(stamp) = record.timestamp_ms() else { + return; + }; + + if stamp < cutoff { + return; + } + + let bucket_start = stamp - (stamp % millis(self.bucket)); + let family = record.family_label(); + + let delta = match record.dedupe_key() { + None => usage, + Some(key) => { + if let Some(counted) = self.seen.get_mut(&key) { + let delta = TokenTotals::delta_from(usage, counted.prev); + counted.prev = usage; + counted.last_bucket = bucket_start; + delta + } else { + self.seen.insert( + key, + Counted { + prev: usage, + last_bucket: bucket_start, + }, + ); + usage + } + } + }; + + if delta == TokenTotals::default() { + return; + } + + self.totals_for_mut(bucket_start, family) + .saturating_add_assign(delta); + + self.cumulative + .entry(family) + .or_default() + .saturating_add_assign(delta); + } + + fn totals_for_mut(&mut self, bucket: u64, family: &'static str) -> &mut TokenTotals { + let entry = self.buckets.entry(bucket).or_default(); + + if let Some(index) = entry.iter().position(|(candidate, _)| *candidate == family) { + return &mut entry[index].1; + } + + entry.push((family, TokenTotals::default())); + + // The push above guarantees a last element; an `expect` here would be a panic path + // in a refresh loop for a case the compiler simply cannot see. + match entry.last_mut() { + Some((_, totals)) => totals, + None => unreachable!("an element was just pushed"), + } + } + + /// Drop buckets, and the dedupe keys pointing at them, that have aged out. + fn evict(&mut self, cutoff: u64) { + let stale = cutoff - (cutoff % millis(self.bucket)); + self.buckets.retain(|start, _| *start >= stale); + // Otherwise a long-running process grows this map forever. + self.seen.retain(|_, counted| counted.last_bucket >= stale); + } + + /// Buckets in the window, oldest first, with empty ones filled in. + /// + /// The gaps matter: a graph drawn from only the buckets that saw traffic would join a + /// point at 10:00 straight to one at 10:40 and draw a plateau across half an hour of + /// silence. `now_ms` fixes the right-hand edge. + #[must_use] + pub(crate) fn buckets_at(&self, now_ms: u64) -> Vec { + let step = millis(self.bucket); + let newest = now_ms - (now_ms % step); + let count = (millis(self.window) / step).max(1); + let oldest = newest.saturating_sub(step.saturating_mul(count - 1)); + + (0..count) + .map(|index| { + let start_ms = oldest + index * step; + + Bucket { + start_ms, + totals: self.buckets.get(&start_ms).cloned().unwrap_or_default(), + } + }) + .collect() + } + + /// Buckets in the window against the system clock. + #[must_use] + pub(crate) fn buckets(&self) -> Vec { + self.buckets_at(now_ms()) + } + + /// Family labels that contributed anything in the window, filtered from `order` to + /// preserve a fixed draw/colour order rather than first-appearance or volume order -- + /// a family going quiet must not repaint the ones that remain. + #[must_use] + pub(crate) fn families_present(&self, order: &[&'static str]) -> Vec<&'static str> { + order + .iter() + .copied() + .filter(|family| { + self.buckets + .values() + .flatten() + .any(|(candidate, totals)| candidate == family && totals.total() > 0) + }) + .collect() + } + + /// The unbounded running cumulative total per family label, in `order`. + /// + /// This never evicts. A caller differences two snapshots of this across refresh ticks + /// to derive a tokens/second rate. + #[must_use] + pub(crate) fn cumulative_totals( + &self, order: &[&'static str], + ) -> Vec<(&'static str, TokenTotals)> { + order + .iter() + .filter_map(|family| self.cumulative.get(family).map(|totals| (*family, *totals))) + .collect() + } +} + +/// Unix epoch milliseconds, now. +fn now_ms() -> u64 { + SystemTime::now() + .duration_since(UNIX_EPOCH) + .map_or(0, millis) +} + +fn millis(duration: Duration) -> u64 { + u64::try_from(duration.as_millis()).unwrap_or(u64::MAX) +} + +#[cfg(test)] +mod tests { + #![allow(clippy::unwrap_used, clippy::expect_used)] + + use std::cell::Cell; + + use super::*; + + /// `2026-08-24T20:00:00.000Z` in epoch millis, a round bucket boundary. + const BASE: u64 = 1_787_601_600_000; + + /// A tiny fixture record: `"
/rollout-*.jsonl` (or `~/.codex/sessions` when +`CODEX_HOME` is unset) plus `archived_sessions/rollout-*.jsonl`. Series are model families: +GPT-5, GPT-4, the o-series (o1/o3/o4), Codex-branded minis, Other, folded from the raw model +id by substring match. + +Counting rules: only `event_msg/token_count` events with a non-null `info` count (a null +`info` is a session-start ping or an aborted turn); the cumulative `info.total_token_usage` +is preferred over `info.last_token_usage`, and this crate tracks the previous cumulative +snapshot per rollout file to take the delta, so a repeated identical snapshot is not +double-counted; reasoning tokens (`reasoning_output_tokens`) are folded into output, priced +the same as output; Codex exposes no cache-write count, so cache-creation is always zero for +this harness. + +### `source = "pi"` + +Reads `~/.pi/agent/sessions//_.jsonl`. Series are providers: +Anthropic, OpenAI, Google, Other, taken directly from each assistant message's own +`provider` field rather than folded from its model id -- pi is multi-provider by design, and +a per-model series would balloon as models rotate. + +Counting rules: every assistant message carries one clean per-turn `usage` object, taken +directly with no running-total or high-water-mark trick needed (unlike Claude); `usage` on +`compaction` and `branch_summary` entries is also counted, since both represent a real +summarization LLM call. + +### `source = "all"` + +Merges the three ledgers above into one, grouped by **harness label** (`Claude`/`Codex`/ +`Pi`) instead of by model family or provider -- every family or provider within one harness +is summed into a single number for that harness. A harness with nothing on disk (never run +on this machine) simply contributes no series, nothing to configure. + +## Key bindings + +`agent_graph` and `agent_stats` take the usual graph bindings. + +| Binding | Action | +| --------- | ---------------------------------------- | +| ++plus++ | Zoom in on chart (decrease time range) | +| ++minus++ | Zoom out on chart (increase time range) | +| ++equal++ | Reset zoom | diff --git a/docs/content/usage/widgets/claude.md b/docs/content/usage/widgets/claude.md deleted file mode 100644 index 84e69442..00000000 --- a/docs/content/usage/widgets/claude.md +++ /dev/null @@ -1,146 +0,0 @@ -# Claude Widgets - -!!! note "Fork addition" - - These widgets are specific to [mon](https://github.com/nredd/mon) and are not part of - upstream bottom. - -Three widgets read [Claude Code](https://claude.com/claude-code) activity off the local -`~/.claude` tree: - -- `claude` -- a table of live sessions -- `claude_graph` -- token throughput over time, by model family -- `claude_stats` -- token spend over the last hour, as stacked bands by model family - -None of them are in the default layout. Add them to your -[layout](../../configuration/config-file/layout.md) to use them, or start from -[`sample_configs/claude_config.toml`](https://github.com/nredd/mon/blob/main/sample_configs/claude_config.toml), -which draws all three and nothing else: - -```console -$ mon -C sample_configs/claude_config.toml --pixel_graphs kitty -``` - -That config also carries the family colour list and the two `use_log` toggles inline, so it -is a reasonable place to retune them. - -## Sessions table - -One row per running session, discovered from `~/.claude/sessions/.json` and pruned with -`kill(pid, 0)` so a session that died without cleaning up does not linger. - -| Column | Source | -| ------ | ------ | -| `Session` | The session's name | -| `Dir` | Last component of its working directory | -| `Model` | Model family, folded from the raw model id | -| `State` | `busy` / `idle` | -| `Tokens` | Every token spent, cache included | -| `Cost` | USD so far | -| `Ctx` | Context-window occupancy | -| `Agents` | Subagent messages produced | - -Sort by clicking a header or pressing the letter in its label. Defaults to `Tokens`, -descending. - -## Token graph - -Token throughput in tokens/second, one line per model family, differenced from the -cumulative totals. - -The y-axis is **logarithmic by default**. Cache reads run into the millions of tokens per -second while fresh input tokens are single digits, so on a linear axis every series but the -largest sits flat on the floor. Set `use_log = false` to switch. - -## Stats graph - -The equivalent of Claude Code's own `/status` stats screen. Claude Code bars token spend by -day; this buckets by **minute over the last hour**, so the shape of a working session is -visible rather than collapsed into a single bar. - -Families are drawn as stacked bands, so the top of the stack is the total spend in that -minute and each band is one family's share. The legend carries each family's total across -the whole window. - -The y-axis is **linear by default**, unlike the token graph. A bucketed total spans a far -narrower range than an instantaneous rate -- a busy minute and a quiet one differ by a factor -of ten, not by five orders of magnitude -- and stacked bands only add up to the total on a -linear axis. Set `stats_use_log = true` if one family dwarfs the rest badly enough to need -it, accepting that the bands stop summing to the visible total. - -Each band is drawn as a **rounded staircase**, holding a bucket's value flat across the -minute it covers. That is not only cosmetic: a bucket is a total over a minute, not a -reading at an instant, so sloping between bucket centres would spread one busy minute over -three and understate its peak. The corner rounding is cosmetic, and is what makes the chart -read the way `/status` does rather than like a bar code. - -Both the fill and the stepping are pixel-path-only. With cell markers the graph degrades to -straight-joined band boundaries, which is still readable -- the same trade the pixel path -makes everywhere else. - -This is the only part of a Claude harvest that reads transcripts outside the live sessions, -so it is skipped entirely unless the widget is in your layout. - -## Where the numbers come from - -Tokens, agent counts, and turn durations are parsed out of the session transcripts under -`~/.claude/projects`. - -`claude_stats` walks that tree itself rather than building on the live sessions, and -attributes each record to a bucket using the record's own timestamp. Two consequences worth -knowing: the window is complete the moment the widget first appears, instead of having to be -accumulated live over an hour before the graph says anything, and it keeps the tokens of -sessions that have since exited -- which the live-session view cannot, since it drops a -session's state as soon as it leaves the registry. Candidate files are filtered by -modification time before being opened, so a tree with a year of transcripts in it still only -opens the handful touched inside the window. - -**Cost, context-window occupancy, and rate limits need the statusline tee.** Claude Code -hands those to the statusline command on stdin and writes them nowhere else on disk, so -`mon` reads them from a cache that your statusline has to populate. Add this near the top of -`~/.claude/statusline.sh`, right after it reads stdin: - -```bash -_mon_cache_payload() { - local dir="${HOME}/.claude/statusline-cache" - mkdir -p "$dir" 2>/dev/null || return 0 - - local key - key=$(printf '%s' "$input" | jq -r '.session_id // .sessionId // empty' 2>/dev/null) - if [ -z "$key" ]; then - key=$(printf '%s' "$input" | jq -r '.workspace.current_dir // .cwd // "unknown"' 2>/dev/null | tr '/._' '---') - fi - [ -n "$key" ] || key="unknown" - - local tmp="${dir}/.${key}.$$" - printf '%s' "$input" >"$tmp" 2>/dev/null || { rm -f "$tmp" 2>/dev/null; return 0; } - mv -f "$tmp" "${dir}/${key}.json" 2>/dev/null || rm -f "$tmp" 2>/dev/null - - find "$dir" -maxdepth 1 -name '*.json' -mtime +1 -delete 2>/dev/null - return 0 -} 2>/dev/null -_mon_cache_payload || true -``` - -Without it the `Cost` and `Ctx` columns read `N/A` and everything else still works. - -!!! warning "The numbers are a close estimate, not billing truth" - - Two gaps, both properties of the data source: - - - Background Haiku calls (session titles and similar) are billed but never written to a - transcript, so they are invisible here. - - A few percent of a long session's tokens are simply not in the tree. Calibrated - against `~/.claude.json`'s `lastModelUsage`, a short session matched exactly on all - four token fields; a 1199-line one matched input, cache-read, and cache-write exactly - with output at 97.7%. - -## Key bindings - -`claude_graph` takes the usual graph bindings. - -| Binding | Action | -| --------- | --------------------------------------- | -| ++plus++ | Zoom in on chart (decrease time range) | -| ++minus++ | Zoom out on chart (increase time range) | -| ++equal++ | Reset zoom | diff --git a/docs/mkdocs.yml b/docs/mkdocs.yml index d8c6d4d5..b39222ae 100644 --- a/docs/mkdocs.yml +++ b/docs/mkdocs.yml @@ -170,7 +170,7 @@ nav: - "Temperature Graph Widget": usage/widgets/temperature-graph.md - "Disk I/O Graph Widget": usage/widgets/disk-io-graph.md - "Power Widget": usage/widgets/power.md - - "Claude Widgets": usage/widgets/claude.md + - "Agent Widgets": usage/widgets/agent.md - "Battery Widget": usage/widgets/battery.md - "Auto-Complete": usage/autocomplete.md - "Configuration": @@ -186,7 +186,7 @@ nav: - "Temperature Graph Widget": configuration/config-file/temperature-graph.md - "Disk I/O Graph Widget": configuration/config-file/disk-io-graph.md - "Power Widget": configuration/config-file/power.md - - "Claude Widgets": configuration/config-file/claude.md + - "Agent Widgets": configuration/config-file/agent.md - "Flags": configuration/config-file/flags.md - "Layout": configuration/config-file/layout.md - "Styling": configuration/config-file/styling.md diff --git a/examples/history_probe.rs b/examples/history_probe.rs index 6e36ce20..5d1d51ef 100644 --- a/examples/history_probe.rs +++ b/examples/history_probe.rs @@ -6,23 +6,24 @@ use std::time::Duration; -use claude_metrics::TokenHistory; +use harness_metrics::{Harness, HarnessLedger}; fn main() { - let Some(home) = std::env::var_os("HOME") else { + let Some(mut ledger) = HarnessLedger::new( + Harness::Claude, + Duration::from_secs(3600), + Duration::from_secs(60), + ) else { eprintln!("no HOME"); return; }; - let root = std::path::Path::new(&home).join(".claude"); - let mut history = TokenHistory::new(&root, Duration::from_secs(3600), Duration::from_secs(60)); - let started = std::time::Instant::now(); - history.refresh(); + ledger.refresh(); let elapsed = started.elapsed(); - let families = history.families(); - let buckets = history.buckets(); + let families = ledger.families_present(); + let buckets = ledger.buckets(); println!("refresh took {elapsed:?}"); println!("families: {families:?}"); @@ -31,7 +32,7 @@ fn main() { for bucket in buckets.iter().filter(|b| b.total() > 0) { let per: Vec = families .iter() - .map(|f| format!("{}={}", f.label(), bucket.total_for(*f))) + .map(|f| format!("{}={}", f, bucket.total_for(f))) .collect(); println!( diff --git a/sample_configs/all_harnesses_config.toml b/sample_configs/all_harnesses_config.toml new file mode 100644 index 00000000..99825ab5 --- /dev/null +++ b/sample_configs/all_harnesses_config.toml @@ -0,0 +1,53 @@ +# Track every coding-agent harness mon knows about, combined, in one view. +# +# `source = "all"` merges Claude Code + Codex CLI + pi into one series set, one line/band +# per harness rather than per model family. This is the direct answer to "how much am I +# spending across every harness right now" -- each harness is still read the same lagging, +# tailed-transcript way as the single-harness configs, just summed together per harness +# instead of split out by model family within one harness. +# +# A harness that has never run on this machine (no `~/.claude`, no `~/.codex` or +# `$CODEX_HOME`, no `~/.pi/agent`) simply contributes no series -- nothing to configure. +# +# Run it with: +# +# cargo run --release -- -C sample_configs/all_harnesses_config.toml --pixel_graphs kitty + +[flags] +rate = "1s" +retention = "1h" +default_time_value = "5m" +time_delta = "30s" +hide_time = true + +[agent] +use_log = true +stats_use_log = false +legend_position = "top-right" + +[styles.agent] +# Harness order: Claude, Codex, Pi. Three entries, not five -- `source = "all"` draws one +# series per harness, not per model family. +colours = ["#3987e5", "#d95926", "#199e70"] + +[styles.widgets] +border_colour = "#4a4a4a" +selected_border_colour = "#c98500" +widget_title = { colour = "#c98500", bold = true } + +[styles.graphs] +graph_colour = "#6a6a6a" +legend_text = { colour = "#b4b4b4" } + +[[row]] +ratio = 58 +[[row.child]] +type = "agent_stats" +source = "all" +default = true + +[[row]] +ratio = 42 +[[row.child]] +type = "agent_graph" +source = "all" diff --git a/sample_configs/claude_config.toml b/sample_configs/claude_config.toml index 33fa89a1..89f32651 100644 --- a/sample_configs/claude_config.toml +++ b/sample_configs/claude_config.toml @@ -1,8 +1,8 @@ # A Claude-Code-only layout for mon. # -# Everything system-related is dropped. What is left is the three Claude widgets, stacked -# largest-first: the per-minute token history (mon's answer to Claude Code's own `/status` -# Models tab), the live token-rate graph underneath it, and the session table at the bottom. +# `agent_graph`/`agent_stats` are harness-agnostic widgets -- this config just points both +# at `source = "claude"`. History on top (mon's answer to Claude Code's own `/status` Models +# tab), the live token-rate graph underneath it. # # Run it with: # @@ -11,13 +11,15 @@ # Drop `--pixel_graphs kitty` outside Ghostty/Kitty/WezTerm and the graphs fall back to # braille cell markers. # -# There is nothing to point at `~/.claude` here. `claude-metrics` resolves `$HOME/.claude` +# There is nothing to point at `~/.claude` here. `harness-metrics` resolves `$HOME/.claude` # itself and reads the transcripts on disk, so this config works as-is on any machine that -# has run Claude Code. +# has run Claude Code. Every harness -- Claude included -- is read a tick behind the actual +# model call: there is no live/PID-registry fast path for any of them, on purpose, so no +# harness gets special-cased over the others. [flags] # One collection tick a second. The history widget rebuilds its window from disk rather -# than accumulating live, so this only paces the rate graph and the session table. +# than accumulating live, so this only paces the rate graph. rate = "1s" # Keep an hour of rate samples so the token graph can zoom out to match the history @@ -29,13 +31,7 @@ time_delta = "30s" # Both graphs carry a legend, so the extra time axis under each one is noise. hide_time = true -# A rule between the session table's header and its rows. -table_gap = "line" - -# Nothing here can stop a process, so take the kill path off the table entirely. -read_only = true - -[claude] +[agent] # The rate graph spans orders of magnitude -- cache reads run into millions of tokens a # second while fresh input tokens are single digits -- so it stays logarithmic. use_log = true @@ -47,7 +43,7 @@ stats_use_log = false # Legends sit top-right on both graphs, out of the way of the rising edge. legend_position = "top-right" -[styles.claude] +[styles.agent] # Model families in fixed order: Opus, Sonnet, Haiku, Fable, Other. These are mon's # built-in defaults, repeated here so they are easy to retune. They are picked for a dark # surface and checked for a lightness band, a chroma floor, adjacent colour-blind @@ -63,22 +59,16 @@ widget_title = { colour = "#d95926", bold = true } graph_colour = "#6a6a6a" legend_text = { colour = "#b4b4b4" } -[styles.tables] -headers = { colour = "#d95926", bold = true } - -# Layout: history on top, rate underneath, sessions at the bottom. -[[row]] -ratio = 44 -[[row.child]] -type = "claude_stats" - +# Layout: history on top, rate underneath. [[row]] -ratio = 32 +ratio = 58 [[row.child]] -type = "claude_graph" +type = "agent_stats" +source = "claude" +default = true [[row]] -ratio = 24 +ratio = 42 [[row.child]] -type = "claude" -default = true +type = "agent_graph" +source = "claude" diff --git a/sample_configs/codex_config.toml b/sample_configs/codex_config.toml new file mode 100644 index 00000000..f6e8ac04 --- /dev/null +++ b/sample_configs/codex_config.toml @@ -0,0 +1,53 @@ +# A Codex-CLI-only layout for mon. +# +# Same shape as `claude_config.toml`, pointed at `source = "codex"` instead. `harness-metrics` +# resolves `$CODEX_HOME` (or `~/.codex`) itself and reads Codex's rollout JSONL transcripts, +# so this config works as-is on any machine that has run the Codex CLI. +# +# There is no sessions table here, for either harness -- Codex writes no live PID registry, +# so there is nothing to build one from. Both harnesses get the same two widgets, read the +# same lagging, tailed-transcript way. +# +# Run it with: +# +# cargo run --release -- -C sample_configs/codex_config.toml --pixel_graphs kitty + +[flags] +rate = "1s" +retention = "1h" +default_time_value = "5m" +time_delta = "30s" +hide_time = true + +[agent] +use_log = true +stats_use_log = false +legend_position = "top-right" + +[styles.agent] +# Codex's model families in fixed order: GPT-5, GPT-4, the o-series (o1/o3/o4), Codex-branded +# minis, then Other. Same palette shape as the Claude config so the two configs read as a +# matched pair. +colours = ["#3987e5", "#d95926", "#199e70", "#c98500", "#d55181"] + +[styles.widgets] +border_colour = "#4a4a4a" +selected_border_colour = "#3987e5" +widget_title = { colour = "#3987e5", bold = true } + +[styles.graphs] +graph_colour = "#6a6a6a" +legend_text = { colour = "#b4b4b4" } + +[[row]] +ratio = 58 +[[row.child]] +type = "agent_stats" +source = "codex" +default = true + +[[row]] +ratio = 42 +[[row.child]] +type = "agent_graph" +source = "codex" diff --git a/sample_configs/pi_config.toml b/sample_configs/pi_config.toml new file mode 100644 index 00000000..f8345a84 --- /dev/null +++ b/sample_configs/pi_config.toml @@ -0,0 +1,52 @@ +# A pi-only layout for mon. +# +# Same shape as `claude_config.toml`/`codex_config.toml`, pointed at `source = "pi"`. +# `harness-metrics` resolves `$HOME/.pi/agent` itself and reads pi's session JSONL +# transcripts, so this config works as-is on any machine that has run pi. +# +# pi's "family" axis is provider (Anthropic, OpenAI, Google, Other), not model -- pi is +# multi-provider by design, and a per-model series would balloon as models rotate, so this +# stays coarser than Claude's or Codex's model-family axis on purpose. +# +# Run it with: +# +# cargo run --release -- -C sample_configs/pi_config.toml --pixel_graphs kitty + +[flags] +rate = "1s" +retention = "1h" +default_time_value = "5m" +time_delta = "30s" +hide_time = true + +[agent] +use_log = true +stats_use_log = false +legend_position = "top-right" + +[styles.agent] +# pi's providers in fixed order: Anthropic, OpenAI, Google, then Other. One fewer entry +# than the Claude/Codex configs since pi only has four buckets, not five. +colours = ["#3987e5", "#d95926", "#199e70", "#d55181"] + +[styles.widgets] +border_colour = "#4a4a4a" +selected_border_colour = "#199e70" +widget_title = { colour = "#199e70", bold = true } + +[styles.graphs] +graph_colour = "#6a6a6a" +legend_text = { colour = "#b4b4b4" } + +[[row]] +ratio = 58 +[[row.child]] +type = "agent_stats" +source = "pi" +default = true + +[[row]] +ratio = 42 +[[row.child]] +type = "agent_graph" +source = "pi" diff --git a/schema/nightly/bottom.json b/schema/nightly/bottom.json index 7d8598e5..475ae7cd 100644 --- a/schema/nightly/bottom.json +++ b/schema/nightly/bottom.json @@ -5,10 +5,10 @@ "description": "https://bottom.pages.dev/nightly/configuration/config-file/", "type": "object", "properties": { - "claude": { + "agent": { "anyOf": [ { - "$ref": "#/$defs/ClaudeConfig" + "$ref": "#/$defs/AgentConfig" }, { "type": "null" @@ -136,6 +136,49 @@ } }, "$defs": { + "AgentConfig": { + "description": "Agent widget configuration.\n\nCovers the token-rate graph (`agent_graph`) and the token-history stats graph\n(`agent_stats`). Both widgets are harness-agnostic: which harness (or harnesses) a\nparticular widget instance reads is set per-instance via the `source` field on that\nwidget's layout entry (`[[row.child]]`), not here -- `source = \"claude\" | \"codex\" |\n\"pi\" | \"all\"`.", + "type": "object", + "properties": { + "legend_position": { + "description": "Where to position the graph legend within the widget.", + "type": [ + "string", + "null" + ] + }, + "stats_use_log": { + "description": "Whether the token-history stats graph uses a logarithmic y-axis. Defaults to false.\n\nOff by default, unlike `use_log`. A bucketed total spans a far narrower range than\nan instantaneous rate -- a busy minute and a quiet one differ by a factor of ten,\nnot by five orders of magnitude -- and stacked bands only sum to the total on a\nlinear axis.", + "type": [ + "boolean", + "null" + ] + }, + "use_log": { + "description": "Whether the token-rate graph uses a logarithmic y-axis. Defaults to true.\n\nOn by default because the series genuinely span orders of magnitude: cache reads run\ninto the millions of tokens per second while fresh input tokens are single digits,\nand on a linear axis everything but the largest series sits flat on the floor.", + "type": [ + "boolean", + "null" + ] + } + } + }, + "AgentStyle": { + "description": "Styling specific to the agent token-rate and token-history graph widgets.\n\nThe `colours` list is indexed by whichever grouping is active for a given widget\ninstance: model-family labels (e.g. Opus, Sonnet, Haiku, Fable, Other) when that\nwidget's `source` names a single harness, or harness labels (`Claude`, `Codex`, `Pi`)\nwhen `source = \"all\"`.", + "type": "object", + "properties": { + "colours": { + "description": "Colour of each series' graph line, indexed by draw order (see the struct docs for\nwhat \"draw order\" means for a given widget's `source`).", + "type": [ + "array", + "null" + ], + "items": { + "$ref": "#/$defs/ColourStr" + } + } + } + }, "BatteryStyle": { "description": "Styling specific to the battery widget.", "type": "object", @@ -175,42 +218,6 @@ } } }, - "ClaudeConfig": { - "description": "Claude widget configuration.\n\nCovers both the sessions table (`claude`) and the token-rate graph (`claude_graph`).", - "type": "object", - "properties": { - "legend_position": { - "description": "Where to position the graph legend within the widget.", - "type": [ - "string", - "null" - ] - }, - "use_log": { - "description": "Whether the token-rate graph uses a logarithmic y-axis. Defaults to true.\n\nOn by default because the series genuinely span orders of magnitude: cache reads run\ninto the millions of tokens per second while fresh input tokens are single digits,\nand on a linear axis everything but the largest series sits flat on the floor.", - "type": [ - "boolean", - "null" - ] - } - } - }, - "ClaudeStyle": { - "description": "Styling specific to the Claude token-rate graph widget.", - "type": "object", - "properties": { - "colours": { - "description": "Colour of each model family's graph line. Read in family order: Opus, Sonnet, Haiku,\nFable, then Other.", - "type": [ - "array", - "null" - ], - "items": { - "$ref": "#/$defs/ColourStr" - } - } - } - }, "ColourStr": { "type": "string" }, @@ -512,6 +519,13 @@ "maximum": 65535, "minimum": 0 }, + "source": { + "description": "Which harness (or harnesses) an `agent_graph`/`agent_stats` widget reads.\n`\"claude\"` | `\"codex\"` | `\"pi\"` | `\"all\"`. Ignored by every other widget type.\nDefaults to `\"claude\"` when omitted.", + "type": [ + "string", + "null" + ] + }, "type": { "type": "string" } @@ -1478,22 +1492,22 @@ "description": "Style-related configs.", "type": "object", "properties": { - "battery": { - "description": "Styling for the battery widget.", + "agent": { + "description": "Styling for the agent token-rate and token-history graph widgets.", "anyOf": [ { - "$ref": "#/$defs/BatteryStyle" + "$ref": "#/$defs/AgentStyle" }, { "type": "null" } ] }, - "claude": { - "description": "Styling for the Claude token-rate graph widget.", + "battery": { + "description": "Styling for the battery widget.", "anyOf": [ { - "$ref": "#/$defs/ClaudeStyle" + "$ref": "#/$defs/BatteryStyle" }, { "type": "null" diff --git a/scripts/schema_gen/Cargo.lock b/scripts/schema_gen/Cargo.lock index 2e9e6693..69ef1159 100644 --- a/scripts/schema_gen/Cargo.lock +++ b/scripts/schema_gen/Cargo.lock @@ -255,12 +255,12 @@ dependencies = [ "clap_complete_fig", "clap_complete_nushell", "clap_mangen", - "claude-metrics", "concat-string", "core-foundation", "crossterm", "ctrlc", "dirs", + "harness-metrics", "humantime", "image", "indexmap", @@ -455,15 +455,6 @@ dependencies = [ "roff", ] -[[package]] -name = "claude-metrics" -version = "0.1.0" -dependencies = [ - "libc", - "serde", - "serde_json", -] - [[package]] name = "color_quant" version = "1.1.0" @@ -955,6 +946,14 @@ dependencies = [ "zerocopy", ] +[[package]] +name = "harness-metrics" +version = "0.1.0" +dependencies = [ + "serde", + "serde_json", +] + [[package]] name = "hashbrown" version = "0.16.1" diff --git a/scripts/smoke b/scripts/smoke index 773369db..338b3875 100755 --- a/scripts/smoke +++ b/scripts/smoke @@ -12,7 +12,7 @@ cargo fmt --all -- --check # NOTE(redd): plain `-D warnings`, not `-W clippy::pedantic`. Upstream bottom is not # pedantic-clean and never claimed to be, so a workspace-wide pedantic run buries real # findings under thousands of upstream nits. Pedantic is enabled per-crate instead, in -# `crates/claude-metrics/Cargo.toml`. +# `crates/harness-metrics/Cargo.toml`. step "cargo clippy --all-targets --all-features --workspace -- -D warnings" cargo clippy --all-targets --all-features --workspace -- -D warnings diff --git a/src/app.rs b/src/app.rs index 90fad822..710dea69 100644 --- a/src/app.rs +++ b/src/app.rs @@ -88,7 +88,7 @@ pub struct AppConfigFields { pub temperature_legend_position: Option, pub disk_io_legend_position: Option, pub power_legend_position: Option, - pub claude_legend_position: Option, + pub agent_legend_position: Option, pub disk_show_unmounted: bool, pub disk_io_graph_show_unmounted: bool, } @@ -184,12 +184,6 @@ impl App { disk.set_table_data(data_source); } } - - for claude in self.states.claude_state.widget_states.values_mut() { - if claude.force_update_data { - claude.set_table_data(&data_source.claude_sessions); - } - } } pub fn reset(&mut self) { @@ -236,7 +230,7 @@ impl App { widget_state.graph.state_mut().reset_zoom(); } - for widget_state in self.states.claude_graph_state.widget_states.values_mut() { + for widget_state in self.states.agent_graph_state.widget_states.values_mut() { widget_state.graph.state_mut().reset_zoom(); } } @@ -2053,10 +2047,10 @@ impl App { { Some(widget_state.graph.state_mut()) } - BottomWidgetType::ClaudeGraph + BottomWidgetType::AgentGraph if let Some(widget_state) = self .states - .claude_graph_state + .agent_graph_state .get_mut_widget_state(self.current_widget.widget_id) => { Some(widget_state.graph.state_mut()) diff --git a/src/app/data/store.rs b/src/app/data/store.rs index 161514e1..cfaa6121 100644 --- a/src/app/data/store.rs +++ b/src/app/data/store.rs @@ -93,16 +93,16 @@ pub struct InnerData { prev_io: FxHashMap<(String, String), (u64, u64)>, pub(crate) disk_harvest: Vec, pub(crate) temp_data: Vec, - pub(crate) claude_sessions: Vec, - pub(crate) claude_rate_limits: Option, - /// Tokens per family over a rolling window, oldest bucket first. Empty unless a widget - /// that draws it is on screen. - pub(crate) claude_history: Vec, - /// Families that contributed anything in that window, in a stable draw order. - pub(crate) claude_history_families: Vec, - /// Cumulative per-family totals from the previous tick, so the graph can difference - /// them into a rate. Keyed by family label. - prev_claude_tokens: FxHashMap, + /// Windowed history per source (`"claude"`, `"codex"`, `"pi"`, `"all"`), oldest bucket + /// first. A source's entry is only present once a harvest has actually carried history + /// for it. + pub(crate) agent_history: std::collections::HashMap<&'static str, Vec>, + /// Labels that contributed anything in that source's window, in a stable draw order. + pub(crate) agent_history_families: std::collections::HashMap<&'static str, Vec<&'static str>>, + /// Cumulative per-label totals from the previous tick, so the graph can difference them + /// into a rate. Keys are already namespaced `"{source_key}::{label}"`; one shared map + /// is fine since the namespacing already keeps sources apart. + prev_agent_tokens: FxHashMap, #[cfg(feature = "battery")] pub(crate) battery_harvest: Vec, @@ -126,11 +126,9 @@ impl Default for InnerData { prev_io: FxHashMap::default(), disk_harvest: Vec::default(), temp_data: Vec::default(), - claude_sessions: Vec::default(), - claude_rate_limits: None, - claude_history: Vec::default(), - claude_history_families: Vec::default(), - prev_claude_tokens: FxHashMap::default(), + agent_history: std::collections::HashMap::default(), + agent_history_families: std::collections::HashMap::default(), + prev_agent_tokens: FxHashMap::default(), #[cfg(feature = "battery")] battery_harvest: Vec::default(), #[cfg(feature = "zfs")] @@ -239,28 +237,31 @@ impl InnerData { } } - if used_widgets.use_claude - && let Some(claude) = data.claude + if used_widgets.use_agent + && let Some(agent) = &data.agent { let elapsed = harvested_time .duration_since(self.last_update_time) .as_secs_f64(); - self.time_series_data.update_claude_tokens( - &claude.totals, - &mut self.prev_claude_tokens, - elapsed, - ); - - self.claude_sessions = claude.sessions; - self.claude_rate_limits = claude.rate_limits; - - // Only overwrite when a harvest actually carried history. A harvest with the - // stats widget off returns none, and clobbering the window with that would - // blank the graph on any tick where the widget happened to be off screen. - if !claude.history.is_empty() { - self.claude_history = claude.history; - self.claude_history_families = claude.history_families; + for source in &agent.sources { + self.time_series_data.update_agent_tokens( + source.key, + &source.cumulative_totals, + &mut self.prev_agent_tokens, + elapsed, + ); + + // Only overwrite when a harvest actually carried history. A harvest with + // the stats widget off returns none, and clobbering the window with that + // would blank the graph on any tick where the widget happened to be off + // screen. + if !source.buckets.is_empty() { + self.agent_history + .insert(source.key, source.buckets.clone()); + self.agent_history_families + .insert(source.key, source.families.clone()); + } } } diff --git a/src/app/data/time_series.rs b/src/app/data/time_series.rs index 687be08c..fe4a7047 100644 --- a/src/app/data/time_series.rs +++ b/src/app/data/time_series.rs @@ -76,8 +76,11 @@ pub struct TimeSeriesData { /// Power draw in Watts, keyed by [`PowerChannel::label`]. pub power: HashMap, - /// Claude token throughput in tokens/second, keyed by model family label. - pub claude_tokens: HashMap, + /// Agent token throughput in tokens/second, keyed by `"{source_key}::{label}"` so two + /// widget instances reading different sources never collide on a label spelled the + /// same way in two backends (every backend has its own `Other` catch-all, for + /// instance). + pub agent_tokens: HashMap, /// Channels that have reported a nonzero reading at least once this run. /// @@ -292,34 +295,45 @@ impl TimeSeriesData { } } - /// Update the Claude token-rate series from cumulative per-family totals. + /// Update one source's agent token-rate series from its cumulative per-label totals. /// /// The collector reports running totals, but a graph of a monotonically climbing line /// says nothing useful. This differences them against the previous sample to get a /// rate. `previous` is owned by the caller so it survives across ticks. - pub fn update_claude_tokens( - &mut self, totals: &[(claude_metrics::ModelFamily, claude_metrics::TokenTotals)], + /// + /// Keys are namespaced `"{source_key}::{label}"` so two widget instances reading + /// different sources never collide on a label spelled the same way in two backends + /// (every backend has its own `Other` catch-all, for instance). + pub fn update_agent_tokens( + &mut self, source_key: &str, totals: &[(&'static str, harness_metrics::TokenTotals)], previous: &mut HashMap, elapsed_secs: f64, ) { + let namespaced = |label: &str| format!("{source_key}::{label}"); + // A zero or negative interval would divide by zero. It happens on the very first // tick, where there is no previous sample to difference against anyway. if elapsed_secs <= 0.0 { - for (family, totals) in totals { - previous.insert(family.label().to_owned(), totals.total()); + for (label, totals) in totals { + previous.insert(namespaced(label), totals.total()); } return; } - let mut not_visited: HashSet = self.claude_tokens.keys().cloned().collect(); + let mut not_visited: HashSet = self + .agent_tokens + .keys() + .filter(|key| key.starts_with(&format!("{source_key}::"))) + .cloned() + .collect(); - for (family, totals) in totals { - let label = family.label().to_owned(); - not_visited.remove(&label); + for (label, totals) in totals { + let key = namespaced(label); + not_visited.remove(&key); let current = totals.total(); - let entry = self.claude_tokens.entry(label.clone()).or_default(); + let entry = self.agent_tokens.entry(key.clone()).or_default(); - match previous.insert(label, current) { + match previous.insert(key, current) { // Totals only ever climb, but a session ending removes its contribution, so // guard the subtraction rather than assuming. Some(before) => { @@ -331,9 +345,9 @@ impl TimeSeriesData { } } - // A family that stopped being used keeps its place in the legend, with a gap. - for label in not_visited { - if let Some(entry) = self.claude_tokens.get_mut(&label) { + // A label that stopped being used keeps its place in the legend, with a gap. + for key in not_visited { + if let Some(entry) = self.agent_tokens.get_mut(&key) { entry.insert_break(); } } @@ -476,7 +490,7 @@ impl TimeSeriesData { } }); - self.claude_tokens.retain(|_, data| { + self.agent_tokens.retain(|_, data| { let _ = data.prune(end); if data.no_elements() { diff --git a/src/app/layout_manager.rs b/src/app/layout_manager.rs index 077222ef..65a3d9ff 100644 --- a/src/app/layout_manager.rs +++ b/src/app/layout_manager.rs @@ -844,6 +844,11 @@ pub struct BottomWidget { /// The value is the direction to bounce, as well as the parent offset. pub parent_reflector: Option<(WidgetDirection, u64)>, + /// Which harness (or harnesses) an `agent_graph`/`agent_stats` widget instance reads. + /// Raw string from the layout config (`"claude"` | `"codex"` | `"pi"` | `"all"`), + /// parsed downstream. Ignored by every other widget type. + pub source: Option, + /// Top left corner when drawn, for mouse click detection. (x, y) /// /// TODO: Replace this with just an Option for top + bottom. @@ -871,6 +876,7 @@ impl BottomWidget { top_left_corner: None, bottom_right_corner: None, ratio_override: None, + source: None, } } @@ -922,6 +928,11 @@ impl BottomWidget { self.parent_reflector = parent_reflector; self } + + pub fn source(mut self, source: Option) -> Self { + self.source = source; + self + } } #[derive(Debug, Clone, Eq, PartialEq, Hash, Default)] @@ -940,9 +951,8 @@ pub enum BottomWidgetType { Disk, DiskIoGraph, Power, - Claude, - ClaudeGraph, - ClaudeStats, + AgentGraph, + AgentStats, BasicCpu, BasicMem, BasicNet, @@ -960,7 +970,7 @@ impl BottomWidgetType { use BottomWidgetType::*; matches!( self, - Cpu | Net | Mem | TempGraph | DiskIoGraph | Power | ClaudeGraph | ClaudeStats + Cpu | Net | Mem | TempGraph | DiskIoGraph | Power | AgentGraph | AgentStats ) } @@ -979,9 +989,8 @@ impl BottomWidgetType { // title, not a compile error. Upstream's `DiskIoGraph` is missing for exactly // that reason -- left alone here to keep this diff to the power widget. Power => "Power", - Claude => "Claude", - ClaudeGraph => "Claude Tokens", - ClaudeStats => "Claude Stats", + AgentGraph => "Agent Tokens", + AgentStats => "Agent Stats", _ => "", } } @@ -1002,9 +1011,8 @@ impl std::str::FromStr for BottomWidgetType { "disk" => Ok(BottomWidgetType::Disk), "disk_io_graph" => Ok(BottomWidgetType::DiskIoGraph), "power" => Ok(BottomWidgetType::Power), - "claude" => Ok(BottomWidgetType::Claude), - "claude_graph" => Ok(BottomWidgetType::ClaudeGraph), - "claude_stats" => Ok(BottomWidgetType::ClaudeStats), + "agent_graph" => Ok(BottomWidgetType::AgentGraph), + "agent_stats" => Ok(BottomWidgetType::AgentStats), "empty" => Ok(BottomWidgetType::Empty), #[cfg(feature = "battery")] "battery" | "batt" => Ok(BottomWidgetType::Battery), @@ -1034,11 +1042,9 @@ Supported widget names: +--------------------------------+ | power | +--------------------------------+ -| claude | -+--------------------------------+ -| claude_graph | +| agent_graph | +--------------------------------+ -| claude_stats | +| agent_stats | +--------------------------------+ | batt, battery | +--------------------------------+ @@ -1072,11 +1078,9 @@ Supported widget names: +--------------------------------+ | power | +--------------------------------+ -| claude | -+--------------------------------+ -| claude_graph | +| agent_graph | +--------------------------------+ -| claude_stats | +| agent_stats | +--------------------------------+ | empty | +--------------------------------+ @@ -1102,10 +1106,11 @@ pub struct UsedWidgets { pub use_disk_io_graph: bool, pub use_battery: bool, pub use_power: bool, - pub use_claude: bool, - /// Whether any widget needs the rolling token history, which is the only part of a - /// Claude harvest that reads transcripts outside the live sessions. - pub use_claude_stats: bool, + pub use_agent: bool, + /// Whether any widget needs the rolling token history, which every `agent_stats` + /// widget always does, and which every harness now reads the exact same lagging + /// tailed-transcript way as the rate graph. + pub use_agent_stats: bool, } #[cfg(test)] @@ -1137,57 +1142,42 @@ mod added_widget_tests { /// There are two such tables, picked by `cfg(feature = "battery")`, so only one is /// compiled into any given build -- this checks whichever one is active, and CI builds /// both feature configurations. - /// Same three catch-all sites, for both Claude widgets. + /// Same catch-all sites, for both agent widgets. #[test] - fn claude_widgets_are_registered_everywhere_a_catch_all_would_hide() { - let table: BottomWidgetType = "claude".parse().expect("`claude` must parse"); - let graph: BottomWidgetType = "claude_graph".parse().expect("`claude_graph` must parse"); + fn agent_widgets_are_registered_everywhere_a_catch_all_would_hide() { + let graph: BottomWidgetType = "agent_graph".parse().expect("`agent_graph` must parse"); + let stats: BottomWidgetType = "agent_stats".parse().expect("`agent_stats` must parse"); - assert_eq!(table, BottomWidgetType::Claude); - assert_eq!(graph, BottomWidgetType::ClaudeGraph); + assert_eq!(graph, BottomWidgetType::AgentGraph); + assert_eq!(stats, BottomWidgetType::AgentStats); - assert!( - !table.is_widget_graph(), - "the sessions table is a table, not a graph -- it must not get timeseries state" - ); assert!( graph.is_widget_graph(), "the token-rate graph must count as a graph or it gets no timeseries state" ); - - let stats: BottomWidgetType = "claude_stats".parse().expect("`claude_stats` must parse"); - assert_eq!(stats, BottomWidgetType::ClaudeStats); assert!( stats.is_widget_graph(), "the stats graph must count as a graph or it gets no timeseries state" ); - assert_eq!(table.get_pretty_name(), "Claude"); - assert_eq!(graph.get_pretty_name(), "Claude Tokens"); - assert_eq!(stats.get_pretty_name(), "Claude Stats"); + assert_eq!(graph.get_pretty_name(), "Agent Tokens"); + assert_eq!(stats.get_pretty_name(), "Agent Stats"); } #[test] - fn claude_widgets_are_listed_in_the_help_table() { + fn agent_widgets_are_listed_in_the_help_table() { let err = "definitely_not_a_widget" .parse::() .expect_err("an unknown widget name must be an error") .to_string(); - for name in ["claude_graph", "claude_stats"] { + for name in ["agent_graph", "agent_stats"] { assert_eq!( err.matches(name).count(), 1, "the widget-name table must list `{name}`. Error was:\n{err}" ); } - - // `claude` also appears inside `claude_graph` and `claude_stats`, hence three. - assert_eq!( - err.matches("claude").count(), - 3, - "the widget-name table must list `claude`. Error was:\n{err}" - ); } #[test] diff --git a/src/app/states.rs b/src/app/states.rs index cb7669a2..84307b35 100644 --- a/src/app/states.rs +++ b/src/app/states.rs @@ -5,8 +5,8 @@ use crate::{ constants, utils::input::InputFieldState, widgets::{ - BatteryWidgetState, ClaudeGraphWidgetState, ClaudeStatsWidgetState, ClaudeWidgetState, - CpuWidgetState, DiskIoGraphWidgetState, DiskTableWidget, MemWidgetState, NetWidgetState, + AgentGraphWidgetState, AgentStatsWidgetState, BatteryWidgetState, CpuWidgetState, + DiskIoGraphWidgetState, DiskTableWidget, MemWidgetState, NetWidgetState, PowerGraphWidgetState, ProcWidgetState, TempGraphWidgetState, TempWidgetState, query::ProcessQuery, }, @@ -22,9 +22,8 @@ pub struct AppWidgetStates { pub disk_state: DiskState, pub disk_io_graph_state: DiskIoGraphStates, pub power_graph_state: PowerGraphStates, - pub claude_state: ClaudeState, - pub claude_graph_state: ClaudeGraphStates, - pub claude_stats_state: ClaudeStatsStates, + pub agent_graph_state: AgentGraphStates, + pub agent_stats_state: AgentStatsStates, pub battery_state: AppBatteryState, pub basic_table_widget_state: Option, } @@ -229,47 +228,32 @@ impl PowerGraphStates { } } -/// Holds per-widget state for all Claude sessions table instances in the layout. -pub struct ClaudeState { - pub widget_states: HashMap, +/// Holds per-widget state for all agent token-rate graph instances in the layout. +pub struct AgentGraphStates { + pub widget_states: HashMap, } -impl ClaudeState { - pub fn init(widget_states: HashMap) -> Self { - ClaudeState { widget_states } +impl AgentGraphStates { + pub fn init(widget_states: HashMap) -> Self { + AgentGraphStates { widget_states } } - pub fn get_mut_widget_state(&mut self, widget_id: u64) -> Option<&mut ClaudeWidgetState> { + pub fn get_mut_widget_state(&mut self, widget_id: u64) -> Option<&mut AgentGraphWidgetState> { self.widget_states.get_mut(&widget_id) } } -/// Holds per-widget state for all Claude token-rate graph instances in the layout. -pub struct ClaudeGraphStates { - pub widget_states: HashMap, +/// Holds per-widget state for all agent token-history stats instances in the layout. +pub struct AgentStatsStates { + pub widget_states: HashMap, } -impl ClaudeGraphStates { - pub fn init(widget_states: HashMap) -> Self { - ClaudeGraphStates { widget_states } +impl AgentStatsStates { + pub fn init(widget_states: HashMap) -> Self { + AgentStatsStates { widget_states } } - pub fn get_mut_widget_state(&mut self, widget_id: u64) -> Option<&mut ClaudeGraphWidgetState> { - self.widget_states.get_mut(&widget_id) - } -} - -/// Holds per-widget state for all Claude token-history stats instances in the layout. -pub struct ClaudeStatsStates { - pub widget_states: HashMap, -} - -impl ClaudeStatsStates { - pub fn init(widget_states: HashMap) -> Self { - ClaudeStatsStates { widget_states } - } - - pub fn get_mut_widget_state(&mut self, widget_id: u64) -> Option<&mut ClaudeStatsWidgetState> { + pub fn get_mut_widget_state(&mut self, widget_id: u64) -> Option<&mut AgentStatsWidgetState> { self.widget_states.get_mut(&widget_id) } } diff --git a/src/canvas.rs b/src/canvas.rs index 71c61e23..1ed78202 100644 --- a/src/canvas.rs +++ b/src/canvas.rs @@ -371,19 +371,13 @@ impl Painter { rect[0], app_state.current_widget.widget_id, ), - Claude => self.draw_claude_table( + AgentStats => self.draw_agent_stats( f, app_state, rect[0], app_state.current_widget.widget_id, ), - ClaudeStats => self.draw_claude_stats( - f, - app_state, - rect[0], - app_state.current_widget.widget_id, - ), - ClaudeGraph => self.draw_claude_graph( + AgentGraph => self.draw_agent_graph( f, app_state, rect[0], @@ -515,14 +509,11 @@ impl Painter { Power => { self.draw_power_graph(f, app_state, vertical_chunks[3], widget_id) } - Claude => { - self.draw_claude_table(f, app_state, vertical_chunks[3], widget_id) - } - ClaudeGraph => { - self.draw_claude_graph(f, app_state, vertical_chunks[3], widget_id) + AgentGraph => { + self.draw_agent_graph(f, app_state, vertical_chunks[3], widget_id) } - ClaudeStats => { - self.draw_claude_stats(f, app_state, vertical_chunks[3], widget_id) + AgentStats => { + self.draw_agent_stats(f, app_state, vertical_chunks[3], widget_id) } _ => {} } @@ -615,13 +606,8 @@ impl Painter { self.draw_disk_io_graph(f, app_state, *draw_loc, widget.widget_id) } Power => self.draw_power_graph(f, app_state, *draw_loc, widget.widget_id), - Claude => self.draw_claude_table(f, app_state, *draw_loc, widget.widget_id), - ClaudeGraph => { - self.draw_claude_graph(f, app_state, *draw_loc, widget.widget_id) - } - ClaudeStats => { - self.draw_claude_stats(f, app_state, *draw_loc, widget.widget_id) - } + AgentGraph => self.draw_agent_graph(f, app_state, *draw_loc, widget.widget_id), + AgentStats => self.draw_agent_stats(f, app_state, *draw_loc, widget.widget_id), _ => {} } } diff --git a/src/canvas/widgets/claude_graph.rs b/src/canvas/widgets/agent_graph.rs similarity index 82% rename from src/canvas/widgets/claude_graph.rs rename to src/canvas/widgets/agent_graph.rs index d4e17a06..1a46f8e2 100644 --- a/src/canvas/widgets/claude_graph.rs +++ b/src/canvas/widgets/agent_graph.rs @@ -1,6 +1,5 @@ use std::borrow::Cow; -use claude_metrics::ModelFamily; use ratatui::{ Frame, layout::{Constraint, Rect}, @@ -17,32 +16,22 @@ use crate::{ components::time_series::GraphDrawCtx, }; -/// Model families in a fixed draw order. -/// -/// The order is also the index into the theme's colour list. Fixed rather than sorted or -/// rank-ordered on purpose: a family that goes quiet and drops out must not repaint the -/// ones that remain. -const FAMILIES: [ModelFamily; 5] = [ - ModelFamily::Opus, - ModelFamily::Sonnet, - ModelFamily::Haiku, - ModelFamily::Fable, - ModelFamily::Other, -]; - impl Painter { - pub fn draw_claude_graph( + pub fn draw_agent_graph( &self, f: &mut Frame<'_>, app_state: &mut App, draw_loc: Rect, widget_id: u64, ) { if let Some(widget_state) = app_state .states - .claude_graph_state + .agent_graph_state .get_mut_widget_state(widget_id) { let shared_data = app_state.data_store.get_data(); - let token_data = &shared_data.time_series_data.claude_tokens; + let token_data = &shared_data.time_series_data.agent_tokens; let times = &shared_data.time_series_data.time; + let source_key = widget_state.source.key(); + let key_for = |label: &str| format!("{source_key}::{label}"); + let border_style = self.get_border_style(widget_id, app_state.current_widget.widget_id); let graph_state = widget_state.graph.state_mut(); let hide_x_labels = should_hide_x_label( @@ -53,17 +42,19 @@ impl Painter { ); let use_log = widget_state.use_log; + let labels = &widget_state.family_labels; - // Only families that have actually produced a series. A family nobody has used + // Only labels that have actually produced a series. A label nobody has used // would otherwise take a legend slot to say nothing. - let present: Vec = FAMILIES - .into_iter() - .filter(|family| token_data.contains_key(family.label())) + let present: Vec<&'static str> = labels + .iter() + .copied() + .filter(|label| token_data.contains_key(&key_for(label))) .collect(); let visible = present .iter() - .filter_map(|family| token_data.get(family.label())); + .filter_map(|label| token_data.get(&key_for(label))); let y_max = widget_state.graph.y_max(visible, times); let (adjusted_y_max, y_labels) = if use_log { @@ -72,16 +63,16 @@ impl Painter { adjust_tokens_linear(y_max) }; - let colours = &self.styles.claude_colour_styles; + let colours = &self.styles.agent_colour_styles; let graph_data: Vec> = present .iter() - .filter_map(|family| { - let values = token_data.get(family.label())?; + .filter_map(|label| { + let values = token_data.get(&key_for(label))?; - // Index by the family's fixed position, not by its position among the - // present ones, so colours stay put as families come and go. - let index = FAMILIES.iter().position(|f| f == family).unwrap_or(0); + // Index by the label's fixed position, not by its position among the + // present ones, so colours stay put as labels come and go. + let index = labels.iter().position(|l| l == label).unwrap_or(0); let style = if colours.is_empty() { Style::default() } else { @@ -92,7 +83,7 @@ impl Painter { Some( GraphData::default() - .name(format!("{:<7}{}", family.label(), format_rate(rate)).into()) + .name(format!("{:<7}{}", label, format_rate(rate)).into()) .style(style) .time(times) .values(values), @@ -113,11 +104,13 @@ impl Painter { ChartScaling::Linear }; + let title = format!(" {} Tokens ", widget_state.source.title_word()); + widget_state.graph.draw( f, draw_loc, GraphDrawCtx { - title: " Claude Tokens ".into(), + title: title.into(), border_style, title_style: self.styles.widget_title_style, graph_style: self.styles.graph_style, @@ -127,7 +120,7 @@ impl Painter { hide_x_labels, is_selected: app_state.current_widget.widget_id == widget_id, is_expanded: app_state.is_expanded, - legend_position: app_state.app_config_fields.claude_legend_position, + legend_position: app_state.app_config_fields.agent_legend_position, legend_constraints: Some(legend_constraints), pixel_renderer: self.pixel_renderer(), last_time: times.last().copied(), @@ -286,18 +279,4 @@ mod tests { assert!(ceiling > 0.0, "a zero ceiling would collapse the y-axis"); assert_eq!(labels.len(), 3); } - - #[test] - fn families_keep_a_stable_colour_index() { - // The whole point of indexing by family rather than by draw position: a family - // going quiet must not recolour the ones still on screen. - assert_eq!( - FAMILIES.iter().position(|f| *f == ModelFamily::Opus), - Some(0) - ); - assert_eq!( - FAMILIES.iter().position(|f| *f == ModelFamily::Fable), - Some(3) - ); - } } diff --git a/src/canvas/widgets/claude_stats.rs b/src/canvas/widgets/agent_stats.rs similarity index 78% rename from src/canvas/widgets/claude_stats.rs rename to src/canvas/widgets/agent_stats.rs index 4e9e995b..426a4918 100644 --- a/src/canvas/widgets/claude_stats.rs +++ b/src/canvas/widgets/agent_stats.rs @@ -1,4 +1,4 @@ -//! Drawing the Claude token-history stats graph. +//! Drawing the agent token-history stats graph. //! //! Claude Code's own `/status` screen bars token spend by day. This is the same idea at a //! far finer grain -- one bucket a minute over the last hour -- so the shape of a working @@ -9,7 +9,7 @@ use std::{ time::{Duration, Instant}, }; -use claude_metrics::{Bucket, ModelFamily}; +use harness_metrics::Bucket; use ratatui::{ Frame, layout::{Constraint, Rect}, @@ -28,17 +28,26 @@ use crate::{ }; impl Painter { - pub fn draw_claude_stats( + pub fn draw_agent_stats( &self, f: &mut Frame<'_>, app_state: &mut App, draw_loc: Rect, widget_id: u64, ) { if let Some(widget_state) = app_state .states - .claude_stats_state + .agent_stats_state .get_mut_widget_state(widget_id) { let shared_data = app_state.data_store.get_data(); - let buckets = shared_data.claude_history.clone(); - let families = shared_data.claude_history_families.clone(); + let source_key = widget_state.source.key(); + let buckets = shared_data + .agent_history + .get(source_key) + .cloned() + .unwrap_or_default(); + let families = shared_data + .agent_history_families + .get(source_key) + .cloned() + .unwrap_or_default(); let border_style = self.get_border_style(widget_id, app_state.current_widget.widget_id); let graph_state = widget_state.graph.state_mut(); @@ -51,19 +60,19 @@ impl Painter { let use_log = widget_state.use_log; - let stacked = Stacked::build(&buckets, &families); + let stacked = Stacked::build(&buckets, &families, &widget_state.family_labels); let (y_max, y_labels) = if use_log { adjust_tokens_log(stacked.peak) } else { adjust_tokens_linear(stacked.peak) }; - let colours = &self.styles.claude_colour_styles; + let colours = &self.styles.agent_colour_styles; // Largest running total first. Each fill reaches down to the axis, so drawing // in descending order lets every later band paint over the lower part of the - // one before it, leaving exactly one visible band per family. Drawing the other - // way round would bury every family under the tallest one's fill. + // one before it, leaving exactly one visible band per label. Drawing the other + // way round would bury every label under the tallest one's fill. let graph_data: Vec> = stacked .layers .iter() @@ -79,7 +88,7 @@ impl Painter { .name( format!( "{:<7}{}", - layer.family.label(), + layer.label, format_tokens_padded(layer.own_total) ) .into(), @@ -109,11 +118,13 @@ impl Painter { ChartScaling::Linear }; + let title = format!(" {} Stats ", widget_state.source.title_word()); + widget_state.graph.draw( f, draw_loc, GraphDrawCtx { - title: " Claude Stats ".into(), + title: title.into(), border_style, title_style: self.styles.widget_title_style, graph_style: self.styles.graph_style, @@ -123,7 +134,7 @@ impl Painter { hide_x_labels, is_selected: app_state.current_widget.widget_id == widget_id, is_expanded: app_state.is_expanded, - legend_position: app_state.app_config_fields.claude_legend_position, + legend_position: app_state.app_config_fields.agent_legend_position, legend_constraints: Some(legend_constraints), pixel_renderer: self.pixel_renderer(), last_time: stacked.times.last().copied(), @@ -146,16 +157,16 @@ impl Painter { } } -/// One family's band: its running total, and what it contributed on its own. +/// One label's band: its running total, and what it contributed on its own. struct Layer { - family: ModelFamily, - /// This family's tokens plus every family below it, per bucket. Plotting running totals + label: &'static str, + /// This label's tokens plus every label below it, per bucket. Plotting running totals /// rather than raw values is what turns overlapping filled areas into a stack. running: ChunkedData, - /// This family's own tokens across the window, for the legend. + /// This label's own tokens across the window, for the legend. own_total: u64, - /// Position in [`ModelFamily::ALL`], so a family that goes quiet cannot repaint the - /// colours of the ones that remain. + /// Position in the widget's fixed label order, so a label that goes quiet cannot + /// repaint the colours of the ones that remain. colour_index: usize, } @@ -176,7 +187,11 @@ impl Stacked { /// done exactly, since `Instant` has no epoch -- the buckets are laid out backwards /// from now at their known spacing. They are contiguous and evenly spaced by /// construction, so this reproduces the real geometry without needing the mapping. - fn build(buckets: &[Bucket], families: &[ModelFamily]) -> Self { + /// + /// `present` is the subset of `order` that actually contributed in this window; + /// `order` is the widget's whole fixed label set, used only to keep colour indices + /// stable as labels come and go. + fn build(buckets: &[Bucket], present: &[&'static str], order: &[&'static str]) -> Self { let now = Instant::now(); let times: Vec = (0..buckets.len()) @@ -188,14 +203,14 @@ impl Stacked { .collect(); let mut running_totals = vec![0.0f64; buckets.len()]; - let mut layers = Vec::with_capacity(families.len()); + let mut layers = Vec::with_capacity(present.len()); - for family in families { + for label in present { let mut running = ChunkedData::default(); let mut own_total = 0u64; for (index, bucket) in buckets.iter().enumerate() { - let own = bucket.total_for(*family); + let own = bucket.total_for(label); own_total = own_total.saturating_add(own); running_totals[index] += own as f64; @@ -203,12 +218,12 @@ impl Stacked { } layers.push(Layer { - family: *family, + label, running, own_total, - colour_index: ModelFamily::ALL + colour_index: order .iter() - .position(|candidate| candidate == family) + .position(|candidate| candidate == label) .unwrap_or(0), }); } @@ -223,7 +238,7 @@ impl Stacked { } } -/// Bucket width, matching what the collector asks `claude-metrics` for. +/// Bucket width, matching what the collector asks `harness-metrics` for. const BUCKET: Duration = Duration::from_secs(60); /// Render a token count compactly. @@ -290,18 +305,18 @@ fn adjust_tokens_log(peak: f64) -> (f64, Vec) { #[cfg(test)] mod tests { - use claude_metrics::TokenTotals; + use harness_metrics::TokenTotals; use super::*; - fn bucket(start_ms: u64, entries: &[(ModelFamily, u64)]) -> Bucket { + fn bucket(start_ms: u64, entries: &[(&'static str, u64)]) -> Bucket { Bucket { start_ms, totals: entries .iter() - .map(|(family, tokens)| { + .map(|(label, tokens)| { ( - *family, + *label, TokenTotals { input: *tokens, ..TokenTotals::default() @@ -319,17 +334,15 @@ mod tests { #[test] fn layers_carry_running_totals_so_filled_areas_stack() { // Each band is drawn as a fill down to the axis, so the value plotted has to be the - // running total. Plotting raw values would draw every family from the axis and bury + // running total. Plotting raw values would draw every label from the axis and bury // all but the largest. let buckets = [ - bucket(0, &[(ModelFamily::Opus, 100), (ModelFamily::Sonnet, 20)]), - bucket( - 60_000, - &[(ModelFamily::Opus, 50), (ModelFamily::Sonnet, 30)], - ), + bucket(0, &[("Opus", 100), ("Sonnet", 20)]), + bucket(60_000, &[("Opus", 50), ("Sonnet", 30)]), ]; - let stacked = Stacked::build(&buckets, &[ModelFamily::Opus, ModelFamily::Sonnet]); + let order = ["Opus", "Sonnet"]; + let stacked = Stacked::build(&buckets, &order, &order); assert_eq!(values(&stacked.layers[0].running), vec![100.0, 50.0]); assert_eq!( @@ -342,25 +355,24 @@ mod tests { #[test] fn the_peak_is_the_tallest_stack_not_the_tallest_family() { // The y-axis has to clear the top of the stack. Scaling to the largest single - // family would clip the band above it straight off the plot. - let buckets = [bucket( - 0, - &[(ModelFamily::Opus, 100), (ModelFamily::Sonnet, 60)], - )]; + // label would clip the band above it straight off the plot. + let buckets = [bucket(0, &[("Opus", 100), ("Sonnet", 60)])]; - let stacked = Stacked::build(&buckets, &[ModelFamily::Opus, ModelFamily::Sonnet]); + let order = ["Opus", "Sonnet"]; + let stacked = Stacked::build(&buckets, &order, &order); assert_eq!(stacked.peak, 160.0); } #[test] - fn a_family_keeps_its_colour_as_others_come_and_go() { - // Indexing by position among the *present* families would repaint every remaining - // band the moment a quiet family dropped out of the window. - let buckets = [bucket(0, &[(ModelFamily::Sonnet, 10)])]; + fn a_label_keeps_its_colour_as_others_come_and_go() { + // Indexing by position among the *present* labels would repaint every remaining + // band the moment a quiet label dropped out of the window. + let buckets = [bucket(0, &[("Sonnet", 10)])]; + let order = ["Opus", "Sonnet"]; - let alone = Stacked::build(&buckets, &[ModelFamily::Sonnet]); - let together = Stacked::build(&buckets, &[ModelFamily::Opus, ModelFamily::Sonnet]); + let alone = Stacked::build(&buckets, &["Sonnet"], &order); + let together = Stacked::build(&buckets, &order, &order); assert_eq!(alone.layers[0].colour_index, 1); assert_eq!(together.layers[1].colour_index, 1); @@ -370,7 +382,7 @@ mod tests { fn times_run_oldest_to_newest_one_bucket_apart() { let buckets = [bucket(0, &[]), bucket(60_000, &[]), bucket(120_000, &[])]; - let stacked = Stacked::build(&buckets, &[]); + let stacked = Stacked::build(&buckets, &[], &[]); assert_eq!(stacked.times.len(), 3); assert!( @@ -382,8 +394,8 @@ mod tests { #[test] fn an_empty_window_still_gets_a_usable_axis() { - // A layout can hold this widget on a machine that has never run Claude Code. - let stacked = Stacked::build(&[], &[]); + // A layout can hold this widget on a machine that has never run any harness. + let stacked = Stacked::build(&[], &[], &[]); assert_eq!(stacked.peak, 0.0); diff --git a/src/canvas/widgets/claude_table.rs b/src/canvas/widgets/claude_table.rs deleted file mode 100644 index 3dc1e4d3..00000000 --- a/src/canvas/widgets/claude_table.rs +++ /dev/null @@ -1,39 +0,0 @@ -use ratatui::{Frame, layout::Rect}; - -use crate::{ - app, - canvas::{ - Painter, - components::data_table::{DrawInfo, SelectionState}, - }, -}; - -impl Painter { - pub fn draw_claude_table( - &self, f: &mut Frame<'_>, app_state: &mut app::App, draw_loc: Rect, widget_id: u64, - ) { - let recalculate_column_widths = app_state.should_get_widget_bounds(); - if let Some(claude_widget_state) = app_state - .states - .claude_state - .widget_states - .get_mut(&widget_id) - { - let is_on_widget = app_state.current_widget.widget_id == widget_id; - - let draw_info = DrawInfo { - loc: draw_loc, - force_redraw: app_state.is_force_redraw, - recalculate_column_widths, - selection_state: SelectionState::new(app_state.is_expanded, is_on_widget), - }; - - claude_widget_state.table.draw( - f, - &draw_info, - app_state.widget_map.get_mut(&widget_id), - self, - ); - } - } -} diff --git a/src/canvas/widgets/mod.rs b/src/canvas/widgets/mod.rs index ef7d1e70..ade81e7b 100644 --- a/src/canvas/widgets/mod.rs +++ b/src/canvas/widgets/mod.rs @@ -1,8 +1,7 @@ use crate::{collection::network::NetworkHarvest, utils::data_units::convert_bytes}; -pub mod claude_graph; -pub mod claude_stats; -pub mod claude_table; +pub mod agent_graph; +pub mod agent_stats; pub mod cpu_basic; pub mod cpu_graph; pub mod disk_io_graph; diff --git a/src/collection.rs b/src/collection.rs index 7e6c6abd..f01e4763 100644 --- a/src/collection.rs +++ b/src/collection.rs @@ -14,9 +14,9 @@ mod linux { pub mod utils; } +pub mod agent; #[cfg(feature = "battery")] pub mod batteries; -pub mod claude; pub mod cpu; pub mod disks; pub mod error; @@ -58,7 +58,7 @@ pub struct Data { #[cfg(feature = "battery")] pub list_of_batteries: Option>, pub power: Option, - pub claude: Option, + pub agent: Option, #[cfg(feature = "zfs")] pub arc: Option, #[cfg(feature = "gpu")] @@ -83,7 +83,7 @@ impl Default for Data { #[cfg(feature = "battery")] list_of_batteries: None, power: None, - claude: None, + agent: None, #[cfg(feature = "zfs")] arc: None, #[cfg(feature = "gpu")] @@ -197,8 +197,8 @@ pub struct DataCollector { #[cfg(target_os = "macos")] power_interval_ms: u32, - /// Built lazily, the first time a layout actually asks for Claude data. - claude_collector: Option, + /// Built lazily, the first time a layout actually asks for agent data. + agent_collector: Option, #[cfg(unix)] user_table: processes::UserTable, @@ -252,7 +252,7 @@ impl DataCollector { power_sampler: None, #[cfg(target_os = "macos")] power_interval_ms: 1000, - claude_collector: None, + agent_collector: None, filters, #[cfg(unix)] user_table: Default::default(), @@ -427,7 +427,7 @@ impl DataCollector { #[cfg(target_os = "macos")] self.update_power(); - self.update_claude(); + self.update_agent(); #[cfg(feature = "gpu")] self.update_gpus(); @@ -636,18 +636,18 @@ impl DataCollector { self.data.power = sampler.latest().cloned(); } - /// Update Claude Code metrics. + /// Update coding-agent metrics. #[inline] - fn update_claude(&mut self) { - if !self.widgets_to_harvest.use_claude { + fn update_agent(&mut self) { + if !self.widgets_to_harvest.use_agent { return; } let collector = self - .claude_collector - .get_or_insert_with(claude::ClaudeCollector::new); + .agent_collector + .get_or_insert_with(agent::AgentCollector::new); - self.data.claude = collector.harvest(self.widgets_to_harvest.use_claude_stats); + self.data.agent = Some(collector.harvest(self.widgets_to_harvest.use_agent_stats)); } /// Update battery information. diff --git a/src/collection/agent.rs b/src/collection/agent.rs new file mode 100644 index 00000000..c1cda72b --- /dev/null +++ b/src/collection/agent.rs @@ -0,0 +1,243 @@ +//! Live coding-agent metrics, read via the `harness-metrics` crate. +//! +//! Every harness (Claude Code, Codex CLI, pi) is read through the exact same lagging, +//! tailed-transcript ledger -- there is no live-session/PID-registry machinery for any of +//! them, Claude included. A refresh re-reads only what each backend's transcripts have +//! appended since the last tick, so a steady-state tick is cheap. +//! +//! A single layout can mix widget instances with *different* `source`s (e.g. one +//! `agent_graph` reading `source = "claude"` next to another reading `source = "all"`), so +//! this collector always computes all four possible source views -- claude, codex, pi, and +//! the merged "all" -- every tick whenever any agent widget is on screen. That is cheap: +//! `HarnessLedger::refresh` is mtime-gated per file, so paying for four views costs one +//! extra `merge_all` call, not four tree walks. + +use std::time::Duration; + +use harness_metrics::{Bucket, Harness, HarnessLedger, TokenTotals, merge_all}; + +use crate::options::OptionError; + +/// How far back the stats history reaches. +const HISTORY_WINDOW: Duration = Duration::from_secs(60 * 60); + +/// How finely that window is divided. +/// +/// A minute is deliberately far finer than Claude Code's own `/status`, which bars by day. +/// At this width an hour is sixty points, which is enough to see the shape of a working +/// session rather than a single flat bar. +const HISTORY_BUCKET: Duration = Duration::from_secs(60); + +/// Which harness (or harnesses) an `agent_graph`/`agent_stats` widget instance reads. +#[derive(Copy, Clone, Debug, PartialEq, Eq)] +pub enum AgentSource { + One(Harness), + All, +} + +impl AgentSource { + /// The stable key used to namespace this source's time-series entries, matching + /// [`AgentSourceData::key`]. + #[must_use] + pub fn key(self) -> &'static str { + match self { + AgentSource::One(Harness::Claude) => "claude", + AgentSource::One(Harness::Codex) => "codex", + AgentSource::One(Harness::Pi) => "pi", + AgentSource::All => "all", + } + } + + /// A short display word for widget titles, e.g. `" {} Tokens "`. + #[must_use] + pub fn title_word(self) -> &'static str { + match self { + AgentSource::One(harness) => harness.label(), + AgentSource::All => "All", + } + } + + /// The fixed, ordered set of series labels this source draws, used as both draw order + /// and colour index. + #[must_use] + pub fn family_labels(self) -> Vec<&'static str> { + match self { + AgentSource::One(Harness::Claude) => harness_metrics::claude::ClaudeFamily::ALL + .iter() + .map(|f| f.label()) + .collect(), + AgentSource::One(Harness::Codex) => harness_metrics::codex::CodexFamily::ALL + .iter() + .map(|f| f.label()) + .collect(), + AgentSource::One(Harness::Pi) => harness_metrics::pi::PiFamily::ALL + .iter() + .map(|f| f.label()) + .collect(), + AgentSource::All => Harness::ALL.iter().map(|h| h.label()).collect(), + } + } +} + +impl Default for AgentSource { + fn default() -> Self { + AgentSource::One(Harness::Claude) + } +} + +impl std::str::FromStr for AgentSource { + type Err = OptionError; + + fn from_str(s: &str) -> Result { + match s.to_ascii_lowercase().as_str() { + "claude" => Ok(AgentSource::One(Harness::Claude)), + "codex" => Ok(AgentSource::One(Harness::Codex)), + "pi" => Ok(AgentSource::One(Harness::Pi)), + "all" => Ok(AgentSource::All), + other => Err(OptionError::config(format!( + "'{other}' is not a valid agent source (expected claude, codex, pi, or all)" + ))), + } + } +} + +/// One source's worth of data for a tick: either a single harness or the "all" merge. +/// +/// `key` is the stable string a widget's parsed `source` matches against ("claude", +/// "codex", "pi", "all") -- used to namespace time-series keys so two widget instances +/// with different `source`s never collide on a family label that happens to be spelled +/// the same way in two backends (e.g. every backend has an `Other` bucket). +#[derive(Clone, Debug, Default)] +pub struct AgentSourceData { + pub key: &'static str, + pub cumulative_totals: Vec<(&'static str, TokenTotals)>, + pub buckets: Vec, + pub families: Vec<&'static str>, +} + +/// A snapshot of everything the agent widgets draw, one entry per possible `source`. +#[derive(Clone, Debug, Default)] +pub struct AgentData { + pub sources: Vec, +} + +/// Owns each harness's reader and turns a refresh into an [`AgentData`] snapshot. +pub struct AgentCollector { + claude: Option, + codex: Option, + pi: Option, +} + +impl std::fmt::Debug for AgentCollector { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("AgentCollector") + .field("claude", &self.claude.is_some()) + .field("codex", &self.codex.is_some()) + .field("pi", &self.pi.is_some()) + .finish() + } +} + +impl Default for AgentCollector { + fn default() -> Self { + Self::new() + } +} + +impl AgentCollector { + /// Build a collector over each harness's default root. + /// + /// A harness with no derivable root (no `$HOME`, and for Codex, no `$CODEX_HOME` + /// either) yields `None` for that harness and every harvest reports it empty. That is a + /// normal state on a machine that has never run that harness, not an error. + pub fn new() -> Self { + Self { + claude: HarnessLedger::new(Harness::Claude, HISTORY_WINDOW, HISTORY_BUCKET), + codex: HarnessLedger::new(Harness::Codex, HISTORY_WINDOW, HISTORY_BUCKET), + pi: HarnessLedger::new(Harness::Pi, HISTORY_WINDOW, HISTORY_BUCKET), + } + } + + /// Re-read and snapshot every possible source. + /// + /// `want_history` gates the windowed bucket scan -- the rate graph needs + /// `cumulative_totals` regardless of whether a stats widget is on screen, but the + /// bucketed history is only worth computing when something will draw it. + pub fn harvest(&mut self, want_history: bool) -> AgentData { + for ledger in [&mut self.claude, &mut self.codex, &mut self.pi] + .into_iter() + .flatten() + { + ledger.refresh(); + } + + let claude_data = source_data("claude", self.claude.as_ref(), want_history); + let codex_data = source_data("codex", self.codex.as_ref(), want_history); + let pi_data = source_data("pi", self.pi.as_ref(), want_history); + + let ledgers: Vec<&HarnessLedger> = [&self.claude, &self.codex, &self.pi] + .into_iter() + .flatten() + .collect(); + let all_data = merged_source_data(&ledgers, want_history); + + AgentData { + sources: vec![claude_data, codex_data, pi_data, all_data], + } + } +} + +/// Build one harness's [`AgentSourceData`], or an empty one if that harness has no ledger. +fn source_data( + key: &'static str, ledger: Option<&HarnessLedger>, want_history: bool, +) -> AgentSourceData { + let Some(ledger) = ledger else { + return AgentSourceData { + key, + ..AgentSourceData::default() + }; + }; + + let cumulative_totals = ledger.cumulative_totals(); + let (buckets, families) = if want_history { + (ledger.buckets(), ledger.families_present()) + } else { + (Vec::new(), Vec::new()) + }; + + AgentSourceData { + key, + cumulative_totals, + buckets, + families, + } +} + +/// Build the merged "all harnesses" [`AgentSourceData`]. +fn merged_source_data(ledgers: &[&HarnessLedger], want_history: bool) -> AgentSourceData { + let merged = merge_all(ledgers); + + let (buckets, families) = if want_history { + let order: Vec<&'static str> = Harness::ALL.iter().map(|h| h.label()).collect(); + let families: Vec<&'static str> = order + .into_iter() + .filter(|label| { + merged + .buckets + .iter() + .any(|bucket| bucket.total_for(label) > 0) + }) + .collect(); + + (merged.buckets, families) + } else { + (Vec::new(), Vec::new()) + }; + + AgentSourceData { + key: "all", + cumulative_totals: merged.cumulative_totals, + buckets, + families, + } +} diff --git a/src/collection/claude.rs b/src/collection/claude.rs deleted file mode 100644 index 2af8e7cf..00000000 --- a/src/collection/claude.rs +++ /dev/null @@ -1,197 +0,0 @@ -//! Live Claude Code metrics, read via the `claude-metrics` crate. -//! -//! Unlike the power sampler this needs no thread of its own. `ClaudeMetrics::refresh` -//! re-reads a handful of small registry files and consumes only what the transcripts have -//! appended since the last call, so a steady-state tick is cheap. -//! -//! The exception is the *first* tick for a session, which reads its whole transcript. On a -//! long session that is a one-time cost of a few hundred milliseconds on the collection -//! thread. Accepted rather than threaded: it happens once per session, and a thread would -//! buy nothing on every subsequent tick. - -use std::time::Duration; - -use claude_metrics::{Bucket, ClaudeMetrics, ModelFamily, Statusline, TokenHistory, TokenTotals}; - -/// One live session, flattened for display. -#[derive(Clone, Debug, Default)] -pub struct ClaudeSession { - /// Session UUID. - pub id: String, - /// Human-facing session name. - pub name: String, - /// Working directory. - pub cwd: String, - /// `busy`, `idle`, and others. - pub status: String, - /// tmux pane id, e.g. `%28`. - pub tmux_pane: String, - /// Model family currently in use, if the statusline cache says. - pub model: Option, - /// Every token this session has spent, cache included. - pub tokens: u64, - /// Subagent messages produced. - pub agents: u64, - /// Cost so far in USD, from the statusline cache. - pub cost_usd: Option, - /// Context-window occupancy as a percentage. - pub context_percent: Option, - /// Most recent turn duration, in milliseconds. - pub last_turn_ms: Option, -} - -/// A snapshot of everything the Claude widgets draw. -#[derive(Clone, Debug, Default)] -pub struct ClaudeData { - /// Live sessions, newest first. - pub sessions: Vec, - /// Cumulative token totals across every live session, by family. - pub totals: Vec<(ModelFamily, TokenTotals)>, - /// Rate limits, taken from whichever session reported most recently. - /// - /// These are account-wide rather than per-session, so any session's copy will do. - pub rate_limits: Option, - /// Tokens per model family over a rolling window, oldest bucket first. - /// - /// Empty unless a widget that draws it is on screen -- see - /// [`crate::app::layout_manager::UsedWidgets::use_claude_stats`]. This is read from the - /// whole `~/.claude/projects` tree rather than from the live sessions, so it keeps the - /// tokens of sessions that have since exited. - pub history: Vec, - /// Families that contributed anything in the window, in a stable draw order. - pub history_families: Vec, -} - -/// Account-wide rate-limit consumption. -#[derive(Clone, Copy, Debug, Default)] -pub struct RateLimits { - /// Rolling 5-hour bucket, as a percentage. - pub five_hour_percent: f64, - /// Unix epoch seconds at which the 5-hour bucket resets. - pub five_hour_resets_at: u64, - /// Rolling 7-day bucket, as a percentage. - pub seven_day_percent: f64, - /// Unix epoch seconds at which the 7-day bucket resets. - pub seven_day_resets_at: u64, -} - -impl From<&Statusline> for RateLimits { - fn from(statusline: &Statusline) -> Self { - Self { - five_hour_percent: statusline.rate_limits.five_hour.used_percentage, - five_hour_resets_at: statusline.rate_limits.five_hour.resets_at, - seven_day_percent: statusline.rate_limits.seven_day.used_percentage, - seven_day_resets_at: statusline.rate_limits.seven_day.resets_at, - } - } -} - -/// How far back the stats history reaches. -const HISTORY_WINDOW: Duration = Duration::from_secs(60 * 60); - -/// How finely that window is divided. -/// -/// A minute is deliberately far finer than Claude Code's own `/status`, which bars by day. -/// At this width an hour is sixty points, which is enough to see the shape of a working -/// session rather than a single flat bar. -const HISTORY_BUCKET: Duration = Duration::from_secs(60); - -/// Owns the reader and turns a refresh into a [`ClaudeData`] snapshot. -#[derive(Debug)] -pub struct ClaudeCollector { - metrics: Option, - /// Built on first use, so a layout without the stats widget never walks the tree. - history: Option, -} - -impl Default for ClaudeCollector { - fn default() -> Self { - Self::new() - } -} - -impl ClaudeCollector { - /// Build a collector over `$HOME/.claude`. - /// - /// With no home directory there is nothing to read and every harvest is empty. That is - /// a normal state on a machine that has never run Claude Code, not an error. - pub fn new() -> Self { - Self { - metrics: ClaudeMetrics::with_default_root(), - history: None, - } - } - - /// Re-read and snapshot. - /// - /// `want_history` gates the rolling-window scan, which is the only part of a harvest - /// that touches transcripts outside the live sessions. A layout with no stats widget - /// should not pay for it. - pub fn harvest(&mut self, want_history: bool) -> Option { - let metrics = self.metrics.as_mut()?; - metrics.refresh(); - - let (history, history_families) = if want_history { - let root = metrics.root().to_path_buf(); - let history = self - .history - .get_or_insert_with(|| TokenHistory::new(root, HISTORY_WINDOW, HISTORY_BUCKET)); - - history.refresh(); - (history.buckets(), history.families()) - } else { - (Vec::new(), Vec::new()) - }; - - let mut rate_limits = None; - - let sessions = metrics - .sessions() - .iter() - .filter_map(|session| { - let id = session.session_id.clone()?; - let statusline = metrics.statusline_for(&id); - - if let Some(statusline) = &statusline - && rate_limits.is_none() - { - rate_limits = Some(RateLimits::from(statusline)); - } - - let tokens = metrics - .totals_for(&id) - .iter() - .map(|(_, totals)| totals.total()) - .sum(); - - Some(ClaudeSession { - name: session.name.clone().unwrap_or_else(|| id.clone()), - cwd: session.cwd.clone().unwrap_or_default(), - status: session.status.clone().unwrap_or_default(), - tmux_pane: session.tmux_pane().unwrap_or_default().to_owned(), - // Prefer the raw model id over the display name -- it is what the - // family table is built to fold. - model: statusline - .as_ref() - .and_then(|s| s.model_id().map(ModelFamily::from_id)), - tokens, - agents: metrics.subagent_messages(&id), - cost_usd: statusline.as_ref().map(|s| s.cost.total_cost_usd), - context_percent: statusline - .as_ref() - .map(|s| s.context_window.used_percentage), - last_turn_ms: metrics.last_turn_duration_ms(&id), - id, - }) - }) - .collect(); - - Some(ClaudeData { - sessions, - totals: metrics.totals_by_model(), - rate_limits, - history, - history_families, - }) - } -} diff --git a/src/lib.rs b/src/lib.rs index 8a5f041a..1e55ddae 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -469,12 +469,6 @@ pub fn start_bottom(enable_error_hook: &mut bool) -> anyhow::Result<()> { } } - if app.used_widgets.use_claude { - for claude in app.states.claude_state.widget_states.values_mut() { - claude.force_data_update(); - } - } - app.update_data(); try_drawing(&mut terminal, &mut app, &mut painter)?; } diff --git a/src/options.rs b/src/options.rs index 1300f1a8..520799dd 100644 --- a/src/options.rs +++ b/src/options.rs @@ -35,6 +35,7 @@ use self::{ use crate::{ app::{filter::Filter, layout_manager::*, *}, canvas::components::time_series::LegendPosition, + collection::agent::AgentSource, components::time_series::TimeseriesConfig, constants::*, utils::data_units::DataUnit, @@ -145,10 +146,10 @@ macro_rules! config_or { /// The default config file sub-path. const DEFAULT_CONFIG_FILE_LOCATION: &str = "bottom/bottom.toml"; -/// The window the Claude stats graph plots, matching what the collector asks -/// `claude-metrics` for. Asking the graph for more would draw dead space before the oldest -/// bucket the history actually holds. -const CLAUDE_STATS_WINDOW_MS: u64 = 60 * 60 * 1000; +/// The window the agent stats graph plots, matching what the collector asks for. +/// Asking the graph for more would draw dead space before the oldest bucket the history +/// actually holds. +const AGENT_STATS_WINDOW_MS: u64 = 60 * 60 * 1000; /// Returns the config path to use. If `override_config_path` is specified, then /// we will use that. If not, then return the "default" config path, which is: @@ -387,9 +388,8 @@ pub(crate) fn init_app(args: BottomArgs, config: Config) -> Result<(App, BottomL let mut disk_state_map: FxHashMap = FxHashMap::default(); let mut disk_io_graph_state_map: FxHashMap = FxHashMap::default(); let mut power_graph_state_map: FxHashMap = FxHashMap::default(); - let mut claude_state_map: FxHashMap = FxHashMap::default(); - let mut claude_graph_state_map: FxHashMap = FxHashMap::default(); - let mut claude_stats_state_map: FxHashMap = FxHashMap::default(); + let mut agent_graph_state_map: FxHashMap = FxHashMap::default(); + let mut agent_stats_state_map: FxHashMap = FxHashMap::default(); let mut battery_state_map: FxHashMap = FxHashMap::default(); let autohide_timer = if autohide_time { @@ -442,7 +442,7 @@ pub(crate) fn init_app(args: BottomArgs, config: Config) -> Result<(App, BottomL let temperature_legend_position = get_temperature_legend_position(config)?; let disk_io_legend_position = get_disk_io_legend_position(config)?; let power_legend_position = get_power_legend_position(config)?; - let claude_legend_position = get_claude_legend_position(config)?; + let agent_legend_position = get_agent_legend_position(config)?; let disk_io_name_filter = match &config.disk_io_graph { Some(cfg) => get_ignore_list(&cfg.name_filter) .context("Update 'disk_io_graph.name_filter' in your config file")?, @@ -551,7 +551,7 @@ pub(crate) fn init_app(args: BottomArgs, config: Config) -> Result<(App, BottomL temperature_legend_position, disk_io_legend_position, power_legend_position, - claude_legend_position, + agent_legend_position, disk_show_unmounted, disk_io_graph_show_unmounted, }; @@ -773,50 +773,68 @@ pub(crate) fn init_app(args: BottomArgs, config: Config) -> Result<(App, BottomL ), ); } - Claude => { - claude_state_map.insert( - widget.widget_id, - ClaudeWidgetState::new(&app_config_fields, &styling), - ); - } - ClaudeGraph => { + AgentGraph => { // Log by default: cache reads run millions of tokens per second // while fresh input tokens are single digits, so a linear axis // flattens everything but the largest series onto the floor. let use_log = config - .claude + .agent .as_ref() .and_then(|c| c.use_log) .unwrap_or(true); - claude_graph_state_map.insert( + let source = widget + .source + .as_deref() + .map(str::parse::) + .transpose()? + .unwrap_or_default(); + + agent_graph_state_map.insert( widget.widget_id, - ClaudeGraphWidgetState::new(ts_config, autohide_timer, use_log), + AgentGraphWidgetState::new( + ts_config, + autohide_timer, + use_log, + source, + ), ); } - ClaudeStats => { + AgentStats => { // Linear by default, unlike the rate graph. A bucketed total // spans a far narrower range than an instantaneous rate, and // stacked bands only sum to the total on a linear axis. let use_log = config - .claude + .agent .as_ref() .and_then(|c| c.stats_use_log) .unwrap_or(false); + let source = widget + .source + .as_deref() + .map(str::parse::) + .transpose()? + .unwrap_or_default(); + // The window is the history's own, not the app-wide default: - // the buckets `claude-metrics` hands over cover exactly an - // hour, and a graph asking for more would draw dead space - // before the oldest bucket. + // the buckets the collector hands over cover exactly an hour, + // and a graph asking for more would draw dead space before the + // oldest bucket. let stats_config = TimeseriesConfig { - default_time_value: CLAUDE_STATS_WINDOW_MS, - retention_ms: CLAUDE_STATS_WINDOW_MS, + default_time_value: AGENT_STATS_WINDOW_MS, + retention_ms: AGENT_STATS_WINDOW_MS, ..ts_config }; - claude_stats_state_map.insert( + agent_stats_state_map.insert( widget.widget_id, - ClaudeStatsWidgetState::new(stats_config, autohide_timer, use_log), + AgentStatsWidgetState::new( + stats_config, + autohide_timer, + use_log, + source, + ), ); } Battery => { @@ -869,10 +887,8 @@ pub(crate) fn init_app(args: BottomArgs, config: Config) -> Result<(App, BottomL use_disk_io_graph: used_widget_set.contains(&DiskIoGraph), use_battery: used_widget_set.contains(&Battery), use_power: used_widget_set.contains(&Power), - use_claude: used_widget_set.contains(&Claude) - || used_widget_set.contains(&ClaudeGraph) - || used_widget_set.contains(&ClaudeStats), - use_claude_stats: used_widget_set.contains(&ClaudeStats), + use_agent: used_widget_set.contains(&AgentGraph) || used_widget_set.contains(&AgentStats), + use_agent_stats: used_widget_set.contains(&AgentStats), }; let (disk_name_filter, disk_mount_filter) = { @@ -914,9 +930,8 @@ pub(crate) fn init_app(args: BottomArgs, config: Config) -> Result<(App, BottomL disk_state: DiskState::init(disk_state_map), disk_io_graph_state: DiskIoGraphStates::init(disk_io_graph_state_map), power_graph_state: PowerGraphStates::init(power_graph_state_map), - claude_state: ClaudeState::init(claude_state_map), - claude_graph_state: ClaudeGraphStates::init(claude_graph_state_map), - claude_stats_state: ClaudeStatsStates::init(claude_stats_state_map), + agent_graph_state: AgentGraphStates::init(agent_graph_state_map), + agent_stats_state: AgentStatsStates::init(agent_stats_state_map), battery_state: AppBatteryState::init(battery_state_map), basic_table_widget_state, }; @@ -1587,15 +1602,15 @@ fn get_pixel_mode(args: &BottomArgs, config: &Config) -> OptionResult Ok(PixelMode::default()) } -fn get_claude_legend_position(config: &Config) -> OptionResult> { +fn get_agent_legend_position(config: &Config) -> OptionResult> { parse_legend_position( None, config - .claude + .agent .as_ref() .and_then(|settings| settings.legend_position.as_ref()), None, - "claude.legend_position", + "agent.legend_position", None, ) } diff --git a/src/options/config.rs b/src/options/config.rs index 4da24323..06c78159 100644 --- a/src/options/config.rs +++ b/src/options/config.rs @@ -1,4 +1,4 @@ -pub mod claude; +pub mod agent; pub mod cpu; pub mod disk; pub mod disk_io_graph; @@ -13,7 +13,7 @@ pub mod style; pub mod temperature; pub mod temperature_graph; -use claude::ClaudeConfig; +use agent::AgentConfig; use disk::DiskConfig; use disk_io_graph::DiskIoGraphConfig; use flags::GeneralConfig; @@ -40,7 +40,7 @@ pub struct Config { pub(crate) disk: Option, pub(crate) disk_io_graph: Option, pub(crate) power: Option, - pub(crate) claude: Option, + pub(crate) agent: Option, pub(crate) temperature: Option, pub(crate) temperature_graph: Option, #[serde(alias = "network")] diff --git a/src/options/config/claude.rs b/src/options/config/agent.rs similarity index 73% rename from src/options/config/claude.rs rename to src/options/config/agent.rs index 71f5158c..9bc4d689 100644 --- a/src/options/config/claude.rs +++ b/src/options/config/agent.rs @@ -1,13 +1,16 @@ use serde::Deserialize; -/// Claude widget configuration. +/// Agent widget configuration. /// -/// Covers the sessions table (`claude`), the token-rate graph (`claude_graph`), and the -/// token-history stats graph (`claude_stats`). +/// Covers the token-rate graph (`agent_graph`) and the token-history stats graph +/// (`agent_stats`). Both widgets are harness-agnostic: which harness (or harnesses) a +/// particular widget instance reads is set per-instance via the `source` field on that +/// widget's layout entry (`[[row.child]]`), not here -- `source = "claude" | "codex" | +/// "pi" | "all"`. #[derive(Clone, Debug, Default, Deserialize)] #[cfg_attr(feature = "generate_schema", derive(schemars::JsonSchema))] #[cfg_attr(test, serde(deny_unknown_fields), derive(PartialEq, Eq))] -pub(crate) struct ClaudeConfig { +pub(crate) struct AgentConfig { /// Whether the token-rate graph uses a logarithmic y-axis. Defaults to true. /// /// On by default because the series genuinely span orders of magnitude: cache reads run diff --git a/src/options/config/layout.rs b/src/options/config/layout.rs index 573d2004..32f04eac 100644 --- a/src/options/config/layout.rs +++ b/src/options/config/layout.rs @@ -115,10 +115,10 @@ impl Row { .total_col_row_ratio(2) .ratio(width_ratio) } - _ => BottomCol::new(vec![BottomColRow::new(vec![BottomWidget::new( - widget_type, - *iter_id, - )])]) + _ => BottomCol::new(vec![BottomColRow::new(vec![ + BottomWidget::new(widget_type, *iter_id) + .source(widget.source.clone()), + ])]) .ratio(width_ratio), }); } @@ -186,10 +186,10 @@ impl Row { total_col_row_ratio += col_row_height_ratio; col_row_children.push( - BottomColRow::new(vec![BottomWidget::new( - widget_type, - *iter_id, - )]) + BottomColRow::new(vec![ + BottomWidget::new(widget_type, *iter_id) + .source(widget.source.clone()), + ]) .ratio(col_row_height_ratio), ) } @@ -238,6 +238,10 @@ pub struct FinalWidget { #[serde(rename = "type")] pub widget_type: String, pub default: Option, + /// Which harness (or harnesses) an `agent_graph`/`agent_stats` widget reads. + /// `"claude"` | `"codex"` | `"pi"` | `"all"`. Ignored by every other widget type. + /// Defaults to `"claude"` when omitted. + pub source: Option, } #[cfg(test)] diff --git a/src/options/config/style.rs b/src/options/config/style.rs index af190403..c2153b3d 100644 --- a/src/options/config/style.rs +++ b/src/options/config/style.rs @@ -1,8 +1,8 @@ //! Config options around styling. +mod agent; mod battery; mod borders; -mod claude; mod cpu; mod disk_io_graph; mod graphs; @@ -17,8 +17,8 @@ mod widgets; use std::borrow::Cow; +use agent::AgentStyle; use battery::BatteryStyle; -use claude::ClaudeStyle; use cpu::CpuStyle; use disk_io_graph::DiskIoGraphStyle; use graphs::GraphStyle; @@ -96,8 +96,8 @@ pub(crate) struct StyleConfig { /// Styling for the power graph widget. pub(crate) power: Option, - /// Styling for the Claude token-rate graph widget. - pub(crate) claude: Option, + /// Styling for the agent token-rate and token-history graph widgets. + pub(crate) agent: Option, /// Styling for the temperature graph widget. pub(crate) temp_graph: Option, @@ -137,7 +137,7 @@ pub struct Styles { pub(crate) disk_io_read_colour_styles: Vec