hades-cli 0.2.0__tar.gz → 0.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. hades_cli-0.2.2/CONTEXT.md +17 -0
  2. {hades_cli-0.2.0 → hades_cli-0.2.2}/PKG-INFO +2 -2
  3. {hades_cli-0.2.0 → hades_cli-0.2.2}/ROADMAP.md +9 -3
  4. hades_cli-0.2.2/docs/adr/0001-token-category-breakdown-precomputed-at-index-time.md +7 -0
  5. hades_cli-0.2.2/docs/adr/0002-hades-serve-uses-stdlib-http-server.md +7 -0
  6. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/_version.py +2 -2
  7. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/cli.py +17 -0
  8. hades_cli-0.2.2/src/hades/commands/reindex.py +10 -0
  9. hades_cli-0.2.2/src/hades/commands/serve.py +117 -0
  10. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/config.py +1 -0
  11. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/db.py +8 -1
  12. hades_cli-0.2.2/src/hades/indexer.py +173 -0
  13. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/models.py +4 -0
  14. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/pricing.py +27 -0
  15. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/sources/base.py +3 -0
  16. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/sources/claude.py +9 -1
  17. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/sources/codex.py +21 -0
  18. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/sources/common.py +16 -0
  19. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/sources/cowork.py +10 -1
  20. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/sources/openclaw.py +21 -0
  21. hades_cli-0.2.2/src/hades/static/serve.html +389 -0
  22. {hades_cli-0.2.0 → hades_cli-0.2.2}/tests/test_codex.py +24 -2
  23. {hades_cli-0.2.0 → hades_cli-0.2.2}/tests/test_indexer.py +63 -3
  24. {hades_cli-0.2.0 → hades_cli-0.2.2}/tests/test_openclaw.py +15 -1
  25. hades_cli-0.2.2/tests/test_serve.py +85 -0
  26. {hades_cli-0.2.0 → hades_cli-0.2.2}/tests/test_tokens.py +20 -0
  27. hades_cli-0.2.0/src/hades/indexer.py +0 -104
  28. {hades_cli-0.2.0 → hades_cli-0.2.2}/.github/workflows/pylint.yml +0 -0
  29. {hades_cli-0.2.0 → hades_cli-0.2.2}/.github/workflows/python-publish.yml +0 -0
  30. {hades_cli-0.2.0 → hades_cli-0.2.2}/.gitignore +0 -0
  31. {hades_cli-0.2.0 → hades_cli-0.2.2}/.python-version +0 -0
  32. {hades_cli-0.2.0 → hades_cli-0.2.2}/README.md +0 -0
  33. {hades_cli-0.2.0 → hades_cli-0.2.2}/SECURITY.md +0 -0
  34. {hades_cli-0.2.0 → hades_cli-0.2.2}/docs/agents/domain.md +0 -0
  35. {hades_cli-0.2.0 → hades_cli-0.2.2}/docs/agents/issue-tracker.md +0 -0
  36. {hades_cli-0.2.0 → hades_cli-0.2.2}/docs/agents/triage-labels.md +0 -0
  37. {hades_cli-0.2.0 → hades_cli-0.2.2}/pyproject.toml +0 -0
  38. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/__init__.py +0 -0
  39. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/classify.py +0 -0
  40. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/commands/__init__.py +0 -0
  41. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/commands/archive.py +0 -0
  42. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/commands/attention.py +0 -0
  43. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/commands/config.py +0 -0
  44. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/commands/export.py +0 -0
  45. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/commands/list.py +0 -0
  46. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/commands/purge.py +0 -0
  47. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/commands/search.py +0 -0
  48. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/commands/show.py +0 -0
  49. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/commands/stats.py +0 -0
  50. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/commands/tools.py +0 -0
  51. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/commands/watch.py +0 -0
  52. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/console.py +0 -0
  53. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/hooks.py +0 -0
  54. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/process_checker.py +0 -0
  55. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/sources/__init__.py +0 -0
  56. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/sources/antigravity.py +0 -0
  57. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/sources/cursor.py +0 -0
  58. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/sources/gemini.py +0 -0
  59. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/transcript.py +0 -0
  60. {hades_cli-0.2.0 → hades_cli-0.2.2}/src/hades/waiting.py +0 -0
  61. {hades_cli-0.2.0 → hades_cli-0.2.2}/tests/__init__.py +0 -0
  62. {hades_cli-0.2.0 → hades_cli-0.2.2}/tests/conftest.py +0 -0
  63. {hades_cli-0.2.0 → hades_cli-0.2.2}/tests/test_antigravity.py +0 -0
  64. {hades_cli-0.2.0 → hades_cli-0.2.2}/tests/test_archive.py +0 -0
  65. {hades_cli-0.2.0 → hades_cli-0.2.2}/tests/test_classify.py +0 -0
  66. {hades_cli-0.2.0 → hades_cli-0.2.2}/tests/test_commands.py +0 -0
  67. {hades_cli-0.2.0 → hades_cli-0.2.2}/tests/test_config.py +0 -0
  68. {hades_cli-0.2.0 → hades_cli-0.2.2}/tests/test_cowork.py +0 -0
  69. {hades_cli-0.2.0 → hades_cli-0.2.2}/tests/test_cursor.py +0 -0
  70. {hades_cli-0.2.0 → hades_cli-0.2.2}/tests/test_db.py +0 -0
  71. {hades_cli-0.2.0 → hades_cli-0.2.2}/tests/test_hooks.py +0 -0
  72. {hades_cli-0.2.0 → hades_cli-0.2.2}/tests/test_notify.py +0 -0
  73. {hades_cli-0.2.0 → hades_cli-0.2.2}/tests/test_pricing.py +0 -0
  74. {hades_cli-0.2.0 → hades_cli-0.2.2}/tests/test_process_checker.py +0 -0
  75. {hades_cli-0.2.0 → hades_cli-0.2.2}/tests/test_second_pass.py +0 -0
  76. {hades_cli-0.2.0 → hades_cli-0.2.2}/tests/test_tools.py +0 -0
  77. {hades_cli-0.2.0 → hades_cli-0.2.2}/tests/test_transcript.py +0 -0
  78. {hades_cli-0.2.0 → hades_cli-0.2.2}/tests/test_waiting.py +0 -0
  79. {hades_cli-0.2.0 → hades_cli-0.2.2}/tests/test_watch.py +0 -0
  80. {hades_cli-0.2.0 → hades_cli-0.2.2}/uv.lock +0 -0
@@ -0,0 +1,17 @@
1
+ # hades
2
+
3
+ A local, zero-config observer over AI coding sessions across tools (Claude Code, Codex, Gemini, Cowork, Cursor, Antigravity, OpenClaw). It reads each tool's transcript files, indexes them into a local SQLite DB, and surfaces state — never launches or orchestrates sessions.
4
+
5
+ ## Language
6
+
7
+ **Source**:
8
+ A per-tool adapter (`BaseSource` subclass) that discovers a tool's transcript files on disk and parses each one into a `Session`. One source per external tool.
9
+ _Avoid_: Scanner, adapter, plugin (reserved for the future plugin-API vision in the v3 roadmap).
10
+
11
+ **Session**:
12
+ The indexed record of one AI coding conversation — one transcript file (or, for Claude, one file within a possibly-resumed chain) reduced to its metadata: tool, project, timing, status, token/cost totals. Not the transcript content itself, which stays on disk at `raw_path`.
13
+ _Avoid_: Conversation, chat, transcript (transcript is the raw file; Session is the indexed record of it).
14
+
15
+ **Token category**:
16
+ One of the four buckets Anthropic's (and Anthropic-shaped) `usage` block reports per turn: input tokens (fresh, uncached), output tokens (generated), cache-write tokens (`cache_creation_input_tokens`), cache-read tokens (`cache_read_input_tokens`). This is the finest-grained "what did the tokens go to" the transcript data actually supports — it is *not* a breakdown by content type (system prompt vs. tool call vs. tool result), which the usage API doesn't report per-block and which hades does not attempt to estimate.
17
+ _Avoid_: Token type, usage breakdown (breakdown is the aggregate view across categories; category is one bucket within it).
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: hades-cli
3
- Version: 0.2.0
3
+ Version: 0.2.2
4
4
  Summary: A local CLI for viewing, searching, and managing AI coding sessions across Claude Code, Codex, Gemini, and Cowork.
5
5
  Project-URL: Homepage, https://github.com/mnghn07/hades
6
6
  Project-URL: Repository, https://github.com/mnghn07/hades
@@ -33,13 +33,19 @@ Attention list becomes actionable instead of read-only. (Dropped `hades resume`
33
33
 
34
34
  - **More sources** — OpenCode, Copilot CLI, Aider still open. Each is ~80 lines given the `BaseSource` ABC; this is where the architecture pays off.
35
35
  - ~~**Cursor**~~ — done: `hades/sources/cursor.py` reads `~/.cursor/projects/*/agent-transcripts/*/*.jsonl` (the CLI agent's own transcripts, not the IDE's chat storage, which lives in a separate VSCode-style SQLite store and wasn't targeted). No `cwd` or per-turn timestamps in the format — `project_path` is decoded from the dash-encoded project dir name (same ambiguity as Claude's decoder for literal hyphens/dots), `last_active_at` uses file mtime (exact — each turn is appended live), `started_at` uses an embedded `<timestamp>` tag on the first user turn when present (~60% of turns have one) else file birthtime. No token/cost data in the format. "Running" status not yet wired into `process_checker._classify` — the live process name for the Cursor CLI agent is unconfirmed, so cursor sessions stay idle/ended. Verified against 428 real transcripts on a dev machine.
36
- - ~~**Per-session token count**~~ — done: `models.Session.token_count`, populated for Claude (summed input/output/cache tokens from each turn's `usage` block) and shown in `list`/`stats` (table + `--json`). Codex/gemini/cowork report 0 — no local sample data to confirm their usage field shape yet.
37
- - ~~**$ cost column**~~ — done: `hades/pricing.py` prices each turn by its own `usage.model` (prefix-matched against a small `$/1M` table for current Claude models; cache write/read derived as 1.25x/0.1x of input price per Anthropic's published multipliers). `models.Session.cost_usd` populated for Claude, summed per session, shown in `list`/`stats` (table + `--json`). Unrecognized models price at $0 rather than guessing. Still open: live-ticking token/cost in `watch`, and token/cost parsing for codex/gemini/cowork once their usage format is confirmed.
36
+ - ~~**Antigravity**~~ and ~~**OpenClaw**~~ — done (2026-08-10): `hades/sources/antigravity.py` and `hades/sources/openclaw.py`, wired into `sources/__init__.py` and `transcript.py`. Codex source extended with token count and cost extraction in the same pass.
37
+ - ~~**Per-session token count**~~ — done: `models.Session.token_count`, populated for Claude (summed input/output/cache tokens from each turn's `usage` block) and shown in `list`/`stats` (table + `--json`). Codex now extracts usage too (2026-08-10); gemini/cowork still report 0 — no local sample data to confirm their usage field shape yet.
38
+ - ~~**$ cost column**~~ — done: `hades/pricing.py` prices each turn by its own `usage.model` (prefix-matched against a small `$/1M` table for current Claude models; cache write/read derived as 1.25x/0.1x of input price per Anthropic's published multipliers). `models.Session.cost_usd` populated for Claude, summed per session, shown in `list`/`stats` (table + `--json`). Unrecognized models price at $0 rather than guessing. Still open: live-ticking token/cost in `watch`, and token/cost parsing for gemini/cowork once their usage format is confirmed.
39
+ - ~~**Per-tool enable/disable**~~ — done (2026-08-10): `hades setup` and `hades tools` commands (`src/hades/commands/tools.py`) let a user toggle which source scanners run; `config.py` persists the enabled-tools set, `indexer.py` and `transcript.py` respect it. Not originally scoped in this roadmap — added to let users opt out of scanning tools they don't use rather than eating the scan cost for every source unconditionally.
40
+ - ~~**Reindex/backfill mechanism**~~ — done (2026-08-31): per-source `parser_version` int (`sources/base.py`), self-healed on every `refresh_index()` pass regardless of mtime, plus a manual `hades reindex [--tool X] [--session ID]` escape hatch (`indexer.py`, `commands/reindex.py`). Also folds in archived sessions, previously invisible to any reparse path since their file moves out of `discover_root`. `ClaudeSource.PARSER_VERSION` bumped to 2 to backfill the 119 pre-existing sessions stuck at `cost_usd=0`. See [#5](https://github.com/mnghn07/hades/issues/5).
38
41
  - **Homebrew tap** — after PyPI validation (per PRD).
39
42
 
40
43
  ## v3 — The ambitious version
41
44
 
42
- - **`hades serve`** — localhost dashboard (sniffly-style) reading the same SQLite DB. The CLI stays the primary interface; the dashboard is a view, not a second product.
45
+ - **Token/cost breakdown** (in progress) — narrowed from the full dashboard idea below during grilling on [#1](https://github.com/mnghn07/hades/issues/1): category-level (input/output/cache-write/cache-read) breakdown by tool and project, delivered as `hades stats --breakdown` (CLI) and a single-purpose `hades serve` page — not the general dashboard. ADR-0001 (breakdown stored at index time) and ADR-0002 (stdlib `http.server`, no framework) settle the storage/serving approach; `docs/CONTEXT.md` has the glossary.
46
+ - ~~**`hades serve` dashboard**~~ — done (2026-08-31), closes [#4](https://github.com/mnghn07/hades/issues/4): the four category columns (`input_tokens`/`output_tokens`/`cache_write_tokens`/`cache_read_tokens`) added to `Session`/schema per ADR-0001, populated by Claude/Codex/Cowork/OpenClaw (Gemini/Cursor/Antigravity have no usage data to extract, same as `token_count` today); each source's `PARSER_VERSION` bumped so the reindex self-heal mechanism backfills existing history for free. `hades serve` (`commands/serve.py`) is a stdlib `ThreadingHTTPServer` serving one static page (`static/serve.html`) plus `/api/summary` and `/api/daily`, both scoped by a `?from=&to=` date range. Sniffly-style multi-page structure (overview / `#/tool/{name}` / `#/project/{name}`), a validated categorical palette and real bar-chart mark specs (dataviz skill), grouped-column token chart + stacked-column cost chart, all resolving #1's open "does v1 need a time-range/trend view" question — yes. Design exploration captured as `docs/prototypes/serve-dashboard.PROTOTYPE.html`.
47
+ - **`hades stats --breakdown` CLI** — still open, [#3](https://github.com/mnghn07/hades/issues/3) (CLI layout design not yet done).
48
+ - **`hades serve` (general dashboard)** — the original sniffly-style ambition below (attention feed, session list, live watch-style updates) stays explicitly out of scope for now; the existing CLI (`list`/`attention`/`watch`) already covers those views adequately. Revisit only if the narrower token/cost page above proves the dashboard surface is worth widening.
43
49
  - **Daemon mode** — `watch` decoupled from a terminal pane; notifications fire even with no terminal open.
44
50
  - **Plugin API** — a source is a pip entry point; the community adds tools without touching core.
45
51
  - **Multi-machine (only if demanded)** — sync the SQLite *index*, never transcripts. The everything-stays-local promise is a selling point; don't spend it cheaply.
@@ -0,0 +1,7 @@
1
+ # Token category breakdown is precomputed at index time, not re-parsed on demand
2
+
3
+ Status: accepted
4
+
5
+ Per-session token category counts (input/output/cache-write/cache-read — see `CONTEXT.md`) will be extracted by each `Source` and stored as columns on the indexed `Session` row, the same way `token_count` and `cost_usd` already are — not recomputed by re-reading raw transcripts each time `stats`/`serve` runs.
6
+
7
+ We considered computing it on demand instead: no schema migration, always fresh. Rejected because it doesn't scale with transcript count/size, requires `raw_path` to remain valid indefinitely, and breaks the precedent every other derived Session field already follows (parse once at scan time, query the index after). A future need to re-derive from raw data (e.g. per-model breakdown) can still read `raw_path` — this ADR only settles where the *category* breakdown lives.
@@ -0,0 +1,7 @@
1
+ # `hades serve` uses stdlib `http.server`, not a web framework
2
+
3
+ Status: accepted
4
+
5
+ `hades serve` starts a foreground localhost HTTP server (stdlib `http.server`) that serves one static HTML/JS page plus a couple of JSON endpoints reading the SQLite index. No Flask/FastAPI/etc. dependency is added.
6
+
7
+ We considered a micro-framework for nicer routing/templating. Rejected: the surface is one page and two read-only endpoints against local data — a framework buys nothing at this size and adds a dependency (plus its own attack surface) to a tool whose value proposition is staying lightweight and local-only (`pyproject.toml` carries no web dependency today; `typer`/`rich`/`sqlite-utils` are all it needs). Revisit only if `serve` grows enough routes/state that hand-rolled dispatch becomes the harder path.
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
18
18
  commit_id: str | None
19
19
  __commit_id__: str | None
20
20
 
21
- __version__ = version = '0.2.0'
22
- __version_tuple__ = version_tuple = (0, 2, 0)
21
+ __version__ = version = '0.2.2'
22
+ __version_tuple__ = version_tuple = (0, 2, 2)
23
23
 
24
24
  __commit_id__ = commit_id = None
@@ -140,6 +140,23 @@ def watch_cmd(
140
140
  cmd_watch(notify=notify)
141
141
 
142
142
 
143
+ @app.command("reindex", help="Force reparse of sessions, bypassing the mtime/parser-version self-heal check.")
144
+ def reindex_cmd(
145
+ tool: Optional[str] = typer.Option(None, "--tool", "-t", help="Only reindex sessions from this tool"),
146
+ session_id: Optional[str] = typer.Option(None, "--session", help="Only reindex this session ID"),
147
+ ):
148
+ from hades.commands.reindex import cmd_reindex
149
+ cmd_reindex(tool=tool, session_id=session_id)
150
+
151
+
152
+ @app.command("serve", help="Run a localhost dashboard showing the token/cost breakdown by tool and project.")
153
+ def serve_cmd(
154
+ port: Optional[int] = typer.Option(None, "--port", "-p", help="Port to listen on (default: config serve_port)"),
155
+ ):
156
+ from hades.commands.serve import cmd_serve
157
+ cmd_serve(port=port)
158
+
159
+
143
160
  @app.command("setup", help="Show which sources are detected on this machine and enabled for scanning.")
144
161
  def setup_cmd():
145
162
  from hades.commands.tools import cmd_setup
@@ -0,0 +1,10 @@
1
+ from hades.db import get_db
2
+ from hades.indexer import force_reindex
3
+
4
+ from hades.console import console
5
+
6
+
7
+ def cmd_reindex(tool: str | None, session_id: str | None) -> None:
8
+ db = get_db()
9
+ count = force_reindex(db, tool=tool, session_id=session_id)
10
+ console.print(f"[green]Reindexed[/green] {count} session(s)")
@@ -0,0 +1,117 @@
1
+ import json
2
+ from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
3
+ from pathlib import Path
4
+ from urllib.parse import parse_qs, urlparse
5
+
6
+ from hades import pricing
7
+ from hades.config import get_config
8
+ from hades.db import get_db
9
+
10
+ from hades.console import console
11
+
12
+ STATIC_DIR = Path(__file__).parent.parent / "static"
13
+
14
+
15
+ def _date_range(query: dict) -> tuple[str, str] | None:
16
+ frm, to = query.get("from", [None])[0], query.get("to", [None])[0]
17
+ if not frm or not to:
18
+ return None
19
+ return frm, to
20
+
21
+
22
+ def _summary_rows(db, frm: str, to: str) -> list[dict]:
23
+ rows = db.execute(
24
+ """SELECT tool, project_path,
25
+ COALESCE(SUM(input_tokens), 0), COALESCE(SUM(output_tokens), 0),
26
+ COALESCE(SUM(cache_write_tokens), 0), COALESCE(SUM(cache_read_tokens), 0),
27
+ COALESCE(SUM(cost_usd), 0.0)
28
+ FROM sessions
29
+ WHERE is_archived IS NOT 1 AND date(last_active_at) BETWEEN ? AND ?
30
+ GROUP BY tool, project_path""",
31
+ [frm, to],
32
+ ).fetchall()
33
+ return [
34
+ {
35
+ "tool": tool, "project": project,
36
+ "input": inp, "output": out, "cache_write": cw, "cache_read": cr,
37
+ "cost_usd": cost or 0.0,
38
+ }
39
+ for tool, project, inp, out, cw, cr, cost in rows
40
+ ]
41
+
42
+
43
+ def _daily_rows(db, frm: str, to: str) -> list[dict]:
44
+ rows = db.execute(
45
+ """SELECT date(last_active_at) as day,
46
+ COALESCE(SUM(input_tokens), 0), COALESCE(SUM(output_tokens), 0),
47
+ COALESCE(SUM(cache_write_tokens), 0), COALESCE(SUM(cache_read_tokens), 0),
48
+ COALESCE(SUM(cost_usd), 0.0)
49
+ FROM sessions
50
+ WHERE is_archived IS NOT 1 AND date(last_active_at) BETWEEN ? AND ?
51
+ GROUP BY day""",
52
+ [frm, to],
53
+ ).fetchall()
54
+ out = []
55
+ for day, inp, out_tok, cw, cr, cost in rows:
56
+ shares = pricing.category_cost_shares(inp, out_tok, cw, cr, cost or 0.0)
57
+ out.append({
58
+ "date": day,
59
+ "input": inp, "output": out_tok, "cache_write": cw, "cache_read": cr,
60
+ "input_cost": shares["input"], "output_cost": shares["output"],
61
+ "cache_write_cost": shares["cache_write"], "cache_read_cost": shares["cache_read"],
62
+ })
63
+ return out
64
+
65
+
66
+ class _Handler(BaseHTTPRequestHandler):
67
+ def log_message(self, format, *args): # pylint: disable=redefined-builtin
68
+ pass # quiet by default; `hades serve` prints its own startup line
69
+
70
+ def _json(self, payload: dict, status: int = 200) -> None:
71
+ body = json.dumps(payload).encode()
72
+ self.send_response(status)
73
+ self.send_header("Content-Type", "application/json")
74
+ self.send_header("Content-Length", str(len(body)))
75
+ self.end_headers()
76
+ self.wfile.write(body)
77
+
78
+ def do_GET(self): # pylint: disable=invalid-name
79
+ parsed = urlparse(self.path)
80
+ query = parse_qs(parsed.query)
81
+
82
+ if parsed.path == "/":
83
+ body = (STATIC_DIR / "serve.html").read_bytes()
84
+ self.send_response(200)
85
+ self.send_header("Content-Type", "text/html")
86
+ self.send_header("Content-Length", str(len(body)))
87
+ self.end_headers()
88
+ self.wfile.write(body)
89
+ return
90
+
91
+ if parsed.path in ("/api/summary", "/api/daily"):
92
+ date_range = _date_range(query)
93
+ if date_range is None:
94
+ self._json({"error": "from and to query params are required (YYYY-MM-DD)"}, status=400)
95
+ return
96
+ frm, to = date_range
97
+ db = get_db()
98
+ if parsed.path == "/api/summary":
99
+ self._json({"rows": _summary_rows(db, frm, to)})
100
+ else:
101
+ self._json({"days": _daily_rows(db, frm, to)})
102
+ return
103
+
104
+ self.send_response(404)
105
+ self.end_headers()
106
+
107
+
108
+ def cmd_serve(port: int | None, host: str = "127.0.0.1") -> None:
109
+ resolved_port = port if port is not None else get_config("serve_port")
110
+ server = ThreadingHTTPServer((host, resolved_port), _Handler)
111
+ console.print(f"[green]hades serve[/green] running at http://{host}:{resolved_port} — Ctrl+C to stop")
112
+ try:
113
+ server.serve_forever()
114
+ except KeyboardInterrupt:
115
+ pass
116
+ finally:
117
+ server.server_close()
@@ -12,6 +12,7 @@ CONFIG_PATH = Path(user_data_dir("hades")) / "config.json"
12
12
 
13
13
  DEFAULTS = {
14
14
  "wait_threshold_minutes": 3,
15
+ "serve_port": 8080,
15
16
  }
16
17
 
17
18
  # Which sources get scanned/indexed. Only Claude is on out of the box;
@@ -34,6 +34,11 @@ def _ensure_schema(db: sqlite_utils.Database) -> None:
34
34
  "waiting_since": str,
35
35
  "token_count": int,
36
36
  "cost_usd": float,
37
+ "parser_version": int,
38
+ "input_tokens": int,
39
+ "output_tokens": int,
40
+ "cache_write_tokens": int,
41
+ "cache_read_tokens": int,
37
42
  }, pk="id")
38
43
  db["sessions"].create_index(["raw_path"], unique=True)
39
44
  else:
@@ -43,7 +48,9 @@ def _ensure_schema(db: sqlite_utils.Database) -> None:
43
48
  ("human_messages", str), ("assistant_messages", str),
44
49
  ("file_mtime", float), ("is_archived", int),
45
50
  ("waiting_since", str), ("token_count", int),
46
- ("cost_usd", float),
51
+ ("cost_usd", float), ("parser_version", int),
52
+ ("input_tokens", int), ("output_tokens", int),
53
+ ("cache_write_tokens", int), ("cache_read_tokens", int),
47
54
  ]:
48
55
  if col not in existing_cols:
49
56
  db["sessions"].add_column(col, col_type)
@@ -0,0 +1,173 @@
1
+ from pathlib import Path
2
+
3
+ import sqlite_utils
4
+
5
+ from hades.config import get_enabled_tools
6
+ from hades.sources import ALL_SOURCES
7
+
8
+
9
+ def refresh_index(db: sqlite_utils.Database) -> None:
10
+ """Scan enabled sources, upsert changed/new/stale-parsed sessions, remove stale rows."""
11
+ enabled_tools = get_enabled_tools()
12
+ indexed_paths: set[str] = set()
13
+ discovered: dict[str, type] = {}
14
+
15
+ for source_cls in ALL_SOURCES:
16
+ if source_cls.name not in enabled_tools:
17
+ # Disabled tool: don't scan it, and don't touch its existing index
18
+ # rows either — same treatment as a source whose store isn't
19
+ # reachable right now (see _remove_stale below).
20
+ continue
21
+ if source_cls.discover_root() is None:
22
+ # Source store not found (unmounted disk, bad env override, tool not
23
+ # installed). Skip it entirely — including stale removal — so a
24
+ # temporarily missing store doesn't wipe its slice of the index.
25
+ continue
26
+ discovered[source_cls.name] = source_cls
27
+
28
+ for path in source_cls.list_files():
29
+ path_str = str(path)
30
+ indexed_paths.add(path_str)
31
+
32
+ try:
33
+ mtime = path.stat().st_mtime
34
+ except OSError:
35
+ continue
36
+
37
+ existing = db.execute(
38
+ "SELECT file_mtime, parser_version FROM sessions WHERE raw_path = ?", [path_str]
39
+ ).fetchone()
40
+
41
+ if existing and existing[0] == mtime and (existing[1] or 0) >= source_cls.PARSER_VERSION:
42
+ continue
43
+
44
+ _reparse_and_upsert(db, source_cls, path, mtime)
45
+
46
+ _reheal_archived(db, discovered)
47
+ _remove_stale(db, indexed_paths, set(discovered))
48
+ db.conn.commit()
49
+
50
+
51
+ def force_reindex(db: sqlite_utils.Database, tool: str | None = None, session_id: str | None = None) -> int:
52
+ """Bypass the mtime/parser_version checks and reparse matching sessions now.
53
+
54
+ Escape hatch for `hades reindex` — the mtime/version check in refresh_index
55
+ self-heals the common case, but a source whose parser changed without a
56
+ version bump (or a one-off re-scrape request) needs a manual trigger.
57
+ """
58
+ source_map = {source_cls.name: source_cls for source_cls in ALL_SOURCES}
59
+
60
+ query = "SELECT id, tool, raw_path FROM sessions WHERE 1=1"
61
+ params: list[str] = []
62
+ if tool:
63
+ query += " AND tool = ?"
64
+ params.append(tool)
65
+ if session_id:
66
+ query += " AND id = ?"
67
+ params.append(session_id)
68
+
69
+ reindexed = 0
70
+ for _id, tool_name, raw_path in db.execute(query, params).fetchall():
71
+ source_cls = source_map.get(tool_name)
72
+ if source_cls is None:
73
+ continue
74
+ path = Path(raw_path)
75
+ if not path.exists():
76
+ continue
77
+ if _reparse_and_upsert(db, source_cls, path, path.stat().st_mtime) is not None:
78
+ reindexed += 1
79
+
80
+ db.conn.commit()
81
+ return reindexed
82
+
83
+
84
+ def _reparse_and_upsert(db: sqlite_utils.Database, source_cls: type, path: Path, mtime: float) -> str | None:
85
+ """Parse `path` and upsert/refresh its row (and FTS entry). Returns the session id, or None."""
86
+ session = source_cls.parse_file(path)
87
+ if session is None:
88
+ return None
89
+
90
+ path_str = str(path)
91
+ human_text, assistant_text = source_cls.extract_messages(path)
92
+
93
+ row = {
94
+ "id": session.id,
95
+ "tool": session.tool,
96
+ "project_path": session.project_path,
97
+ "started_at": session.started_at.isoformat(),
98
+ "last_active_at": session.last_active_at.isoformat(),
99
+ "message_count": session.message_count,
100
+ "token_count": session.token_count,
101
+ "cost_usd": session.cost_usd,
102
+ "input_tokens": session.input_tokens,
103
+ "output_tokens": session.output_tokens,
104
+ "cache_write_tokens": session.cache_write_tokens,
105
+ "cache_read_tokens": session.cache_read_tokens,
106
+ "status": session.status,
107
+ "raw_path": path_str,
108
+ "title": session.title,
109
+ "file_mtime": mtime,
110
+ "human_messages": human_text,
111
+ "assistant_messages": assistant_text,
112
+ "parser_version": source_cls.PARSER_VERSION,
113
+ }
114
+
115
+ # A previous version may have stored this file under a different id
116
+ # (or this id under a different file). Clear both — including the
117
+ # FTS row, which would otherwise be orphaned — before upserting.
118
+ for (old_id,) in db.execute(
119
+ "SELECT id FROM sessions WHERE raw_path = ? AND id != ?",
120
+ [path_str, session.id],
121
+ ).fetchall():
122
+ db.execute("DELETE FROM sessions WHERE id = ?", [old_id])
123
+ db.execute("DELETE FROM sessions_fts WHERE id = ?", [old_id])
124
+ db["sessions"].upsert(row, pk="id")
125
+
126
+ db.execute("DELETE FROM sessions_fts WHERE id = ?", [session.id])
127
+ db.execute(
128
+ "INSERT INTO sessions_fts (id, human_messages, assistant_messages) VALUES (?, ?, ?)",
129
+ [session.id, human_text, assistant_text],
130
+ )
131
+ return session.id
132
+
133
+
134
+ def _reheal_archived(db: sqlite_utils.Database, discovered: dict[str, type]) -> None:
135
+ """Reparse archived sessions whose parser_version is behind — they're never
136
+ visited by the live-file scan above since their file moved out of the
137
+ source's discover_root into the archive.
138
+ """
139
+ if "sessions" not in db.table_names():
140
+ return
141
+
142
+ for tool_name, source_cls in discovered.items():
143
+ rows = db.execute(
144
+ "SELECT raw_path, parser_version FROM sessions WHERE is_archived = 1 AND tool = ?", [tool_name]
145
+ ).fetchall()
146
+ for raw_path, parser_version in rows:
147
+ if (parser_version or 0) >= source_cls.PARSER_VERSION:
148
+ continue
149
+ path = Path(raw_path)
150
+ if not path.exists():
151
+ continue
152
+ _reparse_and_upsert(db, source_cls, path, path.stat().st_mtime)
153
+
154
+
155
+ def _remove_stale(db: sqlite_utils.Database, indexed_paths: set[str], discovered_tools: set[str]) -> None:
156
+ """Delete index rows for session files that no longer exist on disk.
157
+
158
+ Only rows belonging to sources that were actually discovered this run are
159
+ considered — an unreachable store must not erase its history.
160
+ """
161
+ if "sessions" not in db.table_names() or not discovered_tools:
162
+ return
163
+
164
+ placeholders = ",".join("?" * len(discovered_tools))
165
+ db_rows = db.execute(
166
+ f"SELECT id, raw_path FROM sessions WHERE is_archived IS NOT 1 AND tool IN ({placeholders})",
167
+ list(discovered_tools),
168
+ ).fetchall()
169
+ stale_rows = [(session_id, path_str) for session_id, path_str in db_rows if path_str not in indexed_paths]
170
+
171
+ for session_id, path_str in stale_rows:
172
+ db.execute("DELETE FROM sessions WHERE raw_path = ?", [path_str])
173
+ db.execute("DELETE FROM sessions_fts WHERE id = ?", [session_id])
@@ -16,3 +16,7 @@ class Session:
16
16
  title: str | None
17
17
  token_count: int = 0
18
18
  cost_usd: float = 0.0
19
+ input_tokens: int = 0
20
+ output_tokens: int = 0
21
+ cache_write_tokens: int = 0
22
+ cache_read_tokens: int = 0
@@ -25,6 +25,33 @@ _PRICES: dict[str, tuple[float, float]] = {
25
25
 
26
26
  CACHE_WRITE_MULTIPLIER = 1.25
27
27
  CACHE_READ_MULTIPLIER = 0.1
28
+ OUTPUT_MULTIPLIER = 5.0 # every priced model above uses a flat 5x input:output ratio
29
+
30
+
31
+ def category_cost_shares(
32
+ input_tokens: int, output_tokens: int, cache_write_tokens: int, cache_read_tokens: int, total_cost_usd: float
33
+ ) -> dict[str, float]:
34
+ """Split an already-known total cost across the four token categories.
35
+
36
+ Sessions store one total cost_usd (priced per-turn by that turn's own
37
+ model), not a per-category dollar amount — so this apportions the total
38
+ by each category's token count weighted by its price multiplier relative
39
+ to input, rather than re-pricing per model (which would need per-day
40
+ model attribution we don't store).
41
+ # ponytail: approximation, not exact per-model pricing. Upgrade path: store
42
+ # per-category cost at index time (like cost_usd) if per-model precision
43
+ # ever matters here.
44
+ """
45
+ weighted = {
46
+ "input": input_tokens,
47
+ "output": output_tokens * OUTPUT_MULTIPLIER,
48
+ "cache_write": cache_write_tokens * CACHE_WRITE_MULTIPLIER,
49
+ "cache_read": cache_read_tokens * CACHE_READ_MULTIPLIER,
50
+ }
51
+ total_weight = sum(weighted.values())
52
+ if total_weight <= 0:
53
+ return dict.fromkeys(weighted, 0.0)
54
+ return {k: total_cost_usd * w / total_weight for k, w in weighted.items()}
28
55
 
29
56
 
30
57
  def price_per_mtok(model: str) -> tuple[float, float] | None:
@@ -8,6 +8,9 @@ class BaseSource(ABC):
8
8
  name: str
9
9
  default_paths: list[Path]
10
10
  env_var: str
11
+ # Bump when parse_file/extract_messages logic changes so refresh_index
12
+ # reparses already-indexed files even though their mtime is unchanged.
13
+ PARSER_VERSION: int = 1
11
14
 
12
15
  @classmethod
13
16
  def discover_root(cls) -> Path | None:
@@ -3,13 +3,16 @@ from pathlib import Path
3
3
 
4
4
  from hades.models import Session
5
5
  from .base import BaseSource
6
- from .common import extract_cost, extract_token_count, parse_timestamps, read_jsonl_dicts
6
+ from .common import extract_cost, extract_token_breakdown, extract_token_count, parse_timestamps, read_jsonl_dicts
7
7
 
8
8
 
9
9
  class ClaudeSource(BaseSource):
10
10
  name = "claude"
11
11
  env_var = "HADES_CLAUDE_PATH"
12
12
  default_paths = [Path("~/.claude/projects")]
13
+ # v2: backfills cost_usd for sessions indexed before that field existed.
14
+ # v3: backfills the four token-category columns (input/output/cache-write/cache-read).
15
+ PARSER_VERSION = 3
13
16
 
14
17
  @classmethod
15
18
  def list_files(cls) -> list[Path]:
@@ -33,6 +36,7 @@ class ClaudeSource(BaseSource):
33
36
  started_at = min(timestamps) if timestamps else datetime.now(timezone.utc)
34
37
  last_active_at = max(timestamps) if timestamps else started_at
35
38
  title = _extract_title(user_msgs)
39
+ breakdown = extract_token_breakdown(assistant_msgs)
36
40
 
37
41
  return Session(
38
42
  # Keyed by file stem, not the inner sessionId: resumed sessions
@@ -49,6 +53,10 @@ class ClaudeSource(BaseSource):
49
53
  title=title,
50
54
  token_count=extract_token_count(assistant_msgs),
51
55
  cost_usd=extract_cost(assistant_msgs),
56
+ input_tokens=breakdown["input"],
57
+ output_tokens=breakdown["output"],
58
+ cache_write_tokens=breakdown["cache_write"],
59
+ cache_read_tokens=breakdown["cache_read"],
52
60
  )
53
61
 
54
62
  @classmethod
@@ -11,6 +11,8 @@ class CodexSource(BaseSource):
11
11
  name = "codex"
12
12
  env_var = "HADES_CODEX_PATH"
13
13
  default_paths = [Path("~/.codex/sessions")]
14
+ # v2: backfills the four token-category columns (input/output/cache-write/cache-read).
15
+ PARSER_VERSION = 2
14
16
 
15
17
  @classmethod
16
18
  def list_files(cls) -> list[Path]:
@@ -28,6 +30,7 @@ class CodexSource(BaseSource):
28
30
  last_active_at = max(timestamps) if timestamps else started_at
29
31
 
30
32
  turns = [r for r in records if _turn_role(r) in ("user", "assistant")]
33
+ breakdown = _extract_token_breakdown(records)
31
34
 
32
35
  return Session(
33
36
  id=f"codex:{path.stem}",
@@ -41,6 +44,10 @@ class CodexSource(BaseSource):
41
44
  title=_extract_title(turns),
42
45
  token_count=_extract_token_count(records),
43
46
  cost_usd=_extract_cost(records),
47
+ input_tokens=breakdown["input"],
48
+ output_tokens=breakdown["output"],
49
+ cache_write_tokens=breakdown["cache_write"],
50
+ cache_read_tokens=breakdown["cache_read"],
44
51
  )
45
52
 
46
53
  @classmethod
@@ -123,6 +130,20 @@ def _extract_token_count(records: list[dict]) -> int:
123
130
  return usage.get("total_tokens", 0) or 0 if usage else 0
124
131
 
125
132
 
133
+ def _extract_token_breakdown(records: list[dict]) -> dict[str, int]:
134
+ """The cumulative usage dict already carries all four categories — no
135
+ summing needed, same as _extract_cost."""
136
+ usage = _last_token_usage(records)
137
+ if not usage:
138
+ return {"input": 0, "output": 0, "cache_write": 0, "cache_read": 0}
139
+ return {
140
+ "input": usage.get("input_tokens", 0) or 0,
141
+ "output": usage.get("output_tokens", 0) or 0,
142
+ "cache_write": usage.get("cache_write_input_tokens", 0) or 0,
143
+ "cache_read": usage.get("cached_input_tokens", 0) or 0,
144
+ }
145
+
146
+
126
147
  def _extract_model(records: list[dict]) -> str | None:
127
148
  for r in records:
128
149
  if r.get("type") != "world_state":
@@ -64,6 +64,22 @@ def extract_token_count(assistant_msgs: list[dict]) -> int:
64
64
  return total
65
65
 
66
66
 
67
+ def extract_token_breakdown(assistant_msgs: list[dict]) -> dict[str, int]:
68
+ """Sum each of the four token categories across every assistant turn's
69
+ usage block. Shared by any source whose transcript format nests a
70
+ Claude-shaped message.usage block (Claude Code and Cowork both do)."""
71
+ totals = {"input": 0, "output": 0, "cache_write": 0, "cache_read": 0}
72
+ for m in assistant_msgs:
73
+ usage = m.get("message", {}).get("usage")
74
+ if not isinstance(usage, dict):
75
+ continue
76
+ totals["input"] += usage.get("input_tokens", 0) or 0
77
+ totals["output"] += usage.get("output_tokens", 0) or 0
78
+ totals["cache_write"] += usage.get("cache_creation_input_tokens", 0) or 0
79
+ totals["cache_read"] += usage.get("cache_read_input_tokens", 0) or 0
80
+ return totals
81
+
82
+
67
83
  def extract_cost(assistant_msgs: list[dict]) -> float:
68
84
  """Sum per-turn USD cost, priced by that turn's own model."""
69
85
  total = 0.0
@@ -3,7 +3,9 @@ from pathlib import Path
3
3
 
4
4
  from hades.models import Session
5
5
  from .base import BaseSource
6
- from .common import extract_cost, extract_token_count, parse_timestamps, read_json_document, read_jsonl_dicts
6
+ from .common import (
7
+ extract_cost, extract_token_breakdown, extract_token_count, parse_timestamps, read_json_document, read_jsonl_dicts,
8
+ )
7
9
 
8
10
 
9
11
  class CoworkSource(BaseSource):
@@ -13,6 +15,8 @@ class CoworkSource(BaseSource):
13
15
  Path("~/Library/Application Support/Claude/local-agent-mode-sessions"),
14
16
  Path("~/.config/claude/local-agent-mode-sessions"),
15
17
  ]
18
+ # v2: backfills the four token-category columns (input/output/cache-write/cache-read).
19
+ PARSER_VERSION = 2
16
20
 
17
21
  @classmethod
18
22
  def list_files(cls) -> list[Path]:
@@ -40,6 +44,7 @@ class CoworkSource(BaseSource):
40
44
  # not a real host project path. The session's own title is the closest
41
45
  # thing to a project label, so it's used for both fields below.
42
46
  title = _read_metadata(path).get("title") or _extract_title(user_msgs)
47
+ breakdown = extract_token_breakdown(assistant_msgs)
43
48
 
44
49
  return Session(
45
50
  id=f"cowork:{path.parent.name}",
@@ -53,6 +58,10 @@ class CoworkSource(BaseSource):
53
58
  title=title,
54
59
  token_count=extract_token_count(assistant_msgs),
55
60
  cost_usd=extract_cost(assistant_msgs),
61
+ input_tokens=breakdown["input"],
62
+ output_tokens=breakdown["output"],
63
+ cache_write_tokens=breakdown["cache_write"],
64
+ cache_read_tokens=breakdown["cache_read"],
56
65
  )
57
66
 
58
67
  @classmethod