transcript-viewer 0.14.0__tar.gz → 0.15.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. transcript_viewer-0.14.0/README.md → transcript_viewer-0.15.0/PKG-INFO +75 -24
  2. transcript_viewer-0.14.0/PKG-INFO → transcript_viewer-0.15.0/README.md +62 -37
  3. transcript_viewer-0.15.0/TODO.md +40 -0
  4. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/pyproject.toml +22 -3
  5. transcript_viewer-0.15.0/src/transcript_viewer/page.html +7042 -0
  6. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/src/transcript_viewer/viewer.py +6 -0
  7. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/tests/page.test.js +1679 -134
  8. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/tests/test_style.py +29 -6
  9. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/tests/test_tidiness.py +24 -0
  10. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/tests/test_viewer.py +14 -0
  11. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/uv.lock +15 -9
  12. transcript_viewer-0.14.0/src/transcript_viewer/page.html +0 -3463
  13. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/.coverage +0 -0
  14. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/.github/workflows/publish.yml +0 -0
  15. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/.github/workflows/test.yml +0 -0
  16. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/.gitignore +0 -0
  17. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/design/Anchor.dc.html +0 -0
  18. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/design/Dense.dc.html +0 -0
  19. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/design/Hover.dc.html +0 -0
  20. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/design/Main.dc.html +0 -0
  21. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/design/Quiet.dc.html +0 -0
  22. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/design/canvas.json +0 -0
  23. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/design/delegation-timeline.html +0 -0
  24. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/src/transcript_viewer/__init__.py +0 -0
  25. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/src/transcript_viewer/ai.py +0 -0
  26. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/src/transcript_viewer/cli.py +0 -0
  27. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/src/transcript_viewer/config.py +0 -0
  28. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/src/transcript_viewer/corpus.py +0 -0
  29. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/src/transcript_viewer/fetch.py +0 -0
  30. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/src/transcript_viewer/library.py +0 -0
  31. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/src/transcript_viewer/store.py +0 -0
  32. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/tests/fixtures/big-command.json +0 -0
  33. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/tests/test_ai.py +0 -0
  34. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/tests/test_cli.py +0 -0
  35. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/tests/test_config.py +0 -0
  36. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/tests/test_corpus.py +0 -0
  37. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/tests/test_fetch.py +0 -0
  38. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/tests/test_library.py +0 -0
  39. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/tests/test_page.py +0 -0
  40. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/tests/test_readme.py +0 -0
  41. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/tests/test_store.py +0 -0
  42. {transcript_viewer-0.14.0 → transcript_viewer-0.15.0}/tests/test_stream.py +0 -0
@@ -1,3 +1,16 @@
1
+ Metadata-Version: 2.5
2
+ Name: transcript-viewer
3
+ Version: 0.15.0
4
+ Summary: Browse agent transcripts in a local, dependency-free web viewer.
5
+ License: MIT
6
+ Requires-Python: >=3.12
7
+ Requires-Dist: atif-make>=0.8.0
8
+ Provides-Extra: ai
9
+ Requires-Dist: anthropic>=0.40; extra == 'ai'
10
+ Provides-Extra: parquet
11
+ Requires-Dist: atif-make[parquet]>=0.8.0; extra == 'parquet'
12
+ Description-Content-Type: text/markdown
13
+
1
14
  # transcript-viewer
2
15
 
3
16
  Browse agent transcripts in a local web viewer — what Claude Code, Codex and
@@ -166,17 +179,16 @@ If that directory is still there, the first run moves it to
166
179
  `~/.transcript-viewer` — index, annotations and stored keys together — and says
167
180
  so. Nothing is left behind and nothing is copied twice.
168
181
 
169
- A run that delegated shows what it farmed out, above its steps, as a **list**
170
- or a **timeline**. The list leads because the common question is what was asked
171
- and what came back, and the median subagent here is two steps — a bar for which
172
- says almost nothing. It carries the task, the kind, the steps, how long it took
173
- and the tools it reached for, sorted however you ask.
174
-
175
- The timeline is for when the question is *when*, and alongside what. A row is
176
- shared by everything that did not overlap, so a second row exists only because
177
- two things ran at once: thirty subagents in one session pack into eight rows,
178
- and a session of fifteen packs into one, which is the honest picture of a run
179
- that never did two things at a time. Either way a subagent is called what it was asked to
182
+ A run that delegated shows what it farmed out, above its steps, as a
183
+ **timeline** or a **list**. The timeline leads: a row is shared by everything
184
+ that did not overlap, so a second row exists only because two things ran at
185
+ once. Thirty subagents in one session pack into eight rows, and a session of
186
+ fifteen packs into one, which is the honest picture of a run that never did two
187
+ things at a time.
188
+
189
+ The list is one chip away, and carries what a chart cannot: the task, the kind,
190
+ the steps, how long it took and the tools it reached for, sorted however you
191
+ ask. Whichever you pick is remembered. Either way a subagent is called what it was asked to
180
192
  do — "Review backend for bugs and DRY" rather than "general-purpose" — wherever
181
193
  that was written down, which is 50 of the 53 Claude Code subagents here; the
182
194
  kind is the fallback, and reads as a handle rather than as prose so the
@@ -186,9 +198,33 @@ in the corpus this was built against were. Two bars in the same column of the
186
198
  chart is what says it; overlap gets no colour of its own, since a second
187
199
  encoding of the same fact is noise.
188
200
 
201
+ A subagent can delegate in turn, and in some harnesses that does not stop at two
202
+ levels. One that did gets a row to itself, labelled, with the number it spawned
203
+ and a control that opens them: its children appear beneath it, indented, packed
204
+ among themselves rather than against the rest of the run. The list does the same
205
+ with indented rows. Shut by default and remembered per subagent, so a run where
206
+ nothing nested — which is every run in the library this was built against — is
207
+ drawn exactly as it was before. The header says how deep the delegation went
208
+ only when it went further than one hop.
209
+
210
+ Under the chart is the shape of the delegation as numbers: every link in it
211
+ sorted into the three kinds it can be — from the main thread, from another
212
+ subagent, between subagents — to scale, with the counts and the ratio. A run is
213
+ not described by its subagent count alone; five spawned by the main thread is a
214
+ fan-out and five where two of them spawned the rest is a hierarchy, and the two
215
+ read identically until the links are counted. Each kind is counted rather than
216
+ assumed, so a format that starts recording one starts showing it here. Across
217
+ the library this was built against the ratio is 100 / 0 / 0 — 52 links in 17
218
+ transcripts, every one a call from the main thread — and 92 further subagents
219
+ that nothing links at all, which is the fact the line exists to surface. Beside
220
+ it: how many delegated in turn, the widest fan-out, and how many were spawned by
221
+ nothing.
222
+
189
223
  Clicking a lane opens the box the transcript already has for it rather than
190
224
  showing the same thing twice, drawing the run out far enough to reach it first:
191
- in one real session every delegation happens after step 1,231.
225
+ in one real session every delegation happens after step 1,231. A nested subagent
226
+ is drawn inside its orchestrator's box, so it is the orchestrator's step that
227
+ has to be painted and the boxes above it are unfolded on the way.
192
228
 
193
229
  A long run paints its first 250 steps at once and fills the rest in behind
194
230
  itself, so there is nothing to press and nothing to wait for. The largest run
@@ -407,29 +443,44 @@ an external file (`--split-subagents`) rather than an embedded trajectory, the
407
443
  viewer says so instead of silently showing nothing.
408
444
 
409
445
  Because branches sit anywhere in a trajectory that may run to thousands of
410
- steps, every session with branches gets a jump list at the top — agent type,
411
- task, step count — and an "only branches" filter. Steps render 250 at a time so
446
+ steps, there is an "only branches" filter. There was also a jump list above the
447
+ run, until the delegation box came to say everything it said and more — and to
448
+ disagree with it, since the list counted top-level calls while the box counts
449
+ every subagent at every depth. Steps render 250 at a time so
412
450
  a 8,000-step session stays responsive.
413
451
 
414
452
  ## Developing alongside atif-make
415
453
 
416
- The two packages are developed together. Installing the viewer editable makes
417
- its own code live:
454
+ The two packages are developed together — a viewer feature usually needs the
455
+ parser to record the thing first — so `atif-make` resolves from the directory
456
+ next door rather than from the index:
418
457
 
419
- ```sh
420
- uv tool install --force --editable .
458
+ ```toml
459
+ [tool.uv.sources]
460
+ atif-make = { path = "../atif-make", editable = true }
421
461
  ```
422
462
 
423
- That alone still resolves `atif-make` from git, so edits to the converter would
424
- not show up. To run with **both** live, add it explicitly:
463
+ `uv run transcript-viewer` therefore runs both working copies, with no extra
464
+ flags. uv drops that declaration when the package is built, so what is published
465
+ still depends on the version range; a checkout without `atif-make` beside it can
466
+ pass `--no-sources`.
467
+
468
+ To make the installed command live too:
425
469
 
426
470
  ```sh
427
- uv run --with-editable ../atif-make transcript-viewer
471
+ uv tool install --force --editable .
428
472
  ```
429
473
 
430
- Use that while changing anything in `atif-make`. Reinstall from the index
431
- (`uv tool install --force transcript-viewer`) when you want to test what users actually
432
- get.
474
+ Reinstall from the index (`uv tool install --force transcript-viewer`) when you
475
+ want to test what users actually get.
476
+
477
+ **The page is re-read on every load; the converter is not.** `page.html` is
478
+ stat'ed per request, so editing it shows on a refresh. Python modules are
479
+ imported once and a converted trajectory is cached in memory for the life of the
480
+ run, so a parser change needs the server restarted. A fix that made a session's
481
+ delegation three levels deep looked like it had done nothing for exactly this
482
+ reason: the page was new, the parser in memory was not, and the viewer drew what
483
+ it was handed.
433
484
 
434
485
  ## Layout
435
486
 
@@ -1,16 +1,3 @@
1
- Metadata-Version: 2.5
2
- Name: transcript-viewer
3
- Version: 0.14.0
4
- Summary: Browse agent transcripts in a local, dependency-free web viewer.
5
- License: MIT
6
- Requires-Python: >=3.12
7
- Requires-Dist: atif-make>=0.7.1
8
- Provides-Extra: ai
9
- Requires-Dist: anthropic>=0.40; extra == 'ai'
10
- Provides-Extra: parquet
11
- Requires-Dist: atif-make[parquet]>=0.5.0; extra == 'parquet'
12
- Description-Content-Type: text/markdown
13
-
14
1
  # transcript-viewer
15
2
 
16
3
  Browse agent transcripts in a local web viewer — what Claude Code, Codex and
@@ -179,17 +166,16 @@ If that directory is still there, the first run moves it to
179
166
  `~/.transcript-viewer` — index, annotations and stored keys together — and says
180
167
  so. Nothing is left behind and nothing is copied twice.
181
168
 
182
- A run that delegated shows what it farmed out, above its steps, as a **list**
183
- or a **timeline**. The list leads because the common question is what was asked
184
- and what came back, and the median subagent here is two steps — a bar for which
185
- says almost nothing. It carries the task, the kind, the steps, how long it took
186
- and the tools it reached for, sorted however you ask.
187
-
188
- The timeline is for when the question is *when*, and alongside what. A row is
189
- shared by everything that did not overlap, so a second row exists only because
190
- two things ran at once: thirty subagents in one session pack into eight rows,
191
- and a session of fifteen packs into one, which is the honest picture of a run
192
- that never did two things at a time. Either way a subagent is called what it was asked to
169
+ A run that delegated shows what it farmed out, above its steps, as a
170
+ **timeline** or a **list**. The timeline leads: a row is shared by everything
171
+ that did not overlap, so a second row exists only because two things ran at
172
+ once. Thirty subagents in one session pack into eight rows, and a session of
173
+ fifteen packs into one, which is the honest picture of a run that never did two
174
+ things at a time.
175
+
176
+ The list is one chip away, and carries what a chart cannot: the task, the kind,
177
+ the steps, how long it took and the tools it reached for, sorted however you
178
+ ask. Whichever you pick is remembered. Either way a subagent is called what it was asked to
193
179
  do — "Review backend for bugs and DRY" rather than "general-purpose" — wherever
194
180
  that was written down, which is 50 of the 53 Claude Code subagents here; the
195
181
  kind is the fallback, and reads as a handle rather than as prose so the
@@ -199,9 +185,33 @@ in the corpus this was built against were. Two bars in the same column of the
199
185
  chart is what says it; overlap gets no colour of its own, since a second
200
186
  encoding of the same fact is noise.
201
187
 
188
+ A subagent can delegate in turn, and in some harnesses that does not stop at two
189
+ levels. One that did gets a row to itself, labelled, with the number it spawned
190
+ and a control that opens them: its children appear beneath it, indented, packed
191
+ among themselves rather than against the rest of the run. The list does the same
192
+ with indented rows. Shut by default and remembered per subagent, so a run where
193
+ nothing nested — which is every run in the library this was built against — is
194
+ drawn exactly as it was before. The header says how deep the delegation went
195
+ only when it went further than one hop.
196
+
197
+ Under the chart is the shape of the delegation as numbers: every link in it
198
+ sorted into the three kinds it can be — from the main thread, from another
199
+ subagent, between subagents — to scale, with the counts and the ratio. A run is
200
+ not described by its subagent count alone; five spawned by the main thread is a
201
+ fan-out and five where two of them spawned the rest is a hierarchy, and the two
202
+ read identically until the links are counted. Each kind is counted rather than
203
+ assumed, so a format that starts recording one starts showing it here. Across
204
+ the library this was built against the ratio is 100 / 0 / 0 — 52 links in 17
205
+ transcripts, every one a call from the main thread — and 92 further subagents
206
+ that nothing links at all, which is the fact the line exists to surface. Beside
207
+ it: how many delegated in turn, the widest fan-out, and how many were spawned by
208
+ nothing.
209
+
202
210
  Clicking a lane opens the box the transcript already has for it rather than
203
211
  showing the same thing twice, drawing the run out far enough to reach it first:
204
- in one real session every delegation happens after step 1,231.
212
+ in one real session every delegation happens after step 1,231. A nested subagent
213
+ is drawn inside its orchestrator's box, so it is the orchestrator's step that
214
+ has to be painted and the boxes above it are unfolded on the way.
205
215
 
206
216
  A long run paints its first 250 steps at once and fills the rest in behind
207
217
  itself, so there is nothing to press and nothing to wait for. The largest run
@@ -420,29 +430,44 @@ an external file (`--split-subagents`) rather than an embedded trajectory, the
420
430
  viewer says so instead of silently showing nothing.
421
431
 
422
432
  Because branches sit anywhere in a trajectory that may run to thousands of
423
- steps, every session with branches gets a jump list at the top — agent type,
424
- task, step count — and an "only branches" filter. Steps render 250 at a time so
433
+ steps, there is an "only branches" filter. There was also a jump list above the
434
+ run, until the delegation box came to say everything it said and more — and to
435
+ disagree with it, since the list counted top-level calls while the box counts
436
+ every subagent at every depth. Steps render 250 at a time so
425
437
  a 8,000-step session stays responsive.
426
438
 
427
439
  ## Developing alongside atif-make
428
440
 
429
- The two packages are developed together. Installing the viewer editable makes
430
- its own code live:
441
+ The two packages are developed together — a viewer feature usually needs the
442
+ parser to record the thing first — so `atif-make` resolves from the directory
443
+ next door rather than from the index:
431
444
 
432
- ```sh
433
- uv tool install --force --editable .
445
+ ```toml
446
+ [tool.uv.sources]
447
+ atif-make = { path = "../atif-make", editable = true }
434
448
  ```
435
449
 
436
- That alone still resolves `atif-make` from git, so edits to the converter would
437
- not show up. To run with **both** live, add it explicitly:
450
+ `uv run transcript-viewer` therefore runs both working copies, with no extra
451
+ flags. uv drops that declaration when the package is built, so what is published
452
+ still depends on the version range; a checkout without `atif-make` beside it can
453
+ pass `--no-sources`.
454
+
455
+ To make the installed command live too:
438
456
 
439
457
  ```sh
440
- uv run --with-editable ../atif-make transcript-viewer
458
+ uv tool install --force --editable .
441
459
  ```
442
460
 
443
- Use that while changing anything in `atif-make`. Reinstall from the index
444
- (`uv tool install --force transcript-viewer`) when you want to test what users actually
445
- get.
461
+ Reinstall from the index (`uv tool install --force transcript-viewer`) when you
462
+ want to test what users actually get.
463
+
464
+ **The page is re-read on every load; the converter is not.** `page.html` is
465
+ stat'ed per request, so editing it shows on a refresh. Python modules are
466
+ imported once and a converted trajectory is cached in memory for the life of the
467
+ run, so a parser change needs the server restarted. A fix that made a session's
468
+ delegation three levels deep looked like it had done nothing for exactly this
469
+ reason: the page was new, the parser in memory was not, and the viewer drew what
470
+ it was handed.
446
471
 
447
472
  ## Layout
448
473
 
@@ -0,0 +1,40 @@
1
+ # Deliberately not built yet
2
+
3
+ Each of these was considered while building something adjacent and left out on
4
+ purpose. The note says what is missing and what would make it worth doing, so a
5
+ later reader can tell a decision from an oversight.
6
+
7
+ ## Inter-agent communication, drawn as arrows
8
+
9
+ The delegation box counts the three kinds of link and shows the ratio
10
+ (`laneShape`). What it does not do is *draw* a sideways link — a subagent
11
+ addressing one that is not its own child would be an arrow between two bars on
12
+ the timeline, and there is nowhere for one to go in the current layout.
13
+
14
+ Worth building when a transcript has one. Across the whole library today the
15
+ ratio reads 100 / 0 / 0 — 52 links in 17 transcripts, all of them a call from
16
+ the main thread — so an arrow layer would have nothing to draw. The count is
17
+ computed rather than assumed, so the day a format records a sibling message the
18
+ number moves on its own and the arrows become the missing half.
19
+
20
+ ## Summarising what a subagent did
21
+
22
+ The delegation box says how many steps a subagent took, how long it ran, and
23
+ which tools it reached for. It does not say what it concluded. Doing that
24
+ honestly means either reading the final assistant turn — which is often a wall
25
+ of text — or generating a summary, which the AI features could do but which
26
+ would then have to be marked as generated everywhere it appears.
27
+
28
+ ## Fleet scale
29
+
30
+ The timeline is designed for the runs in the library: tens of subagents over
31
+ hours. Reports of runs with **115 agents over ~1,170 minutes** are the case it
32
+ is not designed for. Two things break first:
33
+
34
+ - packed rows stop compressing, because at that many concurrent agents almost
35
+ everything overlaps something;
36
+ - the 7px minimum bar width (`.bar{min-width:7px}`) stops being a floor and
37
+ starts being most of the chart.
38
+
39
+ Both point at the same missing thing: **zoom**, and a way to collapse a group of
40
+ lanes into one summary lane. Neither is a tweak to the current layout.
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "transcript-viewer"
3
- version = "0.14.0"
3
+ version = "0.15.0"
4
4
  description = "Browse agent transcripts in a local, dependency-free web viewer."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.12"
@@ -10,14 +10,18 @@ license = { text = "MIT" }
10
10
  # The floor is not politeness: this viewer imports atif_make.container, which
11
11
  # is how a benchmark published as one JSON array of a thousand runs becomes a
12
12
  # thousand sessions. An older atif-make fails to import at all.
13
- dependencies = ["atif-make>=0.7.1"]
13
+ # 0.8.0 is where a Claude Code session that delegated more than one level deep
14
+ # started converting to a tree. Against an older one the chart still draws, and
15
+ # draws a flat row of subagents that nothing spawned — which reads as the
16
+ # viewer being broken rather than the converter being old.
17
+ dependencies = ["atif-make>=0.8.0"]
14
18
 
15
19
  [project.optional-dependencies]
16
20
  # Claude-backed explanations are opt-in, so the default install stays
17
21
  # dependency-free: pip install "transcript-viewer[ai]"
18
22
  # Datasets published as Parquet — METR's transcripts are the one so far. The
19
23
  # reader lives in atif-make, which only needs it to split a shard.
20
- parquet = ["atif-make[parquet]>=0.5.0"]
24
+ parquet = ["atif-make[parquet]>=0.8.0"]
21
25
  ai = ["anthropic>=0.40"]
22
26
 
23
27
  [project.scripts]
@@ -26,6 +30,21 @@ transcript-viewer = "transcript_viewer.cli:main"
26
30
  [dependency-groups]
27
31
  dev = ["pytest>=8.0"]
28
32
 
33
+ # Develop against the converter in the next directory, not the release.
34
+ #
35
+ # The two repos change together — a viewer feature usually needs the parser to
36
+ # record the thing first — and resolving `atif-make` from PyPI meant every such
37
+ # change looked like it had failed. A parser fix that made a session's
38
+ # delegation three levels deep was invisible here for exactly that reason: the
39
+ # page was local, the converter was 0.7.1, and the viewer faithfully drew what
40
+ # the old parser handed it.
41
+ #
42
+ # uv drops `tool.uv.sources` when this package is built, so what is published
43
+ # still depends on the version range above. A checkout without atif-make beside
44
+ # it can pass `--no-sources`.
45
+ [tool.uv.sources]
46
+ atif-make = { path = "../atif-make", editable = true }
47
+
29
48
 
30
49
  [build-system]
31
50
  requires = ["hatchling"]