transcript-viewer 0.5.0__tar.gz → 0.6.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. transcript_viewer-0.5.0/README.md → transcript_viewer-0.6.1/PKG-INFO +45 -8
  2. transcript_viewer-0.5.0/PKG-INFO → transcript_viewer-0.6.1/README.md +32 -19
  3. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/pyproject.toml +8 -15
  4. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/src/transcript_viewer/cli.py +4 -3
  5. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/src/transcript_viewer/corpus.py +36 -7
  6. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/src/transcript_viewer/fetch.py +4 -1
  7. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/src/transcript_viewer/library.py +35 -3
  8. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/src/transcript_viewer/page.html +117 -7
  9. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/src/transcript_viewer/viewer.py +41 -11
  10. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/tests/page.test.js +83 -8
  11. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/tests/test_corpus.py +63 -0
  12. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/tests/test_readme.py +48 -0
  13. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/tests/test_viewer.py +34 -0
  14. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/uv.lock +54 -5
  15. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/.coverage +0 -0
  16. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/.github/workflows/publish.yml +0 -0
  17. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/.github/workflows/test.yml +0 -0
  18. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/.gitignore +0 -0
  19. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/src/transcript_viewer/__init__.py +0 -0
  20. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/src/transcript_viewer/ai.py +0 -0
  21. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/src/transcript_viewer/config.py +0 -0
  22. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/src/transcript_viewer/store.py +0 -0
  23. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/tests/test_ai.py +0 -0
  24. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/tests/test_cli.py +0 -0
  25. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/tests/test_config.py +0 -0
  26. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/tests/test_fetch.py +0 -0
  27. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/tests/test_library.py +0 -0
  28. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/tests/test_page.py +0 -0
  29. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/tests/test_store.py +0 -0
  30. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/tests/test_stream.py +0 -0
  31. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/tests/test_style.py +0 -0
  32. {transcript_viewer-0.5.0 → transcript_viewer-0.6.1}/tests/test_tidiness.py +0 -0
@@ -1,17 +1,40 @@
1
+ Metadata-Version: 2.5
2
+ Name: transcript-viewer
3
+ Version: 0.6.1
4
+ Summary: Browse agent transcripts in a local, dependency-free web viewer.
5
+ License: MIT
6
+ Requires-Python: >=3.12
7
+ Requires-Dist: atif-make>=0.5.0
8
+ Provides-Extra: ai
9
+ Requires-Dist: anthropic>=0.40; extra == 'ai'
10
+ Provides-Extra: parquet
11
+ Requires-Dist: atif-make[parquet]>=0.5.0; extra == 'parquet'
12
+ Description-Content-Type: text/markdown
13
+
1
14
  # transcript-viewer
2
15
 
3
- Browse [ATIF v1.7](https://github.com/harbor-framework/harbor/blob/main/rfcs/0001-trajectory-format.md)
4
- trajectories in a local web viewer.
16
+ Browse agent transcripts in a local web viewer — what Claude Code, Codex and
17
+ Copilot write as they work.
5
18
 
6
- Conversion lives in [`atif-make`](https://github.com/jammastergirish/atif-make); this package depends on it and
7
- adds only the browser interface.
19
+ Logs are converted on the way in to
20
+ [ATIF v1.7](https://github.com/harbor-framework/harbor/blob/main/rfcs/0001-trajectory-format.md)
21
+ by [`atif-make`](https://github.com/jammastergirish/atif-make), which this package depends on and which does
22
+ all of the reading. What is added here is only the browser interface, so a
23
+ format this cannot open is a parser missing from `atif-make` rather than
24
+ anything to change in the viewer.
8
25
 
9
26
  ## Install
10
27
 
11
28
  ```sh
12
- uv tool install transcript-viewer # pulls atif-make automatically
29
+ uv tool install transcript-viewer # pulls atif-make automatically
30
+ uv tool install "transcript-viewer[parquet]" # + datasets published as Parquet
31
+ uv tool install "transcript-viewer[ai,parquet]" # + the optional Claude features
13
32
  ```
14
33
 
34
+ The extras are only needed for what they name: `parquet` for a dataset that
35
+ ships as Parquet rather than JSON, `ai` for the summarise and ask features. The
36
+ viewer works without either.
37
+
15
38
  ```sh
16
39
  transcript-viewer # the library, empty on a first run
17
40
  transcript-viewer path/to/session.jsonl # one log
@@ -137,6 +160,14 @@ Files are deleted only where they are the viewer's own copy under
137
160
  `~/.transcript-viewer/opened/`; a session found on this machine, or downloaded into a folder
138
161
  of yours, keeps its file.
139
162
 
163
+ Some datasets ship as a single file holding many runs — ATBench publishes a
164
+ thousand agent trajectories as one JSON array, METR's as Parquet shards
165
+ (install with `uv tool install "transcript-viewer[parquet]"` to read those). Those are opened rather than
166
+ listed: the file is split into a transcript apiece and each is indexed on its
167
+ own, so a download of one file becomes a thousand sessions you can read. The
168
+ pieces are kept beside what they came from, so removing the folder removes them
169
+ too.
170
+
140
171
  This tool used to be called `atif-view` and kept all of that under `~/.atif`.
141
172
  If that directory is still there, the first run moves it to
142
173
  `~/.transcript-viewer` — index, annotations and stored keys together — and says
@@ -368,7 +399,7 @@ get.
368
399
  ```
369
400
  src/transcript_viewer/
370
401
  page.html the whole front end — 2,232 lines of HTML, CSS and JavaScript
371
- viewer.py the HTTP server: fifteen endpoints over the page and the library
402
+ viewer.py the HTTP server: sixteen endpoints over the page and the library
372
403
  corpus.py the index — what is on this machine, and where it came from
373
404
  library.py what you decide about a session: title, tags, stars, summaries
374
405
  fetch.py bringing transcripts in from Hugging Face, GitHub, S3 or a link
@@ -376,13 +407,18 @@ src/transcript_viewer/
376
407
  config.py settings and tokens
377
408
  store.py small JSON files under ~/.transcript-viewer, written so a crash cannot truncate them
378
409
  cli.py the command line
410
+ page.html the whole interface: one page, no build step, no dependencies
379
411
  ```
380
412
 
381
413
  The page is a file rather than a string inside `viewer.py`, which is where it
382
414
  used to live. Two thirds of that module was CSS and JavaScript typed as though
383
415
  it were Python — a template expression once shipped inside static HTML and
384
416
  rendered its own source, which is harder to miss in a file that knows what it
385
- is. It is read once at import and served from memory, and the wheel carries it.
417
+ is. It is re-read whenever it changes, so editing the interface and reloading
418
+ shows the edit — twice it did not, and read as an edit that had failed. The
419
+ cost is a stat per page load, and none on what the page then calls; for an
420
+ installed copy the file never changes and the stat always says so. The wheel
421
+ carries it.
386
422
 
387
423
  ## Tests
388
424
 
@@ -398,7 +434,8 @@ uv run --with-editable ../atif-make pytest
398
434
  ```
399
435
 
400
436
  Add `--extra ai` to either command to exercise the AI paths against a real SDK;
401
- the tests stub the model call, so this never contacts the API.
437
+ the tests stub the model call, so this never contacts the API. `--extra parquet`
438
+ does the same for the Parquet path, which is otherwise skipped.
402
439
 
403
440
  The suite starts a real server on an ephemeral port and exercises the endpoints,
404
441
  including that it binds loopback only. It is isolated from your own library,
@@ -1,28 +1,27 @@
1
- Metadata-Version: 2.5
2
- Name: transcript-viewer
3
- Version: 0.5.0
4
- Summary: Browse agent transcripts in a local, dependency-free web viewer.
5
- License: MIT
6
- Requires-Python: >=3.12
7
- Requires-Dist: atif-make>=0.2.0
8
- Provides-Extra: ai
9
- Requires-Dist: anthropic>=0.40; extra == 'ai'
10
- Description-Content-Type: text/markdown
11
-
12
1
  # transcript-viewer
13
2
 
14
- Browse [ATIF v1.7](https://github.com/harbor-framework/harbor/blob/main/rfcs/0001-trajectory-format.md)
15
- trajectories in a local web viewer.
3
+ Browse agent transcripts in a local web viewer — what Claude Code, Codex and
4
+ Copilot write as they work.
16
5
 
17
- Conversion lives in [`atif-make`](https://github.com/jammastergirish/atif-make); this package depends on it and
18
- adds only the browser interface.
6
+ Logs are converted on the way in to
7
+ [ATIF v1.7](https://github.com/harbor-framework/harbor/blob/main/rfcs/0001-trajectory-format.md)
8
+ by [`atif-make`](https://github.com/jammastergirish/atif-make), which this package depends on and which does
9
+ all of the reading. What is added here is only the browser interface, so a
10
+ format this cannot open is a parser missing from `atif-make` rather than
11
+ anything to change in the viewer.
19
12
 
20
13
  ## Install
21
14
 
22
15
  ```sh
23
- uv tool install transcript-viewer # pulls atif-make automatically
16
+ uv tool install transcript-viewer # pulls atif-make automatically
17
+ uv tool install "transcript-viewer[parquet]" # + datasets published as Parquet
18
+ uv tool install "transcript-viewer[ai,parquet]" # + the optional Claude features
24
19
  ```
25
20
 
21
+ The extras are only needed for what they name: `parquet` for a dataset that
22
+ ships as Parquet rather than JSON, `ai` for the summarise and ask features. The
23
+ viewer works without either.
24
+
26
25
  ```sh
27
26
  transcript-viewer # the library, empty on a first run
28
27
  transcript-viewer path/to/session.jsonl # one log
@@ -148,6 +147,14 @@ Files are deleted only where they are the viewer's own copy under
148
147
  `~/.transcript-viewer/opened/`; a session found on this machine, or downloaded into a folder
149
148
  of yours, keeps its file.
150
149
 
150
+ Some datasets ship as a single file holding many runs — ATBench publishes a
151
+ thousand agent trajectories as one JSON array, METR's as Parquet shards
152
+ (install with `uv tool install "transcript-viewer[parquet]"` to read those). Those are opened rather than
153
+ listed: the file is split into a transcript apiece and each is indexed on its
154
+ own, so a download of one file becomes a thousand sessions you can read. The
155
+ pieces are kept beside what they came from, so removing the folder removes them
156
+ too.
157
+
151
158
  This tool used to be called `atif-view` and kept all of that under `~/.atif`.
152
159
  If that directory is still there, the first run moves it to
153
160
  `~/.transcript-viewer` — index, annotations and stored keys together — and says
@@ -379,7 +386,7 @@ get.
379
386
  ```
380
387
  src/transcript_viewer/
381
388
  page.html the whole front end — 2,232 lines of HTML, CSS and JavaScript
382
- viewer.py the HTTP server: fifteen endpoints over the page and the library
389
+ viewer.py the HTTP server: sixteen endpoints over the page and the library
383
390
  corpus.py the index — what is on this machine, and where it came from
384
391
  library.py what you decide about a session: title, tags, stars, summaries
385
392
  fetch.py bringing transcripts in from Hugging Face, GitHub, S3 or a link
@@ -387,13 +394,18 @@ src/transcript_viewer/
387
394
  config.py settings and tokens
388
395
  store.py small JSON files under ~/.transcript-viewer, written so a crash cannot truncate them
389
396
  cli.py the command line
397
+ page.html the whole interface: one page, no build step, no dependencies
390
398
  ```
391
399
 
392
400
  The page is a file rather than a string inside `viewer.py`, which is where it
393
401
  used to live. Two thirds of that module was CSS and JavaScript typed as though
394
402
  it were Python — a template expression once shipped inside static HTML and
395
403
  rendered its own source, which is harder to miss in a file that knows what it
396
- is. It is read once at import and served from memory, and the wheel carries it.
404
+ is. It is re-read whenever it changes, so editing the interface and reloading
405
+ shows the edit — twice it did not, and read as an edit that had failed. The
406
+ cost is a stat per page load, and none on what the page then calls; for an
407
+ installed copy the file never changes and the stat always says so. The wheel
408
+ carries it.
397
409
 
398
410
  ## Tests
399
411
 
@@ -409,7 +421,8 @@ uv run --with-editable ../atif-make pytest
409
421
  ```
410
422
 
411
423
  Add `--extra ai` to either command to exercise the AI paths against a real SDK;
412
- the tests stub the model call, so this never contacts the API.
424
+ the tests stub the model call, so this never contacts the API. `--extra parquet`
425
+ does the same for the Parquet path, which is otherwise skipped.
413
426
 
414
427
  The suite starts a real server on an ephemeral port and exercises the endpoints,
415
428
  including that it binds loopback only. It is isolated from your own library,
@@ -1,19 +1,23 @@
1
1
  [project]
2
2
  name = "transcript-viewer"
3
- version = "0.5.0"
3
+ version = "0.6.1"
4
4
  description = "Browse agent transcripts in a local, dependency-free web viewer."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.12"
7
7
  license = { text = "MIT" }
8
8
  # The only dependency is the converter. The viewer itself is stdlib http.server
9
9
  # plus one self-contained HTML page.
10
- # 0.2.0 is the floor, not politeness: Entry gained the content key this
11
- # viewer addresses sessions by, so an older atif-make crashes on the index.
12
- dependencies = ["atif-make>=0.2.0"]
10
+ # The floor is not politeness: this viewer imports atif_make.container, which
11
+ # is how a benchmark published as one JSON array of a thousand runs becomes a
12
+ # thousand sessions. An older atif-make fails to import at all.
13
+ dependencies = ["atif-make>=0.5.0"]
13
14
 
14
15
  [project.optional-dependencies]
15
16
  # Claude-backed explanations are opt-in, so the default install stays
16
17
  # dependency-free: pip install "transcript-viewer[ai]"
18
+ # Datasets published as Parquet — METR's transcripts are the one so far. The
19
+ # reader lives in atif-make, which only needs it to split a shard.
20
+ parquet = ["atif-make[parquet]>=0.5.0"]
17
21
  ai = ["anthropic>=0.40"]
18
22
 
19
23
  [project.scripts]
@@ -22,17 +26,6 @@ transcript-viewer = "transcript_viewer.cli:main"
22
26
  [dependency-groups]
23
27
  dev = ["pytest>=8.0"]
24
28
 
25
- # The two packages move together, so track atif-make's `main` rather than a
26
- # release. Once atif-make is on PyPI you can delete this block entirely: the
27
- # `dependencies` entry above already resolves from the index.
28
- #
29
- # uv still records an exact commit in uv.lock, which is what keeps a given
30
- # checkout reproducible. To pick up newer atif-make commits, re-resolve:
31
- #
32
- # uv lock --upgrade-package atif-make
33
- #
34
- [tool.uv.sources]
35
- atif-make = { git = "https://github.com/jammastergirish/atif-make", branch = "main" }
36
29
 
37
30
  [build-system]
38
31
  requires = ["hatchling"]
@@ -7,7 +7,7 @@ import sys
7
7
  from pathlib import Path
8
8
 
9
9
  from . import corpus, store
10
- from atif_make.archive import is_archive
10
+ from atif_make.container import is_container
11
11
 
12
12
  from .viewer import serve
13
13
 
@@ -39,10 +39,11 @@ def cmd_view(args: argparse.Namespace) -> int:
39
39
  if not path.exists():
40
40
  print(f"transcript-viewer: no such path: {path}", file=sys.stderr)
41
41
  return 2
42
- # A directory or archive holds many sessions; a plain file holds one.
42
+ # A directory or a container holds many sessions; a plain file
43
+ # holds one.
43
44
  entries = (
44
45
  corpus.scan([path])
45
- if path.is_dir() or is_archive(path)
46
+ if path.is_dir() or is_container(path)
46
47
  else _single_entry(path)
47
48
  )
48
49
  # An explicit path that holds nothing is a mistake worth reporting; an
@@ -16,7 +16,7 @@ from dataclasses import fields as dataclass_fields
16
16
  from datetime import datetime, timezone
17
17
  from pathlib import Path
18
18
 
19
- from atif_make.archive import extract, is_archive
19
+ from atif_make.container import is_container, open_container
20
20
 
21
21
  from . import store
22
22
  from atif_make.convert import AGENTS
@@ -32,6 +32,13 @@ INDEX_PATH = store.ROOT / "index.json"
32
32
  # whoever asks, so a plain rescan classifies them correctly on its own.
33
33
  OPENED_ROOT = store.ROOT / "opened"
34
34
 
35
+ # Where a collection's pieces are kept. A benchmark published as one JSON array
36
+ # becomes a transcript per file, and those files are what the index records, so
37
+ # they need somewhere that outlives the process that made them. Keyed by the
38
+ # source's content, so re-indexing the same download reuses the split and a
39
+ # changed file gets a new one.
40
+ SPLIT_ROOT = store.ROOT / "split"
41
+
35
42
  # A first line shorter than this is structural rather than identifying — a
36
43
  # pretty-printed JSON document opens with a bare "{".
37
44
 
@@ -251,6 +258,20 @@ def merge(existing: list[Entry], new: list[Entry]) -> list[Entry]:
251
258
  return sorted(out, key=lambda e: e.modified, reverse=True)
252
259
 
253
260
 
261
+ def _split_home(path: Path) -> Path:
262
+ """Where this file's pieces belong, if it turns out to hold many.
263
+
264
+ A download is unpacked beside itself on arrival, so prefer that: the pieces
265
+ are already there, they sit with the thing they came from, and they go when
266
+ the folder does. Anything else — a file being scanned where writing beside
267
+ it would be rude — goes under the viewer's own directory instead.
268
+ """
269
+ beside = path.with_suffix("")
270
+ if beside.is_dir() and any(beside.iterdir()):
271
+ return beside
272
+ return SPLIT_ROOT / content_key(path)[:16]
273
+
274
+
254
275
  def scan(roots: list[Path] | None = None, origin: str = "scanned") -> list[Entry]:
255
276
  """Find every convertible log under ``roots``. A root may be a single file."""
256
277
  entries: list[Entry] = []
@@ -260,10 +281,11 @@ def scan(roots: list[Path] | None = None, origin: str = "scanned") -> list[Entry
260
281
  if not root.exists():
261
282
  continue
262
283
  if root.is_file():
263
- # A zip or tarball is a container of logs, not a log.
264
- if is_archive(root):
284
+ # A zip, a tarball, or a benchmark published as one JSON array of a
285
+ # thousand runs: a container of logs, not a log.
286
+ if is_container(root):
265
287
  try:
266
- root = extract(root)
288
+ root = open_container(root, _split_home(root))
267
289
  except (ValueError, OSError):
268
290
  continue
269
291
  else:
@@ -275,11 +297,11 @@ def scan(roots: list[Path] | None = None, origin: str = "scanned") -> list[Entry
275
297
  # bucket of agent runs is mostly zips — so look inside them too. Each is
276
298
  # unpacked once and its contents scanned in place.
277
299
  roots_here = [root]
278
- for archive in sorted(root.rglob("*")):
279
- if not archive.is_file() or not is_archive(archive):
300
+ for item in sorted(root.rglob("*")):
301
+ if not item.is_file() or not is_container(item):
280
302
  continue
281
303
  try:
282
- roots_here.append(extract(archive))
304
+ roots_here.append(open_container(item, _split_home(item)))
283
305
  except (ValueError, OSError):
284
306
  continue
285
307
 
@@ -289,11 +311,18 @@ def scan(roots: list[Path] | None = None, origin: str = "scanned") -> list[Entry
289
311
  sorted(where.rglob("*.jsonl"))
290
312
  + sorted(where.rglob("*.har"))
291
313
  + sorted(where.rglob("*.json"))
314
+ # Not a log, but a container of them, opened above.
315
+ + sorted(where.rglob("*.parquet"))
292
316
  )
293
317
  for path in candidates:
294
318
  resolved = path.resolve()
295
319
  if resolved in seen:
296
320
  continue
321
+ # A container was opened above and its contents are already in this
322
+ # list. Indexing it as well would offer a session that cannot be
323
+ # opened, since it holds a thousand transcripts rather than one.
324
+ if is_container(path):
325
+ continue
297
326
  # Subagent traces are reached through their parent, not indexed alone.
298
327
  if path.parent.name == "subagents":
299
328
  continue
@@ -59,8 +59,11 @@ HOSTS = {
59
59
  # session than loose JSONL, so excluding archives made S3 useless for exactly
60
60
  # the case it was added for.
61
61
  LOGS = (".jsonl", ".json", ".har")
62
+ # A published dataset is a container of runs rather than a log, but it is still
63
+ # a thing worth fetching: METR's transcripts come as Parquet shards.
64
+ DATASETS = (".parquet",)
62
65
  ARCHIVES = (".zip", ".tgz", ".tar", ".tar.gz", ".tar.bz2", ".tar.xz", ".gz")
63
- SUFFIXES = LOGS + ARCHIVES
66
+ SUFFIXES = LOGS + DATASETS + ARCHIVES
64
67
 
65
68
  # Downloads land beside where the viewer was launched, so they are visible and
66
69
  # usable by other tools rather than buried in a dot-directory.
@@ -51,17 +51,49 @@ def _now() -> str:
51
51
  return datetime.now(UTC).isoformat(timespec="seconds").replace("+00:00", "Z")
52
52
 
53
53
 
54
- def load(path: Path | None = None) -> dict[str, dict]:
55
- """Every annotation, by key. A missing or damaged file reads as empty."""
56
- data = store.read_json(path or LIBRARY_PATH)
54
+ # What was parsed, and the state of the file it came from. The library is read
55
+ # far more often than it is written — a page load asks about every session in
56
+ # it — and it grows with the corpus, so parsing a megabyte of JSON per question
57
+ # is the difference between a page appearing and a page taking four seconds.
58
+ _parsed: tuple[int, int, dict[str, dict]] | None = None
59
+
60
+
61
+ def _read(path: Path) -> dict[str, dict]:
62
+ data = store.read_json(path)
57
63
  entries = data.get("entries")
58
64
  if not isinstance(entries, dict):
59
65
  return {}
60
66
  return {k: v for k, v in entries.items() if isinstance(v, dict)}
61
67
 
62
68
 
69
+ def load(path: Path | None = None) -> dict[str, dict]:
70
+ """Every annotation, by key. A missing or damaged file reads as empty.
71
+
72
+ The real library is cached against its own timestamp and size, so a change
73
+ made by anything — this process or another — is picked up on the next read.
74
+ A caller gets its own outer dict, since `remove` pops from what it is given
75
+ and callers should not be able to edit the cache by accident.
76
+ """
77
+ global _parsed
78
+ path = path or LIBRARY_PATH
79
+ if path != LIBRARY_PATH:
80
+ return _read(path)
81
+
82
+ try:
83
+ info = path.stat()
84
+ stamp = (info.st_mtime_ns, info.st_size)
85
+ except OSError:
86
+ return _read(path)
87
+
88
+ if _parsed is None or (_parsed[0], _parsed[1]) != stamp:
89
+ _parsed = (stamp[0], stamp[1], _read(path))
90
+ return dict(_parsed[2])
91
+
92
+
63
93
  def save(entries: dict[str, dict], path: Path | None = None) -> None:
64
94
  """Write the whole library atomically."""
95
+ global _parsed
96
+ _parsed = None # the next read re-stamps against the file just written
65
97
  store.write_json(path or LIBRARY_PATH, {"version": VERSION, "entries": entries})
66
98
 
67
99
 
@@ -430,8 +430,33 @@ pre.json{line-height:1.45}
430
430
  .more button{border:1px solid var(--line);background:var(--panel);color:var(--dim);border-radius:999px;
431
431
  padding:8px 18px;font:inherit;font-size:12.5px;cursor:pointer}
432
432
  .more button:hover{background:var(--sunk);color:var(--ink)}
433
+
434
+ /* Shown only if opening takes long enough to notice — a library of a few
435
+ thousand sessions is a couple of megabytes, and a blank window while that
436
+ arrives reads as broken rather than busy. */
437
+ #boot{position:fixed;inset:0;z-index:60;display:flex;flex-direction:column;
438
+ align-items:center;justify-content:center;gap:14px;background:var(--bg)}
439
+ #boot[hidden]{display:none}
440
+ #bootbar{width:min(320px,52vw);height:3px;border-radius:3px;background:var(--line);
441
+ overflow:hidden}
442
+ #bootbar i{display:block;height:100%;width:0;border-radius:3px;background:var(--accent);
443
+ transition:width .18s ease-out}
444
+ /* Nothing to measure yet: sweep rather than sit at zero, which reads as stuck. */
445
+ #bootbar.waiting i{width:35%;animation:sweep 1.1s ease-in-out infinite}
446
+ @keyframes sweep{0%{transform:translateX(-100%)}100%{transform:translateX(320%)}}
447
+ #boottext{margin:0;font:12px/1.5 var(--font-mono);color:var(--muted);
448
+ letter-spacing:.02em;text-align:center;padding:0 20px}
449
+ @media (prefers-reduced-motion:reduce){
450
+ #bootbar i{transition:none}
451
+ #bootbar.waiting i{animation:none;width:100%;opacity:.5}
452
+ }
433
453
  </style>
434
454
 
455
+ <div id="boot" hidden>
456
+ <div id="bootbar"><i></i></div>
457
+ <p id="boottext">Opening your library…</p>
458
+ </div>
459
+
435
460
  <header id="top">
436
461
  <span class="mark">Transcript Viewer</span>
437
462
  <div id="crumb"></div>
@@ -932,10 +957,77 @@ addEventListener("keydown",e=>{
932
957
  if(e.key==="\\"&&e.target.tagName!=="INPUT"){e.preventDefault();toggleSide()}
933
958
  });
934
959
 
935
- fetch("/api/index").then(r=>r.json()).then(d=>{
936
- INDEX=d.sessions||[];GROUPS=d.groups||[];TAGS=d.tags||[];AI=d.ai||{};DOWNLOADS=d.downloads||DOWNLOADS;
937
- count.textContent=INDEX.length+" sessions";drawList();showLibrary();
938
- });
960
+ /* Opening the library has three parts worth naming: waiting for the server to
961
+ answer, reading the answer, and drawing it. Held back briefly so a fast open
962
+ does not flash a bar at you. */
963
+ const boot={el:null,bar:null,text:null,timer:0};
964
+
965
+ function bootShow(){
966
+ boot.el=document.getElementById("boot");
967
+ boot.bar=document.getElementById("bootbar");
968
+ boot.text=document.getElementById("boottext");
969
+ if(!boot.el)return;
970
+ boot.timer=setTimeout(()=>{if(boot.el.hidden)boot.el.hidden=false},160);
971
+ }
972
+
973
+ function bootSay(what,done,total){
974
+ if(boot.text)boot.text.textContent=what;
975
+ if(!boot.bar)return;
976
+ if(total>0){
977
+ boot.bar.classList.remove("waiting");
978
+ boot.bar.firstElementChild.style.width=Math.min(100,done/total*100)+"%";
979
+ }else{
980
+ boot.bar.classList.add("waiting");
981
+ }
982
+ }
983
+
984
+ function bootDone(){
985
+ clearTimeout(boot.timer);
986
+ if(boot.el)boot.el.hidden=true;
987
+ }
988
+
989
+ /* Read the response as it arrives, so the bar tracks real bytes rather than
990
+ guessing at them. Without a length to measure against it sweeps instead of
991
+ claiming progress it cannot know. */
992
+ async function readIndex(){
993
+ const response=await fetch("/api/index");
994
+ if(!response.ok)throw new Error("the server answered "+response.status);
995
+ const total=Number(response.headers.get("Content-Length")||0);
996
+ if(!response.body||!total)return response.json();
997
+
998
+ const reader=response.body.getReader();
999
+ const chunks=[];let done=0;
1000
+ for(;;){
1001
+ const piece=await reader.read();
1002
+ if(piece.done)break;
1003
+ chunks.push(piece.value);done+=piece.value.length;
1004
+ bootSay("Reading your library… "+bytes(done)+" of "+bytes(total),done,total);
1005
+ }
1006
+ const joined=new Uint8Array(done);let at=0;
1007
+ for(const chunk of chunks){joined.set(chunk,at);at+=chunk.length}
1008
+ bootSay("Making sense of it…",0,0);
1009
+ return JSON.parse(new TextDecoder().decode(joined));
1010
+ }
1011
+
1012
+ async function openLibrary(){
1013
+ bootShow();
1014
+ bootSay("Looking for your sessions…",0,0);
1015
+ try{
1016
+ const d=await readIndex();
1017
+ INDEX=d.sessions||[];GROUPS=d.groups||[];TAGS=d.tags||[];AI=d.ai||{};DOWNLOADS=d.downloads||DOWNLOADS;
1018
+ bootSay("Drawing "+INDEX.length+" sessions…",0,0);
1019
+ // Let that line paint before the work that blocks the thread.
1020
+ await new Promise(r=>requestAnimationFrame(()=>requestAnimationFrame(r)));
1021
+ count.textContent=INDEX.length+" sessions";drawList();showLibrary();
1022
+ }catch(err){
1023
+ // Leave it on screen: a blank window would say nothing at all.
1024
+ bootSay("Could not open your library — "+((err&&err.message)||err),0,0);
1025
+ return;
1026
+ }
1027
+ bootDone();
1028
+ }
1029
+
1030
+ openLibrary();
939
1031
  q.oninput=drawList;
940
1032
 
941
1033
  /* Diwan's collectionTree builds a forest carrying depth, then flattens it
@@ -1328,17 +1420,35 @@ const clearSecret=name=>settings({name,clear:true});
1328
1420
  files, and nobody should find that out by pressing a button once. */
1329
1421
  let PLANNED=null;
1330
1422
 
1423
+ /* Places that really do hold transcripts, checked rather than guessed: each was
1424
+ fetched and converted before it was listed. Gated ones are marked, since a
1425
+ token you have not accepted terms for fails at the door rather than here. */
1331
1426
  const EXAMPLES=[
1427
+ {name:"ATBench",
1428
+ url:"https://huggingface.co/datasets/AI45Research/ATBench/tree/main/ATBench",
1429
+ note:"1,000 labelled agent trajectories"},
1430
+ {name:"ATBench-500",
1431
+ url:"https://huggingface.co/datasets/AI45Research/ATBench/tree/main/ATBench500",
1432
+ note:"the smaller config, 500 trajectories"},
1332
1433
  {name:"SLEIGHT-Bench",
1333
- url:"https://huggingface.co/datasets/sleightbench/SLEIGHT-Bench/tree/main/attacks"},
1334
- {name:"Redwood agent transcripts", url:"s3://rr-agent-transcripts"},
1434
+ url:"https://huggingface.co/datasets/sleightbench/SLEIGHT-Bench/tree/main/attacks",
1435
+ note:"gated — accept the terms on Hugging Face first"},
1436
+ {name:"METR MALT",
1437
+ url:"https://huggingface.co/datasets/metr-evals/malt-transcripts-public/tree/main/data",
1438
+ note:"gated — accept the terms first; shards are large, so pick one"},
1439
+ {name:"METR MALT (vague CoT)",
1440
+ url:"https://huggingface.co/datasets/metr-evals/malt-transcripts-public/tree/main/vague_cot",
1441
+ note:"the same runs with the chain of thought made vaguer"},
1442
+ {name:"Redwood agent transcripts",
1443
+ url:"s3://rr-agent-transcripts",
1444
+ note:"needs an AWS profile set in Settings"},
1335
1445
  ];
1336
1446
 
1337
1447
  function drawExamples(){
1338
1448
  const host=document.getElementById("egs");
1339
1449
  if(!host)return;
1340
1450
  host.innerHTML=`<span>Try:</span>`+EXAMPLES.map((e,i)=>
1341
- `<button onclick="useExample(${i})" title="${esc(e.url)}">${esc(e.name)}</button>`
1451
+ `<button onclick="useExample(${i})" title="${esc(e.note?e.note+" — "+e.url:e.url)}">${esc(e.name)}</button>`
1342
1452
  ).join("");
1343
1453
  }
1344
1454
 
@@ -26,7 +26,7 @@ from urllib.parse import parse_qs, unquote, urlparse
26
26
 
27
27
  from atif_make.atif import ContentPart, Trajectory
28
28
  from . import corpus
29
- from atif_make.archive import extract, is_archive
29
+ from atif_make.container import is_container, open_container
30
30
  from atif_make.convert import convert
31
31
  from transcript_viewer.corpus import Entry, scan
32
32
 
@@ -35,7 +35,33 @@ from . import ai, config, fetch, library
35
35
  # The page is a real .html file rather than a string in here: it is 2,232 lines
36
36
  # of CSS and JavaScript, which is not Python and should not be typed as though
37
37
  # it were. Read once at import, and served from memory.
38
- PAGE = (Path(__file__).parent / "page.html").read_text(encoding="utf-8")
38
+ PAGE_PATH = Path(__file__).parent / "page.html"
39
+
40
+ # What was read, and the state of the file it was read from.
41
+ _page: tuple[float, int, str] | None = None
42
+
43
+
44
+ def page() -> str:
45
+ """The interface, re-read when the file behind it changes.
46
+
47
+ The whole interface is that one file, so editing it and seeing nothing
48
+ change until the server is restarted reads as the edit not having worked.
49
+ Twice now it has. The cost is one stat per page load — not per request, and
50
+ not on anything the page then calls — and for an installed copy, where the
51
+ file never changes, it is a stat that always says the same thing.
52
+ """
53
+ global _page
54
+ try:
55
+ info = PAGE_PATH.stat()
56
+ stamp = (info.st_mtime, info.st_size)
57
+ except OSError:
58
+ # If it cannot be stat'd, what was already read is better than nothing.
59
+ if _page is not None:
60
+ return _page[2]
61
+ raise
62
+ if _page is None or (_page[0], _page[1]) != stamp:
63
+ _page = (stamp[0], stamp[1], PAGE_PATH.read_text(encoding="utf-8"))
64
+ return _page[2]
39
65
 
40
66
 
41
67
  def _point_images_at_server(trajectory: Trajectory, index: str) -> None:
@@ -305,7 +331,7 @@ class _Handler(BaseHTTPRequestHandler):
305
331
  return {
306
332
  e.key
307
333
  for e in self.entries
308
- if (at := _group(e, library.get(e.key).get("source", "")))
334
+ if (at := _group(e, library.load().get(e.key, {}).get("source", "")))
309
335
  and (at == name or at.startswith(f"{name}/"))
310
336
  }
311
337
 
@@ -560,12 +586,13 @@ class _Handler(BaseHTTPRequestHandler):
560
586
  home = corpus.OPENED_ROOT / corpus.content_key(staged)
561
587
  try:
562
588
  home.mkdir(parents=True, exist_ok=True)
563
- if is_archive(staged):
589
+ if is_container(staged):
564
590
  # Keep what is inside, not the container: the logs are what get
565
591
  # indexed, and unpacking here keeps their paths stable and any
566
- # sibling images resolvable. Storing the zip would mean
567
- # re-extracting to a temp directory on every start.
568
- unpacked = extract(staged)
592
+ # sibling images resolvable. Storing the zip — or a benchmark's
593
+ # thousand-run JSON array — would mean opening it again on every
594
+ # start.
595
+ unpacked = open_container(staged)
569
596
  for item in unpacked.iterdir():
570
597
  shutil.move(str(item), home / item.name)
571
598
  target = home
@@ -662,10 +689,10 @@ class _Handler(BaseHTTPRequestHandler):
662
689
  # names the archive happens to use internally.
663
690
  unpacked_from: dict[Path, Path] = {}
664
691
  for item in sorted(home.rglob("*")):
665
- if not item.is_file() or not is_archive(item):
692
+ if not item.is_file() or not is_container(item):
666
693
  continue
667
694
  try:
668
- unpacked = extract(item)
695
+ unpacked = open_container(item)
669
696
  except (ValueError, OSError):
670
697
  continue
671
698
  beside = item.with_suffix("")
@@ -827,10 +854,13 @@ class _Handler(BaseHTTPRequestHandler):
827
854
  url = urlparse(self.path)
828
855
 
829
856
  if url.path == "/":
830
- self._send(PAGE.encode(), "text/html; charset=utf-8")
857
+ self._send(page().encode(), "text/html; charset=utf-8")
831
858
  return
832
859
 
833
860
  if url.path == "/api/index":
861
+ # Read once. Asking per session made a page load parse the whole
862
+ # library as many times as there were sessions in it.
863
+ records = library.load()
834
864
  rows = [
835
865
  {
836
866
  "key": e.key,
@@ -844,7 +874,7 @@ class _Handler(BaseHTTPRequestHandler):
844
874
  "size_bytes": e.size_bytes,
845
875
  "subagents": e.subagents,
846
876
  "session_title": e.session_title,
847
- "group": _group(e, library.get(e.key).get("source", "")),
877
+ "group": _group(e, records.get(e.key, {}).get("source", "")),
848
878
  }
849
879
  for e in self.entries
850
880
  # A downloaded folder the reader deleted, or a drive not mounted
@@ -1165,20 +1165,39 @@ test("the examples fill the field and look straight away", async () => {
1165
1165
  globalThis.__calls.push(url);
1166
1166
  return {ok:true,json:async()=>({nodes:[]})};
1167
1167
  };
1168
- return useExample(1);`);
1169
- assert.strictEqual(fields.urlin.value, "s3://rr-agent-transcripts");
1168
+ const i = EXAMPLES.findIndex(e=>e.url.startsWith("s3://"));
1169
+ globalThis.__url = EXAMPLES[i].url;
1170
+ return useExample(i);`);
1171
+ assert.strictEqual(fields.urlin.value, globalThis.__url);
1170
1172
  assert.ok(
1171
1173
  calls.includes("/api/browse"),
1172
1174
  "it filled the field but did not look",
1173
1175
  );
1174
1176
  });
1175
1177
 
1176
- test("both examples are offered by name", () => {
1177
- const names = run(`return EXAMPLES.map(e=>e.name);`);
1178
- assert.deepStrictEqual(names, ["SLEIGHT-Bench", "Redwood agent transcripts"]);
1179
- const urls = run(`return EXAMPLES.map(e=>e.url);`);
1180
- assert.match(urls[0], /^https:\/\/huggingface\.co\/datasets\/sleightbench\//);
1181
- assert.match(urls[1], /^s3:\/\//);
1178
+ test("every example names a place the fetcher can actually reach", () => {
1179
+ const egs = run(`return EXAMPLES;`);
1180
+ assert.ok(egs.length >= 2, "there should be more than one thing to try");
1181
+ for (const e of egs) {
1182
+ assert.ok(e.name, "an example without a name is a blank button");
1183
+ assert.match(
1184
+ e.url,
1185
+ /^(https:\/\/(huggingface\.co|github\.com)\/|s3:\/\/)/,
1186
+ `${e.name}: not a scheme the fetcher handles`,
1187
+ );
1188
+ }
1189
+ // Both kinds are worth offering: one needs no credentials, one does.
1190
+ assert.ok(egs.some(e => e.url.startsWith("https://huggingface.co/")));
1191
+ assert.ok(egs.some(e => e.url.startsWith("s3://")));
1192
+ });
1193
+
1194
+ test("an example that needs credentials says so", () => {
1195
+ const egs = run(`return EXAMPLES;`);
1196
+ for (const e of egs) {
1197
+ if (e.url.startsWith("s3://") || /sleightbench/.test(e.url)) {
1198
+ assert.ok(e.note, `${e.name} needs something first and does not say so`);
1199
+ }
1200
+ }
1182
1201
  });
1183
1202
 
1184
1203
  test("a selection is measured rather than guessed at", async () => {
@@ -1405,3 +1424,59 @@ test("everything reads as All sessions", () => {
1405
1424
  console.log(`\n${tests.length - failed} passed, ${failed} failed`);
1406
1425
  process.exit(failed ? 1 : 0);
1407
1426
  })();
1427
+
1428
+ test("opening the library reports each stage it goes through", async () => {
1429
+ const said = [];
1430
+ const el = { hidden: true };
1431
+ const bar = {
1432
+ classList: { add() {}, remove() {} },
1433
+ firstElementChild: { style: {} },
1434
+ };
1435
+ const text = { textContent: "" };
1436
+ globalThis.__boot = { boot: el, bootbar: bar, boottext: text, said };
1437
+
1438
+ const seen = run(`
1439
+ document.getElementById = id => globalThis.__boot[id] || {textContent:"",style:{},classList:{add(){},remove(){}}};
1440
+ globalThis.requestAnimationFrame = fn => fn();
1441
+ globalThis.fetch = async () => ({
1442
+ ok: true,
1443
+ headers: { get: () => "0" },
1444
+ body: null,
1445
+ json: async () => ({sessions: [], groups: [], tags: [], ai: {}}),
1446
+ });
1447
+ const stages = [];
1448
+ const realSay = bootSay;
1449
+ bootSay = (what) => { stages.push(what); realSay(what, 0, 0); };
1450
+ return openLibrary().then(() => stages);
1451
+ `);
1452
+
1453
+ return seen.then((stages) => {
1454
+ assert.ok(
1455
+ stages.some((s) => /Looking for your sessions/.test(s)),
1456
+ "it should say it is looking before it has anything",
1457
+ );
1458
+ assert.ok(
1459
+ stages.some((s) => /Drawing/.test(s)),
1460
+ "it should say it is drawing once it has the data",
1461
+ );
1462
+ });
1463
+ });
1464
+
1465
+ test("a failure to open leaves the reason on screen", async () => {
1466
+ const text = { textContent: "" };
1467
+ const el = { hidden: false };
1468
+ globalThis.__boot = {
1469
+ boot: el,
1470
+ bootbar: { classList: { add() {}, remove() {} }, firstElementChild: { style: {} } },
1471
+ boottext: text,
1472
+ };
1473
+
1474
+ await run(`
1475
+ document.getElementById = id => globalThis.__boot[id] || {textContent:"",style:{},classList:{add(){},remove(){}}};
1476
+ globalThis.fetch = async () => ({ ok: false, status: 500 });
1477
+ return openLibrary();
1478
+ `);
1479
+
1480
+ assert.match(text.textContent, /Could not open your library/);
1481
+ assert.strictEqual(el.hidden, false, "a blank window would say nothing at all");
1482
+ });
@@ -433,3 +433,66 @@ def test_only_claude_code_is_asked_for_a_name(tmp_path):
433
433
  entry = corpus.describe(log)
434
434
  assert entry.format.startswith("codex")
435
435
  assert entry.session_title is None
436
+
437
+
438
+ def _atbench_rows(n: int) -> list[dict]:
439
+ """Rows shaped like the real dataset, enough to be recognised as one."""
440
+ return [
441
+ {
442
+ "id": i,
443
+ "label": i % 2,
444
+ "risk_source": "indirect_prompt_injection",
445
+ "failure_mode": "unauthorized_information_disclosure",
446
+ "tool_used": [{"name": "f", "description": "d", "parameters": {}}],
447
+ "contents": [[
448
+ {"role": "user", "content": f"do thing {i}"},
449
+ {"role": "agent", "thought": "", "action": 'Complete{"response": "done"}'},
450
+ ]],
451
+ }
452
+ for i in range(1, n + 1)
453
+ ]
454
+
455
+
456
+ def test_a_file_of_many_transcripts_indexes_as_many(tmp_path):
457
+ """A benchmark published as one JSON array is a container, not a session."""
458
+ source = tmp_path / "test.json"
459
+ source.write_text(json.dumps(_atbench_rows(5)))
460
+
461
+ entries = corpus.scan([source])
462
+ assert len(entries) == 5
463
+ assert {e.format for e in entries} == {"atbench"}
464
+
465
+
466
+ def test_the_container_itself_is_not_offered_as_a_session(tmp_path):
467
+ """It cannot be opened — it holds five transcripts, not one."""
468
+ source = tmp_path / "test.json"
469
+ source.write_text(json.dumps(_atbench_rows(5)))
470
+
471
+ entries = corpus.scan([tmp_path])
472
+ assert len(entries) == 5
473
+ assert str(source) not in {e.path for e in entries}
474
+
475
+
476
+ def test_the_pieces_keep_the_same_paths_across_scans(tmp_path):
477
+ """Paths are recorded in the index, so a scan must not move them."""
478
+ source = tmp_path / "test.json"
479
+ source.write_text(json.dumps(_atbench_rows(4)))
480
+
481
+ first = {e.path for e in corpus.scan([source])}
482
+ second = {e.path for e in corpus.scan([source])}
483
+ assert first == second
484
+ assert all(Path(p).exists() for p in first)
485
+
486
+
487
+ def test_pieces_unpacked_beside_a_download_are_used_as_they_are(tmp_path):
488
+ """A fetch unpacks beside the file; scanning must not split it again."""
489
+ source = tmp_path / "test.json"
490
+ source.write_text(json.dumps(_atbench_rows(3)))
491
+ beside = tmp_path / "test"
492
+ beside.mkdir()
493
+ for i, row in enumerate(_atbench_rows(3), start=1):
494
+ (beside / f"test-{i:05d}-{i}.json").write_text(json.dumps([row]))
495
+
496
+ entries = corpus.scan([tmp_path])
497
+ assert len(entries) == 3
498
+ assert all(str(beside) in e.path for e in entries)
@@ -9,6 +9,7 @@ Deliberately shallow: presence of a marker, not a parse. A test that tried to
9
9
  verify the prose itself would be a worse copy of the code.
10
10
  """
11
11
 
12
+ import re
12
13
  from pathlib import Path
13
14
 
14
15
  import pytest
@@ -40,6 +41,8 @@ CLAIMS = [
40
41
  ("quadratic", "HISTORY_TURNS", "src/transcript_viewer/ai.py"),
41
42
  ("links into the transcript", "function jumpToStep", "src/transcript_viewer/page.html"),
42
43
  ("--extra ai", "ai = [", "pyproject.toml"),
44
+ ("re-read whenever it changes", "def page()", "src/transcript_viewer/viewer.py"),
45
+ ("stat per page load", "PAGE_PATH.stat()", "src/transcript_viewer/viewer.py"),
43
46
  ]
44
47
 
45
48
 
@@ -138,3 +141,48 @@ def test_every_measured_number_is_labelled_as_measured():
138
141
  """Figures in the README came from running something, not from a guess."""
139
142
  for figure in ["3.8 ms", "58 are sent entire", "202 of 202"]:
140
143
  assert figure in README, f"a measured figure went missing: {figure}"
144
+
145
+
146
+ def test_the_endpoint_count_is_the_real_one():
147
+ """A number in prose drifts silently; count it instead of trusting it."""
148
+ words = {
149
+ 12: "twelve", 13: "thirteen", 14: "fourteen", 15: "fifteen",
150
+ 16: "sixteen", 17: "seventeen", 18: "eighteen",
151
+ }
152
+ source = Path("src/transcript_viewer/viewer.py").read_text()
153
+ actual = len(set(re.findall(r'"(/api/[a-z_-]+)"', source)))
154
+ readme = Path("README.md").read_text()
155
+ assert f"{words[actual]} endpoints" in readme, (
156
+ f"the README does not say {words[actual]} endpoints, but there are {actual}"
157
+ )
158
+
159
+
160
+ def test_every_module_appears_in_the_map():
161
+ """A file nobody documented is a file nobody knows to look at."""
162
+ readme = Path("README.md").read_text()
163
+ for path in sorted(Path("src/transcript_viewer").iterdir()):
164
+ if path.name.startswith("_") or path.suffix not in {".py", ".html"}:
165
+ continue
166
+ assert f" {path.name}" in readme, f"{path.name} is missing from the module map"
167
+
168
+
169
+ def test_every_extra_offered_in_the_readme_exists():
170
+ """An install command that names an extra nobody defined simply fails."""
171
+ project = Path("pyproject.toml").read_text()
172
+ defined = set(re.findall(r"^([a-z][a-z0-9-]*) = \[", project, re.MULTILINE))
173
+ readme = Path("README.md").read_text()
174
+ offered = set()
175
+ for group in re.findall(r'transcript-viewer\[([a-z,]+)\]', readme):
176
+ offered.update(group.split(","))
177
+ missing = offered - defined
178
+ assert not missing, f"the README offers extras that do not exist: {missing}"
179
+
180
+
181
+ def test_every_extra_that_exists_is_mentioned():
182
+ """An extra nobody is told about may as well not be there."""
183
+ project = Path("pyproject.toml").read_text()
184
+ block = project.split("[project.optional-dependencies]")[1].split("\n[")[0]
185
+ defined = set(re.findall(r"^([a-z][a-z0-9-]*) = \[", block, re.MULTILINE))
186
+ readme = Path("README.md").read_text()
187
+ unmentioned = {name for name in defined if name not in readme}
188
+ assert not unmentioned, f"extras the README never mentions: {unmentioned}"
@@ -911,3 +911,37 @@ def test_clearing_an_empty_library_is_harmless(server):
911
911
  _delete(server + "/api/library?all=1")
912
912
  status, payload = _delete(server + "/api/library?all=1")
913
913
  assert status == 200 and payload["removed"] == 0
914
+
915
+
916
+ def test_the_page_is_re_read_when_it_changes(tmp_path, monkeypatch):
917
+ """Editing the interface and seeing nothing change reads as a broken edit."""
918
+ from transcript_viewer import viewer
919
+
920
+ page_file = tmp_path / "page.html"
921
+ page_file.write_text("<p>first</p>")
922
+ monkeypatch.setattr(viewer, "PAGE_PATH", page_file)
923
+ monkeypatch.setattr(viewer, "_page", None)
924
+
925
+ assert viewer.page() == "<p>first</p>"
926
+
927
+ # A same-size rewrite is the hard case: only the timestamp separates them.
928
+ page_file.write_text("<p>secnd</p>")
929
+ import os
930
+ stamp = page_file.stat().st_mtime + 10
931
+ os.utime(page_file, (stamp, stamp))
932
+
933
+ assert viewer.page() == "<p>secnd</p>"
934
+
935
+
936
+ def test_an_unreadable_page_serves_what_was_already_read(tmp_path, monkeypatch):
937
+ """A file being written to should not take the interface down mid-save."""
938
+ from transcript_viewer import viewer
939
+
940
+ page_file = tmp_path / "page.html"
941
+ page_file.write_text("<p>kept</p>")
942
+ monkeypatch.setattr(viewer, "PAGE_PATH", page_file)
943
+ monkeypatch.setattr(viewer, "_page", None)
944
+ assert viewer.page() == "<p>kept</p>"
945
+
946
+ page_file.unlink()
947
+ assert viewer.page() == "<p>kept</p>"
@@ -44,8 +44,17 @@ wheels = [
44
44
 
45
45
  [[package]]
46
46
  name = "atif-make"
47
- version = "0.4.0"
48
- source = { git = "https://github.com/jammastergirish/atif-make?branch=main#42672b0ac3ff5b559dee46f0656f352e560b6877" }
47
+ version = "0.5.0"
48
+ source = { registry = "https://pypi.org/simple" }
49
+ sdist = { url = "https://files.pythonhosted.org/packages/60/2c/9e5d1b1ba278fde5135b8066487e029dd0620161b6a220f5cb2a118a0333/atif_make-0.5.0.tar.gz", hash = "sha256:d4c3122a63b553099f0843d5d1c666136b0bccb1c11e8ebd2b4f9c4a9200cb5e", size = 215842, upload-time = "2026-08-23T05:15:04.439Z" }
50
+ wheels = [
51
+ { url = "https://files.pythonhosted.org/packages/00/47/f9a09d10b3d53600c43fa1b22c1d01cb731e628b3572c2e40109de195cfb/atif_make-0.5.0-py3-none-any.whl", hash = "sha256:b52fac5489f5cbc34f947f5d8b21b51a6a04276420d908a96f938e1bf7443791", size = 49769, upload-time = "2026-08-23T05:15:03.265Z" },
52
+ ]
53
+
54
+ [package.optional-dependencies]
55
+ parquet = [
56
+ { name = "pyarrow" },
57
+ ]
49
58
 
50
59
  [[package]]
51
60
  name = "colorama"
@@ -217,6 +226,42 @@ wheels = [
217
226
  { url = "https://files.pythonhosted.org/packages/54/20/4d324d65cc6d9205fabedc306948156824eb9f0ee1633355a8f7ec5c66bf/pluggy-1.6.0-py3-none-any.whl", hash = "sha256:e920276dd6813095e9377c0bc5566d94c932c33b27a3e3945d8389c374dd4746", size = 20538, upload-time = "2025-05-15T12:30:06.134Z" },
218
227
  ]
219
228
 
229
+ [[package]]
230
+ name = "pyarrow"
231
+ version = "25.0.1"
232
+ source = { registry = "https://pypi.org/simple" }
233
+ sdist = { url = "https://files.pythonhosted.org/packages/3d/e3/27f57f80141379d60defe6703eb50a707325706f07fedfd1312c7a751995/pyarrow-25.0.1.tar.gz", hash = "sha256:9150a83248bfed9813ea3c3af74c3856c1984d444aa28e58bf7733b9750ddf6a", size = 1201653, upload-time = "2026-08-10T12:40:53.904Z" }
234
+ wheels = [
235
+ { url = "https://files.pythonhosted.org/packages/a6/e2/9ab15b88cbfac28e16419ce5439ec29234c5172cb8259301b4ba639bdec0/pyarrow-25.0.1-cp312-cp312-macosx_12_0_arm64.whl", hash = "sha256:df961f2e7ae9cf496459259d798652c70625f6c080650d6952f8c04053c58ee9", size = 35861559, upload-time = "2026-08-10T12:38:02.567Z" },
236
+ { url = "https://files.pythonhosted.org/packages/58/79/a0036dbe1eabe1f73127427342f1d99982584c4a2cde2651d6c93499c6f6/pyarrow-25.0.1-cp312-cp312-macosx_12_0_x86_64.whl", hash = "sha256:cc4aa407fde9fc660be3939e49ea31f50f3e9fec17c0ec63159f7711edd3efc9", size = 37628383, upload-time = "2026-08-10T12:38:09.083Z" },
237
+ { url = "https://files.pythonhosted.org/packages/13/49/d93a57d375f4bf0cf82913dd6bb54acafde83dd993be2282c81ac5616cad/pyarrow-25.0.1-cp312-cp312-manylinux_2_28_aarch64.whl", hash = "sha256:4340f0ba6c1d2e13f21658de1d7c662ca2545018568d0030a1e9afca159d87e3", size = 46820190, upload-time = "2026-08-10T12:38:15.458Z" },
238
+ { url = "https://files.pythonhosted.org/packages/60/c9/711ca85d79f1ec98f29a5eae2b051e25b4ecec5de3e3c0e2d5c5dcb15664/pyarrow-25.0.1-cp312-cp312-manylinux_2_28_x86_64.whl", hash = "sha256:5389cdf79447ed1515c9e31620e6e1e2302249564d603f2ad727d4f6d313e4c3", size = 50102437, upload-time = "2026-08-10T12:38:22.487Z" },
239
+ { url = "https://files.pythonhosted.org/packages/80/53/8fb8359ff17cfb6263a1cf3ebf7caec9fe197de118719e84fcb1d0618026/pyarrow-25.0.1-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:d51592cb7561e87877c506113e7adbf1342ab579e6c21f0ef44b8ba41cb74c80", size = 49942424, upload-time = "2026-08-10T12:38:28.755Z" },
240
+ { url = "https://files.pythonhosted.org/packages/e8/83/4e5ae02a9341571b18a6fca380ac7a58ce6ddae7ab3c060208c0a1e79f02/pyarrow-25.0.1-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:6109c94d8b9f3b17a041daca16cacb2f651ad8f1ef70a4232c2c0f37a23da2a8", size = 53144206, upload-time = "2026-08-10T12:38:34.862Z" },
241
+ { url = "https://files.pythonhosted.org/packages/65/ee/197cbf47e49f83e6ebeb946a5259a48a638dea27ac774db42fe78022179d/pyarrow-25.0.1-cp312-cp312-win_amd64.whl", hash = "sha256:8858d7bfc22e3f51529aeaa4077225029724623e4595dc9eff8c793935c34140", size = 27953934, upload-time = "2026-08-10T12:38:39.808Z" },
242
+ { url = "https://files.pythonhosted.org/packages/cc/8d/8f271a7a034c834910ec925d56fa4b29733b1380f5289419f5aaa3b02777/pyarrow-25.0.1-cp313-cp313-macosx_12_0_arm64.whl", hash = "sha256:c7c534ec03c358a76ea3e505e74c1b6aef290af90c444dfd092dbfe23e755b85", size = 35855328, upload-time = "2026-08-10T12:38:45.489Z" },
243
+ { url = "https://files.pythonhosted.org/packages/d2/cd/5bac242f4e841b9971d5eb94fdfe2577e2b70be983e27401e72055786037/pyarrow-25.0.1-cp313-cp313-macosx_12_0_x86_64.whl", hash = "sha256:dda9470024204d7bbf2042b47c6e8a0e47a3eeb8e34405882dfaea6577e0c153", size = 37622415, upload-time = "2026-08-10T12:38:51.107Z" },
244
+ { url = "https://files.pythonhosted.org/packages/63/1f/96d03b4e1506524f7087adb0fd6b2f69f0c9c7aaff1ec36d8030082e15a5/pyarrow-25.0.1-cp313-cp313-manylinux_2_28_aarch64.whl", hash = "sha256:44a9120ce5bd81936b8ab9a88076e3fd47c2c6838e0e43630fed83626aca81d9", size = 46813813, upload-time = "2026-08-10T12:38:57.773Z" },
245
+ { url = "https://files.pythonhosted.org/packages/98/d6/33a411115b61dbfc16ad6ad73e71730f6fea654ee3667673bc53ab0e2fe7/pyarrow-25.0.1-cp313-cp313-manylinux_2_28_x86_64.whl", hash = "sha256:0befcf816e45a1af33ac775a9970b749e4868a230c7372f0ae5e932bee27039f", size = 50104452, upload-time = "2026-08-10T12:39:04.579Z" },
246
+ { url = "https://files.pythonhosted.org/packages/33/ae/b1b97c9ca87f9f9ddbb5230c798df94eccce61bd79b9b45458c69a478588/pyarrow-25.0.1-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:3f89685964f46e4216103c75483aac0c0692a5f72212d7ca835adba5ede56ce3", size = 49951343, upload-time = "2026-08-10T12:39:11.8Z" },
247
+ { url = "https://files.pythonhosted.org/packages/98/9e/a112df5cfd5a68cb1d9fc31cfe38c28d5aec9f10865ce37ecef2e4450873/pyarrow-25.0.1-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:6943e2fe7954d29d84de45d29d34c8dc36ce96570e67d89aa9976e650a4a9138", size = 53144784, upload-time = "2026-08-10T12:39:20.503Z" },
248
+ { url = "https://files.pythonhosted.org/packages/31/24/97e8bd98f1e3b07e2ba08bcdff690674fbe16d69a7d2712cc3884665e615/pyarrow-25.0.1-cp313-cp313-win_amd64.whl", hash = "sha256:31e49a7888fcdf3a835da33ae777f6bb9a866334e5a789282fc26dcf426f7f15", size = 27870159, upload-time = "2026-08-10T12:39:26.161Z" },
249
+ { url = "https://files.pythonhosted.org/packages/36/4c/b525824ad3094076919273cd97db61fb3d78252dee76fa3b8dc8f76774aa/pyarrow-25.0.1-cp314-cp314-macosx_12_0_arm64.whl", hash = "sha256:bf0b672390cdcb640d7288f96b826d71ff4e9abb254a86c89890baf51a29cee6", size = 35885255, upload-time = "2026-08-10T12:39:32.366Z" },
250
+ { url = "https://files.pythonhosted.org/packages/08/62/448bb0e940de41aec31d1a956e63ad9c54afdf122a103cc3ab20c2a3ce33/pyarrow-25.0.1-cp314-cp314-macosx_12_0_x86_64.whl", hash = "sha256:38a9a4b4b9613380e200641891495a56c3d5a98a092db4a870af9975e220471d", size = 37644461, upload-time = "2026-08-10T12:39:38.142Z" },
251
+ { url = "https://files.pythonhosted.org/packages/6e/9a/13587e38bd4806fd218f50fd13b8903fab60588a699ff0c406372e5b4043/pyarrow-25.0.1-cp314-cp314-manylinux_2_28_aarch64.whl", hash = "sha256:0b726ad7e7b669be982b0c71c07fe4b037d654354130da79a7902a669e93a66b", size = 46877146, upload-time = "2026-08-10T12:39:43.722Z" },
252
+ { url = "https://files.pythonhosted.org/packages/8d/61/1c5d1229fa21da4cff5365e41e57177aaac57c563c727f35419b8513d1c1/pyarrow-25.0.1-cp314-cp314-manylinux_2_28_x86_64.whl", hash = "sha256:9171748cdf796972d85a4b60157c279913e242992e350c90c7450182a9838b2a", size = 50131616, upload-time = "2026-08-10T12:39:49.304Z" },
253
+ { url = "https://files.pythonhosted.org/packages/43/20/291e1d65cc0b09aa19f03cf25cf51a2f5fa94b5db315178f2d254ed5cad4/pyarrow-25.0.1-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:b7a296aac7a71fa0886c08e155ddb6c636a50013f801f6178daafa0f9e726188", size = 50008879, upload-time = "2026-08-10T12:39:56.891Z" },
254
+ { url = "https://files.pythonhosted.org/packages/8b/7c/1b7c9ec28e76576337e4f97b31141c9a181b89b6d1d6221e9d8205621a58/pyarrow-25.0.1-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:0fe7c8b6c03969b49c8c66182e4a18e3819ab92d07cfab5d8370c531b9369ef0", size = 53170864, upload-time = "2026-08-10T12:40:04.918Z" },
255
+ { url = "https://files.pythonhosted.org/packages/b7/75/f3d789dc06011a765d14d86bda799cf72ac1d715b6a6edecaa0d73d95062/pyarrow-25.0.1-cp314-cp314-win_amd64.whl", hash = "sha256:f729cfdbd36fd99d543b67a914d2de044c84ebe45be8b34902b299b608c15c8f", size = 28620729, upload-time = "2026-08-10T12:40:51.41Z" },
256
+ { url = "https://files.pythonhosted.org/packages/fc/05/647a8ee6f7c2662feb6921315617bc04dcd6034763fb61b1199720bf6162/pyarrow-25.0.1-cp314-cp314t-macosx_12_0_arm64.whl", hash = "sha256:59a2de54c0cbd954da861eee4d1d330f8e909c45b53455baef696380f2c55033", size = 36130288, upload-time = "2026-08-10T12:40:11.014Z" },
257
+ { url = "https://files.pythonhosted.org/packages/93/f8/c9ee997554d7bea94520667dd1933f109ac1da3ee3556d2b49381e023484/pyarrow-25.0.1-cp314-cp314t-macosx_12_0_x86_64.whl", hash = "sha256:35935cd5de130aa5cf4dea052a63e6bf2e17006c35c3a468194242b9b2bf5956", size = 37762187, upload-time = "2026-08-10T12:40:16.592Z" },
258
+ { url = "https://files.pythonhosted.org/packages/a2/08/a28c01c7fe9e96e8233ce2d13df1d402f4f999f848f51d2daacd6bb4c036/pyarrow-25.0.1-cp314-cp314t-manylinux_2_28_aarch64.whl", hash = "sha256:f3831aaa25c67a99f99dc8b05873cb9d64560390372e2aa197ce9dd4a3f06a44", size = 46888003, upload-time = "2026-08-10T12:40:23.242Z" },
259
+ { url = "https://files.pythonhosted.org/packages/1b/b9/58612e977d28dc58c878448866838369ee8da2f1e7cc8ed2c84b952aafee/pyarrow-25.0.1-cp314-cp314t-manylinux_2_28_x86_64.whl", hash = "sha256:6a1fdfc6659b6b19022f2e50627fb5cf7156a66c46bf4299379955cbe742382a", size = 50079036, upload-time = "2026-08-10T12:40:29.169Z" },
260
+ { url = "https://files.pythonhosted.org/packages/72/13/66e1402dcc860e1dc2760b1e0292c9a569b62b3bccab69def1b3e907d006/pyarrow-25.0.1-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:169d3429d5be7c752125890620f75a60776d38b0035eddae939651640822332e", size = 50040226, upload-time = "2026-08-10T12:40:35.186Z" },
261
+ { url = "https://files.pythonhosted.org/packages/78/10/3f1a5497a7ef732ab0f03ecca3e66d89d9c0f57fdc61b4794c456b781f01/pyarrow-25.0.1-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:119297a6dc197e45d9c6d4415f7814a67ffa36c180d26f68c154c58067ae782d", size = 53149035, upload-time = "2026-08-10T12:40:41.454Z" },
262
+ { url = "https://files.pythonhosted.org/packages/93/c0/37d4a7e8e2f7a6076283673d5298018ca26478b934c6ee369e10505ab32c/pyarrow-25.0.1-cp314-cp314t-win_amd64.whl", hash = "sha256:4288f27577352d608ca08553b0865e4a9b3aa14820c5d95b53337218d609835b", size = 28753071, upload-time = "2026-08-10T12:40:46.623Z" },
263
+ ]
264
+
220
265
  [[package]]
221
266
  name = "pydantic"
222
267
  version = "2.13.4"
@@ -343,7 +388,7 @@ wheels = [
343
388
 
344
389
  [[package]]
345
390
  name = "transcript-viewer"
346
- version = "0.5.0"
391
+ version = "0.6.1"
347
392
  source = { editable = "." }
348
393
  dependencies = [
349
394
  { name = "atif-make" },
@@ -353,6 +398,9 @@ dependencies = [
353
398
  ai = [
354
399
  { name = "anthropic" },
355
400
  ]
401
+ parquet = [
402
+ { name = "atif-make", extra = ["parquet"] },
403
+ ]
356
404
 
357
405
  [package.dev-dependencies]
358
406
  dev = [
@@ -362,9 +410,10 @@ dev = [
362
410
  [package.metadata]
363
411
  requires-dist = [
364
412
  { name = "anthropic", marker = "extra == 'ai'", specifier = ">=0.40" },
365
- { name = "atif-make", git = "https://github.com/jammastergirish/atif-make?branch=main" },
413
+ { name = "atif-make", specifier = ">=0.5.0" },
414
+ { name = "atif-make", extras = ["parquet"], marker = "extra == 'parquet'", specifier = ">=0.5.0" },
366
415
  ]
367
- provides-extras = ["ai"]
416
+ provides-extras = ["parquet", "ai"]
368
417
 
369
418
  [package.metadata.requires-dev]
370
419
  dev = [{ name = "pytest", specifier = ">=8.0" }]