transcript-viewer 0.15.0__tar.gz → 0.16.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. transcript_viewer-0.15.0/README.md → transcript_viewer-0.16.0/PKG-INFO +106 -16
  2. transcript_viewer-0.15.0/PKG-INFO → transcript_viewer-0.16.0/README.md +91 -29
  3. transcript_viewer-0.16.0/postgresql:/postgres:x@localhost/postgres +0 -0
  4. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/pyproject.toml +5 -1
  5. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/src/transcript_viewer/ai.py +8 -1
  6. transcript_viewer-0.16.0/src/transcript_viewer/cache.py +115 -0
  7. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/src/transcript_viewer/cli.py +72 -2
  8. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/src/transcript_viewer/config.py +10 -4
  9. transcript_viewer-0.16.0/src/transcript_viewer/db.py +369 -0
  10. transcript_viewer-0.16.0/src/transcript_viewer/deploy.py +177 -0
  11. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/src/transcript_viewer/fetch.py +20 -0
  12. transcript_viewer-0.16.0/src/transcript_viewer/identity.py +54 -0
  13. transcript_viewer-0.16.0/src/transcript_viewer/ingest.py +119 -0
  14. transcript_viewer-0.16.0/src/transcript_viewer/library.py +301 -0
  15. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/src/transcript_viewer/page.html +237 -71
  16. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/src/transcript_viewer/viewer.py +225 -124
  17. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/tests/page.test.js +355 -58
  18. transcript_viewer-0.16.0/tests/test_authorisation.py +186 -0
  19. transcript_viewer-0.16.0/tests/test_cache.py +85 -0
  20. transcript_viewer-0.16.0/tests/test_db.py +409 -0
  21. transcript_viewer-0.16.0/tests/test_deploy.py +135 -0
  22. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/tests/test_fetch.py +64 -0
  23. transcript_viewer-0.16.0/tests/test_identity.py +76 -0
  24. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/tests/test_library.py +56 -21
  25. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/tests/test_readme.py +5 -1
  26. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/tests/test_style.py +46 -0
  27. transcript_viewer-0.16.0/tests/test_sync.py +95 -0
  28. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/tests/test_viewer.py +29 -22
  29. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/uv.lock +73 -2
  30. transcript_viewer-0.15.0/src/transcript_viewer/library.py +0 -201
  31. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/.coverage +0 -0
  32. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/.github/workflows/publish.yml +0 -0
  33. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/.github/workflows/test.yml +0 -0
  34. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/.gitignore +0 -0
  35. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/TODO.md +0 -0
  36. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/design/Anchor.dc.html +0 -0
  37. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/design/Dense.dc.html +0 -0
  38. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/design/Hover.dc.html +0 -0
  39. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/design/Main.dc.html +0 -0
  40. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/design/Quiet.dc.html +0 -0
  41. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/design/canvas.json +0 -0
  42. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/design/delegation-timeline.html +0 -0
  43. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/src/transcript_viewer/__init__.py +0 -0
  44. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/src/transcript_viewer/corpus.py +0 -0
  45. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/src/transcript_viewer/store.py +0 -0
  46. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/tests/fixtures/big-command.json +0 -0
  47. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/tests/test_ai.py +0 -0
  48. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/tests/test_cli.py +0 -0
  49. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/tests/test_config.py +0 -0
  50. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/tests/test_corpus.py +0 -0
  51. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/tests/test_page.py +0 -0
  52. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/tests/test_store.py +0 -0
  53. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/tests/test_stream.py +0 -0
  54. {transcript_viewer-0.15.0 → transcript_viewer-0.16.0}/tests/test_tidiness.py +0 -0
@@ -1,3 +1,18 @@
1
+ Metadata-Version: 2.5
2
+ Name: transcript-viewer
3
+ Version: 0.16.0
4
+ Summary: Browse agent transcripts in a local, dependency-free web viewer.
5
+ License: MIT
6
+ Requires-Python: >=3.12
7
+ Requires-Dist: atif-make>=0.8.0
8
+ Provides-Extra: ai
9
+ Requires-Dist: anthropic>=0.40; extra == 'ai'
10
+ Provides-Extra: parquet
11
+ Requires-Dist: atif-make[parquet]>=0.8.0; extra == 'parquet'
12
+ Provides-Extra: postgres
13
+ Requires-Dist: psycopg[binary]>=3.1; extra == 'postgres'
14
+ Description-Content-Type: text/markdown
15
+
1
16
  # transcript-viewer
2
17
 
3
18
  Browse agent transcripts in a local web viewer — what Claude Code, Codex and
@@ -10,7 +25,20 @@ all of the reading. What is added here is only the browser interface, so a
10
25
  format this cannot open is a parser missing from `atif-make` rather than
11
26
  anything to change in the viewer.
12
27
 
13
- ## Install
28
+ ## Run it
29
+
30
+ No install, nothing to undo:
31
+
32
+ ```sh
33
+ uvx transcript-viewer # opens on 127.0.0.1, and that is it
34
+ uvx transcript-viewer path/to/session.jsonl
35
+ ```
36
+
37
+ `uvx` fetches it, runs it, and leaves nothing behind. It is the honest way to
38
+ try a tool that is going to read your agent logs — have a look first, and keep
39
+ it only if it earns the disk.
40
+
41
+ To keep it:
14
42
 
15
43
  ```sh
16
44
  uv tool install transcript-viewer # pulls atif-make automatically
@@ -20,15 +48,23 @@ uv tool install "transcript-viewer[ai,parquet]" # + the optional Claude fea
20
48
 
21
49
  The extras are only needed for what they name: `parquet` for a dataset that
22
50
  ships as Parquet rather than JSON, `ai` for the summarise and ask features. The
23
- viewer works without either.
51
+ viewer works without either. With `uvx`, ask for one with
52
+ `uvx --from "transcript-viewer[parquet]" transcript-viewer`.
24
53
 
25
54
  ```sh
26
55
  transcript-viewer # the library, empty on a first run
27
56
  transcript-viewer path/to/session.jsonl # one log
28
57
  transcript-viewer bundle.zip # a bundle someone sent you
29
58
  transcript-viewer --port 8080 --no-open
59
+ transcript-viewer sync s3://bucket/runs # take what is new, and only what is new
30
60
  ```
31
61
 
62
+ Everything is yours and stays here. It binds `127.0.0.1` and nothing else, the
63
+ library lives in `~/.transcript-viewer`, no account is involved, and nothing
64
+ leaves the machine unless you ask it to — a fetch you typed, or a summary you
65
+ pressed for with your own API key. There is no telemetry to turn off because
66
+ there is none to write.
67
+
32
68
  Nothing appears on a first run. Sessions arrive two ways:
33
69
 
34
70
  - **⟳ on the Local folder** finds what Claude Code, Codex and Copilot have
@@ -322,17 +358,17 @@ which of the two is in use, and **Remove** clears the stored one.
322
358
 
323
359
  With no key configured, every AI control is hidden and the endpoint refuses.
324
360
 
325
- Each transcript also has its own **With AI support** switch. Turn it off and
326
- that session's controls disappear — useful when a transcript holds something that should not
327
- leave the machine. The server enforces it too, so a switched-off transcript is
328
- refused even if a request is made directly.
329
-
330
361
  ## Themes
331
362
 
332
- Three, from Diwan: `paper`, `cool` (both light) and `dark`. The control in the
333
- top bar cycles them, as Diwan's own header does, and the choice is remembered
334
- per browser. Note that "light and dark" is really three modes here — two of
335
- Diwan's palettes are light.
363
+ Three from Diwan — `paper`, `cool` (both light) and `dark` — and a pair,
364
+ `redwood` and `redwoodDark`, taken from Redwood Research's own design tokens for
365
+ transcripts that end up in a Redwood deck. Both come from that file as written:
366
+ the dark one is Redwood's own dark palette, not the light one dimmed, which is
367
+ why its green is `#43b184` where the light one's is `#1a7d5c`. The control in
368
+ the top bar cycles them, as Diwan's own header does, and the choice is
369
+ remembered per browser; a remembered name this build no longer has falls back to
370
+ `paper` rather than to an unstyled page. Note that "light and dark" is really
371
+ five modes here, three of them light.
336
372
 
337
373
  ## What it shows
338
374
 
@@ -436,6 +472,56 @@ disagree with it, since the list counted top-level calls while the box counts
436
472
  every subagent at every depth. Steps render 250 at a time so
437
473
  a 8,000-step session stays responsive.
438
474
 
475
+ ## Running it for more than one person
476
+
477
+ Everything above is the whole tool. This section is for the one case it is not:
478
+ a copy of it serving a team rather than sitting on a laptop.
479
+
480
+ Nothing here is on by default, and a copy told nothing behaves exactly as the
481
+ one you have — loopback, nobody signed in, annotations in SQLite beside the
482
+ index. There is a test that walks every route asserting that, because it is what
483
+ keeps this one program rather than two.
484
+
485
+ A `deploy.toml` beside the library says the rest, and every setting can come
486
+ from the environment instead, which is what a container has:
487
+
488
+ ```toml
489
+ [server]
490
+ host = "0.0.0.0"
491
+
492
+ [identity]
493
+ # The header whatever stands in front puts the signed-in person in. Trusted
494
+ # only because it is named here — a header this was not told to expect is one
495
+ # anybody can send.
496
+ header = "x-amzn-oidc-identity"
497
+ admins = ["you@example.com"]
498
+
499
+ [storage]
500
+ database = "postgresql://…" # needs transcript-viewer[postgres]
501
+
502
+ [secrets]
503
+ mode = "server" # the operator's keys, not each person's
504
+
505
+ [ai]
506
+ enabled = "off" # never send a transcript to a model
507
+
508
+ [sources]
509
+ urls = ["s3://bucket/runs"] # what `transcript-viewer sync` follows
510
+ ```
511
+
512
+ The sign-in is not implemented here and should not be: a load balancer or a mesh
513
+ in front does it, and hands over one header. That is why the header must be
514
+ named — if the server can be reached around whatever authenticated the request,
515
+ anyone can claim to be anyone.
516
+
517
+ `postgres` is for this and only this. Annotations live in SQLite otherwise,
518
+ which is in the standard library and needs nothing installed — but two copies of
519
+ the server with two SQLite files means one person's stars exist on one of them
520
+ and not the other, which is a wrong answer rather than a slow one.
521
+
522
+ Labels are shared and stars are personal: a title, a tag or a note is the team's
523
+ one description of a transcript, and what you starred is yours.
524
+
439
525
  ## Developing alongside atif-make
440
526
 
441
527
  The two packages are developed together — a viewer feature usually needs the
@@ -473,16 +559,20 @@ it was handed.
473
559
 
474
560
  ```
475
561
  src/transcript_viewer/
476
- page.html the whole front end — 2,232 lines of HTML, CSS and JavaScript
562
+ page.html the whole interface: one page, no build step, no dependencies
477
563
  viewer.py the HTTP server: sixteen endpoints over the page and the library
478
564
  corpus.py the index — what is on this machine, and where it came from
479
565
  library.py what you decide about a session: title, tags, stars, summaries
566
+ db.py where those decisions are kept — SQLite, or PostgreSQL when shared
567
+ cache.py what has been converted lately, kept only while there is room
480
568
  fetch.py bringing transcripts in from Hugging Face, GitHub, S3 or a link
569
+ ingest.py what happens to them after: unpack, index, record where they came from
481
570
  ai.py the optional Claude-backed features
482
- config.py settings and tokens
571
+ config.py settings and tokens — what a person types in
572
+ deploy.py how this copy is set up — what an operator decided, before anyone arrived
573
+ identity.py who is asking, when anyone is
483
574
  store.py small JSON files under ~/.transcript-viewer, written so a crash cannot truncate them
484
575
  cli.py the command line
485
- page.html the whole interface: one page, no build step, no dependencies
486
576
  ```
487
577
 
488
578
  The page is a file rather than a string inside `viewer.py`, which is where it
@@ -525,8 +615,8 @@ trajectory pane stuck on "Converting…". Those checks live in
525
615
 
526
616
  The AI tests never call the API. Most stub the model call and check the part
527
617
  that matters when it is wrong — that nothing is sent unasked, that a summary is
528
- paid for once, that a stored key never reaches a response, and that a transcript
529
- switched off is refused by the server rather than merely hidden.
618
+ paid for once, that a stored key never reaches a response, and that the endpoint
619
+ refuses outright when no credential is configured.
530
620
 
531
621
  `tests/test_stream.py` is the exception: it drives the one function that does
532
622
  touch the SDK, using a fake client but the SDK's real exception classes, so a
@@ -1,16 +1,3 @@
1
- Metadata-Version: 2.5
2
- Name: transcript-viewer
3
- Version: 0.15.0
4
- Summary: Browse agent transcripts in a local, dependency-free web viewer.
5
- License: MIT
6
- Requires-Python: >=3.12
7
- Requires-Dist: atif-make>=0.8.0
8
- Provides-Extra: ai
9
- Requires-Dist: anthropic>=0.40; extra == 'ai'
10
- Provides-Extra: parquet
11
- Requires-Dist: atif-make[parquet]>=0.8.0; extra == 'parquet'
12
- Description-Content-Type: text/markdown
13
-
14
1
  # transcript-viewer
15
2
 
16
3
  Browse agent transcripts in a local web viewer — what Claude Code, Codex and
@@ -23,7 +10,20 @@ all of the reading. What is added here is only the browser interface, so a
23
10
  format this cannot open is a parser missing from `atif-make` rather than
24
11
  anything to change in the viewer.
25
12
 
26
- ## Install
13
+ ## Run it
14
+
15
+ No install, nothing to undo:
16
+
17
+ ```sh
18
+ uvx transcript-viewer # opens on 127.0.0.1, and that is it
19
+ uvx transcript-viewer path/to/session.jsonl
20
+ ```
21
+
22
+ `uvx` fetches it, runs it, and leaves nothing behind. It is the honest way to
23
+ try a tool that is going to read your agent logs — have a look first, and keep
24
+ it only if it earns the disk.
25
+
26
+ To keep it:
27
27
 
28
28
  ```sh
29
29
  uv tool install transcript-viewer # pulls atif-make automatically
@@ -33,15 +33,23 @@ uv tool install "transcript-viewer[ai,parquet]" # + the optional Claude fea
33
33
 
34
34
  The extras are only needed for what they name: `parquet` for a dataset that
35
35
  ships as Parquet rather than JSON, `ai` for the summarise and ask features. The
36
- viewer works without either.
36
+ viewer works without either. With `uvx`, ask for one with
37
+ `uvx --from "transcript-viewer[parquet]" transcript-viewer`.
37
38
 
38
39
  ```sh
39
40
  transcript-viewer # the library, empty on a first run
40
41
  transcript-viewer path/to/session.jsonl # one log
41
42
  transcript-viewer bundle.zip # a bundle someone sent you
42
43
  transcript-viewer --port 8080 --no-open
44
+ transcript-viewer sync s3://bucket/runs # take what is new, and only what is new
43
45
  ```
44
46
 
47
+ Everything is yours and stays here. It binds `127.0.0.1` and nothing else, the
48
+ library lives in `~/.transcript-viewer`, no account is involved, and nothing
49
+ leaves the machine unless you ask it to — a fetch you typed, or a summary you
50
+ pressed for with your own API key. There is no telemetry to turn off because
51
+ there is none to write.
52
+
45
53
  Nothing appears on a first run. Sessions arrive two ways:
46
54
 
47
55
  - **⟳ on the Local folder** finds what Claude Code, Codex and Copilot have
@@ -335,17 +343,17 @@ which of the two is in use, and **Remove** clears the stored one.
335
343
 
336
344
  With no key configured, every AI control is hidden and the endpoint refuses.
337
345
 
338
- Each transcript also has its own **With AI support** switch. Turn it off and
339
- that session's controls disappear — useful when a transcript holds something that should not
340
- leave the machine. The server enforces it too, so a switched-off transcript is
341
- refused even if a request is made directly.
342
-
343
346
  ## Themes
344
347
 
345
- Three, from Diwan: `paper`, `cool` (both light) and `dark`. The control in the
346
- top bar cycles them, as Diwan's own header does, and the choice is remembered
347
- per browser. Note that "light and dark" is really three modes here — two of
348
- Diwan's palettes are light.
348
+ Three from Diwan — `paper`, `cool` (both light) and `dark` — and a pair,
349
+ `redwood` and `redwoodDark`, taken from Redwood Research's own design tokens for
350
+ transcripts that end up in a Redwood deck. Both come from that file as written:
351
+ the dark one is Redwood's own dark palette, not the light one dimmed, which is
352
+ why its green is `#43b184` where the light one's is `#1a7d5c`. The control in
353
+ the top bar cycles them, as Diwan's own header does, and the choice is
354
+ remembered per browser; a remembered name this build no longer has falls back to
355
+ `paper` rather than to an unstyled page. Note that "light and dark" is really
356
+ five modes here, three of them light.
349
357
 
350
358
  ## What it shows
351
359
 
@@ -449,6 +457,56 @@ disagree with it, since the list counted top-level calls while the box counts
449
457
  every subagent at every depth. Steps render 250 at a time so
450
458
  a 8,000-step session stays responsive.
451
459
 
460
+ ## Running it for more than one person
461
+
462
+ Everything above is the whole tool. This section is for the one case it is not:
463
+ a copy of it serving a team rather than sitting on a laptop.
464
+
465
+ Nothing here is on by default, and a copy told nothing behaves exactly as the
466
+ one you have — loopback, nobody signed in, annotations in SQLite beside the
467
+ index. There is a test that walks every route asserting that, because it is what
468
+ keeps this one program rather than two.
469
+
470
+ A `deploy.toml` beside the library says the rest, and every setting can come
471
+ from the environment instead, which is what a container has:
472
+
473
+ ```toml
474
+ [server]
475
+ host = "0.0.0.0"
476
+
477
+ [identity]
478
+ # The header whatever stands in front puts the signed-in person in. Trusted
479
+ # only because it is named here — a header this was not told to expect is one
480
+ # anybody can send.
481
+ header = "x-amzn-oidc-identity"
482
+ admins = ["you@example.com"]
483
+
484
+ [storage]
485
+ database = "postgresql://…" # needs transcript-viewer[postgres]
486
+
487
+ [secrets]
488
+ mode = "server" # the operator's keys, not each person's
489
+
490
+ [ai]
491
+ enabled = "off" # never send a transcript to a model
492
+
493
+ [sources]
494
+ urls = ["s3://bucket/runs"] # what `transcript-viewer sync` follows
495
+ ```
496
+
497
+ The sign-in is not implemented here and should not be: a load balancer or a mesh
498
+ in front does it, and hands over one header. That is why the header must be
499
+ named — if the server can be reached around whatever authenticated the request,
500
+ anyone can claim to be anyone.
501
+
502
+ `postgres` is for this and only this. Annotations live in SQLite otherwise,
503
+ which is in the standard library and needs nothing installed — but two copies of
504
+ the server with two SQLite files means one person's stars exist on one of them
505
+ and not the other, which is a wrong answer rather than a slow one.
506
+
507
+ Labels are shared and stars are personal: a title, a tag or a note is the team's
508
+ one description of a transcript, and what you starred is yours.
509
+
452
510
  ## Developing alongside atif-make
453
511
 
454
512
  The two packages are developed together — a viewer feature usually needs the
@@ -486,16 +544,20 @@ it was handed.
486
544
 
487
545
  ```
488
546
  src/transcript_viewer/
489
- page.html the whole front end — 2,232 lines of HTML, CSS and JavaScript
547
+ page.html the whole interface: one page, no build step, no dependencies
490
548
  viewer.py the HTTP server: sixteen endpoints over the page and the library
491
549
  corpus.py the index — what is on this machine, and where it came from
492
550
  library.py what you decide about a session: title, tags, stars, summaries
551
+ db.py where those decisions are kept — SQLite, or PostgreSQL when shared
552
+ cache.py what has been converted lately, kept only while there is room
493
553
  fetch.py bringing transcripts in from Hugging Face, GitHub, S3 or a link
554
+ ingest.py what happens to them after: unpack, index, record where they came from
494
555
  ai.py the optional Claude-backed features
495
- config.py settings and tokens
556
+ config.py settings and tokens — what a person types in
557
+ deploy.py how this copy is set up — what an operator decided, before anyone arrived
558
+ identity.py who is asking, when anyone is
496
559
  store.py small JSON files under ~/.transcript-viewer, written so a crash cannot truncate them
497
560
  cli.py the command line
498
- page.html the whole interface: one page, no build step, no dependencies
499
561
  ```
500
562
 
501
563
  The page is a file rather than a string inside `viewer.py`, which is where it
@@ -538,8 +600,8 @@ trajectory pane stuck on "Converting…". Those checks live in
538
600
 
539
601
  The AI tests never call the API. Most stub the model call and check the part
540
602
  that matters when it is wrong — that nothing is sent unasked, that a summary is
541
- paid for once, that a stored key never reaches a response, and that a transcript
542
- switched off is refused by the server rather than merely hidden.
603
+ paid for once, that a stored key never reaches a response, and that the endpoint
604
+ refuses outright when no credential is configured.
543
605
 
544
606
  `tests/test_stream.py` is the exception: it drives the one function that does
545
607
  touch the SDK, using a fake client but the SDK's real exception classes, so a
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "transcript-viewer"
3
- version = "0.15.0"
3
+ version = "0.16.0"
4
4
  description = "Browse agent transcripts in a local, dependency-free web viewer."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.12"
@@ -23,6 +23,10 @@ dependencies = ["atif-make>=0.8.0"]
23
23
  # reader lives in atif-make, which only needs it to split a shard.
24
24
  parquet = ["atif-make[parquet]>=0.8.0"]
25
25
  ai = ["anthropic>=0.40"]
26
+ # Only for a deployment serving more than one person. Two copies of the server
27
+ # with two SQLite files means one person's stars exist on one and not the other,
28
+ # which is a wrong answer rather than a slow one.
29
+ postgres = ["psycopg[binary]>=3.1"]
26
30
 
27
31
  [project.scripts]
28
32
  transcript-viewer = "transcript_viewer.cli:main"
@@ -17,7 +17,7 @@ import re
17
17
  from collections.abc import Iterator
18
18
  from typing import Any
19
19
 
20
- from . import config
20
+ from . import config, deploy
21
21
 
22
22
  MODEL = "claude-opus-5"
23
23
 
@@ -57,7 +57,14 @@ def status() -> tuple[bool, str]:
57
57
 
58
58
  A bare boolean was not enough: a saved key with no SDK installed looks
59
59
  exactly like no key at all, which reads as "the save didn't work".
60
+
61
+ A deployment that has turned this off is answered first and without
62
+ qualification. Everything below is about whether a call *could* be made;
63
+ this is about whether it may be, and the two must not be confused — a
64
+ refusal that reads like a missing dependency invites someone to install it.
60
65
  """
66
+ if deploy.current().ai != "on":
67
+ return False, "this viewer does not send transcripts to a model"
61
68
  try:
62
69
  _client()
63
70
  except Unavailable as exc:
@@ -0,0 +1,115 @@
1
+ """What has been converted lately, kept only while there is room for it.
2
+
3
+ Converting is expensive — seconds, and hundreds of megabytes for the largest
4
+ session here — so the answer is worth keeping. It was kept in a dictionary on
5
+ the handler class that nothing ever removed from, which is fine for an afternoon
6
+ on a laptop and is a leak everywhere else: open enough transcripts and the
7
+ process holds every one of them until it is restarted. In a container with a
8
+ memory limit that is not a leak but a crash.
9
+
10
+ So: least-recently-used, with a budget in bytes rather than in entries. Entries
11
+ are the wrong unit when one of them is a 143 MB rollout and the next is a
12
+ four-step session; a count that holds the small ones comfortably will hold two
13
+ of the large ones and stop.
14
+
15
+ Nothing here persists. A restart re-converts, which costs seconds once — the
16
+ alternative is a store of converted output that grows on disk instead of in
17
+ memory and needs its own eviction, and that is worth building when a restart is
18
+ frequent enough to notice, not before.
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ import json
24
+ import threading
25
+ from collections import OrderedDict
26
+ from typing import Any
27
+
28
+ # Enough for the large sessions people actually open, and small enough that a
29
+ # modest container does not have to be sized around it. Measured against the
30
+ # corpus this was built on: the biggest single converted trajectory is ~34 MB,
31
+ # so this holds a working set of several without ever holding all of them.
32
+ DEFAULT_BUDGET = 256 * 1024 * 1024
33
+
34
+
35
+ def weigh(value: Any) -> int:
36
+ """Roughly how much room something takes.
37
+
38
+ `len(json.dumps(...))` rather than anything exact: the real cost of a nested
39
+ dict is not knowable without walking it, this is within a small factor of
40
+ it, and the number is only ever compared against a budget that is itself a
41
+ judgement. Bytes are counted for anything already bytes, since that is what
42
+ an image is and serialising it would be both wrong and enormous.
43
+ """
44
+ if isinstance(value, (bytes, bytearray, memoryview)):
45
+ return len(value)
46
+ if (
47
+ isinstance(value, dict)
48
+ and value
49
+ and all(isinstance(v, (bytes, bytearray)) for v in value.values())
50
+ ):
51
+ return sum(len(v) for v in value.values())
52
+ try:
53
+ return len(json.dumps(value, default=str))
54
+ except (TypeError, ValueError):
55
+ # Something unserialisable is still taking up room; guess rather than
56
+ # refuse, because refusing would mean not caching it at all.
57
+ return 1024
58
+
59
+
60
+ class Kept:
61
+ """A bounded store of what was asked for recently.
62
+
63
+ Thread-safe because the server is threaded and two requests for the same
64
+ unconverted session arrive together often — both will convert, and both will
65
+ store, and that is cheaper than holding a lock across a conversion.
66
+ """
67
+
68
+ def __init__(self, budget: int = DEFAULT_BUDGET):
69
+ self.budget = budget
70
+ self._items: OrderedDict[str, Any] = OrderedDict()
71
+ self._sizes: dict[str, int] = {}
72
+ self._lock = threading.Lock()
73
+ self.held = 0
74
+
75
+ def get(self, key: str, default: Any = None) -> Any:
76
+ with self._lock:
77
+ if key not in self._items:
78
+ return default
79
+ self._items.move_to_end(key)
80
+ return self._items[key]
81
+
82
+ def __contains__(self, key: str) -> bool:
83
+ with self._lock:
84
+ return key in self._items
85
+
86
+ def __len__(self) -> int:
87
+ with self._lock:
88
+ return len(self._items)
89
+
90
+ def put(self, key: str, value: Any) -> None:
91
+ size = weigh(value)
92
+ with self._lock:
93
+ if key in self._items:
94
+ self.held -= self._sizes.pop(key)
95
+ del self._items[key]
96
+ # Something larger than the whole budget is not cached rather than
97
+ # evicting everything to make room for one thing nobody else can
98
+ # then share.
99
+ if size > self.budget:
100
+ return
101
+ self._items[key] = value
102
+ self._sizes[key] = size
103
+ self.held += size
104
+ while self.held > self.budget and len(self._items) > 1:
105
+ oldest, _ = self._items.popitem(last=False)
106
+ self.held -= self._sizes.pop(oldest)
107
+
108
+ def __setitem__(self, key: str, value: Any) -> None:
109
+ self.put(key, value)
110
+
111
+ def clear(self) -> None:
112
+ with self._lock:
113
+ self._items.clear()
114
+ self._sizes.clear()
115
+ self.held = 0
@@ -6,7 +6,7 @@ import argparse
6
6
  import sys
7
7
  from pathlib import Path
8
8
 
9
- from . import corpus, store
9
+ from . import config, corpus, db, deploy, fetch, ingest, store, viewer
10
10
  from atif_make.container import is_container
11
11
 
12
12
  from .viewer import serve
@@ -34,6 +34,15 @@ def cmd_view(args: argparse.Namespace) -> int:
34
34
  file=sys.stderr,
35
35
  )
36
36
 
37
+ # Annotations used to be one JSON document. Import it here, after any rename
38
+ # has settled, so a library left by either older shape is picked up on the
39
+ # first run that finds it and never looks like it went missing.
40
+ if moved := db.adopt_json():
41
+ print(
42
+ f"transcript-viewer: moved {moved} records into {db.DB_PATH.name}",
43
+ file=sys.stderr,
44
+ )
45
+
37
46
  if args.input:
38
47
  path = Path(args.input).expanduser()
39
48
  if not path.exists():
@@ -70,6 +79,50 @@ def cmd_view(args: argparse.Namespace) -> int:
70
79
  return 0
71
80
 
72
81
 
82
+ def cmd_sync(args: argparse.Namespace) -> int:
83
+ """Ask each source what it has now, and take what is not here yet.
84
+
85
+ A command rather than a thread inside the server: it can be run by a
86
+ schedule, by hand, or by neither, and a fetch that wedges takes nothing
87
+ down with it. The same pipeline the Add dialog uses, so a place followed
88
+ automatically and a place fetched by hand end in the same state.
89
+ """
90
+ urls = args.url or list(deploy.current().sources)
91
+ if not urls:
92
+ print(
93
+ "transcript-viewer: nothing to sync. Name a URL, or list them under "
94
+ f"[sources] in {deploy.CONFIG_PATH}.",
95
+ file=sys.stderr,
96
+ )
97
+ return 2
98
+
99
+ entries = corpus.load()
100
+ into = fetch.destination(None)
101
+ trouble = 0
102
+ for url in urls:
103
+ try:
104
+ plan = fetch.plan(url, config.tokens())
105
+ except fetch.FetchError as exc:
106
+ print(f"transcript-viewer: {url}: {exc}", file=sys.stderr)
107
+ trouble = 1
108
+ continue
109
+
110
+ print(f"{url} — {len(plan.files)} files", file=sys.stderr)
111
+ for frame in ingest.bring(plan, into, entries, viewer._source_of):
112
+ if frame["t"] == "error":
113
+ print(f"transcript-viewer: {url}: {frame['error']}", file=sys.stderr)
114
+ trouble = 1
115
+ elif frame["t"] == "added":
116
+ # `seen` counts what the place holds; `added` what was not here
117
+ # already, which is the only number a schedule cares about.
118
+ print(
119
+ f"{url} — {frame['added']} new of {frame['seen']}",
120
+ file=sys.stderr,
121
+ )
122
+ entries = corpus.load()
123
+ return trouble
124
+
125
+
73
126
  def build_parser() -> argparse.ArgumentParser:
74
127
  parser = argparse.ArgumentParser(
75
128
  prog="transcript-viewer",
@@ -84,8 +137,25 @@ def build_parser() -> argparse.ArgumentParser:
84
137
  return parser
85
138
 
86
139
 
140
+ def sync_parser() -> argparse.ArgumentParser:
141
+ parser = argparse.ArgumentParser(
142
+ prog="transcript-viewer sync",
143
+ description="Take what is new from each source.",
144
+ )
145
+ parser.add_argument("url", nargs="*",
146
+ help="where to look (default: the configured sources)")
147
+ parser.set_defaults(func=cmd_sync)
148
+ return parser
149
+
150
+
87
151
  def main(argv: list[str] | None = None) -> int:
88
- args = build_parser().parse_args(list(sys.argv[1:] if argv is None else argv))
152
+ argv = list(sys.argv[1:] if argv is None else argv)
153
+ # One subcommand, matched before parsing rather than with subparsers: the
154
+ # ordinary form takes a path, and argparse cannot be given both an optional
155
+ # positional and a subcommand without one swallowing the other. A file
156
+ # actually named `sync` is reachable as `./sync`.
157
+ parser = sync_parser() if argv[:1] == ["sync"] else build_parser()
158
+ args = parser.parse_args(argv[1:] if argv[:1] == ["sync"] else argv)
89
159
  return args.func(args)
90
160
 
91
161
 
@@ -20,7 +20,7 @@ from dataclasses import dataclass
20
20
  from pathlib import Path
21
21
  from typing import Any
22
22
 
23
- from . import store
23
+ from . import deploy, store
24
24
 
25
25
  CONFIG_PATH = store.ROOT / "config.json"
26
26
  VERSION = 1
@@ -105,11 +105,17 @@ def secret(name: str, path: Path | None = None) -> str | None:
105
105
 
106
106
  A token typed into settings wins: it is the more explicit, more recent act,
107
107
  and `source()` shows which is in use so the precedence is never a surprise.
108
+
109
+ Unless the operator supplies them. Then the file is not consulted at all —
110
+ not merely outranked. A deployment's credentials are its own, and a key
111
+ left in the file by an earlier local run would otherwise outrank the one
112
+ the operator configured and quietly spend the wrong account.
108
113
  """
109
114
  spec = _spec(name)
110
- value = load(path).get(spec.field)
111
- if isinstance(value, str) and value:
112
- return value
115
+ if deploy.current().secrets != "server":
116
+ value = load(path).get(spec.field)
117
+ if isinstance(value, str) and value:
118
+ return value
113
119
  for variable in spec.env:
114
120
  from_env = os.environ.get(variable)
115
121
  if from_env: