transcripto 0.1.4__tar.gz → 0.1.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {transcripto-0.1.4/transcripto.egg-info → transcripto-0.1.5}/PKG-INFO +6 -6
- {transcripto-0.1.4 → transcripto-0.1.5}/README.md +5 -5
- {transcripto-0.1.4 → transcripto-0.1.5}/pyproject.toml +1 -1
- {transcripto-0.1.4 → transcripto-0.1.5/transcripto.egg-info}/PKG-INFO +6 -6
- {transcripto-0.1.4 → transcripto-0.1.5}/transcripto.py +219 -43
- {transcripto-0.1.4 → transcripto-0.1.5}/LICENSE +0 -0
- {transcripto-0.1.4 → transcripto-0.1.5}/setup.cfg +0 -0
- {transcripto-0.1.4 → transcripto-0.1.5}/transcripto.egg-info/SOURCES.txt +0 -0
- {transcripto-0.1.4 → transcripto-0.1.5}/transcripto.egg-info/dependency_links.txt +0 -0
- {transcripto-0.1.4 → transcripto-0.1.5}/transcripto.egg-info/entry_points.txt +0 -0
- {transcripto-0.1.4 → transcripto-0.1.5}/transcripto.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: transcripto
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.5
|
|
4
4
|
Summary: Search everything your coding agents ever did, grade your own prompts, and price your decisions. Local, stdlib-only, your data never leaves the machine.
|
|
5
5
|
Author: Oscar Morke
|
|
6
6
|
License: MIT
|
|
@@ -29,7 +29,7 @@ your disk and never opens a socket.
|
|
|
29
29
|
uvx transcripto coach
|
|
30
30
|
```
|
|
31
31
|
|
|
32
|
-
> **which build you got.** `uvx transcripto --version` should print **0.1.
|
|
32
|
+
> **which build you got.** `uvx transcripto --version` should print **0.1.4**. If `--version`
|
|
33
33
|
> is not a recognised flag at all you are on 0.1.1, which predates `trace`, Cursor support and
|
|
34
34
|
> this README. `uvx --refresh transcripto` forces a fresh resolve past uv's cache.
|
|
35
35
|
>
|
|
@@ -324,10 +324,10 @@ thinking about X across ALL my sessions", in your own words only.
|
|
|
324
324
|
$ transcripto find USER-JOURNEY.md # run 2026-08-29
|
|
325
325
|
USER-JOURNEY.md 4 touches across sessions (3 were writes/edits)
|
|
326
326
|
|
|
327
|
-
2026-08-20 WROTE ~/CODE/
|
|
328
|
-
2026-08-21 WROTE
|
|
329
|
-
2026-08-27 read ~/CODE/
|
|
330
|
-
2026-08-27 WROTE ~/CODE/
|
|
327
|
+
2026-08-20 WROTE ~/CODE/demo/USER-JOURNEY.md abd9e871
|
|
328
|
+
2026-08-21 WROTE ~/CODE/demo/docs/onboarding-notes.md 0f845ede
|
|
329
|
+
2026-08-27 read ~/CODE/demo-api/docs/USER-JOURNEY.md cddfde29
|
|
330
|
+
2026-08-27 WROTE ~/CODE/demo-api/docs/USER-JOURNEY.md cddfde29
|
|
331
331
|
```
|
|
332
332
|
|
|
333
333
|
the file you lost, found across every session you ever ran, with the session id
|
|
@@ -11,7 +11,7 @@ your disk and never opens a socket.
|
|
|
11
11
|
uvx transcripto coach
|
|
12
12
|
```
|
|
13
13
|
|
|
14
|
-
> **which build you got.** `uvx transcripto --version` should print **0.1.
|
|
14
|
+
> **which build you got.** `uvx transcripto --version` should print **0.1.4**. If `--version`
|
|
15
15
|
> is not a recognised flag at all you are on 0.1.1, which predates `trace`, Cursor support and
|
|
16
16
|
> this README. `uvx --refresh transcripto` forces a fresh resolve past uv's cache.
|
|
17
17
|
>
|
|
@@ -306,10 +306,10 @@ thinking about X across ALL my sessions", in your own words only.
|
|
|
306
306
|
$ transcripto find USER-JOURNEY.md # run 2026-08-29
|
|
307
307
|
USER-JOURNEY.md 4 touches across sessions (3 were writes/edits)
|
|
308
308
|
|
|
309
|
-
2026-08-20 WROTE ~/CODE/
|
|
310
|
-
2026-08-21 WROTE
|
|
311
|
-
2026-08-27 read ~/CODE/
|
|
312
|
-
2026-08-27 WROTE ~/CODE/
|
|
309
|
+
2026-08-20 WROTE ~/CODE/demo/USER-JOURNEY.md abd9e871
|
|
310
|
+
2026-08-21 WROTE ~/CODE/demo/docs/onboarding-notes.md 0f845ede
|
|
311
|
+
2026-08-27 read ~/CODE/demo-api/docs/USER-JOURNEY.md cddfde29
|
|
312
|
+
2026-08-27 WROTE ~/CODE/demo-api/docs/USER-JOURNEY.md cddfde29
|
|
313
313
|
```
|
|
314
314
|
|
|
315
315
|
the file you lost, found across every session you ever ran, with the session id
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "transcripto"
|
|
7
|
-
version = "0.1.
|
|
7
|
+
version = "0.1.5"
|
|
8
8
|
description = "Search everything your coding agents ever did, grade your own prompts, and price your decisions. Local, stdlib-only, your data never leaves the machine."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.9"
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: transcripto
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.5
|
|
4
4
|
Summary: Search everything your coding agents ever did, grade your own prompts, and price your decisions. Local, stdlib-only, your data never leaves the machine.
|
|
5
5
|
Author: Oscar Morke
|
|
6
6
|
License: MIT
|
|
@@ -29,7 +29,7 @@ your disk and never opens a socket.
|
|
|
29
29
|
uvx transcripto coach
|
|
30
30
|
```
|
|
31
31
|
|
|
32
|
-
> **which build you got.** `uvx transcripto --version` should print **0.1.
|
|
32
|
+
> **which build you got.** `uvx transcripto --version` should print **0.1.4**. If `--version`
|
|
33
33
|
> is not a recognised flag at all you are on 0.1.1, which predates `trace`, Cursor support and
|
|
34
34
|
> this README. `uvx --refresh transcripto` forces a fresh resolve past uv's cache.
|
|
35
35
|
>
|
|
@@ -324,10 +324,10 @@ thinking about X across ALL my sessions", in your own words only.
|
|
|
324
324
|
$ transcripto find USER-JOURNEY.md # run 2026-08-29
|
|
325
325
|
USER-JOURNEY.md 4 touches across sessions (3 were writes/edits)
|
|
326
326
|
|
|
327
|
-
2026-08-20 WROTE ~/CODE/
|
|
328
|
-
2026-08-21 WROTE
|
|
329
|
-
2026-08-27 read ~/CODE/
|
|
330
|
-
2026-08-27 WROTE ~/CODE/
|
|
327
|
+
2026-08-20 WROTE ~/CODE/demo/USER-JOURNEY.md abd9e871
|
|
328
|
+
2026-08-21 WROTE ~/CODE/demo/docs/onboarding-notes.md 0f845ede
|
|
329
|
+
2026-08-27 read ~/CODE/demo-api/docs/USER-JOURNEY.md cddfde29
|
|
330
|
+
2026-08-27 WROTE ~/CODE/demo-api/docs/USER-JOURNEY.md cddfde29
|
|
331
331
|
```
|
|
332
332
|
|
|
333
333
|
the file you lost, found across every session you ever ran, with the session id
|
|
@@ -5,7 +5,7 @@ Indexes coding-agent transcripts (Claude Code ~/.claude/projects, Codex ~/.codex
|
|
|
5
5
|
into a local SQLite full-text index. Stdlib only. No network. Your data never
|
|
6
6
|
leaves the machine.
|
|
7
7
|
"""
|
|
8
|
-
import sys, os, json, glob, re, sqlite3, argparse
|
|
8
|
+
import sys, os, json, glob, re, sqlite3, argparse, math
|
|
9
9
|
from datetime import datetime, timezone
|
|
10
10
|
|
|
11
11
|
# The tool ships under two console scripts (`transcripto` and `trace`) and is also
|
|
@@ -25,7 +25,7 @@ PROG = _prog()
|
|
|
25
25
|
# packaging. A stranger who reads the README on GitHub and installs from PyPI can be
|
|
26
26
|
# holding a different build than the one the README describes, and until this flag
|
|
27
27
|
# existed there was no way for them to tell which.
|
|
28
|
-
VERSION = "0.1.
|
|
28
|
+
VERSION = "0.1.5"
|
|
29
29
|
|
|
30
30
|
USAGE = """
|
|
31
31
|
%(p)s index build / refresh the index (incremental)
|
|
@@ -285,22 +285,42 @@ def _day(ts):
|
|
|
285
285
|
|
|
286
286
|
|
|
287
287
|
def cmd_search(args):
|
|
288
|
+
"""Your own prompts first, and never call a machine record "you".
|
|
289
|
+
|
|
290
|
+
The label used to be `"you" if role == "user"`. On a real corpus 93,004 rows
|
|
291
|
+
carry role='user' and only 4,218 were typed: the label was wrong on 95.5% of
|
|
292
|
+
them, on the second command a curious reader runs, in a tool whose whole
|
|
293
|
+
argument is the difference between what you wrote and what the machine did.
|
|
294
|
+
`is_human` already existed and every other command used it.
|
|
295
|
+
|
|
296
|
+
Hits you typed sort first, because those are the ones worth finding.
|
|
297
|
+
"""
|
|
288
298
|
con = connect()
|
|
289
299
|
try:
|
|
290
300
|
rows = con.execute(
|
|
291
|
-
"SELECT m.ts,m.project,m.role,m.cwd,
|
|
301
|
+
"SELECT m.ts,m.project,m.role,m.is_human,m.cwd,"
|
|
302
|
+
" snippet(messages_fts,0,'\033[1m','\033[0m','…',14)"
|
|
292
303
|
" FROM messages_fts JOIN messages m ON m.id=messages_fts.rowid"
|
|
293
|
-
" WHERE messages_fts MATCH ?
|
|
304
|
+
" WHERE messages_fts MATCH ?"
|
|
305
|
+
" ORDER BY m.is_human DESC, m.ts DESC LIMIT ?",
|
|
294
306
|
(_match(args.query), args.limit)).fetchall()
|
|
295
307
|
except sqlite3.OperationalError as e:
|
|
296
308
|
print("search error:", e); return
|
|
297
309
|
if not rows:
|
|
298
310
|
print("no matches. try `%s index` first, or broader terms." % PROG); return
|
|
299
|
-
for ts, proj, role, cwd, snip in rows:
|
|
300
|
-
repo = os.path.basename(cwd) if cwd else proj
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
311
|
+
for ts, proj, role, human, cwd, snip in rows:
|
|
312
|
+
repo = os.path.basename(cwd.rstrip("/")) if cwd else proj
|
|
313
|
+
if human:
|
|
314
|
+
who, colour = "you", "\033[1m"
|
|
315
|
+
elif role == "user":
|
|
316
|
+
# role='user' but nobody typed it: a tool result or a harness injection.
|
|
317
|
+
who, colour = "harness", "\033[2m"
|
|
318
|
+
else:
|
|
319
|
+
who, colour = "agent", "\033[2m"
|
|
320
|
+
print("\033[2m%s\033[0m \033[36m%-16s\033[0m %s%-7s\033[0m %s"
|
|
321
|
+
% (_day(ts), repo[:16], colour, who, " ".join(snip.split())))
|
|
322
|
+
typed = sum(1 for r in rows if r[3])
|
|
323
|
+
print("\n\033[2m%d of %d hits are prompts you typed.\033[0m" % (typed, len(rows)))
|
|
304
324
|
|
|
305
325
|
|
|
306
326
|
def _repo(cwd, proj):
|
|
@@ -469,41 +489,112 @@ def cmd_trace(args):
|
|
|
469
489
|
|
|
470
490
|
|
|
471
491
|
def cmd_sessions(args):
|
|
492
|
+
"""List the sessions YOU typed in. The rest get one line, not eighteen rows.
|
|
493
|
+
|
|
494
|
+
This used to list every session by recency. On a real corpus 1,306 of 1,530
|
|
495
|
+
sessions contain no typed prompt at all, so 8 of 18 rows printed
|
|
496
|
+
"(no prompt you typed)" on the command most likely to be somebody's first
|
|
497
|
+
screenshot. Those rows were not a rendering bug. 85% is the finding, and
|
|
498
|
+
listing machine sessions in date order was burying it.
|
|
499
|
+
|
|
500
|
+
--all brings them back, because sometimes you want the subagent run.
|
|
501
|
+
"""
|
|
472
502
|
con = connect()
|
|
503
|
+
HUMAN = ("is_human=1 AND text!='' AND text NOT LIKE '<command-name>%'"
|
|
504
|
+
" AND text NOT LIKE '<local-command-%'")
|
|
505
|
+
having = "" if getattr(args, "all", False) else (
|
|
506
|
+
" HAVING SUM(CASE WHEN " + HUMAN + " THEN 1 ELSE 0 END) > 0")
|
|
473
507
|
rows = con.execute(
|
|
474
|
-
"SELECT session_id,project,MAX(ts) mx,COUNT(*)
|
|
475
|
-
"
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
# it the column filled up with `<command-name>/clear</command-name>` — a slash
|
|
480
|
-
# command the harness wrote, printed under a heading that says "opening ask".
|
|
481
|
-
# Slash commands clear is_human on most harnesses but not all, so the
|
|
482
|
-
# wrapper is excluded by name too, and a run of them is skipped rather than
|
|
483
|
-
# taken as the title.
|
|
508
|
+
"SELECT session_id,project,MAX(ts) mx,COUNT(*),"
|
|
509
|
+
" SUM(CASE WHEN " + HUMAN + " THEN 1 ELSE 0 END) typed,"
|
|
510
|
+
" MAX(cwd) FROM messages GROUP BY session_id" + having +
|
|
511
|
+
" ORDER BY mx DESC LIMIT ?", (args.limit,)).fetchall()
|
|
512
|
+
for sid, proj, mx, cnt, typed, cwd in rows:
|
|
484
513
|
t = con.execute("SELECT text FROM messages WHERE session_id=? AND role='user'"
|
|
485
|
-
" AND
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
|
|
489
|
-
print("\033[2m%s\033[0m \033[36m%-
|
|
490
|
-
% (_day(mx),
|
|
514
|
+
" AND " + HUMAN + " ORDER BY ts LIMIT 1", (sid,)).fetchone()
|
|
515
|
+
title = (t[0][:78].replace("\n", " ") if t
|
|
516
|
+
else "\033[2mno prompt you typed — agent-only run\033[0m")
|
|
517
|
+
where = os.path.basename((cwd or proj or "").rstrip("/")) or (proj or "")
|
|
518
|
+
print("\033[2m%s\033[0m \033[36m%-18s\033[0m %3d typed \033[2m/%-5d\033[0m %s"
|
|
519
|
+
% (_day(mx), where[:18], typed or 0, cnt, title))
|
|
520
|
+
if not rows:
|
|
521
|
+
total = con.execute("SELECT COUNT(DISTINCT session_id) FROM messages").fetchone()[0]
|
|
522
|
+
if not total:
|
|
523
|
+
print("Your index is empty. `%s index` found no transcripts to read." % PROG)
|
|
524
|
+
else:
|
|
525
|
+
print("None of your %s indexed sessions has a prompt you typed in it."
|
|
526
|
+
% f"{total:,}")
|
|
527
|
+
print("\033[2mthat is the finding, not an empty list. `--all` lists them.\033[0m")
|
|
528
|
+
return
|
|
529
|
+
if not getattr(args, "all", False):
|
|
530
|
+
quiet = con.execute("SELECT COUNT(*) FROM (SELECT session_id FROM messages"
|
|
531
|
+
" GROUP BY session_id HAVING SUM(CASE WHEN " + HUMAN +
|
|
532
|
+
" THEN 1 ELSE 0 END)=0)").fetchone()[0]
|
|
533
|
+
ses = con.execute("SELECT COUNT(DISTINCT session_id) FROM messages").fetchone()[0]
|
|
534
|
+
if quiet:
|
|
535
|
+
print("\n\033[2m%s of your %s sessions have no prompt you typed in them (%.0f%%)."
|
|
536
|
+
" They ran\n without you. `--all` lists them.\033[0m"
|
|
537
|
+
% (f"{quiet:,}", f"{ses:,}", 100 * quiet / ses if ses else 0))
|
|
491
538
|
|
|
492
539
|
|
|
493
540
|
def cmd_stats(args):
|
|
541
|
+
"""Rank what the operator typed, not what the machine emitted.
|
|
542
|
+
|
|
543
|
+
This used to be COUNT(*) FROM messages GROUP BY project. On a real corpus that
|
|
544
|
+
put a single home-directory bucket on top with 110,569 rows and filled eight of
|
|
545
|
+
twelve rows with `wf_*` workflow hashes: the tool ranking its own internals.
|
|
546
|
+
Two things were wrong. It counted machine messages, in a product whose argument
|
|
547
|
+
is that the typed share is the scarce part. And `project` is the harness's
|
|
548
|
+
folder encoding, so every typed prompt from one machine lands in one bucket;
|
|
549
|
+
`cwd` is where work happened and `cost` already keys on it.
|
|
550
|
+
"""
|
|
494
551
|
con = connect()
|
|
495
|
-
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
499
|
-
|
|
552
|
+
HUMAN = ("is_human=1 AND text!='' AND text NOT LIKE '<command-name>%'"
|
|
553
|
+
" AND text NOT LIKE '<local-command-%'")
|
|
554
|
+
typed = con.execute("SELECT COUNT(*) FROM messages WHERE " + HUMAN).fetchone()[0]
|
|
555
|
+
tot = con.execute("SELECT COUNT(*) FROM messages").fetchone()[0]
|
|
556
|
+
# The empty state is the one a stranger sees first, straight after install.
|
|
557
|
+
# "0 of 0 ... 0.0%" is a ratio over nothing, printed by a tool whose argument
|
|
558
|
+
# is that a rate needs a denominator.
|
|
559
|
+
if not tot:
|
|
560
|
+
print("Your index is empty, so there is no share to show and this prints none.")
|
|
561
|
+
print("\033[2m`%s index` found no transcripts under ~/.claude/projects. either"
|
|
562
|
+
" no agent has\n run on this machine yet, or they are somewhere else."
|
|
563
|
+
"\033[0m" % PROG)
|
|
564
|
+
return
|
|
565
|
+
|
|
566
|
+
# The headline, because it is the whole argument and it used to be a footnote.
|
|
567
|
+
# Name the population. `coach` reads the raw JSONL and reports the same
|
|
568
|
+
# numerator over ~499k RECORDS, which is 0.8%. This reads the index, which
|
|
569
|
+
# keeps ~205k MESSAGES, and the same numerator over that is 2.1%. Both are
|
|
570
|
+
# true of different populations, and a tool that prints two shares for one
|
|
571
|
+
# claim without saying which population it read is doing the thing this tool
|
|
572
|
+
# exists to catch.
|
|
573
|
+
print("\033[1m%s of the %s messages in your index are things you typed. %.1f%%\033[0m"
|
|
574
|
+
% (f"{typed:,}", f"{tot:,}", 100 * typed / tot if tot else 0.0))
|
|
575
|
+
print("\033[2mthe rest is the machine answering. `coach` counts raw transcript records"
|
|
576
|
+
" instead\n of indexed messages, so its share is smaller; same numerator, wider"
|
|
577
|
+
" population.\033[0m")
|
|
578
|
+
|
|
579
|
+
print("\n\033[1mwhere you typed them\033[0m")
|
|
580
|
+
rows = con.execute("SELECT cwd,COUNT(*) n FROM messages WHERE " + HUMAN +
|
|
581
|
+
" AND cwd IS NOT NULL AND cwd!='' GROUP BY cwd"
|
|
582
|
+
" ORDER BY n DESC LIMIT 10").fetchall()
|
|
583
|
+
for cwd, n in rows:
|
|
584
|
+
print(" %5d %s" % (n, os.path.basename(cwd.rstrip("/")) or cwd))
|
|
585
|
+
|
|
586
|
+
print("\n\033[1mfiles your prompts moved most\033[0m")
|
|
500
587
|
for name, c in con.execute("SELECT name,COUNT(*) c FROM files WHERE action IN('write','edit')"
|
|
501
|
-
" GROUP BY name ORDER BY c DESC LIMIT
|
|
588
|
+
" GROUP BY name ORDER BY c DESC LIMIT 10"):
|
|
502
589
|
print(" %5d %s" % (c, name))
|
|
503
|
-
|
|
590
|
+
|
|
504
591
|
ses = con.execute("SELECT COUNT(DISTINCT session_id) FROM messages").fetchone()[0]
|
|
592
|
+
quiet = con.execute("SELECT COUNT(*) FROM (SELECT session_id FROM messages"
|
|
593
|
+
" GROUP BY session_id HAVING SUM(CASE WHEN " + HUMAN +
|
|
594
|
+
" THEN 1 ELSE 0 END)=0)").fetchone()[0]
|
|
505
595
|
fil = con.execute("SELECT COUNT(DISTINCT path) FROM files").fetchone()[0]
|
|
506
|
-
print("\n%
|
|
596
|
+
print("\n%s sessions · %s of them you never typed in · %s files touched"
|
|
597
|
+
% (f"{ses:,}", f"{quiet:,}", f"{fil:,}"))
|
|
507
598
|
|
|
508
599
|
|
|
509
600
|
# ─────────────────────────────────────────────────────────────────────────────
|
|
@@ -793,6 +884,7 @@ _STOP = {"the", "a", "an", "to", "of", "in", "into", "on", "for", "and", "or",
|
|
|
793
884
|
"case", "cases", "set", "result", "nothing", "empty", "load", "box"}
|
|
794
885
|
|
|
795
886
|
MIN_PATTERN_N = 8 # a habit is only rankable once you have this many episodes
|
|
887
|
+
MAX_CI_WIDTH = 0.30 # and its 95% interval must be narrower than this to be advice
|
|
796
888
|
MIN_EPISODES_TO_RANK = 30 # below this, refuse to rank the corpus at all — a
|
|
797
889
|
# rate over a handful of episodes is the exact thing
|
|
798
890
|
# this tool exists to distrust. Raised from 3 after a
|
|
@@ -1558,17 +1650,55 @@ def prompt_patterns(text):
|
|
|
1558
1650
|
return tags
|
|
1559
1651
|
|
|
1560
1652
|
|
|
1561
|
-
def
|
|
1653
|
+
def _wilson(k, n, z=1.96):
|
|
1654
|
+
"""95% Wilson score interval for a proportion. Returns (lo, hi)."""
|
|
1655
|
+
if not n:
|
|
1656
|
+
return 0.0, 0.0
|
|
1657
|
+
p = k / n
|
|
1658
|
+
denom = 1 + z * z / n
|
|
1659
|
+
centre = p + z * z / (2 * n)
|
|
1660
|
+
margin = z * math.sqrt(p * (1 - p) / n + z * z / (4 * n * n))
|
|
1661
|
+
return (centre - margin) / denom, (centre + margin) / denom
|
|
1662
|
+
|
|
1663
|
+
|
|
1664
|
+
def rank_patterns(episodes, baseline=None):
|
|
1665
|
+
"""Rank habits by survival, and refuse to rank one we cannot tell from average.
|
|
1666
|
+
|
|
1667
|
+
MIN_PATTERN_N alone is not enough. On a real corpus (2026-09-04, 2244 episodes)
|
|
1668
|
+
a band of n=40 at 55% sat fifth in "do more of these" beside a band of n=815,
|
|
1669
|
+
and its 95% interval was [39.8, 69.3] — it straddled the 47% corpus baseline and
|
|
1670
|
+
overlapped the WORST band. The tool told its user to write vaguer prompts.
|
|
1671
|
+
|
|
1672
|
+
So a habit is ranked only when its interval EXCLUDES the corpus baseline, i.e.
|
|
1673
|
+
when we can actually say it differs from average. Everything else is still
|
|
1674
|
+
measured and still printed, just not under a heading that says "do more of these".
|
|
1675
|
+
A denominator too small to rank does not get ranked.
|
|
1676
|
+
"""
|
|
1562
1677
|
buckets = {}
|
|
1563
1678
|
for ep in episodes:
|
|
1564
1679
|
for tag in prompt_patterns(ep["opener"]):
|
|
1565
1680
|
buckets.setdefault(tag, []).append(ep)
|
|
1681
|
+
if baseline is None:
|
|
1682
|
+
baseline = (sum(1 for e in episodes if e["survived"]) / len(episodes)
|
|
1683
|
+
if episodes else 0.0)
|
|
1566
1684
|
out = []
|
|
1567
1685
|
for tag, eps in buckets.items():
|
|
1568
1686
|
n = len(eps)
|
|
1569
1687
|
durable = sum(1 for e in eps if e["survived"])
|
|
1688
|
+
lo, hi = _wilson(durable, n)
|
|
1570
1689
|
out.append({"pattern": tag, "n": n, "survived": durable,
|
|
1571
1690
|
"survival_rate": durable / n if n else 0.0,
|
|
1691
|
+
"ci_lo": lo, "ci_hi": hi,
|
|
1692
|
+
# Two conditions, both necessary. (1) the interval must EXCLUDE
|
|
1693
|
+
# the corpus baseline, or we cannot say the habit differs from
|
|
1694
|
+
# average. (2) the interval must be narrower than MAX_CI_WIDTH,
|
|
1695
|
+
# because an estimate of "somewhere between 49% and 94%" is not
|
|
1696
|
+
# advice at any position in a list. n>=8 alone allowed both a
|
|
1697
|
+
# baseline-straddling band and a 45pp-wide one to be printed
|
|
1698
|
+
# under "do more of these".
|
|
1699
|
+
"distinguishable": (n >= MIN_PATTERN_N
|
|
1700
|
+
and not (lo <= baseline <= hi)
|
|
1701
|
+
and (hi - lo) <= MAX_CI_WIDTH),
|
|
1572
1702
|
"rankable": n >= MIN_PATTERN_N})
|
|
1573
1703
|
out.sort(key=lambda p: (p["survival_rate"], p["n"]), reverse=True)
|
|
1574
1704
|
return out
|
|
@@ -1649,8 +1779,13 @@ def coach(roots=None, harness=None, verified_human=False):
|
|
|
1649
1779
|
corrections += corr
|
|
1650
1780
|
episodes += extract_episodes(rows, source=p, pasted=pasted)
|
|
1651
1781
|
|
|
1652
|
-
|
|
1653
|
-
|
|
1782
|
+
_coach_baseline = (sum(1 for e in episodes if e["survived"]) / len(episodes)
|
|
1783
|
+
if episodes else 0.0)
|
|
1784
|
+
patterns = rank_patterns(episodes, baseline=_coach_baseline)
|
|
1785
|
+
rankable = [p for p in patterns if p["distinguishable"]]
|
|
1786
|
+
# Measured, but the interval straddles the corpus baseline, so we cannot say it
|
|
1787
|
+
# differs from average. Printed under its own heading, never as advice.
|
|
1788
|
+
indistinct = [p for p in patterns if p["rankable"] and not p["distinguishable"]]
|
|
1654
1789
|
durable = sum(1 for e in episodes if e["survived"])
|
|
1655
1790
|
tiers = {t: sum(1 for e in episodes if e["tier"] == t)
|
|
1656
1791
|
for t in ("commit", "artifact", "reverted", "none")}
|
|
@@ -1673,7 +1808,8 @@ def coach(roots=None, harness=None, verified_human=False):
|
|
|
1673
1808
|
"episodes": len(episodes), "durable": durable,
|
|
1674
1809
|
"durable_rate": round(durable / len(episodes), 3) if episodes else 0.0,
|
|
1675
1810
|
"tiers": tiers,
|
|
1676
|
-
"top_patterns": rankable
|
|
1811
|
+
"top_patterns": [p for p in rankable
|
|
1812
|
+
if p["survival_rate"] > _coach_baseline][:5],
|
|
1677
1813
|
# The share block needs named habits by name, and a habit can land in either
|
|
1678
1814
|
# half of the most/least split depending on corpus. Give it the whole list
|
|
1679
1815
|
# rather than making it guess which slice to look in.
|
|
@@ -1686,12 +1822,18 @@ def coach(roots=None, harness=None, verified_human=False):
|
|
|
1686
1822
|
# first outside run was the one that saw it. Slicing from index 5 makes the
|
|
1687
1823
|
# two lists disjoint by construction at every corpus size, and is
|
|
1688
1824
|
# byte-identical to the old output once there are >=10 rankable habits.
|
|
1689
|
-
"
|
|
1825
|
+
# Direction, not list position. A band is "survives least" because it sits
|
|
1826
|
+
# BELOW the corpus baseline, never because it fell outside the first five
|
|
1827
|
+
# slots. Position slicing put a 42% band under "do more of these" the moment
|
|
1828
|
+
# the baseline filter shortened the list.
|
|
1829
|
+
"bottom_patterns": [p for p in rankable
|
|
1830
|
+
if p["survival_rate"] <= _coach_baseline][-5:][::-1],
|
|
1690
1831
|
"best_prompt": _pick(episodes, True,
|
|
1691
1832
|
lambda e: (e["score"], -e["corrective_turns"],
|
|
1692
1833
|
e["assistant_turns"])),
|
|
1693
1834
|
"worst_prompt": _pick(episodes, False,
|
|
1694
1835
|
lambda e: (e["corrective_turns"], e["assistant_turns"])),
|
|
1836
|
+
"indistinct_patterns": indistinct,
|
|
1695
1837
|
"sparse": len(rankable) < 5,
|
|
1696
1838
|
"rankable_corpus": len(episodes) >= MIN_EPISODES_TO_RANK,
|
|
1697
1839
|
"proxy": ("survival = a durable Write/Edit or an un-reverted git commit "
|
|
@@ -1757,6 +1899,13 @@ def cmd_coach(args):
|
|
|
1757
1899
|
print(" %s coach --root <dir> point at a folder of .jsonl transcripts\n" % PROG)
|
|
1758
1900
|
return
|
|
1759
1901
|
B, D = "\033[1m", "\033[0m"
|
|
1902
|
+
# The share you typed is the argument the whole tool rests on, and it used to
|
|
1903
|
+
# sit in grey at the bottom under the fold. It is the first thing now.
|
|
1904
|
+
print("\n %s%s of the %s records in your transcripts are things you typed. %.2f%%%s"
|
|
1905
|
+
% (B, f"{r['human_turns']:,}", f"{r['total_records']:,}",
|
|
1906
|
+
r["human_pct"], D))
|
|
1907
|
+
print(" \033[2mthe rest is the machine answering you. this grades the %s.\033[0m"
|
|
1908
|
+
% f"{r['human_turns']:,}")
|
|
1760
1909
|
print("\n %sYOUR PROMPT HABITS, GRADED%s (offline, your machine only)\n" % (B, D))
|
|
1761
1910
|
w = r["worst_prompt"]
|
|
1762
1911
|
if w:
|
|
@@ -1778,16 +1927,29 @@ def cmd_coach(args):
|
|
|
1778
1927
|
# one exists to catch. So no ranking. The two things below need no sample
|
|
1779
1928
|
# size — one episode each — and they are the honest half of the output.
|
|
1780
1929
|
print("\n %sNot enough episodes to rank your habits yet.%s" % (B, D))
|
|
1781
|
-
print(" You have %s
|
|
1930
|
+
print(" You have %s. At %s the tool starts looking, and a habit ranks once its"
|
|
1782
1931
|
% (r["episodes"], MIN_EPISODES_TO_RANK))
|
|
1932
|
+
print(" own count is large enough to separate it from your average, which is a"
|
|
1933
|
+
"\n bigger number than %s. A survival percentage over"
|
|
1934
|
+
% MIN_EPISODES_TO_RANK)
|
|
1783
1935
|
print(" a handful of episodes is the number this tool was built to distrust,")
|
|
1784
1936
|
print(" so it will not print one. Your raw survival and your single best and")
|
|
1785
1937
|
print(" worst prompt need no sample size — here they are.")
|
|
1786
1938
|
else:
|
|
1787
|
-
if r["sparse"]:
|
|
1939
|
+
if r["sparse"] and r["top_patterns"]:
|
|
1788
1940
|
print("\n \033[2m(few habits cleared the %s-episode minimum, showing what is "
|
|
1789
1941
|
"rankable)\033[0m" % MIN_PATTERN_N)
|
|
1790
|
-
|
|
1942
|
+
# An empty header under a promise is a claim about rows that do not exist.
|
|
1943
|
+
# The baseline-overlap rule can empty this list entirely on a small corpus,
|
|
1944
|
+
# where no single band's interval clears the user's own average. Say that,
|
|
1945
|
+
# rather than printing "do more of these" over nothing.
|
|
1946
|
+
if not r["top_patterns"]:
|
|
1947
|
+
print("\n %sNo habit beats your own average by enough to call it.%s" % (B, D))
|
|
1948
|
+
print(" \033[2mEvery band's 95%% interval straddles your %d%% baseline. That is a"
|
|
1949
|
+
"\n finding, not a gap: at this corpus size the differences are noise.\033[0m"
|
|
1950
|
+
% round(r["durable_rate"] * 100))
|
|
1951
|
+
else:
|
|
1952
|
+
print("\n %sSURVIVES MOST%s do more of these:" % (B, D))
|
|
1791
1953
|
for p in r["top_patterns"]:
|
|
1792
1954
|
print(" %3d%% (%s/%s) %s" % (round(p["survival_rate"] * 100),
|
|
1793
1955
|
p["survived"], p["n"], p["pattern"]))
|
|
@@ -1799,9 +1961,20 @@ def cmd_coach(args):
|
|
|
1799
1961
|
for p in r["bottom_patterns"]:
|
|
1800
1962
|
print(" %3d%% (%s/%s) %s" % (round(p["survival_rate"] * 100),
|
|
1801
1963
|
p["survived"], p["n"], p["pattern"]))
|
|
1802
|
-
|
|
1964
|
+
elif r["top_patterns"]:
|
|
1803
1965
|
print("\n \033[2m(every rankable habit is listed above; too few to split "
|
|
1804
1966
|
"into a most/least pair)\033[0m")
|
|
1967
|
+
# Measured, but the 95% interval straddles your own survival rate, so the
|
|
1968
|
+
# honest statement is "cannot tell this apart from your average", not a
|
|
1969
|
+
# ranking. These used to appear under "do more of these".
|
|
1970
|
+
if r.get("indistinct_patterns"):
|
|
1971
|
+
print("\n \033[2mMEASURED, NOT RANKED the 95%% interval straddles your "
|
|
1972
|
+
"%d%% baseline, so these cannot be told apart from your average:\033[0m"
|
|
1973
|
+
% round(r["durable_rate"] * 100))
|
|
1974
|
+
for p in r["indistinct_patterns"]:
|
|
1975
|
+
print(" \033[2m%3d%% (%s/%s) [%d-%d%%] %s\033[0m"
|
|
1976
|
+
% (round(p["survival_rate"] * 100), p["survived"], p["n"],
|
|
1977
|
+
round(p["ci_lo"] * 100), round(p["ci_hi"] * 100), p["pattern"]))
|
|
1805
1978
|
print("\n \033[2mharness %s · %s transcript(s), %s records · %s typed by you "
|
|
1806
1979
|
"(%s%%)\033[0m"
|
|
1807
1980
|
% (r["harness"], r["files"], format(r["total_records"], ","),
|
|
@@ -2026,7 +2199,10 @@ def main():
|
|
|
2026
2199
|
s = sub.add_parser("trace"); s.add_argument("query"); s.add_argument("-n", "--limit", type=int, default=10)
|
|
2027
2200
|
s.add_argument("-a", "--all", action="store_true", help="show every file, not the first 8")
|
|
2028
2201
|
s.set_defaults(fn=cmd_trace)
|
|
2029
|
-
s = sub.add_parser("sessions"); s.add_argument("-n", "--limit", type=int, default=30)
|
|
2202
|
+
s = sub.add_parser("sessions"); s.add_argument("-n", "--limit", type=int, default=30)
|
|
2203
|
+
s.add_argument("--all", action="store_true",
|
|
2204
|
+
help="include sessions you never typed in (85%% of them, on a real corpus)")
|
|
2205
|
+
s.set_defaults(fn=cmd_sessions)
|
|
2030
2206
|
sub.add_parser("stats").set_defaults(fn=cmd_stats)
|
|
2031
2207
|
s = sub.add_parser("cost")
|
|
2032
2208
|
s.add_argument("--days", type=int, default=30, help="window in days (0 = all time)")
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|