transcripto 0.1.4__tar.gz → 0.1.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: transcripto
3
- Version: 0.1.4
3
+ Version: 0.1.5
4
4
  Summary: Search everything your coding agents ever did, grade your own prompts, and price your decisions. Local, stdlib-only, your data never leaves the machine.
5
5
  Author: Oscar Morke
6
6
  License: MIT
@@ -29,7 +29,7 @@ your disk and never opens a socket.
29
29
  uvx transcripto coach
30
30
  ```
31
31
 
32
- > **which build you got.** `uvx transcripto --version` should print **0.1.3**. If `--version`
32
+ > **which build you got.** `uvx transcripto --version` should print **0.1.4**. If `--version`
33
33
  > is not a recognised flag at all you are on 0.1.1, which predates `trace`, Cursor support and
34
34
  > this README. `uvx --refresh transcripto` forces a fresh resolve past uv's cache.
35
35
  >
@@ -324,10 +324,10 @@ thinking about X across ALL my sessions", in your own words only.
324
324
  $ transcripto find USER-JOURNEY.md # run 2026-08-29
325
325
  USER-JOURNEY.md 4 touches across sessions (3 were writes/edits)
326
326
 
327
- 2026-08-20 WROTE ~/CODE/mountain-of-helicon-main/USER-JOURNEY.md abd9e871
328
- 2026-08-21 WROTE ~/…/Obsidian LIFE/00 Dashboard/suite-user-journey.md 0f845ede
329
- 2026-08-27 read ~/CODE/hack-fleet-ata/docs/USER-JOURNEY.md cddfde29
330
- 2026-08-27 WROTE ~/CODE/hack-fleet-ata/docs/USER-JOURNEY.md cddfde29
327
+ 2026-08-20 WROTE ~/CODE/demo/USER-JOURNEY.md abd9e871
328
+ 2026-08-21 WROTE ~/CODE/demo/docs/onboarding-notes.md 0f845ede
329
+ 2026-08-27 read ~/CODE/demo-api/docs/USER-JOURNEY.md cddfde29
330
+ 2026-08-27 WROTE ~/CODE/demo-api/docs/USER-JOURNEY.md cddfde29
331
331
  ```
332
332
 
333
333
  the file you lost, found across every session you ever ran, with the session id
@@ -11,7 +11,7 @@ your disk and never opens a socket.
11
11
  uvx transcripto coach
12
12
  ```
13
13
 
14
- > **which build you got.** `uvx transcripto --version` should print **0.1.3**. If `--version`
14
+ > **which build you got.** `uvx transcripto --version` should print **0.1.4**. If `--version`
15
15
  > is not a recognised flag at all you are on 0.1.1, which predates `trace`, Cursor support and
16
16
  > this README. `uvx --refresh transcripto` forces a fresh resolve past uv's cache.
17
17
  >
@@ -306,10 +306,10 @@ thinking about X across ALL my sessions", in your own words only.
306
306
  $ transcripto find USER-JOURNEY.md # run 2026-08-29
307
307
  USER-JOURNEY.md 4 touches across sessions (3 were writes/edits)
308
308
 
309
- 2026-08-20 WROTE ~/CODE/mountain-of-helicon-main/USER-JOURNEY.md abd9e871
310
- 2026-08-21 WROTE ~/…/Obsidian LIFE/00 Dashboard/suite-user-journey.md 0f845ede
311
- 2026-08-27 read ~/CODE/hack-fleet-ata/docs/USER-JOURNEY.md cddfde29
312
- 2026-08-27 WROTE ~/CODE/hack-fleet-ata/docs/USER-JOURNEY.md cddfde29
309
+ 2026-08-20 WROTE ~/CODE/demo/USER-JOURNEY.md abd9e871
310
+ 2026-08-21 WROTE ~/CODE/demo/docs/onboarding-notes.md 0f845ede
311
+ 2026-08-27 read ~/CODE/demo-api/docs/USER-JOURNEY.md cddfde29
312
+ 2026-08-27 WROTE ~/CODE/demo-api/docs/USER-JOURNEY.md cddfde29
313
313
  ```
314
314
 
315
315
  the file you lost, found across every session you ever ran, with the session id
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "transcripto"
7
- version = "0.1.4"
7
+ version = "0.1.5"
8
8
  description = "Search everything your coding agents ever did, grade your own prompts, and price your decisions. Local, stdlib-only, your data never leaves the machine."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.9"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: transcripto
3
- Version: 0.1.4
3
+ Version: 0.1.5
4
4
  Summary: Search everything your coding agents ever did, grade your own prompts, and price your decisions. Local, stdlib-only, your data never leaves the machine.
5
5
  Author: Oscar Morke
6
6
  License: MIT
@@ -29,7 +29,7 @@ your disk and never opens a socket.
29
29
  uvx transcripto coach
30
30
  ```
31
31
 
32
- > **which build you got.** `uvx transcripto --version` should print **0.1.3**. If `--version`
32
+ > **which build you got.** `uvx transcripto --version` should print **0.1.4**. If `--version`
33
33
  > is not a recognised flag at all you are on 0.1.1, which predates `trace`, Cursor support and
34
34
  > this README. `uvx --refresh transcripto` forces a fresh resolve past uv's cache.
35
35
  >
@@ -324,10 +324,10 @@ thinking about X across ALL my sessions", in your own words only.
324
324
  $ transcripto find USER-JOURNEY.md # run 2026-08-29
325
325
  USER-JOURNEY.md 4 touches across sessions (3 were writes/edits)
326
326
 
327
- 2026-08-20 WROTE ~/CODE/mountain-of-helicon-main/USER-JOURNEY.md abd9e871
328
- 2026-08-21 WROTE ~/…/Obsidian LIFE/00 Dashboard/suite-user-journey.md 0f845ede
329
- 2026-08-27 read ~/CODE/hack-fleet-ata/docs/USER-JOURNEY.md cddfde29
330
- 2026-08-27 WROTE ~/CODE/hack-fleet-ata/docs/USER-JOURNEY.md cddfde29
327
+ 2026-08-20 WROTE ~/CODE/demo/USER-JOURNEY.md abd9e871
328
+ 2026-08-21 WROTE ~/CODE/demo/docs/onboarding-notes.md 0f845ede
329
+ 2026-08-27 read ~/CODE/demo-api/docs/USER-JOURNEY.md cddfde29
330
+ 2026-08-27 WROTE ~/CODE/demo-api/docs/USER-JOURNEY.md cddfde29
331
331
  ```
332
332
 
333
333
  the file you lost, found across every session you ever ran, with the session id
@@ -5,7 +5,7 @@ Indexes coding-agent transcripts (Claude Code ~/.claude/projects, Codex ~/.codex
5
5
  into a local SQLite full-text index. Stdlib only. No network. Your data never
6
6
  leaves the machine.
7
7
  """
8
- import sys, os, json, glob, re, sqlite3, argparse
8
+ import sys, os, json, glob, re, sqlite3, argparse, math
9
9
  from datetime import datetime, timezone
10
10
 
11
11
  # The tool ships under two console scripts (`transcripto` and `trace`) and is also
@@ -25,7 +25,7 @@ PROG = _prog()
25
25
  # packaging. A stranger who reads the README on GitHub and installs from PyPI can be
26
26
  # holding a different build than the one the README describes, and until this flag
27
27
  # existed there was no way for them to tell which.
28
- VERSION = "0.1.4"
28
+ VERSION = "0.1.5"
29
29
 
30
30
  USAGE = """
31
31
  %(p)s index build / refresh the index (incremental)
@@ -285,22 +285,42 @@ def _day(ts):
285
285
 
286
286
 
287
287
  def cmd_search(args):
288
+ """Your own prompts first, and never call a machine record "you".
289
+
290
+ The label used to be `"you" if role == "user"`. On a real corpus 93,004 rows
291
+ carry role='user' and only 4,218 were typed: the label was wrong on 95.5% of
292
+ them, on the second command a curious reader runs, in a tool whose whole
293
+ argument is the difference between what you wrote and what the machine did.
294
+ `is_human` already existed and every other command used it.
295
+
296
+ Hits you typed sort first, because those are the ones worth finding.
297
+ """
288
298
  con = connect()
289
299
  try:
290
300
  rows = con.execute(
291
- "SELECT m.ts,m.project,m.role,m.cwd,snippet(messages_fts,0,'\033[1m','\033[0m','…',14)"
301
+ "SELECT m.ts,m.project,m.role,m.is_human,m.cwd,"
302
+ " snippet(messages_fts,0,'\033[1m','\033[0m','…',14)"
292
303
  " FROM messages_fts JOIN messages m ON m.id=messages_fts.rowid"
293
- " WHERE messages_fts MATCH ? ORDER BY m.ts DESC LIMIT ?",
304
+ " WHERE messages_fts MATCH ?"
305
+ " ORDER BY m.is_human DESC, m.ts DESC LIMIT ?",
294
306
  (_match(args.query), args.limit)).fetchall()
295
307
  except sqlite3.OperationalError as e:
296
308
  print("search error:", e); return
297
309
  if not rows:
298
310
  print("no matches. try `%s index` first, or broader terms." % PROG); return
299
- for ts, proj, role, cwd, snip in rows:
300
- repo = os.path.basename(cwd) if cwd else proj
301
- who = "you" if role == "user" else "agent"
302
- print("\033[2m%s\033[0m \033[36m%s\033[0m %-5s %s"
303
- % (_day(ts), repo, who, " ".join(snip.split())))
311
+ for ts, proj, role, human, cwd, snip in rows:
312
+ repo = os.path.basename(cwd.rstrip("/")) if cwd else proj
313
+ if human:
314
+ who, colour = "you", "\033[1m"
315
+ elif role == "user":
316
+ # role='user' but nobody typed it: a tool result or a harness injection.
317
+ who, colour = "harness", "\033[2m"
318
+ else:
319
+ who, colour = "agent", "\033[2m"
320
+ print("\033[2m%s\033[0m \033[36m%-16s\033[0m %s%-7s\033[0m %s"
321
+ % (_day(ts), repo[:16], colour, who, " ".join(snip.split())))
322
+ typed = sum(1 for r in rows if r[3])
323
+ print("\n\033[2m%d of %d hits are prompts you typed.\033[0m" % (typed, len(rows)))
304
324
 
305
325
 
306
326
  def _repo(cwd, proj):
@@ -469,41 +489,112 @@ def cmd_trace(args):
469
489
 
470
490
 
471
491
  def cmd_sessions(args):
492
+ """List the sessions YOU typed in. The rest get one line, not eighteen rows.
493
+
494
+ This used to list every session by recency. On a real corpus 1,306 of 1,530
495
+ sessions contain no typed prompt at all, so 8 of 18 rows printed
496
+ "(no prompt you typed)" on the command most likely to be somebody's first
497
+ screenshot. Those rows were not a rendering bug. 85% is the finding, and
498
+ listing machine sessions in date order was burying it.
499
+
500
+ --all brings them back, because sometimes you want the subagent run.
501
+ """
472
502
  con = connect()
503
+ HUMAN = ("is_human=1 AND text!='' AND text NOT LIKE '<command-name>%'"
504
+ " AND text NOT LIKE '<local-command-%'")
505
+ having = "" if getattr(args, "all", False) else (
506
+ " HAVING SUM(CASE WHEN " + HUMAN + " THEN 1 ELSE 0 END) > 0")
473
507
  rows = con.execute(
474
- "SELECT session_id,project,MAX(ts) mx,COUNT(*) FROM messages"
475
- " GROUP BY session_id ORDER BY mx DESC LIMIT ?", (args.limit,)).fetchall()
476
- for sid, proj, mx, cnt in rows:
477
- # Same gate the rest of the tool sells: the opening ask is the first turn
478
- # the OPERATOR typed (is_human=1), not the first `type: user` record. Without
479
- # it the column filled up with `<command-name>/clear</command-name>` — a slash
480
- # command the harness wrote, printed under a heading that says "opening ask".
481
- # Slash commands clear is_human on most harnesses but not all, so the
482
- # wrapper is excluded by name too, and a run of them is skipped rather than
483
- # taken as the title.
508
+ "SELECT session_id,project,MAX(ts) mx,COUNT(*),"
509
+ " SUM(CASE WHEN " + HUMAN + " THEN 1 ELSE 0 END) typed,"
510
+ " MAX(cwd) FROM messages GROUP BY session_id" + having +
511
+ " ORDER BY mx DESC LIMIT ?", (args.limit,)).fetchall()
512
+ for sid, proj, mx, cnt, typed, cwd in rows:
484
513
  t = con.execute("SELECT text FROM messages WHERE session_id=? AND role='user'"
485
- " AND is_human=1 AND text!='' AND text NOT LIKE '<command-name>%'"
486
- " AND text NOT LIKE '<local-command-%' ORDER BY ts LIMIT 1",
487
- (sid,)).fetchone()
488
- title = (t[0][:90].replace("\n", " ") if t else "\033[2m(no prompt you typed)\033[0m")
489
- print("\033[2m%s\033[0m \033[36m%-22s\033[0m %4d msg %s"
490
- % (_day(mx), proj[:22], cnt, title))
514
+ " AND " + HUMAN + " ORDER BY ts LIMIT 1", (sid,)).fetchone()
515
+ title = (t[0][:78].replace("\n", " ") if t
516
+ else "\033[2mno prompt you typed — agent-only run\033[0m")
517
+ where = os.path.basename((cwd or proj or "").rstrip("/")) or (proj or "")
518
+ print("\033[2m%s\033[0m \033[36m%-18s\033[0m %3d typed \033[2m/%-5d\033[0m %s"
519
+ % (_day(mx), where[:18], typed or 0, cnt, title))
520
+ if not rows:
521
+ total = con.execute("SELECT COUNT(DISTINCT session_id) FROM messages").fetchone()[0]
522
+ if not total:
523
+ print("Your index is empty. `%s index` found no transcripts to read." % PROG)
524
+ else:
525
+ print("None of your %s indexed sessions has a prompt you typed in it."
526
+ % f"{total:,}")
527
+ print("\033[2mthat is the finding, not an empty list. `--all` lists them.\033[0m")
528
+ return
529
+ if not getattr(args, "all", False):
530
+ quiet = con.execute("SELECT COUNT(*) FROM (SELECT session_id FROM messages"
531
+ " GROUP BY session_id HAVING SUM(CASE WHEN " + HUMAN +
532
+ " THEN 1 ELSE 0 END)=0)").fetchone()[0]
533
+ ses = con.execute("SELECT COUNT(DISTINCT session_id) FROM messages").fetchone()[0]
534
+ if quiet:
535
+ print("\n\033[2m%s of your %s sessions have no prompt you typed in them (%.0f%%)."
536
+ " They ran\n without you. `--all` lists them.\033[0m"
537
+ % (f"{quiet:,}", f"{ses:,}", 100 * quiet / ses if ses else 0))
491
538
 
492
539
 
493
540
  def cmd_stats(args):
541
+ """Rank what the operator typed, not what the machine emitted.
542
+
543
+ This used to be COUNT(*) FROM messages GROUP BY project. On a real corpus that
544
+ put a single home-directory bucket on top with 110,569 rows and filled eight of
545
+ twelve rows with `wf_*` workflow hashes: the tool ranking its own internals.
546
+ Two things were wrong. It counted machine messages, in a product whose argument
547
+ is that the typed share is the scarce part. And `project` is the harness's
548
+ folder encoding, so every typed prompt from one machine lands in one bucket;
549
+ `cwd` is where work happened and `cost` already keys on it.
550
+ """
494
551
  con = connect()
495
- print("\033[1mprojects by activity\033[0m")
496
- for proj, c in con.execute("SELECT project,COUNT(*) c FROM messages GROUP BY project"
497
- " ORDER BY c DESC LIMIT 12"):
498
- print(" %5d %s" % (c, proj))
499
- print("\n\033[1mmost-written files\033[0m")
552
+ HUMAN = ("is_human=1 AND text!='' AND text NOT LIKE '<command-name>%'"
553
+ " AND text NOT LIKE '<local-command-%'")
554
+ typed = con.execute("SELECT COUNT(*) FROM messages WHERE " + HUMAN).fetchone()[0]
555
+ tot = con.execute("SELECT COUNT(*) FROM messages").fetchone()[0]
556
+ # The empty state is the one a stranger sees first, straight after install.
557
+ # "0 of 0 ... 0.0%" is a ratio over nothing, printed by a tool whose argument
558
+ # is that a rate needs a denominator.
559
+ if not tot:
560
+ print("Your index is empty, so there is no share to show and this prints none.")
561
+ print("\033[2m`%s index` found no transcripts under ~/.claude/projects. either"
562
+ " no agent has\n run on this machine yet, or they are somewhere else."
563
+ "\033[0m" % PROG)
564
+ return
565
+
566
+ # The headline, because it is the whole argument and it used to be a footnote.
567
+ # Name the population. `coach` reads the raw JSONL and reports the same
568
+ # numerator over ~499k RECORDS, which is 0.8%. This reads the index, which
569
+ # keeps ~205k MESSAGES, and the same numerator over that is 2.1%. Both are
570
+ # true of different populations, and a tool that prints two shares for one
571
+ # claim without saying which population it read is doing the thing this tool
572
+ # exists to catch.
573
+ print("\033[1m%s of the %s messages in your index are things you typed. %.1f%%\033[0m"
574
+ % (f"{typed:,}", f"{tot:,}", 100 * typed / tot if tot else 0.0))
575
+ print("\033[2mthe rest is the machine answering. `coach` counts raw transcript records"
576
+ " instead\n of indexed messages, so its share is smaller; same numerator, wider"
577
+ " population.\033[0m")
578
+
579
+ print("\n\033[1mwhere you typed them\033[0m")
580
+ rows = con.execute("SELECT cwd,COUNT(*) n FROM messages WHERE " + HUMAN +
581
+ " AND cwd IS NOT NULL AND cwd!='' GROUP BY cwd"
582
+ " ORDER BY n DESC LIMIT 10").fetchall()
583
+ for cwd, n in rows:
584
+ print(" %5d %s" % (n, os.path.basename(cwd.rstrip("/")) or cwd))
585
+
586
+ print("\n\033[1mfiles your prompts moved most\033[0m")
500
587
  for name, c in con.execute("SELECT name,COUNT(*) c FROM files WHERE action IN('write','edit')"
501
- " GROUP BY name ORDER BY c DESC LIMIT 12"):
588
+ " GROUP BY name ORDER BY c DESC LIMIT 10"):
502
589
  print(" %5d %s" % (c, name))
503
- tot = con.execute("SELECT COUNT(*) FROM messages").fetchone()[0]
590
+
504
591
  ses = con.execute("SELECT COUNT(DISTINCT session_id) FROM messages").fetchone()[0]
592
+ quiet = con.execute("SELECT COUNT(*) FROM (SELECT session_id FROM messages"
593
+ " GROUP BY session_id HAVING SUM(CASE WHEN " + HUMAN +
594
+ " THEN 1 ELSE 0 END)=0)").fetchone()[0]
505
595
  fil = con.execute("SELECT COUNT(DISTINCT path) FROM files").fetchone()[0]
506
- print("\n%d messages · %d sessions · %d distinct files tracked" % (tot, ses, fil))
596
+ print("\n%s sessions · %s of them you never typed in · %s files touched"
597
+ % (f"{ses:,}", f"{quiet:,}", f"{fil:,}"))
507
598
 
508
599
 
509
600
  # ─────────────────────────────────────────────────────────────────────────────
@@ -793,6 +884,7 @@ _STOP = {"the", "a", "an", "to", "of", "in", "into", "on", "for", "and", "or",
793
884
  "case", "cases", "set", "result", "nothing", "empty", "load", "box"}
794
885
 
795
886
  MIN_PATTERN_N = 8 # a habit is only rankable once you have this many episodes
887
+ MAX_CI_WIDTH = 0.30 # and its 95% interval must be narrower than this to be advice
796
888
  MIN_EPISODES_TO_RANK = 30 # below this, refuse to rank the corpus at all — a
797
889
  # rate over a handful of episodes is the exact thing
798
890
  # this tool exists to distrust. Raised from 3 after a
@@ -1558,17 +1650,55 @@ def prompt_patterns(text):
1558
1650
  return tags
1559
1651
 
1560
1652
 
1561
- def rank_patterns(episodes):
1653
+ def _wilson(k, n, z=1.96):
1654
+ """95% Wilson score interval for a proportion. Returns (lo, hi)."""
1655
+ if not n:
1656
+ return 0.0, 0.0
1657
+ p = k / n
1658
+ denom = 1 + z * z / n
1659
+ centre = p + z * z / (2 * n)
1660
+ margin = z * math.sqrt(p * (1 - p) / n + z * z / (4 * n * n))
1661
+ return (centre - margin) / denom, (centre + margin) / denom
1662
+
1663
+
1664
+ def rank_patterns(episodes, baseline=None):
1665
+ """Rank habits by survival, and refuse to rank one we cannot tell from average.
1666
+
1667
+ MIN_PATTERN_N alone is not enough. On a real corpus (2026-09-04, 2244 episodes)
1668
+ a band of n=40 at 55% sat fifth in "do more of these" beside a band of n=815,
1669
+ and its 95% interval was [39.8, 69.3] — it straddled the 47% corpus baseline and
1670
+ overlapped the WORST band. The tool told its user to write vaguer prompts.
1671
+
1672
+ So a habit is ranked only when its interval EXCLUDES the corpus baseline, i.e.
1673
+ when we can actually say it differs from average. Everything else is still
1674
+ measured and still printed, just not under a heading that says "do more of these".
1675
+ A denominator too small to rank does not get ranked.
1676
+ """
1562
1677
  buckets = {}
1563
1678
  for ep in episodes:
1564
1679
  for tag in prompt_patterns(ep["opener"]):
1565
1680
  buckets.setdefault(tag, []).append(ep)
1681
+ if baseline is None:
1682
+ baseline = (sum(1 for e in episodes if e["survived"]) / len(episodes)
1683
+ if episodes else 0.0)
1566
1684
  out = []
1567
1685
  for tag, eps in buckets.items():
1568
1686
  n = len(eps)
1569
1687
  durable = sum(1 for e in eps if e["survived"])
1688
+ lo, hi = _wilson(durable, n)
1570
1689
  out.append({"pattern": tag, "n": n, "survived": durable,
1571
1690
  "survival_rate": durable / n if n else 0.0,
1691
+ "ci_lo": lo, "ci_hi": hi,
1692
+ # Two conditions, both necessary. (1) the interval must EXCLUDE
1693
+ # the corpus baseline, or we cannot say the habit differs from
1694
+ # average. (2) the interval must be narrower than MAX_CI_WIDTH,
1695
+ # because an estimate of "somewhere between 49% and 94%" is not
1696
+ # advice at any position in a list. n>=8 alone allowed both a
1697
+ # baseline-straddling band and a 45pp-wide one to be printed
1698
+ # under "do more of these".
1699
+ "distinguishable": (n >= MIN_PATTERN_N
1700
+ and not (lo <= baseline <= hi)
1701
+ and (hi - lo) <= MAX_CI_WIDTH),
1572
1702
  "rankable": n >= MIN_PATTERN_N})
1573
1703
  out.sort(key=lambda p: (p["survival_rate"], p["n"]), reverse=True)
1574
1704
  return out
@@ -1649,8 +1779,13 @@ def coach(roots=None, harness=None, verified_human=False):
1649
1779
  corrections += corr
1650
1780
  episodes += extract_episodes(rows, source=p, pasted=pasted)
1651
1781
 
1652
- patterns = rank_patterns(episodes)
1653
- rankable = [p for p in patterns if p["rankable"]]
1782
+ _coach_baseline = (sum(1 for e in episodes if e["survived"]) / len(episodes)
1783
+ if episodes else 0.0)
1784
+ patterns = rank_patterns(episodes, baseline=_coach_baseline)
1785
+ rankable = [p for p in patterns if p["distinguishable"]]
1786
+ # Measured, but the interval straddles the corpus baseline, so we cannot say it
1787
+ # differs from average. Printed under its own heading, never as advice.
1788
+ indistinct = [p for p in patterns if p["rankable"] and not p["distinguishable"]]
1654
1789
  durable = sum(1 for e in episodes if e["survived"])
1655
1790
  tiers = {t: sum(1 for e in episodes if e["tier"] == t)
1656
1791
  for t in ("commit", "artifact", "reverted", "none")}
@@ -1673,7 +1808,8 @@ def coach(roots=None, harness=None, verified_human=False):
1673
1808
  "episodes": len(episodes), "durable": durable,
1674
1809
  "durable_rate": round(durable / len(episodes), 3) if episodes else 0.0,
1675
1810
  "tiers": tiers,
1676
- "top_patterns": rankable[:5],
1811
+ "top_patterns": [p for p in rankable
1812
+ if p["survival_rate"] > _coach_baseline][:5],
1677
1813
  # The share block needs named habits by name, and a habit can land in either
1678
1814
  # half of the most/least split depending on corpus. Give it the whole list
1679
1815
  # rather than making it guess which slice to look in.
@@ -1686,12 +1822,18 @@ def coach(roots=None, harness=None, verified_human=False):
1686
1822
  # first outside run was the one that saw it. Slicing from index 5 makes the
1687
1823
  # two lists disjoint by construction at every corpus size, and is
1688
1824
  # byte-identical to the old output once there are >=10 rankable habits.
1689
- "bottom_patterns": rankable[5:][-5:][::-1],
1825
+ # Direction, not list position. A band is "survives least" because it sits
1826
+ # BELOW the corpus baseline, never because it fell outside the first five
1827
+ # slots. Position slicing put a 42% band under "do more of these" the moment
1828
+ # the baseline filter shortened the list.
1829
+ "bottom_patterns": [p for p in rankable
1830
+ if p["survival_rate"] <= _coach_baseline][-5:][::-1],
1690
1831
  "best_prompt": _pick(episodes, True,
1691
1832
  lambda e: (e["score"], -e["corrective_turns"],
1692
1833
  e["assistant_turns"])),
1693
1834
  "worst_prompt": _pick(episodes, False,
1694
1835
  lambda e: (e["corrective_turns"], e["assistant_turns"])),
1836
+ "indistinct_patterns": indistinct,
1695
1837
  "sparse": len(rankable) < 5,
1696
1838
  "rankable_corpus": len(episodes) >= MIN_EPISODES_TO_RANK,
1697
1839
  "proxy": ("survival = a durable Write/Edit or an un-reverted git commit "
@@ -1757,6 +1899,13 @@ def cmd_coach(args):
1757
1899
  print(" %s coach --root <dir> point at a folder of .jsonl transcripts\n" % PROG)
1758
1900
  return
1759
1901
  B, D = "\033[1m", "\033[0m"
1902
+ # The share you typed is the argument the whole tool rests on, and it used to
1903
+ # sit in grey at the bottom under the fold. It is the first thing now.
1904
+ print("\n %s%s of the %s records in your transcripts are things you typed. %.2f%%%s"
1905
+ % (B, f"{r['human_turns']:,}", f"{r['total_records']:,}",
1906
+ r["human_pct"], D))
1907
+ print(" \033[2mthe rest is the machine answering you. this grades the %s.\033[0m"
1908
+ % f"{r['human_turns']:,}")
1760
1909
  print("\n %sYOUR PROMPT HABITS, GRADED%s (offline, your machine only)\n" % (B, D))
1761
1910
  w = r["worst_prompt"]
1762
1911
  if w:
@@ -1778,16 +1927,29 @@ def cmd_coach(args):
1778
1927
  # one exists to catch. So no ranking. The two things below need no sample
1779
1928
  # size — one episode each — and they are the honest half of the output.
1780
1929
  print("\n %sNot enough episodes to rank your habits yet.%s" % (B, D))
1781
- print(" You have %s; habits become rankable at %s. A survival percentage over"
1930
+ print(" You have %s. At %s the tool starts looking, and a habit ranks once its"
1782
1931
  % (r["episodes"], MIN_EPISODES_TO_RANK))
1932
+ print(" own count is large enough to separate it from your average, which is a"
1933
+ "\n bigger number than %s. A survival percentage over"
1934
+ % MIN_EPISODES_TO_RANK)
1783
1935
  print(" a handful of episodes is the number this tool was built to distrust,")
1784
1936
  print(" so it will not print one. Your raw survival and your single best and")
1785
1937
  print(" worst prompt need no sample size — here they are.")
1786
1938
  else:
1787
- if r["sparse"]:
1939
+ if r["sparse"] and r["top_patterns"]:
1788
1940
  print("\n \033[2m(few habits cleared the %s-episode minimum, showing what is "
1789
1941
  "rankable)\033[0m" % MIN_PATTERN_N)
1790
- print("\n %sSURVIVES MOST%s do more of these:" % (B, D))
1942
+ # An empty header under a promise is a claim about rows that do not exist.
1943
+ # The baseline-overlap rule can empty this list entirely on a small corpus,
1944
+ # where no single band's interval clears the user's own average. Say that,
1945
+ # rather than printing "do more of these" over nothing.
1946
+ if not r["top_patterns"]:
1947
+ print("\n %sNo habit beats your own average by enough to call it.%s" % (B, D))
1948
+ print(" \033[2mEvery band's 95%% interval straddles your %d%% baseline. That is a"
1949
+ "\n finding, not a gap: at this corpus size the differences are noise.\033[0m"
1950
+ % round(r["durable_rate"] * 100))
1951
+ else:
1952
+ print("\n %sSURVIVES MOST%s do more of these:" % (B, D))
1791
1953
  for p in r["top_patterns"]:
1792
1954
  print(" %3d%% (%s/%s) %s" % (round(p["survival_rate"] * 100),
1793
1955
  p["survived"], p["n"], p["pattern"]))
@@ -1799,9 +1961,20 @@ def cmd_coach(args):
1799
1961
  for p in r["bottom_patterns"]:
1800
1962
  print(" %3d%% (%s/%s) %s" % (round(p["survival_rate"] * 100),
1801
1963
  p["survived"], p["n"], p["pattern"]))
1802
- else:
1964
+ elif r["top_patterns"]:
1803
1965
  print("\n \033[2m(every rankable habit is listed above; too few to split "
1804
1966
  "into a most/least pair)\033[0m")
1967
+ # Measured, but the 95% interval straddles your own survival rate, so the
1968
+ # honest statement is "cannot tell this apart from your average", not a
1969
+ # ranking. These used to appear under "do more of these".
1970
+ if r.get("indistinct_patterns"):
1971
+ print("\n \033[2mMEASURED, NOT RANKED the 95%% interval straddles your "
1972
+ "%d%% baseline, so these cannot be told apart from your average:\033[0m"
1973
+ % round(r["durable_rate"] * 100))
1974
+ for p in r["indistinct_patterns"]:
1975
+ print(" \033[2m%3d%% (%s/%s) [%d-%d%%] %s\033[0m"
1976
+ % (round(p["survival_rate"] * 100), p["survived"], p["n"],
1977
+ round(p["ci_lo"] * 100), round(p["ci_hi"] * 100), p["pattern"]))
1805
1978
  print("\n \033[2mharness %s · %s transcript(s), %s records · %s typed by you "
1806
1979
  "(%s%%)\033[0m"
1807
1980
  % (r["harness"], r["files"], format(r["total_records"], ","),
@@ -2026,7 +2199,10 @@ def main():
2026
2199
  s = sub.add_parser("trace"); s.add_argument("query"); s.add_argument("-n", "--limit", type=int, default=10)
2027
2200
  s.add_argument("-a", "--all", action="store_true", help="show every file, not the first 8")
2028
2201
  s.set_defaults(fn=cmd_trace)
2029
- s = sub.add_parser("sessions"); s.add_argument("-n", "--limit", type=int, default=30); s.set_defaults(fn=cmd_sessions)
2202
+ s = sub.add_parser("sessions"); s.add_argument("-n", "--limit", type=int, default=30)
2203
+ s.add_argument("--all", action="store_true",
2204
+ help="include sessions you never typed in (85%% of them, on a real corpus)")
2205
+ s.set_defaults(fn=cmd_sessions)
2030
2206
  sub.add_parser("stats").set_defaults(fn=cmd_stats)
2031
2207
  s = sub.add_parser("cost")
2032
2208
  s.add_argument("--days", type=int, default=30, help="window in days (0 = all time)")
File without changes
File without changes