tldr-experts 0.3.1 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/CHANGELOG.md +2396 -0
  2. package/README.md +133 -17
  3. package/dist/hooks/answer-capture.js +7 -8
  4. package/dist/hooks/budget-gate.js +83 -16
  5. package/dist/hooks/{chunk-zdxgragg.js → chunk-14zn51kh.js} +171 -16
  6. package/dist/hooks/{chunk-xpxe94qe.js → chunk-3w55tp71.js} +75 -27
  7. package/dist/hooks/{chunk-afamdvyn.js → chunk-5556vjt5.js} +1 -1
  8. package/dist/hooks/{chunk-ytvmc5ns.js → chunk-889tybxc.js} +26 -28
  9. package/dist/hooks/{chunk-9gb21660.js → chunk-a5dq2dcp.js} +54 -14
  10. package/dist/hooks/{chunk-ztczwtj0.js → chunk-a6rpj2cp.js} +835 -16
  11. package/dist/hooks/{chunk-3g61yg59.js → chunk-bvm6vjrt.js} +1 -1
  12. package/dist/hooks/{chunk-0z27twdk.js → chunk-hcrbr430.js} +10 -5
  13. package/dist/hooks/{chunk-3t91gvpp.js → chunk-nadqsr3w.js} +8 -2
  14. package/dist/hooks/{chunk-458wgg9j.js → chunk-q8d3sff9.js} +86 -11
  15. package/dist/hooks/{chunk-9kkm6q0t.js → chunk-qw73rdbr.js} +35 -3
  16. package/dist/hooks/{chunk-ybacnpxd.js → chunk-sae7sqty.js} +5 -0
  17. package/dist/hooks/{chunk-s5qsb4k6.js → chunk-v1c1hpb8.js} +1 -1
  18. package/dist/hooks/claim-sources.js +32 -21
  19. package/dist/hooks/dod-gate.js +7 -6
  20. package/dist/hooks/no-reask.js +9 -9
  21. package/dist/hooks/session-start.js +46 -21
  22. package/dist/hooks/statusline.js +8 -9
  23. package/dist/tldrx.js +19667 -10978
  24. package/package.json +4 -2
  25. package/plugin/.claude-plugin/plugin.json +2 -2
  26. package/plugin/skills/tldrx/SKILL.md +1 -1
  27. package/stages/build/stage.yml +2 -1
  28. package/stages/plan/stage.md +11 -0
  29. package/stages/watch/stage.md +4 -0
  30. package/templates/expert.md +13 -1
  31. package/templates/watcher.md +8 -1
  32. package/workflows/bugfix.yml +8 -4
  33. package/workflows/docs.yml +8 -4
  34. package/workflows/feature.yml +8 -4
  35. package/workflows/hotfix.yml +8 -4
  36. package/workflows/integration.yml +8 -4
  37. package/workflows/migration.yml +8 -4
  38. package/workflows/performance.yml +8 -4
  39. package/workflows/prototype.yml +8 -4
  40. package/workflows/refactor.yml +8 -4
  41. package/workflows/security-patch.yml +8 -4
  42. package/workflows/spike.yml +8 -4
  43. package/workflows/upgrade.yml +8 -4
  44. package/dist/hooks/chunk-sznsenee.js +0 -503
  45. package/templates/epic.md +0 -38
  46. package/templates/story.md +0 -55
package/README.md CHANGED
@@ -1,8 +1,8 @@
1
1
  # tldr-experts
2
2
 
3
- [![npm](https://img.shields.io/npm/v/tldr-experts?label=npm%20tldr-experts)](https://www.npmjs.com/package/tldr-experts) [![ci](https://github.com/ederwii/tldr-experts/actions/workflows/ci.yml/badge.svg)](https://github.com/ederwii/tldr-experts/actions/workflows/ci.yml) ![status](https://img.shields.io/badge/status-alpha-orange)
3
+ [![npm](https://img.shields.io/npm/v/tldr-experts?label=npm%20tldr-experts)](https://www.npmjs.com/package/tldr-experts) [![ci](https://github.com/ederwii/tldr-experts/actions/workflows/ci.yml/badge.svg)](https://github.com/ederwii/tldr-experts/actions/workflows/ci.yml) ![status](https://img.shields.io/badge/status-beta-blue)
4
4
 
5
- **A lightweight, file-based AI development workflow.** Open source, tool-agnostic in design, piloted on Claude Code. **Alpha:** every command is implemented and verified by running it; interfaces may change without notice, and `tldrx --help` is the authoritative surface.
5
+ **An evidence-first, file-based AI development framework: five stages, a gate on every one, and every claim cited or refused.** Open source, tool-agnostic in design, piloted on Claude Code. **Beta:** every command is implemented and verified by running it, the `version: 1` file formats only grow from here, and `tldrx --help` is the authoritative command surface.
6
6
 
7
7
  One loop — *Investigate → Handoff → Interview → Gate* — five phases, **what · how · plan · build ·
8
8
  watch**, one stage per command, each stopping at a gate you own; the files ARE the state, the
@@ -13,10 +13,6 @@ do: a command that cannot do the thing exits non-zero and says which thing.
13
13
 
14
14
  ## Quick start
15
15
 
16
- > **Not on npm yet.** Every published version was unpublished on 2026-08-29 (`npm view tldr-experts
17
- > version` → `E404 Unpublished`) and there is no `v0.3.0` tag, so the `npm i -g` line 404s until
18
- > `scripts/release.sh 0.3.0` is run. Until then: clone and `bun link`, or `bun <repo>/bin/tldrx.ts <cmd>`.
19
-
20
16
  ```bash
21
17
  npm i -g tldr-experts # installs `tldrx` (short) and `tldr-experts` (same binary)
22
18
  cd your-project
@@ -25,6 +21,16 @@ tldrx init # detect repos, map the code, write .tldrx/, ask only
25
21
  tldrx install --claude # write the skill, hooks and status line into ./.claude/
26
22
  ```
27
23
 
24
+ Later: **`tldrx update`** pulls the newest published version and prints the CHANGELOG between the
25
+ one you had and the one you now have. Any command will tell you, in one line, when there is a newer
26
+ one — off the hot path, cached for a day, silent when it cannot reach the registry, and never in
27
+ `--json` output or during a hook. Turn it off with `TLDRX_UPDATE_CHECK=off`.
28
+
29
+ **Never used it before?** `tldrx learn` teaches the loop by running it: eight chapters, ~15 minutes,
30
+ in a throwaway sandbox with a toy repo and a stand-in agent. Every command in it is the real one —
31
+ `init`, `run new`, `next`, `approve`, a Build that cuts a branch and runs a real DoD — so nothing it
32
+ shows you can drift from what the binary does, and it costs $0.00 and touches nothing you own.
33
+
28
34
  Then open Claude Code there and type **`/tldrx`**. It runs `tldrx status`, finds what is already
29
35
  waiting on you — unanswered setup questions, a proposed split nobody decided, a run waiting on a gate,
30
36
  an expert no stage can lean on yet — and walks you through it one item at a time, asking every decision
@@ -39,6 +45,79 @@ tldrx run auto # `next`, over and over, until something actually need
39
45
  dependencies, so an installed `tldrx` needs only Node; Bun builds it. Full walkthrough:
40
46
  [`docs/guide/01-quick-start.md`](docs/guide/01-quick-start.md).
41
47
 
48
+ ## Trying it: three ways to run
49
+
50
+ `tldrx run auto` and `tldrx run attend host` read like two speeds of the same thing. They are
51
+ opposites and they do not compose. **`auto` is an engine, not a lock**: a headless loop in which
52
+ the *framework* spawns a metered sub-agent, stage after stage. **`attend host` is a lock, not an
53
+ engine**: it sets one field, spends nothing and runs no stage, and from then on the framework never
54
+ spawns on that run — every turn is a `--prepare` / `--commit` handshake with a session you drive.
55
+ `run auto` on an attended run is refused outright (exit `1`); a bare `tldrx next` there exits `4`
56
+ and names the `--prepare` command instead.
57
+
58
+ | | who executes each turn | what a turn costs | where it stops |
59
+ |---|---|---|---|
60
+ | `tldrx run auto` | the framework — `claude -p`, spawned stage after stage | metered per spawn, rolled up by `tldrx cost` | the first human gate or open question (`4`), stage failure (`5`), ceiling (`2`) |
61
+ | `tldrx run attend host`, driven from a session | your session's own sub-agents | host-billed; the framework records `cost_usd: null, metered: false` | every turn — `--prepare` writes the bundle, `--commit` settles it |
62
+ | the same, under a **mandate** | your session's own sub-agents | host-billed | a new product decision, a ceiling raise, a boundary exit — nothing else |
63
+
64
+ - **A small run you were going to watch anyway** → `run auto`. One command, and it stops the moment it needs you.
65
+ - **A Claude Code session already open, and you care about cost or quality** → `run attend host`, driven from it: the context is warm, the turns are host-billed, and the framework writes the Build reviewer's bundle rather than spawning a second reader beside one you are already paying for.
66
+ - **Overnight, hands off, and you still want the adversarial check** → `run attend host` plus a mandate, below.
67
+ - **CI or cron** → `run auto`. It is the only one of the three with no session behind it.
68
+
69
+ ### Overnight, with the checking kept
70
+
71
+ Two commands and a prompt — and the prompt now ships with the package:
72
+
73
+ ```bash
74
+ tldrx drive --unattended # print the mandate; paste it into the session that drives the run
75
+ tldrx drive --attended # the same disciplines, but every gate stays yours to sign
76
+ ```
77
+
78
+ `tldrx drive` needs no workspace, opens no run and writes nothing: it prints the discipline the
79
+ first real runs were driven by — the three-role protocol (developer → a **fresh** adversarial
80
+ reviewer, never the author → the host verifying both in the code, not in their reports), evidence
81
+ labelled `measured` / `inferred` / `assumed`, product questions parked rather than decided, the
82
+ reviewer calibrated to the story's stakes, and the cost declared once. The two modes differ in
83
+ exactly two places: who drives the turns, and who may close a gate.
84
+
85
+ The rest of this section is what that mandate says, in the shape you would type it by hand.
86
+
87
+ ```bash
88
+ tldrx run new payments --scope feature --budget 25 \
89
+ --attended-by host --gates what:agent,plan:agent,build:agent,watch:agent
90
+ tldrx run attend host 260101-payments # or flip a run that is already open
91
+ ```
92
+
93
+ `--gates` **replaces the workflow's gates wholesale**, and a stage you leave out of the list becomes
94
+ `auto` — so name every gate you want signed. Then, in the session, the mandate:
95
+
96
+ > Act as my unattended verification gate on run `260101-payments`, until it reaches its last gate.
97
+ >
98
+ > Drive every stage yourself — `tldrx next --prepare 260101-payments`, then
99
+ > `tldrx next --commit 260101-payments` — dispatching your own sub-agents for the turns. The
100
+ > framework must never spawn.
101
+ >
102
+ > For every build story, run an INDEPENDENT adversarial review through the `--review` handshake:
103
+ > `tldrx next --prepare --review`, one read-only sub-agent over the diff, then
104
+ > `tldrx next --commit --review`. Its job is to find what the developer got wrong, not to agree
105
+ > with it.
106
+ >
107
+ > Approve a gate only after you have checked it yourself — that the citations resolve, that every
108
+ > touched path is one this run declared, and that the diff matches the stories it claims to
109
+ > implement — and write that check down as evidence: `tldrx gate template`, fill it in, then
110
+ > `tldrx approve --as-agent`.
111
+ >
112
+ > Interrupt me ONLY for a new product decision, a budget-ceiling raise, or work that has to go
113
+ > outside the declared boundary. Everything else you decide, and log.
114
+ >
115
+ > Never push. The final merge is mine.
116
+
117
+ The whole chapter — the three switches, what "never spawns" is enforced by, the review handshake,
118
+ the fix list, the evidence note and the four fallthroughs:
119
+ [10 Unattended mode](docs/guide/10-unattended-mode.md).
120
+
42
121
  ## How much human is in the loop
43
122
 
44
123
  Every stage ends at a gate; what you choose is **who closes it**. `human` waits for `tldrx approve`.
@@ -78,7 +157,14 @@ Those are the shipped defaults, and every scope keeps at least one human gate. O
78
157
  signs something it should not have, `tldrx reject --stage <phase>/<stage> --note "…"` revokes it, moves
79
158
  the cursor back and marks the later stages `stale`. When it is one BUILD STORY you disagree with — a
80
159
  story two reviewers refused, which is terminal for the rest of the run —
81
- `tldrx story reopen <id> --note "…"` gives that one story another run of attempts and nothing else.
160
+ `tldrx story reopen <id> --note "…"` gives that one story another run of attempts and nothing else;
161
+ for one named defect in a story already `done`, `--for-fix` opens a fix round instead — no attempt
162
+ consumed, the same DoD and the same reviewer, one open round at a time.
163
+ To move who may close a gate after `run new` froze it, `tldrx run gates set <stage>:<policy> --note "…"`
164
+ is the only sanctioned way, and it records the old→new value with your reason.
165
+ When you fix `.tldrx/workspace.yml` mid-run and the approved stories still cite the old command strings,
166
+ `tldrx plan sync-dod` rewrites just their dod lines — renames followed, removed commands dropped, and
167
+ anything with no ancestor in the file's history flagged rather than guessed at.
82
168
  What an auto gate cannot do: [`docs/guide/03-runs-and-gates.md`](docs/guide/03-runs-and-gates.md).
83
169
 
84
170
  ## What you see while it runs
@@ -127,7 +213,7 @@ context 83.7 KB of 160.0 KB (~23.8k tok, 12% of sonnet's ~200.0k window)
127
213
  ```
128
214
 
129
215
  Over `prompt_max_bytes` the stage is **refused** (exit 2) before anything spawns; `max_reads` stops
130
- the sub-agent at a read ceiling; `--effort` changes what a turn costs. What `--max-budget-usd` does is
216
+ the sub-agent at a read ceiling; `--effort` changes what a turn costs. What `--max-usd` does is
131
217
  end a run *after* the turn it is already in — measured, one turn spent **$5.15** against a **$1.50**
132
218
  ceiling — so size the prompt for the money you are willing to lose. Afterwards `tldrx cost [--all]` adds up what was actually charged, per attempt, per stage, per
133
219
  run, read off `agent.result` events and nothing else. Retries are never merged — a retry is
@@ -137,22 +223,40 @@ Details: [`docs/guide/06-budgets-and-cost.md`](docs/guide/06-budgets-and-cost.md
137
223
 
138
224
  ## Several runs
139
225
 
140
- With several runs open and no id, every run-targeting command **refuses rather than guessing**,
141
- exits `2`, and lists them — `tldrx next: 3 runs are open — pass one:`. That means "you left off
142
- the id", not "it broke". Pass a positional `<run>` on `next`, `run status`, `cost`, `replay` and
143
- `retro`; `--run <id>` on the rest. `tldrx run status` with several open lists them all, exit `0`.
226
+ With several runs open and no id, a run-targeting command **refuses rather than guessing** and
227
+ lists them — `tldrx next: 3 runs are open — pass one:`. That means "you left off the id", not "it
228
+ broke". Two of them refuse differently and it is worth knowing which: `tldrx run status` is not a
229
+ refusal at all — it lists every open run and exits `0`, and it is the screen you read to find the
230
+ id the others want — and `tldrx cost` refuses at exit `1`, not `2`.
231
+
232
+ Most run-targeting commands take the id either way, a positional `<run>` or `--run <id>`: `next`,
233
+ `cost`, `note`, `gate template`, `questions`, `budget show`, `ship`, `tickets`, and `run attend` ·
234
+ `status` · `estimate` · `auto` · `unlock` · `cancel`. `replay` and `retro` take the positional only
235
+ — `--run` there is an unknown flag. `approve`, `reject`, `answer`, `interview`, `plan`,
236
+ `story reopen`, `watch` and `run gates set` take `--run <id>` only.
237
+
238
+ `tldrx retro --all` goes the other way: it reads **every** run in the workspace and prints one
239
+ table of what keeps catching you — finding class × count × how many runs × one example with its
240
+ citation — mined from the review logs, the fix lists, `retro.md` and the `story.reopened` reasons.
241
+ Strictly read-only: it writes nothing, anywhere.
144
242
 
145
243
  ## What to commit
146
244
 
147
245
  **Both `.tldrx/` and `tldrx-work/`.** The files are the state — the map, the facts, the questions and
148
246
  their answers, `run.yml`, `budget.yml`, `events.jsonl`, the handoffs, the plan — so a teammate who clones
149
- the repo gets the run. The block `tldrx init` appends to `.gitignore` excludes five paths and nothing
150
- else, because those five are machine-local or regenerated: `.tldrx/graphify-out/`, `.tldrx/cache/`,
151
- `.tldrx/worktrees/`, `tldrx-work/*/.lock`, `tldrx-work/*/.agent/`.
247
+ the repo gets the run. The block `tldrx init` appends to `.gitignore` excludes eight paths and nothing
248
+ else, because those eight are machine-local, regenerated, or a backup git already holds the history
249
+ of: `.tldrx/graphify-out/`, `.tldrx/cache/`, `.tldrx/worktrees/`, `tldrx-work/*/.lock`,
250
+ `tldrx-work/*/.agent/`, `tldrx-work/*/*.bak`, `.tldrx/memory/*.bak` and
251
+ `.claude/settings.json.bak-tldrx-*`.
152
252
 
153
253
  ## Documentation
154
254
 
155
- The guide, in `docs/guide/`: [1 Quick start](docs/guide/01-quick-start.md) ·
255
+ **[The documentation site](https://ederwii.github.io/tldr-experts/)** is the place to start if you have
256
+ never used this: a landing page, a Quickstart and one short page per concept, written for a reader
257
+ rather than for an agent. Source in [`docs-site/`](docs-site/).
258
+
259
+ The reference guide, in `docs/guide/`: [1 Quick start](docs/guide/01-quick-start.md) ·
156
260
  [2 The loop](docs/guide/02-the-loop.md) (the four steps, what a stage file controls, the two execution modes) ·
157
261
  [3 Runs and gates](docs/guide/03-runs-and-gates.md) (`run new`→`retro`, gate policy, `run auto`, unlock/cancel, dashboard, tickets) ·
158
262
  [4 Experts](docs/guide/04-experts.md) (loading rules, role experts, training, levels) ·
@@ -166,6 +270,12 @@ Design docs: [`docs/concept.md`](docs/concept.md) (why) · [`docs/spec.md`](docs
166
270
  open decisions) · [`docs/ROADMAP.md`](docs/ROADMAP.md) (next) · [`CHANGELOG.md`](CHANGELOG.md) (shipped) ·
167
271
  [`docs/dashboard-model.md`](docs/dashboard-model.md).
168
272
 
273
+ Contributing: **[`CONTRIBUTING.md`](CONTRIBUTING.md)** — the loop a change goes through, the four gates
274
+ and what CI actually runs, the red-first test rules, and
275
+ [how to contribute a model-provider config](CONTRIBUTING.md#contributing-a-model-provider-config)
276
+ (the `TLDRX_CLAUDE_BIN` seam, the `stream-json` transcript contract, and what a generic provider
277
+ would have to supply).
278
+
169
279
  ## Releases and status tags
170
280
 
171
281
  Install name is **`tldr-experts`**; it installs two commands, **`tldrx`** (short) and `tldr-experts` (same binary).
@@ -175,6 +285,8 @@ back on the registry is 0.3.0.
175
285
 
176
286
  | Version | Date | Status | Contains |
177
287
  |---|---|---|---|
288
+ | 0.5.0 | 2026-09-02 | `beta` | `tldrx drive` and its own preflight, `watch check` / `watch arm`, `questions cards`, `plan schema`, `retro --all` (with its findings fed back into every reviewer prompt) and `story reopen --for-fix`; `tldrx update` plus a cached newer-version notice; the dashboard reads `budget.yml` and `events.jsonl` — operator notes, reopens and retries, the per-phase budget panel, and a host-attended run metered in tokens against `ceiling_host_tokens`; a rejected review envelope no longer burns a story attempt; `ship` opens one PR per repo; merge-wave lock + ref guard; five golden-transcript evals, one per stage; `CONTRIBUTING.md` and a model-provider contract |
289
+ | 0.4.0 | 2026-09-01 | `beta` | FIRST BETA — 40-issue hardening burn (DoD pre-flight + `plan sync-dod`, merge-wave lock + gated-HEAD, load-aware tests, claim-sources across all outputs), `tldrx learn` 8-chapter sandbox tutorial (cold-player QA), `tldrx ship` / `tldrx note` / `run gates set`, budget policies + dual-economy wiring, single integration branch for chained epics, epic worktrees live to run close, bilingual docs site |
178
290
  | 0.3.1 | 2026-08-31 | `alpha` | Unattended mode (gates_policy agent, review handshake, fixlist, decision cards, dual economy), 6 contact fixes from the first feature-scope runs, colored init, training repair round |
179
291
  | 0.3.0 | 2026-08-30 | `alpha` | expert training with provenance, auto gates with an undo, `tldrx status`, seed triage, the token economy (context ledger, `max_reads`, `cost`, `estimate`), `install --claude`, `interview`, the ticket mirror, `--help` with flags and exit codes |
180
292
  | 0.2.0 | 2026-08-29 | `alpha` | Build executor (worktree + branch per story, epic branches, DoD gate, reviewer), Watch cards, live dashboard |
@@ -188,6 +300,10 @@ path documented; `stable` = 1.0, semver from here on. The badge above shows the
188
300
 
189
301
  ## Releasing
190
302
 
191
- **One command: `scripts/release.sh X.Y.Z --tag alpha`.** It is the only sanctioned path — a Claude Code hook denies hand-made `git tag` / `npm publish`, and `publish.yml` re-runs the same checks. Checklist and judgement calls: `docs/RELEASING.md`.
303
+ **One command: `scripts/release.sh X.Y.Z --tag beta`.** The tag is not optional in practice: omit
304
+ `--tag` and the script writes `alpha`, which is no longer this project's status. It is the only
305
+ sanctioned path — a Claude Code hook denies hand-made `git tag` / `npm publish`, and `publish.yml`
306
+ runs `release-check.sh --ci` (the file checks only) plus its own typecheck, tests and build.
307
+ Checklist and judgement calls: `docs/RELEASING.md`.
192
308
 
193
309
  MIT, © 2026 Alan Martinez — a placeholder made while scaffolding; change it freely before anything ships.
@@ -1,15 +1,15 @@
1
1
  #!/usr/bin/env node
2
2
  import {
3
3
  FactsStore
4
- } from "./chunk-ytvmc5ns.js";
4
+ } from "./chunk-889tybxc.js";
5
5
  import {
6
6
  parseHookInput,
7
7
  readStdin
8
- } from "./chunk-s5qsb4k6.js";
8
+ } from "./chunk-v1c1hpb8.js";
9
9
  import {
10
10
  EventLog
11
- } from "./chunk-ybacnpxd.js";
12
- import"./chunk-9kkm6q0t.js";
11
+ } from "./chunk-sae7sqty.js";
12
+ import"./chunk-qw73rdbr.js";
13
13
  import {
14
14
  MAX_FACT_CHARS,
15
15
  detectAnswered,
@@ -17,13 +17,12 @@ import {
17
17
  recordAnswer,
18
18
  replaceBlock,
19
19
  serializeQuestions
20
- } from "./chunk-3t91gvpp.js";
21
- import"./chunk-sznsenee.js";
20
+ } from "./chunk-nadqsr3w.js";
22
21
  import"./chunk-39zh2e44.js";
23
22
  import {
24
23
  PROJECT_WORK_DIR,
25
24
  factsPath
26
- } from "./chunk-ztczwtj0.js";
25
+ } from "./chunk-a6rpj2cp.js";
27
26
 
28
27
  // src/hooks/answer-capture.ts
29
28
  import { existsSync as existsSync2 } from "fs";
@@ -68,7 +67,7 @@ function filePathOf(payload) {
68
67
  }
69
68
 
70
69
  // src/hooks/lib/workspace.ts
71
- import { dirname, isAbsolute, join, resolve, sep } from "node:path";
70
+ import { basename, dirname, isAbsolute, join, resolve, sep } from "node:path";
72
71
  function locateWork(filePath) {
73
72
  if (filePath === "")
74
73
  return null;
@@ -1,47 +1,51 @@
1
1
  #!/usr/bin/env node
2
2
  import {
3
3
  budgetGateDeny
4
- } from "./chunk-9gb21660.js";
4
+ } from "./chunk-a5dq2dcp.js";
5
5
  import {
6
6
  allow,
7
7
  deny,
8
8
  readPayload,
9
9
  runHook,
10
10
  toolInput
11
- } from "./chunk-3g61yg59.js";
12
- import"./chunk-s5qsb4k6.js";
11
+ } from "./chunk-bvm6vjrt.js";
12
+ import"./chunk-v1c1hpb8.js";
13
13
  import {
14
14
  asRunBudget,
15
15
  currentActor,
16
16
  cursorStage,
17
17
  economyFor,
18
+ hostTokensIn,
19
+ isAttendedByHostView,
18
20
  isHostTokens,
19
21
  loadRunView,
20
22
  newestActiveRun,
21
23
  nowRfc3339,
22
24
  raiseCommand,
23
25
  remainingWork,
26
+ renderRunEconomies,
27
+ runSpend,
24
28
  shortBy,
25
29
  validateRunBudget,
26
- wouldExceed
27
- } from "./chunk-zdxgragg.js";
30
+ wouldExceed,
31
+ wouldExceedHostTokens
32
+ } from "./chunk-14zn51kh.js";
28
33
  import {
29
34
  EventLog
30
- } from "./chunk-ybacnpxd.js";
31
- import"./chunk-458wgg9j.js";
32
- import"./chunk-0z27twdk.js";
35
+ } from "./chunk-sae7sqty.js";
36
+ import"./chunk-hcrbr430.js";
33
37
  import {
34
38
  noteDeprecations
35
- } from "./chunk-3t91gvpp.js";
36
- import"./chunk-sznsenee.js";
39
+ } from "./chunk-nadqsr3w.js";
37
40
  import"./chunk-39zh2e44.js";
41
+ import"./chunk-q8d3sff9.js";
38
42
  import {
39
43
  PROJECT_WORK_DIR,
40
44
  findWorkspaceRoot,
41
45
  locateWork,
42
46
  parseYaml,
43
47
  stageYamlPath
44
- } from "./chunk-ztczwtj0.js";
48
+ } from "./chunk-a6rpj2cp.js";
45
49
 
46
50
  // src/hooks/budget-gate.ts
47
51
  import { existsSync as existsSync2, readFileSync as readFileSync2, statSync } from "node:fs";
@@ -69,7 +73,9 @@ function loadRunBudget(runDir) {
69
73
  var SPAWN_RE = /^(claude -p|tldrx next|tldrx run auto|tldrx expert train|tldrx seed triage)\b/;
70
74
  var RUN_ARG_RE = /--run[= ]([\w.-]+)/;
71
75
  var DEFAULT_TRAIN_USD = 2;
76
+ var DEFAULT_FULL_TRAIN_USD = 3;
72
77
  var DEFAULT_TRIAGE_USD = 1;
78
+ var FULL_MODE_RE = /--mode[= ]full\b/;
73
79
  var MAX_USD_RE = /--max-usd[= ]([0-9]+(?:\.[0-9]+)?)/;
74
80
  var MAX_BUDGET_RE = /--max-budget-usd[= ]([0-9]+(?:\.[0-9]+)?)/;
75
81
  await runHook("budget-gate", async () => {
@@ -107,6 +113,7 @@ await runHook("budget-gate", async () => {
107
113
  }
108
114
  if (budget === null)
109
115
  failClosed(command, `${view.dir}/budget.yml is missing or unreadable`);
116
+ const attended = isAttendedByHostView(view);
110
117
  const stage = cursorStage(view);
111
118
  const declared = stage?.budget_usd ?? stageBudgetFromLibrary(root, view.cursor.stage);
112
119
  const work = declared === null ? null : remainingWork({
@@ -116,18 +123,59 @@ await runHook("budget-gate", async () => {
116
123
  stageSpentUsd: stage?.cost_usd ?? 0,
117
124
  perAgentMaxUsd: budget.per_agent_max_usd,
118
125
  maxUsd: null,
119
- economy: economyFor(budget, view.cursor.phase)
126
+ economy: economyFor(budget, view.cursor.phase),
127
+ attended
120
128
  });
121
129
  const estimate = estimateFor(command, work === null ? null : work.usd);
122
130
  if (estimate <= 0)
123
131
  return;
132
+ const economies = renderRunEconomies(view);
133
+ const spend = runSpend(view);
124
134
  if (isHostTokens(budget, view.cursor.phase)) {
125
- process.stderr.write(`tldrx hook budget-gate: ${view.cursor.phase} is priced in \`host-tokens\` — ` + "no dollar ceiling to enforce here; `tldrx next` refuses a headless spawn on it.\n");
135
+ const tokens = wouldExceedHostTokens(budget, view.cursor.phase, hostTokensIn(view, view.cursor.phase));
136
+ const over = tokens !== null && tokens.over ? ` ${view.cursor.phase} is OVER its host-token ceiling: ` + `${String(tokens.spent)} declared of ${String(tokens.ceiling)} allowed.` : "";
137
+ const stops = tokens !== null && tokens.blocked && !attended;
138
+ if (over !== "") {
139
+ recordBudgetEvent(view, view.cursor.stage, stops ? "budget.blocked" : "budget.warned", {
140
+ phase: view.cursor.phase,
141
+ scope: tokens?.scope ?? "phase",
142
+ economy: "host-tokens",
143
+ attended_by: view.attended_by,
144
+ host_tokens: tokens?.spent ?? 0,
145
+ ceiling_tokens: tokens?.ceiling ?? 0,
146
+ estimate_usd: estimate,
147
+ metered_usd: spend.meteredUsd,
148
+ unmetered_tasks: spend.unmeteredTasks
149
+ });
150
+ }
151
+ if (stops && tokens !== null) {
152
+ deny(`[tldrx] budget-gate: refusing to start stage "${view.cursor.stage}" — phase ${view.cursor.phase} is ` + `priced in \`host-tokens\` and has declared ${String(tokens.spent)} of ${String(tokens.ceiling)} ` + "allowed. Raise that phase's ceiling in budget.yml (under this economy the number is a TOKEN " + "allowance), or set `on_host_tokens_exceed: warn` to go back to a note." + `${economies === null ? "" : `
153
+ ${economies}`}`);
154
+ }
155
+ process.stderr.write(`tldrx hook budget-gate: ${view.cursor.phase} is priced in \`host-tokens\` — ` + "no dollar ceiling to enforce here; `tldrx next` refuses a headless spawn on it." + over + `${economies === null ? "" : ` ${economies}`}
156
+ `);
126
157
  return;
127
158
  }
128
159
  const decision = wouldExceed(budget, view.cursor.phase, estimate);
129
160
  if (!decision.blocked)
130
161
  return;
162
+ if (attended) {
163
+ recordBudgetEvent(view, view.cursor.stage, "budget.warned", {
164
+ phase: view.cursor.phase,
165
+ scope: decision.scope,
166
+ remaining_usd: decision.remaining,
167
+ ceiling_usd: decision.ceiling,
168
+ estimate_usd: decision.estimate,
169
+ economy: economyFor(budget, view.cursor.phase),
170
+ attended_by: view.attended_by,
171
+ metered_usd: spend.meteredUsd,
172
+ host_tokens: spend.hostTokens,
173
+ unmetered_tasks: spend.unmeteredTasks
174
+ });
175
+ process.stderr.write(`tldrx hook budget-gate: ${view.cursor.phase} has $${decision.remaining.toFixed(2)} left of ` + `$${decision.ceiling.toFixed(2)} and the stage estimate is $${estimate.toFixed(2)} — NOT refusing, ` + "because this run is attended_by: host and the framework spawns nothing on it." + `${economies === null ? "" : ` ${economies}`}
176
+ `);
177
+ return;
178
+ }
131
179
  new EventLog(join2(view.dir, "events.jsonl")).tryAppend({
132
180
  ts: nowRfc3339(),
133
181
  run: view.run,
@@ -141,18 +189,37 @@ await runHook("budget-gate", async () => {
141
189
  remaining_usd: decision.remaining,
142
190
  ceiling_usd: decision.ceiling,
143
191
  estimate_usd: decision.estimate,
144
- blocked_by: currentActor()
192
+ blocked_by: currentActor(),
193
+ economy: economyFor(budget, view.cursor.phase),
194
+ attended_by: view.attended_by,
195
+ metered_usd: spend.meteredUsd,
196
+ host_tokens: spend.hostTokens,
197
+ unmetered_tasks: spend.unmeteredTasks
145
198
  }
146
199
  });
147
- deny(budgetGateDeny(view.cursor.stage, view.cursor.phase, decision.remaining, decision.ceiling, estimate, raiseCommand(view.run, view.cursor.phase, shortBy(estimate, decision.remaining))));
200
+ deny(budgetGateDeny(view.cursor.stage, view.cursor.phase, decision.remaining, decision.ceiling, estimate, raiseCommand(view.run, view.cursor.phase, shortBy(estimate, decision.remaining))) + (economies === null ? "" : `
201
+ ${economies}`));
148
202
  });
203
+ function recordBudgetEvent(view, stage, type, payload) {
204
+ new EventLog(join2(view.dir, "events.jsonl")).tryAppend({
205
+ ts: nowRfc3339(),
206
+ run: view.run,
207
+ stage,
208
+ type,
209
+ actor: "hook:budget-gate",
210
+ cost_usd: 0,
211
+ payload
212
+ });
213
+ }
149
214
  function estimateFor(command, stageBudget) {
150
215
  const flagged = Number(MAX_USD_RE.exec(command)?.[1] ?? MAX_BUDGET_RE.exec(command)?.[1] ?? NaN);
151
216
  if (/^tldrx run auto\b/.test(command)) {
152
217
  return Number.isFinite(flagged) ? flagged : stageBudget ?? 0;
153
218
  }
154
219
  if (/^tldrx expert train\b/.test(command)) {
155
- return Number.isFinite(flagged) ? flagged : DEFAULT_TRAIN_USD;
220
+ if (Number.isFinite(flagged))
221
+ return flagged;
222
+ return FULL_MODE_RE.test(command) ? DEFAULT_FULL_TRAIN_USD : DEFAULT_TRAIN_USD;
156
223
  }
157
224
  if (/^tldrx seed triage\b/.test(command)) {
158
225
  return Number.isFinite(flagged) ? flagged : DEFAULT_TRIAGE_USD;