tldr-experts 0.3.1 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +2396 -0
- package/README.md +133 -17
- package/dist/hooks/answer-capture.js +7 -8
- package/dist/hooks/budget-gate.js +83 -16
- package/dist/hooks/{chunk-zdxgragg.js → chunk-14zn51kh.js} +171 -16
- package/dist/hooks/{chunk-xpxe94qe.js → chunk-3w55tp71.js} +75 -27
- package/dist/hooks/{chunk-afamdvyn.js → chunk-5556vjt5.js} +1 -1
- package/dist/hooks/{chunk-ytvmc5ns.js → chunk-889tybxc.js} +26 -28
- package/dist/hooks/{chunk-9gb21660.js → chunk-a5dq2dcp.js} +54 -14
- package/dist/hooks/{chunk-ztczwtj0.js → chunk-a6rpj2cp.js} +835 -16
- package/dist/hooks/{chunk-3g61yg59.js → chunk-bvm6vjrt.js} +1 -1
- package/dist/hooks/{chunk-0z27twdk.js → chunk-hcrbr430.js} +10 -5
- package/dist/hooks/{chunk-3t91gvpp.js → chunk-nadqsr3w.js} +8 -2
- package/dist/hooks/{chunk-458wgg9j.js → chunk-q8d3sff9.js} +86 -11
- package/dist/hooks/{chunk-9kkm6q0t.js → chunk-qw73rdbr.js} +35 -3
- package/dist/hooks/{chunk-ybacnpxd.js → chunk-sae7sqty.js} +5 -0
- package/dist/hooks/{chunk-s5qsb4k6.js → chunk-v1c1hpb8.js} +1 -1
- package/dist/hooks/claim-sources.js +32 -21
- package/dist/hooks/dod-gate.js +7 -6
- package/dist/hooks/no-reask.js +9 -9
- package/dist/hooks/session-start.js +46 -21
- package/dist/hooks/statusline.js +8 -9
- package/dist/tldrx.js +19667 -10978
- package/package.json +4 -2
- package/plugin/.claude-plugin/plugin.json +2 -2
- package/plugin/skills/tldrx/SKILL.md +1 -1
- package/stages/build/stage.yml +2 -1
- package/stages/plan/stage.md +11 -0
- package/stages/watch/stage.md +4 -0
- package/templates/expert.md +13 -1
- package/templates/watcher.md +8 -1
- package/workflows/bugfix.yml +8 -4
- package/workflows/docs.yml +8 -4
- package/workflows/feature.yml +8 -4
- package/workflows/hotfix.yml +8 -4
- package/workflows/integration.yml +8 -4
- package/workflows/migration.yml +8 -4
- package/workflows/performance.yml +8 -4
- package/workflows/prototype.yml +8 -4
- package/workflows/refactor.yml +8 -4
- package/workflows/security-patch.yml +8 -4
- package/workflows/spike.yml +8 -4
- package/workflows/upgrade.yml +8 -4
- package/dist/hooks/chunk-sznsenee.js +0 -503
- package/templates/epic.md +0 -38
- package/templates/story.md +0 -55
package/README.md
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
# tldr-experts
|
|
2
2
|
|
|
3
|
-
[](https://www.npmjs.com/package/tldr-experts) [](https://github.com/ederwii/tldr-experts/actions/workflows/ci.yml) ](https://www.npmjs.com/package/tldr-experts) [](https://github.com/ederwii/tldr-experts/actions/workflows/ci.yml) 
|
|
4
4
|
|
|
5
|
-
**
|
|
5
|
+
**An evidence-first, file-based AI development framework: five stages, a gate on every one, and every claim cited or refused.** Open source, tool-agnostic in design, piloted on Claude Code. **Beta:** every command is implemented and verified by running it, the `version: 1` file formats only grow from here, and `tldrx --help` is the authoritative command surface.
|
|
6
6
|
|
|
7
7
|
One loop — *Investigate → Handoff → Interview → Gate* — five phases, **what · how · plan · build ·
|
|
8
8
|
watch**, one stage per command, each stopping at a gate you own; the files ARE the state, the
|
|
@@ -13,10 +13,6 @@ do: a command that cannot do the thing exits non-zero and says which thing.
|
|
|
13
13
|
|
|
14
14
|
## Quick start
|
|
15
15
|
|
|
16
|
-
> **Not on npm yet.** Every published version was unpublished on 2026-08-29 (`npm view tldr-experts
|
|
17
|
-
> version` → `E404 Unpublished`) and there is no `v0.3.0` tag, so the `npm i -g` line 404s until
|
|
18
|
-
> `scripts/release.sh 0.3.0` is run. Until then: clone and `bun link`, or `bun <repo>/bin/tldrx.ts <cmd>`.
|
|
19
|
-
|
|
20
16
|
```bash
|
|
21
17
|
npm i -g tldr-experts # installs `tldrx` (short) and `tldr-experts` (same binary)
|
|
22
18
|
cd your-project
|
|
@@ -25,6 +21,16 @@ tldrx init # detect repos, map the code, write .tldrx/, ask only
|
|
|
25
21
|
tldrx install --claude # write the skill, hooks and status line into ./.claude/
|
|
26
22
|
```
|
|
27
23
|
|
|
24
|
+
Later: **`tldrx update`** pulls the newest published version and prints the CHANGELOG between the
|
|
25
|
+
one you had and the one you now have. Any command will tell you, in one line, when there is a newer
|
|
26
|
+
one — off the hot path, cached for a day, silent when it cannot reach the registry, and never in
|
|
27
|
+
`--json` output or during a hook. Turn it off with `TLDRX_UPDATE_CHECK=off`.
|
|
28
|
+
|
|
29
|
+
**Never used it before?** `tldrx learn` teaches the loop by running it: eight chapters, ~15 minutes,
|
|
30
|
+
in a throwaway sandbox with a toy repo and a stand-in agent. Every command in it is the real one —
|
|
31
|
+
`init`, `run new`, `next`, `approve`, a Build that cuts a branch and runs a real DoD — so nothing it
|
|
32
|
+
shows you can drift from what the binary does, and it costs $0.00 and touches nothing you own.
|
|
33
|
+
|
|
28
34
|
Then open Claude Code there and type **`/tldrx`**. It runs `tldrx status`, finds what is already
|
|
29
35
|
waiting on you — unanswered setup questions, a proposed split nobody decided, a run waiting on a gate,
|
|
30
36
|
an expert no stage can lean on yet — and walks you through it one item at a time, asking every decision
|
|
@@ -39,6 +45,79 @@ tldrx run auto # `next`, over and over, until something actually need
|
|
|
39
45
|
dependencies, so an installed `tldrx` needs only Node; Bun builds it. Full walkthrough:
|
|
40
46
|
[`docs/guide/01-quick-start.md`](docs/guide/01-quick-start.md).
|
|
41
47
|
|
|
48
|
+
## Trying it: three ways to run
|
|
49
|
+
|
|
50
|
+
`tldrx run auto` and `tldrx run attend host` read like two speeds of the same thing. They are
|
|
51
|
+
opposites and they do not compose. **`auto` is an engine, not a lock**: a headless loop in which
|
|
52
|
+
the *framework* spawns a metered sub-agent, stage after stage. **`attend host` is a lock, not an
|
|
53
|
+
engine**: it sets one field, spends nothing and runs no stage, and from then on the framework never
|
|
54
|
+
spawns on that run — every turn is a `--prepare` / `--commit` handshake with a session you drive.
|
|
55
|
+
`run auto` on an attended run is refused outright (exit `1`); a bare `tldrx next` there exits `4`
|
|
56
|
+
and names the `--prepare` command instead.
|
|
57
|
+
|
|
58
|
+
| | who executes each turn | what a turn costs | where it stops |
|
|
59
|
+
|---|---|---|---|
|
|
60
|
+
| `tldrx run auto` | the framework — `claude -p`, spawned stage after stage | metered per spawn, rolled up by `tldrx cost` | the first human gate or open question (`4`), stage failure (`5`), ceiling (`2`) |
|
|
61
|
+
| `tldrx run attend host`, driven from a session | your session's own sub-agents | host-billed; the framework records `cost_usd: null, metered: false` | every turn — `--prepare` writes the bundle, `--commit` settles it |
|
|
62
|
+
| the same, under a **mandate** | your session's own sub-agents | host-billed | a new product decision, a ceiling raise, a boundary exit — nothing else |
|
|
63
|
+
|
|
64
|
+
- **A small run you were going to watch anyway** → `run auto`. One command, and it stops the moment it needs you.
|
|
65
|
+
- **A Claude Code session already open, and you care about cost or quality** → `run attend host`, driven from it: the context is warm, the turns are host-billed, and the framework writes the Build reviewer's bundle rather than spawning a second reader beside one you are already paying for.
|
|
66
|
+
- **Overnight, hands off, and you still want the adversarial check** → `run attend host` plus a mandate, below.
|
|
67
|
+
- **CI or cron** → `run auto`. It is the only one of the three with no session behind it.
|
|
68
|
+
|
|
69
|
+
### Overnight, with the checking kept
|
|
70
|
+
|
|
71
|
+
Two commands and a prompt — and the prompt now ships with the package:
|
|
72
|
+
|
|
73
|
+
```bash
|
|
74
|
+
tldrx drive --unattended # print the mandate; paste it into the session that drives the run
|
|
75
|
+
tldrx drive --attended # the same disciplines, but every gate stays yours to sign
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
`tldrx drive` needs no workspace, opens no run and writes nothing: it prints the discipline the
|
|
79
|
+
first real runs were driven by — the three-role protocol (developer → a **fresh** adversarial
|
|
80
|
+
reviewer, never the author → the host verifying both in the code, not in their reports), evidence
|
|
81
|
+
labelled `measured` / `inferred` / `assumed`, product questions parked rather than decided, the
|
|
82
|
+
reviewer calibrated to the story's stakes, and the cost declared once. The two modes differ in
|
|
83
|
+
exactly two places: who drives the turns, and who may close a gate.
|
|
84
|
+
|
|
85
|
+
The rest of this section is what that mandate says, in the shape you would type it by hand.
|
|
86
|
+
|
|
87
|
+
```bash
|
|
88
|
+
tldrx run new payments --scope feature --budget 25 \
|
|
89
|
+
--attended-by host --gates what:agent,plan:agent,build:agent,watch:agent
|
|
90
|
+
tldrx run attend host 260101-payments # or flip a run that is already open
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
`--gates` **replaces the workflow's gates wholesale**, and a stage you leave out of the list becomes
|
|
94
|
+
`auto` — so name every gate you want signed. Then, in the session, the mandate:
|
|
95
|
+
|
|
96
|
+
> Act as my unattended verification gate on run `260101-payments`, until it reaches its last gate.
|
|
97
|
+
>
|
|
98
|
+
> Drive every stage yourself — `tldrx next --prepare 260101-payments`, then
|
|
99
|
+
> `tldrx next --commit 260101-payments` — dispatching your own sub-agents for the turns. The
|
|
100
|
+
> framework must never spawn.
|
|
101
|
+
>
|
|
102
|
+
> For every build story, run an INDEPENDENT adversarial review through the `--review` handshake:
|
|
103
|
+
> `tldrx next --prepare --review`, one read-only sub-agent over the diff, then
|
|
104
|
+
> `tldrx next --commit --review`. Its job is to find what the developer got wrong, not to agree
|
|
105
|
+
> with it.
|
|
106
|
+
>
|
|
107
|
+
> Approve a gate only after you have checked it yourself — that the citations resolve, that every
|
|
108
|
+
> touched path is one this run declared, and that the diff matches the stories it claims to
|
|
109
|
+
> implement — and write that check down as evidence: `tldrx gate template`, fill it in, then
|
|
110
|
+
> `tldrx approve --as-agent`.
|
|
111
|
+
>
|
|
112
|
+
> Interrupt me ONLY for a new product decision, a budget-ceiling raise, or work that has to go
|
|
113
|
+
> outside the declared boundary. Everything else you decide, and log.
|
|
114
|
+
>
|
|
115
|
+
> Never push. The final merge is mine.
|
|
116
|
+
|
|
117
|
+
The whole chapter — the three switches, what "never spawns" is enforced by, the review handshake,
|
|
118
|
+
the fix list, the evidence note and the four fallthroughs:
|
|
119
|
+
[10 Unattended mode](docs/guide/10-unattended-mode.md).
|
|
120
|
+
|
|
42
121
|
## How much human is in the loop
|
|
43
122
|
|
|
44
123
|
Every stage ends at a gate; what you choose is **who closes it**. `human` waits for `tldrx approve`.
|
|
@@ -78,7 +157,14 @@ Those are the shipped defaults, and every scope keeps at least one human gate. O
|
|
|
78
157
|
signs something it should not have, `tldrx reject --stage <phase>/<stage> --note "…"` revokes it, moves
|
|
79
158
|
the cursor back and marks the later stages `stale`. When it is one BUILD STORY you disagree with — a
|
|
80
159
|
story two reviewers refused, which is terminal for the rest of the run —
|
|
81
|
-
`tldrx story reopen <id> --note "…"` gives that one story another run of attempts and nothing else
|
|
160
|
+
`tldrx story reopen <id> --note "…"` gives that one story another run of attempts and nothing else;
|
|
161
|
+
for one named defect in a story already `done`, `--for-fix` opens a fix round instead — no attempt
|
|
162
|
+
consumed, the same DoD and the same reviewer, one open round at a time.
|
|
163
|
+
To move who may close a gate after `run new` froze it, `tldrx run gates set <stage>:<policy> --note "…"`
|
|
164
|
+
is the only sanctioned way, and it records the old→new value with your reason.
|
|
165
|
+
When you fix `.tldrx/workspace.yml` mid-run and the approved stories still cite the old command strings,
|
|
166
|
+
`tldrx plan sync-dod` rewrites just their dod lines — renames followed, removed commands dropped, and
|
|
167
|
+
anything with no ancestor in the file's history flagged rather than guessed at.
|
|
82
168
|
What an auto gate cannot do: [`docs/guide/03-runs-and-gates.md`](docs/guide/03-runs-and-gates.md).
|
|
83
169
|
|
|
84
170
|
## What you see while it runs
|
|
@@ -127,7 +213,7 @@ context 83.7 KB of 160.0 KB (~23.8k tok, 12% of sonnet's ~200.0k window)
|
|
|
127
213
|
```
|
|
128
214
|
|
|
129
215
|
Over `prompt_max_bytes` the stage is **refused** (exit 2) before anything spawns; `max_reads` stops
|
|
130
|
-
the sub-agent at a read ceiling; `--effort` changes what a turn costs. What `--max-
|
|
216
|
+
the sub-agent at a read ceiling; `--effort` changes what a turn costs. What `--max-usd` does is
|
|
131
217
|
end a run *after* the turn it is already in — measured, one turn spent **$5.15** against a **$1.50**
|
|
132
218
|
ceiling — so size the prompt for the money you are willing to lose. Afterwards `tldrx cost [--all]` adds up what was actually charged, per attempt, per stage, per
|
|
133
219
|
run, read off `agent.result` events and nothing else. Retries are never merged — a retry is
|
|
@@ -137,22 +223,40 @@ Details: [`docs/guide/06-budgets-and-cost.md`](docs/guide/06-budgets-and-cost.md
|
|
|
137
223
|
|
|
138
224
|
## Several runs
|
|
139
225
|
|
|
140
|
-
With several runs open and no id,
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
226
|
+
With several runs open and no id, a run-targeting command **refuses rather than guessing** and
|
|
227
|
+
lists them — `tldrx next: 3 runs are open — pass one:`. That means "you left off the id", not "it
|
|
228
|
+
broke". Two of them refuse differently and it is worth knowing which: `tldrx run status` is not a
|
|
229
|
+
refusal at all — it lists every open run and exits `0`, and it is the screen you read to find the
|
|
230
|
+
id the others want — and `tldrx cost` refuses at exit `1`, not `2`.
|
|
231
|
+
|
|
232
|
+
Most run-targeting commands take the id either way, a positional `<run>` or `--run <id>`: `next`,
|
|
233
|
+
`cost`, `note`, `gate template`, `questions`, `budget show`, `ship`, `tickets`, and `run attend` ·
|
|
234
|
+
`status` · `estimate` · `auto` · `unlock` · `cancel`. `replay` and `retro` take the positional only
|
|
235
|
+
— `--run` there is an unknown flag. `approve`, `reject`, `answer`, `interview`, `plan`,
|
|
236
|
+
`story reopen`, `watch` and `run gates set` take `--run <id>` only.
|
|
237
|
+
|
|
238
|
+
`tldrx retro --all` goes the other way: it reads **every** run in the workspace and prints one
|
|
239
|
+
table of what keeps catching you — finding class × count × how many runs × one example with its
|
|
240
|
+
citation — mined from the review logs, the fix lists, `retro.md` and the `story.reopened` reasons.
|
|
241
|
+
Strictly read-only: it writes nothing, anywhere.
|
|
144
242
|
|
|
145
243
|
## What to commit
|
|
146
244
|
|
|
147
245
|
**Both `.tldrx/` and `tldrx-work/`.** The files are the state — the map, the facts, the questions and
|
|
148
246
|
their answers, `run.yml`, `budget.yml`, `events.jsonl`, the handoffs, the plan — so a teammate who clones
|
|
149
|
-
the repo gets the run. The block `tldrx init` appends to `.gitignore` excludes
|
|
150
|
-
else, because those
|
|
151
|
-
`.tldrx/worktrees/`, `tldrx-work/*/.lock`,
|
|
247
|
+
the repo gets the run. The block `tldrx init` appends to `.gitignore` excludes eight paths and nothing
|
|
248
|
+
else, because those eight are machine-local, regenerated, or a backup git already holds the history
|
|
249
|
+
of: `.tldrx/graphify-out/`, `.tldrx/cache/`, `.tldrx/worktrees/`, `tldrx-work/*/.lock`,
|
|
250
|
+
`tldrx-work/*/.agent/`, `tldrx-work/*/*.bak`, `.tldrx/memory/*.bak` and
|
|
251
|
+
`.claude/settings.json.bak-tldrx-*`.
|
|
152
252
|
|
|
153
253
|
## Documentation
|
|
154
254
|
|
|
155
|
-
The
|
|
255
|
+
**[The documentation site](https://ederwii.github.io/tldr-experts/)** is the place to start if you have
|
|
256
|
+
never used this: a landing page, a Quickstart and one short page per concept, written for a reader
|
|
257
|
+
rather than for an agent. Source in [`docs-site/`](docs-site/).
|
|
258
|
+
|
|
259
|
+
The reference guide, in `docs/guide/`: [1 Quick start](docs/guide/01-quick-start.md) ·
|
|
156
260
|
[2 The loop](docs/guide/02-the-loop.md) (the four steps, what a stage file controls, the two execution modes) ·
|
|
157
261
|
[3 Runs and gates](docs/guide/03-runs-and-gates.md) (`run new`→`retro`, gate policy, `run auto`, unlock/cancel, dashboard, tickets) ·
|
|
158
262
|
[4 Experts](docs/guide/04-experts.md) (loading rules, role experts, training, levels) ·
|
|
@@ -166,6 +270,12 @@ Design docs: [`docs/concept.md`](docs/concept.md) (why) · [`docs/spec.md`](docs
|
|
|
166
270
|
open decisions) · [`docs/ROADMAP.md`](docs/ROADMAP.md) (next) · [`CHANGELOG.md`](CHANGELOG.md) (shipped) ·
|
|
167
271
|
[`docs/dashboard-model.md`](docs/dashboard-model.md).
|
|
168
272
|
|
|
273
|
+
Contributing: **[`CONTRIBUTING.md`](CONTRIBUTING.md)** — the loop a change goes through, the four gates
|
|
274
|
+
and what CI actually runs, the red-first test rules, and
|
|
275
|
+
[how to contribute a model-provider config](CONTRIBUTING.md#contributing-a-model-provider-config)
|
|
276
|
+
(the `TLDRX_CLAUDE_BIN` seam, the `stream-json` transcript contract, and what a generic provider
|
|
277
|
+
would have to supply).
|
|
278
|
+
|
|
169
279
|
## Releases and status tags
|
|
170
280
|
|
|
171
281
|
Install name is **`tldr-experts`**; it installs two commands, **`tldrx`** (short) and `tldr-experts` (same binary).
|
|
@@ -175,6 +285,8 @@ back on the registry is 0.3.0.
|
|
|
175
285
|
|
|
176
286
|
| Version | Date | Status | Contains |
|
|
177
287
|
|---|---|---|---|
|
|
288
|
+
| 0.5.0 | 2026-09-02 | `beta` | `tldrx drive` and its own preflight, `watch check` / `watch arm`, `questions cards`, `plan schema`, `retro --all` (with its findings fed back into every reviewer prompt) and `story reopen --for-fix`; `tldrx update` plus a cached newer-version notice; the dashboard reads `budget.yml` and `events.jsonl` — operator notes, reopens and retries, the per-phase budget panel, and a host-attended run metered in tokens against `ceiling_host_tokens`; a rejected review envelope no longer burns a story attempt; `ship` opens one PR per repo; merge-wave lock + ref guard; five golden-transcript evals, one per stage; `CONTRIBUTING.md` and a model-provider contract |
|
|
289
|
+
| 0.4.0 | 2026-09-01 | `beta` | FIRST BETA — 40-issue hardening burn (DoD pre-flight + `plan sync-dod`, merge-wave lock + gated-HEAD, load-aware tests, claim-sources across all outputs), `tldrx learn` 8-chapter sandbox tutorial (cold-player QA), `tldrx ship` / `tldrx note` / `run gates set`, budget policies + dual-economy wiring, single integration branch for chained epics, epic worktrees live to run close, bilingual docs site |
|
|
178
290
|
| 0.3.1 | 2026-08-31 | `alpha` | Unattended mode (gates_policy agent, review handshake, fixlist, decision cards, dual economy), 6 contact fixes from the first feature-scope runs, colored init, training repair round |
|
|
179
291
|
| 0.3.0 | 2026-08-30 | `alpha` | expert training with provenance, auto gates with an undo, `tldrx status`, seed triage, the token economy (context ledger, `max_reads`, `cost`, `estimate`), `install --claude`, `interview`, the ticket mirror, `--help` with flags and exit codes |
|
|
180
292
|
| 0.2.0 | 2026-08-29 | `alpha` | Build executor (worktree + branch per story, epic branches, DoD gate, reviewer), Watch cards, live dashboard |
|
|
@@ -188,6 +300,10 @@ path documented; `stable` = 1.0, semver from here on. The badge above shows the
|
|
|
188
300
|
|
|
189
301
|
## Releasing
|
|
190
302
|
|
|
191
|
-
**One command: `scripts/release.sh X.Y.Z --tag
|
|
303
|
+
**One command: `scripts/release.sh X.Y.Z --tag beta`.** The tag is not optional in practice: omit
|
|
304
|
+
`--tag` and the script writes `alpha`, which is no longer this project's status. It is the only
|
|
305
|
+
sanctioned path — a Claude Code hook denies hand-made `git tag` / `npm publish`, and `publish.yml`
|
|
306
|
+
runs `release-check.sh --ci` (the file checks only) plus its own typecheck, tests and build.
|
|
307
|
+
Checklist and judgement calls: `docs/RELEASING.md`.
|
|
192
308
|
|
|
193
309
|
MIT, © 2026 Alan Martinez — a placeholder made while scaffolding; change it freely before anything ships.
|
|
@@ -1,15 +1,15 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
import {
|
|
3
3
|
FactsStore
|
|
4
|
-
} from "./chunk-
|
|
4
|
+
} from "./chunk-889tybxc.js";
|
|
5
5
|
import {
|
|
6
6
|
parseHookInput,
|
|
7
7
|
readStdin
|
|
8
|
-
} from "./chunk-
|
|
8
|
+
} from "./chunk-v1c1hpb8.js";
|
|
9
9
|
import {
|
|
10
10
|
EventLog
|
|
11
|
-
} from "./chunk-
|
|
12
|
-
import"./chunk-
|
|
11
|
+
} from "./chunk-sae7sqty.js";
|
|
12
|
+
import"./chunk-qw73rdbr.js";
|
|
13
13
|
import {
|
|
14
14
|
MAX_FACT_CHARS,
|
|
15
15
|
detectAnswered,
|
|
@@ -17,13 +17,12 @@ import {
|
|
|
17
17
|
recordAnswer,
|
|
18
18
|
replaceBlock,
|
|
19
19
|
serializeQuestions
|
|
20
|
-
} from "./chunk-
|
|
21
|
-
import"./chunk-sznsenee.js";
|
|
20
|
+
} from "./chunk-nadqsr3w.js";
|
|
22
21
|
import"./chunk-39zh2e44.js";
|
|
23
22
|
import {
|
|
24
23
|
PROJECT_WORK_DIR,
|
|
25
24
|
factsPath
|
|
26
|
-
} from "./chunk-
|
|
25
|
+
} from "./chunk-a6rpj2cp.js";
|
|
27
26
|
|
|
28
27
|
// src/hooks/answer-capture.ts
|
|
29
28
|
import { existsSync as existsSync2 } from "fs";
|
|
@@ -68,7 +67,7 @@ function filePathOf(payload) {
|
|
|
68
67
|
}
|
|
69
68
|
|
|
70
69
|
// src/hooks/lib/workspace.ts
|
|
71
|
-
import { dirname, isAbsolute, join, resolve, sep } from "node:path";
|
|
70
|
+
import { basename, dirname, isAbsolute, join, resolve, sep } from "node:path";
|
|
72
71
|
function locateWork(filePath) {
|
|
73
72
|
if (filePath === "")
|
|
74
73
|
return null;
|
|
@@ -1,47 +1,51 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
import {
|
|
3
3
|
budgetGateDeny
|
|
4
|
-
} from "./chunk-
|
|
4
|
+
} from "./chunk-a5dq2dcp.js";
|
|
5
5
|
import {
|
|
6
6
|
allow,
|
|
7
7
|
deny,
|
|
8
8
|
readPayload,
|
|
9
9
|
runHook,
|
|
10
10
|
toolInput
|
|
11
|
-
} from "./chunk-
|
|
12
|
-
import"./chunk-
|
|
11
|
+
} from "./chunk-bvm6vjrt.js";
|
|
12
|
+
import"./chunk-v1c1hpb8.js";
|
|
13
13
|
import {
|
|
14
14
|
asRunBudget,
|
|
15
15
|
currentActor,
|
|
16
16
|
cursorStage,
|
|
17
17
|
economyFor,
|
|
18
|
+
hostTokensIn,
|
|
19
|
+
isAttendedByHostView,
|
|
18
20
|
isHostTokens,
|
|
19
21
|
loadRunView,
|
|
20
22
|
newestActiveRun,
|
|
21
23
|
nowRfc3339,
|
|
22
24
|
raiseCommand,
|
|
23
25
|
remainingWork,
|
|
26
|
+
renderRunEconomies,
|
|
27
|
+
runSpend,
|
|
24
28
|
shortBy,
|
|
25
29
|
validateRunBudget,
|
|
26
|
-
wouldExceed
|
|
27
|
-
|
|
30
|
+
wouldExceed,
|
|
31
|
+
wouldExceedHostTokens
|
|
32
|
+
} from "./chunk-14zn51kh.js";
|
|
28
33
|
import {
|
|
29
34
|
EventLog
|
|
30
|
-
} from "./chunk-
|
|
31
|
-
import"./chunk-
|
|
32
|
-
import"./chunk-0z27twdk.js";
|
|
35
|
+
} from "./chunk-sae7sqty.js";
|
|
36
|
+
import"./chunk-hcrbr430.js";
|
|
33
37
|
import {
|
|
34
38
|
noteDeprecations
|
|
35
|
-
} from "./chunk-
|
|
36
|
-
import"./chunk-sznsenee.js";
|
|
39
|
+
} from "./chunk-nadqsr3w.js";
|
|
37
40
|
import"./chunk-39zh2e44.js";
|
|
41
|
+
import"./chunk-q8d3sff9.js";
|
|
38
42
|
import {
|
|
39
43
|
PROJECT_WORK_DIR,
|
|
40
44
|
findWorkspaceRoot,
|
|
41
45
|
locateWork,
|
|
42
46
|
parseYaml,
|
|
43
47
|
stageYamlPath
|
|
44
|
-
} from "./chunk-
|
|
48
|
+
} from "./chunk-a6rpj2cp.js";
|
|
45
49
|
|
|
46
50
|
// src/hooks/budget-gate.ts
|
|
47
51
|
import { existsSync as existsSync2, readFileSync as readFileSync2, statSync } from "node:fs";
|
|
@@ -69,7 +73,9 @@ function loadRunBudget(runDir) {
|
|
|
69
73
|
var SPAWN_RE = /^(claude -p|tldrx next|tldrx run auto|tldrx expert train|tldrx seed triage)\b/;
|
|
70
74
|
var RUN_ARG_RE = /--run[= ]([\w.-]+)/;
|
|
71
75
|
var DEFAULT_TRAIN_USD = 2;
|
|
76
|
+
var DEFAULT_FULL_TRAIN_USD = 3;
|
|
72
77
|
var DEFAULT_TRIAGE_USD = 1;
|
|
78
|
+
var FULL_MODE_RE = /--mode[= ]full\b/;
|
|
73
79
|
var MAX_USD_RE = /--max-usd[= ]([0-9]+(?:\.[0-9]+)?)/;
|
|
74
80
|
var MAX_BUDGET_RE = /--max-budget-usd[= ]([0-9]+(?:\.[0-9]+)?)/;
|
|
75
81
|
await runHook("budget-gate", async () => {
|
|
@@ -107,6 +113,7 @@ await runHook("budget-gate", async () => {
|
|
|
107
113
|
}
|
|
108
114
|
if (budget === null)
|
|
109
115
|
failClosed(command, `${view.dir}/budget.yml is missing or unreadable`);
|
|
116
|
+
const attended = isAttendedByHostView(view);
|
|
110
117
|
const stage = cursorStage(view);
|
|
111
118
|
const declared = stage?.budget_usd ?? stageBudgetFromLibrary(root, view.cursor.stage);
|
|
112
119
|
const work = declared === null ? null : remainingWork({
|
|
@@ -116,18 +123,59 @@ await runHook("budget-gate", async () => {
|
|
|
116
123
|
stageSpentUsd: stage?.cost_usd ?? 0,
|
|
117
124
|
perAgentMaxUsd: budget.per_agent_max_usd,
|
|
118
125
|
maxUsd: null,
|
|
119
|
-
economy: economyFor(budget, view.cursor.phase)
|
|
126
|
+
economy: economyFor(budget, view.cursor.phase),
|
|
127
|
+
attended
|
|
120
128
|
});
|
|
121
129
|
const estimate = estimateFor(command, work === null ? null : work.usd);
|
|
122
130
|
if (estimate <= 0)
|
|
123
131
|
return;
|
|
132
|
+
const economies = renderRunEconomies(view);
|
|
133
|
+
const spend = runSpend(view);
|
|
124
134
|
if (isHostTokens(budget, view.cursor.phase)) {
|
|
125
|
-
|
|
135
|
+
const tokens = wouldExceedHostTokens(budget, view.cursor.phase, hostTokensIn(view, view.cursor.phase));
|
|
136
|
+
const over = tokens !== null && tokens.over ? ` ${view.cursor.phase} is OVER its host-token ceiling: ` + `${String(tokens.spent)} declared of ${String(tokens.ceiling)} allowed.` : "";
|
|
137
|
+
const stops = tokens !== null && tokens.blocked && !attended;
|
|
138
|
+
if (over !== "") {
|
|
139
|
+
recordBudgetEvent(view, view.cursor.stage, stops ? "budget.blocked" : "budget.warned", {
|
|
140
|
+
phase: view.cursor.phase,
|
|
141
|
+
scope: tokens?.scope ?? "phase",
|
|
142
|
+
economy: "host-tokens",
|
|
143
|
+
attended_by: view.attended_by,
|
|
144
|
+
host_tokens: tokens?.spent ?? 0,
|
|
145
|
+
ceiling_tokens: tokens?.ceiling ?? 0,
|
|
146
|
+
estimate_usd: estimate,
|
|
147
|
+
metered_usd: spend.meteredUsd,
|
|
148
|
+
unmetered_tasks: spend.unmeteredTasks
|
|
149
|
+
});
|
|
150
|
+
}
|
|
151
|
+
if (stops && tokens !== null) {
|
|
152
|
+
deny(`[tldrx] budget-gate: refusing to start stage "${view.cursor.stage}" — phase ${view.cursor.phase} is ` + `priced in \`host-tokens\` and has declared ${String(tokens.spent)} of ${String(tokens.ceiling)} ` + "allowed. Raise that phase's ceiling in budget.yml (under this economy the number is a TOKEN " + "allowance), or set `on_host_tokens_exceed: warn` to go back to a note." + `${economies === null ? "" : `
|
|
153
|
+
${economies}`}`);
|
|
154
|
+
}
|
|
155
|
+
process.stderr.write(`tldrx hook budget-gate: ${view.cursor.phase} is priced in \`host-tokens\` — ` + "no dollar ceiling to enforce here; `tldrx next` refuses a headless spawn on it." + over + `${economies === null ? "" : ` ${economies}`}
|
|
156
|
+
`);
|
|
126
157
|
return;
|
|
127
158
|
}
|
|
128
159
|
const decision = wouldExceed(budget, view.cursor.phase, estimate);
|
|
129
160
|
if (!decision.blocked)
|
|
130
161
|
return;
|
|
162
|
+
if (attended) {
|
|
163
|
+
recordBudgetEvent(view, view.cursor.stage, "budget.warned", {
|
|
164
|
+
phase: view.cursor.phase,
|
|
165
|
+
scope: decision.scope,
|
|
166
|
+
remaining_usd: decision.remaining,
|
|
167
|
+
ceiling_usd: decision.ceiling,
|
|
168
|
+
estimate_usd: decision.estimate,
|
|
169
|
+
economy: economyFor(budget, view.cursor.phase),
|
|
170
|
+
attended_by: view.attended_by,
|
|
171
|
+
metered_usd: spend.meteredUsd,
|
|
172
|
+
host_tokens: spend.hostTokens,
|
|
173
|
+
unmetered_tasks: spend.unmeteredTasks
|
|
174
|
+
});
|
|
175
|
+
process.stderr.write(`tldrx hook budget-gate: ${view.cursor.phase} has $${decision.remaining.toFixed(2)} left of ` + `$${decision.ceiling.toFixed(2)} and the stage estimate is $${estimate.toFixed(2)} — NOT refusing, ` + "because this run is attended_by: host and the framework spawns nothing on it." + `${economies === null ? "" : ` ${economies}`}
|
|
176
|
+
`);
|
|
177
|
+
return;
|
|
178
|
+
}
|
|
131
179
|
new EventLog(join2(view.dir, "events.jsonl")).tryAppend({
|
|
132
180
|
ts: nowRfc3339(),
|
|
133
181
|
run: view.run,
|
|
@@ -141,18 +189,37 @@ await runHook("budget-gate", async () => {
|
|
|
141
189
|
remaining_usd: decision.remaining,
|
|
142
190
|
ceiling_usd: decision.ceiling,
|
|
143
191
|
estimate_usd: decision.estimate,
|
|
144
|
-
blocked_by: currentActor()
|
|
192
|
+
blocked_by: currentActor(),
|
|
193
|
+
economy: economyFor(budget, view.cursor.phase),
|
|
194
|
+
attended_by: view.attended_by,
|
|
195
|
+
metered_usd: spend.meteredUsd,
|
|
196
|
+
host_tokens: spend.hostTokens,
|
|
197
|
+
unmetered_tasks: spend.unmeteredTasks
|
|
145
198
|
}
|
|
146
199
|
});
|
|
147
|
-
deny(budgetGateDeny(view.cursor.stage, view.cursor.phase, decision.remaining, decision.ceiling, estimate, raiseCommand(view.run, view.cursor.phase, shortBy(estimate, decision.remaining)))
|
|
200
|
+
deny(budgetGateDeny(view.cursor.stage, view.cursor.phase, decision.remaining, decision.ceiling, estimate, raiseCommand(view.run, view.cursor.phase, shortBy(estimate, decision.remaining))) + (economies === null ? "" : `
|
|
201
|
+
${economies}`));
|
|
148
202
|
});
|
|
203
|
+
function recordBudgetEvent(view, stage, type, payload) {
|
|
204
|
+
new EventLog(join2(view.dir, "events.jsonl")).tryAppend({
|
|
205
|
+
ts: nowRfc3339(),
|
|
206
|
+
run: view.run,
|
|
207
|
+
stage,
|
|
208
|
+
type,
|
|
209
|
+
actor: "hook:budget-gate",
|
|
210
|
+
cost_usd: 0,
|
|
211
|
+
payload
|
|
212
|
+
});
|
|
213
|
+
}
|
|
149
214
|
function estimateFor(command, stageBudget) {
|
|
150
215
|
const flagged = Number(MAX_USD_RE.exec(command)?.[1] ?? MAX_BUDGET_RE.exec(command)?.[1] ?? NaN);
|
|
151
216
|
if (/^tldrx run auto\b/.test(command)) {
|
|
152
217
|
return Number.isFinite(flagged) ? flagged : stageBudget ?? 0;
|
|
153
218
|
}
|
|
154
219
|
if (/^tldrx expert train\b/.test(command)) {
|
|
155
|
-
|
|
220
|
+
if (Number.isFinite(flagged))
|
|
221
|
+
return flagged;
|
|
222
|
+
return FULL_MODE_RE.test(command) ? DEFAULT_FULL_TRAIN_USD : DEFAULT_TRAIN_USD;
|
|
156
223
|
}
|
|
157
224
|
if (/^tldrx seed triage\b/.test(command)) {
|
|
158
225
|
return Number.isFinite(flagged) ? flagged : DEFAULT_TRIAGE_USD;
|