tldr-experts 0.28.1 → 0.29.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,236 @@
1
1
  # Changelog
2
2
 
3
+ ## 0.29.0 — 2026-09-15
4
+
5
+ ### Changed
6
+
7
+ - **A reviewer's `changes` verdict must cite its evidence, or it is refused as a malformed
8
+ envelope (closes #326).** MEASURED on a field run: a reviewer returned `changes` with summary
9
+ "test" and findings `["a","b"]`, and the framework recorded it as a real review — it spent the
10
+ story's attempt, blocked `done`, and would have quoted "test" to the next developer as why the
11
+ story is not done. Nothing checked a `changes` verdict's content; only a `refuted` fix-list
12
+ finding had to carry a citation. Owner decision (2026-09-14): a `changes` now pays the same
13
+ price — its `summary`, or one line of a finding, must END with an `[src: …]` token the §2.8
14
+ parser reads (the diff line that is wrong, the unmet acceptance criterion as
15
+ `03-plan/stories/<id>.md:<line>`, or `absent:<path>` for missing work). One that cites nothing
16
+ is a fault in the REPORT, so it takes the existing bounded format retry — re-prompted with the
17
+ refusal, no attempt spent, two corrections — and the third is recorded as `changes` exactly as
18
+ before. This deliberately reverses the part of #79's pin that called a citation-less `changes` a
19
+ judgement about the work; a `changes` that DOES cite still costs the attempt. The reviewer prompt
20
+ now states the rule under `changes` and carries the `[src: …]` grammar on every review, not only
21
+ when `fixlist` is on the table. The citation is read, not resolved: a token that parses but
22
+ points at nothing still passes, so this refuses the hollow envelope, not a wrong one.
23
+ - **`tldrx run auto` now moves a blocked phase's shortfall out of finished phases by default
24
+ (closes #330).** The rebalance shipped opt-in in #314 because a phase ceiling is a person's
25
+ decision about money. An audit of `run auto`'s intervention episodes across two client
26
+ workspaces ranked stage/phase sizing as the #2 cause of a human stepping in: 22 episodes, 27
27
+ `budget.raised`, 7 `budget.blocked` (counts the audit measured, cited from #330), and not one raise note changed the work
28
+ (the audit's inferred reading of those notes). In a validation run the flag moved money twice
29
+ with nobody present. So a person was signing a move the framework could already prove safe.
30
+ Now a bare `run auto` makes it, and `--no-rebalance-finished` turns it off for a launch whose
31
+ phase ceilings must mean exactly what they say. `--rebalance-finished` is still accepted, and
32
+ passing both flags is exit 1, by name. The rules are unchanged: the run ceiling never grows, no
33
+ move passes a recorded grant, donors must be finished and metered with no unmetered turn, the
34
+ move is the exact shortfall or nothing, and each move is one `budget.raised` with the same
35
+ `source: run auto --rebalance-finished`. `tldrx next` on its own still never rebalances. The
36
+ `tldrx drive` mandate is unchanged on purpose: its unattended mode is refused `run auto`
37
+ outright (`attended_by: host`), and its attended mode drives `tldrx next`. No mandate-driven
38
+ turn ever reaches the move, so a sentence about it would teach something the driver cannot do.
39
+ - **A headless fix list now buys its fix round in the same run, and the approving reviewer's
40
+ signature closes what it was shown (#327, owner decision "A: ronda + auto-cierre").** A reviewer
41
+ that signed with a `fix-now` finding used to park the story at `review`, on the grounds that
42
+ routing a fix list needed a host — measured live, that park stranded the story and the two
43
+ stories of the next wave behind it until a person intervened. Now the story is requeued without
44
+ spending an attempt, the developer is handed the open findings, and the next reviewer is shown
45
+ each one verbatim with the rule "approve only if every one is fixed". Its `approve` rewrites each
46
+ finding it was shown as `Resolved: yes <commit> — auto-closed: …` naming the reviewer session,
47
+ attempt and run — the existing line grammar, so every reader parses it unchanged, and checked
48
+ against git before the story may settle, so a sha that is not on the story branch reopens it. A
49
+ finding that reached the file after the prompt was rendered stays open and still blocks. The
50
+ test that pinned the park ("a headless run … parks the story") is inverted, citing the decision.
51
+ A host review is shown the same findings and closes nothing. No new event, `run.yml` field or
52
+ CLI verb.
53
+ - **An `auto` Build gate no longer stops on paths outside the declared surface — it signs, and
54
+ carries them into the PR (closes #331).** A hold-surface audit of two client workspaces measured
55
+ `boundary` in 48 of 78 Build `held_by` values, the only reason on 6, and 7 intervention episodes
56
+ on it; every one ended in an approval or a `story widen` (the "no real catch" reading is inferred
57
+ from their notes). A condition that stops an unattended run and is always waved through is a
58
+ cost, not a check — but dropping it would lose the one question a reviewer asks first. So the
59
+ fact moves from the gate to the review: the auto note renders the boundary with "does not hold an
60
+ auto gate — it is carried into the PR body for review" and ends in `· warned by: boundary`;
61
+ `gate.requested` gains an additive `warned_by` beside `held_by` (present only when non-empty);
62
+ `tldrx next` prints a `warning: boundary — …` line; `tldrx run status` marks the gate row
63
+ `— warning: boundary`; and `tldrx ship` renders `## Outside declared scope — review these`,
64
+ every outside path, grouped under the story whose own measured `story.touches_widened` row names
65
+ it, the rest under `no story's measured diff names these`. One derivation: the gate and the PR
66
+ body both call `evaluateBoundary`, which now returns the uncapped list beside its detail. An
67
+ `agent` gate still falls to a person on a boundary, and a `human` gate is unchanged. **One
68
+ carve-out keeps the hold where the blast radius is not product code:** an outside path in
69
+ `SENSITIVE_PATH_CLASSES` (one exported constant beside `evaluateBoundary` — `ci`, `secrets`,
70
+ `infra`, `dependencies`) still holds an auto gate, named with its class
71
+ (`app:.github/workflows/deploy.yml [ci]`); declaring it in a story's `touches:` clears it. The
72
+ per-story grouping needs nobody to widen: the executor writes its own measured
73
+ `story.touches_widened` row for each story as it settles, and a real unattended Build then
74
+ `ship` puts the path under the story that wrote it. No exit
75
+ code changed; a check-only refusal with a boundary warning beside it is now eligible for
76
+ `run auto`'s bounded re-run, since the boundary no longer holds it.
77
+
78
+ ### Fixed
79
+
80
+ - **A Build gate now names the story a dependent WAITS on, not only the stories that are
81
+ `blocked` (closes #303).** #280 correctly left a dependent of a `review`/`in_progress` story at
82
+ `todo` instead of recording a terminal `blocked` — but the gate only ever named literal
83
+ `blocked` stories (`storiesView`'s `firstBlocked`), so the sentence saying what S2 waits on
84
+ reached `04-build/handoff.md`'s `## Unknowns` and nowhere an operator watches: `gate.requested`,
85
+ its notification and the terminal `stories:` line said `S2 not started` and named no
86
+ dependency. That is the one thing to act on — `story reopen` refuses a story already `todo`,
87
+ and `run auto --until-done` does not relaunch an awaiting-human exit. MEASURED on the #280
88
+ fixture (S1 parked at `review` by a dying reviewer): the payload carried no key for S2 at all.
89
+ Now `gate.requested` and its notification carry three ADDITIVE keys, `waiting_story`,
90
+ `waiting_on` and `waiting_reason` — absent, never null, when nothing waits — and the summary
91
+ and terminal line say S2 waits on S1 (`review`). The wait is derived with the Build loop's own
92
+ `decidingHold`/`dependencyIsPending`, so a story that will BLOCK on a terminal dependency is
93
+ never reported as merely waiting, and `waiting_reason` is `dependencyWaitReason` — the
94
+ `## Unknowns` sentence, byte for byte. `story reopen`'s `depends_on` reader moved to
95
+ `buildProgress.ts` so both ask one reader. No event version or dashboard model bump.
96
+ - **`run auto --until-done` stops a `host-tokens` refusal in tokens, not in dollar words it
97
+ never measured (closes #270).** Every `budget.blocked` went through one dollar-shaped
98
+ sentence, reading `remaining_usd` / `estimate_usd` through a helper that defaults a missing
99
+ field to `0`. The two `host-tokens` writers carry neither field, so a token refusal stopped
100
+ the supervisor with `remaining_usd $0.00 < estimate_usd $0.00` — a confident figure nothing
101
+ measured — and sent the operator to `tldrx budget raise`, a dollar command for a ceiling that
102
+ is a token allowance (and for the headless refusal, a raise that changes nothing: the way out
103
+ is `--prepare` or re-pricing to `metered-usd`, as the row's own reason already says). The stop
104
+ line now switches on `economy`, as the dashboard model already did: a `host-tokens` row names
105
+ the `host_tokens` / `ceiling_tokens` it carries and quotes its `reason`, and a dollar row
106
+ missing its figures says they are not recorded rather than printing `$0.00`. The verdict is
107
+ unchanged — a `budget.blocked` was never relaunched and still is not.
108
+ - **`budget show`, `run estimate` and the `budget-gate` hook price the remaining work with the
109
+ stage's own `attempts:` (closes #214).** `tldrx next`'s brake has resolved the stage's
110
+ `attempts:` since the key existed; these three readers asked `remainingWork` with the shipped
111
+ 2. On an `attempts: 1` stage that reserved a second developer turn and a second reviewer the
112
+ executor never dispatches, so the page said BLOCKED and the hook DENIED a `tldrx next` the
113
+ brake itself allowed. Reproduced before the fix on the Build fixture with `attempts: 1`, two
114
+ of three stories done and $5.00 left: the brake allowed ($4.40 of work), while `budget show`
115
+ and `run estimate` quoted $6.40 and the hook denied `tldrx next` on "the stage estimate is
116
+ $6.40". The hook and `budget show` now resolve the cursor stage through the same tolerant
117
+ `buildStageDefaults` the `dod-gate` hook already calls on its PreToolUse path (it now takes
118
+ the stage id; an unreadable preset still gives the shipped 2), and `run estimate` passes the
119
+ stage spec it already loads. `reviewer_share`, `story_cap_multiplier` and
120
+ `story_cap_floor_usd` are still asked with their defaults by these three readers — the same
121
+ disagreement on three more knobs, measured and filed as #333; this change moves `attempts` only.
122
+ - **An in-session Watch turn is recorded as UNMETERED, not as a measured `$0.00` (closes #224).**
123
+ The `--commit` path read the result envelope's own `cost_usd` and defaulted it to `0` — and a
124
+ host session has no reason to fill that field in — while `--cost-usd`, the flag the host
125
+ declares what its sub-agent cost with, was never read anywhere in the Watch executor. Measured
126
+ on disk across two live workspaces: 56 watch task rows at `cost_usd: 0.0` with no `metered` key,
127
+ one of them written by a current release, so every zero was counted by `budget.spent_usd` as a
128
+ real measurement and the lower-bound labelling that keys off `metered: false` could not fire for
129
+ the whole of 05-watch. Watch now uses the same three-value contract Build and the single-agent
130
+ stages already use: `--cost-usd` first, then the envelope's figure, and with NEITHER
131
+ `cost_usd: null` + `metered: false` — a named absence instead of a confident zero. A declared
132
+ `--tokens` rides along on the row the same way. A headless watch turn, which this process really
133
+ does meter, is untouched.
134
+ - **A Watch spawn now records the ceiling it was given (closes #190).** The executor computed a
135
+ per-feature ceiling, handed it to the sub-agent and emitted no `agent.spawned` at all, so a
136
+ watcher feature was the one turn in the framework whose measured cost could not be read against
137
+ what it was allowed to cost — every other stage reconciles `agent.spawned.max_budget_usd`
138
+ against the `agent.result` it paired with. Watch now emits one `agent.spawned` per feature
139
+ before the spawn (a turn that dies is still a turn the ceiling was committed to), carrying
140
+ `phase`, `role: developer`, `model`, `effort`, `max_budget_usd` and `key` — the same `key` the
141
+ `agent.result` for that row already carries, which is what joins the two ends of one turn.
142
+ Deliberately not `story:`: that is the field `tldrx cost --stories` counts a row as a Build
143
+ story by, and a Watch feature id is not a story, so that report is unchanged and whether it
144
+ grows a Watch axis stays a separate decision.
145
+ - **The conflict-turn marker guard no longer counts an unreadable path as a clean one, bounds its
146
+ scan, and reads modified files too (closes #324).** #286's guard caught every read failure and
147
+ returned "holds nothing": a guard failing OPEN, and silently, which is the one direction an audit
148
+ record may never fail in. A path that could not be read is now NAMED with its reason — the errno,
149
+ or the cap that skipped it — and only `ENOENT` stays silent, because a file that is gone cannot
150
+ carry a marker into a commit. The scan is bounded for the first time: 2,000 paths and 4 MiB per
151
+ file, neither of which can bite a normal worktree (this whole repo is 1,037 tracked files and its
152
+ largest is a 766 KB `CHANGELOG.md`, measured on 5256fab), and a path past either cap surfaces the
153
+ same way an unreadable one does rather than passing as clean — an un-ignored generated tree used
154
+ to make the guard as slow and as hungry as that tree was big, with nothing afterwards to say it
155
+ had been. The scope now also includes files MODIFIED since the handed sha, closing the hole where
156
+ a developer pastes a conflicted hunk, markers and all, into an already-tracked file; that
157
+ widening was measured before it was made, not assumed — every `M` row of 1,060 commits of this
158
+ repo's history, 8,366 rows, each row's content against the marker regex, 0 false positives (one
159
+ repo is not every repo). What to DO about an unchecked path is one decision at one call site
160
+ (`UNCHECKED_PATH_POLICY` in `src/core/build/git.ts`): it refuses the attempt with the paths
161
+ named, and a `warn` policy that lets the story walk on with them named on the Build lines is
162
+ implemented beside it.
163
+ - **A developer's `notes` written as an ARRAY of strings is now joined with newlines instead of
164
+ read as `""` by both readers (closes #217).** Measured on a 0.14.3 security-patch run: the
165
+ in-session developer wrote one note per array element, and the two readers' `typeof … ===
166
+ "string"` ternaries — `toEnvelope` on the spawned path, `readResult` on the `--commit` path —
167
+ each threw the whole field away. `notes` is the envelope's one free-text channel, so a turn's
168
+ stated caveats (on that run, what a security patch could NOT verify) vanished from `run.yml`,
169
+ the story log and the handoff with nothing on the `--commit` line saying so — a record wrong in
170
+ the dangerous direction. The elements ARE the notes, so they are joined rather than refused:
171
+ refusing would spend the turn again over a file whose content is not in doubt, on the reader
172
+ that is tolerant by design. Silence was the actual defect, so `tldrx next --commit --check` now
173
+ says which of the three readings a file gets — joined (with the count), dropped by index for a
174
+ non-string element the way `outputs[i]` already is, or still `""` for a `notes` that is neither
175
+ a string nor an array. All three paths, the check included, go through ONE function
176
+ (`readNotes`, `envelope.ts`), so the rehearsal cannot disagree with the commit.
177
+ - **The `budget-gate` hook no longer refuses the `run auto` launch that would fix the shortfall
178
+ (closes #321).** It priced a `tldrx run auto` spawn against the cursor phase's own ceiling
179
+ alone. On a phase already short, it denied the launch and wrote `budget.blocked` before the
180
+ loop could run its in-process rebalance, even when finished phases held the money. With the
181
+ rebalance on by default, that turned into the common case. Reproduced on the hook fixture
182
+ before the fix: 02-how was $2.39 short, finished 01-what held $2.86 unspent, and
183
+ `tldrx run auto` was denied. Now, when finished phases cover the WHOLE shortfall under the
184
+ loop's own rules, the gate allows and says why on stderr, writing no event. It checks that
185
+ with the same `planRebalance`, `--take-from` validation and grant verdict the loop uses, read
186
+ against the full `run.yml` so a stale donor still counts as unfinished. The gate itself moves
187
+ nothing; the move and its `budget.raised` are the loop's. Denied exactly as before:
188
+ `tldrx next`, a launch with `--no-rebalance-finished`, a run-ceiling shortfall, a shortfall
189
+ the finished phases cover only in part, a stale donor, and a move past a recorded grant.
190
+ - **A re-reviewed story's `changes` is consumed instead of parking the story (closes #327).** A
191
+ story whose earlier reviewer died or was unfunded is re-reviewed alone, and that path returned
192
+ right after the verdict — so a `changes` with attempts left parked at `review` and its
193
+ dependents never started, on the serial path and in a wave alike. It now enters the attempt loop
194
+ (or the wave's next round) under the same ledger bound every other requeue uses.
195
+ - **A story `blocked` on an approving review whose fix list was closed later now settles `done`
196
+ (closes #329).** Measured live: a fix landed, the reviewer approved, the story blocked on a
197
+ `Resolved: no` nobody had rewritten — and after a person rewrote it, `story reopen` dispatched a
198
+ developer with nothing to change and `--as-is` refused a branch already on its epic. Before the
199
+ frontier walk, `tldrx next` and `--prepare` now settle such a story `done` with no agent spawned,
200
+ after holding every `Resolved: yes` to git, and say why in one line. The refusal that blocks a
201
+ story on its fix list names that remedy instead of `tldrx story reopen`.
202
+ - **A parallel Build wave no longer hands two developers the same unspent money (closes #325).**
203
+ Every per-story cap reads the stage's METERED remainder, which is only right when stories run
204
+ one at a time. Measured live: a $21.60 stage spawned S1 under $21.00 and S2 under $16.50 in one
205
+ wave before either had metered a cent, spend landed at $33.07, and S2's reviewer was refused on
206
+ "$0.00 left" — the build ended 0 of 4. Reproduced on the fake agent before the fix: two lanes on
207
+ a $20 stage were spawned at $9.00 each, $22 claimed with their review floors. Now each lane is
208
+ capped at the stage's remainder less the caps of the lanes still running and a $2.00 reviewer
209
+ floor for every story whose review is still ahead, its own included. A lane that bound would
210
+ put under the least its developer is already allowed (`story_cap_floor_usd` for a priced story,
211
+ its uniform share otherwise) waits for a lane to finish, and the run prints `S2: not dispatched
212
+ beside S1 yet —` with the remainder, the reservation, the floors and the bound; with nothing in
213
+ flight nothing can be freed by waiting, so it runs at that floor instead of stalling. The
214
+ consequence to know: a small stage can now serialise a wide wave — measured, the third of three
215
+ unpriced stories on an $8 stage waits for the other two to meter. `--parallel 1` is
216
+ byte-identical (golden unchanged). No new config key, event or field. Not in this
217
+ change: sizing the stage from the plan once it exists, and `budget show`
218
+ naming what its estimate excludes — both named on #325.
219
+ - **The Plan prompt states the per-item character cap where each capped field is described, and
220
+ an over-cap finding now survives into the retry (closes #328).** Two plan stages failed on the
221
+ same cap in one day: a 683-character `test_plan` item, and a `wave_cap_reason` — held to the
222
+ same cap at the gate, and stated nowhere the Plan agent reads before writing it. The
223
+ `acceptance` and `test_plan` rows now say "one sentence each; split a long … into several
224
+ items", and the Plan-shape wave-cap rule and the Caps list name `wave_cap_reason`'s cap — all
225
+ rendered from `MAX_ITEM_CHARS`, never a typed number. The retry half was measured, not assumed:
226
+ a failed check's reason reaches the next `## Previous attempt` squeezed to one line with its
227
+ middle elided, and on a one-defect plan the old message lost "split it into several items" to
228
+ that ellipsis. The finding is now `<n> characters (cap <cap>) — split it into several items`,
229
+ short enough that the field, the length, the cap and the instruction reach the retry; an
230
+ over-cap `wave_cap_reason` is told its length and the cap too. A prompt byte change by
231
+ design; no Build golden moves. Not in this change: the facilitator's one-line squeeze itself,
232
+ which still elides whatever a longer multi-finding reason puts in the middle.
233
+
3
234
  ## 0.28.1 — 2026-09-15
4
235
 
5
236
  ### Fixed
package/README.md CHANGED
@@ -335,6 +335,7 @@ back on the registry is 0.3.0.
335
335
 
336
336
  | Version | Date | Status | Contains |
337
337
  |---|---|---|---|
338
+ | 0.29.0 | 2026-09-15 | `beta` | Sixteen changes that make an unattended run trust its own gates and its own numbers. Verdict integrity: a reviewer's `changes` must cite evidence (#326), a re-reviewed story's `changes` is consumed instead of parking it (#327), and a story blocked on a stale fix list settles `done` once resolved (#329). Fix-list flow: a headless fix list now buys its own fix round, and an approving reviewer's signature closes what it was shown (#327). Money and phase sizing: `run auto` rebalances a blocked phase's shortfall out of finished phases by default (#330), the `budget-gate` hook stops denying the launch that would fix the shortfall (#321), a Build gate no longer stops on paths outside the declared surface but carries them into the PR instead (#331), a parallel wave no longer hands two developers the same unspent money (#325), `budget show`/`run estimate`/the budget-gate hook price remaining work off the stage's own `attempts:` (#214), `run auto --until-done` states a token refusal in tokens rather than invented dollar words (#270), a Watch turn is recorded as unmetered rather than a false `$0.00` (#224), and a Watch spawn now records the ceiling it was given (#190). Dependency and safety: a Build gate names the story a dependent waits on (#303), the conflict-turn marker guard names unreadable paths, bounds its scan, and covers modified files too (#324), a developer's array-valued `notes` is joined instead of silently dropped (#217), and the Plan prompt states its per-item character cap where each field is described so an over-cap finding survives into the retry (#328). |
338
339
  | 0.28.1 | 2026-09-15 | `beta` | Two fixes so a reopened story's reason reaches the agents that need it. The developer and reviewer prompts now carry `## Why this story was reopened` — who signed it, whether it's a fix round, the note verbatim — off the same ledger value #308 reads, instead of the note reaching only a report line and #308's check (closes #322). `tldrx note --help` now says an operator note reaches no agent and names `dispatch-notes.md` as the channel that does (closes #151). |
339
340
  | 0.28.0 | 2026-09-15 | `beta` | Four changes that keep an unattended run moving instead of stranding on a person. A story whose merge-up to its epic conflicts in at most three of its own touched files gets one automated conflict turn instead of a person, guarded against committing conflict markers (#286). `story reopen` on a blocked story now releases every dependent it alone was holding, in the same command (#312). Seed's `Recommended:` line and the loop's question parser now share one grammar, so a citation-only recommendation no longer parks a run that `seed check` had already passed (#323). And an `absent:` needle over `facts.yml` no longer trips on a fact's own recording metadata, while an `auto` gate refused only by failed checks gets one automatic re-run before waiting for a person (#231). |
340
341
  | 0.27.0 | 2026-09-14 | `beta` | Five changes for unattended runs, from a planning audit and live field measurement. The `plan` check now refuses a plan with more waves than the framework carries per run (unless `waves.yml` names a `wave_cap_reason`), refuses a story scheduled before its `depends_on` allows, and flags a dod command siblings carry that one story doesn't (#316–#319). `tldrx run auto --rebalance-finished` moves a blocked phase's exact shortfall out of a finished phase's unspent ceiling before refusing, through `budget raise --take-from` itself, so the run ceiling never grows and no grant is ever assumed; every money refusal now names the unspent total and the exact command to fix it (#314). `tldrx ship` fetches and merges the base into the epic and re-runs the `done` stories' DoD before opening a PR, so a stale or red epic refuses instead of shipping a PR that comes back red (#315). A reviewer's budget floor rises from $1.00 to $2.00, matching what completed reviews actually cost, so a review no longer dies mid-diff on a floor that was never measured (#307). And a developer whose DoD goes red gets the story's next attempt instead of blocking on the first miss, with the kept output handed to the next attempt and the bound counted from `events.jsonl` so it holds across process restarts; a refused developer, a cap death, or a broken dod command still blocks immediately (#313). |
@@ -1,24 +1,23 @@
1
1
  #!/usr/bin/env node
2
2
  import {
3
3
  conflictOf
4
- } from "./chunk-c62y35wh.js";
4
+ } from "./chunk-erasdab5.js";
5
5
  import {
6
6
  FactsStore,
7
7
  formatJaccard
8
- } from "./chunk-48ctk154.js";
8
+ } from "./chunk-gr83pq7x.js";
9
9
  import {
10
10
  parseHookInput,
11
11
  readStdin
12
- } from "./chunk-khjm356r.js";
12
+ } from "./chunk-z9hv47ps.js";
13
13
  import {
14
14
  EventLog,
15
15
  PHASE_ID_RE
16
- } from "./chunk-abswp1vt.js";
16
+ } from "./chunk-t6gbx3jg.js";
17
17
  import {
18
18
  PHASE_IDS
19
- } from "./chunk-veffvzw5.js";
20
- import"./chunk-g7wz339n.js";
21
- import"./chunk-zvxgdssb.js";
19
+ } from "./chunk-g4d2v4jc.js";
20
+ import"./chunk-857jhjmb.js";
22
21
  import {
23
22
  ADVISORY_KEY,
24
23
  MAX_FACT_CHARS,
@@ -28,7 +27,7 @@ import {
28
27
  renderQuestionBlock,
29
28
  replaceBlock,
30
29
  serializeQuestions
31
- } from "./chunk-khy21x6e.js";
30
+ } from "./chunk-yanmz0vz.js";
32
31
  import {
33
32
  ITERATION_ONLY_SLOT,
34
33
  PROJECT_FRAMEWORK_DIR,
@@ -38,7 +37,7 @@ import {
38
37
  isScopedTemplate,
39
38
  parseYaml,
40
39
  scopedSlotOf
41
- } from "./chunk-f6t1xc3f.js";
40
+ } from "./chunk-qmp5p77c.js";
42
41
 
43
42
  // src/hooks/answer-capture.ts
44
43
  import { existsSync as existsSync5 } from "fs";
@@ -1,31 +1,36 @@
1
1
  #!/usr/bin/env node
2
2
  import {
3
3
  budgetGateDeny
4
- } from "./chunk-q3d4xdhb.js";
4
+ } from "./chunk-mg3aj0sa.js";
5
5
  import {
6
6
  allow,
7
7
  deny,
8
8
  readPayload,
9
9
  runHook,
10
10
  toolInput
11
- } from "./chunk-tna1bz8r.js";
12
- import"./chunk-khjm356r.js";
11
+ } from "./chunk-pne7afnf.js";
12
+ import"./chunk-z9hv47ps.js";
13
13
  import {
14
+ REBALANCE_SOURCE,
15
+ applyRebalance,
14
16
  asRunBudget,
15
17
  currentActor,
18
+ describeRebalance,
16
19
  economyFor,
17
20
  isHostTokens,
18
21
  nowRfc3339,
22
+ planRebalance,
19
23
  raiseCommand,
24
+ raiseGrantVerdict,
20
25
  remainingWork,
21
26
  shortBy,
22
27
  validateRunBudget,
23
28
  wouldExceed,
24
29
  wouldExceedHostTokens
25
- } from "./chunk-e618k6te.js";
30
+ } from "./chunk-w3d9kxcp.js";
26
31
  import {
27
32
  EventLog
28
- } from "./chunk-abswp1vt.js";
33
+ } from "./chunk-t6gbx3jg.js";
29
34
  import {
30
35
  cursorStage,
31
36
  hostTokensIn,
@@ -34,18 +39,20 @@ import {
34
39
  newestActiveRun,
35
40
  renderRunEconomies,
36
41
  runSpend
37
- } from "./chunk-hws2gnxj.js";
38
- import"./chunk-g7wz339n.js";
42
+ } from "./chunk-0d4csx4s.js";
43
+ import {
44
+ buildStageDefaults
45
+ } from "./chunk-g4d2v4jc.js";
39
46
  import {
40
47
  noteDeprecations
41
- } from "./chunk-khy21x6e.js";
48
+ } from "./chunk-yanmz0vz.js";
42
49
  import {
43
50
  PROJECT_WORK_DIR,
44
51
  findWorkspaceRoot,
45
52
  locateWork,
46
53
  parseYaml,
47
54
  stageYamlPath
48
- } from "./chunk-f6t1xc3f.js";
55
+ } from "./chunk-qmp5p77c.js";
49
56
 
50
57
  // src/hooks/budget-gate.ts
51
58
  import { existsSync as existsSync2, readFileSync as readFileSync2, statSync } from "node:fs";
@@ -72,6 +79,8 @@ function loadRunBudget(runDir) {
72
79
  // src/hooks/budget-gate.ts
73
80
  var SPAWN_RE = /^(claude -p|tldrx next|tldrx run auto|tldrx expert train|tldrx seed triage)\b/;
74
81
  var RUN_ARG_RE = /--run[= ]([\w.-]+)/;
82
+ var RUN_AUTO_RE = /^tldrx run auto\b/;
83
+ var NO_REBALANCE_RE = /(?:^|\s)--no-rebalance-finished(?:[\s=]|$)/;
75
84
  var DEFAULT_TRAIN_USD = 2;
76
85
  var DEFAULT_FULL_TRAIN_USD = 3;
77
86
  var DEFAULT_TRIAGE_USD = 1;
@@ -116,6 +125,7 @@ await runHook("budget-gate", async () => {
116
125
  const attended = isAttendedByHostView(view);
117
126
  const stage = cursorStage(view);
118
127
  const declared = stage?.budget_usd ?? stageBudgetFromLibrary(root, view.cursor.stage);
128
+ const attempts = buildStageDefaults(root, view.scope, view.cursor.stage).attempts;
119
129
  const work = declared === null ? null : remainingWork({
120
130
  runDir: view.dir,
121
131
  phaseId: view.cursor.phase,
@@ -124,7 +134,8 @@ await runHook("budget-gate", async () => {
124
134
  perAgentMaxUsd: budget.per_agent_max_usd,
125
135
  maxUsd: null,
126
136
  economy: economyFor(budget, view.cursor.phase),
127
- attended
137
+ attended,
138
+ attempts
128
139
  });
129
140
  const estimate = estimateFor(command, work === null ? null : work.usd);
130
141
  if (estimate <= 0)
@@ -156,7 +167,7 @@ ${economies}`}`);
156
167
  `);
157
168
  return;
158
169
  }
159
- const decision = wouldExceed(budget, view.cursor.phase, estimate);
170
+ const decision = wouldExceed(budget, view.cursor.phase, estimate, attempts);
160
171
  if (!decision.blocked)
161
172
  return;
162
173
  if (attended) {
@@ -176,6 +187,15 @@ ${economies}`}`);
176
187
  `);
177
188
  return;
178
189
  }
190
+ if (decision.scope === "phase" && RUN_AUTO_RE.test(command) && !NO_REBALANCE_RE.test(command)) {
191
+ const short = shortBy(estimate, decision.remaining);
192
+ const covered = finishedPhasesCover(view.dir, budget, view.cursor.phase, short);
193
+ if (covered !== null) {
194
+ process.stderr.write(`tldrx hook budget-gate: ${view.cursor.phase} has $${decision.remaining.toFixed(2)} left of ` + `$${decision.ceiling.toFixed(2)} and the stage estimate is $${estimate.toFixed(2)} — NOT refusing: ` + `\`run auto\` rebalances finished phases before it refuses, and ${covered}, which covers the ` + `$${short.toFixed(2)} shortfall. The loop makes that move on its own record (\`budget.raised\`, ` + `source: ${REBALANCE_SOURCE}); launch with --no-rebalance-finished to be refused here instead.` + `${economies === null ? "" : ` ${economies}`}
195
+ `);
196
+ return;
197
+ }
198
+ }
179
199
  new EventLog(join2(view.dir, "events.jsonl")).tryAppend({
180
200
  ts: nowRfc3339(),
181
201
  run: view.run,
@@ -226,6 +246,22 @@ function estimateFor(command, stageBudget) {
226
246
  }
227
247
  return stageBudget ?? 0;
228
248
  }
249
+ function finishedPhasesCover(runDir, budget, phaseId, shortUsd) {
250
+ try {
251
+ const doc = parseYaml(readFileSync2(join2(runDir, "run.yml"), "utf8"));
252
+ if (doc === null || !Array.isArray(doc.phases))
253
+ return null;
254
+ const plan = planRebalance(budget, doc, phaseId, shortUsd);
255
+ if (plan.moves.length === 0)
256
+ return null;
257
+ const applied = applyRebalance(budget, plan);
258
+ if (applied.outcomes.some((outcome) => raiseGrantVerdict(budget, outcome).exceeds))
259
+ return null;
260
+ return describeRebalance(plan);
261
+ } catch {
262
+ return null;
263
+ }
264
+ }
229
265
  function failClosed(command, why) {
230
266
  deny(`[tldrx] budget-gate: refusing \`${command.slice(0, 80)}\` — this gate could not read the budget ` + `it is supposed to enforce (${why}).
231
267
  ` + "It fails CLOSED: a spend nothing can check is exactly the one that must not start. Fix the run's " + "budget.yml, or pass `--run <id>` so the gate knows which run to charge.");
@@ -1,7 +1,7 @@
1
1
  import {
2
2
  listRunDirs,
3
3
  parseYaml
4
- } from "./chunk-f6t1xc3f.js";
4
+ } from "./chunk-qmp5p77c.js";
5
5
 
6
6
  // src/hooks/lib/runFile.ts
7
7
  import { existsSync, readFileSync } from "node:fs";
@@ -1,6 +1,6 @@
1
1
  import {
2
2
  toolInput
3
- } from "./chunk-tna1bz8r.js";
3
+ } from "./chunk-pne7afnf.js";
4
4
 
5
5
  // src/hooks/lib/wouldBe.ts
6
6
  import { existsSync, readFileSync } from "node:fs";
@@ -1,7 +1,7 @@
1
1
  import {
2
2
  PROJECT_FRAMEWORK_DIR,
3
3
  PROJECT_WORK_DIR
4
- } from "./chunk-f6t1xc3f.js";
4
+ } from "./chunk-qmp5p77c.js";
5
5
 
6
6
  // src/core/lock/workspaceLock.ts
7
7
  import { existsSync as existsSync2, mkdirSync as mkdirSync2, openSync, readFileSync as readFileSync2, rmSync as rmSync2, writeSync, closeSync } from "node:fs";
@@ -1,15 +1,15 @@
1
1
  import {
2
2
  findDuplicate
3
- } from "./chunk-48ctk154.js";
3
+ } from "./chunk-gr83pq7x.js";
4
4
  import {
5
5
  evidencePath,
6
6
  gateEvidencePath,
7
7
  parseEvidence
8
- } from "./chunk-abswp1vt.js";
8
+ } from "./chunk-t6gbx3jg.js";
9
9
  import {
10
10
  openBlocks,
11
11
  parseQuestions
12
- } from "./chunk-khy21x6e.js";
12
+ } from "./chunk-yanmz0vz.js";
13
13
 
14
14
  // src/core/distill/distill.ts
15
15
  var CONFLICT_THRESHOLD = 0.6;
@@ -1,9 +1,3 @@
1
- import {
2
- GatePolicyError,
3
- STAGE_TUNING_DEFAULTS,
4
- parseWorkflowGates,
5
- readStageTuning
6
- } from "./chunk-g7wz339n.js";
7
1
  import {
8
2
  PROJECT_FRAMEWORK_DIR,
9
3
  STAGES_DIR,
@@ -12,12 +6,109 @@ import {
12
6
  isRecord,
13
7
  loadWorkspace,
14
8
  parseYaml
15
- } from "./chunk-f6t1xc3f.js";
9
+ } from "./chunk-qmp5p77c.js";
16
10
 
17
11
  // src/core/run/workflowPreset.ts
18
12
  import { existsSync, readFileSync } from "node:fs";
19
13
  import { join } from "node:path";
20
14
 
15
+ // src/core/run/stagePolicy.ts
16
+ function validateStagePolicy(value, stageIds, issues, spec) {
17
+ if (value === undefined || value === null)
18
+ return;
19
+ if (!isRecord(value)) {
20
+ issues.push({
21
+ path: spec.field,
22
+ message: `expected a mapping of stage id -> ${spec.policies.join(" | ")}`
23
+ });
24
+ return;
25
+ }
26
+ const known = new Set(stageIds);
27
+ for (const [key, raw] of Object.entries(value)) {
28
+ if (typeof raw !== "string" || !spec.policies.includes(raw)) {
29
+ issues.push({
30
+ path: `${spec.field}.${key}`,
31
+ message: `expected one of ${spec.policies.join(" | ")}, got ${JSON.stringify(raw)}`
32
+ });
33
+ }
34
+ if (known.size > 0 && !known.has(key)) {
35
+ issues.push({ path: `${spec.field}.${key}`, message: `names no stage in this run` });
36
+ }
37
+ }
38
+ }
39
+
40
+ // src/core/run/gatePolicy.ts
41
+ var GATE_POLICIES = ["human", "auto", "agent"];
42
+ var RESERVED_GATE_KEYS = ["collapse"];
43
+
44
+ class GatePolicyError extends Error {
45
+ }
46
+ function isGatePolicy(value) {
47
+ return typeof value === "string" && GATE_POLICIES.includes(value);
48
+ }
49
+ function parseWorkflowGates(value, stageIds, source) {
50
+ if (value === undefined || value === null)
51
+ return {};
52
+ if (!isRecord(value)) {
53
+ throw new GatePolicyError(`${source}: gates must be a mapping of stage id -> ${GATE_POLICIES.join(" | ")}`);
54
+ }
55
+ const known = new Set(stageIds);
56
+ const out = {};
57
+ for (const [key, raw] of Object.entries(value)) {
58
+ if (RESERVED_GATE_KEYS.includes(key))
59
+ continue;
60
+ if (!known.has(key)) {
61
+ throw new GatePolicyError(`${source}: gates names '${key}', which is not one of this workflow's stages (${stageIds.join(", ")})`);
62
+ }
63
+ if (!isGatePolicy(raw)) {
64
+ throw new GatePolicyError(`${source}: gates.${key} must be ${GATE_POLICIES.join(" | ")}, got ${JSON.stringify(raw)}`);
65
+ }
66
+ out[key] = raw;
67
+ }
68
+ return out;
69
+ }
70
+ function validateGatesPolicy(value, stageIds, issues) {
71
+ validateStagePolicy(value, stageIds, issues, { field: "gates_policy", policies: GATE_POLICIES });
72
+ }
73
+
74
+ // src/core/schemas/stageTuning.ts
75
+ var STAGE_TUNING_DEFAULTS = {
76
+ attempts: 2,
77
+ fixlistRounds: 1,
78
+ reviewerShare: 0.25,
79
+ gateSignerShare: 0.25,
80
+ storyCapMultiplier: 3,
81
+ storyCapFloorUsd: 4
82
+ };
83
+ var STAGE_TUNING_RANGES = {
84
+ attempts: { key: "attempts", min: 1, max: 5, integer: true },
85
+ fixlistRounds: { key: "fixlist_rounds", min: 0, max: 3, integer: true },
86
+ reviewerShare: { key: "reviewer_share", min: 0, max: 1, integer: false },
87
+ gateSignerShare: { key: "gate_signer_share", min: 0, max: 1, integer: false },
88
+ storyCapMultiplier: { key: "story_cap_multiplier", min: 1, max: 20, integer: false },
89
+ storyCapFloorUsd: { key: "story_cap_floor_usd", min: 0, max: 200, integer: false }
90
+ };
91
+ var FIELDS = Object.keys(STAGE_TUNING_RANGES);
92
+ function inTuningRange(range, value) {
93
+ if (!Number.isFinite(value))
94
+ return false;
95
+ if (range.integer && !Number.isInteger(value))
96
+ return false;
97
+ return value >= range.min && value <= range.max;
98
+ }
99
+ function readStageTuning(doc) {
100
+ if (!isRecord(doc))
101
+ return STAGE_TUNING_DEFAULTS;
102
+ const out = { ...STAGE_TUNING_DEFAULTS };
103
+ for (const field of FIELDS) {
104
+ const range = STAGE_TUNING_RANGES[field];
105
+ const value = doc[range.key];
106
+ if (typeof value === "number" && inTuningRange(range, value))
107
+ out[field] = value;
108
+ }
109
+ return out;
110
+ }
111
+
21
112
  // src/core/schemas/stage.ts
22
113
  var EFFORT_LEVELS = ["low", "medium", "high", "xhigh", "max"];
23
114
  function isEffortLevel(value) {
@@ -30,11 +121,11 @@ var PRECONDITION_TIMEOUT_S = 60;
30
121
  var PHASE_IDS = ["01-what", "02-how", "03-plan", "04-build", "05-watch"];
31
122
  var PHASE_SEGMENT_RE = /^0[1-5]-[a-z]+$/;
32
123
  var DEFAULT_TIMEOUT_S = 7200;
33
- function buildStageDefaults(root, scope) {
124
+ function buildStageDefaults(root, scope, stageId) {
34
125
  const fallback = { attempts: STAGE_TUNING_DEFAULTS.attempts, timeoutS: DEFAULT_TIMEOUT_S };
35
126
  try {
36
127
  const preset = loadWorkflowPreset(root, scope);
37
- const stage = preset.stages.find((s) => s.phase === PHASE_IDS[3]);
128
+ const stage = (stageId === undefined ? undefined : preset.stages.find((s) => s.id === stageId)) ?? preset.stages.find((s) => s.phase === PHASE_IDS[3]);
38
129
  if (stage === undefined)
39
130
  return fallback;
40
131
  return { attempts: stage.attempts, timeoutS: stage.timeout_s };
@@ -275,4 +366,4 @@ function isString(value) {
275
366
  return typeof value === "string";
276
367
  }
277
368
 
278
- export { PHASE_IDS, DEFAULT_TIMEOUT_S, buildStageDefaults };
369
+ export { validateStagePolicy, GATE_POLICIES, validateGatesPolicy, STAGE_TUNING_DEFAULTS, PHASE_IDS, DEFAULT_TIMEOUT_S, buildStageDefaults };