tldr-experts 0.28.0 → 0.29.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +257 -0
- package/README.md +2 -0
- package/dist/hooks/answer-capture.js +8 -9
- package/dist/hooks/budget-gate.js +47 -11
- package/dist/hooks/{chunk-hws2gnxj.js → chunk-0d4csx4s.js} +1 -1
- package/dist/hooks/{chunk-zrwq6r7k.js → chunk-7xascvee.js} +1 -1
- package/dist/hooks/{chunk-zvxgdssb.js → chunk-857jhjmb.js} +1 -1
- package/dist/hooks/{chunk-c62y35wh.js → chunk-erasdab5.js} +3 -3
- package/dist/hooks/{chunk-veffvzw5.js → chunk-g4d2v4jc.js} +101 -10
- package/dist/hooks/{chunk-kx45kt9n.js → chunk-ggjxes7m.js} +11 -7
- package/dist/hooks/{chunk-48ctk154.js → chunk-gr83pq7x.js} +3 -3
- package/dist/hooks/{chunk-q3d4xdhb.js → chunk-mg3aj0sa.js} +1 -1
- package/dist/hooks/{chunk-tna1bz8r.js → chunk-pne7afnf.js} +1 -1
- package/dist/hooks/{chunk-f6t1xc3f.js → chunk-qmp5p77c.js} +2 -2
- package/dist/hooks/{chunk-abswp1vt.js → chunk-t6gbx3jg.js} +2 -2
- package/dist/hooks/{chunk-xhd77h0p.js → chunk-w3d9kxcp.js} +190 -5
- package/dist/hooks/{chunk-khy21x6e.js → chunk-yanmz0vz.js} +1 -1
- package/dist/hooks/{chunk-khjm356r.js → chunk-z9hv47ps.js} +1 -1
- package/dist/hooks/claim-sources.js +5 -5
- package/dist/hooks/dod-gate.js +9 -10
- package/dist/hooks/no-reask.js +8 -8
- package/dist/hooks/session-start.js +12 -13
- package/dist/hooks/statusline.js +8 -9
- package/dist/tldrx.js +898 -260
- package/package.json +1 -1
- package/plugin/.claude-plugin/plugin.json +1 -1
- package/dist/hooks/chunk-g7wz339n.js +0 -102
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,262 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.29.0 — 2026-09-15
|
|
4
|
+
|
|
5
|
+
### Changed
|
|
6
|
+
|
|
7
|
+
- **A reviewer's `changes` verdict must cite its evidence, or it is refused as a malformed
|
|
8
|
+
envelope (closes #326).** MEASURED on a field run: a reviewer returned `changes` with summary
|
|
9
|
+
"test" and findings `["a","b"]`, and the framework recorded it as a real review — it spent the
|
|
10
|
+
story's attempt, blocked `done`, and would have quoted "test" to the next developer as why the
|
|
11
|
+
story is not done. Nothing checked a `changes` verdict's content; only a `refuted` fix-list
|
|
12
|
+
finding had to carry a citation. Owner decision (2026-09-14): a `changes` now pays the same
|
|
13
|
+
price — its `summary`, or one line of a finding, must END with an `[src: …]` token the §2.8
|
|
14
|
+
parser reads (the diff line that is wrong, the unmet acceptance criterion as
|
|
15
|
+
`03-plan/stories/<id>.md:<line>`, or `absent:<path>` for missing work). One that cites nothing
|
|
16
|
+
is a fault in the REPORT, so it takes the existing bounded format retry — re-prompted with the
|
|
17
|
+
refusal, no attempt spent, two corrections — and the third is recorded as `changes` exactly as
|
|
18
|
+
before. This deliberately reverses the part of #79's pin that called a citation-less `changes` a
|
|
19
|
+
judgement about the work; a `changes` that DOES cite still costs the attempt. The reviewer prompt
|
|
20
|
+
now states the rule under `changes` and carries the `[src: …]` grammar on every review, not only
|
|
21
|
+
when `fixlist` is on the table. The citation is read, not resolved: a token that parses but
|
|
22
|
+
points at nothing still passes, so this refuses the hollow envelope, not a wrong one.
|
|
23
|
+
- **`tldrx run auto` now moves a blocked phase's shortfall out of finished phases by default
|
|
24
|
+
(closes #330).** The rebalance shipped opt-in in #314 because a phase ceiling is a person's
|
|
25
|
+
decision about money. An audit of `run auto`'s intervention episodes across two client
|
|
26
|
+
workspaces ranked stage/phase sizing as the #2 cause of a human stepping in: 22 episodes, 27
|
|
27
|
+
`budget.raised`, 7 `budget.blocked` (counts the audit measured, cited from #330), and not one raise note changed the work
|
|
28
|
+
(the audit's inferred reading of those notes). In a validation run the flag moved money twice
|
|
29
|
+
with nobody present. So a person was signing a move the framework could already prove safe.
|
|
30
|
+
Now a bare `run auto` makes it, and `--no-rebalance-finished` turns it off for a launch whose
|
|
31
|
+
phase ceilings must mean exactly what they say. `--rebalance-finished` is still accepted, and
|
|
32
|
+
passing both flags is exit 1, by name. The rules are unchanged: the run ceiling never grows, no
|
|
33
|
+
move passes a recorded grant, donors must be finished and metered with no unmetered turn, the
|
|
34
|
+
move is the exact shortfall or nothing, and each move is one `budget.raised` with the same
|
|
35
|
+
`source: run auto --rebalance-finished`. `tldrx next` on its own still never rebalances. The
|
|
36
|
+
`tldrx drive` mandate is unchanged on purpose: its unattended mode is refused `run auto`
|
|
37
|
+
outright (`attended_by: host`), and its attended mode drives `tldrx next`. No mandate-driven
|
|
38
|
+
turn ever reaches the move, so a sentence about it would teach something the driver cannot do.
|
|
39
|
+
- **A headless fix list now buys its fix round in the same run, and the approving reviewer's
|
|
40
|
+
signature closes what it was shown (#327, owner decision "A: ronda + auto-cierre").** A reviewer
|
|
41
|
+
that signed with a `fix-now` finding used to park the story at `review`, on the grounds that
|
|
42
|
+
routing a fix list needed a host — measured live, that park stranded the story and the two
|
|
43
|
+
stories of the next wave behind it until a person intervened. Now the story is requeued without
|
|
44
|
+
spending an attempt, the developer is handed the open findings, and the next reviewer is shown
|
|
45
|
+
each one verbatim with the rule "approve only if every one is fixed". Its `approve` rewrites each
|
|
46
|
+
finding it was shown as `Resolved: yes <commit> — auto-closed: …` naming the reviewer session,
|
|
47
|
+
attempt and run — the existing line grammar, so every reader parses it unchanged, and checked
|
|
48
|
+
against git before the story may settle, so a sha that is not on the story branch reopens it. A
|
|
49
|
+
finding that reached the file after the prompt was rendered stays open and still blocks. The
|
|
50
|
+
test that pinned the park ("a headless run … parks the story") is inverted, citing the decision.
|
|
51
|
+
A host review is shown the same findings and closes nothing. No new event, `run.yml` field or
|
|
52
|
+
CLI verb.
|
|
53
|
+
- **An `auto` Build gate no longer stops on paths outside the declared surface — it signs, and
|
|
54
|
+
carries them into the PR (closes #331).** A hold-surface audit of two client workspaces measured
|
|
55
|
+
`boundary` in 48 of 78 Build `held_by` values, the only reason on 6, and 7 intervention episodes
|
|
56
|
+
on it; every one ended in an approval or a `story widen` (the "no real catch" reading is inferred
|
|
57
|
+
from their notes). A condition that stops an unattended run and is always waved through is a
|
|
58
|
+
cost, not a check — but dropping it would lose the one question a reviewer asks first. So the
|
|
59
|
+
fact moves from the gate to the review: the auto note renders the boundary with "does not hold an
|
|
60
|
+
auto gate — it is carried into the PR body for review" and ends in `· warned by: boundary`;
|
|
61
|
+
`gate.requested` gains an additive `warned_by` beside `held_by` (present only when non-empty);
|
|
62
|
+
`tldrx next` prints a `warning: boundary — …` line; `tldrx run status` marks the gate row
|
|
63
|
+
`— warning: boundary`; and `tldrx ship` renders `## Outside declared scope — review these`,
|
|
64
|
+
every outside path, grouped under the story whose own measured `story.touches_widened` row names
|
|
65
|
+
it, the rest under `no story's measured diff names these`. One derivation: the gate and the PR
|
|
66
|
+
body both call `evaluateBoundary`, which now returns the uncapped list beside its detail. An
|
|
67
|
+
`agent` gate still falls to a person on a boundary, and a `human` gate is unchanged. **One
|
|
68
|
+
carve-out keeps the hold where the blast radius is not product code:** an outside path in
|
|
69
|
+
`SENSITIVE_PATH_CLASSES` (one exported constant beside `evaluateBoundary` — `ci`, `secrets`,
|
|
70
|
+
`infra`, `dependencies`) still holds an auto gate, named with its class
|
|
71
|
+
(`app:.github/workflows/deploy.yml [ci]`); declaring it in a story's `touches:` clears it. The
|
|
72
|
+
per-story grouping needs nobody to widen: the executor writes its own measured
|
|
73
|
+
`story.touches_widened` row for each story as it settles, and a real unattended Build then
|
|
74
|
+
`ship` puts the path under the story that wrote it. No exit
|
|
75
|
+
code changed; a check-only refusal with a boundary warning beside it is now eligible for
|
|
76
|
+
`run auto`'s bounded re-run, since the boundary no longer holds it.
|
|
77
|
+
|
|
78
|
+
### Fixed
|
|
79
|
+
|
|
80
|
+
- **A Build gate now names the story a dependent WAITS on, not only the stories that are
|
|
81
|
+
`blocked` (closes #303).** #280 correctly left a dependent of a `review`/`in_progress` story at
|
|
82
|
+
`todo` instead of recording a terminal `blocked` — but the gate only ever named literal
|
|
83
|
+
`blocked` stories (`storiesView`'s `firstBlocked`), so the sentence saying what S2 waits on
|
|
84
|
+
reached `04-build/handoff.md`'s `## Unknowns` and nowhere an operator watches: `gate.requested`,
|
|
85
|
+
its notification and the terminal `stories:` line said `S2 not started` and named no
|
|
86
|
+
dependency. That is the one thing to act on — `story reopen` refuses a story already `todo`,
|
|
87
|
+
and `run auto --until-done` does not relaunch an awaiting-human exit. MEASURED on the #280
|
|
88
|
+
fixture (S1 parked at `review` by a dying reviewer): the payload carried no key for S2 at all.
|
|
89
|
+
Now `gate.requested` and its notification carry three ADDITIVE keys, `waiting_story`,
|
|
90
|
+
`waiting_on` and `waiting_reason` — absent, never null, when nothing waits — and the summary
|
|
91
|
+
and terminal line say S2 waits on S1 (`review`). The wait is derived with the Build loop's own
|
|
92
|
+
`decidingHold`/`dependencyIsPending`, so a story that will BLOCK on a terminal dependency is
|
|
93
|
+
never reported as merely waiting, and `waiting_reason` is `dependencyWaitReason` — the
|
|
94
|
+
`## Unknowns` sentence, byte for byte. `story reopen`'s `depends_on` reader moved to
|
|
95
|
+
`buildProgress.ts` so both ask one reader. No event version or dashboard model bump.
|
|
96
|
+
- **`run auto --until-done` stops a `host-tokens` refusal in tokens, not in dollar words it
|
|
97
|
+
never measured (closes #270).** Every `budget.blocked` went through one dollar-shaped
|
|
98
|
+
sentence, reading `remaining_usd` / `estimate_usd` through a helper that defaults a missing
|
|
99
|
+
field to `0`. The two `host-tokens` writers carry neither field, so a token refusal stopped
|
|
100
|
+
the supervisor with `remaining_usd $0.00 < estimate_usd $0.00` — a confident figure nothing
|
|
101
|
+
measured — and sent the operator to `tldrx budget raise`, a dollar command for a ceiling that
|
|
102
|
+
is a token allowance (and for the headless refusal, a raise that changes nothing: the way out
|
|
103
|
+
is `--prepare` or re-pricing to `metered-usd`, as the row's own reason already says). The stop
|
|
104
|
+
line now switches on `economy`, as the dashboard model already did: a `host-tokens` row names
|
|
105
|
+
the `host_tokens` / `ceiling_tokens` it carries and quotes its `reason`, and a dollar row
|
|
106
|
+
missing its figures says they are not recorded rather than printing `$0.00`. The verdict is
|
|
107
|
+
unchanged — a `budget.blocked` was never relaunched and still is not.
|
|
108
|
+
- **`budget show`, `run estimate` and the `budget-gate` hook price the remaining work with the
|
|
109
|
+
stage's own `attempts:` (closes #214).** `tldrx next`'s brake has resolved the stage's
|
|
110
|
+
`attempts:` since the key existed; these three readers asked `remainingWork` with the shipped
|
|
111
|
+
2. On an `attempts: 1` stage that reserved a second developer turn and a second reviewer the
|
|
112
|
+
executor never dispatches, so the page said BLOCKED and the hook DENIED a `tldrx next` the
|
|
113
|
+
brake itself allowed. Reproduced before the fix on the Build fixture with `attempts: 1`, two
|
|
114
|
+
of three stories done and $5.00 left: the brake allowed ($4.40 of work), while `budget show`
|
|
115
|
+
and `run estimate` quoted $6.40 and the hook denied `tldrx next` on "the stage estimate is
|
|
116
|
+
$6.40". The hook and `budget show` now resolve the cursor stage through the same tolerant
|
|
117
|
+
`buildStageDefaults` the `dod-gate` hook already calls on its PreToolUse path (it now takes
|
|
118
|
+
the stage id; an unreadable preset still gives the shipped 2), and `run estimate` passes the
|
|
119
|
+
stage spec it already loads. `reviewer_share`, `story_cap_multiplier` and
|
|
120
|
+
`story_cap_floor_usd` are still asked with their defaults by these three readers — the same
|
|
121
|
+
disagreement on three more knobs, measured and filed as #333; this change moves `attempts` only.
|
|
122
|
+
- **An in-session Watch turn is recorded as UNMETERED, not as a measured `$0.00` (closes #224).**
|
|
123
|
+
The `--commit` path read the result envelope's own `cost_usd` and defaulted it to `0` — and a
|
|
124
|
+
host session has no reason to fill that field in — while `--cost-usd`, the flag the host
|
|
125
|
+
declares what its sub-agent cost with, was never read anywhere in the Watch executor. Measured
|
|
126
|
+
on disk across two live workspaces: 56 watch task rows at `cost_usd: 0.0` with no `metered` key,
|
|
127
|
+
one of them written by a current release, so every zero was counted by `budget.spent_usd` as a
|
|
128
|
+
real measurement and the lower-bound labelling that keys off `metered: false` could not fire for
|
|
129
|
+
the whole of 05-watch. Watch now uses the same three-value contract Build and the single-agent
|
|
130
|
+
stages already use: `--cost-usd` first, then the envelope's figure, and with NEITHER
|
|
131
|
+
`cost_usd: null` + `metered: false` — a named absence instead of a confident zero. A declared
|
|
132
|
+
`--tokens` rides along on the row the same way. A headless watch turn, which this process really
|
|
133
|
+
does meter, is untouched.
|
|
134
|
+
- **A Watch spawn now records the ceiling it was given (closes #190).** The executor computed a
|
|
135
|
+
per-feature ceiling, handed it to the sub-agent and emitted no `agent.spawned` at all, so a
|
|
136
|
+
watcher feature was the one turn in the framework whose measured cost could not be read against
|
|
137
|
+
what it was allowed to cost — every other stage reconciles `agent.spawned.max_budget_usd`
|
|
138
|
+
against the `agent.result` it paired with. Watch now emits one `agent.spawned` per feature
|
|
139
|
+
before the spawn (a turn that dies is still a turn the ceiling was committed to), carrying
|
|
140
|
+
`phase`, `role: developer`, `model`, `effort`, `max_budget_usd` and `key` — the same `key` the
|
|
141
|
+
`agent.result` for that row already carries, which is what joins the two ends of one turn.
|
|
142
|
+
Deliberately not `story:`: that is the field `tldrx cost --stories` counts a row as a Build
|
|
143
|
+
story by, and a Watch feature id is not a story, so that report is unchanged and whether it
|
|
144
|
+
grows a Watch axis stays a separate decision.
|
|
145
|
+
- **The conflict-turn marker guard no longer counts an unreadable path as a clean one, bounds its
|
|
146
|
+
scan, and reads modified files too (closes #324).** #286's guard caught every read failure and
|
|
147
|
+
returned "holds nothing": a guard failing OPEN, and silently, which is the one direction an audit
|
|
148
|
+
record may never fail in. A path that could not be read is now NAMED with its reason — the errno,
|
|
149
|
+
or the cap that skipped it — and only `ENOENT` stays silent, because a file that is gone cannot
|
|
150
|
+
carry a marker into a commit. The scan is bounded for the first time: 2,000 paths and 4 MiB per
|
|
151
|
+
file, neither of which can bite a normal worktree (this whole repo is 1,037 tracked files and its
|
|
152
|
+
largest is a 766 KB `CHANGELOG.md`, measured on 5256fab), and a path past either cap surfaces the
|
|
153
|
+
same way an unreadable one does rather than passing as clean — an un-ignored generated tree used
|
|
154
|
+
to make the guard as slow and as hungry as that tree was big, with nothing afterwards to say it
|
|
155
|
+
had been. The scope now also includes files MODIFIED since the handed sha, closing the hole where
|
|
156
|
+
a developer pastes a conflicted hunk, markers and all, into an already-tracked file; that
|
|
157
|
+
widening was measured before it was made, not assumed — every `M` row of 1,060 commits of this
|
|
158
|
+
repo's history, 8,366 rows, each row's content against the marker regex, 0 false positives (one
|
|
159
|
+
repo is not every repo). What to DO about an unchecked path is one decision at one call site
|
|
160
|
+
(`UNCHECKED_PATH_POLICY` in `src/core/build/git.ts`): it refuses the attempt with the paths
|
|
161
|
+
named, and a `warn` policy that lets the story walk on with them named on the Build lines is
|
|
162
|
+
implemented beside it.
|
|
163
|
+
- **A developer's `notes` written as an ARRAY of strings is now joined with newlines instead of
|
|
164
|
+
read as `""` by both readers (closes #217).** Measured on a 0.14.3 security-patch run: the
|
|
165
|
+
in-session developer wrote one note per array element, and the two readers' `typeof … ===
|
|
166
|
+
"string"` ternaries — `toEnvelope` on the spawned path, `readResult` on the `--commit` path —
|
|
167
|
+
each threw the whole field away. `notes` is the envelope's one free-text channel, so a turn's
|
|
168
|
+
stated caveats (on that run, what a security patch could NOT verify) vanished from `run.yml`,
|
|
169
|
+
the story log and the handoff with nothing on the `--commit` line saying so — a record wrong in
|
|
170
|
+
the dangerous direction. The elements ARE the notes, so they are joined rather than refused:
|
|
171
|
+
refusing would spend the turn again over a file whose content is not in doubt, on the reader
|
|
172
|
+
that is tolerant by design. Silence was the actual defect, so `tldrx next --commit --check` now
|
|
173
|
+
says which of the three readings a file gets — joined (with the count), dropped by index for a
|
|
174
|
+
non-string element the way `outputs[i]` already is, or still `""` for a `notes` that is neither
|
|
175
|
+
a string nor an array. All three paths, the check included, go through ONE function
|
|
176
|
+
(`readNotes`, `envelope.ts`), so the rehearsal cannot disagree with the commit.
|
|
177
|
+
- **The `budget-gate` hook no longer refuses the `run auto` launch that would fix the shortfall
|
|
178
|
+
(closes #321).** It priced a `tldrx run auto` spawn against the cursor phase's own ceiling
|
|
179
|
+
alone. On a phase already short, it denied the launch and wrote `budget.blocked` before the
|
|
180
|
+
loop could run its in-process rebalance, even when finished phases held the money. With the
|
|
181
|
+
rebalance on by default, that turned into the common case. Reproduced on the hook fixture
|
|
182
|
+
before the fix: 02-how was $2.39 short, finished 01-what held $2.86 unspent, and
|
|
183
|
+
`tldrx run auto` was denied. Now, when finished phases cover the WHOLE shortfall under the
|
|
184
|
+
loop's own rules, the gate allows and says why on stderr, writing no event. It checks that
|
|
185
|
+
with the same `planRebalance`, `--take-from` validation and grant verdict the loop uses, read
|
|
186
|
+
against the full `run.yml` so a stale donor still counts as unfinished. The gate itself moves
|
|
187
|
+
nothing; the move and its `budget.raised` are the loop's. Denied exactly as before:
|
|
188
|
+
`tldrx next`, a launch with `--no-rebalance-finished`, a run-ceiling shortfall, a shortfall
|
|
189
|
+
the finished phases cover only in part, a stale donor, and a move past a recorded grant.
|
|
190
|
+
- **A re-reviewed story's `changes` is consumed instead of parking the story (closes #327).** A
|
|
191
|
+
story whose earlier reviewer died or was unfunded is re-reviewed alone, and that path returned
|
|
192
|
+
right after the verdict — so a `changes` with attempts left parked at `review` and its
|
|
193
|
+
dependents never started, on the serial path and in a wave alike. It now enters the attempt loop
|
|
194
|
+
(or the wave's next round) under the same ledger bound every other requeue uses.
|
|
195
|
+
- **A story `blocked` on an approving review whose fix list was closed later now settles `done`
|
|
196
|
+
(closes #329).** Measured live: a fix landed, the reviewer approved, the story blocked on a
|
|
197
|
+
`Resolved: no` nobody had rewritten — and after a person rewrote it, `story reopen` dispatched a
|
|
198
|
+
developer with nothing to change and `--as-is` refused a branch already on its epic. Before the
|
|
199
|
+
frontier walk, `tldrx next` and `--prepare` now settle such a story `done` with no agent spawned,
|
|
200
|
+
after holding every `Resolved: yes` to git, and say why in one line. The refusal that blocks a
|
|
201
|
+
story on its fix list names that remedy instead of `tldrx story reopen`.
|
|
202
|
+
- **A parallel Build wave no longer hands two developers the same unspent money (closes #325).**
|
|
203
|
+
Every per-story cap reads the stage's METERED remainder, which is only right when stories run
|
|
204
|
+
one at a time. Measured live: a $21.60 stage spawned S1 under $21.00 and S2 under $16.50 in one
|
|
205
|
+
wave before either had metered a cent, spend landed at $33.07, and S2's reviewer was refused on
|
|
206
|
+
"$0.00 left" — the build ended 0 of 4. Reproduced on the fake agent before the fix: two lanes on
|
|
207
|
+
a $20 stage were spawned at $9.00 each, $22 claimed with their review floors. Now each lane is
|
|
208
|
+
capped at the stage's remainder less the caps of the lanes still running and a $2.00 reviewer
|
|
209
|
+
floor for every story whose review is still ahead, its own included. A lane that bound would
|
|
210
|
+
put under the least its developer is already allowed (`story_cap_floor_usd` for a priced story,
|
|
211
|
+
its uniform share otherwise) waits for a lane to finish, and the run prints `S2: not dispatched
|
|
212
|
+
beside S1 yet —` with the remainder, the reservation, the floors and the bound; with nothing in
|
|
213
|
+
flight nothing can be freed by waiting, so it runs at that floor instead of stalling. The
|
|
214
|
+
consequence to know: a small stage can now serialise a wide wave — measured, the third of three
|
|
215
|
+
unpriced stories on an $8 stage waits for the other two to meter. `--parallel 1` is
|
|
216
|
+
byte-identical (golden unchanged). No new config key, event or field. Not in this
|
|
217
|
+
change: sizing the stage from the plan once it exists, and `budget show`
|
|
218
|
+
naming what its estimate excludes — both named on #325.
|
|
219
|
+
- **The Plan prompt states the per-item character cap where each capped field is described, and
|
|
220
|
+
an over-cap finding now survives into the retry (closes #328).** Two plan stages failed on the
|
|
221
|
+
same cap in one day: a 683-character `test_plan` item, and a `wave_cap_reason` — held to the
|
|
222
|
+
same cap at the gate, and stated nowhere the Plan agent reads before writing it. The
|
|
223
|
+
`acceptance` and `test_plan` rows now say "one sentence each; split a long … into several
|
|
224
|
+
items", and the Plan-shape wave-cap rule and the Caps list name `wave_cap_reason`'s cap — all
|
|
225
|
+
rendered from `MAX_ITEM_CHARS`, never a typed number. The retry half was measured, not assumed:
|
|
226
|
+
a failed check's reason reaches the next `## Previous attempt` squeezed to one line with its
|
|
227
|
+
middle elided, and on a one-defect plan the old message lost "split it into several items" to
|
|
228
|
+
that ellipsis. The finding is now `<n> characters (cap <cap>) — split it into several items`,
|
|
229
|
+
short enough that the field, the length, the cap and the instruction reach the retry; an
|
|
230
|
+
over-cap `wave_cap_reason` is told its length and the cap too. A prompt byte change by
|
|
231
|
+
design; no Build golden moves. Not in this change: the facilitator's one-line squeeze itself,
|
|
232
|
+
which still elides whatever a longer multi-finding reason puts in the middle.
|
|
233
|
+
|
|
234
|
+
## 0.28.1 — 2026-09-15
|
|
235
|
+
|
|
236
|
+
### Fixed
|
|
237
|
+
|
|
238
|
+
- **The note a story was reopened with now reaches the developer and the reviewer (closes #322).**
|
|
239
|
+
On a field run a person reopened a story with a note naming the exact gap two reviews had
|
|
240
|
+
refused it for; the next developer made a docstring-only commit and a reviewer approved it, and
|
|
241
|
+
neither turn had been shown the note. It was written only into `story.reopened` and read back
|
|
242
|
+
for a report line and #308's no-diff check — while the refusal for a `--for-fix` with no note
|
|
243
|
+
told the operator the note "is what scopes the fix round, and what the reviewer reads". Measured
|
|
244
|
+
before the fix: 0 of the post-reopen developer and reviewer prompts contained the note, for a
|
|
245
|
+
plain reopen and for `--for-fix`, and after a plain reopen the developer prompt had no
|
|
246
|
+
`## Previous attempt` either, so it was told nothing about why the story came back. Both
|
|
247
|
+
prompts now carry `## Why this story was reopened` — who signed it, whether it is a fix round,
|
|
248
|
+
the note verbatim — read off the same ledger value #308 already reads, so the prompt and the
|
|
249
|
+
refusal cannot name different notes; a story nobody reopened renders nothing new. The
|
|
250
|
+
reviewer's copy tells it that a diff leaving the named gap as it was is `changes`, but that
|
|
251
|
+
instruction is **prompt-advisory**: the framework does not enforce a `changes` verdict against
|
|
252
|
+
the note, and an `approve` is still accepted as the reviewer's judgement.
|
|
253
|
+
- **`tldrx note --help` says where an operator note goes, and where it does not (closes #151,
|
|
254
|
+
docs only).** An `operator_note` is read by `run status`, `replay` and the dashboard and by no
|
|
255
|
+
prompt — so a note an owner wrote for the next stage's agents silently reached nobody. The gap
|
|
256
|
+
was the unwritten contract, not a missing reader: the help and the CLI reference now say the
|
|
257
|
+
note is for people and name `.agent/<stage>/dispatch-notes.md` as the channel that reaches
|
|
258
|
+
agents.
|
|
259
|
+
|
|
3
260
|
## 0.28.0 — 2026-09-15
|
|
4
261
|
|
|
5
262
|
### Added
|
package/README.md
CHANGED
|
@@ -335,6 +335,8 @@ back on the registry is 0.3.0.
|
|
|
335
335
|
|
|
336
336
|
| Version | Date | Status | Contains |
|
|
337
337
|
|---|---|---|---|
|
|
338
|
+
| 0.29.0 | 2026-09-15 | `beta` | Sixteen changes that make an unattended run trust its own gates and its own numbers. Verdict integrity: a reviewer's `changes` must cite evidence (#326), a re-reviewed story's `changes` is consumed instead of parking it (#327), and a story blocked on a stale fix list settles `done` once resolved (#329). Fix-list flow: a headless fix list now buys its own fix round, and an approving reviewer's signature closes what it was shown (#327). Money and phase sizing: `run auto` rebalances a blocked phase's shortfall out of finished phases by default (#330), the `budget-gate` hook stops denying the launch that would fix the shortfall (#321), a Build gate no longer stops on paths outside the declared surface but carries them into the PR instead (#331), a parallel wave no longer hands two developers the same unspent money (#325), `budget show`/`run estimate`/the budget-gate hook price remaining work off the stage's own `attempts:` (#214), `run auto --until-done` states a token refusal in tokens rather than invented dollar words (#270), a Watch turn is recorded as unmetered rather than a false `$0.00` (#224), and a Watch spawn now records the ceiling it was given (#190). Dependency and safety: a Build gate names the story a dependent waits on (#303), the conflict-turn marker guard names unreadable paths, bounds its scan, and covers modified files too (#324), a developer's array-valued `notes` is joined instead of silently dropped (#217), and the Plan prompt states its per-item character cap where each field is described so an over-cap finding survives into the retry (#328). |
|
|
339
|
+
| 0.28.1 | 2026-09-15 | `beta` | Two fixes so a reopened story's reason reaches the agents that need it. The developer and reviewer prompts now carry `## Why this story was reopened` — who signed it, whether it's a fix round, the note verbatim — off the same ledger value #308 reads, instead of the note reaching only a report line and #308's check (closes #322). `tldrx note --help` now says an operator note reaches no agent and names `dispatch-notes.md` as the channel that does (closes #151). |
|
|
338
340
|
| 0.28.0 | 2026-09-15 | `beta` | Four changes that keep an unattended run moving instead of stranding on a person. A story whose merge-up to its epic conflicts in at most three of its own touched files gets one automated conflict turn instead of a person, guarded against committing conflict markers (#286). `story reopen` on a blocked story now releases every dependent it alone was holding, in the same command (#312). Seed's `Recommended:` line and the loop's question parser now share one grammar, so a citation-only recommendation no longer parks a run that `seed check` had already passed (#323). And an `absent:` needle over `facts.yml` no longer trips on a fact's own recording metadata, while an `auto` gate refused only by failed checks gets one automatic re-run before waiting for a person (#231). |
|
|
339
341
|
| 0.27.0 | 2026-09-14 | `beta` | Five changes for unattended runs, from a planning audit and live field measurement. The `plan` check now refuses a plan with more waves than the framework carries per run (unless `waves.yml` names a `wave_cap_reason`), refuses a story scheduled before its `depends_on` allows, and flags a dod command siblings carry that one story doesn't (#316–#319). `tldrx run auto --rebalance-finished` moves a blocked phase's exact shortfall out of a finished phase's unspent ceiling before refusing, through `budget raise --take-from` itself, so the run ceiling never grows and no grant is ever assumed; every money refusal now names the unspent total and the exact command to fix it (#314). `tldrx ship` fetches and merges the base into the epic and re-runs the `done` stories' DoD before opening a PR, so a stale or red epic refuses instead of shipping a PR that comes back red (#315). A reviewer's budget floor rises from $1.00 to $2.00, matching what completed reviews actually cost, so a review no longer dies mid-diff on a floor that was never measured (#307). And a developer whose DoD goes red gets the story's next attempt instead of blocking on the first miss, with the kept output handed to the next attempt and the bound counted from `events.jsonl` so it holds across process restarts; a refused developer, a cap death, or a broken dod command still blocks immediately (#313). |
|
|
340
342
|
| 0.26.1 | 2026-09-14 | `beta` | Three refusals that stop a bad state from settling quietly. A Build fix round whose developer lands no diff is refused before the DoD instead of closing `done` on an unmoved tree, because `commitIfDirty` hands back the old head sha on a clean commit and the existing "no commit to review" gate never fires for it — `workSince` is now asked on the success path too, for any story a person put back with a note. `tldrx ship` refuses an epic carrying a story the reviewer marked `changes`: it asks the ledger's `lastMerge` for every story not `done` and exits 2 naming the story, the merge and the handoff's reason, instead of opening a PR whose "Not done" list hid rejected code already merged into the diff. And a stage-failure line keeps the branch it names when the worktree path is wider than the line: the refusal puts the branches before the path, and `oneLine` keeps a line's head and tail instead of dropping the identifiers that sat at the end. |
|
|
@@ -1,24 +1,23 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
import {
|
|
3
3
|
conflictOf
|
|
4
|
-
} from "./chunk-
|
|
4
|
+
} from "./chunk-erasdab5.js";
|
|
5
5
|
import {
|
|
6
6
|
FactsStore,
|
|
7
7
|
formatJaccard
|
|
8
|
-
} from "./chunk-
|
|
8
|
+
} from "./chunk-gr83pq7x.js";
|
|
9
9
|
import {
|
|
10
10
|
parseHookInput,
|
|
11
11
|
readStdin
|
|
12
|
-
} from "./chunk-
|
|
12
|
+
} from "./chunk-z9hv47ps.js";
|
|
13
13
|
import {
|
|
14
14
|
EventLog,
|
|
15
15
|
PHASE_ID_RE
|
|
16
|
-
} from "./chunk-
|
|
16
|
+
} from "./chunk-t6gbx3jg.js";
|
|
17
17
|
import {
|
|
18
18
|
PHASE_IDS
|
|
19
|
-
} from "./chunk-
|
|
20
|
-
import"./chunk-
|
|
21
|
-
import"./chunk-zvxgdssb.js";
|
|
19
|
+
} from "./chunk-g4d2v4jc.js";
|
|
20
|
+
import"./chunk-857jhjmb.js";
|
|
22
21
|
import {
|
|
23
22
|
ADVISORY_KEY,
|
|
24
23
|
MAX_FACT_CHARS,
|
|
@@ -28,7 +27,7 @@ import {
|
|
|
28
27
|
renderQuestionBlock,
|
|
29
28
|
replaceBlock,
|
|
30
29
|
serializeQuestions
|
|
31
|
-
} from "./chunk-
|
|
30
|
+
} from "./chunk-yanmz0vz.js";
|
|
32
31
|
import {
|
|
33
32
|
ITERATION_ONLY_SLOT,
|
|
34
33
|
PROJECT_FRAMEWORK_DIR,
|
|
@@ -38,7 +37,7 @@ import {
|
|
|
38
37
|
isScopedTemplate,
|
|
39
38
|
parseYaml,
|
|
40
39
|
scopedSlotOf
|
|
41
|
-
} from "./chunk-
|
|
40
|
+
} from "./chunk-qmp5p77c.js";
|
|
42
41
|
|
|
43
42
|
// src/hooks/answer-capture.ts
|
|
44
43
|
import { existsSync as existsSync5 } from "fs";
|
|
@@ -1,31 +1,36 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
import {
|
|
3
3
|
budgetGateDeny
|
|
4
|
-
} from "./chunk-
|
|
4
|
+
} from "./chunk-mg3aj0sa.js";
|
|
5
5
|
import {
|
|
6
6
|
allow,
|
|
7
7
|
deny,
|
|
8
8
|
readPayload,
|
|
9
9
|
runHook,
|
|
10
10
|
toolInput
|
|
11
|
-
} from "./chunk-
|
|
12
|
-
import"./chunk-
|
|
11
|
+
} from "./chunk-pne7afnf.js";
|
|
12
|
+
import"./chunk-z9hv47ps.js";
|
|
13
13
|
import {
|
|
14
|
+
REBALANCE_SOURCE,
|
|
15
|
+
applyRebalance,
|
|
14
16
|
asRunBudget,
|
|
15
17
|
currentActor,
|
|
18
|
+
describeRebalance,
|
|
16
19
|
economyFor,
|
|
17
20
|
isHostTokens,
|
|
18
21
|
nowRfc3339,
|
|
22
|
+
planRebalance,
|
|
19
23
|
raiseCommand,
|
|
24
|
+
raiseGrantVerdict,
|
|
20
25
|
remainingWork,
|
|
21
26
|
shortBy,
|
|
22
27
|
validateRunBudget,
|
|
23
28
|
wouldExceed,
|
|
24
29
|
wouldExceedHostTokens
|
|
25
|
-
} from "./chunk-
|
|
30
|
+
} from "./chunk-w3d9kxcp.js";
|
|
26
31
|
import {
|
|
27
32
|
EventLog
|
|
28
|
-
} from "./chunk-
|
|
33
|
+
} from "./chunk-t6gbx3jg.js";
|
|
29
34
|
import {
|
|
30
35
|
cursorStage,
|
|
31
36
|
hostTokensIn,
|
|
@@ -34,18 +39,20 @@ import {
|
|
|
34
39
|
newestActiveRun,
|
|
35
40
|
renderRunEconomies,
|
|
36
41
|
runSpend
|
|
37
|
-
} from "./chunk-
|
|
38
|
-
import
|
|
42
|
+
} from "./chunk-0d4csx4s.js";
|
|
43
|
+
import {
|
|
44
|
+
buildStageDefaults
|
|
45
|
+
} from "./chunk-g4d2v4jc.js";
|
|
39
46
|
import {
|
|
40
47
|
noteDeprecations
|
|
41
|
-
} from "./chunk-
|
|
48
|
+
} from "./chunk-yanmz0vz.js";
|
|
42
49
|
import {
|
|
43
50
|
PROJECT_WORK_DIR,
|
|
44
51
|
findWorkspaceRoot,
|
|
45
52
|
locateWork,
|
|
46
53
|
parseYaml,
|
|
47
54
|
stageYamlPath
|
|
48
|
-
} from "./chunk-
|
|
55
|
+
} from "./chunk-qmp5p77c.js";
|
|
49
56
|
|
|
50
57
|
// src/hooks/budget-gate.ts
|
|
51
58
|
import { existsSync as existsSync2, readFileSync as readFileSync2, statSync } from "node:fs";
|
|
@@ -72,6 +79,8 @@ function loadRunBudget(runDir) {
|
|
|
72
79
|
// src/hooks/budget-gate.ts
|
|
73
80
|
var SPAWN_RE = /^(claude -p|tldrx next|tldrx run auto|tldrx expert train|tldrx seed triage)\b/;
|
|
74
81
|
var RUN_ARG_RE = /--run[= ]([\w.-]+)/;
|
|
82
|
+
var RUN_AUTO_RE = /^tldrx run auto\b/;
|
|
83
|
+
var NO_REBALANCE_RE = /(?:^|\s)--no-rebalance-finished(?:[\s=]|$)/;
|
|
75
84
|
var DEFAULT_TRAIN_USD = 2;
|
|
76
85
|
var DEFAULT_FULL_TRAIN_USD = 3;
|
|
77
86
|
var DEFAULT_TRIAGE_USD = 1;
|
|
@@ -116,6 +125,7 @@ await runHook("budget-gate", async () => {
|
|
|
116
125
|
const attended = isAttendedByHostView(view);
|
|
117
126
|
const stage = cursorStage(view);
|
|
118
127
|
const declared = stage?.budget_usd ?? stageBudgetFromLibrary(root, view.cursor.stage);
|
|
128
|
+
const attempts = buildStageDefaults(root, view.scope, view.cursor.stage).attempts;
|
|
119
129
|
const work = declared === null ? null : remainingWork({
|
|
120
130
|
runDir: view.dir,
|
|
121
131
|
phaseId: view.cursor.phase,
|
|
@@ -124,7 +134,8 @@ await runHook("budget-gate", async () => {
|
|
|
124
134
|
perAgentMaxUsd: budget.per_agent_max_usd,
|
|
125
135
|
maxUsd: null,
|
|
126
136
|
economy: economyFor(budget, view.cursor.phase),
|
|
127
|
-
attended
|
|
137
|
+
attended,
|
|
138
|
+
attempts
|
|
128
139
|
});
|
|
129
140
|
const estimate = estimateFor(command, work === null ? null : work.usd);
|
|
130
141
|
if (estimate <= 0)
|
|
@@ -156,7 +167,7 @@ ${economies}`}`);
|
|
|
156
167
|
`);
|
|
157
168
|
return;
|
|
158
169
|
}
|
|
159
|
-
const decision = wouldExceed(budget, view.cursor.phase, estimate);
|
|
170
|
+
const decision = wouldExceed(budget, view.cursor.phase, estimate, attempts);
|
|
160
171
|
if (!decision.blocked)
|
|
161
172
|
return;
|
|
162
173
|
if (attended) {
|
|
@@ -176,6 +187,15 @@ ${economies}`}`);
|
|
|
176
187
|
`);
|
|
177
188
|
return;
|
|
178
189
|
}
|
|
190
|
+
if (decision.scope === "phase" && RUN_AUTO_RE.test(command) && !NO_REBALANCE_RE.test(command)) {
|
|
191
|
+
const short = shortBy(estimate, decision.remaining);
|
|
192
|
+
const covered = finishedPhasesCover(view.dir, budget, view.cursor.phase, short);
|
|
193
|
+
if (covered !== null) {
|
|
194
|
+
process.stderr.write(`tldrx hook budget-gate: ${view.cursor.phase} has $${decision.remaining.toFixed(2)} left of ` + `$${decision.ceiling.toFixed(2)} and the stage estimate is $${estimate.toFixed(2)} — NOT refusing: ` + `\`run auto\` rebalances finished phases before it refuses, and ${covered}, which covers the ` + `$${short.toFixed(2)} shortfall. The loop makes that move on its own record (\`budget.raised\`, ` + `source: ${REBALANCE_SOURCE}); launch with --no-rebalance-finished to be refused here instead.` + `${economies === null ? "" : ` ${economies}`}
|
|
195
|
+
`);
|
|
196
|
+
return;
|
|
197
|
+
}
|
|
198
|
+
}
|
|
179
199
|
new EventLog(join2(view.dir, "events.jsonl")).tryAppend({
|
|
180
200
|
ts: nowRfc3339(),
|
|
181
201
|
run: view.run,
|
|
@@ -226,6 +246,22 @@ function estimateFor(command, stageBudget) {
|
|
|
226
246
|
}
|
|
227
247
|
return stageBudget ?? 0;
|
|
228
248
|
}
|
|
249
|
+
function finishedPhasesCover(runDir, budget, phaseId, shortUsd) {
|
|
250
|
+
try {
|
|
251
|
+
const doc = parseYaml(readFileSync2(join2(runDir, "run.yml"), "utf8"));
|
|
252
|
+
if (doc === null || !Array.isArray(doc.phases))
|
|
253
|
+
return null;
|
|
254
|
+
const plan = planRebalance(budget, doc, phaseId, shortUsd);
|
|
255
|
+
if (plan.moves.length === 0)
|
|
256
|
+
return null;
|
|
257
|
+
const applied = applyRebalance(budget, plan);
|
|
258
|
+
if (applied.outcomes.some((outcome) => raiseGrantVerdict(budget, outcome).exceeds))
|
|
259
|
+
return null;
|
|
260
|
+
return describeRebalance(plan);
|
|
261
|
+
} catch {
|
|
262
|
+
return null;
|
|
263
|
+
}
|
|
264
|
+
}
|
|
229
265
|
function failClosed(command, why) {
|
|
230
266
|
deny(`[tldrx] budget-gate: refusing \`${command.slice(0, 80)}\` — this gate could not read the budget ` + `it is supposed to enforce (${why}).
|
|
231
267
|
` + "It fails CLOSED: a spend nothing can check is exactly the one that must not start. Fix the run's " + "budget.yml, or pass `--run <id>` so the gate knows which run to charge.");
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import {
|
|
2
2
|
PROJECT_FRAMEWORK_DIR,
|
|
3
3
|
PROJECT_WORK_DIR
|
|
4
|
-
} from "./chunk-
|
|
4
|
+
} from "./chunk-qmp5p77c.js";
|
|
5
5
|
|
|
6
6
|
// src/core/lock/workspaceLock.ts
|
|
7
7
|
import { existsSync as existsSync2, mkdirSync as mkdirSync2, openSync, readFileSync as readFileSync2, rmSync as rmSync2, writeSync, closeSync } from "node:fs";
|
|
@@ -1,15 +1,15 @@
|
|
|
1
1
|
import {
|
|
2
2
|
findDuplicate
|
|
3
|
-
} from "./chunk-
|
|
3
|
+
} from "./chunk-gr83pq7x.js";
|
|
4
4
|
import {
|
|
5
5
|
evidencePath,
|
|
6
6
|
gateEvidencePath,
|
|
7
7
|
parseEvidence
|
|
8
|
-
} from "./chunk-
|
|
8
|
+
} from "./chunk-t6gbx3jg.js";
|
|
9
9
|
import {
|
|
10
10
|
openBlocks,
|
|
11
11
|
parseQuestions
|
|
12
|
-
} from "./chunk-
|
|
12
|
+
} from "./chunk-yanmz0vz.js";
|
|
13
13
|
|
|
14
14
|
// src/core/distill/distill.ts
|
|
15
15
|
var CONFLICT_THRESHOLD = 0.6;
|