froid-loop 0.11.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. froid_loop/__init__.py +11 -0
  2. froid_loop/__main__.py +12 -0
  3. froid_loop/adapters/__init__.py +3 -0
  4. froid_loop/adapters/base.py +254 -0
  5. froid_loop/adapters/entrypoints.py +63 -0
  6. froid_loop/adapters/env_fault.py +290 -0
  7. froid_loop/adapters/generic.py +2013 -0
  8. froid_loop/adapters/mock.py +49 -0
  9. froid_loop/adapters/multiplexer.py +914 -0
  10. froid_loop/adapters/opencode_http.py +1687 -0
  11. froid_loop/adapters/profile.py +650 -0
  12. froid_loop/adapters/psmux_backend.py +1428 -0
  13. froid_loop/adapters/registry.py +322 -0
  14. froid_loop/adapters/tmux_backend.py +35 -0
  15. froid_loop/adapters/tmux_base.py +630 -0
  16. froid_loop/checks.py +187 -0
  17. froid_loop/cli.py +5041 -0
  18. froid_loop/data/__init__.py +0 -0
  19. froid_loop/data/froid_loop_hook.py +228 -0
  20. froid_loop/data/froid_loop_probe_hook.py +88 -0
  21. froid_loop/data/plugins/example/plugin.toml +21 -0
  22. froid_loop/data/plugins/tea/plugin.toml +184 -0
  23. froid_loop/data/plugins/tea/tea_plugin.py +258 -0
  24. froid_loop/data/plugins/unity/plugin.toml +140 -0
  25. froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef +16 -0
  26. froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef.meta +7 -0
  27. froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs +221 -0
  28. froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs.meta +11 -0
  29. froid_loop/data/plugins/unity/unity_assets/_folders/Editor.meta +8 -0
  30. froid_loop/data/plugins/unity/unity_assets/_folders/FroidLoop.meta +8 -0
  31. froid_loop/data/plugins/unity/unity_cleanup.py +125 -0
  32. froid_loop/data/plugins/unity/unity_dialog_probe.py +239 -0
  33. froid_loop/data/plugins/unity/unity_facts.md +17 -0
  34. froid_loop/data/plugins/unity/unity_plugin.py +415 -0
  35. froid_loop/data/plugins/unity/unity_quiesce.py +234 -0
  36. froid_loop/data/plugins/unity/unity_ready.py +230 -0
  37. froid_loop/data/plugins/unity/unity_seed_assets.py +298 -0
  38. froid_loop/data/plugins/unity/unity_setup.py +551 -0
  39. froid_loop/data/plugins/unity/unity_teardown.py +362 -0
  40. froid_loop/data/profiles/antigravity.toml +52 -0
  41. froid_loop/data/profiles/claude.toml +85 -0
  42. froid_loop/data/profiles/codex.toml +22 -0
  43. froid_loop/data/profiles/copilot.toml +52 -0
  44. froid_loop/data/profiles/gemini.toml +26 -0
  45. froid_loop/data/profiles/opencode.toml +54 -0
  46. froid_loop/data/settings/core.toml +458 -0
  47. froid_loop/data/skills/README.md +93 -0
  48. froid_loop/data/skills/froid-loop-resolve/SKILL.md +288 -0
  49. froid_loop/data/skills/froid-loop-setup/SKILL.md +161 -0
  50. froid_loop/data/skills/froid-loop-setup/assets/module-help.csv +3 -0
  51. froid_loop/data/skills/froid-loop-setup/assets/module.yaml +19 -0
  52. froid_loop/data/skills/froid-loop-sweep/SKILL.md +100 -0
  53. froid_loop/data/skills/froid-loop-sweep/automation-mode.md +127 -0
  54. froid_loop/data/skills/froid-loop-sweep/deferred-work-format.md +302 -0
  55. froid_loop/data/skills/froid-loop-sweep/migration-mode.md +86 -0
  56. froid_loop/decisions.py +202 -0
  57. froid_loop/deferredwork.py +2282 -0
  58. froid_loop/devcontract.py +892 -0
  59. froid_loop/diagnostics.py +1104 -0
  60. froid_loop/documents.py +532 -0
  61. froid_loop/engine.py +7732 -0
  62. froid_loop/envvars.py +111 -0
  63. froid_loop/escalation.py +225 -0
  64. froid_loop/events.py +266 -0
  65. froid_loop/fences.py +103 -0
  66. froid_loop/froidconfig.py +226 -0
  67. froid_loop/frontmatter.py +526 -0
  68. froid_loop/gates.py +133 -0
  69. froid_loop/install.py +2936 -0
  70. froid_loop/journal.py +178 -0
  71. froid_loop/machine.py +148 -0
  72. froid_loop/model.py +898 -0
  73. froid_loop/operatoractions.py +474 -0
  74. froid_loop/platform_util.py +1490 -0
  75. froid_loop/plugins/__init__.py +64 -0
  76. froid_loop/plugins/bus.py +259 -0
  77. froid_loop/plugins/context.py +319 -0
  78. froid_loop/plugins/loader.py +145 -0
  79. froid_loop/plugins/manifest.py +279 -0
  80. froid_loop/plugins/model.py +296 -0
  81. froid_loop/plugins/registry.py +245 -0
  82. froid_loop/plugins/trust.py +75 -0
  83. froid_loop/policy.py +1569 -0
  84. froid_loop/probe.py +1044 -0
  85. froid_loop/process_host.py +408 -0
  86. froid_loop/recovery_flow.py +1561 -0
  87. froid_loop/resolve.py +283 -0
  88. froid_loop/runs.py +4715 -0
  89. froid_loop/runsetup.py +1293 -0
  90. froid_loop/sanitize.py +593 -0
  91. froid_loop/settings_schema.py +276 -0
  92. froid_loop/signals.py +160 -0
  93. froid_loop/sprintstatus.py +609 -0
  94. froid_loop/statemachine.py +57 -0
  95. froid_loop/stories.py +615 -0
  96. froid_loop/stories_engine.py +796 -0
  97. froid_loop/sweep.py +1892 -0
  98. froid_loop/tokens.py +196 -0
  99. froid_loop/tui/__init__.py +11 -0
  100. froid_loop/tui/app.py +1584 -0
  101. froid_loop/tui/data.py +840 -0
  102. froid_loop/tui/launch.py +1003 -0
  103. froid_loop/tui/screens/__init__.py +1 -0
  104. froid_loop/tui/screens/dashboard.py +1071 -0
  105. froid_loop/tui/screens/modals.py +943 -0
  106. froid_loop/tui/screens/settings_screen.py +477 -0
  107. froid_loop/tui/settings.py +135 -0
  108. froid_loop/tui/widgets.py +981 -0
  109. froid_loop/verify.py +4545 -0
  110. froid_loop/workspace.py +320 -0
  111. froid_loop/worktree_flow.py +2301 -0
  112. froid_loop-0.11.1.dist-info/METADATA +728 -0
  113. froid_loop-0.11.1.dist-info/RECORD +116 -0
  114. froid_loop-0.11.1.dist-info/WHEEL +4 -0
  115. froid_loop-0.11.1.dist-info/entry_points.txt +2 -0
  116. froid_loop-0.11.1.dist-info/licenses/LICENSE +30 -0
@@ -0,0 +1,2282 @@
1
+ """Deterministic reading and editing of the deferred-work ledger.
2
+
3
+ The ledger (`{implementation_artifacts}/deferred-work.md`) is append-only
4
+ markdown in the canonical form documented at
5
+ froid-loop-sweep/deferred-work-format.md: `### DW-<seq>: <title>` headings with
6
+ `origin:`/`location:`/`reason:`/`status:` field lines. The one sanctioned
7
+ rewrite is :func:`archive_closed`, which moves closed entries verbatim to a
8
+ sibling archive file and leaves id-preserving stubs. Pre-#2651 dev primitives
9
+ and the attended `froid-build` append flatter entries here directly, which the
10
+ orchestrator normalizes on sweep; the current unattended primitive records its
11
+ findings in the spec's frontmatter instead and the engine harvests them into
12
+ canonical entries. The
13
+ orchestrator never trusts an LLM to have edited it — status flips and decision
14
+ records happen here, and gates re-read the file from disk.
15
+
16
+ Concurrency (#286/#469): every mutator below is a read->edit->write of the whole
17
+ file, so two orchestrator processes — a second `froid-loop run`, a run plus a
18
+ sweep, a run plus the TUI decision modal, a run plus `sweep --archive` — would
19
+ otherwise both read, both edit, and let the last atomic write win. Each leaf
20
+ mutator therefore runs its whole read->edit->write under :func:`ledger_lock`, a
21
+ cross-process mutex on an out-of-repo sidecar. Readers stay lock-free on
22
+ purpose: every writer replaces the file atomically, so a reader already sees one
23
+ whole version or another, and taking the lock to read would buy nothing while
24
+ adding a way to deadlock. Out of scope by #286's own non-goals: the dev/review
25
+ LLM session writes this file directly and does NOT take the lock — orchestrator
26
+ writes are sequenced against sessions today, so the exposure this closes is
27
+ orchestrator-vs-orchestrator.
28
+
29
+ What the hold covers is every read that decides the PUBLISHED BYTES, which is
30
+ not quite every read (#736). A mutator handed work that turns out to be a no-op
31
+ — ids that are all already done, a decision on an entry that is not there,
32
+ specs that all dedupe, nothing eligible to archive — may answer from ONE
33
+ advisory read taken before the lock, running the same pure decision helper the
34
+ locked pass runs so the two cannot drift. Only a "would write nothing" answer
35
+ is acted on, and such a call linearizes at the probe read: it publishes no
36
+ bytes, so there is nothing for a rival to interleave with. Every other answer,
37
+ and any fault during the probe, falls through to the hold, which re-reads and
38
+ decides authoritatively. This is what keeps a no-op from failing on a lock it
39
+ never needed — an `OSError` from acquisition, or a
40
+ :class:`~froid_loop.runs.StateRootError` from deriving the sidecar path where no
41
+ state root exists — which a replayed rollback, a re-run sweep and
42
+ ``sweep --archive`` all reach routinely.
43
+ """
44
+
45
+ from __future__ import annotations
46
+
47
+ import hashlib
48
+ import re
49
+ import threading
50
+ from bisect import bisect_right
51
+ from collections.abc import Iterator, Sequence
52
+ from contextlib import contextmanager
53
+ from dataclasses import dataclass
54
+ from datetime import date as calendar_date
55
+ from pathlib import Path
56
+
57
+ from . import sprintstatus
58
+ from .fences import fenced_spans
59
+ from .platform_util import atomic_write_text, file_lock, neutralize_surrogates
60
+
61
+ HEADING_RE = re.compile(r"^### (DW-\d+): (.+?)\s*$", re.MULTILINE)
62
+ # Where a canonical entry ENDS, in every shape CommonMark spells an ATX heading:
63
+ # up to three spaces of indent, a space OR a tab after the hashes, and an empty
64
+ # heading (`##` alone — the separator may be the end of the line). A fourth space
65
+ # of indent is an indented code block rather than a heading, so those lines
66
+ # deliberately keep absorbing, and the indent class is spaces only for that same
67
+ # reason — a leading tab is four columns, so `\t## Notes` is a code block too.
68
+ # Read as permissively as the syntax is, because a missed boundary here does not
69
+ # merely lose a section header: the span runs on and the next section's
70
+ # `status:`/`gate:` lines are read as this entry's, so an open entry that never
71
+ # declared a gate takes a story hostage and the operator finds no gate in the
72
+ # entry the refusal names (#516). Where a miss would WRITE, the strict
73
+ # column-zero reading is the right one (`devcontract`'s destructive edits pin it
74
+ # deliberately); this only decides how far a read reaches.
75
+ # A lookahead rather than a consuming group: `parse_ledger` reads `.start()`, and
76
+ # holding the match to the opener leaves nothing free to grow a dependency on
77
+ # where the separator ended.
78
+ ANY_HEADING_RE = re.compile(r"^ {0,3}#{1,6}(?=[ \t]|$)", re.MULTILINE)
79
+ # The flat appender's opening line, in the two forms this module needs it: as a
80
+ # bullet in the raw ledger (FLAT_ENTRY_RE, the canonical-span boundary in
81
+ # parse_ledger) and as bullet *content* after `_BULLET_RE` has stripped the
82
+ # marker (`_FLAT_SOURCE_RE`, legacy section). One shape, two anchors — they have
83
+ # to agree, or a block the legacy parser recognizes stays invisible to it (#304).
84
+ # Keyed on the opening line alone, deliberately: also requiring the block's
85
+ # `summary:`/`evidence:` lines would narrow the boundary below the parser's own
86
+ # recognition, leaving the bug in place for every partial shape it accepts.
87
+ _FLAT_SOURCE_BODY = r"source_spec:[ \t]"
88
+ FLAT_ENTRY_RE = re.compile(rf"^[-*][ \t]+{_FLAT_SOURCE_BODY}", re.IGNORECASE | re.MULTILINE)
89
+ STATUS_RE = re.compile(r"^status:[ \t]*(.*)$", re.MULTILINE)
90
+ # The mechanical half of a hard gate. An entry could always *say* it blocked a
91
+ # story — `HARD GATE: must land before 3-2` in the reason line — and saying it
92
+ # stopped nothing: the queue picked the story up anyway, and the gate surfaced
93
+ # afterwards in the diff of work built on a leg nobody had wired. `gate:` names
94
+ # the blocked story keys in a form a check can match, so the claim can refuse.
95
+ # Parsed exactly like `status:`: a field line, read inside `parse_ledger`'s
96
+ # canonical span, so a line under a flat-append bullet belongs to that block and
97
+ # not to the entry above it.
98
+ GATE_RE = re.compile(r"^gate:[ \t]*(.*)$", re.MULTILINE)
99
+ # A story key as either queue spells one: a sprint key (`3-2-invite-link`), the
100
+ # stories-mode id it starts with (`3-2`), or a bare slug. Whitespace and
101
+ # separators are deliberately out — a token nothing can match is the same silent
102
+ # no-op the field exists to end, so it is surfaced rather than dropped.
103
+ GATE_TOKEN_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._-]*$")
104
+ # The second half of "can this token gate anything", and a different miss from the
105
+ # one above: `GATE_TOKEN_RE` rejects the spellings a *line* cannot carry (a space,
106
+ # a bare separator), this rejects the ones no *key* can carry. `gate: 3.2` passes
107
+ # the first and can never match `gates_story` against any legal key, so it used to
108
+ # report a green `ok` while gating nothing — the field's own silent no-op, one
109
+ # keystroke away from the shape that works.
110
+ #
111
+ # Two arms, because FROID spells a story key two ways and they are NOT
112
+ # interchangeable. A stories-mode id is alphanumeric segments joined by single
113
+ # dashes (`_STORIES_ID_RE`), so `3.2` and `3_2` are out. A sprint key's slug is
114
+ # unconstrained (`sprintstatus.STORY_RE`'s trailing group), so `3-2-foo.bar` and
115
+ # `3-2-a_b` are LEGAL keys that gate correctly — which is why this is a
116
+ # whole-token shape test and not a ban on `.`/`_`. Only those characters in the
117
+ # *number* prefix are unmatchable; banning them outright would refuse real gates.
118
+ #
119
+ # Sound in the direction that matters: a token matching either arm is itself a
120
+ # legal key, so a story it could gate can exist. `gates_story`'s prefix and split
121
+ # arms only ever extend a key rightward past a `-`, and every such prefix of a
122
+ # legal key matches one of these arms too.
123
+ #
124
+ # `sprintstatus` is imported for its regex; `stories.ID_RE` is copied rather than
125
+ # imported because `stories` imports *this* module (a cycle). The copy is pinned
126
+ # to the original by a drift test rather than to a comment.
127
+ _STORIES_ID_RE = re.compile(r"^[A-Za-z0-9]+(-[A-Za-z0-9]+)*$")
128
+ # The tokens `gates_story`'s split arm may fire for: a bare `<epic>-<story>`, both
129
+ # numeric. `sprintstatus.STORY_RE` attaches the split letter straight after the
130
+ # story *number*, so that is the only token a split can extend. "Ends in a digit"
131
+ # is a weaker test that reads the same shape into a slug — `3-2-v2` would take the
132
+ # arm and refuse `3-2-v2a-followup`, a different and legal key.
133
+ _SPLITTABLE_TOKEN_RE = re.compile(r"^\d+-\d+$")
134
+ # A `gate:` line the strict field pattern above will never see. `GATE_RE` is
135
+ # anchored to a lowercase `gate:` in column 0, exactly like `status:`, and that
136
+ # strictness fails in opposite directions for the two fields: a missed `status:`
137
+ # leaves an entry unresolved, which now gates conservatively, while a missed
138
+ # `gate:` leaves no gate at all. `Gate: 3-2` and an indented ` gate: 3-2` are
139
+ # therefore surfaced as unenforceable rather than silently absent — and surfaced
140
+ # rather than *accepted*, because accepting an indented line would read a fenced
141
+ # example inside an entry as a live gate and refuse a story nobody meant to block.
142
+ _GATE_NEAR_RE = re.compile(r"^[ \t]*gate[ \t]*:", re.IGNORECASE | re.MULTILINE)
143
+ # The prose convention `gate:` replaces, matched anywhere on a line rather than
144
+ # at its start: real ledgers hard-wrap their `reason:` prose, so the declaration
145
+ # routinely lands mid-line and a line-anchored pattern misses exactly the entries
146
+ # that have one. The quote lookbehind is what keeps that from over-firing — an
147
+ # entry *citing* the phrase (`names a "HARD GATE: ..."`) is discussion, not a
148
+ # declaration — and the colon does the rest of the work, since a sentence about
149
+ # "this HARD GATE is textual only" never reaches the pattern at all.
150
+ # The class covers the backtick and the curly quotes as well as the ASCII pair:
151
+ # a ledger is markdown, so `HARD GATE:` is the citation form an author reaches for
152
+ # first, and an LLM-written entry curls its quotes. Missing them made the warning
153
+ # fire on entries documenting the convention — including this repo's own docs.
154
+ # The lookbehind only reaches an *inline* citation, though; the block form of the
155
+ # same quoting is a fence, and no character precedes a line inside one. Callers
156
+ # read this through `declares_prose_gate`, which masks those out.
157
+ HARD_GATE_PROSE_RE = re.compile(r"""(?<!["'`«“”‘’])HARD GATE:""")
158
+ # Everything `str.splitlines()` splits on, not `\n` alone (#305). The writers
159
+ # below interpolate their arguments into a line-oriented file, so a break in a
160
+ # value injects ledger lines. The C1/Unicode members are load-bearing rather
161
+ # than decorative: `parse_legacy` scans with `splitlines()` while `parse_ledger`
162
+ # matches with `re.MULTILINE`, so a U+2028 splits an entry for one reader and is
163
+ # invisible to the other — the two then disagree about what the ledger says.
164
+ LINE_BREAK_RE = re.compile(r"[\n\r\v\f\x1c-\x1e\x85\u2028\u2029]+")
165
+ # The writers' date shape. Deliberately a separate literal from the legacy
166
+ # parser's `_DATE_TOKEN_RE`, which happens to look similar today: that one
167
+ # decides whether a freeform heading is a dated section, and tightening what the
168
+ # orchestrator will *write* must never quietly retune what `parse_legacy` reads.
169
+ # Spelled `[0-9]` rather than `\d`, which also matches Arabic-Indic, fullwidth
170
+ # and mathematical digit forms — the ledger's readers understand none of them.
171
+ _ISO_DATE_RE = re.compile(r"[0-9]{4}-[0-9]{2}-[0-9]{2}")
172
+
173
+
174
+ @dataclass(frozen=True)
175
+ class DWEntry:
176
+ id: str
177
+ title: str
178
+ status: str # the status field value, "" when the line is missing
179
+ body: str # full entry text including the heading
180
+ span: tuple[int, int] # char offsets of the entry in the ledger text
181
+ # Body-relative offsets of the line `status` was read from; None when the
182
+ # entry has no status line. Carried rather than re-derived because the reader
183
+ # picks the status with a fence-aware lookup at file scope, and a writer that
184
+ # ran `STATUS_RE.search(body)` again would pick the *first* raw match instead.
185
+ # Those differ exactly when an entry quotes an example above its live status:
186
+ # the writer rewrote the quoted line, the reader kept reporting the real
187
+ # `status: open`, and the close reported success while the entry — and any
188
+ # `gate:` it carries — stayed open forever. No default: `parse_ledger` is the
189
+ # only constructor, and a fallback here would silently restore that split.
190
+ status_span: tuple[int, int] | None
191
+ # The whole-file fence index the entry was carved with, so the gate scans can
192
+ # ask the question the heading and status reads already ask at file scope. A
193
+ # body slice cannot see a fence opened above the heading, and the two views
194
+ # disagree: under a stray unclosed ``` above the heading, whole-file scope
195
+ # treats the opener as text (so this entry EXISTS) while the body sees a later
196
+ # matched `~~~` pair as a real fence and reads a live `gate:` as an example.
197
+ # That direction loses a gate in silence, which is what the field exists to
198
+ # end. No default, for `status_span`'s reason: the fallback IS the bug.
199
+ examples: _Examples
200
+
201
+ @property
202
+ def open(self) -> bool:
203
+ return self.status.split()[0] == "open" if self.status else False
204
+
205
+ @property
206
+ def done(self) -> bool:
207
+ """Whether the entry has landed.
208
+
209
+ Deliberately NOT ``not open``. A status line the format does not
210
+ understand — ``status: opne``, or no status line at all — is neither open
211
+ nor done, and the readers that ask want *opposite* answers about it:
212
+ :func:`open_ids` drops it (it may already be finished), while a gate on it
213
+ has to hold (it may not be). Deriving one from the other is what let
214
+ ``gate:`` fail open on a one-character typo — the entry read as closed, so
215
+ the gate was skipped and ``validate`` reported an all-clear naming it.
216
+ """
217
+ return self.status.split()[0] == "done" if self.status else False
218
+
219
+
220
+ @dataclass(frozen=True)
221
+ class _Examples:
222
+ """The ledger's fenced worked examples, indexed for repeated offset queries.
223
+
224
+ ``fenced_spans`` returns its ranges in increasing order and non-overlapping (a
225
+ fence cannot open inside an open one), so a query is a binary search for the
226
+ last span starting at or before the offset. Kept as an index rather than a bare
227
+ list because both scales are in play at once: a parse asks once per heading and
228
+ several times per entry, so a linear membership test would leave the parse
229
+ quadratic whenever a ledger's examples grow with its entries.
230
+ """
231
+
232
+ spans: tuple[tuple[int, int], ...]
233
+ starts: tuple[int, ...]
234
+
235
+ def covers(self, offset: int) -> bool:
236
+ i = bisect_right(self.starts, offset)
237
+ return i > 0 and offset < self.spans[i - 1][1]
238
+
239
+
240
+ def _example_spans(text: str) -> _Examples:
241
+ """The ledger's fenced worked examples, as offset ranges.
242
+
243
+ Read at WHOLE-FILE scope, which is the whole point. A fence that opens above
244
+ a quoted ``### DW-n:`` heading is stranded in the *previous* entry once spans
245
+ are carved, so an entry-local query reads the example as live — a phantom
246
+ entry whose ``gate:`` refuses a story nobody deferred. `deferred-work-format.md`
247
+ ships exactly that shape (a complete entry inside a ```markdown fence), so
248
+ quoting it into a ledger is the expected trigger, not a corner case.
249
+
250
+ ``unclosed_hides_rest=False`` repeats the answer `gates()` gives one level
251
+ down, and here for a stronger reason: under ``True`` a single stray opener
252
+ would erase every heading below it, dropping real open work out of
253
+ ``open_ids()`` in silence. A phantom entry from an unterminated fence is
254
+ today's behaviour and is visible; a vanished ledger is neither.
255
+
256
+ Walked once per :func:`parse_ledger` and passed down to the offset checks. The
257
+ walk covers the whole file, so recomputing it per offset made the parse
258
+ quadratic in the number of entries — and `Engine._refuse_gated_story` re-parses
259
+ before every story dispatch, so a mature ledger paid it on the dispatch path.
260
+ """
261
+ spans = tuple(fenced_spans(text, unclosed_hides_rest=False))
262
+ return _Examples(spans=spans, starts=tuple(s for s, _ in spans))
263
+
264
+
265
+ def _example(examples: _Examples, offset: int) -> bool:
266
+ """Whether ``offset`` sits in a fenced worked example rather than the ledger.
267
+
268
+ Takes the index rather than the text: the answer must come from the same
269
+ whole-file walk for every offset in one parse, and a signature that re-derived
270
+ it per call is what made that expensive enough to matter.
271
+ """
272
+ return examples.covers(offset)
273
+
274
+
275
+ def _unfenced(
276
+ pattern: re.Pattern[str],
277
+ text: str,
278
+ start: int,
279
+ end: int,
280
+ examples: _Examples,
281
+ ) -> re.Match[str] | None:
282
+ """First match of ``pattern`` within ``text[start:end]`` that is not quoted.
283
+
284
+ Not `search()` plus a check: the first match may be the quoted one, and the
285
+ real boundary sits after it. Bounded by ``endpos`` so a match beyond the span
286
+ cannot claim it, while ``examples`` still describes fence state from offset 0.
287
+ """
288
+ for m in pattern.finditer(text, start, end):
289
+ if not _example(examples, m.start()):
290
+ return m
291
+ return None
292
+
293
+
294
+ def parse_ledger(text: str) -> list[DWEntry]:
295
+ """Extract DW entries; non-conforming sections are skipped, an entry
296
+ without a status line parses with status "" (not open).
297
+
298
+ Fenced matches are skipped by every scan below, not just the heading one: a
299
+ heading or flat bullet quoted inside an example must not start an entry, end
300
+ one, or bound a block out of one. Filtering only the headings would trade the
301
+ phantom entry for a truncation — a fenced ``## heading`` would still cut a
302
+ real entry short at its own boundary, and a `gate:` line below the example
303
+ would fall outside the span and stop gating, which is the failure this field
304
+ exists to end.
305
+ """
306
+ entries = []
307
+ examples = _example_spans(text)
308
+ headings = [m for m in HEADING_RE.finditer(text) if not _example(examples, m.start())]
309
+ for i, m in enumerate(headings):
310
+ end = headings[i + 1].start() if i + 1 < len(headings) else len(text)
311
+ # an entry also ends at any intervening heading (e.g. a "## Deferred
312
+ # from:" section header between freeform and DW-format content)
313
+ other = _unfenced(ANY_HEADING_RE, text, m.end(), end, examples)
314
+ if other:
315
+ end = other.start()
316
+ # ...and at a flat appender block, which belongs to no canonical entry
317
+ # (#304). This span is what parse_legacy() masks out before scanning, so
318
+ # absorbing the block hides the finding from every reader of the ledger.
319
+ # Searched from the entry's own `status:` line, never from above it:
320
+ # truncating over the status leaves the entry reading as neither open nor
321
+ # done (open_ids() drops it, classify() calls it malformed), which trades
322
+ # one lost flat block for one lost tracked entry. An entry with no status
323
+ # line has nothing to protect, so the whole span is fair game.
324
+ status_m = _unfenced(STATUS_RE, text, m.end(), end, examples)
325
+ flat = _unfenced(
326
+ FLAT_ENTRY_RE, text, status_m.end() if status_m else m.end(), end, examples
327
+ )
328
+ if flat:
329
+ end = flat.start()
330
+ body = text[m.start() : end]
331
+ # Re-read rather than reuse the probe above: `end` may have moved, and the
332
+ # status must be the one inside the final span. Searched over `text` at
333
+ # absolute offsets because `_example` reads fence state from the top of the
334
+ # file — a body slice cannot see an opener that sits above the heading.
335
+ status_m = _unfenced(STATUS_RE, text, m.start(), end, examples)
336
+ entries.append(
337
+ DWEntry(
338
+ id=m.group(1),
339
+ title=m.group(2),
340
+ status=status_m.group(1).strip() if status_m else "",
341
+ body=body,
342
+ span=(m.start(), end),
343
+ status_span=(
344
+ (status_m.start() - m.start(), status_m.end() - m.start()) if status_m else None
345
+ ),
346
+ examples=examples,
347
+ )
348
+ )
349
+ return entries
350
+
351
+
352
+ def open_ids(text: str) -> set[str]:
353
+ return {e.id for e in parse_ledger(text) if e.open}
354
+
355
+
356
+ @dataclass(frozen=True)
357
+ class EntryGates:
358
+ """One entry's ``gate:`` declaration, split by what a check can act on.
359
+
360
+ Every shape that is not an enforceable token is reported by ``validate``,
361
+ because none of them is a *weaker* gate than a valid one — each is the prose
362
+ gate again wearing the field's clothes, and silence about it is what let the
363
+ story run. ``lines`` is what distinguishes "declared nothing usable" from
364
+ "declared nothing at all": an entry with no ``gate:`` line has made no claim,
365
+ while ``gate:`` with an empty value has made one and inertly.
366
+
367
+ ``empty`` counts those inert lines individually rather than folding them into
368
+ an entry-wide verdict, because the two coexist: ``gate: 3-2`` followed by a
369
+ bare ``gate:`` has both a gate in force and a line that names nothing, and an
370
+ aggregate answer can only report one of them. Reporting the tokens and
371
+ swallowing the empty line is the worse half to lose — the operator who wrote
372
+ it believes a second story is held back.
373
+ """
374
+
375
+ tokens: tuple[str, ...] = ()
376
+ malformed: tuple[str, ...] = ()
377
+ lines: int = 0
378
+ empty: int = 0
379
+ near_miss: int = 0
380
+
381
+ @property
382
+ def inert(self) -> bool:
383
+ """Every ``gate:`` line named nothing — ``gate:`` or ``gate: ,`` and no others."""
384
+ return self.lines > 0 and not self.tokens and not self.malformed
385
+
386
+
387
+ def _quoted(entry: DWEntry, offset: int) -> bool:
388
+ """Whether a BODY-relative ``offset`` sits in a fenced example.
389
+
390
+ The single rule every gate scan in this module reads through, so that a fence
391
+ means the same thing to all of them: an entry documenting the field quotes it,
392
+ and a quoted example is not a declaration. Sharing it is the point — the prose
393
+ scan was left on the raw body once, on the reasoning that a warning is cheap
394
+ and its quote lookbehind was guard enough. It is not: that lookbehind reaches
395
+ an inline citation only, so an entry explaining the old convention in a fenced
396
+ block was told to convert a gate it was not declaring.
397
+
398
+ Asked at FILE scope, like the heading and status reads in :func:`parse_ledger`
399
+ and for the same reason: a body slice cannot see a fence opened above the
400
+ heading, so the two views can disagree about the same line. They disagree in
401
+ the direction that matters — a stray unclosed ``` above the heading leaves the
402
+ entry standing at file scope while the body reads a later matched ``~~~`` pair
403
+ as a real fence, masking a live ``gate:`` into an example. A gate lost in
404
+ silence is the failure this field exists to end; a spurious refusal in an entry
405
+ whose markdown is already malformed is the cheaper wrong answer.
406
+ """
407
+ return entry.examples.covers(entry.span[0] + offset)
408
+
409
+
410
+ def declares_prose_gate(entry: DWEntry) -> bool:
411
+ """Whether the entry declares a gate in the pre-``gate:`` prose convention.
412
+
413
+ :data:`HARD_GATE_PROSE_RE` filtered the way every other gate scan here is
414
+ filtered. Lives beside them rather than at the caller so the fence rule has
415
+ one implementation: ``validate`` is the only reader today, and a second one
416
+ reaching for the bare pattern would reintroduce exactly the half-applied rule
417
+ this replaced.
418
+ """
419
+ return any(not _quoted(entry, m.start()) for m in HARD_GATE_PROSE_RE.finditer(entry.body))
420
+
421
+
422
+ def gates(entry: DWEntry) -> EntryGates:
423
+ """Every ``gate:`` token in one entry's canonical span, order-preserving.
424
+
425
+ Multiple ``gate:`` lines union: an entry blocking three stories may list them
426
+ on one line or on three, and to a line-oriented file neither spelling is the
427
+ wrong one. Within a line the separator is a comma, and only a comma — a
428
+ space-separated ``gate: 3-2 3-3`` lands in ``malformed`` rather than being
429
+ read leniently, so the operator is told the spelling gated nothing instead of
430
+ finding out from a story that ran.
431
+
432
+ Duplicates collapse (an id repeated across lines is one claim, not two);
433
+ empty items drop, so a trailing separator is not a token — but the *line* is
434
+ still counted, which is how an all-empty declaration stays reportable.
435
+
436
+ ``near_miss`` counts the lines this function deliberately did NOT read as a
437
+ declaration: a `gate:` the strict field anchor misses (see
438
+ :data:`_GATE_NEAR_RE`). They are counted rather than parsed so the operator is
439
+ told the spelling gated nothing — the same trade the space-separated token
440
+ makes, one level up.
441
+ """
442
+ tokens: list[str] = []
443
+ malformed: list[str] = []
444
+ lines = 0
445
+ empty = 0
446
+
447
+ # Both scans below skip fenced matches: an entry documenting this field quotes
448
+ # it, and a quoted example is not a declaration — a fenced `gate: 3-2` sits in
449
+ # column 0, right where the anchor looks, and the answer here is a *refusal*.
450
+ for m in GATE_RE.finditer(entry.body):
451
+ if _quoted(entry, m.start()):
452
+ continue
453
+ lines += 1
454
+ named = False
455
+ for raw in m.group(1).split(","):
456
+ token = raw.strip()
457
+ if not token:
458
+ continue
459
+ named = True
460
+ bucket = tokens if _matchable_token(token) else malformed
461
+ if token not in bucket:
462
+ bucket.append(token)
463
+ if not named:
464
+ empty += 1
465
+ near_miss = sum(
466
+ # `^` puts every match at a line start, so this asks whether the same line
467
+ # would have satisfied `GATE_RE` — i.e. whether it is the canonical spelling
468
+ # already counted above — without re-running the anchor against a slice.
469
+ not entry.body.startswith("gate:", m.start())
470
+ for m in _GATE_NEAR_RE.finditer(entry.body)
471
+ if not _quoted(entry, m.start())
472
+ )
473
+ return EntryGates(
474
+ tokens=tuple(tokens),
475
+ malformed=tuple(malformed),
476
+ lines=lines,
477
+ empty=empty,
478
+ near_miss=near_miss,
479
+ )
480
+
481
+
482
+ def _matchable_token(token: str) -> bool:
483
+ """Whether ``token`` could gate any legal story key — the test that decides
484
+ :attr:`EntryGates.tokens` vs :attr:`EntryGates.malformed`.
485
+
486
+ Both halves are required and neither implies the other: ``GATE_TOKEN_RE``
487
+ alone admits ``3.2``, which nothing can match, and the key shapes alone admit
488
+ ``3-2 3-3`` via the sprint slug, which is one token pretending to be two.
489
+ """
490
+ if not GATE_TOKEN_RE.match(token):
491
+ return False
492
+ return bool(_STORIES_ID_RE.match(token) or sprintstatus.STORY_RE.match(token))
493
+
494
+
495
+ def gates_story(token: str, story_key: str) -> bool:
496
+ """Whether ``token`` gates ``story_key``: equal, or its prefix at a key boundary.
497
+
498
+ The prefix arm is what lets one token reach both queues — stories mode keys on
499
+ the bare id (``3-2``) while sprint mode keys on the full ``3-2-invite-link``,
500
+ and an author gating "story 3-2" means the story, not the spelling. The
501
+ boundary is required rather than a bare ``startswith`` so ``3-2`` cannot sweep
502
+ in its numeric neighbours: ``3-20-later`` is a different story.
503
+
504
+ Two boundaries count, because FROID spells a story key two ways. The plain one
505
+ is ``-``. The other is a **split**: ``sprintstatus.STORY_RE`` lets an oversized
506
+ story become ``3-2a-...`` / ``3-2b-...`` at breakdown time, and a token that
507
+ only knew ``-`` would lose its gate the moment the gated story was split —
508
+ silently, which is the worst thing a gate can do. One lowercase ASCII letter
509
+ followed by ``-`` is therefore also a boundary. Exactly one letter, and the
510
+ ``-`` after it is required, so ``3-2ab-x`` and a bare ``3-2a`` are not swept in.
511
+
512
+ The split arm applies only to a token that *is* a bare ``<epic>-<story>``,
513
+ because that is the only place a split letter can attach: ``STORY_RE`` puts it
514
+ straight after the story *number*. Without that guard the arm reads any
515
+ trailing letter as a split and gates a story nobody named — ``stories.ID_RE``
516
+ admits word ids, so ``gate: auth`` refused ``authz-login``, and a hard failure
517
+ on an unrelated story is the one way this check can be worse than the prose it
518
+ replaced. "Ends in a digit" is the same guard written too loosely: the digit
519
+ can belong to a *slug*, so ``gate: 3-2-v2`` took the arm and refused
520
+ ``3-2-v2a-followup`` — a different, legal key — and ``gate: 3`` refused the
521
+ distinct stories id ``3a-task``.
522
+ """
523
+ if story_key == token or story_key.startswith(f"{token}-"):
524
+ return True
525
+ # The `startswith` guard is load-bearing, not redundant with the slice below:
526
+ # `story_key[len(token):]` says nothing about what preceded it, so without it
527
+ # `3-2` would gate `9-9a-x` on the tail alone.
528
+ if not story_key.startswith(token) or not _SPLITTABLE_TOKEN_RE.match(token):
529
+ return False
530
+ rest = story_key[len(token) :]
531
+ return len(rest) >= 2 and "a" <= rest[0] <= "z" and rest[1] == "-"
532
+
533
+
534
+ def parse_declaration(raw: object) -> tuple[tuple[str, ...], str | None]:
535
+ """The single reading of a ``closes_deferred:`` declaration (#234), shared by
536
+ the ``stories.yaml`` parser, the engine's close hook, and ``validate``.
537
+
538
+ Returns the normalized ids plus an error describing a wrong *container*.
539
+ Missing / YAML-null is an empty declaration, not an error.
540
+
541
+ Strict about the container, lenient about each item. A bare
542
+ ``closes_deferred: DW-1`` is a schema error rather than a silently-wrapped
543
+ single id — a string is iterable, so a lenient reading would quietly turn one
544
+ id into a list of characters — while items are ``str()``-normalized and
545
+ stripped, because an LLM-authored manifest may emit an unquoted ``DW-1`` as a
546
+ string but a bare ``5`` as an int. Blanks drop and duplicates collapse
547
+ (order-preserving): both are noise, not a contradiction.
548
+
549
+ Callers decide the severity: the manifest parser raises, the engine journals,
550
+ ``validate`` warns. What they must NOT do is disagree — before this, a wrong
551
+ container was a hard schema error in ``stories.yaml`` and a silent empty
552
+ declaration in frontmatter, so the same mistake either failed the parse or
553
+ vanished depending on which file it was made in.
554
+
555
+ Whether an id names a real entry is not decided here; that needs the ledger
556
+ (:func:`classify`).
557
+ """
558
+ if raw is None:
559
+ return (), None
560
+ if not isinstance(raw, list):
561
+ return (), f"must be a list of deferred-work ids (got {type(raw).__name__})"
562
+ return tuple(dict.fromkeys(item for item in (str(x).strip() for x in raw) if item)), None
563
+
564
+
565
+ @dataclass(frozen=True)
566
+ class Declared:
567
+ """How declared ids line up against one ledger snapshot (#234).
568
+
569
+ Four outcomes, not two, because "not open" hides two very different cases.
570
+ ``already_done`` is a satisfied declaration — a resume re-driving a close that
571
+ already landed — and must stay silent. ``malformed`` is an entry that exists
572
+ but carries neither an ``open`` nor a ``done`` status: nothing can be marked,
573
+ and saying nothing would leave the operator believing it was.
574
+
575
+ ``duplicates`` cross-cuts the other four: it names the declared ids the ledger
576
+ carries more than once, whichever bucket they landed in. A duplicate id is a
577
+ corrupt ledger (#286), and the entry this classification describes is only one
578
+ of them — so the close is reported, never silent.
579
+ """
580
+
581
+ open_ids: tuple[str, ...] = ()
582
+ already_done: tuple[str, ...] = ()
583
+ unknown: tuple[str, ...] = ()
584
+ malformed: tuple[str, ...] = ()
585
+ duplicates: tuple[str, ...] = ()
586
+
587
+
588
+ def classify(text: str, ids: Sequence[str]) -> Declared:
589
+ """Partition `ids` against a single ledger snapshot, preserving order.
590
+
591
+ Classifying from a snapshot rather than from :func:`mark_done`'s return value
592
+ is deliberate: that return conflates "already done" with "absent from the
593
+ ledger", and those need opposite treatment (silence vs. a warning).
594
+
595
+ **The FIRST entry of a duplicated id wins**, because that is the one
596
+ :func:`_find_entry` — and so every mutation in this module — acts on. Indexing
597
+ last-wins instead made the two disagree, and a ledger carrying one `DW-1` open
598
+ and another done then closed nothing while saying nothing, in either order: a
599
+ done-first ledger classified the id `open`, sent it to
600
+ :func:`mark_done_many`, and had :func:`_apply_done` refuse the done copy it
601
+ found first (marked nothing, so not even an unmatched warning); an open-first
602
+ ledger classified it `already_done` and never attempted the write at all
603
+ (#284 round-6 review, finding 4). The duplicate itself is reported through
604
+ ``duplicates`` rather than swallowed — one id naming two entries is a fault
605
+ about the ledger, not an answer about the work."""
606
+ by_id: dict[str, DWEntry] = {}
607
+ duplicated: set[str] = set()
608
+ for e in parse_ledger(text):
609
+ if e.id in by_id:
610
+ duplicated.add(e.id)
611
+ continue # first wins: `_find_entry` mutates that one
612
+ by_id[e.id] = e
613
+ buckets: dict[str, list[str]] = {"open": [], "done": [], "unknown": [], "malformed": []}
614
+ for dw_id in ids:
615
+ entry = by_id.get(dw_id)
616
+ if entry is None:
617
+ buckets["unknown"].append(dw_id)
618
+ continue
619
+ word = entry.status.split()[0] if entry.status else ""
620
+ buckets[word if word in ("open", "done") else "malformed"].append(dw_id)
621
+ return Declared(
622
+ open_ids=tuple(buckets["open"]),
623
+ already_done=tuple(buckets["done"]),
624
+ unknown=tuple(buckets["unknown"]),
625
+ malformed=tuple(buckets["malformed"]),
626
+ duplicates=tuple(dw_id for dw_id in dict.fromkeys(ids) if dw_id in duplicated),
627
+ )
628
+
629
+
630
+ def _find_entry(text: str, dw_id: str) -> DWEntry | None:
631
+ for entry in parse_ledger(text):
632
+ if entry.id == dw_id:
633
+ return entry
634
+ return None
635
+
636
+
637
+ def _insert_after_status(text: str, entry: DWEntry, line: str) -> str:
638
+ """Insert a field line right after the entry's status line (or at the end
639
+ of the entry when no status line exists)."""
640
+ if entry.status_span:
641
+ pos = entry.span[0] + entry.status_span[1]
642
+ return text[:pos] + "\n" + line + text[pos:]
643
+ insert_at = entry.span[0] + len(entry.body.rstrip())
644
+ return text[:insert_at] + "\n" + line + text[insert_at:]
645
+
646
+
647
+ def _one_line(value: str) -> str:
648
+ """Collapse every run of line-break characters in `value` to a single space.
649
+
650
+ The whole of the #305 fix. These writers interpolate their arguments into a
651
+ line-oriented file, so a value carrying a break mints a phantom
652
+ `### DW-<n>` entry, truncates the entry's span at :data:`FLAT_ENTRY_RE` and
653
+ re-surfaces the tail as a legacy item, or leaves the entry carrying two
654
+ `status:` lines.
655
+
656
+ Note what the last shape does *not* do: `STATUS_RE` takes the first match, so
657
+ an injected `status:` never changes what `parse_ledger` reports. A test that
658
+ asserts on `entry.status` therefore passes with this guard deleted — the
659
+ observable is the line structure.
660
+
661
+ Sanitizes; never raises, and nothing upstream rejects on a break either. The
662
+ close paths call these writers bare (`sweep._close_resolved`,
663
+ `decisions.apply_pre_answer`), so a `ValueError` would end the sweep as
664
+ crashed; refusing the same text back at `validate_triage` only moved the
665
+ stoppage to a pause. Collapsing is lossless enough — the ledger wants one
666
+ line anyway — so this is the fix, and the skill docs are guidance that
667
+ reduces occurrences without gating on them.
668
+
669
+ That contract covers one hazard more than the break collapse alone, which is
670
+ what the `neutralize_surrogates` pass in front of it buys (#329). A lone
671
+ surrogate is not a line break, so it sailed through untouched — but it has
672
+ no UTF-8 encoding, and `atomic_write_text`'s strict encode raises
673
+ `UnicodeEncodeError` (a `ValueError` subclass) on it, from inside those same
674
+ bare close-path calls. It arrives the way the break did: a triage
675
+ `result.json` is cached with `json.dumps`, whose `ensure_ascii` keeps the
676
+ code point a harmless `\\ud800` escape, and the reload's `json.loads` revives
677
+ the real thing into `ResolvedEntry.evidence` and on into the `mark_done`
678
+ note. Refusing it upstream would only move the stoppage again — same
679
+ doctrine, same answer.
680
+
681
+ A value with neither a break nor a surrogate is returned **untouched**, so an
682
+ existing ledger is never reformatted and a clean write is byte-identical to
683
+ before the guard; each pass keeps its own fast path, so the common value is
684
+ scanned twice and copied never. The trailing `.strip()` removes all
685
+ surrounding whitespace, not merely the space a leading or trailing break left
686
+ behind — which is why it must stay on the far side of that fast path.
687
+
688
+ A break-only value therefore sanitizes to `""`. Keeping it non-empty *here*
689
+ could only yield bare whitespace, which trades an unfindable entry for an
690
+ unidentifiable one, so each caller handles its own empties — and by two
691
+ different strategies, which is why neither belongs in this helper.
692
+ :func:`append_entry` **substitutes**, naming a vanished title
693
+ `(untitled DW-<n>)` so the id it just burned stays findable.
694
+ :func:`append_decision` **drops**, shedding the ` — ` separator along with an
695
+ empty detail rather than promising one that is not there. Its `label` needs
696
+ neither: every member of :data:`LINE_BREAK_RE` is `str.isspace()`, and
697
+ `validate_triage` builds each `DecisionOption` with `.strip() or key`, so a
698
+ break-only label has already become the option key before it arrives.
699
+
700
+ A surrogate-only value, by contrast, sanitizes to a truthy `"�"`, so neither
701
+ caller's empty-handling fires for it. That is the point of replacing rather
702
+ than stripping: a title reading `�` still says *something unencodable was
703
+ here*, where a vanished one would silently become `(untitled DW-<n>)`."""
704
+ value = neutralize_surrogates(value)
705
+ if not LINE_BREAK_RE.search(value):
706
+ return value
707
+ return LINE_BREAK_RE.sub(" ", value).strip()
708
+
709
+
710
+ def _iso_date_or_none(value: str) -> str | None:
711
+ """`value` when it is a strict ISO ``YYYY-MM-DD`` calendar date, else None.
712
+
713
+ The shared shape of the ledger's two date checks, so a skip-not-raise caller
714
+ (:func:`_close_date`) and a raise caller (:func:`_require_iso_date`) cannot
715
+ drift apart on what counts as a close date. The regex is not redundant with
716
+ ``date.fromisoformat``: since 3.11 that also accepts ``20260611`` and ISO
717
+ week dates, neither of which the ledger's own readers recognize, and it is
718
+ the regex — via ``[0-9]`` — that pins the digits to ASCII. ``fromisoformat``
719
+ in turn rejects the well-shaped impossible day (``2026-02-30``) that no
720
+ pattern can catch."""
721
+ if not _ISO_DATE_RE.fullmatch(value):
722
+ return None
723
+ try:
724
+ calendar_date.fromisoformat(value)
725
+ except ValueError:
726
+ return None
727
+ return value
728
+
729
+
730
+ def _require_iso_date(value: str) -> None:
731
+ """Raise unless `value` is a strict ISO `YYYY-MM-DD` calendar date.
732
+
733
+ Raising is right here and wrong for free text: `date` is orchestrator-owned
734
+ (`Engine._today()`), never model-authored, so a bad value is a programmer
735
+ bug. Letting it through writes a `status:` line that reads as neither open
736
+ nor done, which `classify` reports as malformed and `open_ids` drops — the
737
+ entry silently leaves the sweep's world."""
738
+ if _iso_date_or_none(value) is None:
739
+ raise ValueError(f"date must be YYYY-MM-DD: {value!r}")
740
+
741
+
742
+ def _require_canonical_status(status: str) -> None:
743
+ """Raise unless `status` is exactly `open` or `done YYYY-MM-DD`.
744
+
745
+ Two halves with two different dependents. The *first word* is what
746
+ :attr:`DWEntry.open` and :func:`classify` branch on, so anything but `open`
747
+ or `done` makes an entry unreadable to both. The *date* is invisible to them
748
+ — they read `status.split()[0]` and cannot tell `done 2026-02-30` from a real
749
+ day — but it is not invisible downstream: the whole status value is carried
750
+ verbatim to readers (the TUI's deferred pane, the `--json` projections), so a
751
+ malformed date is rendered to a human as though it were one."""
752
+ if status == "open":
753
+ return
754
+ if status.startswith("done "):
755
+ _require_iso_date(status.removeprefix("done "))
756
+ return
757
+ raise ValueError(f"status must be 'open' or 'done YYYY-MM-DD': {status!r}")
758
+
759
+
760
+ def _operation_digest(operation_id: str) -> str:
761
+ """Encode a stable close-operation id as one ledger-safe token."""
762
+ if not operation_id:
763
+ raise ValueError("operation_id must not be empty")
764
+ return hashlib.sha256(operation_id.encode("utf-8")).hexdigest()
765
+
766
+
767
+ # Per-thread reentrancy guard for :func:`ledger_lock`. `file_lock` is per open
768
+ # fd, so a second acquisition from the same process does not merely queue — on
769
+ # POSIX `flock` it blocks forever against a lock this very thread holds, with no
770
+ # timeout and no traceback. Thread-local rather than a plain module global
771
+ # because the state being tracked is "does THIS thread already hold it", and two
772
+ # threads legitimately contend through the OS lock.
773
+ _LOCK_STATE = threading.local()
774
+
775
+
776
+ @contextmanager
777
+ def ledger_lock(path: Path) -> Iterator[None]:
778
+ """Cross-process mutual exclusion for one ledger (#286/#469).
779
+
780
+ Held only around a single read->edit->write of `path` — never across a
781
+ subprocess, a coding-CLI session, or an operator pause. That is an acceptance
782
+ criterion of #286 rather than a style preference: `file_lock`'s Windows
783
+ branch gives up after ~10 s and raises, so a holder that waits on anything
784
+ slower converts a contended run into a failed one. It is also why the
785
+ engine's rollback/restore windows, which span git spawns, get compare-and-set
786
+ semantics instead of a lock around the window.
787
+
788
+ Acquired in exactly two strata: the leaf mutators in this module, and the
789
+ engine's CAS restores, which do pure in-memory text work under the hold.
790
+ Never call a mutator while holding it — every mutator takes this lock itself,
791
+ and the nested acquisition would deadlock.
792
+
793
+ Nesting raises :class:`RuntimeError` rather than deadlocking. The guard is
794
+ deliberately path-agnostic: two *different* ledgers would not self-deadlock
795
+ on the OS lock, but nesting is still a lock-ordering hazard, and no caller
796
+ has a reason to hold two ledgers at once. The lock file itself lives out of
797
+ the repository — see :func:`~froid_loop.runs.lock_path_for` for why a sidecar
798
+ beside the tracked ledger would be committed by the engine's own `git add
799
+ -A`. Propagates `OSError` from acquisition and
800
+ :class:`~froid_loop.runs.StateRootError` when no state root can be derived: a
801
+ write that could not be serialized must fail loudly, not proceed unlocked.
802
+ """
803
+ # Lazy, and it has to stay lazy: `runs` imports `verify`, which imports this
804
+ # module, so a top-level import here closes the cycle.
805
+ from . import runs
806
+
807
+ if getattr(_LOCK_STATE, "held", False):
808
+ raise RuntimeError("ledger lock is not reentrant")
809
+ lock_path = runs.lock_path_for(path)
810
+ _LOCK_STATE.held = True
811
+ try:
812
+ with file_lock(lock_path):
813
+ yield
814
+ finally:
815
+ _LOCK_STATE.held = False
816
+
817
+
818
+ def _apply_done(
819
+ text: str,
820
+ dw_id: str,
821
+ date: str,
822
+ note: str,
823
+ *,
824
+ undo_owner: str | None = None,
825
+ ) -> str | None:
826
+ """Flip one entry to `status: done <date>` + a resolution note *within* `text`.
827
+ None when the entry is missing or not open. The entry is re-located after the
828
+ status rewrite because that edit shifts every later span offset.
829
+
830
+ The note is sanitized here, at the point of interpolation, rather than on
831
+ :func:`mark_done`: that is a one-id wrapper over :func:`mark_done_many`, which
832
+ `Engine._apply_deferred_closes` calls directly, so a wrapper-side guard would
833
+ never see a story close (#305). `date` is validated by the sole caller, at its
834
+ entry, so the check does not depend on a ledger existing."""
835
+ note = _one_line(note)
836
+ entry = _find_entry(text, dw_id)
837
+ if entry is None or not entry.open:
838
+ return None
839
+ assert entry.status_span is not None # open implies a status line
840
+ start = entry.span[0] + entry.status_span[0]
841
+ end = entry.span[0] + entry.status_span[1]
842
+ previous_status_line = entry.body[entry.status_span[0] : entry.status_span[1]]
843
+ if undo_owner is not None and LINE_BREAK_RE.search(previous_status_line):
844
+ # An undo marker must never preserve a value that becomes more than one line
845
+ # under the ledger readers' shared splitlines semantics. Standard closes
846
+ # retain their existing behavior; the undo-capable path refuses the mark.
847
+ return None
848
+ done_status_line = f"status: done {date}"
849
+ text = text[:start] + done_status_line + text[end:]
850
+ entry = _find_entry(text, dw_id)
851
+ assert entry is not None
852
+ tail = f"resolution: {note}"
853
+ if undo_owner is not None:
854
+ # The owner digest makes this close distinguishable from an earlier run
855
+ # that reused its human-readable note. The encoded prior line makes the
856
+ # undo lossless for parser-accepted spacing and annotations. Hex keeps
857
+ # every payload on one ASCII line, including Unicode annotations.
858
+ previous_status_hex = previous_status_line.encode("utf-8").hex()
859
+ tail += f"\nresolution-undo: {undo_owner} {date} {previous_status_hex}"
860
+ return _insert_after_status(text, entry, tail)
861
+
862
+
863
+ def _apply_done_many(
864
+ text: str,
865
+ dw_ids: Sequence[str],
866
+ date: str,
867
+ note: str,
868
+ notes: Sequence[str] | None,
869
+ undo_owner: str | None,
870
+ ) -> tuple[str, list[str]]:
871
+ """Fold every id in `dw_ids` through :func:`_apply_done` *within* `text`,
872
+ returning the new text and the ids actually flipped, in the order given.
873
+
874
+ Pure — text in, text out, no `Path` and no I/O — and it is ONE body for the
875
+ advisory pre-lock probe and the locked pass, so the two cannot drift: the
876
+ argument :func:`_apply_append`'s extraction already makes for the batched
877
+ appender. The whole decision lives here, `undo_owner` included, because the
878
+ reopenable arm's LINE_BREAK refusal (:func:`_apply_done`) can be the only
879
+ reason a batch flips nothing — a probe that scanned for open entries by hand
880
+ would answer "would write" where this answers "would not"."""
881
+ marked: list[str] = []
882
+ for index, dw_id in enumerate(dw_ids):
883
+ entry_note = note if notes is None else notes[index]
884
+ updated = _apply_done(text, dw_id, date, entry_note, undo_owner=undo_owner)
885
+ if updated is None:
886
+ continue
887
+ text = updated
888
+ marked.append(dw_id)
889
+ return text, marked
890
+
891
+
892
+ def _mark_done_many(
893
+ path: Path,
894
+ dw_ids: Sequence[str],
895
+ date: str,
896
+ note: str,
897
+ *,
898
+ operation_id: str | None = None,
899
+ notes: Sequence[str] | None = None,
900
+ ) -> list[str]:
901
+ """Shared atomic implementation for the public close operations.
902
+
903
+ ONE locked read->edit->write: the whole cycle runs under the cross-process
904
+ ledger lock (#286/#469), so concurrent mutators — a second run, a sweep, the
905
+ TUI decision modal, ``sweep --archive`` — serialize here rather than trading
906
+ last-write-wins. Validation stays ABOVE the lock, so a programmer bug reports
907
+ itself without first waiting on another process.
908
+
909
+ A batch that would flip nothing — every id missing, already done, or refused
910
+ by the reopenable arm's line-break guard — is answered from the advisory
911
+ pre-lock probe instead, with no acquisition at all (#736). The probe folds
912
+ the ids through :func:`_apply_done_many`, the same helper the locked pass
913
+ uses, so it cannot answer "no write" where the authority would write.
914
+
915
+ ``notes`` supplies a per-id resolution note, positionally paired with
916
+ ``dw_ids``; ``note`` is the fallback for every id when it is None. A length
917
+ mismatch raises before any I/O rather than closing a prefix under the wrong
918
+ evidence — the pairing is positional, so a short list is a caller bug that
919
+ would otherwise mis-attribute notes silently.
920
+ """
921
+ _require_iso_date(date)
922
+ if notes is not None and len(notes) != len(dw_ids):
923
+ raise ValueError(f"notes must be one per dw_id: {len(notes)} for {len(dw_ids)} ids")
924
+ undo_owner = _operation_digest(operation_id) if operation_id is not None else None
925
+ if not dw_ids:
926
+ # Nothing to serialize against, so nothing to take a lock for — the same
927
+ # early return `append_entries` makes, for the same reason. Below the
928
+ # validation above, so an empty batch still reports a bad date or a bad
929
+ # operation id; above the lock, so a caller that batches an empty set
930
+ # cannot start failing on a lock it never needed. The per-id loop this
931
+ # primitive replaced took no lock at all when handed nothing, and that
932
+ # identity is part of what "byte-identical to the serial sequence" buys.
933
+ return []
934
+ if not path.is_file():
935
+ # No ledger, no entry to flip, so no write and no lock — the order
936
+ # `archive_closed` already keeps for its own missing-ledger case. The
937
+ # recheck under the hold below stays: creation can race this answer.
938
+ return []
939
+ try:
940
+ # ADVISORY pre-lock probe (#736): one read, and the same pure decision
941
+ # the locked pass makes. Only a "would write nothing" answer is acted on
942
+ # — the call then serializes at this read. Anything else, including any
943
+ # fault here, falls through to the hold, which re-reads and decides.
944
+ probe = path.read_text(encoding="utf-8")
945
+ if not _apply_done_many(probe, dw_ids, date, note, notes, undo_owner)[1]:
946
+ return []
947
+ except Exception: # nosec B110 - ADVISORY probe: a fault here must decide nothing
948
+ pass
949
+ with ledger_lock(path):
950
+ if not path.is_file():
951
+ return []
952
+ text = path.read_text(encoding="utf-8")
953
+ text, marked = _apply_done_many(text, dw_ids, date, note, notes, undo_owner)
954
+ if not marked:
955
+ return []
956
+ atomic_write_text(path, text)
957
+ return marked
958
+
959
+
960
+ def mark_done_many(
961
+ path: Path,
962
+ dw_ids: Sequence[str],
963
+ date: str,
964
+ note: str,
965
+ *,
966
+ notes: Sequence[str] | None = None,
967
+ ) -> list[str]:
968
+ """Flip every entry in `dw_ids` to `status: done <date>` + a resolution note,
969
+ in ONE read and ONE atomic write. Returns the ids actually flipped (missing
970
+ and already-done ids are skipped), in the order given.
971
+
972
+ ``notes[i]`` overrides `note` for ``dw_ids[i]`` — the shape a caller closing
973
+ several entries under per-entry evidence needs, which otherwise costs one
974
+ read-modify-write cycle per id. A length mismatch raises `ValueError` before
975
+ any I/O.
976
+
977
+ All-or-nothing on purpose. A per-id read-modify-write loop leaves marks on
978
+ disk when it raises partway through several ids — a half-applied closure the
979
+ caller never gets to journal, so the ledger claims resolutions the run has no
980
+ record of. Here a failure writes nothing, and the returned list is exactly
981
+ what landed.
982
+
983
+ The write goes through :func:`~froid_loop.platform_util.atomic_write_text`
984
+ rather than a bare tmp+replace: swapping a fresh inode over the ledger
985
+ otherwise resets its mode (a ``0600`` ledger silently becoming world-readable)
986
+ and turns a symlinked ledger into a regular file.
987
+
988
+ ``date`` is validated before the ``is_file`` short-circuit so a programmer bug
989
+ fails the same way whether or not a ledger happens to exist — a guard that
990
+ only fires when the file is present is one an absent fixture hides."""
991
+ return _mark_done_many(path, dw_ids, date, note, notes=notes)
992
+
993
+
994
+ def mark_done_many_reopenable(
995
+ path: Path,
996
+ dw_ids: Sequence[str],
997
+ date: str,
998
+ note: str,
999
+ operation_id: str,
1000
+ ) -> list[str]:
1001
+ """Close entries atomically with a durable, operation-specific undo marker.
1002
+
1003
+ ``operation_id`` must be stable and recomputable across crash/replay from
1004
+ already-persisted identity (for example ``run_id`` + ``story_key``), never an
1005
+ ephemeral random value. Only entries actually flipped receive its marker;
1006
+ skipped, already-done ids therefore cannot be reopened by this operation.
1007
+
1008
+ The ordinary :func:`mark_done_many` deliberately emits no marker and retains
1009
+ its existing ledger format. Use this variant only for a transaction with a
1010
+ later rollback leg.
1011
+ """
1012
+ return _mark_done_many(path, dw_ids, date, note, operation_id=operation_id)
1013
+
1014
+
1015
+ def mark_done(path: Path, dw_id: str, date: str, note: str) -> bool:
1016
+ """Flip one entry to `status: done <date>` and record a resolution note.
1017
+ Returns False (no write) when the entry is missing or already done."""
1018
+ return bool(mark_done_many(path, [dw_id], date, note))
1019
+
1020
+
1021
+ _MARK_DONE_TAIL_RE = re.compile(
1022
+ r"\nresolution:[ \t]*(.*)"
1023
+ r"\nresolution-undo:[ \t]*([0-9a-f]{64})[ \t]+"
1024
+ r"([0-9]{4}-[0-9]{2}-[0-9]{2})[ \t]+([0-9a-f]+)$",
1025
+ re.MULTILINE,
1026
+ )
1027
+
1028
+
1029
+ def _apply_open(text: str, dw_id: str, note: str, undo_owner: str) -> str | None:
1030
+ """Undo one reopenable close *within* `text`. None when the entry is missing,
1031
+ already open, or does not carry this operation's adjacent resolution and
1032
+ undo-marker lines.
1033
+
1034
+ Pure by construction — text in, text out, no `Path` and no I/O — which is
1035
+ what keeps :func:`mark_open_many` able to run it several times inside a
1036
+ single :func:`ledger_lock` hold. A version of this that touched the file
1037
+ would have to take the lock itself, and the nested acquisition is exactly the
1038
+ self-deadlock the guard on `ledger_lock` exists to convert into an error.
1039
+
1040
+ A standard or earlier close has no matching marker and cannot be reopened
1041
+ merely because it reused the same human-readable note.
1042
+
1043
+ A live ``archived:`` stamp is demoted to :data:`_ARCHIVED_BODY_FIELD` rather
1044
+ than dropped: the reopened entry is no longer archived, but the body its
1045
+ close moved out still is, and that line is the only thing a later triage has
1046
+ to find it with."""
1047
+ entry = _find_entry(text, dw_id)
1048
+ if entry is None or entry.open:
1049
+ return None
1050
+ if entry.status_span is None:
1051
+ # parse_ledger deliberately tolerates status-less entries. This primitive
1052
+ # is later called from _defer, where an AttributeError would crash the run
1053
+ # instead of completing the deferral.
1054
+ return None
1055
+ status_line = entry.body[entry.status_span[0] : entry.status_span[1]]
1056
+ try:
1057
+ _require_canonical_status(entry.status)
1058
+ except ValueError:
1059
+ # Only a canonical status written by mark_done is eligible for undo.
1060
+ # Preserve malformed or human-authored statuses for validation/reporting.
1061
+ return None
1062
+ res_m = _MARK_DONE_TAIL_RE.match(entry.body, entry.status_span[1])
1063
+ if res_m is None:
1064
+ return None
1065
+ if res_m.group(1).strip() != _one_line(note).strip() or res_m.group(2) != undo_owner:
1066
+ return None
1067
+ if status_line != f"status: done {res_m.group(3)}":
1068
+ return None
1069
+ try:
1070
+ previous_status_line = bytes.fromhex(res_m.group(4)).decode("utf-8")
1071
+ except (UnicodeDecodeError, ValueError):
1072
+ return None
1073
+ if LINE_BREAK_RE.search(previous_status_line):
1074
+ return None
1075
+ previous_status_m = STATUS_RE.fullmatch(previous_status_line)
1076
+ previous_status = previous_status_m.group(1).strip() if previous_status_m else ""
1077
+ if not previous_status or previous_status.split()[0] != "open":
1078
+ return None
1079
+ start = entry.span[0] + entry.status_span[0]
1080
+ end = entry.span[0] + res_m.end()
1081
+ # Demote the entry's live `archived:` stamps along with the close they
1082
+ # describe, rather than deleting them. A stub's stamp says "this body lives
1083
+ # in the archive file"; once the close is undone the body is here and the
1084
+ # line is a lie, and leaving it standing is not merely untidy — status +
1085
+ # undo tail + stamp is the exact `_STUB_BODY_RE` shape, so the next
1086
+ # reopenable close reconstitutes a stub `archive_closed` skips forever,
1087
+ # stranding the entry outside every future archive (#711).
1088
+ #
1089
+ # Cutting the line outright strands the entry a second way: a stub keeps
1090
+ # neither `location:` nor `reason:` (`_PRESERVED_FIELD_RE`), so the stamp is
1091
+ # the reopened entry's ONLY route back to the body, and triage arrives with
1092
+ # a heading and nothing to triage (#711 review). Renaming the field keeps
1093
+ # both properties — the value still narrows to the archive block, an id
1094
+ # owning several once a re-closure is archived too, while the renamed line
1095
+ # matches neither `_ARCHIVED_FIELD_RE` nor `_STUB_BODY_RE`, so the entry
1096
+ # reads as live and re-archives normally. Rehydrating the body here
1097
+ # instead was the alternative and is worse: several blocks per id is by
1098
+ # design, so a rollback's reopen would have to guess which one, and a wrong
1099
+ # guess overwrites live content with a stale body.
1100
+ #
1101
+ # Cuts are disjoint (an `^archived:` line cannot start inside the status
1102
+ # line or its adjacent tail) and applied back-to-front so earlier offsets
1103
+ # stay valid.
1104
+ cuts = [(start, end, previous_status_line)]
1105
+ for cut_start, cut_end in _archived_line_spans(entry):
1106
+ # Everything after the field name — value, spacing and the terminating
1107
+ # newline — carries over verbatim; the span starts at the anchor, so
1108
+ # the first colon is the field's own.
1109
+ stamp = entry.body[cut_start:cut_end].split(":", 1)[1]
1110
+ cuts.append(
1111
+ (
1112
+ entry.span[0] + cut_start,
1113
+ entry.span[0] + cut_end,
1114
+ f"{_ARCHIVED_BODY_FIELD}{stamp}",
1115
+ )
1116
+ )
1117
+ for cut_start, cut_end, replacement in sorted(cuts, reverse=True):
1118
+ text = text[:cut_start] + replacement + text[cut_end:]
1119
+ return text
1120
+
1121
+
1122
+ def _apply_open_many(
1123
+ text: str, dw_ids: Sequence[str], note: str, undo_owner: str
1124
+ ) -> tuple[str, list[str]]:
1125
+ """Fold every id in `dw_ids` through :func:`_apply_open` *within* `text`,
1126
+ returning the new text and the ids actually reopened, in the order given.
1127
+
1128
+ Pure — text in, text out, no `Path` and no I/O — and ONE body for the
1129
+ advisory pre-lock probe and the locked pass, so the two cannot drift. The
1130
+ `undo_owner` match is part of the decision: an entry closed by a different
1131
+ operation is skipped here, which is what makes "no id was eligible" a
1132
+ question only this fold can answer."""
1133
+ reopened: list[str] = []
1134
+ for dw_id in dw_ids:
1135
+ updated = _apply_open(text, dw_id, note, undo_owner)
1136
+ if updated is None:
1137
+ continue
1138
+ text = updated
1139
+ reopened.append(dw_id)
1140
+ return text, reopened
1141
+
1142
+
1143
+ def mark_open_many(path: Path, dw_ids: Sequence[str], note: str, operation_id: str) -> list[str]:
1144
+ """Undo every close in `dw_ids` written by :func:`mark_done_many_reopenable`
1145
+ under `operation_id`, in ONE read and ONE atomic write. Returns the ids
1146
+ actually reopened, in the order given; missing and ineligible ids are
1147
+ skipped, and an entry whose marker does not match this operation is left
1148
+ exactly as it was.
1149
+
1150
+ ONE locked read->edit->write: the whole cycle runs under the cross-process
1151
+ ledger lock (#286/#469), so concurrent mutators — a second run, a sweep, the
1152
+ TUI decision modal, ``sweep --archive`` — serialize here rather than trading
1153
+ last-write-wins. A per-id loop over :func:`mark_open` would instead take the
1154
+ lock once per id, leaving a rival writer a window between every pair of
1155
+ undos in what a rollback needs to be one step.
1156
+
1157
+ Nothing is written when no id was eligible, and no lock is taken either
1158
+ (#736): a replayed rollback over already-reopened entries is answered from
1159
+ one advisory read, so it leaves the file untouched rather than rewriting it
1160
+ byte-for-byte, and cannot fail on a lock it had no write to serialize."""
1161
+ undo_owner = _operation_digest(operation_id)
1162
+ if not dw_ids:
1163
+ # No ids, no lock — see `_mark_done_many`. The `operation_id` above is
1164
+ # still validated, so an empty reopen cannot smuggle a bad one through.
1165
+ return []
1166
+ if not path.is_file():
1167
+ # No ledger, no close to undo — see `_mark_done_many`. Rechecked under
1168
+ # the hold below.
1169
+ return []
1170
+ try:
1171
+ # ADVISORY pre-lock probe (#736): one read, and the same pure decision
1172
+ # the locked pass makes. Only a "would write nothing" answer is acted on
1173
+ # — the call then serializes at this read. Anything else, including any
1174
+ # fault here, falls through to the hold, which re-reads and decides.
1175
+ probe = path.read_text(encoding="utf-8")
1176
+ if not _apply_open_many(probe, dw_ids, note, undo_owner)[1]:
1177
+ return []
1178
+ except Exception: # nosec B110 - ADVISORY probe: a fault here must decide nothing
1179
+ pass
1180
+ with ledger_lock(path):
1181
+ if not path.is_file():
1182
+ return []
1183
+ text = path.read_text(encoding="utf-8")
1184
+ text, reopened = _apply_open_many(text, dw_ids, note, undo_owner)
1185
+ if not reopened:
1186
+ return []
1187
+ atomic_write_text(path, text)
1188
+ return reopened
1189
+
1190
+
1191
+ def mark_open(path: Path, dw_id: str, note: str, operation_id: str) -> bool:
1192
+ """Undo one close written by :func:`mark_done_many_reopenable`.
1193
+
1194
+ A one-id wrapper over :func:`mark_open_many`, which is where the lock is
1195
+ taken and the contract documented. It delegates rather than duplicating the
1196
+ read->edit->write so that one public call is exactly one acquisition — a
1197
+ wrapper that took the lock itself and then called the batch would nest, and
1198
+ `ledger_lock` raises on that rather than deadlocking."""
1199
+ return bool(mark_open_many(path, [dw_id], note, operation_id))
1200
+
1201
+
1202
+ def _apply_decision(text: str, dw_id: str, date: str, label: str, detail: str) -> str | None:
1203
+ """Insert one `decision: <date> <label> — <detail>` line *within* `text`,
1204
+ right after the entry's status line. None when the entry is missing.
1205
+
1206
+ Pure — text in, text out, no `Path` and no I/O — so :func:`record_decision`
1207
+ can run it and :func:`_apply_done` against the same in-memory text inside a
1208
+ single :func:`ledger_lock` hold. Applies to a done entry as readily as an
1209
+ open one: a decision is a record of what a human chose, not a status change.
1210
+
1211
+ `label` and `detail` come from a triage session's `DecisionOption`, so they
1212
+ are sanitized to one line rather than refused — see :func:`_one_line`. This
1213
+ is also where a build option's `intent` gets flattened, since it reaches the
1214
+ ledger only as `detail = option.resolution or option.intent`."""
1215
+ entry = _find_entry(text, dw_id)
1216
+ if entry is None:
1217
+ return None
1218
+ label = _one_line(label)
1219
+ # Sanitize before the emptiness test, never after: a break-only detail
1220
+ # collapses to "" and must then drop the separator with it, or the entry
1221
+ # carries a dangling `— ` promising a detail that is not there.
1222
+ detail = _one_line(detail)
1223
+ detail_part = f" — {detail}" if detail else ""
1224
+ return _insert_after_status(text, entry, f"decision: {date} {label}{detail_part}")
1225
+
1226
+
1227
+ def record_decision(
1228
+ path: Path,
1229
+ dw_id: str,
1230
+ date: str,
1231
+ label: str,
1232
+ detail: str,
1233
+ *,
1234
+ close_note: str | None = None,
1235
+ ) -> bool:
1236
+ """Record a human decision on one entry and, when `close_note` is given, act
1237
+ on it by flipping the entry to `status: done <date>` — both in ONE read and
1238
+ ONE atomic write. Returns True when the entry was found (and therefore
1239
+ carries a decision line), False when it was not.
1240
+
1241
+ ONE locked read->edit->write: the whole cycle runs under the cross-process
1242
+ ledger lock (#286/#469), so concurrent mutators — a second run, a sweep, the
1243
+ TUI decision modal, ``sweep --archive`` — serialize here rather than trading
1244
+ last-write-wins. That is the reason the pair is one primitive at all: as
1245
+ separate :func:`append_decision` and :func:`mark_done` calls it is two
1246
+ acquisitions with a window between them, and a rival writer landing in that
1247
+ window sees an entry whose decision says "close it" and whose status still
1248
+ says open.
1249
+
1250
+ The decision line is inserted BEFORE the close is applied, which is not a
1251
+ preference: :func:`_apply_done` writes its `resolution:` line immediately
1252
+ after the status line, and :data:`_MARK_DONE_TAIL_RE` — what
1253
+ :func:`_apply_open` matches an undo marker with — anchors on exactly that
1254
+ adjacency. Applying the close first would leave the decision line between
1255
+ status and resolution and make a reopenable close unreopenable. Ordered this
1256
+ way the bytes are identical to the serial pair's.
1257
+
1258
+ An already-done (or missing-status) entry skips only the close half: the
1259
+ decision line still lands, because a decision recorded on an entry someone
1260
+ else already closed is still what the human chose. `close_note` is the
1261
+ resolution note for the flip, distinct from `detail`, which is the decision's
1262
+ own rationale.
1263
+
1264
+ Precondition: `date` is ISO `YYYY-MM-DD` — one check for both halves, since
1265
+ the decision line and the close share it; anything else raises `ValueError`,
1266
+ checked before the ``is_file`` short-circuit so an absent ledger cannot hide
1267
+ the bug.
1268
+
1269
+ A missing ledger, and a `dw_id` no entry carries, are both answered False
1270
+ without taking the lock (#736) — there is no write to serialize, and the
1271
+ TUI decision modal reaching a stale id should not fail on an acquisition.
1272
+ The probe runs :func:`_apply_decision`, the same helper the locked pass
1273
+ runs, which is None exactly when the entry is missing.
1274
+
1275
+ The write goes through :func:`~froid_loop.platform_util.atomic_write_text` for
1276
+ the reasons documented on :func:`mark_done_many`, plus one this sibling shares
1277
+ with it: a bare ``Path.write_text`` truncates *before* it encodes, so any
1278
+ failure between the two — an unencodable value, ``ENOSPC``, ``EIO`` — leaves a
1279
+ zero-byte ledger where every entry used to be (#328).
1280
+ """
1281
+ _require_iso_date(date)
1282
+ if not path.is_file():
1283
+ # No ledger, no entry to record against — see `_mark_done_many`.
1284
+ # Rechecked under the hold below.
1285
+ return False
1286
+ try:
1287
+ # ADVISORY pre-lock probe (#736): one read, and the same pure decision
1288
+ # the locked pass makes. Only a "would write nothing" answer is acted on
1289
+ # — the call then serializes at this read. Anything else, including any
1290
+ # fault here, falls through to the hold, which re-reads and decides.
1291
+ probe = path.read_text(encoding="utf-8")
1292
+ if _apply_decision(probe, dw_id, date, label, detail) is None:
1293
+ return False
1294
+ except Exception: # nosec B110 - ADVISORY probe: a fault here must decide nothing
1295
+ pass
1296
+ with ledger_lock(path):
1297
+ if not path.is_file():
1298
+ return False
1299
+ text = path.read_text(encoding="utf-8")
1300
+ updated = _apply_decision(text, dw_id, date, label, detail)
1301
+ if updated is None:
1302
+ return False
1303
+ text = updated
1304
+ if close_note is not None:
1305
+ closed = _apply_done(text, dw_id, date, close_note)
1306
+ if closed is not None:
1307
+ text = closed
1308
+ atomic_write_text(path, text)
1309
+ return True
1310
+
1311
+
1312
+ def append_decision(path: Path, dw_id: str, date: str, label: str, detail: str) -> bool:
1313
+ """Record a human decision on an entry without changing its status.
1314
+
1315
+ The no-close case of :func:`record_decision`, which is where the lock is
1316
+ taken and the contract documented. It delegates rather than duplicating the
1317
+ read->edit->write so that one public call is exactly one acquisition."""
1318
+ return record_decision(path, dw_id, date, label, detail)
1319
+
1320
+
1321
+ DW_ID_RE = re.compile(r"\bDW-(\d+)\b")
1322
+
1323
+
1324
+ def next_seq(text: str) -> int:
1325
+ """The next free DW sequence number — one past the highest DW-<n> anywhere
1326
+ in the ledger (malformed entries included, so a number is never reused and
1327
+ the sweep numbering check stays satisfied)."""
1328
+ nums = [int(m.group(1)) for m in DW_ID_RE.finditer(text)]
1329
+ return (max(nums) + 1) if nums else 1
1330
+
1331
+
1332
+ def field_line_present(body: str, field: str, value: str) -> bool:
1333
+ """True when `body` has a `field:` line whose value is exactly `value`,
1334
+ matching the shapes append_entry writes (plain, or backtick-wrapped as for
1335
+ `source_spec:`). Anchored per-line so an incidental substring elsewhere in
1336
+ the body (e.g. inside `reason:`) never counts as a match."""
1337
+ v = re.escape(value)
1338
+ return re.search(rf"(?m)^{re.escape(field)}:[ \t]*`?{v}`?[ \t]*$", body) is not None
1339
+
1340
+
1341
+ @dataclass(frozen=True)
1342
+ class EntrySpec:
1343
+ """One :func:`append_entry` call's arguments as data, for the batched writer.
1344
+
1345
+ The defaults are that function's defaults, so a spec built from the same
1346
+ values produces the same entry. Frozen because :func:`append_entries`
1347
+ validates the whole sequence before it takes the lock and then trusts what it
1348
+ validated — a spec mutated in between would be written unchecked."""
1349
+
1350
+ title: str
1351
+ origin: str
1352
+ source_spec: str
1353
+ reason: str
1354
+ location: str = "n/a"
1355
+ status: str = "open"
1356
+ severity: str | None = None
1357
+
1358
+
1359
+ def _apply_append(text: str, spec: EntrySpec) -> tuple[str, str | None]:
1360
+ """Append one canonical `### DW-<seq>` entry *within* `text`, returning the
1361
+ new text and the id minted — or `text` unchanged and None when an open entry
1362
+ already carries the same `origin:` marker and `source_spec:`.
1363
+
1364
+ Pure — text in, text out, no `Path` and no I/O — which is what lets
1365
+ :func:`append_entries` run it once per spec against the text as it evolves,
1366
+ inside a single :func:`ledger_lock` hold. Both halves that make a batch
1367
+ differ from a loop read that evolving text: `next_seq` mints past the entry
1368
+ the previous spec just added, so ids are sequential rather than colliding,
1369
+ and the idempotence scan sees it too, so two identical specs in one call
1370
+ dedupe against each other exactly as a serial pair would.
1371
+
1372
+ Free text is sanitized (:func:`_one_line`) **before** the idempotence scan,
1373
+ which compares the caller's value against the stored one via
1374
+ :func:`field_line_present`: sanitizing afterwards would compare a raw value
1375
+ against a sanitized line, so every replay of the same multiline defer would
1376
+ miss its own entry and append another.
1377
+
1378
+ The scan is deliberately open-only: a closed entry with the same marker does
1379
+ not suppress the append, because the work has come back."""
1380
+ given_title = bool(spec.title)
1381
+ title = _one_line(spec.title)
1382
+ origin = _one_line(spec.origin)
1383
+ source_spec = _one_line(spec.source_spec)
1384
+ reason = _one_line(spec.reason)
1385
+ location = _one_line(spec.location)
1386
+ for entry in parse_ledger(text):
1387
+ if (
1388
+ entry.open
1389
+ and field_line_present(entry.body, "origin", origin)
1390
+ and field_line_present(entry.body, "source_spec", source_spec)
1391
+ ):
1392
+ return text, None
1393
+ dw_id = f"DW-{next_seq(text)}"
1394
+ if given_title and not title.strip():
1395
+ # A break-only title sanitizes to nothing, and `### DW-<n>: ` is a
1396
+ # heading `HEADING_RE`'s `(.+?)` does not match: the caller is handed an
1397
+ # id no reader can find while `next_seq` has already burned it.
1398
+ #
1399
+ # Tested with `.strip()`, not `not title`: a title of `" "` carries no
1400
+ # break at all, so `_one_line` returns it unchanged by the byte-identity
1401
+ # fast path and it stays truthy. It parses, but renders blank in
1402
+ # `status`, `--json` and the TUI — the unidentifiable half of the same
1403
+ # problem, reached without ever touching the sanitizer.
1404
+ #
1405
+ # Scoped to a title that *had* content: an already-empty one keeps its
1406
+ # long-standing behavior, and the invariant is about non-empty values.
1407
+ title = f"(untitled {dw_id})"
1408
+ lines = [
1409
+ f"### {dw_id}: {title}",
1410
+ f"origin: {origin}",
1411
+ f"location: {location}",
1412
+ f"source_spec: `{source_spec}`",
1413
+ ]
1414
+ if spec.severity:
1415
+ lines.append(f"severity: {spec.severity}")
1416
+ lines.append(f"reason: {reason}")
1417
+ lines.append(f"status: {spec.status}")
1418
+ block = "\n".join(lines) + "\n"
1419
+ # exactly one blank line between the previous content and the new entry
1420
+ if text == "" or text.endswith("\n\n"):
1421
+ sep = ""
1422
+ elif text.endswith("\n"):
1423
+ sep = "\n"
1424
+ else:
1425
+ sep = "\n\n"
1426
+ return text + sep + block, dw_id
1427
+
1428
+
1429
+ def _apply_appends(text: str, specs: Sequence[EntrySpec]) -> tuple[str, list[str | None]]:
1430
+ """Fold every spec through :func:`_apply_append` *within* `text`, returning
1431
+ the new text and one minted id per spec — None where the spec deduped
1432
+ against an open entry that already carries its marker.
1433
+
1434
+ Pure — text in, text out, no `Path` and no I/O — and ONE body for the
1435
+ advisory pre-lock probe and the locked pass, so the two cannot drift. Each
1436
+ spec sees the text the previous one produced, which is what makes ids
1437
+ sequential and lets two identical specs in one call dedupe against each
1438
+ other; see :func:`_apply_append` for why that evolution is load-bearing."""
1439
+ minted: list[str | None] = []
1440
+ for spec in specs:
1441
+ text, dw_id = _apply_append(text, spec)
1442
+ minted.append(dw_id)
1443
+ return text, minted
1444
+
1445
+
1446
+ def append_entries(path: Path, specs: Sequence[EntrySpec]) -> list[str | None]:
1447
+ """Append every entry in `specs` in ONE read and ONE atomic write, returning
1448
+ each spec's minted id — or None in its position when that spec deduped
1449
+ against an already-open entry. Creates the ledger (and parent dir) if it does
1450
+ not yet exist.
1451
+
1452
+ A thin wrapper over :func:`append_entries_published`, for the callers that
1453
+ only need the ids. One acquisition, in the leaf.
1454
+ """
1455
+ return append_entries_published(path, specs)[0]
1456
+
1457
+
1458
+ def append_entries_published(
1459
+ path: Path, specs: Sequence[EntrySpec]
1460
+ ) -> tuple[list[str | None], str | None]:
1461
+ """:func:`append_entries`, additionally handing back the text it published —
1462
+ or None when it wrote nothing, because every spec deduped or `specs` was
1463
+ empty.
1464
+
1465
+ For a caller that has to record WHAT IT WROTE rather than what the file holds
1466
+ afterwards. Reading the ledger back after this returns is a different
1467
+ question with the same answer only when nobody else wrote in between: the
1468
+ lock is released before the read, so a concurrent mutator's bytes would be
1469
+ folded into the caller's own anchor. That matters for
1470
+ ``post_engine_ledger_digest``, whose whole job is to say "these bytes are
1471
+ ours" — counting a rival's write as ours would have the pre-harvest restore
1472
+ retract it, which is the loss this module exists to prevent (#286). Taking
1473
+ the text from inside the hold removes the window rather than narrowing it.
1474
+
1475
+ The returned text is what was handed to
1476
+ :func:`~froid_loop.platform_util.atomic_write_text`, so a digest of it equals
1477
+ a digest of a later ``read_text`` of the file: the writer's text mode
1478
+ translates the newlines on the way out and ``read_text`` normalizes them back
1479
+ on the way in.
1480
+
1481
+ ONE locked read->edit->write: the whole cycle runs under the cross-process
1482
+ ledger lock (#286/#469), so concurrent mutators — a second run, a sweep, the
1483
+ TUI decision modal, ``sweep --archive`` — serialize here rather than trading
1484
+ last-write-wins. The hold spans every `next_seq` mint as well as every
1485
+ idempotence scan, which is what stops two concurrent appenders reading the
1486
+ same highest id and both minting it (#469).
1487
+
1488
+ Byte-identical to a serial :func:`append_entry` loop over the same specs,
1489
+ because each spec is applied to the text the previous one produced rather
1490
+ than to the text this call read. That is what a naive batch gets wrong: minted
1491
+ against the original text, every spec in one call would claim the same id.
1492
+
1493
+ ALL specs are validated — the `status` and `severity` enumerations, which are
1494
+ orchestrator-owned and so raise rather than sanitize — before the lock is
1495
+ taken and before anything is written. All-or-nothing: a bad spec anywhere in
1496
+ the sequence leaves the ledger exactly as it was, rather than committing the
1497
+ prefix that happened to precede it. Validating above the lock also means a
1498
+ programmer bug reports itself without first waiting on another process.
1499
+
1500
+ Nothing is written when every spec dedupes, and no lock is taken either
1501
+ (#736): a replayed defer is answered from one advisory read that runs
1502
+ :func:`_apply_appends`, the same helper the locked pass runs, so it leaves
1503
+ the file untouched rather than rewriting it byte-for-byte. Deliberately NO
1504
+ missing-ledger guard, unlike its sibling mutators: an absent ledger here
1505
+ means CREATE, which is a write, and a write must take the lock.
1506
+
1507
+ The write goes through :func:`~froid_loop.platform_util.atomic_write_text` for
1508
+ the reasons documented on :func:`mark_done_many`, plus one this sibling shares
1509
+ with it: a bare ``Path.write_text`` truncates *before* it encodes, so any
1510
+ failure between the two — an unencodable value, ``ENOSPC``, ``EIO`` — leaves a
1511
+ zero-byte ledger where every entry used to be (#328).
1512
+ """
1513
+ for spec in specs:
1514
+ _require_canonical_status(spec.status)
1515
+ # The whitelist is derived from the legacy parser's alias table (defined
1516
+ # below; resolved at call time) so what this writer emits and what
1517
+ # `field_severity` normalizes to cannot drift apart.
1518
+ if spec.severity and spec.severity not in _CANONICAL_SEVERITIES:
1519
+ raise ValueError(
1520
+ f"severity must be one of {sorted(_CANONICAL_SEVERITIES)}: {spec.severity!r}"
1521
+ )
1522
+ if not specs:
1523
+ # Nothing to serialize against, so nothing to take a lock for.
1524
+ return [], None
1525
+ try:
1526
+ # ADVISORY pre-lock probe (#736): one read — shaped exactly like the
1527
+ # locked one, absence included — and the same pure decision the locked
1528
+ # pass makes. Only a "would write nothing" answer is acted on, and here
1529
+ # that is every spec deduping, which is also the only case where the
1530
+ # published text is the text already on disk. Anything else, including
1531
+ # any fault here, falls through to the hold, which re-reads and decides.
1532
+ probe = path.read_text(encoding="utf-8") if path.is_file() else ""
1533
+ minted = _apply_appends(probe, specs)[1]
1534
+ if all(dw_id is None for dw_id in minted):
1535
+ return minted, None
1536
+ except Exception: # nosec B110 - ADVISORY probe: a fault here must decide nothing
1537
+ pass
1538
+ with ledger_lock(path):
1539
+ text = path.read_text(encoding="utf-8") if path.is_file() else ""
1540
+ text, minted = _apply_appends(text, specs)
1541
+ if all(dw_id is None for dw_id in minted):
1542
+ return minted, None
1543
+ path.parent.mkdir(parents=True, exist_ok=True)
1544
+ atomic_write_text(path, text)
1545
+ # Returned from INSIDE the hold: this is the published text by
1546
+ # construction, not a read-back that a rival could have moved.
1547
+ return minted, text
1548
+
1549
+
1550
+ def append_entry(
1551
+ path: Path,
1552
+ *,
1553
+ title: str,
1554
+ origin: str,
1555
+ source_spec: str,
1556
+ reason: str,
1557
+ location: str = "n/a",
1558
+ status: str = "open",
1559
+ severity: str | None = None,
1560
+ ) -> str | None:
1561
+ """Append a new canonical `### DW-<seq>` entry numbered past the highest
1562
+ existing DW id, returning the new id (e.g. "DW-42").
1563
+
1564
+ Idempotent: returns None without writing when an open entry already carries
1565
+ the same `origin:` marker and `source_spec:` — so re-running the same defer
1566
+ (e.g. a second sweep of the same story) never duplicates the entry. Creates
1567
+ the ledger (and parent dir) if it does not yet exist.
1568
+
1569
+ The one-spec case of :func:`append_entries`, which is where the lock is taken
1570
+ and the contract documented. It delegates rather than duplicating the
1571
+ read->edit->write so that one public call is exactly one acquisition."""
1572
+ return append_entries(
1573
+ path,
1574
+ [
1575
+ EntrySpec(
1576
+ title=title,
1577
+ origin=origin,
1578
+ source_spec=source_spec,
1579
+ reason=reason,
1580
+ location=location,
1581
+ status=status,
1582
+ severity=severity,
1583
+ )
1584
+ ],
1585
+ )[0]
1586
+
1587
+
1588
+ ARCHIVE_REL = "deferred-work-archive.md"
1589
+ # The archive sibling is never locked in its own right: :func:`archive_closed`
1590
+ # is the only writer, and it holds the LEDGER's :func:`ledger_lock` across both
1591
+ # writes (#286/#469). Any future writer of this file must take that same lock.
1592
+ # A stub left by a prior archive_closed run carries this field. The next run
1593
+ # reads it to skip entries whose body has already been moved — without it,
1594
+ # every run would re-archive the stub (a heading + status line) and the
1595
+ # archive would accumulate duplicates.
1596
+ _ARCHIVED_FIELD_RE = re.compile(r"^archived:", re.MULTILINE)
1597
+
1598
+ # What :func:`mark_open` leaves where that stamp was. A reopened entry is not
1599
+ # archived — its body is back in the ledger — but the body the undone close
1600
+ # moved out still is, and this line is what a triage session follows to it.
1601
+ # Deliberately a different field name: `archived:` means "the body is
1602
+ # elsewhere", which a reopened entry must not claim, and a line matching
1603
+ # `_ARCHIVED_FIELD_RE` here would rebuild the exact `_STUB_BODY_RE` shape on
1604
+ # the next reopenable close.
1605
+ _ARCHIVED_BODY_FIELD = "archived-body:"
1606
+
1607
+
1608
+ def _archived_line_spans(entry: DWEntry) -> list[tuple[int, int]]:
1609
+ """Body-relative spans of the entry's live ``archived:`` field lines, each
1610
+ covering the whole line including its terminating newline.
1611
+
1612
+ Reads through :func:`_quoted` for the same reason every gate scan does:
1613
+ an entry documenting the archive field in a fenced example carries the
1614
+ line in column 0, right where the anchor looks, and without the fence
1615
+ check a quoted ``archived:`` would be mistaken for the real thing. The one
1616
+ place that rule is written, so the three questions asked about the field —
1617
+ is this entry archived, what does its body say apart from the stamp, and
1618
+ which bytes must a reopen rename — cannot answer it differently.
1619
+
1620
+ Whole lines rather than match starts because both cutting callers remove
1621
+ the line, and a span ending at the anchor would leave the stamp's value
1622
+ behind as orphaned text.
1623
+ """
1624
+ spans: list[tuple[int, int]] = []
1625
+ for m in _ARCHIVED_FIELD_RE.finditer(entry.body):
1626
+ if _quoted(entry, m.start()):
1627
+ continue
1628
+ line_end = entry.body.find("\n", m.end())
1629
+ spans.append((m.start(), len(entry.body) if line_end == -1 else line_end + 1))
1630
+ return spans
1631
+
1632
+
1633
+ def _is_archived(entry: DWEntry) -> bool:
1634
+ """Whether the entry carries a live ``archived:`` field line (not a quoted
1635
+ example), marking it as touched by :func:`archive_closed` — a stub in the
1636
+ live ledger, or an archived body in the archive file.
1637
+ """
1638
+ return bool(_archived_line_spans(entry))
1639
+
1640
+
1641
+ def _body_without_archived(entry: DWEntry) -> str:
1642
+ """The entry's body with its live ``archived:`` stamps and trailing blank
1643
+ lines removed — the comparison key for :func:`archive_closed`'s
1644
+ crash-recovery skip.
1645
+
1646
+ An archived twin is its ledger entry plus exactly one ``archived:`` line,
1647
+ so the two are the same content only once that line is discounted; trailing
1648
+ newlines go with it because they record where the entry sat in its file,
1649
+ not what it says. Everything else is compared verbatim, deliberately: the
1650
+ cheap wrong answer is archiving a body twice, and the expensive one is
1651
+ deciding a divergent re-closure was already saved and dropping it (#711).
1652
+ """
1653
+ body = entry.body
1654
+ for start, end in reversed(_archived_line_spans(entry)):
1655
+ body = body[:start] + body[end:]
1656
+ return body.rstrip("\n")
1657
+
1658
+
1659
+ def _archived_stamp(entry: DWEntry) -> str | None:
1660
+ """The value of the entry's first live ``archived:`` field line, or None
1661
+ when it carries none.
1662
+
1663
+ Read from an *archive* twin, this is what a stub pointing at that block
1664
+ must carry — and what :func:`mark_open` demotes into an `archived-body:`
1665
+ pointer. The archive holds several blocks per id by design, so the stamp
1666
+ narrows rather than identifies: two closures archived on one day share it,
1667
+ and the append-only file's order is the tie-break (later block, later
1668
+ closure).
1669
+ """
1670
+ spans = _archived_line_spans(entry)
1671
+ if not spans:
1672
+ return None
1673
+ start, end = spans[0]
1674
+ return entry.body[start:end].split(":", 1)[1].strip()
1675
+
1676
+
1677
+ # Field lines a stub must carry when the archived body had them, because
1678
+ # downstream readers key on them regardless of status: `gate:` (validate's
1679
+ # closed-entry gate report deliberately keeps speaking), `origin:` +
1680
+ # `source_spec:` (the engine's status-agnostic harvest-replay dedupe), and the
1681
+ # reopenable-close undo tail (`mark_open`'s adjacency requirement).
1682
+ _PRESERVED_FIELD_RE = re.compile(r"^(gate:.*|origin:.*|source_spec:.*)$", re.MULTILINE)
1683
+
1684
+ # The exact stub shape :func:`archive_closed` leaves in the live ledger.
1685
+ # A done entry that merely carries a hand-written `archived:` line does NOT
1686
+ # match — it is a real entry, not a stub, and must still be archived.
1687
+ _STUB_BODY_RE = re.compile(
1688
+ r"### .*: .*\n\n"
1689
+ r"status: done [0-9]{4}-[0-9]{2}-[0-9]{2}\n"
1690
+ # Separators mirror `_MARK_DONE_TAIL_RE`, which tolerates tabs: that regex
1691
+ # decides what `_preserved_stub_lines` copies into the stub verbatim, so a
1692
+ # stricter shape here reads a stub this module just wrote as a live entry
1693
+ # and re-archives it on every run, forever, appending nothing (#711).
1694
+ r"(?:resolution:[ \t]*[^\n]*\nresolution-undo:[ \t]*[0-9a-f]{64}[ \t]+[^\n]*\n)?"
1695
+ r"(?:(?:gate:|origin:|source_spec:)[^\n]*\n)*"
1696
+ r"archived: [^\n]*\n"
1697
+ r"\n?"
1698
+ )
1699
+
1700
+
1701
+ def _is_stub(entry: DWEntry) -> bool:
1702
+ """Whether the entry is a stub left by a prior :func:`archive_closed` run.
1703
+
1704
+ Shape-based rather than `archived:`-line-based: a done entry a human
1705
+ annotated with a stray unfenced ``archived:`` line is a real entry whose
1706
+ body still belongs in the live ledger — skipping it forever on the strength
1707
+ of one line would silently exclude it from every future archive.
1708
+ """
1709
+ return entry.done and _STUB_BODY_RE.fullmatch(entry.body.rstrip("\n") + "\n") is not None
1710
+
1711
+
1712
+ def _preserved_stub_lines(entry: DWEntry) -> list[str]:
1713
+ """The load-bearing field lines a stub must keep from the archived body.
1714
+
1715
+ Scanned fence-aware like every field read in this module: a fenced example
1716
+ documenting `origin:` is not a declaration. The undo tail is read with the
1717
+ same adjacency regex :func:`mark_open` will later use against the stub, so
1718
+ what qualifies here is exactly what remains undoable there.
1719
+ """
1720
+ lines = [
1721
+ entry.body[m.start() : m.end()]
1722
+ for m in _PRESERVED_FIELD_RE.finditer(entry.body)
1723
+ if not _quoted(entry, m.start())
1724
+ ]
1725
+ if entry.status_span is not None:
1726
+ tail = _MARK_DONE_TAIL_RE.match(entry.body, entry.status_span[1])
1727
+ if tail is not None:
1728
+ lines = [tail.group(0).lstrip("\n")] + lines
1729
+ return lines
1730
+
1731
+
1732
+ def _close_date(entry: DWEntry) -> str | None:
1733
+ """The ISO close date from a ``done <date>`` status, or None when the
1734
+ entry is not done, is done without a date suffix, or carries a date
1735
+ that does not match the ISO ``YYYY-MM-DD`` shape.
1736
+
1737
+ Entries closed with a bare ``status: done`` (no date) or a hand-edited
1738
+ non-ISO date are skipped by :func:`archive_closed`: there is no close
1739
+ date to compare against a ``--before`` cutoff, and the stub the function
1740
+ leaves in the ledger needs one to stay readable as done.
1741
+ """
1742
+ if not entry.done:
1743
+ return None
1744
+ parts = entry.status.split()
1745
+ if len(parts) != 2: # exactly `done YYYY-MM-DD` — extra tokens are not a close date
1746
+ return None
1747
+ # Same shape check as `_require_iso_date` (well-formed regex AND a real
1748
+ # calendar day), skip-not-raise: a hand-edited close is data, not a bug.
1749
+ return _iso_date_or_none(parts[1])
1750
+
1751
+
1752
+ def _eligible_for_archive(text: str, before: str | None) -> list[tuple[DWEntry, str]]:
1753
+ """Every entry in `text` :func:`archive_closed` would move, paired with its
1754
+ close date, in ledger order.
1755
+
1756
+ Pure — text in, entries out, no `Path` and no I/O — and ONE body for the
1757
+ advisory pre-lock probe and the locked pass, so the two cannot drift. Three
1758
+ skips make up the decision: an entry that is not done, or done without a
1759
+ date, has nothing to compare or to stamp a stub with; `before` excludes
1760
+ entries closed on or after the cutoff; and a stub from a prior run is
1761
+ already archived."""
1762
+ to_archive: list[tuple[DWEntry, str]] = []
1763
+ for entry in parse_ledger(text):
1764
+ close_date = _close_date(entry)
1765
+ if close_date is None:
1766
+ continue # not done, or done without a date
1767
+ if before is not None and close_date >= before:
1768
+ continue # closed on or after the cutoff
1769
+ if _is_stub(entry):
1770
+ continue # stub from a prior archive_closed run
1771
+ to_archive.append((entry, close_date))
1772
+ return to_archive
1773
+
1774
+
1775
+ def archive_closed(
1776
+ path: Path,
1777
+ *,
1778
+ before: str | None = None,
1779
+ archive_date: str | None = None,
1780
+ dry_run: bool = False,
1781
+ ) -> list[str]:
1782
+ """Move closed (``status: done <date>``) ledger entries to a sibling
1783
+ archive file (:data:`ARCHIVE_REL`), replacing each with a minimal stub
1784
+ that preserves the DW- id for grep and ``closes_deferred``
1785
+ cross-references.
1786
+
1787
+ Returns the list of archived ids, in ledger order. ``dry_run=True``
1788
+ returns the ids that *would* be archived without writing anything.
1789
+
1790
+ Each archived entry's body is preserved verbatim in the archive file,
1791
+ with an ``archived: <date>`` field line appended after the entry's status
1792
+ line. The stub left in the live ledger keeps the heading, a ``status:
1793
+ done <date>`` line (so :func:`parse_ledger` reads it as done and
1794
+ :func:`open_ids` drops it), an ``archived: <date>`` line (so a subsequent
1795
+ run skips it rather than re-archiving the stub), and the entry's
1796
+ load-bearing field lines — ``gate:``, ``origin:``/``source_spec:``, and
1797
+ the reopenable-close undo tail — because downstream readers key on those
1798
+ regardless of status (validate's closed-gate report, the engine's
1799
+ harvest-replay dedupe, and sweep bundle rollback respectively).
1800
+
1801
+ ``before`` (ISO ``YYYY-MM-DD``) archives only entries closed strictly
1802
+ *before* that date. Entries with ``status: done`` (no date) are always
1803
+ skipped — there is no close date to compare against a cutoff or to stamp
1804
+ the stub with. Open and legacy entries are never touched.
1805
+
1806
+ Dates are validated with :func:`_require_iso_date` (same validation as
1807
+ the existing close-path writers), ahead of the ``is_file`` short-circuit
1808
+ so a programmer bug fails the same way whether or not a ledger exists.
1809
+ Both writes — the trimmed ledger and the appended archive — go through
1810
+ :func:`atomic_write_text`, the same primitive every ledger writer uses.
1811
+ The archive file accumulates on repeat runs: new entries are appended to
1812
+ the existing file, never overwritten, and stubs from a prior run are
1813
+ skipped by their exact stub shape. A stub's ``archived:`` date names the
1814
+ archive block holding its body, so an entry recovered from a crashed run
1815
+ is stamped with the date already on that block rather than with this run's.
1816
+
1817
+ The whole read->edit->write runs under the cross-process ledger lock
1818
+ (#286/#469): concurrent mutators — a second run, a sweep, the TUI decision
1819
+ modal, ``sweep --archive`` — serialize here rather than trading
1820
+ last-write-wins. ONE acquisition spans BOTH writes — the
1821
+ archive sibling has no lock of its own precisely because it is only ever
1822
+ written under its ledger's lock — and an ELIGIBLE ``dry_run`` runs inside
1823
+ the hold too, so there is one code path rather than a locked and an unlocked
1824
+ one. A run with nothing eligible is the exception, and only because it is
1825
+ not a code path at all: the advisory pre-lock probe (#736) answers it with
1826
+ the empty list before either branch is reached, so ``sweep --archive`` over
1827
+ a ledger holding nothing closed keeps reporting success where the state root
1828
+ cannot be derived or the lock cannot be taken.
1829
+ """
1830
+ if before is not None:
1831
+ _require_iso_date(before)
1832
+ if archive_date is not None:
1833
+ _require_iso_date(archive_date)
1834
+ if not path.is_file():
1835
+ # No ledger means no write, and so no lock — the order
1836
+ # `sprintstatus.advance` already keeps for its own missing-board case.
1837
+ # Acquiring first would turn "there is nothing to archive", which
1838
+ # `froid-loop sweep --archive` reports as SUCCESS, into a failure wherever
1839
+ # the state root cannot be derived: a released behavior, changed by a lock
1840
+ # taken for a file that is not there. Rechecked under the hold below,
1841
+ # deletion being able to race this answer.
1842
+ return []
1843
+ try:
1844
+ # ADVISORY pre-lock probe (#736): one read, and the same pure decision
1845
+ # the locked pass makes. Only a "would write nothing" answer is acted on
1846
+ # — the call then serializes at this read. Above the `dry_run` branch on
1847
+ # purpose, so a nothing-eligible dry run skips the lock too; an ELIGIBLE
1848
+ # dry run still runs under the hold, where the one code path is. Anything
1849
+ # else, including any fault here, falls through to that hold.
1850
+ probe = path.read_text(encoding="utf-8")
1851
+ if not _eligible_for_archive(probe, before):
1852
+ return []
1853
+ except Exception: # nosec B110 - ADVISORY probe: a fault here must decide nothing
1854
+ pass
1855
+ with ledger_lock(path):
1856
+ if not path.is_file():
1857
+ return []
1858
+ text = path.read_text(encoding="utf-8")
1859
+ to_archive = _eligible_for_archive(text, before)
1860
+ if not to_archive:
1861
+ return []
1862
+ archived_ids = [e.id for e, _ in to_archive]
1863
+ if dry_run:
1864
+ return archived_ids
1865
+ stamp = archive_date or calendar_date.today().isoformat()
1866
+ archive_path = path.parent / ARCHIVE_REL
1867
+ existing = archive_path.read_text(encoding="utf-8") if archive_path.is_file() else ""
1868
+ # Append an `archived:` line after each entry's status line. The status
1869
+ # span is body-relative, so the insertion works within the body slice —
1870
+ # same offset math as `_insert_after_status`, applied to the body.
1871
+ #
1872
+ # Crash recovery: the archive is written BEFORE the ledger (see below), so
1873
+ # a crash between the two writes leaves the ledger with full entries whose
1874
+ # bodies are already in the archive. A retry must still stub those ledger
1875
+ # entries (completing the interrupted operation) but must NOT append their
1876
+ # bodies again — an append-only archive accumulating duplicates. Entries
1877
+ # whose parsed archive twin carries a live (non-fenced) ``archived:``
1878
+ # field are therefore skipped here and only replaced with stubs below.
1879
+ #
1880
+ # The twin must match in BODY, not merely in id and close date. A DW id is
1881
+ # reusable across closures (`mark_open` reopens, a re-close follows) and a
1882
+ # closed entry still accepts writes (`append_decision` does not read
1883
+ # status), so id + date names a *closure slot*, not its content: reopened
1884
+ # and re-closed the same day with a new resolution, or annotated with a
1885
+ # decision after its body was archived, the ledger entry and its twin
1886
+ # differ. Skipping on the slot alone stubbed that entry over its own
1887
+ # content while reporting the id as archived — the body reached neither
1888
+ # file (#711). A body that differs is appended instead; the archive holds
1889
+ # several blocks per id by design, and over-archiving is recoverable where
1890
+ # a silent drop is not.
1891
+ archive_blocks: list[str] = []
1892
+ already_archived = {
1893
+ e.id: ((_close_date(e), _body_without_archived(e)), _archived_stamp(e))
1894
+ for e in parse_ledger(existing)
1895
+ if _is_archived(e)
1896
+ } # fence-aware: a quoted example in the archive is not a real body
1897
+ # A recovered entry's stub is stamped with the date already on its archived
1898
+ # body, not with this run's. The two diverge whenever the retry lands on a
1899
+ # later day than the crashed run, and the stamp is not decoration: it is
1900
+ # what picks one of an id's several archive blocks — for a reader following
1901
+ # the stub, and for the `archived-body:` pointer `mark_open` demotes that
1902
+ # stamp into, which is a reopened entry's only route back to its body
1903
+ # (#711 review). A stub naming a date no block carries resolves to nothing.
1904
+ recovered_stamps: dict[str, str] = {}
1905
+ for entry, close_date in to_archive:
1906
+ twin = already_archived.get(entry.id)
1907
+ if twin is not None and twin[0] == (close_date, _body_without_archived(entry)):
1908
+ # this closure's body is already archived (crashed prior run)
1909
+ if twin[1] is not None:
1910
+ recovered_stamps[entry.id] = twin[1]
1911
+ continue
1912
+ body = entry.body
1913
+ assert entry.status_span is not None # done with a date implies a status line
1914
+ pos = entry.status_span[1]
1915
+ body = body[:pos] + f"\narchived: {stamp}" + body[pos:]
1916
+ archive_blocks.append(body)
1917
+ # Appended, never prepended: for one id the file's order is closure order,
1918
+ # which is the documented tie-break when two closures were archived on the
1919
+ # same day and so carry the same stamp (#711 review).
1920
+ if archive_blocks:
1921
+ if existing == "" or existing.endswith("\n\n"):
1922
+ sep = ""
1923
+ elif existing.endswith("\n"):
1924
+ sep = "\n"
1925
+ else:
1926
+ sep = "\n\n"
1927
+ archive_content = existing + sep + "".join(archive_blocks)
1928
+ else:
1929
+ archive_content = existing # pure crash-recovery pass: only stub the ledger
1930
+ # Replace each archived entry's span with a stub, working backwards so
1931
+ # earlier spans are unaffected by later replacements — the same
1932
+ # text-surgery pattern as `_apply_done`, applied to multiple entries.
1933
+ for entry, close_date in reversed(to_archive):
1934
+ preserved = "".join(f"{line}\n" for line in _preserved_stub_lines(entry))
1935
+ stub = (
1936
+ f"### {entry.id}: {entry.title}\n\n"
1937
+ f"status: done {close_date}\n"
1938
+ f"{preserved}"
1939
+ f"archived: {recovered_stamps.get(entry.id, stamp)}\n\n"
1940
+ )
1941
+ start, end = entry.span
1942
+ text = text[:start] + stub + text[end:]
1943
+ # Write the archive BEFORE the ledger: a crash between writes leaves the
1944
+ # archive with extra content (harmless — the archive is append-only) and
1945
+ # the ledger unchanged (safe — the bodies are still in the live file).
1946
+ # Writing the ledger first would leave stubs in the ledger with no bodies
1947
+ # in the archive — content lost.
1948
+ atomic_write_text(archive_path, archive_content)
1949
+ atomic_write_text(path, text)
1950
+ return archived_ids
1951
+
1952
+
1953
+ # ------------------------------------------------------------------- legacy
1954
+ #
1955
+ # Ledgers written before the DW format (older FROID-method projects) are
1956
+ # freeform markdown: "## Deferred from: ..." sections holding id'd or
1957
+ # strikethrough bullets, "### D-1.2-003: title — RESOLVED" entry headings,
1958
+ # topic sections closed with "(... — DONE)". parse_legacy() reads them
1959
+ # tolerantly so the TUI can display them and a sweep can migrate them; the
1960
+ # strict DW contract above is untouched — legacy items have no status line
1961
+ # to flip, so mark_done/open_ids never see them.
1962
+
1963
+ # Severity is extracted forgivingly (the ledger is LLM-written): a
1964
+ # `severity:`/`priority:` field line in any case, plain or bold-bulleted
1965
+ # ("- **Severity:** high"), common synonyms accepted.
1966
+ SEVERITY_ALIASES = {
1967
+ "critical": "critical",
1968
+ "blocker": "critical",
1969
+ "high": "high",
1970
+ "major": "high",
1971
+ "medium": "medium",
1972
+ "med": "medium",
1973
+ "moderate": "medium",
1974
+ "low": "low",
1975
+ "minor": "low",
1976
+ "trivial": "low",
1977
+ }
1978
+ # What every alias above normalizes to, and so the only values `append_entry` may
1979
+ # write. Derived rather than restated: a hand-copied whitelist drifts the moment
1980
+ # an alias is added for a new canonical level.
1981
+ _CANONICAL_SEVERITIES = frozenset(SEVERITY_ALIASES.values())
1982
+
1983
+ SEVERITY_FIELD_RE = re.compile(
1984
+ r"^[ \t]*(?:[-*][ \t]+)?(?:\*\*)?(?:severity|priority)[ \t]*:[ \t]*(?:\*\*)?[ \t]*"
1985
+ r"([A-Za-z][\w-]*)",
1986
+ re.IGNORECASE | re.MULTILINE,
1987
+ )
1988
+
1989
+
1990
+ def field_severity(body: str) -> str | None:
1991
+ m = SEVERITY_FIELD_RE.search(body)
1992
+ return SEVERITY_ALIASES.get(m.group(1).lower()) if m else None
1993
+
1994
+
1995
+ @dataclass(frozen=True)
1996
+ class LegacyEntry:
1997
+ key: str # stable content-derived identity, unique within the file
1998
+ id: str # native id ("W2", "D-CAP-001", "0-1"), "" when the item has none
1999
+ title: str # cleaned one-line title (markers/strikethrough stripped)
2000
+ done: bool
2001
+ severity: str | None # normalized critical/high/medium/low, None unknown
2002
+ body: str # the bullet/heading block verbatim
2003
+ section: str # enclosing ##/### heading text, "" at top level
2004
+ span: tuple[int, int] # char offsets in the ledger text
2005
+
2006
+
2007
+ _DONE_WORDS = r"(?:DONE|RESOLVED|CLOSED|VERIFIED|DOCUMENTED|FIXED)"
2008
+ _LINE_HEADING_RE = re.compile(r"^(#{1,6})[ \t]+(.*?)[ \t]*$")
2009
+ # a single whitespace-free digit-bearing token before ":" or "—" makes a
2010
+ # heading an entry ("### D-CAP-001: title", "## D-8.6-001 — title");
2011
+ # "## Epic 0: ..." has a space and "## 2026-06-09 — ..." is a date, so: section
2012
+ _ENTRY_HEADING_RE = re.compile(r"^(~~)?([^\s:*~]*\d[^\s:*~]*)(?::[ \t]+|[ \t]+[—–][ \t]+)(.+)$")
2013
+ _DATE_TOKEN_RE = re.compile(r"\d{4}-\d{2}-\d{2}")
2014
+ _SECTION_DONE_RE = re.compile(rf"(?:—|–|-|\()[ \t]*{_DONE_WORDS}\b[^)]*\)?[ \t]*$")
2015
+ _TITLE_DONE_SUFFIX_RE = re.compile(rf"[ \t]*(?:—|–|-)[ \t]*{_DONE_WORDS}[ \t]*$")
2016
+ _BARE_DONE_SUFFIX_RE = re.compile(rf"[ \t]*{_DONE_WORDS}\b.*$")
2017
+ _BOLD_DONE_RE = re.compile(rf"\*\*{_DONE_WORDS}\b[^*]*\*\*")
2018
+ _BRACKET_DONE_RE = re.compile(rf"\[{_DONE_WORDS}\]")
2019
+ _DONE_PREFIX_RE = re.compile(rf"^\*\*{_DONE_WORDS}\b[^*]*\*\*:?[ \t]*")
2020
+ # "- W-1.2-c — CLOSED: ..." / "CLOSED 2026-06-11 (story 1.11). ..."
2021
+ _LEAD_DONE_RE = re.compile(rf"^{_DONE_WORDS}\b")
2022
+ _LEAD_DONE_STRIP_RE = re.compile(
2023
+ rf"^{_DONE_WORDS}\b(?:[ \t]+\d{{4}}-\d{{2}}-\d{{2}})?(?:[ \t]*\([^)]*\))?[ \t]*[:.—–-]?[ \t]*"
2024
+ )
2025
+ # The flat review appender — pre-#2651 dev primitives and the attended
2026
+ # `froid-build` — writes a flat block per finding:
2027
+ # - source_spec: `spec-foo.md`
2028
+ # summary: <one sentence>
2029
+ # evidence: <why this is real>
2030
+ # We recognize it so the `summary` becomes the title (not the source_spec path)
2031
+ # and the entry migrates cleanly into the canonical `### DW-<seq>` shape. The
2032
+ # opening line comes from the same `_FLAT_SOURCE_BODY` as FLAT_ENTRY_RE, which
2033
+ # bounds canonical spans on it — see that constant for why they must not drift.
2034
+ _FLAT_SOURCE_RE = re.compile(rf"^{_FLAT_SOURCE_BODY}", re.IGNORECASE)
2035
+ _FLAT_SUMMARY_RE = re.compile(r"^[ \t]*summary:[ \t]*(.*)$", re.IGNORECASE | re.MULTILINE)
2036
+ _BULLET_RE = re.compile(r"^[-*][ \t]+(.*)$")
2037
+ _ITEM_ID_RE = re.compile(
2038
+ r"^(?:\*\*)?([^\s:*~]*\d[^\s:*~]*)(?:\*\*)?(?:[ \t]*[—–][ \t]+|:[ \t]+|[ \t]+-[ \t]+)"
2039
+ )
2040
+ _BRACKET_TOKEN_RE = re.compile(r"\[([A-Za-z]+)[^\]]*\]")
2041
+ _LEAD_BOLD_RE = re.compile(r"^\*\*(.+?)\*\*")
2042
+ _STRUCK_LINE_RE = re.compile(r"^~~(.*)~~")
2043
+ _TRAIL_BRACKET_RE = re.compile(r"[ \t]*\[[^\]]+\][ \t.]*$")
2044
+
2045
+
2046
+ def _bracket_severity(s: str) -> str | None:
2047
+ for m in _BRACKET_TOKEN_RE.finditer(s):
2048
+ sev = SEVERITY_ALIASES.get(m.group(1).lower())
2049
+ if sev:
2050
+ return sev
2051
+ return None
2052
+
2053
+
2054
+ def _clean_title(s: str) -> str:
2055
+ return " ".join(s.replace("**", "").split())
2056
+
2057
+
2058
+ def _item_entry(first: str, body: str, section: str, section_done: bool) -> dict:
2059
+ """Interpret one bullet item; returns the pre-key entry fields."""
2060
+ if _FLAT_SOURCE_RE.match(first):
2061
+ # flat appender block (legacy/attended era): title is the `summary`
2062
+ sm = _FLAT_SUMMARY_RE.search(body)
2063
+ summary = sm.group(1).strip() if sm else ""
2064
+ return {
2065
+ "id": "",
2066
+ "title": _clean_title(summary) if summary else _clean_title(first),
2067
+ "done": section_done,
2068
+ "severity": field_severity(body),
2069
+ "section": section,
2070
+ }
2071
+ content = first
2072
+ struck = False
2073
+ m = _STRUCK_LINE_RE.match(content)
2074
+ if m: # "~~text~~ DONE" / "~~text~~ → resolution" on the first line
2075
+ struck = True
2076
+ content = m.group(1)
2077
+ elif content.startswith("~~") and "~~" in body[2:]:
2078
+ struck = True # strikethrough closes on a later line
2079
+ content = content[2:]
2080
+ item_id = ""
2081
+ m = _ITEM_ID_RE.match(content)
2082
+ if m:
2083
+ item_id = m.group(1)
2084
+ content = content[m.end() :]
2085
+ done = (
2086
+ struck
2087
+ or section_done
2088
+ or bool(_LEAD_DONE_RE.match(content))
2089
+ or bool(_BOLD_DONE_RE.search(body))
2090
+ or bool(_BRACKET_DONE_RE.search(body))
2091
+ )
2092
+ content = _DONE_PREFIX_RE.sub("", content)
2093
+ content = _LEAD_DONE_STRIP_RE.sub("", content)
2094
+ while True: # trailing "[MINOR]" / "[CLOSED]" tokens are not title text
2095
+ trimmed = _TRAIL_BRACKET_RE.sub("", content)
2096
+ if trimmed == content:
2097
+ break
2098
+ content = trimmed
2099
+ bold = _LEAD_BOLD_RE.match(content)
2100
+ if bold and len(bold.group(1).split()) >= 3:
2101
+ title = bold.group(1) # notey: the bold phrase is the title
2102
+ else:
2103
+ title = content
2104
+ return {
2105
+ "id": item_id,
2106
+ "title": _clean_title(title),
2107
+ "done": done,
2108
+ "severity": _bracket_severity(body) or field_severity(body),
2109
+ "section": section,
2110
+ }
2111
+
2112
+
2113
+ def _heading_entry(struck: bool, hid: str, rest: str, body: str, section: str) -> dict:
2114
+ """Interpret one '### D-1: title' entry heading (story-maker shape)."""
2115
+ title = rest
2116
+ done = struck
2117
+ m = _TITLE_DONE_SUFFIX_RE.search(title)
2118
+ if m:
2119
+ done = True
2120
+ title = title[: m.start()]
2121
+ if struck:
2122
+ title = _BARE_DONE_SUFFIX_RE.sub("", title.replace("~~", ""))
2123
+ return {
2124
+ "id": hid,
2125
+ "title": _clean_title(title),
2126
+ "done": done,
2127
+ "severity": field_severity(body) or _bracket_severity(body),
2128
+ "section": section,
2129
+ }
2130
+
2131
+
2132
+ def parse_legacy(text: str) -> list[LegacyEntry]:
2133
+ """Extract legacy (non-DW) deferred items. Canonical DW entries and fenced
2134
+ examples are masked out first, so mixed ledgers parse both ways without
2135
+ overlap and a quoted example contributes nothing to either reading.
2136
+
2137
+ The fenced half is not symmetry for its own sake. `parse_ledger` used to hand
2138
+ a quoted example over as a phantom canonical entry, whose span masked the
2139
+ example here by accident; once it stopped doing that, the same quotation
2140
+ surfaced on this side instead — a bullet or `### DW-n:` heading inside a fence
2141
+ read as a legacy finding (#514).
2142
+ """
2143
+ masked = text
2144
+ # `unclosed_hides_rest=False` for the reason the canonical side uses it: one
2145
+ # stray opener must not blank every legacy finding below it out of view. The
2146
+ # delimiter lines survive as a lone backtick or tilde plus spaces, which no
2147
+ # pattern below can start an item on — masking them too made no test disagree.
2148
+ spans = [e.span for e in parse_ledger(text)] + fenced_spans(text, unclosed_hides_rest=False)
2149
+ for s, t in spans:
2150
+ masked = masked[:s] + re.sub(r"[^\n]", " ", masked[s:t]) + masked[t:]
2151
+
2152
+ found: list[tuple[dict, tuple[int, int]]] = []
2153
+ section = ""
2154
+ section_done = False
2155
+ # a done section with no items yet: emitted as its own done entry unless
2156
+ # bullets, an entry heading, or a deeper child heading claim it first
2157
+ pending: dict | None = None # {"level", "fields", "span"}
2158
+ item: dict | None = None # accumulating bullet or entry heading
2159
+
2160
+ def close_item(end: int) -> None:
2161
+ nonlocal item
2162
+ if item is None:
2163
+ return
2164
+ body = text[item["start"] : end].rstrip()
2165
+ span = (item["start"], item["start"] + len(body))
2166
+ if item["kind"] == "item":
2167
+ fields = _item_entry(item["first"], body, item["section"], item["section_done"])
2168
+ else:
2169
+ fields = _heading_entry(
2170
+ item["struck"], item["hid"], item["rest"], body, item["section"]
2171
+ )
2172
+ found.append((fields, span))
2173
+ item = None
2174
+
2175
+ def emit_pending() -> None:
2176
+ nonlocal pending
2177
+ if pending is not None:
2178
+ found.append((pending["fields"], pending["span"]))
2179
+ pending = None
2180
+
2181
+ offset = 0
2182
+ for line in text.splitlines(keepends=True):
2183
+ masked_line = masked[offset : offset + len(line)].rstrip("\n")
2184
+ hm = _LINE_HEADING_RE.match(masked_line)
2185
+ if hm:
2186
+ level = len(hm.group(1))
2187
+ close_item(offset)
2188
+ if level == 1:
2189
+ emit_pending()
2190
+ section, section_done = "", False
2191
+ elif level in (2, 3):
2192
+ if pending is not None and level > pending["level"]:
2193
+ pending = None # a child heading: the parent is structure
2194
+ else:
2195
+ emit_pending()
2196
+ em = _ENTRY_HEADING_RE.match(hm.group(2))
2197
+ if em and _DATE_TOKEN_RE.fullmatch(em.group(2)):
2198
+ em = None # "## 2026-06-09 — ..." is a dated section
2199
+ if em:
2200
+ pending = None
2201
+ item = {
2202
+ "kind": "heading",
2203
+ "start": offset,
2204
+ "struck": bool(em.group(1)),
2205
+ "hid": em.group(2),
2206
+ "rest": em.group(3),
2207
+ "section": section,
2208
+ "section_done": section_done,
2209
+ }
2210
+ else:
2211
+ htext = hm.group(2)
2212
+ struck = htext.startswith("~~") and "~~" in htext[2:]
2213
+ section = _clean_title(htext.replace("~~", ""))
2214
+ section_done = struck or bool(_SECTION_DONE_RE.search(htext))
2215
+ if section_done:
2216
+ pending = {
2217
+ "level": level,
2218
+ "span": (offset, offset + len(line.rstrip("\n"))),
2219
+ "fields": {
2220
+ "id": "",
2221
+ "title": section,
2222
+ "done": True,
2223
+ "severity": None,
2224
+ "section": "",
2225
+ },
2226
+ }
2227
+ offset += len(line)
2228
+ continue
2229
+ if item is not None and item["kind"] == "heading":
2230
+ if masked_line.strip() == "---" or (masked_line.strip() == "" and line.strip() != ""):
2231
+ close_item(offset) # rule, or a masked canonical entry
2232
+ offset += len(line)
2233
+ continue
2234
+ bm = _BULLET_RE.match(masked_line)
2235
+ if bm:
2236
+ close_item(offset)
2237
+ pending = None
2238
+ item = {
2239
+ "kind": "item",
2240
+ "start": offset,
2241
+ "first": bm.group(1),
2242
+ "section": section,
2243
+ "section_done": section_done,
2244
+ }
2245
+ elif masked_line.strip() in ("", "---"):
2246
+ # a masked canonical entry reads as blank: it still bounds the item
2247
+ if masked_line.strip() == "---" or line.strip() != masked_line.strip():
2248
+ close_item(offset)
2249
+ elif masked_line[0] in " \t":
2250
+ pass # indented continuation of the current item
2251
+ else:
2252
+ close_item(offset) # column-0 prose ends an item, emits nothing
2253
+ offset += len(line)
2254
+ close_item(len(text))
2255
+ emit_pending()
2256
+
2257
+ entries: list[LegacyEntry] = []
2258
+ counts: dict[str, int] = {}
2259
+ for fields, span in found:
2260
+ base = hashlib.sha1(
2261
+ f"{fields['section']}\0{fields['id'] or fields['title']}".encode(),
2262
+ usedforsecurity=False, # display/identity key, not a credential
2263
+ ).hexdigest()[:10]
2264
+ n = counts.get(base, 0) + 1
2265
+ counts[base] = n
2266
+ entries.append(
2267
+ LegacyEntry(
2268
+ key=base if n == 1 else f"{base}-{n}",
2269
+ id=fields["id"],
2270
+ title=fields["title"],
2271
+ done=fields["done"],
2272
+ severity=fields["severity"],
2273
+ body=text[span[0] : span[1]],
2274
+ section=fields["section"],
2275
+ span=span,
2276
+ )
2277
+ )
2278
+ return entries
2279
+
2280
+
2281
+ def has_legacy(text: str) -> bool:
2282
+ return bool(parse_legacy(text))