froid-loop 0.11.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- froid_loop/__init__.py +11 -0
- froid_loop/__main__.py +12 -0
- froid_loop/adapters/__init__.py +3 -0
- froid_loop/adapters/base.py +254 -0
- froid_loop/adapters/entrypoints.py +63 -0
- froid_loop/adapters/env_fault.py +290 -0
- froid_loop/adapters/generic.py +2013 -0
- froid_loop/adapters/mock.py +49 -0
- froid_loop/adapters/multiplexer.py +914 -0
- froid_loop/adapters/opencode_http.py +1687 -0
- froid_loop/adapters/profile.py +650 -0
- froid_loop/adapters/psmux_backend.py +1428 -0
- froid_loop/adapters/registry.py +322 -0
- froid_loop/adapters/tmux_backend.py +35 -0
- froid_loop/adapters/tmux_base.py +630 -0
- froid_loop/checks.py +187 -0
- froid_loop/cli.py +5041 -0
- froid_loop/data/__init__.py +0 -0
- froid_loop/data/froid_loop_hook.py +228 -0
- froid_loop/data/froid_loop_probe_hook.py +88 -0
- froid_loop/data/plugins/example/plugin.toml +21 -0
- froid_loop/data/plugins/tea/plugin.toml +184 -0
- froid_loop/data/plugins/tea/tea_plugin.py +258 -0
- froid_loop/data/plugins/unity/plugin.toml +140 -0
- froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef +16 -0
- froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef.meta +7 -0
- froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs +221 -0
- froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs.meta +11 -0
- froid_loop/data/plugins/unity/unity_assets/_folders/Editor.meta +8 -0
- froid_loop/data/plugins/unity/unity_assets/_folders/FroidLoop.meta +8 -0
- froid_loop/data/plugins/unity/unity_cleanup.py +125 -0
- froid_loop/data/plugins/unity/unity_dialog_probe.py +239 -0
- froid_loop/data/plugins/unity/unity_facts.md +17 -0
- froid_loop/data/plugins/unity/unity_plugin.py +415 -0
- froid_loop/data/plugins/unity/unity_quiesce.py +234 -0
- froid_loop/data/plugins/unity/unity_ready.py +230 -0
- froid_loop/data/plugins/unity/unity_seed_assets.py +298 -0
- froid_loop/data/plugins/unity/unity_setup.py +551 -0
- froid_loop/data/plugins/unity/unity_teardown.py +362 -0
- froid_loop/data/profiles/antigravity.toml +52 -0
- froid_loop/data/profiles/claude.toml +85 -0
- froid_loop/data/profiles/codex.toml +22 -0
- froid_loop/data/profiles/copilot.toml +52 -0
- froid_loop/data/profiles/gemini.toml +26 -0
- froid_loop/data/profiles/opencode.toml +54 -0
- froid_loop/data/settings/core.toml +458 -0
- froid_loop/data/skills/README.md +93 -0
- froid_loop/data/skills/froid-loop-resolve/SKILL.md +288 -0
- froid_loop/data/skills/froid-loop-setup/SKILL.md +161 -0
- froid_loop/data/skills/froid-loop-setup/assets/module-help.csv +3 -0
- froid_loop/data/skills/froid-loop-setup/assets/module.yaml +19 -0
- froid_loop/data/skills/froid-loop-sweep/SKILL.md +100 -0
- froid_loop/data/skills/froid-loop-sweep/automation-mode.md +127 -0
- froid_loop/data/skills/froid-loop-sweep/deferred-work-format.md +302 -0
- froid_loop/data/skills/froid-loop-sweep/migration-mode.md +86 -0
- froid_loop/decisions.py +202 -0
- froid_loop/deferredwork.py +2282 -0
- froid_loop/devcontract.py +892 -0
- froid_loop/diagnostics.py +1104 -0
- froid_loop/documents.py +532 -0
- froid_loop/engine.py +7732 -0
- froid_loop/envvars.py +111 -0
- froid_loop/escalation.py +225 -0
- froid_loop/events.py +266 -0
- froid_loop/fences.py +103 -0
- froid_loop/froidconfig.py +226 -0
- froid_loop/frontmatter.py +526 -0
- froid_loop/gates.py +133 -0
- froid_loop/install.py +2936 -0
- froid_loop/journal.py +178 -0
- froid_loop/machine.py +148 -0
- froid_loop/model.py +898 -0
- froid_loop/operatoractions.py +474 -0
- froid_loop/platform_util.py +1490 -0
- froid_loop/plugins/__init__.py +64 -0
- froid_loop/plugins/bus.py +259 -0
- froid_loop/plugins/context.py +319 -0
- froid_loop/plugins/loader.py +145 -0
- froid_loop/plugins/manifest.py +279 -0
- froid_loop/plugins/model.py +296 -0
- froid_loop/plugins/registry.py +245 -0
- froid_loop/plugins/trust.py +75 -0
- froid_loop/policy.py +1569 -0
- froid_loop/probe.py +1044 -0
- froid_loop/process_host.py +408 -0
- froid_loop/recovery_flow.py +1561 -0
- froid_loop/resolve.py +283 -0
- froid_loop/runs.py +4715 -0
- froid_loop/runsetup.py +1293 -0
- froid_loop/sanitize.py +593 -0
- froid_loop/settings_schema.py +276 -0
- froid_loop/signals.py +160 -0
- froid_loop/sprintstatus.py +609 -0
- froid_loop/statemachine.py +57 -0
- froid_loop/stories.py +615 -0
- froid_loop/stories_engine.py +796 -0
- froid_loop/sweep.py +1892 -0
- froid_loop/tokens.py +196 -0
- froid_loop/tui/__init__.py +11 -0
- froid_loop/tui/app.py +1584 -0
- froid_loop/tui/data.py +840 -0
- froid_loop/tui/launch.py +1003 -0
- froid_loop/tui/screens/__init__.py +1 -0
- froid_loop/tui/screens/dashboard.py +1071 -0
- froid_loop/tui/screens/modals.py +943 -0
- froid_loop/tui/screens/settings_screen.py +477 -0
- froid_loop/tui/settings.py +135 -0
- froid_loop/tui/widgets.py +981 -0
- froid_loop/verify.py +4545 -0
- froid_loop/workspace.py +320 -0
- froid_loop/worktree_flow.py +2301 -0
- froid_loop-0.11.1.dist-info/METADATA +728 -0
- froid_loop-0.11.1.dist-info/RECORD +116 -0
- froid_loop-0.11.1.dist-info/WHEEL +4 -0
- froid_loop-0.11.1.dist-info/entry_points.txt +2 -0
- froid_loop-0.11.1.dist-info/licenses/LICENSE +30 -0
|
@@ -0,0 +1,2282 @@
|
|
|
1
|
+
"""Deterministic reading and editing of the deferred-work ledger.
|
|
2
|
+
|
|
3
|
+
The ledger (`{implementation_artifacts}/deferred-work.md`) is append-only
|
|
4
|
+
markdown in the canonical form documented at
|
|
5
|
+
froid-loop-sweep/deferred-work-format.md: `### DW-<seq>: <title>` headings with
|
|
6
|
+
`origin:`/`location:`/`reason:`/`status:` field lines. The one sanctioned
|
|
7
|
+
rewrite is :func:`archive_closed`, which moves closed entries verbatim to a
|
|
8
|
+
sibling archive file and leaves id-preserving stubs. Pre-#2651 dev primitives
|
|
9
|
+
and the attended `froid-build` append flatter entries here directly, which the
|
|
10
|
+
orchestrator normalizes on sweep; the current unattended primitive records its
|
|
11
|
+
findings in the spec's frontmatter instead and the engine harvests them into
|
|
12
|
+
canonical entries. The
|
|
13
|
+
orchestrator never trusts an LLM to have edited it — status flips and decision
|
|
14
|
+
records happen here, and gates re-read the file from disk.
|
|
15
|
+
|
|
16
|
+
Concurrency (#286/#469): every mutator below is a read->edit->write of the whole
|
|
17
|
+
file, so two orchestrator processes — a second `froid-loop run`, a run plus a
|
|
18
|
+
sweep, a run plus the TUI decision modal, a run plus `sweep --archive` — would
|
|
19
|
+
otherwise both read, both edit, and let the last atomic write win. Each leaf
|
|
20
|
+
mutator therefore runs its whole read->edit->write under :func:`ledger_lock`, a
|
|
21
|
+
cross-process mutex on an out-of-repo sidecar. Readers stay lock-free on
|
|
22
|
+
purpose: every writer replaces the file atomically, so a reader already sees one
|
|
23
|
+
whole version or another, and taking the lock to read would buy nothing while
|
|
24
|
+
adding a way to deadlock. Out of scope by #286's own non-goals: the dev/review
|
|
25
|
+
LLM session writes this file directly and does NOT take the lock — orchestrator
|
|
26
|
+
writes are sequenced against sessions today, so the exposure this closes is
|
|
27
|
+
orchestrator-vs-orchestrator.
|
|
28
|
+
|
|
29
|
+
What the hold covers is every read that decides the PUBLISHED BYTES, which is
|
|
30
|
+
not quite every read (#736). A mutator handed work that turns out to be a no-op
|
|
31
|
+
— ids that are all already done, a decision on an entry that is not there,
|
|
32
|
+
specs that all dedupe, nothing eligible to archive — may answer from ONE
|
|
33
|
+
advisory read taken before the lock, running the same pure decision helper the
|
|
34
|
+
locked pass runs so the two cannot drift. Only a "would write nothing" answer
|
|
35
|
+
is acted on, and such a call linearizes at the probe read: it publishes no
|
|
36
|
+
bytes, so there is nothing for a rival to interleave with. Every other answer,
|
|
37
|
+
and any fault during the probe, falls through to the hold, which re-reads and
|
|
38
|
+
decides authoritatively. This is what keeps a no-op from failing on a lock it
|
|
39
|
+
never needed — an `OSError` from acquisition, or a
|
|
40
|
+
:class:`~froid_loop.runs.StateRootError` from deriving the sidecar path where no
|
|
41
|
+
state root exists — which a replayed rollback, a re-run sweep and
|
|
42
|
+
``sweep --archive`` all reach routinely.
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
from __future__ import annotations
|
|
46
|
+
|
|
47
|
+
import hashlib
|
|
48
|
+
import re
|
|
49
|
+
import threading
|
|
50
|
+
from bisect import bisect_right
|
|
51
|
+
from collections.abc import Iterator, Sequence
|
|
52
|
+
from contextlib import contextmanager
|
|
53
|
+
from dataclasses import dataclass
|
|
54
|
+
from datetime import date as calendar_date
|
|
55
|
+
from pathlib import Path
|
|
56
|
+
|
|
57
|
+
from . import sprintstatus
|
|
58
|
+
from .fences import fenced_spans
|
|
59
|
+
from .platform_util import atomic_write_text, file_lock, neutralize_surrogates
|
|
60
|
+
|
|
61
|
+
HEADING_RE = re.compile(r"^### (DW-\d+): (.+?)\s*$", re.MULTILINE)
|
|
62
|
+
# Where a canonical entry ENDS, in every shape CommonMark spells an ATX heading:
|
|
63
|
+
# up to three spaces of indent, a space OR a tab after the hashes, and an empty
|
|
64
|
+
# heading (`##` alone — the separator may be the end of the line). A fourth space
|
|
65
|
+
# of indent is an indented code block rather than a heading, so those lines
|
|
66
|
+
# deliberately keep absorbing, and the indent class is spaces only for that same
|
|
67
|
+
# reason — a leading tab is four columns, so `\t## Notes` is a code block too.
|
|
68
|
+
# Read as permissively as the syntax is, because a missed boundary here does not
|
|
69
|
+
# merely lose a section header: the span runs on and the next section's
|
|
70
|
+
# `status:`/`gate:` lines are read as this entry's, so an open entry that never
|
|
71
|
+
# declared a gate takes a story hostage and the operator finds no gate in the
|
|
72
|
+
# entry the refusal names (#516). Where a miss would WRITE, the strict
|
|
73
|
+
# column-zero reading is the right one (`devcontract`'s destructive edits pin it
|
|
74
|
+
# deliberately); this only decides how far a read reaches.
|
|
75
|
+
# A lookahead rather than a consuming group: `parse_ledger` reads `.start()`, and
|
|
76
|
+
# holding the match to the opener leaves nothing free to grow a dependency on
|
|
77
|
+
# where the separator ended.
|
|
78
|
+
ANY_HEADING_RE = re.compile(r"^ {0,3}#{1,6}(?=[ \t]|$)", re.MULTILINE)
|
|
79
|
+
# The flat appender's opening line, in the two forms this module needs it: as a
|
|
80
|
+
# bullet in the raw ledger (FLAT_ENTRY_RE, the canonical-span boundary in
|
|
81
|
+
# parse_ledger) and as bullet *content* after `_BULLET_RE` has stripped the
|
|
82
|
+
# marker (`_FLAT_SOURCE_RE`, legacy section). One shape, two anchors — they have
|
|
83
|
+
# to agree, or a block the legacy parser recognizes stays invisible to it (#304).
|
|
84
|
+
# Keyed on the opening line alone, deliberately: also requiring the block's
|
|
85
|
+
# `summary:`/`evidence:` lines would narrow the boundary below the parser's own
|
|
86
|
+
# recognition, leaving the bug in place for every partial shape it accepts.
|
|
87
|
+
_FLAT_SOURCE_BODY = r"source_spec:[ \t]"
|
|
88
|
+
FLAT_ENTRY_RE = re.compile(rf"^[-*][ \t]+{_FLAT_SOURCE_BODY}", re.IGNORECASE | re.MULTILINE)
|
|
89
|
+
STATUS_RE = re.compile(r"^status:[ \t]*(.*)$", re.MULTILINE)
|
|
90
|
+
# The mechanical half of a hard gate. An entry could always *say* it blocked a
|
|
91
|
+
# story — `HARD GATE: must land before 3-2` in the reason line — and saying it
|
|
92
|
+
# stopped nothing: the queue picked the story up anyway, and the gate surfaced
|
|
93
|
+
# afterwards in the diff of work built on a leg nobody had wired. `gate:` names
|
|
94
|
+
# the blocked story keys in a form a check can match, so the claim can refuse.
|
|
95
|
+
# Parsed exactly like `status:`: a field line, read inside `parse_ledger`'s
|
|
96
|
+
# canonical span, so a line under a flat-append bullet belongs to that block and
|
|
97
|
+
# not to the entry above it.
|
|
98
|
+
GATE_RE = re.compile(r"^gate:[ \t]*(.*)$", re.MULTILINE)
|
|
99
|
+
# A story key as either queue spells one: a sprint key (`3-2-invite-link`), the
|
|
100
|
+
# stories-mode id it starts with (`3-2`), or a bare slug. Whitespace and
|
|
101
|
+
# separators are deliberately out — a token nothing can match is the same silent
|
|
102
|
+
# no-op the field exists to end, so it is surfaced rather than dropped.
|
|
103
|
+
GATE_TOKEN_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._-]*$")
|
|
104
|
+
# The second half of "can this token gate anything", and a different miss from the
|
|
105
|
+
# one above: `GATE_TOKEN_RE` rejects the spellings a *line* cannot carry (a space,
|
|
106
|
+
# a bare separator), this rejects the ones no *key* can carry. `gate: 3.2` passes
|
|
107
|
+
# the first and can never match `gates_story` against any legal key, so it used to
|
|
108
|
+
# report a green `ok` while gating nothing — the field's own silent no-op, one
|
|
109
|
+
# keystroke away from the shape that works.
|
|
110
|
+
#
|
|
111
|
+
# Two arms, because FROID spells a story key two ways and they are NOT
|
|
112
|
+
# interchangeable. A stories-mode id is alphanumeric segments joined by single
|
|
113
|
+
# dashes (`_STORIES_ID_RE`), so `3.2` and `3_2` are out. A sprint key's slug is
|
|
114
|
+
# unconstrained (`sprintstatus.STORY_RE`'s trailing group), so `3-2-foo.bar` and
|
|
115
|
+
# `3-2-a_b` are LEGAL keys that gate correctly — which is why this is a
|
|
116
|
+
# whole-token shape test and not a ban on `.`/`_`. Only those characters in the
|
|
117
|
+
# *number* prefix are unmatchable; banning them outright would refuse real gates.
|
|
118
|
+
#
|
|
119
|
+
# Sound in the direction that matters: a token matching either arm is itself a
|
|
120
|
+
# legal key, so a story it could gate can exist. `gates_story`'s prefix and split
|
|
121
|
+
# arms only ever extend a key rightward past a `-`, and every such prefix of a
|
|
122
|
+
# legal key matches one of these arms too.
|
|
123
|
+
#
|
|
124
|
+
# `sprintstatus` is imported for its regex; `stories.ID_RE` is copied rather than
|
|
125
|
+
# imported because `stories` imports *this* module (a cycle). The copy is pinned
|
|
126
|
+
# to the original by a drift test rather than to a comment.
|
|
127
|
+
_STORIES_ID_RE = re.compile(r"^[A-Za-z0-9]+(-[A-Za-z0-9]+)*$")
|
|
128
|
+
# The tokens `gates_story`'s split arm may fire for: a bare `<epic>-<story>`, both
|
|
129
|
+
# numeric. `sprintstatus.STORY_RE` attaches the split letter straight after the
|
|
130
|
+
# story *number*, so that is the only token a split can extend. "Ends in a digit"
|
|
131
|
+
# is a weaker test that reads the same shape into a slug — `3-2-v2` would take the
|
|
132
|
+
# arm and refuse `3-2-v2a-followup`, a different and legal key.
|
|
133
|
+
_SPLITTABLE_TOKEN_RE = re.compile(r"^\d+-\d+$")
|
|
134
|
+
# A `gate:` line the strict field pattern above will never see. `GATE_RE` is
|
|
135
|
+
# anchored to a lowercase `gate:` in column 0, exactly like `status:`, and that
|
|
136
|
+
# strictness fails in opposite directions for the two fields: a missed `status:`
|
|
137
|
+
# leaves an entry unresolved, which now gates conservatively, while a missed
|
|
138
|
+
# `gate:` leaves no gate at all. `Gate: 3-2` and an indented ` gate: 3-2` are
|
|
139
|
+
# therefore surfaced as unenforceable rather than silently absent — and surfaced
|
|
140
|
+
# rather than *accepted*, because accepting an indented line would read a fenced
|
|
141
|
+
# example inside an entry as a live gate and refuse a story nobody meant to block.
|
|
142
|
+
_GATE_NEAR_RE = re.compile(r"^[ \t]*gate[ \t]*:", re.IGNORECASE | re.MULTILINE)
|
|
143
|
+
# The prose convention `gate:` replaces, matched anywhere on a line rather than
|
|
144
|
+
# at its start: real ledgers hard-wrap their `reason:` prose, so the declaration
|
|
145
|
+
# routinely lands mid-line and a line-anchored pattern misses exactly the entries
|
|
146
|
+
# that have one. The quote lookbehind is what keeps that from over-firing — an
|
|
147
|
+
# entry *citing* the phrase (`names a "HARD GATE: ..."`) is discussion, not a
|
|
148
|
+
# declaration — and the colon does the rest of the work, since a sentence about
|
|
149
|
+
# "this HARD GATE is textual only" never reaches the pattern at all.
|
|
150
|
+
# The class covers the backtick and the curly quotes as well as the ASCII pair:
|
|
151
|
+
# a ledger is markdown, so `HARD GATE:` is the citation form an author reaches for
|
|
152
|
+
# first, and an LLM-written entry curls its quotes. Missing them made the warning
|
|
153
|
+
# fire on entries documenting the convention — including this repo's own docs.
|
|
154
|
+
# The lookbehind only reaches an *inline* citation, though; the block form of the
|
|
155
|
+
# same quoting is a fence, and no character precedes a line inside one. Callers
|
|
156
|
+
# read this through `declares_prose_gate`, which masks those out.
|
|
157
|
+
HARD_GATE_PROSE_RE = re.compile(r"""(?<!["'`«“”‘’])HARD GATE:""")
|
|
158
|
+
# Everything `str.splitlines()` splits on, not `\n` alone (#305). The writers
|
|
159
|
+
# below interpolate their arguments into a line-oriented file, so a break in a
|
|
160
|
+
# value injects ledger lines. The C1/Unicode members are load-bearing rather
|
|
161
|
+
# than decorative: `parse_legacy` scans with `splitlines()` while `parse_ledger`
|
|
162
|
+
# matches with `re.MULTILINE`, so a U+2028 splits an entry for one reader and is
|
|
163
|
+
# invisible to the other — the two then disagree about what the ledger says.
|
|
164
|
+
LINE_BREAK_RE = re.compile(r"[\n\r\v\f\x1c-\x1e\x85\u2028\u2029]+")
|
|
165
|
+
# The writers' date shape. Deliberately a separate literal from the legacy
|
|
166
|
+
# parser's `_DATE_TOKEN_RE`, which happens to look similar today: that one
|
|
167
|
+
# decides whether a freeform heading is a dated section, and tightening what the
|
|
168
|
+
# orchestrator will *write* must never quietly retune what `parse_legacy` reads.
|
|
169
|
+
# Spelled `[0-9]` rather than `\d`, which also matches Arabic-Indic, fullwidth
|
|
170
|
+
# and mathematical digit forms — the ledger's readers understand none of them.
|
|
171
|
+
_ISO_DATE_RE = re.compile(r"[0-9]{4}-[0-9]{2}-[0-9]{2}")
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
@dataclass(frozen=True)
|
|
175
|
+
class DWEntry:
|
|
176
|
+
id: str
|
|
177
|
+
title: str
|
|
178
|
+
status: str # the status field value, "" when the line is missing
|
|
179
|
+
body: str # full entry text including the heading
|
|
180
|
+
span: tuple[int, int] # char offsets of the entry in the ledger text
|
|
181
|
+
# Body-relative offsets of the line `status` was read from; None when the
|
|
182
|
+
# entry has no status line. Carried rather than re-derived because the reader
|
|
183
|
+
# picks the status with a fence-aware lookup at file scope, and a writer that
|
|
184
|
+
# ran `STATUS_RE.search(body)` again would pick the *first* raw match instead.
|
|
185
|
+
# Those differ exactly when an entry quotes an example above its live status:
|
|
186
|
+
# the writer rewrote the quoted line, the reader kept reporting the real
|
|
187
|
+
# `status: open`, and the close reported success while the entry — and any
|
|
188
|
+
# `gate:` it carries — stayed open forever. No default: `parse_ledger` is the
|
|
189
|
+
# only constructor, and a fallback here would silently restore that split.
|
|
190
|
+
status_span: tuple[int, int] | None
|
|
191
|
+
# The whole-file fence index the entry was carved with, so the gate scans can
|
|
192
|
+
# ask the question the heading and status reads already ask at file scope. A
|
|
193
|
+
# body slice cannot see a fence opened above the heading, and the two views
|
|
194
|
+
# disagree: under a stray unclosed ``` above the heading, whole-file scope
|
|
195
|
+
# treats the opener as text (so this entry EXISTS) while the body sees a later
|
|
196
|
+
# matched `~~~` pair as a real fence and reads a live `gate:` as an example.
|
|
197
|
+
# That direction loses a gate in silence, which is what the field exists to
|
|
198
|
+
# end. No default, for `status_span`'s reason: the fallback IS the bug.
|
|
199
|
+
examples: _Examples
|
|
200
|
+
|
|
201
|
+
@property
|
|
202
|
+
def open(self) -> bool:
|
|
203
|
+
return self.status.split()[0] == "open" if self.status else False
|
|
204
|
+
|
|
205
|
+
@property
|
|
206
|
+
def done(self) -> bool:
|
|
207
|
+
"""Whether the entry has landed.
|
|
208
|
+
|
|
209
|
+
Deliberately NOT ``not open``. A status line the format does not
|
|
210
|
+
understand — ``status: opne``, or no status line at all — is neither open
|
|
211
|
+
nor done, and the readers that ask want *opposite* answers about it:
|
|
212
|
+
:func:`open_ids` drops it (it may already be finished), while a gate on it
|
|
213
|
+
has to hold (it may not be). Deriving one from the other is what let
|
|
214
|
+
``gate:`` fail open on a one-character typo — the entry read as closed, so
|
|
215
|
+
the gate was skipped and ``validate`` reported an all-clear naming it.
|
|
216
|
+
"""
|
|
217
|
+
return self.status.split()[0] == "done" if self.status else False
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
@dataclass(frozen=True)
|
|
221
|
+
class _Examples:
|
|
222
|
+
"""The ledger's fenced worked examples, indexed for repeated offset queries.
|
|
223
|
+
|
|
224
|
+
``fenced_spans`` returns its ranges in increasing order and non-overlapping (a
|
|
225
|
+
fence cannot open inside an open one), so a query is a binary search for the
|
|
226
|
+
last span starting at or before the offset. Kept as an index rather than a bare
|
|
227
|
+
list because both scales are in play at once: a parse asks once per heading and
|
|
228
|
+
several times per entry, so a linear membership test would leave the parse
|
|
229
|
+
quadratic whenever a ledger's examples grow with its entries.
|
|
230
|
+
"""
|
|
231
|
+
|
|
232
|
+
spans: tuple[tuple[int, int], ...]
|
|
233
|
+
starts: tuple[int, ...]
|
|
234
|
+
|
|
235
|
+
def covers(self, offset: int) -> bool:
|
|
236
|
+
i = bisect_right(self.starts, offset)
|
|
237
|
+
return i > 0 and offset < self.spans[i - 1][1]
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def _example_spans(text: str) -> _Examples:
|
|
241
|
+
"""The ledger's fenced worked examples, as offset ranges.
|
|
242
|
+
|
|
243
|
+
Read at WHOLE-FILE scope, which is the whole point. A fence that opens above
|
|
244
|
+
a quoted ``### DW-n:`` heading is stranded in the *previous* entry once spans
|
|
245
|
+
are carved, so an entry-local query reads the example as live — a phantom
|
|
246
|
+
entry whose ``gate:`` refuses a story nobody deferred. `deferred-work-format.md`
|
|
247
|
+
ships exactly that shape (a complete entry inside a ```markdown fence), so
|
|
248
|
+
quoting it into a ledger is the expected trigger, not a corner case.
|
|
249
|
+
|
|
250
|
+
``unclosed_hides_rest=False`` repeats the answer `gates()` gives one level
|
|
251
|
+
down, and here for a stronger reason: under ``True`` a single stray opener
|
|
252
|
+
would erase every heading below it, dropping real open work out of
|
|
253
|
+
``open_ids()`` in silence. A phantom entry from an unterminated fence is
|
|
254
|
+
today's behaviour and is visible; a vanished ledger is neither.
|
|
255
|
+
|
|
256
|
+
Walked once per :func:`parse_ledger` and passed down to the offset checks. The
|
|
257
|
+
walk covers the whole file, so recomputing it per offset made the parse
|
|
258
|
+
quadratic in the number of entries — and `Engine._refuse_gated_story` re-parses
|
|
259
|
+
before every story dispatch, so a mature ledger paid it on the dispatch path.
|
|
260
|
+
"""
|
|
261
|
+
spans = tuple(fenced_spans(text, unclosed_hides_rest=False))
|
|
262
|
+
return _Examples(spans=spans, starts=tuple(s for s, _ in spans))
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
def _example(examples: _Examples, offset: int) -> bool:
|
|
266
|
+
"""Whether ``offset`` sits in a fenced worked example rather than the ledger.
|
|
267
|
+
|
|
268
|
+
Takes the index rather than the text: the answer must come from the same
|
|
269
|
+
whole-file walk for every offset in one parse, and a signature that re-derived
|
|
270
|
+
it per call is what made that expensive enough to matter.
|
|
271
|
+
"""
|
|
272
|
+
return examples.covers(offset)
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def _unfenced(
|
|
276
|
+
pattern: re.Pattern[str],
|
|
277
|
+
text: str,
|
|
278
|
+
start: int,
|
|
279
|
+
end: int,
|
|
280
|
+
examples: _Examples,
|
|
281
|
+
) -> re.Match[str] | None:
|
|
282
|
+
"""First match of ``pattern`` within ``text[start:end]`` that is not quoted.
|
|
283
|
+
|
|
284
|
+
Not `search()` plus a check: the first match may be the quoted one, and the
|
|
285
|
+
real boundary sits after it. Bounded by ``endpos`` so a match beyond the span
|
|
286
|
+
cannot claim it, while ``examples`` still describes fence state from offset 0.
|
|
287
|
+
"""
|
|
288
|
+
for m in pattern.finditer(text, start, end):
|
|
289
|
+
if not _example(examples, m.start()):
|
|
290
|
+
return m
|
|
291
|
+
return None
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
def parse_ledger(text: str) -> list[DWEntry]:
|
|
295
|
+
"""Extract DW entries; non-conforming sections are skipped, an entry
|
|
296
|
+
without a status line parses with status "" (not open).
|
|
297
|
+
|
|
298
|
+
Fenced matches are skipped by every scan below, not just the heading one: a
|
|
299
|
+
heading or flat bullet quoted inside an example must not start an entry, end
|
|
300
|
+
one, or bound a block out of one. Filtering only the headings would trade the
|
|
301
|
+
phantom entry for a truncation — a fenced ``## heading`` would still cut a
|
|
302
|
+
real entry short at its own boundary, and a `gate:` line below the example
|
|
303
|
+
would fall outside the span and stop gating, which is the failure this field
|
|
304
|
+
exists to end.
|
|
305
|
+
"""
|
|
306
|
+
entries = []
|
|
307
|
+
examples = _example_spans(text)
|
|
308
|
+
headings = [m for m in HEADING_RE.finditer(text) if not _example(examples, m.start())]
|
|
309
|
+
for i, m in enumerate(headings):
|
|
310
|
+
end = headings[i + 1].start() if i + 1 < len(headings) else len(text)
|
|
311
|
+
# an entry also ends at any intervening heading (e.g. a "## Deferred
|
|
312
|
+
# from:" section header between freeform and DW-format content)
|
|
313
|
+
other = _unfenced(ANY_HEADING_RE, text, m.end(), end, examples)
|
|
314
|
+
if other:
|
|
315
|
+
end = other.start()
|
|
316
|
+
# ...and at a flat appender block, which belongs to no canonical entry
|
|
317
|
+
# (#304). This span is what parse_legacy() masks out before scanning, so
|
|
318
|
+
# absorbing the block hides the finding from every reader of the ledger.
|
|
319
|
+
# Searched from the entry's own `status:` line, never from above it:
|
|
320
|
+
# truncating over the status leaves the entry reading as neither open nor
|
|
321
|
+
# done (open_ids() drops it, classify() calls it malformed), which trades
|
|
322
|
+
# one lost flat block for one lost tracked entry. An entry with no status
|
|
323
|
+
# line has nothing to protect, so the whole span is fair game.
|
|
324
|
+
status_m = _unfenced(STATUS_RE, text, m.end(), end, examples)
|
|
325
|
+
flat = _unfenced(
|
|
326
|
+
FLAT_ENTRY_RE, text, status_m.end() if status_m else m.end(), end, examples
|
|
327
|
+
)
|
|
328
|
+
if flat:
|
|
329
|
+
end = flat.start()
|
|
330
|
+
body = text[m.start() : end]
|
|
331
|
+
# Re-read rather than reuse the probe above: `end` may have moved, and the
|
|
332
|
+
# status must be the one inside the final span. Searched over `text` at
|
|
333
|
+
# absolute offsets because `_example` reads fence state from the top of the
|
|
334
|
+
# file — a body slice cannot see an opener that sits above the heading.
|
|
335
|
+
status_m = _unfenced(STATUS_RE, text, m.start(), end, examples)
|
|
336
|
+
entries.append(
|
|
337
|
+
DWEntry(
|
|
338
|
+
id=m.group(1),
|
|
339
|
+
title=m.group(2),
|
|
340
|
+
status=status_m.group(1).strip() if status_m else "",
|
|
341
|
+
body=body,
|
|
342
|
+
span=(m.start(), end),
|
|
343
|
+
status_span=(
|
|
344
|
+
(status_m.start() - m.start(), status_m.end() - m.start()) if status_m else None
|
|
345
|
+
),
|
|
346
|
+
examples=examples,
|
|
347
|
+
)
|
|
348
|
+
)
|
|
349
|
+
return entries
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
def open_ids(text: str) -> set[str]:
|
|
353
|
+
return {e.id for e in parse_ledger(text) if e.open}
|
|
354
|
+
|
|
355
|
+
|
|
356
|
+
@dataclass(frozen=True)
|
|
357
|
+
class EntryGates:
|
|
358
|
+
"""One entry's ``gate:`` declaration, split by what a check can act on.
|
|
359
|
+
|
|
360
|
+
Every shape that is not an enforceable token is reported by ``validate``,
|
|
361
|
+
because none of them is a *weaker* gate than a valid one — each is the prose
|
|
362
|
+
gate again wearing the field's clothes, and silence about it is what let the
|
|
363
|
+
story run. ``lines`` is what distinguishes "declared nothing usable" from
|
|
364
|
+
"declared nothing at all": an entry with no ``gate:`` line has made no claim,
|
|
365
|
+
while ``gate:`` with an empty value has made one and inertly.
|
|
366
|
+
|
|
367
|
+
``empty`` counts those inert lines individually rather than folding them into
|
|
368
|
+
an entry-wide verdict, because the two coexist: ``gate: 3-2`` followed by a
|
|
369
|
+
bare ``gate:`` has both a gate in force and a line that names nothing, and an
|
|
370
|
+
aggregate answer can only report one of them. Reporting the tokens and
|
|
371
|
+
swallowing the empty line is the worse half to lose — the operator who wrote
|
|
372
|
+
it believes a second story is held back.
|
|
373
|
+
"""
|
|
374
|
+
|
|
375
|
+
tokens: tuple[str, ...] = ()
|
|
376
|
+
malformed: tuple[str, ...] = ()
|
|
377
|
+
lines: int = 0
|
|
378
|
+
empty: int = 0
|
|
379
|
+
near_miss: int = 0
|
|
380
|
+
|
|
381
|
+
@property
|
|
382
|
+
def inert(self) -> bool:
|
|
383
|
+
"""Every ``gate:`` line named nothing — ``gate:`` or ``gate: ,`` and no others."""
|
|
384
|
+
return self.lines > 0 and not self.tokens and not self.malformed
|
|
385
|
+
|
|
386
|
+
|
|
387
|
+
def _quoted(entry: DWEntry, offset: int) -> bool:
|
|
388
|
+
"""Whether a BODY-relative ``offset`` sits in a fenced example.
|
|
389
|
+
|
|
390
|
+
The single rule every gate scan in this module reads through, so that a fence
|
|
391
|
+
means the same thing to all of them: an entry documenting the field quotes it,
|
|
392
|
+
and a quoted example is not a declaration. Sharing it is the point — the prose
|
|
393
|
+
scan was left on the raw body once, on the reasoning that a warning is cheap
|
|
394
|
+
and its quote lookbehind was guard enough. It is not: that lookbehind reaches
|
|
395
|
+
an inline citation only, so an entry explaining the old convention in a fenced
|
|
396
|
+
block was told to convert a gate it was not declaring.
|
|
397
|
+
|
|
398
|
+
Asked at FILE scope, like the heading and status reads in :func:`parse_ledger`
|
|
399
|
+
and for the same reason: a body slice cannot see a fence opened above the
|
|
400
|
+
heading, so the two views can disagree about the same line. They disagree in
|
|
401
|
+
the direction that matters — a stray unclosed ``` above the heading leaves the
|
|
402
|
+
entry standing at file scope while the body reads a later matched ``~~~`` pair
|
|
403
|
+
as a real fence, masking a live ``gate:`` into an example. A gate lost in
|
|
404
|
+
silence is the failure this field exists to end; a spurious refusal in an entry
|
|
405
|
+
whose markdown is already malformed is the cheaper wrong answer.
|
|
406
|
+
"""
|
|
407
|
+
return entry.examples.covers(entry.span[0] + offset)
|
|
408
|
+
|
|
409
|
+
|
|
410
|
+
def declares_prose_gate(entry: DWEntry) -> bool:
|
|
411
|
+
"""Whether the entry declares a gate in the pre-``gate:`` prose convention.
|
|
412
|
+
|
|
413
|
+
:data:`HARD_GATE_PROSE_RE` filtered the way every other gate scan here is
|
|
414
|
+
filtered. Lives beside them rather than at the caller so the fence rule has
|
|
415
|
+
one implementation: ``validate`` is the only reader today, and a second one
|
|
416
|
+
reaching for the bare pattern would reintroduce exactly the half-applied rule
|
|
417
|
+
this replaced.
|
|
418
|
+
"""
|
|
419
|
+
return any(not _quoted(entry, m.start()) for m in HARD_GATE_PROSE_RE.finditer(entry.body))
|
|
420
|
+
|
|
421
|
+
|
|
422
|
+
def gates(entry: DWEntry) -> EntryGates:
|
|
423
|
+
"""Every ``gate:`` token in one entry's canonical span, order-preserving.
|
|
424
|
+
|
|
425
|
+
Multiple ``gate:`` lines union: an entry blocking three stories may list them
|
|
426
|
+
on one line or on three, and to a line-oriented file neither spelling is the
|
|
427
|
+
wrong one. Within a line the separator is a comma, and only a comma — a
|
|
428
|
+
space-separated ``gate: 3-2 3-3`` lands in ``malformed`` rather than being
|
|
429
|
+
read leniently, so the operator is told the spelling gated nothing instead of
|
|
430
|
+
finding out from a story that ran.
|
|
431
|
+
|
|
432
|
+
Duplicates collapse (an id repeated across lines is one claim, not two);
|
|
433
|
+
empty items drop, so a trailing separator is not a token — but the *line* is
|
|
434
|
+
still counted, which is how an all-empty declaration stays reportable.
|
|
435
|
+
|
|
436
|
+
``near_miss`` counts the lines this function deliberately did NOT read as a
|
|
437
|
+
declaration: a `gate:` the strict field anchor misses (see
|
|
438
|
+
:data:`_GATE_NEAR_RE`). They are counted rather than parsed so the operator is
|
|
439
|
+
told the spelling gated nothing — the same trade the space-separated token
|
|
440
|
+
makes, one level up.
|
|
441
|
+
"""
|
|
442
|
+
tokens: list[str] = []
|
|
443
|
+
malformed: list[str] = []
|
|
444
|
+
lines = 0
|
|
445
|
+
empty = 0
|
|
446
|
+
|
|
447
|
+
# Both scans below skip fenced matches: an entry documenting this field quotes
|
|
448
|
+
# it, and a quoted example is not a declaration — a fenced `gate: 3-2` sits in
|
|
449
|
+
# column 0, right where the anchor looks, and the answer here is a *refusal*.
|
|
450
|
+
for m in GATE_RE.finditer(entry.body):
|
|
451
|
+
if _quoted(entry, m.start()):
|
|
452
|
+
continue
|
|
453
|
+
lines += 1
|
|
454
|
+
named = False
|
|
455
|
+
for raw in m.group(1).split(","):
|
|
456
|
+
token = raw.strip()
|
|
457
|
+
if not token:
|
|
458
|
+
continue
|
|
459
|
+
named = True
|
|
460
|
+
bucket = tokens if _matchable_token(token) else malformed
|
|
461
|
+
if token not in bucket:
|
|
462
|
+
bucket.append(token)
|
|
463
|
+
if not named:
|
|
464
|
+
empty += 1
|
|
465
|
+
near_miss = sum(
|
|
466
|
+
# `^` puts every match at a line start, so this asks whether the same line
|
|
467
|
+
# would have satisfied `GATE_RE` — i.e. whether it is the canonical spelling
|
|
468
|
+
# already counted above — without re-running the anchor against a slice.
|
|
469
|
+
not entry.body.startswith("gate:", m.start())
|
|
470
|
+
for m in _GATE_NEAR_RE.finditer(entry.body)
|
|
471
|
+
if not _quoted(entry, m.start())
|
|
472
|
+
)
|
|
473
|
+
return EntryGates(
|
|
474
|
+
tokens=tuple(tokens),
|
|
475
|
+
malformed=tuple(malformed),
|
|
476
|
+
lines=lines,
|
|
477
|
+
empty=empty,
|
|
478
|
+
near_miss=near_miss,
|
|
479
|
+
)
|
|
480
|
+
|
|
481
|
+
|
|
482
|
+
def _matchable_token(token: str) -> bool:
|
|
483
|
+
"""Whether ``token`` could gate any legal story key — the test that decides
|
|
484
|
+
:attr:`EntryGates.tokens` vs :attr:`EntryGates.malformed`.
|
|
485
|
+
|
|
486
|
+
Both halves are required and neither implies the other: ``GATE_TOKEN_RE``
|
|
487
|
+
alone admits ``3.2``, which nothing can match, and the key shapes alone admit
|
|
488
|
+
``3-2 3-3`` via the sprint slug, which is one token pretending to be two.
|
|
489
|
+
"""
|
|
490
|
+
if not GATE_TOKEN_RE.match(token):
|
|
491
|
+
return False
|
|
492
|
+
return bool(_STORIES_ID_RE.match(token) or sprintstatus.STORY_RE.match(token))
|
|
493
|
+
|
|
494
|
+
|
|
495
|
+
def gates_story(token: str, story_key: str) -> bool:
|
|
496
|
+
"""Whether ``token`` gates ``story_key``: equal, or its prefix at a key boundary.
|
|
497
|
+
|
|
498
|
+
The prefix arm is what lets one token reach both queues — stories mode keys on
|
|
499
|
+
the bare id (``3-2``) while sprint mode keys on the full ``3-2-invite-link``,
|
|
500
|
+
and an author gating "story 3-2" means the story, not the spelling. The
|
|
501
|
+
boundary is required rather than a bare ``startswith`` so ``3-2`` cannot sweep
|
|
502
|
+
in its numeric neighbours: ``3-20-later`` is a different story.
|
|
503
|
+
|
|
504
|
+
Two boundaries count, because FROID spells a story key two ways. The plain one
|
|
505
|
+
is ``-``. The other is a **split**: ``sprintstatus.STORY_RE`` lets an oversized
|
|
506
|
+
story become ``3-2a-...`` / ``3-2b-...`` at breakdown time, and a token that
|
|
507
|
+
only knew ``-`` would lose its gate the moment the gated story was split —
|
|
508
|
+
silently, which is the worst thing a gate can do. One lowercase ASCII letter
|
|
509
|
+
followed by ``-`` is therefore also a boundary. Exactly one letter, and the
|
|
510
|
+
``-`` after it is required, so ``3-2ab-x`` and a bare ``3-2a`` are not swept in.
|
|
511
|
+
|
|
512
|
+
The split arm applies only to a token that *is* a bare ``<epic>-<story>``,
|
|
513
|
+
because that is the only place a split letter can attach: ``STORY_RE`` puts it
|
|
514
|
+
straight after the story *number*. Without that guard the arm reads any
|
|
515
|
+
trailing letter as a split and gates a story nobody named — ``stories.ID_RE``
|
|
516
|
+
admits word ids, so ``gate: auth`` refused ``authz-login``, and a hard failure
|
|
517
|
+
on an unrelated story is the one way this check can be worse than the prose it
|
|
518
|
+
replaced. "Ends in a digit" is the same guard written too loosely: the digit
|
|
519
|
+
can belong to a *slug*, so ``gate: 3-2-v2`` took the arm and refused
|
|
520
|
+
``3-2-v2a-followup`` — a different, legal key — and ``gate: 3`` refused the
|
|
521
|
+
distinct stories id ``3a-task``.
|
|
522
|
+
"""
|
|
523
|
+
if story_key == token or story_key.startswith(f"{token}-"):
|
|
524
|
+
return True
|
|
525
|
+
# The `startswith` guard is load-bearing, not redundant with the slice below:
|
|
526
|
+
# `story_key[len(token):]` says nothing about what preceded it, so without it
|
|
527
|
+
# `3-2` would gate `9-9a-x` on the tail alone.
|
|
528
|
+
if not story_key.startswith(token) or not _SPLITTABLE_TOKEN_RE.match(token):
|
|
529
|
+
return False
|
|
530
|
+
rest = story_key[len(token) :]
|
|
531
|
+
return len(rest) >= 2 and "a" <= rest[0] <= "z" and rest[1] == "-"
|
|
532
|
+
|
|
533
|
+
|
|
534
|
+
def parse_declaration(raw: object) -> tuple[tuple[str, ...], str | None]:
|
|
535
|
+
"""The single reading of a ``closes_deferred:`` declaration (#234), shared by
|
|
536
|
+
the ``stories.yaml`` parser, the engine's close hook, and ``validate``.
|
|
537
|
+
|
|
538
|
+
Returns the normalized ids plus an error describing a wrong *container*.
|
|
539
|
+
Missing / YAML-null is an empty declaration, not an error.
|
|
540
|
+
|
|
541
|
+
Strict about the container, lenient about each item. A bare
|
|
542
|
+
``closes_deferred: DW-1`` is a schema error rather than a silently-wrapped
|
|
543
|
+
single id — a string is iterable, so a lenient reading would quietly turn one
|
|
544
|
+
id into a list of characters — while items are ``str()``-normalized and
|
|
545
|
+
stripped, because an LLM-authored manifest may emit an unquoted ``DW-1`` as a
|
|
546
|
+
string but a bare ``5`` as an int. Blanks drop and duplicates collapse
|
|
547
|
+
(order-preserving): both are noise, not a contradiction.
|
|
548
|
+
|
|
549
|
+
Callers decide the severity: the manifest parser raises, the engine journals,
|
|
550
|
+
``validate`` warns. What they must NOT do is disagree — before this, a wrong
|
|
551
|
+
container was a hard schema error in ``stories.yaml`` and a silent empty
|
|
552
|
+
declaration in frontmatter, so the same mistake either failed the parse or
|
|
553
|
+
vanished depending on which file it was made in.
|
|
554
|
+
|
|
555
|
+
Whether an id names a real entry is not decided here; that needs the ledger
|
|
556
|
+
(:func:`classify`).
|
|
557
|
+
"""
|
|
558
|
+
if raw is None:
|
|
559
|
+
return (), None
|
|
560
|
+
if not isinstance(raw, list):
|
|
561
|
+
return (), f"must be a list of deferred-work ids (got {type(raw).__name__})"
|
|
562
|
+
return tuple(dict.fromkeys(item for item in (str(x).strip() for x in raw) if item)), None
|
|
563
|
+
|
|
564
|
+
|
|
565
|
+
@dataclass(frozen=True)
|
|
566
|
+
class Declared:
|
|
567
|
+
"""How declared ids line up against one ledger snapshot (#234).
|
|
568
|
+
|
|
569
|
+
Four outcomes, not two, because "not open" hides two very different cases.
|
|
570
|
+
``already_done`` is a satisfied declaration — a resume re-driving a close that
|
|
571
|
+
already landed — and must stay silent. ``malformed`` is an entry that exists
|
|
572
|
+
but carries neither an ``open`` nor a ``done`` status: nothing can be marked,
|
|
573
|
+
and saying nothing would leave the operator believing it was.
|
|
574
|
+
|
|
575
|
+
``duplicates`` cross-cuts the other four: it names the declared ids the ledger
|
|
576
|
+
carries more than once, whichever bucket they landed in. A duplicate id is a
|
|
577
|
+
corrupt ledger (#286), and the entry this classification describes is only one
|
|
578
|
+
of them — so the close is reported, never silent.
|
|
579
|
+
"""
|
|
580
|
+
|
|
581
|
+
open_ids: tuple[str, ...] = ()
|
|
582
|
+
already_done: tuple[str, ...] = ()
|
|
583
|
+
unknown: tuple[str, ...] = ()
|
|
584
|
+
malformed: tuple[str, ...] = ()
|
|
585
|
+
duplicates: tuple[str, ...] = ()
|
|
586
|
+
|
|
587
|
+
|
|
588
|
+
def classify(text: str, ids: Sequence[str]) -> Declared:
|
|
589
|
+
"""Partition `ids` against a single ledger snapshot, preserving order.
|
|
590
|
+
|
|
591
|
+
Classifying from a snapshot rather than from :func:`mark_done`'s return value
|
|
592
|
+
is deliberate: that return conflates "already done" with "absent from the
|
|
593
|
+
ledger", and those need opposite treatment (silence vs. a warning).
|
|
594
|
+
|
|
595
|
+
**The FIRST entry of a duplicated id wins**, because that is the one
|
|
596
|
+
:func:`_find_entry` — and so every mutation in this module — acts on. Indexing
|
|
597
|
+
last-wins instead made the two disagree, and a ledger carrying one `DW-1` open
|
|
598
|
+
and another done then closed nothing while saying nothing, in either order: a
|
|
599
|
+
done-first ledger classified the id `open`, sent it to
|
|
600
|
+
:func:`mark_done_many`, and had :func:`_apply_done` refuse the done copy it
|
|
601
|
+
found first (marked nothing, so not even an unmatched warning); an open-first
|
|
602
|
+
ledger classified it `already_done` and never attempted the write at all
|
|
603
|
+
(#284 round-6 review, finding 4). The duplicate itself is reported through
|
|
604
|
+
``duplicates`` rather than swallowed — one id naming two entries is a fault
|
|
605
|
+
about the ledger, not an answer about the work."""
|
|
606
|
+
by_id: dict[str, DWEntry] = {}
|
|
607
|
+
duplicated: set[str] = set()
|
|
608
|
+
for e in parse_ledger(text):
|
|
609
|
+
if e.id in by_id:
|
|
610
|
+
duplicated.add(e.id)
|
|
611
|
+
continue # first wins: `_find_entry` mutates that one
|
|
612
|
+
by_id[e.id] = e
|
|
613
|
+
buckets: dict[str, list[str]] = {"open": [], "done": [], "unknown": [], "malformed": []}
|
|
614
|
+
for dw_id in ids:
|
|
615
|
+
entry = by_id.get(dw_id)
|
|
616
|
+
if entry is None:
|
|
617
|
+
buckets["unknown"].append(dw_id)
|
|
618
|
+
continue
|
|
619
|
+
word = entry.status.split()[0] if entry.status else ""
|
|
620
|
+
buckets[word if word in ("open", "done") else "malformed"].append(dw_id)
|
|
621
|
+
return Declared(
|
|
622
|
+
open_ids=tuple(buckets["open"]),
|
|
623
|
+
already_done=tuple(buckets["done"]),
|
|
624
|
+
unknown=tuple(buckets["unknown"]),
|
|
625
|
+
malformed=tuple(buckets["malformed"]),
|
|
626
|
+
duplicates=tuple(dw_id for dw_id in dict.fromkeys(ids) if dw_id in duplicated),
|
|
627
|
+
)
|
|
628
|
+
|
|
629
|
+
|
|
630
|
+
def _find_entry(text: str, dw_id: str) -> DWEntry | None:
|
|
631
|
+
for entry in parse_ledger(text):
|
|
632
|
+
if entry.id == dw_id:
|
|
633
|
+
return entry
|
|
634
|
+
return None
|
|
635
|
+
|
|
636
|
+
|
|
637
|
+
def _insert_after_status(text: str, entry: DWEntry, line: str) -> str:
|
|
638
|
+
"""Insert a field line right after the entry's status line (or at the end
|
|
639
|
+
of the entry when no status line exists)."""
|
|
640
|
+
if entry.status_span:
|
|
641
|
+
pos = entry.span[0] + entry.status_span[1]
|
|
642
|
+
return text[:pos] + "\n" + line + text[pos:]
|
|
643
|
+
insert_at = entry.span[0] + len(entry.body.rstrip())
|
|
644
|
+
return text[:insert_at] + "\n" + line + text[insert_at:]
|
|
645
|
+
|
|
646
|
+
|
|
647
|
+
def _one_line(value: str) -> str:
|
|
648
|
+
"""Collapse every run of line-break characters in `value` to a single space.
|
|
649
|
+
|
|
650
|
+
The whole of the #305 fix. These writers interpolate their arguments into a
|
|
651
|
+
line-oriented file, so a value carrying a break mints a phantom
|
|
652
|
+
`### DW-<n>` entry, truncates the entry's span at :data:`FLAT_ENTRY_RE` and
|
|
653
|
+
re-surfaces the tail as a legacy item, or leaves the entry carrying two
|
|
654
|
+
`status:` lines.
|
|
655
|
+
|
|
656
|
+
Note what the last shape does *not* do: `STATUS_RE` takes the first match, so
|
|
657
|
+
an injected `status:` never changes what `parse_ledger` reports. A test that
|
|
658
|
+
asserts on `entry.status` therefore passes with this guard deleted — the
|
|
659
|
+
observable is the line structure.
|
|
660
|
+
|
|
661
|
+
Sanitizes; never raises, and nothing upstream rejects on a break either. The
|
|
662
|
+
close paths call these writers bare (`sweep._close_resolved`,
|
|
663
|
+
`decisions.apply_pre_answer`), so a `ValueError` would end the sweep as
|
|
664
|
+
crashed; refusing the same text back at `validate_triage` only moved the
|
|
665
|
+
stoppage to a pause. Collapsing is lossless enough — the ledger wants one
|
|
666
|
+
line anyway — so this is the fix, and the skill docs are guidance that
|
|
667
|
+
reduces occurrences without gating on them.
|
|
668
|
+
|
|
669
|
+
That contract covers one hazard more than the break collapse alone, which is
|
|
670
|
+
what the `neutralize_surrogates` pass in front of it buys (#329). A lone
|
|
671
|
+
surrogate is not a line break, so it sailed through untouched — but it has
|
|
672
|
+
no UTF-8 encoding, and `atomic_write_text`'s strict encode raises
|
|
673
|
+
`UnicodeEncodeError` (a `ValueError` subclass) on it, from inside those same
|
|
674
|
+
bare close-path calls. It arrives the way the break did: a triage
|
|
675
|
+
`result.json` is cached with `json.dumps`, whose `ensure_ascii` keeps the
|
|
676
|
+
code point a harmless `\\ud800` escape, and the reload's `json.loads` revives
|
|
677
|
+
the real thing into `ResolvedEntry.evidence` and on into the `mark_done`
|
|
678
|
+
note. Refusing it upstream would only move the stoppage again — same
|
|
679
|
+
doctrine, same answer.
|
|
680
|
+
|
|
681
|
+
A value with neither a break nor a surrogate is returned **untouched**, so an
|
|
682
|
+
existing ledger is never reformatted and a clean write is byte-identical to
|
|
683
|
+
before the guard; each pass keeps its own fast path, so the common value is
|
|
684
|
+
scanned twice and copied never. The trailing `.strip()` removes all
|
|
685
|
+
surrounding whitespace, not merely the space a leading or trailing break left
|
|
686
|
+
behind — which is why it must stay on the far side of that fast path.
|
|
687
|
+
|
|
688
|
+
A break-only value therefore sanitizes to `""`. Keeping it non-empty *here*
|
|
689
|
+
could only yield bare whitespace, which trades an unfindable entry for an
|
|
690
|
+
unidentifiable one, so each caller handles its own empties — and by two
|
|
691
|
+
different strategies, which is why neither belongs in this helper.
|
|
692
|
+
:func:`append_entry` **substitutes**, naming a vanished title
|
|
693
|
+
`(untitled DW-<n>)` so the id it just burned stays findable.
|
|
694
|
+
:func:`append_decision` **drops**, shedding the ` — ` separator along with an
|
|
695
|
+
empty detail rather than promising one that is not there. Its `label` needs
|
|
696
|
+
neither: every member of :data:`LINE_BREAK_RE` is `str.isspace()`, and
|
|
697
|
+
`validate_triage` builds each `DecisionOption` with `.strip() or key`, so a
|
|
698
|
+
break-only label has already become the option key before it arrives.
|
|
699
|
+
|
|
700
|
+
A surrogate-only value, by contrast, sanitizes to a truthy `"�"`, so neither
|
|
701
|
+
caller's empty-handling fires for it. That is the point of replacing rather
|
|
702
|
+
than stripping: a title reading `�` still says *something unencodable was
|
|
703
|
+
here*, where a vanished one would silently become `(untitled DW-<n>)`."""
|
|
704
|
+
value = neutralize_surrogates(value)
|
|
705
|
+
if not LINE_BREAK_RE.search(value):
|
|
706
|
+
return value
|
|
707
|
+
return LINE_BREAK_RE.sub(" ", value).strip()
|
|
708
|
+
|
|
709
|
+
|
|
710
|
+
def _iso_date_or_none(value: str) -> str | None:
|
|
711
|
+
"""`value` when it is a strict ISO ``YYYY-MM-DD`` calendar date, else None.
|
|
712
|
+
|
|
713
|
+
The shared shape of the ledger's two date checks, so a skip-not-raise caller
|
|
714
|
+
(:func:`_close_date`) and a raise caller (:func:`_require_iso_date`) cannot
|
|
715
|
+
drift apart on what counts as a close date. The regex is not redundant with
|
|
716
|
+
``date.fromisoformat``: since 3.11 that also accepts ``20260611`` and ISO
|
|
717
|
+
week dates, neither of which the ledger's own readers recognize, and it is
|
|
718
|
+
the regex — via ``[0-9]`` — that pins the digits to ASCII. ``fromisoformat``
|
|
719
|
+
in turn rejects the well-shaped impossible day (``2026-02-30``) that no
|
|
720
|
+
pattern can catch."""
|
|
721
|
+
if not _ISO_DATE_RE.fullmatch(value):
|
|
722
|
+
return None
|
|
723
|
+
try:
|
|
724
|
+
calendar_date.fromisoformat(value)
|
|
725
|
+
except ValueError:
|
|
726
|
+
return None
|
|
727
|
+
return value
|
|
728
|
+
|
|
729
|
+
|
|
730
|
+
def _require_iso_date(value: str) -> None:
|
|
731
|
+
"""Raise unless `value` is a strict ISO `YYYY-MM-DD` calendar date.
|
|
732
|
+
|
|
733
|
+
Raising is right here and wrong for free text: `date` is orchestrator-owned
|
|
734
|
+
(`Engine._today()`), never model-authored, so a bad value is a programmer
|
|
735
|
+
bug. Letting it through writes a `status:` line that reads as neither open
|
|
736
|
+
nor done, which `classify` reports as malformed and `open_ids` drops — the
|
|
737
|
+
entry silently leaves the sweep's world."""
|
|
738
|
+
if _iso_date_or_none(value) is None:
|
|
739
|
+
raise ValueError(f"date must be YYYY-MM-DD: {value!r}")
|
|
740
|
+
|
|
741
|
+
|
|
742
|
+
def _require_canonical_status(status: str) -> None:
|
|
743
|
+
"""Raise unless `status` is exactly `open` or `done YYYY-MM-DD`.
|
|
744
|
+
|
|
745
|
+
Two halves with two different dependents. The *first word* is what
|
|
746
|
+
:attr:`DWEntry.open` and :func:`classify` branch on, so anything but `open`
|
|
747
|
+
or `done` makes an entry unreadable to both. The *date* is invisible to them
|
|
748
|
+
— they read `status.split()[0]` and cannot tell `done 2026-02-30` from a real
|
|
749
|
+
day — but it is not invisible downstream: the whole status value is carried
|
|
750
|
+
verbatim to readers (the TUI's deferred pane, the `--json` projections), so a
|
|
751
|
+
malformed date is rendered to a human as though it were one."""
|
|
752
|
+
if status == "open":
|
|
753
|
+
return
|
|
754
|
+
if status.startswith("done "):
|
|
755
|
+
_require_iso_date(status.removeprefix("done "))
|
|
756
|
+
return
|
|
757
|
+
raise ValueError(f"status must be 'open' or 'done YYYY-MM-DD': {status!r}")
|
|
758
|
+
|
|
759
|
+
|
|
760
|
+
def _operation_digest(operation_id: str) -> str:
|
|
761
|
+
"""Encode a stable close-operation id as one ledger-safe token."""
|
|
762
|
+
if not operation_id:
|
|
763
|
+
raise ValueError("operation_id must not be empty")
|
|
764
|
+
return hashlib.sha256(operation_id.encode("utf-8")).hexdigest()
|
|
765
|
+
|
|
766
|
+
|
|
767
|
+
# Per-thread reentrancy guard for :func:`ledger_lock`. `file_lock` is per open
|
|
768
|
+
# fd, so a second acquisition from the same process does not merely queue — on
|
|
769
|
+
# POSIX `flock` it blocks forever against a lock this very thread holds, with no
|
|
770
|
+
# timeout and no traceback. Thread-local rather than a plain module global
|
|
771
|
+
# because the state being tracked is "does THIS thread already hold it", and two
|
|
772
|
+
# threads legitimately contend through the OS lock.
|
|
773
|
+
_LOCK_STATE = threading.local()
|
|
774
|
+
|
|
775
|
+
|
|
776
|
+
@contextmanager
|
|
777
|
+
def ledger_lock(path: Path) -> Iterator[None]:
|
|
778
|
+
"""Cross-process mutual exclusion for one ledger (#286/#469).
|
|
779
|
+
|
|
780
|
+
Held only around a single read->edit->write of `path` — never across a
|
|
781
|
+
subprocess, a coding-CLI session, or an operator pause. That is an acceptance
|
|
782
|
+
criterion of #286 rather than a style preference: `file_lock`'s Windows
|
|
783
|
+
branch gives up after ~10 s and raises, so a holder that waits on anything
|
|
784
|
+
slower converts a contended run into a failed one. It is also why the
|
|
785
|
+
engine's rollback/restore windows, which span git spawns, get compare-and-set
|
|
786
|
+
semantics instead of a lock around the window.
|
|
787
|
+
|
|
788
|
+
Acquired in exactly two strata: the leaf mutators in this module, and the
|
|
789
|
+
engine's CAS restores, which do pure in-memory text work under the hold.
|
|
790
|
+
Never call a mutator while holding it — every mutator takes this lock itself,
|
|
791
|
+
and the nested acquisition would deadlock.
|
|
792
|
+
|
|
793
|
+
Nesting raises :class:`RuntimeError` rather than deadlocking. The guard is
|
|
794
|
+
deliberately path-agnostic: two *different* ledgers would not self-deadlock
|
|
795
|
+
on the OS lock, but nesting is still a lock-ordering hazard, and no caller
|
|
796
|
+
has a reason to hold two ledgers at once. The lock file itself lives out of
|
|
797
|
+
the repository — see :func:`~froid_loop.runs.lock_path_for` for why a sidecar
|
|
798
|
+
beside the tracked ledger would be committed by the engine's own `git add
|
|
799
|
+
-A`. Propagates `OSError` from acquisition and
|
|
800
|
+
:class:`~froid_loop.runs.StateRootError` when no state root can be derived: a
|
|
801
|
+
write that could not be serialized must fail loudly, not proceed unlocked.
|
|
802
|
+
"""
|
|
803
|
+
# Lazy, and it has to stay lazy: `runs` imports `verify`, which imports this
|
|
804
|
+
# module, so a top-level import here closes the cycle.
|
|
805
|
+
from . import runs
|
|
806
|
+
|
|
807
|
+
if getattr(_LOCK_STATE, "held", False):
|
|
808
|
+
raise RuntimeError("ledger lock is not reentrant")
|
|
809
|
+
lock_path = runs.lock_path_for(path)
|
|
810
|
+
_LOCK_STATE.held = True
|
|
811
|
+
try:
|
|
812
|
+
with file_lock(lock_path):
|
|
813
|
+
yield
|
|
814
|
+
finally:
|
|
815
|
+
_LOCK_STATE.held = False
|
|
816
|
+
|
|
817
|
+
|
|
818
|
+
def _apply_done(
|
|
819
|
+
text: str,
|
|
820
|
+
dw_id: str,
|
|
821
|
+
date: str,
|
|
822
|
+
note: str,
|
|
823
|
+
*,
|
|
824
|
+
undo_owner: str | None = None,
|
|
825
|
+
) -> str | None:
|
|
826
|
+
"""Flip one entry to `status: done <date>` + a resolution note *within* `text`.
|
|
827
|
+
None when the entry is missing or not open. The entry is re-located after the
|
|
828
|
+
status rewrite because that edit shifts every later span offset.
|
|
829
|
+
|
|
830
|
+
The note is sanitized here, at the point of interpolation, rather than on
|
|
831
|
+
:func:`mark_done`: that is a one-id wrapper over :func:`mark_done_many`, which
|
|
832
|
+
`Engine._apply_deferred_closes` calls directly, so a wrapper-side guard would
|
|
833
|
+
never see a story close (#305). `date` is validated by the sole caller, at its
|
|
834
|
+
entry, so the check does not depend on a ledger existing."""
|
|
835
|
+
note = _one_line(note)
|
|
836
|
+
entry = _find_entry(text, dw_id)
|
|
837
|
+
if entry is None or not entry.open:
|
|
838
|
+
return None
|
|
839
|
+
assert entry.status_span is not None # open implies a status line
|
|
840
|
+
start = entry.span[0] + entry.status_span[0]
|
|
841
|
+
end = entry.span[0] + entry.status_span[1]
|
|
842
|
+
previous_status_line = entry.body[entry.status_span[0] : entry.status_span[1]]
|
|
843
|
+
if undo_owner is not None and LINE_BREAK_RE.search(previous_status_line):
|
|
844
|
+
# An undo marker must never preserve a value that becomes more than one line
|
|
845
|
+
# under the ledger readers' shared splitlines semantics. Standard closes
|
|
846
|
+
# retain their existing behavior; the undo-capable path refuses the mark.
|
|
847
|
+
return None
|
|
848
|
+
done_status_line = f"status: done {date}"
|
|
849
|
+
text = text[:start] + done_status_line + text[end:]
|
|
850
|
+
entry = _find_entry(text, dw_id)
|
|
851
|
+
assert entry is not None
|
|
852
|
+
tail = f"resolution: {note}"
|
|
853
|
+
if undo_owner is not None:
|
|
854
|
+
# The owner digest makes this close distinguishable from an earlier run
|
|
855
|
+
# that reused its human-readable note. The encoded prior line makes the
|
|
856
|
+
# undo lossless for parser-accepted spacing and annotations. Hex keeps
|
|
857
|
+
# every payload on one ASCII line, including Unicode annotations.
|
|
858
|
+
previous_status_hex = previous_status_line.encode("utf-8").hex()
|
|
859
|
+
tail += f"\nresolution-undo: {undo_owner} {date} {previous_status_hex}"
|
|
860
|
+
return _insert_after_status(text, entry, tail)
|
|
861
|
+
|
|
862
|
+
|
|
863
|
+
def _apply_done_many(
|
|
864
|
+
text: str,
|
|
865
|
+
dw_ids: Sequence[str],
|
|
866
|
+
date: str,
|
|
867
|
+
note: str,
|
|
868
|
+
notes: Sequence[str] | None,
|
|
869
|
+
undo_owner: str | None,
|
|
870
|
+
) -> tuple[str, list[str]]:
|
|
871
|
+
"""Fold every id in `dw_ids` through :func:`_apply_done` *within* `text`,
|
|
872
|
+
returning the new text and the ids actually flipped, in the order given.
|
|
873
|
+
|
|
874
|
+
Pure — text in, text out, no `Path` and no I/O — and it is ONE body for the
|
|
875
|
+
advisory pre-lock probe and the locked pass, so the two cannot drift: the
|
|
876
|
+
argument :func:`_apply_append`'s extraction already makes for the batched
|
|
877
|
+
appender. The whole decision lives here, `undo_owner` included, because the
|
|
878
|
+
reopenable arm's LINE_BREAK refusal (:func:`_apply_done`) can be the only
|
|
879
|
+
reason a batch flips nothing — a probe that scanned for open entries by hand
|
|
880
|
+
would answer "would write" where this answers "would not"."""
|
|
881
|
+
marked: list[str] = []
|
|
882
|
+
for index, dw_id in enumerate(dw_ids):
|
|
883
|
+
entry_note = note if notes is None else notes[index]
|
|
884
|
+
updated = _apply_done(text, dw_id, date, entry_note, undo_owner=undo_owner)
|
|
885
|
+
if updated is None:
|
|
886
|
+
continue
|
|
887
|
+
text = updated
|
|
888
|
+
marked.append(dw_id)
|
|
889
|
+
return text, marked
|
|
890
|
+
|
|
891
|
+
|
|
892
|
+
def _mark_done_many(
|
|
893
|
+
path: Path,
|
|
894
|
+
dw_ids: Sequence[str],
|
|
895
|
+
date: str,
|
|
896
|
+
note: str,
|
|
897
|
+
*,
|
|
898
|
+
operation_id: str | None = None,
|
|
899
|
+
notes: Sequence[str] | None = None,
|
|
900
|
+
) -> list[str]:
|
|
901
|
+
"""Shared atomic implementation for the public close operations.
|
|
902
|
+
|
|
903
|
+
ONE locked read->edit->write: the whole cycle runs under the cross-process
|
|
904
|
+
ledger lock (#286/#469), so concurrent mutators — a second run, a sweep, the
|
|
905
|
+
TUI decision modal, ``sweep --archive`` — serialize here rather than trading
|
|
906
|
+
last-write-wins. Validation stays ABOVE the lock, so a programmer bug reports
|
|
907
|
+
itself without first waiting on another process.
|
|
908
|
+
|
|
909
|
+
A batch that would flip nothing — every id missing, already done, or refused
|
|
910
|
+
by the reopenable arm's line-break guard — is answered from the advisory
|
|
911
|
+
pre-lock probe instead, with no acquisition at all (#736). The probe folds
|
|
912
|
+
the ids through :func:`_apply_done_many`, the same helper the locked pass
|
|
913
|
+
uses, so it cannot answer "no write" where the authority would write.
|
|
914
|
+
|
|
915
|
+
``notes`` supplies a per-id resolution note, positionally paired with
|
|
916
|
+
``dw_ids``; ``note`` is the fallback for every id when it is None. A length
|
|
917
|
+
mismatch raises before any I/O rather than closing a prefix under the wrong
|
|
918
|
+
evidence — the pairing is positional, so a short list is a caller bug that
|
|
919
|
+
would otherwise mis-attribute notes silently.
|
|
920
|
+
"""
|
|
921
|
+
_require_iso_date(date)
|
|
922
|
+
if notes is not None and len(notes) != len(dw_ids):
|
|
923
|
+
raise ValueError(f"notes must be one per dw_id: {len(notes)} for {len(dw_ids)} ids")
|
|
924
|
+
undo_owner = _operation_digest(operation_id) if operation_id is not None else None
|
|
925
|
+
if not dw_ids:
|
|
926
|
+
# Nothing to serialize against, so nothing to take a lock for — the same
|
|
927
|
+
# early return `append_entries` makes, for the same reason. Below the
|
|
928
|
+
# validation above, so an empty batch still reports a bad date or a bad
|
|
929
|
+
# operation id; above the lock, so a caller that batches an empty set
|
|
930
|
+
# cannot start failing on a lock it never needed. The per-id loop this
|
|
931
|
+
# primitive replaced took no lock at all when handed nothing, and that
|
|
932
|
+
# identity is part of what "byte-identical to the serial sequence" buys.
|
|
933
|
+
return []
|
|
934
|
+
if not path.is_file():
|
|
935
|
+
# No ledger, no entry to flip, so no write and no lock — the order
|
|
936
|
+
# `archive_closed` already keeps for its own missing-ledger case. The
|
|
937
|
+
# recheck under the hold below stays: creation can race this answer.
|
|
938
|
+
return []
|
|
939
|
+
try:
|
|
940
|
+
# ADVISORY pre-lock probe (#736): one read, and the same pure decision
|
|
941
|
+
# the locked pass makes. Only a "would write nothing" answer is acted on
|
|
942
|
+
# — the call then serializes at this read. Anything else, including any
|
|
943
|
+
# fault here, falls through to the hold, which re-reads and decides.
|
|
944
|
+
probe = path.read_text(encoding="utf-8")
|
|
945
|
+
if not _apply_done_many(probe, dw_ids, date, note, notes, undo_owner)[1]:
|
|
946
|
+
return []
|
|
947
|
+
except Exception: # nosec B110 - ADVISORY probe: a fault here must decide nothing
|
|
948
|
+
pass
|
|
949
|
+
with ledger_lock(path):
|
|
950
|
+
if not path.is_file():
|
|
951
|
+
return []
|
|
952
|
+
text = path.read_text(encoding="utf-8")
|
|
953
|
+
text, marked = _apply_done_many(text, dw_ids, date, note, notes, undo_owner)
|
|
954
|
+
if not marked:
|
|
955
|
+
return []
|
|
956
|
+
atomic_write_text(path, text)
|
|
957
|
+
return marked
|
|
958
|
+
|
|
959
|
+
|
|
960
|
+
def mark_done_many(
|
|
961
|
+
path: Path,
|
|
962
|
+
dw_ids: Sequence[str],
|
|
963
|
+
date: str,
|
|
964
|
+
note: str,
|
|
965
|
+
*,
|
|
966
|
+
notes: Sequence[str] | None = None,
|
|
967
|
+
) -> list[str]:
|
|
968
|
+
"""Flip every entry in `dw_ids` to `status: done <date>` + a resolution note,
|
|
969
|
+
in ONE read and ONE atomic write. Returns the ids actually flipped (missing
|
|
970
|
+
and already-done ids are skipped), in the order given.
|
|
971
|
+
|
|
972
|
+
``notes[i]`` overrides `note` for ``dw_ids[i]`` — the shape a caller closing
|
|
973
|
+
several entries under per-entry evidence needs, which otherwise costs one
|
|
974
|
+
read-modify-write cycle per id. A length mismatch raises `ValueError` before
|
|
975
|
+
any I/O.
|
|
976
|
+
|
|
977
|
+
All-or-nothing on purpose. A per-id read-modify-write loop leaves marks on
|
|
978
|
+
disk when it raises partway through several ids — a half-applied closure the
|
|
979
|
+
caller never gets to journal, so the ledger claims resolutions the run has no
|
|
980
|
+
record of. Here a failure writes nothing, and the returned list is exactly
|
|
981
|
+
what landed.
|
|
982
|
+
|
|
983
|
+
The write goes through :func:`~froid_loop.platform_util.atomic_write_text`
|
|
984
|
+
rather than a bare tmp+replace: swapping a fresh inode over the ledger
|
|
985
|
+
otherwise resets its mode (a ``0600`` ledger silently becoming world-readable)
|
|
986
|
+
and turns a symlinked ledger into a regular file.
|
|
987
|
+
|
|
988
|
+
``date`` is validated before the ``is_file`` short-circuit so a programmer bug
|
|
989
|
+
fails the same way whether or not a ledger happens to exist — a guard that
|
|
990
|
+
only fires when the file is present is one an absent fixture hides."""
|
|
991
|
+
return _mark_done_many(path, dw_ids, date, note, notes=notes)
|
|
992
|
+
|
|
993
|
+
|
|
994
|
+
def mark_done_many_reopenable(
|
|
995
|
+
path: Path,
|
|
996
|
+
dw_ids: Sequence[str],
|
|
997
|
+
date: str,
|
|
998
|
+
note: str,
|
|
999
|
+
operation_id: str,
|
|
1000
|
+
) -> list[str]:
|
|
1001
|
+
"""Close entries atomically with a durable, operation-specific undo marker.
|
|
1002
|
+
|
|
1003
|
+
``operation_id`` must be stable and recomputable across crash/replay from
|
|
1004
|
+
already-persisted identity (for example ``run_id`` + ``story_key``), never an
|
|
1005
|
+
ephemeral random value. Only entries actually flipped receive its marker;
|
|
1006
|
+
skipped, already-done ids therefore cannot be reopened by this operation.
|
|
1007
|
+
|
|
1008
|
+
The ordinary :func:`mark_done_many` deliberately emits no marker and retains
|
|
1009
|
+
its existing ledger format. Use this variant only for a transaction with a
|
|
1010
|
+
later rollback leg.
|
|
1011
|
+
"""
|
|
1012
|
+
return _mark_done_many(path, dw_ids, date, note, operation_id=operation_id)
|
|
1013
|
+
|
|
1014
|
+
|
|
1015
|
+
def mark_done(path: Path, dw_id: str, date: str, note: str) -> bool:
|
|
1016
|
+
"""Flip one entry to `status: done <date>` and record a resolution note.
|
|
1017
|
+
Returns False (no write) when the entry is missing or already done."""
|
|
1018
|
+
return bool(mark_done_many(path, [dw_id], date, note))
|
|
1019
|
+
|
|
1020
|
+
|
|
1021
|
+
_MARK_DONE_TAIL_RE = re.compile(
|
|
1022
|
+
r"\nresolution:[ \t]*(.*)"
|
|
1023
|
+
r"\nresolution-undo:[ \t]*([0-9a-f]{64})[ \t]+"
|
|
1024
|
+
r"([0-9]{4}-[0-9]{2}-[0-9]{2})[ \t]+([0-9a-f]+)$",
|
|
1025
|
+
re.MULTILINE,
|
|
1026
|
+
)
|
|
1027
|
+
|
|
1028
|
+
|
|
1029
|
+
def _apply_open(text: str, dw_id: str, note: str, undo_owner: str) -> str | None:
|
|
1030
|
+
"""Undo one reopenable close *within* `text`. None when the entry is missing,
|
|
1031
|
+
already open, or does not carry this operation's adjacent resolution and
|
|
1032
|
+
undo-marker lines.
|
|
1033
|
+
|
|
1034
|
+
Pure by construction — text in, text out, no `Path` and no I/O — which is
|
|
1035
|
+
what keeps :func:`mark_open_many` able to run it several times inside a
|
|
1036
|
+
single :func:`ledger_lock` hold. A version of this that touched the file
|
|
1037
|
+
would have to take the lock itself, and the nested acquisition is exactly the
|
|
1038
|
+
self-deadlock the guard on `ledger_lock` exists to convert into an error.
|
|
1039
|
+
|
|
1040
|
+
A standard or earlier close has no matching marker and cannot be reopened
|
|
1041
|
+
merely because it reused the same human-readable note.
|
|
1042
|
+
|
|
1043
|
+
A live ``archived:`` stamp is demoted to :data:`_ARCHIVED_BODY_FIELD` rather
|
|
1044
|
+
than dropped: the reopened entry is no longer archived, but the body its
|
|
1045
|
+
close moved out still is, and that line is the only thing a later triage has
|
|
1046
|
+
to find it with."""
|
|
1047
|
+
entry = _find_entry(text, dw_id)
|
|
1048
|
+
if entry is None or entry.open:
|
|
1049
|
+
return None
|
|
1050
|
+
if entry.status_span is None:
|
|
1051
|
+
# parse_ledger deliberately tolerates status-less entries. This primitive
|
|
1052
|
+
# is later called from _defer, where an AttributeError would crash the run
|
|
1053
|
+
# instead of completing the deferral.
|
|
1054
|
+
return None
|
|
1055
|
+
status_line = entry.body[entry.status_span[0] : entry.status_span[1]]
|
|
1056
|
+
try:
|
|
1057
|
+
_require_canonical_status(entry.status)
|
|
1058
|
+
except ValueError:
|
|
1059
|
+
# Only a canonical status written by mark_done is eligible for undo.
|
|
1060
|
+
# Preserve malformed or human-authored statuses for validation/reporting.
|
|
1061
|
+
return None
|
|
1062
|
+
res_m = _MARK_DONE_TAIL_RE.match(entry.body, entry.status_span[1])
|
|
1063
|
+
if res_m is None:
|
|
1064
|
+
return None
|
|
1065
|
+
if res_m.group(1).strip() != _one_line(note).strip() or res_m.group(2) != undo_owner:
|
|
1066
|
+
return None
|
|
1067
|
+
if status_line != f"status: done {res_m.group(3)}":
|
|
1068
|
+
return None
|
|
1069
|
+
try:
|
|
1070
|
+
previous_status_line = bytes.fromhex(res_m.group(4)).decode("utf-8")
|
|
1071
|
+
except (UnicodeDecodeError, ValueError):
|
|
1072
|
+
return None
|
|
1073
|
+
if LINE_BREAK_RE.search(previous_status_line):
|
|
1074
|
+
return None
|
|
1075
|
+
previous_status_m = STATUS_RE.fullmatch(previous_status_line)
|
|
1076
|
+
previous_status = previous_status_m.group(1).strip() if previous_status_m else ""
|
|
1077
|
+
if not previous_status or previous_status.split()[0] != "open":
|
|
1078
|
+
return None
|
|
1079
|
+
start = entry.span[0] + entry.status_span[0]
|
|
1080
|
+
end = entry.span[0] + res_m.end()
|
|
1081
|
+
# Demote the entry's live `archived:` stamps along with the close they
|
|
1082
|
+
# describe, rather than deleting them. A stub's stamp says "this body lives
|
|
1083
|
+
# in the archive file"; once the close is undone the body is here and the
|
|
1084
|
+
# line is a lie, and leaving it standing is not merely untidy — status +
|
|
1085
|
+
# undo tail + stamp is the exact `_STUB_BODY_RE` shape, so the next
|
|
1086
|
+
# reopenable close reconstitutes a stub `archive_closed` skips forever,
|
|
1087
|
+
# stranding the entry outside every future archive (#711).
|
|
1088
|
+
#
|
|
1089
|
+
# Cutting the line outright strands the entry a second way: a stub keeps
|
|
1090
|
+
# neither `location:` nor `reason:` (`_PRESERVED_FIELD_RE`), so the stamp is
|
|
1091
|
+
# the reopened entry's ONLY route back to the body, and triage arrives with
|
|
1092
|
+
# a heading and nothing to triage (#711 review). Renaming the field keeps
|
|
1093
|
+
# both properties — the value still narrows to the archive block, an id
|
|
1094
|
+
# owning several once a re-closure is archived too, while the renamed line
|
|
1095
|
+
# matches neither `_ARCHIVED_FIELD_RE` nor `_STUB_BODY_RE`, so the entry
|
|
1096
|
+
# reads as live and re-archives normally. Rehydrating the body here
|
|
1097
|
+
# instead was the alternative and is worse: several blocks per id is by
|
|
1098
|
+
# design, so a rollback's reopen would have to guess which one, and a wrong
|
|
1099
|
+
# guess overwrites live content with a stale body.
|
|
1100
|
+
#
|
|
1101
|
+
# Cuts are disjoint (an `^archived:` line cannot start inside the status
|
|
1102
|
+
# line or its adjacent tail) and applied back-to-front so earlier offsets
|
|
1103
|
+
# stay valid.
|
|
1104
|
+
cuts = [(start, end, previous_status_line)]
|
|
1105
|
+
for cut_start, cut_end in _archived_line_spans(entry):
|
|
1106
|
+
# Everything after the field name — value, spacing and the terminating
|
|
1107
|
+
# newline — carries over verbatim; the span starts at the anchor, so
|
|
1108
|
+
# the first colon is the field's own.
|
|
1109
|
+
stamp = entry.body[cut_start:cut_end].split(":", 1)[1]
|
|
1110
|
+
cuts.append(
|
|
1111
|
+
(
|
|
1112
|
+
entry.span[0] + cut_start,
|
|
1113
|
+
entry.span[0] + cut_end,
|
|
1114
|
+
f"{_ARCHIVED_BODY_FIELD}{stamp}",
|
|
1115
|
+
)
|
|
1116
|
+
)
|
|
1117
|
+
for cut_start, cut_end, replacement in sorted(cuts, reverse=True):
|
|
1118
|
+
text = text[:cut_start] + replacement + text[cut_end:]
|
|
1119
|
+
return text
|
|
1120
|
+
|
|
1121
|
+
|
|
1122
|
+
def _apply_open_many(
|
|
1123
|
+
text: str, dw_ids: Sequence[str], note: str, undo_owner: str
|
|
1124
|
+
) -> tuple[str, list[str]]:
|
|
1125
|
+
"""Fold every id in `dw_ids` through :func:`_apply_open` *within* `text`,
|
|
1126
|
+
returning the new text and the ids actually reopened, in the order given.
|
|
1127
|
+
|
|
1128
|
+
Pure — text in, text out, no `Path` and no I/O — and ONE body for the
|
|
1129
|
+
advisory pre-lock probe and the locked pass, so the two cannot drift. The
|
|
1130
|
+
`undo_owner` match is part of the decision: an entry closed by a different
|
|
1131
|
+
operation is skipped here, which is what makes "no id was eligible" a
|
|
1132
|
+
question only this fold can answer."""
|
|
1133
|
+
reopened: list[str] = []
|
|
1134
|
+
for dw_id in dw_ids:
|
|
1135
|
+
updated = _apply_open(text, dw_id, note, undo_owner)
|
|
1136
|
+
if updated is None:
|
|
1137
|
+
continue
|
|
1138
|
+
text = updated
|
|
1139
|
+
reopened.append(dw_id)
|
|
1140
|
+
return text, reopened
|
|
1141
|
+
|
|
1142
|
+
|
|
1143
|
+
def mark_open_many(path: Path, dw_ids: Sequence[str], note: str, operation_id: str) -> list[str]:
|
|
1144
|
+
"""Undo every close in `dw_ids` written by :func:`mark_done_many_reopenable`
|
|
1145
|
+
under `operation_id`, in ONE read and ONE atomic write. Returns the ids
|
|
1146
|
+
actually reopened, in the order given; missing and ineligible ids are
|
|
1147
|
+
skipped, and an entry whose marker does not match this operation is left
|
|
1148
|
+
exactly as it was.
|
|
1149
|
+
|
|
1150
|
+
ONE locked read->edit->write: the whole cycle runs under the cross-process
|
|
1151
|
+
ledger lock (#286/#469), so concurrent mutators — a second run, a sweep, the
|
|
1152
|
+
TUI decision modal, ``sweep --archive`` — serialize here rather than trading
|
|
1153
|
+
last-write-wins. A per-id loop over :func:`mark_open` would instead take the
|
|
1154
|
+
lock once per id, leaving a rival writer a window between every pair of
|
|
1155
|
+
undos in what a rollback needs to be one step.
|
|
1156
|
+
|
|
1157
|
+
Nothing is written when no id was eligible, and no lock is taken either
|
|
1158
|
+
(#736): a replayed rollback over already-reopened entries is answered from
|
|
1159
|
+
one advisory read, so it leaves the file untouched rather than rewriting it
|
|
1160
|
+
byte-for-byte, and cannot fail on a lock it had no write to serialize."""
|
|
1161
|
+
undo_owner = _operation_digest(operation_id)
|
|
1162
|
+
if not dw_ids:
|
|
1163
|
+
# No ids, no lock — see `_mark_done_many`. The `operation_id` above is
|
|
1164
|
+
# still validated, so an empty reopen cannot smuggle a bad one through.
|
|
1165
|
+
return []
|
|
1166
|
+
if not path.is_file():
|
|
1167
|
+
# No ledger, no close to undo — see `_mark_done_many`. Rechecked under
|
|
1168
|
+
# the hold below.
|
|
1169
|
+
return []
|
|
1170
|
+
try:
|
|
1171
|
+
# ADVISORY pre-lock probe (#736): one read, and the same pure decision
|
|
1172
|
+
# the locked pass makes. Only a "would write nothing" answer is acted on
|
|
1173
|
+
# — the call then serializes at this read. Anything else, including any
|
|
1174
|
+
# fault here, falls through to the hold, which re-reads and decides.
|
|
1175
|
+
probe = path.read_text(encoding="utf-8")
|
|
1176
|
+
if not _apply_open_many(probe, dw_ids, note, undo_owner)[1]:
|
|
1177
|
+
return []
|
|
1178
|
+
except Exception: # nosec B110 - ADVISORY probe: a fault here must decide nothing
|
|
1179
|
+
pass
|
|
1180
|
+
with ledger_lock(path):
|
|
1181
|
+
if not path.is_file():
|
|
1182
|
+
return []
|
|
1183
|
+
text = path.read_text(encoding="utf-8")
|
|
1184
|
+
text, reopened = _apply_open_many(text, dw_ids, note, undo_owner)
|
|
1185
|
+
if not reopened:
|
|
1186
|
+
return []
|
|
1187
|
+
atomic_write_text(path, text)
|
|
1188
|
+
return reopened
|
|
1189
|
+
|
|
1190
|
+
|
|
1191
|
+
def mark_open(path: Path, dw_id: str, note: str, operation_id: str) -> bool:
|
|
1192
|
+
"""Undo one close written by :func:`mark_done_many_reopenable`.
|
|
1193
|
+
|
|
1194
|
+
A one-id wrapper over :func:`mark_open_many`, which is where the lock is
|
|
1195
|
+
taken and the contract documented. It delegates rather than duplicating the
|
|
1196
|
+
read->edit->write so that one public call is exactly one acquisition — a
|
|
1197
|
+
wrapper that took the lock itself and then called the batch would nest, and
|
|
1198
|
+
`ledger_lock` raises on that rather than deadlocking."""
|
|
1199
|
+
return bool(mark_open_many(path, [dw_id], note, operation_id))
|
|
1200
|
+
|
|
1201
|
+
|
|
1202
|
+
def _apply_decision(text: str, dw_id: str, date: str, label: str, detail: str) -> str | None:
|
|
1203
|
+
"""Insert one `decision: <date> <label> — <detail>` line *within* `text`,
|
|
1204
|
+
right after the entry's status line. None when the entry is missing.
|
|
1205
|
+
|
|
1206
|
+
Pure — text in, text out, no `Path` and no I/O — so :func:`record_decision`
|
|
1207
|
+
can run it and :func:`_apply_done` against the same in-memory text inside a
|
|
1208
|
+
single :func:`ledger_lock` hold. Applies to a done entry as readily as an
|
|
1209
|
+
open one: a decision is a record of what a human chose, not a status change.
|
|
1210
|
+
|
|
1211
|
+
`label` and `detail` come from a triage session's `DecisionOption`, so they
|
|
1212
|
+
are sanitized to one line rather than refused — see :func:`_one_line`. This
|
|
1213
|
+
is also where a build option's `intent` gets flattened, since it reaches the
|
|
1214
|
+
ledger only as `detail = option.resolution or option.intent`."""
|
|
1215
|
+
entry = _find_entry(text, dw_id)
|
|
1216
|
+
if entry is None:
|
|
1217
|
+
return None
|
|
1218
|
+
label = _one_line(label)
|
|
1219
|
+
# Sanitize before the emptiness test, never after: a break-only detail
|
|
1220
|
+
# collapses to "" and must then drop the separator with it, or the entry
|
|
1221
|
+
# carries a dangling `— ` promising a detail that is not there.
|
|
1222
|
+
detail = _one_line(detail)
|
|
1223
|
+
detail_part = f" — {detail}" if detail else ""
|
|
1224
|
+
return _insert_after_status(text, entry, f"decision: {date} {label}{detail_part}")
|
|
1225
|
+
|
|
1226
|
+
|
|
1227
|
+
def record_decision(
|
|
1228
|
+
path: Path,
|
|
1229
|
+
dw_id: str,
|
|
1230
|
+
date: str,
|
|
1231
|
+
label: str,
|
|
1232
|
+
detail: str,
|
|
1233
|
+
*,
|
|
1234
|
+
close_note: str | None = None,
|
|
1235
|
+
) -> bool:
|
|
1236
|
+
"""Record a human decision on one entry and, when `close_note` is given, act
|
|
1237
|
+
on it by flipping the entry to `status: done <date>` — both in ONE read and
|
|
1238
|
+
ONE atomic write. Returns True when the entry was found (and therefore
|
|
1239
|
+
carries a decision line), False when it was not.
|
|
1240
|
+
|
|
1241
|
+
ONE locked read->edit->write: the whole cycle runs under the cross-process
|
|
1242
|
+
ledger lock (#286/#469), so concurrent mutators — a second run, a sweep, the
|
|
1243
|
+
TUI decision modal, ``sweep --archive`` — serialize here rather than trading
|
|
1244
|
+
last-write-wins. That is the reason the pair is one primitive at all: as
|
|
1245
|
+
separate :func:`append_decision` and :func:`mark_done` calls it is two
|
|
1246
|
+
acquisitions with a window between them, and a rival writer landing in that
|
|
1247
|
+
window sees an entry whose decision says "close it" and whose status still
|
|
1248
|
+
says open.
|
|
1249
|
+
|
|
1250
|
+
The decision line is inserted BEFORE the close is applied, which is not a
|
|
1251
|
+
preference: :func:`_apply_done` writes its `resolution:` line immediately
|
|
1252
|
+
after the status line, and :data:`_MARK_DONE_TAIL_RE` — what
|
|
1253
|
+
:func:`_apply_open` matches an undo marker with — anchors on exactly that
|
|
1254
|
+
adjacency. Applying the close first would leave the decision line between
|
|
1255
|
+
status and resolution and make a reopenable close unreopenable. Ordered this
|
|
1256
|
+
way the bytes are identical to the serial pair's.
|
|
1257
|
+
|
|
1258
|
+
An already-done (or missing-status) entry skips only the close half: the
|
|
1259
|
+
decision line still lands, because a decision recorded on an entry someone
|
|
1260
|
+
else already closed is still what the human chose. `close_note` is the
|
|
1261
|
+
resolution note for the flip, distinct from `detail`, which is the decision's
|
|
1262
|
+
own rationale.
|
|
1263
|
+
|
|
1264
|
+
Precondition: `date` is ISO `YYYY-MM-DD` — one check for both halves, since
|
|
1265
|
+
the decision line and the close share it; anything else raises `ValueError`,
|
|
1266
|
+
checked before the ``is_file`` short-circuit so an absent ledger cannot hide
|
|
1267
|
+
the bug.
|
|
1268
|
+
|
|
1269
|
+
A missing ledger, and a `dw_id` no entry carries, are both answered False
|
|
1270
|
+
without taking the lock (#736) — there is no write to serialize, and the
|
|
1271
|
+
TUI decision modal reaching a stale id should not fail on an acquisition.
|
|
1272
|
+
The probe runs :func:`_apply_decision`, the same helper the locked pass
|
|
1273
|
+
runs, which is None exactly when the entry is missing.
|
|
1274
|
+
|
|
1275
|
+
The write goes through :func:`~froid_loop.platform_util.atomic_write_text` for
|
|
1276
|
+
the reasons documented on :func:`mark_done_many`, plus one this sibling shares
|
|
1277
|
+
with it: a bare ``Path.write_text`` truncates *before* it encodes, so any
|
|
1278
|
+
failure between the two — an unencodable value, ``ENOSPC``, ``EIO`` — leaves a
|
|
1279
|
+
zero-byte ledger where every entry used to be (#328).
|
|
1280
|
+
"""
|
|
1281
|
+
_require_iso_date(date)
|
|
1282
|
+
if not path.is_file():
|
|
1283
|
+
# No ledger, no entry to record against — see `_mark_done_many`.
|
|
1284
|
+
# Rechecked under the hold below.
|
|
1285
|
+
return False
|
|
1286
|
+
try:
|
|
1287
|
+
# ADVISORY pre-lock probe (#736): one read, and the same pure decision
|
|
1288
|
+
# the locked pass makes. Only a "would write nothing" answer is acted on
|
|
1289
|
+
# — the call then serializes at this read. Anything else, including any
|
|
1290
|
+
# fault here, falls through to the hold, which re-reads and decides.
|
|
1291
|
+
probe = path.read_text(encoding="utf-8")
|
|
1292
|
+
if _apply_decision(probe, dw_id, date, label, detail) is None:
|
|
1293
|
+
return False
|
|
1294
|
+
except Exception: # nosec B110 - ADVISORY probe: a fault here must decide nothing
|
|
1295
|
+
pass
|
|
1296
|
+
with ledger_lock(path):
|
|
1297
|
+
if not path.is_file():
|
|
1298
|
+
return False
|
|
1299
|
+
text = path.read_text(encoding="utf-8")
|
|
1300
|
+
updated = _apply_decision(text, dw_id, date, label, detail)
|
|
1301
|
+
if updated is None:
|
|
1302
|
+
return False
|
|
1303
|
+
text = updated
|
|
1304
|
+
if close_note is not None:
|
|
1305
|
+
closed = _apply_done(text, dw_id, date, close_note)
|
|
1306
|
+
if closed is not None:
|
|
1307
|
+
text = closed
|
|
1308
|
+
atomic_write_text(path, text)
|
|
1309
|
+
return True
|
|
1310
|
+
|
|
1311
|
+
|
|
1312
|
+
def append_decision(path: Path, dw_id: str, date: str, label: str, detail: str) -> bool:
|
|
1313
|
+
"""Record a human decision on an entry without changing its status.
|
|
1314
|
+
|
|
1315
|
+
The no-close case of :func:`record_decision`, which is where the lock is
|
|
1316
|
+
taken and the contract documented. It delegates rather than duplicating the
|
|
1317
|
+
read->edit->write so that one public call is exactly one acquisition."""
|
|
1318
|
+
return record_decision(path, dw_id, date, label, detail)
|
|
1319
|
+
|
|
1320
|
+
|
|
1321
|
+
DW_ID_RE = re.compile(r"\bDW-(\d+)\b")
|
|
1322
|
+
|
|
1323
|
+
|
|
1324
|
+
def next_seq(text: str) -> int:
|
|
1325
|
+
"""The next free DW sequence number — one past the highest DW-<n> anywhere
|
|
1326
|
+
in the ledger (malformed entries included, so a number is never reused and
|
|
1327
|
+
the sweep numbering check stays satisfied)."""
|
|
1328
|
+
nums = [int(m.group(1)) for m in DW_ID_RE.finditer(text)]
|
|
1329
|
+
return (max(nums) + 1) if nums else 1
|
|
1330
|
+
|
|
1331
|
+
|
|
1332
|
+
def field_line_present(body: str, field: str, value: str) -> bool:
|
|
1333
|
+
"""True when `body` has a `field:` line whose value is exactly `value`,
|
|
1334
|
+
matching the shapes append_entry writes (plain, or backtick-wrapped as for
|
|
1335
|
+
`source_spec:`). Anchored per-line so an incidental substring elsewhere in
|
|
1336
|
+
the body (e.g. inside `reason:`) never counts as a match."""
|
|
1337
|
+
v = re.escape(value)
|
|
1338
|
+
return re.search(rf"(?m)^{re.escape(field)}:[ \t]*`?{v}`?[ \t]*$", body) is not None
|
|
1339
|
+
|
|
1340
|
+
|
|
1341
|
+
@dataclass(frozen=True)
|
|
1342
|
+
class EntrySpec:
|
|
1343
|
+
"""One :func:`append_entry` call's arguments as data, for the batched writer.
|
|
1344
|
+
|
|
1345
|
+
The defaults are that function's defaults, so a spec built from the same
|
|
1346
|
+
values produces the same entry. Frozen because :func:`append_entries`
|
|
1347
|
+
validates the whole sequence before it takes the lock and then trusts what it
|
|
1348
|
+
validated — a spec mutated in between would be written unchecked."""
|
|
1349
|
+
|
|
1350
|
+
title: str
|
|
1351
|
+
origin: str
|
|
1352
|
+
source_spec: str
|
|
1353
|
+
reason: str
|
|
1354
|
+
location: str = "n/a"
|
|
1355
|
+
status: str = "open"
|
|
1356
|
+
severity: str | None = None
|
|
1357
|
+
|
|
1358
|
+
|
|
1359
|
+
def _apply_append(text: str, spec: EntrySpec) -> tuple[str, str | None]:
|
|
1360
|
+
"""Append one canonical `### DW-<seq>` entry *within* `text`, returning the
|
|
1361
|
+
new text and the id minted — or `text` unchanged and None when an open entry
|
|
1362
|
+
already carries the same `origin:` marker and `source_spec:`.
|
|
1363
|
+
|
|
1364
|
+
Pure — text in, text out, no `Path` and no I/O — which is what lets
|
|
1365
|
+
:func:`append_entries` run it once per spec against the text as it evolves,
|
|
1366
|
+
inside a single :func:`ledger_lock` hold. Both halves that make a batch
|
|
1367
|
+
differ from a loop read that evolving text: `next_seq` mints past the entry
|
|
1368
|
+
the previous spec just added, so ids are sequential rather than colliding,
|
|
1369
|
+
and the idempotence scan sees it too, so two identical specs in one call
|
|
1370
|
+
dedupe against each other exactly as a serial pair would.
|
|
1371
|
+
|
|
1372
|
+
Free text is sanitized (:func:`_one_line`) **before** the idempotence scan,
|
|
1373
|
+
which compares the caller's value against the stored one via
|
|
1374
|
+
:func:`field_line_present`: sanitizing afterwards would compare a raw value
|
|
1375
|
+
against a sanitized line, so every replay of the same multiline defer would
|
|
1376
|
+
miss its own entry and append another.
|
|
1377
|
+
|
|
1378
|
+
The scan is deliberately open-only: a closed entry with the same marker does
|
|
1379
|
+
not suppress the append, because the work has come back."""
|
|
1380
|
+
given_title = bool(spec.title)
|
|
1381
|
+
title = _one_line(spec.title)
|
|
1382
|
+
origin = _one_line(spec.origin)
|
|
1383
|
+
source_spec = _one_line(spec.source_spec)
|
|
1384
|
+
reason = _one_line(spec.reason)
|
|
1385
|
+
location = _one_line(spec.location)
|
|
1386
|
+
for entry in parse_ledger(text):
|
|
1387
|
+
if (
|
|
1388
|
+
entry.open
|
|
1389
|
+
and field_line_present(entry.body, "origin", origin)
|
|
1390
|
+
and field_line_present(entry.body, "source_spec", source_spec)
|
|
1391
|
+
):
|
|
1392
|
+
return text, None
|
|
1393
|
+
dw_id = f"DW-{next_seq(text)}"
|
|
1394
|
+
if given_title and not title.strip():
|
|
1395
|
+
# A break-only title sanitizes to nothing, and `### DW-<n>: ` is a
|
|
1396
|
+
# heading `HEADING_RE`'s `(.+?)` does not match: the caller is handed an
|
|
1397
|
+
# id no reader can find while `next_seq` has already burned it.
|
|
1398
|
+
#
|
|
1399
|
+
# Tested with `.strip()`, not `not title`: a title of `" "` carries no
|
|
1400
|
+
# break at all, so `_one_line` returns it unchanged by the byte-identity
|
|
1401
|
+
# fast path and it stays truthy. It parses, but renders blank in
|
|
1402
|
+
# `status`, `--json` and the TUI — the unidentifiable half of the same
|
|
1403
|
+
# problem, reached without ever touching the sanitizer.
|
|
1404
|
+
#
|
|
1405
|
+
# Scoped to a title that *had* content: an already-empty one keeps its
|
|
1406
|
+
# long-standing behavior, and the invariant is about non-empty values.
|
|
1407
|
+
title = f"(untitled {dw_id})"
|
|
1408
|
+
lines = [
|
|
1409
|
+
f"### {dw_id}: {title}",
|
|
1410
|
+
f"origin: {origin}",
|
|
1411
|
+
f"location: {location}",
|
|
1412
|
+
f"source_spec: `{source_spec}`",
|
|
1413
|
+
]
|
|
1414
|
+
if spec.severity:
|
|
1415
|
+
lines.append(f"severity: {spec.severity}")
|
|
1416
|
+
lines.append(f"reason: {reason}")
|
|
1417
|
+
lines.append(f"status: {spec.status}")
|
|
1418
|
+
block = "\n".join(lines) + "\n"
|
|
1419
|
+
# exactly one blank line between the previous content and the new entry
|
|
1420
|
+
if text == "" or text.endswith("\n\n"):
|
|
1421
|
+
sep = ""
|
|
1422
|
+
elif text.endswith("\n"):
|
|
1423
|
+
sep = "\n"
|
|
1424
|
+
else:
|
|
1425
|
+
sep = "\n\n"
|
|
1426
|
+
return text + sep + block, dw_id
|
|
1427
|
+
|
|
1428
|
+
|
|
1429
|
+
def _apply_appends(text: str, specs: Sequence[EntrySpec]) -> tuple[str, list[str | None]]:
|
|
1430
|
+
"""Fold every spec through :func:`_apply_append` *within* `text`, returning
|
|
1431
|
+
the new text and one minted id per spec — None where the spec deduped
|
|
1432
|
+
against an open entry that already carries its marker.
|
|
1433
|
+
|
|
1434
|
+
Pure — text in, text out, no `Path` and no I/O — and ONE body for the
|
|
1435
|
+
advisory pre-lock probe and the locked pass, so the two cannot drift. Each
|
|
1436
|
+
spec sees the text the previous one produced, which is what makes ids
|
|
1437
|
+
sequential and lets two identical specs in one call dedupe against each
|
|
1438
|
+
other; see :func:`_apply_append` for why that evolution is load-bearing."""
|
|
1439
|
+
minted: list[str | None] = []
|
|
1440
|
+
for spec in specs:
|
|
1441
|
+
text, dw_id = _apply_append(text, spec)
|
|
1442
|
+
minted.append(dw_id)
|
|
1443
|
+
return text, minted
|
|
1444
|
+
|
|
1445
|
+
|
|
1446
|
+
def append_entries(path: Path, specs: Sequence[EntrySpec]) -> list[str | None]:
|
|
1447
|
+
"""Append every entry in `specs` in ONE read and ONE atomic write, returning
|
|
1448
|
+
each spec's minted id — or None in its position when that spec deduped
|
|
1449
|
+
against an already-open entry. Creates the ledger (and parent dir) if it does
|
|
1450
|
+
not yet exist.
|
|
1451
|
+
|
|
1452
|
+
A thin wrapper over :func:`append_entries_published`, for the callers that
|
|
1453
|
+
only need the ids. One acquisition, in the leaf.
|
|
1454
|
+
"""
|
|
1455
|
+
return append_entries_published(path, specs)[0]
|
|
1456
|
+
|
|
1457
|
+
|
|
1458
|
+
def append_entries_published(
|
|
1459
|
+
path: Path, specs: Sequence[EntrySpec]
|
|
1460
|
+
) -> tuple[list[str | None], str | None]:
|
|
1461
|
+
""":func:`append_entries`, additionally handing back the text it published —
|
|
1462
|
+
or None when it wrote nothing, because every spec deduped or `specs` was
|
|
1463
|
+
empty.
|
|
1464
|
+
|
|
1465
|
+
For a caller that has to record WHAT IT WROTE rather than what the file holds
|
|
1466
|
+
afterwards. Reading the ledger back after this returns is a different
|
|
1467
|
+
question with the same answer only when nobody else wrote in between: the
|
|
1468
|
+
lock is released before the read, so a concurrent mutator's bytes would be
|
|
1469
|
+
folded into the caller's own anchor. That matters for
|
|
1470
|
+
``post_engine_ledger_digest``, whose whole job is to say "these bytes are
|
|
1471
|
+
ours" — counting a rival's write as ours would have the pre-harvest restore
|
|
1472
|
+
retract it, which is the loss this module exists to prevent (#286). Taking
|
|
1473
|
+
the text from inside the hold removes the window rather than narrowing it.
|
|
1474
|
+
|
|
1475
|
+
The returned text is what was handed to
|
|
1476
|
+
:func:`~froid_loop.platform_util.atomic_write_text`, so a digest of it equals
|
|
1477
|
+
a digest of a later ``read_text`` of the file: the writer's text mode
|
|
1478
|
+
translates the newlines on the way out and ``read_text`` normalizes them back
|
|
1479
|
+
on the way in.
|
|
1480
|
+
|
|
1481
|
+
ONE locked read->edit->write: the whole cycle runs under the cross-process
|
|
1482
|
+
ledger lock (#286/#469), so concurrent mutators — a second run, a sweep, the
|
|
1483
|
+
TUI decision modal, ``sweep --archive`` — serialize here rather than trading
|
|
1484
|
+
last-write-wins. The hold spans every `next_seq` mint as well as every
|
|
1485
|
+
idempotence scan, which is what stops two concurrent appenders reading the
|
|
1486
|
+
same highest id and both minting it (#469).
|
|
1487
|
+
|
|
1488
|
+
Byte-identical to a serial :func:`append_entry` loop over the same specs,
|
|
1489
|
+
because each spec is applied to the text the previous one produced rather
|
|
1490
|
+
than to the text this call read. That is what a naive batch gets wrong: minted
|
|
1491
|
+
against the original text, every spec in one call would claim the same id.
|
|
1492
|
+
|
|
1493
|
+
ALL specs are validated — the `status` and `severity` enumerations, which are
|
|
1494
|
+
orchestrator-owned and so raise rather than sanitize — before the lock is
|
|
1495
|
+
taken and before anything is written. All-or-nothing: a bad spec anywhere in
|
|
1496
|
+
the sequence leaves the ledger exactly as it was, rather than committing the
|
|
1497
|
+
prefix that happened to precede it. Validating above the lock also means a
|
|
1498
|
+
programmer bug reports itself without first waiting on another process.
|
|
1499
|
+
|
|
1500
|
+
Nothing is written when every spec dedupes, and no lock is taken either
|
|
1501
|
+
(#736): a replayed defer is answered from one advisory read that runs
|
|
1502
|
+
:func:`_apply_appends`, the same helper the locked pass runs, so it leaves
|
|
1503
|
+
the file untouched rather than rewriting it byte-for-byte. Deliberately NO
|
|
1504
|
+
missing-ledger guard, unlike its sibling mutators: an absent ledger here
|
|
1505
|
+
means CREATE, which is a write, and a write must take the lock.
|
|
1506
|
+
|
|
1507
|
+
The write goes through :func:`~froid_loop.platform_util.atomic_write_text` for
|
|
1508
|
+
the reasons documented on :func:`mark_done_many`, plus one this sibling shares
|
|
1509
|
+
with it: a bare ``Path.write_text`` truncates *before* it encodes, so any
|
|
1510
|
+
failure between the two — an unencodable value, ``ENOSPC``, ``EIO`` — leaves a
|
|
1511
|
+
zero-byte ledger where every entry used to be (#328).
|
|
1512
|
+
"""
|
|
1513
|
+
for spec in specs:
|
|
1514
|
+
_require_canonical_status(spec.status)
|
|
1515
|
+
# The whitelist is derived from the legacy parser's alias table (defined
|
|
1516
|
+
# below; resolved at call time) so what this writer emits and what
|
|
1517
|
+
# `field_severity` normalizes to cannot drift apart.
|
|
1518
|
+
if spec.severity and spec.severity not in _CANONICAL_SEVERITIES:
|
|
1519
|
+
raise ValueError(
|
|
1520
|
+
f"severity must be one of {sorted(_CANONICAL_SEVERITIES)}: {spec.severity!r}"
|
|
1521
|
+
)
|
|
1522
|
+
if not specs:
|
|
1523
|
+
# Nothing to serialize against, so nothing to take a lock for.
|
|
1524
|
+
return [], None
|
|
1525
|
+
try:
|
|
1526
|
+
# ADVISORY pre-lock probe (#736): one read — shaped exactly like the
|
|
1527
|
+
# locked one, absence included — and the same pure decision the locked
|
|
1528
|
+
# pass makes. Only a "would write nothing" answer is acted on, and here
|
|
1529
|
+
# that is every spec deduping, which is also the only case where the
|
|
1530
|
+
# published text is the text already on disk. Anything else, including
|
|
1531
|
+
# any fault here, falls through to the hold, which re-reads and decides.
|
|
1532
|
+
probe = path.read_text(encoding="utf-8") if path.is_file() else ""
|
|
1533
|
+
minted = _apply_appends(probe, specs)[1]
|
|
1534
|
+
if all(dw_id is None for dw_id in minted):
|
|
1535
|
+
return minted, None
|
|
1536
|
+
except Exception: # nosec B110 - ADVISORY probe: a fault here must decide nothing
|
|
1537
|
+
pass
|
|
1538
|
+
with ledger_lock(path):
|
|
1539
|
+
text = path.read_text(encoding="utf-8") if path.is_file() else ""
|
|
1540
|
+
text, minted = _apply_appends(text, specs)
|
|
1541
|
+
if all(dw_id is None for dw_id in minted):
|
|
1542
|
+
return minted, None
|
|
1543
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
1544
|
+
atomic_write_text(path, text)
|
|
1545
|
+
# Returned from INSIDE the hold: this is the published text by
|
|
1546
|
+
# construction, not a read-back that a rival could have moved.
|
|
1547
|
+
return minted, text
|
|
1548
|
+
|
|
1549
|
+
|
|
1550
|
+
def append_entry(
|
|
1551
|
+
path: Path,
|
|
1552
|
+
*,
|
|
1553
|
+
title: str,
|
|
1554
|
+
origin: str,
|
|
1555
|
+
source_spec: str,
|
|
1556
|
+
reason: str,
|
|
1557
|
+
location: str = "n/a",
|
|
1558
|
+
status: str = "open",
|
|
1559
|
+
severity: str | None = None,
|
|
1560
|
+
) -> str | None:
|
|
1561
|
+
"""Append a new canonical `### DW-<seq>` entry numbered past the highest
|
|
1562
|
+
existing DW id, returning the new id (e.g. "DW-42").
|
|
1563
|
+
|
|
1564
|
+
Idempotent: returns None without writing when an open entry already carries
|
|
1565
|
+
the same `origin:` marker and `source_spec:` — so re-running the same defer
|
|
1566
|
+
(e.g. a second sweep of the same story) never duplicates the entry. Creates
|
|
1567
|
+
the ledger (and parent dir) if it does not yet exist.
|
|
1568
|
+
|
|
1569
|
+
The one-spec case of :func:`append_entries`, which is where the lock is taken
|
|
1570
|
+
and the contract documented. It delegates rather than duplicating the
|
|
1571
|
+
read->edit->write so that one public call is exactly one acquisition."""
|
|
1572
|
+
return append_entries(
|
|
1573
|
+
path,
|
|
1574
|
+
[
|
|
1575
|
+
EntrySpec(
|
|
1576
|
+
title=title,
|
|
1577
|
+
origin=origin,
|
|
1578
|
+
source_spec=source_spec,
|
|
1579
|
+
reason=reason,
|
|
1580
|
+
location=location,
|
|
1581
|
+
status=status,
|
|
1582
|
+
severity=severity,
|
|
1583
|
+
)
|
|
1584
|
+
],
|
|
1585
|
+
)[0]
|
|
1586
|
+
|
|
1587
|
+
|
|
1588
|
+
ARCHIVE_REL = "deferred-work-archive.md"
|
|
1589
|
+
# The archive sibling is never locked in its own right: :func:`archive_closed`
|
|
1590
|
+
# is the only writer, and it holds the LEDGER's :func:`ledger_lock` across both
|
|
1591
|
+
# writes (#286/#469). Any future writer of this file must take that same lock.
|
|
1592
|
+
# A stub left by a prior archive_closed run carries this field. The next run
|
|
1593
|
+
# reads it to skip entries whose body has already been moved — without it,
|
|
1594
|
+
# every run would re-archive the stub (a heading + status line) and the
|
|
1595
|
+
# archive would accumulate duplicates.
|
|
1596
|
+
_ARCHIVED_FIELD_RE = re.compile(r"^archived:", re.MULTILINE)
|
|
1597
|
+
|
|
1598
|
+
# What :func:`mark_open` leaves where that stamp was. A reopened entry is not
|
|
1599
|
+
# archived — its body is back in the ledger — but the body the undone close
|
|
1600
|
+
# moved out still is, and this line is what a triage session follows to it.
|
|
1601
|
+
# Deliberately a different field name: `archived:` means "the body is
|
|
1602
|
+
# elsewhere", which a reopened entry must not claim, and a line matching
|
|
1603
|
+
# `_ARCHIVED_FIELD_RE` here would rebuild the exact `_STUB_BODY_RE` shape on
|
|
1604
|
+
# the next reopenable close.
|
|
1605
|
+
_ARCHIVED_BODY_FIELD = "archived-body:"
|
|
1606
|
+
|
|
1607
|
+
|
|
1608
|
+
def _archived_line_spans(entry: DWEntry) -> list[tuple[int, int]]:
|
|
1609
|
+
"""Body-relative spans of the entry's live ``archived:`` field lines, each
|
|
1610
|
+
covering the whole line including its terminating newline.
|
|
1611
|
+
|
|
1612
|
+
Reads through :func:`_quoted` for the same reason every gate scan does:
|
|
1613
|
+
an entry documenting the archive field in a fenced example carries the
|
|
1614
|
+
line in column 0, right where the anchor looks, and without the fence
|
|
1615
|
+
check a quoted ``archived:`` would be mistaken for the real thing. The one
|
|
1616
|
+
place that rule is written, so the three questions asked about the field —
|
|
1617
|
+
is this entry archived, what does its body say apart from the stamp, and
|
|
1618
|
+
which bytes must a reopen rename — cannot answer it differently.
|
|
1619
|
+
|
|
1620
|
+
Whole lines rather than match starts because both cutting callers remove
|
|
1621
|
+
the line, and a span ending at the anchor would leave the stamp's value
|
|
1622
|
+
behind as orphaned text.
|
|
1623
|
+
"""
|
|
1624
|
+
spans: list[tuple[int, int]] = []
|
|
1625
|
+
for m in _ARCHIVED_FIELD_RE.finditer(entry.body):
|
|
1626
|
+
if _quoted(entry, m.start()):
|
|
1627
|
+
continue
|
|
1628
|
+
line_end = entry.body.find("\n", m.end())
|
|
1629
|
+
spans.append((m.start(), len(entry.body) if line_end == -1 else line_end + 1))
|
|
1630
|
+
return spans
|
|
1631
|
+
|
|
1632
|
+
|
|
1633
|
+
def _is_archived(entry: DWEntry) -> bool:
|
|
1634
|
+
"""Whether the entry carries a live ``archived:`` field line (not a quoted
|
|
1635
|
+
example), marking it as touched by :func:`archive_closed` — a stub in the
|
|
1636
|
+
live ledger, or an archived body in the archive file.
|
|
1637
|
+
"""
|
|
1638
|
+
return bool(_archived_line_spans(entry))
|
|
1639
|
+
|
|
1640
|
+
|
|
1641
|
+
def _body_without_archived(entry: DWEntry) -> str:
|
|
1642
|
+
"""The entry's body with its live ``archived:`` stamps and trailing blank
|
|
1643
|
+
lines removed — the comparison key for :func:`archive_closed`'s
|
|
1644
|
+
crash-recovery skip.
|
|
1645
|
+
|
|
1646
|
+
An archived twin is its ledger entry plus exactly one ``archived:`` line,
|
|
1647
|
+
so the two are the same content only once that line is discounted; trailing
|
|
1648
|
+
newlines go with it because they record where the entry sat in its file,
|
|
1649
|
+
not what it says. Everything else is compared verbatim, deliberately: the
|
|
1650
|
+
cheap wrong answer is archiving a body twice, and the expensive one is
|
|
1651
|
+
deciding a divergent re-closure was already saved and dropping it (#711).
|
|
1652
|
+
"""
|
|
1653
|
+
body = entry.body
|
|
1654
|
+
for start, end in reversed(_archived_line_spans(entry)):
|
|
1655
|
+
body = body[:start] + body[end:]
|
|
1656
|
+
return body.rstrip("\n")
|
|
1657
|
+
|
|
1658
|
+
|
|
1659
|
+
def _archived_stamp(entry: DWEntry) -> str | None:
|
|
1660
|
+
"""The value of the entry's first live ``archived:`` field line, or None
|
|
1661
|
+
when it carries none.
|
|
1662
|
+
|
|
1663
|
+
Read from an *archive* twin, this is what a stub pointing at that block
|
|
1664
|
+
must carry — and what :func:`mark_open` demotes into an `archived-body:`
|
|
1665
|
+
pointer. The archive holds several blocks per id by design, so the stamp
|
|
1666
|
+
narrows rather than identifies: two closures archived on one day share it,
|
|
1667
|
+
and the append-only file's order is the tie-break (later block, later
|
|
1668
|
+
closure).
|
|
1669
|
+
"""
|
|
1670
|
+
spans = _archived_line_spans(entry)
|
|
1671
|
+
if not spans:
|
|
1672
|
+
return None
|
|
1673
|
+
start, end = spans[0]
|
|
1674
|
+
return entry.body[start:end].split(":", 1)[1].strip()
|
|
1675
|
+
|
|
1676
|
+
|
|
1677
|
+
# Field lines a stub must carry when the archived body had them, because
|
|
1678
|
+
# downstream readers key on them regardless of status: `gate:` (validate's
|
|
1679
|
+
# closed-entry gate report deliberately keeps speaking), `origin:` +
|
|
1680
|
+
# `source_spec:` (the engine's status-agnostic harvest-replay dedupe), and the
|
|
1681
|
+
# reopenable-close undo tail (`mark_open`'s adjacency requirement).
|
|
1682
|
+
_PRESERVED_FIELD_RE = re.compile(r"^(gate:.*|origin:.*|source_spec:.*)$", re.MULTILINE)
|
|
1683
|
+
|
|
1684
|
+
# The exact stub shape :func:`archive_closed` leaves in the live ledger.
|
|
1685
|
+
# A done entry that merely carries a hand-written `archived:` line does NOT
|
|
1686
|
+
# match — it is a real entry, not a stub, and must still be archived.
|
|
1687
|
+
_STUB_BODY_RE = re.compile(
|
|
1688
|
+
r"### .*: .*\n\n"
|
|
1689
|
+
r"status: done [0-9]{4}-[0-9]{2}-[0-9]{2}\n"
|
|
1690
|
+
# Separators mirror `_MARK_DONE_TAIL_RE`, which tolerates tabs: that regex
|
|
1691
|
+
# decides what `_preserved_stub_lines` copies into the stub verbatim, so a
|
|
1692
|
+
# stricter shape here reads a stub this module just wrote as a live entry
|
|
1693
|
+
# and re-archives it on every run, forever, appending nothing (#711).
|
|
1694
|
+
r"(?:resolution:[ \t]*[^\n]*\nresolution-undo:[ \t]*[0-9a-f]{64}[ \t]+[^\n]*\n)?"
|
|
1695
|
+
r"(?:(?:gate:|origin:|source_spec:)[^\n]*\n)*"
|
|
1696
|
+
r"archived: [^\n]*\n"
|
|
1697
|
+
r"\n?"
|
|
1698
|
+
)
|
|
1699
|
+
|
|
1700
|
+
|
|
1701
|
+
def _is_stub(entry: DWEntry) -> bool:
|
|
1702
|
+
"""Whether the entry is a stub left by a prior :func:`archive_closed` run.
|
|
1703
|
+
|
|
1704
|
+
Shape-based rather than `archived:`-line-based: a done entry a human
|
|
1705
|
+
annotated with a stray unfenced ``archived:`` line is a real entry whose
|
|
1706
|
+
body still belongs in the live ledger — skipping it forever on the strength
|
|
1707
|
+
of one line would silently exclude it from every future archive.
|
|
1708
|
+
"""
|
|
1709
|
+
return entry.done and _STUB_BODY_RE.fullmatch(entry.body.rstrip("\n") + "\n") is not None
|
|
1710
|
+
|
|
1711
|
+
|
|
1712
|
+
def _preserved_stub_lines(entry: DWEntry) -> list[str]:
|
|
1713
|
+
"""The load-bearing field lines a stub must keep from the archived body.
|
|
1714
|
+
|
|
1715
|
+
Scanned fence-aware like every field read in this module: a fenced example
|
|
1716
|
+
documenting `origin:` is not a declaration. The undo tail is read with the
|
|
1717
|
+
same adjacency regex :func:`mark_open` will later use against the stub, so
|
|
1718
|
+
what qualifies here is exactly what remains undoable there.
|
|
1719
|
+
"""
|
|
1720
|
+
lines = [
|
|
1721
|
+
entry.body[m.start() : m.end()]
|
|
1722
|
+
for m in _PRESERVED_FIELD_RE.finditer(entry.body)
|
|
1723
|
+
if not _quoted(entry, m.start())
|
|
1724
|
+
]
|
|
1725
|
+
if entry.status_span is not None:
|
|
1726
|
+
tail = _MARK_DONE_TAIL_RE.match(entry.body, entry.status_span[1])
|
|
1727
|
+
if tail is not None:
|
|
1728
|
+
lines = [tail.group(0).lstrip("\n")] + lines
|
|
1729
|
+
return lines
|
|
1730
|
+
|
|
1731
|
+
|
|
1732
|
+
def _close_date(entry: DWEntry) -> str | None:
|
|
1733
|
+
"""The ISO close date from a ``done <date>`` status, or None when the
|
|
1734
|
+
entry is not done, is done without a date suffix, or carries a date
|
|
1735
|
+
that does not match the ISO ``YYYY-MM-DD`` shape.
|
|
1736
|
+
|
|
1737
|
+
Entries closed with a bare ``status: done`` (no date) or a hand-edited
|
|
1738
|
+
non-ISO date are skipped by :func:`archive_closed`: there is no close
|
|
1739
|
+
date to compare against a ``--before`` cutoff, and the stub the function
|
|
1740
|
+
leaves in the ledger needs one to stay readable as done.
|
|
1741
|
+
"""
|
|
1742
|
+
if not entry.done:
|
|
1743
|
+
return None
|
|
1744
|
+
parts = entry.status.split()
|
|
1745
|
+
if len(parts) != 2: # exactly `done YYYY-MM-DD` — extra tokens are not a close date
|
|
1746
|
+
return None
|
|
1747
|
+
# Same shape check as `_require_iso_date` (well-formed regex AND a real
|
|
1748
|
+
# calendar day), skip-not-raise: a hand-edited close is data, not a bug.
|
|
1749
|
+
return _iso_date_or_none(parts[1])
|
|
1750
|
+
|
|
1751
|
+
|
|
1752
|
+
def _eligible_for_archive(text: str, before: str | None) -> list[tuple[DWEntry, str]]:
|
|
1753
|
+
"""Every entry in `text` :func:`archive_closed` would move, paired with its
|
|
1754
|
+
close date, in ledger order.
|
|
1755
|
+
|
|
1756
|
+
Pure — text in, entries out, no `Path` and no I/O — and ONE body for the
|
|
1757
|
+
advisory pre-lock probe and the locked pass, so the two cannot drift. Three
|
|
1758
|
+
skips make up the decision: an entry that is not done, or done without a
|
|
1759
|
+
date, has nothing to compare or to stamp a stub with; `before` excludes
|
|
1760
|
+
entries closed on or after the cutoff; and a stub from a prior run is
|
|
1761
|
+
already archived."""
|
|
1762
|
+
to_archive: list[tuple[DWEntry, str]] = []
|
|
1763
|
+
for entry in parse_ledger(text):
|
|
1764
|
+
close_date = _close_date(entry)
|
|
1765
|
+
if close_date is None:
|
|
1766
|
+
continue # not done, or done without a date
|
|
1767
|
+
if before is not None and close_date >= before:
|
|
1768
|
+
continue # closed on or after the cutoff
|
|
1769
|
+
if _is_stub(entry):
|
|
1770
|
+
continue # stub from a prior archive_closed run
|
|
1771
|
+
to_archive.append((entry, close_date))
|
|
1772
|
+
return to_archive
|
|
1773
|
+
|
|
1774
|
+
|
|
1775
|
+
def archive_closed(
|
|
1776
|
+
path: Path,
|
|
1777
|
+
*,
|
|
1778
|
+
before: str | None = None,
|
|
1779
|
+
archive_date: str | None = None,
|
|
1780
|
+
dry_run: bool = False,
|
|
1781
|
+
) -> list[str]:
|
|
1782
|
+
"""Move closed (``status: done <date>``) ledger entries to a sibling
|
|
1783
|
+
archive file (:data:`ARCHIVE_REL`), replacing each with a minimal stub
|
|
1784
|
+
that preserves the DW- id for grep and ``closes_deferred``
|
|
1785
|
+
cross-references.
|
|
1786
|
+
|
|
1787
|
+
Returns the list of archived ids, in ledger order. ``dry_run=True``
|
|
1788
|
+
returns the ids that *would* be archived without writing anything.
|
|
1789
|
+
|
|
1790
|
+
Each archived entry's body is preserved verbatim in the archive file,
|
|
1791
|
+
with an ``archived: <date>`` field line appended after the entry's status
|
|
1792
|
+
line. The stub left in the live ledger keeps the heading, a ``status:
|
|
1793
|
+
done <date>`` line (so :func:`parse_ledger` reads it as done and
|
|
1794
|
+
:func:`open_ids` drops it), an ``archived: <date>`` line (so a subsequent
|
|
1795
|
+
run skips it rather than re-archiving the stub), and the entry's
|
|
1796
|
+
load-bearing field lines — ``gate:``, ``origin:``/``source_spec:``, and
|
|
1797
|
+
the reopenable-close undo tail — because downstream readers key on those
|
|
1798
|
+
regardless of status (validate's closed-gate report, the engine's
|
|
1799
|
+
harvest-replay dedupe, and sweep bundle rollback respectively).
|
|
1800
|
+
|
|
1801
|
+
``before`` (ISO ``YYYY-MM-DD``) archives only entries closed strictly
|
|
1802
|
+
*before* that date. Entries with ``status: done`` (no date) are always
|
|
1803
|
+
skipped — there is no close date to compare against a cutoff or to stamp
|
|
1804
|
+
the stub with. Open and legacy entries are never touched.
|
|
1805
|
+
|
|
1806
|
+
Dates are validated with :func:`_require_iso_date` (same validation as
|
|
1807
|
+
the existing close-path writers), ahead of the ``is_file`` short-circuit
|
|
1808
|
+
so a programmer bug fails the same way whether or not a ledger exists.
|
|
1809
|
+
Both writes — the trimmed ledger and the appended archive — go through
|
|
1810
|
+
:func:`atomic_write_text`, the same primitive every ledger writer uses.
|
|
1811
|
+
The archive file accumulates on repeat runs: new entries are appended to
|
|
1812
|
+
the existing file, never overwritten, and stubs from a prior run are
|
|
1813
|
+
skipped by their exact stub shape. A stub's ``archived:`` date names the
|
|
1814
|
+
archive block holding its body, so an entry recovered from a crashed run
|
|
1815
|
+
is stamped with the date already on that block rather than with this run's.
|
|
1816
|
+
|
|
1817
|
+
The whole read->edit->write runs under the cross-process ledger lock
|
|
1818
|
+
(#286/#469): concurrent mutators — a second run, a sweep, the TUI decision
|
|
1819
|
+
modal, ``sweep --archive`` — serialize here rather than trading
|
|
1820
|
+
last-write-wins. ONE acquisition spans BOTH writes — the
|
|
1821
|
+
archive sibling has no lock of its own precisely because it is only ever
|
|
1822
|
+
written under its ledger's lock — and an ELIGIBLE ``dry_run`` runs inside
|
|
1823
|
+
the hold too, so there is one code path rather than a locked and an unlocked
|
|
1824
|
+
one. A run with nothing eligible is the exception, and only because it is
|
|
1825
|
+
not a code path at all: the advisory pre-lock probe (#736) answers it with
|
|
1826
|
+
the empty list before either branch is reached, so ``sweep --archive`` over
|
|
1827
|
+
a ledger holding nothing closed keeps reporting success where the state root
|
|
1828
|
+
cannot be derived or the lock cannot be taken.
|
|
1829
|
+
"""
|
|
1830
|
+
if before is not None:
|
|
1831
|
+
_require_iso_date(before)
|
|
1832
|
+
if archive_date is not None:
|
|
1833
|
+
_require_iso_date(archive_date)
|
|
1834
|
+
if not path.is_file():
|
|
1835
|
+
# No ledger means no write, and so no lock — the order
|
|
1836
|
+
# `sprintstatus.advance` already keeps for its own missing-board case.
|
|
1837
|
+
# Acquiring first would turn "there is nothing to archive", which
|
|
1838
|
+
# `froid-loop sweep --archive` reports as SUCCESS, into a failure wherever
|
|
1839
|
+
# the state root cannot be derived: a released behavior, changed by a lock
|
|
1840
|
+
# taken for a file that is not there. Rechecked under the hold below,
|
|
1841
|
+
# deletion being able to race this answer.
|
|
1842
|
+
return []
|
|
1843
|
+
try:
|
|
1844
|
+
# ADVISORY pre-lock probe (#736): one read, and the same pure decision
|
|
1845
|
+
# the locked pass makes. Only a "would write nothing" answer is acted on
|
|
1846
|
+
# — the call then serializes at this read. Above the `dry_run` branch on
|
|
1847
|
+
# purpose, so a nothing-eligible dry run skips the lock too; an ELIGIBLE
|
|
1848
|
+
# dry run still runs under the hold, where the one code path is. Anything
|
|
1849
|
+
# else, including any fault here, falls through to that hold.
|
|
1850
|
+
probe = path.read_text(encoding="utf-8")
|
|
1851
|
+
if not _eligible_for_archive(probe, before):
|
|
1852
|
+
return []
|
|
1853
|
+
except Exception: # nosec B110 - ADVISORY probe: a fault here must decide nothing
|
|
1854
|
+
pass
|
|
1855
|
+
with ledger_lock(path):
|
|
1856
|
+
if not path.is_file():
|
|
1857
|
+
return []
|
|
1858
|
+
text = path.read_text(encoding="utf-8")
|
|
1859
|
+
to_archive = _eligible_for_archive(text, before)
|
|
1860
|
+
if not to_archive:
|
|
1861
|
+
return []
|
|
1862
|
+
archived_ids = [e.id for e, _ in to_archive]
|
|
1863
|
+
if dry_run:
|
|
1864
|
+
return archived_ids
|
|
1865
|
+
stamp = archive_date or calendar_date.today().isoformat()
|
|
1866
|
+
archive_path = path.parent / ARCHIVE_REL
|
|
1867
|
+
existing = archive_path.read_text(encoding="utf-8") if archive_path.is_file() else ""
|
|
1868
|
+
# Append an `archived:` line after each entry's status line. The status
|
|
1869
|
+
# span is body-relative, so the insertion works within the body slice —
|
|
1870
|
+
# same offset math as `_insert_after_status`, applied to the body.
|
|
1871
|
+
#
|
|
1872
|
+
# Crash recovery: the archive is written BEFORE the ledger (see below), so
|
|
1873
|
+
# a crash between the two writes leaves the ledger with full entries whose
|
|
1874
|
+
# bodies are already in the archive. A retry must still stub those ledger
|
|
1875
|
+
# entries (completing the interrupted operation) but must NOT append their
|
|
1876
|
+
# bodies again — an append-only archive accumulating duplicates. Entries
|
|
1877
|
+
# whose parsed archive twin carries a live (non-fenced) ``archived:``
|
|
1878
|
+
# field are therefore skipped here and only replaced with stubs below.
|
|
1879
|
+
#
|
|
1880
|
+
# The twin must match in BODY, not merely in id and close date. A DW id is
|
|
1881
|
+
# reusable across closures (`mark_open` reopens, a re-close follows) and a
|
|
1882
|
+
# closed entry still accepts writes (`append_decision` does not read
|
|
1883
|
+
# status), so id + date names a *closure slot*, not its content: reopened
|
|
1884
|
+
# and re-closed the same day with a new resolution, or annotated with a
|
|
1885
|
+
# decision after its body was archived, the ledger entry and its twin
|
|
1886
|
+
# differ. Skipping on the slot alone stubbed that entry over its own
|
|
1887
|
+
# content while reporting the id as archived — the body reached neither
|
|
1888
|
+
# file (#711). A body that differs is appended instead; the archive holds
|
|
1889
|
+
# several blocks per id by design, and over-archiving is recoverable where
|
|
1890
|
+
# a silent drop is not.
|
|
1891
|
+
archive_blocks: list[str] = []
|
|
1892
|
+
already_archived = {
|
|
1893
|
+
e.id: ((_close_date(e), _body_without_archived(e)), _archived_stamp(e))
|
|
1894
|
+
for e in parse_ledger(existing)
|
|
1895
|
+
if _is_archived(e)
|
|
1896
|
+
} # fence-aware: a quoted example in the archive is not a real body
|
|
1897
|
+
# A recovered entry's stub is stamped with the date already on its archived
|
|
1898
|
+
# body, not with this run's. The two diverge whenever the retry lands on a
|
|
1899
|
+
# later day than the crashed run, and the stamp is not decoration: it is
|
|
1900
|
+
# what picks one of an id's several archive blocks — for a reader following
|
|
1901
|
+
# the stub, and for the `archived-body:` pointer `mark_open` demotes that
|
|
1902
|
+
# stamp into, which is a reopened entry's only route back to its body
|
|
1903
|
+
# (#711 review). A stub naming a date no block carries resolves to nothing.
|
|
1904
|
+
recovered_stamps: dict[str, str] = {}
|
|
1905
|
+
for entry, close_date in to_archive:
|
|
1906
|
+
twin = already_archived.get(entry.id)
|
|
1907
|
+
if twin is not None and twin[0] == (close_date, _body_without_archived(entry)):
|
|
1908
|
+
# this closure's body is already archived (crashed prior run)
|
|
1909
|
+
if twin[1] is not None:
|
|
1910
|
+
recovered_stamps[entry.id] = twin[1]
|
|
1911
|
+
continue
|
|
1912
|
+
body = entry.body
|
|
1913
|
+
assert entry.status_span is not None # done with a date implies a status line
|
|
1914
|
+
pos = entry.status_span[1]
|
|
1915
|
+
body = body[:pos] + f"\narchived: {stamp}" + body[pos:]
|
|
1916
|
+
archive_blocks.append(body)
|
|
1917
|
+
# Appended, never prepended: for one id the file's order is closure order,
|
|
1918
|
+
# which is the documented tie-break when two closures were archived on the
|
|
1919
|
+
# same day and so carry the same stamp (#711 review).
|
|
1920
|
+
if archive_blocks:
|
|
1921
|
+
if existing == "" or existing.endswith("\n\n"):
|
|
1922
|
+
sep = ""
|
|
1923
|
+
elif existing.endswith("\n"):
|
|
1924
|
+
sep = "\n"
|
|
1925
|
+
else:
|
|
1926
|
+
sep = "\n\n"
|
|
1927
|
+
archive_content = existing + sep + "".join(archive_blocks)
|
|
1928
|
+
else:
|
|
1929
|
+
archive_content = existing # pure crash-recovery pass: only stub the ledger
|
|
1930
|
+
# Replace each archived entry's span with a stub, working backwards so
|
|
1931
|
+
# earlier spans are unaffected by later replacements — the same
|
|
1932
|
+
# text-surgery pattern as `_apply_done`, applied to multiple entries.
|
|
1933
|
+
for entry, close_date in reversed(to_archive):
|
|
1934
|
+
preserved = "".join(f"{line}\n" for line in _preserved_stub_lines(entry))
|
|
1935
|
+
stub = (
|
|
1936
|
+
f"### {entry.id}: {entry.title}\n\n"
|
|
1937
|
+
f"status: done {close_date}\n"
|
|
1938
|
+
f"{preserved}"
|
|
1939
|
+
f"archived: {recovered_stamps.get(entry.id, stamp)}\n\n"
|
|
1940
|
+
)
|
|
1941
|
+
start, end = entry.span
|
|
1942
|
+
text = text[:start] + stub + text[end:]
|
|
1943
|
+
# Write the archive BEFORE the ledger: a crash between writes leaves the
|
|
1944
|
+
# archive with extra content (harmless — the archive is append-only) and
|
|
1945
|
+
# the ledger unchanged (safe — the bodies are still in the live file).
|
|
1946
|
+
# Writing the ledger first would leave stubs in the ledger with no bodies
|
|
1947
|
+
# in the archive — content lost.
|
|
1948
|
+
atomic_write_text(archive_path, archive_content)
|
|
1949
|
+
atomic_write_text(path, text)
|
|
1950
|
+
return archived_ids
|
|
1951
|
+
|
|
1952
|
+
|
|
1953
|
+
# ------------------------------------------------------------------- legacy
|
|
1954
|
+
#
|
|
1955
|
+
# Ledgers written before the DW format (older FROID-method projects) are
|
|
1956
|
+
# freeform markdown: "## Deferred from: ..." sections holding id'd or
|
|
1957
|
+
# strikethrough bullets, "### D-1.2-003: title — RESOLVED" entry headings,
|
|
1958
|
+
# topic sections closed with "(... — DONE)". parse_legacy() reads them
|
|
1959
|
+
# tolerantly so the TUI can display them and a sweep can migrate them; the
|
|
1960
|
+
# strict DW contract above is untouched — legacy items have no status line
|
|
1961
|
+
# to flip, so mark_done/open_ids never see them.
|
|
1962
|
+
|
|
1963
|
+
# Severity is extracted forgivingly (the ledger is LLM-written): a
|
|
1964
|
+
# `severity:`/`priority:` field line in any case, plain or bold-bulleted
|
|
1965
|
+
# ("- **Severity:** high"), common synonyms accepted.
|
|
1966
|
+
SEVERITY_ALIASES = {
|
|
1967
|
+
"critical": "critical",
|
|
1968
|
+
"blocker": "critical",
|
|
1969
|
+
"high": "high",
|
|
1970
|
+
"major": "high",
|
|
1971
|
+
"medium": "medium",
|
|
1972
|
+
"med": "medium",
|
|
1973
|
+
"moderate": "medium",
|
|
1974
|
+
"low": "low",
|
|
1975
|
+
"minor": "low",
|
|
1976
|
+
"trivial": "low",
|
|
1977
|
+
}
|
|
1978
|
+
# What every alias above normalizes to, and so the only values `append_entry` may
|
|
1979
|
+
# write. Derived rather than restated: a hand-copied whitelist drifts the moment
|
|
1980
|
+
# an alias is added for a new canonical level.
|
|
1981
|
+
_CANONICAL_SEVERITIES = frozenset(SEVERITY_ALIASES.values())
|
|
1982
|
+
|
|
1983
|
+
SEVERITY_FIELD_RE = re.compile(
|
|
1984
|
+
r"^[ \t]*(?:[-*][ \t]+)?(?:\*\*)?(?:severity|priority)[ \t]*:[ \t]*(?:\*\*)?[ \t]*"
|
|
1985
|
+
r"([A-Za-z][\w-]*)",
|
|
1986
|
+
re.IGNORECASE | re.MULTILINE,
|
|
1987
|
+
)
|
|
1988
|
+
|
|
1989
|
+
|
|
1990
|
+
def field_severity(body: str) -> str | None:
|
|
1991
|
+
m = SEVERITY_FIELD_RE.search(body)
|
|
1992
|
+
return SEVERITY_ALIASES.get(m.group(1).lower()) if m else None
|
|
1993
|
+
|
|
1994
|
+
|
|
1995
|
+
@dataclass(frozen=True)
|
|
1996
|
+
class LegacyEntry:
|
|
1997
|
+
key: str # stable content-derived identity, unique within the file
|
|
1998
|
+
id: str # native id ("W2", "D-CAP-001", "0-1"), "" when the item has none
|
|
1999
|
+
title: str # cleaned one-line title (markers/strikethrough stripped)
|
|
2000
|
+
done: bool
|
|
2001
|
+
severity: str | None # normalized critical/high/medium/low, None unknown
|
|
2002
|
+
body: str # the bullet/heading block verbatim
|
|
2003
|
+
section: str # enclosing ##/### heading text, "" at top level
|
|
2004
|
+
span: tuple[int, int] # char offsets in the ledger text
|
|
2005
|
+
|
|
2006
|
+
|
|
2007
|
+
_DONE_WORDS = r"(?:DONE|RESOLVED|CLOSED|VERIFIED|DOCUMENTED|FIXED)"
|
|
2008
|
+
_LINE_HEADING_RE = re.compile(r"^(#{1,6})[ \t]+(.*?)[ \t]*$")
|
|
2009
|
+
# a single whitespace-free digit-bearing token before ":" or "—" makes a
|
|
2010
|
+
# heading an entry ("### D-CAP-001: title", "## D-8.6-001 — title");
|
|
2011
|
+
# "## Epic 0: ..." has a space and "## 2026-06-09 — ..." is a date, so: section
|
|
2012
|
+
_ENTRY_HEADING_RE = re.compile(r"^(~~)?([^\s:*~]*\d[^\s:*~]*)(?::[ \t]+|[ \t]+[—–][ \t]+)(.+)$")
|
|
2013
|
+
_DATE_TOKEN_RE = re.compile(r"\d{4}-\d{2}-\d{2}")
|
|
2014
|
+
_SECTION_DONE_RE = re.compile(rf"(?:—|–|-|\()[ \t]*{_DONE_WORDS}\b[^)]*\)?[ \t]*$")
|
|
2015
|
+
_TITLE_DONE_SUFFIX_RE = re.compile(rf"[ \t]*(?:—|–|-)[ \t]*{_DONE_WORDS}[ \t]*$")
|
|
2016
|
+
_BARE_DONE_SUFFIX_RE = re.compile(rf"[ \t]*{_DONE_WORDS}\b.*$")
|
|
2017
|
+
_BOLD_DONE_RE = re.compile(rf"\*\*{_DONE_WORDS}\b[^*]*\*\*")
|
|
2018
|
+
_BRACKET_DONE_RE = re.compile(rf"\[{_DONE_WORDS}\]")
|
|
2019
|
+
_DONE_PREFIX_RE = re.compile(rf"^\*\*{_DONE_WORDS}\b[^*]*\*\*:?[ \t]*")
|
|
2020
|
+
# "- W-1.2-c — CLOSED: ..." / "CLOSED 2026-06-11 (story 1.11). ..."
|
|
2021
|
+
_LEAD_DONE_RE = re.compile(rf"^{_DONE_WORDS}\b")
|
|
2022
|
+
_LEAD_DONE_STRIP_RE = re.compile(
|
|
2023
|
+
rf"^{_DONE_WORDS}\b(?:[ \t]+\d{{4}}-\d{{2}}-\d{{2}})?(?:[ \t]*\([^)]*\))?[ \t]*[:.—–-]?[ \t]*"
|
|
2024
|
+
)
|
|
2025
|
+
# The flat review appender — pre-#2651 dev primitives and the attended
|
|
2026
|
+
# `froid-build` — writes a flat block per finding:
|
|
2027
|
+
# - source_spec: `spec-foo.md`
|
|
2028
|
+
# summary: <one sentence>
|
|
2029
|
+
# evidence: <why this is real>
|
|
2030
|
+
# We recognize it so the `summary` becomes the title (not the source_spec path)
|
|
2031
|
+
# and the entry migrates cleanly into the canonical `### DW-<seq>` shape. The
|
|
2032
|
+
# opening line comes from the same `_FLAT_SOURCE_BODY` as FLAT_ENTRY_RE, which
|
|
2033
|
+
# bounds canonical spans on it — see that constant for why they must not drift.
|
|
2034
|
+
_FLAT_SOURCE_RE = re.compile(rf"^{_FLAT_SOURCE_BODY}", re.IGNORECASE)
|
|
2035
|
+
_FLAT_SUMMARY_RE = re.compile(r"^[ \t]*summary:[ \t]*(.*)$", re.IGNORECASE | re.MULTILINE)
|
|
2036
|
+
_BULLET_RE = re.compile(r"^[-*][ \t]+(.*)$")
|
|
2037
|
+
_ITEM_ID_RE = re.compile(
|
|
2038
|
+
r"^(?:\*\*)?([^\s:*~]*\d[^\s:*~]*)(?:\*\*)?(?:[ \t]*[—–][ \t]+|:[ \t]+|[ \t]+-[ \t]+)"
|
|
2039
|
+
)
|
|
2040
|
+
_BRACKET_TOKEN_RE = re.compile(r"\[([A-Za-z]+)[^\]]*\]")
|
|
2041
|
+
_LEAD_BOLD_RE = re.compile(r"^\*\*(.+?)\*\*")
|
|
2042
|
+
_STRUCK_LINE_RE = re.compile(r"^~~(.*)~~")
|
|
2043
|
+
_TRAIL_BRACKET_RE = re.compile(r"[ \t]*\[[^\]]+\][ \t.]*$")
|
|
2044
|
+
|
|
2045
|
+
|
|
2046
|
+
def _bracket_severity(s: str) -> str | None:
|
|
2047
|
+
for m in _BRACKET_TOKEN_RE.finditer(s):
|
|
2048
|
+
sev = SEVERITY_ALIASES.get(m.group(1).lower())
|
|
2049
|
+
if sev:
|
|
2050
|
+
return sev
|
|
2051
|
+
return None
|
|
2052
|
+
|
|
2053
|
+
|
|
2054
|
+
def _clean_title(s: str) -> str:
|
|
2055
|
+
return " ".join(s.replace("**", "").split())
|
|
2056
|
+
|
|
2057
|
+
|
|
2058
|
+
def _item_entry(first: str, body: str, section: str, section_done: bool) -> dict:
|
|
2059
|
+
"""Interpret one bullet item; returns the pre-key entry fields."""
|
|
2060
|
+
if _FLAT_SOURCE_RE.match(first):
|
|
2061
|
+
# flat appender block (legacy/attended era): title is the `summary`
|
|
2062
|
+
sm = _FLAT_SUMMARY_RE.search(body)
|
|
2063
|
+
summary = sm.group(1).strip() if sm else ""
|
|
2064
|
+
return {
|
|
2065
|
+
"id": "",
|
|
2066
|
+
"title": _clean_title(summary) if summary else _clean_title(first),
|
|
2067
|
+
"done": section_done,
|
|
2068
|
+
"severity": field_severity(body),
|
|
2069
|
+
"section": section,
|
|
2070
|
+
}
|
|
2071
|
+
content = first
|
|
2072
|
+
struck = False
|
|
2073
|
+
m = _STRUCK_LINE_RE.match(content)
|
|
2074
|
+
if m: # "~~text~~ DONE" / "~~text~~ → resolution" on the first line
|
|
2075
|
+
struck = True
|
|
2076
|
+
content = m.group(1)
|
|
2077
|
+
elif content.startswith("~~") and "~~" in body[2:]:
|
|
2078
|
+
struck = True # strikethrough closes on a later line
|
|
2079
|
+
content = content[2:]
|
|
2080
|
+
item_id = ""
|
|
2081
|
+
m = _ITEM_ID_RE.match(content)
|
|
2082
|
+
if m:
|
|
2083
|
+
item_id = m.group(1)
|
|
2084
|
+
content = content[m.end() :]
|
|
2085
|
+
done = (
|
|
2086
|
+
struck
|
|
2087
|
+
or section_done
|
|
2088
|
+
or bool(_LEAD_DONE_RE.match(content))
|
|
2089
|
+
or bool(_BOLD_DONE_RE.search(body))
|
|
2090
|
+
or bool(_BRACKET_DONE_RE.search(body))
|
|
2091
|
+
)
|
|
2092
|
+
content = _DONE_PREFIX_RE.sub("", content)
|
|
2093
|
+
content = _LEAD_DONE_STRIP_RE.sub("", content)
|
|
2094
|
+
while True: # trailing "[MINOR]" / "[CLOSED]" tokens are not title text
|
|
2095
|
+
trimmed = _TRAIL_BRACKET_RE.sub("", content)
|
|
2096
|
+
if trimmed == content:
|
|
2097
|
+
break
|
|
2098
|
+
content = trimmed
|
|
2099
|
+
bold = _LEAD_BOLD_RE.match(content)
|
|
2100
|
+
if bold and len(bold.group(1).split()) >= 3:
|
|
2101
|
+
title = bold.group(1) # notey: the bold phrase is the title
|
|
2102
|
+
else:
|
|
2103
|
+
title = content
|
|
2104
|
+
return {
|
|
2105
|
+
"id": item_id,
|
|
2106
|
+
"title": _clean_title(title),
|
|
2107
|
+
"done": done,
|
|
2108
|
+
"severity": _bracket_severity(body) or field_severity(body),
|
|
2109
|
+
"section": section,
|
|
2110
|
+
}
|
|
2111
|
+
|
|
2112
|
+
|
|
2113
|
+
def _heading_entry(struck: bool, hid: str, rest: str, body: str, section: str) -> dict:
|
|
2114
|
+
"""Interpret one '### D-1: title' entry heading (story-maker shape)."""
|
|
2115
|
+
title = rest
|
|
2116
|
+
done = struck
|
|
2117
|
+
m = _TITLE_DONE_SUFFIX_RE.search(title)
|
|
2118
|
+
if m:
|
|
2119
|
+
done = True
|
|
2120
|
+
title = title[: m.start()]
|
|
2121
|
+
if struck:
|
|
2122
|
+
title = _BARE_DONE_SUFFIX_RE.sub("", title.replace("~~", ""))
|
|
2123
|
+
return {
|
|
2124
|
+
"id": hid,
|
|
2125
|
+
"title": _clean_title(title),
|
|
2126
|
+
"done": done,
|
|
2127
|
+
"severity": field_severity(body) or _bracket_severity(body),
|
|
2128
|
+
"section": section,
|
|
2129
|
+
}
|
|
2130
|
+
|
|
2131
|
+
|
|
2132
|
+
def parse_legacy(text: str) -> list[LegacyEntry]:
|
|
2133
|
+
"""Extract legacy (non-DW) deferred items. Canonical DW entries and fenced
|
|
2134
|
+
examples are masked out first, so mixed ledgers parse both ways without
|
|
2135
|
+
overlap and a quoted example contributes nothing to either reading.
|
|
2136
|
+
|
|
2137
|
+
The fenced half is not symmetry for its own sake. `parse_ledger` used to hand
|
|
2138
|
+
a quoted example over as a phantom canonical entry, whose span masked the
|
|
2139
|
+
example here by accident; once it stopped doing that, the same quotation
|
|
2140
|
+
surfaced on this side instead — a bullet or `### DW-n:` heading inside a fence
|
|
2141
|
+
read as a legacy finding (#514).
|
|
2142
|
+
"""
|
|
2143
|
+
masked = text
|
|
2144
|
+
# `unclosed_hides_rest=False` for the reason the canonical side uses it: one
|
|
2145
|
+
# stray opener must not blank every legacy finding below it out of view. The
|
|
2146
|
+
# delimiter lines survive as a lone backtick or tilde plus spaces, which no
|
|
2147
|
+
# pattern below can start an item on — masking them too made no test disagree.
|
|
2148
|
+
spans = [e.span for e in parse_ledger(text)] + fenced_spans(text, unclosed_hides_rest=False)
|
|
2149
|
+
for s, t in spans:
|
|
2150
|
+
masked = masked[:s] + re.sub(r"[^\n]", " ", masked[s:t]) + masked[t:]
|
|
2151
|
+
|
|
2152
|
+
found: list[tuple[dict, tuple[int, int]]] = []
|
|
2153
|
+
section = ""
|
|
2154
|
+
section_done = False
|
|
2155
|
+
# a done section with no items yet: emitted as its own done entry unless
|
|
2156
|
+
# bullets, an entry heading, or a deeper child heading claim it first
|
|
2157
|
+
pending: dict | None = None # {"level", "fields", "span"}
|
|
2158
|
+
item: dict | None = None # accumulating bullet or entry heading
|
|
2159
|
+
|
|
2160
|
+
def close_item(end: int) -> None:
|
|
2161
|
+
nonlocal item
|
|
2162
|
+
if item is None:
|
|
2163
|
+
return
|
|
2164
|
+
body = text[item["start"] : end].rstrip()
|
|
2165
|
+
span = (item["start"], item["start"] + len(body))
|
|
2166
|
+
if item["kind"] == "item":
|
|
2167
|
+
fields = _item_entry(item["first"], body, item["section"], item["section_done"])
|
|
2168
|
+
else:
|
|
2169
|
+
fields = _heading_entry(
|
|
2170
|
+
item["struck"], item["hid"], item["rest"], body, item["section"]
|
|
2171
|
+
)
|
|
2172
|
+
found.append((fields, span))
|
|
2173
|
+
item = None
|
|
2174
|
+
|
|
2175
|
+
def emit_pending() -> None:
|
|
2176
|
+
nonlocal pending
|
|
2177
|
+
if pending is not None:
|
|
2178
|
+
found.append((pending["fields"], pending["span"]))
|
|
2179
|
+
pending = None
|
|
2180
|
+
|
|
2181
|
+
offset = 0
|
|
2182
|
+
for line in text.splitlines(keepends=True):
|
|
2183
|
+
masked_line = masked[offset : offset + len(line)].rstrip("\n")
|
|
2184
|
+
hm = _LINE_HEADING_RE.match(masked_line)
|
|
2185
|
+
if hm:
|
|
2186
|
+
level = len(hm.group(1))
|
|
2187
|
+
close_item(offset)
|
|
2188
|
+
if level == 1:
|
|
2189
|
+
emit_pending()
|
|
2190
|
+
section, section_done = "", False
|
|
2191
|
+
elif level in (2, 3):
|
|
2192
|
+
if pending is not None and level > pending["level"]:
|
|
2193
|
+
pending = None # a child heading: the parent is structure
|
|
2194
|
+
else:
|
|
2195
|
+
emit_pending()
|
|
2196
|
+
em = _ENTRY_HEADING_RE.match(hm.group(2))
|
|
2197
|
+
if em and _DATE_TOKEN_RE.fullmatch(em.group(2)):
|
|
2198
|
+
em = None # "## 2026-06-09 — ..." is a dated section
|
|
2199
|
+
if em:
|
|
2200
|
+
pending = None
|
|
2201
|
+
item = {
|
|
2202
|
+
"kind": "heading",
|
|
2203
|
+
"start": offset,
|
|
2204
|
+
"struck": bool(em.group(1)),
|
|
2205
|
+
"hid": em.group(2),
|
|
2206
|
+
"rest": em.group(3),
|
|
2207
|
+
"section": section,
|
|
2208
|
+
"section_done": section_done,
|
|
2209
|
+
}
|
|
2210
|
+
else:
|
|
2211
|
+
htext = hm.group(2)
|
|
2212
|
+
struck = htext.startswith("~~") and "~~" in htext[2:]
|
|
2213
|
+
section = _clean_title(htext.replace("~~", ""))
|
|
2214
|
+
section_done = struck or bool(_SECTION_DONE_RE.search(htext))
|
|
2215
|
+
if section_done:
|
|
2216
|
+
pending = {
|
|
2217
|
+
"level": level,
|
|
2218
|
+
"span": (offset, offset + len(line.rstrip("\n"))),
|
|
2219
|
+
"fields": {
|
|
2220
|
+
"id": "",
|
|
2221
|
+
"title": section,
|
|
2222
|
+
"done": True,
|
|
2223
|
+
"severity": None,
|
|
2224
|
+
"section": "",
|
|
2225
|
+
},
|
|
2226
|
+
}
|
|
2227
|
+
offset += len(line)
|
|
2228
|
+
continue
|
|
2229
|
+
if item is not None and item["kind"] == "heading":
|
|
2230
|
+
if masked_line.strip() == "---" or (masked_line.strip() == "" and line.strip() != ""):
|
|
2231
|
+
close_item(offset) # rule, or a masked canonical entry
|
|
2232
|
+
offset += len(line)
|
|
2233
|
+
continue
|
|
2234
|
+
bm = _BULLET_RE.match(masked_line)
|
|
2235
|
+
if bm:
|
|
2236
|
+
close_item(offset)
|
|
2237
|
+
pending = None
|
|
2238
|
+
item = {
|
|
2239
|
+
"kind": "item",
|
|
2240
|
+
"start": offset,
|
|
2241
|
+
"first": bm.group(1),
|
|
2242
|
+
"section": section,
|
|
2243
|
+
"section_done": section_done,
|
|
2244
|
+
}
|
|
2245
|
+
elif masked_line.strip() in ("", "---"):
|
|
2246
|
+
# a masked canonical entry reads as blank: it still bounds the item
|
|
2247
|
+
if masked_line.strip() == "---" or line.strip() != masked_line.strip():
|
|
2248
|
+
close_item(offset)
|
|
2249
|
+
elif masked_line[0] in " \t":
|
|
2250
|
+
pass # indented continuation of the current item
|
|
2251
|
+
else:
|
|
2252
|
+
close_item(offset) # column-0 prose ends an item, emits nothing
|
|
2253
|
+
offset += len(line)
|
|
2254
|
+
close_item(len(text))
|
|
2255
|
+
emit_pending()
|
|
2256
|
+
|
|
2257
|
+
entries: list[LegacyEntry] = []
|
|
2258
|
+
counts: dict[str, int] = {}
|
|
2259
|
+
for fields, span in found:
|
|
2260
|
+
base = hashlib.sha1(
|
|
2261
|
+
f"{fields['section']}\0{fields['id'] or fields['title']}".encode(),
|
|
2262
|
+
usedforsecurity=False, # display/identity key, not a credential
|
|
2263
|
+
).hexdigest()[:10]
|
|
2264
|
+
n = counts.get(base, 0) + 1
|
|
2265
|
+
counts[base] = n
|
|
2266
|
+
entries.append(
|
|
2267
|
+
LegacyEntry(
|
|
2268
|
+
key=base if n == 1 else f"{base}-{n}",
|
|
2269
|
+
id=fields["id"],
|
|
2270
|
+
title=fields["title"],
|
|
2271
|
+
done=fields["done"],
|
|
2272
|
+
severity=fields["severity"],
|
|
2273
|
+
body=text[span[0] : span[1]],
|
|
2274
|
+
section=fields["section"],
|
|
2275
|
+
span=span,
|
|
2276
|
+
)
|
|
2277
|
+
)
|
|
2278
|
+
return entries
|
|
2279
|
+
|
|
2280
|
+
|
|
2281
|
+
def has_legacy(text: str) -> bool:
|
|
2282
|
+
return bool(parse_legacy(text))
|