pentimento 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pentimento/__init__.py ADDED
File without changes
pentimento/__main__.py ADDED
@@ -0,0 +1,6 @@
1
+ import sys
2
+
3
+ from pentimento.cli import main
4
+
5
+ if __name__ == "__main__":
6
+ sys.exit(main())
pentimento/backfill.py ADDED
@@ -0,0 +1,139 @@
1
+ """Derive and write frontmatter for plans that don't have it."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from pentimento import frontmatter, lineage, status, times, vocabulary
6
+ from pentimento import plan as plan_module
7
+
8
+ _PROGRESS_RANK = {s: i for i, s in enumerate(vocabulary.PROGRESS_ORDER)}
9
+
10
+
11
+ def _created_date(target) -> str:
12
+ return times.local_date(target.created_at) or target.started[:10]
13
+
14
+
15
+ def _derive_project(target, sessions) -> str | None:
16
+ session = sessions.get(target.id)
17
+ if session and session.project:
18
+ return session.project
19
+ return target.fields.get("project")
20
+
21
+
22
+ def _resolves_to_cycle(plan_id: str, parent_id: str, fields_by_id: dict[str, dict]) -> bool:
23
+ seen = {plan_id}
24
+ current = parent_id
25
+ while current is not None:
26
+ if current in seen:
27
+ return True
28
+ seen.add(current)
29
+ current = fields_by_id.get(current, {}).get("parent")
30
+ return False
31
+
32
+
33
+ def derive_fields(
34
+ target,
35
+ candidates,
36
+ sessions,
37
+ *,
38
+ rederive: bool = False,
39
+ recreate: bool = False,
40
+ max_status: str | None = None,
41
+ ) -> dict[str, str]:
42
+ """Fields to backfill for `target`."""
43
+ fields = dict(target.fields)
44
+ fields.setdefault("intent", vocabulary.DEFAULT_INTENT)
45
+ fields.setdefault("created", _created_date(target))
46
+ if recreate:
47
+ fields["created"] = _created_date(target)
48
+
49
+ existing_status = fields.get("status")
50
+ derived_status = status.derive_status(target.body)
51
+ if max_status and _PROGRESS_RANK.get(derived_status, -1) > _PROGRESS_RANK[max_status]:
52
+ derived_status = max_status
53
+ if existing_status is None:
54
+ fields["status"] = derived_status
55
+ elif existing_status != vocabulary.SUPERSEDED:
56
+ if rederive:
57
+ fields["status"] = derived_status
58
+ else:
59
+ existing_rank = _PROGRESS_RANK.get(existing_status, -1)
60
+ derived_rank = _PROGRESS_RANK.get(derived_status, -1)
61
+ if derived_rank > existing_rank:
62
+ fields["status"] = derived_status
63
+
64
+ if rederive or "project" not in fields:
65
+ project = _derive_project(target, sessions)
66
+ if project:
67
+ fields["project"] = project
68
+
69
+ if rederive:
70
+ parent_id = lineage.derive_parent(
71
+ target, candidates, sessions, project=fields.get("project")
72
+ )
73
+ if parent_id:
74
+ fields["parent"] = parent_id
75
+ else:
76
+ fields.pop("parent", None)
77
+ elif "parent" not in fields:
78
+ parent_id = lineage.derive_parent(
79
+ target, candidates, sessions, project=fields.get("project")
80
+ )
81
+ if parent_id:
82
+ fields["parent"] = parent_id
83
+
84
+ return fields
85
+
86
+
87
+ def run(
88
+ plans,
89
+ sessions=None,
90
+ *,
91
+ dry_run: bool = False,
92
+ rederive: bool = False,
93
+ recreate: bool = False,
94
+ max_status: str | None = None,
95
+ only=None,
96
+ ) -> list[str]:
97
+ """Backfill frontmatter across `plans`. Returns ids that were changed.
98
+
99
+ `only`, when given, restricts writes to those ids; derivation still spans
100
+ `plans` entire, because `lineage.derive_parent` resolves against the whole
101
+ corpus and `_resolves_to_cycle` needs every plan's new fields.
102
+ """
103
+ sessions = sessions or {}
104
+ new_fields_by_path = {
105
+ target.path: derive_fields(
106
+ target,
107
+ plans,
108
+ sessions,
109
+ rederive=rederive,
110
+ recreate=recreate,
111
+ max_status=max_status,
112
+ )
113
+ for target in plans
114
+ }
115
+ # Cycle traversal walks `parent` id references, so it needs an id-keyed view.
116
+ # A shared id makes the choice of which plan's fields represent that id
117
+ # arbitrary here, but that ambiguity is inherent to duplicate ids, not
118
+ # introduced by this map -- it does not affect which plan's fields get
119
+ # written, which is keyed by path above.
120
+ new_fields_by_id = {target.id: new_fields_by_path[target.path] for target in plans}
121
+
122
+ for target in plans:
123
+ new_fields = new_fields_by_path[target.path]
124
+ parent_id = new_fields.get("parent")
125
+ if parent_id and _resolves_to_cycle(target.id, parent_id, new_fields_by_id):
126
+ new_fields.pop("parent", None)
127
+
128
+ changed = []
129
+ for target in plans:
130
+ if only is not None and target.id not in only:
131
+ continue
132
+ new_fields = new_fields_by_path[target.path]
133
+ if frontmatter.serialize(new_fields, target.body, target.extras) == target.text:
134
+ continue
135
+ changed.append(target.id)
136
+ if not dry_run:
137
+ target.fields = new_fields
138
+ plan_module.save(target, keep_mtime=True)
139
+ return changed
pentimento/cache.py ADDED
@@ -0,0 +1,54 @@
1
+ """Memoize per-file derived data across `pentimento` invocations.
2
+
3
+ A cache miss, a corrupt file, or an unwritable cache directory all degrade
4
+ to a full parse rather than an error.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import json
10
+ import os
11
+ import tempfile
12
+ from pathlib import Path
13
+
14
+ VERSION = 1
15
+
16
+
17
+ def cache_dir() -> Path:
18
+ base = os.environ.get("XDG_CACHE_HOME")
19
+ root = Path(base) if base else Path.home() / ".cache"
20
+ return root / "pentimento"
21
+
22
+
23
+ def key(path: Path) -> str:
24
+ stat = path.stat()
25
+ return f"{path}:{stat.st_size}:{stat.st_mtime_ns}"
26
+
27
+
28
+ def read(namespace: str) -> dict:
29
+ try:
30
+ data = json.loads((cache_dir() / f"{namespace}.json").read_text(encoding="utf-8"))
31
+ except (OSError, ValueError):
32
+ return {}
33
+
34
+ if data.get("version") != VERSION:
35
+ return {}
36
+
37
+ entries = data.get("entries")
38
+ return entries if isinstance(entries, dict) else {}
39
+
40
+
41
+ def write(namespace: str, entries: dict) -> None:
42
+ directory = cache_dir()
43
+ try:
44
+ directory.mkdir(parents=True, exist_ok=True)
45
+ fd, tmp_name = tempfile.mkstemp(dir=directory, prefix=f".{namespace}.", suffix=".json.tmp")
46
+ try:
47
+ with os.fdopen(fd, "w", encoding="utf-8") as handle:
48
+ handle.write(json.dumps({"version": VERSION, "entries": entries}))
49
+ os.replace(tmp_name, directory / f"{namespace}.json")
50
+ except OSError:
51
+ os.unlink(tmp_name)
52
+ raise
53
+ except OSError:
54
+ pass
pentimento/check.py ADDED
@@ -0,0 +1,197 @@
1
+ """Validate the plan corpus for lineage and vocabulary defects."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import dataclasses
6
+
7
+ from pentimento import counts
8
+ from pentimento import status as status_module
9
+ from pentimento import tags as tags_module
10
+ from pentimento import touches as touches_module
11
+ from pentimento import vocabulary as vocabulary_module
12
+
13
+ _HISTORY_ELIGIBLE_STATUSES = vocabulary_module.UNWORKED_STATUSES
14
+
15
+
16
+ @dataclasses.dataclass
17
+ class Finding:
18
+ plan_id: str
19
+ code: str
20
+ message: str
21
+
22
+
23
+ def _dangling_parents(plans, by_id):
24
+ return [p for p in plans if p.parent and p.parent not in by_id]
25
+
26
+
27
+ def _self_parents(plans):
28
+ return [p for p in plans if p.parent == p.id]
29
+
30
+
31
+ def _cross_project_parents(plans, by_id):
32
+ findings = []
33
+ for p in plans:
34
+ if not p.parent:
35
+ continue
36
+ parent = by_id.get(p.parent)
37
+ if parent is None:
38
+ continue
39
+ if parent.project != p.project:
40
+ findings.append(p)
41
+ return findings
42
+
43
+
44
+ def _in_cycle(plan, by_id):
45
+ if plan.parent == plan.id:
46
+ return False
47
+ seen = set()
48
+ current = plan
49
+ while current is not None and current.parent:
50
+ if current.id in seen:
51
+ return current.id == plan.id
52
+ seen.add(current.id)
53
+ current = by_id.get(current.parent)
54
+ return False
55
+
56
+
57
+ def _cycle_members(plans, by_id):
58
+ return [p for p in plans if p.parent and _in_cycle(p, by_id)]
59
+
60
+
61
+ def duplicate_ids(plans):
62
+ seen = set()
63
+ duplicates = []
64
+ for p in plans:
65
+ if p.id in seen:
66
+ duplicates.append(p)
67
+ else:
68
+ seen.add(p.id)
69
+ return duplicates
70
+
71
+
72
+ def _off_vocabulary_status(plans):
73
+ return [p for p in plans if p.status not in vocabulary_module.STATUS_ORDER]
74
+
75
+
76
+ def _off_vocabulary_intent(plans):
77
+ return [p for p in plans if p.intent not in vocabulary_module.INTENT_VALUES]
78
+
79
+
80
+ def _missing_title(plans):
81
+ return [p for p in plans if not p.has_title]
82
+
83
+
84
+ def _malformed_tags(plans):
85
+ return [p for p in plans if any(not tags_module.is_valid(t) for t in p.tags)]
86
+
87
+
88
+ def _missing_progress(plans):
89
+ return [p for p in plans if status_module.progress_section(p.body) is None]
90
+
91
+
92
+ def _underived_project(plans, sessions):
93
+ findings = []
94
+ for p in plans:
95
+ if p.project:
96
+ continue
97
+ session = sessions.get(p.id)
98
+ if session and session.project:
99
+ findings.append(p)
100
+ return findings
101
+
102
+
103
+ def _status_behind_history(plans, touches):
104
+ findings = []
105
+ for p in plans:
106
+ if p.status not in _HISTORY_ELIGIBLE_STATUSES:
107
+ continue
108
+ worked = touches_module.worked(touches.get(p.id, []), p.id)
109
+ if not worked:
110
+ continue
111
+ sessions_worked = len({t.session for t in worked})
112
+ message = (
113
+ f"status {p.status!r} but "
114
+ f"{counts.plural(sessions_worked, 'later session')} worked this plan; "
115
+ f"see `pentimento history {p.id}`"
116
+ )
117
+ findings.append(Finding(p.id, "status-behind-history", message))
118
+ return findings
119
+
120
+
121
+ _PROGRESS_RANK = {s: i for i, s in enumerate(vocabulary_module.PROGRESS_ORDER)}
122
+
123
+
124
+ def _status_behind_progress(plans):
125
+ findings = []
126
+ for p in plans:
127
+ if p.status not in _PROGRESS_RANK:
128
+ continue
129
+ derived = status_module.derive_status(p.body)
130
+ if derived not in _PROGRESS_RANK:
131
+ continue
132
+ if _PROGRESS_RANK[derived] <= _PROGRESS_RANK[p.status]:
133
+ continue
134
+ message = (
135
+ f"status {p.status!r} but '## Progress' derives {derived!r}; run pentimento backfill"
136
+ )
137
+ findings.append(Finding(p.id, "status-behind-progress", message))
138
+ return findings
139
+
140
+
141
+ def run(plans, sessions=None, touches=None, skips=None) -> list[Finding]:
142
+ """Return structured findings; an empty list means a clean corpus.
143
+
144
+ `sessions`, when given, enables the `underived-project` finding; it
145
+ stays silent by default so Cursor plans and session-less plans, where
146
+ an empty `project` is a legitimate state, are not flagged. `touches`,
147
+ when given, enables `status-behind-history` the same way. `skips`, when
148
+ given, is the `(source, path, error)` list `corpus.load_all` collected
149
+ for files it could not read; each becomes an `unreadable-file` finding.
150
+ """
151
+ sessions = sessions or {}
152
+ touches = touches or {}
153
+ skips = skips or []
154
+ by_id = {p.id: p for p in plans}
155
+ findings = []
156
+
157
+ for _source, path, exc in skips:
158
+ message = f"could not read {path}: {exc}"
159
+ findings.append(Finding(path.name, "unreadable-file", message))
160
+
161
+ for p in _dangling_parents(plans, by_id):
162
+ message = f"parent {p.parent!r} does not resolve to a plan"
163
+ findings.append(Finding(p.id, "dangling-parent", message))
164
+ for p in _self_parents(plans):
165
+ findings.append(Finding(p.id, "self-parent", "parent is itself"))
166
+ for p in _cross_project_parents(plans, by_id):
167
+ message = f"parent {p.parent!r} is in a different project"
168
+ findings.append(Finding(p.id, "cross-project-parent", message))
169
+ for p in _cycle_members(plans, by_id):
170
+ message = "parent chain cycles back to itself"
171
+ findings.append(Finding(p.id, "cycle", message))
172
+ for p in duplicate_ids(plans):
173
+ message = f"duplicate id across sources (second occurrence from {p.source})"
174
+ findings.append(Finding(p.id, "duplicate-id", message))
175
+ for p in _off_vocabulary_status(plans):
176
+ message = f"status {p.status!r} is outside {vocabulary_module.STATUS_ORDER}"
177
+ findings.append(Finding(p.id, "off-vocabulary-status", message))
178
+ for p in _off_vocabulary_intent(plans):
179
+ message = f"intent {p.intent!r} is outside {vocabulary_module.INTENT_VALUES}"
180
+ findings.append(Finding(p.id, "off-vocabulary-intent", message))
181
+ for p in _missing_title(plans):
182
+ message = "body has no H1 title; falling back to the plan id"
183
+ findings.append(Finding(p.id, "missing-title", message))
184
+ for p in _malformed_tags(plans):
185
+ bad = [t for t in p.tags if not tags_module.is_valid(t)]
186
+ message = f"malformed tag(s) {bad!r}"
187
+ findings.append(Finding(p.id, "malformed-tag", message))
188
+ for p in _missing_progress(plans):
189
+ message = "body has no '## Progress' heading; status can't be derived"
190
+ findings.append(Finding(p.id, "missing-progress", message))
191
+ for p in _underived_project(plans, sessions):
192
+ message = f"session supplies project {sessions[p.id].project!r} but frontmatter has none"
193
+ findings.append(Finding(p.id, "underived-project", message))
194
+ findings.extend(_status_behind_history(plans, touches))
195
+ findings.extend(_status_behind_progress(plans))
196
+
197
+ return findings