@akinet/akidevrule 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +835 -0
- package/LICENSE +21 -0
- package/README.md +356 -0
- package/claude/CLAUDE.md +40 -0
- package/claude/agents/aki-challenger.md +38 -0
- package/claude/agents/aki-conduct.md +54 -0
- package/claude/agents/aki-hands.md +59 -0
- package/claude/agents/aki-judge.md +37 -0
- package/claude/agents/aki-maker.md +36 -0
- package/claude/fragments/settings.akidoc.fragment.json +15 -0
- package/claude/hooks/aki-update-check.mjs +160 -0
- package/claude/hooks/aki_version_check.mjs +83 -0
- package/docs/ref/macos-codesign-tcc.md +59 -0
- package/install.mjs +1067 -0
- package/install.ps1 +11 -0
- package/install.sh +12 -0
- package/package.json +52 -0
- package/payload/GEMINI.md +147 -0
- package/payload/METHOD-audit-flow.md +147 -0
- package/payload/METHOD-audit-subtraction.md +67 -0
- package/payload/METHOD-audit-zero-trust.md +49 -0
- package/payload/METHOD-deep-think.md +172 -0
- package/payload/METHOD-proportionality.md +62 -0
- package/payload/METHOD-ux-psych.md +60 -0
- package/payload/RULE-agent-behavior.md +138 -0
- package/payload/RULE-biz.md +51 -0
- package/payload/RULE-coding.md +130 -0
- package/payload/RULE-content-write.md +54 -0
- package/payload/RULE-db-design.md +26 -0
- package/payload/RULE-docs.md +144 -0
- package/payload/RULE-pattern-core.md +80 -0
- package/payload/RULE-release.md +215 -0
- package/payload/RULE-seo.md +173 -0
- package/payload/RULE-stack-akiNuxtCf.md +179 -0
- package/payload/RULE-stack-tauri.md +59 -0
- package/payload/RULE-ui-pattern.md +167 -0
- package/payload/index.md +91 -0
- package/skills/aki-article-writer/SKILL.md +50 -0
- package/skills/aki-article-writer/references/article-workflow.md +377 -0
- package/skills/akidevsync-notes/SKILL.md +48 -0
- package/skills/akidevsync-notes/scripts/notes_cli.py +212 -0
- package/skills/akiflow/SKILL.md +221 -0
- package/skills/akiflow/references/harness-facts.md +215 -0
- package/skills/akiflow/scripts/council-cost.sh +4 -0
- package/skills/akiflow/scripts/council-open.sh +4 -0
- package/skills/akiflow/scripts/council-read.sh +4 -0
- package/skills/akiflow/scripts/council-verify.sh +4 -0
- package/skills/akiflow/scripts/council_cost.py +149 -0
- package/skills/akiflow/scripts/council_open.py +323 -0
- package/skills/akiflow/scripts/council_read.py +148 -0
- package/skills/akiflow/scripts/council_verify.py +315 -0
- package/skills/akiflow/scripts/scythe.py +307 -0
- package/skills/akiflow/scripts/scythe.sh +4 -0
- package/skills/akigitcommit/SKILL.md +85 -0
- package/skills/akihelp/SKILL.md +47 -0
- package/skills/akihtmlreport/SKILL.md +59 -0
- package/skills/akilint/SKILL.md +29 -0
- package/skills/akirule/SKILL.md +155 -0
- package/skills/akiship/SKILL.md +55 -0
- package/skills/akithink/SKILL.md +59 -0
|
@@ -0,0 +1,315 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
# council_verify.py — mechanical closure gate over the room's own compliance artifacts.
|
|
3
|
+
#
|
|
4
|
+
# Usage: council_verify.py <session-dir>
|
|
5
|
+
# <session-dir> a council workspace holding chat.md + checklist.md
|
|
6
|
+
#
|
|
7
|
+
# Checks only what a script can check — the classes real runs lost silently:
|
|
8
|
+
# 1. anchor — chat.md carries a non-empty '## anchor' block
|
|
9
|
+
# 2. REQ quotes — every 'REQ-<n>' line in checklist.md quotes a fragment found in that anchor
|
|
10
|
+
# 3. ghost seats — every owner/challenger named in checklist.md left a trace somewhere in the session
|
|
11
|
+
# 4. rule receipts — every seat emitted a '[RULES]' line
|
|
12
|
+
# 5. evidence tags — every seat used FACT/CONSTRAINT/ASSUMPTION at least once
|
|
13
|
+
# 6. reminders — every REMIND-<n> has a later ACK or OVERRULE
|
|
14
|
+
# 7. REQ coverage — every REQ-<n> in the ledger is named by some item's 'covers:' line
|
|
15
|
+
#
|
|
16
|
+
# A seat's evidence is its chat.md turns plus its own <seat>.md at any depth. Nothing to check prints SKIP, never PASS.
|
|
17
|
+
# NOT checked, deliberately: the presence of any named seat. Roster composition is judgment; evidence is not.
|
|
18
|
+
# Exit 0 = no FAIL. Exit 1 = at least one FAIL.
|
|
19
|
+
|
|
20
|
+
import re
|
|
21
|
+
import sys
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def read_text(path: Path) -> str:
|
|
26
|
+
return path.read_text(encoding='utf-8', errors='replace')
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _strip_comments(text: str) -> str:
|
|
30
|
+
return re.sub(r'<!--.*?-->', '', text, flags=re.DOTALL)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _req_ids(text: str) -> list[str]:
|
|
34
|
+
"""REQ ids in first-seen order, expanding the compact run form `REQ-2,3` to REQ-2 and REQ-3."""
|
|
35
|
+
ids: list[str] = []
|
|
36
|
+
for run in re.findall(r'REQ-[0-9]+(?:[ \t]*,[ \t]*[0-9]+)*', _strip_comments(text)):
|
|
37
|
+
for num in re.findall(r'[0-9]+', run):
|
|
38
|
+
if f'REQ-{num}' not in ids:
|
|
39
|
+
ids.append(f'REQ-{num}')
|
|
40
|
+
return ids
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def read_mode(chat: str) -> str:
|
|
44
|
+
"""Mode from chat.md's line-2 stamp, duplicating council_open.py::read_mode() — the scripts
|
|
45
|
+
are deliberately standalone. No match (pre-dispatch room, malformed file) reads as council."""
|
|
46
|
+
lines = chat.splitlines()
|
|
47
|
+
if len(lines) < 2:
|
|
48
|
+
return "council"
|
|
49
|
+
match = re.search(r"mode[ \t]+(\S+)", lines[1])
|
|
50
|
+
return match.group(1).rstrip("`") if match else "council"
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
LANE_HEADING = re.compile(r"^#{0,6}[ \t]*LANE[ \t]+\S")
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def get_lane_names(checklist: str) -> list[str]:
|
|
57
|
+
"""Dispatch's trace unit is the lane, not its worker — the lane is what dispatch partitions
|
|
58
|
+
into, and 'worker' is roster/cost metadata that two lanes may legitimately share. A LANE
|
|
59
|
+
heading's text after '·' is slugified (lowercase, spaces→dashes) so it matches both a
|
|
60
|
+
<lane>.md file stem and a turn header's agent field."""
|
|
61
|
+
names: set[str] = set()
|
|
62
|
+
for line in _strip_comments(checklist).splitlines():
|
|
63
|
+
if not LANE_HEADING.match(line):
|
|
64
|
+
continue
|
|
65
|
+
_, sep, short = line.partition('·')
|
|
66
|
+
if not sep:
|
|
67
|
+
continue
|
|
68
|
+
slug = re.sub(r'[ \t]+', '-', short.strip().lower())
|
|
69
|
+
slug = re.sub(r'[^a-z0-9-]', '', slug)
|
|
70
|
+
if slug:
|
|
71
|
+
names.add(slug)
|
|
72
|
+
return sorted(names)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def extract_anchor(chat: str) -> str:
|
|
76
|
+
"""Extract text between '## anchor' and the next '## ' heading, stripping HTML comments."""
|
|
77
|
+
block_lines = []
|
|
78
|
+
in_block = False
|
|
79
|
+
for line in chat.splitlines():
|
|
80
|
+
if re.match(r'^## anchor', line):
|
|
81
|
+
in_block = True
|
|
82
|
+
continue
|
|
83
|
+
if in_block and re.match(r'^## ', line):
|
|
84
|
+
break
|
|
85
|
+
if in_block:
|
|
86
|
+
block_lines.append(line)
|
|
87
|
+
return _strip_comments('\n'.join(block_lines))
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def get_posters(chat: str) -> list[str]:
|
|
91
|
+
"""Agents that posted at least one turn (### <time> <agent> #<n>)."""
|
|
92
|
+
posters: set[str] = set()
|
|
93
|
+
for line in chat.splitlines():
|
|
94
|
+
if line.startswith('### '):
|
|
95
|
+
parts = line.split()
|
|
96
|
+
if len(parts) >= 3:
|
|
97
|
+
posters.add(parts[2])
|
|
98
|
+
return sorted(posters)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def get_seat_files(session_dir: Path) -> dict[str, str]:
|
|
102
|
+
"""Text of every non-core .md at any depth, keyed by stem — live rooms group seat files into phase subdirectories.
|
|
103
|
+
|
|
104
|
+
A file only supplies evidence for a seat the checklist or chat already names; it never adds a seat of its own,
|
|
105
|
+
or a lead's summary.md would be conscripted into a roster it was never part of."""
|
|
106
|
+
seats: dict[str, str] = {}
|
|
107
|
+
for path in sorted(session_dir.rglob('*.md')):
|
|
108
|
+
if path.name in ('chat.md', 'checklist.md'):
|
|
109
|
+
continue
|
|
110
|
+
seats[path.stem] = seats.get(path.stem, '') + read_text(path)
|
|
111
|
+
return seats
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def get_declared(checklist: str, mode: str) -> list[str]:
|
|
115
|
+
"""Seat names the lead assigned in checklist.md.
|
|
116
|
+
|
|
117
|
+
Council: owner/challenger/worker field values ARE the seat identity.
|
|
118
|
+
Dispatch: the lane is the unit of dispatch, so trace by lane name instead — a lane's
|
|
119
|
+
'worker' is a substrate detail, and two lanes sharing a worker type must still trace
|
|
120
|
+
separately or one file/turn silently covers both."""
|
|
121
|
+
if mode == "dispatch":
|
|
122
|
+
return get_lane_names(checklist)
|
|
123
|
+
declared: set[str] = set()
|
|
124
|
+
for line in checklist.splitlines():
|
|
125
|
+
# Items declare fields block-form, lanes declare them as markdown bullets — a leading dash must not hide a worker from the ghost-seat check.
|
|
126
|
+
m = re.match(r'^[ \t]*(?:[-*][ \t]+)?(owner|challenger|worker):[ \t]*(\S+)', line, re.IGNORECASE)
|
|
127
|
+
if m:
|
|
128
|
+
name = m.group(2).split()[0] # first token only
|
|
129
|
+
if re.match(r'^[a-z0-9][a-z0-9-]*$', name):
|
|
130
|
+
declared.add(name)
|
|
131
|
+
return sorted(declared)
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def check_anchor(anchor: str) -> tuple[bool, str]:
|
|
135
|
+
if anchor.replace('\n', '').replace(' ', '').strip():
|
|
136
|
+
return True, "PASS anchor: the owner's verbatim message is pinned"
|
|
137
|
+
return False, "FAIL anchor: chat.md has no non-empty '## anchor' block — the room has nothing to be measured against"
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def check_req_quotes(checklist: str, anchor: str) -> tuple[bool, list[str]]:
|
|
141
|
+
unquoted: list[str] = []
|
|
142
|
+
unfound: list[str] = []
|
|
143
|
+
for line in checklist.splitlines():
|
|
144
|
+
if not re.search(r'^[ \t]*[-*]?[ \t]*REQ-[0-9]+', line):
|
|
145
|
+
continue
|
|
146
|
+
id_match = re.search(r'REQ-[0-9]+', line)
|
|
147
|
+
req_id = id_match.group() if id_match else ''
|
|
148
|
+
frag_match = re.search(r'^[^"]*"([^"]+)"', line)
|
|
149
|
+
if not frag_match:
|
|
150
|
+
unquoted.append(req_id)
|
|
151
|
+
elif frag_match.group(1) not in anchor:
|
|
152
|
+
unfound.append(req_id)
|
|
153
|
+
|
|
154
|
+
messages: list[str] = []
|
|
155
|
+
ok = True
|
|
156
|
+
if unquoted:
|
|
157
|
+
messages.append(f"FAIL req-quotes: no \"quoted fragment\" on: {' '.join(unquoted)}")
|
|
158
|
+
ok = False
|
|
159
|
+
if unfound:
|
|
160
|
+
messages.append(f"FAIL req-quotes: quoted text not found in the anchor block: {' '.join(unfound)}")
|
|
161
|
+
ok = False
|
|
162
|
+
if ok:
|
|
163
|
+
messages.append("PASS req-quotes: every REQ quotes the owner's own words")
|
|
164
|
+
return ok, messages
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def check_ghost_seats(declared: list[str], traced: dict[str, str]) -> tuple[bool, str]:
|
|
168
|
+
ghosts = [n for n in declared if n not in traced]
|
|
169
|
+
if ghosts:
|
|
170
|
+
return False, f"FAIL ghost-seats: declared but left no turn and no file: {' '.join(ghosts)}"
|
|
171
|
+
if not declared:
|
|
172
|
+
return True, "SKIP ghost-seats: checklist.md declares no owner/challenger"
|
|
173
|
+
return True, "PASS ghost-seats: every declared owner/challenger left a trace"
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _agent_blocks(chat: str, agent: str) -> str:
|
|
177
|
+
"""Return the concatenated text of all turns posted by agent."""
|
|
178
|
+
result: list[str] = []
|
|
179
|
+
capturing = False
|
|
180
|
+
for line in chat.splitlines():
|
|
181
|
+
if line.startswith('### '):
|
|
182
|
+
parts = line.split()
|
|
183
|
+
capturing = len(parts) >= 3 and parts[2] == agent
|
|
184
|
+
elif capturing:
|
|
185
|
+
result.append(line)
|
|
186
|
+
return '\n'.join(result)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def check_rule_receipts(traced: dict[str, str]) -> tuple[bool, str]:
|
|
190
|
+
noreceipt = [name for name, text in sorted(traced.items()) if '[RULES]' not in text]
|
|
191
|
+
if noreceipt:
|
|
192
|
+
return False, f"FAIL rule-receipts: no [RULES] line anywhere from: {' '.join(noreceipt)}"
|
|
193
|
+
if not traced:
|
|
194
|
+
return True, "SKIP rule-receipts: no seat left a trace to check"
|
|
195
|
+
return True, "PASS rule-receipts: every seat reported what it loaded"
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def check_evidence_tags(traced: dict[str, str]) -> tuple[bool, str]:
|
|
199
|
+
untagged = [
|
|
200
|
+
name for name, text in sorted(traced.items())
|
|
201
|
+
if not re.search(r'FACT|CONSTRAINT|ASSUMPTION', text)
|
|
202
|
+
]
|
|
203
|
+
if untagged:
|
|
204
|
+
return False, f"FAIL evidence-tags: no FACT/CONSTRAINT/ASSUMPTION anywhere from: {' '.join(untagged)}"
|
|
205
|
+
if not traced:
|
|
206
|
+
return True, "SKIP evidence-tags: no seat left a trace to check"
|
|
207
|
+
return True, "PASS evidence-tags: every seat tagged evidence"
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def check_req_coverage(checklist: str) -> tuple[bool, str]:
|
|
211
|
+
"""Diff the ledger against the items — the lead's omissions, found without asking the lead.
|
|
212
|
+
|
|
213
|
+
Parsed by section rather than by line shape: both the block form (`covers:` on its own line) and
|
|
214
|
+
the one-line pipe form (`ITEM 5 · … | covers REQ-1 | …`) are in real use, and a parser that only
|
|
215
|
+
knows one of them fails open — the silent direction for a coverage check."""
|
|
216
|
+
ledger_text, items_text = [], []
|
|
217
|
+
target = None
|
|
218
|
+
for line in checklist.splitlines():
|
|
219
|
+
if line.startswith('## '):
|
|
220
|
+
head = line[3:].strip().lower()
|
|
221
|
+
# Anchored, not substring: '## REQ with no item' contains 'item' and would otherwise route an explicitly-uncovered REQ into the covered set — a silent pass.
|
|
222
|
+
target = ledger_text if 'ledger' in head else (items_text if head.startswith('item') or head.startswith('lane') else None)
|
|
223
|
+
continue
|
|
224
|
+
if target is not None:
|
|
225
|
+
target.append(line)
|
|
226
|
+
|
|
227
|
+
ledger = _req_ids('\n'.join(ledger_text))
|
|
228
|
+
covered = set(_req_ids('\n'.join(items_text)))
|
|
229
|
+
if not ledger:
|
|
230
|
+
return False, "FAIL req-coverage: no REQ-<n> lines in checklist.md — the ledger was never written"
|
|
231
|
+
orphans = [r for r in ledger if r not in covered]
|
|
232
|
+
if orphans:
|
|
233
|
+
return False, f"FAIL req-coverage: no item covers: {' '.join(orphans)}"
|
|
234
|
+
return True, f"PASS req-coverage: all {len(ledger)} REQs owned by an item"
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
def check_reminders(chat: str) -> tuple[bool, str]:
|
|
238
|
+
remind_ids = set(re.findall(r'REMIND-[0-9]+', chat))
|
|
239
|
+
open_reminds = []
|
|
240
|
+
for rid in sorted(remind_ids):
|
|
241
|
+
if not re.search(rf'(ACK|OVERRULE) {re.escape(rid)}\b', chat):
|
|
242
|
+
open_reminds.append(rid)
|
|
243
|
+
if open_reminds:
|
|
244
|
+
return False, f"FAIL reminders: no ACK/OVERRULE for: {' '.join(open_reminds)}"
|
|
245
|
+
return True, "PASS reminders: every REMIND answered (or none issued)"
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def main() -> None:
|
|
249
|
+
if len(sys.argv) != 2 or not sys.argv[1]:
|
|
250
|
+
print("usage: council_verify.py <session-dir> (must contain chat.md + checklist.md)", file=sys.stderr)
|
|
251
|
+
sys.exit(2)
|
|
252
|
+
|
|
253
|
+
session_dir = Path(sys.argv[1])
|
|
254
|
+
chat_path = session_dir / 'chat.md'
|
|
255
|
+
checklist_path = session_dir / 'checklist.md'
|
|
256
|
+
|
|
257
|
+
if not chat_path.is_file() or not checklist_path.is_file():
|
|
258
|
+
print("usage: council_verify.py <session-dir> (must contain chat.md + checklist.md)", file=sys.stderr)
|
|
259
|
+
sys.exit(2)
|
|
260
|
+
|
|
261
|
+
chat = read_text(chat_path)
|
|
262
|
+
checklist = read_text(checklist_path)
|
|
263
|
+
|
|
264
|
+
anchor = extract_anchor(chat)
|
|
265
|
+
seat_files = get_seat_files(session_dir)
|
|
266
|
+
declared = get_declared(checklist, read_mode(chat))
|
|
267
|
+
roster = sorted(set(get_posters(chat)) | set(declared))
|
|
268
|
+
traced = {
|
|
269
|
+
name: text for name in roster
|
|
270
|
+
if (text := f"{_agent_blocks(chat, name)}\n{seat_files.get(name, '')}".strip())
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
fail = False
|
|
274
|
+
|
|
275
|
+
# 1. anchor
|
|
276
|
+
ok, msg = check_anchor(anchor)
|
|
277
|
+
print(msg)
|
|
278
|
+
fail = fail or not ok
|
|
279
|
+
|
|
280
|
+
# 2. REQ quotes
|
|
281
|
+
ok, msgs = check_req_quotes(checklist, anchor)
|
|
282
|
+
for msg in msgs:
|
|
283
|
+
print(msg)
|
|
284
|
+
fail = fail or not ok
|
|
285
|
+
|
|
286
|
+
# 3. ghost seats
|
|
287
|
+
ok, msg = check_ghost_seats(declared, traced)
|
|
288
|
+
print(msg)
|
|
289
|
+
fail = fail or not ok
|
|
290
|
+
|
|
291
|
+
# 4. rule receipts
|
|
292
|
+
ok, msg = check_rule_receipts(traced)
|
|
293
|
+
print(msg)
|
|
294
|
+
fail = fail or not ok
|
|
295
|
+
|
|
296
|
+
# 5. evidence tags
|
|
297
|
+
ok, msg = check_evidence_tags(traced)
|
|
298
|
+
print(msg)
|
|
299
|
+
fail = fail or not ok
|
|
300
|
+
|
|
301
|
+
# 6. unanswered reminders
|
|
302
|
+
ok, msg = check_reminders(chat)
|
|
303
|
+
print(msg)
|
|
304
|
+
fail = fail or not ok
|
|
305
|
+
|
|
306
|
+
# 7. REQ coverage
|
|
307
|
+
ok, msg = check_req_coverage(checklist)
|
|
308
|
+
print(msg)
|
|
309
|
+
fail = fail or not ok
|
|
310
|
+
|
|
311
|
+
sys.exit(1 if fail else 0)
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
if __name__ == "__main__":
|
|
315
|
+
main()
|
|
@@ -0,0 +1,307 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
# scythe.py — mechanical lint for the greppable penalty-card classes (RULE-agent-behavior.md §0).
|
|
3
|
+
# Detects [WRAP] (hard-wrapped code comments / markdown prose) and [YAP] (oversize comments — flagged "review", never a verdict).
|
|
4
|
+
# [FLUFF] (density) is content judgment and deliberately out of scope for a script.
|
|
5
|
+
# Usage: scythe.py [--all] <file|dir> [...] A dir expands to its git-tracked files; outside a repo, to find(1).
|
|
6
|
+
# Output: [TAG] path:line[-line] | short label Exit: 0 clean · 1 findings · 2 usage error.
|
|
7
|
+
# Past 40 findings (SCYTHE_CAP) output becomes a capped list plus per-tag and per-file counts; --all prints everything.
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import os
|
|
12
|
+
import re
|
|
13
|
+
import sys
|
|
14
|
+
import subprocess
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
SLASH_EXT = {'ts', 'tsx', 'js', 'jsx', 'mjs', 'rs', 'go', 'c', 'h', 'cc', 'cpp', 'swift', 'kt', 'scss'}
|
|
18
|
+
HASH_EXT = {'sh', 'bash', 'py', 'rb', 'toml', 'yaml', 'yml'}
|
|
19
|
+
DASH_EXT = {'sql'}
|
|
20
|
+
HTML_EXT = {'vue', 'html'}
|
|
21
|
+
|
|
22
|
+
# Directive/exempt prefixes inside a comment's text portion — mirrors the awk exempt list.
|
|
23
|
+
_EXEMPT_TEXT = re.compile(
|
|
24
|
+
r'^(@|!|#|eslint|prettier|biome|ts-|type:|noqa|pylint|ruff|shellcheck|fmt:|region|endregion|-{3,}|={3,}|\*)'
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _scan_comment_runs(path: str, numbered: list[tuple[int, str]], marker: re.Pattern, in_header: bool) -> list[str]:
|
|
29
|
+
"""Comment-run logic shared by whole code files and fenced code blocks inside markdown."""
|
|
30
|
+
findings: list[str] = []
|
|
31
|
+
|
|
32
|
+
header = in_header
|
|
33
|
+
rs = 0 # run size (number of comment lines accumulated)
|
|
34
|
+
rstart = 0 # 1-based line number where the run started
|
|
35
|
+
cont = False # True if a lowercase-continuation line was seen
|
|
36
|
+
|
|
37
|
+
def flush() -> None:
|
|
38
|
+
nonlocal rs
|
|
39
|
+
if rs == 0:
|
|
40
|
+
return
|
|
41
|
+
if header:
|
|
42
|
+
rs = 0
|
|
43
|
+
return
|
|
44
|
+
if rs >= 3:
|
|
45
|
+
findings.append(f"[YAP] {path}:{rstart}-{rstart + rs - 1} | {rs}-line comment block (review)")
|
|
46
|
+
elif rs == 2 and cont:
|
|
47
|
+
findings.append(f"[WRAP] {path}:{rstart}-{rstart + 1} | wrapped comment (rejoin)")
|
|
48
|
+
rs = 0
|
|
49
|
+
|
|
50
|
+
for lineno, raw in numbered:
|
|
51
|
+
m = marker.search(raw)
|
|
52
|
+
if m:
|
|
53
|
+
# Text after the marker, leading spaces/tabs dropped — mirrors the awk marker match's trailing [ \t]* consumption.
|
|
54
|
+
text = raw[m.end():].lstrip(' \t')
|
|
55
|
+
if not text or _EXEMPT_TEXT.match(text):
|
|
56
|
+
flush()
|
|
57
|
+
# blank-ish structural line — do NOT clear header here (matches awk: next skips header=0)
|
|
58
|
+
continue
|
|
59
|
+
if rs == 0:
|
|
60
|
+
rstart = lineno
|
|
61
|
+
cont = False
|
|
62
|
+
rs += 1
|
|
63
|
+
# continuation: second+ line whose text starts with a lowercase letter (awk: /^[a-z]/)
|
|
64
|
+
# Vietnamese multibyte words (được, ở, …) are matched by \p{Ll} — use re.UNICODE default.
|
|
65
|
+
if rs > 1 and re.match(r'^[a-z\u00c0-\u024f\u1e00-\u1eff]', text):
|
|
66
|
+
cont = True
|
|
67
|
+
if len(raw) > 200 and not header:
|
|
68
|
+
findings.append(f"[YAP] {path}:{lineno} | comment {len(raw)} chars (review)")
|
|
69
|
+
continue
|
|
70
|
+
flush()
|
|
71
|
+
if raw.strip():
|
|
72
|
+
header = False
|
|
73
|
+
|
|
74
|
+
flush()
|
|
75
|
+
return findings
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _lint_code(path: str, marker: re.Pattern) -> list[str]:
|
|
79
|
+
try:
|
|
80
|
+
lines = Path(path).read_text(encoding='utf-8', errors='replace').splitlines()
|
|
81
|
+
except OSError as e:
|
|
82
|
+
print(f"scythe: cannot read {path}: {e}", file=sys.stderr)
|
|
83
|
+
return []
|
|
84
|
+
return _scan_comment_runs(path, list(enumerate(lines, 1)), marker, True)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _marker_pattern(ext: str) -> re.Pattern | None:
|
|
88
|
+
if ext in SLASH_EXT:
|
|
89
|
+
return re.compile(r'^[ \t]*(//)')
|
|
90
|
+
if ext in HASH_EXT:
|
|
91
|
+
return re.compile(r'^[ \t]*(#)')
|
|
92
|
+
if ext in DASH_EXT:
|
|
93
|
+
return re.compile(r'^[ \t]*(--)')
|
|
94
|
+
if ext in HTML_EXT:
|
|
95
|
+
return re.compile(r'^[ \t]*(//)|(<!--)')
|
|
96
|
+
return None
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
# --- Markdown detector -------------------------------------------------------
|
|
100
|
+
|
|
101
|
+
# Fence info strings that name a real language, mapped to the extension whose comment marker they share.
|
|
102
|
+
# An untagged fence stays exempt: it usually holds verbatim output, where a "comment" is not authored text.
|
|
103
|
+
_FENCE_LANG = {
|
|
104
|
+
'ts': 'ts', 'typescript': 'ts', 'tsx': 'tsx', 'js': 'js', 'javascript': 'js', 'jsx': 'jsx', 'mjs': 'mjs',
|
|
105
|
+
'rs': 'rs', 'rust': 'rs', 'go': 'go', 'golang': 'go', 'c': 'c', 'h': 'h', 'cc': 'cc', 'cpp': 'cpp',
|
|
106
|
+
'c++': 'cpp', 'swift': 'swift', 'kt': 'kt', 'kotlin': 'kt', 'scss': 'scss',
|
|
107
|
+
'sh': 'sh', 'bash': 'sh', 'shell': 'sh', 'zsh': 'sh', 'py': 'py', 'python': 'py', 'rb': 'rb', 'ruby': 'rb',
|
|
108
|
+
'toml': 'toml', 'yaml': 'yaml', 'yml': 'yml',
|
|
109
|
+
'sql': 'sql', 'vue': 'vue', 'html': 'html',
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _fence_marker(info: str) -> re.Pattern | None:
|
|
114
|
+
return _marker_pattern(_FENCE_LANG.get(info.lower(), ''))
|
|
115
|
+
|
|
116
|
+
def _blockish(line: str) -> bool:
|
|
117
|
+
return bool(
|
|
118
|
+
re.match(r'^[ \t]*$', line)
|
|
119
|
+
or re.match(r'^#', line)
|
|
120
|
+
or re.match(r'^[ \t]*\|', line)
|
|
121
|
+
or re.match(r'^>', line)
|
|
122
|
+
or re.match(r'^<', line)
|
|
123
|
+
or re.match(r'^[ \t]{4}', line)
|
|
124
|
+
or re.match(r'^[ \t]*(---+|\*\*\*+|___+)[ \t]*$', line)
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _listitem(line: str) -> bool:
|
|
129
|
+
return bool(re.match(r'^[ \t]*([-*+]|[0-9]+[.)]|[a-z][.)])[ \t]', line))
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
# sig() classifies the structural marker of a line for the same-marker-run exemption.
|
|
133
|
+
# Returns "@", "bracket", "keyline", or "" (no marker → plain prose).
|
|
134
|
+
def _sig(line: str) -> str:
|
|
135
|
+
stripped = line.lstrip()
|
|
136
|
+
if stripped.startswith('@'):
|
|
137
|
+
return '@'
|
|
138
|
+
if stripped.startswith('['):
|
|
139
|
+
return 'bracket'
|
|
140
|
+
# One-or-two-word ASCII label closed by a colon (bold optional): **Key:** or Key:
|
|
141
|
+
if re.match(r'^\**[A-Za-z_][A-Za-z0-9_ -]*\**:([ \t]|$)', stripped):
|
|
142
|
+
return 'keyline'
|
|
143
|
+
return ''
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
# Terminal-punctuation pattern — a line ending with these chars is NOT a wrapped line.
|
|
147
|
+
_TERMINAL = re.compile(r"""([.!?:;…→]|-->)[)"'\]*_`]*[ \t]*$""")
|
|
148
|
+
_TRAILING_SPACES = re.compile(r' $')
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def _lint_md(path: str) -> list[str]:
|
|
152
|
+
"""Detect [WRAP] in markdown/mdx files."""
|
|
153
|
+
findings = []
|
|
154
|
+
try:
|
|
155
|
+
lines = Path(path).read_text(encoding='utf-8', errors='replace').splitlines()
|
|
156
|
+
except OSError as e:
|
|
157
|
+
print(f"scythe: cannot read {path}: {e}", file=sys.stderr)
|
|
158
|
+
return findings
|
|
159
|
+
|
|
160
|
+
front = False
|
|
161
|
+
fence = False
|
|
162
|
+
fence_marker: re.Pattern | None = None
|
|
163
|
+
fence_body: list[tuple[int, str]] = []
|
|
164
|
+
prev = ''
|
|
165
|
+
prevset = False
|
|
166
|
+
|
|
167
|
+
for lineno, raw in enumerate(lines, 1):
|
|
168
|
+
if lineno == 1 and re.match(r'^---[ \t]*$', raw):
|
|
169
|
+
front = True
|
|
170
|
+
continue
|
|
171
|
+
if front:
|
|
172
|
+
if re.match(r'^---[ \t]*$', raw):
|
|
173
|
+
front = False
|
|
174
|
+
continue
|
|
175
|
+
fm = re.match(r'^[ \t]*(?:```|~~~)[ \t]*([A-Za-z0-9_+#-]*)', raw)
|
|
176
|
+
if fm:
|
|
177
|
+
if fence:
|
|
178
|
+
if fence_marker is not None:
|
|
179
|
+
findings.extend(_scan_comment_runs(path, fence_body, fence_marker, False))
|
|
180
|
+
fence, fence_marker, fence_body = False, None, []
|
|
181
|
+
else:
|
|
182
|
+
fence, fence_marker, fence_body = True, _fence_marker(fm.group(1)), []
|
|
183
|
+
# A fence is a block boundary: the lines either side of it are not a wrapped pair.
|
|
184
|
+
prevset = False
|
|
185
|
+
continue
|
|
186
|
+
if fence:
|
|
187
|
+
if fence_marker is not None:
|
|
188
|
+
fence_body.append((lineno, raw))
|
|
189
|
+
continue
|
|
190
|
+
|
|
191
|
+
if prevset:
|
|
192
|
+
if (
|
|
193
|
+
not _blockish(prev)
|
|
194
|
+
and not _blockish(raw)
|
|
195
|
+
and not _listitem(raw)
|
|
196
|
+
and not _TERMINAL.search(prev)
|
|
197
|
+
and not _TRAILING_SPACES.search(prev)
|
|
198
|
+
# Two adjacent lines with the same non-empty structural marker are a machine-parsed run — exempt.
|
|
199
|
+
and not (_sig(prev) != '' and _sig(prev) == _sig(raw))
|
|
200
|
+
):
|
|
201
|
+
findings.append(f"[WRAP] {path}:{lineno - 1}-{lineno} | wrapped prose (rejoin)")
|
|
202
|
+
|
|
203
|
+
prev = raw
|
|
204
|
+
prevset = True
|
|
205
|
+
|
|
206
|
+
return findings
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def lint_file(path: str) -> list[str]:
|
|
210
|
+
p = Path(path)
|
|
211
|
+
if p.suffix in ('.md', '.mdx'):
|
|
212
|
+
return _lint_md(path)
|
|
213
|
+
ext = p.suffix.lstrip('.')
|
|
214
|
+
pat = _marker_pattern(ext)
|
|
215
|
+
if pat is None:
|
|
216
|
+
return []
|
|
217
|
+
return _lint_code(path, pat)
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def collect_files(target: str) -> list[str]:
|
|
221
|
+
p = Path(target)
|
|
222
|
+
if p.is_file():
|
|
223
|
+
return [str(p)]
|
|
224
|
+
if p.is_dir():
|
|
225
|
+
try:
|
|
226
|
+
result = subprocess.run(
|
|
227
|
+
['git', '-C', str(p), 'ls-files'],
|
|
228
|
+
capture_output=True, text=True, check=True
|
|
229
|
+
)
|
|
230
|
+
base = str(p).rstrip('/')
|
|
231
|
+
out = []
|
|
232
|
+
for f in result.stdout.splitlines():
|
|
233
|
+
full = f"{base}/{f}"
|
|
234
|
+
if Path(full).is_file():
|
|
235
|
+
out.append(full)
|
|
236
|
+
return out
|
|
237
|
+
except (subprocess.CalledProcessError, FileNotFoundError):
|
|
238
|
+
# Not a git repo or git not available — fall back to find.
|
|
239
|
+
return [str(f) for f in p.rglob('*') if f.is_file()]
|
|
240
|
+
print(f"scythe: no such path: {target}", file=sys.stderr)
|
|
241
|
+
sys.exit(2)
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def main() -> None:
|
|
245
|
+
show_all = False
|
|
246
|
+
cap = int(os.environ.get('SCYTHE_CAP', '40'))
|
|
247
|
+
targets: list[str] = []
|
|
248
|
+
|
|
249
|
+
for arg in sys.argv[1:]:
|
|
250
|
+
if arg == '--all':
|
|
251
|
+
show_all = True
|
|
252
|
+
else:
|
|
253
|
+
targets.append(arg)
|
|
254
|
+
|
|
255
|
+
if not targets:
|
|
256
|
+
print("usage: scythe.py [--all] <file|dir> [...]", file=sys.stderr)
|
|
257
|
+
sys.exit(2)
|
|
258
|
+
|
|
259
|
+
# Overlapping targets are a normal call shape, so identity is the real path: one file, one entry.
|
|
260
|
+
seen: dict[str, str] = {}
|
|
261
|
+
for t in targets:
|
|
262
|
+
for f in collect_files(t):
|
|
263
|
+
seen.setdefault(os.path.realpath(f), f)
|
|
264
|
+
|
|
265
|
+
all_findings: list[str] = []
|
|
266
|
+
for f in seen.values():
|
|
267
|
+
all_findings.extend(lint_file(f))
|
|
268
|
+
|
|
269
|
+
if not all_findings:
|
|
270
|
+
sys.exit(0)
|
|
271
|
+
|
|
272
|
+
total = len(all_findings)
|
|
273
|
+
|
|
274
|
+
if not show_all and total > cap:
|
|
275
|
+
for line in all_findings[:cap]:
|
|
276
|
+
print(line)
|
|
277
|
+
print(f"--- {total - cap} more findings suppressed ---")
|
|
278
|
+
|
|
279
|
+
# Per-tag totals (first space-delimited token is the tag, e.g. "[WRAP]").
|
|
280
|
+
tag_counts: dict[str, int] = {}
|
|
281
|
+
for line in all_findings:
|
|
282
|
+
tag = line.split()[0]
|
|
283
|
+
tag_counts[tag] = tag_counts.get(tag, 0) + 1
|
|
284
|
+
parts = ' '.join(f"{t} {c}" for t, c in tag_counts.items())
|
|
285
|
+
print(f"{parts} (total {total})")
|
|
286
|
+
|
|
287
|
+
# Worst-5 files: extract path by stripping ":digits..." suffix from the second token.
|
|
288
|
+
print("worst files:")
|
|
289
|
+
file_counts: dict[str, int] = {}
|
|
290
|
+
for line in all_findings:
|
|
291
|
+
tokens = line.split()
|
|
292
|
+
if len(tokens) >= 2:
|
|
293
|
+
file_path = re.sub(r':[0-9].*', '', tokens[1])
|
|
294
|
+
file_counts[file_path] = file_counts.get(file_path, 0) + 1
|
|
295
|
+
worst = sorted(file_counts.items(), key=lambda x: -x[1])[:5]
|
|
296
|
+
for file_path, count in worst:
|
|
297
|
+
print(f"{count:6d} {file_path}")
|
|
298
|
+
print("Narrow the path, or pass --all for every finding.")
|
|
299
|
+
else:
|
|
300
|
+
for line in all_findings:
|
|
301
|
+
print(line)
|
|
302
|
+
|
|
303
|
+
sys.exit(1)
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
if __name__ == "__main__":
|
|
307
|
+
main()
|