super-ux 0.28.0 → 0.30.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,957 @@
1
+ #!/usr/bin/env python3
2
+ """Lint a project's brand pack against `brand-contract v1` (stdlib only).
3
+
4
+ `ux_lint.py` checks the UX chain against itself; `ux_doctor.py` catches a
5
+ chain written to an old contract. This is the third question, about a
6
+ different artifact: does the text the product actually ships match the voice
7
+ the product wrote down?
8
+
9
+ It checks only what a machine can prove -- a banned word, one action under
10
+ two names, a number with no sourced fact, a field over its limit, a blocked
11
+ crawler. Everything evaluative -- tone drift, whether a claim lands, whether
12
+ the voice has overshot into its own failure mode -- belongs to the `copy`
13
+ scope of `ux-audit`, which reads the same pack and answers with evidence.
14
+
15
+ Read-only by default. `--fix` applies only the changes that cannot be wrong.
16
+
17
+ python3 brand_lint.py [path] # report (default docs/brand)
18
+ python3 brand_lint.py [path] --fix # apply the safe subset
19
+ python3 brand_lint.py [path] --brief # one line, for sweeping projects
20
+ python3 brand_lint.py [path] --json # machine-readable findings
21
+
22
+ Exit codes: 0 clean, 1 warnings only, 2 any error.
23
+ """
24
+
25
+ from __future__ import annotations
26
+
27
+ import argparse
28
+ import json
29
+ import re
30
+ import sys
31
+ from collections import namedtuple
32
+ from pathlib import Path
33
+
34
+ CONTRACT = "brand-contract"
35
+ CONTRACT_VERSION = "v1"
36
+
37
+ SEVERITY_ERROR = "error"
38
+ SEVERITY_WARN = "warn"
39
+
40
+ Finding = namedtuple("Finding", "code severity path line message")
41
+
42
+ # The files the contract owns. `locales/` is a directory and optional; a
43
+ # project with one locale legitimately has none.
44
+ CONTRACT_FILES = (
45
+ "voice.md", "terminology.md", "facts.md",
46
+ "channels.md", "strings.md", "README.md",
47
+ )
48
+
49
+ SOURCE_KEYS = ("ui", "marketing", "store", "robots", "locales")
50
+
51
+ MARKER_RE = re.compile(rf"^Contract:\s*{CONTRACT}\s*(v\d+)\s*$", re.M)
52
+
53
+
54
+ def unfilled(value: str) -> bool:
55
+ """A template placeholder, not data.
56
+
57
+ Templates ship worked examples so the shape is unambiguous, and a project
58
+ mid-fill has some rows done and some not. `<...>` means "nobody has filled
59
+ this in yet" -- reporting it as a defect would make every freshly seeded
60
+ project fail on its own scaffolding, which teaches people to ignore the
61
+ linter on day one.
62
+ """
63
+ value = value.strip()
64
+ return not value or (value.startswith("<") and value.endswith(">"))
65
+
66
+
67
+ def read(path: Path) -> str | None:
68
+ try:
69
+ return path.read_text(encoding="utf-8")
70
+ except OSError:
71
+ return None
72
+
73
+
74
+ def header_field(text: str, key: str) -> str | None:
75
+ """A `Key: value` line from a file's header block."""
76
+ match = re.search(rf"^{re.escape(key)}:\s*(.+?)\s*$", text, re.M)
77
+ return match.group(1) if match else None
78
+
79
+
80
+ def table_rows(text: str) -> list[list[str]]:
81
+ """Every pipe-table data row, as trimmed cell lists.
82
+
83
+ Separator rows (`|---|---|`) and header rows are indistinguishable from
84
+ data by shape alone, so the separator is dropped and the caller decides
85
+ what the first surviving row means.
86
+ """
87
+ rows = []
88
+ for line in text.splitlines():
89
+ line = line.strip()
90
+ if not line.startswith("|") or not line.endswith("|"):
91
+ continue
92
+ cells = [c.strip() for c in line.strip("|").split("|")]
93
+ if all(re.fullmatch(r":?-{2,}:?", c) for c in cells if c):
94
+ continue
95
+ rows.append(cells)
96
+ return rows
97
+
98
+
99
+ def load_sources(brand_dir: Path) -> dict[str, list[str]]:
100
+ """The `Sources:` block from README.md -- key -> glob patterns.
101
+
102
+ The linter cannot guess where a project keeps its text, and guessing
103
+ wrong produces the worst possible output: a clean report about a surface
104
+ that was never read. So an absent block is a finding (B006) and an
105
+ absent key means its checks are skipped and counted as skipped.
106
+ """
107
+ text = read(brand_dir / "README.md") or ""
108
+ block = re.search(r"^Sources:\s*$(.*?)(?=^\S|\Z)", text, re.M | re.S)
109
+ if not block:
110
+ return {}
111
+ sources: dict[str, list[str]] = {}
112
+ for line in block.group(1).splitlines():
113
+ entry = re.match(r"^\s+(\w+):\s*(.+?)\s*$", line)
114
+ if not entry:
115
+ continue
116
+ key, value = entry.group(1), entry.group(2)
117
+ if key in SOURCE_KEYS:
118
+ sources[key] = [p.strip() for p in value.split() if p.strip()]
119
+ return sources
120
+
121
+
122
+ def check_contract(brand_dir: Path) -> list[Finding]:
123
+ """B001-B006 -- the pack announces its contract and what to scan."""
124
+ findings: list[Finding] = []
125
+ versions: dict[str, str] = {}
126
+
127
+ for path in sorted(brand_dir.rglob("*.md")):
128
+ rel = path.relative_to(brand_dir).as_posix()
129
+ text = read(path) or ""
130
+ marker = MARKER_RE.search(text)
131
+ if not marker:
132
+ findings.append(Finding(
133
+ "B001", SEVERITY_ERROR, rel, 1,
134
+ f"no `Contract: {CONTRACT} {CONTRACT_VERSION}` marker -- "
135
+ f"without it a pack written to an old contract is "
136
+ f"indistinguishable from a current one",
137
+ ))
138
+ continue
139
+ versions[rel] = marker.group(1)
140
+
141
+ distinct = set(versions.values())
142
+ if len(distinct) > 1:
143
+ listed = ", ".join(
144
+ f"{rel} {ver}" for rel, ver in sorted(versions.items())
145
+ )
146
+ findings.append(Finding(
147
+ "B002", SEVERITY_ERROR, "", 0,
148
+ f"mixed contract versions in one pack: {listed}",
149
+ ))
150
+
151
+ voice = read(brand_dir / "voice.md") or ""
152
+ status = header_field(voice, "Status")
153
+ strings = read(brand_dir / "strings.md") or ""
154
+ agreed = [r for r in table_rows(strings) if r and r[-1] == "agreed"]
155
+ if status == "draft" and agreed:
156
+ findings.append(Finding(
157
+ "B003", SEVERITY_WARN, "voice.md", 1,
158
+ f"voice.md is `draft` while strings.md already has "
159
+ f"{len(agreed)} agreed string(s) -- they were agreed against a "
160
+ f"voice nobody approved",
161
+ ))
162
+
163
+ derived = header_field(voice, "Derived-from")
164
+ if derived and derived != "inferred" and not unfilled(derived):
165
+ foundation = read(brand_dir.parent / "ux" / "foundation.md")
166
+ ids = [i.strip() for i in derived.split(",") if i.strip()]
167
+ if foundation is not None:
168
+ for ident in ids:
169
+ if ident not in foundation:
170
+ findings.append(Finding(
171
+ "B004", SEVERITY_ERROR, "voice.md", 1,
172
+ f"Derived-from references `{ident}`, which is not in "
173
+ f"docs/ux/foundation.md -- the trace is broken",
174
+ ))
175
+ calibrated = header_field(voice, "Last calibrated")
176
+ if foundation is not None and calibrated:
177
+ try:
178
+ stamp = (brand_dir.parent / "ux" / "foundation.md").stat().st_mtime
179
+ import datetime
180
+
181
+ changed = datetime.date.fromtimestamp(stamp).isoformat()
182
+ if changed > calibrated:
183
+ findings.append(Finding(
184
+ "B005", SEVERITY_WARN, "voice.md", 1,
185
+ f"foundation.md changed on {changed}, after the voice "
186
+ f"was last calibrated on {calibrated}",
187
+ ))
188
+ except OSError:
189
+ pass
190
+
191
+ if not load_sources(brand_dir):
192
+ findings.append(Finding(
193
+ "B006", SEVERITY_ERROR, "README.md", 1,
194
+ "no `Sources:` block -- the linter has nothing to scan, and a "
195
+ "clean report over a surface it never read is worse than no "
196
+ "report",
197
+ ))
198
+
199
+ return findings
200
+
201
+
202
+ def registry(brand_dir: Path) -> list[dict]:
203
+ """`strings.md` data rows as dicts, header dropped."""
204
+ rows = []
205
+ for cells in table_rows(read(brand_dir / "strings.md") or ""):
206
+ if len(cells) < 5 or cells[0].strip().lower() == "key":
207
+ continue
208
+ if unfilled(cells[0]) or unfilled(cells[1]) or unfilled(cells[2]):
209
+ continue
210
+ rows.append({
211
+ "key": cells[0], "text": cells[1], "location": cells[2],
212
+ "scenario": cells[3], "status": cells[4],
213
+ })
214
+ return rows
215
+
216
+
217
+ def dictionary(brand_dir: Path) -> tuple[list, list, list]:
218
+ """(banned, product terms, entity names) from `terminology.md`."""
219
+ text = read(brand_dir / "terminology.md") or ""
220
+ sections: dict[str, list[str]] = {}
221
+ current = None
222
+ for line in text.splitlines():
223
+ heading = re.match(r"^##\s+(.*?)\s*$", line)
224
+ if heading:
225
+ current = heading.group(1)
226
+ sections[current] = []
227
+ elif current:
228
+ sections[current].append(line)
229
+
230
+ def rows(fragment: str, header: str) -> list[list[str]]:
231
+ for title, lines in sections.items():
232
+ if fragment.lower() in title.lower():
233
+ return [
234
+ r for r in table_rows("\n".join(lines))
235
+ if r and r[0].strip().lower() != header
236
+ ]
237
+ return []
238
+
239
+ banned = [r[0] for r in rows("Banned", "word or phrase") if r[0]]
240
+ terms = [
241
+ (r[0], r[1]) for r in rows("Product terms", "our term") if len(r) > 1
242
+ ]
243
+ entities = [
244
+ (r[0], r[1]) for r in rows("Entity and tier", "name") if len(r) > 1
245
+ ]
246
+ return banned, terms, entities
247
+
248
+
249
+ def _alternatives(cell: str) -> list[str]:
250
+ """A comma-separated cell as a list, placeholders dropped."""
251
+ out = []
252
+ for part in cell.split(","):
253
+ part = part.strip()
254
+ if part and not part.startswith("<"):
255
+ out.append(part)
256
+ return out
257
+
258
+
259
+ def _mentions(needle: str, haystack: str) -> bool:
260
+ return bool(re.search(rf"(?<!\w){re.escape(needle)}(?!\w)", haystack, re.I))
261
+
262
+
263
+ def check_terminology(brand_dir: Path) -> list[Finding]:
264
+ """B010-B012 -- the dictionary is law, in the interface and in copy."""
265
+ findings: list[Finding] = []
266
+ banned, terms, entities = dictionary(brand_dir)
267
+ for row in registry(brand_dir):
268
+ text = row["text"]
269
+ for word in banned:
270
+ if _mentions(word, text):
271
+ findings.append(Finding(
272
+ "B010", SEVERITY_ERROR, row["location"], 0,
273
+ f"`{row['key']}` uses the banned word `{word}`: "
274
+ f"\"{text}\"",
275
+ ))
276
+ for ours, generic_cell in terms:
277
+ for generic in _alternatives(generic_cell):
278
+ if _mentions(generic, text):
279
+ findings.append(Finding(
280
+ "B011", SEVERITY_ERROR, row["location"], 0,
281
+ f"`{row['key']}` says `{generic}` where the product "
282
+ f"term is `{ours}`: \"{text}\"",
283
+ ))
284
+ for name, wrong_cell in entities:
285
+ for wrong in _alternatives(wrong_cell):
286
+ if wrong and wrong in text:
287
+ findings.append(Finding(
288
+ "B012", SEVERITY_ERROR, row["location"], 0,
289
+ f"`{row['key']}` spells the entity `{name}` as "
290
+ f"`{wrong}`: \"{text}\"",
291
+ ))
292
+ return findings
293
+
294
+
295
+ WEAK_LABELS = {
296
+ "ok", "yes", "no", "submit", "done", "go", "click here",
297
+ "learn more", "get started", "continue",
298
+ }
299
+
300
+ LITERAL_RE = re.compile(r"""["']([^"'\n]{3,60})["']""")
301
+
302
+
303
+ def _looks_like_copy(literal: str) -> bool:
304
+ """A quoted literal that could plausibly be user-visible text."""
305
+ if not literal or literal[0].islower() and " " not in literal:
306
+ return False
307
+ return bool(re.match(r"^[A-Z]", literal)) or " " in literal
308
+
309
+
310
+ def check_consistency(brand_dir: Path, sources: dict) -> list[Finding]:
311
+ """B020-B025 -- the registry, the code and the casing agree."""
312
+ findings: list[Finding] = []
313
+ root = brand_dir.parent.parent
314
+ rows = registry(brand_dir)
315
+ _, _, entities = dictionary(brand_dir)
316
+ entity_words = {name for name, _ in entities}
317
+
318
+ by_key: dict[str, set] = {}
319
+ for row in rows:
320
+ by_key.setdefault(row["key"], set()).add(row["text"])
321
+ for key, texts in sorted(by_key.items()):
322
+ if len(texts) > 1:
323
+ listed = " / ".join(f'"{t}"' for t in sorted(texts))
324
+ findings.append(Finding(
325
+ "B020", SEVERITY_ERROR, "strings.md", 0,
326
+ f"one action, two names -- `{key}` is {listed}. An action "
327
+ f"keeps one name across the whole flow",
328
+ ))
329
+
330
+ for row in rows:
331
+ location = row["location"]
332
+ file_part = location.split(":")[0]
333
+ target = root / file_part
334
+ if not target.is_file():
335
+ findings.append(Finding(
336
+ "B023", SEVERITY_ERROR, location, 0,
337
+ f"`{row['key']}` points at {file_part}, which does not exist",
338
+ ))
339
+ continue
340
+
341
+ body = read(target) or ""
342
+ literals = [
343
+ lit for lit in LITERAL_RE.findall(body) if _looks_like_copy(lit)
344
+ ]
345
+ if literals:
346
+ if row["text"] not in body:
347
+ findings.append(Finding(
348
+ "B021", SEVERITY_ERROR, location, 0,
349
+ f"`{row['key']}` is \"{row['text']}\" in the registry, "
350
+ f"but that text is not in {file_part}",
351
+ ))
352
+ known = {r["text"] for r in rows}
353
+ for lit in literals:
354
+ if lit not in known:
355
+ findings.append(Finding(
356
+ "B022", SEVERITY_WARN, f"{file_part}", 0,
357
+ f"\"{lit}\" is in the code with no registry row -- "
358
+ f"agree it or retire it",
359
+ ))
360
+
361
+ for row in rows:
362
+ words = row["text"].split()
363
+ for word in words[1:]:
364
+ bare = word.strip(".,:;!?()")
365
+ if not bare or bare in entity_words:
366
+ continue
367
+ if bare.isupper() and len(bare) <= 4:
368
+ continue
369
+ if bare[0].isupper():
370
+ findings.append(Finding(
371
+ "B024", SEVERITY_ERROR, row["location"], 0,
372
+ f"`{row['key']}` is not sentence case: \"{row['text']}\"",
373
+ ))
374
+ break
375
+ if row["key"].startswith("button.") and \
376
+ row["text"].strip().lower() in WEAK_LABELS:
377
+ findings.append(Finding(
378
+ "B025", SEVERITY_WARN, row["location"], 0,
379
+ f"`{row['key']}` is \"{row['text']}\" -- a button says what "
380
+ f"happens, not that something happens",
381
+ ))
382
+
383
+ return findings
384
+
385
+
386
+ FRONT_MATTER_RE = re.compile(r"^---\n(.*?)\n---\n", re.S)
387
+ NUMBER_RE = re.compile(r"\d+\s?%|[$€£]\s?\d[\d,.]*|\b\d{3,}\b")
388
+ SUPERLATIVES = (
389
+ "the best", "best-in-class", "leading", "fastest", "most trusted",
390
+ "#1", "number one", "world-class", "unmatched",
391
+ )
392
+
393
+
394
+ def _front_matter(text: str) -> tuple[dict, str]:
395
+ match = FRONT_MATTER_RE.match(text)
396
+ if not match:
397
+ return {}, text
398
+ fields = {}
399
+ for line in match.group(1).splitlines():
400
+ pair = re.match(r"^([\w-]+):\s*(.*)$", line)
401
+ if pair:
402
+ fields[pair.group(1)] = pair.group(2).strip()
403
+ return fields, text[match.end():]
404
+
405
+
406
+ def documents(brand_dir: Path, sources: dict, key: str) -> list[tuple]:
407
+ """(relative path, front matter, body) for one declared source key."""
408
+ root = brand_dir.parent.parent
409
+ out = []
410
+ for pattern in sources.get(key, []):
411
+ for path in sorted(root.glob(pattern)):
412
+ if not path.is_file():
413
+ continue
414
+ fields, body = _front_matter(read(path) or "")
415
+ out.append((path.relative_to(root).as_posix(), fields, body))
416
+ return out
417
+
418
+
419
+ def surfaces(brand_dir: Path) -> dict[str, dict]:
420
+ """`channels.md` records: surface name -> field dict."""
421
+ text = read(brand_dir / "channels.md") or ""
422
+ out: dict[str, dict] = {}
423
+ for match in re.finditer(r"^### (.+?)$\s*```(.*?)```", text, re.M | re.S):
424
+ record = {}
425
+ for line in match.group(2).splitlines():
426
+ pair = re.match(r"^([A-Za-z][A-Za-z ]*?):\s*(.*)$", line)
427
+ if pair:
428
+ record[pair.group(1).strip()] = pair.group(2).strip()
429
+ out[match.group(1).strip()] = record
430
+ return out
431
+
432
+
433
+ def facts(brand_dir: Path) -> list[dict]:
434
+ rows = []
435
+ for cells in table_rows(read(brand_dir / "facts.md") or ""):
436
+ if len(cells) < 6 or cells[0].strip().lower() == "fact":
437
+ continue
438
+ if unfilled(cells[0]) or unfilled(cells[1]):
439
+ continue
440
+ rows.append({
441
+ "fact": cells[0], "value": cells[1], "source": cells[2],
442
+ "checked": cells[3], "review": cells[4], "public": cells[5],
443
+ })
444
+ return rows
445
+
446
+
447
+ def _today() -> str:
448
+ import datetime
449
+
450
+ return datetime.date.today().isoformat()
451
+
452
+
453
+ def check_facts(brand_dir: Path, sources: dict) -> list[Finding]:
454
+ """B030-B032 -- every figure traces to a row, every row to a source."""
455
+ findings: list[Finding] = []
456
+ rows = facts(brand_dir)
457
+ known = " ".join(r["value"] for r in rows if r["public"].lower() != "no")
458
+
459
+ for row in rows:
460
+ if unfilled(row["source"]):
461
+ findings.append(Finding(
462
+ "B031", SEVERITY_WARN, "facts.md", 0,
463
+ f"`{row['fact']}` has no source -- an unsourced fact is an "
464
+ f"opinion with a number on it",
465
+ ))
466
+ elif row["review"] and not row["review"].startswith("<") \
467
+ and row["review"] < _today():
468
+ findings.append(Finding(
469
+ "B031", SEVERITY_WARN, "facts.md", 0,
470
+ f"`{row['fact']}` was due for review on {row['review']}",
471
+ ))
472
+
473
+ for path, _fields, body in documents(brand_dir, sources, "marketing"):
474
+ for number in NUMBER_RE.findall(body):
475
+ compact = number.replace(" ", "")
476
+ if compact not in known.replace(" ", ""):
477
+ findings.append(Finding(
478
+ "B030", SEVERITY_ERROR, path, 0,
479
+ f"`{number}` appears in public copy with no row in "
480
+ f"facts.md -- a number nobody can check is a claim "
481
+ f"nobody should make",
482
+ ))
483
+ for paragraph in re.split(r"\n\s*\n", body):
484
+ lowered = paragraph.lower()
485
+ for superlative in SUPERLATIVES:
486
+ if superlative in lowered and not re.search(r"\d", paragraph):
487
+ findings.append(Finding(
488
+ "B032", SEVERITY_ERROR, path, 0,
489
+ f"`{superlative}` with nothing beside it to back it",
490
+ ))
491
+ break
492
+ return findings
493
+
494
+
495
+ def _limits(record: dict) -> dict[str, int]:
496
+ out: dict[str, int] = {}
497
+ for part in record.get("Limits", "").split(","):
498
+ pair = re.match(r"^\s*([\w ]+?)\s+(\d+)\s*$", part)
499
+ if pair:
500
+ out[pair.group(1).strip().lower()] = int(pair.group(2))
501
+ return out
502
+
503
+
504
+ def _coefficient(brand_dir: Path, locale: str | None) -> float:
505
+ if not locale:
506
+ return 1.0
507
+ text = read(brand_dir / "locales" / f"{locale}.md") or ""
508
+ value = header_field(text, "Length coefficient")
509
+ try:
510
+ return float(value) if value else 1.0
511
+ except ValueError:
512
+ return 1.0
513
+
514
+
515
+ def check_channels(brand_dir: Path, sources: dict) -> list[Finding]:
516
+ """B040-B043 -- platform physics, applied with the locale's coefficient."""
517
+ findings: list[Finding] = []
518
+ records = surfaces(brand_dir)
519
+
520
+ for key in ("marketing", "store"):
521
+ for path, fields, body in documents(brand_dir, sources, key):
522
+ record = records.get(fields.get("surface", ""))
523
+ if not record:
524
+ continue
525
+ locale = fields.get("locale")
526
+ factor = _coefficient(brand_dir, locale)
527
+ for name, limit in _limits(record).items():
528
+ value = body if name == "body" else fields.get(name, "")
529
+ allowed = int(limit * factor)
530
+ if value and len(value.strip()) > allowed:
531
+ # Same overflow, two codes on purpose: B040 is the
532
+ # primary-locale case, B073 the one the coefficient
533
+ # created. They are fixed differently -- one shortens the
534
+ # string, the other questions the original design.
535
+ code = "B073" if locale else "B040"
536
+ findings.append(Finding(
537
+ code, SEVERITY_ERROR, path, 0,
538
+ f"{name} is {len(value.strip())} characters, over the "
539
+ f"{allowed} this surface allows"
540
+ + (f" for `{locale}` (coefficient {factor})"
541
+ if locale else ""),
542
+ ))
543
+
544
+ physics = record.get("Forbidden", "").split("|")[0].lower()
545
+ if "link in body" in physics and re.search(r"https?://", body):
546
+ findings.append(Finding(
547
+ "B042", SEVERITY_ERROR, path, 0,
548
+ "link in the post body -- this surface suppresses reach "
549
+ "for it; the convention is the first reply",
550
+ ))
551
+ cap = re.search(r"max (\d+) hashtags", physics)
552
+ if cap:
553
+ used = re.findall(r"(?<!\w)#\w+", body)
554
+ if len(used) > int(cap.group(1)):
555
+ findings.append(Finding(
556
+ "B043", SEVERITY_WARN, path, 0,
557
+ f"{len(used)} hashtags, over the {cap.group(1)} this "
558
+ f"surface tolerates",
559
+ ))
560
+
561
+ keywords = fields.get("keywords")
562
+ if keywords is not None:
563
+ findings.extend(_ios_keywords(path, fields, keywords))
564
+ return findings
565
+
566
+
567
+ def _ios_keywords(path: str, fields: dict, raw: str) -> list[Finding]:
568
+ """B041 -- the four rules that recover a third of the 100-character field."""
569
+ findings = []
570
+ if ", " in raw:
571
+ findings.append(Finding(
572
+ "B041", SEVERITY_ERROR, path, 0,
573
+ "space after a comma in the keyword field -- each one is a "
574
+ "character bought for nothing",
575
+ ))
576
+ terms = [t.strip() for t in raw.split(",") if t.strip()]
577
+ singulars = {t[:-1] for t in terms if t.endswith("s")}
578
+ for term in terms:
579
+ if term in singulars:
580
+ findings.append(Finding(
581
+ "B041", SEVERITY_ERROR, path, 0,
582
+ f"`{term}` and its plural are both listed -- the store "
583
+ f"matches both forms from the singular",
584
+ ))
585
+ title = fields.get("title", "").lower()
586
+ for term in terms:
587
+ if term.lower() in title.split():
588
+ findings.append(Finding(
589
+ "B041", SEVERITY_ERROR, path, 0,
590
+ f"`{term}` is already in the title, which is indexed at "
591
+ f"higher weight -- the field spends it twice",
592
+ ))
593
+ return findings
594
+
595
+
596
+ AI_AGENTS = ("GPTBot", "ClaudeBot", "PerplexityBot", "Google-Extended")
597
+
598
+ FILLER_OPENERS = (
599
+ "in today's digital landscape", "in the ever-evolving world",
600
+ "in today's fast-paced", "in an increasingly", "in the modern era",
601
+ )
602
+
603
+ # S1 markers only -- the ones decisive on their own. The full catalogue,
604
+ # with S2 and S3, is in references/ai-tells.md.
605
+ S1_MARKERS = (
606
+ "delve", "it is important to note", "it's important to note",
607
+ "it is worth noting", "in conclusion", "needless to say",
608
+ "landscape of", "leverage the", "robust and", "seamless integration",
609
+ "crucial to", "navigate the complexities",
610
+ )
611
+
612
+ STOPWORDS = {
613
+ "the", "and", "for", "with", "that", "this", "from", "your", "you",
614
+ "are", "was", "were", "have", "has", "had", "not", "but", "all",
615
+ "can", "will", "into", "than", "then", "them", "they", "our", "its",
616
+ "some", "other", "here", "more", "most", "when", "what", "which",
617
+ }
618
+
619
+ SENSITIVE_PREFIXES = ("error.", "destructive.", "billing.", "paywall.")
620
+ EMOJI_RE = re.compile(
621
+ "[\U0001F300-\U0001FAFF☀-➿️]"
622
+ )
623
+
624
+
625
+ def check_bot_safety(brand_dir: Path, sources: dict) -> list[Finding]:
626
+ """B050-B054 -- do not write text that looks like gaming a crawler."""
627
+ findings: list[Finding] = []
628
+ channels = read(brand_dir / "channels.md") or ""
629
+ targets_ai = "AI search: target" in channels
630
+
631
+ if targets_ai and "robots" in sources:
632
+ for path, _fields, _body in [
633
+ (p, {}, "") for p in sources["robots"]
634
+ ]:
635
+ robots = read(brand_dir.parent.parent / path) or ""
636
+ blocks = []
637
+ agent = None
638
+ for line in robots.splitlines():
639
+ head = re.match(r"^User-agent:\s*(.+?)\s*$", line, re.I)
640
+ if head:
641
+ agent = head.group(1).strip()
642
+ elif re.match(r"^Disallow:\s*/\s*$", line, re.I) and agent:
643
+ if agent in AI_AGENTS:
644
+ blocks.append(agent)
645
+ if blocks:
646
+ findings.append(Finding(
647
+ "B050", SEVERITY_ERROR, path, 0,
648
+ f"channels.md declares AI search a target while "
649
+ f"{', '.join(blocks)} is blocked here -- content quality "
650
+ f"is irrelevant to a crawler that never arrives",
651
+ ))
652
+
653
+ records = surfaces(brand_dir)
654
+ for path, fields, body in documents(brand_dir, sources, "marketing"):
655
+ words = [w.lower().strip(".,:;!?()\"'") for w in body.split()]
656
+ real = [w for w in words if len(w) > 3 and w not in STOPWORDS]
657
+ if len(real) >= 40:
658
+ counts: dict[str, int] = {}
659
+ for word in real:
660
+ counts[word] = counts.get(word, 0) + 1
661
+ for word, count in sorted(counts.items()):
662
+ if count / len(words) > 0.01:
663
+ findings.append(Finding(
664
+ "B051", SEVERITY_ERROR, path, 0,
665
+ f"`{word}` is {count / len(words):.1%} of the "
666
+ f"document -- above 1% reads as stuffing, which "
667
+ f"lowers citation likelihood rather than raising it",
668
+ ))
669
+ break
670
+
671
+ opening = body.strip().lower()[:120]
672
+ for filler in FILLER_OPENERS:
673
+ if opening.startswith(filler) or f"\n{filler}" in opening:
674
+ findings.append(Finding(
675
+ "B052", SEVERITY_ERROR, path, 0,
676
+ f"filler opener \"{filler}…\" -- it delays the answer "
677
+ f"past the point where extraction happens",
678
+ ))
679
+ break
680
+
681
+ record = records.get(fields.get("surface", ""))
682
+ if record and "author" in record.get("Proof", "").lower():
683
+ if not fields.get("author"):
684
+ findings.append(Finding(
685
+ "B053", SEVERITY_WARN, path, 0,
686
+ "no named author, and this surface makes claims that "
687
+ "need one",
688
+ ))
689
+
690
+ title = fields.get("title", "")
691
+ promised = re.match(r"^\s*(\d+)\b", title)
692
+ if promised:
693
+ items = len(re.findall(r"^\s*(?:[-*]|\d+\.)\s+", body, re.M))
694
+ headings = len(re.findall(r"^#{2,}\s+", body, re.M))
695
+ if max(items, headings) < int(promised.group(1)):
696
+ findings.append(Finding(
697
+ "B054", SEVERITY_WARN, path, 0,
698
+ f"the title promises {promised.group(1)} and the body "
699
+ f"delivers {max(items, headings)}",
700
+ ))
701
+ return findings
702
+
703
+
704
+ def check_ai_tells(brand_dir: Path, sources: dict) -> list[Finding]:
705
+ """B060-B061 -- machine-drafting markers, and the one absolute ban."""
706
+ findings: list[Finding] = []
707
+
708
+ for path, _fields, body in documents(brand_dir, sources, "marketing"):
709
+ lowered = body.lower()
710
+ hits = [m for m in S1_MARKERS if m in lowered]
711
+ if not hits:
712
+ continue
713
+ grade = "B" if len(hits) < 3 else "C"
714
+ severity = SEVERITY_ERROR if len(hits) >= 3 else SEVERITY_WARN
715
+ findings.append(Finding(
716
+ "B060", severity, path, 0,
717
+ f"{len(hits)} S1 marker(s) -- {', '.join(sorted(hits))}. "
718
+ f"Naturalness grade {grade}",
719
+ ))
720
+
721
+ for row in registry(brand_dir):
722
+ if not row["key"].startswith(SENSITIVE_PREFIXES):
723
+ continue
724
+ text = row["text"]
725
+ reason = None
726
+ if "!" in text:
727
+ reason = "an exclamation mark"
728
+ elif EMOJI_RE.search(text):
729
+ reason = "an emoji"
730
+ if reason:
731
+ findings.append(Finding(
732
+ "B061", SEVERITY_ERROR, row["location"], 0,
733
+ f"`{row['key']}` carries {reason}. The user is losing data, "
734
+ f"access or money on this surface; levity reads as mockery "
735
+ f"of a loss the product caused",
736
+ ))
737
+ return findings
738
+
739
+
740
+ def declared_locales(brand_dir: Path) -> tuple[str | None, list[str]]:
741
+ """(primary, others) from voice.md's `Locales:` line."""
742
+ text = read(brand_dir / "voice.md") or ""
743
+ raw = header_field(text, "Locales") or ""
744
+ primary, others = None, []
745
+ for part in raw.split(","):
746
+ part = part.strip()
747
+ if not part:
748
+ continue
749
+ code = part.split()[0]
750
+ if "(primary)" in part:
751
+ primary = code
752
+ else:
753
+ others.append(code)
754
+ return primary, others
755
+
756
+
757
+ def check_locales(brand_dir: Path, sources: dict) -> list[Finding]:
758
+ """B070-B072 -- a locale may lag, but it may not hide that it lags."""
759
+ findings: list[Finding] = []
760
+ primary, others = declared_locales(brand_dir)
761
+ root = brand_dir.parent.parent
762
+
763
+ for code in others:
764
+ if not (brand_dir / "locales" / f"{code}.md").is_file():
765
+ findings.append(Finding(
766
+ "B070", SEVERITY_ERROR, "voice.md", 0,
767
+ f"`{code}` is declared but has no locales/{code}.md -- "
768
+ f"nothing records its address form, humor level or "
769
+ f"length coefficient",
770
+ ))
771
+
772
+ threshold = header_field(read(brand_dir / "voice.md") or "",
773
+ "Locale parity threshold")
774
+ limit = 0.0
775
+ if threshold:
776
+ try:
777
+ limit = float(threshold.rstrip("%")) / 100
778
+ except ValueError:
779
+ limit = 0.0
780
+
781
+ if primary and limit and "locales" in sources:
782
+ catalogues: dict[str, set] = {}
783
+ for pattern in sources["locales"]:
784
+ for path in sorted(root.glob(pattern)):
785
+ try:
786
+ data = json.loads(read(path) or "{}")
787
+ except json.JSONDecodeError:
788
+ continue
789
+ if isinstance(data, dict):
790
+ catalogues[path.stem] = set(data)
791
+ base = catalogues.get(primary, set())
792
+ for code in others:
793
+ keys = catalogues.get(code)
794
+ if base and keys is not None:
795
+ parity = len(keys & base) / len(base)
796
+ if parity < limit:
797
+ findings.append(Finding(
798
+ "B071", SEVERITY_WARN, f"locales/{code}.md", 0,
799
+ f"{code} covers {parity:.0%} of {primary}, under the "
800
+ f"declared {limit:.0%} -- {len(base - keys)} string(s) "
801
+ f"behind",
802
+ ))
803
+
804
+ for code in others:
805
+ text = read(brand_dir / "locales" / f"{code}.md") or ""
806
+ for cells in table_rows(text):
807
+ if len(cells) < 2 or cells[0].startswith("<"):
808
+ continue
809
+ if cells[0].strip().lower() in ("primary", "term"):
810
+ continue
811
+ if cells[0] and cells[0] == cells[1]:
812
+ findings.append(Finding(
813
+ "B072", SEVERITY_WARN, f"locales/{code}.md", 0,
814
+ f"`{cells[0]}` is unchanged from the primary -- a "
815
+ f"word-for-word rendering translates the words and not "
816
+ f"the job the string does",
817
+ ))
818
+ return findings
819
+
820
+
821
+ FIXABLE = ("B024", "B041", "B023")
822
+
823
+
824
+ def apply_fixes(brand_dir: Path, findings: list[Finding]) -> int:
825
+ """The subset that cannot be wrong. Everything else needs a human.
826
+
827
+ B024 normalises casing, B041 tightens the iOS keyword field, and B023
828
+ re-points a registry row when the string is unchanged and matches
829
+ exactly one new location. Anything requiring a judgement about meaning
830
+ is reported, never rewritten.
831
+ """
832
+ rewritten = 0
833
+ root = brand_dir.parent.parent
834
+ strings_path = brand_dir / "strings.md"
835
+ text = read(strings_path)
836
+ if text is None:
837
+ return 0
838
+ original = text
839
+
840
+ _, _, entities = dictionary(brand_dir)
841
+ entity_words = {name for name, _ in entities}
842
+ for finding in findings:
843
+ if finding.code != "B024":
844
+ continue
845
+ for row in registry(brand_dir):
846
+ words = row["text"].split()
847
+ fixed = [words[0]] + [
848
+ w if (w.strip(".,:;!?()") in entity_words
849
+ or (w.isupper() and len(w) <= 4))
850
+ else w[0].lower() + w[1:]
851
+ for w in words[1:]
852
+ ]
853
+ replacement = " ".join(fixed)
854
+ if replacement != row["text"]:
855
+ text = text.replace(
856
+ f"| {row['text']} |", f"| {replacement} |"
857
+ )
858
+
859
+ for finding in findings:
860
+ if finding.code != "B023":
861
+ continue
862
+ row_key = finding.message.split("`")[1]
863
+ for row in registry(brand_dir):
864
+ if row["key"] != row_key:
865
+ continue
866
+ matches = [
867
+ p for p in root.rglob("*")
868
+ if p.is_file() and p.suffix in (".ts", ".tsx", ".js", ".jsx")
869
+ and row["text"] in (read(p) or "")
870
+ ]
871
+ if len(matches) == 1:
872
+ new = matches[0].relative_to(root).as_posix()
873
+ text = text.replace(row["location"], f"{new}:1")
874
+
875
+ if text != original:
876
+ strings_path.write_text(text, encoding="utf-8")
877
+ rewritten += 1
878
+
879
+ for path, fields, _body in documents(brand_dir, load_sources(brand_dir),
880
+ "store"):
881
+ raw = fields.get("keywords")
882
+ if not raw or ", " not in raw:
883
+ continue
884
+ target = root / path
885
+ body = read(target) or ""
886
+ tightened = raw.replace(", ", ",")
887
+ target.write_text(
888
+ body.replace(f"keywords: {raw}", f"keywords: {tightened}"),
889
+ encoding="utf-8",
890
+ )
891
+ rewritten += 1
892
+
893
+ return rewritten
894
+
895
+
896
+ def run(brand_dir: Path, fix: bool = False) -> list[Finding]:
897
+ """Every check, in order. `fix` is applied by the caller via apply_fixes."""
898
+ sources = load_sources(brand_dir)
899
+ findings: list[Finding] = []
900
+ findings.extend(check_contract(brand_dir))
901
+ findings.extend(check_terminology(brand_dir))
902
+ findings.extend(check_consistency(brand_dir, sources))
903
+ findings.extend(check_facts(brand_dir, sources))
904
+ findings.extend(check_channels(brand_dir, sources))
905
+ findings.extend(check_bot_safety(brand_dir, sources))
906
+ findings.extend(check_ai_tells(brand_dir, sources))
907
+ findings.extend(check_locales(brand_dir, sources))
908
+ return findings
909
+
910
+
911
+ def report(findings: list[Finding], brief: bool, as_json: bool) -> None:
912
+ if as_json:
913
+ print(json.dumps([f._asdict() for f in findings], indent=2))
914
+ return
915
+ errors = [f for f in findings if f.severity == SEVERITY_ERROR]
916
+ warns = [f for f in findings if f.severity == SEVERITY_WARN]
917
+ if brief:
918
+ state = "clean" if not findings else f"{len(errors)}E {len(warns)}W"
919
+ print(f"brand: {state}")
920
+ return
921
+ for finding in findings:
922
+ where = f"{finding.path}:{finding.line}" if finding.path else "pack"
923
+ tag = "ERROR" if finding.severity == SEVERITY_ERROR else "warn "
924
+ print(f"{tag} {finding.code} {where}: {finding.message}")
925
+ if not findings:
926
+ print("brand pack is clean")
927
+ else:
928
+ print(f"\n{len(errors)} error(s), {len(warns)} warning(s)")
929
+
930
+
931
+ def main(argv: list[str] | None = None) -> int:
932
+ parser = argparse.ArgumentParser(description=__doc__)
933
+ parser.add_argument("path", nargs="?", default="docs/brand")
934
+ parser.add_argument("--fix", action="store_true")
935
+ parser.add_argument("--brief", action="store_true")
936
+ parser.add_argument("--json", action="store_true")
937
+ args = parser.parse_args(argv)
938
+
939
+ brand_dir = Path(args.path)
940
+ if not brand_dir.is_dir():
941
+ print(f"no brand pack at {brand_dir} -- run /brand-init")
942
+ return 2
943
+
944
+ findings = run(brand_dir, fix=args.fix)
945
+ if args.fix:
946
+ rewritten = apply_fixes(brand_dir, findings)
947
+ print(f"--fix rewrote {rewritten} file(s)")
948
+ findings = run(brand_dir)
949
+ report(findings, args.brief, args.json)
950
+
951
+ if any(f.severity == SEVERITY_ERROR for f in findings):
952
+ return 2
953
+ return 1 if findings else 0
954
+
955
+
956
+ if __name__ == "__main__":
957
+ sys.exit(main())