macca-method 2.0.0 → 2.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/.agents/macca-lock.json +1 -1
  2. package/.agents/skills/_shared/references/brainstorm-session.md +5 -5
  3. package/.agents/skills/_shared/references/fix-mode.md +18 -3
  4. package/.agents/skills/_shared/references/human-loop.md +1 -1
  5. package/.agents/skills/_shared/references/invocation-policy.md +20 -20
  6. package/.agents/skills/_shared/references/language-config.md +7 -5
  7. package/.agents/skills/_shared/references/output-ownership.md +11 -11
  8. package/.agents/skills/_shared/references/scope-rules.md +1 -1
  9. package/.agents/skills/_shared/references/skill-catalog.md +20 -20
  10. package/.agents/skills/_shared/scripts/validate-skills.py +37 -15
  11. package/.agents/skills/add-feature/SKILL.md +11 -3
  12. package/.agents/skills/antislop-copywriting/SKILL.md +372 -0
  13. package/.agents/skills/brainstorm-api/SKILL.md +33 -19
  14. package/.agents/skills/brainstorm-api/assets/api.template.md +35 -15
  15. package/.agents/skills/brainstorm-architecture/SKILL.md +40 -18
  16. package/.agents/skills/brainstorm-architecture/assets/architecture.template.md +44 -25
  17. package/.agents/skills/brainstorm-prd/SKILL.md +52 -20
  18. package/.agents/skills/brainstorm-prd/assets/PRD.template.md +47 -23
  19. package/.agents/skills/brainstorm-rules/SKILL.md +39 -22
  20. package/.agents/skills/brainstorm-rules/assets/rules.template.md +32 -18
  21. package/.agents/skills/brainstorm-schema/SKILL.md +23 -11
  22. package/.agents/skills/brainstorm-schema/assets/schema.template.md +25 -10
  23. package/.agents/skills/brainstorm-styleguide/SKILL.md +40 -22
  24. package/.agents/skills/brainstorm-styleguide/assets/StyleGuide.template.md +78 -60
  25. package/.agents/skills/brainstorm-task/SKILL.md +29 -14
  26. package/.agents/skills/brainstorm-task/assets/Task.template.md +29 -18
  27. package/.agents/skills/bug-fix/SKILL.md +39 -8
  28. package/.agents/skills/code-review/SKILL.md +9 -7
  29. package/.agents/skills/code-review/references/review-checklist.md +21 -10
  30. package/.agents/skills/developer/SKILL.md +10 -0
  31. package/.agents/skills/developer/references/execute-task.md +13 -7
  32. package/.agents/skills/developer/references/onboarding.md +1 -1
  33. package/.agents/skills/help/SKILL.md +39 -23
  34. package/.agents/skills/meet/SKILL.md +11 -4
  35. package/.agents/skills/quick-dev/SKILL.md +29 -22
  36. package/.agents/skills/release-readiness/SKILL.md +19 -13
  37. package/.agents/skills/skill-creator/LICENSE.txt +202 -0
  38. package/.agents/skills/skill-creator/SKILL.md +485 -0
  39. package/.agents/skills/skill-creator/agents/analyzer.md +274 -0
  40. package/.agents/skills/skill-creator/agents/comparator.md +202 -0
  41. package/.agents/skills/skill-creator/agents/grader.md +223 -0
  42. package/.agents/skills/skill-creator/assets/eval_review.html +146 -0
  43. package/.agents/skills/skill-creator/eval-viewer/generate_review.py +471 -0
  44. package/.agents/skills/skill-creator/eval-viewer/viewer.html +1325 -0
  45. package/.agents/skills/skill-creator/references/schemas.md +441 -0
  46. package/.agents/skills/skill-creator/scripts/__init__.py +0 -0
  47. package/.agents/skills/skill-creator/scripts/aggregate_benchmark.py +401 -0
  48. package/.agents/skills/skill-creator/scripts/generate_report.py +326 -0
  49. package/.agents/skills/skill-creator/scripts/improve_description.py +247 -0
  50. package/.agents/skills/skill-creator/scripts/package_skill.py +136 -0
  51. package/.agents/skills/skill-creator/scripts/quick_validate.py +103 -0
  52. package/.agents/skills/skill-creator/scripts/run_eval.py +310 -0
  53. package/.agents/skills/skill-creator/scripts/run_loop.py +328 -0
  54. package/.agents/skills/skill-creator/scripts/utils.py +47 -0
  55. package/.agents/skills/spec-audit/SKILL.md +33 -4
  56. package/.agents/skills/spec-compliance/SKILL.md +36 -21
  57. package/.agents/skills/spec-init/SKILL.md +31 -17
  58. package/README.md +165 -129
  59. package/bin/macca-method.js +1378 -1077
  60. package/package.json +40 -40
  61. package/scripts/run-skill-validator.js +27 -9
  62. package/scripts/test-install.js +611 -337
  63. package/scripts/test-upgrade-legacy.js +131 -76
  64. package/scripts/validate-skill-behavior.js +175 -64
@@ -0,0 +1,328 @@
1
+ #!/usr/bin/env python3
2
+ """Run the eval + improve loop until all pass or max iterations reached.
3
+
4
+ Combines run_eval.py and improve_description.py in a loop, tracking history
5
+ and returning the best description found. Supports train/test split to prevent
6
+ overfitting.
7
+ """
8
+
9
+ import argparse
10
+ import json
11
+ import random
12
+ import sys
13
+ import tempfile
14
+ import time
15
+ import webbrowser
16
+ from pathlib import Path
17
+
18
+ from scripts.generate_report import generate_html
19
+ from scripts.improve_description import improve_description
20
+ from scripts.run_eval import find_project_root, run_eval
21
+ from scripts.utils import parse_skill_md
22
+
23
+
24
+ def split_eval_set(eval_set: list[dict], holdout: float, seed: int = 42) -> tuple[list[dict], list[dict]]:
25
+ """Split eval set into train and test sets, stratified by should_trigger."""
26
+ random.seed(seed)
27
+
28
+ # Separate by should_trigger
29
+ trigger = [e for e in eval_set if e["should_trigger"]]
30
+ no_trigger = [e for e in eval_set if not e["should_trigger"]]
31
+
32
+ # Shuffle each group
33
+ random.shuffle(trigger)
34
+ random.shuffle(no_trigger)
35
+
36
+ # Calculate split points
37
+ n_trigger_test = max(1, int(len(trigger) * holdout))
38
+ n_no_trigger_test = max(1, int(len(no_trigger) * holdout))
39
+
40
+ # Split
41
+ test_set = trigger[:n_trigger_test] + no_trigger[:n_no_trigger_test]
42
+ train_set = trigger[n_trigger_test:] + no_trigger[n_no_trigger_test:]
43
+
44
+ return train_set, test_set
45
+
46
+
47
+ def run_loop(
48
+ eval_set: list[dict],
49
+ skill_path: Path,
50
+ description_override: str | None,
51
+ num_workers: int,
52
+ timeout: int,
53
+ max_iterations: int,
54
+ runs_per_query: int,
55
+ trigger_threshold: float,
56
+ holdout: float,
57
+ model: str,
58
+ verbose: bool,
59
+ live_report_path: Path | None = None,
60
+ log_dir: Path | None = None,
61
+ ) -> dict:
62
+ """Run the eval + improvement loop."""
63
+ project_root = find_project_root()
64
+ name, original_description, content = parse_skill_md(skill_path)
65
+ current_description = description_override or original_description
66
+
67
+ # Split into train/test if holdout > 0
68
+ if holdout > 0:
69
+ train_set, test_set = split_eval_set(eval_set, holdout)
70
+ if verbose:
71
+ print(f"Split: {len(train_set)} train, {len(test_set)} test (holdout={holdout})", file=sys.stderr)
72
+ else:
73
+ train_set = eval_set
74
+ test_set = []
75
+
76
+ history = []
77
+ exit_reason = "unknown"
78
+
79
+ for iteration in range(1, max_iterations + 1):
80
+ if verbose:
81
+ print(f"\n{'='*60}", file=sys.stderr)
82
+ print(f"Iteration {iteration}/{max_iterations}", file=sys.stderr)
83
+ print(f"Description: {current_description}", file=sys.stderr)
84
+ print(f"{'='*60}", file=sys.stderr)
85
+
86
+ # Evaluate train + test together in one batch for parallelism
87
+ all_queries = train_set + test_set
88
+ t0 = time.time()
89
+ all_results = run_eval(
90
+ eval_set=all_queries,
91
+ skill_name=name,
92
+ description=current_description,
93
+ num_workers=num_workers,
94
+ timeout=timeout,
95
+ project_root=project_root,
96
+ runs_per_query=runs_per_query,
97
+ trigger_threshold=trigger_threshold,
98
+ model=model,
99
+ )
100
+ eval_elapsed = time.time() - t0
101
+
102
+ # Split results back into train/test by matching queries
103
+ train_queries_set = {q["query"] for q in train_set}
104
+ train_result_list = [r for r in all_results["results"] if r["query"] in train_queries_set]
105
+ test_result_list = [r for r in all_results["results"] if r["query"] not in train_queries_set]
106
+
107
+ train_passed = sum(1 for r in train_result_list if r["pass"])
108
+ train_total = len(train_result_list)
109
+ train_summary = {"passed": train_passed, "failed": train_total - train_passed, "total": train_total}
110
+ train_results = {"results": train_result_list, "summary": train_summary}
111
+
112
+ if test_set:
113
+ test_passed = sum(1 for r in test_result_list if r["pass"])
114
+ test_total = len(test_result_list)
115
+ test_summary = {"passed": test_passed, "failed": test_total - test_passed, "total": test_total}
116
+ test_results = {"results": test_result_list, "summary": test_summary}
117
+ else:
118
+ test_results = None
119
+ test_summary = None
120
+
121
+ history.append({
122
+ "iteration": iteration,
123
+ "description": current_description,
124
+ "train_passed": train_summary["passed"],
125
+ "train_failed": train_summary["failed"],
126
+ "train_total": train_summary["total"],
127
+ "train_results": train_results["results"],
128
+ "test_passed": test_summary["passed"] if test_summary else None,
129
+ "test_failed": test_summary["failed"] if test_summary else None,
130
+ "test_total": test_summary["total"] if test_summary else None,
131
+ "test_results": test_results["results"] if test_results else None,
132
+ # For backward compat with report generator
133
+ "passed": train_summary["passed"],
134
+ "failed": train_summary["failed"],
135
+ "total": train_summary["total"],
136
+ "results": train_results["results"],
137
+ })
138
+
139
+ # Write live report if path provided
140
+ if live_report_path:
141
+ partial_output = {
142
+ "original_description": original_description,
143
+ "best_description": current_description,
144
+ "best_score": "in progress",
145
+ "iterations_run": len(history),
146
+ "holdout": holdout,
147
+ "train_size": len(train_set),
148
+ "test_size": len(test_set),
149
+ "history": history,
150
+ }
151
+ live_report_path.write_text(generate_html(partial_output, auto_refresh=True, skill_name=name))
152
+
153
+ if verbose:
154
+ def print_eval_stats(label, results, elapsed):
155
+ pos = [r for r in results if r["should_trigger"]]
156
+ neg = [r for r in results if not r["should_trigger"]]
157
+ tp = sum(r["triggers"] for r in pos)
158
+ pos_runs = sum(r["runs"] for r in pos)
159
+ fn = pos_runs - tp
160
+ fp = sum(r["triggers"] for r in neg)
161
+ neg_runs = sum(r["runs"] for r in neg)
162
+ tn = neg_runs - fp
163
+ total = tp + tn + fp + fn
164
+ precision = tp / (tp + fp) if (tp + fp) > 0 else 1.0
165
+ recall = tp / (tp + fn) if (tp + fn) > 0 else 1.0
166
+ accuracy = (tp + tn) / total if total > 0 else 0.0
167
+ print(f"{label}: {tp+tn}/{total} correct, precision={precision:.0%} recall={recall:.0%} accuracy={accuracy:.0%} ({elapsed:.1f}s)", file=sys.stderr)
168
+ for r in results:
169
+ status = "PASS" if r["pass"] else "FAIL"
170
+ rate_str = f"{r['triggers']}/{r['runs']}"
171
+ print(f" [{status}] rate={rate_str} expected={r['should_trigger']}: {r['query'][:60]}", file=sys.stderr)
172
+
173
+ print_eval_stats("Train", train_results["results"], eval_elapsed)
174
+ if test_summary:
175
+ print_eval_stats("Test ", test_results["results"], 0)
176
+
177
+ if train_summary["failed"] == 0:
178
+ exit_reason = f"all_passed (iteration {iteration})"
179
+ if verbose:
180
+ print(f"\nAll train queries passed on iteration {iteration}!", file=sys.stderr)
181
+ break
182
+
183
+ if iteration == max_iterations:
184
+ exit_reason = f"max_iterations ({max_iterations})"
185
+ if verbose:
186
+ print(f"\nMax iterations reached ({max_iterations}).", file=sys.stderr)
187
+ break
188
+
189
+ # Improve the description based on train results
190
+ if verbose:
191
+ print(f"\nImproving description...", file=sys.stderr)
192
+
193
+ t0 = time.time()
194
+ # Strip test scores from history so improvement model can't see them
195
+ blinded_history = [
196
+ {k: v for k, v in h.items() if not k.startswith("test_")}
197
+ for h in history
198
+ ]
199
+ new_description = improve_description(
200
+ skill_name=name,
201
+ skill_content=content,
202
+ current_description=current_description,
203
+ eval_results=train_results,
204
+ history=blinded_history,
205
+ model=model,
206
+ log_dir=log_dir,
207
+ iteration=iteration,
208
+ )
209
+ improve_elapsed = time.time() - t0
210
+
211
+ if verbose:
212
+ print(f"Proposed ({improve_elapsed:.1f}s): {new_description}", file=sys.stderr)
213
+
214
+ current_description = new_description
215
+
216
+ # Find the best iteration by TEST score (or train if no test set)
217
+ if test_set:
218
+ best = max(history, key=lambda h: h["test_passed"] or 0)
219
+ best_score = f"{best['test_passed']}/{best['test_total']}"
220
+ else:
221
+ best = max(history, key=lambda h: h["train_passed"])
222
+ best_score = f"{best['train_passed']}/{best['train_total']}"
223
+
224
+ if verbose:
225
+ print(f"\nExit reason: {exit_reason}", file=sys.stderr)
226
+ print(f"Best score: {best_score} (iteration {best['iteration']})", file=sys.stderr)
227
+
228
+ return {
229
+ "exit_reason": exit_reason,
230
+ "original_description": original_description,
231
+ "best_description": best["description"],
232
+ "best_score": best_score,
233
+ "best_train_score": f"{best['train_passed']}/{best['train_total']}",
234
+ "best_test_score": f"{best['test_passed']}/{best['test_total']}" if test_set else None,
235
+ "final_description": current_description,
236
+ "iterations_run": len(history),
237
+ "holdout": holdout,
238
+ "train_size": len(train_set),
239
+ "test_size": len(test_set),
240
+ "history": history,
241
+ }
242
+
243
+
244
+ def main():
245
+ parser = argparse.ArgumentParser(description="Run eval + improve loop")
246
+ parser.add_argument("--eval-set", required=True, help="Path to eval set JSON file")
247
+ parser.add_argument("--skill-path", required=True, help="Path to skill directory")
248
+ parser.add_argument("--description", default=None, help="Override starting description")
249
+ parser.add_argument("--num-workers", type=int, default=10, help="Number of parallel workers")
250
+ parser.add_argument("--timeout", type=int, default=30, help="Timeout per query in seconds")
251
+ parser.add_argument("--max-iterations", type=int, default=5, help="Max improvement iterations")
252
+ parser.add_argument("--runs-per-query", type=int, default=3, help="Number of runs per query")
253
+ parser.add_argument("--trigger-threshold", type=float, default=0.5, help="Trigger rate threshold")
254
+ parser.add_argument("--holdout", type=float, default=0.4, help="Fraction of eval set to hold out for testing (0 to disable)")
255
+ parser.add_argument("--model", required=True, help="Model for improvement")
256
+ parser.add_argument("--verbose", action="store_true", help="Print progress to stderr")
257
+ parser.add_argument("--report", default="auto", help="Generate HTML report at this path (default: 'auto' for temp file, 'none' to disable)")
258
+ parser.add_argument("--results-dir", default=None, help="Save all outputs (results.json, report.html, log.txt) to a timestamped subdirectory here")
259
+ args = parser.parse_args()
260
+
261
+ eval_set = json.loads(Path(args.eval_set).read_text())
262
+ skill_path = Path(args.skill_path)
263
+
264
+ if not (skill_path / "SKILL.md").exists():
265
+ print(f"Error: No SKILL.md found at {skill_path}", file=sys.stderr)
266
+ sys.exit(1)
267
+
268
+ name, _, _ = parse_skill_md(skill_path)
269
+
270
+ # Set up live report path
271
+ if args.report != "none":
272
+ if args.report == "auto":
273
+ timestamp = time.strftime("%Y%m%d_%H%M%S")
274
+ live_report_path = Path(tempfile.gettempdir()) / f"skill_description_report_{skill_path.name}_{timestamp}.html"
275
+ else:
276
+ live_report_path = Path(args.report)
277
+ # Open the report immediately so the user can watch
278
+ live_report_path.write_text("<html><body><h1>Starting optimization loop...</h1><meta http-equiv='refresh' content='5'></body></html>")
279
+ webbrowser.open(str(live_report_path))
280
+ else:
281
+ live_report_path = None
282
+
283
+ # Determine output directory (create before run_loop so logs can be written)
284
+ if args.results_dir:
285
+ timestamp = time.strftime("%Y-%m-%d_%H%M%S")
286
+ results_dir = Path(args.results_dir) / timestamp
287
+ results_dir.mkdir(parents=True, exist_ok=True)
288
+ else:
289
+ results_dir = None
290
+
291
+ log_dir = results_dir / "logs" if results_dir else None
292
+
293
+ output = run_loop(
294
+ eval_set=eval_set,
295
+ skill_path=skill_path,
296
+ description_override=args.description,
297
+ num_workers=args.num_workers,
298
+ timeout=args.timeout,
299
+ max_iterations=args.max_iterations,
300
+ runs_per_query=args.runs_per_query,
301
+ trigger_threshold=args.trigger_threshold,
302
+ holdout=args.holdout,
303
+ model=args.model,
304
+ verbose=args.verbose,
305
+ live_report_path=live_report_path,
306
+ log_dir=log_dir,
307
+ )
308
+
309
+ # Save JSON output
310
+ json_output = json.dumps(output, indent=2)
311
+ print(json_output)
312
+ if results_dir:
313
+ (results_dir / "results.json").write_text(json_output)
314
+
315
+ # Write final HTML report (without auto-refresh)
316
+ if live_report_path:
317
+ live_report_path.write_text(generate_html(output, auto_refresh=False, skill_name=name))
318
+ print(f"\nReport: {live_report_path}", file=sys.stderr)
319
+
320
+ if results_dir and live_report_path:
321
+ (results_dir / "report.html").write_text(generate_html(output, auto_refresh=False, skill_name=name))
322
+
323
+ if results_dir:
324
+ print(f"Results saved to: {results_dir}", file=sys.stderr)
325
+
326
+
327
+ if __name__ == "__main__":
328
+ main()
@@ -0,0 +1,47 @@
1
+ """Shared utilities for skill-creator scripts."""
2
+
3
+ from pathlib import Path
4
+
5
+
6
+
7
+ def parse_skill_md(skill_path: Path) -> tuple[str, str, str]:
8
+ """Parse a SKILL.md file, returning (name, description, full_content)."""
9
+ content = (skill_path / "SKILL.md").read_text()
10
+ lines = content.split("\n")
11
+
12
+ if lines[0].strip() != "---":
13
+ raise ValueError("SKILL.md missing frontmatter (no opening ---)")
14
+
15
+ end_idx = None
16
+ for i, line in enumerate(lines[1:], start=1):
17
+ if line.strip() == "---":
18
+ end_idx = i
19
+ break
20
+
21
+ if end_idx is None:
22
+ raise ValueError("SKILL.md missing frontmatter (no closing ---)")
23
+
24
+ name = ""
25
+ description = ""
26
+ frontmatter_lines = lines[1:end_idx]
27
+ i = 0
28
+ while i < len(frontmatter_lines):
29
+ line = frontmatter_lines[i]
30
+ if line.startswith("name:"):
31
+ name = line[len("name:"):].strip().strip('"').strip("'")
32
+ elif line.startswith("description:"):
33
+ value = line[len("description:"):].strip()
34
+ # Handle YAML multiline indicators (>, |, >-, |-)
35
+ if value in (">", "|", ">-", "|-"):
36
+ continuation_lines: list[str] = []
37
+ i += 1
38
+ while i < len(frontmatter_lines) and (frontmatter_lines[i].startswith(" ") or frontmatter_lines[i].startswith("\t")):
39
+ continuation_lines.append(frontmatter_lines[i].strip())
40
+ i += 1
41
+ description = " ".join(continuation_lines)
42
+ continue
43
+ else:
44
+ description = value.strip('"').strip("'")
45
+ i += 1
46
+
47
+ return name, description, content
@@ -11,14 +11,16 @@ metadata:
11
11
 
12
12
  ## Shared Runtime Setup
13
13
 
14
+ Paths written as `../...` below are relative to this SKILL.md's own folder, not the project's working directory - resolve them as a sibling of the folder that contains this file.
15
+
14
16
  At startup:
15
17
 
16
18
  1. Read `../_shared/references/language-config.md`.
17
19
  2. Read `../_shared/references/fix-mode.md`.
18
20
  3. Read `../_shared/references/human-loop.md`.
19
- 3. If this message answers this skill's active correction gate, resume directly under the Approval Resume Protocol. Do not rerun startup or the audit.
20
- 4. Otherwise, read `codeReviewPreferences.fixMode` from `.agents/developer-config.json`. If it is missing, treat it as `"report-first"`. Announce: `[Fix mode: report-first]` or `[Fix mode: fix-then-report]`.
21
- 5. Use `languagePreferences.communication.normalized` for audit reports.
21
+ 4. If this message answers this skill's active correction gate, resume directly under the Approval Resume Protocol. Do not rerun startup or the audit.
22
+ 5. Otherwise, read `codeReviewPreferences.fixMode` from `.agents/developer-config.json`. If it is missing, treat it as `"report-first"`. Announce: `[Fix mode: report-first]` or `[Fix mode: fix-then-report]`.
23
+ 6. Use `languagePreferences.communication.normalized` for audit reports.
22
24
 
23
25
  ---
24
26
 
@@ -33,6 +35,7 @@ Run as `@Fachri` (Tech Lead). Use the shared persona profile in `../_shared/refe
33
35
  You are **@Fachri - Tech Lead** and **Spec Reviewer**. Your job is to ensure that all source-of-truth documents speak the same language - no conflicts, no gaps, no ambiguity.
34
36
 
35
37
  Two audit modes:
38
+
36
39
  - **Project Mode** - audit `project-context/` documents
37
40
  - **Framework Mode** - audit the MACCA framework itself (README, skill docs, workflows)
38
41
 
@@ -57,18 +60,23 @@ To change it: update `codeReviewPreferences.fixMode` in `.agents/developer-confi
57
60
  Determine the mode from user context:
58
61
 
59
62
  ### Project Mode
63
+
60
64
  Audit `project-context/` documents. Use it when:
65
+
61
66
  - The user is checking spec alignment before coding
62
67
  - They just finished spec documents and want a pre-check
63
68
  - Spec audit is part of the normal workflow
64
69
 
65
70
  ### Framework Mode
71
+
66
72
  Audit MACCA itself (README, skill docs, workflows). Use it when:
73
+
67
74
  - The user wants to refine MACCA
68
75
  - They suspect instruction drift between skills
69
76
  - They want to verify alignment across README, `help`, and the workflows
70
77
 
71
78
  Before continuing, show the target:
79
+
72
80
  ```
73
81
  Mode: [Project / Framework]
74
82
  Auditing: [short list of main documents being checked]
@@ -83,6 +91,7 @@ Default valid prefixes (Project Mode): `FEAT-*`, `BR-*`, `NFR-*`, `AC-*`, `US-*`
83
91
  ### Project Mode
84
92
 
85
93
  Read everything available in `project-context/`:
94
+
86
95
  - `PRD.md` - features, business rules, acceptance criteria, non-goals
87
96
  - `architecture.md` - tech stack, folder structure, patterns
88
97
  - `schema.md` - tables, columns, types, relationships
@@ -97,7 +106,7 @@ Read everything that exists. Note ID patterns if they are used.
97
106
 
98
107
  ### Framework Mode
99
108
 
100
- Resolve the active skill installation root. First read `_shared/references/skill-catalog.md`, `invocation-policy.md`, `output-ownership.md`, `scope-rules.md`, and the MACCA README only when this is the MACCA source repository.
109
+ Resolve the active skill installation root. First read `../_shared/references/skill-catalog.md`, `../_shared/references/invocation-policy.md`, `../_shared/references/output-ownership.md`, `../_shared/references/scope-rules.md`, and the MACCA README only when this is the MACCA source repository.
101
110
 
102
111
  Compare compact contracts first. Read full `SKILL.md`, local references/assets, or installer code only for skills/relationships flagged by that comparison or explicitly named by the user. Do not load the entire collection by default. Do not treat an application's README as MACCA documentation.
103
112
 
@@ -108,48 +117,57 @@ Compare compact contracts first. Read full `SKILL.md`, local references/assets,
108
117
  ### Project Mode
109
118
 
110
119
  **SA-01: PRD ↔ architecture**
120
+
111
121
  - Does the architecture support the PRD NFRs (performance, security, accessibility)?
112
122
  - Do the PRD constraints fit the chosen tech stack?
113
123
  - Do PRD success metrics map to architecture observability signals where technical instrumentation is required?
114
124
  - Does product rollout align with deployment, rollback, and recovery constraints?
115
125
 
116
126
  **SA-02: PRD ↔ schema**
127
+
117
128
  - Does every persisted PRD entity have a datastore-native representation (table, collection, aggregate, node, stream, or equivalent)?
118
129
  - Do validation, relationship, consistency, and retention rules reflect PRD business rules?
119
130
 
120
131
  **SA-03: PRD ↔ api**
132
+
121
133
  - Does every integration-facing PRD feature have supporting operations (endpoint, query/mutation, procedure, event, or equivalent)?
122
134
  - Does `api.md` contain operations for PRD non-goals?
123
135
 
124
136
  **SA-04: PRD ↔ Task.md**
137
+
125
138
  - Is every PRD feature mapped to >=1 task?
126
139
  - Does Task.md include tasks for features not in the PRD (scope creep)? If the feature is recorded in `## Approved Scope Delta`, mark it as `pending formal spec sync`, not a direct conflict.
127
140
  - Do PRD IDs (`FEAT-*`, `BR-*`) appear in Task.md traceability?
128
141
  - Do rollout, analytics, degraded behavior, and applicable NFR work have tasks or explicit N/A decisions?
129
142
 
130
143
  **SA-05: schema ↔ api**
144
+
131
145
  - Does every persisted input/output field in `api.md` map to the data contract where appropriate?
132
146
  - Do response types match schema types?
133
147
  - If schema/api traceability is used, does it reference real PRD IDs?
134
148
  - Do API retry/idempotency assumptions align with schema concurrency and consistency rules?
135
149
 
136
150
  **SA-06: architecture ↔ rules**
151
+
137
152
  - Are architectural patterns (e.g. repository pattern) required in `rules.md`?
138
153
  - Do any rules conflict with the chosen architecture?
139
154
  - Do logging, migration, feature-flag, generated-code, and secret rules exist only when their architecture/schema mechanisms apply?
140
155
 
141
156
  **SA-07: architecture ↔ schema**
157
+
142
158
  - Does schema notation fit the architecture's database choice?
143
159
  - Is schema style consistent with the architecture's ORM choice?
144
160
  - Do tenancy, scale, migration, backup, and recovery assumptions align?
145
161
 
146
162
  **SA-08: StyleGuide ↔ PRD**
163
+
147
164
  - Does the CSS framework in StyleGuide match any PRD mention?
148
165
  - Are all PRD pages/features covered by StyleGuide components?
149
166
  - Do accessibility targets and supported locales match the PRD?
150
167
  - Do operational UI states cover PRD failure/degraded behavior where UI is involved?
151
168
 
152
169
  **SA-09: Task.md ↔ all specs**
170
+
153
171
  - Do task references point to real spec sections?
154
172
  - Do task acceptance criteria match PRD acceptance criteria?
155
173
  - If task traceability IDs are used, do they reference real PRD/schema/api/rules IDs?
@@ -160,38 +178,47 @@ Compare compact contracts first. Read full `SKILL.md`, local references/assets,
160
178
  ### Framework Mode
161
179
 
162
180
  **SA-F01: README ↔ skill descriptions**
181
+
163
182
  - Are skill names, personas, and functions the same in README and `SKILL.md`?
164
183
  - Do README summaries differ from the actual skill descriptions?
165
184
 
166
185
  **SA-F02: README ↔ workflow order**
186
+
167
187
  - Does the README workflow match skill prerequisites?
168
188
  - Does the README suggest an order that conflicts with skill instructions?
169
189
 
170
190
  **SA-F03: help ↔ README**
191
+
171
192
  - Does `help` recommend the same next-step workflow as README?
172
193
  - Does `help` contain an alternative path that changes the core workflow order without reason?
173
194
 
174
195
  **SA-F04: Skill prerequisite consistency**
196
+
175
197
  - Are `brainstorm-*`, `developer`, `spec-init`, `spec-compliance`, and `code-review` aligned on prerequisites?
176
198
  - Does one skill allow a step that another skill marks invalid?
177
199
 
178
200
  **SA-F05: Output file naming consistency**
201
+
179
202
  - Are output names (`PRD.md`, `Task.md`, etc.) the same across all skills?
180
203
  - Are output locations (`project-context/`, `.agents/`, elsewhere) named consistently?
181
204
 
182
205
  **SA-F06: Cross-skill handoff**
206
+
183
207
  - Does the "next step" from skill A match the entry point of skill B?
184
208
  - Are there dead ends, loops, or mismatched handoffs?
185
209
 
186
210
  **SA-F07: Persona consistency**
211
+
187
212
  - Are personas, roles, and assigned skills consistent across README, `meet`, and skill frontmatter?
188
213
  - Does any skill name the wrong owner?
189
214
 
190
215
  **SA-F08: Enforcement & order consistency**
216
+
191
217
  - Are "spec-compliance before code-review," "update Task.md," and "confirm before bug-log" stated consistently everywhere?
192
218
  - Does any instruction weaken a mandatory gate elsewhere?
193
219
 
194
220
  **SA-F09: Terminology consistency**
221
+
195
222
  - Are terms such as `spec`, `project-context/`, `phase`, `task`, `Batch Generate`, and `Project Audit` used with the same meaning everywhere?
196
223
  - Is any concept defined differently in 2+ places?
197
224
 
@@ -259,11 +286,13 @@ Fix Manifest (only when corrections were requested):
259
286
  ```
260
287
 
261
288
  If there are no issues:
289
+
262
290
  ```
263
291
  ✅ All documents in this audit mode are consistent - no conflicts, inconsistencies, or ambiguities were found.
264
292
  ```
265
293
 
266
294
  **Apply fixes:**
295
+
267
296
  - Audit-only invocation: report only, regardless of `fixMode`; do not offer a mutation gate unless the user requests corrections.
268
297
  - Correction requested + `fix-then-report`: apply actionable corrections, validate all affected document pairs, then report.
269
298
  - Correction requested + `report-first`: show the summary and shared gate. On approval, resume directly under the Approval Resume Protocol.