@mmerterden/multi-agent-pipeline 20.6.0 → 20.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/CHANGELOG.md +18 -0
  2. package/docs/facts.json +1 -1
  3. package/manifest.json +71 -71
  4. package/package.json +1 -1
  5. package/pipeline/multi-agent-refs/features/design-conformance.md +62 -64
  6. package/pipeline/scripts/_notices.mjs +11 -0
  7. package/pipeline/scripts/gen-skills-index.mjs +1 -1
  8. package/pipeline/skills/.skill-manifest.json +39 -39
  9. package/pipeline/skills/shared/README.md +1 -1
  10. package/pipeline/skills/shared/external/agent-introspection-debugging/SKILL.md +1 -0
  11. package/pipeline/skills/shared/external/android-architecture/SKILL.md +2 -0
  12. package/pipeline/skills/shared/external/android-performance/SKILL.md +2 -0
  13. package/pipeline/skills/shared/external/android-security/SKILL.md +2 -0
  14. package/pipeline/skills/shared/external/backlog/BACKLOG.md +1 -1
  15. package/pipeline/skills/shared/external/backlog/SKILL.md +56 -33
  16. package/pipeline/skills/shared/external/ci-cd-pipelines/SKILL.md +1 -0
  17. package/pipeline/skills/shared/external/compose-components/SKILL.md +2 -0
  18. package/pipeline/skills/shared/external/compose-navigation/SKILL.md +3 -2
  19. package/pipeline/skills/shared/external/compose-testing/SKILL.md +2 -0
  20. package/pipeline/skills/shared/external/council/SKILL.md +1 -0
  21. package/pipeline/skills/shared/external/css-modern/SKILL.md +1 -0
  22. package/pipeline/skills/shared/external/database-patterns/SKILL.md +1 -0
  23. package/pipeline/skills/shared/external/evidence-github/SKILL.md +2 -0
  24. package/pipeline/skills/shared/external/evidence-registry/SKILL.md +2 -0
  25. package/pipeline/skills/shared/external/gradle-kotlin-dsl/SKILL.md +2 -0
  26. package/pipeline/skills/shared/external/html-semantic/SKILL.md +1 -0
  27. package/pipeline/skills/shared/external/humanizer/SKILL.md +1 -0
  28. package/pipeline/skills/shared/external/ios-coding-standard/SKILL.md +1 -0
  29. package/pipeline/skills/shared/external/ios-module-structure/SKILL.md +1 -0
  30. package/pipeline/skills/shared/external/ios-security/SKILL.md +2 -0
  31. package/pipeline/skills/shared/external/localization-reuse-map/SKILL.md +91 -283
  32. package/pipeline/skills/shared/external/localization-reuse-map/reference/format-and-output.md +119 -151
  33. package/pipeline/skills/shared/external/localization-reuse-map/reference/publish-and-snapshot.md +60 -90
  34. package/pipeline/skills/shared/external/localization-reuse-map/reference/sources-and-recipes.md +119 -156
  35. package/pipeline/skills/shared/external/localization-reuse-map/scripts/build-artifact.py +726 -787
  36. package/pipeline/skills/shared/external/localization-reuse-map/scripts/build-spreadsheet.py +253 -288
  37. package/pipeline/skills/shared/external/localization-reuse-map/scripts/fetch-annotations.py +243 -304
  38. package/pipeline/skills/shared/external/localization-reuse-map/scripts/fetch-legacy-labels.py +88 -104
  39. package/pipeline/skills/shared/external/localization-reuse-map/scripts/publish-confluence.py +181 -235
  40. package/pipeline/skills/shared/external/localization-reuse-map/scripts/render-key-shots.py +198 -263
  41. package/pipeline/skills/shared/external/localization-reuse-map/scripts/render-overlay.py +461 -466
  42. package/pipeline/skills/shared/external/localization-reuse-map/scripts/resolve-legacy-values.py +145 -151
  43. package/pipeline/skills/shared/external/localization-reuse-map/scripts/resolve-new-values.py +123 -141
  44. package/pipeline/skills/shared/external/localization-reuse-map/scripts/scan-screen-keys.py +146 -157
  45. package/pipeline/skills/shared/external/localization-reuse-map/scripts/snapshot-resources.sh +22 -19
  46. package/pipeline/skills/shared/external/localization-reuse-map/scripts/verify-map.py +156 -140
  47. package/pipeline/skills/shared/external/nextjs-app-router/SKILL.md +1 -0
  48. package/pipeline/skills/shared/external/play-store-review/SKILL.md +2 -0
  49. package/pipeline/skills/shared/external/python-patterns/SKILL.md +1 -0
  50. package/pipeline/skills/shared/external/react-best-practices/SKILL.md +1 -0
  51. package/pipeline/skills/shared/external/rest-api-design/SKILL.md +1 -0
  52. package/pipeline/skills/shared/external/retrofit-networking/SKILL.md +2 -0
  53. package/pipeline/skills/shared/external/room-database/SKILL.md +2 -0
  54. package/pipeline/skills/shared/external/search-first/SKILL.md +1 -0
  55. package/pipeline/skills/shared/external/signal-community/SKILL.md +2 -0
  56. package/pipeline/skills/shared/external/skill-creator/SKILL.md +80 -41
  57. package/pipeline/skills/shared/external/skill-creator/audit.md +63 -59
  58. package/pipeline/skills/shared/external/skill-creator/checklist.md +28 -20
  59. package/pipeline/skills/shared/external/skill-creator/examples.md +40 -40
  60. package/pipeline/skills/shared/external/skill-creator/label-check.md +48 -36
  61. package/pipeline/skills/shared/external/skill-creator/scripts/audit-panel.js +91 -100
  62. package/pipeline/skills/shared/external/skill-creator/template.md +51 -39
  63. package/pipeline/skills/shared/external/tailwind-css/SKILL.md +1 -0
  64. package/pipeline/skills/shared/external/testing-backend/SKILL.md +1 -0
  65. package/pipeline/skills/shared/external/typescript-patterns/SKILL.md +1 -0
  66. package/pipeline/skills/shared/external/vue-composition/SKILL.md +1 -0
  67. package/pipeline/skills/shared/external/web-accessibility/SKILL.md +1 -0
  68. package/pipeline/skills/shared/external/web-performance/SKILL.md +1 -0
  69. package/pipeline/skills/shared/external/web-testing/SKILL.md +1 -0
@@ -1,59 +1,63 @@
1
- # Multi-architect audit - adversarial quality gate for a skill
2
-
3
- The [checklist](checklist.md) is the self-review floor. This is the **high-confidence
4
- gate**: a panel of N independent reviewers, each a *distinct lens*, reads the skill
5
- against the rubric + ground truth, then a synthesizer dedups and rules. You **loop
6
- fix → re-audit until it passes** (no blockers/majors). Run it before shipping a new
7
- or substantially-changed skill, or whenever a skill must be trusted.
8
-
9
- Use the [Workflow tool](scripts/audit-panel.js) to fan the panel out - it returns
10
- `{remaining, clearsBar, verdict}`.
11
-
12
- ## The loop (run to convergence)
13
- 1. **Panel** - N agents, one lens each, read-only. Each returns specific findings
14
- (severity · file:location · problem · concrete fix). Empty if its lens is clean.
15
- 2. **Synthesize** - dedup overlaps; **drop false positives** (verify each against the
16
- real files/repo before keeping); classify blocker / major / minor / nit;
17
- `clearsBar = no blockers and no majors remain`.
18
- 3. **If not clean** - apply the do-now set (blockers + majors + cheap high-value
19
- minors), then run a **fresh** panel (see "Fresh each round") and repeat.
20
- 4. **Stop** when `clearsBar` is true. The later/nit list is optional polish.
21
-
22
- ## Rules that make it trustworthy
23
- - **Fresh panel each round.** After a fix, re-run the panel anew so agents read the
24
- *current* files - do not reuse a prior run's cached results (they reflect the old
25
- state and will report fixed issues as open / miss regressions).
26
- - **Verify before keeping.** Every finding must be confirmed against the actual file
27
- or repo. The synthesizer drops anything it can't reproduce. Prefer findings that
28
- cite a line and a reproduction.
29
- - **Ground-truth, not vibes.** Give the panel the rubric ([SKILL.md](SKILL.md) +
30
- [checklist.md](checklist.md)), an exemplar skill, and the real source the skill
31
- describes (e.g. the iOS repo). Agents read; they don't assume.
32
- - **Watch for regressions.** A fix can break something - the red-team lens and the
33
- fresh re-audit exist to catch it (e.g. a broadened check that now false-positives).
34
- - **clearsBar gates the merge.** Don't merge with an open blocker/major.
35
-
36
- ## Lenses - pick the set for the task
37
- One distinct lens per agent (+ a synthesizer). Two ready-made sets:
38
-
39
- **General skill audit** - discovery · leanness · shape/layer · scripts · accuracy-vs-source · domain-correctness · wiring · boundaries · red-team → synthesis.
40
-
41
- **Prompt-engineering audit** (when the concern is *prompt quality* - descriptions,
42
- instructions, triggering):
43
- 1. **description-as-discovery** - what + when + trigger words + example; would it activate on a real task?
44
- 2. **disambiguation** - distinct from siblings; a reader can tell when to pick this vs a neighbor.
45
- 3. **trigger coverage** - the real user phrasings (and synonyms) are covered.
46
- 4. **instruction clarity** - steps are unambiguous, ordered, single-interpretation; no vague directives.
47
- 5. **degrees of freedom / layer fit** - reference=descriptive (no numbered procedure), workflow=spine, tool=thin-index; freedom matches the task.
48
- 6. **progressive disclosure / leanness** - SKILL.md is an index; detail bundled one level deep; no duplication.
49
- 7. **token efficiency** - every line is load-bearing; nothing restates what the model already knows.
50
- 8. **consistency** - terminology, naming, paths, cross-references, examples all agree.
51
- 9. **anti-patterns / misfire** - would the model over-apply it, skip a gate, hallucinate a path, or follow it into a wrong output?
52
- 10. **red-team** - adversarially try to make it not trigger, trigger wrongly, or produce broken output; hunt overclaims (says X, the script doesn't).
53
-
54
- Scale the panel to risk: a quick check is ~3 lenses; a thorough gate is ~10.
55
-
56
- ## Scope of fixes per round
57
- Fix **blockers + majors** always; fold in **cheap, high-value minors** while you're
58
- in the file. Defer speculative nits and pre-existing/out-of-scope drift to the
59
- "later" list (and `log()`/note them - don't silently drop coverage).
1
+ # Multi-architect audit
2
+
3
+ The checklist is what one author catches alone. The audit adds a panel of independent, read-only reviewers, each looking through one lens, and a synthesizer that turns their output into a decision. New skills and substantial changes go through it before merge.
4
+
5
+ ## The loop
6
+
7
+ 1. **Panel.** Every lens reviews the skill and returns findings, each with a severity, a `file:location`, the problem and a concrete fix. A clean lens returns an empty list.
8
+ 2. **Synthesis.** One agent merges duplicates, opens the real files to discard false positives, classifies the rest as blocker, major, minor or nit, and sets `clearsBar` to true only when no blocker and no major is left.
9
+ 3. **Fix.** If the bar is not cleared, apply the do-now set and start a **new** panel.
10
+ 4. **Stop** once `clearsBar` is true. Remaining nits are optional polish.
11
+
12
+ ### What to fix in a round
13
+
14
+ - Always: every blocker and every major.
15
+ - Also, when cheap and clearly valuable: minors.
16
+ - Later, on a written list: speculative nits and drift that was already there or sits outside this change. Record them; never drop them silently.
17
+
18
+ ## Rules that keep the result trustworthy
19
+
20
+ - **Fresh panel every round.** Cached results describe the files as they were; a resumed run audits the old version.
21
+ - **Verify before keeping.** Each finding is checked against the file before it counts. Prefer findings that cite a line and a way to reproduce.
22
+ - **Give the panel ground truth:** the rubric (skill-creator SKILL.md plus checklist.md), an exemplar skill of the same layer, and the actual source the skill describes.
23
+ - **Watch the fixes.** A fix can break something else. The red-team lens and the fresh re-audit are there to catch that.
24
+ - **`clearsBar` gates the merge.**
25
+
26
+ ## Lens sets
27
+
28
+ Scale the panel to the risk: about three lenses for a quick check, about ten for a full gate.
29
+
30
+ ### General
31
+
32
+ discovery, leanness, shape and layer, scripts, accuracy against the source, domain correctness, wiring, boundaries, red team, then synthesis.
33
+
34
+ ### Prompt engineering (the script's default ten)
35
+
36
+ | Lens | Asks |
37
+ |---|---|
38
+ | discovery | Does the description alone get the skill loaded? |
39
+ | disambiguation | Could it be confused with a sibling? |
40
+ | triggers | Which real phrasings and synonyms are missing? |
41
+ | clarity | Is every instruction unambiguous and actionable? |
42
+ | layer | Does the degree of freedom fit the declared layer? |
43
+ | leanness | Is SKILL.md an index with detail one level down? |
44
+ | tokens | Which lines fail the cut test? |
45
+ | consistency | Do terms, names, paths, cross-references and examples agree? |
46
+ | misfire | Where would it over-apply, skip its gate, invent a path, or produce the wrong output? |
47
+ | redteam | Make it fail to trigger, trigger wrongly, or break its output; find prose that claims what the script does not do. |
48
+
49
+ ## Running it with the script
50
+
51
+ `scripts/audit-panel.js` is a Workflow-tool template (workflow name `skill-audit-panel`, phases Audit and Synthesis). It takes no arguments; edit its config block first:
52
+
53
+ | Constant | Set to |
54
+ |---|---|
55
+ | `TARGET` | absolute paths of the SKILL.md and each bundled file and script |
56
+ | `RUBRIC` | path to skill-creator's SKILL.md plus checklist.md |
57
+ | `EXEMPLAR` | a known-good skill of the same layer |
58
+ | `GROUND_TRUTH` | the real source the skill documents, with the key facts to check |
59
+ | `LENSES` | label and brief pairs; defaults to the ten above |
60
+
61
+ It runs one agent per lens in parallel (labelled `audit:<lens>`), tags each finding with its lens, and hands them all to a `synthesis` agent. The result is `{ remaining, clearsBar, verdict }`: the surviving findings with the lenses that raised them, the gate boolean, and a one-paragraph verdict.
62
+
63
+ After applying fixes, launch it again as a new run. Do not resume the old one; resumed agents return cached results.
@@ -1,32 +1,40 @@
1
1
  # Pre-ship checklist
2
2
 
3
- Run before considering a skill done.
3
+ The floor for every skill. New or substantially changed skills also go through the audit loop in audit.md.
4
4
 
5
5
  ## Discovery
6
- - [ ] `description` says **what + when + trigger words**, third person, and names the platform (iOS/SwiftUI).
7
- - [ ] Description disambiguates from sibling skills (a reader can tell when to pick this vs a neighbor).
8
- - [ ] Tested: in a fresh task that should trigger it, does the model actually reach for it? If not, sharpen the description.
6
+
7
+ - [ ] The description says what the skill does, when to use it, and lists trigger phrases, in third person.
8
+ - [ ] It names the platform where that matters.
9
+ - [ ] It separates this skill from its siblings.
10
+ - [ ] Tried on a new task it ought to catch, the skill really fires.
9
11
 
10
12
  ## Leanness
11
- - [ ] `SKILL.md` reads like an index, ~1 page (hard cap ~500 lines).
12
- - [ ] Detail lives in bundled files **one level deep** - no nested-deeper references.
13
- - [ ] Every line survives "would Claude err without this?" - no over-explaining, no restating what the model knows.
14
- - [ ] Deterministic code is a **script**, not prose asking the model to regenerate it.
13
+
14
+ - [ ] SKILL.md is roughly one page and works as an index, far below the 500-line cap.
15
+ - [ ] Detail lives in bundled files exactly one level below SKILL.md.
16
+ - [ ] Every line passes the cut test.
17
+ - [ ] Anything deterministic ships as a script instead of prose the model must re-derive.
15
18
 
16
19
  ## Shape
17
- - [ ] Layer is explicit (reference or workflow) and matches the content.
18
- - [ ] **Reference** skill has **no numbered procedures** - heuristics + pointers only.
19
- - [ ] **Workflow** skill's numbered steps are **only** the deterministic/irreversible spine, and it has a **verification gate**.
20
- - [ ] Conventions **point to canonical files** rather than pasting code that will rot.
21
20
 
22
- ## Naming & wiring
23
- - [ ] Name has **no platform prefix**; reference = aspect noun, workflow = verb-phrase; not vague/reserved.
24
- - [ ] Registered in the plugin manifest (`<plugin>/.claude-plugin/plugin.json` `skills` array) and the marketplace version bumped (`.github/plugin/marketplace.json`).
25
- - [ ] Listed in the `index` skill's route table.
21
+ - [ ] The layer is declared and the content matches it.
22
+ - [ ] Numbered procedures never appear in a reference skill.
23
+ - [ ] A workflow skill's steps are only the deterministic spine, and it ends in a verification gate.
24
+ - [ ] Conventions point at canonical files instead of pasting code that will go stale.
25
+
26
+ ## Naming and wiring
27
+
28
+ - [ ] No platform prefix; noun for reference, verb phrase for workflow; not vague, not reserved.
29
+ - [ ] Registered in the `skills` array of `<plugin>/.claude-plugin/plugin.json`.
30
+ - [ ] Plugin version bumped in `.github/plugin/marketplace.json`.
31
+ - [ ] A row in the plugin's `index` skill route table.
26
32
 
27
33
  ## House rules
28
- - [ ] No AI/Claude/Anthropic attribution anywhere in the skill or its examples.
29
- - [ ] Escape hatches name the skill to hand off to when this one doesn't apply.
30
34
 
31
- ## High-confidence gate (new / substantially-changed skills)
32
- - [ ] Ran the multi-architect audit ([audit.md](audit.md)) and looped fix → re-audit until `clearsBar` (no blockers/majors).
35
+ - [ ] Neither the skill nor its examples carry AI, Claude or Anthropic attribution.
36
+ - [ ] Each escape hatch says which skill takes over.
37
+
38
+ ## Confidence gate
39
+
40
+ - [ ] New or substantially changed: the multi-architect audit ran, fixes were applied, and a fresh panel returned `clearsBar: true`.
@@ -1,65 +1,65 @@
1
- # Examples - descriptions & shape
1
+ # Examples
2
2
 
3
- ## Descriptions: the discovery lever
3
+ ## Descriptions
4
+
5
+ ### Weak
4
6
 
5
- **Bad (vague - rarely triggers):**
6
7
  ```yaml
7
- description: Helps with navigation
8
+ description: Helps with navigation stuff.
8
9
  ```
9
- Why it fails: no "when", no trigger words, no platform - the router can't tell what task this matches.
10
10
 
11
- **Good (what + when + triggers + platform):**
11
+ No situation, no trigger words, no platform. The router has nothing to match a request against, so the skill stays unloaded.
12
+
13
+ ### Strong, reference layer
14
+
12
15
  ```yaml
13
- description: "In-app navigation for a SwiftUI iOS app - the coordinator-per-domain pattern, the tab shell, cross-domain navigation, deep links, and modals. Reach for this when adding a screen, a tab, a deep link, or a modal, or deciding how one screen reaches another."
16
+ description: "Navigation for iOS / SwiftUI: NavigationStack paths, tab structure, deep links, sheets and full-screen modals, and where routing state lives. Use when adding a screen, a tab or a deep link, presenting a modal, or debugging a push that does nothing. Triggers: 'add a new screen', 'open this from a deep link', 'present as a sheet'."
14
17
  ```
15
- Why it works: states the domain, lists concrete triggers ("adding a screen, a tab, a deep link"), names the platform, and implies the boundary.
16
18
 
17
- **Workflow example (action + when to run):**
19
+ Coverage, concrete situations, the platform, and quoted phrases a developer would type. The domain words keep it apart from sibling skills on state or overlays.
20
+
21
+ ### Strong, workflow layer
22
+
18
23
  ```yaml
19
- description: "Branch, commit, push, and open a PR for this repo. Owns the git/PR spine and house guardrails - type-first branch names, no AI attribution, no force-push, correct base branch. Reach for this whenever a change is ready to ship."
24
+ description: "Creates a branch, commits, pushes and opens a pull request with the house guardrails: type-first branch names, conventional commit messages with no AI attribution, never a force push, and the correct base branch. Use when work is ready to ship. Triggers: 'open a PR', 'commit and push this', 'create a branch for this ticket'."
20
25
  ```
21
26
 
22
- ### Recipe
23
- `<what it covers/does> + <concrete when/trigger phrases> + <platform> + <implied boundary vs siblings>`
27
+ States the action, the guardrails that make it safe, when it applies, and how users ask for it.
24
28
 
25
- ---
29
+ ## Layer shape
26
30
 
27
- ## Shape: reference vs workflow
31
+ ### Wrong: a reference skill written as keystrokes
28
32
 
29
- **Wrong - a reference skill drifting into a procedure:**
30
33
  ```markdown
31
- ## Procedures
32
- 1. Open the file
33
- 2. Add the wrapper
34
- 3. Register the route
35
- 4. Build and run
34
+ ## Adding a tab
35
+ 1. Open AppTabView.swift.
36
+ 2. Add a case to the Tab enum.
37
+ 3. Add a .tabItem block under the last one.
38
+ 4. Build.
36
39
  ```
37
- A reference (knowledge) skill should describe *the rule and where the canonical example lives*, not script the keystrokes.
38
40
 
39
- **Right - reference stays descriptive:**
41
+ A reference skill describes; it does not dictate. These steps go stale the first time the file changes.
42
+
43
+ ### Right: the rule plus where to look
44
+
40
45
  ```markdown
41
- ## Decision rules
42
- - New screen in a domain → add `<Name>Screen` under its Presentation layer,
43
- wrap content in the domain's container, route via the domain coordinator.
44
- Canonical: `<repo>/Features/Profile/ProfileScreen.swift`.
46
+ ## Conventions
47
+ - Tabs are cases of the root `Tab` enum; each case owns its own navigation path.
48
+ Canonical file: `App/Sources/AppTabView.swift`.
45
49
  ```
46
50
 
47
- **Right - workflow earns its numbered steps** (deterministic/irreversible spine):
51
+ ### Right: a workflow that earns its steps
52
+
48
53
  ```markdown
49
54
  ## Procedures
50
- 1. Pre-flight: `git status` clean? on base branch?
51
- 2. Branch: `git checkout -b <type>/<scope>/<desc>`
52
- 3. Commit: stage; conventional message; strip any AI co-author trailer.
53
- 4. Push; never force-push a protected branch.
54
- 5. PR: `gh pr create --base <base> ...` (no AI footer).
55
+ 1. Pre-flight: `git status` is clean and the base branch is correct.
56
+ 2. Create `{type}/{scope}-{short-description}`.
57
+ 3. Commit with a conventional message; strip any AI co-author trailer.
58
+ 4. Push without `--force`; never to a protected branch.
59
+ 5. `gh pr create --base {base}` with no AI footer in the body.
60
+
55
61
  ## Verification
56
- - `git log -1` shows clean message, no AI trailer, correct author.
62
+ - `git log -1` shows a clean message, no AI trailer, and the expected author.
57
63
  ```
58
64
 
59
- ---
60
-
61
- ## Leanness: cut what the model already knows
62
-
63
- **Bloat:** "PDF files are documents that contain text and images. To read them you first need a library that can parse the binary format..."
64
-
65
- **Lean:** "Extract text from PDFs with `pdfplumber`."
65
+ Every step is deterministic or irreversible, and the skill proves its result before it finishes.
@@ -1,43 +1,55 @@
1
- # Label intent check - does a title carry its own meaning?
1
+ # Label check
2
2
 
3
- A label (rule title, section header, principle/constraint name) is the handle a reader grasps *before* the body. If the handle leaks the wrong shape - ambiguous, one-sided, or off-topic - the reader forms the wrong model and the body has to fight it. This is a cheap test that catches that early: have **fresh subagents infer the rule from the label alone**, then compare against your intended meaning. Convergence = the label is self-carrying.
3
+ A reader forms a model of a rule from its label before reading the body. If the label points the wrong way (ambiguous, covering only one side, or about something adjacent), that wrong model sticks. This test measures what a label actually conveys by asking agents that never see the body.
4
4
 
5
5
  ## When to run it
6
- - Naming a **strict rule** or constraint inside a skill (the kind a reader must obey).
7
- - A **section header** that gates non-trivial content (phase/principle/constraint names).
8
- - Any label meant to convey the **whole intent** of its body (not a summary - the intent).
9
- - Skip it for purely organizational labels ("Examples", "See also") - they carry no intent.
10
-
11
- ## Inputs (collect first)
12
- 1. **Candidate label** - the exact string under test.
13
- 2. **Surrounding context** - one paragraph: which skill/doc, what adjacent labels cover, the domain concepts in play.
14
- 3. **Intended meaning** - 1-3 sentences of ground truth (what the body says / what the reader should walk away believing). **Not** shown to the inference agents. If you can't state it in 3 sentences, the body is unclear - fix that first.
15
-
16
- ## Procedure
17
- **One iteration = 5 parallel subagents**, each given ONLY the label + context (never the body), each forced to commit:
18
- ```
19
- Context: <surrounding context - 1 paragraph>.
20
- The label under test is: **"<candidate label>"**
21
- You have NOT seen the body. From ONLY this label + context, what do you infer it
22
- enforces / requires / means? 1-2 concrete sentences, under 80 words. Commit - no
23
- hedging, no clarifying questions.
24
- ```
25
- Launch all 5 in the **same** message (parallel) so they're independent by construction.
26
-
27
- **Convergence (per iteration), comparing the 5 inferences to the intended meaning:**
28
- - **PASS** - ≥4/5 cover *all* key intent beats (including the bidirectional / edge / scope beat, if the intent has one). Wording drift is fine.
29
- - **PARTIAL** - they consistently catch one side of a bidirectional rule but miss the other, or over-extend into an adjacent concern. Revise the label, re-run.
30
- - **FAIL** - inferences diverge wildly or miss the core beat. Revise, re-run.
31
-
32
- **Idempotency (after a PASS):** run **2 more rounds**, 5 fresh agents each (3×5 = 15 inferences), same prompt. Require **all 3 rounds PASS** with no new systematic miss - confirms the PASS wasn't luck. A regression in round 2/3 → treat as PARTIAL and iterate.
33
-
34
- **Termination:** converged (3/3 PASS) → apply the label. 3 iterations without a PASS → stop, hand back the best candidate + the persistent miss; don't force a label that won't stick.
6
+
7
+ - Names of strict rules and constraints.
8
+ - Section headers that gate non-trivial content.
9
+ - Any label that is meant to carry the whole intent on its own, including top-level skill names.
10
+
11
+ Purely organisational headers ("Examples", "Setup") do not need it.
12
+
13
+ ## Inputs
14
+
15
+ 1. The candidate label.
16
+ 2. One paragraph of the surrounding context, as a reader would meet it.
17
+ 3. The intended meaning in one to three sentences. This stays with you; the agents never see it. If you cannot state it in three sentences, fix the body before testing the label.
18
+
19
+ ## A round
20
+
21
+ Launch five subagents **in one message**, so none anchors on another. Each receives only the label and the context paragraph, never the body, and is told to commit to what the label means in one or two sentences, under 80 words, with no hedging and no questions back.
22
+
23
+ ## Grading
24
+
25
+ | Grade | Condition | Next |
26
+ |---|---|---|
27
+ | PASS | At least 4 of 5 inferences cover every key beat of the intent, including the bidirectional, edge or scope beat. Different wording is fine. | Idempotency rounds |
28
+ | PARTIAL | The inferences consistently catch one side of a two-sided rule, or stretch into a neighbouring concern. | Revise, run again |
29
+ | FAIL | Inferences diverge, or miss the core beat. | Revise, run again |
30
+
31
+ ## Idempotency
32
+
33
+ After a first PASS, run two more rounds with five fresh agents each: 15 inferences in total. All three rounds must pass with no new systematic miss. A later round that regresses counts as PARTIAL.
34
+
35
+ ## Stopping
36
+
37
+ - Three of three rounds pass: apply the label.
38
+ - Three revisions without a PASS: stop, and hand back the best candidate together with the miss that kept recurring.
35
39
 
36
40
  ## Defaults
37
- 5 agents/round · 3 rounds · 80-word inference cap · max 3 iterations before handing back. High-stakes labels (top-level skill names, protocol names) → 7 agents × 3 rounds; throwaway section headers → 3 × 2.
41
+
42
+ | Stakes | Agents | Rounds |
43
+ |---|---|---|
44
+ | normal | 5 | 3 |
45
+ | high (top-level skill names, protocol names) | 7 | 3 |
46
+ | throwaway header | 3 | 2 |
47
+
48
+ Word cap 80; at most 3 revision iterations.
38
49
 
39
50
  ## Anti-patterns
40
- - **Giving agents the body** - defeats the test (you're checking the label's *self-carrying* power).
41
- - **Running sequentially** - they'd anchor on each other; must be parallel.
42
- - **Tweaking one word on a FAIL** - step back and rethink the concept the label names, don't polish a wrong handle.
43
- - **Accepting "mostly right"** - a label that reliably misses a beat misleads every reader the same way. Fix it, or accept the rule is one-sided and split the body.
51
+
52
+ - Showing the agents the body. That tests the body, not the label.
53
+ - Running agents one after another; later ones anchor on earlier answers.
54
+ - Changing one word after a FAIL. A FAIL means the concept behind the label is off; rethink it.
55
+ - Accepting "mostly right". For a one-sided result, either fix the label or split the rule so each part has a label that fits.
@@ -1,119 +1,110 @@
1
- // audit-panel.js - Workflow template for the multi-architect skill audit.
2
- //
3
- // Run with the Workflow tool. Edit the CONFIG block for your target, then launch.
4
- // Returns { remaining, clearsBar, verdict }. If clearsBar is false, apply the
5
- // `remaining` blockers/majors, then RE-RUN A FRESH copy (do NOT resume - agents
6
- // must read the current files). Loop until clearsBar. See ../audit.md.
7
-
8
1
  export const meta = {
9
- name: "skill-audit-panel",
10
- description: "Adversarial N-architect audit of a skill against the skill-creator rubric",
11
- phases: [{ title: "Audit" }, { title: "Synthesis" }],
12
- };
13
-
14
- // ── CONFIG - edit these ──────────────────────────────────────────────────────
15
- const TARGET = `
16
- - <abs path to SKILL.md>
17
- - <abs path to each bundled file / script>
18
- `;
19
- const RUBRIC = `<abs path to reference/skill-creator/SKILL.md> + checklist.md`;
20
- const EXEMPLAR = `<abs path to a known-good skill of the same layer>`;
21
- const GROUND_TRUTH = `<abs path to the real source the skill describes (e.g. the iOS repo), with key facts>`;
22
- // Pick a lens set from audit.md (general or prompt-engineering). [label, brief] pairs:
23
- const LENSES = [
24
- [
25
- "discovery",
26
- "description-as-discovery: what+when+trigger words+example; would it activate on a real task?",
27
- ],
28
- [
29
- "disambiguation",
30
- "distinct from sibling skills; a reader can tell when to pick this vs a neighbor.",
31
- ],
32
- ["triggers", "real user phrasings + synonyms are covered."],
33
- ["clarity", "instructions unambiguous, ordered, single-interpretation; no vague directives."],
34
- [
35
- "layer",
36
- "degrees of freedom match the layer (reference=descriptive / workflow=spine / tool=thin-index).",
2
+ name: 'skill-audit-panel',
3
+ description: 'Runs an independent, multi-lens review of a single plugin skill, scored on the skill-creator rubric',
4
+ whenToUse: 'Gate before shipping a skill that is new or heavily reworked; start a new run once each round of fixes lands',
5
+ phases: [
6
+ { title: 'Audit', detail: 'one read-only reviewer per lens' },
7
+ { title: 'Synthesis', detail: 'dedup, verify, classify, decide clearsBar' },
37
8
  ],
38
- ["leanness", "SKILL.md is an index; detail bundled one level deep; no duplication or bloat."],
39
- ["tokens", "every line load-bearing; nothing restates what the model already knows."],
40
- ["consistency", "terminology, naming, paths, cross-refs, examples all agree."],
41
- [
42
- "misfire",
43
- "would the model over-apply, skip a gate, hallucinate a path, or follow it to a wrong output?",
44
- ],
45
- [
46
- "redteam",
47
- "adversarially make it not trigger / trigger wrongly / produce broken output; hunt overclaims (says X, script does not).",
48
- ],
49
- ];
50
- // ─────────────────────────────────────────────────────────────────────────────
9
+ }
51
10
 
52
- const COMMON = `
53
- You are auditing a skill. Read the CURRENT files and report ONLY genuine, specific,
54
- reproducible findings for YOUR lens. Verify against the rubric + ground truth - do
55
- not assume. Read-only; do NOT edit. If your lens is clean, return an empty array.
11
+ const TARGET = [
12
+ '/abs/path/to/plugins/<plugin>/skills/<layer>/<skill>/SKILL.md',
13
+ '/abs/path/to/plugins/<plugin>/skills/<layer>/<skill>/reference.md',
14
+ ]
15
+ const RUBRIC = '/abs/path/to/plugins/ai-common-toolkit/skills/knowledge/skill-creator/SKILL.md and checklist.md beside it'
16
+ const EXEMPLAR = '/abs/path/to/an/exemplar/skill/in/this/layer/SKILL.md'
17
+ const GROUND_TRUTH = '/abs/path/to/source/code/described/by/the/skill (name the key facts worth checking)'
56
18
 
57
- TARGET (the skill under audit):${TARGET}
58
- RUBRIC (the bar): ${RUBRIC}
59
- EXEMPLAR (known-good of this layer): ${EXEMPLAR}
60
- GROUND TRUTH (what the skill describes): ${GROUND_TRUTH}
61
-
62
- Each finding: severity (blocker|major|minor|nit), file:location, the problem, the concrete fix.
63
- `;
19
+ const LENSES = [
20
+ { label: 'discovery', brief: 'Does the description alone get the skill loaded? What + when + trigger words + a sample trigger, third person, platform named.' },
21
+ { label: 'disambiguation', brief: 'Could a router confuse this skill with a sibling? Are the boundaries and hand-offs explicit in the description and header?' },
22
+ { label: 'triggers', brief: 'Which real phrasings and synonyms a user would type are missing from the description?' },
23
+ { label: 'clarity', brief: 'Is every instruction unambiguous and actionable, with no step that two readers would execute differently?' },
24
+ { label: 'layer', brief: 'Does the content match its declared layer (reference descriptive, workflow imperative with a verification gate, tool a thin index over a script)?' },
25
+ { label: 'leanness', brief: 'Is SKILL.md an index of about one page, with detail one level deep in bundled files and deterministic code in scripts?' },
26
+ { label: 'tokens', brief: 'Which lines fail the cut test because the model would act correctly without them?' },
27
+ { label: 'consistency', brief: 'Do terms, names, paths, cross-references and examples agree with each other and with the files that exist?' },
28
+ { label: 'misfire', brief: 'Where would the model over-apply the skill, skip its gate, invent a path, or produce the wrong output?' },
29
+ { label: 'redteam', brief: 'Try to make it fail to trigger, trigger wrongly, or break its output; hunt for prose that claims what the scripts do not do.' },
30
+ ]
64
31
 
65
32
  const FINDING = {
66
- type: "object",
33
+ type: 'object',
67
34
  additionalProperties: false,
35
+ required: ['severity', 'location', 'problem', 'fix'],
68
36
  properties: {
69
- severity: { type: "string", enum: ["blocker", "major", "minor", "nit"] },
70
- location: { type: "string" },
71
- problem: { type: "string" },
72
- fix: { type: "string" },
37
+ severity: { enum: ['blocker', 'major', 'minor', 'nit'], type: 'string' },
38
+ location: { type: 'string', description: 'file:line or file#heading' },
39
+ problem: { type: 'string' },
40
+ fix: { type: 'string', description: 'a concrete change, not advice' },
73
41
  },
74
- required: ["severity", "location", "problem", "fix"],
75
- };
76
- const LENS_SCHEMA = {
77
- type: "object",
78
- additionalProperties: false,
79
- properties: { lens: { type: "string" }, findings: { type: "array", items: FINDING } },
80
- required: ["lens", "findings"],
81
- };
42
+ }
82
43
 
83
- phase("Audit");
84
- const results = await parallel(
85
- LENSES.map(
86
- ([lens, brief]) =>
87
- () =>
88
- agent(`${COMMON}\n\nYOUR LENS - ${lens}: ${brief}`, {
89
- label: `audit:${lens}`,
90
- phase: "Audit",
91
- schema: LENS_SCHEMA,
92
- }),
93
- ),
94
- );
95
- const all = results.filter(Boolean).flatMap((r) => r.findings.map((f) => ({ ...f, lens: r.lens })));
44
+ const LENS_RESULT = {
45
+ type: 'object',
46
+ additionalProperties: false,
47
+ required: ['lens', 'findings'],
48
+ properties: {
49
+ lens: { type: 'string' },
50
+ findings: { type: 'array', items: FINDING },
51
+ },
52
+ }
96
53
 
97
- phase("Synthesis");
98
- const SYNTH = {
99
- type: "object",
54
+ const SYNTHESIS = {
55
+ type: 'object',
100
56
  additionalProperties: false,
57
+ required: ['remaining', 'clearsBar', 'verdict'],
101
58
  properties: {
102
59
  remaining: {
103
- type: "array",
60
+ type: 'array',
104
61
  items: {
105
- type: "object",
62
+ type: 'object',
106
63
  additionalProperties: false,
107
- properties: { ...FINDING.properties, lenses: { type: "string" } },
108
- required: [...FINDING.required, "lenses"],
64
+ required: ['severity', 'location', 'problem', 'fix', 'lenses'],
65
+ properties: { lenses: { type: 'string', description: 'every lens that reported it, comma-separated' }, ...FINDING.properties },
109
66
  },
110
67
  },
111
- clearsBar: { type: "boolean" },
112
- verdict: { type: "string" },
68
+ clearsBar: { type: 'boolean' },
69
+ verdict: { type: 'string', description: 'one paragraph' },
113
70
  },
114
- required: ["remaining", "clearsBar", "verdict"],
115
- };
116
- return await agent(
117
- `${COMMON}\n\nYou are the SYNTHESIS architect. Below are ${all.length} raw findings from ${LENSES.length} lenses. Dedup overlaps (note which lenses raised each). DROP false positives - verify against the files before keeping; if you cannot reproduce it, drop it. Classify each survivor; set clearsBar=true ONLY if no blockers and no majors remain. One-paragraph verdict.\n\nRAW:\n${JSON.stringify(all, null, 2)}`,
118
- { label: "synthesis", phase: "Synthesis", schema: SYNTH },
119
- );
71
+ }
72
+
73
+ const PREAMBLE = [
74
+ 'You are one reviewer on an independent audit panel for a plugin skill. Stay read-only: do not edit any file.',
75
+ `Read the current state of every target file now; do not rely on anything you saw earlier:\n- ${TARGET.join('\n- ')}`,
76
+ `Rubric: ${RUBRIC}`,
77
+ `Exemplar of the same layer: ${EXEMPLAR}`,
78
+ `Ground truth the skill describes: ${GROUND_TRUTH}`,
79
+ 'Report only genuine findings for your lens that you can reproduce by pointing at a line. Check each one against the rubric and the ground truth before reporting it.',
80
+ 'Each finding needs a severity (blocker, major, minor, nit), a location, the problem, and a concrete fix. If your lens finds nothing, return an empty findings array.',
81
+ ].join('\n\n')
82
+
83
+ phase('Audit')
84
+ const lensOutputs = await parallel(LENSES.map((lensDef) => () =>
85
+ agent(`${PREAMBLE}\n\nYour lens: ${lensDef.label}. ${lensDef.brief}\nSet "lens" to "${lensDef.label}".`, {
86
+ label: `audit:${lensDef.label}`,
87
+ phase: 'Audit',
88
+ schema: LENS_RESULT,
89
+ })))
90
+
91
+ const raw = lensOutputs
92
+ .filter(Boolean)
93
+ .flatMap((out) => out.findings.map((item) => ({ ...item, lens: out.lens })))
94
+ log(`${raw.length} raw finding(s) from ${lensOutputs.filter(Boolean).length}/${LENSES.length} lenses`)
95
+
96
+ phase('Synthesis')
97
+ const synthesis = await agent([
98
+ 'You synthesize an audit panel\'s findings for a plugin skill. Stay read-only.',
99
+ `Target files:\n- ${TARGET.join('\n- ')}`,
100
+ `Rubric: ${RUBRIC}`,
101
+ `Ground truth: ${GROUND_TRUTH}`,
102
+ `Raw findings (JSON):\n${JSON.stringify(raw, null, 2)}`,
103
+ 'Merge duplicates and record every lens that raised each merged item in "lenses".',
104
+ 'Open the cited file for each finding and drop any you cannot reproduce there.',
105
+ 'Classify what survives as blocker, major, minor or nit.',
106
+ 'Set clearsBar to true only when no blocker and no major remains.',
107
+ 'Write a one-paragraph verdict.',
108
+ ].join('\n\n'), { label: 'synthesis', phase: 'Synthesis', schema: SYNTHESIS })
109
+
110
+ return synthesis