@mmerterden/multi-agent-pipeline 20.6.0 → 20.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +18 -0
- package/docs/facts.json +1 -1
- package/manifest.json +71 -71
- package/package.json +1 -1
- package/pipeline/multi-agent-refs/features/design-conformance.md +62 -64
- package/pipeline/scripts/_notices.mjs +11 -0
- package/pipeline/scripts/gen-skills-index.mjs +1 -1
- package/pipeline/skills/.skill-manifest.json +39 -39
- package/pipeline/skills/shared/README.md +1 -1
- package/pipeline/skills/shared/external/agent-introspection-debugging/SKILL.md +1 -0
- package/pipeline/skills/shared/external/android-architecture/SKILL.md +2 -0
- package/pipeline/skills/shared/external/android-performance/SKILL.md +2 -0
- package/pipeline/skills/shared/external/android-security/SKILL.md +2 -0
- package/pipeline/skills/shared/external/backlog/BACKLOG.md +1 -1
- package/pipeline/skills/shared/external/backlog/SKILL.md +56 -33
- package/pipeline/skills/shared/external/ci-cd-pipelines/SKILL.md +1 -0
- package/pipeline/skills/shared/external/compose-components/SKILL.md +2 -0
- package/pipeline/skills/shared/external/compose-navigation/SKILL.md +3 -2
- package/pipeline/skills/shared/external/compose-testing/SKILL.md +2 -0
- package/pipeline/skills/shared/external/council/SKILL.md +1 -0
- package/pipeline/skills/shared/external/css-modern/SKILL.md +1 -0
- package/pipeline/skills/shared/external/database-patterns/SKILL.md +1 -0
- package/pipeline/skills/shared/external/evidence-github/SKILL.md +2 -0
- package/pipeline/skills/shared/external/evidence-registry/SKILL.md +2 -0
- package/pipeline/skills/shared/external/gradle-kotlin-dsl/SKILL.md +2 -0
- package/pipeline/skills/shared/external/html-semantic/SKILL.md +1 -0
- package/pipeline/skills/shared/external/humanizer/SKILL.md +1 -0
- package/pipeline/skills/shared/external/ios-coding-standard/SKILL.md +1 -0
- package/pipeline/skills/shared/external/ios-module-structure/SKILL.md +1 -0
- package/pipeline/skills/shared/external/ios-security/SKILL.md +2 -0
- package/pipeline/skills/shared/external/localization-reuse-map/SKILL.md +91 -283
- package/pipeline/skills/shared/external/localization-reuse-map/reference/format-and-output.md +119 -151
- package/pipeline/skills/shared/external/localization-reuse-map/reference/publish-and-snapshot.md +60 -90
- package/pipeline/skills/shared/external/localization-reuse-map/reference/sources-and-recipes.md +119 -156
- package/pipeline/skills/shared/external/localization-reuse-map/scripts/build-artifact.py +726 -787
- package/pipeline/skills/shared/external/localization-reuse-map/scripts/build-spreadsheet.py +253 -288
- package/pipeline/skills/shared/external/localization-reuse-map/scripts/fetch-annotations.py +243 -304
- package/pipeline/skills/shared/external/localization-reuse-map/scripts/fetch-legacy-labels.py +88 -104
- package/pipeline/skills/shared/external/localization-reuse-map/scripts/publish-confluence.py +181 -235
- package/pipeline/skills/shared/external/localization-reuse-map/scripts/render-key-shots.py +198 -263
- package/pipeline/skills/shared/external/localization-reuse-map/scripts/render-overlay.py +461 -466
- package/pipeline/skills/shared/external/localization-reuse-map/scripts/resolve-legacy-values.py +145 -151
- package/pipeline/skills/shared/external/localization-reuse-map/scripts/resolve-new-values.py +123 -141
- package/pipeline/skills/shared/external/localization-reuse-map/scripts/scan-screen-keys.py +146 -157
- package/pipeline/skills/shared/external/localization-reuse-map/scripts/snapshot-resources.sh +22 -19
- package/pipeline/skills/shared/external/localization-reuse-map/scripts/verify-map.py +156 -140
- package/pipeline/skills/shared/external/nextjs-app-router/SKILL.md +1 -0
- package/pipeline/skills/shared/external/play-store-review/SKILL.md +2 -0
- package/pipeline/skills/shared/external/python-patterns/SKILL.md +1 -0
- package/pipeline/skills/shared/external/react-best-practices/SKILL.md +1 -0
- package/pipeline/skills/shared/external/rest-api-design/SKILL.md +1 -0
- package/pipeline/skills/shared/external/retrofit-networking/SKILL.md +2 -0
- package/pipeline/skills/shared/external/room-database/SKILL.md +2 -0
- package/pipeline/skills/shared/external/search-first/SKILL.md +1 -0
- package/pipeline/skills/shared/external/signal-community/SKILL.md +2 -0
- package/pipeline/skills/shared/external/skill-creator/SKILL.md +80 -41
- package/pipeline/skills/shared/external/skill-creator/audit.md +63 -59
- package/pipeline/skills/shared/external/skill-creator/checklist.md +28 -20
- package/pipeline/skills/shared/external/skill-creator/examples.md +40 -40
- package/pipeline/skills/shared/external/skill-creator/label-check.md +48 -36
- package/pipeline/skills/shared/external/skill-creator/scripts/audit-panel.js +91 -100
- package/pipeline/skills/shared/external/skill-creator/template.md +51 -39
- package/pipeline/skills/shared/external/tailwind-css/SKILL.md +1 -0
- package/pipeline/skills/shared/external/testing-backend/SKILL.md +1 -0
- package/pipeline/skills/shared/external/typescript-patterns/SKILL.md +1 -0
- package/pipeline/skills/shared/external/vue-composition/SKILL.md +1 -0
- package/pipeline/skills/shared/external/web-accessibility/SKILL.md +1 -0
- package/pipeline/skills/shared/external/web-performance/SKILL.md +1 -0
- package/pipeline/skills/shared/external/web-testing/SKILL.md +1 -0
|
@@ -1,59 +1,63 @@
|
|
|
1
|
-
# Multi-architect audit
|
|
2
|
-
|
|
3
|
-
The
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
`
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
- **
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
1
|
+
# Multi-architect audit
|
|
2
|
+
|
|
3
|
+
The checklist is what one author catches alone. The audit adds a panel of independent, read-only reviewers, each looking through one lens, and a synthesizer that turns their output into a decision. New skills and substantial changes go through it before merge.
|
|
4
|
+
|
|
5
|
+
## The loop
|
|
6
|
+
|
|
7
|
+
1. **Panel.** Every lens reviews the skill and returns findings, each with a severity, a `file:location`, the problem and a concrete fix. A clean lens returns an empty list.
|
|
8
|
+
2. **Synthesis.** One agent merges duplicates, opens the real files to discard false positives, classifies the rest as blocker, major, minor or nit, and sets `clearsBar` to true only when no blocker and no major is left.
|
|
9
|
+
3. **Fix.** If the bar is not cleared, apply the do-now set and start a **new** panel.
|
|
10
|
+
4. **Stop** once `clearsBar` is true. Remaining nits are optional polish.
|
|
11
|
+
|
|
12
|
+
### What to fix in a round
|
|
13
|
+
|
|
14
|
+
- Always: every blocker and every major.
|
|
15
|
+
- Also, when cheap and clearly valuable: minors.
|
|
16
|
+
- Later, on a written list: speculative nits and drift that was already there or sits outside this change. Record them; never drop them silently.
|
|
17
|
+
|
|
18
|
+
## Rules that keep the result trustworthy
|
|
19
|
+
|
|
20
|
+
- **Fresh panel every round.** Cached results describe the files as they were; a resumed run audits the old version.
|
|
21
|
+
- **Verify before keeping.** Each finding is checked against the file before it counts. Prefer findings that cite a line and a way to reproduce.
|
|
22
|
+
- **Give the panel ground truth:** the rubric (skill-creator SKILL.md plus checklist.md), an exemplar skill of the same layer, and the actual source the skill describes.
|
|
23
|
+
- **Watch the fixes.** A fix can break something else. The red-team lens and the fresh re-audit are there to catch that.
|
|
24
|
+
- **`clearsBar` gates the merge.**
|
|
25
|
+
|
|
26
|
+
## Lens sets
|
|
27
|
+
|
|
28
|
+
Scale the panel to the risk: about three lenses for a quick check, about ten for a full gate.
|
|
29
|
+
|
|
30
|
+
### General
|
|
31
|
+
|
|
32
|
+
discovery, leanness, shape and layer, scripts, accuracy against the source, domain correctness, wiring, boundaries, red team, then synthesis.
|
|
33
|
+
|
|
34
|
+
### Prompt engineering (the script's default ten)
|
|
35
|
+
|
|
36
|
+
| Lens | Asks |
|
|
37
|
+
|---|---|
|
|
38
|
+
| discovery | Does the description alone get the skill loaded? |
|
|
39
|
+
| disambiguation | Could it be confused with a sibling? |
|
|
40
|
+
| triggers | Which real phrasings and synonyms are missing? |
|
|
41
|
+
| clarity | Is every instruction unambiguous and actionable? |
|
|
42
|
+
| layer | Does the degree of freedom fit the declared layer? |
|
|
43
|
+
| leanness | Is SKILL.md an index with detail one level down? |
|
|
44
|
+
| tokens | Which lines fail the cut test? |
|
|
45
|
+
| consistency | Do terms, names, paths, cross-references and examples agree? |
|
|
46
|
+
| misfire | Where would it over-apply, skip its gate, invent a path, or produce the wrong output? |
|
|
47
|
+
| redteam | Make it fail to trigger, trigger wrongly, or break its output; find prose that claims what the script does not do. |
|
|
48
|
+
|
|
49
|
+
## Running it with the script
|
|
50
|
+
|
|
51
|
+
`scripts/audit-panel.js` is a Workflow-tool template (workflow name `skill-audit-panel`, phases Audit and Synthesis). It takes no arguments; edit its config block first:
|
|
52
|
+
|
|
53
|
+
| Constant | Set to |
|
|
54
|
+
|---|---|
|
|
55
|
+
| `TARGET` | absolute paths of the SKILL.md and each bundled file and script |
|
|
56
|
+
| `RUBRIC` | path to skill-creator's SKILL.md plus checklist.md |
|
|
57
|
+
| `EXEMPLAR` | a known-good skill of the same layer |
|
|
58
|
+
| `GROUND_TRUTH` | the real source the skill documents, with the key facts to check |
|
|
59
|
+
| `LENSES` | label and brief pairs; defaults to the ten above |
|
|
60
|
+
|
|
61
|
+
It runs one agent per lens in parallel (labelled `audit:<lens>`), tags each finding with its lens, and hands them all to a `synthesis` agent. The result is `{ remaining, clearsBar, verdict }`: the surviving findings with the lenses that raised them, the gate boolean, and a one-paragraph verdict.
|
|
62
|
+
|
|
63
|
+
After applying fixes, launch it again as a new run. Do not resume the old one; resumed agents return cached results.
|
|
@@ -1,32 +1,40 @@
|
|
|
1
1
|
# Pre-ship checklist
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
The floor for every skill. New or substantially changed skills also go through the audit loop in audit.md.
|
|
4
4
|
|
|
5
5
|
## Discovery
|
|
6
|
-
|
|
7
|
-
- [ ]
|
|
8
|
-
- [ ]
|
|
6
|
+
|
|
7
|
+
- [ ] The description says what the skill does, when to use it, and lists trigger phrases, in third person.
|
|
8
|
+
- [ ] It names the platform where that matters.
|
|
9
|
+
- [ ] It separates this skill from its siblings.
|
|
10
|
+
- [ ] Tried on a new task it ought to catch, the skill really fires.
|
|
9
11
|
|
|
10
12
|
## Leanness
|
|
11
|
-
|
|
12
|
-
- [ ]
|
|
13
|
-
- [ ]
|
|
14
|
-
- [ ]
|
|
13
|
+
|
|
14
|
+
- [ ] SKILL.md is roughly one page and works as an index, far below the 500-line cap.
|
|
15
|
+
- [ ] Detail lives in bundled files exactly one level below SKILL.md.
|
|
16
|
+
- [ ] Every line passes the cut test.
|
|
17
|
+
- [ ] Anything deterministic ships as a script instead of prose the model must re-derive.
|
|
15
18
|
|
|
16
19
|
## Shape
|
|
17
|
-
- [ ] Layer is explicit (reference or workflow) and matches the content.
|
|
18
|
-
- [ ] **Reference** skill has **no numbered procedures** - heuristics + pointers only.
|
|
19
|
-
- [ ] **Workflow** skill's numbered steps are **only** the deterministic/irreversible spine, and it has a **verification gate**.
|
|
20
|
-
- [ ] Conventions **point to canonical files** rather than pasting code that will rot.
|
|
21
20
|
|
|
22
|
-
|
|
23
|
-
- [ ]
|
|
24
|
-
- [ ]
|
|
25
|
-
- [ ]
|
|
21
|
+
- [ ] The layer is declared and the content matches it.
|
|
22
|
+
- [ ] Numbered procedures never appear in a reference skill.
|
|
23
|
+
- [ ] A workflow skill's steps are only the deterministic spine, and it ends in a verification gate.
|
|
24
|
+
- [ ] Conventions point at canonical files instead of pasting code that will go stale.
|
|
25
|
+
|
|
26
|
+
## Naming and wiring
|
|
27
|
+
|
|
28
|
+
- [ ] No platform prefix; noun for reference, verb phrase for workflow; not vague, not reserved.
|
|
29
|
+
- [ ] Registered in the `skills` array of `<plugin>/.claude-plugin/plugin.json`.
|
|
30
|
+
- [ ] Plugin version bumped in `.github/plugin/marketplace.json`.
|
|
31
|
+
- [ ] A row in the plugin's `index` skill route table.
|
|
26
32
|
|
|
27
33
|
## House rules
|
|
28
|
-
- [ ] No AI/Claude/Anthropic attribution anywhere in the skill or its examples.
|
|
29
|
-
- [ ] Escape hatches name the skill to hand off to when this one doesn't apply.
|
|
30
34
|
|
|
31
|
-
|
|
32
|
-
- [ ]
|
|
35
|
+
- [ ] Neither the skill nor its examples carry AI, Claude or Anthropic attribution.
|
|
36
|
+
- [ ] Each escape hatch says which skill takes over.
|
|
37
|
+
|
|
38
|
+
## Confidence gate
|
|
39
|
+
|
|
40
|
+
- [ ] New or substantially changed: the multi-architect audit ran, fixes were applied, and a fresh panel returned `clearsBar: true`.
|
|
@@ -1,65 +1,65 @@
|
|
|
1
|
-
# Examples
|
|
1
|
+
# Examples
|
|
2
2
|
|
|
3
|
-
## Descriptions
|
|
3
|
+
## Descriptions
|
|
4
|
+
|
|
5
|
+
### Weak
|
|
4
6
|
|
|
5
|
-
**Bad (vague - rarely triggers):**
|
|
6
7
|
```yaml
|
|
7
|
-
description: Helps with navigation
|
|
8
|
+
description: Helps with navigation stuff.
|
|
8
9
|
```
|
|
9
|
-
Why it fails: no "when", no trigger words, no platform - the router can't tell what task this matches.
|
|
10
10
|
|
|
11
|
-
|
|
11
|
+
No situation, no trigger words, no platform. The router has nothing to match a request against, so the skill stays unloaded.
|
|
12
|
+
|
|
13
|
+
### Strong, reference layer
|
|
14
|
+
|
|
12
15
|
```yaml
|
|
13
|
-
description: "
|
|
16
|
+
description: "Navigation for iOS / SwiftUI: NavigationStack paths, tab structure, deep links, sheets and full-screen modals, and where routing state lives. Use when adding a screen, a tab or a deep link, presenting a modal, or debugging a push that does nothing. Triggers: 'add a new screen', 'open this from a deep link', 'present as a sheet'."
|
|
14
17
|
```
|
|
15
|
-
Why it works: states the domain, lists concrete triggers ("adding a screen, a tab, a deep link"), names the platform, and implies the boundary.
|
|
16
18
|
|
|
17
|
-
|
|
19
|
+
Coverage, concrete situations, the platform, and quoted phrases a developer would type. The domain words keep it apart from sibling skills on state or overlays.
|
|
20
|
+
|
|
21
|
+
### Strong, workflow layer
|
|
22
|
+
|
|
18
23
|
```yaml
|
|
19
|
-
description: "
|
|
24
|
+
description: "Creates a branch, commits, pushes and opens a pull request with the house guardrails: type-first branch names, conventional commit messages with no AI attribution, never a force push, and the correct base branch. Use when work is ready to ship. Triggers: 'open a PR', 'commit and push this', 'create a branch for this ticket'."
|
|
20
25
|
```
|
|
21
26
|
|
|
22
|
-
|
|
23
|
-
`<what it covers/does> + <concrete when/trigger phrases> + <platform> + <implied boundary vs siblings>`
|
|
27
|
+
States the action, the guardrails that make it safe, when it applies, and how users ask for it.
|
|
24
28
|
|
|
25
|
-
|
|
29
|
+
## Layer shape
|
|
26
30
|
|
|
27
|
-
|
|
31
|
+
### Wrong: a reference skill written as keystrokes
|
|
28
32
|
|
|
29
|
-
**Wrong - a reference skill drifting into a procedure:**
|
|
30
33
|
```markdown
|
|
31
|
-
##
|
|
32
|
-
1. Open
|
|
33
|
-
2. Add the
|
|
34
|
-
3.
|
|
35
|
-
4. Build
|
|
34
|
+
## Adding a tab
|
|
35
|
+
1. Open AppTabView.swift.
|
|
36
|
+
2. Add a case to the Tab enum.
|
|
37
|
+
3. Add a .tabItem block under the last one.
|
|
38
|
+
4. Build.
|
|
36
39
|
```
|
|
37
|
-
A reference (knowledge) skill should describe *the rule and where the canonical example lives*, not script the keystrokes.
|
|
38
40
|
|
|
39
|
-
|
|
41
|
+
A reference skill describes; it does not dictate. These steps go stale the first time the file changes.
|
|
42
|
+
|
|
43
|
+
### Right: the rule plus where to look
|
|
44
|
+
|
|
40
45
|
```markdown
|
|
41
|
-
##
|
|
42
|
-
-
|
|
43
|
-
|
|
44
|
-
Canonical: `<repo>/Features/Profile/ProfileScreen.swift`.
|
|
46
|
+
## Conventions
|
|
47
|
+
- Tabs are cases of the root `Tab` enum; each case owns its own navigation path.
|
|
48
|
+
Canonical file: `App/Sources/AppTabView.swift`.
|
|
45
49
|
```
|
|
46
50
|
|
|
47
|
-
|
|
51
|
+
### Right: a workflow that earns its steps
|
|
52
|
+
|
|
48
53
|
```markdown
|
|
49
54
|
## Procedures
|
|
50
|
-
1. Pre-flight: `git status` clean
|
|
51
|
-
2.
|
|
52
|
-
3. Commit
|
|
53
|
-
4. Push
|
|
54
|
-
5.
|
|
55
|
+
1. Pre-flight: `git status` is clean and the base branch is correct.
|
|
56
|
+
2. Create `{type}/{scope}-{short-description}`.
|
|
57
|
+
3. Commit with a conventional message; strip any AI co-author trailer.
|
|
58
|
+
4. Push without `--force`; never to a protected branch.
|
|
59
|
+
5. `gh pr create --base {base}` with no AI footer in the body.
|
|
60
|
+
|
|
55
61
|
## Verification
|
|
56
|
-
- `git log -1` shows clean message, no AI trailer,
|
|
62
|
+
- `git log -1` shows a clean message, no AI trailer, and the expected author.
|
|
57
63
|
```
|
|
58
64
|
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
## Leanness: cut what the model already knows
|
|
62
|
-
|
|
63
|
-
**Bloat:** "PDF files are documents that contain text and images. To read them you first need a library that can parse the binary format..."
|
|
64
|
-
|
|
65
|
-
**Lean:** "Extract text from PDFs with `pdfplumber`."
|
|
65
|
+
Every step is deterministic or irreversible, and the skill proves its result before it finishes.
|
|
@@ -1,43 +1,55 @@
|
|
|
1
|
-
# Label
|
|
1
|
+
# Label check
|
|
2
2
|
|
|
3
|
-
A
|
|
3
|
+
A reader forms a model of a rule from its label before reading the body. If the label points the wrong way (ambiguous, covering only one side, or about something adjacent), that wrong model sticks. This test measures what a label actually conveys by asking agents that never see the body.
|
|
4
4
|
|
|
5
5
|
## When to run it
|
|
6
|
-
|
|
7
|
-
-
|
|
8
|
-
-
|
|
9
|
-
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
6
|
+
|
|
7
|
+
- Names of strict rules and constraints.
|
|
8
|
+
- Section headers that gate non-trivial content.
|
|
9
|
+
- Any label that is meant to carry the whole intent on its own, including top-level skill names.
|
|
10
|
+
|
|
11
|
+
Purely organisational headers ("Examples", "Setup") do not need it.
|
|
12
|
+
|
|
13
|
+
## Inputs
|
|
14
|
+
|
|
15
|
+
1. The candidate label.
|
|
16
|
+
2. One paragraph of the surrounding context, as a reader would meet it.
|
|
17
|
+
3. The intended meaning in one to three sentences. This stays with you; the agents never see it. If you cannot state it in three sentences, fix the body before testing the label.
|
|
18
|
+
|
|
19
|
+
## A round
|
|
20
|
+
|
|
21
|
+
Launch five subagents **in one message**, so none anchors on another. Each receives only the label and the context paragraph, never the body, and is told to commit to what the label means in one or two sentences, under 80 words, with no hedging and no questions back.
|
|
22
|
+
|
|
23
|
+
## Grading
|
|
24
|
+
|
|
25
|
+
| Grade | Condition | Next |
|
|
26
|
+
|---|---|---|
|
|
27
|
+
| PASS | At least 4 of 5 inferences cover every key beat of the intent, including the bidirectional, edge or scope beat. Different wording is fine. | Idempotency rounds |
|
|
28
|
+
| PARTIAL | The inferences consistently catch one side of a two-sided rule, or stretch into a neighbouring concern. | Revise, run again |
|
|
29
|
+
| FAIL | Inferences diverge, or miss the core beat. | Revise, run again |
|
|
30
|
+
|
|
31
|
+
## Idempotency
|
|
32
|
+
|
|
33
|
+
After a first PASS, run two more rounds with five fresh agents each: 15 inferences in total. All three rounds must pass with no new systematic miss. A later round that regresses counts as PARTIAL.
|
|
34
|
+
|
|
35
|
+
## Stopping
|
|
36
|
+
|
|
37
|
+
- Three of three rounds pass: apply the label.
|
|
38
|
+
- Three revisions without a PASS: stop, and hand back the best candidate together with the miss that kept recurring.
|
|
35
39
|
|
|
36
40
|
## Defaults
|
|
37
|
-
|
|
41
|
+
|
|
42
|
+
| Stakes | Agents | Rounds |
|
|
43
|
+
|---|---|---|
|
|
44
|
+
| normal | 5 | 3 |
|
|
45
|
+
| high (top-level skill names, protocol names) | 7 | 3 |
|
|
46
|
+
| throwaway header | 3 | 2 |
|
|
47
|
+
|
|
48
|
+
Word cap 80; at most 3 revision iterations.
|
|
38
49
|
|
|
39
50
|
## Anti-patterns
|
|
40
|
-
|
|
41
|
-
-
|
|
42
|
-
-
|
|
43
|
-
-
|
|
51
|
+
|
|
52
|
+
- Showing the agents the body. That tests the body, not the label.
|
|
53
|
+
- Running agents one after another; later ones anchor on earlier answers.
|
|
54
|
+
- Changing one word after a FAIL. A FAIL means the concept behind the label is off; rethink it.
|
|
55
|
+
- Accepting "mostly right". For a one-sided result, either fix the label or split the rule so each part has a label that fits.
|
|
@@ -1,119 +1,110 @@
|
|
|
1
|
-
// audit-panel.js - Workflow template for the multi-architect skill audit.
|
|
2
|
-
//
|
|
3
|
-
// Run with the Workflow tool. Edit the CONFIG block for your target, then launch.
|
|
4
|
-
// Returns { remaining, clearsBar, verdict }. If clearsBar is false, apply the
|
|
5
|
-
// `remaining` blockers/majors, then RE-RUN A FRESH copy (do NOT resume - agents
|
|
6
|
-
// must read the current files). Loop until clearsBar. See ../audit.md.
|
|
7
|
-
|
|
8
1
|
export const meta = {
|
|
9
|
-
name:
|
|
10
|
-
description:
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
const TARGET = `
|
|
16
|
-
- <abs path to SKILL.md>
|
|
17
|
-
- <abs path to each bundled file / script>
|
|
18
|
-
`;
|
|
19
|
-
const RUBRIC = `<abs path to reference/skill-creator/SKILL.md> + checklist.md`;
|
|
20
|
-
const EXEMPLAR = `<abs path to a known-good skill of the same layer>`;
|
|
21
|
-
const GROUND_TRUTH = `<abs path to the real source the skill describes (e.g. the iOS repo), with key facts>`;
|
|
22
|
-
// Pick a lens set from audit.md (general or prompt-engineering). [label, brief] pairs:
|
|
23
|
-
const LENSES = [
|
|
24
|
-
[
|
|
25
|
-
"discovery",
|
|
26
|
-
"description-as-discovery: what+when+trigger words+example; would it activate on a real task?",
|
|
27
|
-
],
|
|
28
|
-
[
|
|
29
|
-
"disambiguation",
|
|
30
|
-
"distinct from sibling skills; a reader can tell when to pick this vs a neighbor.",
|
|
31
|
-
],
|
|
32
|
-
["triggers", "real user phrasings + synonyms are covered."],
|
|
33
|
-
["clarity", "instructions unambiguous, ordered, single-interpretation; no vague directives."],
|
|
34
|
-
[
|
|
35
|
-
"layer",
|
|
36
|
-
"degrees of freedom match the layer (reference=descriptive / workflow=spine / tool=thin-index).",
|
|
2
|
+
name: 'skill-audit-panel',
|
|
3
|
+
description: 'Runs an independent, multi-lens review of a single plugin skill, scored on the skill-creator rubric',
|
|
4
|
+
whenToUse: 'Gate before shipping a skill that is new or heavily reworked; start a new run once each round of fixes lands',
|
|
5
|
+
phases: [
|
|
6
|
+
{ title: 'Audit', detail: 'one read-only reviewer per lens' },
|
|
7
|
+
{ title: 'Synthesis', detail: 'dedup, verify, classify, decide clearsBar' },
|
|
37
8
|
],
|
|
38
|
-
|
|
39
|
-
["tokens", "every line load-bearing; nothing restates what the model already knows."],
|
|
40
|
-
["consistency", "terminology, naming, paths, cross-refs, examples all agree."],
|
|
41
|
-
[
|
|
42
|
-
"misfire",
|
|
43
|
-
"would the model over-apply, skip a gate, hallucinate a path, or follow it to a wrong output?",
|
|
44
|
-
],
|
|
45
|
-
[
|
|
46
|
-
"redteam",
|
|
47
|
-
"adversarially make it not trigger / trigger wrongly / produce broken output; hunt overclaims (says X, script does not).",
|
|
48
|
-
],
|
|
49
|
-
];
|
|
50
|
-
// ─────────────────────────────────────────────────────────────────────────────
|
|
9
|
+
}
|
|
51
10
|
|
|
52
|
-
const
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
11
|
+
const TARGET = [
|
|
12
|
+
'/abs/path/to/plugins/<plugin>/skills/<layer>/<skill>/SKILL.md',
|
|
13
|
+
'/abs/path/to/plugins/<plugin>/skills/<layer>/<skill>/reference.md',
|
|
14
|
+
]
|
|
15
|
+
const RUBRIC = '/abs/path/to/plugins/ai-common-toolkit/skills/knowledge/skill-creator/SKILL.md and checklist.md beside it'
|
|
16
|
+
const EXEMPLAR = '/abs/path/to/an/exemplar/skill/in/this/layer/SKILL.md'
|
|
17
|
+
const GROUND_TRUTH = '/abs/path/to/source/code/described/by/the/skill (name the key facts worth checking)'
|
|
56
18
|
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
19
|
+
const LENSES = [
|
|
20
|
+
{ label: 'discovery', brief: 'Does the description alone get the skill loaded? What + when + trigger words + a sample trigger, third person, platform named.' },
|
|
21
|
+
{ label: 'disambiguation', brief: 'Could a router confuse this skill with a sibling? Are the boundaries and hand-offs explicit in the description and header?' },
|
|
22
|
+
{ label: 'triggers', brief: 'Which real phrasings and synonyms a user would type are missing from the description?' },
|
|
23
|
+
{ label: 'clarity', brief: 'Is every instruction unambiguous and actionable, with no step that two readers would execute differently?' },
|
|
24
|
+
{ label: 'layer', brief: 'Does the content match its declared layer (reference descriptive, workflow imperative with a verification gate, tool a thin index over a script)?' },
|
|
25
|
+
{ label: 'leanness', brief: 'Is SKILL.md an index of about one page, with detail one level deep in bundled files and deterministic code in scripts?' },
|
|
26
|
+
{ label: 'tokens', brief: 'Which lines fail the cut test because the model would act correctly without them?' },
|
|
27
|
+
{ label: 'consistency', brief: 'Do terms, names, paths, cross-references and examples agree with each other and with the files that exist?' },
|
|
28
|
+
{ label: 'misfire', brief: 'Where would the model over-apply the skill, skip its gate, invent a path, or produce the wrong output?' },
|
|
29
|
+
{ label: 'redteam', brief: 'Try to make it fail to trigger, trigger wrongly, or break its output; hunt for prose that claims what the scripts do not do.' },
|
|
30
|
+
]
|
|
64
31
|
|
|
65
32
|
const FINDING = {
|
|
66
|
-
type:
|
|
33
|
+
type: 'object',
|
|
67
34
|
additionalProperties: false,
|
|
35
|
+
required: ['severity', 'location', 'problem', 'fix'],
|
|
68
36
|
properties: {
|
|
69
|
-
severity: {
|
|
70
|
-
location: { type:
|
|
71
|
-
problem: { type:
|
|
72
|
-
fix: { type:
|
|
37
|
+
severity: { enum: ['blocker', 'major', 'minor', 'nit'], type: 'string' },
|
|
38
|
+
location: { type: 'string', description: 'file:line or file#heading' },
|
|
39
|
+
problem: { type: 'string' },
|
|
40
|
+
fix: { type: 'string', description: 'a concrete change, not advice' },
|
|
73
41
|
},
|
|
74
|
-
|
|
75
|
-
};
|
|
76
|
-
const LENS_SCHEMA = {
|
|
77
|
-
type: "object",
|
|
78
|
-
additionalProperties: false,
|
|
79
|
-
properties: { lens: { type: "string" }, findings: { type: "array", items: FINDING } },
|
|
80
|
-
required: ["lens", "findings"],
|
|
81
|
-
};
|
|
42
|
+
}
|
|
82
43
|
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
}),
|
|
93
|
-
),
|
|
94
|
-
);
|
|
95
|
-
const all = results.filter(Boolean).flatMap((r) => r.findings.map((f) => ({ ...f, lens: r.lens })));
|
|
44
|
+
const LENS_RESULT = {
|
|
45
|
+
type: 'object',
|
|
46
|
+
additionalProperties: false,
|
|
47
|
+
required: ['lens', 'findings'],
|
|
48
|
+
properties: {
|
|
49
|
+
lens: { type: 'string' },
|
|
50
|
+
findings: { type: 'array', items: FINDING },
|
|
51
|
+
},
|
|
52
|
+
}
|
|
96
53
|
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
type: "object",
|
|
54
|
+
const SYNTHESIS = {
|
|
55
|
+
type: 'object',
|
|
100
56
|
additionalProperties: false,
|
|
57
|
+
required: ['remaining', 'clearsBar', 'verdict'],
|
|
101
58
|
properties: {
|
|
102
59
|
remaining: {
|
|
103
|
-
type:
|
|
60
|
+
type: 'array',
|
|
104
61
|
items: {
|
|
105
|
-
type:
|
|
62
|
+
type: 'object',
|
|
106
63
|
additionalProperties: false,
|
|
107
|
-
|
|
108
|
-
|
|
64
|
+
required: ['severity', 'location', 'problem', 'fix', 'lenses'],
|
|
65
|
+
properties: { lenses: { type: 'string', description: 'every lens that reported it, comma-separated' }, ...FINDING.properties },
|
|
109
66
|
},
|
|
110
67
|
},
|
|
111
|
-
clearsBar: { type:
|
|
112
|
-
verdict: { type:
|
|
68
|
+
clearsBar: { type: 'boolean' },
|
|
69
|
+
verdict: { type: 'string', description: 'one paragraph' },
|
|
113
70
|
},
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
const PREAMBLE = [
|
|
74
|
+
'You are one reviewer on an independent audit panel for a plugin skill. Stay read-only: do not edit any file.',
|
|
75
|
+
`Read the current state of every target file now; do not rely on anything you saw earlier:\n- ${TARGET.join('\n- ')}`,
|
|
76
|
+
`Rubric: ${RUBRIC}`,
|
|
77
|
+
`Exemplar of the same layer: ${EXEMPLAR}`,
|
|
78
|
+
`Ground truth the skill describes: ${GROUND_TRUTH}`,
|
|
79
|
+
'Report only genuine findings for your lens that you can reproduce by pointing at a line. Check each one against the rubric and the ground truth before reporting it.',
|
|
80
|
+
'Each finding needs a severity (blocker, major, minor, nit), a location, the problem, and a concrete fix. If your lens finds nothing, return an empty findings array.',
|
|
81
|
+
].join('\n\n')
|
|
82
|
+
|
|
83
|
+
phase('Audit')
|
|
84
|
+
const lensOutputs = await parallel(LENSES.map((lensDef) => () =>
|
|
85
|
+
agent(`${PREAMBLE}\n\nYour lens: ${lensDef.label}. ${lensDef.brief}\nSet "lens" to "${lensDef.label}".`, {
|
|
86
|
+
label: `audit:${lensDef.label}`,
|
|
87
|
+
phase: 'Audit',
|
|
88
|
+
schema: LENS_RESULT,
|
|
89
|
+
})))
|
|
90
|
+
|
|
91
|
+
const raw = lensOutputs
|
|
92
|
+
.filter(Boolean)
|
|
93
|
+
.flatMap((out) => out.findings.map((item) => ({ ...item, lens: out.lens })))
|
|
94
|
+
log(`${raw.length} raw finding(s) from ${lensOutputs.filter(Boolean).length}/${LENSES.length} lenses`)
|
|
95
|
+
|
|
96
|
+
phase('Synthesis')
|
|
97
|
+
const synthesis = await agent([
|
|
98
|
+
'You synthesize an audit panel\'s findings for a plugin skill. Stay read-only.',
|
|
99
|
+
`Target files:\n- ${TARGET.join('\n- ')}`,
|
|
100
|
+
`Rubric: ${RUBRIC}`,
|
|
101
|
+
`Ground truth: ${GROUND_TRUTH}`,
|
|
102
|
+
`Raw findings (JSON):\n${JSON.stringify(raw, null, 2)}`,
|
|
103
|
+
'Merge duplicates and record every lens that raised each merged item in "lenses".',
|
|
104
|
+
'Open the cited file for each finding and drop any you cannot reproduce there.',
|
|
105
|
+
'Classify what survives as blocker, major, minor or nit.',
|
|
106
|
+
'Set clearsBar to true only when no blocker and no major remains.',
|
|
107
|
+
'Write a one-paragraph verdict.',
|
|
108
|
+
].join('\n\n'), { label: 'synthesis', phase: 'Synthesis', schema: SYNTHESIS })
|
|
109
|
+
|
|
110
|
+
return synthesis
|