@heretek-ai/epistemic-swarm 0.6.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +20 -0
- package/.claude-plugin/plugin.json +4 -2
- package/MARKETPLACE.md +28 -0
- package/install.sh +1 -1
- package/package.json +1 -1
- package/plugins/darkharvest/.claude-plugin/plugin.json +15 -0
- package/plugins/darkharvest/agents/harvest-proponent.md +22 -0
- package/plugins/darkharvest/agents/harvest-redteam.md +20 -0
- package/plugins/darkharvest/evals/teardown-verdict/graders/license-line.md +6 -0
- package/plugins/darkharvest/evals/teardown-verdict/graders/skill-fired.md +5 -0
- package/plugins/darkharvest/evals/teardown-verdict/prompt.md +6 -0
- package/plugins/darkharvest/skills/darkharvest/SKILL.md +70 -0
- package/plugins/darkharvest/skills/darkharvest/scripts/harvest.py +252 -0
- package/plugins/factory/.claude-plugin/plugin.json +15 -0
- package/plugins/factory/agents/factory-manager.md +22 -0
- package/plugins/factory/agents/programmer.md +16 -0
- package/plugins/factory/agents/qa-adversarial.md +17 -0
- package/plugins/factory/agents/qa-functional.md +17 -0
- package/plugins/factory/evals/gate-halt/graders/gates-first.md +6 -0
- package/plugins/factory/evals/gate-halt/graders/skill-fired.md +5 -0
- package/plugins/factory/evals/gate-halt/prompt.md +6 -0
- package/plugins/factory/skills/factory/SKILL.md +51 -0
- package/plugins/factory/skills/factory/scripts/factory.py +212 -0
- package/runner/__pycache__/__init__.cpython-311.pyc +0 -0
- package/runner/__pycache__/auctioneer.cpython-311.pyc +0 -0
- package/runner/__pycache__/auditor_engine.cpython-311.pyc +0 -0
- package/runner/__pycache__/claim_store.cpython-311.pyc +0 -0
- package/runner/__pycache__/claim_witness.cpython-311.pyc +0 -0
- package/runner/__pycache__/living_dossiers.cpython-311.pyc +0 -0
- package/runner/__pycache__/mcp_protocol.cpython-311.pyc +0 -0
- package/runner/__pycache__/mcp_server.cpython-311.pyc +0 -0
- package/runner/__pycache__/path_safety.cpython-311.pyc +0 -0
- package/runner/__pycache__/pcrb.cpython-311.pyc +0 -0
- package/runner/__pycache__/pcrb_verify.cpython-311.pyc +0 -0
- package/runner/__pycache__/refinement.cpython-311.pyc +0 -0
- package/runner/__pycache__/research_swarm.cpython-311.pyc +0 -0
- package/runner/__pycache__/state_machine.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_auction_order.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_bet1_spike.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claim_store.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claim_witness.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claude_plugin.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_domain_packs.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_factory.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_fleet_seam.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_living_dossiers.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_mcp_server.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_opencode_ux.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_pcrb.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_refinement.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_swarm.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_sweep_regressions.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_webcache.cpython-311.pyc +0 -0
- package/runner/tests/test_claude_plugin.py +170 -0
- package/scripts/__pycache__/bet1_advisory_spike.cpython-311.pyc +0 -0
- package/scripts/__pycache__/build_adapters.cpython-311.pyc +0 -0
- package/scripts/__pycache__/divergence_experiment.cpython-311.pyc +0 -0
- package/scripts/build_adapters.py +69 -0
- package/skills/epistemic_search/scripts/__pycache__/search.cpython-311.pyc +0 -0
- package/skills/research_cache/__pycache__/__init__.cpython-311.pyc +0 -0
- package/skills/research_cache/__pycache__/hasher.cpython-311.pyc +0 -0
- package/skills/swarm_config/__pycache__/__init__.cpython-311.pyc +0 -0
- package/skills/swarm_config/__pycache__/configure.cpython-311.pyc +0 -0
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: factory
|
|
3
|
+
description: Coding-factory Manager loop. Use when user invokes /factory or /domainexpansion, or wants grill-gated phased builds with programmer spawns and dual QA. Manager grills until frontier settled, runs brainstorm plus darkharvest swarms per gate, synthesizes .roadmap phases, spawns programmer per phase, and enforces dual-QA retry bounds. Never writes code itself; never bypasses explicit user sign-off (except inside /domainexpansion count).
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Factory — Manager Loop & Domain Expansion
|
|
7
|
+
|
|
8
|
+
## 1. Roles (OpenCode profiles in `config/opencode-snippet.json`)
|
|
9
|
+
|
|
10
|
+
- **manager** (primary): owns gates, grilling, swarm dispatch, roadmap synthesis, tiebreaks.
|
|
11
|
+
Task allowlist: `factory-*`, `programmer`, `qa-a`, `qa-b`, `brainstormer`, `darkharvester` deny `*` otherwise.
|
|
12
|
+
- **programmer** (subagent): implements exactly one phase brief. May call `iumbtems_verify_quote`; may NOT invoke swarms or other programmers.
|
|
13
|
+
- **qa-a / qa-b** (subagents): same model, DIVERGED prompts (functional-correctness vs adversarial edge-case). Read-only plus test execution; never edit.
|
|
14
|
+
- **researcher** = existing `iumbtems_brainstorm` + `iumbtems_darkharvest` swarms (no new profile).
|
|
15
|
+
|
|
16
|
+
## 2. Gate protocol (max 5 swarm cycles per gate)
|
|
17
|
+
|
|
18
|
+
1. Grill until `.factory/frontier.json` settled (grilling skill).
|
|
19
|
+
2. Run brainstorm + darkharvest swarms (mock-first on fixtures).
|
|
20
|
+
3. Manager synthesizes `.roadmap/<phase>/` (GOAL.md + dossier.json).
|
|
21
|
+
4. Explicit user `approve` advances; anything else regrills (cycle counter in `.factory/state.json`).
|
|
22
|
+
|
|
23
|
+
## 3. Phase contract (`.roadmap/<phase>/`)
|
|
24
|
+
|
|
25
|
+
- `GOAL.md`: human-readable goal + acceptance criteria.
|
|
26
|
+
- `dossier.json`: `{phase, goal, evidence:[{hash, quote}], acceptance[], brief, verdict, hashes}`. Every harvest/claim entry needs a SHA-256 source hash or `file://` pointer or it is purged to NEGATIVE_KNOWLEDGE.
|
|
27
|
+
- Programmer receives the phase dossier (Manager chooses freeform vs strict brief but MUST cite phase hashes).
|
|
28
|
+
|
|
29
|
+
## 4. QA protocol (3 retries, then escalate)
|
|
30
|
+
|
|
31
|
+
- Both QA seats run per phase; disagreements go to manager tiebreak.
|
|
32
|
+
- `skills/factory/scripts/factory.py` tracks `qa_retries` per phase in `.factory/state.json`. On 3rd rejection: halt phase, return to manager with both QA reports (retry loop per plan; manager may regrill scope or escalate to user).
|
|
33
|
+
- QA verdicts: `pass | fail(reason) | conditional(note)`.
|
|
34
|
+
|
|
35
|
+
## 5. Domain expansion (`/domainexpansion <n>`)
|
|
36
|
+
|
|
37
|
+
- Bypasses per-loop gates; stops on count OR `.factory/STOP` file OR user kill, whichever first.
|
|
38
|
+
- Each loop: agents propose direction → quick swarm check → implement → QA → next.
|
|
39
|
+
- `factory.py` enforces: refuse `n < 1`, cap `n` at `--max-loops` default 10, check STOP file before every loop.
|
|
40
|
+
|
|
41
|
+
## 6. State layout (three dirs, distinct jobs)
|
|
42
|
+
|
|
43
|
+
- `.factory/`: run state (`state.json`, `frontier.json`, `STOP` kill-file). Gitignored runtime state.
|
|
44
|
+
- `.roadmap/`: output (phase dirs). Committed.
|
|
45
|
+
- `.research/`: evidence (swarm dossiers, source cache). Gitignored (existing rule).
|
|
46
|
+
|
|
47
|
+
## 7. Anti-patterns
|
|
48
|
+
|
|
49
|
+
- No code writes by manager; no swarm invocation by programmer; no edits by QA.
|
|
50
|
+
- No gate bypass outside `/domainexpansion`; no uncapped loops.
|
|
51
|
+
- No VERIFIED claims without hashes, even in expansion proposals.
|
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
Factory run-state helper: phase dossiers, QA retry bounds, expansion loop guard.
|
|
4
|
+
|
|
5
|
+
All state lives under .factory/ (gitignored runtime state). Phase output goes
|
|
6
|
+
to .roadmap/<phase>/. Evidence stays in .research/. Read-only w.r.t. repo code.
|
|
7
|
+
|
|
8
|
+
Usage:
|
|
9
|
+
python3 skills/factory/scripts/factory.py init --run <name>
|
|
10
|
+
python3 skills/factory/scripts/factory.py phase-add --run <name> --phase 01-auth --goal "..." --accept "..."
|
|
11
|
+
python3 skills/factory/scripts/factory.py qa-record --run <name> --phase 01-auth --seat qa-a --verdict fail --reason "..."
|
|
12
|
+
python3 skills/factory/scripts/factory.py expansion --run <name> --loops 10 [--max-loops 10]
|
|
13
|
+
python3 skills/factory/scripts/factory.py stop --run <name> # write STOP kill-file
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
import argparse
|
|
17
|
+
import json
|
|
18
|
+
import sys
|
|
19
|
+
from datetime import datetime, timezone
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
|
|
22
|
+
PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent.parent
|
|
23
|
+
FACTORY_DIR = PROJECT_ROOT / ".factory"
|
|
24
|
+
ROADMAP_DIR = PROJECT_ROOT / ".roadmap"
|
|
25
|
+
|
|
26
|
+
MAX_QA_RETRIES = 3
|
|
27
|
+
MAX_EXPANSION_LOOPS = 10
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def now():
|
|
31
|
+
return datetime.now(timezone.utc).isoformat()
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def run_dir(run):
|
|
35
|
+
return FACTORY_DIR / run
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def load_state(run):
|
|
39
|
+
p = run_dir(run) / "state.json"
|
|
40
|
+
if not p.exists():
|
|
41
|
+
raise SystemExit(f"No factory run '{run}'. Run `factory.py init` first.")
|
|
42
|
+
return json.loads(p.read_text(encoding="utf-8"))
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def save_state(run, state):
|
|
46
|
+
p = run_dir(run) / "state.json"
|
|
47
|
+
p.write_text(json.dumps(state, indent=2), encoding="utf-8")
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def cmd_init(args):
|
|
51
|
+
d = run_dir(args.run)
|
|
52
|
+
d.mkdir(parents=True, exist_ok=True)
|
|
53
|
+
state = {
|
|
54
|
+
"run": args.run,
|
|
55
|
+
"created_at": now(),
|
|
56
|
+
"gate_cycles": 0,
|
|
57
|
+
"phases": {},
|
|
58
|
+
"expansion": {"loops_done": 0, "loops_planned": 0},
|
|
59
|
+
}
|
|
60
|
+
save_state(args.run, state)
|
|
61
|
+
print(f"✅ Factory run '{args.run}' initialised at {d}")
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def cmd_phase_add(args):
|
|
65
|
+
state = load_state(args.run)
|
|
66
|
+
if args.phase in state["phases"]:
|
|
67
|
+
raise SystemExit(f"Phase '{args.phase}' already exists.")
|
|
68
|
+
state["phases"][args.phase] = {
|
|
69
|
+
"goal": args.goal,
|
|
70
|
+
"acceptance": [a.strip() for a in args.accept.split(";") if a.strip()],
|
|
71
|
+
"status": "briefed",
|
|
72
|
+
"qa_retries": 0,
|
|
73
|
+
"qa_reports": [],
|
|
74
|
+
}
|
|
75
|
+
save_state(args.run, state)
|
|
76
|
+
# Phase output: GOAL.md + dossier.json skeleton (evidence filled by manager).
|
|
77
|
+
phase_dir = ROADMAP_DIR / args.phase
|
|
78
|
+
phase_dir.mkdir(parents=True, exist_ok=True)
|
|
79
|
+
(phase_dir / "GOAL.md").write_text(
|
|
80
|
+
f"# {args.phase}: {args.goal}\n\n## Acceptance\n"
|
|
81
|
+
+ "".join(f"- [ ] {a}\n" for a in state["phases"][args.phase]["acceptance"]),
|
|
82
|
+
encoding="utf-8",
|
|
83
|
+
)
|
|
84
|
+
(phase_dir / "dossier.json").write_text(
|
|
85
|
+
json.dumps(
|
|
86
|
+
{
|
|
87
|
+
"phase": args.phase,
|
|
88
|
+
"goal": args.goal,
|
|
89
|
+
"evidence": [],
|
|
90
|
+
"acceptance": state["phases"][args.phase]["acceptance"],
|
|
91
|
+
"brief": "",
|
|
92
|
+
"verdict": "briefed",
|
|
93
|
+
"hashes": [],
|
|
94
|
+
},
|
|
95
|
+
indent=2,
|
|
96
|
+
),
|
|
97
|
+
encoding="utf-8",
|
|
98
|
+
)
|
|
99
|
+
print(f"✅ Phase '{args.phase}' briefed → {phase_dir}")
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def cmd_qa_record(args):
|
|
103
|
+
state = load_state(args.run)
|
|
104
|
+
phase = state["phases"].get(args.phase)
|
|
105
|
+
if phase is None:
|
|
106
|
+
raise SystemExit(f"Unknown phase '{args.phase}'.")
|
|
107
|
+
if args.verdict not in ("pass", "fail", "conditional"):
|
|
108
|
+
raise SystemExit("verdict must be pass|fail|conditional.")
|
|
109
|
+
phase["qa_reports"].append(
|
|
110
|
+
{
|
|
111
|
+
"seat": args.seat,
|
|
112
|
+
"verdict": args.verdict,
|
|
113
|
+
"reason": args.reason or "",
|
|
114
|
+
"at": now(),
|
|
115
|
+
}
|
|
116
|
+
)
|
|
117
|
+
if args.verdict == "fail":
|
|
118
|
+
phase["qa_retries"] += 1
|
|
119
|
+
if phase["qa_retries"] >= MAX_QA_RETRIES:
|
|
120
|
+
phase["status"] = "escalated"
|
|
121
|
+
save_state(args.run, state)
|
|
122
|
+
print(
|
|
123
|
+
f"🛑 Phase '{args.phase}' ESCALATED after "
|
|
124
|
+
f"{MAX_QA_RETRIES} QA failures — back to manager."
|
|
125
|
+
)
|
|
126
|
+
return 2
|
|
127
|
+
phase["status"] = "retrying"
|
|
128
|
+
elif args.verdict == "pass":
|
|
129
|
+
phase["status"] = "signed-off"
|
|
130
|
+
else:
|
|
131
|
+
phase["status"] = "conditional"
|
|
132
|
+
save_state(args.run, state)
|
|
133
|
+
print(
|
|
134
|
+
f"✅ QA recorded: {args.phase} [{args.seat}] → {args.verdict} "
|
|
135
|
+
f"(status={phase['status']}, retries={phase['qa_retries']})"
|
|
136
|
+
)
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def stop_requested(run):
|
|
140
|
+
return (run_dir(run) / "STOP").exists()
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def cmd_expansion(args):
|
|
144
|
+
state = load_state(args.run)
|
|
145
|
+
n = args.loops
|
|
146
|
+
if n < 1:
|
|
147
|
+
raise SystemExit("loops must be >= 1.")
|
|
148
|
+
cap = args.max_loops or MAX_EXPANSION_LOOPS
|
|
149
|
+
n = min(n, cap)
|
|
150
|
+
state["expansion"]["loops_planned"] = n
|
|
151
|
+
save_state(args.run, state)
|
|
152
|
+
done = 0
|
|
153
|
+
for i in range(1, n + 1):
|
|
154
|
+
if stop_requested(args.run):
|
|
155
|
+
print(f"🛑 STOP file present — halting after {done}/{n} loops.")
|
|
156
|
+
break
|
|
157
|
+
done += 1
|
|
158
|
+
print(
|
|
159
|
+
f"🔁 Expansion loop {done}/{n}: agents propose → swarm check → "
|
|
160
|
+
f"implement → QA (Manager drives each step)."
|
|
161
|
+
)
|
|
162
|
+
state = load_state(args.run)
|
|
163
|
+
state["expansion"]["loops_done"] += done
|
|
164
|
+
save_state(args.run, state)
|
|
165
|
+
print(f"✅ Expansion finished {done} loop(s).")
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def cmd_stop(args):
|
|
169
|
+
(run_dir(args.run)).mkdir(parents=True, exist_ok=True)
|
|
170
|
+
(run_dir(args.run) / "STOP").write_text(
|
|
171
|
+
f"stop requested at {now()}\n", encoding="utf-8"
|
|
172
|
+
)
|
|
173
|
+
print(f"🛑 STOP file written for run '{args.run}'.")
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def main():
|
|
177
|
+
ap = argparse.ArgumentParser(description="Factory run-state helper")
|
|
178
|
+
sub = ap.add_subparsers(dest="command", required=True)
|
|
179
|
+
|
|
180
|
+
p = sub.add_parser("init")
|
|
181
|
+
p.add_argument("--run", required=True)
|
|
182
|
+
p = sub.add_parser("phase-add")
|
|
183
|
+
p.add_argument("--run", required=True)
|
|
184
|
+
p.add_argument("--phase", required=True)
|
|
185
|
+
p.add_argument("--goal", required=True)
|
|
186
|
+
p.add_argument("--accept", default="")
|
|
187
|
+
p = sub.add_parser("qa-record")
|
|
188
|
+
p.add_argument("--run", required=True)
|
|
189
|
+
p.add_argument("--phase", required=True)
|
|
190
|
+
p.add_argument("--seat", required=True)
|
|
191
|
+
p.add_argument("--verdict", required=True)
|
|
192
|
+
p.add_argument("--reason", default="")
|
|
193
|
+
p = sub.add_parser("expansion")
|
|
194
|
+
p.add_argument("--run", required=True)
|
|
195
|
+
p.add_argument("--loops", type=int, required=True)
|
|
196
|
+
p.add_argument("--max-loops", type=int, default=MAX_EXPANSION_LOOPS)
|
|
197
|
+
p = sub.add_parser("stop")
|
|
198
|
+
p.add_argument("--run", required=True)
|
|
199
|
+
|
|
200
|
+
args = ap.parse_args()
|
|
201
|
+
code = {
|
|
202
|
+
"init": cmd_init,
|
|
203
|
+
"phase-add": cmd_phase_add,
|
|
204
|
+
"qa-record": cmd_qa_record,
|
|
205
|
+
"expansion": cmd_expansion,
|
|
206
|
+
"stop": cmd_stop,
|
|
207
|
+
}[args.command](args)
|
|
208
|
+
sys.exit(code if isinstance(code, int) else 0)
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
if __name__ == "__main__":
|
|
212
|
+
main()
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Tests for Claude Code plugin surfaces: manifests, agents, modular sync, evals.
|
|
3
|
+
|
|
4
|
+
Per https://code.claude.com/docs/en/plugins/create (manifest + layout),
|
|
5
|
+
/components (skills/agents/hooks/MCP), and /plugin-evals (suite layout).
|
|
6
|
+
Behavioral eval runs are billable and stay in CI (plugin-evals.yml); here we
|
|
7
|
+
assert suite structure only.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
import unittest
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent
|
|
15
|
+
if str(PROJECT_ROOT) not in __import__("sys").path:
|
|
16
|
+
__import__("sys").path.insert(0, str(PROJECT_ROOT))
|
|
17
|
+
|
|
18
|
+
MODULAR = {
|
|
19
|
+
"socratic-grilling": "grilling",
|
|
20
|
+
"research-cache": "research_cache",
|
|
21
|
+
"darkharvest": "darkharvest",
|
|
22
|
+
"factory": "factory",
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def parse_frontmatter(path):
|
|
27
|
+
text = Path(path).read_text(encoding="utf-8")
|
|
28
|
+
if not text.startswith("---"):
|
|
29
|
+
return {}
|
|
30
|
+
end = text.find("---", 3)
|
|
31
|
+
if end < 0:
|
|
32
|
+
return {}
|
|
33
|
+
data = {}
|
|
34
|
+
for line in text[3:end].strip().splitlines():
|
|
35
|
+
if ":" in line:
|
|
36
|
+
k, v = line.split(":", 1)
|
|
37
|
+
data[k.strip()] = v.strip()
|
|
38
|
+
return data
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class TestClaudeManifests(unittest.TestCase):
|
|
42
|
+
def test_root_manifest_lists_all_nine_skills(self):
|
|
43
|
+
manifest = json.loads(
|
|
44
|
+
(PROJECT_ROOT / ".claude-plugin" / "plugin.json").read_text()
|
|
45
|
+
)
|
|
46
|
+
self.assertEqual(manifest["name"], "epistemic-swarm")
|
|
47
|
+
for skill in (
|
|
48
|
+
"grilling",
|
|
49
|
+
"research_cache",
|
|
50
|
+
"epistemic_search",
|
|
51
|
+
"swarm_config",
|
|
52
|
+
"code_audit",
|
|
53
|
+
"oss_scout",
|
|
54
|
+
"brainstorming",
|
|
55
|
+
"darkharvest",
|
|
56
|
+
"factory",
|
|
57
|
+
):
|
|
58
|
+
self.assertIn(f"./skills/{skill}", manifest["skills"])
|
|
59
|
+
|
|
60
|
+
def test_marketplace_entries_resolve(self):
|
|
61
|
+
marketplace = json.loads(
|
|
62
|
+
(PROJECT_ROOT / ".claude-plugin" / "marketplace.json").read_text()
|
|
63
|
+
)
|
|
64
|
+
names = [p["name"] for p in marketplace["plugins"]]
|
|
65
|
+
for name in (
|
|
66
|
+
"epistemic-swarm",
|
|
67
|
+
"socratic-grilling",
|
|
68
|
+
"research-cache",
|
|
69
|
+
"darkharvest",
|
|
70
|
+
"factory",
|
|
71
|
+
):
|
|
72
|
+
self.assertIn(name, names)
|
|
73
|
+
for plugin in marketplace["plugins"]:
|
|
74
|
+
if plugin["name"] == "epistemic-swarm":
|
|
75
|
+
continue
|
|
76
|
+
plugin_json = (
|
|
77
|
+
PROJECT_ROOT
|
|
78
|
+
/ plugin["source"].lstrip("./")
|
|
79
|
+
/ ".claude-plugin"
|
|
80
|
+
/ "plugin.json"
|
|
81
|
+
)
|
|
82
|
+
self.assertTrue(plugin_json.is_file(), f"missing {plugin_json}")
|
|
83
|
+
data = json.loads(plugin_json.read_text())
|
|
84
|
+
self.assertEqual(data["name"], plugin["name"])
|
|
85
|
+
|
|
86
|
+
def test_modular_skills_in_sync(self):
|
|
87
|
+
for mod, skill in MODULAR.items():
|
|
88
|
+
src = PROJECT_ROOT / "skills" / skill
|
|
89
|
+
dst = PROJECT_ROOT / "plugins" / mod / "skills" / skill
|
|
90
|
+
self.assertTrue((dst / "SKILL.md").is_file(), f"missing {dst}")
|
|
91
|
+
src_files = {
|
|
92
|
+
p.relative_to(src)
|
|
93
|
+
for p in src.rglob("*")
|
|
94
|
+
if p.is_file() and "__pycache__" not in p.parts
|
|
95
|
+
}
|
|
96
|
+
dst_files = {
|
|
97
|
+
p.relative_to(dst)
|
|
98
|
+
for p in dst.rglob("*")
|
|
99
|
+
if p.is_file() and "__pycache__" not in p.parts
|
|
100
|
+
}
|
|
101
|
+
self.assertEqual(src_files, dst_files, f"drift in {mod}")
|
|
102
|
+
for rel in src_files:
|
|
103
|
+
self.assertEqual(
|
|
104
|
+
(src / rel).read_bytes(),
|
|
105
|
+
(dst / rel).read_bytes(),
|
|
106
|
+
f"content drift: {mod}/{rel}",
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
class TestClaudeAgents(unittest.TestCase):
|
|
111
|
+
AGENTS = {
|
|
112
|
+
"agents/alpha-thesis.md": "alpha-thesis",
|
|
113
|
+
"agents/beta-antithesis.md": "beta-antithesis",
|
|
114
|
+
"agents/epistemic-auditor.md": "epistemic-auditor",
|
|
115
|
+
"plugins/darkharvest/agents/harvest-proponent.md": "harvest-proponent",
|
|
116
|
+
"plugins/darkharvest/agents/harvest-redteam.md": "harvest-redteam",
|
|
117
|
+
"plugins/factory/agents/factory-manager.md": "factory-manager",
|
|
118
|
+
"plugins/factory/agents/programmer.md": "programmer",
|
|
119
|
+
"plugins/factory/agents/qa-functional.md": "qa-functional",
|
|
120
|
+
"plugins/factory/agents/qa-adversarial.md": "qa-adversarial",
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
def test_agent_frontmatter(self):
|
|
124
|
+
for rel, expected_name in self.AGENTS.items():
|
|
125
|
+
fm = parse_frontmatter(PROJECT_ROOT / rel)
|
|
126
|
+
self.assertEqual(fm.get("name"), expected_name, rel)
|
|
127
|
+
self.assertTrue(fm.get("description"), f"no description: {rel}")
|
|
128
|
+
self.assertIn("model", fm, f"no model: {rel}")
|
|
129
|
+
for banned in ("permissionMode", "hooks", "mcpServers", "initialPrompt"):
|
|
130
|
+
self.assertNotIn(banned, fm, f"banned field in {rel}")
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
class TestClaudeEvals(unittest.TestCase):
|
|
134
|
+
CASES = {
|
|
135
|
+
"evals/grill-fires": "grilling",
|
|
136
|
+
"evals/darkharvest-fires": "darkharvest",
|
|
137
|
+
"evals/factory-gate": "factory",
|
|
138
|
+
"plugins/darkharvest/evals/teardown-verdict": "darkharvest",
|
|
139
|
+
"plugins/factory/evals/gate-halt": "factory",
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
def test_eval_suite_structure(self):
|
|
143
|
+
for case, skill in self.CASES.items():
|
|
144
|
+
prompt = PROJECT_ROOT / case / "prompt.md"
|
|
145
|
+
self.assertTrue(prompt.is_file(), f"missing {prompt}")
|
|
146
|
+
body = prompt.read_text(encoding="utf-8")
|
|
147
|
+
self.assertIn("allowed_tools", body)
|
|
148
|
+
graders = sorted((PROJECT_ROOT / case / "graders").glob("*.md"))
|
|
149
|
+
self.assertTrue(graders, f"no graders in {case}")
|
|
150
|
+
kinds = set()
|
|
151
|
+
for g in graders:
|
|
152
|
+
fm = parse_frontmatter(g)
|
|
153
|
+
self.assertIn(
|
|
154
|
+
fm.get("type"),
|
|
155
|
+
(
|
|
156
|
+
"tool_used",
|
|
157
|
+
"llm",
|
|
158
|
+
"regex",
|
|
159
|
+
"tool_order",
|
|
160
|
+
"file_exists",
|
|
161
|
+
"baseline",
|
|
162
|
+
),
|
|
163
|
+
f"bad type in {g}",
|
|
164
|
+
)
|
|
165
|
+
kinds.add(fm.get("type"))
|
|
166
|
+
self.assertIn("tool_used", kinds, f"no Skill-fired grader in {case}")
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
if __name__ == "__main__":
|
|
170
|
+
unittest.main()
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
@@ -43,6 +43,17 @@ CANONICAL_SKILLS = [
|
|
|
43
43
|
"factory",
|
|
44
44
|
]
|
|
45
45
|
|
|
46
|
+
# Modular Claude Code plugins mirror canonical skills as full copies (Claude
|
|
47
|
+
# Code loads plugin-local skills/, not the repo skills/ tree). Rule: copies
|
|
48
|
+
# are generated — never hand-edit under plugins/<mod>/skills/. Rebuild with
|
|
49
|
+
# `python3 scripts/build_adapters.py`; `--check` fails CI on drift.
|
|
50
|
+
MODULAR_PLUGINS = {
|
|
51
|
+
"socratic-grilling": "grilling",
|
|
52
|
+
"research-cache": "research_cache",
|
|
53
|
+
"darkharvest": "darkharvest",
|
|
54
|
+
"factory": "factory",
|
|
55
|
+
}
|
|
56
|
+
|
|
46
57
|
# Skill -> the MCP tool(s) that now carry its programmatic surface.
|
|
47
58
|
SKILL_TOOLS = {
|
|
48
59
|
"grilling": ["iumbtems_socratic_frontier"],
|
|
@@ -133,9 +144,66 @@ def build():
|
|
|
133
144
|
if _build_skill_stub(target_base, skill):
|
|
134
145
|
count += 1
|
|
135
146
|
print(f"✅ Built {count} thin skill stubs across {len(TARGETS)} targets.")
|
|
147
|
+
synced = sync_modular_plugins()
|
|
148
|
+
print(f"✅ Synced {synced} modular Claude Code plugin skill copies.")
|
|
136
149
|
return 0
|
|
137
150
|
|
|
138
151
|
|
|
152
|
+
def _iter_modular_files(src: Path, dst: Path):
|
|
153
|
+
"""Yield (relpath, src_bytes|None, dst_bytes|None) for sync/check."""
|
|
154
|
+
src_files = {}
|
|
155
|
+
for p in sorted(src.rglob("*")):
|
|
156
|
+
if p.is_file() and "__pycache__" not in p.parts:
|
|
157
|
+
src_files[p.relative_to(src)] = p
|
|
158
|
+
dst_files = {}
|
|
159
|
+
if dst.exists():
|
|
160
|
+
for p in sorted(dst.rglob("*")):
|
|
161
|
+
if p.is_file() and "__pycache__" not in p.parts:
|
|
162
|
+
dst_files[p.relative_to(dst)] = p
|
|
163
|
+
for rel in sorted(set(src_files) | set(dst_files)):
|
|
164
|
+
s = src_files.get(rel)
|
|
165
|
+
d = dst_files.get(rel)
|
|
166
|
+
yield rel, (s.read_bytes() if s else None), (d.read_bytes() if d else None)
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def sync_modular_plugins() -> int:
|
|
170
|
+
synced = 0
|
|
171
|
+
for mod, skill in MODULAR_PLUGINS.items():
|
|
172
|
+
src = PROJECT_ROOT / "skills" / skill
|
|
173
|
+
dst = PROJECT_ROOT / "plugins" / mod / "skills" / skill
|
|
174
|
+
if not (src / SKILL_FILENAME).exists():
|
|
175
|
+
print(f"[WARN] canonical skill missing: skills/{skill}", file=sys.stderr)
|
|
176
|
+
continue
|
|
177
|
+
if dst.exists():
|
|
178
|
+
shutil.rmtree(dst)
|
|
179
|
+
dst.mkdir(parents=True, exist_ok=True)
|
|
180
|
+
for rel, content, _ in _iter_modular_files(src, dst):
|
|
181
|
+
if content is None:
|
|
182
|
+
continue
|
|
183
|
+
out = dst / rel
|
|
184
|
+
out.parent.mkdir(parents=True, exist_ok=True)
|
|
185
|
+
out.write_bytes(content)
|
|
186
|
+
shutil.copymode(src / rel, out)
|
|
187
|
+
synced += 1
|
|
188
|
+
return synced
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def _check_modular_plugins() -> List[str]:
|
|
192
|
+
errors = []
|
|
193
|
+
for mod, skill in MODULAR_PLUGINS.items():
|
|
194
|
+
src = PROJECT_ROOT / "skills" / skill
|
|
195
|
+
dst = PROJECT_ROOT / "plugins" / mod / "skills" / skill
|
|
196
|
+
if not dst.exists():
|
|
197
|
+
errors.append(f"MISSING modular copy: plugins/{mod}/skills/{skill}/")
|
|
198
|
+
continue
|
|
199
|
+
for rel, s_bytes, d_bytes in _iter_modular_files(src, dst):
|
|
200
|
+
if s_bytes is None:
|
|
201
|
+
errors.append(f"STRAY FILE in plugins/{mod}/skills/{skill}/{rel}")
|
|
202
|
+
elif s_bytes != d_bytes:
|
|
203
|
+
errors.append(f"OUT OF SYNC: plugins/{mod}/skills/{skill}/{rel}")
|
|
204
|
+
return errors
|
|
205
|
+
|
|
206
|
+
|
|
139
207
|
def _check_skill_stub(target_rel: str, skill: str) -> List[str]:
|
|
140
208
|
errors = []
|
|
141
209
|
dst = PROJECT_ROOT / target_rel / skill / SKILL_FILENAME
|
|
@@ -176,6 +244,7 @@ def check():
|
|
|
176
244
|
for target_rel, _mode in TARGETS:
|
|
177
245
|
for skill in CANONICAL_SKILLS:
|
|
178
246
|
errors.extend(_check_skill_stub(target_rel, skill))
|
|
247
|
+
errors.extend(_check_modular_plugins())
|
|
179
248
|
errors.extend(_check_package_json())
|
|
180
249
|
if errors:
|
|
181
250
|
print("❌ Adapter check failed:")
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|