@heretek-ai/epistemic-swarm 0.5.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/skills/factory/SKILL.md +13 -0
- package/.claude-plugin/marketplace.json +20 -0
- package/.claude-plugin/plugin.json +4 -2
- package/.omp/commands/domainexpansion.md +9 -0
- package/.omp/commands/factory.md +9 -0
- package/MARKETPLACE.md +28 -0
- package/README.md +9 -0
- package/bin/cli.js +21 -0
- package/config/opencode-snippet.json +88 -1
- package/install.sh +1 -1
- package/package.json +5 -3
- package/plugins/antigravity/skills/factory/SKILL.md +13 -0
- package/plugins/codex/skills/factory/SKILL.md +13 -0
- package/plugins/darkharvest/.claude-plugin/plugin.json +15 -0
- package/plugins/darkharvest/agents/harvest-proponent.md +22 -0
- package/plugins/darkharvest/agents/harvest-redteam.md +20 -0
- package/plugins/darkharvest/evals/teardown-verdict/graders/license-line.md +6 -0
- package/plugins/darkharvest/evals/teardown-verdict/graders/skill-fired.md +5 -0
- package/plugins/darkharvest/evals/teardown-verdict/prompt.md +6 -0
- package/plugins/darkharvest/skills/darkharvest/SKILL.md +70 -0
- package/plugins/darkharvest/skills/darkharvest/scripts/harvest.py +252 -0
- package/plugins/factory/.claude-plugin/plugin.json +15 -0
- package/plugins/factory/agents/factory-manager.md +22 -0
- package/plugins/factory/agents/programmer.md +16 -0
- package/plugins/factory/agents/qa-adversarial.md +17 -0
- package/plugins/factory/agents/qa-functional.md +17 -0
- package/plugins/factory/evals/gate-halt/graders/gates-first.md +6 -0
- package/plugins/factory/evals/gate-halt/graders/skill-fired.md +5 -0
- package/plugins/factory/evals/gate-halt/prompt.md +6 -0
- package/plugins/factory/skills/factory/SKILL.md +51 -0
- package/plugins/factory/skills/factory/scripts/factory.py +212 -0
- package/plugins/gemini/commands/domainexpansion.toml +7 -0
- package/plugins/gemini/commands/factory.toml +8 -0
- package/plugins/gemini/skills/factory/SKILL.md +13 -0
- package/plugins/opencode/index.js +35 -0
- package/runner/__pycache__/__init__.cpython-311.pyc +0 -0
- package/runner/__pycache__/auctioneer.cpython-311.pyc +0 -0
- package/runner/__pycache__/auditor_engine.cpython-311.pyc +0 -0
- package/runner/__pycache__/claim_store.cpython-311.pyc +0 -0
- package/runner/__pycache__/claim_witness.cpython-311.pyc +0 -0
- package/runner/__pycache__/living_dossiers.cpython-311.pyc +0 -0
- package/runner/__pycache__/mcp_protocol.cpython-311.pyc +0 -0
- package/runner/__pycache__/mcp_server.cpython-311.pyc +0 -0
- package/runner/__pycache__/path_safety.cpython-311.pyc +0 -0
- package/runner/__pycache__/pcrb.cpython-311.pyc +0 -0
- package/runner/__pycache__/pcrb_verify.cpython-311.pyc +0 -0
- package/runner/__pycache__/refinement.cpython-311.pyc +0 -0
- package/runner/__pycache__/research_swarm.cpython-311.pyc +0 -0
- package/runner/__pycache__/state_machine.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_auction_order.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_bet1_spike.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claim_store.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claim_witness.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claude_plugin.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_domain_packs.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_factory.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_fleet_seam.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_living_dossiers.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_mcp_server.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_opencode_ux.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_pcrb.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_refinement.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_swarm.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_sweep_regressions.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_webcache.cpython-311.pyc +0 -0
- package/runner/tests/test_claude_plugin.py +170 -0
- package/runner/tests/test_factory.py +199 -0
- package/runner/tests/test_opencode_ux.py +8 -0
- package/runner/tests/test_swarm.py +2 -0
- package/scripts/__pycache__/bet1_advisory_spike.cpython-311.pyc +0 -0
- package/scripts/__pycache__/build_adapters.cpython-311.pyc +0 -0
- package/scripts/__pycache__/divergence_experiment.cpython-311.pyc +0 -0
- package/scripts/build_adapters.py +76 -0
- package/skills/epistemic_search/scripts/__pycache__/search.cpython-311.pyc +0 -0
- package/skills/factory/SKILL.md +51 -0
- package/skills/factory/scripts/factory.py +212 -0
- package/skills/research_cache/__pycache__/__init__.cpython-311.pyc +0 -0
- package/skills/research_cache/__pycache__/hasher.cpython-311.pyc +0 -0
- package/skills/swarm_config/__pycache__/__init__.cpython-311.pyc +0 -0
- package/skills/swarm_config/__pycache__/configure.cpython-311.pyc +0 -0
|
@@ -0,0 +1,252 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
Darkharvest fetch helper: clone-to-temp + allowlisted scan + SHA-256 cache.
|
|
4
|
+
|
|
5
|
+
Read-only. Never installs dependencies. Never writes outside .research/
|
|
6
|
+
(except the temp clone, which is dropped). Enforces per-repo caps so
|
|
7
|
+
`--max-repos 10 --depth 3` style defaults cannot OOM the host.
|
|
8
|
+
|
|
9
|
+
Usage:
|
|
10
|
+
python3 skills/darkharvest/scripts/harvest.py --repo <url> [--show-context]
|
|
11
|
+
python3 skills/darkharvest/scripts/harvest.py --repo <url> --export <path>
|
|
12
|
+
|
|
13
|
+
Closed / non-cloneable targets: prints a metadata-only stub with a
|
|
14
|
+
[NEGATIVE_KNOWLEDGE] gap note instead of failing.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
import argparse
|
|
18
|
+
import json
|
|
19
|
+
import os
|
|
20
|
+
import re
|
|
21
|
+
import shutil
|
|
22
|
+
import subprocess
|
|
23
|
+
import sys
|
|
24
|
+
import tempfile
|
|
25
|
+
from datetime import datetime, timezone
|
|
26
|
+
from pathlib import Path
|
|
27
|
+
|
|
28
|
+
PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent.parent
|
|
29
|
+
if str(PROJECT_ROOT) not in sys.path:
|
|
30
|
+
sys.path.insert(0, str(PROJECT_ROOT))
|
|
31
|
+
|
|
32
|
+
from skills.research_cache.hasher import SourceHasher # noqa: E402
|
|
33
|
+
|
|
34
|
+
SKIP_DIRS = {
|
|
35
|
+
".git",
|
|
36
|
+
"node_modules",
|
|
37
|
+
"__pycache__",
|
|
38
|
+
".venv",
|
|
39
|
+
"venv",
|
|
40
|
+
"dist",
|
|
41
|
+
"build",
|
|
42
|
+
"target",
|
|
43
|
+
".next",
|
|
44
|
+
".research",
|
|
45
|
+
"coverage",
|
|
46
|
+
}
|
|
47
|
+
MANIFEST_FILES = [
|
|
48
|
+
"package.json",
|
|
49
|
+
"pyproject.toml",
|
|
50
|
+
"Cargo.toml",
|
|
51
|
+
"go.mod",
|
|
52
|
+
"requirements.txt",
|
|
53
|
+
"LICENSE",
|
|
54
|
+
"LICENSE.md",
|
|
55
|
+
"README.md",
|
|
56
|
+
]
|
|
57
|
+
TEXT_EXTS = {".md", ".json", ".toml", ".yaml", ".yml", ".txt", ".py", ".js", ".ts"}
|
|
58
|
+
|
|
59
|
+
DEFAULT_MAX_FILES = 120
|
|
60
|
+
DEFAULT_MAX_BYTES = 400_000
|
|
61
|
+
DEFAULT_TIMEOUT = 120
|
|
62
|
+
|
|
63
|
+
SPDX_MAP = {
|
|
64
|
+
"mit": "MIT",
|
|
65
|
+
"apache-2.0": "Apache-2.0",
|
|
66
|
+
"apache 2.0": "Apache-2.0",
|
|
67
|
+
"bsd-3-clause": "BSD-3-Clause",
|
|
68
|
+
"bsd 3-clause": "BSD-3-Clause",
|
|
69
|
+
"isc": "ISC",
|
|
70
|
+
"gpl": "GPL",
|
|
71
|
+
"agpl": "AGPL",
|
|
72
|
+
"mpl": "MPL",
|
|
73
|
+
"lgpl": "LGPL",
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def sh(cmd, cwd=None, timeout=DEFAULT_TIMEOUT):
|
|
78
|
+
try:
|
|
79
|
+
r = subprocess.run(
|
|
80
|
+
cmd,
|
|
81
|
+
capture_output=True,
|
|
82
|
+
text=True,
|
|
83
|
+
cwd=str(cwd or PROJECT_ROOT),
|
|
84
|
+
timeout=timeout,
|
|
85
|
+
)
|
|
86
|
+
return r.returncode, (r.stdout or "").strip()
|
|
87
|
+
except Exception as exc: # noqa: BLE001
|
|
88
|
+
return 1, f"error: {exc}"
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def detect_license(text):
|
|
92
|
+
low = (text or "").lower()
|
|
93
|
+
for key, spdx in SPDX_MAP.items():
|
|
94
|
+
if key in low:
|
|
95
|
+
return spdx
|
|
96
|
+
return "UNKNOWN"
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def scan_tree(root, max_files=DEFAULT_MAX_FILES, max_bytes=DEFAULT_MAX_BYTES):
|
|
100
|
+
entries = []
|
|
101
|
+
total_bytes = 0
|
|
102
|
+
manifests = {}
|
|
103
|
+
license_text = ""
|
|
104
|
+
readme_head = ""
|
|
105
|
+
for dirpath, dirnames, filenames in os.walk(root):
|
|
106
|
+
dirnames[:] = [d for d in dirnames if d not in SKIP_DIRS]
|
|
107
|
+
rel = Path(dirpath).relative_to(root)
|
|
108
|
+
depth = len(rel.parts) if str(rel) != "." else 0
|
|
109
|
+
if depth > 3:
|
|
110
|
+
dirnames[:] = []
|
|
111
|
+
continue
|
|
112
|
+
for f in sorted(filenames):
|
|
113
|
+
if len(entries) >= max_files or total_bytes >= max_bytes:
|
|
114
|
+
return entries, manifests, license_text, readme_head, True
|
|
115
|
+
fp = Path(dirpath) / f
|
|
116
|
+
try:
|
|
117
|
+
size = fp.stat().st_size
|
|
118
|
+
except OSError:
|
|
119
|
+
continue
|
|
120
|
+
if size > 100_000:
|
|
121
|
+
continue
|
|
122
|
+
rel_p = str(rel / f) if str(rel) != "." else f
|
|
123
|
+
entries.append(rel_p)
|
|
124
|
+
if f in MANIFEST_FILES or Path(f).suffix in TEXT_EXTS:
|
|
125
|
+
try:
|
|
126
|
+
content = fp.read_text(encoding="utf-8", errors="replace")
|
|
127
|
+
except OSError:
|
|
128
|
+
continue
|
|
129
|
+
total_bytes += len(content.encode("utf-8", errors="replace"))
|
|
130
|
+
if f in MANIFEST_FILES and f not in manifests:
|
|
131
|
+
manifests[f] = content[:4000]
|
|
132
|
+
if f.lower().startswith("license") and not license_text:
|
|
133
|
+
license_text = content[:4000]
|
|
134
|
+
if f.lower() == "readme.md" and not readme_head:
|
|
135
|
+
readme_head = content[:4000]
|
|
136
|
+
return entries, manifests, license_text, readme_head, False
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def harvest_repo(
|
|
140
|
+
url,
|
|
141
|
+
max_files=DEFAULT_MAX_FILES,
|
|
142
|
+
max_bytes=DEFAULT_MAX_BYTES,
|
|
143
|
+
timeout=DEFAULT_TIMEOUT,
|
|
144
|
+
base_dir=".research",
|
|
145
|
+
):
|
|
146
|
+
started = datetime.now(timezone.utc).isoformat()
|
|
147
|
+
tmp = tempfile.mkdtemp(prefix="darkharvest-")
|
|
148
|
+
try:
|
|
149
|
+
code, out = sh(["git", "clone", "--depth", "1", url, tmp], timeout=timeout)
|
|
150
|
+
if code != 0:
|
|
151
|
+
return {
|
|
152
|
+
"url": url,
|
|
153
|
+
"status": "metadata-only",
|
|
154
|
+
"generated_at": started,
|
|
155
|
+
"negative_knowledge": (
|
|
156
|
+
f"[NEGATIVE_KNOWLEDGE: clone failed for {url}; "
|
|
157
|
+
f"used metadata-only row. Detail: {out[:200]}]"
|
|
158
|
+
),
|
|
159
|
+
}
|
|
160
|
+
entries, manifests, license_text, readme_head, truncated = scan_tree(
|
|
161
|
+
Path(tmp), max_files=max_files, max_bytes=max_bytes
|
|
162
|
+
)
|
|
163
|
+
spdx = detect_license(license_text + manifests.get("package.json", ""))
|
|
164
|
+
harvestable = (
|
|
165
|
+
"depend-or-vendor"
|
|
166
|
+
if spdx in ("MIT", "Apache-2.0", "BSD-3-Clause", "ISC")
|
|
167
|
+
else "clean-room-rebuild-only"
|
|
168
|
+
)
|
|
169
|
+
hasher = SourceHasher(Path(base_dir))
|
|
170
|
+
cache_blob = (
|
|
171
|
+
f"# {url}\n\nSPDX: {spdx}\n\n"
|
|
172
|
+
f"## README head\n{readme_head[:2000]}\n\n"
|
|
173
|
+
f"## Tree sample ({len(entries)} entries)\n" + "\n".join(entries[:60])
|
|
174
|
+
)
|
|
175
|
+
source_hash = hasher.store_source(url, cache_blob, f"Darkharvest {url}")
|
|
176
|
+
warnings = []
|
|
177
|
+
if spdx in ("GPL", "AGPL"):
|
|
178
|
+
warnings.append(
|
|
179
|
+
"⚠️ LICENSE: strong copyleft — spec rebuild only, never vendor"
|
|
180
|
+
)
|
|
181
|
+
elif spdx == "UNKNOWN":
|
|
182
|
+
warnings.append("⚠️ LICENSE: unknown — treat as clean-room-rebuild-only")
|
|
183
|
+
if truncated:
|
|
184
|
+
warnings.append(
|
|
185
|
+
"⚠️ SCAN: truncated by file/byte caps; treat as partial evidence"
|
|
186
|
+
)
|
|
187
|
+
return {
|
|
188
|
+
"url": url,
|
|
189
|
+
"status": "scanned",
|
|
190
|
+
"generated_at": started,
|
|
191
|
+
"spdx": spdx,
|
|
192
|
+
"harvest_policy": harvestable,
|
|
193
|
+
"tree_entries": len(entries),
|
|
194
|
+
"truncated": truncated,
|
|
195
|
+
"source_hash": source_hash,
|
|
196
|
+
"readme_head": readme_head[:1500],
|
|
197
|
+
"manifests_seen": sorted(manifests.keys()),
|
|
198
|
+
"warnings": warnings,
|
|
199
|
+
}
|
|
200
|
+
finally:
|
|
201
|
+
shutil.rmtree(tmp, ignore_errors=True)
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def main():
|
|
205
|
+
ap = argparse.ArgumentParser(description="Darkharvest competitor fetch helper")
|
|
206
|
+
ap.add_argument("--repo", required=True, help="Competitor repo URL to scan")
|
|
207
|
+
ap.add_argument("--show-context", action="store_true")
|
|
208
|
+
ap.add_argument("--export", default="")
|
|
209
|
+
ap.add_argument("--max-files", type=int, default=DEFAULT_MAX_FILES)
|
|
210
|
+
ap.add_argument("--max-bytes", type=int, default=DEFAULT_MAX_BYTES)
|
|
211
|
+
ap.add_argument("--timeout", type=int, default=DEFAULT_TIMEOUT)
|
|
212
|
+
ap.add_argument("--dir", default=".research")
|
|
213
|
+
args = ap.parse_args()
|
|
214
|
+
|
|
215
|
+
if not re.match(r"^https?://", args.repo):
|
|
216
|
+
print(f"WARN: {args.repo} is not an http(s) URL; metadata-only stub emitted.")
|
|
217
|
+
result = {
|
|
218
|
+
"url": args.repo,
|
|
219
|
+
"status": "metadata-only",
|
|
220
|
+
"generated_at": datetime.now(timezone.utc).isoformat(),
|
|
221
|
+
"negative_knowledge": (
|
|
222
|
+
f"[NEGATIVE_KNOWLEDGE: non-cloneable target {args.repo}]"
|
|
223
|
+
),
|
|
224
|
+
}
|
|
225
|
+
else:
|
|
226
|
+
result = harvest_repo(
|
|
227
|
+
args.repo,
|
|
228
|
+
max_files=args.max_files,
|
|
229
|
+
max_bytes=args.max_bytes,
|
|
230
|
+
timeout=args.timeout,
|
|
231
|
+
base_dir=args.dir,
|
|
232
|
+
)
|
|
233
|
+
|
|
234
|
+
if args.show_context or not args.export:
|
|
235
|
+
print(f"\n🌑 [Darkharvest] {result['url']} → {result['status']}")
|
|
236
|
+
for key in ("spdx", "harvest_policy", "tree_entries", "source_hash"):
|
|
237
|
+
if key in result:
|
|
238
|
+
print(f" {key}: {result[key]}")
|
|
239
|
+
for w in result.get("warnings", []):
|
|
240
|
+
print(f" {w}")
|
|
241
|
+
if result.get("negative_knowledge"):
|
|
242
|
+
print(f" {result['negative_knowledge'][:200]}")
|
|
243
|
+
|
|
244
|
+
if args.export:
|
|
245
|
+
out = Path(os.path.realpath(args.export))
|
|
246
|
+
out.parent.mkdir(parents=True, exist_ok=True)
|
|
247
|
+
out.write_text(json.dumps(result, indent=2), encoding="utf-8")
|
|
248
|
+
print(f"✅ Harvest scan exported to {out}")
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
if __name__ == "__main__":
|
|
252
|
+
main()
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "factory",
|
|
3
|
+
"version": "0.1.0",
|
|
4
|
+
"description": "Coding-factory Manager loop with grill-gated phased builds and dual QA for Claude Code.",
|
|
5
|
+
"author": {
|
|
6
|
+
"name": "Heretek AI",
|
|
7
|
+
"email": "dev@heretek.ai"
|
|
8
|
+
},
|
|
9
|
+
"license": "Apache-2.0",
|
|
10
|
+
"repository": "https://github.com/Heretek-AI/IUMBTEMS",
|
|
11
|
+
"homepage": "https://github.com/Heretek-AI/IUMBTEMS#readme",
|
|
12
|
+
"skills": [
|
|
13
|
+
"./skills/factory"
|
|
14
|
+
]
|
|
15
|
+
}
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: factory-manager
|
|
3
|
+
description: Coding-factory Manager. Owns grill gates, swarm dispatch, roadmap synthesis, and QA tiebreaks. Never writes code. Use to drive /factory runs.
|
|
4
|
+
model: sonnet
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
You are the Factory Manager in the factory plugin.
|
|
8
|
+
|
|
9
|
+
Loop: grill the user until `.factory/frontier.json` is settled (max 5
|
|
10
|
+
brainstorm+darkharvest swarm cycles per gate); explicit user approve advances
|
|
11
|
+
each gate. Per gate, dispatch both swarms (mock-first on fixtures), then
|
|
12
|
+
synthesize `.roadmap/<phase>/` as GOAL.md plus dossier.json
|
|
13
|
+
(goal/evidence/acceptance/brief/verdict/hashes — every claim needs a VERIFIED
|
|
14
|
+
hash or file pointer). Spawn the programmer subagent per phase with the phase
|
|
15
|
+
dossier (cite phase hashes). Run qa-a and qa-b per phase; track retries with
|
|
16
|
+
`skills/factory/scripts/factory.py` (3 failures escalate back to you with both
|
|
17
|
+
QA reports). Tiebreak QA disagreements; require explicit user sign-off per phase.
|
|
18
|
+
|
|
19
|
+
Never write implementation code yourself. Never bypass gates outside an
|
|
20
|
+
active `/domainexpansion` count. Never accept unverified harvest verdicts.
|
|
21
|
+
|
|
22
|
+
Full specification: `skills/factory/SKILL.md` in the plugin root.
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: programmer
|
|
3
|
+
description: Factory implementer. Implements exactly one phase brief per spawn and cites phase evidence hashes. Never invokes swarms or other programmers. Use per .roadmap phase.
|
|
4
|
+
model: sonnet
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
You are the Factory Programmer in the factory plugin.
|
|
8
|
+
|
|
9
|
+
You receive one phase dossier (GOAL.md + dossier.json). Implement exactly what
|
|
10
|
+
the brief specifies — no scope expansion, no drive-by refactors. Cite the
|
|
11
|
+
phase evidence hashes for any borrowed pattern. If the brief is ambiguous or
|
|
12
|
+
its acceptance criteria are untestable, stop and return questions to the
|
|
13
|
+
manager instead of guessing. Never invoke brainstorm/darkharvest swarms, never
|
|
14
|
+
spawn other agents, never edit outside the phase scope.
|
|
15
|
+
|
|
16
|
+
Full specification: `skills/factory/SKILL.md` in the plugin root.
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: qa-adversarial
|
|
3
|
+
description: Factory QA, adversarial seat. Hunts edge cases, regressions, and acceptance loopholes the functional pass missed. Read-only plus test execution. Use alongside qa-functional per phase.
|
|
4
|
+
model: sonnet
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
You are QA-B (adversarial seat) in the factory plugin.
|
|
8
|
+
|
|
9
|
+
Assume the functional pass missed something. Attack the phase: edge-case
|
|
10
|
+
inputs, regressions in neighboring behavior, acceptance criteria that are
|
|
11
|
+
vacuously satisfiable, error paths, and resource/timeout boundaries. Read-only
|
|
12
|
+
plus test execution — never edit code. Verdicts: `pass | fail(reason) |
|
|
13
|
+
conditional(note)`, each with a reproduction. Your prompt is deliberately
|
|
14
|
+
diverged from qa-functional: it proves what works, you hunt what breaks.
|
|
15
|
+
Disagreements go to the factory-manager tiebreak.
|
|
16
|
+
|
|
17
|
+
Full specification: `skills/factory/SKILL.md` in the plugin root.
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: qa-functional
|
|
3
|
+
description: Factory QA, functional seat. Verifies phase acceptance criteria pass on the real surface. Read-only plus test execution. Use alongside qa-adversarial per phase.
|
|
4
|
+
model: sonnet
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
You are QA-A (functional seat) in the factory plugin.
|
|
8
|
+
|
|
9
|
+
Verify each acceptance criterion in the phase dossier by executing it on the
|
|
10
|
+
real surface (run the tests, exercise the feature, inspect the output).
|
|
11
|
+
Read-only plus test execution — never edit code to make a check pass. Verdicts:
|
|
12
|
+
`pass | fail(reason) | conditional(note)`, each tied to a specific criterion.
|
|
13
|
+
Record via the manager; three failures on one phase escalate it back to the
|
|
14
|
+
manager with both QA reports attached. Your prompt is deliberately diverged
|
|
15
|
+
from qa-adversarial: you prove what works, it hunts what breaks.
|
|
16
|
+
|
|
17
|
+
Full specification: `skills/factory/SKILL.md` in the plugin root.
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
---
|
|
2
|
+
type: llm
|
|
3
|
+
---
|
|
4
|
+
|
|
5
|
+
PASS if the reply proposes phased scope with explicit approval gates (or asks gate/phase-clarifying questions first) instead of writing implementation code immediately.
|
|
6
|
+
FAIL if the reply starts implementing code with no phase breakdown and no user sign-off step.
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: factory
|
|
3
|
+
description: Coding-factory Manager loop. Use when user invokes /factory or /domainexpansion, or wants grill-gated phased builds with programmer spawns and dual QA. Manager grills until frontier settled, runs brainstorm plus darkharvest swarms per gate, synthesizes .roadmap phases, spawns programmer per phase, and enforces dual-QA retry bounds. Never writes code itself; never bypasses explicit user sign-off (except inside /domainexpansion count).
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Factory — Manager Loop & Domain Expansion
|
|
7
|
+
|
|
8
|
+
## 1. Roles (OpenCode profiles in `config/opencode-snippet.json`)
|
|
9
|
+
|
|
10
|
+
- **manager** (primary): owns gates, grilling, swarm dispatch, roadmap synthesis, tiebreaks.
|
|
11
|
+
Task allowlist: `factory-*`, `programmer`, `qa-a`, `qa-b`, `brainstormer`, `darkharvester` deny `*` otherwise.
|
|
12
|
+
- **programmer** (subagent): implements exactly one phase brief. May call `iumbtems_verify_quote`; may NOT invoke swarms or other programmers.
|
|
13
|
+
- **qa-a / qa-b** (subagents): same model, DIVERGED prompts (functional-correctness vs adversarial edge-case). Read-only plus test execution; never edit.
|
|
14
|
+
- **researcher** = existing `iumbtems_brainstorm` + `iumbtems_darkharvest` swarms (no new profile).
|
|
15
|
+
|
|
16
|
+
## 2. Gate protocol (max 5 swarm cycles per gate)
|
|
17
|
+
|
|
18
|
+
1. Grill until `.factory/frontier.json` settled (grilling skill).
|
|
19
|
+
2. Run brainstorm + darkharvest swarms (mock-first on fixtures).
|
|
20
|
+
3. Manager synthesizes `.roadmap/<phase>/` (GOAL.md + dossier.json).
|
|
21
|
+
4. Explicit user `approve` advances; anything else regrills (cycle counter in `.factory/state.json`).
|
|
22
|
+
|
|
23
|
+
## 3. Phase contract (`.roadmap/<phase>/`)
|
|
24
|
+
|
|
25
|
+
- `GOAL.md`: human-readable goal + acceptance criteria.
|
|
26
|
+
- `dossier.json`: `{phase, goal, evidence:[{hash, quote}], acceptance[], brief, verdict, hashes}`. Every harvest/claim entry needs a SHA-256 source hash or `file://` pointer or it is purged to NEGATIVE_KNOWLEDGE.
|
|
27
|
+
- Programmer receives the phase dossier (Manager chooses freeform vs strict brief but MUST cite phase hashes).
|
|
28
|
+
|
|
29
|
+
## 4. QA protocol (3 retries, then escalate)
|
|
30
|
+
|
|
31
|
+
- Both QA seats run per phase; disagreements go to manager tiebreak.
|
|
32
|
+
- `skills/factory/scripts/factory.py` tracks `qa_retries` per phase in `.factory/state.json`. On 3rd rejection: halt phase, return to manager with both QA reports (retry loop per plan; manager may regrill scope or escalate to user).
|
|
33
|
+
- QA verdicts: `pass | fail(reason) | conditional(note)`.
|
|
34
|
+
|
|
35
|
+
## 5. Domain expansion (`/domainexpansion <n>`)
|
|
36
|
+
|
|
37
|
+
- Bypasses per-loop gates; stops on count OR `.factory/STOP` file OR user kill, whichever first.
|
|
38
|
+
- Each loop: agents propose direction → quick swarm check → implement → QA → next.
|
|
39
|
+
- `factory.py` enforces: refuse `n < 1`, cap `n` at `--max-loops` default 10, check STOP file before every loop.
|
|
40
|
+
|
|
41
|
+
## 6. State layout (three dirs, distinct jobs)
|
|
42
|
+
|
|
43
|
+
- `.factory/`: run state (`state.json`, `frontier.json`, `STOP` kill-file). Gitignored runtime state.
|
|
44
|
+
- `.roadmap/`: output (phase dirs). Committed.
|
|
45
|
+
- `.research/`: evidence (swarm dossiers, source cache). Gitignored (existing rule).
|
|
46
|
+
|
|
47
|
+
## 7. Anti-patterns
|
|
48
|
+
|
|
49
|
+
- No code writes by manager; no swarm invocation by programmer; no edits by QA.
|
|
50
|
+
- No gate bypass outside `/domainexpansion`; no uncapped loops.
|
|
51
|
+
- No VERIFIED claims without hashes, even in expansion proposals.
|
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
Factory run-state helper: phase dossiers, QA retry bounds, expansion loop guard.
|
|
4
|
+
|
|
5
|
+
All state lives under .factory/ (gitignored runtime state). Phase output goes
|
|
6
|
+
to .roadmap/<phase>/. Evidence stays in .research/. Read-only w.r.t. repo code.
|
|
7
|
+
|
|
8
|
+
Usage:
|
|
9
|
+
python3 skills/factory/scripts/factory.py init --run <name>
|
|
10
|
+
python3 skills/factory/scripts/factory.py phase-add --run <name> --phase 01-auth --goal "..." --accept "..."
|
|
11
|
+
python3 skills/factory/scripts/factory.py qa-record --run <name> --phase 01-auth --seat qa-a --verdict fail --reason "..."
|
|
12
|
+
python3 skills/factory/scripts/factory.py expansion --run <name> --loops 10 [--max-loops 10]
|
|
13
|
+
python3 skills/factory/scripts/factory.py stop --run <name> # write STOP kill-file
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
import argparse
|
|
17
|
+
import json
|
|
18
|
+
import sys
|
|
19
|
+
from datetime import datetime, timezone
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
|
|
22
|
+
PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent.parent
|
|
23
|
+
FACTORY_DIR = PROJECT_ROOT / ".factory"
|
|
24
|
+
ROADMAP_DIR = PROJECT_ROOT / ".roadmap"
|
|
25
|
+
|
|
26
|
+
MAX_QA_RETRIES = 3
|
|
27
|
+
MAX_EXPANSION_LOOPS = 10
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def now():
|
|
31
|
+
return datetime.now(timezone.utc).isoformat()
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def run_dir(run):
|
|
35
|
+
return FACTORY_DIR / run
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def load_state(run):
|
|
39
|
+
p = run_dir(run) / "state.json"
|
|
40
|
+
if not p.exists():
|
|
41
|
+
raise SystemExit(f"No factory run '{run}'. Run `factory.py init` first.")
|
|
42
|
+
return json.loads(p.read_text(encoding="utf-8"))
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def save_state(run, state):
|
|
46
|
+
p = run_dir(run) / "state.json"
|
|
47
|
+
p.write_text(json.dumps(state, indent=2), encoding="utf-8")
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def cmd_init(args):
|
|
51
|
+
d = run_dir(args.run)
|
|
52
|
+
d.mkdir(parents=True, exist_ok=True)
|
|
53
|
+
state = {
|
|
54
|
+
"run": args.run,
|
|
55
|
+
"created_at": now(),
|
|
56
|
+
"gate_cycles": 0,
|
|
57
|
+
"phases": {},
|
|
58
|
+
"expansion": {"loops_done": 0, "loops_planned": 0},
|
|
59
|
+
}
|
|
60
|
+
save_state(args.run, state)
|
|
61
|
+
print(f"✅ Factory run '{args.run}' initialised at {d}")
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def cmd_phase_add(args):
|
|
65
|
+
state = load_state(args.run)
|
|
66
|
+
if args.phase in state["phases"]:
|
|
67
|
+
raise SystemExit(f"Phase '{args.phase}' already exists.")
|
|
68
|
+
state["phases"][args.phase] = {
|
|
69
|
+
"goal": args.goal,
|
|
70
|
+
"acceptance": [a.strip() for a in args.accept.split(";") if a.strip()],
|
|
71
|
+
"status": "briefed",
|
|
72
|
+
"qa_retries": 0,
|
|
73
|
+
"qa_reports": [],
|
|
74
|
+
}
|
|
75
|
+
save_state(args.run, state)
|
|
76
|
+
# Phase output: GOAL.md + dossier.json skeleton (evidence filled by manager).
|
|
77
|
+
phase_dir = ROADMAP_DIR / args.phase
|
|
78
|
+
phase_dir.mkdir(parents=True, exist_ok=True)
|
|
79
|
+
(phase_dir / "GOAL.md").write_text(
|
|
80
|
+
f"# {args.phase}: {args.goal}\n\n## Acceptance\n"
|
|
81
|
+
+ "".join(f"- [ ] {a}\n" for a in state["phases"][args.phase]["acceptance"]),
|
|
82
|
+
encoding="utf-8",
|
|
83
|
+
)
|
|
84
|
+
(phase_dir / "dossier.json").write_text(
|
|
85
|
+
json.dumps(
|
|
86
|
+
{
|
|
87
|
+
"phase": args.phase,
|
|
88
|
+
"goal": args.goal,
|
|
89
|
+
"evidence": [],
|
|
90
|
+
"acceptance": state["phases"][args.phase]["acceptance"],
|
|
91
|
+
"brief": "",
|
|
92
|
+
"verdict": "briefed",
|
|
93
|
+
"hashes": [],
|
|
94
|
+
},
|
|
95
|
+
indent=2,
|
|
96
|
+
),
|
|
97
|
+
encoding="utf-8",
|
|
98
|
+
)
|
|
99
|
+
print(f"✅ Phase '{args.phase}' briefed → {phase_dir}")
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def cmd_qa_record(args):
|
|
103
|
+
state = load_state(args.run)
|
|
104
|
+
phase = state["phases"].get(args.phase)
|
|
105
|
+
if phase is None:
|
|
106
|
+
raise SystemExit(f"Unknown phase '{args.phase}'.")
|
|
107
|
+
if args.verdict not in ("pass", "fail", "conditional"):
|
|
108
|
+
raise SystemExit("verdict must be pass|fail|conditional.")
|
|
109
|
+
phase["qa_reports"].append(
|
|
110
|
+
{
|
|
111
|
+
"seat": args.seat,
|
|
112
|
+
"verdict": args.verdict,
|
|
113
|
+
"reason": args.reason or "",
|
|
114
|
+
"at": now(),
|
|
115
|
+
}
|
|
116
|
+
)
|
|
117
|
+
if args.verdict == "fail":
|
|
118
|
+
phase["qa_retries"] += 1
|
|
119
|
+
if phase["qa_retries"] >= MAX_QA_RETRIES:
|
|
120
|
+
phase["status"] = "escalated"
|
|
121
|
+
save_state(args.run, state)
|
|
122
|
+
print(
|
|
123
|
+
f"🛑 Phase '{args.phase}' ESCALATED after "
|
|
124
|
+
f"{MAX_QA_RETRIES} QA failures — back to manager."
|
|
125
|
+
)
|
|
126
|
+
return 2
|
|
127
|
+
phase["status"] = "retrying"
|
|
128
|
+
elif args.verdict == "pass":
|
|
129
|
+
phase["status"] = "signed-off"
|
|
130
|
+
else:
|
|
131
|
+
phase["status"] = "conditional"
|
|
132
|
+
save_state(args.run, state)
|
|
133
|
+
print(
|
|
134
|
+
f"✅ QA recorded: {args.phase} [{args.seat}] → {args.verdict} "
|
|
135
|
+
f"(status={phase['status']}, retries={phase['qa_retries']})"
|
|
136
|
+
)
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def stop_requested(run):
|
|
140
|
+
return (run_dir(run) / "STOP").exists()
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def cmd_expansion(args):
|
|
144
|
+
state = load_state(args.run)
|
|
145
|
+
n = args.loops
|
|
146
|
+
if n < 1:
|
|
147
|
+
raise SystemExit("loops must be >= 1.")
|
|
148
|
+
cap = args.max_loops or MAX_EXPANSION_LOOPS
|
|
149
|
+
n = min(n, cap)
|
|
150
|
+
state["expansion"]["loops_planned"] = n
|
|
151
|
+
save_state(args.run, state)
|
|
152
|
+
done = 0
|
|
153
|
+
for i in range(1, n + 1):
|
|
154
|
+
if stop_requested(args.run):
|
|
155
|
+
print(f"🛑 STOP file present — halting after {done}/{n} loops.")
|
|
156
|
+
break
|
|
157
|
+
done += 1
|
|
158
|
+
print(
|
|
159
|
+
f"🔁 Expansion loop {done}/{n}: agents propose → swarm check → "
|
|
160
|
+
f"implement → QA (Manager drives each step)."
|
|
161
|
+
)
|
|
162
|
+
state = load_state(args.run)
|
|
163
|
+
state["expansion"]["loops_done"] += done
|
|
164
|
+
save_state(args.run, state)
|
|
165
|
+
print(f"✅ Expansion finished {done} loop(s).")
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def cmd_stop(args):
|
|
169
|
+
(run_dir(args.run)).mkdir(parents=True, exist_ok=True)
|
|
170
|
+
(run_dir(args.run) / "STOP").write_text(
|
|
171
|
+
f"stop requested at {now()}\n", encoding="utf-8"
|
|
172
|
+
)
|
|
173
|
+
print(f"🛑 STOP file written for run '{args.run}'.")
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def main():
|
|
177
|
+
ap = argparse.ArgumentParser(description="Factory run-state helper")
|
|
178
|
+
sub = ap.add_subparsers(dest="command", required=True)
|
|
179
|
+
|
|
180
|
+
p = sub.add_parser("init")
|
|
181
|
+
p.add_argument("--run", required=True)
|
|
182
|
+
p = sub.add_parser("phase-add")
|
|
183
|
+
p.add_argument("--run", required=True)
|
|
184
|
+
p.add_argument("--phase", required=True)
|
|
185
|
+
p.add_argument("--goal", required=True)
|
|
186
|
+
p.add_argument("--accept", default="")
|
|
187
|
+
p = sub.add_parser("qa-record")
|
|
188
|
+
p.add_argument("--run", required=True)
|
|
189
|
+
p.add_argument("--phase", required=True)
|
|
190
|
+
p.add_argument("--seat", required=True)
|
|
191
|
+
p.add_argument("--verdict", required=True)
|
|
192
|
+
p.add_argument("--reason", default="")
|
|
193
|
+
p = sub.add_parser("expansion")
|
|
194
|
+
p.add_argument("--run", required=True)
|
|
195
|
+
p.add_argument("--loops", type=int, required=True)
|
|
196
|
+
p.add_argument("--max-loops", type=int, default=MAX_EXPANSION_LOOPS)
|
|
197
|
+
p = sub.add_parser("stop")
|
|
198
|
+
p.add_argument("--run", required=True)
|
|
199
|
+
|
|
200
|
+
args = ap.parse_args()
|
|
201
|
+
code = {
|
|
202
|
+
"init": cmd_init,
|
|
203
|
+
"phase-add": cmd_phase_add,
|
|
204
|
+
"qa-record": cmd_qa_record,
|
|
205
|
+
"expansion": cmd_expansion,
|
|
206
|
+
"stop": cmd_stop,
|
|
207
|
+
}[args.command](args)
|
|
208
|
+
sys.exit(code if isinstance(code, int) else 0)
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
if __name__ == "__main__":
|
|
212
|
+
main()
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
description = "Autonomous agent-guided self-improvement loop (count-flagged)"
|
|
2
|
+
prompt = """Run the IUMBTEMS domain-expansion loop for {{args}} loops (max 10).
|
|
3
|
+
|
|
4
|
+
Bypasses per-loop gates; stops on count OR .factory/STOP file OR user kill.
|
|
5
|
+
Each loop: agents propose direction, quick swarm check, implement, dual-QA verify.
|
|
6
|
+
Enforce via: python3 skills/factory/scripts/factory.py expansion --run <run> --loops {{args}}.
|
|
7
|
+
"""
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
description = "Coding-factory Manager loop with grill-gated phased builds"
|
|
2
|
+
prompt = """Run the IUMBTEMS coding-factory Manager loop for: {{args}}.
|
|
3
|
+
|
|
4
|
+
1. Grill until .factory/frontier.json is settled (max 5 swarm cycles per gate); explicit user approve advances each gate.
|
|
5
|
+
2. Per gate: python3 runner/research_swarm.py --mode brainstorm plus --mode darkharvest (mock-first), then synthesize .roadmap/<phase>/ GOAL.md + dossier.json.
|
|
6
|
+
3. Programmer subagent per phase; qa-a plus qa-b per phase; retries via python3 skills/factory/scripts/factory.py (3 failures escalate).
|
|
7
|
+
Report: .roadmap/ phases plus .factory/state.json.
|
|
8
|
+
"""
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
# Factory (thin adapter stub)
|
|
2
|
+
|
|
3
|
+
This file is a POINTER, not the implementation. It exists so harness skill
|
|
4
|
+
discovery finds an entry; the real skill lives in the IUMBTEMS repo.
|
|
5
|
+
|
|
6
|
+
- Canonical prose & scripts: `skills/factory/`
|
|
7
|
+
- Canonical programmatic surface: `python3 runner/mcp_server.py` (stdio MCP),
|
|
8
|
+
or one-shot: `python3 runner/mcp_server.py call <tool> '{...json...}'`
|
|
9
|
+
- MCP tools for this skill: `iumbtems_brainstorm`, `iumbtems_darkharvest`, `iumbtems_socratic_frontier`
|
|
10
|
+
|
|
11
|
+
Epistemic rules apply regardless of harness: tag claims as
|
|
12
|
+
`[VERIFIED: <hash>]`, `[INFERRED: <reasoning>]`, `[HYPOTHESIS: <test>]`, or
|
|
13
|
+
`[NEGATIVE_KNOWLEDGE: <query>]`. Writes go only to `.research/`.
|