@heretek-ai/epistemic-swarm 0.7.3 → 0.7.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/.omp/commands/domainexpansion.md +1 -1
- package/.omp/commands/factory.md +1 -1
- package/README.md +10 -9
- package/config/opencode-snippet.json +5 -5
- package/extensions/pi/index.js +28 -2
- package/package.json +1 -1
- package/plugins/factory/skills/factory/SKILL.md +10 -0
- package/plugins/factory/skills/factory/scripts/factory.py +39 -5
- package/plugins/gemini/commands/domainexpansion.toml +1 -1
- package/plugins/gemini/commands/factory.toml +1 -1
- package/plugins/opencode/index.js +43 -3
- package/prompts/antigravity_repo_factcheck.md +143 -0
- package/runner/__pycache__/__init__.cpython-311.pyc +0 -0
- package/runner/__pycache__/auctioneer.cpython-311.pyc +0 -0
- package/runner/__pycache__/auditor_engine.cpython-311.pyc +0 -0
- package/runner/__pycache__/claim_store.cpython-311.pyc +0 -0
- package/runner/__pycache__/claim_witness.cpython-311.pyc +0 -0
- package/runner/__pycache__/living_dossiers.cpython-311.pyc +0 -0
- package/runner/__pycache__/mcp_protocol.cpython-311.pyc +0 -0
- package/runner/__pycache__/mcp_server.cpython-311.pyc +0 -0
- package/runner/__pycache__/path_safety.cpython-311.pyc +0 -0
- package/runner/__pycache__/pcrb.cpython-311.pyc +0 -0
- package/runner/__pycache__/pcrb_verify.cpython-311.pyc +0 -0
- package/runner/__pycache__/refinement.cpython-311.pyc +0 -0
- package/runner/__pycache__/research_swarm.cpython-311.pyc +0 -0
- package/runner/__pycache__/state_machine.cpython-311.pyc +0 -0
- package/runner/mcp_server.py +121 -0
- package/runner/tests/__pycache__/test_auction_order.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_backends.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_bet1_spike.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claim_store.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claim_witness.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claude_plugin.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_domain_packs.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_factory.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_fleet_seam.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_living_dossiers.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_mcp_server.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_opencode_ux.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_pcrb.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_refinement.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_swarm.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_sweep_regressions.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_webcache.cpython-311.pyc +0 -0
- package/runner/tests/test_factory.py +51 -0
- package/runner/tests/test_mcp_server.py +1 -0
- package/runner/tests/test_opencode_ux.py +2 -2
- package/runner/tests/test_swarm.py +1 -0
- package/scripts/__pycache__/bet1_advisory_spike.cpython-311.pyc +0 -0
- package/scripts/__pycache__/build_adapters.cpython-311.pyc +0 -0
- package/scripts/__pycache__/divergence_experiment.cpython-311.pyc +0 -0
- package/skills/epistemic_search/scripts/__pycache__/search.cpython-311.pyc +0 -0
- package/skills/factory/SKILL.md +10 -0
- package/skills/factory/scripts/factory.py +39 -5
- package/skills/research_cache/__pycache__/__init__.cpython-311.pyc +0 -0
- package/skills/research_cache/__pycache__/hasher.cpython-311.pyc +0 -0
- package/skills/swarm_config/__pycache__/__init__.cpython-311.pyc +0 -0
- package/skills/swarm_config/__pycache__/configure.cpython-311.pyc +0 -0
|
@@ -6,4 +6,4 @@ Run the IUMBTEMS domain-expansion loop for `$1` loops (max 10):
|
|
|
6
6
|
|
|
7
7
|
Bypasses per-loop gates; stops on count OR `.factory/STOP` file OR user kill.
|
|
8
8
|
Each loop: agents propose direction, quick swarm check, implement, dual-QA verify.
|
|
9
|
-
Enforce via `
|
|
9
|
+
Enforce via the `iumbtems_factory` tool (expansion command with loops/max_loops).
|
package/.omp/commands/factory.md
CHANGED
|
@@ -6,4 +6,4 @@ Run the IUMBTEMS coding-factory Manager loop for `$1`:
|
|
|
6
6
|
|
|
7
7
|
1. Grill until `.factory/frontier.json` is settled (max 5 swarm cycles per gate); explicit user approve advances each gate.
|
|
8
8
|
2. Per gate: `python3 runner/research_swarm.py --mode brainstorm` plus `--mode darkharvest` (mock-first), then synthesize `.roadmap/<phase>/` GOAL.md + dossier.json.
|
|
9
|
-
3. Programmer subagent per phase; qa-a plus qa-b per phase; retries tracked via `
|
|
9
|
+
3. Programmer subagent per phase; qa-a plus qa-b per phase; retries tracked via the `iumbtems_factory` tool (phase-add / qa-record; 3 failures escalate). Never invoke factory helper scripts by relative path.
|
package/README.md
CHANGED
|
@@ -75,19 +75,20 @@ pi install npm:@heretek-ai/epistemic-swarm
|
|
|
75
75
|
- OMP (`omp.sh`, oh-my-pi) shares the same entry point: `omp install npm:@heretek-ai/epistemic-swarm`, project commands in `.omp/commands/` (`/swarm`, `/grill`, `/audit`, `/scout`, `/brainstorming`, `/swarm-config`), prompts in `.omp/prompts/`, hooks in `.omp/hooks/pre|post/`.
|
|
76
76
|
- All commands automatically respect `.research/config.json`.
|
|
77
77
|
|
|
78
|
-
### 3. OpenCode V2 (
|
|
79
|
-
Enable IUMBTEMS in your `~/.config/opencode/opencode.
|
|
80
|
-
```
|
|
78
|
+
### 3. OpenCode V2 ([opencode.ai/v2/docs](https://opencode.ai/v2/docs))
|
|
79
|
+
Enable IUMBTEMS in your `~/.config/opencode/opencode.jsonc` or project `opencode.jsonc`. You can configure settings declaratively using native OpenCode V2 syntax:
|
|
80
|
+
```jsonc
|
|
81
81
|
{
|
|
82
|
-
"
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
82
|
+
"$schema": "https://opencode.ai/config.json",
|
|
83
|
+
"plugins": [
|
|
84
|
+
{
|
|
85
|
+
"package": "@heretek-ai/epistemic-swarm",
|
|
86
|
+
"options": {
|
|
86
87
|
"search_engine": "duckduckgo",
|
|
87
88
|
"max_iterations": 2,
|
|
88
89
|
"mode": "research"
|
|
89
90
|
}
|
|
90
|
-
|
|
91
|
+
}
|
|
91
92
|
]
|
|
92
93
|
}
|
|
93
94
|
```
|
|
@@ -197,7 +198,7 @@ Every factual claim in IUMBTEMS carries an explicit evidentiary tag:
|
|
|
197
198
|
1. **Discovery Tier**: SearXNG (unbiased metasearch) and Brave Search API.
|
|
198
199
|
2. **Extraction Tier**: Firecrawl (headless JavaScript rendering, DOM cleaning, Markdown extraction).
|
|
199
200
|
3. **Academic Tier**: Semantic Scholar / arXiv MCPs for DOI citation resolution.
|
|
200
|
-
4. **Caching Tier**: Content-addressed SHA-256 storage (`skills/
|
|
201
|
+
4. **Caching Tier**: Content-addressed SHA-256 storage (`skills/research_cache/hasher.py`).
|
|
201
202
|
|
|
202
203
|
### Local Infrastructure (Optional)
|
|
203
204
|
Run local SearXNG and Firecrawl instances via Docker Compose:
|
|
@@ -7,15 +7,15 @@
|
|
|
7
7
|
"enabled": true
|
|
8
8
|
}
|
|
9
9
|
},
|
|
10
|
-
"
|
|
11
|
-
|
|
12
|
-
"@heretek-ai/epistemic-swarm",
|
|
13
|
-
{
|
|
10
|
+
"plugins": [
|
|
11
|
+
{
|
|
12
|
+
"package": "@heretek-ai/epistemic-swarm",
|
|
13
|
+
"options": {
|
|
14
14
|
"search_engine": "duckduckgo",
|
|
15
15
|
"max_iterations": 2,
|
|
16
16
|
"mode": "research"
|
|
17
17
|
}
|
|
18
|
-
|
|
18
|
+
}
|
|
19
19
|
],
|
|
20
20
|
"agent": {
|
|
21
21
|
"code-auditor": {
|
package/extensions/pi/index.js
CHANGED
|
@@ -210,8 +210,7 @@ export default function initPiExtension(pi) {
|
|
|
210
210
|
} catch { /* alias is best-effort across pi/omp versions */ }
|
|
211
211
|
|
|
212
212
|
// /darkharvest: Product competitor teardown with harvest verdicts
|
|
213
|
-
pi.registerCommand('darkharvest', {
|
|
214
|
-
description: 'Product competitor teardown: seed inspirations, expand to adjacents, emit harvest verdicts',
|
|
213
|
+
pi.registerCommand('darkharvest', { description: 'Product competitor teardown: seed inspirations, expand to adjacents, emit harvest verdicts',
|
|
215
214
|
usage: '/darkharvest <product-arena> [--seeds <urls>]',
|
|
216
215
|
handler: async (args, ctx) => {
|
|
217
216
|
const objective = args.trim();
|
|
@@ -316,6 +315,33 @@ export default function initPiExtension(pi) {
|
|
|
316
315
|
return { content: [{ type: 'text', text: r.text }] };
|
|
317
316
|
}
|
|
318
317
|
});
|
|
318
|
+
|
|
319
|
+
// Tool: iumbtems_factory (run-state helper, no script paths)
|
|
320
|
+
pi.registerTool({
|
|
321
|
+
name: 'iumbtems_factory',
|
|
322
|
+
description: 'Drive factory run state: init / phase-add / qa-record / expansion / stop',
|
|
323
|
+
parameters: {
|
|
324
|
+
type: 'object',
|
|
325
|
+
properties: {
|
|
326
|
+
command: { type: 'string', enum: ['init', 'phase-add', 'qa-record', 'expansion', 'stop'] },
|
|
327
|
+
run: { type: 'string', description: 'Factory run name' },
|
|
328
|
+
phase: { type: 'string' },
|
|
329
|
+
goal: { type: 'string' },
|
|
330
|
+
accept: { type: 'string' },
|
|
331
|
+
seat: { type: 'string' },
|
|
332
|
+
verdict: { type: 'string', enum: ['pass', 'fail', 'conditional'] },
|
|
333
|
+
reason: { type: 'string' },
|
|
334
|
+
loops: { type: 'integer' },
|
|
335
|
+
max_loops: { type: 'integer' },
|
|
336
|
+
project_dir: { type: 'string' }
|
|
337
|
+
},
|
|
338
|
+
required: ['command']
|
|
339
|
+
},
|
|
340
|
+
execute: async (args = {}) => {
|
|
341
|
+
const r = callMcp('iumbtems_factory', args, {});
|
|
342
|
+
return { content: [{ type: 'text', text: r.text }] };
|
|
343
|
+
}
|
|
344
|
+
});
|
|
319
345
|
}
|
|
320
346
|
}
|
|
321
347
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@heretek-ai/epistemic-swarm",
|
|
3
|
-
"version": "0.7.
|
|
3
|
+
"version": "0.7.5",
|
|
4
4
|
"description": "IUMBTEMS: I Use My Brain To Express My Self — High-Integrity Dialectic Research Agent Harness for Claude Code, OpenCode V2, Pi, OMP (oh-my-pi), Gemini CLI, Codex CLI, and AntiGravity",
|
|
5
5
|
"main": "bin/cli.js",
|
|
6
6
|
"bin": {
|
|
@@ -13,6 +13,16 @@ description: Coding-factory Manager loop. Use when user invokes /factory or /dom
|
|
|
13
13
|
- **qa-a / qa-b** (subagents): same model, DIVERGED prompts (functional-correctness vs adversarial edge-case). Read-only plus test execution; never edit.
|
|
14
14
|
- **researcher** = existing `iumbtems_brainstorm` + `iumbtems_darkharvest` swarms (no new profile).
|
|
15
15
|
|
|
16
|
+
## 1b. Driving run state (no filesystem paths)
|
|
17
|
+
|
|
18
|
+
Use the **`iumbtems_factory` MCP tool** for all run-state changes — never a
|
|
19
|
+
relative `skills/factory/scripts/factory.py` path (the toolchain lives in the
|
|
20
|
+
npm cache in consuming projects, and relative paths broke live: the manager
|
|
21
|
+
agent ran `find / -name factory.py`). The tool wraps the same helper and
|
|
22
|
+
resolves the project via `project_dir` argument, `IUMBTEMS_PROJECT_DIR`, or
|
|
23
|
+
the session cwd. It supports `init`, `phase-add`, `qa-record`, `expansion`,
|
|
24
|
+
`stop`, and returns `status: escalated` (exit 2) on the 3rd QA failure.
|
|
25
|
+
|
|
16
26
|
## 2. Gate protocol (max 5 swarm cycles per gate)
|
|
17
27
|
|
|
18
28
|
1. Grill until `.factory/frontier.json` settled (grilling skill).
|
|
@@ -2,8 +2,13 @@
|
|
|
2
2
|
"""
|
|
3
3
|
Factory run-state helper: phase dossiers, QA retry bounds, expansion loop guard.
|
|
4
4
|
|
|
5
|
-
All state lives under
|
|
6
|
-
to
|
|
5
|
+
All state lives under <project>/.factory/ (gitignored runtime state). Phase
|
|
6
|
+
output goes to <project>/.roadmap/<phase>/. Evidence stays in <project>/
|
|
7
|
+
.research/. Read-only w.r.t. repo code.
|
|
8
|
+
|
|
9
|
+
Project directory resolution: --project-dir > IUMBTEMS_PROJECT_DIR env > the
|
|
10
|
+
toolchain repo root (local-dev default). A consuming project must never leak
|
|
11
|
+
state into the IUMBTEMS checkout.
|
|
7
12
|
|
|
8
13
|
Usage:
|
|
9
14
|
python3 skills/factory/scripts/factory.py init --run <name>
|
|
@@ -15,13 +20,25 @@ Usage:
|
|
|
15
20
|
|
|
16
21
|
import argparse
|
|
17
22
|
import json
|
|
23
|
+
import os
|
|
18
24
|
import sys
|
|
19
25
|
from datetime import datetime, timezone
|
|
20
26
|
from pathlib import Path
|
|
21
27
|
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
28
|
+
REPO_ROOT = Path(__file__).resolve().parent.parent.parent.parent
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def resolve_project_root(cli_value=None):
|
|
32
|
+
"""--project-dir > IUMBTEMS_PROJECT_DIR > repo root (local-dev default)."""
|
|
33
|
+
for candidate in (cli_value, os.environ.get("IUMBTEMS_PROJECT_DIR")):
|
|
34
|
+
if candidate and Path(candidate).is_dir():
|
|
35
|
+
return Path(candidate).resolve()
|
|
36
|
+
return REPO_ROOT
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
PROJECT_DIR = resolve_project_root()
|
|
40
|
+
FACTORY_DIR = PROJECT_DIR / ".factory"
|
|
41
|
+
ROADMAP_DIR = PROJECT_DIR / ".roadmap"
|
|
25
42
|
|
|
26
43
|
MAX_QA_RETRIES = 3
|
|
27
44
|
MAX_EXPANSION_LOOPS = 10
|
|
@@ -173,31 +190,48 @@ def cmd_stop(args):
|
|
|
173
190
|
print(f"🛑 STOP file written for run '{args.run}'.")
|
|
174
191
|
|
|
175
192
|
|
|
193
|
+
def _add_common(parser):
|
|
194
|
+
parser.add_argument(
|
|
195
|
+
"--project-dir",
|
|
196
|
+
default=None,
|
|
197
|
+
help="Project the factory state belongs to (default: IUMBTEMS_PROJECT_DIR env, else repo root)",
|
|
198
|
+
)
|
|
199
|
+
|
|
200
|
+
|
|
176
201
|
def main():
|
|
202
|
+
global FACTORY_DIR, ROADMAP_DIR
|
|
177
203
|
ap = argparse.ArgumentParser(description="Factory run-state helper")
|
|
178
204
|
sub = ap.add_subparsers(dest="command", required=True)
|
|
179
205
|
|
|
180
206
|
p = sub.add_parser("init")
|
|
181
207
|
p.add_argument("--run", required=True)
|
|
208
|
+
_add_common(p)
|
|
182
209
|
p = sub.add_parser("phase-add")
|
|
183
210
|
p.add_argument("--run", required=True)
|
|
184
211
|
p.add_argument("--phase", required=True)
|
|
185
212
|
p.add_argument("--goal", required=True)
|
|
186
213
|
p.add_argument("--accept", default="")
|
|
214
|
+
_add_common(p)
|
|
187
215
|
p = sub.add_parser("qa-record")
|
|
188
216
|
p.add_argument("--run", required=True)
|
|
189
217
|
p.add_argument("--phase", required=True)
|
|
190
218
|
p.add_argument("--seat", required=True)
|
|
191
219
|
p.add_argument("--verdict", required=True)
|
|
192
220
|
p.add_argument("--reason", default="")
|
|
221
|
+
_add_common(p)
|
|
193
222
|
p = sub.add_parser("expansion")
|
|
194
223
|
p.add_argument("--run", required=True)
|
|
195
224
|
p.add_argument("--loops", type=int, required=True)
|
|
196
225
|
p.add_argument("--max-loops", type=int, default=MAX_EXPANSION_LOOPS)
|
|
226
|
+
_add_common(p)
|
|
197
227
|
p = sub.add_parser("stop")
|
|
198
228
|
p.add_argument("--run", required=True)
|
|
229
|
+
_add_common(p)
|
|
199
230
|
|
|
200
231
|
args = ap.parse_args()
|
|
232
|
+
root = resolve_project_root(args.project_dir)
|
|
233
|
+
FACTORY_DIR = root / ".factory"
|
|
234
|
+
ROADMAP_DIR = root / ".roadmap"
|
|
201
235
|
code = {
|
|
202
236
|
"init": cmd_init,
|
|
203
237
|
"phase-add": cmd_phase_add,
|
|
@@ -3,5 +3,5 @@ prompt = """Run the IUMBTEMS domain-expansion loop for {{args}} loops (max 10).
|
|
|
3
3
|
|
|
4
4
|
Bypasses per-loop gates; stops on count OR .factory/STOP file OR user kill.
|
|
5
5
|
Each loop: agents propose direction, quick swarm check, implement, dual-QA verify.
|
|
6
|
-
Enforce via
|
|
6
|
+
Enforce via the iumbtems_factory tool (command: expansion, with loops/max_loops).
|
|
7
7
|
"""
|
|
@@ -3,6 +3,6 @@ prompt = """Run the IUMBTEMS coding-factory Manager loop for: {{args}}.
|
|
|
3
3
|
|
|
4
4
|
1. Grill until .factory/frontier.json is settled (max 5 swarm cycles per gate); explicit user approve advances each gate.
|
|
5
5
|
2. Per gate: python3 runner/research_swarm.py --mode brainstorm plus --mode darkharvest (mock-first), then synthesize .roadmap/<phase>/ GOAL.md + dossier.json.
|
|
6
|
-
3. Programmer subagent per phase; qa-a plus qa-b per phase; retries via
|
|
6
|
+
3. Programmer subagent per phase; qa-a plus qa-b per phase; retries via the iumbtems_factory tool (phase-add / qa-record; 3 failures escalate). Never invoke factory helper scripts by relative path.
|
|
7
7
|
Report: .roadmap/ phases plus .factory/state.json.
|
|
8
8
|
"""
|
|
@@ -102,6 +102,7 @@ export const IUMBTEMS_TOOL_NAMES = [
|
|
|
102
102
|
'iumbtems_oss_scout',
|
|
103
103
|
'iumbtems_brainstorm',
|
|
104
104
|
'iumbtems_darkharvest',
|
|
105
|
+
'iumbtems_factory',
|
|
105
106
|
'iumbtems_verify_quote',
|
|
106
107
|
'iumbtems_socratic_frontier',
|
|
107
108
|
'iumbtems_reindex_claims',
|
|
@@ -291,6 +292,31 @@ const TOOL_CATALOG = [
|
|
|
291
292
|
required: ['objective'],
|
|
292
293
|
},
|
|
293
294
|
},
|
|
295
|
+
{
|
|
296
|
+
name: 'iumbtems_factory',
|
|
297
|
+
description:
|
|
298
|
+
'Drive factory run state: init / phase-add / qa-record / expansion / stop. State goes to <project>/.factory and <project>/.roadmap; no helper-script path needed.',
|
|
299
|
+
input: {
|
|
300
|
+
type: 'object',
|
|
301
|
+
properties: {
|
|
302
|
+
command: {
|
|
303
|
+
type: 'string',
|
|
304
|
+
enum: ['init', 'phase-add', 'qa-record', 'expansion', 'stop'],
|
|
305
|
+
},
|
|
306
|
+
run: { type: 'string', description: 'Factory run name' },
|
|
307
|
+
phase: { type: 'string', description: 'Phase id (e.g. 01-auth)' },
|
|
308
|
+
goal: { type: 'string' },
|
|
309
|
+
accept: { type: 'string', description: 'Semicolon-separated acceptance criteria' },
|
|
310
|
+
seat: { type: 'string', description: 'QA seat (qa-a | qa-b)' },
|
|
311
|
+
verdict: { type: 'string', enum: ['pass', 'fail', 'conditional'] },
|
|
312
|
+
reason: { type: 'string' },
|
|
313
|
+
loops: { type: 'integer', description: 'Expansion loop count (1-10)' },
|
|
314
|
+
max_loops: { type: 'integer' },
|
|
315
|
+
project_dir: { type: 'string', description: 'Project root (default: IUMBTEMS_PROJECT_DIR or cwd)' },
|
|
316
|
+
},
|
|
317
|
+
required: ['command'],
|
|
318
|
+
},
|
|
319
|
+
},
|
|
294
320
|
{
|
|
295
321
|
name: 'iumbtems_verify_quote',
|
|
296
322
|
description:
|
|
@@ -551,6 +577,7 @@ export const OPENCODE_COMMANDS = [
|
|
|
551
577
|
description: 'Coding-factory Manager loop: grill-gated phased build with programmer spawns and dual QA',
|
|
552
578
|
usage: '/factory <product-arena>',
|
|
553
579
|
agent: 'manager',
|
|
580
|
+
subagent: false,
|
|
554
581
|
subtask: false,
|
|
555
582
|
template: [
|
|
556
583
|
'Run the IUMBTEMS coding-factory Manager loop as the manager agent.',
|
|
@@ -558,7 +585,7 @@ export const OPENCODE_COMMANDS = [
|
|
|
558
585
|
'If $ARGUMENTS is empty, ask the user what to build first; never proceed on placeholder input.',
|
|
559
586
|
'1. Grill the user until .factory/frontier.json is settled (max 5 brainstorm+darkharvest swarm cycles per gate); explicit user approve advances each gate.',
|
|
560
587
|
'2. Per gate run iumbtems_brainstorm and iumbtems_darkharvest (mock_mode only for dry runs), then synthesize .roadmap/<phase>/ GOAL.md + dossier.json (goal/evidence/acceptance/brief/verdict/hashes; every claim needs a VERIFIED hash).',
|
|
561
|
-
'3. Spawn the programmer subagent per phase with the phase dossier (cite phase hashes); run qa-a and qa-b (diverged prompts) per phase; track retries with
|
|
588
|
+
'3. Spawn the programmer subagent per phase with the phase dossier (cite phase hashes); run qa-a and qa-b (diverged prompts) per phase; track retries with the iumbtems_factory tool (command: phase-add / qa-record; 3 failures escalate to manager). Never invoke factory helper scripts by relative path.',
|
|
562
589
|
'4. Manager tiebreaks QA disagreements; explicit user sign-off closes each phase.',
|
|
563
590
|
].join('\n'),
|
|
564
591
|
},
|
|
@@ -567,12 +594,13 @@ export const OPENCODE_COMMANDS = [
|
|
|
567
594
|
description: 'Autonomous agent-guided self-improvement loop over the codebase (count-flagged)',
|
|
568
595
|
usage: '/domainexpansion <n>',
|
|
569
596
|
agent: 'manager',
|
|
597
|
+
subagent: false,
|
|
570
598
|
subtask: false,
|
|
571
599
|
template: [
|
|
572
600
|
'Run the IUMBTEMS domain-expansion loop as the manager agent.',
|
|
573
601
|
'Loops: $ARGUMENTS (integer count, max 10)',
|
|
574
602
|
'If $ARGUMENTS is not a positive integer, ask the user for the loop count first.',
|
|
575
|
-
'1. Bypass per-loop gates; stop on count OR .factory/STOP file OR user kill, whichever first (enforce via
|
|
603
|
+
'1. Bypass per-loop gates; stop on count OR .factory/STOP file OR user kill, whichever first (enforce via the iumbtems_factory tool, command: expansion, with loops/max_loops).',
|
|
576
604
|
'2. Each loop: agents propose direction, quick iumbtems_brainstorm/iumbtems_darkharvest check, implement via programmer spawn, dual-QA verify.',
|
|
577
605
|
'3. All expansion proposals carry the strict VERIFIED evidence bar; log every loop to .factory/state.json.',
|
|
578
606
|
].join('\n'),
|
|
@@ -593,7 +621,14 @@ export function commandCatalog() {
|
|
|
593
621
|
if (!cmd?.name) continue;
|
|
594
622
|
out[cmd.name] = { description: cmd.description, template: cmd.template };
|
|
595
623
|
if (cmd.agent) out[cmd.name].agent = cmd.agent;
|
|
624
|
+
if (cmd.subagent !== undefined) out[cmd.name].subagent = cmd.subagent;
|
|
596
625
|
if (cmd.subtask !== undefined) out[cmd.name].subtask = cmd.subtask;
|
|
626
|
+
if (out[cmd.name].subagent === undefined && out[cmd.name].subtask !== undefined) {
|
|
627
|
+
out[cmd.name].subagent = out[cmd.name].subtask;
|
|
628
|
+
}
|
|
629
|
+
if (out[cmd.name].subtask === undefined && out[cmd.name].subagent !== undefined) {
|
|
630
|
+
out[cmd.name].subtask = out[cmd.name].subagent;
|
|
631
|
+
}
|
|
597
632
|
}
|
|
598
633
|
return out;
|
|
599
634
|
}
|
|
@@ -863,7 +898,12 @@ async function registerHostCommands(host) {
|
|
|
863
898
|
name: cmd.name,
|
|
864
899
|
description: cmd.description,
|
|
865
900
|
...(cmd.agent ? { agent: cmd.agent } : {}),
|
|
866
|
-
...(cmd.
|
|
901
|
+
...(cmd.subagent !== undefined || cmd.subtask !== undefined
|
|
902
|
+
? {
|
|
903
|
+
subagent: cmd.subagent !== undefined ? cmd.subagent : cmd.subtask,
|
|
904
|
+
subtask: cmd.subtask !== undefined ? cmd.subtask : cmd.subagent,
|
|
905
|
+
}
|
|
906
|
+
: {}),
|
|
867
907
|
execute: async (input) => {
|
|
868
908
|
const args = input?.prompt?.text || '';
|
|
869
909
|
const prompt = (typeof input?.prompt === 'object' && input?.prompt !== null) ? input.prompt : {};
|
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
# IUMBTEMS REPOSITORY FACT-CHECKING REVIEW PROMPT FOR ANTIGRAVITY
|
|
2
|
+
|
|
3
|
+
> **Role & Protocol**: You are operating as the **Epistemic Swarm Auditor** within Google AntiGravity. Your mission is to conduct a rigorous, evidentiary, live web-search fact-checking review of the **IUMBTEMS** repository (`https://github.com/Heretek-AI/IUMBTEMS` / local workspace).
|
|
4
|
+
>
|
|
5
|
+
> You are governed by the **Epistemic Integrity Protocol** defined in this repository: your internal parametric memory is strictly quarantined as untrusted heuristic guidance. You are prohibited from presenting unverified parametric recollections as established empirical facts. Every factual assertion must be verified against live reality via web searching and URL extraction.
|
|
6
|
+
|
|
7
|
+
---
|
|
8
|
+
|
|
9
|
+
## 1. MANDATORY TAGGING TAXONOMY
|
|
10
|
+
|
|
11
|
+
Every factual statement, version assertion, API specification claim, or benchmark metric MUST carry an explicit epistemic tag:
|
|
12
|
+
|
|
13
|
+
- `[VERIFIED: <URL | "verbatim quote excerpt">]`
|
|
14
|
+
- Backed by live web content fetched during this session via `search_web` or `read_url_content`.
|
|
15
|
+
- Must include the exact URL and an exact verbatim quote substring from the retrieved page.
|
|
16
|
+
- `[INFERRED: <Parent Tags> -> <Deductive Reasoning>]`
|
|
17
|
+
- Deductive conclusion derived directly from cited `[VERIFIED]` premises.
|
|
18
|
+
- `[HYPOTHESIS: <Measurable Falsification Condition>]`
|
|
19
|
+
- Unverified projection or speculation; requires an empirical test that would disprove it.
|
|
20
|
+
- `[NEGATIVE_KNOWLEDGE: <Search Query>]`
|
|
21
|
+
- Rigorous confirmation that an exhaustive live web search yielded zero supporting evidence.
|
|
22
|
+
|
|
23
|
+
---
|
|
24
|
+
|
|
25
|
+
## 2. REPOSITORY AUDIT TARGETS
|
|
26
|
+
|
|
27
|
+
Inspect the local codebase (`README.md`, `AGENTS.md`, `MARKETPLACE.md`, `package.json`, `prompts/`, and `plugins/`) and fact-check the following four empirical domains using `search_web` and `read_url_content`:
|
|
28
|
+
|
|
29
|
+
### Domain 1: Package Registry & Release Veracity
|
|
30
|
+
- **Claims in Repo**: The project is published on npm as `@heretek-ai/epistemic-swarm` under Apache-2.0, providing binary `iumbtems`.
|
|
31
|
+
- **Fact-Checking Action**:
|
|
32
|
+
- Search `https://registry.npmjs.org/@heretek-ai%2Fepistemic-swarm` or search the web for npm package `@heretek-ai/epistemic-swarm`.
|
|
33
|
+
- Verify: Does the package exist on npm? What is the latest published version? Does it match `package.json`? Does it expose the `iumbtems` binary?
|
|
34
|
+
|
|
35
|
+
### Domain 2: Peer Agent Harness Compatibility Claims
|
|
36
|
+
- **Claims in Repo**:
|
|
37
|
+
1. **OpenCode V2**: Claims plugin integration in `plugins/opencode/index.js`, using slash commands (`/swarm`, `/grill`, `/audit`), agent profiles (`config/opencode-snippet.json`), and notes that OpenCode lacks a pre-execution webfetch hook.
|
|
38
|
+
2. **Pi & OMP**: Claims native install via `pi install npm:@heretek-ai/epistemic-swarm` and `omp install npm:@heretek-ai/epistemic-swarm` with command blocks in `package.json`.
|
|
39
|
+
3. **Claude Code**: Claims marketplace support via `claude plugin marketplace add Heretek-AI/IUMBTEMS` and `.claude-plugin/marketplace.json`.
|
|
40
|
+
- **Fact-Checking Action**:
|
|
41
|
+
- Search official documentation and repos for OpenCode (`opencode.ai`), Pi (`pi.dev`), and Claude Code plugin specs.
|
|
42
|
+
- Verify: Are the configuration formats, CLI command syntaxes, and plugin manifest schemas valid against current upstream specifications?
|
|
43
|
+
|
|
44
|
+
### Domain 3: Cited Academic & Algorithmic Benchmarks
|
|
45
|
+
- **Claims in Repo** (found in `prompts/base_epistemic_system.md`, `prompts/orchestrator.md`, etc.):
|
|
46
|
+
1. Llama-3-70B context window (131,072 tokens) and GQA across 8 KV heads (`arXiv:2407.21783`).
|
|
47
|
+
2. Tip5 hash vs. Poseidon hash SNARK witness generation benchmarks.
|
|
48
|
+
3. Zero-dependency Raft consensus and DuckDuckGo Lite HTML scraping behavior.
|
|
49
|
+
- **Fact-Checking Action**:
|
|
50
|
+
- Search arXiv and web sources for the cited papers and benchmarks.
|
|
51
|
+
- Verify: Are the numbers, citations, and DOIs authentic, or were any placeholder/synthetic examples presented as real citations?
|
|
52
|
+
|
|
53
|
+
### Domain 4: License, Security & Dependency Invariants
|
|
54
|
+
- **Claims in Repo**: Apache-2.0 clean-room licensing, permissive-only vendoring, no AGPL/GPL contamination.
|
|
55
|
+
- **Fact-Checking Action**:
|
|
56
|
+
- Inspect dependencies in `package.json` and python scripts.
|
|
57
|
+
- Verify license status of key referenced dependencies via web search.
|
|
58
|
+
|
|
59
|
+
---
|
|
60
|
+
|
|
61
|
+
## 3. SINGLE-SESSION AGENTIC EXECUTION WORKFLOW
|
|
62
|
+
|
|
63
|
+
Execute the fact-checking mission autonomously in three sequential phases:
|
|
64
|
+
|
|
65
|
+
```
|
|
66
|
+
[Phase 1: Alpha (Affirmative)] ──> [Phase 2: Beta (Adversary)] ──> [Phase 3: Epistemic Auditor]
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
### Phase 1: Alpha (The Affirmative Grounding)
|
|
70
|
+
1. Read the local claims in `README.md` and `package.json` using `view_file`.
|
|
71
|
+
2. Formulate targeted search queries and execute them using `search_web`.
|
|
72
|
+
3. Fetch full source pages using `read_url_content` for key results.
|
|
73
|
+
4. Extract verbatim evidence excerpts corroborating the repository's claims.
|
|
74
|
+
5. Tag all confirmed claims with `[VERIFIED: <URL | "quote">]`.
|
|
75
|
+
|
|
76
|
+
### Phase 2: Beta (The Adversarial Red Team)
|
|
77
|
+
1. Execute inverted and adversarial queries to hunt for discrepancies, breaking changes, and invalid claims:
|
|
78
|
+
- `"<package> deprecated"`, `"<command> error"`, `"<paper> critique"`
|
|
79
|
+
- Check if any upstream harness APIs (OpenCode, Pi, Claude Code) have deprecated or altered the interfaces IUMBTEMS relies on.
|
|
80
|
+
- Hunt for missing packages, unfulfilled promises, or exaggerated marketing statements.
|
|
81
|
+
2. If claimed features or benchmarks cannot be found online, log them as `[NEGATIVE_KNOWLEDGE: <query>]`.
|
|
82
|
+
3. If an assertion is disproven by current live documentation, document the exact contradiction.
|
|
83
|
+
|
|
84
|
+
### Phase 3: Epistemic Auditor & Mathematical Synthesis
|
|
85
|
+
1. Perform character-for-character verification between extracted quotes and source URLs.
|
|
86
|
+
2. Compute the **Epistemic Score**:
|
|
87
|
+
$$\mathcal{E} = \frac{1.0 \times N_{\text{verified}} + 0.5 \times N_{\text{neg\_knowledge}} - 2.5 \times N_{\text{rejected}}}{N_{\text{verified}} + N_{\text{inferred}} + N_{\text{hypothesis}} + N_{\text{rejected}}}$$
|
|
88
|
+
*(Threshold: $\mathcal{E} \ge 0.65$ to certify empirical grounding).*
|
|
89
|
+
3. Compute the **Divergence Score**:
|
|
90
|
+
$$D = \frac{|\text{Contradicted Claims}|}{|\text{Total Scope Claims}|}$$
|
|
91
|
+
4. Output the final synthesis report as a Markdown Artifact or structured response.
|
|
92
|
+
|
|
93
|
+
---
|
|
94
|
+
|
|
95
|
+
## 4. OUTPUT FORMAT SPECIFICATION
|
|
96
|
+
|
|
97
|
+
Your final output must follow this structure:
|
|
98
|
+
|
|
99
|
+
```markdown
|
|
100
|
+
# Epistemic Fact-Checking Audit: IUMBTEMS Repository
|
|
101
|
+
|
|
102
|
+
## Executive Summary
|
|
103
|
+
- **Overall Verdict**: [CERTIFIED (Score >= 0.65) | AUDIT_WARNING: LOW_EMPIRICAL_GROUNDING]
|
|
104
|
+
- **Epistemic Score ($\mathcal{E}$)**: `<score>`
|
|
105
|
+
- **Dialectic Divergence ($D$)**: `<score>`
|
|
106
|
+
- **Total Claims Audited**: `<count>` (Verified: `<count>`, Rejected: `<count>`, Negative Knowledge: `<count>`)
|
|
107
|
+
|
|
108
|
+
---
|
|
109
|
+
|
|
110
|
+
## Evidentiary Audit Ledger
|
|
111
|
+
|
|
112
|
+
### 1. Package & Distribution Veracity
|
|
113
|
+
- Claim: ...
|
|
114
|
+
- Status: [VERIFIED | REJECTED | NEGATIVE_KNOWLEDGE]
|
|
115
|
+
- Evidence: [VERIFIED: https://... | "Verbatim quote..."]
|
|
116
|
+
- Notes: ...
|
|
117
|
+
|
|
118
|
+
### 2. Multi-Harness Compatibility (OpenCode, Pi, OMP, Claude Code)
|
|
119
|
+
...
|
|
120
|
+
|
|
121
|
+
### 3. Academic & Benchmark Integrity
|
|
122
|
+
...
|
|
123
|
+
|
|
124
|
+
### 4. License & Contamination Safety
|
|
125
|
+
...
|
|
126
|
+
|
|
127
|
+
---
|
|
128
|
+
|
|
129
|
+
## Divergence & Contradiction Matrix
|
|
130
|
+
| Dimension | Affirmative Claim (Alpha) | Adversarial Finding (Beta) | Adjudicated Truth |
|
|
131
|
+
| :--- | :--- | :--- | :--- |
|
|
132
|
+
| ... | ... | ... | ... |
|
|
133
|
+
|
|
134
|
+
---
|
|
135
|
+
|
|
136
|
+
## Actionable Remediations
|
|
137
|
+
1. [P0/P1/P2] Specific changes required in `README.md`, `package.json`, or code to align with verified live reality.
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
---
|
|
141
|
+
|
|
142
|
+
## 5. EXECUTION DIRECTIVE
|
|
143
|
+
Begin Phase 1 immediately: inspect local claims, invoke `search_web` to verify npm and harness registries, then proceed through Phase 2 and Phase 3 without stopping.
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
package/runner/mcp_server.py
CHANGED
|
@@ -192,6 +192,87 @@ def _handle_darkharvest(args: Dict[str, Any]) -> str:
|
|
|
192
192
|
return _run_swarm_mode("darkharvest", args)
|
|
193
193
|
|
|
194
194
|
|
|
195
|
+
def _handle_factory(args: Dict[str, Any]) -> str:
|
|
196
|
+
"""Drive factory run state (init / phase-add / qa-record / expansion / stop).
|
|
197
|
+
|
|
198
|
+
Wraps skills/factory/scripts/factory.py so agents never need a filesystem
|
|
199
|
+
path to the helper: the npm-installed toolchain lives outside the project,
|
|
200
|
+
and relative `skills/...` paths broke in consuming projects (observed
|
|
201
|
+
live: the manager agent ran `find / -name factory.py`).
|
|
202
|
+
"""
|
|
203
|
+
import subprocess
|
|
204
|
+
|
|
205
|
+
command = str(args.get("command") or "").strip()
|
|
206
|
+
if command not in ("init", "phase-add", "qa-record", "expansion", "stop"):
|
|
207
|
+
raise ValueError(
|
|
208
|
+
"command must be one of: init, phase-add, qa-record, expansion, stop"
|
|
209
|
+
)
|
|
210
|
+
script = Path(PROJECT_ROOT) / "skills" / "factory" / "scripts" / "factory.py"
|
|
211
|
+
if not script.is_file():
|
|
212
|
+
raise FileNotFoundError(f"factory helper not found at {script}")
|
|
213
|
+
|
|
214
|
+
project = (
|
|
215
|
+
args.get("project_dir") or os.environ.get("IUMBTEMS_PROJECT_DIR") or os.getcwd()
|
|
216
|
+
)
|
|
217
|
+
cmd = [sys.executable, str(script), command, "--project-dir", str(project)]
|
|
218
|
+
if command == "init":
|
|
219
|
+
cmd += ["--run", str(args.get("run") or "")]
|
|
220
|
+
elif command == "phase-add":
|
|
221
|
+
cmd += [
|
|
222
|
+
"--run",
|
|
223
|
+
str(args.get("run") or ""),
|
|
224
|
+
"--phase",
|
|
225
|
+
str(args.get("phase") or ""),
|
|
226
|
+
"--goal",
|
|
227
|
+
str(args.get("goal") or ""),
|
|
228
|
+
"--accept",
|
|
229
|
+
str(args.get("accept") or ""),
|
|
230
|
+
]
|
|
231
|
+
elif command == "qa-record":
|
|
232
|
+
cmd += [
|
|
233
|
+
"--run",
|
|
234
|
+
str(args.get("run") or ""),
|
|
235
|
+
"--phase",
|
|
236
|
+
str(args.get("phase") or ""),
|
|
237
|
+
"--seat",
|
|
238
|
+
str(args.get("seat") or ""),
|
|
239
|
+
"--verdict",
|
|
240
|
+
str(args.get("verdict") or ""),
|
|
241
|
+
]
|
|
242
|
+
if args.get("reason"):
|
|
243
|
+
cmd += ["--reason", str(args["reason"])]
|
|
244
|
+
elif command == "expansion":
|
|
245
|
+
cmd += [
|
|
246
|
+
"--run",
|
|
247
|
+
str(args.get("run") or ""),
|
|
248
|
+
"--loops",
|
|
249
|
+
str(int(args.get("loops") or 1)),
|
|
250
|
+
]
|
|
251
|
+
if args.get("max_loops"):
|
|
252
|
+
cmd += ["--max-loops", str(int(args["max_loops"]))]
|
|
253
|
+
elif command == "stop":
|
|
254
|
+
cmd += ["--run", str(args.get("run") or "")]
|
|
255
|
+
|
|
256
|
+
proc = subprocess.run(
|
|
257
|
+
cmd,
|
|
258
|
+
capture_output=True,
|
|
259
|
+
text=True,
|
|
260
|
+
shell=False,
|
|
261
|
+
stdin=subprocess.DEVNULL,
|
|
262
|
+
cwd=str(project),
|
|
263
|
+
)
|
|
264
|
+
payload = {
|
|
265
|
+
"status": "ok" if proc.returncode == 0 else "error",
|
|
266
|
+
"command": command,
|
|
267
|
+
"returncode": proc.returncode,
|
|
268
|
+
"project_dir": str(project),
|
|
269
|
+
"output": (proc.stdout or proc.stderr or "").strip()[-4000:],
|
|
270
|
+
}
|
|
271
|
+
if proc.returncode == 2:
|
|
272
|
+
payload["status"] = "escalated"
|
|
273
|
+
return _tool_text(payload)
|
|
274
|
+
|
|
275
|
+
|
|
195
276
|
def _handle_verify_quote(args: Dict[str, Any]) -> str:
|
|
196
277
|
"""Verify a verbatim quote against the content-addressed source cache."""
|
|
197
278
|
from skills.research_cache.hasher import SourceHasher
|
|
@@ -541,6 +622,46 @@ def build_tools() -> List[ToolSpec]:
|
|
|
541
622
|
},
|
|
542
623
|
handler=_handle_darkharvest,
|
|
543
624
|
),
|
|
625
|
+
ToolSpec(
|
|
626
|
+
name="iumbtems_factory",
|
|
627
|
+
description="Drive factory run state: init / phase-add / qa-record / expansion / stop. State goes to <project>/.factory and <project>/.roadmap; no filesystem path to helper scripts required.",
|
|
628
|
+
input_schema={
|
|
629
|
+
"type": "object",
|
|
630
|
+
"properties": {
|
|
631
|
+
"command": {
|
|
632
|
+
"type": "string",
|
|
633
|
+
"enum": ["init", "phase-add", "qa-record", "expansion", "stop"],
|
|
634
|
+
},
|
|
635
|
+
"run": {"type": "string", "description": "Factory run name"},
|
|
636
|
+
"phase": {
|
|
637
|
+
"type": "string",
|
|
638
|
+
"description": "Phase id (e.g. 01-auth)",
|
|
639
|
+
},
|
|
640
|
+
"goal": {"type": "string"},
|
|
641
|
+
"accept": {
|
|
642
|
+
"type": "string",
|
|
643
|
+
"description": "Semicolon-separated acceptance criteria",
|
|
644
|
+
},
|
|
645
|
+
"seat": {"type": "string", "description": "QA seat (qa-a | qa-b)"},
|
|
646
|
+
"verdict": {
|
|
647
|
+
"type": "string",
|
|
648
|
+
"enum": ["pass", "fail", "conditional"],
|
|
649
|
+
},
|
|
650
|
+
"reason": {"type": "string"},
|
|
651
|
+
"loops": {
|
|
652
|
+
"type": "integer",
|
|
653
|
+
"description": "Expansion loop count (1-10)",
|
|
654
|
+
},
|
|
655
|
+
"max_loops": {"type": "integer"},
|
|
656
|
+
"project_dir": {
|
|
657
|
+
"type": "string",
|
|
658
|
+
"description": "Project root (default: IUMBTEMS_PROJECT_DIR or cwd)",
|
|
659
|
+
},
|
|
660
|
+
},
|
|
661
|
+
"required": ["command"],
|
|
662
|
+
},
|
|
663
|
+
handler=_handle_factory,
|
|
664
|
+
),
|
|
544
665
|
ToolSpec(
|
|
545
666
|
name="iumbtems_verify_quote",
|
|
546
667
|
description="Audit a verbatim citation against the SHA-256 source cache (.research/sources/<hash>.md).",
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
"""Tests for the factory loop: run-state helper, QA retry bounds, plugin surface."""
|
|
3
3
|
|
|
4
4
|
import json
|
|
5
|
+
import os
|
|
5
6
|
import subprocess
|
|
6
7
|
import sys
|
|
7
8
|
import tempfile
|
|
@@ -164,6 +165,52 @@ class TestFactoryHelper(unittest.TestCase):
|
|
|
164
165
|
self.assertIn("./skills/factory", pkg["pi"]["skills"])
|
|
165
166
|
self.assertIn("./skills/factory", pkg["omp"]["skills"])
|
|
166
167
|
|
|
168
|
+
def test_project_dir_env_resolution(self):
|
|
169
|
+
"""Factory state must land in the target project, never the checkout."""
|
|
170
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
171
|
+
env = dict(os.environ, IUMBTEMS_PROJECT_DIR=tmp)
|
|
172
|
+
r = subprocess.run(
|
|
173
|
+
[
|
|
174
|
+
sys.executable,
|
|
175
|
+
"skills/factory/scripts/factory.py",
|
|
176
|
+
"init",
|
|
177
|
+
"--run",
|
|
178
|
+
"env-run",
|
|
179
|
+
],
|
|
180
|
+
capture_output=True,
|
|
181
|
+
text=True,
|
|
182
|
+
cwd=str(PROJECT_ROOT),
|
|
183
|
+
env=env,
|
|
184
|
+
)
|
|
185
|
+
self.assertEqual(r.returncode, 0, r.stderr)
|
|
186
|
+
self.assertTrue(
|
|
187
|
+
(Path(tmp) / ".factory" / "env-run" / "state.json").exists()
|
|
188
|
+
)
|
|
189
|
+
|
|
190
|
+
def test_mcp_factory_tool_project_dir(self):
|
|
191
|
+
"""iumbtems_factory drives state in the given project (no script path)."""
|
|
192
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
193
|
+
r = subprocess.run(
|
|
194
|
+
[
|
|
195
|
+
sys.executable,
|
|
196
|
+
"runner/mcp_server.py",
|
|
197
|
+
"call",
|
|
198
|
+
"iumbtems_factory",
|
|
199
|
+
json.dumps(
|
|
200
|
+
{"command": "init", "run": "mcp-run", "project_dir": tmp}
|
|
201
|
+
),
|
|
202
|
+
],
|
|
203
|
+
capture_output=True,
|
|
204
|
+
text=True,
|
|
205
|
+
cwd=str(PROJECT_ROOT),
|
|
206
|
+
)
|
|
207
|
+
self.assertEqual(r.returncode, 0, r.stderr)
|
|
208
|
+
payload = json.loads(r.stdout)
|
|
209
|
+
self.assertEqual(payload["status"], "ok")
|
|
210
|
+
self.assertTrue(
|
|
211
|
+
(Path(tmp) / ".factory" / "mcp-run" / "state.json").exists()
|
|
212
|
+
)
|
|
213
|
+
|
|
167
214
|
def test_snippet_factory_roster(self):
|
|
168
215
|
with open(PROJECT_ROOT / "config" / "opencode-snippet.json") as f:
|
|
169
216
|
snippet = json.load(f)
|
|
@@ -191,8 +238,12 @@ class TestFactoryHelper(unittest.TestCase):
|
|
|
191
238
|
self.assertEqual(res.returncode, 0, res.stderr)
|
|
192
239
|
data = last_json_object(res.stdout)
|
|
193
240
|
self.assertEqual(data["factory"]["agent"], "manager")
|
|
241
|
+
self.assertEqual(data["factory"]["subagent"], False)
|
|
242
|
+
self.assertEqual(data["factory"]["subtask"], False)
|
|
194
243
|
self.assertIn("$ARGUMENTS", data["factory"]["template"])
|
|
195
244
|
self.assertEqual(data["expansion"]["agent"], "manager")
|
|
245
|
+
self.assertEqual(data["expansion"]["subagent"], False)
|
|
246
|
+
self.assertEqual(data["expansion"]["subtask"], False)
|
|
196
247
|
|
|
197
248
|
|
|
198
249
|
if __name__ == "__main__":
|
|
@@ -214,7 +214,7 @@ class TestOpenCodeCommandCatalog(unittest.TestCase):
|
|
|
214
214
|
self.assertEqual(res.returncode, 0, f"tool map test failed: {res.stderr}")
|
|
215
215
|
data = last_json_object(res.stdout)
|
|
216
216
|
self.assertEqual(sorted(data["keys"]), sorted(data["canonical"]))
|
|
217
|
-
self.assertEqual(len(data["keys"]),
|
|
217
|
+
self.assertEqual(len(data["keys"]), 15)
|
|
218
218
|
self.assertTrue(all(data["ok"]))
|
|
219
219
|
|
|
220
220
|
|
|
@@ -618,7 +618,7 @@ class TestOpenCodeV2Transforms(unittest.TestCase):
|
|
|
618
618
|
self.assertIn(c, data["commands"])
|
|
619
619
|
self.assertNotIn("goal", data["commands"])
|
|
620
620
|
self.assertEqual(data["cmdExec"], "function")
|
|
621
|
-
self.assertEqual(len(data["tools"]),
|
|
621
|
+
self.assertEqual(len(data["tools"]), 15)
|
|
622
622
|
self.assertIn("iumbtems_brainstorm", data["tools"])
|
|
623
623
|
self.assertIn("iumbtems_darkharvest", data["tools"])
|
|
624
624
|
self.assertEqual(data["toolExec"], "function")
|
|
@@ -381,6 +381,7 @@ In our experiments, the 70B parameter model was trained on 15.0 trillion tokens.
|
|
|
381
381
|
"iumbtems_oss_scout",
|
|
382
382
|
"iumbtems_brainstorm",
|
|
383
383
|
"iumbtems_darkharvest",
|
|
384
|
+
"iumbtems_factory",
|
|
384
385
|
"iumbtems_verify_quote",
|
|
385
386
|
"iumbtems_socratic_frontier",
|
|
386
387
|
"iumbtems_reindex_claims",
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
package/skills/factory/SKILL.md
CHANGED
|
@@ -13,6 +13,16 @@ description: Coding-factory Manager loop. Use when user invokes /factory or /dom
|
|
|
13
13
|
- **qa-a / qa-b** (subagents): same model, DIVERGED prompts (functional-correctness vs adversarial edge-case). Read-only plus test execution; never edit.
|
|
14
14
|
- **researcher** = existing `iumbtems_brainstorm` + `iumbtems_darkharvest` swarms (no new profile).
|
|
15
15
|
|
|
16
|
+
## 1b. Driving run state (no filesystem paths)
|
|
17
|
+
|
|
18
|
+
Use the **`iumbtems_factory` MCP tool** for all run-state changes — never a
|
|
19
|
+
relative `skills/factory/scripts/factory.py` path (the toolchain lives in the
|
|
20
|
+
npm cache in consuming projects, and relative paths broke live: the manager
|
|
21
|
+
agent ran `find / -name factory.py`). The tool wraps the same helper and
|
|
22
|
+
resolves the project via `project_dir` argument, `IUMBTEMS_PROJECT_DIR`, or
|
|
23
|
+
the session cwd. It supports `init`, `phase-add`, `qa-record`, `expansion`,
|
|
24
|
+
`stop`, and returns `status: escalated` (exit 2) on the 3rd QA failure.
|
|
25
|
+
|
|
16
26
|
## 2. Gate protocol (max 5 swarm cycles per gate)
|
|
17
27
|
|
|
18
28
|
1. Grill until `.factory/frontier.json` settled (grilling skill).
|
|
@@ -2,8 +2,13 @@
|
|
|
2
2
|
"""
|
|
3
3
|
Factory run-state helper: phase dossiers, QA retry bounds, expansion loop guard.
|
|
4
4
|
|
|
5
|
-
All state lives under
|
|
6
|
-
to
|
|
5
|
+
All state lives under <project>/.factory/ (gitignored runtime state). Phase
|
|
6
|
+
output goes to <project>/.roadmap/<phase>/. Evidence stays in <project>/
|
|
7
|
+
.research/. Read-only w.r.t. repo code.
|
|
8
|
+
|
|
9
|
+
Project directory resolution: --project-dir > IUMBTEMS_PROJECT_DIR env > the
|
|
10
|
+
toolchain repo root (local-dev default). A consuming project must never leak
|
|
11
|
+
state into the IUMBTEMS checkout.
|
|
7
12
|
|
|
8
13
|
Usage:
|
|
9
14
|
python3 skills/factory/scripts/factory.py init --run <name>
|
|
@@ -15,13 +20,25 @@ Usage:
|
|
|
15
20
|
|
|
16
21
|
import argparse
|
|
17
22
|
import json
|
|
23
|
+
import os
|
|
18
24
|
import sys
|
|
19
25
|
from datetime import datetime, timezone
|
|
20
26
|
from pathlib import Path
|
|
21
27
|
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
28
|
+
REPO_ROOT = Path(__file__).resolve().parent.parent.parent.parent
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def resolve_project_root(cli_value=None):
|
|
32
|
+
"""--project-dir > IUMBTEMS_PROJECT_DIR > repo root (local-dev default)."""
|
|
33
|
+
for candidate in (cli_value, os.environ.get("IUMBTEMS_PROJECT_DIR")):
|
|
34
|
+
if candidate and Path(candidate).is_dir():
|
|
35
|
+
return Path(candidate).resolve()
|
|
36
|
+
return REPO_ROOT
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
PROJECT_DIR = resolve_project_root()
|
|
40
|
+
FACTORY_DIR = PROJECT_DIR / ".factory"
|
|
41
|
+
ROADMAP_DIR = PROJECT_DIR / ".roadmap"
|
|
25
42
|
|
|
26
43
|
MAX_QA_RETRIES = 3
|
|
27
44
|
MAX_EXPANSION_LOOPS = 10
|
|
@@ -173,31 +190,48 @@ def cmd_stop(args):
|
|
|
173
190
|
print(f"🛑 STOP file written for run '{args.run}'.")
|
|
174
191
|
|
|
175
192
|
|
|
193
|
+
def _add_common(parser):
|
|
194
|
+
parser.add_argument(
|
|
195
|
+
"--project-dir",
|
|
196
|
+
default=None,
|
|
197
|
+
help="Project the factory state belongs to (default: IUMBTEMS_PROJECT_DIR env, else repo root)",
|
|
198
|
+
)
|
|
199
|
+
|
|
200
|
+
|
|
176
201
|
def main():
|
|
202
|
+
global FACTORY_DIR, ROADMAP_DIR
|
|
177
203
|
ap = argparse.ArgumentParser(description="Factory run-state helper")
|
|
178
204
|
sub = ap.add_subparsers(dest="command", required=True)
|
|
179
205
|
|
|
180
206
|
p = sub.add_parser("init")
|
|
181
207
|
p.add_argument("--run", required=True)
|
|
208
|
+
_add_common(p)
|
|
182
209
|
p = sub.add_parser("phase-add")
|
|
183
210
|
p.add_argument("--run", required=True)
|
|
184
211
|
p.add_argument("--phase", required=True)
|
|
185
212
|
p.add_argument("--goal", required=True)
|
|
186
213
|
p.add_argument("--accept", default="")
|
|
214
|
+
_add_common(p)
|
|
187
215
|
p = sub.add_parser("qa-record")
|
|
188
216
|
p.add_argument("--run", required=True)
|
|
189
217
|
p.add_argument("--phase", required=True)
|
|
190
218
|
p.add_argument("--seat", required=True)
|
|
191
219
|
p.add_argument("--verdict", required=True)
|
|
192
220
|
p.add_argument("--reason", default="")
|
|
221
|
+
_add_common(p)
|
|
193
222
|
p = sub.add_parser("expansion")
|
|
194
223
|
p.add_argument("--run", required=True)
|
|
195
224
|
p.add_argument("--loops", type=int, required=True)
|
|
196
225
|
p.add_argument("--max-loops", type=int, default=MAX_EXPANSION_LOOPS)
|
|
226
|
+
_add_common(p)
|
|
197
227
|
p = sub.add_parser("stop")
|
|
198
228
|
p.add_argument("--run", required=True)
|
|
229
|
+
_add_common(p)
|
|
199
230
|
|
|
200
231
|
args = ap.parse_args()
|
|
232
|
+
root = resolve_project_root(args.project_dir)
|
|
233
|
+
FACTORY_DIR = root / ".factory"
|
|
234
|
+
ROADMAP_DIR = root / ".roadmap"
|
|
201
235
|
code = {
|
|
202
236
|
"init": cmd_init,
|
|
203
237
|
"phase-add": cmd_phase_add,
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|