@heretek-ai/epistemic-swarm 0.4.3 → 0.4.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/install.sh +25 -0
- package/package.json +1 -1
- package/plugins/antigravity/plugin.json +1 -1
- package/plugins/gemini/gemini-extension.json +1 -1
- package/plugins/opencode/index.js +44 -15
- package/plugins/opencode/tui.js +40 -18
- package/runner/__pycache__/__init__.cpython-311.pyc +0 -0
- package/runner/__pycache__/auctioneer.cpython-311.pyc +0 -0
- package/runner/__pycache__/auditor_engine.cpython-311.pyc +0 -0
- package/runner/__pycache__/claim_store.cpython-311.pyc +0 -0
- package/runner/__pycache__/claim_witness.cpython-311.pyc +0 -0
- package/runner/__pycache__/living_dossiers.cpython-311.pyc +0 -0
- package/runner/__pycache__/mcp_protocol.cpython-311.pyc +0 -0
- package/runner/__pycache__/mcp_server.cpython-311.pyc +0 -0
- package/runner/__pycache__/pcrb.cpython-311.pyc +0 -0
- package/runner/__pycache__/pcrb_verify.cpython-311.pyc +0 -0
- package/runner/__pycache__/refinement.cpython-311.pyc +0 -0
- package/runner/__pycache__/research_swarm.cpython-311.pyc +0 -0
- package/runner/__pycache__/state_machine.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_auction_order.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_bet1_spike.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claim_store.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claim_witness.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_domain_packs.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_fleet_seam.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_living_dossiers.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_mcp_server.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_opencode_ux.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_pcrb.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_refinement.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_swarm.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_webcache.cpython-311.pyc +0 -0
- package/runner/tests/test_bet1_spike.py +95 -0
- package/runner/tests/test_opencode_ux.py +76 -0
- package/scripts/__pycache__/bet1_advisory_spike.cpython-311.pyc +0 -0
- package/scripts/bet1_advisory_spike.py +426 -0
- package/skills/research_cache/__pycache__/__init__.cpython-311.pyc +0 -0
- package/skills/research_cache/__pycache__/hasher.cpython-311.pyc +0 -0
- package/skills/swarm_config/__pycache__/__init__.cpython-311.pyc +0 -0
- package/skills/swarm_config/__pycache__/configure.cpython-311.pyc +0 -0
package/install.sh
CHANGED
|
@@ -177,6 +177,31 @@ try:
|
|
|
177
177
|
print(f" + Added agent: {name}")
|
|
178
178
|
else:
|
|
179
179
|
print(f" ℹ️ Agent already configured: {name}")
|
|
180
|
+
|
|
181
|
+
# Skills: snippet paths are repo-relative ("./skills/x") which breaks
|
|
182
|
+
# from ~/.config — merge as absolute paths, repairing relative entries.
|
|
183
|
+
import os as _os
|
|
184
|
+
wanted = []
|
|
185
|
+
for sp in snippet.get("skills", {}).get("paths", []):
|
|
186
|
+
wanted.append(sp if _os.path.isabs(sp) else str(repo / sp.lstrip("./")))
|
|
187
|
+
skills_cfg = cfg.setdefault("skills", {})
|
|
188
|
+
paths = skills_cfg.setdefault("paths", [])
|
|
189
|
+
# Repair repo-relative entries in place, then append missing absolutes.
|
|
190
|
+
for i, existing in enumerate(list(paths)):
|
|
191
|
+
for abs_sp in wanted:
|
|
192
|
+
if existing == abs_sp:
|
|
193
|
+
break
|
|
194
|
+
rel = "./" + str(Path(abs_sp).relative_to(repo)) if str(abs_sp).startswith(str(repo)) else None
|
|
195
|
+
if rel and existing == rel:
|
|
196
|
+
paths[i] = abs_sp
|
|
197
|
+
print(f" ~ Repaired relative skill path: {rel} -> {abs_sp}")
|
|
198
|
+
break
|
|
199
|
+
for abs_sp in wanted:
|
|
200
|
+
if abs_sp not in paths:
|
|
201
|
+
paths.append(abs_sp)
|
|
202
|
+
print(f" + Added skill path: {abs_sp}")
|
|
203
|
+
else:
|
|
204
|
+
print(f" ℹ️ Skill path already configured: {abs_sp}")
|
|
180
205
|
except Exception as e:
|
|
181
206
|
print(f" ⚠️ Non-critical warning merging agents: {e}")
|
|
182
207
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@heretek-ai/epistemic-swarm",
|
|
3
|
-
"version": "0.4.
|
|
3
|
+
"version": "0.4.4",
|
|
4
4
|
"description": "IUMBTEMS: I Use My Brain To Express My Self — High-Integrity Dialectic Research Agent Harness for Claude Code, OpenCode V2, Pi, OMP (oh-my-pi), Gemini CLI, Codex CLI, and AntiGravity",
|
|
5
5
|
"main": "bin/cli.js",
|
|
6
6
|
"bin": {
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"$schema": "https://antigravity.google/schemas/v1/plugin.json",
|
|
3
3
|
"name": "epistemic-swarm",
|
|
4
|
-
"version": "0.4.
|
|
4
|
+
"version": "0.4.4",
|
|
5
5
|
"description": "IUMBTEMS Epistemic Swarm: dialectic research, code audit, OSS scout, and lateral brainstorming for AntiGravity.",
|
|
6
6
|
"author": "Heretek AI",
|
|
7
7
|
"license": "Apache-2.0",
|
|
@@ -11,7 +11,7 @@
|
|
|
11
11
|
* catalog. Schemas mirror runner/mcp_server.py build_tools() — keep in sync.
|
|
12
12
|
*/
|
|
13
13
|
|
|
14
|
-
import {
|
|
14
|
+
import { spawn } from 'child_process';
|
|
15
15
|
import { readFileSync } from 'fs';
|
|
16
16
|
import path from 'path';
|
|
17
17
|
import { fileURLToPath } from 'url';
|
|
@@ -30,21 +30,40 @@ try {
|
|
|
30
30
|
// keep fallback; version is informational only
|
|
31
31
|
}
|
|
32
32
|
|
|
33
|
-
/**
|
|
33
|
+
/**
|
|
34
|
+
* Uniform dispatch: one-shot MCP call, return OpenCode's {content, status}.
|
|
35
|
+
* Async (non-blocking spawn) so long swarms/audits never freeze the host
|
|
36
|
+
* event loop; cancellation remains the host's prerogative.
|
|
37
|
+
*/
|
|
34
38
|
function callMcp(tool, args = {}, cwd) {
|
|
35
|
-
|
|
36
|
-
'
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
39
|
+
return new Promise((resolve) => {
|
|
40
|
+
let stdout = '';
|
|
41
|
+
let stderr = '';
|
|
42
|
+
let settled = false;
|
|
43
|
+
const done = (content, status) => {
|
|
44
|
+
if (settled) return;
|
|
45
|
+
settled = true;
|
|
46
|
+
resolve({ content, status });
|
|
47
|
+
};
|
|
48
|
+
let child;
|
|
49
|
+
try {
|
|
50
|
+
child = spawn(
|
|
51
|
+
'python3',
|
|
52
|
+
[MCP_SERVER, 'call', tool, JSON.stringify(args || {})],
|
|
53
|
+
{
|
|
54
|
+
cwd: cwd || process.cwd(),
|
|
55
|
+
env: { ...process.env, PYTHONPATH: PKG_ROOT },
|
|
56
|
+
}
|
|
57
|
+
);
|
|
58
|
+
} catch (err) {
|
|
59
|
+
done(String((err && err.message) || err), 'error');
|
|
60
|
+
return;
|
|
42
61
|
}
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
};
|
|
62
|
+
child.stdout.on('data', (d) => { stdout += String(d); });
|
|
63
|
+
child.stderr.on('data', (d) => { stderr += String(d); });
|
|
64
|
+
child.on('error', (err) => done(String((err && err.message) || err), 'error'));
|
|
65
|
+
child.on('close', (code) => done(stdout || stderr, code === 0 ? 'success' : 'error'));
|
|
66
|
+
});
|
|
48
67
|
}
|
|
49
68
|
|
|
50
69
|
/** Catalog mirrors runner/mcp_server.py build_tools(). */
|
|
@@ -351,9 +370,11 @@ export const OPENCODE_COMMANDS = [
|
|
|
351
370
|
{
|
|
352
371
|
name: 'swarm',
|
|
353
372
|
description: 'Run IUMBTEMS dialectic research swarm on an objective',
|
|
373
|
+
usage: '/swarm <objective>',
|
|
354
374
|
template: [
|
|
355
375
|
'Run an IUMBTEMS dialectic research swarm.',
|
|
356
376
|
'Objective: $ARGUMENTS',
|
|
377
|
+
'If $ARGUMENTS is empty, ask the user for the research objective first; never call the tool with placeholder, empty, or literal "<objective>" arguments.',
|
|
357
378
|
'1. If the objective is ambiguous, frame it first via iumbtems_socratic_frontier.',
|
|
358
379
|
'2. Execute iumbtems_swarm_research with {"objective": "<objective>"} (pass "mock_mode": true only for dry runs).',
|
|
359
380
|
'3. Summarize .research/final_synthesis.md, preserving [VERIFIED:<hash>] pointers.',
|
|
@@ -362,6 +383,7 @@ export const OPENCODE_COMMANDS = [
|
|
|
362
383
|
{
|
|
363
384
|
name: 'grill',
|
|
364
385
|
description: 'Launch Socratic grilling and decision tree frontier exploration',
|
|
386
|
+
usage: '/grill [--objective <text>]',
|
|
365
387
|
template: [
|
|
366
388
|
'Launch Socratic grilling on the decision frontier.',
|
|
367
389
|
'Objective: $ARGUMENTS (may be empty to inspect the current frontier)',
|
|
@@ -372,6 +394,7 @@ export const OPENCODE_COMMANDS = [
|
|
|
372
394
|
{
|
|
373
395
|
name: 'swarm-config',
|
|
374
396
|
description: 'Inspect or update Epistemic Swarm parameters (engine, depth, mode)',
|
|
397
|
+
usage: '/swarm-config [--engine <e>] [--depth <d>] [--mode <m>] [--show]',
|
|
375
398
|
template: [
|
|
376
399
|
'Inspect or update the Epistemic Swarm configuration.',
|
|
377
400
|
'Arguments: $ARGUMENTS (may be empty to inspect current settings)',
|
|
@@ -381,9 +404,11 @@ export const OPENCODE_COMMANDS = [
|
|
|
381
404
|
{
|
|
382
405
|
name: 'audit',
|
|
383
406
|
description: 'Run dialectic codebase architectural and security audit with line-level proof',
|
|
407
|
+
usage: '/audit <target_path_or_scope>',
|
|
384
408
|
template: [
|
|
385
409
|
'Run an IUMBTEMS dialectic codebase audit (structural architect vs adversarial red-teamer).',
|
|
386
410
|
'Target: $ARGUMENTS (path, component, or empty for full-repository architecture and vulnerability audit)',
|
|
411
|
+
'If $ARGUMENTS names a path that does not exist, ask the user to clarify the target first; never audit a placeholder path.',
|
|
387
412
|
'1. Execute iumbtems_code_audit with {"target": "<target>"} (pass "mock_mode": true only for dry runs).',
|
|
388
413
|
'2. Summarize .research/code_audit_report.md with line-level proof pointers.',
|
|
389
414
|
].join('\n'),
|
|
@@ -391,9 +416,11 @@ export const OPENCODE_COMMANDS = [
|
|
|
391
416
|
{
|
|
392
417
|
name: 'scout',
|
|
393
418
|
description: 'Scout open-source libraries, audit copyleft licenses, generate clean-room blueprints',
|
|
419
|
+
usage: '/scout <feature_or_algorithm>',
|
|
394
420
|
template: [
|
|
395
421
|
'Scout open-source solutions for the requested capability.',
|
|
396
422
|
'Feature: $ARGUMENTS',
|
|
423
|
+
'If $ARGUMENTS is empty, ask the user for the feature or algorithm first; never call the tool with placeholder, empty, or literal "<feature>" arguments.',
|
|
397
424
|
'1. Execute iumbtems_oss_scout with {"feature": "<feature>"} (pass "mock_mode": true only for dry runs).',
|
|
398
425
|
'2. Report mature candidates, GPL/AGPL copyleft risks, and the clean-room blueprint in .research/oss_scout_report.md.',
|
|
399
426
|
].join('\n'),
|
|
@@ -401,6 +428,7 @@ export const OPENCODE_COMMANDS = [
|
|
|
401
428
|
{
|
|
402
429
|
name: 'brainstorming',
|
|
403
430
|
description: 'Lateral brainstorming: novel feature vectors, paradigm inversions, falsifiable spikes',
|
|
431
|
+
usage: '/brainstorming <ambiguous-prompt>',
|
|
404
432
|
template: [
|
|
405
433
|
'Run lateral brainstorming (divergent what-if ideation, never bug-fix lists).',
|
|
406
434
|
'Prompt: $ARGUMENTS (defaults to "Where do we go from here?" when empty)',
|
|
@@ -411,6 +439,7 @@ export const OPENCODE_COMMANDS = [
|
|
|
411
439
|
{
|
|
412
440
|
name: 'brainstorm',
|
|
413
441
|
description: 'Alias for /brainstorming',
|
|
442
|
+
usage: '/brainstorm <ambiguous-prompt>',
|
|
414
443
|
template: [
|
|
415
444
|
'Alias for /brainstorming: run lateral brainstorming (divergent what-if ideation, never bug-fix lists).',
|
|
416
445
|
'Prompt: $ARGUMENTS (defaults to "Where do we go from here?" when empty)',
|
|
@@ -640,7 +669,7 @@ export function createOpenCodePlugin(context = {}) {
|
|
|
640
669
|
updates.divergence_threshold = opts.divergence_threshold;
|
|
641
670
|
}
|
|
642
671
|
if (Object.keys(updates).length > 0) {
|
|
643
|
-
callMcp('iumbtems_config', updates, appContext?.cwd);
|
|
672
|
+
await callMcp('iumbtems_config', updates, appContext?.cwd);
|
|
644
673
|
}
|
|
645
674
|
}
|
|
646
675
|
// Phase C: idle staleness nudge (opt out with {staleness_nudge: false}).
|
package/plugins/opencode/tui.js
CHANGED
|
@@ -15,6 +15,7 @@ import { readFileSync, readdirSync, statSync } from 'fs';
|
|
|
15
15
|
import path from 'path';
|
|
16
16
|
import { createElement, insert, setProp } from '@opentui/solid';
|
|
17
17
|
import { createEffect, createSignal, onCleanup } from 'solid-js';
|
|
18
|
+
import { OPENCODE_COMMANDS } from './index.js';
|
|
18
19
|
|
|
19
20
|
const PLUGIN_ID = 'heretek.iumbtems.epistemic-swarm.tui';
|
|
20
21
|
const REFRESH_MS = 5000;
|
|
@@ -172,26 +173,47 @@ function SwarmSidebar(api, root, sessionID) {
|
|
|
172
173
|
}
|
|
173
174
|
|
|
174
175
|
function SwarmKeymapLayer(api) {
|
|
176
|
+
const toast = (title, message) => {
|
|
177
|
+
try {
|
|
178
|
+
api.ui.toast.show({ title, message, variant: 'info', duration: 4000 });
|
|
179
|
+
} catch {
|
|
180
|
+
/* toast is best-effort */
|
|
181
|
+
}
|
|
182
|
+
};
|
|
183
|
+
const commands = [
|
|
184
|
+
{
|
|
185
|
+
id: 'iumbtems.swarm-status',
|
|
186
|
+
title: 'Swarm status',
|
|
187
|
+
description: 'Show IUMBTEMS workspace status (mode, reports, frontier, STALE claims)',
|
|
188
|
+
group: 'IUMBTEMS',
|
|
189
|
+
palette: true,
|
|
190
|
+
run: () => {
|
|
191
|
+
const root = resolveRoot(api);
|
|
192
|
+
toast('IUMBTEMS', summarize(readSwarmStatus(root)));
|
|
193
|
+
},
|
|
194
|
+
},
|
|
195
|
+
];
|
|
196
|
+
// Every slash command gets a palette entry (usage toast; the host has no
|
|
197
|
+
// programmatic slash-invoke API, so discovery + usage guidance is the win).
|
|
198
|
+
for (const cmd of OPENCODE_COMMANDS || []) {
|
|
199
|
+
if (!cmd || !cmd.name) continue;
|
|
200
|
+
commands.push({
|
|
201
|
+
id: `iumbtems.command.${cmd.name}`,
|
|
202
|
+
title: `/${cmd.name}`,
|
|
203
|
+
description: cmd.description || cmd.name,
|
|
204
|
+
group: 'IUMBTEMS',
|
|
205
|
+
palette: true,
|
|
206
|
+
run: () => {
|
|
207
|
+
toast(
|
|
208
|
+
`IUMBTEMS /${cmd.name}`,
|
|
209
|
+
`${cmd.usage || cmd.description || ''} — type /${cmd.name} in the prompt to run.`
|
|
210
|
+
);
|
|
211
|
+
},
|
|
212
|
+
});
|
|
213
|
+
}
|
|
175
214
|
api.keymap.layer(() => ({
|
|
176
215
|
mode: 'global',
|
|
177
|
-
commands
|
|
178
|
-
{
|
|
179
|
-
id: 'iumbtems.swarm-status',
|
|
180
|
-
title: 'Swarm status',
|
|
181
|
-
description: 'Show IUMBTEMS workspace status (mode, reports, frontier, STALE claims)',
|
|
182
|
-
group: 'IUMBTEMS',
|
|
183
|
-
palette: true,
|
|
184
|
-
run: () => {
|
|
185
|
-
const root = resolveRoot(api);
|
|
186
|
-
api.ui.toast.show({
|
|
187
|
-
title: 'IUMBTEMS',
|
|
188
|
-
message: summarize(readSwarmStatus(root)),
|
|
189
|
-
variant: 'info',
|
|
190
|
-
duration: 4000,
|
|
191
|
-
});
|
|
192
|
-
},
|
|
193
|
-
},
|
|
194
|
-
],
|
|
216
|
+
commands,
|
|
195
217
|
}));
|
|
196
218
|
return null;
|
|
197
219
|
}
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""CI-safe tests for the Bet-1 advisory spike harness (no LLM, no network)."""
|
|
3
|
+
|
|
4
|
+
import json
|
|
5
|
+
import subprocess
|
|
6
|
+
import sys
|
|
7
|
+
import tempfile
|
|
8
|
+
import unittest
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent
|
|
12
|
+
if str(PROJECT_ROOT) not in sys.path:
|
|
13
|
+
sys.path.insert(0, str(PROJECT_ROOT))
|
|
14
|
+
|
|
15
|
+
from scripts.bet1_advisory_spike import ( # noqa: E402
|
|
16
|
+
advisory_comment,
|
|
17
|
+
generate,
|
|
18
|
+
score_fixture,
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class TestBet1Harness(unittest.TestCase):
|
|
23
|
+
def test_generate_counts_and_layout(self):
|
|
24
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
25
|
+
manifest = generate(Path(tmp))
|
|
26
|
+
self.assertEqual(len(manifest), 20)
|
|
27
|
+
defective = [e for e in manifest if not e["clean"]]
|
|
28
|
+
clean = [e for e in manifest if e["clean"]]
|
|
29
|
+
self.assertEqual(len(defective), 16)
|
|
30
|
+
self.assertEqual(len(clean), 4)
|
|
31
|
+
for entry in manifest:
|
|
32
|
+
self.assertTrue((Path(tmp) / entry["id"] / entry["file"]).exists())
|
|
33
|
+
|
|
34
|
+
def test_generate_deterministic(self):
|
|
35
|
+
with tempfile.TemporaryDirectory() as t1, tempfile.TemporaryDirectory() as t2:
|
|
36
|
+
m1 = generate(Path(t1))
|
|
37
|
+
m2 = generate(Path(t2))
|
|
38
|
+
self.assertEqual(m1, m2)
|
|
39
|
+
|
|
40
|
+
def test_scorer_hit_and_tolerance(self):
|
|
41
|
+
entry = {"id": "x", "file": "users.py", "line": 6, "clean": False}
|
|
42
|
+
self.assertTrue(score_fixture(entry, "see users.py:6 SQLi")["hit"])
|
|
43
|
+
self.assertTrue(score_fixture(entry, "users.py L8 concat")["hit"])
|
|
44
|
+
self.assertFalse(score_fixture(entry, "users.py:40 unrelated")["hit"])
|
|
45
|
+
self.assertFalse(score_fixture(entry, "no file cited")["hit"])
|
|
46
|
+
|
|
47
|
+
def test_scorer_clean_flags_fp(self):
|
|
48
|
+
entry = {"id": "c", "file": "util.py", "line": None, "clean": True}
|
|
49
|
+
self.assertTrue(score_fixture(entry, "issue in util.py:2")["false_positive"])
|
|
50
|
+
self.assertFalse(
|
|
51
|
+
score_fixture(entry, "all clear, nothing found")["false_positive"]
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
def test_advisory_is_non_blocking_shape(self):
|
|
55
|
+
comment = advisory_comment("bet1-01", "findings text")
|
|
56
|
+
self.assertIn("non-blocking", comment)
|
|
57
|
+
self.assertIn("bet1-01", comment)
|
|
58
|
+
self.assertIn("findings text", comment)
|
|
59
|
+
|
|
60
|
+
def test_cli_generate_and_score_exit_zero(self):
|
|
61
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
62
|
+
for argv in (["--generate", "--dir", tmp], ["--score", "--dir", tmp]):
|
|
63
|
+
res = subprocess.run(
|
|
64
|
+
[
|
|
65
|
+
sys.executable,
|
|
66
|
+
str(PROJECT_ROOT / "scripts" / "bet1_advisory_spike.py"),
|
|
67
|
+
]
|
|
68
|
+
+ argv,
|
|
69
|
+
capture_output=True,
|
|
70
|
+
text=True,
|
|
71
|
+
timeout=120,
|
|
72
|
+
)
|
|
73
|
+
self.assertEqual(res.returncode, 0, res.stderr)
|
|
74
|
+
summary = json.loads(
|
|
75
|
+
subprocess.run(
|
|
76
|
+
[
|
|
77
|
+
sys.executable,
|
|
78
|
+
str(PROJECT_ROOT / "scripts" / "bet1_advisory_spike.py"),
|
|
79
|
+
"--score",
|
|
80
|
+
"--dir",
|
|
81
|
+
tmp,
|
|
82
|
+
],
|
|
83
|
+
capture_output=True,
|
|
84
|
+
text=True,
|
|
85
|
+
timeout=120,
|
|
86
|
+
).stdout
|
|
87
|
+
)
|
|
88
|
+
self.assertEqual(summary["fixtures"], 20)
|
|
89
|
+
# No advisories written yet -> all misses, no false positives.
|
|
90
|
+
self.assertEqual(summary["hits"], 0)
|
|
91
|
+
self.assertEqual(summary["false_positives"], 0)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
if __name__ == "__main__":
|
|
95
|
+
unittest.main()
|
|
@@ -75,6 +75,39 @@ class TestOpenCodeCommandCatalog(unittest.TestCase):
|
|
|
75
75
|
self.assertIn(c, data["registered"])
|
|
76
76
|
self.assertTrue(data["templates"][c])
|
|
77
77
|
|
|
78
|
+
def test_empty_arg_guards(self):
|
|
79
|
+
res = run_node(
|
|
80
|
+
"""
|
|
81
|
+
import { OPENCODE_COMMANDS } from "./plugins/opencode/index.js";
|
|
82
|
+
const guarded = OPENCODE_COMMANDS.filter(c => /empty/i.test(c.template) && /ask the user/i.test(c.template)).map(c => c.name);
|
|
83
|
+
const withUsage = OPENCODE_COMMANDS.filter(c => !!c.usage).map(c => c.name);
|
|
84
|
+
console.log(JSON.stringify({guarded, withUsage, total: OPENCODE_COMMANDS.length}));
|
|
85
|
+
"""
|
|
86
|
+
)
|
|
87
|
+
self.assertEqual(res.returncode, 0, f"guard test failed: {res.stderr}")
|
|
88
|
+
data = last_json_object(res.stdout)
|
|
89
|
+
for c in ["swarm", "scout", "audit"]:
|
|
90
|
+
self.assertIn(c, data["guarded"])
|
|
91
|
+
self.assertEqual(len(data["withUsage"]), data["total"])
|
|
92
|
+
|
|
93
|
+
def test_async_execute_resolves(self):
|
|
94
|
+
res = run_node(
|
|
95
|
+
"""
|
|
96
|
+
import plugin from "./plugins/opencode/index.js";
|
|
97
|
+
const shell = await plugin.server();
|
|
98
|
+
let ticked = false;
|
|
99
|
+
const timer = setInterval(() => { ticked = true; }, 5);
|
|
100
|
+
const r = await shell.tool.iumbtems_config.execute({});
|
|
101
|
+
clearInterval(timer);
|
|
102
|
+
console.log(JSON.stringify({status: r.status, hasContent: typeof r.content === "string" && r.content.length > 0, loopAlive: ticked}));
|
|
103
|
+
"""
|
|
104
|
+
)
|
|
105
|
+
self.assertEqual(res.returncode, 0, f"async execute test failed: {res.stderr}")
|
|
106
|
+
data = last_json_object(res.stdout)
|
|
107
|
+
self.assertEqual(data["status"], "success")
|
|
108
|
+
self.assertTrue(data["hasContent"])
|
|
109
|
+
self.assertTrue(data["loopAlive"])
|
|
110
|
+
|
|
78
111
|
def test_registration_never_overwrites(self):
|
|
79
112
|
res = run_node(
|
|
80
113
|
"""
|
|
@@ -180,6 +213,13 @@ class TestInstallOpenCode(unittest.TestCase):
|
|
|
180
213
|
cmd = cfg["mcp"]["iumbtems"]["command"]
|
|
181
214
|
self.assertTrue(Path(cmd[-1]).is_absolute())
|
|
182
215
|
self.assertEqual(cmd[-1], str(PROJECT_ROOT / "runner" / "mcp_server.py"))
|
|
216
|
+
# Skills merged as absolute repo paths (relative entries repaired).
|
|
217
|
+
skill_paths = cfg.get("skills", {}).get("paths", [])
|
|
218
|
+
self.assertTrue(skill_paths)
|
|
219
|
+
for sp in skill_paths:
|
|
220
|
+
self.assertTrue(Path(sp).is_absolute(), f"relative skill path: {sp}")
|
|
221
|
+
self.assertIn(str(PROJECT_ROOT / "skills" / "grilling"), skill_paths)
|
|
222
|
+
self.assertFalse(any(not Path(sp).is_absolute() for sp in skill_paths))
|
|
183
223
|
# Second run changes nothing.
|
|
184
224
|
before = oc_path.read_bytes()
|
|
185
225
|
second = self._run_install(home)
|
|
@@ -285,6 +325,42 @@ class TestOpenCodeTui(unittest.TestCase):
|
|
|
285
325
|
self.assertEqual(data["suspect"], 1)
|
|
286
326
|
self.assertEqual(data["requeued"], 2)
|
|
287
327
|
|
|
328
|
+
def test_palette_entries_for_all_commands(self):
|
|
329
|
+
res = run_node(
|
|
330
|
+
"""
|
|
331
|
+
import { setupTui } from "./plugins/opencode/tui.js";
|
|
332
|
+
import { OPENCODE_COMMANDS } from "./plugins/opencode/index.js";
|
|
333
|
+
const renders = [];
|
|
334
|
+
let factory = null;
|
|
335
|
+
const toasts = [];
|
|
336
|
+
const mock = {
|
|
337
|
+
theme: {}, directory: process.cwd(),
|
|
338
|
+
ui: { slot: (arg) => { renders.push(arg); return () => {}; }, toast: { show: (t) => toasts.push(t) }, router: { current: () => ({type: "other"}) } },
|
|
339
|
+
keymap: { layer: (fn) => { factory = fn; } }
|
|
340
|
+
};
|
|
341
|
+
setupTui(mock);
|
|
342
|
+
const appSlot = renders.find(r => r && r.append === "app");
|
|
343
|
+
if (!appSlot) throw new Error("app slot not registered");
|
|
344
|
+
appSlot.render();
|
|
345
|
+
const spec = factory();
|
|
346
|
+
const palette = spec.commands.filter(c => c.palette);
|
|
347
|
+
for (const c of palette) c.run();
|
|
348
|
+
console.log(JSON.stringify({
|
|
349
|
+
total: spec.commands.length,
|
|
350
|
+
palette: palette.length,
|
|
351
|
+
groups: [...new Set(palette.map(c => c.group))],
|
|
352
|
+
toasts: toasts.length,
|
|
353
|
+
expected: OPENCODE_COMMANDS.length + 1
|
|
354
|
+
}));
|
|
355
|
+
"""
|
|
356
|
+
)
|
|
357
|
+
self.assertEqual(res.returncode, 0, f"palette test failed: {res.stderr}")
|
|
358
|
+
data = last_json_object(res.stdout)
|
|
359
|
+
self.assertEqual(data["total"], data["expected"])
|
|
360
|
+
self.assertEqual(data["palette"], data["expected"])
|
|
361
|
+
self.assertEqual(data["groups"], ["IUMBTEMS"])
|
|
362
|
+
self.assertEqual(data["toasts"], data["expected"])
|
|
363
|
+
|
|
288
364
|
|
|
289
365
|
class TestOpenCodeLifecycle(unittest.TestCase):
|
|
290
366
|
@staticmethod
|
|
Binary file
|
|
@@ -0,0 +1,426 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Bet-1 advisory spike: proof-carrying review on a synthetic PR fixture set.
|
|
3
|
+
|
|
4
|
+
Advisory mode ONLY: findings are formatted as PR comments, never block, and
|
|
5
|
+
the process always exits 0. Measures: seeded-defect hit rate (finding cites
|
|
6
|
+
seeded file + line within +/-3) and clean-fixture false-positive rate.
|
|
7
|
+
|
|
8
|
+
python3 scripts/bet1_advisory_spike.py --generate --dir /tmp/bet1
|
|
9
|
+
python3 scripts/bet1_advisory_spike.py --run --mock --dir /tmp/bet1
|
|
10
|
+
python3 scripts/bet1_advisory_spike.py --run --live --only bet1-01,bet1-02 --dir /tmp/bet1
|
|
11
|
+
python3 scripts/bet1_advisory_spike.py --score --dir /tmp/bet1
|
|
12
|
+
|
|
13
|
+
Live runs invoke real audits (LLM spend); mock runs exercise dispatch only.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
import argparse
|
|
17
|
+
import json
|
|
18
|
+
import re
|
|
19
|
+
import subprocess
|
|
20
|
+
import sys
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
|
|
23
|
+
PROJECT_ROOT = Path(__file__).resolve().parent.parent
|
|
24
|
+
LINE_TOLERANCE = 3
|
|
25
|
+
|
|
26
|
+
# name -> (filename, lines, seeded_line_1based, description)
|
|
27
|
+
FIXTURES = {
|
|
28
|
+
"hardcoded-secret-1": (
|
|
29
|
+
"auth.py",
|
|
30
|
+
[
|
|
31
|
+
'API_KEY = "sk-live-51H7x9Q2mZ4v8K3n" # production key',
|
|
32
|
+
"",
|
|
33
|
+
"def headers():",
|
|
34
|
+
' return {"Authorization": f"Bearer {API_KEY}"}',
|
|
35
|
+
],
|
|
36
|
+
1,
|
|
37
|
+
"hardcoded production secret",
|
|
38
|
+
),
|
|
39
|
+
"hardcoded-secret-2": (
|
|
40
|
+
"config.py",
|
|
41
|
+
[
|
|
42
|
+
"import os",
|
|
43
|
+
"",
|
|
44
|
+
'DB_PASSWORD = "P@ssw0rd-2024-prod"',
|
|
45
|
+
"",
|
|
46
|
+
"def dsn():",
|
|
47
|
+
' return f"postgres://admin:{DB_PASSWORD}@db:5432/app"',
|
|
48
|
+
],
|
|
49
|
+
3,
|
|
50
|
+
"hardcoded database password",
|
|
51
|
+
),
|
|
52
|
+
"sqli-concat-1": (
|
|
53
|
+
"users.py",
|
|
54
|
+
[
|
|
55
|
+
"import sqlite3",
|
|
56
|
+
"",
|
|
57
|
+
"def get_user(db, username):",
|
|
58
|
+
" conn = sqlite3.connect(db)",
|
|
59
|
+
" cur = conn.cursor()",
|
|
60
|
+
' query = "SELECT * FROM users WHERE name = \'" + username + "\'"',
|
|
61
|
+
" cur.execute(query)",
|
|
62
|
+
" return cur.fetchall()",
|
|
63
|
+
],
|
|
64
|
+
6,
|
|
65
|
+
"SQL string concatenation with user input",
|
|
66
|
+
),
|
|
67
|
+
"sqli-concat-2": (
|
|
68
|
+
"orders.py",
|
|
69
|
+
[
|
|
70
|
+
"import sqlite3",
|
|
71
|
+
"",
|
|
72
|
+
"def get_orders(db, status, limit):",
|
|
73
|
+
" conn = sqlite3.connect(db)",
|
|
74
|
+
" sql = f\"SELECT * FROM orders WHERE status='{status}' LIMIT {limit}\"",
|
|
75
|
+
" return conn.execute(sql).fetchall()",
|
|
76
|
+
],
|
|
77
|
+
5,
|
|
78
|
+
"f-string SQL with user input",
|
|
79
|
+
),
|
|
80
|
+
"race-counter-1": (
|
|
81
|
+
"counter.py",
|
|
82
|
+
[
|
|
83
|
+
"import threading",
|
|
84
|
+
"",
|
|
85
|
+
"total = 0",
|
|
86
|
+
"",
|
|
87
|
+
"def add(n):",
|
|
88
|
+
" global total",
|
|
89
|
+
" total += n",
|
|
90
|
+
"",
|
|
91
|
+
"threads = [threading.Thread(target=add, args=(i,)) for i in range(100)]",
|
|
92
|
+
"[t.start() for t in threads]",
|
|
93
|
+
],
|
|
94
|
+
7,
|
|
95
|
+
"unsynchronized shared counter across threads",
|
|
96
|
+
),
|
|
97
|
+
"race-counter-2": (
|
|
98
|
+
"cache.py",
|
|
99
|
+
[
|
|
100
|
+
"import threading",
|
|
101
|
+
"",
|
|
102
|
+
"_cache = {}",
|
|
103
|
+
"",
|
|
104
|
+
"def get_or_load(key, loader):",
|
|
105
|
+
" if key not in _cache:",
|
|
106
|
+
" _cache[key] = loader()",
|
|
107
|
+
" return _cache[key]",
|
|
108
|
+
],
|
|
109
|
+
7,
|
|
110
|
+
"check-then-act race on shared dict",
|
|
111
|
+
),
|
|
112
|
+
"auth-bypass-1": (
|
|
113
|
+
"login.py",
|
|
114
|
+
[
|
|
115
|
+
"def is_admin(user):",
|
|
116
|
+
" return True",
|
|
117
|
+
"",
|
|
118
|
+
"def delete_account(user, target):",
|
|
119
|
+
" if is_admin(user):",
|
|
120
|
+
" return db.delete(target)",
|
|
121
|
+
" raise PermissionError()",
|
|
122
|
+
],
|
|
123
|
+
2,
|
|
124
|
+
"authorization check always returns True",
|
|
125
|
+
),
|
|
126
|
+
"auth-bypass-2": (
|
|
127
|
+
"views.py",
|
|
128
|
+
[
|
|
129
|
+
"def admin_panel(request):",
|
|
130
|
+
' if request.args.get("debug") == "1":',
|
|
131
|
+
" return render_admin()",
|
|
132
|
+
" if not request.user or not request.user.is_staff:",
|
|
133
|
+
" abort(403)",
|
|
134
|
+
" return render_admin()",
|
|
135
|
+
],
|
|
136
|
+
2,
|
|
137
|
+
"debug query param bypasses staff check",
|
|
138
|
+
),
|
|
139
|
+
"off-by-one-1": (
|
|
140
|
+
"paging.py",
|
|
141
|
+
[
|
|
142
|
+
"def page(items, n, size):",
|
|
143
|
+
" start = n * size",
|
|
144
|
+
" return items[start:start + size + 1]",
|
|
145
|
+
],
|
|
146
|
+
3,
|
|
147
|
+
"page slice returns size+1 items",
|
|
148
|
+
),
|
|
149
|
+
"off-by-one-2": (
|
|
150
|
+
"retry.py",
|
|
151
|
+
[
|
|
152
|
+
"def fetch_with_retry(url, tries=3):",
|
|
153
|
+
" for i in range(tries + 1):",
|
|
154
|
+
" try:",
|
|
155
|
+
" return http_get(url)",
|
|
156
|
+
" except NetError:",
|
|
157
|
+
" continue",
|
|
158
|
+
" raise GiveUp()",
|
|
159
|
+
],
|
|
160
|
+
2,
|
|
161
|
+
"loop runs tries+1 attempts, not tries",
|
|
162
|
+
),
|
|
163
|
+
"exec-input-1": (
|
|
164
|
+
"render.py",
|
|
165
|
+
[
|
|
166
|
+
"def render(template_name, context):",
|
|
167
|
+
' expr = "f\'" + open(template_name).read() + "\'"',
|
|
168
|
+
" return eval(expr, {}, context)",
|
|
169
|
+
],
|
|
170
|
+
3,
|
|
171
|
+
"eval on template file content with context",
|
|
172
|
+
),
|
|
173
|
+
"exec-input-2": (
|
|
174
|
+
"calc.py",
|
|
175
|
+
[
|
|
176
|
+
"def calculate(formula, variables):",
|
|
177
|
+
' return eval(formula, {"__builtins__": {}}, variables)',
|
|
178
|
+
],
|
|
179
|
+
2,
|
|
180
|
+
"eval on user-supplied formula string",
|
|
181
|
+
),
|
|
182
|
+
"weak-crypto-1": (
|
|
183
|
+
"passwords.py",
|
|
184
|
+
[
|
|
185
|
+
"import hashlib",
|
|
186
|
+
"",
|
|
187
|
+
"def store_pw(username, password):",
|
|
188
|
+
" digest = hashlib.md5(password.encode()).hexdigest()",
|
|
189
|
+
" db.save(username, digest)",
|
|
190
|
+
],
|
|
191
|
+
4,
|
|
192
|
+
"MD5 for password hashing, no salt",
|
|
193
|
+
),
|
|
194
|
+
"weak-crypto-2": (
|
|
195
|
+
"tokens.py",
|
|
196
|
+
[
|
|
197
|
+
"import random",
|
|
198
|
+
"",
|
|
199
|
+
"def new_token():",
|
|
200
|
+
' return "%016x" % random.getrandbits(64)',
|
|
201
|
+
],
|
|
202
|
+
4,
|
|
203
|
+
"Mersenne Twister for security tokens",
|
|
204
|
+
),
|
|
205
|
+
"path-traversal-1": (
|
|
206
|
+
"files.py",
|
|
207
|
+
[
|
|
208
|
+
"import os",
|
|
209
|
+
"",
|
|
210
|
+
'BASE = "/srv/uploads"',
|
|
211
|
+
"",
|
|
212
|
+
"def read_upload(name):",
|
|
213
|
+
" with open(os.path.join(BASE, name)) as f:",
|
|
214
|
+
" return f.read()",
|
|
215
|
+
],
|
|
216
|
+
6,
|
|
217
|
+
"unsanitized join allows ../ escape",
|
|
218
|
+
),
|
|
219
|
+
"path-traversal-2": (
|
|
220
|
+
"export.py",
|
|
221
|
+
[
|
|
222
|
+
"import os",
|
|
223
|
+
"import shutil",
|
|
224
|
+
"",
|
|
225
|
+
"def export_report(name, dest_dir):",
|
|
226
|
+
' src = os.path.join("/srv/reports", name)',
|
|
227
|
+
" return shutil.copy(src, dest_dir)",
|
|
228
|
+
],
|
|
229
|
+
5,
|
|
230
|
+
"unsanitized report name allows ../ escape",
|
|
231
|
+
),
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
CLEAN_FIXTURES = {
|
|
235
|
+
"clean-1": (
|
|
236
|
+
"util.py",
|
|
237
|
+
[
|
|
238
|
+
"def clamp(value, lo, hi):",
|
|
239
|
+
" return max(lo, min(hi, value))",
|
|
240
|
+
],
|
|
241
|
+
"no defect: pure bounds clamp",
|
|
242
|
+
),
|
|
243
|
+
"clean-2": (
|
|
244
|
+
"queries.py",
|
|
245
|
+
[
|
|
246
|
+
"import sqlite3",
|
|
247
|
+
"",
|
|
248
|
+
"def get_user(db, username):",
|
|
249
|
+
" conn = sqlite3.connect(db)",
|
|
250
|
+
" return conn.execute(",
|
|
251
|
+
' "SELECT * FROM users WHERE name = ?", (username,)',
|
|
252
|
+
" ).fetchall()",
|
|
253
|
+
],
|
|
254
|
+
"no defect: parameterized query",
|
|
255
|
+
),
|
|
256
|
+
"clean-3": (
|
|
257
|
+
"authn.py",
|
|
258
|
+
[
|
|
259
|
+
"import hashlib",
|
|
260
|
+
"import hmac",
|
|
261
|
+
"import secrets",
|
|
262
|
+
"",
|
|
263
|
+
"def check(password, stored):",
|
|
264
|
+
' digest = hashlib.pbkdf2_hmac("sha256", password.encode(), stored["salt"], 600_000)',
|
|
265
|
+
' return hmac.compare_digest(digest, stored["digest"])',
|
|
266
|
+
],
|
|
267
|
+
"no defect: salted KDF with constant-time compare",
|
|
268
|
+
),
|
|
269
|
+
"clean-4": (
|
|
270
|
+
"counter_safe.py",
|
|
271
|
+
[
|
|
272
|
+
"import threading",
|
|
273
|
+
"",
|
|
274
|
+
"_lock = threading.Lock()",
|
|
275
|
+
"total = 0",
|
|
276
|
+
"",
|
|
277
|
+
"def add(n):",
|
|
278
|
+
" global total",
|
|
279
|
+
" with _lock:",
|
|
280
|
+
" total += n",
|
|
281
|
+
],
|
|
282
|
+
"no defect: locked counter",
|
|
283
|
+
),
|
|
284
|
+
}
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
def generate(root):
|
|
288
|
+
root = Path(root)
|
|
289
|
+
manifest = []
|
|
290
|
+
for fid, (fname, lines, line, desc) in FIXTURES.items():
|
|
291
|
+
d = root / fid
|
|
292
|
+
d.mkdir(parents=True, exist_ok=True)
|
|
293
|
+
d.joinpath(fname).write_text("\n".join(lines) + "\n", encoding="utf-8")
|
|
294
|
+
manifest.append(
|
|
295
|
+
{
|
|
296
|
+
"id": fid,
|
|
297
|
+
"file": fname,
|
|
298
|
+
"line": line,
|
|
299
|
+
"class": fid.rsplit("-", 1)[0],
|
|
300
|
+
"defect": desc,
|
|
301
|
+
"clean": False,
|
|
302
|
+
}
|
|
303
|
+
)
|
|
304
|
+
for fid, (fname, lines, desc) in CLEAN_FIXTURES.items():
|
|
305
|
+
d = root / fid
|
|
306
|
+
d.mkdir(parents=True, exist_ok=True)
|
|
307
|
+
d.joinpath(fname).write_text("\n".join(lines) + "\n", encoding="utf-8")
|
|
308
|
+
manifest.append(
|
|
309
|
+
{
|
|
310
|
+
"id": fid,
|
|
311
|
+
"file": fname,
|
|
312
|
+
"line": None,
|
|
313
|
+
"class": "clean",
|
|
314
|
+
"defect": desc,
|
|
315
|
+
"clean": True,
|
|
316
|
+
}
|
|
317
|
+
)
|
|
318
|
+
root.joinpath("manifest.json").write_text(
|
|
319
|
+
json.dumps(manifest, indent=2), encoding="utf-8"
|
|
320
|
+
)
|
|
321
|
+
print(f"generated {len(manifest)} fixtures in {root}")
|
|
322
|
+
return manifest
|
|
323
|
+
|
|
324
|
+
|
|
325
|
+
def run_audit(target_dir, mock, timeout=600):
|
|
326
|
+
"""Invoke the real dispatch path. Advisory: findings only, never blocks."""
|
|
327
|
+
args = {"target": str(target_dir)}
|
|
328
|
+
if mock:
|
|
329
|
+
args["mock_mode"] = True
|
|
330
|
+
proc = subprocess.run(
|
|
331
|
+
[
|
|
332
|
+
sys.executable,
|
|
333
|
+
str(PROJECT_ROOT / "runner" / "mcp_server.py"),
|
|
334
|
+
"call",
|
|
335
|
+
"iumbtems_code_audit",
|
|
336
|
+
json.dumps(args),
|
|
337
|
+
],
|
|
338
|
+
capture_output=True,
|
|
339
|
+
text=True,
|
|
340
|
+
timeout=timeout,
|
|
341
|
+
cwd=str(PROJECT_ROOT),
|
|
342
|
+
)
|
|
343
|
+
return proc.stdout + proc.stderr
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def advisory_comment(fixture_id, findings):
|
|
347
|
+
return (
|
|
348
|
+
f"### IUMBTEMS advisory review: `{fixture_id}`\n\n"
|
|
349
|
+
f"Automated findings below are **non-blocking**. Verify each cited line before acting.\n\n"
|
|
350
|
+
f"{findings}\n"
|
|
351
|
+
)
|
|
352
|
+
|
|
353
|
+
|
|
354
|
+
LINE_RE = re.compile(r"(?:^|[\s(:\[\"'])(?:line\s+)?L?(\d{1,4})\b", re.IGNORECASE)
|
|
355
|
+
|
|
356
|
+
|
|
357
|
+
def score_fixture(entry, findings):
|
|
358
|
+
"""Hit = seeded filename cited with a line number within tolerance."""
|
|
359
|
+
if entry["clean"]:
|
|
360
|
+
cited_own_file = entry["file"] in findings
|
|
361
|
+
return {"hit": False, "false_positive": cited_own_file}
|
|
362
|
+
if entry["file"] not in findings:
|
|
363
|
+
return {"hit": False, "false_positive": False}
|
|
364
|
+
lines = {int(n) for n in LINE_RE.findall(findings)}
|
|
365
|
+
hit = any(abs(n - entry["line"]) <= LINE_TOLERANCE for n in lines)
|
|
366
|
+
return {"hit": hit, "false_positive": False}
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
def score_all(root):
|
|
370
|
+
manifest = json.loads((Path(root) / "manifest.json").read_text(encoding="utf-8"))
|
|
371
|
+
results = []
|
|
372
|
+
for entry in manifest:
|
|
373
|
+
comment_file = Path(root) / entry["id"] / "advisory.md"
|
|
374
|
+
findings = (
|
|
375
|
+
comment_file.read_text(encoding="utf-8") if comment_file.exists() else ""
|
|
376
|
+
)
|
|
377
|
+
results.append({"id": entry["id"], **score_fixture(entry, findings)})
|
|
378
|
+
defective = [r for r, e in zip(results, manifest) if not e["clean"]]
|
|
379
|
+
clean = [r for r, e in zip(results, manifest) if e["clean"]]
|
|
380
|
+
hits = sum(1 for r in defective if r["hit"])
|
|
381
|
+
fps = sum(1 for r in clean if r["false_positive"])
|
|
382
|
+
return {
|
|
383
|
+
"fixtures": len(manifest),
|
|
384
|
+
"defective": len(defective),
|
|
385
|
+
"hits": hits,
|
|
386
|
+
"hit_rate": round(hits / max(len(defective), 1), 3),
|
|
387
|
+
"false_positives": fps,
|
|
388
|
+
"clean_total": len(clean),
|
|
389
|
+
"misses": [r["id"] for r in defective if not r["hit"]],
|
|
390
|
+
"results": results,
|
|
391
|
+
}
|
|
392
|
+
|
|
393
|
+
|
|
394
|
+
def main(argv=None):
|
|
395
|
+
ap = argparse.ArgumentParser()
|
|
396
|
+
ap.add_argument("--dir", default="/tmp/bet1")
|
|
397
|
+
ap.add_argument("--generate", action="store_true")
|
|
398
|
+
ap.add_argument("--run", action="store_true")
|
|
399
|
+
ap.add_argument("--score", action="store_true")
|
|
400
|
+
ap.add_argument("--mock", action="store_true")
|
|
401
|
+
ap.add_argument("--live", action="store_true")
|
|
402
|
+
ap.add_argument("--only", default="")
|
|
403
|
+
args = ap.parse_args(argv)
|
|
404
|
+
root = Path(args.dir)
|
|
405
|
+
|
|
406
|
+
if args.generate or (not args.run and not args.score):
|
|
407
|
+
generate(root)
|
|
408
|
+
if args.run:
|
|
409
|
+
manifest = json.loads((root / "manifest.json").read_text(encoding="utf-8"))
|
|
410
|
+
only = {s.strip() for s in args.only.split(",") if s.strip()}
|
|
411
|
+
for entry in manifest:
|
|
412
|
+
if only and entry["id"] not in only:
|
|
413
|
+
continue
|
|
414
|
+
findings = run_audit(root / entry["id"], mock=args.mock and not args.live)
|
|
415
|
+
(root / entry["id"] / "advisory.md").write_text(
|
|
416
|
+
advisory_comment(entry["id"], findings), encoding="utf-8"
|
|
417
|
+
)
|
|
418
|
+
print(f"advisory written: {entry['id']}")
|
|
419
|
+
if args.score:
|
|
420
|
+
summary = score_all(root)
|
|
421
|
+
print(json.dumps(summary, indent=2))
|
|
422
|
+
return 0 # advisory: never blocks
|
|
423
|
+
|
|
424
|
+
|
|
425
|
+
if __name__ == "__main__":
|
|
426
|
+
sys.exit(main())
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|