@claude-flow/cli 3.38.12 → 3.38.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/.proven-config-version +1 -0
- package/.claude/helpers/.helpers-version +1 -1
- package/.claude/helpers/helpers.manifest.json +2 -2
- package/.claude/helpers/statusline.cjs +0 -0
- package/.claude/proven-config.json +42 -0
- package/catalog-manifest.json +4 -4
- package/dist/src/mcp-tools/hooks-tools.js +6 -1
- package/dist/src/ruvector/lattice-wasm.d.ts +14 -0
- package/dist/src/ruvector/lattice-wasm.js +144 -0
- package/dist/src/services/flywheel-receipt.d.ts +10 -0
- package/dist/src/services/flywheel-receipt.js +82 -7
- package/dist/src/services/flywheel-transaction.js +10 -1
- package/node_modules/@claude-flow/codex/dist/cli.js +0 -0
- package/node_modules/@claude-flow/plugin-agent-federation/dist/bin.js +0 -0
- package/node_modules/@claude-flow/security/dist/input-validator.d.ts +6 -6
- package/package.json +1 -1
- package/plugins/ruflo-metaharness/.claude-flow/daemon-state.json +178 -0
- package/plugins/ruflo-metaharness/.claude-flow/daemon.pid +1 -0
- package/plugins/ruflo-metaharness/.claude-flow/data/pending-insights.jsonl +5 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/daemon.log +269 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783604774864_ozbujc_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783604774864_ozbujc_result.log +108 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783605513587_ulvmpb_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783605513587_ulvmpb_result.log +209 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783606368867_ahysui_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783606368867_ahysui_result.log +192 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783607120257_lh05rb_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783607120257_lh05rb_result.log +13 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783608020362_j3096j_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783608020362_j3096j_result.log +120 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783608776347_261b61_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783608776347_261b61_result.log +85 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783609621359_s5i6ye_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783609621359_s5i6ye_result.log +13 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783610087998_qihv9v_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783610087998_qihv9v_result.log +17 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783610773920_qlzmxo_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783610773920_qlzmxo_result.log +17 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783611376090_xpqf1z_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783611376090_xpqf1z_result.log +16 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783612097184_9rqfor_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783612097184_9rqfor_result.log +138 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783612811574_4u602j_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783612811574_4u602j_result.log +16 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783613750487_a46ttn_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783613750487_a46ttn_result.log +107 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783614289360_figwc0_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783614289360_figwc0_result.log +200 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783615067640_l7tm6a_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783615067640_l7tm6a_result.log +54 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783615825308_44nor5_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783615825308_44nor5_result.log +85 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783616524771_ut1ftw_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783616524771_ut1ftw_result.log +266 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783617323039_fs3x5a_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783617323039_fs3x5a_result.log +56 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783618049184_1f4yah_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783618049184_1f4yah_result.log +96 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783618793925_4ee7tf_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783618793925_4ee7tf_result.log +481 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783619642574_yjr4mm_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783619642574_yjr4mm_result.log +104 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783620392771_oduto0_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783620392771_oduto0_result.log +148 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783621183670_gd0p1x_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783621183670_gd0p1x_result.log +111 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783621738388_z4k48b_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783621738388_z4k48b_result.log +89 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783622493677_zwc35w_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783622493677_zwc35w_result.log +207 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783604894861_v6n3ut_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783604894861_v6n3ut_result.log +66 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783605934532_9h8ikb_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783605934532_9h8ikb_result.log +68 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783607181736_t12f4y_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783607181736_t12f4y_result.log +78 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783608341266_zhk0fl_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783608341266_zhk0fl_result.log +68 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783609420180_7xw817_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783609420180_7xw817_result.log +72 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783610535307_2pxofp_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783610535307_2pxofp_result.log +17 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783611437445_ofwnpb_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783611437445_ofwnpb_result.log +60 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783612556827_1cb112_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783612556827_1cb112_result.log +56 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783613579531_18xiax_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783613579531_18xiax_result.log +74 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783614650481_2zdz7w_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783614650481_2zdz7w_result.log +68 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783615742328_rjj69d_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783615742328_rjj69d_result.log +65 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783616767230_1iad99_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783616767230_1iad99_result.log +68 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783617783537_rqagku_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783617783537_rqagku_result.log +56 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783618817343_r1lbhr_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783618817343_r1lbhr_result.log +65 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783619907773_8msfw3_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783619907773_8msfw3_result.log +77 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783621009302_hybdum_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783621009302_hybdum_result.log +64 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783622083681_nznpnx_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783622083681_nznpnx_result.log +56 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783623208340_olsbaw_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783605134860_9jssz9_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783605134860_9jssz9_result.log +69 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783606516743_zftbaa_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783606516743_zftbaa_result.log +92 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783608038317_jrd66e_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783608038317_jrd66e_result.log +92 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783609388416_mc3zoe_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783609388416_mc3zoe_result.log +82 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783610821384_tzqvlb_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783610821384_tzqvlb_result.log +17 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783612023285_buygpo_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783612023285_buygpo_result.log +57 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783613599301_6f78cw_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783613599301_6f78cw_result.log +60 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783615000844_v95ues_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783615000844_v95ues_result.log +69 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783616341693_bl5d9o_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783616341693_bl5d9o_result.log +64 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783617832831_ha6s8d_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783617832831_ha6s8d_result.log +42 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783619384959_s5iiwf_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783619384959_s5iiwf_result.log +47 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783620946263_d7ovai_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783620946263_d7ovai_result.log +52 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783622473846_t839e5_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783622473846_t839e5_result.log +96 -0
- package/plugins/ruflo-metaharness/.claude-flow/metrics/backup.json +7 -0
- package/plugins/ruflo-metaharness/.claude-flow/metrics/codebase-map.json +11 -0
- package/plugins/ruflo-metaharness/.claude-flow/metrics/consolidation.json +16 -0
- package/plugins/ruflo-metaharness/.claude-flow/metrics/harness-loop.json +83 -0
- package/plugins/ruflo-metaharness/.claude-flow/metrics/performance.json +55 -0
- package/plugins/ruflo-metaharness/.claude-flow/metrics/security-audit.json +206 -0
- package/plugins/ruflo-metaharness/.claude-flow/metrics/test-gaps.json +95 -0
- package/plugins/ruflo-metaharness/.claude-flow/neural/stats.json +6 -0
- package/plugins/ruflo-metaharness/skills/harness-gepa/SKILL.md +1 -1
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
[2026-07-09T14:43:08.375Z] RESULT
|
|
2
|
+
============================================================
|
|
3
|
+
{
|
|
4
|
+
"success": true,
|
|
5
|
+
"output": "## Test Coverage Analysis — `plugins/ruflo-metaharness`\n\n**Infrastructure note:** there's no `package.json`/jest/vitest here. The \"test suite\" is `scripts/test-*.mjs` (6 files) using a hand-rolled `assert()` + pass/fail counter, gated by `scripts/smoke.sh`. All skeletons below follow that existing convention rather than introducing a new framework — nothing here is exported as a library, so most gaps require black-box `execFileSync` tests, not unit imports.\n\nI verified every claim below directly against source (this plugin already had an auto-generated `testgaps` report from today in `.claude-flow/metrics/test-gaps.json` — I cross-checked it rather than trusting it blindly, and corrected one inaccuracy, noted at #5).\n\n### 1. `threat-model.mjs` — stale local `SEVERITY_RANK` (real bug, highest priority)\n\n`scripts/threat-model.mjs:9` keeps its own `{clean:0,low:1,medium:2,high:3}` instead of importing the canonical table from `_harness.mjs:261` (`{clean,info,low,medium,warn,high,error,critical}`). If the harness reports `worst:\"critical\"`, `SEVERITY_RANK['critical']` is `undefined`, so `undefined >= threshold` → `false` — `--fail-on high` **never trips for a critical finding**.\n\n```js\n// scripts/test-threat-model-severity.mjs\nimport { execFileSync } from 'node:child_process';\nlet passed = 0, failed = 0;\nconst assert = (c, l) => c ? passed++ : (failed++, console.log(`✗ ${l}`));\n\n// Requires a mock/fixture harness binary that returns worst:\"critical\"\nconst out = execFileSync('node', ['scripts/threat-model.mjs', '--path', 'FIXTURE_CRITICAL', '--fail-on', 'high'], { encoding: 'utf8' });\nconst payload = JSON.parse(out);\nassert(payload.alert.triggered === true, 'worst=critical must trigger --fail-on high (currently silently false)');\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);\n```\nFix: `import { SEVERITY_RANK } from './_harness.mjs'` instead of the local literal.\n\n### 2. `router-parallel-analyze.mjs` — one malformed JSONL line kills the whole run\n\n`scripts/router-parallel-analyze.mjs:90-95`: the `.map(l => JSON.parse(l))` has no per-line try/catch; the *outer* catch aborts with exit 2 for the entire file on a single bad line, discarding all valid rows.\n\n```js\n// scripts/test-router-parallel-malformed.mjs\nimport { execFileSync } from 'node:child_process';\nimport { writeFileSync, mkdtempSync } from 'node:fs';\nimport { join } from 'node:path';\nimport { tmpdir } from 'node:os';\nlet passed = 0, failed = 0;\nconst assert = (c, l) => c ? passed++ : (failed++, console.log(`✗ ${l}`));\n\nconst dir = mkdtempSync(join(tmpdir(), 'rpa-'));\nconst file = join(dir, 'router-parallel.jsonl');\nconst goodRow = JSON.stringify({ bandit: {}, ser: {}, outcome: {} }); // must match the real filter shape (bandit && ser && outcome)\nwriteFileSync(file, Array(35).fill(goodRow).join('\\n') + '\\nNOT VALID JSON\\n');\n\nlet exitCode = 0;\ntry {\n execFileSync('node', ['scripts/router-parallel-analyze.mjs', '--input', file], { encoding: 'utf8' });\n} catch (e) { exitCode = e.status; }\n\nassert(exitCode !== 2, 'one malformed JSONL line should not abort analysis of the other 35 valid rows');\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);\n```\nFix: wrap the per-line parse individually, skip+count bad lines instead of aborting.\n\n### 3. `evolve.mjs --diagnose` subsystem — zero coverage\n\n`looksLikeGepaTranscript`, `extractGepaTranscripts`, `summarizeTraces`, `buildDiagnosis` (evolve.mjs:146-239, ~90 lines) are module-private (no `export`) and never exercised by any of the 6 existing test scripts. Must be tested black-box via CLI.\n\n```js\n// scripts/test-evolve-diagnose.mjs\nimport { execFileSync } from 'node:child_process';\nimport { mkdtempSync, writeFileSync, mkdirSync } from 'node:fs';\nimport { tmpdir } from 'node:os';\nimport { join } from 'node:path';\nlet passed = 0, failed = 0;\nconst assert = (c, l) => c ? passed++ : (failed++, console.log(`✗ ${l}`));\n\nconst runsDir = mkdtempSync(join(tmpdir(), 'evolve-diag-'));\nmkdirSync(join(runsDir, '.metaharness', 'runs'), { recursive: true });\nwriteFileSync(join(runsDir, '.metaharness', 'runs', 'winner.json'), '{\"id\":\"v1\"}');\nwriteFileSync(join(runsDir, '.metaharness', 'runs', 'v1.json'), JSON.stringify({ id: 'v1', transcript: [] }));\n\nconst out = execFileSync('node', ['scripts/evolve.mjs', runsDir, '--diagnose', '--sandbox', 'mock'], { encoding: 'utf8' });\nassert(typeof JSON.parse(out).diagnosis === 'object', 'diagnosis object present for a valid winner+run record');\n\n// corrupt winner.json must degrade, not throw\nwriteFileSync(join(runsDir, '.metaharness', 'runs', 'winner.json'), '{not json');\nconst out2 = execFileSync('node', ['scripts/evolve.mjs', runsDir, '--diagnose', '--sandbox', 'mock'], { encoding: 'utf8' });\nassert(!!JSON.parse(out2).diagnosis, 'corrupt winner.json does not crash --diagnose');\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);\n```\n\n### 4. `gepa.mjs` — `validate`/`render`/`analyze` ops untested\n\nConfirmed all 4 ops (`genome|validate|render|analyze`, gepa.mjs:104) exist but only `genome` is exercised indirectly via MCP. `--alert-on-invalid` exit-1 path and the non-array `--transcript` exit-2 path (gepa.mjs:135-136) are unhit.\n\n```js\n// scripts/test-gepa-ops.mjs\nimport { execFileSync } from 'node:child_process';\nlet passed = 0, failed = 0;\nconst assert = (c, l) => c ? passed++ : (failed++, console.log(`✗ ${l}`));\n\nfor (const op of ['validate', 'render']) {\n const out = execFileSync('node', ['scripts/gepa.mjs', '--op', op], { encoding: 'utf8' });\n const payload = JSON.parse(out);\n assert(payload.op === op || payload.degraded, `op=${op} returns a shaped payload or degrades cleanly`);\n}\n\ntry {\n execFileSync('node', ['scripts/gepa.mjs', '--op', 'analyze', '--transcript', '{\"not\":\"array\"}'], { encoding: 'utf8' });\n assert(false, 'non-array --transcript should exit non-zero');\n} catch (e) { assert(e.status === 2, 'non-array --transcript exits 2'); }\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);\n```\n\n### 5. `mcp-scan.mjs` — string-shaped finding produces silent corruption, **not a crash** (correction)\n\nThe auto-generated report claimed `findings.map` \"throws\" on a raw-string array element. I checked the map body (`mcp-scan.mjs:62-66`): `{...f, message: ...}` on a string `f` spreads its characters into numeric keys and `f.message`/`f.title`/`f.detail` are all `undefined` on a string — so `message` silently falls back to `''`, no throw. The real bug is **silent data corruption** (a finding becomes `{0:'a',1:' ',...,message:''}`), which is worse for debugging than a crash since it never surfaces.\n\n```js\n// scripts/test-mcp-scan-malformed-finding.mjs\nimport { parseMcpScanText } from './_harness.mjs'; // if exported; else drive via fixture stdout + execFileSync\n// Fixture: upstream JSON with findings: [\"a raw string\", {severity:'high', message:'ok'}]\n// Assert: string elements are filtered out or coerced to a real object BEFORE spread,\n// not silently spread into {0:'a',1:' ',...}.\n```\nRecommend a guard: `rawFindings.filter(f => f && typeof f === 'object')` before the `.map`.\n\n### 6. `security-bench.mjs` — `parseSecurityBenchMarkdown` not exported, no isolated test\n\nConfirmed not exported (`security-bench.mjs:77`). Fragile to upstream markdown/emoji/table drift.\n\n```js\n// scripts/test-security-bench-parse.mjs\n// Requires: export function parseSecurityBenchMarkdown(md) { ... }\nimport { parseSecurityBenchMarkdown } from './security-bench.mjs';\nlet passed = 0, failed = 0;\nconst assert = (c, l) => c ? passed++ : (failed++, console.log(`✗ ${l}`));\n\nconst wellFormed = `## Overall: ✅ PASS\\n| Gate | Status |\\n|---|---|\\n| injection | ✅ |\\n`;\nassert(parseSecurityBenchMarkdown(wellFormed).gates.length > 0, 'well-formed markdown parses gates');\n\nconst malformed = `unexpected upstream format with no matching tables`;\nconst r2 = parseSecurityBenchMarkdown(malformed);\nassert(r2.gates.length === 0, 'malformed markdown returns empty gates, not throw');\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);\n```\n\n### 7. Missing error-handling tests (quick wins, all confirmed against source)\n\n| File:line | Confirmed issue | Test scenario |\n|---|---|---|\n| `audit-trend.mjs:140` | `fingerprint()` truncates `message` at 80 chars via `.slice(0,80)` | Two findings identical in the first 80 chars but differing after — assert they are NOT deduped as \"unchanged\" |\n| `similarity.mjs:51,114` | `Number(process.argv[++i])` on `--alert-below abc` → `NaN`; `result.overall < NaN` is always `false` | `execFileSync(['similarity.mjs',...,'--alert-below','abc'])` should exit non-zero (reject bad arg), not silently succeed |\n| `redblue.mjs:90` | `!next.startsWith('-')` guard means `--some-flag -5` drops `-5` as the value (mistaken for a bare flag) | Passthrough arg with a negative-number value must be preserved, not swallowed |\n| `learn.mjs:92` | `checkout-required` detection is a fixed status string keyed off upstream stderr wording | Fixture stderr with slightly different wording — assert it doesn't silently misreport `status:'failed'` |\n| `_harness.mjs:269` | `rankSeverity` lowercases but doesn't trim — `rankSeverity(' high ')` → 0, not 3 | `assert(rankSeverity(' high ') === rankSeverity('high'))` — currently fails, flags the footgun |\n\n### 8. Integration test gaps\n\n- **`oia-audit.mjs`**: only the \"all 5 subprocesses degraded\" path is tested (`test-graceful-degradation.mjs`). No test for 1–4 of 5 degrading while the composite still reports healthy at the top level.\n- **`mint.mjs`**: the documented pre-0.1.13 `--target`-ignored workaround has no regression test against the currently pinned metaharness version — if upstream fixes it, the workaround's continued correctness is unverified either way.\n- **`test-parallel-pipeline.mjs`**: its TS-source check is a `grep`, not a runtime test — `recordPair`/`recordPairOutcome` in `router-parallel-recorder.ts` are pattern-matched as text, never actually executed against compiled `dist/`.\n- **`test-pipeline-roundtrip.mjs`**: `REPO_ROOT = dirname(dirname(dirname(SCRIPTS_DIR)))` is a fragile path assumption — relocating the plugin silently points at the wrong tree instead of failing loudly. Add a sentinel-file assertion (e.g., root `CLAUDE.md` exists) before proceeding.\n\n### Priority order if picking a subset\n\n1. Fix `threat-model.mjs` `SEVERITY_RANK` (#1) — real bug, silently swallows critical findings\n2. Fix `router-parallel-analyze.mjs` malformed-JSONL abort (#2) — real bug, discards good data\n3. Fix `mcp-scan.mjs` string-finding silent corruption (#5) — real bug, worse than the report's crash claim\n4. `evolve.mjs --diagnose` coverage (#3) — largest untested surface\n5. `gepa.mjs` ops (#4), then the quick-win table (#7) and integration gaps (#8)\n",
|
|
6
|
+
"parsedOutput": {
|
|
7
|
+
"sections": [
|
|
8
|
+
{
|
|
9
|
+
"title": "Test Coverage Analysis — `plugins/ruflo-metaharness`",
|
|
10
|
+
"content": "\n**Infrastructure note:** there's no `package.json`/jest/vitest here. The \"test suite\" is `scripts/test-*.mjs` (6 files) using a hand-rolled `assert()` + pass/fail counter, gated by `scripts/smoke.sh`. All skeletons below follow that existing convention rather than introducing a new framework — nothing here is exported as a library, so most gaps require black-box `execFileSync` tests, not unit imports.\n\nI verified every claim below directly against source (this plugin already had an auto-generated `testgaps` report from today in `.claude-flow/metrics/test-gaps.json` — I cross-checked it rather than trusting it blindly, and corrected one inaccuracy, noted at #5).\n\n",
|
|
11
|
+
"level": 2
|
|
12
|
+
},
|
|
13
|
+
{
|
|
14
|
+
"title": "1. `threat-model.mjs` — stale local `SEVERITY_RANK` (real bug, highest priority)",
|
|
15
|
+
"content": "\n`scripts/threat-model.mjs:9` keeps its own `{clean:0,low:1,medium:2,high:3}` instead of importing the canonical table from `_harness.mjs:261` (`{clean,info,low,medium,warn,high,error,critical}`). If the harness reports `worst:\"critical\"`, `SEVERITY_RANK['critical']` is `undefined`, so `undefined >= threshold` → `false` — `--fail-on high` **never trips for a critical finding**.\n\n```js\n// scripts/test-threat-model-severity.mjs\nimport { execFileSync } from 'node:child_process';\nlet passed = 0, failed = 0;\nconst assert = (c, l) => c ? passed++ : (failed++, console.log(`✗ ${l}`));\n\n// Requires a mock/fixture harness binary that returns worst:\"critical\"\nconst out = execFileSync('node', ['scripts/threat-model.mjs', '--path', 'FIXTURE_CRITICAL', '--fail-on', 'high'], { encoding: 'utf8' });\nconst payload = JSON.parse(out);\nassert(payload.alert.triggered === true, 'worst=critical must trigger --fail-on high (currently silently false)');\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);\n```\nFix: `import { SEVERITY_RANK } from './_harness.mjs'` instead of the local literal.\n\n",
|
|
16
|
+
"level": 3
|
|
17
|
+
},
|
|
18
|
+
{
|
|
19
|
+
"title": "2. `router-parallel-analyze.mjs` — one malformed JSONL line kills the whole run",
|
|
20
|
+
"content": "\n`scripts/router-parallel-analyze.mjs:90-95`: the `.map(l => JSON.parse(l))` has no per-line try/catch; the *outer* catch aborts with exit 2 for the entire file on a single bad line, discarding all valid rows.\n\n```js\n// scripts/test-router-parallel-malformed.mjs\nimport { execFileSync } from 'node:child_process';\nimport { writeFileSync, mkdtempSync } from 'node:fs';\nimport { join } from 'node:path';\nimport { tmpdir } from 'node:os';\nlet passed = 0, failed = 0;\nconst assert = (c, l) => c ? passed++ : (failed++, console.log(`✗ ${l}`));\n\nconst dir = mkdtempSync(join(tmpdir(), 'rpa-'));\nconst file = join(dir, 'router-parallel.jsonl');\nconst goodRow = JSON.stringify({ bandit: {}, ser: {}, outcome: {} }); // must match the real filter shape (bandit && ser && outcome)\nwriteFileSync(file, Array(35).fill(goodRow).join('\\n') + '\\nNOT VALID JSON\\n');\n\nlet exitCode = 0;\ntry {\n execFileSync('node', ['scripts/router-parallel-analyze.mjs', '--input', file], { encoding: 'utf8' });\n} catch (e) { exitCode = e.status; }\n\nassert(exitCode !== 2, 'one malformed JSONL line should not abort analysis of the other 35 valid rows');\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);\n```\nFix: wrap the per-line parse individually, skip+count bad lines instead of aborting.\n\n",
|
|
21
|
+
"level": 3
|
|
22
|
+
},
|
|
23
|
+
{
|
|
24
|
+
"title": "3. `evolve.mjs --diagnose` subsystem — zero coverage",
|
|
25
|
+
"content": "\n`looksLikeGepaTranscript`, `extractGepaTranscripts`, `summarizeTraces`, `buildDiagnosis` (evolve.mjs:146-239, ~90 lines) are module-private (no `export`) and never exercised by any of the 6 existing test scripts. Must be tested black-box via CLI.\n\n```js\n// scripts/test-evolve-diagnose.mjs\nimport { execFileSync } from 'node:child_process';\nimport { mkdtempSync, writeFileSync, mkdirSync } from 'node:fs';\nimport { tmpdir } from 'node:os';\nimport { join } from 'node:path';\nlet passed = 0, failed = 0;\nconst assert = (c, l) => c ? passed++ : (failed++, console.log(`✗ ${l}`));\n\nconst runsDir = mkdtempSync(join(tmpdir(), 'evolve-diag-'));\nmkdirSync(join(runsDir, '.metaharness', 'runs'), { recursive: true });\nwriteFileSync(join(runsDir, '.metaharness', 'runs', 'winner.json'), '{\"id\":\"v1\"}');\nwriteFileSync(join(runsDir, '.metaharness', 'runs', 'v1.json'), JSON.stringify({ id: 'v1', transcript: [] }));\n\nconst out = execFileSync('node', ['scripts/evolve.mjs', runsDir, '--diagnose', '--sandbox', 'mock'], { encoding: 'utf8' });\nassert(typeof JSON.parse(out).diagnosis === 'object', 'diagnosis object present for a valid winner+run record');\n\n// corrupt winner.json must degrade, not throw\nwriteFileSync(join(runsDir, '.metaharness', 'runs', 'winner.json'), '{not json');\nconst out2 = execFileSync('node', ['scripts/evolve.mjs', runsDir, '--diagnose', '--sandbox', 'mock'], { encoding: 'utf8' });\nassert(!!JSON.parse(out2).diagnosis, 'corrupt winner.json does not crash --diagnose');\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);\n```\n\n",
|
|
26
|
+
"level": 3
|
|
27
|
+
},
|
|
28
|
+
{
|
|
29
|
+
"title": "4. `gepa.mjs` — `validate`/`render`/`analyze` ops untested",
|
|
30
|
+
"content": "\nConfirmed all 4 ops (`genome|validate|render|analyze`, gepa.mjs:104) exist but only `genome` is exercised indirectly via MCP. `--alert-on-invalid` exit-1 path and the non-array `--transcript` exit-2 path (gepa.mjs:135-136) are unhit.\n\n```js\n// scripts/test-gepa-ops.mjs\nimport { execFileSync } from 'node:child_process';\nlet passed = 0, failed = 0;\nconst assert = (c, l) => c ? passed++ : (failed++, console.log(`✗ ${l}`));\n\nfor (const op of ['validate', 'render']) {\n const out = execFileSync('node', ['scripts/gepa.mjs', '--op', op], { encoding: 'utf8' });\n const payload = JSON.parse(out);\n assert(payload.op === op || payload.degraded, `op=${op} returns a shaped payload or degrades cleanly`);\n}\n\ntry {\n execFileSync('node', ['scripts/gepa.mjs', '--op', 'analyze', '--transcript', '{\"not\":\"array\"}'], { encoding: 'utf8' });\n assert(false, 'non-array --transcript should exit non-zero');\n} catch (e) { assert(e.status === 2, 'non-array --transcript exits 2'); }\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);\n```\n\n",
|
|
31
|
+
"level": 3
|
|
32
|
+
},
|
|
33
|
+
{
|
|
34
|
+
"title": "5. `mcp-scan.mjs` — string-shaped finding produces silent corruption, **not a crash** (correction)",
|
|
35
|
+
"content": "\nThe auto-generated report claimed `findings.map` \"throws\" on a raw-string array element. I checked the map body (`mcp-scan.mjs:62-66`): `{...f, message: ...}` on a string `f` spreads its characters into numeric keys and `f.message`/`f.title`/`f.detail` are all `undefined` on a string — so `message` silently falls back to `''`, no throw. The real bug is **silent data corruption** (a finding becomes `{0:'a',1:' ',...,message:''}`), which is worse for debugging than a crash since it never surfaces.\n\n```js\n// scripts/test-mcp-scan-malformed-finding.mjs\nimport { parseMcpScanText } from './_harness.mjs'; // if exported; else drive via fixture stdout + execFileSync\n// Fixture: upstream JSON with findings: [\"a raw string\", {severity:'high', message:'ok'}]\n// Assert: string elements are filtered out or coerced to a real object BEFORE spread,\n// not silently spread into {0:'a',1:' ',...}.\n```\nRecommend a guard: `rawFindings.filter(f => f && typeof f === 'object')` before the `.map`.\n\n",
|
|
36
|
+
"level": 3
|
|
37
|
+
},
|
|
38
|
+
{
|
|
39
|
+
"title": "6. `security-bench.mjs` — `parseSecurityBenchMarkdown` not exported, no isolated test",
|
|
40
|
+
"content": "\nConfirmed not exported (`security-bench.mjs:77`). Fragile to upstream markdown/emoji/table drift.\n\n```js\n// scripts/test-security-bench-parse.mjs\n// Requires: export function parseSecurityBenchMarkdown(md) { ... }\nimport { parseSecurityBenchMarkdown } from './security-bench.mjs';\nlet passed = 0, failed = 0;\nconst assert = (c, l) => c ? passed++ : (failed++, console.log(`✗ ${l}`));\n\nconst wellFormed = `## Overall: ✅ PASS\\n| Gate | Status |\\n|---|---|\\n| injection | ✅ |\\n`;\nassert(parseSecurityBenchMarkdown(wellFormed).gates.length > 0, 'well-formed markdown parses gates');\n\nconst malformed = `unexpected upstream format with no matching tables`;\nconst r2 = parseSecurityBenchMarkdown(malformed);\nassert(r2.gates.length === 0, 'malformed markdown returns empty gates, not throw');\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);\n```\n\n",
|
|
41
|
+
"level": 3
|
|
42
|
+
},
|
|
43
|
+
{
|
|
44
|
+
"title": "7. Missing error-handling tests (quick wins, all confirmed against source)",
|
|
45
|
+
"content": "\n| File:line | Confirmed issue | Test scenario |\n|---|---|---|\n| `audit-trend.mjs:140` | `fingerprint()` truncates `message` at 80 chars via `.slice(0,80)` | Two findings identical in the first 80 chars but differing after — assert they are NOT deduped as \"unchanged\" |\n| `similarity.mjs:51,114` | `Number(process.argv[++i])` on `--alert-below abc` → `NaN`; `result.overall < NaN` is always `false` | `execFileSync(['similarity.mjs',...,'--alert-below','abc'])` should exit non-zero (reject bad arg), not silently succeed |\n| `redblue.mjs:90` | `!next.startsWith('-')` guard means `--some-flag -5` drops `-5` as the value (mistaken for a bare flag) | Passthrough arg with a negative-number value must be preserved, not swallowed |\n| `learn.mjs:92` | `checkout-required` detection is a fixed status string keyed off upstream stderr wording | Fixture stderr with slightly different wording — assert it doesn't silently misreport `status:'failed'` |\n| `_harness.mjs:269` | `rankSeverity` lowercases but doesn't trim — `rankSeverity(' high ')` → 0, not 3 | `assert(rankSeverity(' high ') === rankSeverity('high'))` — currently fails, flags the footgun |\n\n",
|
|
46
|
+
"level": 3
|
|
47
|
+
},
|
|
48
|
+
{
|
|
49
|
+
"title": "8. Integration test gaps",
|
|
50
|
+
"content": "\n- **`oia-audit.mjs`**: only the \"all 5 subprocesses degraded\" path is tested (`test-graceful-degradation.mjs`). No test for 1–4 of 5 degrading while the composite still reports healthy at the top level.\n- **`mint.mjs`**: the documented pre-0.1.13 `--target`-ignored workaround has no regression test against the currently pinned metaharness version — if upstream fixes it, the workaround's continued correctness is unverified either way.\n- **`test-parallel-pipeline.mjs`**: its TS-source check is a `grep`, not a runtime test — `recordPair`/`recordPairOutcome` in `router-parallel-recorder.ts` are pattern-matched as text, never actually executed against compiled `dist/`.\n- **`test-pipeline-roundtrip.mjs`**: `REPO_ROOT = dirname(dirname(dirname(SCRIPTS_DIR)))` is a fragile path assumption — relocating the plugin silently points at the wrong tree instead of failing loudly. Add a sentinel-file assertion (e.g., root `CLAUDE.md` exists) before proceeding.\n\n",
|
|
51
|
+
"level": 3
|
|
52
|
+
},
|
|
53
|
+
{
|
|
54
|
+
"title": "Priority order if picking a subset",
|
|
55
|
+
"content": "1. Fix `threat-model.mjs` `SEVERITY_RANK` (#1) — real bug, silently swallows critical findings\n2. Fix `router-parallel-analyze.mjs` malformed-JSONL abort (#2) — real bug, discards good data\n3. Fix `mcp-scan.mjs` string-finding silent corruption (#5) — real bug, worse than the report's crash claim\n4. `evolve.mjs --diagnose` coverage (#3) — largest untested surface\n5. `gepa.mjs` ops (#4), then the quick-win table (#7) and integration gaps (#8)",
|
|
56
|
+
"level": 3
|
|
57
|
+
}
|
|
58
|
+
],
|
|
59
|
+
"codeBlocks": [
|
|
60
|
+
{
|
|
61
|
+
"language": "js",
|
|
62
|
+
"code": "// scripts/test-threat-model-severity.mjs\nimport { execFileSync } from 'node:child_process';\nlet passed = 0, failed = 0;\nconst assert = (c, l) => c ? passed++ : (failed++, console.log(`✗ ${l}`));\n\n// Requires a mock/fixture harness binary that returns worst:\"critical\"\nconst out = execFileSync('node', ['scripts/threat-model.mjs', '--path', 'FIXTURE_CRITICAL', '--fail-on', 'high'], { encoding: 'utf8' });\nconst payload = JSON.parse(out);\nassert(payload.alert.triggered === true, 'worst=critical must trigger --fail-on high (currently silently false)');\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);"
|
|
63
|
+
},
|
|
64
|
+
{
|
|
65
|
+
"language": "js",
|
|
66
|
+
"code": "// scripts/test-router-parallel-malformed.mjs\nimport { execFileSync } from 'node:child_process';\nimport { writeFileSync, mkdtempSync } from 'node:fs';\nimport { join } from 'node:path';\nimport { tmpdir } from 'node:os';\nlet passed = 0, failed = 0;\nconst assert = (c, l) => c ? passed++ : (failed++, console.log(`✗ ${l}`));\n\nconst dir = mkdtempSync(join(tmpdir(), 'rpa-'));\nconst file = join(dir, 'router-parallel.jsonl');\nconst goodRow = JSON.stringify({ bandit: {}, ser: {}, outcome: {} }); // must match the real filter shape (bandit && ser && outcome)\nwriteFileSync(file, Array(35).fill(goodRow).join('\\n') + '\\nNOT VALID JSON\\n');\n\nlet exitCode = 0;\ntry {\n execFileSync('node', ['scripts/router-parallel-analyze.mjs', '--input', file], { encoding: 'utf8' });\n} catch (e) { exitCode = e.status; }\n\nassert(exitCode !== 2, 'one malformed JSONL line should not abort analysis of the other 35 valid rows');\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);"
|
|
67
|
+
},
|
|
68
|
+
{
|
|
69
|
+
"language": "js",
|
|
70
|
+
"code": "// scripts/test-evolve-diagnose.mjs\nimport { execFileSync } from 'node:child_process';\nimport { mkdtempSync, writeFileSync, mkdirSync } from 'node:fs';\nimport { tmpdir } from 'node:os';\nimport { join } from 'node:path';\nlet passed = 0, failed = 0;\nconst assert = (c, l) => c ? passed++ : (failed++, console.log(`✗ ${l}`));\n\nconst runsDir = mkdtempSync(join(tmpdir(), 'evolve-diag-'));\nmkdirSync(join(runsDir, '.metaharness', 'runs'), { recursive: true });\nwriteFileSync(join(runsDir, '.metaharness', 'runs', 'winner.json'), '{\"id\":\"v1\"}');\nwriteFileSync(join(runsDir, '.metaharness', 'runs', 'v1.json'), JSON.stringify({ id: 'v1', transcript: [] }));\n\nconst out = execFileSync('node', ['scripts/evolve.mjs', runsDir, '--diagnose', '--sandbox', 'mock'], { encoding: 'utf8' });\nassert(typeof JSON.parse(out).diagnosis === 'object', 'diagnosis object present for a valid winner+run record');\n\n// corrupt winner.json must degrade, not throw\nwriteFileSync(join(runsDir, '.metaharness', 'runs', 'winner.json'), '{not json');\nconst out2 = execFileSync('node', ['scripts/evolve.mjs', runsDir, '--diagnose', '--sandbox', 'mock'], { encoding: 'utf8' });\nassert(!!JSON.parse(out2).diagnosis, 'corrupt winner.json does not crash --diagnose');\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);"
|
|
71
|
+
},
|
|
72
|
+
{
|
|
73
|
+
"language": "js",
|
|
74
|
+
"code": "// scripts/test-gepa-ops.mjs\nimport { execFileSync } from 'node:child_process';\nlet passed = 0, failed = 0;\nconst assert = (c, l) => c ? passed++ : (failed++, console.log(`✗ ${l}`));\n\nfor (const op of ['validate', 'render']) {\n const out = execFileSync('node', ['scripts/gepa.mjs', '--op', op], { encoding: 'utf8' });\n const payload = JSON.parse(out);\n assert(payload.op === op || payload.degraded, `op=${op} returns a shaped payload or degrades cleanly`);\n}\n\ntry {\n execFileSync('node', ['scripts/gepa.mjs', '--op', 'analyze', '--transcript', '{\"not\":\"array\"}'], { encoding: 'utf8' });\n assert(false, 'non-array --transcript should exit non-zero');\n} catch (e) { assert(e.status === 2, 'non-array --transcript exits 2'); }\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);"
|
|
75
|
+
},
|
|
76
|
+
{
|
|
77
|
+
"language": "js",
|
|
78
|
+
"code": "// scripts/test-mcp-scan-malformed-finding.mjs\nimport { parseMcpScanText } from './_harness.mjs'; // if exported; else drive via fixture stdout + execFileSync\n// Fixture: upstream JSON with findings: [\"a raw string\", {severity:'high', message:'ok'}]\n// Assert: string elements are filtered out or coerced to a real object BEFORE spread,\n// not silently spread into {0:'a',1:' ',...}."
|
|
79
|
+
},
|
|
80
|
+
{
|
|
81
|
+
"language": "js",
|
|
82
|
+
"code": "// scripts/test-security-bench-parse.mjs\n// Requires: export function parseSecurityBenchMarkdown(md) { ... }\nimport { parseSecurityBenchMarkdown } from './security-bench.mjs';\nlet passed = 0, failed = 0;\nconst assert = (c, l) => c ? passed++ : (failed++, console.log(`✗ ${l}`));\n\nconst wellFormed = `## Overall: ✅ PASS\\n| Gate | Status |\\n|---|---|\\n| injection | ✅ |\\n`;\nassert(parseSecurityBenchMarkdown(wellFormed).gates.length > 0, 'well-formed markdown parses gates');\n\nconst malformed = `unexpected upstream format with no matching tables`;\nconst r2 = parseSecurityBenchMarkdown(malformed);\nassert(r2.gates.length === 0, 'malformed markdown returns empty gates, not throw');\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);"
|
|
83
|
+
}
|
|
84
|
+
]
|
|
85
|
+
},
|
|
86
|
+
"durationMs": 150058,
|
|
87
|
+
"model": "sonnet",
|
|
88
|
+
"sandboxMode": "permissive",
|
|
89
|
+
"workerType": "testgaps",
|
|
90
|
+
"timestamp": "2026-07-09T14:43:08.375Z",
|
|
91
|
+
"executionId": "testgaps_1783608038317_jrd66e"
|
|
92
|
+
}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
[2026-07-09T15:03:08.417Z] PROMPT
|
|
2
|
+
============================================================
|
|
3
|
+
Analyze test coverage and identify gaps:
|
|
4
|
+
- Find untested functions and classes
|
|
5
|
+
- Identify edge cases not covered
|
|
6
|
+
- Suggest new test scenarios
|
|
7
|
+
- Check for missing error handling tests
|
|
8
|
+
- Identify integration test gaps
|
|
9
|
+
|
|
10
|
+
For each gap, provide a test skeleton.
|
|
11
|
+
|
|
12
|
+
## Instructions
|
|
13
|
+
|
|
14
|
+
Analyze the codebase and provide your response following the format specified in the task.
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
[2026-07-09T15:07:01.357Z] RESULT
|
|
2
|
+
============================================================
|
|
3
|
+
{
|
|
4
|
+
"success": true,
|
|
5
|
+
"output": "## Test coverage analysis — `plugins/ruflo-metaharness`\n\nThe plugin has no framework — tests are plain `assert()`-based `.mjs` scripts run via `node scripts/test-*.mjs` (see `test-similarity.mjs` for the convention: `assert(cond, label)`, `--format json`, exit 1 on failure). Existing coverage is almost entirely **black-box CLI/subprocess tests** (`test-graceful-degradation.mjs`, `test-pipeline-roundtrip.mjs`, `test-parallel-pipeline.mjs`, `test-mcp-tools.mjs`) plus one genuine unit-test file (`test-similarity.mjs`, covering `_similarity.mjs` + parts of `_harness.mjs`). Benchmarks (`bench*.mjs`) exercise code paths but only assert performance, not correctness.\n\n### 1. Untested module: `_invoke.mjs` (highest priority — zero direct tests)\n\nThis is the shared plumbing layer (~15 scripts depend on it) and every exported function is pure or near-pure, yet none has a direct unit test — only indirect, best-effort coverage through full CLI subprocess runs.\n\n```js\n// scripts/test-invoke.mjs — unit tests for _invoke.mjs (currently untested)\nimport {\n classifyDegraded, injectJson, parseTrailingJson,\n satisfiesTildeRange, cacheBaseDir, DEGRADED_RX,\n} from './_invoke.mjs';\n\nlet passed = 0, failed = 0;\nfunction assert(cond, label) { cond ? passed++ : (failed++, console.log(`✗ ${label}`)); }\n\n// classifyDegraded\nassert(classifyDegraded('', null, 'x').reason === 'x-timeout', 'null exitCode -> timeout');\nassert(classifyDegraded('npm ERR! 404', 1, 'x').degraded === true, 'npm ERR matches DEGRADED_RX');\nassert(classifyDegraded('boom', 1, 'x').degraded === false, 'unmatched stderr -> not degraded');\nassert(classifyDegraded(undefined, 1, 'x').degraded === false, 'undefined stderr does not throw');\n\n// injectJson\nassert(injectJson(['--foo'], true).includes('--json'), 'appends --json when wanted');\nassert(injectJson(['--json'], true).filter(a => a === '--json').length === 1, 'no duplicate --json');\nassert(injectJson(['--foo'], false).includes('--json') === false, 'no --json when not wanted');\nassert(injectJson(['--foo'], true) !== ['--foo'], 'does not mutate/alias input array');\n\n// parseTrailingJson\nassert(parseTrailingJson('progress {a:1} Result: {\"ok\":true}').ok === true, 'grabs LAST json block, not first-looking-like-json text');\nassert(parseTrailingJson('{\"a\":{\"b\":1}}').a.b === 1, 'nested object via greedy fallback');\nassert(parseTrailingJson('no json here') === null, 'no JSON -> null');\nassert(parseTrailingJson('') === null, 'empty string -> null');\nassert(parseTrailingJson(undefined) === null, 'undefined stdout does not throw');\n\n// satisfiesTildeRange\nassert(satisfiesTildeRange('0.3.2', '~0.3.0') === true, 'patch >= pin patch satisfies');\nassert(satisfiesTildeRange('0.3.0', '~0.3.1') === false, 'patch < pin patch fails');\nassert(satisfiesTildeRange('0.4.0', '~0.3.0') === false, 'minor mismatch fails');\nassert(satisfiesTildeRange('not-a-version', '~0.3.0') === false, 'garbage version -> false, no throw');\nassert(satisfiesTildeRange('0.3.0', 'not-a-range') === false, 'garbage pin -> false, no throw');\n\n// cacheBaseDir (env override test seam)\nprocess.env.RUFLO_METAHARNESS_CACHE_BASE = '/tmp/x';\nassert(cacheBaseDir() === '/tmp/x', 'RUFLO_METAHARNESS_CACHE_BASE override respected');\ndelete process.env.RUFLO_METAHARNESS_CACHE_BASE;\n\nconsole.log(`${passed} passed, ${failed} failed`);\nprocess.exit(failed > 0 ? 1 : 0);\n```\n\nAlso missing: `findLocalPackageDir` (stale-major/minor-in-ancestor-node_modules is *skipped*, per its docstring — untested), `ensureCachedInstall` (cache-hit-short-circuit vs cold-install path — can be unit tested by pointing `RUFLO_METAHARNESS_CACHE_BASE` at a temp dir and stubbing `pkg`), and `importOptionalLibrary`'s \"recoverable vs rethrow\" branch (a non-`MODULE_NOT_FOUND` error, e.g. a syntax error in the imported module, should propagate — currently unverified).\n\n### 2. Untested safety guard: `mint.mjs` `safetyChecks()`\n\n`mint.mjs` has destructive-action guards (refuse to target the repo root or a path inside it, `--name`/`--template` required, default `--target` to a tmpdir) but `safetyChecks()` isn't exported and is explicitly excluded from `test-graceful-degradation.mjs`'s skill list (\"mint requires a `--name` argv to even start\"). **No test exercises the refusal paths at all.**\n\n```js\n// scripts/test-mint-safety.mjs — subprocess-level test since safetyChecks() isn't exported\nimport { spawnSync } from 'node:child_process';\nimport { join, dirname } from 'node:path';\nimport { fileURLToPath } from 'node:url';\n\nconst SCRIPT = join(dirname(fileURLToPath(import.meta.url)), 'mint.mjs');\nfunction run(args) { return spawnSync('node', [SCRIPT, ...args], { encoding: 'utf-8' }); }\n\nlet passed = 0, failed = 0;\nfunction assert(cond, label) { cond ? passed++ : (failed++, console.log(`✗ ${label}`)); }\n\nassert(run(['--template', 'minimal']).status === 2, 'missing --name exits 2');\nassert(run(['--name', 'x']).status === 2, 'missing --template exits 2');\n\nconst repoRoot = process.cwd();\nconst r1 = run(['--name', 'x', '--template', 'minimal', '--target', repoRoot]);\nassert(r1.status === 2 && /refusing to write to project root/.test(r1.stderr), 'refuses exact repo root');\n\nconst r2 = run(['--name', 'x', '--template', 'minimal', '--target', join(repoRoot, 'sub')]);\nassert(r2.status === 2 && /refusing to write inside/.test(r2.stderr), 'refuses path inside repo root');\n\nconsole.log(`${passed} passed, ${failed} failed`);\nprocess.exit(failed > 0 ? 1 : 0);\n```\n\n**Recommendation independent of the test:** export `safetyChecks` from `mint.mjs` so it can be unit-tested without spawning `node` each time (subprocess tests here cost ~100-300ms/case vs <1ms for an in-process call).\n\n### 3. Documented-but-unfixed bugs with no regression test\n\nPer `test-graceful-degradation.mjs`'s own comments (iter 55), three known bugs are carved out of the drill rather than fixed+covered:\n- `mint`'s degraded payload doesn't include the literal `\"degraded\": true` marker.\n- `drift-from-history` exits `2` (config error) instead of `3` (test-cannot-run) when there's no audit history yet.\n- `oia-audit` can exceed the 180s local timeout in the unreachable-registry case.\n\nNone of these have a regression test asserting the *correct* future behavior — when fixed, nothing will catch a re-regression.\n\n```js\n// Add to test-graceful-degradation.mjs once fixed, or as a standalone xfail-style test:\nconst mintR = run('mint.mjs', ['--name', 'x', '--template', 'minimal', '--target', '/tmp/mint-drill']);\nconst mintJson = JSON.parse(mintR.stdout);\nassert(mintJson.degraded === true, 'mint degraded payload includes literal degraded:true');\n\nconst dfhR = run('drift-from-history.mjs', ['--format', 'json']); // no prior audit records\nassert(dfhR.exitCode === 3, 'drift-from-history exits 3 (test-cannot-run) on empty history, not 2');\n```\n\n### 4. `audit-list.mjs` — untested pure function `parseDurationMs()`\n\n```js\n// scripts/test-audit-list.mjs\n// parseDurationMs isn't exported today — export it first, then:\nimport { parseDurationMs } from './audit-list.mjs';\nassert(parseDurationMs('30d') === 30 * 86_400_000, '30d parses to ms');\nassert(parseDurationMs('1w') !== null, 'w unit supported');\nassert(parseDurationMs('abc') === null, 'garbage spec returns null, not throw');\nassert(parseDurationMs('') === null, 'empty spec returns null');\nassert(parseDurationMs('-5d') === null, 'negative count rejected'); // regex requires \\d+, confirm no coercion bug\n```\n\n### 5. `router-parallel-analyze.mjs` — partial-pass / boundary cases missing\n\n`test-parallel-pipeline.mjs` only checks the all-pass and all-fail fixtures for the 3-criteria AND-gate. It never verifies **AND semantics under partial failure** (e.g. quality passes, cost fails, latency passes → must still be non-promotable) or **exact-threshold boundaries** (`qualityImprovementPct === 2` should fail since the check is `> 2`, not `>= 2`).\n\n```js\n// Extend test-parallel-pipeline.mjs (or a new test-router-gate.mjs) with fixtures where:\n// - exactly 2 of 3 criteria pass -> verdict.promotable === false (guards against an accidental OR)\n// - qualityImprovementPct is exactly 2.00, usdIncreasePct exactly 1.00, latencyIncreasePct exactly 5.00\n// -> all three individually false (boundary is exclusive)\n```\n\n### 6. Integration gaps: `redblue.mjs`, `learn.mjs`, `gepa.mjs`\n\n`test-mcp-tools.mjs` only proves the MCP tool wrapper's handler is *callable* with minimal input (structural contract), not that `redblue.mjs`/`learn.mjs`/`gepa.mjs`'s actual CLI behavior is correct end-to-end (e.g. redblue's 4-gate human-gated submit flow, `--mock-judge` vs real-judge cost capping, `gepa --op analyze` failure-mode classification). `test-graceful-degradation.mjs` also doesn't include these three in its skill list, so their degraded-path behavior (`_redblue.mjs`'s `emitRedblueDegradedJsonAndExit`, `_darwin.mjs`'s `importGepa`) is unverified with an unreachable registry.\n\n```js\n// Extend test-graceful-degradation.mjs's `skills` array:\n{ name: 'redblue', args: ['run', '--mock-judge', '--tests', '1', '--format', 'json'] },\n{ name: 'gepa', args: ['--op', 'genome', '--format', 'json'] },\n```\n\n### Summary\n\n| Gap | File | Severity |\n|---|---|---|\n| Zero unit tests for shared plumbing (`classifyDegraded`, `parseTrailingJson`, `satisfiesTildeRange`, etc.) | `_invoke.mjs` | High — used by ~15 scripts |\n| Destructive-action guard untested | `mint.mjs` `safetyChecks()` | High — safety-critical, currently unreachable by any test |\n| 3 known bugs with no regression test | `mint.mjs`, `drift-from-history.mjs`, `oia-audit.mjs` | Medium — will silently re-regress |\n| Pure helper untested | `audit-list.mjs` `parseDurationMs()` | Low |\n| AND-gate boundary/partial-fail cases | `router-parallel-analyze.mjs` | Medium — could hide an OR-vs-AND logic bug |\n| Degraded-path untested for 3 scripts | `redblue.mjs`, `learn.mjs`, `gepa.mjs` | Medium — ADR-150 constraint unverified for these |\n\nI didn't run these skeletons — they're illustrative gaps to fill, following the existing plain-assert convention. Want me to actually write and wire up any of these as real files (e.g. `test-invoke.mjs` and `test-mint-safety.mjs` are the two highest-value ones)?\n",
|
|
6
|
+
"parsedOutput": {
|
|
7
|
+
"sections": [
|
|
8
|
+
{
|
|
9
|
+
"title": "Test coverage analysis — `plugins/ruflo-metaharness`",
|
|
10
|
+
"content": "\nThe plugin has no framework — tests are plain `assert()`-based `.mjs` scripts run via `node scripts/test-*.mjs` (see `test-similarity.mjs` for the convention: `assert(cond, label)`, `--format json`, exit 1 on failure). Existing coverage is almost entirely **black-box CLI/subprocess tests** (`test-graceful-degradation.mjs`, `test-pipeline-roundtrip.mjs`, `test-parallel-pipeline.mjs`, `test-mcp-tools.mjs`) plus one genuine unit-test file (`test-similarity.mjs`, covering `_similarity.mjs` + parts of `_harness.mjs`). Benchmarks (`bench*.mjs`) exercise code paths but only assert performance, not correctness.\n\n",
|
|
11
|
+
"level": 2
|
|
12
|
+
},
|
|
13
|
+
{
|
|
14
|
+
"title": "1. Untested module: `_invoke.mjs` (highest priority — zero direct tests)",
|
|
15
|
+
"content": "\nThis is the shared plumbing layer (~15 scripts depend on it) and every exported function is pure or near-pure, yet none has a direct unit test — only indirect, best-effort coverage through full CLI subprocess runs.\n\n```js\n// scripts/test-invoke.mjs — unit tests for _invoke.mjs (currently untested)\nimport {\n classifyDegraded, injectJson, parseTrailingJson,\n satisfiesTildeRange, cacheBaseDir, DEGRADED_RX,\n} from './_invoke.mjs';\n\nlet passed = 0, failed = 0;\nfunction assert(cond, label) { cond ? passed++ : (failed++, console.log(`✗ ${label}`)); }\n\n// classifyDegraded\nassert(classifyDegraded('', null, 'x').reason === 'x-timeout', 'null exitCode -> timeout');\nassert(classifyDegraded('npm ERR! 404', 1, 'x').degraded === true, 'npm ERR matches DEGRADED_RX');\nassert(classifyDegraded('boom', 1, 'x').degraded === false, 'unmatched stderr -> not degraded');\nassert(classifyDegraded(undefined, 1, 'x').degraded === false, 'undefined stderr does not throw');\n\n// injectJson\nassert(injectJson(['--foo'], true).includes('--json'), 'appends --json when wanted');\nassert(injectJson(['--json'], true).filter(a => a === '--json').length === 1, 'no duplicate --json');\nassert(injectJson(['--foo'], false).includes('--json') === false, 'no --json when not wanted');\nassert(injectJson(['--foo'], true) !== ['--foo'], 'does not mutate/alias input array');\n\n// parseTrailingJson\nassert(parseTrailingJson('progress {a:1} Result: {\"ok\":true}').ok === true, 'grabs LAST json block, not first-looking-like-json text');\nassert(parseTrailingJson('{\"a\":{\"b\":1}}').a.b === 1, 'nested object via greedy fallback');\nassert(parseTrailingJson('no json here') === null, 'no JSON -> null');\nassert(parseTrailingJson('') === null, 'empty string -> null');\nassert(parseTrailingJson(undefined) === null, 'undefined stdout does not throw');\n\n// satisfiesTildeRange\nassert(satisfiesTildeRange('0.3.2', '~0.3.0') === true, 'patch >= pin patch satisfies');\nassert(satisfiesTildeRange('0.3.0', '~0.3.1') === false, 'patch < pin patch fails');\nassert(satisfiesTildeRange('0.4.0', '~0.3.0') === false, 'minor mismatch fails');\nassert(satisfiesTildeRange('not-a-version', '~0.3.0') === false, 'garbage version -> false, no throw');\nassert(satisfiesTildeRange('0.3.0', 'not-a-range') === false, 'garbage pin -> false, no throw');\n\n// cacheBaseDir (env override test seam)\nprocess.env.RUFLO_METAHARNESS_CACHE_BASE = '/tmp/x';\nassert(cacheBaseDir() === '/tmp/x', 'RUFLO_METAHARNESS_CACHE_BASE override respected');\ndelete process.env.RUFLO_METAHARNESS_CACHE_BASE;\n\nconsole.log(`${passed} passed, ${failed} failed`);\nprocess.exit(failed > 0 ? 1 : 0);\n```\n\nAlso missing: `findLocalPackageDir` (stale-major/minor-in-ancestor-node_modules is *skipped*, per its docstring — untested), `ensureCachedInstall` (cache-hit-short-circuit vs cold-install path — can be unit tested by pointing `RUFLO_METAHARNESS_CACHE_BASE` at a temp dir and stubbing `pkg`), and `importOptionalLibrary`'s \"recoverable vs rethrow\" branch (a non-`MODULE_NOT_FOUND` error, e.g. a syntax error in the imported module, should propagate — currently unverified).\n\n",
|
|
16
|
+
"level": 3
|
|
17
|
+
},
|
|
18
|
+
{
|
|
19
|
+
"title": "2. Untested safety guard: `mint.mjs` `safetyChecks()`",
|
|
20
|
+
"content": "\n`mint.mjs` has destructive-action guards (refuse to target the repo root or a path inside it, `--name`/`--template` required, default `--target` to a tmpdir) but `safetyChecks()` isn't exported and is explicitly excluded from `test-graceful-degradation.mjs`'s skill list (\"mint requires a `--name` argv to even start\"). **No test exercises the refusal paths at all.**\n\n```js\n// scripts/test-mint-safety.mjs — subprocess-level test since safetyChecks() isn't exported\nimport { spawnSync } from 'node:child_process';\nimport { join, dirname } from 'node:path';\nimport { fileURLToPath } from 'node:url';\n\nconst SCRIPT = join(dirname(fileURLToPath(import.meta.url)), 'mint.mjs');\nfunction run(args) { return spawnSync('node', [SCRIPT, ...args], { encoding: 'utf-8' }); }\n\nlet passed = 0, failed = 0;\nfunction assert(cond, label) { cond ? passed++ : (failed++, console.log(`✗ ${label}`)); }\n\nassert(run(['--template', 'minimal']).status === 2, 'missing --name exits 2');\nassert(run(['--name', 'x']).status === 2, 'missing --template exits 2');\n\nconst repoRoot = process.cwd();\nconst r1 = run(['--name', 'x', '--template', 'minimal', '--target', repoRoot]);\nassert(r1.status === 2 && /refusing to write to project root/.test(r1.stderr), 'refuses exact repo root');\n\nconst r2 = run(['--name', 'x', '--template', 'minimal', '--target', join(repoRoot, 'sub')]);\nassert(r2.status === 2 && /refusing to write inside/.test(r2.stderr), 'refuses path inside repo root');\n\nconsole.log(`${passed} passed, ${failed} failed`);\nprocess.exit(failed > 0 ? 1 : 0);\n```\n\n**Recommendation independent of the test:** export `safetyChecks` from `mint.mjs` so it can be unit-tested without spawning `node` each time (subprocess tests here cost ~100-300ms/case vs <1ms for an in-process call).\n\n",
|
|
21
|
+
"level": 3
|
|
22
|
+
},
|
|
23
|
+
{
|
|
24
|
+
"title": "3. Documented-but-unfixed bugs with no regression test",
|
|
25
|
+
"content": "\nPer `test-graceful-degradation.mjs`'s own comments (iter 55), three known bugs are carved out of the drill rather than fixed+covered:\n- `mint`'s degraded payload doesn't include the literal `\"degraded\": true` marker.\n- `drift-from-history` exits `2` (config error) instead of `3` (test-cannot-run) when there's no audit history yet.\n- `oia-audit` can exceed the 180s local timeout in the unreachable-registry case.\n\nNone of these have a regression test asserting the *correct* future behavior — when fixed, nothing will catch a re-regression.\n\n```js\n// Add to test-graceful-degradation.mjs once fixed, or as a standalone xfail-style test:\nconst mintR = run('mint.mjs', ['--name', 'x', '--template', 'minimal', '--target', '/tmp/mint-drill']);\nconst mintJson = JSON.parse(mintR.stdout);\nassert(mintJson.degraded === true, 'mint degraded payload includes literal degraded:true');\n\nconst dfhR = run('drift-from-history.mjs', ['--format', 'json']); // no prior audit records\nassert(dfhR.exitCode === 3, 'drift-from-history exits 3 (test-cannot-run) on empty history, not 2');\n```\n\n",
|
|
26
|
+
"level": 3
|
|
27
|
+
},
|
|
28
|
+
{
|
|
29
|
+
"title": "4. `audit-list.mjs` — untested pure function `parseDurationMs()`",
|
|
30
|
+
"content": "\n```js\n// scripts/test-audit-list.mjs\n// parseDurationMs isn't exported today — export it first, then:\nimport { parseDurationMs } from './audit-list.mjs';\nassert(parseDurationMs('30d') === 30 * 86_400_000, '30d parses to ms');\nassert(parseDurationMs('1w') !== null, 'w unit supported');\nassert(parseDurationMs('abc') === null, 'garbage spec returns null, not throw');\nassert(parseDurationMs('') === null, 'empty spec returns null');\nassert(parseDurationMs('-5d') === null, 'negative count rejected'); // regex requires \\d+, confirm no coercion bug\n```\n\n",
|
|
31
|
+
"level": 3
|
|
32
|
+
},
|
|
33
|
+
{
|
|
34
|
+
"title": "5. `router-parallel-analyze.mjs` — partial-pass / boundary cases missing",
|
|
35
|
+
"content": "\n`test-parallel-pipeline.mjs` only checks the all-pass and all-fail fixtures for the 3-criteria AND-gate. It never verifies **AND semantics under partial failure** (e.g. quality passes, cost fails, latency passes → must still be non-promotable) or **exact-threshold boundaries** (`qualityImprovementPct === 2` should fail since the check is `> 2`, not `>= 2`).\n\n```js\n// Extend test-parallel-pipeline.mjs (or a new test-router-gate.mjs) with fixtures where:\n// - exactly 2 of 3 criteria pass -> verdict.promotable === false (guards against an accidental OR)\n// - qualityImprovementPct is exactly 2.00, usdIncreasePct exactly 1.00, latencyIncreasePct exactly 5.00\n// -> all three individually false (boundary is exclusive)\n```\n\n",
|
|
36
|
+
"level": 3
|
|
37
|
+
},
|
|
38
|
+
{
|
|
39
|
+
"title": "6. Integration gaps: `redblue.mjs`, `learn.mjs`, `gepa.mjs`",
|
|
40
|
+
"content": "\n`test-mcp-tools.mjs` only proves the MCP tool wrapper's handler is *callable* with minimal input (structural contract), not that `redblue.mjs`/`learn.mjs`/`gepa.mjs`'s actual CLI behavior is correct end-to-end (e.g. redblue's 4-gate human-gated submit flow, `--mock-judge` vs real-judge cost capping, `gepa --op analyze` failure-mode classification). `test-graceful-degradation.mjs` also doesn't include these three in its skill list, so their degraded-path behavior (`_redblue.mjs`'s `emitRedblueDegradedJsonAndExit`, `_darwin.mjs`'s `importGepa`) is unverified with an unreachable registry.\n\n```js\n// Extend test-graceful-degradation.mjs's `skills` array:\n{ name: 'redblue', args: ['run', '--mock-judge', '--tests', '1', '--format', 'json'] },\n{ name: 'gepa', args: ['--op', 'genome', '--format', 'json'] },\n```\n\n",
|
|
41
|
+
"level": 3
|
|
42
|
+
},
|
|
43
|
+
{
|
|
44
|
+
"title": "Summary",
|
|
45
|
+
"content": "| Gap | File | Severity |\n|---|---|---|\n| Zero unit tests for shared plumbing (`classifyDegraded`, `parseTrailingJson`, `satisfiesTildeRange`, etc.) | `_invoke.mjs` | High — used by ~15 scripts |\n| Destructive-action guard untested | `mint.mjs` `safetyChecks()` | High — safety-critical, currently unreachable by any test |\n| 3 known bugs with no regression test | `mint.mjs`, `drift-from-history.mjs`, `oia-audit.mjs` | Medium — will silently re-regress |\n| Pure helper untested | `audit-list.mjs` `parseDurationMs()` | Low |\n| AND-gate boundary/partial-fail cases | `router-parallel-analyze.mjs` | Medium — could hide an OR-vs-AND logic bug |\n| Degraded-path untested for 3 scripts | `redblue.mjs`, `learn.mjs`, `gepa.mjs` | Medium — ADR-150 constraint unverified for these |\n\nI didn't run these skeletons — they're illustrative gaps to fill, following the existing plain-assert convention. Want me to actually write and wire up any of these as real files (e.g. `test-invoke.mjs` and `test-mint-safety.mjs` are the two highest-value ones)?",
|
|
46
|
+
"level": 3
|
|
47
|
+
}
|
|
48
|
+
],
|
|
49
|
+
"codeBlocks": [
|
|
50
|
+
{
|
|
51
|
+
"language": "js",
|
|
52
|
+
"code": "// scripts/test-invoke.mjs — unit tests for _invoke.mjs (currently untested)\nimport {\n classifyDegraded, injectJson, parseTrailingJson,\n satisfiesTildeRange, cacheBaseDir, DEGRADED_RX,\n} from './_invoke.mjs';\n\nlet passed = 0, failed = 0;\nfunction assert(cond, label) { cond ? passed++ : (failed++, console.log(`✗ ${label}`)); }\n\n// classifyDegraded\nassert(classifyDegraded('', null, 'x').reason === 'x-timeout', 'null exitCode -> timeout');\nassert(classifyDegraded('npm ERR! 404', 1, 'x').degraded === true, 'npm ERR matches DEGRADED_RX');\nassert(classifyDegraded('boom', 1, 'x').degraded === false, 'unmatched stderr -> not degraded');\nassert(classifyDegraded(undefined, 1, 'x').degraded === false, 'undefined stderr does not throw');\n\n// injectJson\nassert(injectJson(['--foo'], true).includes('--json'), 'appends --json when wanted');\nassert(injectJson(['--json'], true).filter(a => a === '--json').length === 1, 'no duplicate --json');\nassert(injectJson(['--foo'], false).includes('--json') === false, 'no --json when not wanted');\nassert(injectJson(['--foo'], true) !== ['--foo'], 'does not mutate/alias input array');\n\n// parseTrailingJson\nassert(parseTrailingJson('progress {a:1} Result: {\"ok\":true}').ok === true, 'grabs LAST json block, not first-looking-like-json text');\nassert(parseTrailingJson('{\"a\":{\"b\":1}}').a.b === 1, 'nested object via greedy fallback');\nassert(parseTrailingJson('no json here') === null, 'no JSON -> null');\nassert(parseTrailingJson('') === null, 'empty string -> null');\nassert(parseTrailingJson(undefined) === null, 'undefined stdout does not throw');\n\n// satisfiesTildeRange\nassert(satisfiesTildeRange('0.3.2', '~0.3.0') === true, 'patch >= pin patch satisfies');\nassert(satisfiesTildeRange('0.3.0', '~0.3.1') === false, 'patch < pin patch fails');\nassert(satisfiesTildeRange('0.4.0', '~0.3.0') === false, 'minor mismatch fails');\nassert(satisfiesTildeRange('not-a-version', '~0.3.0') === false, 'garbage version -> false, no throw');\nassert(satisfiesTildeRange('0.3.0', 'not-a-range') === false, 'garbage pin -> false, no throw');\n\n// cacheBaseDir (env override test seam)\nprocess.env.RUFLO_METAHARNESS_CACHE_BASE = '/tmp/x';\nassert(cacheBaseDir() === '/tmp/x', 'RUFLO_METAHARNESS_CACHE_BASE override respected');\ndelete process.env.RUFLO_METAHARNESS_CACHE_BASE;\n\nconsole.log(`${passed} passed, ${failed} failed`);\nprocess.exit(failed > 0 ? 1 : 0);"
|
|
53
|
+
},
|
|
54
|
+
{
|
|
55
|
+
"language": "js",
|
|
56
|
+
"code": "// scripts/test-mint-safety.mjs — subprocess-level test since safetyChecks() isn't exported\nimport { spawnSync } from 'node:child_process';\nimport { join, dirname } from 'node:path';\nimport { fileURLToPath } from 'node:url';\n\nconst SCRIPT = join(dirname(fileURLToPath(import.meta.url)), 'mint.mjs');\nfunction run(args) { return spawnSync('node', [SCRIPT, ...args], { encoding: 'utf-8' }); }\n\nlet passed = 0, failed = 0;\nfunction assert(cond, label) { cond ? passed++ : (failed++, console.log(`✗ ${label}`)); }\n\nassert(run(['--template', 'minimal']).status === 2, 'missing --name exits 2');\nassert(run(['--name', 'x']).status === 2, 'missing --template exits 2');\n\nconst repoRoot = process.cwd();\nconst r1 = run(['--name', 'x', '--template', 'minimal', '--target', repoRoot]);\nassert(r1.status === 2 && /refusing to write to project root/.test(r1.stderr), 'refuses exact repo root');\n\nconst r2 = run(['--name', 'x', '--template', 'minimal', '--target', join(repoRoot, 'sub')]);\nassert(r2.status === 2 && /refusing to write inside/.test(r2.stderr), 'refuses path inside repo root');\n\nconsole.log(`${passed} passed, ${failed} failed`);\nprocess.exit(failed > 0 ? 1 : 0);"
|
|
57
|
+
},
|
|
58
|
+
{
|
|
59
|
+
"language": "js",
|
|
60
|
+
"code": "// Add to test-graceful-degradation.mjs once fixed, or as a standalone xfail-style test:\nconst mintR = run('mint.mjs', ['--name', 'x', '--template', 'minimal', '--target', '/tmp/mint-drill']);\nconst mintJson = JSON.parse(mintR.stdout);\nassert(mintJson.degraded === true, 'mint degraded payload includes literal degraded:true');\n\nconst dfhR = run('drift-from-history.mjs', ['--format', 'json']); // no prior audit records\nassert(dfhR.exitCode === 3, 'drift-from-history exits 3 (test-cannot-run) on empty history, not 2');"
|
|
61
|
+
},
|
|
62
|
+
{
|
|
63
|
+
"language": "js",
|
|
64
|
+
"code": "// scripts/test-audit-list.mjs\n// parseDurationMs isn't exported today — export it first, then:\nimport { parseDurationMs } from './audit-list.mjs';\nassert(parseDurationMs('30d') === 30 * 86_400_000, '30d parses to ms');\nassert(parseDurationMs('1w') !== null, 'w unit supported');\nassert(parseDurationMs('abc') === null, 'garbage spec returns null, not throw');\nassert(parseDurationMs('') === null, 'empty spec returns null');\nassert(parseDurationMs('-5d') === null, 'negative count rejected'); // regex requires \\d+, confirm no coercion bug"
|
|
65
|
+
},
|
|
66
|
+
{
|
|
67
|
+
"language": "js",
|
|
68
|
+
"code": "// Extend test-parallel-pipeline.mjs (or a new test-router-gate.mjs) with fixtures where:\n// - exactly 2 of 3 criteria pass -> verdict.promotable === false (guards against an accidental OR)\n// - qualityImprovementPct is exactly 2.00, usdIncreasePct exactly 1.00, latencyIncreasePct exactly 5.00\n// -> all three individually false (boundary is exclusive)"
|
|
69
|
+
},
|
|
70
|
+
{
|
|
71
|
+
"language": "js",
|
|
72
|
+
"code": "// Extend test-graceful-degradation.mjs's `skills` array:\n{ name: 'redblue', args: ['run', '--mock-judge', '--tests', '1', '--format', 'json'] },\n{ name: 'gepa', args: ['--op', 'genome', '--format', 'json'] },"
|
|
73
|
+
}
|
|
74
|
+
]
|
|
75
|
+
},
|
|
76
|
+
"durationMs": 232941,
|
|
77
|
+
"model": "sonnet",
|
|
78
|
+
"sandboxMode": "permissive",
|
|
79
|
+
"workerType": "testgaps",
|
|
80
|
+
"timestamp": "2026-07-09T15:07:01.357Z",
|
|
81
|
+
"executionId": "testgaps_1783609388416_mc3zoe"
|
|
82
|
+
}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
[2026-07-09T15:27:01.385Z] PROMPT
|
|
2
|
+
============================================================
|
|
3
|
+
Analyze test coverage and identify gaps:
|
|
4
|
+
- Find untested functions and classes
|
|
5
|
+
- Identify edge cases not covered
|
|
6
|
+
- Suggest new test scenarios
|
|
7
|
+
- Check for missing error handling tests
|
|
8
|
+
- Identify integration test gaps
|
|
9
|
+
|
|
10
|
+
For each gap, provide a test skeleton.
|
|
11
|
+
|
|
12
|
+
## Instructions
|
|
13
|
+
|
|
14
|
+
Analyze the codebase and provide your response following the format specified in the task.
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
[2026-07-09T15:27:03.262Z] RESULT
|
|
2
|
+
============================================================
|
|
3
|
+
{
|
|
4
|
+
"success": false,
|
|
5
|
+
"output": "You've hit your session limit · resets 2pm (America/Toronto)\n",
|
|
6
|
+
"parsedOutput": {
|
|
7
|
+
"sections": [],
|
|
8
|
+
"codeBlocks": []
|
|
9
|
+
},
|
|
10
|
+
"durationMs": 1878,
|
|
11
|
+
"model": "sonnet",
|
|
12
|
+
"sandboxMode": "permissive",
|
|
13
|
+
"workerType": "testgaps",
|
|
14
|
+
"timestamp": "2026-07-09T15:27:03.262Z",
|
|
15
|
+
"executionId": "testgaps_1783610821384_tzqvlb",
|
|
16
|
+
"error": "Process exited with code 1"
|
|
17
|
+
}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
[2026-07-09T15:47:03.286Z] PROMPT
|
|
2
|
+
============================================================
|
|
3
|
+
Analyze test coverage and identify gaps:
|
|
4
|
+
- Find untested functions and classes
|
|
5
|
+
- Identify edge cases not covered
|
|
6
|
+
- Suggest new test scenarios
|
|
7
|
+
- Check for missing error handling tests
|
|
8
|
+
- Identify integration test gaps
|
|
9
|
+
|
|
10
|
+
For each gap, provide a test skeleton.
|
|
11
|
+
|
|
12
|
+
## Instructions
|
|
13
|
+
|
|
14
|
+
Analyze the codebase and provide your response following the format specified in the task.
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
[2026-07-09T15:53:19.262Z] RESULT
|
|
2
|
+
============================================================
|
|
3
|
+
{
|
|
4
|
+
"success": true,
|
|
5
|
+
"output": "## Test Coverage Analysis — `plugins/ruflo-metaharness`\n\nNo formal test framework exists here — just 6 ad-hoc `test-*.mjs` smoke/E2E scripts covering ~34 plain `.mjs` files (~7,000 lines). Full report:\n\n## Current Coverage Summary\n- **`test-graceful-degradation.mjs`** — drills ADR-150 rule #3 (registry unreachable → all scripts degrade cleanly) across 8 of the scripts. Misses `mint`, `similarity`, `redblue`, `learn`, `gepa`, `drift-from-history`, `router-parallel-analyze`.\n- **`test-mcp-tools.mjs`** — smoke-tests the 15 MCP tool handlers (envelope shape only for most; deeper checks for 4 tools).\n- **`test-parallel-pipeline.mjs`** — E2E for the router parallel-logging AND-gate, using a hand-synthesized JSONL fixture (not the real TS writer).\n- **`test-pipeline-roundtrip.mjs`** — deepest test: real `oia-audit --dry-run` chained through `audit-trend` → `drift-from-history`, 12 stages.\n- **`test-similarity.mjs`** — pure unit tests of `_similarity.mjs` internals + a frozen regression anchor.\n- **`test-with-openrouter.mjs`** — real-money E2E via GCP-secret-gated OpenRouter key + **upstream** `metaharness` CLI directly (bypasses this plugin's `mint.mjs` wrapper entirely).\n\n## Highest-priority finding — likely a real bug\n`threat-model.mjs` defines its own local `SEVERITY_RANK = {clean,low,medium,high}` instead of importing the shared frozen table from `_harness.mjs` (which uses `info/warn/error/critical`). Since the two vocabularies don't match, `SEVERITY_RANK[worst] >= threshold` evaluates to `undefined >= N === false` whenever upstream reports `critical` — meaning `--fail-on high` **silently never fires** on a critical finding. No test catches this today.\n\n## Untested functions/classes (selected)\n| File | Function | Why it matters |\n|---|---|---|\n| `mint.mjs` | `safetyChecks()` | Path-traversal refusals never exercised by any script in-repo |\n| `_invoke.mjs` | `parseTrailingJson()` | Exists specifically to fix a documented past bug (progress-line JSON shadowing); no regression test |\n| `_invoke.mjs` | `satisfiesTildeRange()` | Pre-release/malformed version strings unhandled by any test |\n| `evolve.mjs` | `buildDiagnosis()`, `extractGepaTranscripts()` | Zero fixture-driven coverage of the 3 fallback branches |\n| `security-bench.mjs` | `parseSecurityBenchMarkdown()` | Regex-based parser never fed a canned markdown fixture |\n| `redblue.mjs` | `buildUpstreamArgs()` | Only the `attack` branch is exercised; `init/run/patch/report` untested |\n| `drift-from-history.mjs` | `--baseline-key` path | Has a documented-but-unfixed exit-code bug (2 vs 3) with no test locking behavior |\n| `gepa.mjs` | `validate`/`render`/`analyze` ops | Only `genome` op tested |\n\n## Edge cases not covered\n- `_similarity.mjs`: zero-length vectors, negative cost inputs (`Math.log10` → NaN risk), all-zero weight overrides\n- `audit-trend.mjs`/`drift-from-history.mjs`: malformed baseline JSON, missing `composite` field, conflicting `--baseline-key`+`--baseline-file`, ignored `--baseline-since`\n- `oia-audit.mjs`: partial degradation (2-of-5 sub-audits down) — only all-degraded/all-healthy tested\n- Concurrency: no test simulates one of 5 parallel sub-audits timing out while others succeed\n\n## Missing error-handling tests\n- `mint.mjs` \"target already exists\" exit-2 guard\n- `gepa.mjs` file-not-found / invalid-JSON `process.exit(2)` branches\n- `evolve.mjs` exit code 99 (safety-disqualified) — zero coverage\n- `_darwin.mjs` subprocess spawn-error path (`npx`/`node` binary missing)\n- `_harness.mjs` `readBinMap()` returning `null` for malformed `package.json`\n\n## Integration gaps\n- `npx @claude-flow/cli memory {store,retrieve,list}` shell-outs — never verified against a real memory backend round-trip\n- `.swarm/router-parallel.jsonl` real writer (`recordPair`/`recordPairOutcome`) never actually invoked — only source-grepped\n- `npx metaharness` cache-install \"stale/partial/corrupted\" middle states — only the two extremes (fully missing / fully available via upstream bypass) are tested\n- OpenRouter real-judge path (`redblue run --patch`, `learn --run`) has no CI-runnable coverage at all\n\n## Suggested new tests (18 concrete scenarios with `node:test` skeletons)\nThe agent produced 18 ready-to-adapt skeletons targeting the gaps above — most notably #1 (threat-model severity bug regression test), #3 (mint path-traversal), #4 (drift-from-history flag precedence), and #12 (security-bench markdown parser). Several target functions (`parseSecurityBenchMarkdown`, `parseDurationMs`, `median`/`pctile`, `buildDiagnosis`) would need an `export` added — harmless since these files run as CLI scripts, not imports — to become unit-testable without full subprocess spawns.\n\nWant me to write these test files (starting with the threat-model severity fix + regression test, since that looks like a live bug) or just leave this as the analysis?\n",
|
|
6
|
+
"parsedOutput": {
|
|
7
|
+
"sections": [
|
|
8
|
+
{
|
|
9
|
+
"title": "Test Coverage Analysis — `plugins/ruflo-metaharness`",
|
|
10
|
+
"content": "\nNo formal test framework exists here — just 6 ad-hoc `test-*.mjs` smoke/E2E scripts covering ~34 plain `.mjs` files (~7,000 lines). Full report:\n\n",
|
|
11
|
+
"level": 2
|
|
12
|
+
},
|
|
13
|
+
{
|
|
14
|
+
"title": "Current Coverage Summary",
|
|
15
|
+
"content": "- **`test-graceful-degradation.mjs`** — drills ADR-150 rule #3 (registry unreachable → all scripts degrade cleanly) across 8 of the scripts. Misses `mint`, `similarity`, `redblue`, `learn`, `gepa`, `drift-from-history`, `router-parallel-analyze`.\n- **`test-mcp-tools.mjs`** — smoke-tests the 15 MCP tool handlers (envelope shape only for most; deeper checks for 4 tools).\n- **`test-parallel-pipeline.mjs`** — E2E for the router parallel-logging AND-gate, using a hand-synthesized JSONL fixture (not the real TS writer).\n- **`test-pipeline-roundtrip.mjs`** — deepest test: real `oia-audit --dry-run` chained through `audit-trend` → `drift-from-history`, 12 stages.\n- **`test-similarity.mjs`** — pure unit tests of `_similarity.mjs` internals + a frozen regression anchor.\n- **`test-with-openrouter.mjs`** — real-money E2E via GCP-secret-gated OpenRouter key + **upstream** `metaharness` CLI directly (bypasses this plugin's `mint.mjs` wrapper entirely).\n\n",
|
|
16
|
+
"level": 2
|
|
17
|
+
},
|
|
18
|
+
{
|
|
19
|
+
"title": "Highest-priority finding — likely a real bug",
|
|
20
|
+
"content": "`threat-model.mjs` defines its own local `SEVERITY_RANK = {clean,low,medium,high}` instead of importing the shared frozen table from `_harness.mjs` (which uses `info/warn/error/critical`). Since the two vocabularies don't match, `SEVERITY_RANK[worst] >= threshold` evaluates to `undefined >= N === false` whenever upstream reports `critical` — meaning `--fail-on high` **silently never fires** on a critical finding. No test catches this today.\n\n",
|
|
21
|
+
"level": 2
|
|
22
|
+
},
|
|
23
|
+
{
|
|
24
|
+
"title": "Untested functions/classes (selected)",
|
|
25
|
+
"content": "| File | Function | Why it matters |\n|---|---|---|\n| `mint.mjs` | `safetyChecks()` | Path-traversal refusals never exercised by any script in-repo |\n| `_invoke.mjs` | `parseTrailingJson()` | Exists specifically to fix a documented past bug (progress-line JSON shadowing); no regression test |\n| `_invoke.mjs` | `satisfiesTildeRange()` | Pre-release/malformed version strings unhandled by any test |\n| `evolve.mjs` | `buildDiagnosis()`, `extractGepaTranscripts()` | Zero fixture-driven coverage of the 3 fallback branches |\n| `security-bench.mjs` | `parseSecurityBenchMarkdown()` | Regex-based parser never fed a canned markdown fixture |\n| `redblue.mjs` | `buildUpstreamArgs()` | Only the `attack` branch is exercised; `init/run/patch/report` untested |\n| `drift-from-history.mjs` | `--baseline-key` path | Has a documented-but-unfixed exit-code bug (2 vs 3) with no test locking behavior |\n| `gepa.mjs` | `validate`/`render`/`analyze` ops | Only `genome` op tested |\n\n",
|
|
26
|
+
"level": 2
|
|
27
|
+
},
|
|
28
|
+
{
|
|
29
|
+
"title": "Edge cases not covered",
|
|
30
|
+
"content": "- `_similarity.mjs`: zero-length vectors, negative cost inputs (`Math.log10` → NaN risk), all-zero weight overrides\n- `audit-trend.mjs`/`drift-from-history.mjs`: malformed baseline JSON, missing `composite` field, conflicting `--baseline-key`+`--baseline-file`, ignored `--baseline-since`\n- `oia-audit.mjs`: partial degradation (2-of-5 sub-audits down) — only all-degraded/all-healthy tested\n- Concurrency: no test simulates one of 5 parallel sub-audits timing out while others succeed\n\n",
|
|
31
|
+
"level": 2
|
|
32
|
+
},
|
|
33
|
+
{
|
|
34
|
+
"title": "Missing error-handling tests",
|
|
35
|
+
"content": "- `mint.mjs` \"target already exists\" exit-2 guard\n- `gepa.mjs` file-not-found / invalid-JSON `process.exit(2)` branches\n- `evolve.mjs` exit code 99 (safety-disqualified) — zero coverage\n- `_darwin.mjs` subprocess spawn-error path (`npx`/`node` binary missing)\n- `_harness.mjs` `readBinMap()` returning `null` for malformed `package.json`\n\n",
|
|
36
|
+
"level": 2
|
|
37
|
+
},
|
|
38
|
+
{
|
|
39
|
+
"title": "Integration gaps",
|
|
40
|
+
"content": "- `npx @claude-flow/cli memory {store,retrieve,list}` shell-outs — never verified against a real memory backend round-trip\n- `.swarm/router-parallel.jsonl` real writer (`recordPair`/`recordPairOutcome`) never actually invoked — only source-grepped\n- `npx metaharness` cache-install \"stale/partial/corrupted\" middle states — only the two extremes (fully missing / fully available via upstream bypass) are tested\n- OpenRouter real-judge path (`redblue run --patch`, `learn --run`) has no CI-runnable coverage at all\n\n",
|
|
41
|
+
"level": 2
|
|
42
|
+
},
|
|
43
|
+
{
|
|
44
|
+
"title": "Suggested new tests (18 concrete scenarios with `node:test` skeletons)",
|
|
45
|
+
"content": "The agent produced 18 ready-to-adapt skeletons targeting the gaps above — most notably #1 (threat-model severity bug regression test), #3 (mint path-traversal), #4 (drift-from-history flag precedence), and #12 (security-bench markdown parser). Several target functions (`parseSecurityBenchMarkdown`, `parseDurationMs`, `median`/`pctile`, `buildDiagnosis`) would need an `export` added — harmless since these files run as CLI scripts, not imports — to become unit-testable without full subprocess spawns.\n\nWant me to write these test files (starting with the threat-model severity fix + regression test, since that looks like a live bug) or just leave this as the analysis?",
|
|
46
|
+
"level": 2
|
|
47
|
+
}
|
|
48
|
+
],
|
|
49
|
+
"codeBlocks": []
|
|
50
|
+
},
|
|
51
|
+
"durationMs": 375976,
|
|
52
|
+
"model": "sonnet",
|
|
53
|
+
"sandboxMode": "permissive",
|
|
54
|
+
"workerType": "testgaps",
|
|
55
|
+
"timestamp": "2026-07-09T15:53:19.261Z",
|
|
56
|
+
"executionId": "testgaps_1783612023285_buygpo"
|
|
57
|
+
}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
[2026-07-09T16:13:19.302Z] PROMPT
|
|
2
|
+
============================================================
|
|
3
|
+
Analyze test coverage and identify gaps:
|
|
4
|
+
- Find untested functions and classes
|
|
5
|
+
- Identify edge cases not covered
|
|
6
|
+
- Suggest new test scenarios
|
|
7
|
+
- Check for missing error handling tests
|
|
8
|
+
- Identify integration test gaps
|
|
9
|
+
|
|
10
|
+
For each gap, provide a test skeleton.
|
|
11
|
+
|
|
12
|
+
## Instructions
|
|
13
|
+
|
|
14
|
+
Analyze the codebase and provide your response following the format specified in the task.
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
[2026-07-09T16:16:40.844Z] RESULT
|
|
2
|
+
============================================================
|
|
3
|
+
{
|
|
4
|
+
"success": true,
|
|
5
|
+
"output": "## Test Coverage Analysis — ruflo-metaharness Plugin\n\nThe plugin has 32 `.mjs` scripts (~6,900 lines) with no formal test runner — existing `test-*.mjs` files are manual smoke/integration drills, not unit tests.\n\n### 1. Untested functions/classes\n- **`_invoke.mjs`** shared plumbing (`parseTrailingJson`, `satisfiesTildeRange`, `classifyDegraded`, `findLocalPackageDir`, `ensureCachedInstall`) — no direct unit test.\n- **`threat-model.mjs:9`** has a *local* `SEVERITY_RANK` that's diverged from the shared one in `_harness.mjs` (missing `info`/`warn`/`error`/`critical`) — the same drift class fixed elsewhere in iter 63, but not here, and nothing catches it.\n- **`evolve.mjs`**: the entire `--diagnose` feature (`looksLikeGepaTranscript`, `extractGepaTranscripts`, `summarizeTraces`, `buildDiagnosis`, lines 146-274) is untested.\n- `evolve.mjs`/`security-bench.mjs` `safetyChecks()` bound validation, `audit-list.mjs:parseDurationMs()`, `security-bench.mjs:parseSecurityBenchMarkdown()` — all cheap pure-logic, all untested.\n- **`redblue.mjs`, `gepa.mjs`, `learn.mjs`, `mint.mjs`** — zero dedicated test coverage at any level.\n\n### 2. Edge cases not covered\nEmpty/zero-record paths (`drift-from-history` \"no records found\", `audit-list` empty namespace), malformed JSON in `audit-trend.mjs:loadRecord()` and `gepa.mjs:readJsonFile()`, missing-file argv guards, and CLI arg-parsing edge cases (invalid `--sandbox`/`--fail-on`/`--alert-on-worst` enums, `redblue` unknown subcommand, `mint` missing required flags) — none exercised, despite failing fast before any network call. Also: `OPENROUTER_API_KEY`-absent path for `redblue run --patch`, and partial-degradation races (2-of-5 subcalls degraded in `oia-audit.mjs:runAllParallel()`).\n\n### 3. Missing error-handling tests\n`audit-list.mjs` memory-command failures silently returning `[]`, `oia-audit.mjs:persist()` store-failure path, `redblue.mjs` report-JSON-parse failure, `ensureCachedInstall()`'s cache-dir-create-failed branch — all untested failure paths.\n\n### 4. Integration test gaps\nNo test ever goes through the actual `npx ruflo metaharness <sub>` CLI dispatcher or `commands/` slash-command layer — everything invokes `scripts/*.mjs` directly with `node`. `mint`, `redblue`, `gepa`, `learn` have no integration coverage at all, not even the ADR-150 graceful-degradation contract check that the other 8 skills get. `drift-from-history`'s `--baseline-since`/`--baseline-key` fast-paths are also untested.\n\n### 5. Test skeletons (Node's built-in `node:test`, matching the project's zero-dependency style)\n\n```js\n// tests/test-argv-validation.mjs\nimport { test } from 'node:test';\nimport assert from 'node:assert/strict';\nimport { spawnSync } from 'node:child_process';\n\ntest('evolve.mjs rejects out-of-range --generations', () => {\n const r = spawnSync('node', ['scripts/evolve.mjs', '--generations', '51']);\n assert.equal(r.status, 2);\n});\n\ntest('redblue.mjs rejects unknown subcommand', () => {\n const r = spawnSync('node', ['scripts/redblue.mjs', 'bogus']);\n assert.equal(r.status, 2);\n});\n```\n\n```js\n// tests/test-invoke.mjs\nimport { test } from 'node:test';\nimport assert from 'node:assert/strict';\nimport { parseTrailingJson, satisfiesTildeRange } from '../scripts/_invoke.mjs';\n\ntest('parseTrailingJson grabs the LAST JSON block, not a progress line', () => {\n assert.deepEqual(parseTrailingJson('progress {\"x\":1}\\nresult {\"final\":true}'), { final: true });\n});\n```\n\n```js\n// tests/test-threat-model-severity-drift.mjs\nimport { test } from 'node:test';\nimport assert from 'node:assert/strict';\nimport { SEVERITY_RANK as SHARED } from '../scripts/_harness.mjs';\nimport { readFileSync } from 'node:fs';\n\ntest('threat-model.mjs should reuse the shared SEVERITY_RANK, not a local copy', () => {\n const src = readFileSync('scripts/threat-model.mjs', 'utf-8');\n assert.ok(Object.keys(SHARED).every(k => src.includes(k)));\n});\n```\n\nHighest-priority fix: the `threat-model.mjs` `SEVERITY_RANK` drift is a real correctness bug (silently ranks `info`/`warn`/`error`/`critical` as `undefined`), not just a coverage gap — worth fixing alongside adding the test. Want me to add `node:test`-based tests for the argv-validation and `_invoke.mjs` gaps, or fix the severity-rank drift first?\n",
|
|
6
|
+
"parsedOutput": {
|
|
7
|
+
"sections": [
|
|
8
|
+
{
|
|
9
|
+
"title": "Test Coverage Analysis — ruflo-metaharness Plugin",
|
|
10
|
+
"content": "\nThe plugin has 32 `.mjs` scripts (~6,900 lines) with no formal test runner — existing `test-*.mjs` files are manual smoke/integration drills, not unit tests.\n\n",
|
|
11
|
+
"level": 2
|
|
12
|
+
},
|
|
13
|
+
{
|
|
14
|
+
"title": "1. Untested functions/classes",
|
|
15
|
+
"content": "- **`_invoke.mjs`** shared plumbing (`parseTrailingJson`, `satisfiesTildeRange`, `classifyDegraded`, `findLocalPackageDir`, `ensureCachedInstall`) — no direct unit test.\n- **`threat-model.mjs:9`** has a *local* `SEVERITY_RANK` that's diverged from the shared one in `_harness.mjs` (missing `info`/`warn`/`error`/`critical`) — the same drift class fixed elsewhere in iter 63, but not here, and nothing catches it.\n- **`evolve.mjs`**: the entire `--diagnose` feature (`looksLikeGepaTranscript`, `extractGepaTranscripts`, `summarizeTraces`, `buildDiagnosis`, lines 146-274) is untested.\n- `evolve.mjs`/`security-bench.mjs` `safetyChecks()` bound validation, `audit-list.mjs:parseDurationMs()`, `security-bench.mjs:parseSecurityBenchMarkdown()` — all cheap pure-logic, all untested.\n- **`redblue.mjs`, `gepa.mjs`, `learn.mjs`, `mint.mjs`** — zero dedicated test coverage at any level.\n\n",
|
|
16
|
+
"level": 3
|
|
17
|
+
},
|
|
18
|
+
{
|
|
19
|
+
"title": "2. Edge cases not covered",
|
|
20
|
+
"content": "Empty/zero-record paths (`drift-from-history` \"no records found\", `audit-list` empty namespace), malformed JSON in `audit-trend.mjs:loadRecord()` and `gepa.mjs:readJsonFile()`, missing-file argv guards, and CLI arg-parsing edge cases (invalid `--sandbox`/`--fail-on`/`--alert-on-worst` enums, `redblue` unknown subcommand, `mint` missing required flags) — none exercised, despite failing fast before any network call. Also: `OPENROUTER_API_KEY`-absent path for `redblue run --patch`, and partial-degradation races (2-of-5 subcalls degraded in `oia-audit.mjs:runAllParallel()`).\n\n",
|
|
21
|
+
"level": 3
|
|
22
|
+
},
|
|
23
|
+
{
|
|
24
|
+
"title": "3. Missing error-handling tests",
|
|
25
|
+
"content": "`audit-list.mjs` memory-command failures silently returning `[]`, `oia-audit.mjs:persist()` store-failure path, `redblue.mjs` report-JSON-parse failure, `ensureCachedInstall()`'s cache-dir-create-failed branch — all untested failure paths.\n\n",
|
|
26
|
+
"level": 3
|
|
27
|
+
},
|
|
28
|
+
{
|
|
29
|
+
"title": "4. Integration test gaps",
|
|
30
|
+
"content": "No test ever goes through the actual `npx ruflo metaharness <sub>` CLI dispatcher or `commands/` slash-command layer — everything invokes `scripts/*.mjs` directly with `node`. `mint`, `redblue`, `gepa`, `learn` have no integration coverage at all, not even the ADR-150 graceful-degradation contract check that the other 8 skills get. `drift-from-history`'s `--baseline-since`/`--baseline-key` fast-paths are also untested.\n\n",
|
|
31
|
+
"level": 3
|
|
32
|
+
},
|
|
33
|
+
{
|
|
34
|
+
"title": "5. Test skeletons (Node's built-in `node:test`, matching the project's zero-dependency style)",
|
|
35
|
+
"content": "```js\n// tests/test-argv-validation.mjs\nimport { test } from 'node:test';\nimport assert from 'node:assert/strict';\nimport { spawnSync } from 'node:child_process';\n\ntest('evolve.mjs rejects out-of-range --generations', () => {\n const r = spawnSync('node', ['scripts/evolve.mjs', '--generations', '51']);\n assert.equal(r.status, 2);\n});\n\ntest('redblue.mjs rejects unknown subcommand', () => {\n const r = spawnSync('node', ['scripts/redblue.mjs', 'bogus']);\n assert.equal(r.status, 2);\n});\n```\n\n```js\n// tests/test-invoke.mjs\nimport { test } from 'node:test';\nimport assert from 'node:assert/strict';\nimport { parseTrailingJson, satisfiesTildeRange } from '../scripts/_invoke.mjs';\n\ntest('parseTrailingJson grabs the LAST JSON block, not a progress line', () => {\n assert.deepEqual(parseTrailingJson('progress {\"x\":1}\\nresult {\"final\":true}'), { final: true });\n});\n```\n\n```js\n// tests/test-threat-model-severity-drift.mjs\nimport { test } from 'node:test';\nimport assert from 'node:assert/strict';\nimport { SEVERITY_RANK as SHARED } from '../scripts/_harness.mjs';\nimport { readFileSync } from 'node:fs';\n\ntest('threat-model.mjs should reuse the shared SEVERITY_RANK, not a local copy', () => {\n const src = readFileSync('scripts/threat-model.mjs', 'utf-8');\n assert.ok(Object.keys(SHARED).every(k => src.includes(k)));\n});\n```\n\nHighest-priority fix: the `threat-model.mjs` `SEVERITY_RANK` drift is a real correctness bug (silently ranks `info`/`warn`/`error`/`critical` as `undefined`), not just a coverage gap — worth fixing alongside adding the test. Want me to add `node:test`-based tests for the argv-validation and `_invoke.mjs` gaps, or fix the severity-rank drift first?",
|
|
36
|
+
"level": 3
|
|
37
|
+
}
|
|
38
|
+
],
|
|
39
|
+
"codeBlocks": [
|
|
40
|
+
{
|
|
41
|
+
"language": "js",
|
|
42
|
+
"code": "// tests/test-argv-validation.mjs\nimport { test } from 'node:test';\nimport assert from 'node:assert/strict';\nimport { spawnSync } from 'node:child_process';\n\ntest('evolve.mjs rejects out-of-range --generations', () => {\n const r = spawnSync('node', ['scripts/evolve.mjs', '--generations', '51']);\n assert.equal(r.status, 2);\n});\n\ntest('redblue.mjs rejects unknown subcommand', () => {\n const r = spawnSync('node', ['scripts/redblue.mjs', 'bogus']);\n assert.equal(r.status, 2);\n});"
|
|
43
|
+
},
|
|
44
|
+
{
|
|
45
|
+
"language": "js",
|
|
46
|
+
"code": "// tests/test-invoke.mjs\nimport { test } from 'node:test';\nimport assert from 'node:assert/strict';\nimport { parseTrailingJson, satisfiesTildeRange } from '../scripts/_invoke.mjs';\n\ntest('parseTrailingJson grabs the LAST JSON block, not a progress line', () => {\n assert.deepEqual(parseTrailingJson('progress {\"x\":1}\\nresult {\"final\":true}'), { final: true });\n});"
|
|
47
|
+
},
|
|
48
|
+
{
|
|
49
|
+
"language": "js",
|
|
50
|
+
"code": "// tests/test-threat-model-severity-drift.mjs\nimport { test } from 'node:test';\nimport assert from 'node:assert/strict';\nimport { SEVERITY_RANK as SHARED } from '../scripts/_harness.mjs';\nimport { readFileSync } from 'node:fs';\n\ntest('threat-model.mjs should reuse the shared SEVERITY_RANK, not a local copy', () => {\n const src = readFileSync('scripts/threat-model.mjs', 'utf-8');\n assert.ok(Object.keys(SHARED).every(k => src.includes(k)));\n});"
|
|
51
|
+
}
|
|
52
|
+
]
|
|
53
|
+
},
|
|
54
|
+
"durationMs": 201543,
|
|
55
|
+
"model": "sonnet",
|
|
56
|
+
"sandboxMode": "permissive",
|
|
57
|
+
"workerType": "testgaps",
|
|
58
|
+
"timestamp": "2026-07-09T16:16:40.844Z",
|
|
59
|
+
"executionId": "testgaps_1783613599301_6f78cw"
|
|
60
|
+
}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
[2026-07-09T16:36:40.845Z] PROMPT
|
|
2
|
+
============================================================
|
|
3
|
+
Analyze test coverage and identify gaps:
|
|
4
|
+
- Find untested functions and classes
|
|
5
|
+
- Identify edge cases not covered
|
|
6
|
+
- Suggest new test scenarios
|
|
7
|
+
- Check for missing error handling tests
|
|
8
|
+
- Identify integration test gaps
|
|
9
|
+
|
|
10
|
+
For each gap, provide a test skeleton.
|
|
11
|
+
|
|
12
|
+
## Instructions
|
|
13
|
+
|
|
14
|
+
Analyze the codebase and provide your response following the format specified in the task.
|