@olegkoval/agent-skills 1.40.0 → 1.41.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/prompts/open-source-publisher.prompt.md +3 -4
- package/.kiro/steering/open-source-publisher.md +3 -4
- package/.windsurf/rules/open-source-publisher.md +3 -4
- package/adapters/claude/olko-github-pr/skills/lekker-review/SKILL.md +15 -17
- package/adapters/claude/olko-github-pr/skills/lekker-review/references/agents/implementation.md +2 -1
- package/adapters/claude/olko-github-pr/skills/lekker-review/references/agents/prover.md +13 -10
- package/adapters/claude/olko-github-pr/skills/lekker-review/references/agents/quality.md +1 -1
- package/adapters/claude/olko-github-pr/skills/lekker-review/references/agents/test-quality.md +7 -0
- package/adapters/claude/olko-github-pr/skills/lekker-review/references/agents/triage-quality.md +1 -1
- package/adapters/claude/olko-github-pr/skills/lekker-review/references/agents/verifier.md +12 -9
- package/adapters/claude/olko-github-pr/skills/lekker-review/references/artifact-page.md +6 -1
- package/adapters/claude/olko-github-pr/skills/lekker-review/references/output-format.md +5 -8
- package/adapters/claude/olko-github-pr/skills/lekker-review/scripts/selftest.mjs +174 -0
- package/adapters/claude/olko-reflection/skills/self-critique/scripts/critique-nudge.mjs +3 -0
- package/adapters/claude/olko-release/skills/open-source-publisher/SKILL.md +4 -5
- package/adapters/cursor/olko-reflection/skills/self-critique/scripts/critique-nudge.mjs +3 -0
- package/adapters/cursor/olko-release/skills/open-source-publisher/SKILL.md +4 -5
- package/adapters/grok/olko-reflection/skills/self-critique/scripts/critique-nudge.mjs +3 -0
- package/adapters/grok/olko-release/skills/open-source-publisher/SKILL.md +4 -5
- package/package.json +1 -1
- package/plugins/olko-apple-kit/.claude-plugin/plugin.json +1 -1
- package/plugins/olko-creative/.claude-plugin/plugin.json +1 -1
- package/plugins/olko-garmin-kit/.claude-plugin/plugin.json +1 -1
- package/plugins/olko-git-tools/.claude-plugin/plugin.json +1 -1
- package/plugins/olko-github-pr/.claude-plugin/plugin.json +1 -1
- package/plugins/olko-github-pr/skills/lekker-review/README.md +9 -9
- package/plugins/olko-github-pr/skills/lekker-review/SKILL.md +15 -17
- package/plugins/olko-github-pr/skills/lekker-review/references/agents/implementation.md +2 -1
- package/plugins/olko-github-pr/skills/lekker-review/references/agents/prover.md +13 -10
- package/plugins/olko-github-pr/skills/lekker-review/references/agents/quality.md +1 -1
- package/plugins/olko-github-pr/skills/lekker-review/references/agents/test-quality.md +7 -0
- package/plugins/olko-github-pr/skills/lekker-review/references/agents/triage-quality.md +1 -1
- package/plugins/olko-github-pr/skills/lekker-review/references/agents/verifier.md +12 -9
- package/plugins/olko-github-pr/skills/lekker-review/references/artifact-page.md +6 -1
- package/plugins/olko-github-pr/skills/lekker-review/references/output-format.md +5 -8
- package/plugins/olko-github-pr/skills/lekker-review/scripts/selftest.mjs +174 -0
- package/plugins/olko-github-pr/skills/lekker-review/workflow.js +217 -75
- package/plugins/olko-obsidian/.claude-plugin/plugin.json +1 -1
- package/plugins/olko-product/.claude-plugin/plugin.json +1 -1
- package/plugins/olko-reflection/.claude-plugin/plugin.json +1 -1
- package/plugins/olko-reflection/skills/self-critique/scripts/critique-nudge.mjs +3 -0
- package/plugins/olko-release/.claude-plugin/plugin.json +1 -1
- package/plugins/olko-release/skills/open-source-publisher/SKILL.md +4 -5
- package/plugins/olko-skill-meta/.claude-plugin/plugin.json +1 -1
- package/plugins/olko-web-ops/.claude-plugin/plugin.json +1 -1
- package/scripts/lib/catalog.mjs +3 -0
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// Zero-agent regression test for lekker-review's PURE logic.
|
|
3
|
+
//
|
|
4
|
+
// Why this exists: every defect found in the 2026-08-31 hardening pass was in
|
|
5
|
+
// pure, synchronous code - the dedup bucket key, the hard-rule exemption gate,
|
|
6
|
+
// the model/effort routing - yet the only way to exercise any of it was a live
|
|
7
|
+
// workflow run costing ~7 agents and 70+ seconds. This runs the same logic in
|
|
8
|
+
// milliseconds with no agents at all. Run it after ANY edit to workflow.js:
|
|
9
|
+
//
|
|
10
|
+
// node ~/.claude/skills/lekker-review/scripts/selftest.mjs
|
|
11
|
+
//
|
|
12
|
+
// It lifts the real functions out of workflow.js by source extraction rather
|
|
13
|
+
// than importing, because workflow.js is written for the Workflow harness (top
|
|
14
|
+
// level `return`, an injected `args` global) and is not importable as a module.
|
|
15
|
+
import { readFileSync } from 'node:fs'
|
|
16
|
+
import { fileURLToPath } from 'node:url'
|
|
17
|
+
import { dirname, join } from 'node:path'
|
|
18
|
+
|
|
19
|
+
const SKILL = dirname(dirname(fileURLToPath(import.meta.url)))
|
|
20
|
+
// Optional arg: a different workflow.js to test. Used to prove this suite
|
|
21
|
+
// actually discriminates - point it at a pre-fix backup and it must FAIL.
|
|
22
|
+
const target = process.argv[2] || join(SKILL, 'workflow.js')
|
|
23
|
+
const src = readFileSync(target, 'utf8')
|
|
24
|
+
console.log(`selftest target: ${target}`)
|
|
25
|
+
|
|
26
|
+
function lift(name) {
|
|
27
|
+
const i = src.indexOf(`function ${name}`)
|
|
28
|
+
if (i === -1) throw new Error(`selftest: function ${name} not found in workflow.js - was it renamed?`)
|
|
29
|
+
let d = 0, j = i
|
|
30
|
+
for (;; j++) {
|
|
31
|
+
if (src[j] === '{') d++
|
|
32
|
+
else if (src[j] === '}') { d--; if (d === 0) break }
|
|
33
|
+
}
|
|
34
|
+
return src.slice(i, j + 1)
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
function liftConst(name) {
|
|
38
|
+
const m = new RegExp(`^const ${name} = .*$`, 'm').exec(src)
|
|
39
|
+
if (!m) throw new Error(`selftest: const ${name} not found in workflow.js`)
|
|
40
|
+
return m[0]
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
const preamble = [
|
|
44
|
+
liftConst('HARD_RULES'),
|
|
45
|
+
(() => { try { return liftConst('SAME_ISSUE_LINE_WINDOW') } catch { return 'const SAME_ISSUE_LINE_WINDOW = 30' } })(),
|
|
46
|
+
"const SEVERITY_RANK = { observation: 0, idiomatic: 1, important: 2, critical: 3 }",
|
|
47
|
+
...['titleTokens', 'sameIssue', 'nearbyLines', 'spanWithinWindow', 'hardRuleCorroborated',
|
|
48
|
+
'isHardRule', 'longest', 'mergeFindings', 'dedup', 'shouldVerify'].map(n => {
|
|
49
|
+
try { return lift(n) } catch { return `function ${n}() { throw new Error('${n} absent from this workflow.js') }` }
|
|
50
|
+
}),
|
|
51
|
+
].join('\n')
|
|
52
|
+
|
|
53
|
+
const { dedup, isHardRule, shouldVerify, sameIssue } =
|
|
54
|
+
new Function(preamble + '\nreturn { dedup, isHardRule, shouldVerify, sameIssue }')()
|
|
55
|
+
|
|
56
|
+
let failed = 0
|
|
57
|
+
function check(name, actual, expected) {
|
|
58
|
+
const a = JSON.stringify(actual), e = JSON.stringify(expected)
|
|
59
|
+
if (a === e) { console.log(` ok ${name}`) }
|
|
60
|
+
else { console.log(` FAIL ${name}\n expected ${e}\n actual ${a}`); failed++ }
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
console.log('\ndedup: the same defect anchored at different lines must merge')
|
|
64
|
+
// Regression: bucketing on `file:line` meant these two were never compared,
|
|
65
|
+
// despite a title similarity of 0.64 against a 0.4 threshold. Observed live.
|
|
66
|
+
const dupes = [
|
|
67
|
+
{ file: 'src/total.ts', line: 14, severity: 'critical', title: 'Off-by-one loop skips the first cart line', badCode: 'for (let i = 1;', description: 'aaa' },
|
|
68
|
+
{ file: 'src/total.ts', line: 9, severity: 'critical', title: 'cartTotal skips the first line item (off-by-one loop start)', badCode: 'for (let i = 1;', description: 'bb' },
|
|
69
|
+
]
|
|
70
|
+
check('two anchors, one issue -> 1 finding', dedup(dupes).length, 1)
|
|
71
|
+
check('merge keeps the highest severity', dedup([
|
|
72
|
+
{ file: 'a.ts', line: 3, severity: 'observation', title: 'Off-by-one loop skips first line', badCode: '', description: '' },
|
|
73
|
+
{ file: 'a.ts', line: 5, severity: 'critical', title: 'Off-by-one loop skips the first line', badCode: '', description: '' },
|
|
74
|
+
])[0].severity, 'critical')
|
|
75
|
+
|
|
76
|
+
console.log('\ndedup: distinct issues must NOT be merged')
|
|
77
|
+
check('similar titles 390 lines apart stay separate', dedup([
|
|
78
|
+
{ file: 'big.ts', line: 10, severity: 'important', title: 'Missing pagination on the products query', badCode: '', description: '' },
|
|
79
|
+
{ file: 'big.ts', line: 400, severity: 'important', title: 'Missing pagination on the orders query', badCode: '', description: '' },
|
|
80
|
+
]).length, 2)
|
|
81
|
+
check('same line, unrelated titles stay separate', dedup([
|
|
82
|
+
{ file: 'a.ts', line: 7, severity: 'important', title: 'Unbounded retry loop hides throttling', badCode: '', description: '' },
|
|
83
|
+
{ file: 'a.ts', line: 7, severity: 'important', title: 'Metafield namespace hardcoded in the query', badCode: '', description: '' },
|
|
84
|
+
]).length, 2)
|
|
85
|
+
// Grouping must not depend on arrival order. With only g[0] compared, findings
|
|
86
|
+
// at 25, 50 and 1 all joined when 25 arrived first, spanning 49 lines.
|
|
87
|
+
const spanCase = [
|
|
88
|
+
{ file: 'a.ts', line: 25, severity: 'important', title: 'Missing pagination on the query', badCode: '', description: '' },
|
|
89
|
+
{ file: 'a.ts', line: 50, severity: 'important', title: 'Missing pagination on the query', badCode: '', description: '' },
|
|
90
|
+
{ file: 'a.ts', line: 1, severity: 'important', title: 'Missing pagination on the query', badCode: '', description: '' },
|
|
91
|
+
]
|
|
92
|
+
check('a group never spans more than the window (25, 50, 1)', dedup(spanCase).length, 2)
|
|
93
|
+
check('the same set in a different order gives the same answer',
|
|
94
|
+
dedup([spanCase[2], spanCase[0], spanCase[1]]).length, dedup(spanCase).length)
|
|
95
|
+
|
|
96
|
+
check('different files never merge', dedup([
|
|
97
|
+
{ file: 'a.ts', line: 7, severity: 'critical', title: 'Off-by-one loop skips the first line', badCode: '', description: '' },
|
|
98
|
+
{ file: 'b.ts', line: 7, severity: 'critical', title: 'Off-by-one loop skips the first line', badCode: '', description: '' },
|
|
99
|
+
]).length, 2)
|
|
100
|
+
|
|
101
|
+
console.log('\nhard rules: a tag must be corroborated to skip verification')
|
|
102
|
+
// Regression: any agent could bypass the verifier by writing rule: "TS-1".
|
|
103
|
+
// Observed live - a test-coverage finding and a comment-policy finding both did.
|
|
104
|
+
check('TS-1 on a comment-policy finding is NOT exempt',
|
|
105
|
+
isHardRule({ rule: 'TS-1', file: 'a.ts', line: 6, title: 'Comment restates the function', badCode: '/** Sum a cart. */', description: 'a comment earns its place' }), false)
|
|
106
|
+
check('TS-1 on a test-coverage finding is NOT exempt',
|
|
107
|
+
isHardRule({ rule: 'TS-1', file: 'a.ts', line: 8, title: 'No test coverage', badCode: 'for (let i = 1;', description: 'zero test files added' }), false)
|
|
108
|
+
check('an unknown rule string is NOT exempt',
|
|
109
|
+
isHardRule({ rule: 'MADE-UP', file: 'a.ts', line: 1, title: 't', badCode: 'x as Foo', description: '' }), false)
|
|
110
|
+
// TS-1 is judged on quoted code only: prose is full of `as` and `any`.
|
|
111
|
+
check('the word "any" in PROSE alone is NOT exempt',
|
|
112
|
+
isHardRule({ rule: 'TS-1', file: 'a.ts', line: 1, title: 'fails on any cart with items', badCode: 'total += lines[i].price', description: 'any agent could trip this' }), false)
|
|
113
|
+
check('the phrase "such as" in prose alone is NOT exempt',
|
|
114
|
+
isHardRule({ rule: 'TS-1', file: 'a.ts', line: 1, title: 'issue', badCode: 'const n = 1', description: 'a primitive such as String is used' }), false)
|
|
115
|
+
|
|
116
|
+
console.log('\nhard rules: genuine violations must STILL be exempt')
|
|
117
|
+
check('TS-1 with a real cast', isHardRule({ rule: 'TS-1', file: 'a.ts', line: 1, title: 'cast', badCode: 'const x = y as Foo;', description: '' }), true)
|
|
118
|
+
check('TS-1 with a real any', isHardRule({ rule: 'TS-1', file: 'a.ts', line: 1, title: 'any', badCode: 'function f(x: any) {}', description: '' }), true)
|
|
119
|
+
check('TS-2 with a .js path', isHardRule({ rule: 'TS-2', file: 'web/thing.js', line: 1, title: 'js added', badCode: '', description: '' }), true)
|
|
120
|
+
check('GQL-1 with a nodes query',isHardRule({ rule: 'GQL-1', file: 'q.graphql', line: 1, title: 'no pageInfo', badCode: 'products { nodes { id } }', description: '' }), true)
|
|
121
|
+
check('PR-1 anchored on the PR title', isHardRule({ rule: 'PR-1', file: 'PR title', line: 1, title: 'missing prefix', badCode: '', description: '' }), true)
|
|
122
|
+
// A cast to a lowercase built-in is as much a TS-1 violation as a cast to a
|
|
123
|
+
// named type. Missing it sent a genuine hard rule to a verifier that cannot
|
|
124
|
+
// answer a policy claim, where it could be dropped.
|
|
125
|
+
for (const cast of ['x as string', 'x as number', 'x as unknown as Foo', 'x as const', 'x as boolean']) {
|
|
126
|
+
check(`TS-1 corroborated by \`${cast}\``,
|
|
127
|
+
isHardRule({ rule: 'TS-1', file: 'a.ts', line: 1, title: 'cast', badCode: cast, description: '' }), true)
|
|
128
|
+
}
|
|
129
|
+
check('TS-1 corroborated by an any annotation',
|
|
130
|
+
isHardRule({ rule: 'TS-1', file: 'a.ts', line: 1, title: 'any', badCode: 'function f(x: any) {}', description: '' }), true)
|
|
131
|
+
check('TS-1 corroborated by an any[] ',
|
|
132
|
+
isHardRule({ rule: 'TS-1', file: 'a.ts', line: 1, title: 'any', badCode: 'const xs: any[] = []', description: '' }), true)
|
|
133
|
+
|
|
134
|
+
console.log('\nverification scope by depth')
|
|
135
|
+
const crit = { severity: 'critical' }, imp = { severity: 'important' }, obs = { severity: 'observation' }
|
|
136
|
+
check('scan verifies nothing', [crit, imp, obs].map(f => shouldVerify(f, 'scan')), [false, false, false])
|
|
137
|
+
check('medium verifies criticals only', [crit, imp, obs].map(f => shouldVerify(f, 'medium')), [true, false, false])
|
|
138
|
+
check('deep verifies crit + important', [crit, imp, obs].map(f => shouldVerify(f, 'deep')), [true, true, false])
|
|
139
|
+
check('a corroborated hard rule is never verified',
|
|
140
|
+
shouldVerify({ severity: 'critical', rule: 'TS-2', file: 'x.js', badCode: '', description: '', title: '' }, 'deep'), false)
|
|
141
|
+
|
|
142
|
+
// Per-host model routing is optional: a deployment may pin models per host via
|
|
143
|
+
// a MODEL_TABLE, or leave every agent() call to name its own model. Test it
|
|
144
|
+
// only when it is present, so this suite runs against either shape.
|
|
145
|
+
const routingAssign = /const\s+MODEL\s*=\s*MODEL_TABLE\s*\[\s*HOST\s*\]/.exec(src)
|
|
146
|
+
const hasRouting = Boolean(routingAssign)
|
|
147
|
+
// A guard that can silently disable itself is worse than no guard. If the file
|
|
148
|
+
// clearly HAS a MODEL_TABLE but the assignment did not parse, that is a failure,
|
|
149
|
+
// not a reason to skip.
|
|
150
|
+
if (!hasRouting && /MODEL_TABLE/.test(src)) {
|
|
151
|
+
check('MODEL_TABLE is present but its assignment was not recognised', false, true)
|
|
152
|
+
}
|
|
153
|
+
if (!hasRouting) {
|
|
154
|
+
console.log('\nmodel + effort routing: not configured in this workflow.js, skipped')
|
|
155
|
+
} else {
|
|
156
|
+
console.log('\nmodel + effort routing is pinned per host, never inherited')
|
|
157
|
+
const routing = new Function('input', [
|
|
158
|
+
src.slice(src.indexOf('const HOST = '), routingAssign.index + routingAssign[0].length),
|
|
159
|
+
'return { HOST, MODEL }',
|
|
160
|
+
].join('\n'))
|
|
161
|
+
check('an absent host arg falls back to the default table', routing({}).MODEL, routing({ host: 'claude' }).MODEL)
|
|
162
|
+
check('an unknown host falls back to the default, not an invalid model', routing({ host: 'nonsense' }).HOST, 'claude')
|
|
163
|
+
check('host matching is case-insensitive', routing({ host: 'CODEX' }).HOST, 'codex')
|
|
164
|
+
// Every role must resolve to a non-empty model, and effort must be set.
|
|
165
|
+
for (const host of ['claude', 'codex']) {
|
|
166
|
+
const m = routing({ host }).MODEL
|
|
167
|
+
const roles = Object.keys(m).filter(k => k !== 'effort')
|
|
168
|
+
check(`${host}: every role resolves to a model`, roles.every(r => typeof m[r] === 'string' && m[r].length > 0), true)
|
|
169
|
+
check(`${host}: effort is set`, typeof m.effort === 'string' && m.effort.length > 0, true)
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
console.log(failed === 0 ? '\nall checks passed\n' : `\n${failed} check(s) FAILED\n`)
|
|
174
|
+
process.exit(failed === 0 ? 0 : 1)
|
|
@@ -8,43 +8,68 @@ export const meta = {
|
|
|
8
8
|
// Schemas
|
|
9
9
|
// ---------------------------------------------------------------------------
|
|
10
10
|
|
|
11
|
+
const FINDINGS_ARRAY_SCHEMA = {
|
|
12
|
+
type: 'array',
|
|
13
|
+
items: {
|
|
14
|
+
type: 'object',
|
|
15
|
+
required: ['file', 'line', 'severity', 'title', 'description', 'badCode', 'fix'],
|
|
16
|
+
properties: {
|
|
17
|
+
file: { type: 'string' },
|
|
18
|
+
line: { type: 'integer' },
|
|
19
|
+
severity: { enum: ['critical', 'important', 'observation', 'idiomatic'] },
|
|
20
|
+
title: { type: 'string' },
|
|
21
|
+
description: { type: 'string' },
|
|
22
|
+
badCode: { type: 'string' },
|
|
23
|
+
fix: { type: 'string' },
|
|
24
|
+
precedent: { type: 'string' },
|
|
25
|
+
rule: { type: 'string', minLength: 1, pattern: '\\S' },
|
|
26
|
+
},
|
|
27
|
+
},
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
const MOCK_SMELLS_SCHEMA = {
|
|
31
|
+
type: 'array',
|
|
32
|
+
items: {
|
|
33
|
+
type: 'object',
|
|
34
|
+
required: ['file', 'line', 'description', 'fix'],
|
|
35
|
+
properties: {
|
|
36
|
+
file: { type: 'string' },
|
|
37
|
+
line: { type: 'integer' },
|
|
38
|
+
description: { type: 'string' },
|
|
39
|
+
fix: { type: 'string' },
|
|
40
|
+
},
|
|
41
|
+
},
|
|
42
|
+
}
|
|
43
|
+
|
|
11
44
|
const FINDINGS_SCHEMA = {
|
|
12
45
|
type: 'object',
|
|
13
46
|
required: ['findings'],
|
|
14
47
|
properties: {
|
|
15
|
-
findings:
|
|
16
|
-
type: 'array',
|
|
17
|
-
items: {
|
|
18
|
-
type: 'object',
|
|
19
|
-
required: ['file', 'line', 'severity', 'title', 'description', 'badCode', 'fix'],
|
|
20
|
-
properties: {
|
|
21
|
-
file: { type: 'string' },
|
|
22
|
-
line: { type: 'integer' },
|
|
23
|
-
severity: { enum: ['critical', 'important', 'observation', 'idiomatic'] },
|
|
24
|
-
title: { type: 'string' },
|
|
25
|
-
description: { type: 'string' },
|
|
26
|
-
badCode: { type: 'string' },
|
|
27
|
-
fix: { type: 'string' },
|
|
28
|
-
precedent: { type: 'string' },
|
|
29
|
-
rule: { enum: ['TS-1', 'TS-2', 'GQL-1', 'PR-1'] },
|
|
30
|
-
},
|
|
31
|
-
},
|
|
32
|
-
},
|
|
48
|
+
findings: FINDINGS_ARRAY_SCHEMA,
|
|
33
49
|
acCoverage: { type: 'string' },
|
|
34
50
|
mutationSlip: { type: 'string' },
|
|
35
51
|
coverageVerdict: { type: 'string' },
|
|
36
|
-
mockSmells:
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
52
|
+
mockSmells: MOCK_SMELLS_SCHEMA,
|
|
53
|
+
},
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
const IMPLEMENTATION_SCHEMA = {
|
|
57
|
+
type: 'object',
|
|
58
|
+
required: ['findings', 'acCoverage'],
|
|
59
|
+
properties: {
|
|
60
|
+
findings: FINDINGS_ARRAY_SCHEMA,
|
|
61
|
+
acCoverage: { type: 'string' },
|
|
62
|
+
},
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
const TEST_QUALITY_SCHEMA = {
|
|
66
|
+
type: 'object',
|
|
67
|
+
required: ['findings', 'coverageVerdict', 'mutationSlip', 'mockSmells'],
|
|
68
|
+
properties: {
|
|
69
|
+
findings: FINDINGS_ARRAY_SCHEMA,
|
|
70
|
+
coverageVerdict: { type: 'string' },
|
|
71
|
+
mutationSlip: { type: 'string' },
|
|
72
|
+
mockSmells: MOCK_SMELLS_SCHEMA,
|
|
48
73
|
},
|
|
49
74
|
}
|
|
50
75
|
|
|
@@ -80,10 +105,11 @@ const CRITIC_SCHEMA = {
|
|
|
80
105
|
|
|
81
106
|
const PROOF_SCHEMA = {
|
|
82
107
|
type: 'object',
|
|
83
|
-
required: ['attempted', 'proven', 'reason'],
|
|
108
|
+
required: ['attempted', 'proven', 'outcome', 'reason'],
|
|
84
109
|
properties: {
|
|
85
110
|
attempted: { type: 'boolean' },
|
|
86
111
|
proven: { type: 'boolean' },
|
|
112
|
+
outcome: { enum: ['proven', 'passed', 'inconclusive', 'not_attempted'] },
|
|
87
113
|
reason: { type: 'string' },
|
|
88
114
|
testCode: { type: 'string' },
|
|
89
115
|
testCommand: { type: 'string' },
|
|
@@ -95,10 +121,8 @@ const PROOF_SCHEMA = {
|
|
|
95
121
|
// Helpers
|
|
96
122
|
// ---------------------------------------------------------------------------
|
|
97
123
|
|
|
98
|
-
const HARD_RULES = ['TS-1', 'TS-2', 'GQL-1', 'PR-1']
|
|
99
|
-
|
|
100
124
|
function isHardRule(finding) {
|
|
101
|
-
return typeof finding.rule === 'string' &&
|
|
125
|
+
return typeof finding.rule === 'string' && finding.rule.trim().length > 0
|
|
102
126
|
}
|
|
103
127
|
|
|
104
128
|
const SEVERITY_RANK = { critical: 3, important: 2, observation: 1, idiomatic: 0 }
|
|
@@ -136,6 +160,13 @@ function mergeFindings(a, b) {
|
|
|
136
160
|
|
|
137
161
|
if (SEVERITY_RANK[b.severity] > SEVERITY_RANK[merged.severity]) {
|
|
138
162
|
merged.severity = b.severity
|
|
163
|
+
for (const field of ['verificationStatus', 'verifierReasoning', 'proof']) {
|
|
164
|
+
if (Object.prototype.hasOwnProperty.call(b, field)) {
|
|
165
|
+
merged[field] = b[field]
|
|
166
|
+
} else {
|
|
167
|
+
delete merged[field]
|
|
168
|
+
}
|
|
169
|
+
}
|
|
139
170
|
}
|
|
140
171
|
|
|
141
172
|
merged.description = longest(a.description, b.description)
|
|
@@ -162,10 +193,43 @@ function mergeFindings(a, b) {
|
|
|
162
193
|
return merged
|
|
163
194
|
}
|
|
164
195
|
|
|
196
|
+
// How far apart two anchor lines may be and still count as the same issue.
|
|
197
|
+
// Reviewers routinely anchor one defect at different lines: the loop header,
|
|
198
|
+
// the body, the function signature. Kept tight enough that two genuinely
|
|
199
|
+
// distinct findings in one file are not merged because their titles rhyme.
|
|
200
|
+
const SAME_ISSUE_LINE_WINDOW = 30
|
|
201
|
+
|
|
202
|
+
function nearbyLines(a, b) {
|
|
203
|
+
const la = Number(a.line)
|
|
204
|
+
const lb = Number(b.line)
|
|
205
|
+
if (!Number.isFinite(la) || !Number.isFinite(lb)) {
|
|
206
|
+
// A finding with no usable line (a PR-level note) only merges with another
|
|
207
|
+
// one at the same missing line, which is what strict equality gives.
|
|
208
|
+
return a.line === b.line
|
|
209
|
+
}
|
|
210
|
+
return Math.abs(la - lb) <= SAME_ISSUE_LINE_WINDOW
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
// Would adding `candidate` keep the whole group inside the window? Uses the
|
|
214
|
+
// group's min and max so the answer never depends on arrival order.
|
|
215
|
+
function spanWithinWindow(group, candidate) {
|
|
216
|
+
const lines = group.concat([candidate]).map(function(f) { return Number(f.line) })
|
|
217
|
+
if (!lines.every(function(n) { return Number.isFinite(n) })) {
|
|
218
|
+
// Any unusable line falls back to the strict pairwise rule.
|
|
219
|
+
return group.every(function(m) { return nearbyLines(m, candidate) })
|
|
220
|
+
}
|
|
221
|
+
return Math.max.apply(null, lines) - Math.min.apply(null, lines) <= SAME_ISSUE_LINE_WINDOW
|
|
222
|
+
}
|
|
223
|
+
|
|
165
224
|
function dedup(allFindings) {
|
|
225
|
+
// Bucket by FILE, not by `file:line`. Bucketing on the exact line meant two
|
|
226
|
+
// reviewers describing one defect at lines 9 and 14 landed in different
|
|
227
|
+
// buckets, so `sameIssue` was never consulted: the issue was verified twice,
|
|
228
|
+
// burning two verifier agents and emitting two inline comments for one
|
|
229
|
+
// problem, which is exactly what this pass exists to prevent.
|
|
166
230
|
const byLocation = new Map()
|
|
167
231
|
for (const f of allFindings) {
|
|
168
|
-
const key =
|
|
232
|
+
const key = String(f.file)
|
|
169
233
|
if (!byLocation.has(key)) {
|
|
170
234
|
byLocation.set(key, [])
|
|
171
235
|
}
|
|
@@ -176,7 +240,13 @@ function dedup(allFindings) {
|
|
|
176
240
|
for (const candidates of byLocation.values()) {
|
|
177
241
|
const groups = []
|
|
178
242
|
for (const candidate of candidates) {
|
|
179
|
-
|
|
243
|
+
// Compare against the whole group's SPAN, not just its first member.
|
|
244
|
+
// Checking only g[0] made grouping order-dependent: findings at 25, 50
|
|
245
|
+
// and 1 all joined when 25 arrived first, leaving a group spanning 49
|
|
246
|
+
// lines despite a 30-line window.
|
|
247
|
+
const match = groups.find(function(g) {
|
|
248
|
+
return sameIssue(g[0], candidate) && spanWithinWindow(g, candidate)
|
|
249
|
+
})
|
|
180
250
|
if (match) {
|
|
181
251
|
match.push(candidate)
|
|
182
252
|
} else {
|
|
@@ -190,21 +260,51 @@ function dedup(allFindings) {
|
|
|
190
260
|
return result
|
|
191
261
|
}
|
|
192
262
|
|
|
193
|
-
function shouldVerify(finding
|
|
194
|
-
//
|
|
195
|
-
//
|
|
196
|
-
//
|
|
197
|
-
|
|
198
|
-
|
|
263
|
+
function shouldVerify(finding) {
|
|
264
|
+
// Review depth controls breadth, not the trust bar. Every finding that can
|
|
265
|
+
// affect the verdict must be checked. Hard rules take the verifier's
|
|
266
|
+
// rule-specific anchor/applicability path instead of its runtime challenges.
|
|
267
|
+
return finding.severity === 'critical' || finding.severity === 'important'
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
function normalizeProof(result) {
|
|
271
|
+
const hasText = function(value) {
|
|
272
|
+
return typeof value === 'string' && value.trim().length > 0
|
|
199
273
|
}
|
|
200
|
-
|
|
201
|
-
|
|
274
|
+
const coherent = (
|
|
275
|
+
result.attempted === true
|
|
276
|
+
&& result.proven === true
|
|
277
|
+
&& result.outcome === 'proven'
|
|
278
|
+
&& hasText(result.testCode)
|
|
279
|
+
&& hasText(result.testCommand)
|
|
280
|
+
&& hasText(result.redOutput)
|
|
281
|
+
) || (
|
|
282
|
+
result.attempted === true
|
|
283
|
+
&& result.proven === false
|
|
284
|
+
&& result.outcome === 'passed'
|
|
285
|
+
&& hasText(result.testCode)
|
|
286
|
+
&& hasText(result.testCommand)
|
|
287
|
+
) || (
|
|
288
|
+
result.attempted === true
|
|
289
|
+
&& result.proven === false
|
|
290
|
+
&& result.outcome === 'inconclusive'
|
|
291
|
+
) || (
|
|
292
|
+
result.attempted === false
|
|
293
|
+
&& result.proven === false
|
|
294
|
+
&& result.outcome === 'not_attempted'
|
|
295
|
+
)
|
|
296
|
+
|
|
297
|
+
if (coherent) {
|
|
298
|
+
return result
|
|
202
299
|
}
|
|
203
|
-
|
|
204
|
-
|
|
300
|
+
|
|
301
|
+
const attempted = result.attempted === true
|
|
302
|
+
return {
|
|
303
|
+
attempted,
|
|
304
|
+
proven: false,
|
|
305
|
+
outcome: attempted ? 'inconclusive' : 'not_attempted',
|
|
306
|
+
reason: `Prover returned an inconsistent proof state; ignored. ${result.reason || ''}`.trim(),
|
|
205
307
|
}
|
|
206
|
-
// deep: critical + important
|
|
207
|
-
return finding.severity === 'critical' || finding.severity === 'important'
|
|
208
308
|
}
|
|
209
309
|
|
|
210
310
|
// Run thunks in sequential batches, giving the per-finding stages a ceiling.
|
|
@@ -345,6 +445,7 @@ const budgetAtStart = budget.spent()
|
|
|
345
445
|
`Read and follow ${promptDir}/verifier.md.`,
|
|
346
446
|
`FINDING (JSON): ${JSON.stringify(finding)}.`,
|
|
347
447
|
`DIFF_FILE=${diffFile}, CONTEXT_FILE=${contextFile}, WORKTREE_PATH=${wtDisplay}.`,
|
|
448
|
+
`HOUSE_RULES_FILE: read key "houseRulesFile" from CONTEXT_FILE, then read that exact path before validating a hard rule.`,
|
|
348
449
|
`Default to dropping when uncertain.`,
|
|
349
450
|
].join(' ')
|
|
350
451
|
}
|
|
@@ -355,16 +456,22 @@ const budgetAtStart = budget.spent()
|
|
|
355
456
|
|
|
356
457
|
async function reviewStage(dim) {
|
|
357
458
|
agentCount++
|
|
459
|
+
let schema = FINDINGS_SCHEMA
|
|
460
|
+
if (dim.key === 'implementation') {
|
|
461
|
+
schema = IMPLEMENTATION_SCHEMA
|
|
462
|
+
} else if (dim.key === 'test-quality') {
|
|
463
|
+
schema = TEST_QUALITY_SCHEMA
|
|
464
|
+
}
|
|
358
465
|
const result = await agent(reviewPrompt(dim), {
|
|
359
466
|
label: `review:${dim.key}`,
|
|
360
467
|
phase: 'Review',
|
|
361
|
-
schema
|
|
468
|
+
schema,
|
|
362
469
|
model: dim.model,
|
|
363
470
|
})
|
|
364
471
|
|
|
365
472
|
if (!result) {
|
|
366
473
|
log(`${dim.key}: agent returned null, skipping`)
|
|
367
|
-
return []
|
|
474
|
+
return { key: dim.key, findings: [] }
|
|
368
475
|
}
|
|
369
476
|
|
|
370
477
|
const findings = (result.findings || []).map(function(f) {
|
|
@@ -374,44 +481,45 @@ const budgetAtStart = budget.spent()
|
|
|
374
481
|
})
|
|
375
482
|
|
|
376
483
|
log(`${dim.key}: ${findings.length} findings, ${
|
|
377
|
-
findings.filter(function(f) { return shouldVerify(f
|
|
484
|
+
findings.filter(function(f) { return shouldVerify(f) }).length
|
|
378
485
|
} to verify`)
|
|
379
486
|
|
|
380
|
-
return
|
|
487
|
+
return {
|
|
488
|
+
key: dim.key,
|
|
489
|
+
findings,
|
|
490
|
+
acCoverage: dim.key === 'implementation' ? result.acCoverage : undefined,
|
|
491
|
+
coverageVerdict: dim.key === 'test-quality' ? result.coverageVerdict : undefined,
|
|
492
|
+
mutationSlip: dim.key === 'test-quality' ? result.mutationSlip : undefined,
|
|
493
|
+
mockSmells: dim.key === 'test-quality' ? result.mockSmells : undefined,
|
|
494
|
+
}
|
|
381
495
|
}
|
|
382
496
|
|
|
383
497
|
// -------------------------------------------------------------------------
|
|
384
498
|
// Verify stage: takes the deduped cross-dimension finding set, splits into
|
|
385
|
-
// in-scope (
|
|
386
|
-
//
|
|
387
|
-
//
|
|
499
|
+
// in-scope (verdict-affecting) and out-of-scope (non-blocking), and runs
|
|
500
|
+
// verifiers in parallel over the in-scope set. Hard rules use the verifier's
|
|
501
|
+
// rule-specific checks and skip only its runtime-oriented challenges.
|
|
388
502
|
// -------------------------------------------------------------------------
|
|
389
503
|
|
|
390
504
|
async function verifyAll(findings) {
|
|
391
505
|
const inScope = findings.filter(function(f) {
|
|
392
|
-
return shouldVerify(f
|
|
506
|
+
return shouldVerify(f)
|
|
393
507
|
})
|
|
394
508
|
const outOfScope = findings.filter(function(f) {
|
|
395
|
-
return !shouldVerify(f
|
|
396
|
-
})
|
|
397
|
-
|
|
398
|
-
const exempted = outOfScope.map(function(f) {
|
|
399
|
-
if (!isHardRule(f)) {
|
|
400
|
-
return f
|
|
401
|
-
}
|
|
402
|
-
hardRuleCount++
|
|
403
|
-
const exempt = Object.assign({}, f)
|
|
404
|
-
exempt.verifierReasoning = `hard rule ${f.rule}: exempt from adversarial verification (standards violation, not a runtime-failure claim)`
|
|
405
|
-
return exempt
|
|
509
|
+
return !shouldVerify(f)
|
|
406
510
|
})
|
|
407
511
|
|
|
408
512
|
if (inScope.length === 0) {
|
|
409
|
-
return
|
|
513
|
+
return outOfScope
|
|
410
514
|
}
|
|
411
515
|
|
|
412
516
|
const verified = await batched(inScope.map(function(finding) {
|
|
413
517
|
return async function() {
|
|
414
518
|
agentCount++
|
|
519
|
+
if (isHardRule(finding)) {
|
|
520
|
+
hardRuleCount++
|
|
521
|
+
}
|
|
522
|
+
|
|
415
523
|
const verdict = await agent(verifierPrompt(finding), {
|
|
416
524
|
label: `verify:${finding.file}:${finding.line}`,
|
|
417
525
|
model: 'sonnet',
|
|
@@ -421,9 +529,12 @@ const budgetAtStart = budget.spent()
|
|
|
421
529
|
})
|
|
422
530
|
|
|
423
531
|
if (!verdict) {
|
|
424
|
-
//
|
|
532
|
+
// Missing verification cannot support a blocking finding.
|
|
533
|
+
downgradedCount++
|
|
425
534
|
const kept = Object.assign({}, finding)
|
|
426
|
-
kept.
|
|
535
|
+
kept.severity = 'observation'
|
|
536
|
+
kept.verificationStatus = 'unavailable'
|
|
537
|
+
kept.verifierReasoning = 'verifier agent returned no usable result; downgraded to non-blocking'
|
|
427
538
|
return kept
|
|
428
539
|
}
|
|
429
540
|
|
|
@@ -436,18 +547,20 @@ const budgetAtStart = budget.spent()
|
|
|
436
547
|
downgradedCount++
|
|
437
548
|
const downgraded = Object.assign({}, finding)
|
|
438
549
|
downgraded.severity = verdict.newSeverity || 'observation'
|
|
550
|
+
downgraded.verificationStatus = 'downgraded'
|
|
439
551
|
downgraded.verifierReasoning = verdict.reasoning
|
|
440
552
|
return downgraded
|
|
441
553
|
}
|
|
442
554
|
|
|
443
555
|
// confirmed
|
|
444
556
|
const confirmed = Object.assign({}, finding)
|
|
557
|
+
confirmed.verificationStatus = isHardRule(finding) ? 'hard-rule-confirmed' : 'confirmed'
|
|
445
558
|
confirmed.verifierReasoning = verdict.reasoning
|
|
446
559
|
return confirmed
|
|
447
560
|
}
|
|
448
561
|
}), FANOUT_PLAN)
|
|
449
562
|
|
|
450
|
-
return
|
|
563
|
+
return outOfScope.concat(verified.filter(Boolean))
|
|
451
564
|
}
|
|
452
565
|
|
|
453
566
|
// -------------------------------------------------------------------------
|
|
@@ -465,10 +578,19 @@ const budgetAtStart = budget.spent()
|
|
|
465
578
|
return function() { return reviewStage(dim) }
|
|
466
579
|
}), REVIEW_PLAN)
|
|
467
580
|
|
|
468
|
-
const
|
|
581
|
+
const reviewResults = reviewed.filter(Boolean)
|
|
582
|
+
const implementationResult = reviewResults.find(function(result) {
|
|
583
|
+
return result.key === 'implementation'
|
|
584
|
+
}) || {}
|
|
585
|
+
const testQualityResult = reviewResults.find(function(result) {
|
|
586
|
+
return result.key === 'test-quality'
|
|
587
|
+
}) || {}
|
|
588
|
+
const deduped = dedup(reviewResults.flatMap(function(result) {
|
|
589
|
+
return result.findings
|
|
590
|
+
}))
|
|
469
591
|
|
|
470
592
|
log(`${deduped.length} unique finding(s) after dedup; verifying ${
|
|
471
|
-
deduped.filter(function(f) { return shouldVerify(f
|
|
593
|
+
deduped.filter(function(f) { return shouldVerify(f) }).length
|
|
472
594
|
}`)
|
|
473
595
|
|
|
474
596
|
phase('Verify')
|
|
@@ -552,11 +674,17 @@ const budgetAtStart = budget.spent()
|
|
|
552
674
|
return []
|
|
553
675
|
}
|
|
554
676
|
|
|
677
|
+
if (isHardRule(newFinding)) {
|
|
678
|
+
hardRuleCount++
|
|
679
|
+
}
|
|
680
|
+
|
|
555
681
|
if (verdict.verdict === 'downgraded') {
|
|
556
682
|
downgradedCount++
|
|
557
683
|
newFinding.severity = verdict.newSeverity || 'observation'
|
|
684
|
+
newFinding.verificationStatus = 'downgraded'
|
|
558
685
|
newFinding.verifierReasoning = verdict.reasoning
|
|
559
686
|
} else {
|
|
687
|
+
newFinding.verificationStatus = isHardRule(newFinding) ? 'hard-rule-confirmed' : 'confirmed'
|
|
560
688
|
newFinding.verifierReasoning = verdict.reasoning
|
|
561
689
|
}
|
|
562
690
|
|
|
@@ -622,9 +750,19 @@ const budgetAtStart = budget.spent()
|
|
|
622
750
|
return
|
|
623
751
|
}
|
|
624
752
|
|
|
625
|
-
|
|
626
|
-
|
|
753
|
+
const proof = normalizeProof(result)
|
|
754
|
+
finding.proof = proof
|
|
755
|
+
if (proof.proven === true) {
|
|
627
756
|
provenCount++
|
|
757
|
+
finding.verificationStatus = 'proven'
|
|
758
|
+
} else if (proof.outcome === 'passed') {
|
|
759
|
+
downgradedCount++
|
|
760
|
+
finding.severity = 'important'
|
|
761
|
+
finding.verificationStatus = 'counter-evidence'
|
|
762
|
+
finding.verifierReasoning = [
|
|
763
|
+
finding.verifierReasoning,
|
|
764
|
+
`Proof test passed: ${proof.reason}`,
|
|
765
|
+
].filter(Boolean).join(' ')
|
|
628
766
|
}
|
|
629
767
|
}
|
|
630
768
|
}), FANOUT_PLAN)
|
|
@@ -648,6 +786,10 @@ const budgetAtStart = budget.spent()
|
|
|
648
786
|
agentCount,
|
|
649
787
|
proveAttemptCount,
|
|
650
788
|
provenCount,
|
|
789
|
+
acCoverage: implementationResult.acCoverage || null,
|
|
790
|
+
coverageVerdict: testQualityResult.coverageVerdict || null,
|
|
791
|
+
mutationSlip: testQualityResult.mutationSlip || null,
|
|
792
|
+
mockSmells: Array.isArray(testQualityResult.mockSmells) ? testQualityResult.mockSmells : [],
|
|
651
793
|
outputTokens: budget.spent() - budgetAtStart,
|
|
652
794
|
turnTokensTotal: budget.spent(),
|
|
653
795
|
}
|