code-foundry 1.22.1 → 1.25.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/workflows/eval.yml +116 -0
- package/.github/workflows/validation-no-codeql.yml +34 -6
- package/.github/workflows/validation.yml +32 -5
- package/.github/workflows/validation_audit_self-ci.yml +1 -0
- package/.github/workflows/validation_self-ci.yml +1 -0
- package/AGENTS.md +1 -0
- package/CHANGELOG.md +42 -0
- package/README.md +3 -0
- package/docs/CONFIGURATION.md +20 -13
- package/docs/EVALS.md +139 -0
- package/docs/WORKFLOWS.md +12 -4
- package/docs/required-capabilities.md +6 -4
- package/package.json +1 -1
- package/src/commands/qualified-publication.mjs +80 -2
- package/src/commands/release-integrity.mjs +37 -14
- package/src/commands/sync.mjs +1 -0
- package/src/lib/eval-envelope.mjs +234 -0
- package/src/lib/merge-queue.mjs +1 -0
- package/src/lib/task-policy.mjs +7 -0
- package/src/lib/validation-policy.mjs +10 -6
- package/src/runtime-core.mjs +154 -0
- package/src/runtime.mjs +2 -0
- package/src/templates/gitignore +2 -0
|
@@ -46,6 +46,24 @@ export function runCommand(command, args, cwd) {
|
|
|
46
46
|
)
|
|
47
47
|
return result.stdout
|
|
48
48
|
}
|
|
49
|
+
|
|
50
|
+
/** Default tolerant runner: GitHub CLI failures surface as retryable misses
|
|
51
|
+
* instead of aborting the publication flow.
|
|
52
|
+
* @param {string} command
|
|
53
|
+
* @param {string[]} args
|
|
54
|
+
* @param {string} [cwd]
|
|
55
|
+
* @returns {{status: number, stdout: string}}
|
|
56
|
+
*/
|
|
57
|
+
function defaultRunTolerant(command, args, cwd) {
|
|
58
|
+
const result = spawnSync(command, args, {
|
|
59
|
+
cwd,
|
|
60
|
+
encoding: 'utf8',
|
|
61
|
+
timeout: 180_000,
|
|
62
|
+
maxBuffer: 8 * 1024 * 1024,
|
|
63
|
+
env: { ...process.env, GH_HOST: 'github.com', GH_PROMPT_DISABLED: '1' },
|
|
64
|
+
})
|
|
65
|
+
return { status: result.error ? 1 : (result.status ?? 1), stdout: result.stdout ?? '' }
|
|
66
|
+
}
|
|
49
67
|
/** @param {Candidate} candidate */
|
|
50
68
|
export function validateCandidate(candidate) {
|
|
51
69
|
ensure(
|
|
@@ -225,10 +243,50 @@ export async function publishQualifiedArchive(candidate, reports, adapters = {})
|
|
|
225
243
|
return { ...before, status: 'published' }
|
|
226
244
|
}
|
|
227
245
|
|
|
246
|
+
/** @typedef {(command:string, args:string[], cwd?:string) => {status: number, stdout: string}} RunTolerant */
|
|
247
|
+
|
|
248
|
+
/** Poll the releases-by-tag endpoint through its eventual-consistency window.
|
|
249
|
+
* Immediately after `gh release edit --draft=false` the endpoint can still
|
|
250
|
+
* return 404; a strict verification in that window aborts an otherwise
|
|
251
|
+
* healthy publication. Bounded retries keep the flow strict: it only proceeds
|
|
252
|
+
* once the endpoint serves the published release with the exact tag.
|
|
253
|
+
* @param {Candidate} candidate
|
|
254
|
+
* @param {{runTolerant?: RunTolerant, delay?: (ms:number)=>Promise<void>, attempts?: number}} [adapters]
|
|
255
|
+
* @returns {Promise<Record<string, any>>}
|
|
256
|
+
*/
|
|
257
|
+
async function waitForPublishedRelease(candidate, adapters = {}) {
|
|
258
|
+
const runTolerant = adapters.runTolerant ?? defaultRunTolerant
|
|
259
|
+
const delay = adapters.delay ?? ((ms) => new Promise((resolve) => setTimeout(resolve, ms)))
|
|
260
|
+
const attempts = adapters.attempts ?? 15
|
|
261
|
+
for (let attempt = 1; attempt <= attempts; attempt += 1) {
|
|
262
|
+
const result = runTolerant('gh', [
|
|
263
|
+
'api',
|
|
264
|
+
'--hostname',
|
|
265
|
+
'github.com',
|
|
266
|
+
`repos/${candidate.repository}/releases/tags/${encodeURIComponent(candidate.tag)}`,
|
|
267
|
+
])
|
|
268
|
+
if (result.status === 0) {
|
|
269
|
+
/** @type {Record<string, any> | null} */
|
|
270
|
+
let release = null
|
|
271
|
+
try {
|
|
272
|
+
release = JSON.parse(result.stdout)
|
|
273
|
+
} catch {
|
|
274
|
+
release = null
|
|
275
|
+
}
|
|
276
|
+
if (release?.draft === false && release?.tag_name === candidate.tag) return release
|
|
277
|
+
}
|
|
278
|
+
if (attempt === attempts) break
|
|
279
|
+
await delay(2_000)
|
|
280
|
+
}
|
|
281
|
+
throw new Error(
|
|
282
|
+
'The published release was not retrievable via the tag endpoint within the consistency window'
|
|
283
|
+
)
|
|
284
|
+
}
|
|
285
|
+
|
|
228
286
|
/** Stage only a previously-created draft: never move tags, overwrite assets, or
|
|
229
287
|
* treat an API/permissions failure as an absent release. Authentication needs
|
|
230
288
|
* immutable-settings read and release write; no setting is mutated.
|
|
231
|
-
* @param {Candidate} candidate @param {string[]} reports @param {{run?:Run, verify?:Verify}} [adapters]
|
|
289
|
+
* @param {Candidate} candidate @param {string[]} reports @param {{run?:Run, runTolerant?:RunTolerant, delay?:(ms:number)=>Promise<void>, attempts?:number, verify?:Verify}} [adapters]
|
|
232
290
|
*/
|
|
233
291
|
export async function stageQualifiedRelease(candidate, reports, adapters = {}) {
|
|
234
292
|
const run = adapters.run ?? runCommand
|
|
@@ -332,7 +390,27 @@ export async function stageQualifiedRelease(candidate, reports, adapters = {}) {
|
|
|
332
390
|
'--repo',
|
|
333
391
|
candidate.repository,
|
|
334
392
|
])
|
|
335
|
-
|
|
393
|
+
// GitHub's release index is eventually consistent right after a draft is
|
|
394
|
+
// published; a strict verification a second later can observe a 404 and
|
|
395
|
+
// abort the publication even though the release is live. Poll until the
|
|
396
|
+
// index serves the release, then verify; the verification cycle itself
|
|
397
|
+
// retries through the same window and still fails closed in the end.
|
|
398
|
+
await waitForPublishedRelease(candidate, {
|
|
399
|
+
runTolerant: adapters.runTolerant,
|
|
400
|
+
delay: adapters.delay,
|
|
401
|
+
attempts: 10,
|
|
402
|
+
})
|
|
403
|
+
/** @type {Record<string, any>} */
|
|
404
|
+
let identity
|
|
405
|
+
for (let attempt = 1; ; attempt += 1) {
|
|
406
|
+
try {
|
|
407
|
+
identity = await (adapters.verify ?? verifier)(candidate)
|
|
408
|
+
break
|
|
409
|
+
} catch (error) {
|
|
410
|
+
if (attempt >= 10) throw error
|
|
411
|
+
await (adapters.delay ?? ((ms) => new Promise((resolve) => setTimeout(resolve, ms))))(5_000)
|
|
412
|
+
}
|
|
413
|
+
}
|
|
336
414
|
ensure(
|
|
337
415
|
identity.status === 'passed' &&
|
|
338
416
|
identity.immutable === true &&
|
|
@@ -7,7 +7,7 @@ import { basename, isAbsolute, relative, resolve, sep } from 'node:path'
|
|
|
7
7
|
import { spawnSync } from 'node:child_process'
|
|
8
8
|
import { pathToFileURL } from 'node:url'
|
|
9
9
|
|
|
10
|
-
/** @typedef {(argv: string[]) => {status: number|null, stdout: string}} Runner */
|
|
10
|
+
/** @typedef {(argv: string[]) => {status: number|null, stdout: string, stderr?: string}} Runner */
|
|
11
11
|
|
|
12
12
|
/** @type {Runner} */
|
|
13
13
|
export function runGh(argv) {
|
|
@@ -18,7 +18,7 @@ export function runGh(argv) {
|
|
|
18
18
|
env: { ...process.env, GH_PROMPT_DISABLED: '1', GH_HOST: 'github.com' },
|
|
19
19
|
})
|
|
20
20
|
if (result.error) throw new Error(`GitHub CLI unavailable: ${result.error.name}`)
|
|
21
|
-
return { status: result.status, stdout: result.stdout ?? '' }
|
|
21
|
+
return { status: result.status, stdout: result.stdout ?? '', stderr: result.stderr ?? '' }
|
|
22
22
|
}
|
|
23
23
|
|
|
24
24
|
/** @param {string} repository */
|
|
@@ -56,7 +56,7 @@ function json(run, argv) {
|
|
|
56
56
|
const result = run(argv)
|
|
57
57
|
if (result.status !== 0)
|
|
58
58
|
throw new Error(
|
|
59
|
-
`GitHub verification command failed (${result.status ?? 'signal'}); state is unverified`
|
|
59
|
+
`GitHub verification command failed (${result.status ?? 'signal'}); state is unverified: ${String(result.stderr ?? '').slice(0, 300) || 'no stderr captured'}`
|
|
60
60
|
)
|
|
61
61
|
try {
|
|
62
62
|
return JSON.parse(result.stdout)
|
|
@@ -197,6 +197,17 @@ export function verifyRelease(options, run = runGh) {
|
|
|
197
197
|
}
|
|
198
198
|
}
|
|
199
199
|
|
|
200
|
+
/** Sync pause used by the post-publication verifier while the release index
|
|
201
|
+
* catches up with a just-published release.
|
|
202
|
+
* @param {number} ms
|
|
203
|
+
*/
|
|
204
|
+
function sleepSync(ms) {
|
|
205
|
+
if (ms <= 0) return
|
|
206
|
+
Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, ms)
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
const RELEASE_VERIFICATION_ATTEMPTS = 6
|
|
210
|
+
|
|
200
211
|
/** @param {string[]} argv @param {Runner} [run] */
|
|
201
212
|
export function integrityCommand(argv, run = runGh) {
|
|
202
213
|
const [command, ...args] = argv
|
|
@@ -228,17 +239,29 @@ export function integrityCommand(argv, run = runGh) {
|
|
|
228
239
|
throw new Error('An argument is not supported by this integrity command')
|
|
229
240
|
if (command === 'settings') return verifyImmutableSetting(values['--repo'] ?? '', run)
|
|
230
241
|
if (command === 'manifest') return assetManifest(values['--root'] ?? process.cwd(), assets)
|
|
231
|
-
if (command === 'release')
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
+
if (command === 'release') {
|
|
243
|
+
const options = {
|
|
244
|
+
repository: values['--repo'] ?? '',
|
|
245
|
+
tag: values['--tag'] ?? '',
|
|
246
|
+
root: values['--root'],
|
|
247
|
+
expectedSha: values['--expected-sha'],
|
|
248
|
+
assets,
|
|
249
|
+
}
|
|
250
|
+
// The lane fires the instant GitHub marks the release published, while the
|
|
251
|
+
// releases-by-tag index can still lag behind. Retry through that window;
|
|
252
|
+
// every strict check still runs on every attempt and the command fails
|
|
253
|
+
// closed once the bounded window is exhausted.
|
|
254
|
+
let lastError
|
|
255
|
+
for (let attempt = 1; attempt <= RELEASE_VERIFICATION_ATTEMPTS; attempt += 1) {
|
|
256
|
+
try {
|
|
257
|
+
return verifyRelease(options, run)
|
|
258
|
+
} catch (error) {
|
|
259
|
+
lastError = error
|
|
260
|
+
if (attempt < RELEASE_VERIFICATION_ATTEMPTS) sleepSync(5_000)
|
|
261
|
+
}
|
|
262
|
+
}
|
|
263
|
+
throw lastError
|
|
264
|
+
}
|
|
242
265
|
throw new Error('Use release-integrity settings, release, or manifest')
|
|
243
266
|
}
|
|
244
267
|
|
package/src/commands/sync.mjs
CHANGED
|
@@ -714,6 +714,7 @@ function renderWorkflow(content, config, repository, ref, rustCodeql, file, self
|
|
|
714
714
|
'codeql-runner': config.codeql_runner ?? config.runner,
|
|
715
715
|
'unit-runner': config.unit_runner,
|
|
716
716
|
'performance-runner': config.performance_runner ?? config.test_runner ?? config.runner,
|
|
717
|
+
'eval-runner': config.eval_runner ?? config.runner,
|
|
717
718
|
}
|
|
718
719
|
for (const [input, value] of Object.entries(runnerInputs)) {
|
|
719
720
|
if (!value) continue
|
|
@@ -0,0 +1,234 @@
|
|
|
1
|
+
// @ts-check
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* The shared eval-report contract. Every `ci eval` run validates the consumer
|
|
5
|
+
* repository's eval report against this envelope before applying budgets.
|
|
6
|
+
* Deterministic probes (Layer 1) and model-agent runs (Layer 2) emit the same
|
|
7
|
+
* envelope, so task outcomes stay comparable across executors and revisions.
|
|
8
|
+
* See docs/EVALS.md for the full field reference.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
export const EVAL_REPORT_FILE = 'eval-results/result.json'
|
|
12
|
+
export const EVAL_SUMMARY_FILE = 'eval-results/summary.json'
|
|
13
|
+
export const EVAL_BUDGET_FILE_DEFAULT = 'eval-budgets.json'
|
|
14
|
+
export const EVAL_REPORT_SCHEMA_VERSION = 1
|
|
15
|
+
export const EVAL_SUMMARY_KIND = 'code-foundry-eval-summary'
|
|
16
|
+
|
|
17
|
+
const FAILURE_CLASSES = ['harness', 'task']
|
|
18
|
+
const STAT_FIELDS = ['mean', 'p50', 'p95', 'max']
|
|
19
|
+
|
|
20
|
+
/** @param {unknown} value @returns {value is number} */
|
|
21
|
+
function isNonNegativeInteger(value) {
|
|
22
|
+
return typeof value === 'number' && Number.isSafeInteger(value) && value >= 0
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
/** @param {unknown} value @returns {value is number} */
|
|
26
|
+
function isFiniteNumber(value) {
|
|
27
|
+
return typeof value === 'number' && Number.isFinite(value)
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
/** @param {unknown} value @returns {boolean} */
|
|
31
|
+
function isStats(value) {
|
|
32
|
+
if (!value || typeof value !== 'object' || Array.isArray(value)) return false
|
|
33
|
+
const record = /** @type {Record<string, unknown>} */ (value)
|
|
34
|
+
if (!isNonNegativeInteger(record.count)) return false
|
|
35
|
+
if (record.count === 0) {
|
|
36
|
+
return (
|
|
37
|
+
Object.keys(record).length === 1 || STAT_FIELDS.every((field) => record[field] === undefined)
|
|
38
|
+
)
|
|
39
|
+
}
|
|
40
|
+
return STAT_FIELDS.every((field) => isFiniteNumber(record[field]))
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Validate one eval report envelope. Returns every violation instead of
|
|
45
|
+
* throwing, so one broken report lists all of its problems at once.
|
|
46
|
+
* @param {unknown} report @returns {{valid: boolean, errors: string[]}}
|
|
47
|
+
*/
|
|
48
|
+
export function validateEvalReport(report) {
|
|
49
|
+
/** @type {string[]} */
|
|
50
|
+
const errors = []
|
|
51
|
+
/** @param {string} message */
|
|
52
|
+
const push = (message) => errors.push(message)
|
|
53
|
+
if (!report || typeof report !== 'object' || Array.isArray(report)) {
|
|
54
|
+
return { valid: false, errors: ['report must be a JSON object'] }
|
|
55
|
+
}
|
|
56
|
+
const value = /** @type {Record<string, unknown>} */ (report)
|
|
57
|
+
if (value.schemaVersion !== EVAL_REPORT_SCHEMA_VERSION)
|
|
58
|
+
push(`schemaVersion: expected ${EVAL_REPORT_SCHEMA_VERSION}`)
|
|
59
|
+
if (value.revision !== undefined && typeof value.revision !== 'string')
|
|
60
|
+
push('revision: must be a string when present')
|
|
61
|
+
if (value.dependencyHash !== undefined && typeof value.dependencyHash !== 'string')
|
|
62
|
+
push('dependencyHash: must be a string when present')
|
|
63
|
+
|
|
64
|
+
const summary = value.summary
|
|
65
|
+
if (!summary || typeof summary !== 'object' || Array.isArray(summary)) {
|
|
66
|
+
push('summary: must be an object')
|
|
67
|
+
return { valid: errors.length === 0, errors }
|
|
68
|
+
}
|
|
69
|
+
const counts = /** @type {Record<string, unknown>} */ (summary)
|
|
70
|
+
for (const field of [
|
|
71
|
+
'taskCount',
|
|
72
|
+
'attempts',
|
|
73
|
+
'passed',
|
|
74
|
+
'failed',
|
|
75
|
+
'harnessFailures',
|
|
76
|
+
'toolCalls',
|
|
77
|
+
'evidenceErrors',
|
|
78
|
+
]) {
|
|
79
|
+
if (!isNonNegativeInteger(counts[field]))
|
|
80
|
+
push(`summary.${field}: must be a non-negative integer`)
|
|
81
|
+
}
|
|
82
|
+
if (
|
|
83
|
+
isNonNegativeInteger(counts.passed) &&
|
|
84
|
+
isNonNegativeInteger(counts.failed) &&
|
|
85
|
+
isNonNegativeInteger(counts.attempts) &&
|
|
86
|
+
counts.passed + counts.failed > counts.attempts
|
|
87
|
+
)
|
|
88
|
+
push('summary.passed/failed: must not exceed attempts')
|
|
89
|
+
if (
|
|
90
|
+
isNonNegativeInteger(counts.harnessFailures) &&
|
|
91
|
+
isNonNegativeInteger(counts.attempts) &&
|
|
92
|
+
counts.harnessFailures > counts.attempts
|
|
93
|
+
)
|
|
94
|
+
push('summary.harnessFailures: must not exceed attempts')
|
|
95
|
+
if (isFiniteNumber(counts.successRate) && (counts.successRate < 0 || counts.successRate > 1))
|
|
96
|
+
push('summary.successRate: must be between 0 and 1')
|
|
97
|
+
for (const field of ['taskDurationMs', 'startupMs', 'stepDurationMs']) {
|
|
98
|
+
if (!isStats(counts[field])) push(`summary.${field}: must be a stats object`)
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
const tasks = value.tasks
|
|
102
|
+
if (!Array.isArray(tasks)) {
|
|
103
|
+
push('tasks: must be an array')
|
|
104
|
+
return { valid: errors.length === 0, errors }
|
|
105
|
+
}
|
|
106
|
+
for (const [index, task] of tasks.entries()) {
|
|
107
|
+
if (!task || typeof task !== 'object' || Array.isArray(task)) {
|
|
108
|
+
push(`tasks[${index}]: must be an object`)
|
|
109
|
+
continue
|
|
110
|
+
}
|
|
111
|
+
const record = /** @type {Record<string, unknown>} */ (task)
|
|
112
|
+
if (typeof record.id !== 'string' || record.id.trim().length === 0)
|
|
113
|
+
push(`tasks[${index}].id: must be a non-empty string`)
|
|
114
|
+
if (!Array.isArray(record.attempts)) {
|
|
115
|
+
push(`tasks[${index}].attempts: must be an array`)
|
|
116
|
+
continue
|
|
117
|
+
}
|
|
118
|
+
for (const [attemptIndex, attempt] of record.attempts.entries()) {
|
|
119
|
+
const path = `tasks[${index}].attempts[${attemptIndex}]`
|
|
120
|
+
if (!attempt || typeof attempt !== 'object') {
|
|
121
|
+
push(`${path}: must be an object`)
|
|
122
|
+
continue
|
|
123
|
+
}
|
|
124
|
+
const entry = /** @type {Record<string, unknown>} */ (attempt)
|
|
125
|
+
if (entry.status !== 'passed' && entry.status !== 'failed')
|
|
126
|
+
push(`${path}.status: must be passed or failed`)
|
|
127
|
+
if (entry.durationMs !== undefined && !isFiniteNumber(entry.durationMs))
|
|
128
|
+
push(`${path}.durationMs: must be a finite number`)
|
|
129
|
+
if (entry.startupMs !== undefined && !isFiniteNumber(entry.startupMs))
|
|
130
|
+
push(`${path}.startupMs: must be a finite number`)
|
|
131
|
+
if (entry.failure) {
|
|
132
|
+
const failure = entry.failure
|
|
133
|
+
if (!failure || typeof failure !== 'object') push(`${path}.failure: must be an object`)
|
|
134
|
+
else if (
|
|
135
|
+
typeof (/** @type {Record<string, unknown>} */ (failure).message) !== 'string' ||
|
|
136
|
+
/** @type {Record<string, unknown>} */ (failure).message === ''
|
|
137
|
+
)
|
|
138
|
+
push(`${path}.failure.message: must be a non-empty string`)
|
|
139
|
+
if (!FAILURE_CLASSES.includes(/** @type {string} */ (entry.failureClass)))
|
|
140
|
+
push(`${path}.failureClass: must be harness or task when a failure is present`)
|
|
141
|
+
} else if (entry.failureClass !== undefined) {
|
|
142
|
+
if (!FAILURE_CLASSES.includes(/** @type {string} */ (entry.failureClass)))
|
|
143
|
+
push(`${path}.failureClass: must be harness or task when present`)
|
|
144
|
+
}
|
|
145
|
+
if (Array.isArray(entry.steps)) {
|
|
146
|
+
for (const [stepIndex, stepEntry] of entry.steps.entries()) {
|
|
147
|
+
const stepPath = `${path}.steps[${stepIndex}]`
|
|
148
|
+
if (!stepEntry || typeof stepEntry !== 'object') {
|
|
149
|
+
push(`${stepPath}: must be an object`)
|
|
150
|
+
continue
|
|
151
|
+
}
|
|
152
|
+
const step = /** @type {Record<string, unknown>} */ (stepEntry)
|
|
153
|
+
if (typeof step.tool !== 'string' || step.tool.trim().length === 0)
|
|
154
|
+
push(`${stepPath}.tool: must be a non-empty string`)
|
|
155
|
+
if (!['running', 'passed', 'failed'].includes(/** @type {string} */ (step.status)))
|
|
156
|
+
push(`${stepPath}.status: must be running, passed, or failed`)
|
|
157
|
+
if (step.durationMs !== undefined && !isFiniteNumber(step.durationMs))
|
|
158
|
+
push(`${stepPath}.durationMs: must be a finite number`)
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
}
|
|
162
|
+
}
|
|
163
|
+
return { valid: errors.length === 0, errors }
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
/**
|
|
167
|
+
* Evaluate the configured budget thresholds against one validated report.
|
|
168
|
+
* Unknown budget keys fail closed so a typo can never silently disable a gate.
|
|
169
|
+
* @param {{summary?: Record<string, unknown>}} report
|
|
170
|
+
* @param {Record<string, unknown>} budgets
|
|
171
|
+
* @returns {{passed: boolean, failures: string[]}}
|
|
172
|
+
*/
|
|
173
|
+
export function evaluateEvalBudgets(report, budgets) {
|
|
174
|
+
if (!report || typeof report !== 'object' || Array.isArray(report))
|
|
175
|
+
return { passed: false, failures: ['report must be a validated eval report object'] }
|
|
176
|
+
const failures = []
|
|
177
|
+
if (!budgets || typeof budgets !== 'object' || Array.isArray(budgets)) {
|
|
178
|
+
return { passed: false, failures: ['budgets must be a JSON object'] }
|
|
179
|
+
}
|
|
180
|
+
const known = new Set([
|
|
181
|
+
'successRate',
|
|
182
|
+
'taskP95Ms',
|
|
183
|
+
'startupP95Ms',
|
|
184
|
+
'stepP95Ms',
|
|
185
|
+
'maxHarnessFailures',
|
|
186
|
+
'maxEvidenceErrors',
|
|
187
|
+
'maxToolCalls',
|
|
188
|
+
])
|
|
189
|
+
for (const key of Object.keys(budgets)) {
|
|
190
|
+
if (!known.has(key)) failures.push(`budgets.${key}: unknown budget key`)
|
|
191
|
+
}
|
|
192
|
+
if (failures.length > 0) return { passed: false, failures }
|
|
193
|
+
/** @type {Record<string, unknown>} */
|
|
194
|
+
const summary = report.summary ?? {}
|
|
195
|
+
const rate = budgets.successRate
|
|
196
|
+
if (rate !== undefined) {
|
|
197
|
+
if (!isFiniteNumber(rate) || rate < 0 || rate > 1)
|
|
198
|
+
return { passed: false, failures: ['budgets.successRate: must be between 0 and 1'] }
|
|
199
|
+
if (isFiniteNumber(summary.successRate) && summary.successRate < rate)
|
|
200
|
+
failures.push(`successRate ${summary.successRate.toFixed(2)} is below budget ${rate}`)
|
|
201
|
+
}
|
|
202
|
+
const p95Fields = /** @type {const} */ ([
|
|
203
|
+
['taskP95Ms', 'taskDurationMs'],
|
|
204
|
+
['startupP95Ms', 'startupMs'],
|
|
205
|
+
['stepP95Ms', 'stepDurationMs'],
|
|
206
|
+
])
|
|
207
|
+
for (const [budgetField, summaryField] of p95Fields) {
|
|
208
|
+
const budget = /** @type {unknown} */ (budgets[budgetField])
|
|
209
|
+
if (budget === undefined) continue
|
|
210
|
+
if (!isFiniteNumber(budget) || budget <= 0)
|
|
211
|
+
return { passed: false, failures: [`budgets.${budgetField}: must be a positive number`] }
|
|
212
|
+
const stats = summary[summaryField]
|
|
213
|
+
if (!stats || typeof stats !== 'object') continue
|
|
214
|
+
const p95 = /** @type {unknown} */ (/** @type {Record<string, unknown>} */ (stats).p95)
|
|
215
|
+
const count = /** @type {Record<string, unknown>} */ (stats).count
|
|
216
|
+
if (!isFiniteNumber(p95) || count === 0) continue
|
|
217
|
+
if (p95 > budget) failures.push(`${summaryField}.p95 ${p95}ms is above budget ${budget}ms`)
|
|
218
|
+
}
|
|
219
|
+
const maxFields = /** @type {const} */ ([
|
|
220
|
+
['maxHarnessFailures', 'harnessFailures'],
|
|
221
|
+
['maxEvidenceErrors', 'evidenceErrors'],
|
|
222
|
+
['maxToolCalls', 'toolCalls'],
|
|
223
|
+
])
|
|
224
|
+
for (const [budgetField, summaryField] of maxFields) {
|
|
225
|
+
const budget = /** @type {unknown} */ (budgets[budgetField])
|
|
226
|
+
if (budget === undefined) continue
|
|
227
|
+
if (!isNonNegativeInteger(budget))
|
|
228
|
+
return { passed: false, failures: [`budgets.${budgetField}: must be a non-negative integer`] }
|
|
229
|
+
const measured = /** @type {unknown} */ (summary[summaryField])
|
|
230
|
+
if (isNonNegativeInteger(measured) && measured > budget)
|
|
231
|
+
failures.push(`${summaryField} ${measured} is above budget ${budget}`)
|
|
232
|
+
}
|
|
233
|
+
return { passed: failures.length === 0, failures }
|
|
234
|
+
}
|
package/src/lib/merge-queue.mjs
CHANGED
|
@@ -153,6 +153,7 @@ jobs:
|
|
|
153
153
|
unit-runner: ${runner('unit_runner', 'ubuntu-slim')}
|
|
154
154
|
performance-runner: ${runner('performance_runner')}
|
|
155
155
|
security-runner: ${runner('security_runner', 'ubuntu-slim')}
|
|
156
|
+
eval-runner: ${runner('eval_runner')}
|
|
156
157
|
codeql-runner: ${runner('codeql_runner')}
|
|
157
158
|
rust-shards: '${JSON.stringify(shards)}'
|
|
158
159
|
rust-threads: '${threads}'
|
package/src/lib/task-policy.mjs
CHANGED
|
@@ -15,6 +15,7 @@ export const TASKS = Object.freeze([
|
|
|
15
15
|
'integration',
|
|
16
16
|
'e2e',
|
|
17
17
|
'smoke',
|
|
18
|
+
'eval',
|
|
18
19
|
'performance',
|
|
19
20
|
])
|
|
20
21
|
|
|
@@ -28,6 +29,7 @@ export const TASK_SCRIPTS = Object.freeze({
|
|
|
28
29
|
integration: ['test:integration'],
|
|
29
30
|
e2e: ['test:e2e', 'e2e'],
|
|
30
31
|
smoke: ['test:smoke', 'smoke'],
|
|
32
|
+
eval: ['eval'],
|
|
31
33
|
performance: ['performance:check', 'perf:check'],
|
|
32
34
|
})
|
|
33
35
|
|
|
@@ -41,6 +43,8 @@ export function readTaskPolicy(root) {
|
|
|
41
43
|
}
|
|
42
44
|
if (!['true', 'false', 'auto'].includes(config.performance ?? 'auto'))
|
|
43
45
|
throw new Error('performance must be true, false, or auto')
|
|
46
|
+
if (!['true', 'false', 'auto'].includes(config.eval ?? 'auto'))
|
|
47
|
+
throw new Error('eval must be true, false, or auto')
|
|
44
48
|
const coverageMode = config.coverage_enforcement ?? 'auto'
|
|
45
49
|
if (!['auto', 'required', 'off'].includes(coverageMode))
|
|
46
50
|
throw new Error('coverage_enforcement must be auto, required, or off')
|
|
@@ -50,6 +54,9 @@ export function readTaskPolicy(root) {
|
|
|
50
54
|
throw new Error('Required performance cannot use performance: false')
|
|
51
55
|
if (config.performance === 'true' && !required.includes('performance'))
|
|
52
56
|
required.push('performance')
|
|
57
|
+
if (required.includes('eval') && config.eval === 'false')
|
|
58
|
+
throw new Error('Required eval cannot use eval: false')
|
|
59
|
+
if (config.eval === 'true' && !required.includes('eval')) required.push('eval')
|
|
53
60
|
const coverageRequired = required.includes('coverage') || coverageMode === 'required'
|
|
54
61
|
if (coverageRequired && !required.includes('unit')) required.push('unit')
|
|
55
62
|
const minimum = Number(config.coverage_minimum ?? '80')
|
|
@@ -22,7 +22,7 @@ export const AGGREGATE_CHECK_NAME = 'Validation / Gate'
|
|
|
22
22
|
/** Job ids owned by the validation orchestrator. Release runs the full audit
|
|
23
23
|
* suite and validates its generated diff inside the gate, so no separate
|
|
24
24
|
* release-policy job id exists. */
|
|
25
|
-
export const VALIDATION_JOBS = ['ci', 'test', 'security', 'codeql']
|
|
25
|
+
export const VALIDATION_JOBS = ['ci', 'test', 'security', 'codeql', 'eval']
|
|
26
26
|
|
|
27
27
|
/** Events that may trigger canonical validation. */
|
|
28
28
|
export const VALIDATION_EVENTS = ['pull_request', 'schedule', 'workflow_dispatch']
|
|
@@ -81,11 +81,15 @@ export function classifyValidationMode(input) {
|
|
|
81
81
|
/** @type {Record<'fast'|'audit'|'release', string[]>} */
|
|
82
82
|
const REQUIRED_JOBS_BY_MODE = {
|
|
83
83
|
fast: ['ci', 'test'],
|
|
84
|
-
audit: ['ci', 'test', 'security', 'codeql'],
|
|
85
|
-
// Release Please pull requests
|
|
86
|
-
//
|
|
87
|
-
//
|
|
88
|
-
|
|
84
|
+
audit: ['ci', 'test', 'security', 'codeql', 'eval'],
|
|
85
|
+
// Release Please pull requests change version metadata only, and the gate
|
|
86
|
+
// validates that diff against the release policy before anything publishes.
|
|
87
|
+
// The content tree was already fully audited by the pull requests that
|
|
88
|
+
// merged into main, and the scheduled audit lane re-covers drift. The
|
|
89
|
+
// release tier therefore requires the fast suite plus CodeQL, whose
|
|
90
|
+
// per-pull-request code scanning results some repository rulesets require
|
|
91
|
+
// to keep the release from deadlocking.
|
|
92
|
+
release: ['ci', 'test', 'codeql'],
|
|
89
93
|
}
|
|
90
94
|
|
|
91
95
|
/**
|