code-foundry 1.22.1 → 1.28.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/workflows/cloudflare-delivery.yml +134 -1
- package/.github/workflows/cloudflare-deploy.yml +140 -1
- package/.github/workflows/consumer-qualification.yml +3 -3
- package/.github/workflows/eval.yml +116 -0
- package/.github/workflows/qualified-foundry-publish.yml +4 -5
- package/.github/workflows/release_self-ci.yml +24 -10
- package/.github/workflows/validation-no-codeql.yml +34 -6
- package/.github/workflows/validation.yml +32 -5
- package/.github/workflows/validation_audit_self-ci.yml +1 -0
- package/.github/workflows/validation_self-ci.yml +7 -1
- package/AGENTS.md +10 -0
- package/CHANGELOG.md +77 -0
- package/README.md +4 -1
- package/docs/CONFIGURATION.md +20 -13
- package/docs/EVALS.md +139 -0
- package/docs/PUBLISHING.md +6 -1
- package/docs/WORKFLOWS.md +27 -4
- package/docs/cloudflare-delivery.md +47 -25
- package/docs/consumer-qualification.md +1 -1
- package/docs/fleet-release-eligibility.md +1 -5
- package/docs/qualified-publication.md +1 -1
- package/docs/required-capabilities.md +6 -4
- package/package.json +3 -3
- package/src/commands/doctor.mjs +5 -2
- package/src/commands/fleet-core.mjs +1 -1
- package/src/commands/qualified-publication.mjs +81 -3
- package/src/commands/release-integrity.mjs +37 -14
- package/src/commands/sync.mjs +3 -2
- package/src/lib/docs-only.mjs +39 -0
- package/src/lib/eval-envelope.mjs +234 -0
- package/src/lib/fleet-manifest.mjs +9 -5
- package/src/lib/merge-queue.mjs +1 -0
- package/src/lib/overlay.mjs +1 -1
- package/src/lib/release-manifest.mjs +1 -1
- package/src/lib/release-policy.mjs +2 -2
- package/src/lib/task-policy.mjs +7 -0
- package/src/lib/validation-policy.mjs +10 -6
- package/src/runtime-core.mjs +187 -10
- package/src/runtime.mjs +2 -0
- package/src/templates/gitignore +2 -0
|
@@ -0,0 +1,234 @@
|
|
|
1
|
+
// @ts-check
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* The shared eval-report contract. Every `ci eval` run validates the consumer
|
|
5
|
+
* repository's eval report against this envelope before applying budgets.
|
|
6
|
+
* Deterministic probes (Layer 1) and model-agent runs (Layer 2) emit the same
|
|
7
|
+
* envelope, so task outcomes stay comparable across executors and revisions.
|
|
8
|
+
* See docs/EVALS.md for the full field reference.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
export const EVAL_REPORT_FILE = 'eval-results/result.json'
|
|
12
|
+
export const EVAL_SUMMARY_FILE = 'eval-results/summary.json'
|
|
13
|
+
export const EVAL_BUDGET_FILE_DEFAULT = 'eval-budgets.json'
|
|
14
|
+
export const EVAL_REPORT_SCHEMA_VERSION = 1
|
|
15
|
+
export const EVAL_SUMMARY_KIND = 'code-foundry-eval-summary'
|
|
16
|
+
|
|
17
|
+
const FAILURE_CLASSES = ['harness', 'task']
|
|
18
|
+
const STAT_FIELDS = ['mean', 'p50', 'p95', 'max']
|
|
19
|
+
|
|
20
|
+
/** @param {unknown} value @returns {value is number} */
|
|
21
|
+
function isNonNegativeInteger(value) {
|
|
22
|
+
return typeof value === 'number' && Number.isSafeInteger(value) && value >= 0
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
/** @param {unknown} value @returns {value is number} */
|
|
26
|
+
function isFiniteNumber(value) {
|
|
27
|
+
return typeof value === 'number' && Number.isFinite(value)
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
/** @param {unknown} value @returns {boolean} */
|
|
31
|
+
function isStats(value) {
|
|
32
|
+
if (!value || typeof value !== 'object' || Array.isArray(value)) return false
|
|
33
|
+
const record = /** @type {Record<string, unknown>} */ (value)
|
|
34
|
+
if (!isNonNegativeInteger(record.count)) return false
|
|
35
|
+
if (record.count === 0) {
|
|
36
|
+
return (
|
|
37
|
+
Object.keys(record).length === 1 || STAT_FIELDS.every((field) => record[field] === undefined)
|
|
38
|
+
)
|
|
39
|
+
}
|
|
40
|
+
return STAT_FIELDS.every((field) => isFiniteNumber(record[field]))
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Validate one eval report envelope. Returns every violation instead of
|
|
45
|
+
* throwing, so one broken report lists all of its problems at once.
|
|
46
|
+
* @param {unknown} report @returns {{valid: boolean, errors: string[]}}
|
|
47
|
+
*/
|
|
48
|
+
export function validateEvalReport(report) {
|
|
49
|
+
/** @type {string[]} */
|
|
50
|
+
const errors = []
|
|
51
|
+
/** @param {string} message */
|
|
52
|
+
const push = (message) => errors.push(message)
|
|
53
|
+
if (!report || typeof report !== 'object' || Array.isArray(report)) {
|
|
54
|
+
return { valid: false, errors: ['report must be a JSON object'] }
|
|
55
|
+
}
|
|
56
|
+
const value = /** @type {Record<string, unknown>} */ (report)
|
|
57
|
+
if (value.schemaVersion !== EVAL_REPORT_SCHEMA_VERSION)
|
|
58
|
+
push(`schemaVersion: expected ${EVAL_REPORT_SCHEMA_VERSION}`)
|
|
59
|
+
if (value.revision !== undefined && typeof value.revision !== 'string')
|
|
60
|
+
push('revision: must be a string when present')
|
|
61
|
+
if (value.dependencyHash !== undefined && typeof value.dependencyHash !== 'string')
|
|
62
|
+
push('dependencyHash: must be a string when present')
|
|
63
|
+
|
|
64
|
+
const summary = value.summary
|
|
65
|
+
if (!summary || typeof summary !== 'object' || Array.isArray(summary)) {
|
|
66
|
+
push('summary: must be an object')
|
|
67
|
+
return { valid: errors.length === 0, errors }
|
|
68
|
+
}
|
|
69
|
+
const counts = /** @type {Record<string, unknown>} */ (summary)
|
|
70
|
+
for (const field of [
|
|
71
|
+
'taskCount',
|
|
72
|
+
'attempts',
|
|
73
|
+
'passed',
|
|
74
|
+
'failed',
|
|
75
|
+
'harnessFailures',
|
|
76
|
+
'toolCalls',
|
|
77
|
+
'evidenceErrors',
|
|
78
|
+
]) {
|
|
79
|
+
if (!isNonNegativeInteger(counts[field]))
|
|
80
|
+
push(`summary.${field}: must be a non-negative integer`)
|
|
81
|
+
}
|
|
82
|
+
if (
|
|
83
|
+
isNonNegativeInteger(counts.passed) &&
|
|
84
|
+
isNonNegativeInteger(counts.failed) &&
|
|
85
|
+
isNonNegativeInteger(counts.attempts) &&
|
|
86
|
+
counts.passed + counts.failed > counts.attempts
|
|
87
|
+
)
|
|
88
|
+
push('summary.passed/failed: must not exceed attempts')
|
|
89
|
+
if (
|
|
90
|
+
isNonNegativeInteger(counts.harnessFailures) &&
|
|
91
|
+
isNonNegativeInteger(counts.attempts) &&
|
|
92
|
+
counts.harnessFailures > counts.attempts
|
|
93
|
+
)
|
|
94
|
+
push('summary.harnessFailures: must not exceed attempts')
|
|
95
|
+
if (isFiniteNumber(counts.successRate) && (counts.successRate < 0 || counts.successRate > 1))
|
|
96
|
+
push('summary.successRate: must be between 0 and 1')
|
|
97
|
+
for (const field of ['taskDurationMs', 'startupMs', 'stepDurationMs']) {
|
|
98
|
+
if (!isStats(counts[field])) push(`summary.${field}: must be a stats object`)
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
const tasks = value.tasks
|
|
102
|
+
if (!Array.isArray(tasks)) {
|
|
103
|
+
push('tasks: must be an array')
|
|
104
|
+
return { valid: errors.length === 0, errors }
|
|
105
|
+
}
|
|
106
|
+
for (const [index, task] of tasks.entries()) {
|
|
107
|
+
if (!task || typeof task !== 'object' || Array.isArray(task)) {
|
|
108
|
+
push(`tasks[${index}]: must be an object`)
|
|
109
|
+
continue
|
|
110
|
+
}
|
|
111
|
+
const record = /** @type {Record<string, unknown>} */ (task)
|
|
112
|
+
if (typeof record.id !== 'string' || record.id.trim().length === 0)
|
|
113
|
+
push(`tasks[${index}].id: must be a non-empty string`)
|
|
114
|
+
if (!Array.isArray(record.attempts)) {
|
|
115
|
+
push(`tasks[${index}].attempts: must be an array`)
|
|
116
|
+
continue
|
|
117
|
+
}
|
|
118
|
+
for (const [attemptIndex, attempt] of record.attempts.entries()) {
|
|
119
|
+
const path = `tasks[${index}].attempts[${attemptIndex}]`
|
|
120
|
+
if (!attempt || typeof attempt !== 'object') {
|
|
121
|
+
push(`${path}: must be an object`)
|
|
122
|
+
continue
|
|
123
|
+
}
|
|
124
|
+
const entry = /** @type {Record<string, unknown>} */ (attempt)
|
|
125
|
+
if (entry.status !== 'passed' && entry.status !== 'failed')
|
|
126
|
+
push(`${path}.status: must be passed or failed`)
|
|
127
|
+
if (entry.durationMs !== undefined && !isFiniteNumber(entry.durationMs))
|
|
128
|
+
push(`${path}.durationMs: must be a finite number`)
|
|
129
|
+
if (entry.startupMs !== undefined && !isFiniteNumber(entry.startupMs))
|
|
130
|
+
push(`${path}.startupMs: must be a finite number`)
|
|
131
|
+
if (entry.failure) {
|
|
132
|
+
const failure = entry.failure
|
|
133
|
+
if (!failure || typeof failure !== 'object') push(`${path}.failure: must be an object`)
|
|
134
|
+
else if (
|
|
135
|
+
typeof (/** @type {Record<string, unknown>} */ (failure).message) !== 'string' ||
|
|
136
|
+
/** @type {Record<string, unknown>} */ (failure).message === ''
|
|
137
|
+
)
|
|
138
|
+
push(`${path}.failure.message: must be a non-empty string`)
|
|
139
|
+
if (!FAILURE_CLASSES.includes(/** @type {string} */ (entry.failureClass)))
|
|
140
|
+
push(`${path}.failureClass: must be harness or task when a failure is present`)
|
|
141
|
+
} else if (entry.failureClass !== undefined) {
|
|
142
|
+
if (!FAILURE_CLASSES.includes(/** @type {string} */ (entry.failureClass)))
|
|
143
|
+
push(`${path}.failureClass: must be harness or task when present`)
|
|
144
|
+
}
|
|
145
|
+
if (Array.isArray(entry.steps)) {
|
|
146
|
+
for (const [stepIndex, stepEntry] of entry.steps.entries()) {
|
|
147
|
+
const stepPath = `${path}.steps[${stepIndex}]`
|
|
148
|
+
if (!stepEntry || typeof stepEntry !== 'object') {
|
|
149
|
+
push(`${stepPath}: must be an object`)
|
|
150
|
+
continue
|
|
151
|
+
}
|
|
152
|
+
const step = /** @type {Record<string, unknown>} */ (stepEntry)
|
|
153
|
+
if (typeof step.tool !== 'string' || step.tool.trim().length === 0)
|
|
154
|
+
push(`${stepPath}.tool: must be a non-empty string`)
|
|
155
|
+
if (!['running', 'passed', 'failed'].includes(/** @type {string} */ (step.status)))
|
|
156
|
+
push(`${stepPath}.status: must be running, passed, or failed`)
|
|
157
|
+
if (step.durationMs !== undefined && !isFiniteNumber(step.durationMs))
|
|
158
|
+
push(`${stepPath}.durationMs: must be a finite number`)
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
}
|
|
162
|
+
}
|
|
163
|
+
return { valid: errors.length === 0, errors }
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
/**
|
|
167
|
+
* Evaluate the configured budget thresholds against one validated report.
|
|
168
|
+
* Unknown budget keys fail closed so a typo can never silently disable a gate.
|
|
169
|
+
* @param {{summary?: Record<string, unknown>}} report
|
|
170
|
+
* @param {Record<string, unknown>} budgets
|
|
171
|
+
* @returns {{passed: boolean, failures: string[]}}
|
|
172
|
+
*/
|
|
173
|
+
export function evaluateEvalBudgets(report, budgets) {
|
|
174
|
+
if (!report || typeof report !== 'object' || Array.isArray(report))
|
|
175
|
+
return { passed: false, failures: ['report must be a validated eval report object'] }
|
|
176
|
+
const failures = []
|
|
177
|
+
if (!budgets || typeof budgets !== 'object' || Array.isArray(budgets)) {
|
|
178
|
+
return { passed: false, failures: ['budgets must be a JSON object'] }
|
|
179
|
+
}
|
|
180
|
+
const known = new Set([
|
|
181
|
+
'successRate',
|
|
182
|
+
'taskP95Ms',
|
|
183
|
+
'startupP95Ms',
|
|
184
|
+
'stepP95Ms',
|
|
185
|
+
'maxHarnessFailures',
|
|
186
|
+
'maxEvidenceErrors',
|
|
187
|
+
'maxToolCalls',
|
|
188
|
+
])
|
|
189
|
+
for (const key of Object.keys(budgets)) {
|
|
190
|
+
if (!known.has(key)) failures.push(`budgets.${key}: unknown budget key`)
|
|
191
|
+
}
|
|
192
|
+
if (failures.length > 0) return { passed: false, failures }
|
|
193
|
+
/** @type {Record<string, unknown>} */
|
|
194
|
+
const summary = report.summary ?? {}
|
|
195
|
+
const rate = budgets.successRate
|
|
196
|
+
if (rate !== undefined) {
|
|
197
|
+
if (!isFiniteNumber(rate) || rate < 0 || rate > 1)
|
|
198
|
+
return { passed: false, failures: ['budgets.successRate: must be between 0 and 1'] }
|
|
199
|
+
if (isFiniteNumber(summary.successRate) && summary.successRate < rate)
|
|
200
|
+
failures.push(`successRate ${summary.successRate.toFixed(2)} is below budget ${rate}`)
|
|
201
|
+
}
|
|
202
|
+
const p95Fields = /** @type {const} */ ([
|
|
203
|
+
['taskP95Ms', 'taskDurationMs'],
|
|
204
|
+
['startupP95Ms', 'startupMs'],
|
|
205
|
+
['stepP95Ms', 'stepDurationMs'],
|
|
206
|
+
])
|
|
207
|
+
for (const [budgetField, summaryField] of p95Fields) {
|
|
208
|
+
const budget = /** @type {unknown} */ (budgets[budgetField])
|
|
209
|
+
if (budget === undefined) continue
|
|
210
|
+
if (!isFiniteNumber(budget) || budget <= 0)
|
|
211
|
+
return { passed: false, failures: [`budgets.${budgetField}: must be a positive number`] }
|
|
212
|
+
const stats = summary[summaryField]
|
|
213
|
+
if (!stats || typeof stats !== 'object') continue
|
|
214
|
+
const p95 = /** @type {unknown} */ (/** @type {Record<string, unknown>} */ (stats).p95)
|
|
215
|
+
const count = /** @type {Record<string, unknown>} */ (stats).count
|
|
216
|
+
if (!isFiniteNumber(p95) || count === 0) continue
|
|
217
|
+
if (p95 > budget) failures.push(`${summaryField}.p95 ${p95}ms is above budget ${budget}ms`)
|
|
218
|
+
}
|
|
219
|
+
const maxFields = /** @type {const} */ ([
|
|
220
|
+
['maxHarnessFailures', 'harnessFailures'],
|
|
221
|
+
['maxEvidenceErrors', 'evidenceErrors'],
|
|
222
|
+
['maxToolCalls', 'toolCalls'],
|
|
223
|
+
])
|
|
224
|
+
for (const [budgetField, summaryField] of maxFields) {
|
|
225
|
+
const budget = /** @type {unknown} */ (budgets[budgetField])
|
|
226
|
+
if (budget === undefined) continue
|
|
227
|
+
if (!isNonNegativeInteger(budget))
|
|
228
|
+
return { passed: false, failures: [`budgets.${budgetField}: must be a non-negative integer`] }
|
|
229
|
+
const measured = /** @type {unknown} */ (summary[summaryField])
|
|
230
|
+
if (isNonNegativeInteger(measured) && measured > budget)
|
|
231
|
+
failures.push(`${summaryField} ${measured} is above budget ${budget}`)
|
|
232
|
+
}
|
|
233
|
+
return { passed: failures.length === 0, failures }
|
|
234
|
+
}
|
|
@@ -277,16 +277,20 @@ export function verifiedPullRequest(pr, required) {
|
|
|
277
277
|
return false
|
|
278
278
|
const checks = pr.statusCheckRollup
|
|
279
279
|
if (!Array.isArray(checks) || !checks.length) return false
|
|
280
|
-
|
|
281
|
-
const state = (check) =>
|
|
282
|
-
check.status === 'COMPLETED' ? check.conclusion : (check.state ?? check.status)
|
|
283
|
-
if (checks.some((check) => !['SUCCESS', 'NEUTRAL', 'SKIPPED'].includes(state(check))))
|
|
280
|
+
if (checks.some((check) => !['SUCCESS', 'NEUTRAL', 'SKIPPED'].includes(checkState(check))))
|
|
284
281
|
return false
|
|
285
282
|
return required.every((name) =>
|
|
286
|
-
checks.some(
|
|
283
|
+
checks.some(
|
|
284
|
+
(check) => (check.name ?? check.context) === name && checkState(check) === 'SUCCESS'
|
|
285
|
+
)
|
|
287
286
|
)
|
|
288
287
|
}
|
|
289
288
|
|
|
289
|
+
/** @param {any} check */
|
|
290
|
+
function checkState(check) {
|
|
291
|
+
return check.status === 'COMPLETED' ? check.conclusion : (check.state ?? check.status)
|
|
292
|
+
}
|
|
293
|
+
|
|
290
294
|
/** @param {string} text */
|
|
291
295
|
function scalarConfig(text) {
|
|
292
296
|
return Object.fromEntries(
|
package/src/lib/merge-queue.mjs
CHANGED
|
@@ -153,6 +153,7 @@ jobs:
|
|
|
153
153
|
unit-runner: ${runner('unit_runner', 'ubuntu-slim')}
|
|
154
154
|
performance-runner: ${runner('performance_runner')}
|
|
155
155
|
security-runner: ${runner('security_runner', 'ubuntu-slim')}
|
|
156
|
+
eval-runner: ${runner('eval_runner')}
|
|
156
157
|
codeql-runner: ${runner('codeql_runner')}
|
|
157
158
|
rust-shards: '${JSON.stringify(shards)}'
|
|
158
159
|
rust-threads: '${threads}'
|
package/src/lib/overlay.mjs
CHANGED
|
@@ -16,7 +16,7 @@ export function customWorkflowFiles(root, standardFiles) {
|
|
|
16
16
|
.filter((file) => file.endsWith('.yml') || file.endsWith('.yaml'))
|
|
17
17
|
.map((file) => `.github/workflows/${file}`)
|
|
18
18
|
.filter((file) => !isManagedPath(standardFiles, file))
|
|
19
|
-
.
|
|
19
|
+
.toSorted()
|
|
20
20
|
}
|
|
21
21
|
|
|
22
22
|
/** @param {string} root @param {Record<string, string>} config */
|
|
@@ -295,8 +295,8 @@ export function buildReleaseRecoveryPlan(input) {
|
|
|
295
295
|
const orphanGitHubReleases = releaseTags.filter(
|
|
296
296
|
(tag) => !tagSet.has(tag) && !tagSet.has(tag.replace(/^v/, ''))
|
|
297
297
|
)
|
|
298
|
-
const latestTag = [...tags].
|
|
299
|
-
const latestPackageVersion = [...input.packageVersions].
|
|
298
|
+
const latestTag = [...tags].toSorted(compareVersions).at(-1) ?? ''
|
|
299
|
+
const latestPackageVersion = [...input.packageVersions].toSorted(compareVersions).at(-1) ?? ''
|
|
300
300
|
return {
|
|
301
301
|
latestTag,
|
|
302
302
|
latestPackageVersion,
|
package/src/lib/task-policy.mjs
CHANGED
|
@@ -15,6 +15,7 @@ export const TASKS = Object.freeze([
|
|
|
15
15
|
'integration',
|
|
16
16
|
'e2e',
|
|
17
17
|
'smoke',
|
|
18
|
+
'eval',
|
|
18
19
|
'performance',
|
|
19
20
|
])
|
|
20
21
|
|
|
@@ -28,6 +29,7 @@ export const TASK_SCRIPTS = Object.freeze({
|
|
|
28
29
|
integration: ['test:integration'],
|
|
29
30
|
e2e: ['test:e2e', 'e2e'],
|
|
30
31
|
smoke: ['test:smoke', 'smoke'],
|
|
32
|
+
eval: ['eval'],
|
|
31
33
|
performance: ['performance:check', 'perf:check'],
|
|
32
34
|
})
|
|
33
35
|
|
|
@@ -41,6 +43,8 @@ export function readTaskPolicy(root) {
|
|
|
41
43
|
}
|
|
42
44
|
if (!['true', 'false', 'auto'].includes(config.performance ?? 'auto'))
|
|
43
45
|
throw new Error('performance must be true, false, or auto')
|
|
46
|
+
if (!['true', 'false', 'auto'].includes(config.eval ?? 'auto'))
|
|
47
|
+
throw new Error('eval must be true, false, or auto')
|
|
44
48
|
const coverageMode = config.coverage_enforcement ?? 'auto'
|
|
45
49
|
if (!['auto', 'required', 'off'].includes(coverageMode))
|
|
46
50
|
throw new Error('coverage_enforcement must be auto, required, or off')
|
|
@@ -50,6 +54,9 @@ export function readTaskPolicy(root) {
|
|
|
50
54
|
throw new Error('Required performance cannot use performance: false')
|
|
51
55
|
if (config.performance === 'true' && !required.includes('performance'))
|
|
52
56
|
required.push('performance')
|
|
57
|
+
if (required.includes('eval') && config.eval === 'false')
|
|
58
|
+
throw new Error('Required eval cannot use eval: false')
|
|
59
|
+
if (config.eval === 'true' && !required.includes('eval')) required.push('eval')
|
|
53
60
|
const coverageRequired = required.includes('coverage') || coverageMode === 'required'
|
|
54
61
|
if (coverageRequired && !required.includes('unit')) required.push('unit')
|
|
55
62
|
const minimum = Number(config.coverage_minimum ?? '80')
|
|
@@ -22,7 +22,7 @@ export const AGGREGATE_CHECK_NAME = 'Validation / Gate'
|
|
|
22
22
|
/** Job ids owned by the validation orchestrator. Release runs the full audit
|
|
23
23
|
* suite and validates its generated diff inside the gate, so no separate
|
|
24
24
|
* release-policy job id exists. */
|
|
25
|
-
export const VALIDATION_JOBS = ['ci', 'test', 'security', 'codeql']
|
|
25
|
+
export const VALIDATION_JOBS = ['ci', 'test', 'security', 'codeql', 'eval']
|
|
26
26
|
|
|
27
27
|
/** Events that may trigger canonical validation. */
|
|
28
28
|
export const VALIDATION_EVENTS = ['pull_request', 'schedule', 'workflow_dispatch']
|
|
@@ -81,11 +81,15 @@ export function classifyValidationMode(input) {
|
|
|
81
81
|
/** @type {Record<'fast'|'audit'|'release', string[]>} */
|
|
82
82
|
const REQUIRED_JOBS_BY_MODE = {
|
|
83
83
|
fast: ['ci', 'test'],
|
|
84
|
-
audit: ['ci', 'test', 'security', 'codeql'],
|
|
85
|
-
// Release Please pull requests
|
|
86
|
-
//
|
|
87
|
-
//
|
|
88
|
-
|
|
84
|
+
audit: ['ci', 'test', 'security', 'codeql', 'eval'],
|
|
85
|
+
// Release Please pull requests change version metadata only, and the gate
|
|
86
|
+
// validates that diff against the release policy before anything publishes.
|
|
87
|
+
// The content tree was already fully audited by the pull requests that
|
|
88
|
+
// merged into main, and the scheduled audit lane re-covers drift. The
|
|
89
|
+
// release tier therefore requires the fast suite plus CodeQL, whose
|
|
90
|
+
// per-pull-request code scanning results some repository rulesets require
|
|
91
|
+
// to keep the release from deadlocking.
|
|
92
|
+
release: ['ci', 'test', 'codeql'],
|
|
89
93
|
}
|
|
90
94
|
|
|
91
95
|
/**
|
package/src/runtime-core.mjs
CHANGED
|
@@ -7,9 +7,17 @@ import { spawnSync } from 'node:child_process'
|
|
|
7
7
|
import { detectPackageManager, resolveProfile } from './lib/profile.mjs'
|
|
8
8
|
import { configured, readConfig } from './lib/config.mjs'
|
|
9
9
|
import { classifyTestFiles } from './lib/test-discovery.mjs'
|
|
10
|
+
import { docsOnlyPullRequest } from './lib/docs-only.mjs'
|
|
10
11
|
import { classifyValidationMode, evaluateValidationGate } from './lib/validation-policy.mjs'
|
|
11
12
|
import { readReleaseConfig, validateGeneratedReleaseDiff } from './lib/release-policy.mjs'
|
|
12
13
|
import { runNodePackagePerformance } from './lib/node-package-performance.mjs'
|
|
14
|
+
import {
|
|
15
|
+
EVAL_BUDGET_FILE_DEFAULT,
|
|
16
|
+
EVAL_REPORT_FILE,
|
|
17
|
+
EVAL_SUMMARY_FILE,
|
|
18
|
+
evaluateEvalBudgets,
|
|
19
|
+
validateEvalReport,
|
|
20
|
+
} from './lib/eval-envelope.mjs'
|
|
13
21
|
|
|
14
22
|
const root = process.cwd()
|
|
15
23
|
const config = readConfig(resolve(root, '.github/code-foundry.yml'))
|
|
@@ -57,6 +65,26 @@ function performanceEnabled() {
|
|
|
57
65
|
return configured(config.performance, 'auto') !== 'false'
|
|
58
66
|
}
|
|
59
67
|
|
|
68
|
+
function evalEnabled() {
|
|
69
|
+
return configured(config.eval, 'auto') !== 'false'
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
function evalCommand() {
|
|
73
|
+
const raw = configured(config.eval_command, '').trim()
|
|
74
|
+
if (!raw) return null
|
|
75
|
+
let command
|
|
76
|
+
try {
|
|
77
|
+
command = JSON.parse(raw)
|
|
78
|
+
} catch {
|
|
79
|
+
throw new Error('eval_command must be a JSON argv array.')
|
|
80
|
+
}
|
|
81
|
+
if (!Array.isArray(command) || command.length === 0)
|
|
82
|
+
throw new Error('eval_command must be a non-empty JSON array.')
|
|
83
|
+
if (!command.every((argument) => typeof argument === 'string' && argument.length > 0))
|
|
84
|
+
throw new Error('eval_command must contain only non-empty strings.')
|
|
85
|
+
return /** @type {string[]} */ (command)
|
|
86
|
+
}
|
|
87
|
+
|
|
60
88
|
function performanceCommands() {
|
|
61
89
|
const raw = configured(config.performance_command, '').trim()
|
|
62
90
|
if (!raw) return []
|
|
@@ -128,6 +156,123 @@ function writePerformanceSummary(startedAt, status, commands, artifacts, error =
|
|
|
128
156
|
)
|
|
129
157
|
}
|
|
130
158
|
|
|
159
|
+
const evalResultsDirectory = 'eval-results'
|
|
160
|
+
|
|
161
|
+
/** @returns {{source: string, argv: string[]}[]} */
|
|
162
|
+
function selectedEvalCommands() {
|
|
163
|
+
const name = ['eval'].find((candidate) => hasScript(candidate))
|
|
164
|
+
if (name) {
|
|
165
|
+
const [manager, args] = packageCommand(['run', name])
|
|
166
|
+
if (!manager) throw new Error(`Cannot run ${name}: select a supported package_manager.`)
|
|
167
|
+
return [{ source: `package-script:${name}`, argv: [manager, ...args] }]
|
|
168
|
+
}
|
|
169
|
+
const command = evalCommand()
|
|
170
|
+
if (!command) throw new Error('No eval script or eval_command was discovered.')
|
|
171
|
+
return [{ source: 'configuration', argv: command }]
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
/** @param {string} startedAt @param {'passed'|'failed'} status @param {{source: string, argv: string[], status: number}[]} commands @param {string[]} artifacts @param {string|null} error @param {{file: string, applied: boolean, failures: string[]}|null} budgets */
|
|
175
|
+
function writeEvalSummary(startedAt, status, commands, artifacts, error = null, budgets = null) {
|
|
176
|
+
const directory = resolve(root, evalResultsDirectory)
|
|
177
|
+
mkdirSync(directory, { recursive: true })
|
|
178
|
+
writeFileSync(
|
|
179
|
+
resolve(directory, 'summary.json'),
|
|
180
|
+
`${JSON.stringify(
|
|
181
|
+
{
|
|
182
|
+
schemaVersion: 1,
|
|
183
|
+
kind: 'code-foundry-eval-summary',
|
|
184
|
+
status,
|
|
185
|
+
startedAt,
|
|
186
|
+
completedAt: new Date().toISOString(),
|
|
187
|
+
commands,
|
|
188
|
+
report: configured(config.eval_report_file, EVAL_REPORT_FILE),
|
|
189
|
+
budgets,
|
|
190
|
+
artifacts,
|
|
191
|
+
error,
|
|
192
|
+
},
|
|
193
|
+
null,
|
|
194
|
+
2
|
|
195
|
+
)}\n`
|
|
196
|
+
)
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
function runEval() {
|
|
200
|
+
if (!evalEnabled()) return
|
|
201
|
+
const startedAt = new Date().toISOString()
|
|
202
|
+
/** @type {{source: string, argv: string[], status: number}[]} */
|
|
203
|
+
const records = []
|
|
204
|
+
const artifacts = [EVAL_REPORT_FILE, EVAL_SUMMARY_FILE]
|
|
205
|
+
const budgetFile = configured(config.eval_budget_file, EVAL_BUDGET_FILE_DEFAULT)
|
|
206
|
+
/** @type {{file: string, applied: boolean, failures: string[]}|null} */
|
|
207
|
+
let budgets = null
|
|
208
|
+
try {
|
|
209
|
+
for (const command of selectedEvalCommands()) {
|
|
210
|
+
const result = spawnSync(command.argv[0], command.argv.slice(1), {
|
|
211
|
+
cwd: root,
|
|
212
|
+
stdio: 'inherit',
|
|
213
|
+
env: process.env,
|
|
214
|
+
})
|
|
215
|
+
if (result.error) throw result.error
|
|
216
|
+
const status = result.status ?? 1
|
|
217
|
+
records.push({ ...command, status })
|
|
218
|
+
if (status !== 0) {
|
|
219
|
+
writeEvalSummary(
|
|
220
|
+
startedAt,
|
|
221
|
+
'failed',
|
|
222
|
+
records,
|
|
223
|
+
artifacts,
|
|
224
|
+
`command exited ${status}`,
|
|
225
|
+
budgets
|
|
226
|
+
)
|
|
227
|
+
process.exitCode = status
|
|
228
|
+
return
|
|
229
|
+
}
|
|
230
|
+
}
|
|
231
|
+
const reportFile = resolve(root, configured(config.eval_report_file, EVAL_REPORT_FILE))
|
|
232
|
+
if (!existsSync(reportFile))
|
|
233
|
+
throw new Error(
|
|
234
|
+
`Eval report was not produced: ${configured(config.eval_report_file, EVAL_REPORT_FILE)}`
|
|
235
|
+
)
|
|
236
|
+
let report
|
|
237
|
+
try {
|
|
238
|
+
report = JSON.parse(readFileSync(reportFile, 'utf8'))
|
|
239
|
+
} catch (error) {
|
|
240
|
+
throw new Error(
|
|
241
|
+
`Eval report is not valid JSON: ${error instanceof Error ? error.message : String(error)}`,
|
|
242
|
+
{ cause: error }
|
|
243
|
+
)
|
|
244
|
+
}
|
|
245
|
+
const envelope = validateEvalReport(report)
|
|
246
|
+
if (!envelope.valid)
|
|
247
|
+
throw new Error(`Eval report violates the contract: ${envelope.errors.join('; ')}`)
|
|
248
|
+
if (existsSync(resolve(root, budgetFile))) {
|
|
249
|
+
const raw = JSON.parse(readFileSync(resolve(root, budgetFile), 'utf8'))
|
|
250
|
+
const gate = evaluateEvalBudgets(report, raw)
|
|
251
|
+
budgets = { file: budgetFile, applied: true, failures: gate.failures }
|
|
252
|
+
if (!gate.passed) {
|
|
253
|
+
for (const failure of gate.failures) console.error(`::error::${failure}`)
|
|
254
|
+
writeEvalSummary(
|
|
255
|
+
startedAt,
|
|
256
|
+
'failed',
|
|
257
|
+
records,
|
|
258
|
+
artifacts,
|
|
259
|
+
`eval budgets failed: ${gate.failures.join('; ')}`,
|
|
260
|
+
budgets
|
|
261
|
+
)
|
|
262
|
+
process.exitCode = 1
|
|
263
|
+
return
|
|
264
|
+
}
|
|
265
|
+
} else {
|
|
266
|
+
budgets = { file: budgetFile, applied: false, failures: [] }
|
|
267
|
+
}
|
|
268
|
+
writeEvalSummary(startedAt, 'passed', records, artifacts, null, budgets)
|
|
269
|
+
} catch (error) {
|
|
270
|
+
const message = error instanceof Error ? error.message : String(error)
|
|
271
|
+
writeEvalSummary(startedAt, 'failed', records, artifacts, message, budgets)
|
|
272
|
+
throw error
|
|
273
|
+
}
|
|
274
|
+
}
|
|
275
|
+
|
|
131
276
|
function runPerformance() {
|
|
132
277
|
if (!performanceEnabled()) return
|
|
133
278
|
const startedAt = new Date().toISOString()
|
|
@@ -172,17 +317,37 @@ function runPerformance() {
|
|
|
172
317
|
}
|
|
173
318
|
}
|
|
174
319
|
|
|
320
|
+
/**
|
|
321
|
+
* Audit-mode pull requests whose diff is docs-only run the fast tier instead.
|
|
322
|
+
* Scheduled and manual audits always keep the full tier; release pull requests
|
|
323
|
+
* are classified separately and always run their lane.
|
|
324
|
+
* @param {string} mode
|
|
325
|
+
*/
|
|
326
|
+
function downgradeDocsOnlyPullRequest(mode) {
|
|
327
|
+
if (mode !== 'audit') return mode
|
|
328
|
+
if ((process.env.FOUNDRY_EVENT_NAME ?? '') !== 'pull_request') return mode
|
|
329
|
+
const baseRef = process.env.FOUNDRY_BASE_REF ?? ''
|
|
330
|
+
if (!baseRef) return mode
|
|
331
|
+
try {
|
|
332
|
+
return docsOnlyPullRequest(root, baseRef) ? 'fast' : mode
|
|
333
|
+
} catch {
|
|
334
|
+
return mode
|
|
335
|
+
}
|
|
336
|
+
}
|
|
337
|
+
|
|
175
338
|
/** @param {string} task */
|
|
176
339
|
function validation(task) {
|
|
177
340
|
if (task === 'mode') {
|
|
178
341
|
writeOutput(
|
|
179
342
|
'mode',
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
343
|
+
downgradeDocsOnlyPullRequest(
|
|
344
|
+
classifyValidationMode({
|
|
345
|
+
eventName: process.env.FOUNDRY_EVENT_NAME ?? '',
|
|
346
|
+
baseRef: process.env.FOUNDRY_BASE_REF ?? '',
|
|
347
|
+
headRef: process.env.FOUNDRY_HEAD_REF ?? '',
|
|
348
|
+
stagingMode: config.staging_validation_mode,
|
|
349
|
+
})
|
|
350
|
+
)
|
|
186
351
|
)
|
|
187
352
|
return
|
|
188
353
|
}
|
|
@@ -195,6 +360,7 @@ function validation(task) {
|
|
|
195
360
|
test: process.env.FOUNDRY_TEST,
|
|
196
361
|
security: process.env.FOUNDRY_SECURITY,
|
|
197
362
|
codeql: process.env.FOUNDRY_CODEQL,
|
|
363
|
+
eval: process.env.FOUNDRY_EVAL,
|
|
198
364
|
},
|
|
199
365
|
})
|
|
200
366
|
if (gate.valid) {
|
|
@@ -391,8 +557,15 @@ function relevant(task) {
|
|
|
391
557
|
integration: ['test:integration'],
|
|
392
558
|
e2e: ['test:e2e', 'e2e'],
|
|
393
559
|
smoke: ['test:smoke', 'smoke'],
|
|
560
|
+
eval: ['eval'],
|
|
394
561
|
performance: ['performance:check', 'perf:check'],
|
|
395
562
|
}[task]
|
|
563
|
+
if (task === 'eval') {
|
|
564
|
+
if (!evalEnabled()) return false
|
|
565
|
+
return Boolean(
|
|
566
|
+
(scripted && scripted.some((candidate) => hasScript(candidate))) || evalCommand()
|
|
567
|
+
)
|
|
568
|
+
}
|
|
396
569
|
if (task === 'performance') {
|
|
397
570
|
if (!performanceEnabled()) return false
|
|
398
571
|
return Boolean(
|
|
@@ -541,10 +714,11 @@ function ci(task) {
|
|
|
541
714
|
hasRootJavascriptProject() &&
|
|
542
715
|
hasOxlintSetup()
|
|
543
716
|
) {
|
|
544
|
-
// Oxlint is the baseline's linter.
|
|
545
|
-
//
|
|
546
|
-
//
|
|
547
|
-
|
|
717
|
+
// Oxlint is the baseline's linter. Warnings are denied so lint debt
|
|
718
|
+
// never accumulates silently. Repositories that use a different linter
|
|
719
|
+
// keep full control through their own `lint` script, which the runScript
|
|
720
|
+
// fallback above already honors.
|
|
721
|
+
runTool('oxlint', ['--deny-warnings'])
|
|
548
722
|
}
|
|
549
723
|
if (hasLanguage('python') && hasRootPythonProject()) runTool('ruff', ['check', '.'])
|
|
550
724
|
if (hasLanguage('rust') && hasRootRustProject())
|
|
@@ -563,6 +737,9 @@ function ci(task) {
|
|
|
563
737
|
run('cargo', ['build', '--all-targets'])
|
|
564
738
|
return
|
|
565
739
|
}
|
|
740
|
+
if (task === 'eval') {
|
|
741
|
+
return runEval()
|
|
742
|
+
}
|
|
566
743
|
if (task === 'performance') {
|
|
567
744
|
return runPerformance()
|
|
568
745
|
}
|
package/src/runtime.mjs
CHANGED
|
@@ -186,6 +186,8 @@ export function runRuntime(args, root = process.cwd(), entry = core) {
|
|
|
186
186
|
report.coverage = evaluateCoverage(root, policy, before)
|
|
187
187
|
report.artifacts.push(...report.coverage.artifacts)
|
|
188
188
|
}
|
|
189
|
+
if (task === 'eval')
|
|
190
|
+
report.artifacts.push('eval-results/summary.json', 'eval-results/result.json')
|
|
189
191
|
if (task === 'performance') report.artifacts.push('performance-results/summary.json')
|
|
190
192
|
report.status = 'passed'
|
|
191
193
|
return 0
|