code-foundry 1.22.1 → 1.25.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/workflows/eval.yml +116 -0
- package/.github/workflows/validation-no-codeql.yml +34 -6
- package/.github/workflows/validation.yml +32 -5
- package/.github/workflows/validation_audit_self-ci.yml +1 -0
- package/.github/workflows/validation_self-ci.yml +1 -0
- package/AGENTS.md +1 -0
- package/CHANGELOG.md +42 -0
- package/README.md +3 -0
- package/docs/CONFIGURATION.md +20 -13
- package/docs/EVALS.md +139 -0
- package/docs/WORKFLOWS.md +12 -4
- package/docs/required-capabilities.md +6 -4
- package/package.json +1 -1
- package/src/commands/qualified-publication.mjs +80 -2
- package/src/commands/release-integrity.mjs +37 -14
- package/src/commands/sync.mjs +1 -0
- package/src/lib/eval-envelope.mjs +234 -0
- package/src/lib/merge-queue.mjs +1 -0
- package/src/lib/task-policy.mjs +7 -0
- package/src/lib/validation-policy.mjs +10 -6
- package/src/runtime-core.mjs +154 -0
- package/src/runtime.mjs +2 -0
- package/src/templates/gitignore +2 -0
package/src/runtime-core.mjs
CHANGED
|
@@ -10,6 +10,13 @@ import { classifyTestFiles } from './lib/test-discovery.mjs'
|
|
|
10
10
|
import { classifyValidationMode, evaluateValidationGate } from './lib/validation-policy.mjs'
|
|
11
11
|
import { readReleaseConfig, validateGeneratedReleaseDiff } from './lib/release-policy.mjs'
|
|
12
12
|
import { runNodePackagePerformance } from './lib/node-package-performance.mjs'
|
|
13
|
+
import {
|
|
14
|
+
EVAL_BUDGET_FILE_DEFAULT,
|
|
15
|
+
EVAL_REPORT_FILE,
|
|
16
|
+
EVAL_SUMMARY_FILE,
|
|
17
|
+
evaluateEvalBudgets,
|
|
18
|
+
validateEvalReport,
|
|
19
|
+
} from './lib/eval-envelope.mjs'
|
|
13
20
|
|
|
14
21
|
const root = process.cwd()
|
|
15
22
|
const config = readConfig(resolve(root, '.github/code-foundry.yml'))
|
|
@@ -57,6 +64,26 @@ function performanceEnabled() {
|
|
|
57
64
|
return configured(config.performance, 'auto') !== 'false'
|
|
58
65
|
}
|
|
59
66
|
|
|
67
|
+
function evalEnabled() {
|
|
68
|
+
return configured(config.eval, 'auto') !== 'false'
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
function evalCommand() {
|
|
72
|
+
const raw = configured(config.eval_command, '').trim()
|
|
73
|
+
if (!raw) return null
|
|
74
|
+
let command
|
|
75
|
+
try {
|
|
76
|
+
command = JSON.parse(raw)
|
|
77
|
+
} catch {
|
|
78
|
+
throw new Error('eval_command must be a JSON argv array.')
|
|
79
|
+
}
|
|
80
|
+
if (!Array.isArray(command) || command.length === 0)
|
|
81
|
+
throw new Error('eval_command must be a non-empty JSON array.')
|
|
82
|
+
if (!command.every((argument) => typeof argument === 'string' && argument.length > 0))
|
|
83
|
+
throw new Error('eval_command must contain only non-empty strings.')
|
|
84
|
+
return /** @type {string[]} */ (command)
|
|
85
|
+
}
|
|
86
|
+
|
|
60
87
|
function performanceCommands() {
|
|
61
88
|
const raw = configured(config.performance_command, '').trim()
|
|
62
89
|
if (!raw) return []
|
|
@@ -128,6 +155,122 @@ function writePerformanceSummary(startedAt, status, commands, artifacts, error =
|
|
|
128
155
|
)
|
|
129
156
|
}
|
|
130
157
|
|
|
158
|
+
const evalResultsDirectory = 'eval-results'
|
|
159
|
+
|
|
160
|
+
/** @returns {{source: string, argv: string[]}[]} */
|
|
161
|
+
function selectedEvalCommands() {
|
|
162
|
+
const name = ['eval'].find((candidate) => hasScript(candidate))
|
|
163
|
+
if (name) {
|
|
164
|
+
const [manager, args] = packageCommand(['run', name])
|
|
165
|
+
if (!manager) throw new Error(`Cannot run ${name}: select a supported package_manager.`)
|
|
166
|
+
return [{ source: `package-script:${name}`, argv: [manager, ...args] }]
|
|
167
|
+
}
|
|
168
|
+
const command = evalCommand()
|
|
169
|
+
if (!command) throw new Error('No eval script or eval_command was discovered.')
|
|
170
|
+
return [{ source: 'configuration', argv: command }]
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
/** @param {string} startedAt @param {'passed'|'failed'} status @param {{source: string, argv: string[], status: number}[]} commands @param {string[]} artifacts @param {string|null} error @param {{file: string, applied: boolean, failures: string[]}|null} budgets */
|
|
174
|
+
function writeEvalSummary(startedAt, status, commands, artifacts, error = null, budgets = null) {
|
|
175
|
+
const directory = resolve(root, evalResultsDirectory)
|
|
176
|
+
mkdirSync(directory, { recursive: true })
|
|
177
|
+
writeFileSync(
|
|
178
|
+
resolve(directory, 'summary.json'),
|
|
179
|
+
`${JSON.stringify(
|
|
180
|
+
{
|
|
181
|
+
schemaVersion: 1,
|
|
182
|
+
kind: 'code-foundry-eval-summary',
|
|
183
|
+
status,
|
|
184
|
+
startedAt,
|
|
185
|
+
completedAt: new Date().toISOString(),
|
|
186
|
+
commands,
|
|
187
|
+
report: configured(config.eval_report_file, EVAL_REPORT_FILE),
|
|
188
|
+
budgets,
|
|
189
|
+
artifacts,
|
|
190
|
+
error,
|
|
191
|
+
},
|
|
192
|
+
null,
|
|
193
|
+
2
|
|
194
|
+
)}\n`
|
|
195
|
+
)
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
function runEval() {
|
|
199
|
+
if (!evalEnabled()) return
|
|
200
|
+
const startedAt = new Date().toISOString()
|
|
201
|
+
/** @type {{source: string, argv: string[], status: number}[]} */
|
|
202
|
+
const records = []
|
|
203
|
+
const artifacts = [EVAL_REPORT_FILE, EVAL_SUMMARY_FILE]
|
|
204
|
+
const budgetFile = configured(config.eval_budget_file, EVAL_BUDGET_FILE_DEFAULT)
|
|
205
|
+
/** @type {{file: string, applied: boolean, failures: string[]}|null} */
|
|
206
|
+
let budgets = null
|
|
207
|
+
try {
|
|
208
|
+
for (const command of selectedEvalCommands()) {
|
|
209
|
+
const result = spawnSync(command.argv[0], command.argv.slice(1), {
|
|
210
|
+
cwd: root,
|
|
211
|
+
stdio: 'inherit',
|
|
212
|
+
env: process.env,
|
|
213
|
+
})
|
|
214
|
+
if (result.error) throw result.error
|
|
215
|
+
const status = result.status ?? 1
|
|
216
|
+
records.push({ ...command, status })
|
|
217
|
+
if (status !== 0) {
|
|
218
|
+
writeEvalSummary(
|
|
219
|
+
startedAt,
|
|
220
|
+
'failed',
|
|
221
|
+
records,
|
|
222
|
+
artifacts,
|
|
223
|
+
`command exited ${status}`,
|
|
224
|
+
budgets
|
|
225
|
+
)
|
|
226
|
+
process.exitCode = status
|
|
227
|
+
return
|
|
228
|
+
}
|
|
229
|
+
}
|
|
230
|
+
const reportFile = resolve(root, configured(config.eval_report_file, EVAL_REPORT_FILE))
|
|
231
|
+
if (!existsSync(reportFile))
|
|
232
|
+
throw new Error(
|
|
233
|
+
`Eval report was not produced: ${configured(config.eval_report_file, EVAL_REPORT_FILE)}`
|
|
234
|
+
)
|
|
235
|
+
let report
|
|
236
|
+
try {
|
|
237
|
+
report = JSON.parse(readFileSync(reportFile, 'utf8'))
|
|
238
|
+
} catch (error) {
|
|
239
|
+
throw new Error(
|
|
240
|
+
`Eval report is not valid JSON: ${error instanceof Error ? error.message : String(error)}`
|
|
241
|
+
)
|
|
242
|
+
}
|
|
243
|
+
const envelope = validateEvalReport(report)
|
|
244
|
+
if (!envelope.valid)
|
|
245
|
+
throw new Error(`Eval report violates the contract: ${envelope.errors.join('; ')}`)
|
|
246
|
+
if (existsSync(resolve(root, budgetFile))) {
|
|
247
|
+
const raw = JSON.parse(readFileSync(resolve(root, budgetFile), 'utf8'))
|
|
248
|
+
const gate = evaluateEvalBudgets(report, raw)
|
|
249
|
+
budgets = { file: budgetFile, applied: true, failures: gate.failures }
|
|
250
|
+
if (!gate.passed) {
|
|
251
|
+
for (const failure of gate.failures) console.error(`::error::${failure}`)
|
|
252
|
+
writeEvalSummary(
|
|
253
|
+
startedAt,
|
|
254
|
+
'failed',
|
|
255
|
+
records,
|
|
256
|
+
artifacts,
|
|
257
|
+
`eval budgets failed: ${gate.failures.join('; ')}`,
|
|
258
|
+
budgets
|
|
259
|
+
)
|
|
260
|
+
process.exitCode = 1
|
|
261
|
+
return
|
|
262
|
+
}
|
|
263
|
+
} else {
|
|
264
|
+
budgets = { file: budgetFile, applied: false, failures: [] }
|
|
265
|
+
}
|
|
266
|
+
writeEvalSummary(startedAt, 'passed', records, artifacts, null, budgets)
|
|
267
|
+
} catch (error) {
|
|
268
|
+
const message = error instanceof Error ? error.message : String(error)
|
|
269
|
+
writeEvalSummary(startedAt, 'failed', records, artifacts, message, budgets)
|
|
270
|
+
throw error
|
|
271
|
+
}
|
|
272
|
+
}
|
|
273
|
+
|
|
131
274
|
function runPerformance() {
|
|
132
275
|
if (!performanceEnabled()) return
|
|
133
276
|
const startedAt = new Date().toISOString()
|
|
@@ -195,6 +338,7 @@ function validation(task) {
|
|
|
195
338
|
test: process.env.FOUNDRY_TEST,
|
|
196
339
|
security: process.env.FOUNDRY_SECURITY,
|
|
197
340
|
codeql: process.env.FOUNDRY_CODEQL,
|
|
341
|
+
eval: process.env.FOUNDRY_EVAL,
|
|
198
342
|
},
|
|
199
343
|
})
|
|
200
344
|
if (gate.valid) {
|
|
@@ -391,8 +535,15 @@ function relevant(task) {
|
|
|
391
535
|
integration: ['test:integration'],
|
|
392
536
|
e2e: ['test:e2e', 'e2e'],
|
|
393
537
|
smoke: ['test:smoke', 'smoke'],
|
|
538
|
+
eval: ['eval'],
|
|
394
539
|
performance: ['performance:check', 'perf:check'],
|
|
395
540
|
}[task]
|
|
541
|
+
if (task === 'eval') {
|
|
542
|
+
if (!evalEnabled()) return false
|
|
543
|
+
return Boolean(
|
|
544
|
+
(scripted && scripted.some((candidate) => hasScript(candidate))) || evalCommand()
|
|
545
|
+
)
|
|
546
|
+
}
|
|
396
547
|
if (task === 'performance') {
|
|
397
548
|
if (!performanceEnabled()) return false
|
|
398
549
|
return Boolean(
|
|
@@ -563,6 +714,9 @@ function ci(task) {
|
|
|
563
714
|
run('cargo', ['build', '--all-targets'])
|
|
564
715
|
return
|
|
565
716
|
}
|
|
717
|
+
if (task === 'eval') {
|
|
718
|
+
return runEval()
|
|
719
|
+
}
|
|
566
720
|
if (task === 'performance') {
|
|
567
721
|
return runPerformance()
|
|
568
722
|
}
|
package/src/runtime.mjs
CHANGED
|
@@ -186,6 +186,8 @@ export function runRuntime(args, root = process.cwd(), entry = core) {
|
|
|
186
186
|
report.coverage = evaluateCoverage(root, policy, before)
|
|
187
187
|
report.artifacts.push(...report.coverage.artifacts)
|
|
188
188
|
}
|
|
189
|
+
if (task === 'eval')
|
|
190
|
+
report.artifacts.push('eval-results/summary.json', 'eval-results/result.json')
|
|
189
191
|
if (task === 'performance') report.artifacts.push('performance-results/summary.json')
|
|
190
192
|
report.status = 'passed'
|
|
191
193
|
return 0
|