code-foundry 1.22.1 → 1.28.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/.github/workflows/cloudflare-delivery.yml +134 -1
  2. package/.github/workflows/cloudflare-deploy.yml +140 -1
  3. package/.github/workflows/consumer-qualification.yml +3 -3
  4. package/.github/workflows/eval.yml +116 -0
  5. package/.github/workflows/qualified-foundry-publish.yml +4 -5
  6. package/.github/workflows/release_self-ci.yml +24 -10
  7. package/.github/workflows/validation-no-codeql.yml +34 -6
  8. package/.github/workflows/validation.yml +32 -5
  9. package/.github/workflows/validation_audit_self-ci.yml +1 -0
  10. package/.github/workflows/validation_self-ci.yml +7 -1
  11. package/AGENTS.md +10 -0
  12. package/CHANGELOG.md +77 -0
  13. package/README.md +4 -1
  14. package/docs/CONFIGURATION.md +20 -13
  15. package/docs/EVALS.md +139 -0
  16. package/docs/PUBLISHING.md +6 -1
  17. package/docs/WORKFLOWS.md +27 -4
  18. package/docs/cloudflare-delivery.md +47 -25
  19. package/docs/consumer-qualification.md +1 -1
  20. package/docs/fleet-release-eligibility.md +1 -5
  21. package/docs/qualified-publication.md +1 -1
  22. package/docs/required-capabilities.md +6 -4
  23. package/package.json +3 -3
  24. package/src/commands/doctor.mjs +5 -2
  25. package/src/commands/fleet-core.mjs +1 -1
  26. package/src/commands/qualified-publication.mjs +81 -3
  27. package/src/commands/release-integrity.mjs +37 -14
  28. package/src/commands/sync.mjs +3 -2
  29. package/src/lib/docs-only.mjs +39 -0
  30. package/src/lib/eval-envelope.mjs +234 -0
  31. package/src/lib/fleet-manifest.mjs +9 -5
  32. package/src/lib/merge-queue.mjs +1 -0
  33. package/src/lib/overlay.mjs +1 -1
  34. package/src/lib/release-manifest.mjs +1 -1
  35. package/src/lib/release-policy.mjs +2 -2
  36. package/src/lib/task-policy.mjs +7 -0
  37. package/src/lib/validation-policy.mjs +10 -6
  38. package/src/runtime-core.mjs +187 -10
  39. package/src/runtime.mjs +2 -0
  40. package/src/templates/gitignore +2 -0
@@ -0,0 +1,234 @@
1
+ // @ts-check
2
+
3
+ /**
4
+ * The shared eval-report contract. Every `ci eval` run validates the consumer
5
+ * repository's eval report against this envelope before applying budgets.
6
+ * Deterministic probes (Layer 1) and model-agent runs (Layer 2) emit the same
7
+ * envelope, so task outcomes stay comparable across executors and revisions.
8
+ * See docs/EVALS.md for the full field reference.
9
+ */
10
+
11
+ export const EVAL_REPORT_FILE = 'eval-results/result.json'
12
+ export const EVAL_SUMMARY_FILE = 'eval-results/summary.json'
13
+ export const EVAL_BUDGET_FILE_DEFAULT = 'eval-budgets.json'
14
+ export const EVAL_REPORT_SCHEMA_VERSION = 1
15
+ export const EVAL_SUMMARY_KIND = 'code-foundry-eval-summary'
16
+
17
+ const FAILURE_CLASSES = ['harness', 'task']
18
+ const STAT_FIELDS = ['mean', 'p50', 'p95', 'max']
19
+
20
+ /** @param {unknown} value @returns {value is number} */
21
+ function isNonNegativeInteger(value) {
22
+ return typeof value === 'number' && Number.isSafeInteger(value) && value >= 0
23
+ }
24
+
25
+ /** @param {unknown} value @returns {value is number} */
26
+ function isFiniteNumber(value) {
27
+ return typeof value === 'number' && Number.isFinite(value)
28
+ }
29
+
30
+ /** @param {unknown} value @returns {boolean} */
31
+ function isStats(value) {
32
+ if (!value || typeof value !== 'object' || Array.isArray(value)) return false
33
+ const record = /** @type {Record<string, unknown>} */ (value)
34
+ if (!isNonNegativeInteger(record.count)) return false
35
+ if (record.count === 0) {
36
+ return (
37
+ Object.keys(record).length === 1 || STAT_FIELDS.every((field) => record[field] === undefined)
38
+ )
39
+ }
40
+ return STAT_FIELDS.every((field) => isFiniteNumber(record[field]))
41
+ }
42
+
43
+ /**
44
+ * Validate one eval report envelope. Returns every violation instead of
45
+ * throwing, so one broken report lists all of its problems at once.
46
+ * @param {unknown} report @returns {{valid: boolean, errors: string[]}}
47
+ */
48
+ export function validateEvalReport(report) {
49
+ /** @type {string[]} */
50
+ const errors = []
51
+ /** @param {string} message */
52
+ const push = (message) => errors.push(message)
53
+ if (!report || typeof report !== 'object' || Array.isArray(report)) {
54
+ return { valid: false, errors: ['report must be a JSON object'] }
55
+ }
56
+ const value = /** @type {Record<string, unknown>} */ (report)
57
+ if (value.schemaVersion !== EVAL_REPORT_SCHEMA_VERSION)
58
+ push(`schemaVersion: expected ${EVAL_REPORT_SCHEMA_VERSION}`)
59
+ if (value.revision !== undefined && typeof value.revision !== 'string')
60
+ push('revision: must be a string when present')
61
+ if (value.dependencyHash !== undefined && typeof value.dependencyHash !== 'string')
62
+ push('dependencyHash: must be a string when present')
63
+
64
+ const summary = value.summary
65
+ if (!summary || typeof summary !== 'object' || Array.isArray(summary)) {
66
+ push('summary: must be an object')
67
+ return { valid: errors.length === 0, errors }
68
+ }
69
+ const counts = /** @type {Record<string, unknown>} */ (summary)
70
+ for (const field of [
71
+ 'taskCount',
72
+ 'attempts',
73
+ 'passed',
74
+ 'failed',
75
+ 'harnessFailures',
76
+ 'toolCalls',
77
+ 'evidenceErrors',
78
+ ]) {
79
+ if (!isNonNegativeInteger(counts[field]))
80
+ push(`summary.${field}: must be a non-negative integer`)
81
+ }
82
+ if (
83
+ isNonNegativeInteger(counts.passed) &&
84
+ isNonNegativeInteger(counts.failed) &&
85
+ isNonNegativeInteger(counts.attempts) &&
86
+ counts.passed + counts.failed > counts.attempts
87
+ )
88
+ push('summary.passed/failed: must not exceed attempts')
89
+ if (
90
+ isNonNegativeInteger(counts.harnessFailures) &&
91
+ isNonNegativeInteger(counts.attempts) &&
92
+ counts.harnessFailures > counts.attempts
93
+ )
94
+ push('summary.harnessFailures: must not exceed attempts')
95
+ if (isFiniteNumber(counts.successRate) && (counts.successRate < 0 || counts.successRate > 1))
96
+ push('summary.successRate: must be between 0 and 1')
97
+ for (const field of ['taskDurationMs', 'startupMs', 'stepDurationMs']) {
98
+ if (!isStats(counts[field])) push(`summary.${field}: must be a stats object`)
99
+ }
100
+
101
+ const tasks = value.tasks
102
+ if (!Array.isArray(tasks)) {
103
+ push('tasks: must be an array')
104
+ return { valid: errors.length === 0, errors }
105
+ }
106
+ for (const [index, task] of tasks.entries()) {
107
+ if (!task || typeof task !== 'object' || Array.isArray(task)) {
108
+ push(`tasks[${index}]: must be an object`)
109
+ continue
110
+ }
111
+ const record = /** @type {Record<string, unknown>} */ (task)
112
+ if (typeof record.id !== 'string' || record.id.trim().length === 0)
113
+ push(`tasks[${index}].id: must be a non-empty string`)
114
+ if (!Array.isArray(record.attempts)) {
115
+ push(`tasks[${index}].attempts: must be an array`)
116
+ continue
117
+ }
118
+ for (const [attemptIndex, attempt] of record.attempts.entries()) {
119
+ const path = `tasks[${index}].attempts[${attemptIndex}]`
120
+ if (!attempt || typeof attempt !== 'object') {
121
+ push(`${path}: must be an object`)
122
+ continue
123
+ }
124
+ const entry = /** @type {Record<string, unknown>} */ (attempt)
125
+ if (entry.status !== 'passed' && entry.status !== 'failed')
126
+ push(`${path}.status: must be passed or failed`)
127
+ if (entry.durationMs !== undefined && !isFiniteNumber(entry.durationMs))
128
+ push(`${path}.durationMs: must be a finite number`)
129
+ if (entry.startupMs !== undefined && !isFiniteNumber(entry.startupMs))
130
+ push(`${path}.startupMs: must be a finite number`)
131
+ if (entry.failure) {
132
+ const failure = entry.failure
133
+ if (!failure || typeof failure !== 'object') push(`${path}.failure: must be an object`)
134
+ else if (
135
+ typeof (/** @type {Record<string, unknown>} */ (failure).message) !== 'string' ||
136
+ /** @type {Record<string, unknown>} */ (failure).message === ''
137
+ )
138
+ push(`${path}.failure.message: must be a non-empty string`)
139
+ if (!FAILURE_CLASSES.includes(/** @type {string} */ (entry.failureClass)))
140
+ push(`${path}.failureClass: must be harness or task when a failure is present`)
141
+ } else if (entry.failureClass !== undefined) {
142
+ if (!FAILURE_CLASSES.includes(/** @type {string} */ (entry.failureClass)))
143
+ push(`${path}.failureClass: must be harness or task when present`)
144
+ }
145
+ if (Array.isArray(entry.steps)) {
146
+ for (const [stepIndex, stepEntry] of entry.steps.entries()) {
147
+ const stepPath = `${path}.steps[${stepIndex}]`
148
+ if (!stepEntry || typeof stepEntry !== 'object') {
149
+ push(`${stepPath}: must be an object`)
150
+ continue
151
+ }
152
+ const step = /** @type {Record<string, unknown>} */ (stepEntry)
153
+ if (typeof step.tool !== 'string' || step.tool.trim().length === 0)
154
+ push(`${stepPath}.tool: must be a non-empty string`)
155
+ if (!['running', 'passed', 'failed'].includes(/** @type {string} */ (step.status)))
156
+ push(`${stepPath}.status: must be running, passed, or failed`)
157
+ if (step.durationMs !== undefined && !isFiniteNumber(step.durationMs))
158
+ push(`${stepPath}.durationMs: must be a finite number`)
159
+ }
160
+ }
161
+ }
162
+ }
163
+ return { valid: errors.length === 0, errors }
164
+ }
165
+
166
+ /**
167
+ * Evaluate the configured budget thresholds against one validated report.
168
+ * Unknown budget keys fail closed so a typo can never silently disable a gate.
169
+ * @param {{summary?: Record<string, unknown>}} report
170
+ * @param {Record<string, unknown>} budgets
171
+ * @returns {{passed: boolean, failures: string[]}}
172
+ */
173
+ export function evaluateEvalBudgets(report, budgets) {
174
+ if (!report || typeof report !== 'object' || Array.isArray(report))
175
+ return { passed: false, failures: ['report must be a validated eval report object'] }
176
+ const failures = []
177
+ if (!budgets || typeof budgets !== 'object' || Array.isArray(budgets)) {
178
+ return { passed: false, failures: ['budgets must be a JSON object'] }
179
+ }
180
+ const known = new Set([
181
+ 'successRate',
182
+ 'taskP95Ms',
183
+ 'startupP95Ms',
184
+ 'stepP95Ms',
185
+ 'maxHarnessFailures',
186
+ 'maxEvidenceErrors',
187
+ 'maxToolCalls',
188
+ ])
189
+ for (const key of Object.keys(budgets)) {
190
+ if (!known.has(key)) failures.push(`budgets.${key}: unknown budget key`)
191
+ }
192
+ if (failures.length > 0) return { passed: false, failures }
193
+ /** @type {Record<string, unknown>} */
194
+ const summary = report.summary ?? {}
195
+ const rate = budgets.successRate
196
+ if (rate !== undefined) {
197
+ if (!isFiniteNumber(rate) || rate < 0 || rate > 1)
198
+ return { passed: false, failures: ['budgets.successRate: must be between 0 and 1'] }
199
+ if (isFiniteNumber(summary.successRate) && summary.successRate < rate)
200
+ failures.push(`successRate ${summary.successRate.toFixed(2)} is below budget ${rate}`)
201
+ }
202
+ const p95Fields = /** @type {const} */ ([
203
+ ['taskP95Ms', 'taskDurationMs'],
204
+ ['startupP95Ms', 'startupMs'],
205
+ ['stepP95Ms', 'stepDurationMs'],
206
+ ])
207
+ for (const [budgetField, summaryField] of p95Fields) {
208
+ const budget = /** @type {unknown} */ (budgets[budgetField])
209
+ if (budget === undefined) continue
210
+ if (!isFiniteNumber(budget) || budget <= 0)
211
+ return { passed: false, failures: [`budgets.${budgetField}: must be a positive number`] }
212
+ const stats = summary[summaryField]
213
+ if (!stats || typeof stats !== 'object') continue
214
+ const p95 = /** @type {unknown} */ (/** @type {Record<string, unknown>} */ (stats).p95)
215
+ const count = /** @type {Record<string, unknown>} */ (stats).count
216
+ if (!isFiniteNumber(p95) || count === 0) continue
217
+ if (p95 > budget) failures.push(`${summaryField}.p95 ${p95}ms is above budget ${budget}ms`)
218
+ }
219
+ const maxFields = /** @type {const} */ ([
220
+ ['maxHarnessFailures', 'harnessFailures'],
221
+ ['maxEvidenceErrors', 'evidenceErrors'],
222
+ ['maxToolCalls', 'toolCalls'],
223
+ ])
224
+ for (const [budgetField, summaryField] of maxFields) {
225
+ const budget = /** @type {unknown} */ (budgets[budgetField])
226
+ if (budget === undefined) continue
227
+ if (!isNonNegativeInteger(budget))
228
+ return { passed: false, failures: [`budgets.${budgetField}: must be a non-negative integer`] }
229
+ const measured = /** @type {unknown} */ (summary[summaryField])
230
+ if (isNonNegativeInteger(measured) && measured > budget)
231
+ failures.push(`${summaryField} ${measured} is above budget ${budget}`)
232
+ }
233
+ return { passed: failures.length === 0, failures }
234
+ }
@@ -277,16 +277,20 @@ export function verifiedPullRequest(pr, required) {
277
277
  return false
278
278
  const checks = pr.statusCheckRollup
279
279
  if (!Array.isArray(checks) || !checks.length) return false
280
- /** @param {any} check */
281
- const state = (check) =>
282
- check.status === 'COMPLETED' ? check.conclusion : (check.state ?? check.status)
283
- if (checks.some((check) => !['SUCCESS', 'NEUTRAL', 'SKIPPED'].includes(state(check))))
280
+ if (checks.some((check) => !['SUCCESS', 'NEUTRAL', 'SKIPPED'].includes(checkState(check))))
284
281
  return false
285
282
  return required.every((name) =>
286
- checks.some((check) => (check.name ?? check.context) === name && state(check) === 'SUCCESS')
283
+ checks.some(
284
+ (check) => (check.name ?? check.context) === name && checkState(check) === 'SUCCESS'
285
+ )
287
286
  )
288
287
  }
289
288
 
289
+ /** @param {any} check */
290
+ function checkState(check) {
291
+ return check.status === 'COMPLETED' ? check.conclusion : (check.state ?? check.status)
292
+ }
293
+
290
294
  /** @param {string} text */
291
295
  function scalarConfig(text) {
292
296
  return Object.fromEntries(
@@ -153,6 +153,7 @@ jobs:
153
153
  unit-runner: ${runner('unit_runner', 'ubuntu-slim')}
154
154
  performance-runner: ${runner('performance_runner')}
155
155
  security-runner: ${runner('security_runner', 'ubuntu-slim')}
156
+ eval-runner: ${runner('eval_runner')}
156
157
  codeql-runner: ${runner('codeql_runner')}
157
158
  rust-shards: '${JSON.stringify(shards)}'
158
159
  rust-threads: '${threads}'
@@ -16,7 +16,7 @@ export function customWorkflowFiles(root, standardFiles) {
16
16
  .filter((file) => file.endsWith('.yml') || file.endsWith('.yaml'))
17
17
  .map((file) => `.github/workflows/${file}`)
18
18
  .filter((file) => !isManagedPath(standardFiles, file))
19
- .sort()
19
+ .toSorted()
20
20
  }
21
21
 
22
22
  /** @param {string} root @param {Record<string, string>} config */
@@ -61,7 +61,7 @@ export function detectReleasePackages(root) {
61
61
  })
62
62
  }
63
63
  })
64
- return packages.sort((a, b) => a.directory.localeCompare(b.directory))
64
+ return packages.toSorted((a, b) => a.directory.localeCompare(b.directory))
65
65
  }
66
66
 
67
67
  /**
@@ -295,8 +295,8 @@ export function buildReleaseRecoveryPlan(input) {
295
295
  const orphanGitHubReleases = releaseTags.filter(
296
296
  (tag) => !tagSet.has(tag) && !tagSet.has(tag.replace(/^v/, ''))
297
297
  )
298
- const latestTag = [...tags].sort(compareVersions).at(-1) ?? ''
299
- const latestPackageVersion = [...input.packageVersions].sort(compareVersions).at(-1) ?? ''
298
+ const latestTag = [...tags].toSorted(compareVersions).at(-1) ?? ''
299
+ const latestPackageVersion = [...input.packageVersions].toSorted(compareVersions).at(-1) ?? ''
300
300
  return {
301
301
  latestTag,
302
302
  latestPackageVersion,
@@ -15,6 +15,7 @@ export const TASKS = Object.freeze([
15
15
  'integration',
16
16
  'e2e',
17
17
  'smoke',
18
+ 'eval',
18
19
  'performance',
19
20
  ])
20
21
 
@@ -28,6 +29,7 @@ export const TASK_SCRIPTS = Object.freeze({
28
29
  integration: ['test:integration'],
29
30
  e2e: ['test:e2e', 'e2e'],
30
31
  smoke: ['test:smoke', 'smoke'],
32
+ eval: ['eval'],
31
33
  performance: ['performance:check', 'perf:check'],
32
34
  })
33
35
 
@@ -41,6 +43,8 @@ export function readTaskPolicy(root) {
41
43
  }
42
44
  if (!['true', 'false', 'auto'].includes(config.performance ?? 'auto'))
43
45
  throw new Error('performance must be true, false, or auto')
46
+ if (!['true', 'false', 'auto'].includes(config.eval ?? 'auto'))
47
+ throw new Error('eval must be true, false, or auto')
44
48
  const coverageMode = config.coverage_enforcement ?? 'auto'
45
49
  if (!['auto', 'required', 'off'].includes(coverageMode))
46
50
  throw new Error('coverage_enforcement must be auto, required, or off')
@@ -50,6 +54,9 @@ export function readTaskPolicy(root) {
50
54
  throw new Error('Required performance cannot use performance: false')
51
55
  if (config.performance === 'true' && !required.includes('performance'))
52
56
  required.push('performance')
57
+ if (required.includes('eval') && config.eval === 'false')
58
+ throw new Error('Required eval cannot use eval: false')
59
+ if (config.eval === 'true' && !required.includes('eval')) required.push('eval')
53
60
  const coverageRequired = required.includes('coverage') || coverageMode === 'required'
54
61
  if (coverageRequired && !required.includes('unit')) required.push('unit')
55
62
  const minimum = Number(config.coverage_minimum ?? '80')
@@ -22,7 +22,7 @@ export const AGGREGATE_CHECK_NAME = 'Validation / Gate'
22
22
  /** Job ids owned by the validation orchestrator. Release runs the full audit
23
23
  * suite and validates its generated diff inside the gate, so no separate
24
24
  * release-policy job id exists. */
25
- export const VALIDATION_JOBS = ['ci', 'test', 'security', 'codeql']
25
+ export const VALIDATION_JOBS = ['ci', 'test', 'security', 'codeql', 'eval']
26
26
 
27
27
  /** Events that may trigger canonical validation. */
28
28
  export const VALIDATION_EVENTS = ['pull_request', 'schedule', 'workflow_dispatch']
@@ -81,11 +81,15 @@ export function classifyValidationMode(input) {
81
81
  /** @type {Record<'fast'|'audit'|'release', string[]>} */
82
82
  const REQUIRED_JOBS_BY_MODE = {
83
83
  fast: ['ci', 'test'],
84
- audit: ['ci', 'test', 'security', 'codeql'],
85
- // Release Please pull requests run the full audit suite so every registered
86
- // validation check succeeds rather than appearing as an expected skip. The
87
- // gate additionally validates the generated release diff.
88
- release: ['ci', 'test', 'security', 'codeql'],
84
+ audit: ['ci', 'test', 'security', 'codeql', 'eval'],
85
+ // Release Please pull requests change version metadata only, and the gate
86
+ // validates that diff against the release policy before anything publishes.
87
+ // The content tree was already fully audited by the pull requests that
88
+ // merged into main, and the scheduled audit lane re-covers drift. The
89
+ // release tier therefore requires the fast suite plus CodeQL, whose
90
+ // per-pull-request code scanning results some repository rulesets require
91
+ // to keep the release from deadlocking.
92
+ release: ['ci', 'test', 'codeql'],
89
93
  }
90
94
 
91
95
  /**
@@ -7,9 +7,17 @@ import { spawnSync } from 'node:child_process'
7
7
  import { detectPackageManager, resolveProfile } from './lib/profile.mjs'
8
8
  import { configured, readConfig } from './lib/config.mjs'
9
9
  import { classifyTestFiles } from './lib/test-discovery.mjs'
10
+ import { docsOnlyPullRequest } from './lib/docs-only.mjs'
10
11
  import { classifyValidationMode, evaluateValidationGate } from './lib/validation-policy.mjs'
11
12
  import { readReleaseConfig, validateGeneratedReleaseDiff } from './lib/release-policy.mjs'
12
13
  import { runNodePackagePerformance } from './lib/node-package-performance.mjs'
14
+ import {
15
+ EVAL_BUDGET_FILE_DEFAULT,
16
+ EVAL_REPORT_FILE,
17
+ EVAL_SUMMARY_FILE,
18
+ evaluateEvalBudgets,
19
+ validateEvalReport,
20
+ } from './lib/eval-envelope.mjs'
13
21
 
14
22
  const root = process.cwd()
15
23
  const config = readConfig(resolve(root, '.github/code-foundry.yml'))
@@ -57,6 +65,26 @@ function performanceEnabled() {
57
65
  return configured(config.performance, 'auto') !== 'false'
58
66
  }
59
67
 
68
+ function evalEnabled() {
69
+ return configured(config.eval, 'auto') !== 'false'
70
+ }
71
+
72
+ function evalCommand() {
73
+ const raw = configured(config.eval_command, '').trim()
74
+ if (!raw) return null
75
+ let command
76
+ try {
77
+ command = JSON.parse(raw)
78
+ } catch {
79
+ throw new Error('eval_command must be a JSON argv array.')
80
+ }
81
+ if (!Array.isArray(command) || command.length === 0)
82
+ throw new Error('eval_command must be a non-empty JSON array.')
83
+ if (!command.every((argument) => typeof argument === 'string' && argument.length > 0))
84
+ throw new Error('eval_command must contain only non-empty strings.')
85
+ return /** @type {string[]} */ (command)
86
+ }
87
+
60
88
  function performanceCommands() {
61
89
  const raw = configured(config.performance_command, '').trim()
62
90
  if (!raw) return []
@@ -128,6 +156,123 @@ function writePerformanceSummary(startedAt, status, commands, artifacts, error =
128
156
  )
129
157
  }
130
158
 
159
+ const evalResultsDirectory = 'eval-results'
160
+
161
+ /** @returns {{source: string, argv: string[]}[]} */
162
+ function selectedEvalCommands() {
163
+ const name = ['eval'].find((candidate) => hasScript(candidate))
164
+ if (name) {
165
+ const [manager, args] = packageCommand(['run', name])
166
+ if (!manager) throw new Error(`Cannot run ${name}: select a supported package_manager.`)
167
+ return [{ source: `package-script:${name}`, argv: [manager, ...args] }]
168
+ }
169
+ const command = evalCommand()
170
+ if (!command) throw new Error('No eval script or eval_command was discovered.')
171
+ return [{ source: 'configuration', argv: command }]
172
+ }
173
+
174
+ /** @param {string} startedAt @param {'passed'|'failed'} status @param {{source: string, argv: string[], status: number}[]} commands @param {string[]} artifacts @param {string|null} error @param {{file: string, applied: boolean, failures: string[]}|null} budgets */
175
+ function writeEvalSummary(startedAt, status, commands, artifacts, error = null, budgets = null) {
176
+ const directory = resolve(root, evalResultsDirectory)
177
+ mkdirSync(directory, { recursive: true })
178
+ writeFileSync(
179
+ resolve(directory, 'summary.json'),
180
+ `${JSON.stringify(
181
+ {
182
+ schemaVersion: 1,
183
+ kind: 'code-foundry-eval-summary',
184
+ status,
185
+ startedAt,
186
+ completedAt: new Date().toISOString(),
187
+ commands,
188
+ report: configured(config.eval_report_file, EVAL_REPORT_FILE),
189
+ budgets,
190
+ artifacts,
191
+ error,
192
+ },
193
+ null,
194
+ 2
195
+ )}\n`
196
+ )
197
+ }
198
+
199
+ function runEval() {
200
+ if (!evalEnabled()) return
201
+ const startedAt = new Date().toISOString()
202
+ /** @type {{source: string, argv: string[], status: number}[]} */
203
+ const records = []
204
+ const artifacts = [EVAL_REPORT_FILE, EVAL_SUMMARY_FILE]
205
+ const budgetFile = configured(config.eval_budget_file, EVAL_BUDGET_FILE_DEFAULT)
206
+ /** @type {{file: string, applied: boolean, failures: string[]}|null} */
207
+ let budgets = null
208
+ try {
209
+ for (const command of selectedEvalCommands()) {
210
+ const result = spawnSync(command.argv[0], command.argv.slice(1), {
211
+ cwd: root,
212
+ stdio: 'inherit',
213
+ env: process.env,
214
+ })
215
+ if (result.error) throw result.error
216
+ const status = result.status ?? 1
217
+ records.push({ ...command, status })
218
+ if (status !== 0) {
219
+ writeEvalSummary(
220
+ startedAt,
221
+ 'failed',
222
+ records,
223
+ artifacts,
224
+ `command exited ${status}`,
225
+ budgets
226
+ )
227
+ process.exitCode = status
228
+ return
229
+ }
230
+ }
231
+ const reportFile = resolve(root, configured(config.eval_report_file, EVAL_REPORT_FILE))
232
+ if (!existsSync(reportFile))
233
+ throw new Error(
234
+ `Eval report was not produced: ${configured(config.eval_report_file, EVAL_REPORT_FILE)}`
235
+ )
236
+ let report
237
+ try {
238
+ report = JSON.parse(readFileSync(reportFile, 'utf8'))
239
+ } catch (error) {
240
+ throw new Error(
241
+ `Eval report is not valid JSON: ${error instanceof Error ? error.message : String(error)}`,
242
+ { cause: error }
243
+ )
244
+ }
245
+ const envelope = validateEvalReport(report)
246
+ if (!envelope.valid)
247
+ throw new Error(`Eval report violates the contract: ${envelope.errors.join('; ')}`)
248
+ if (existsSync(resolve(root, budgetFile))) {
249
+ const raw = JSON.parse(readFileSync(resolve(root, budgetFile), 'utf8'))
250
+ const gate = evaluateEvalBudgets(report, raw)
251
+ budgets = { file: budgetFile, applied: true, failures: gate.failures }
252
+ if (!gate.passed) {
253
+ for (const failure of gate.failures) console.error(`::error::${failure}`)
254
+ writeEvalSummary(
255
+ startedAt,
256
+ 'failed',
257
+ records,
258
+ artifacts,
259
+ `eval budgets failed: ${gate.failures.join('; ')}`,
260
+ budgets
261
+ )
262
+ process.exitCode = 1
263
+ return
264
+ }
265
+ } else {
266
+ budgets = { file: budgetFile, applied: false, failures: [] }
267
+ }
268
+ writeEvalSummary(startedAt, 'passed', records, artifacts, null, budgets)
269
+ } catch (error) {
270
+ const message = error instanceof Error ? error.message : String(error)
271
+ writeEvalSummary(startedAt, 'failed', records, artifacts, message, budgets)
272
+ throw error
273
+ }
274
+ }
275
+
131
276
  function runPerformance() {
132
277
  if (!performanceEnabled()) return
133
278
  const startedAt = new Date().toISOString()
@@ -172,17 +317,37 @@ function runPerformance() {
172
317
  }
173
318
  }
174
319
 
320
+ /**
321
+ * Audit-mode pull requests whose diff is docs-only run the fast tier instead.
322
+ * Scheduled and manual audits always keep the full tier; release pull requests
323
+ * are classified separately and always run their lane.
324
+ * @param {string} mode
325
+ */
326
+ function downgradeDocsOnlyPullRequest(mode) {
327
+ if (mode !== 'audit') return mode
328
+ if ((process.env.FOUNDRY_EVENT_NAME ?? '') !== 'pull_request') return mode
329
+ const baseRef = process.env.FOUNDRY_BASE_REF ?? ''
330
+ if (!baseRef) return mode
331
+ try {
332
+ return docsOnlyPullRequest(root, baseRef) ? 'fast' : mode
333
+ } catch {
334
+ return mode
335
+ }
336
+ }
337
+
175
338
  /** @param {string} task */
176
339
  function validation(task) {
177
340
  if (task === 'mode') {
178
341
  writeOutput(
179
342
  'mode',
180
- classifyValidationMode({
181
- eventName: process.env.FOUNDRY_EVENT_NAME ?? '',
182
- baseRef: process.env.FOUNDRY_BASE_REF ?? '',
183
- headRef: process.env.FOUNDRY_HEAD_REF ?? '',
184
- stagingMode: config.staging_validation_mode,
185
- })
343
+ downgradeDocsOnlyPullRequest(
344
+ classifyValidationMode({
345
+ eventName: process.env.FOUNDRY_EVENT_NAME ?? '',
346
+ baseRef: process.env.FOUNDRY_BASE_REF ?? '',
347
+ headRef: process.env.FOUNDRY_HEAD_REF ?? '',
348
+ stagingMode: config.staging_validation_mode,
349
+ })
350
+ )
186
351
  )
187
352
  return
188
353
  }
@@ -195,6 +360,7 @@ function validation(task) {
195
360
  test: process.env.FOUNDRY_TEST,
196
361
  security: process.env.FOUNDRY_SECURITY,
197
362
  codeql: process.env.FOUNDRY_CODEQL,
363
+ eval: process.env.FOUNDRY_EVAL,
198
364
  },
199
365
  })
200
366
  if (gate.valid) {
@@ -391,8 +557,15 @@ function relevant(task) {
391
557
  integration: ['test:integration'],
392
558
  e2e: ['test:e2e', 'e2e'],
393
559
  smoke: ['test:smoke', 'smoke'],
560
+ eval: ['eval'],
394
561
  performance: ['performance:check', 'perf:check'],
395
562
  }[task]
563
+ if (task === 'eval') {
564
+ if (!evalEnabled()) return false
565
+ return Boolean(
566
+ (scripted && scripted.some((candidate) => hasScript(candidate))) || evalCommand()
567
+ )
568
+ }
396
569
  if (task === 'performance') {
397
570
  if (!performanceEnabled()) return false
398
571
  return Boolean(
@@ -541,10 +714,11 @@ function ci(task) {
541
714
  hasRootJavascriptProject() &&
542
715
  hasOxlintSetup()
543
716
  ) {
544
- // Oxlint is the baseline's linter. Repositories that use a different
545
- // linter keep full control through their own `lint` script, which the
546
- // runScript fallback above already honors.
547
- runTool('oxlint', [])
717
+ // Oxlint is the baseline's linter. Warnings are denied so lint debt
718
+ // never accumulates silently. Repositories that use a different linter
719
+ // keep full control through their own `lint` script, which the runScript
720
+ // fallback above already honors.
721
+ runTool('oxlint', ['--deny-warnings'])
548
722
  }
549
723
  if (hasLanguage('python') && hasRootPythonProject()) runTool('ruff', ['check', '.'])
550
724
  if (hasLanguage('rust') && hasRootRustProject())
@@ -563,6 +737,9 @@ function ci(task) {
563
737
  run('cargo', ['build', '--all-targets'])
564
738
  return
565
739
  }
740
+ if (task === 'eval') {
741
+ return runEval()
742
+ }
566
743
  if (task === 'performance') {
567
744
  return runPerformance()
568
745
  }
package/src/runtime.mjs CHANGED
@@ -186,6 +186,8 @@ export function runRuntime(args, root = process.cwd(), entry = core) {
186
186
  report.coverage = evaluateCoverage(root, policy, before)
187
187
  report.artifacts.push(...report.coverage.artifacts)
188
188
  }
189
+ if (task === 'eval')
190
+ report.artifacts.push('eval-results/summary.json', 'eval-results/result.json')
189
191
  if (task === 'performance') report.artifacts.push('performance-results/summary.json')
190
192
  report.status = 'passed'
191
193
  return 0
@@ -41,6 +41,8 @@ htmlcov/
41
41
  artifacts/
42
42
  performance-results.json
43
43
  performance-results/
44
+ eval-results.json
45
+ eval-results/
44
46
  cache/
45
47
  !.github/actions/cache/
46
48
  !.github/actions/cache/action.yml