code-foundry 1.20.2 → 1.25.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/.github/CONTRIBUTING.md +5 -2
  2. package/.github/code-foundry.yml +1 -0
  3. package/.github/release-please-foundry.json +56 -0
  4. package/.github/workflows/cloudflare-delivery.yml +9 -0
  5. package/.github/workflows/cloudflare-deploy.yml +8 -1
  6. package/.github/workflows/eval.yml +116 -0
  7. package/.github/workflows/opencode-security_self-ci.yml +1 -1
  8. package/.github/workflows/qualified-foundry-publish.yml +6 -7
  9. package/.github/workflows/release.yml +58 -10
  10. package/.github/workflows/release_self-ci.yml +211 -8
  11. package/.github/workflows/validation-no-codeql.yml +34 -6
  12. package/.github/workflows/validation.yml +32 -5
  13. package/.github/workflows/validation_audit_self-ci.yml +1 -0
  14. package/.github/workflows/validation_self-ci.yml +1 -0
  15. package/.gitignore +1 -1
  16. package/AGENTS.md +5 -1
  17. package/CHANGELOG.md +95 -0
  18. package/README.md +23 -17
  19. package/docs/CONFIGURATION.md +189 -152
  20. package/docs/EVALS.md +139 -0
  21. package/docs/EXTENSIONS.md +28 -7
  22. package/docs/INITIALIZATION.md +13 -9
  23. package/docs/PERFORMANCE.md +67 -59
  24. package/docs/PUBLISHING.md +39 -14
  25. package/docs/README.md +37 -22
  26. package/docs/RELEASES.md +18 -9
  27. package/docs/WORKFLOWS.md +23 -4
  28. package/docs/agent-validation.md +6 -5
  29. package/docs/cloudflare-delivery.md +20 -8
  30. package/docs/consumer-qualification.md +11 -9
  31. package/docs/fleet-release-eligibility.md +14 -15
  32. package/docs/fleet-rollouts.md +3 -3
  33. package/docs/merge-queues.md +21 -21
  34. package/docs/product-quality.md +9 -10
  35. package/docs/qualified-publication.md +166 -93
  36. package/docs/release-integrity.md +7 -6
  37. package/docs/required-capabilities.md +24 -15
  38. package/package.json +1 -1
  39. package/src/commands/cloudflare-delivery.mjs +6 -1
  40. package/src/commands/qualified-publication.mjs +103 -12
  41. package/src/commands/release-integrity.mjs +46 -14
  42. package/src/commands/sync.mjs +30 -24
  43. package/src/lib/eval-envelope.mjs +234 -0
  44. package/src/lib/merge-queue.mjs +1 -0
  45. package/src/lib/product-quality.mjs +220 -16
  46. package/src/lib/task-policy.mjs +7 -0
  47. package/src/lib/validation-policy.mjs +10 -6
  48. package/src/runtime-core.mjs +166 -8
  49. package/src/runtime.mjs +3 -1
  50. package/src/templates/gitignore +3 -1
@@ -81,21 +81,226 @@ function decode(value) {
81
81
  })
82
82
  }
83
83
 
84
+ /**
85
+ * Find an HTML comment terminator, including the parser's comment-end-bang
86
+ * form. An unterminated comment is treated as extending to the end of input.
87
+ * @param {string} html
88
+ * @param {number} start
89
+ * @returns {{ start: number, length: number } | null}
90
+ */
91
+ function findCommentEnd(html, start) {
92
+ const normal = html.indexOf('-->', start + 4)
93
+ const bang = html.indexOf('--!>', start + 4)
94
+ if (normal === -1 && bang === -1) return null
95
+ if (bang !== -1 && (normal === -1 || bang < normal)) return { start: bang, length: 4 }
96
+ return { start: normal, length: 3 }
97
+ }
98
+
99
+ /**
100
+ * Remove comments without leaving an unmatched comment opener that can expose
101
+ * markup to the lightweight tag reader.
102
+ * @param {string} html
103
+ */
104
+ function stripHtmlComments(html) {
105
+ let clean = ''
106
+ let cursor = 0
107
+ while (cursor < html.length) {
108
+ const start = html.indexOf('<!--', cursor)
109
+ if (start === -1) return clean + html.slice(cursor)
110
+ clean += html.slice(cursor, start)
111
+ const end = findCommentEnd(html, start)
112
+ if (!end) return clean
113
+ cursor = end.start + end.length
114
+ }
115
+ return clean
116
+ }
117
+
118
+ /** @param {string} html @param {number} start */
119
+ function findTagEnd(html, start) {
120
+ let quote = ''
121
+ for (let index = start + 1; index < html.length; index += 1) {
122
+ const character = html[index]
123
+ if (quote) {
124
+ if (character === quote) quote = ''
125
+ } else if (character === '"' || character === "'") {
126
+ quote = character
127
+ } else if (character === '>') {
128
+ return index
129
+ }
130
+ }
131
+ return -1
132
+ }
133
+
134
+ /** @param {string | undefined} character */
135
+ function isTagBoundary(character) {
136
+ return (
137
+ character === undefined || character === '>' || character === '/' || character.trim() === ''
138
+ )
139
+ }
140
+
141
+ /**
142
+ * Locate a named HTML tag while accepting parser-tolerated end-tag junk such
143
+ * as `</script\\t\\n data>`.
144
+ * @param {string} html
145
+ * @param {string} name
146
+ * @param {number} from
147
+ * @param {boolean} closing
148
+ * @returns {{ start: number, end: number } | null}
149
+ */
150
+ function findNamedTag(html, name, from, closing) {
151
+ const lowerName = name.toLowerCase()
152
+ for (let start = html.indexOf('<', from); start !== -1; start = html.indexOf('<', start + 1)) {
153
+ let nameStart = start + 1
154
+ if (closing) {
155
+ if (html[nameStart] !== '/') continue
156
+ nameStart += 1
157
+ } else if (html[nameStart] === '/' || html[nameStart] === '!' || html[nameStart] === '?') {
158
+ continue
159
+ }
160
+ if (html.slice(nameStart, nameStart + name.length).toLowerCase() !== lowerName) continue
161
+ if (!isTagBoundary(html[nameStart + name.length])) continue
162
+ return { start, end: findTagEnd(html, start) }
163
+ }
164
+ return null
165
+ }
166
+
167
+ const RAW_TEXT_TAGS = [
168
+ 'script',
169
+ 'style',
170
+ 'textarea',
171
+ 'template',
172
+ 'xmp',
173
+ 'iframe',
174
+ 'noembed',
175
+ 'noframes',
176
+ 'noscript',
177
+ ]
178
+
179
+ /**
180
+ * @param {string} html
181
+ * @param {number} from
182
+ * @returns {{ start: number, end: number, name: string } | null}
183
+ */
184
+ function findNextRawTextTag(html, from) {
185
+ let next = null
186
+ for (const name of RAW_TEXT_TAGS) {
187
+ const candidate = findNamedTag(html, name, from, false)
188
+ if (candidate && (!next || candidate.start < next.start)) {
189
+ next = { ...candidate, name }
190
+ }
191
+ }
192
+ return next
193
+ }
194
+
195
+ /**
196
+ * Remove comments and raw-text contents in document order. Comment markers
197
+ * inside raw-text elements are data, not document comments.
198
+ * @param {string} html
199
+ */
200
+ function stripCommentsAndRawText(html) {
201
+ let clean = ''
202
+ let cursor = 0
203
+ while (cursor < html.length) {
204
+ const commentStart = html.indexOf('<!--', cursor)
205
+ const rawText = findNextRawTextTag(html, cursor)
206
+ if (commentStart === -1 && !rawText) return clean + html.slice(cursor)
207
+ if (commentStart !== -1 && (!rawText || commentStart < rawText.start)) {
208
+ clean += html.slice(cursor, commentStart)
209
+ const commentEnd = findCommentEnd(html, commentStart)
210
+ if (!commentEnd) return clean
211
+ cursor = commentEnd.start + commentEnd.length
212
+ continue
213
+ }
214
+ if (!rawText) return clean + html.slice(cursor)
215
+ clean += html.slice(cursor, rawText.start)
216
+ if (rawText.end === -1) return clean
217
+ clean += html.slice(rawText.start, rawText.end + 1)
218
+ const close = findNamedTag(html, rawText.name, rawText.end + 1, true)
219
+ if (!close || close.end === -1) return clean
220
+ cursor = close.end + 1
221
+ }
222
+ return clean
223
+ }
224
+
225
+ /**
226
+ * Remove angle-bracket markup from title text. A dangling `<` is discarded
227
+ * through EOF instead of being left as a potentially active tag prefix.
228
+ * @param {string} value
229
+ */
230
+ function stripAngleBracketMarkup(value) {
231
+ let clean = ''
232
+ let cursor = 0
233
+ while (cursor < value.length) {
234
+ const start = value.indexOf('<', cursor)
235
+ if (start === -1) return clean + value.slice(cursor)
236
+ clean += value.slice(cursor, start)
237
+ const end = value.indexOf('>', start + 1)
238
+ if (end === -1) return clean
239
+ cursor = end + 1
240
+ }
241
+ return clean
242
+ }
243
+
244
+ /**
245
+ * Rewrite title bodies without relying on a multi-character HTML-filtering
246
+ * regexp. Malformed title elements are truncated so their contents cannot be
247
+ * interpreted as page metadata.
248
+ * @param {string} html
249
+ */
250
+ function rewriteTitleMarkup(html) {
251
+ let clean = ''
252
+ let cursor = 0
253
+ while (cursor < html.length) {
254
+ const open = findNamedTag(html, 'title', cursor, false)
255
+ if (!open) return clean + html.slice(cursor)
256
+ clean += html.slice(cursor, open.start)
257
+ if (open.end === -1) return clean
258
+ clean += html.slice(open.start, open.end + 1)
259
+ const close = findNamedTag(html, 'title', open.end + 1, true)
260
+ if (!close || close.end === -1) return clean
261
+ clean += stripAngleBracketMarkup(html.slice(open.end + 1, close.start))
262
+ clean += html.slice(close.start, close.end + 1)
263
+ cursor = close.end + 1
264
+ }
265
+ return clean
266
+ }
267
+
268
+ /**
269
+ * Return script bodies while respecting comments and parser-tolerated script
270
+ * end tags. An unclosed script consumes the remainder of the document.
271
+ * @param {string} html
272
+ */
273
+ function readInlineScripts(html) {
274
+ /** @type {string[]} */
275
+ const scripts = []
276
+ let cursor = 0
277
+ while (cursor < html.length) {
278
+ const script = findNamedTag(html, 'script', cursor, false)
279
+ const commentStart = html.indexOf('<!--', cursor)
280
+ if (commentStart !== -1 && (!script || commentStart < script.start)) {
281
+ const commentEnd = findCommentEnd(html, commentStart)
282
+ if (!commentEnd) return scripts
283
+ cursor = commentEnd.start + commentEnd.length
284
+ continue
285
+ }
286
+ if (!script || script.end === -1) return scripts
287
+ const contentStart = script.end + 1
288
+ const candidateClose = findNamedTag(html, 'script', contentStart, true)
289
+ const close = candidateClose?.end === -1 ? null : candidateClose
290
+ const contentEnd = close?.start ?? html.length
291
+ scripts.push(html.slice(contentStart, contentEnd))
292
+ if (!close || close.end === -1) return scripts
293
+ cursor = close.end + 1
294
+ }
295
+ return scripts
296
+ }
297
+
84
298
  /** Conservative generated-HTML tag reader, not a DOM implementation.
85
299
  * Raw-text HTML elements and comments cannot create fake metadata or links.
86
300
  * @param {string} html
87
301
  */
88
302
  export function readTags(html) {
89
- const clean = html
90
- .replace(/<!--[\s\S]*?-->/g, '')
91
- .replace(
92
- /(<(script|style|textarea|template|xmp|iframe|noembed|noframes|noscript)\b(?:"[^"]*"|'[^']*'|[^'">])*>)[\s\S]*?<\/\2\s*>/gi,
93
- '$1'
94
- )
95
- .replace(
96
- /(<title\b(?:"[^"]*"|'[^']*'|[^'">])*>)([\s\S]*?)<\/title\s*>/gi,
97
- (_, open, content) => `${open}${content.replace(/<[^>]*>/g, '')}</title>`
98
- )
303
+ const clean = rewriteTitleMarkup(stripCommentsAndRawText(html))
99
304
  /** @type {Array<{ name: string, attrs: Record<string,string> }>} */
100
305
  const tags = []
101
306
  for (const match of clean.matchAll(/<([a-z][a-z0-9:-]*)\b((?:"[^"]*"|'[^']*'|[^'">])*)>/gi)) {
@@ -245,11 +450,10 @@ export function checkStaticSite(root, profile) {
245
450
  requireValue(lstatSync(file).isFile(), 'Budget assets must be regular files')
246
451
  return sum + statSync(file).size
247
452
  }, 0)
248
- const inlineBytes = [
249
- ...readFileSync(page.file, 'utf8')
250
- .replace(/<!--[\s\S]*?-->/g, '')
251
- .matchAll(/<script\b(?:"[^"]*"|'[^']*'|[^'">])*?>([\s\S]*?)<\/script\s*>/gi),
252
- ].reduce((sum, match) => sum + Buffer.byteLength(match[1]), 0)
453
+ const inlineBytes = readInlineScripts(readFileSync(page.file, 'utf8')).reduce(
454
+ (sum, script) => sum + Buffer.byteLength(script),
455
+ 0
456
+ )
253
457
  const result = {
254
458
  route: path,
255
459
  htmlBytes: statSync(page.file).size,
@@ -267,7 +471,7 @@ export function checkStaticSite(root, profile) {
267
471
  if (profile.sitemap) {
268
472
  const sitemap = readFileSync(ownedPath(dist, profile.sitemap), 'utf8')
269
473
  requireValue(!/<!DOCTYPE|<!ENTITY/i.test(sitemap), 'Sitemap entities are not supported')
270
- const sitemapWithoutComments = sitemap.replace(/<!--[\s\S]*?-->/g, '')
474
+ const sitemapWithoutComments = stripHtmlComments(sitemap)
271
475
  const locations = new Set(
272
476
  [
273
477
  ...sitemapWithoutComments.matchAll(
@@ -15,6 +15,7 @@ export const TASKS = Object.freeze([
15
15
  'integration',
16
16
  'e2e',
17
17
  'smoke',
18
+ 'eval',
18
19
  'performance',
19
20
  ])
20
21
 
@@ -28,6 +29,7 @@ export const TASK_SCRIPTS = Object.freeze({
28
29
  integration: ['test:integration'],
29
30
  e2e: ['test:e2e', 'e2e'],
30
31
  smoke: ['test:smoke', 'smoke'],
32
+ eval: ['eval'],
31
33
  performance: ['performance:check', 'perf:check'],
32
34
  })
33
35
 
@@ -41,6 +43,8 @@ export function readTaskPolicy(root) {
41
43
  }
42
44
  if (!['true', 'false', 'auto'].includes(config.performance ?? 'auto'))
43
45
  throw new Error('performance must be true, false, or auto')
46
+ if (!['true', 'false', 'auto'].includes(config.eval ?? 'auto'))
47
+ throw new Error('eval must be true, false, or auto')
44
48
  const coverageMode = config.coverage_enforcement ?? 'auto'
45
49
  if (!['auto', 'required', 'off'].includes(coverageMode))
46
50
  throw new Error('coverage_enforcement must be auto, required, or off')
@@ -50,6 +54,9 @@ export function readTaskPolicy(root) {
50
54
  throw new Error('Required performance cannot use performance: false')
51
55
  if (config.performance === 'true' && !required.includes('performance'))
52
56
  required.push('performance')
57
+ if (required.includes('eval') && config.eval === 'false')
58
+ throw new Error('Required eval cannot use eval: false')
59
+ if (config.eval === 'true' && !required.includes('eval')) required.push('eval')
53
60
  const coverageRequired = required.includes('coverage') || coverageMode === 'required'
54
61
  if (coverageRequired && !required.includes('unit')) required.push('unit')
55
62
  const minimum = Number(config.coverage_minimum ?? '80')
@@ -22,7 +22,7 @@ export const AGGREGATE_CHECK_NAME = 'Validation / Gate'
22
22
  /** Job ids owned by the validation orchestrator. Release runs the full audit
23
23
  * suite and validates its generated diff inside the gate, so no separate
24
24
  * release-policy job id exists. */
25
- export const VALIDATION_JOBS = ['ci', 'test', 'security', 'codeql']
25
+ export const VALIDATION_JOBS = ['ci', 'test', 'security', 'codeql', 'eval']
26
26
 
27
27
  /** Events that may trigger canonical validation. */
28
28
  export const VALIDATION_EVENTS = ['pull_request', 'schedule', 'workflow_dispatch']
@@ -81,11 +81,15 @@ export function classifyValidationMode(input) {
81
81
  /** @type {Record<'fast'|'audit'|'release', string[]>} */
82
82
  const REQUIRED_JOBS_BY_MODE = {
83
83
  fast: ['ci', 'test'],
84
- audit: ['ci', 'test', 'security', 'codeql'],
85
- // Release Please pull requests run the full audit suite so every registered
86
- // validation check succeeds rather than appearing as an expected skip. The
87
- // gate additionally validates the generated release diff.
88
- release: ['ci', 'test', 'security', 'codeql'],
84
+ audit: ['ci', 'test', 'security', 'codeql', 'eval'],
85
+ // Release Please pull requests change version metadata only, and the gate
86
+ // validates that diff against the release policy before anything publishes.
87
+ // The content tree was already fully audited by the pull requests that
88
+ // merged into main, and the scheduled audit lane re-covers drift. The
89
+ // release tier therefore requires the fast suite plus CodeQL, whose
90
+ // per-pull-request code scanning results some repository rulesets require
91
+ // to keep the release from deadlocking.
92
+ release: ['ci', 'test', 'codeql'],
89
93
  }
90
94
 
91
95
  /**
@@ -10,6 +10,13 @@ import { classifyTestFiles } from './lib/test-discovery.mjs'
10
10
  import { classifyValidationMode, evaluateValidationGate } from './lib/validation-policy.mjs'
11
11
  import { readReleaseConfig, validateGeneratedReleaseDiff } from './lib/release-policy.mjs'
12
12
  import { runNodePackagePerformance } from './lib/node-package-performance.mjs'
13
+ import {
14
+ EVAL_BUDGET_FILE_DEFAULT,
15
+ EVAL_REPORT_FILE,
16
+ EVAL_SUMMARY_FILE,
17
+ evaluateEvalBudgets,
18
+ validateEvalReport,
19
+ } from './lib/eval-envelope.mjs'
13
20
 
14
21
  const root = process.cwd()
15
22
  const config = readConfig(resolve(root, '.github/code-foundry.yml'))
@@ -57,6 +64,26 @@ function performanceEnabled() {
57
64
  return configured(config.performance, 'auto') !== 'false'
58
65
  }
59
66
 
67
+ function evalEnabled() {
68
+ return configured(config.eval, 'auto') !== 'false'
69
+ }
70
+
71
+ function evalCommand() {
72
+ const raw = configured(config.eval_command, '').trim()
73
+ if (!raw) return null
74
+ let command
75
+ try {
76
+ command = JSON.parse(raw)
77
+ } catch {
78
+ throw new Error('eval_command must be a JSON argv array.')
79
+ }
80
+ if (!Array.isArray(command) || command.length === 0)
81
+ throw new Error('eval_command must be a non-empty JSON array.')
82
+ if (!command.every((argument) => typeof argument === 'string' && argument.length > 0))
83
+ throw new Error('eval_command must contain only non-empty strings.')
84
+ return /** @type {string[]} */ (command)
85
+ }
86
+
60
87
  function performanceCommands() {
61
88
  const raw = configured(config.performance_command, '').trim()
62
89
  if (!raw) return []
@@ -98,7 +125,8 @@ function selectedPerformanceCommands() {
98
125
  const name = ['performance:check', 'perf:check'].find((candidate) => hasScript(candidate))
99
126
  if (name) {
100
127
  const [manager, args] = packageCommand(['run', name])
101
- return manager ? [{ source: `package-script:${name}`, argv: [manager, ...args] }] : []
128
+ if (!manager) throw new Error(`Cannot run ${name}: select a supported package_manager.`)
129
+ return [{ source: `package-script:${name}`, argv: [manager, ...args] }]
102
130
  }
103
131
  return performanceCommands().map((argv) => ({ source: 'configuration', argv }))
104
132
  }
@@ -127,6 +155,122 @@ function writePerformanceSummary(startedAt, status, commands, artifacts, error =
127
155
  )
128
156
  }
129
157
 
158
+ const evalResultsDirectory = 'eval-results'
159
+
160
+ /** @returns {{source: string, argv: string[]}[]} */
161
+ function selectedEvalCommands() {
162
+ const name = ['eval'].find((candidate) => hasScript(candidate))
163
+ if (name) {
164
+ const [manager, args] = packageCommand(['run', name])
165
+ if (!manager) throw new Error(`Cannot run ${name}: select a supported package_manager.`)
166
+ return [{ source: `package-script:${name}`, argv: [manager, ...args] }]
167
+ }
168
+ const command = evalCommand()
169
+ if (!command) throw new Error('No eval script or eval_command was discovered.')
170
+ return [{ source: 'configuration', argv: command }]
171
+ }
172
+
173
+ /** @param {string} startedAt @param {'passed'|'failed'} status @param {{source: string, argv: string[], status: number}[]} commands @param {string[]} artifacts @param {string|null} error @param {{file: string, applied: boolean, failures: string[]}|null} budgets */
174
+ function writeEvalSummary(startedAt, status, commands, artifacts, error = null, budgets = null) {
175
+ const directory = resolve(root, evalResultsDirectory)
176
+ mkdirSync(directory, { recursive: true })
177
+ writeFileSync(
178
+ resolve(directory, 'summary.json'),
179
+ `${JSON.stringify(
180
+ {
181
+ schemaVersion: 1,
182
+ kind: 'code-foundry-eval-summary',
183
+ status,
184
+ startedAt,
185
+ completedAt: new Date().toISOString(),
186
+ commands,
187
+ report: configured(config.eval_report_file, EVAL_REPORT_FILE),
188
+ budgets,
189
+ artifacts,
190
+ error,
191
+ },
192
+ null,
193
+ 2
194
+ )}\n`
195
+ )
196
+ }
197
+
198
+ function runEval() {
199
+ if (!evalEnabled()) return
200
+ const startedAt = new Date().toISOString()
201
+ /** @type {{source: string, argv: string[], status: number}[]} */
202
+ const records = []
203
+ const artifacts = [EVAL_REPORT_FILE, EVAL_SUMMARY_FILE]
204
+ const budgetFile = configured(config.eval_budget_file, EVAL_BUDGET_FILE_DEFAULT)
205
+ /** @type {{file: string, applied: boolean, failures: string[]}|null} */
206
+ let budgets = null
207
+ try {
208
+ for (const command of selectedEvalCommands()) {
209
+ const result = spawnSync(command.argv[0], command.argv.slice(1), {
210
+ cwd: root,
211
+ stdio: 'inherit',
212
+ env: process.env,
213
+ })
214
+ if (result.error) throw result.error
215
+ const status = result.status ?? 1
216
+ records.push({ ...command, status })
217
+ if (status !== 0) {
218
+ writeEvalSummary(
219
+ startedAt,
220
+ 'failed',
221
+ records,
222
+ artifacts,
223
+ `command exited ${status}`,
224
+ budgets
225
+ )
226
+ process.exitCode = status
227
+ return
228
+ }
229
+ }
230
+ const reportFile = resolve(root, configured(config.eval_report_file, EVAL_REPORT_FILE))
231
+ if (!existsSync(reportFile))
232
+ throw new Error(
233
+ `Eval report was not produced: ${configured(config.eval_report_file, EVAL_REPORT_FILE)}`
234
+ )
235
+ let report
236
+ try {
237
+ report = JSON.parse(readFileSync(reportFile, 'utf8'))
238
+ } catch (error) {
239
+ throw new Error(
240
+ `Eval report is not valid JSON: ${error instanceof Error ? error.message : String(error)}`
241
+ )
242
+ }
243
+ const envelope = validateEvalReport(report)
244
+ if (!envelope.valid)
245
+ throw new Error(`Eval report violates the contract: ${envelope.errors.join('; ')}`)
246
+ if (existsSync(resolve(root, budgetFile))) {
247
+ const raw = JSON.parse(readFileSync(resolve(root, budgetFile), 'utf8'))
248
+ const gate = evaluateEvalBudgets(report, raw)
249
+ budgets = { file: budgetFile, applied: true, failures: gate.failures }
250
+ if (!gate.passed) {
251
+ for (const failure of gate.failures) console.error(`::error::${failure}`)
252
+ writeEvalSummary(
253
+ startedAt,
254
+ 'failed',
255
+ records,
256
+ artifacts,
257
+ `eval budgets failed: ${gate.failures.join('; ')}`,
258
+ budgets
259
+ )
260
+ process.exitCode = 1
261
+ return
262
+ }
263
+ } else {
264
+ budgets = { file: budgetFile, applied: false, failures: [] }
265
+ }
266
+ writeEvalSummary(startedAt, 'passed', records, artifacts, null, budgets)
267
+ } catch (error) {
268
+ const message = error instanceof Error ? error.message : String(error)
269
+ writeEvalSummary(startedAt, 'failed', records, artifacts, message, budgets)
270
+ throw error
271
+ }
272
+ }
273
+
130
274
  function runPerformance() {
131
275
  if (!performanceEnabled()) return
132
276
  const startedAt = new Date().toISOString()
@@ -194,6 +338,7 @@ function validation(task) {
194
338
  test: process.env.FOUNDRY_TEST,
195
339
  security: process.env.FOUNDRY_SECURITY,
196
340
  codeql: process.env.FOUNDRY_CODEQL,
341
+ eval: process.env.FOUNDRY_EVAL,
197
342
  },
198
343
  })
199
344
  if (gate.valid) {
@@ -390,8 +535,15 @@ function relevant(task) {
390
535
  integration: ['test:integration'],
391
536
  e2e: ['test:e2e', 'e2e'],
392
537
  smoke: ['test:smoke', 'smoke'],
538
+ eval: ['eval'],
393
539
  performance: ['performance:check', 'perf:check'],
394
540
  }[task]
541
+ if (task === 'eval') {
542
+ if (!evalEnabled()) return false
543
+ return Boolean(
544
+ (scripted && scripted.some((candidate) => hasScript(candidate))) || evalCommand()
545
+ )
546
+ }
395
547
  if (task === 'performance') {
396
548
  if (!performanceEnabled()) return false
397
549
  return Boolean(
@@ -562,6 +714,9 @@ function ci(task) {
562
714
  run('cargo', ['build', '--all-targets'])
563
715
  return
564
716
  }
717
+ if (task === 'eval') {
718
+ return runEval()
719
+ }
565
720
  if (task === 'performance') {
566
721
  return runPerformance()
567
722
  }
@@ -600,15 +755,18 @@ function ci(task) {
600
755
 
601
756
  if (hasLanguage('rust') && hasRootRustProject()) {
602
757
  if (task === 'unit') {
603
- if (existsSync(resolve(root, 'src/lib.rs'))) run('cargo', ['test', '--lib'])
604
- if (existsSync(resolve(root, 'src/main.rs'))) run('cargo', ['test', '--bin', packageName()])
605
- if (!existsSync(resolve(root, 'src/lib.rs')) && !existsSync(resolve(root, 'src/main.rs')))
606
- run('cargo', ['test'])
758
+ // Batch exact prior targets through one Cargo graph.
759
+ const args = ['test']
760
+ if (existsSync(resolve(root, 'src/lib.rs'))) args.push('--lib')
761
+ if (existsSync(resolve(root, 'src/main.rs'))) args.push('--bin', packageName())
762
+ run('cargo', args)
607
763
  } else {
608
- for (const file of rustTests) {
764
+ // Keep --test selection narrow; --tests includes unit targets.
765
+ const targets = rustTests.flatMap((file) => {
609
766
  const match = file.match(/^tests\/(.+)\.rs$/)
610
- if (match && !match[1].includes('/')) run('cargo', ['test', '--test', match[1]])
611
- }
767
+ return match && !match[1].includes('/') ? ['--test', match[1]] : []
768
+ })
769
+ if (targets.length > 0) run('cargo', ['test', ...targets])
612
770
  }
613
771
  }
614
772
  }
package/src/runtime.mjs CHANGED
@@ -81,7 +81,7 @@ function sourceSha(root) {
81
81
  }
82
82
 
83
83
  /**
84
- * Stable public runtime; the original ecosystem executor is kept private and unchanged.
84
+ * Stable public runtime; the native executor stays separate from policy and evidence.
85
85
  * Tests can substitute an executor without installing consumer dependencies.
86
86
  * @param {string[]} args @param {string} [root] @param {string} [entry]
87
87
  * @returns {number}
@@ -186,6 +186,8 @@ export function runRuntime(args, root = process.cwd(), entry = core) {
186
186
  report.coverage = evaluateCoverage(root, policy, before)
187
187
  report.artifacts.push(...report.coverage.artifacts)
188
188
  }
189
+ if (task === 'eval')
190
+ report.artifacts.push('eval-results/summary.json', 'eval-results/result.json')
189
191
  if (task === 'performance') report.artifacts.push('performance-results/summary.json')
190
192
  report.status = 'passed'
191
193
  return 0
@@ -41,12 +41,14 @@ htmlcov/
41
41
  artifacts/
42
42
  performance-results.json
43
43
  performance-results/
44
+ eval-results.json
45
+ eval-results/
44
46
  cache/
45
47
  !.github/actions/cache/
46
48
  !.github/actions/cache/action.yml
47
49
  typechain/
48
50
  coverage.json
49
- performance-results/
51
+ .code-foundry/
50
52
  tests/.bin/
51
53
  **/.app/
52
54