code-foundry 1.22.1 → 1.28.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/.github/workflows/cloudflare-delivery.yml +134 -1
  2. package/.github/workflows/cloudflare-deploy.yml +140 -1
  3. package/.github/workflows/consumer-qualification.yml +3 -3
  4. package/.github/workflows/eval.yml +116 -0
  5. package/.github/workflows/qualified-foundry-publish.yml +4 -5
  6. package/.github/workflows/release_self-ci.yml +24 -10
  7. package/.github/workflows/validation-no-codeql.yml +34 -6
  8. package/.github/workflows/validation.yml +32 -5
  9. package/.github/workflows/validation_audit_self-ci.yml +1 -0
  10. package/.github/workflows/validation_self-ci.yml +7 -1
  11. package/AGENTS.md +10 -0
  12. package/CHANGELOG.md +77 -0
  13. package/README.md +4 -1
  14. package/docs/CONFIGURATION.md +20 -13
  15. package/docs/EVALS.md +139 -0
  16. package/docs/PUBLISHING.md +6 -1
  17. package/docs/WORKFLOWS.md +27 -4
  18. package/docs/cloudflare-delivery.md +47 -25
  19. package/docs/consumer-qualification.md +1 -1
  20. package/docs/fleet-release-eligibility.md +1 -5
  21. package/docs/qualified-publication.md +1 -1
  22. package/docs/required-capabilities.md +6 -4
  23. package/package.json +3 -3
  24. package/src/commands/doctor.mjs +5 -2
  25. package/src/commands/fleet-core.mjs +1 -1
  26. package/src/commands/qualified-publication.mjs +81 -3
  27. package/src/commands/release-integrity.mjs +37 -14
  28. package/src/commands/sync.mjs +3 -2
  29. package/src/lib/docs-only.mjs +39 -0
  30. package/src/lib/eval-envelope.mjs +234 -0
  31. package/src/lib/fleet-manifest.mjs +9 -5
  32. package/src/lib/merge-queue.mjs +1 -0
  33. package/src/lib/overlay.mjs +1 -1
  34. package/src/lib/release-manifest.mjs +1 -1
  35. package/src/lib/release-policy.mjs +2 -2
  36. package/src/lib/task-policy.mjs +7 -0
  37. package/src/lib/validation-policy.mjs +10 -6
  38. package/src/runtime-core.mjs +187 -10
  39. package/src/runtime.mjs +2 -0
  40. package/src/templates/gitignore +2 -0
@@ -42,6 +42,14 @@ on:
42
42
  required: false
43
43
  type: string
44
44
  default: ubuntu-slim
45
+ eval-runner:
46
+ description: >-
47
+ Runner used by the eval harness. Browser evals need a
48
+ Chrome-capable runner such as ubuntu-latest; ubuntu-slim does not
49
+ ship a browser.
50
+ required: false
51
+ type: string
52
+ default: ubuntu-latest
45
53
  codeql-runner:
46
54
  description: Runner used by CodeQL jobs.
47
55
  required: false
@@ -106,28 +114,47 @@ jobs:
106
114
  runner: ${{ inputs.test-runner }}
107
115
  unit-runner: ${{ inputs.unit-runner }}
108
116
  performance-runner: ${{ inputs.performance-runner }}
109
- unit-only: ${{ inputs.mode == 'fast' }}
117
+ # Release pull requests change version metadata only; their content was
118
+ # audited on the pull requests that merged into main. Integration, E2E,
119
+ # smoke, and performance lanes stay in the audit and scheduled lanes.
120
+ unit-only: ${{ inputs.mode == 'fast' || inputs.mode == 'release' }}
110
121
  artifact-prefix: ${{ inputs.artifact-prefix }}
111
122
  secrets:
112
123
  TURBO_TOKEN: ${{ secrets.TURBO_TOKEN }}
113
124
 
114
125
  security:
115
126
  name: Security
116
- if: vars.CI_BILLING_PAUSED != 'true' && (inputs.mode == 'audit' || inputs.mode == 'release')
127
+ if: vars.CI_BILLING_PAUSED != 'true' && inputs.mode == 'audit'
117
128
  uses: ./.github/workflows/security.yml
118
129
  with:
119
130
  runtime-repository: ${{ inputs.runtime-repository }}
120
131
  runtime-ref: ${{ inputs.runtime-ref }}
121
132
  runner: ${{ inputs.security-runner }}
122
133
 
134
+ eval:
135
+ name: Eval
136
+ # Behavior evals cover content changes; the lean release lane skips them
137
+ # because its diff is version metadata only. The reusable workflow skips
138
+ # cleanly when the repository has no eval harness.
139
+ if: vars.CI_BILLING_PAUSED != 'true' && inputs.mode == 'audit'
140
+ uses: ./.github/workflows/eval.yml
141
+ with:
142
+ runtime-repository: ${{ inputs.runtime-repository }}
143
+ runtime-ref: ${{ inputs.runtime-ref }}
144
+ runner: ${{ inputs.eval-runner }}
145
+ artifact-prefix: ${{ inputs.artifact-prefix }}
146
+
123
147
  # Aggregate check: "Validation / Gate" (caller job name + this job name).
124
148
  # Always evaluates; only jobs required for the mode must have succeeded.
125
- # The release tier runs the available CI/test/security suite. Its generated
126
- # release diff policy executes as a conditional step below, so no separate
127
- # release-policy job renders skipped checks on ordinary pull requests.
149
+ # The release tier requires the available fast suite; its diff is version
150
+ # metadata (validated against the release policy below), its content was
151
+ # audited on the pull requests that merged into main, and the scheduled
152
+ # audit lane re-covers drift. Its generated release diff policy executes
153
+ # as a conditional step below, so no separate release-policy job renders
154
+ # skipped checks on ordinary pull requests.
128
155
  gate:
129
156
  name: Gate
130
- needs: [ci, test, security]
157
+ needs: [ci, test, security, eval]
131
158
  if: vars.CI_BILLING_PAUSED != 'true' && always()
132
159
  runs-on: ubuntu-slim
133
160
  timeout-minutes: 10
@@ -136,6 +163,7 @@ jobs:
136
163
  FOUNDRY_CI: ${{ needs.ci.result }}
137
164
  FOUNDRY_TEST: ${{ needs.test.result }}
138
165
  FOUNDRY_SECURITY: ${{ needs.security.result }}
166
+ FOUNDRY_EVAL: ${{ needs.eval.result }}
139
167
  # This workflow is selected only for repositories whose canonical
140
168
  # configuration explicitly disables unavailable CodeQL. Treat that
141
169
  # policy decision as satisfied without registering a skipped PR check.
@@ -42,6 +42,14 @@ on:
42
42
  required: false
43
43
  type: string
44
44
  default: ubuntu-slim
45
+ eval-runner:
46
+ description: >-
47
+ Runner used by the eval harness. Browser evals need a
48
+ Chrome-capable runner such as ubuntu-latest; ubuntu-slim does not
49
+ ship a browser.
50
+ required: false
51
+ type: string
52
+ default: ubuntu-latest
45
53
  codeql-runner:
46
54
  description: Runner used by CodeQL jobs.
47
55
  required: false
@@ -106,14 +114,17 @@ jobs:
106
114
  runner: ${{ inputs.test-runner }}
107
115
  unit-runner: ${{ inputs.unit-runner }}
108
116
  performance-runner: ${{ inputs.performance-runner }}
109
- unit-only: ${{ inputs.mode == 'fast' }}
117
+ # Release pull requests change version metadata only; their content was
118
+ # audited on the pull requests that merged into main. Integration, E2E,
119
+ # smoke, and performance lanes stay in the audit and scheduled lanes.
120
+ unit-only: ${{ inputs.mode == 'fast' || inputs.mode == 'release' }}
110
121
  artifact-prefix: ${{ inputs.artifact-prefix }}
111
122
  secrets:
112
123
  TURBO_TOKEN: ${{ secrets.TURBO_TOKEN }}
113
124
 
114
125
  security:
115
126
  name: Security
116
- if: vars.CI_BILLING_PAUSED != 'true' && (inputs.mode == 'audit' || inputs.mode == 'release')
127
+ if: vars.CI_BILLING_PAUSED != 'true' && inputs.mode == 'audit'
117
128
  uses: ./.github/workflows/security.yml
118
129
  with:
119
130
  runtime-repository: ${{ inputs.runtime-repository }}
@@ -134,15 +145,30 @@ jobs:
134
145
  rust-threads: ${{ inputs.rust-threads }}
135
146
  rust-max-parallel: ${{ inputs.rust-max-parallel }}
136
147
 
148
+ eval:
149
+ name: Eval
150
+ # Behavior evals cover content changes; the lean release lane skips them
151
+ # because its diff is version metadata only. The reusable workflow skips
152
+ # cleanly when the repository has no eval harness.
153
+ if: vars.CI_BILLING_PAUSED != 'true' && inputs.mode == 'audit'
154
+ uses: ./.github/workflows/eval.yml
155
+ with:
156
+ runtime-repository: ${{ inputs.runtime-repository }}
157
+ runtime-ref: ${{ inputs.runtime-ref }}
158
+ runner: ${{ inputs.eval-runner }}
159
+ artifact-prefix: ${{ inputs.artifact-prefix }}
160
+
137
161
  # Aggregate check: "Validation / Gate" (caller job name + this job name).
138
162
  # Always evaluates; only jobs required for the mode must have succeeded.
139
- # The release tier runs the full audit suite so Release Please pull requests
140
- # expose no neutral suite checks. Its generated release diff policy executes
163
+ # The release tier requires the fast suite plus CodeQL: its diff is version
164
+ # metadata (validated against the release policy below), its content was
165
+ # audited on the pull requests that merged into main, and the scheduled
166
+ # audit lane re-covers drift. Its generated release diff policy executes
141
167
  # as a conditional step below, so ordinary pull requests still render no
142
168
  # skipped release-policy row.
143
169
  gate:
144
170
  name: Gate
145
- needs: [ci, test, security, codeql]
171
+ needs: [ci, test, security, codeql, eval]
146
172
  if: vars.CI_BILLING_PAUSED != 'true' && always()
147
173
  runs-on: ubuntu-slim
148
174
  timeout-minutes: 10
@@ -152,6 +178,7 @@ jobs:
152
178
  FOUNDRY_TEST: ${{ needs.test.result }}
153
179
  FOUNDRY_SECURITY: ${{ needs.security.result }}
154
180
  FOUNDRY_CODEQL: ${{ needs.codeql.result }}
181
+ FOUNDRY_EVAL: ${{ needs.eval.result }}
155
182
  steps:
156
183
  - name: Checkout runtime
157
184
  uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
@@ -34,6 +34,7 @@ jobs:
34
34
  unit-runner: ubuntu-slim
35
35
  performance-runner: ubuntu-latest
36
36
  security-runner: ubuntu-slim
37
+ eval-runner: ubuntu-latest
37
38
  codeql-runner: ubuntu-latest
38
39
  rust-shards: '["all"]'
39
40
  rust-threads: '1'
@@ -64,7 +64,12 @@ jobs:
64
64
  validation:
65
65
  name: Validation
66
66
  needs: mode
67
- if: vars.CI_BILLING_PAUSED != 'true' && github.event_name == 'pull_request' && github.event.pull_request.draft == false
67
+ # Bot-authored pull requests (Dependabot and friends) validate only on an
68
+ # explicit ready transition: dependency pushes never allocate validation
69
+ # runners by themselves. A human pushing to a bot branch also runs, since
70
+ # the sender is no longer a bot. Release Please heads are managed by the
71
+ # release workflow and always validate.
72
+ if: vars.CI_BILLING_PAUSED != 'true' && github.event_name == 'pull_request' && github.event.pull_request.draft == false && (github.event.pull_request.user.type != 'Bot' || github.event.sender.type != 'Bot' || startsWith(github.event.pull_request.head.ref, 'release-please--branches--main'))
68
73
  # Reusable workflows can only maintain or reduce the caller job's scopes.
69
74
  # The audit tier needs security-events: write for CodeQL uploads.
70
75
  permissions:
@@ -85,6 +90,7 @@ jobs:
85
90
  unit-runner: ubuntu-slim
86
91
  performance-runner: ubuntu-latest
87
92
  security-runner: ubuntu-slim
93
+ eval-runner: ubuntu-latest
88
94
  codeql-runner: ubuntu-latest
89
95
  rust-shards: '["all"]'
90
96
  rust-threads: '1'
package/AGENTS.md CHANGED
@@ -17,6 +17,15 @@ They complement `CONTRIBUTING.md`. More specific instructions in nested `AGENTS.
17
17
 
18
18
  This repository may contain TypeScript, Rust, Python, or any combination of them. Detect the active stack from the files present; do not assume every check applies.
19
19
 
20
+ ## Contributing back
21
+
22
+ Consumers are encouraged to help improve this open-source project. Open a small,
23
+ focused pull request for bug fixes, performance improvements, documentation, or
24
+ other narrowly scoped changes. For larger feature requests or architectural
25
+ changes, create an issue first so the proposal can be discussed and scoped.
26
+ Contributions should help make the tool as performant, reliable, and helpful as
27
+ possible for everyone.
28
+
20
29
  ## Read before acting
21
30
 
22
31
  Before editing:
@@ -149,6 +158,7 @@ node src/runtime.mjs ci unit
149
158
  node src/runtime.mjs ci integration
150
159
  node src/runtime.mjs ci e2e
151
160
  node src/runtime.mjs ci smoke
161
+ node src/runtime.mjs ci eval
152
162
  node src/runtime.mjs ci performance
153
163
  Security and dependency audits run through the GitHub Security workflow.
154
164
  ```
package/CHANGELOG.md CHANGED
@@ -1,5 +1,82 @@
1
1
  # Changelog
2
2
 
3
+ ## [1.28.2](https://github.com/0xPlayerOne/code-foundry/compare/v1.28.1...v1.28.2) (2026-09-10)
4
+
5
+
6
+ ### Bug Fixes
7
+
8
+ * **release:** pass one qualification report per required node ([#597](https://github.com/0xPlayerOne/code-foundry/issues/597)) ([5ceb92c](https://github.com/0xPlayerOne/code-foundry/commit/5ceb92cab3afc97145ef068e6b5e33399d99b9e5))
9
+
10
+ ## [1.28.1](https://github.com/0xPlayerOne/code-foundry/compare/v1.28.0...v1.28.1) (2026-09-10)
11
+
12
+
13
+ ### Bug Fixes
14
+
15
+ * **cloudflare:** remove duplicate deployment records ([#595](https://github.com/0xPlayerOne/code-foundry/issues/595)) ([d9a2a4b](https://github.com/0xPlayerOne/code-foundry/commit/d9a2a4b0363bb9a9ded91d3368bdb7b9b580330f))
16
+
17
+ ## [1.28.0](https://github.com/0xPlayerOne/code-foundry/compare/v1.27.0...v1.28.0) (2026-09-10)
18
+
19
+
20
+ ### Features
21
+
22
+ * **release:** qualify only when a release or a stuck draft needs it ([#591](https://github.com/0xPlayerOne/code-foundry/issues/591)) ([4cce045](https://github.com/0xPlayerOne/code-foundry/commit/4cce045228648e6e705ad1e89c59eaf20a72a538))
23
+
24
+ ## [1.27.0](https://github.com/0xPlayerOne/code-foundry/compare/v1.26.0...v1.27.0) (2026-09-10)
25
+
26
+
27
+ ### Features
28
+
29
+ * **validation:** downgrade docs-only PRs to fast and gate bot pushes ([#590](https://github.com/0xPlayerOne/code-foundry/issues/590)) ([df774f5](https://github.com/0xPlayerOne/code-foundry/commit/df774f5d1e4a8a6c5f174219eb2b24ef012f6fd3))
30
+
31
+ ## [1.26.0](https://github.com/0xPlayerOne/code-foundry/compare/v1.25.3...v1.26.0) (2026-09-10)
32
+
33
+
34
+ ### Features
35
+
36
+ * **node:** support Node 24 and 26, drop 20 and 22 ([#589](https://github.com/0xPlayerOne/code-foundry/issues/589)) ([ee3e294](https://github.com/0xPlayerOne/code-foundry/commit/ee3e2943889ffd06a8d4cd4ccb75a3f36c46c0fd))
37
+
38
+ ## [1.25.3](https://github.com/0xPlayerOne/code-foundry/compare/v1.25.2...v1.25.3) (2026-09-10)
39
+
40
+
41
+ ### Bug Fixes
42
+
43
+ * **release:** cover the attestation lag in publication retries ([#587](https://github.com/0xPlayerOne/code-foundry/issues/587)) ([7ae5e4b](https://github.com/0xPlayerOne/code-foundry/commit/7ae5e4bb95f68cbf4e2b48778776a0396c877271))
44
+
45
+ ## [1.25.2](https://github.com/0xPlayerOne/code-foundry/compare/v1.25.1...v1.25.2) (2026-09-10)
46
+
47
+
48
+ ### Bug Fixes
49
+
50
+ * **release:** retry integrity verification through the publication consistency window ([#585](https://github.com/0xPlayerOne/code-foundry/issues/585)) ([5b997dd](https://github.com/0xPlayerOne/code-foundry/commit/5b997dda7e272b59a386eeeda115e60ad284f6ce))
51
+
52
+ ## [1.25.1](https://github.com/0xPlayerOne/code-foundry/compare/v1.25.0...v1.25.1) (2026-09-10)
53
+
54
+
55
+ ### Bug Fixes
56
+
57
+ * **release:** wait for the release index before verifying a published release ([#583](https://github.com/0xPlayerOne/code-foundry/issues/583)) ([7823c14](https://github.com/0xPlayerOne/code-foundry/commit/7823c1473d8c5e7c17b039dc755ff5c8d1dba39c))
58
+
59
+ ## [1.25.0](https://github.com/0xPlayerOne/code-foundry/compare/v1.24.0...v1.25.0) (2026-09-10)
60
+
61
+
62
+ ### Features
63
+
64
+ * **eval:** run the eval tier as a managed Validation / Eval lane ([#580](https://github.com/0xPlayerOne/code-foundry/issues/580)) ([9414889](https://github.com/0xPlayerOne/code-foundry/commit/94148894da9a627a6bd7c71c846848c198c1dfec))
65
+
66
+ ## [1.24.0](https://github.com/0xPlayerOne/code-foundry/compare/v1.23.0...v1.24.0) (2026-09-10)
67
+
68
+
69
+ ### Features
70
+
71
+ * **validation:** run a lean release lane for Release Please pull requests ([#579](https://github.com/0xPlayerOne/code-foundry/issues/579)) ([a38ca75](https://github.com/0xPlayerOne/code-foundry/commit/a38ca75c7da577af8368467b8b026d0ec563fc41))
72
+
73
+ ## [1.23.0](https://github.com/0xPlayerOne/code-foundry/compare/v1.22.1...v1.23.0) (2026-09-09)
74
+
75
+
76
+ ### Features
77
+
78
+ * add ci eval tier with shared report contract and budget gate ([#576](https://github.com/0xPlayerOne/code-foundry/issues/576)) ([7d881dd](https://github.com/0xPlayerOne/code-foundry/commit/7d881dd1bc6ace8e48dffff45b1a22a12bdae1c9))
79
+
3
80
  ## [1.22.1](https://github.com/0xPlayerOne/code-foundry/compare/v1.22.0...v1.22.1) (2026-09-09)
4
81
 
5
82
 
package/README.md CHANGED
@@ -39,6 +39,9 @@ contract. For normal updates, edit that file and run `npx code-foundry sync`.
39
39
  audits, Draft Guard, Draft PR, Release PR, and Release.
40
40
  - A deterministic performance lane with ordered command support, stable result
41
41
  artifacts, and an optional shared Node package budget profile.
42
+ - A deterministic eval tier (`ci eval`) that runs a repository's behavior-eval
43
+ harness against a shared report contract with optional budgets, keeping task
44
+ outcomes comparable across revisions and executors.
42
45
  - An opt-in product-quality runner for static sites, web apps, Workers, and
43
46
  published packages.
44
47
  - A small `.githooks/pre-commit` launcher with language-aware formatting and
@@ -132,7 +135,7 @@ is merged; npm publication is opt-in through `npm_publish: true` and supports
132
135
  npm trusted publishing or an `NPM_TOKEN` fallback.
133
136
 
134
137
  Code Foundry's own release caller is stricter: it qualifies the package across
135
- Node 20, 22, and 24, stages the exact qualified archive, publishes the
138
+ Node 24 and 26, stages the exact qualified archive, publishes the
136
139
  immutable GitHub Release, and publishes that archive through the verified
137
140
  publisher. See
138
141
  [Consumer qualification](docs/consumer-qualification.md) and [Qualified
@@ -89,22 +89,29 @@ See [Merge queue validation](merge-queues.md) before enabling it.
89
89
 
90
90
  ## Validation and quality
91
91
 
92
- | Key | Values | Purpose |
93
- | ------------------------- | ---------------------------------------------- | ------------------------------------------------------------------------------- |
94
- | `performance` | `auto`, `true`, `false` | Discover, require, or disable performance checks. |
95
- | `performance_command` | JSON argv array or array of argv arrays | Ordered commands for a non-package performance harness. |
96
- | `performance_profile` | empty or `node-package` | Shared package import, memory, archive, and dependency audit. |
97
- | `performance_budget_file` | repository-relative path | Budget file for `node-package`; defaults to `performance-package-budgets.json`. |
98
- | `required_capabilities` | comma-separated task names | Fail closed when a required task or coverage evidence is unavailable. |
99
- | `coverage_enforcement` | `auto`, `required`, `off` | Shared coverage-report policy. |
100
- | `coverage_minimum` | `0`–`100` | Minimum percentage; defaults to `80`. |
101
- | `coverage_metrics` | `lines`, `functions`, `branches`, `statements` | Metrics checked by the coverage gate. |
102
- | `coverage_report` | comma-separated repository paths | Istanbul JSON summary or LCOV evidence files. |
92
+ | Key | Values | Purpose |
93
+ | ------------------------- | ---------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------- |
94
+ | `performance` | `auto`, `true`, `false` | Discover, require, or disable performance checks. |
95
+ | `performance_command` | JSON argv array or array of argv arrays | Ordered commands for a non-package performance harness. |
96
+ | `performance_profile` | empty or `node-package` | Shared package import, memory, archive, and dependency audit. |
97
+ | `performance_budget_file` | repository-relative path | Budget file for `node-package`; defaults to `performance-package-budgets.json`. |
98
+ | `eval` | `auto`, `true`, `false` | Discover, require, or disable the deterministic eval tier. |
99
+ | `eval_command` | JSON argv array | Explicit harness command when there is no `eval` package script. |
100
+ | `eval_report_file` | repository-relative path | Report validated against the eval contract; defaults to `eval-results/result.json`. |
101
+ | `eval_budget_file` | repository-relative path | Optional budget file; defaults to `eval-budgets.json`. |
102
+ | `eval_runner` | runner label | Runner for the `Validation / Eval` lane; defaults to the repository runner. Browser evals need a Chrome-capable runner such as `ubuntu-latest`. |
103
+ | `required_capabilities` | comma-separated task names | Fail closed when a required task or coverage evidence is unavailable. |
104
+ | `coverage_enforcement` | `auto`, `required`, `off` | Shared coverage-report policy. |
105
+ | `coverage_minimum` | `0`–`100` | Minimum percentage; defaults to `80`. |
106
+ | `coverage_metrics` | `lines`, `functions`, `branches`, `statements` | Metrics checked by the coverage gate. |
107
+ | `coverage_report` | comma-separated repository paths | Istanbul JSON summary or LCOV evidence files. |
103
108
 
104
109
  Supported task capabilities are `format`, `lint`, `type_check`, `build`, `unit`,
105
- `integration`, `e2e`, `smoke`, and `performance`. `coverage` is a policy
110
+ `integration`, `e2e`, `smoke`, `eval`, and `performance`. `coverage` is a policy
106
111
  capability that also requires unit tests. See [Required capabilities and task
107
- evidence](required-capabilities.md).
112
+ evidence](required-capabilities.md). The eval tier runs the repository's own
113
+ harness against the shared report contract and optional budgets; see
114
+ [Evals](EVALS.md).
108
115
 
109
116
  The shared performance job discovers `performance:check`, then `perf:check`,
110
117
  in JavaScript repositories. Other repositories can provide one command or an
package/docs/EVALS.md ADDED
@@ -0,0 +1,139 @@
1
+ # Evals
2
+
3
+ `ci eval` is the repository's deterministic behavior-evaluation tier. It runs the
4
+ repository's own eval harness and consumes its result against a shared contract,
5
+ so task outcomes stay comparable across revisions and — later — across
6
+ executors (deterministic today, model-agent when a repository adopts one).
7
+
8
+ Evals are tests' measurement-oriented sibling: tests verify a specified
9
+ behavior (pass/fail); evals record how well the system does at repeated probes
10
+ (success rates, timing percentiles, bounded evidence) and compare those numbers
11
+ against pinned baselines. The `performance` tier is the resource-metric
12
+ special case of the same pattern.
13
+
14
+ ## How the runtime finds your harness
15
+
16
+ `ci eval` discovers its subject in this order and skips when neither exists:
17
+
18
+ 1. A package script named `eval`.
19
+ 2. An explicit `eval_command` (JSON argv array) in `.github/code-foundry.yml`.
20
+
21
+ Both mechanisms use `eval: auto` by default: the tier runs when a subject is
22
+ present and skips cleanly otherwise. Set `eval: true` to require it (the
23
+ validation policy then treats a missing subject as an error), or `eval: false`
24
+ to disable discovery.
25
+
26
+ In managed validation the eval tier runs as its own `Validation / Eval` lane
27
+ during audit-mode runs, with the receipt retained as a task artifact. Browser
28
+ eval harnesses need a Chrome-capable runner: set `eval_runner: ubuntu-latest`
29
+ in `.github/code-foundry.yml` (the default when unset is the repository's
30
+ default runner, which may not ship a browser). The lane uploads the eval
31
+ report alongside the task receipt whenever a run fails, so reviewers get the
32
+ measured numbers with the red check. Performance's task profile and gating
33
+ rules apply unchanged; evals are an optional, non-gating tier like
34
+ performance.
35
+
36
+ | Config key | Values | Meaning |
37
+ | ------------------ | --------------------------------- | ---------------------------------------------------------------- |
38
+ | `eval` | `auto` (default), `true`, `false` | Enable, require, or disable the eval tier. |
39
+ | `eval_command` | JSON argv array | Explicit harness command when there is no `eval` package script. |
40
+ | `eval_report_file` | repository-relative path | The report to validate; defaults to `eval-results/result.json`. |
41
+ | `eval_budget_file` | repository-relative path | Optional budget file; defaults to `eval-budgets.json`. |
42
+
43
+ ## The report contract
44
+
45
+ Your harness writes `eval-results/result.json`. The runtime validates the
46
+ envelope before budgets; a report that violates the contract fails the tier
47
+ with every violation listed (fail closed, typos can never pass silently):
48
+
49
+ ```jsonc
50
+ {
51
+ "schemaVersion": 1, // contract version, currently 1
52
+ "revision": "<git sha>", // optional but strongly recommended
53
+ "dependencyHash": "<sha256>", // optional; pins the dependency set
54
+ "summary": {
55
+ "taskCount": 7,
56
+ "attempts": 7,
57
+ "passed": 7,
58
+ "failed": 0,
59
+ "harnessFailures": 0, // environment broke; not a task regression
60
+ "successRate": 1.0,
61
+ "toolCalls": 31,
62
+ "evidenceErrors": 0,
63
+ "taskDurationMs": { "count": 7, "mean": 2.1, "p50": 1.8, "p95": 3.0, "max": 3.4 },
64
+ "startupMs": { "count": 7, "mean": 0.3, "p50": 0.3, "p95": 0.4, "max": 0.4 },
65
+ "stepDurationMs": { "count": 31, "mean": 0.2, "p50": 0.1, "p95": 0.6, "max": 0.6 },
66
+ },
67
+ "tasks": [
68
+ {
69
+ "id": "form-submit",
70
+ "attempts": [
71
+ {
72
+ "iteration": 1,
73
+ "status": "passed",
74
+ "durationMs": 2.1,
75
+ "startupMs": 0.3,
76
+ "steps": [{ "tool": "fill_form", "status": "passed", "durationMs": 0.4 }],
77
+ "checks": ["fill_form completed"],
78
+ "metrics": { "formFields": 2 },
79
+ },
80
+ ],
81
+ },
82
+ ],
83
+ }
84
+ ```
85
+
86
+ Rules the validator enforces:
87
+
88
+ - `schemaVersion` must match the contract version; `revision`/`dependencyHash`
89
+ must be strings when present.
90
+ - Summary counts are non-negative integers; `passed + failed` cannot exceed
91
+ `attempts`; `successRate` is between 0 and 1.
92
+ - `harnessFailures` counts attempts where the environment itself broke (browser
93
+ never started, zero steps succeeded, cleanup failed). It must not exceed
94
+ `attempts`. Fix the environment; do not treat these as task regressions.
95
+ - Failed attempts carry a bounded `failure` (non-empty `message`) plus a
96
+ `failureClass` of `harness` or `task`.
97
+ - `taskDurationMs`, `startupMs`, and `stepDurationMs` are stats objects with
98
+ `count`, `mean`, `p50`, `p95`, `max` (or bare `{ "count": 0 }`).
99
+
100
+ A reference implementation ships in
101
+ [`pi-browser-use`](https://github.com/0xPlayerOne/pi-browser-use)
102
+ (`scripts/eval.mjs` + `docs/eval-results.md`): deterministic browser-tool tasks
103
+ that emit this exact envelope.
104
+
105
+ ## Budgets
106
+
107
+ Create `eval-budgets.json` (or point `eval_budget_file` at another file) to
108
+ gate on the measured numbers:
109
+
110
+ ```json
111
+ {
112
+ "successRate": 1.0,
113
+ "stepP95Ms": 1000,
114
+ "taskP95Ms": 45000,
115
+ "maxHarnessFailures": 0
116
+ }
117
+ ```
118
+
119
+ Supported budgets: `successRate`, `taskP95Ms`, `startupP95Ms`, `stepP95Ms`,
120
+ `maxHarnessFailures`, `maxEvidenceErrors`, `maxToolCalls`. Unknown keys fail
121
+ closed. When the budget file is absent the tier validates the contract only.
122
+ Budget files are committed configuration; only `eval-results/` is a local
123
+ artifact.
124
+
125
+ The runtime writes `eval-results/summary.json` with the executed commands, the
126
+ budget outcome, and the artifact list, mirroring the performance summary.
127
+
128
+ ## Determinism rules for eval tasks
129
+
130
+ - Fixed fixtures and no network. A task's outcome must depend only on the code
131
+ under test.
132
+ - Bounded evidence: failure text and screenshots are capped; a failing task
133
+ never produces unbounded output.
134
+ - Cleanup failures are surfaced, never swallowed silently.
135
+ - Every attempt records the revision and dependency hash it ran against.
136
+ - Keep model-in-the-loop runs out of this tier. Reuse the same task IDs in a
137
+ separate scheduled harness with frozen model/reasoning/judge baselines when
138
+ you need agent-capability measurement; its variance is why it must never
139
+ gate pull requests.
@@ -37,7 +37,7 @@ registry version and its provenance link before treating the setup as complete.
37
37
  The self-hosted package follows a stricter contract than generated consumer
38
38
  callers:
39
39
 
40
- 1. Pack the candidate once and qualify that archive across Node 20, 22, and 24.
40
+ 1. Pack the candidate once and qualify that archive across Node 24 and 26.
41
41
  2. Create a draft GitHub Release and attach the exact qualified archive plus its
42
42
  qualification receipt.
43
43
  3. Publish the immutable GitHub Release after verifying the tag, source commit,
@@ -50,6 +50,11 @@ the bytes selected from the same workflow run and attempt. See [Consumer
50
50
  qualification](consumer-qualification.md) and [Qualified publication](qualified-publication.md)
51
51
  for the complete contract, retries, and recovery rules.
52
52
 
53
+ The main-push pipeline runs Release Please first, then qualifies only when the
54
+ push created a release or recovery found a stuck draft. Feature merges pay for
55
+ the cheap release-please and recovery probes; the runner-heavy matrix runs on
56
+ release merges, which re-qualify the exact tree they publish.
57
+
53
58
  ## GitHub Releases and GitHub Packages
54
59
 
55
60
  A GitHub Release is metadata attached to a Git tag. It is independent of npm and
package/docs/WORKFLOWS.md CHANGED
@@ -27,7 +27,24 @@ ready transition. Release Please version heads are excluded because the
27
27
  release workflow owns their state. A separate draft-control caller listens for
28
28
  `converted_to_draft` and cancels queued or running pull-request workflows.
29
29
  Marking a pull request ready starts validation, and each new commit on a ready
30
- pull request starts it again for the current head.
30
+ pull request starts it again for the current head. Audit-mode runs additionally
31
+ execute the eval lane (`Validation / Eval`) for repositories that ship an eval
32
+ harness; see [Evals](EVALS.md).
33
+
34
+ Two waste-avoidance rules keep validation minutes honest without weakening
35
+ confidence:
36
+
37
+ - **Docs-only pull requests run the fast tier.** When an audit-classified pull
38
+ request changes only markdown, `docs/`, and license roots, the mode
39
+ classifier downgrades it to fast. Any code, lockfile, workflow,
40
+ configuration, or packaging change — or any diff that cannot be computed
41
+ deterministically — keeps the audit tier. Scheduled and manual audits always
42
+ run the full tier.
43
+ - **Bot-authored pull requests validate only on ready transitions.**
44
+ Dependabot and other bot pushes never allocate validation runners; a
45
+ maintainer marks the (Guard-drafted) pull request ready — or pushes to its
46
+ branch, which changes the sender — to run validation. Release Please heads
47
+ are exempt because the release workflow merges them through its own lane.
31
48
 
32
49
  The separate `validation-audit.yml` caller is pinned to the configured released
33
50
  runtime and handles scheduled and manual audits:
@@ -40,9 +57,15 @@ workflow_dispatch:
40
57
 
41
58
  In the `staging-release` topology, pull requests into `staging` run the fast
42
59
  tier, ordinary pull requests into `main` run the full audit tier, and exact
43
- Release Please pull requests into `main` run the full audit tier plus the
44
- release-diff policy. This keeps repository rulesets satisfiable for the release
45
- commit without exposing neutral or skipped suite checks. In the
60
+ Release Please pull requests into `main` run a lean release lane — CI, unit
61
+ tests, and CodeQL plus the release-diff policy. Their diff is version
62
+ metadata only: the content was fully audited on the pull requests that
63
+ merged into `main`, and the scheduled audit lane re-covers drift, so the
64
+ release lane keeps repository rulesets satisfiable without re-running the
65
+ runner-heavy suites. The managed branch rulesets require only the aggregate
66
+ `Validation / Gate` check, so skipped non-required jobs never deadlock the
67
+ release; do not hand-require individual job contexts on release branches.
68
+ In the
46
69
  `direct` topology (the default) every pull request targets `main` and runs the
47
70
  full audit tier, because there is no integration branch for a fast pass.
48
71
  Scheduled and manual runs select the audit tier in both topologies. Draft PR