code-foundry 1.22.1 → 1.28.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/workflows/cloudflare-delivery.yml +134 -1
- package/.github/workflows/cloudflare-deploy.yml +140 -1
- package/.github/workflows/consumer-qualification.yml +3 -3
- package/.github/workflows/eval.yml +116 -0
- package/.github/workflows/qualified-foundry-publish.yml +4 -5
- package/.github/workflows/release_self-ci.yml +24 -10
- package/.github/workflows/validation-no-codeql.yml +34 -6
- package/.github/workflows/validation.yml +32 -5
- package/.github/workflows/validation_audit_self-ci.yml +1 -0
- package/.github/workflows/validation_self-ci.yml +7 -1
- package/AGENTS.md +10 -0
- package/CHANGELOG.md +77 -0
- package/README.md +4 -1
- package/docs/CONFIGURATION.md +20 -13
- package/docs/EVALS.md +139 -0
- package/docs/PUBLISHING.md +6 -1
- package/docs/WORKFLOWS.md +27 -4
- package/docs/cloudflare-delivery.md +47 -25
- package/docs/consumer-qualification.md +1 -1
- package/docs/fleet-release-eligibility.md +1 -5
- package/docs/qualified-publication.md +1 -1
- package/docs/required-capabilities.md +6 -4
- package/package.json +3 -3
- package/src/commands/doctor.mjs +5 -2
- package/src/commands/fleet-core.mjs +1 -1
- package/src/commands/qualified-publication.mjs +81 -3
- package/src/commands/release-integrity.mjs +37 -14
- package/src/commands/sync.mjs +3 -2
- package/src/lib/docs-only.mjs +39 -0
- package/src/lib/eval-envelope.mjs +234 -0
- package/src/lib/fleet-manifest.mjs +9 -5
- package/src/lib/merge-queue.mjs +1 -0
- package/src/lib/overlay.mjs +1 -1
- package/src/lib/release-manifest.mjs +1 -1
- package/src/lib/release-policy.mjs +2 -2
- package/src/lib/task-policy.mjs +7 -0
- package/src/lib/validation-policy.mjs +10 -6
- package/src/runtime-core.mjs +187 -10
- package/src/runtime.mjs +2 -0
- package/src/templates/gitignore +2 -0
|
@@ -42,6 +42,14 @@ on:
|
|
|
42
42
|
required: false
|
|
43
43
|
type: string
|
|
44
44
|
default: ubuntu-slim
|
|
45
|
+
eval-runner:
|
|
46
|
+
description: >-
|
|
47
|
+
Runner used by the eval harness. Browser evals need a
|
|
48
|
+
Chrome-capable runner such as ubuntu-latest; ubuntu-slim does not
|
|
49
|
+
ship a browser.
|
|
50
|
+
required: false
|
|
51
|
+
type: string
|
|
52
|
+
default: ubuntu-latest
|
|
45
53
|
codeql-runner:
|
|
46
54
|
description: Runner used by CodeQL jobs.
|
|
47
55
|
required: false
|
|
@@ -106,28 +114,47 @@ jobs:
|
|
|
106
114
|
runner: ${{ inputs.test-runner }}
|
|
107
115
|
unit-runner: ${{ inputs.unit-runner }}
|
|
108
116
|
performance-runner: ${{ inputs.performance-runner }}
|
|
109
|
-
|
|
117
|
+
# Release pull requests change version metadata only; their content was
|
|
118
|
+
# audited on the pull requests that merged into main. Integration, E2E,
|
|
119
|
+
# smoke, and performance lanes stay in the audit and scheduled lanes.
|
|
120
|
+
unit-only: ${{ inputs.mode == 'fast' || inputs.mode == 'release' }}
|
|
110
121
|
artifact-prefix: ${{ inputs.artifact-prefix }}
|
|
111
122
|
secrets:
|
|
112
123
|
TURBO_TOKEN: ${{ secrets.TURBO_TOKEN }}
|
|
113
124
|
|
|
114
125
|
security:
|
|
115
126
|
name: Security
|
|
116
|
-
if: vars.CI_BILLING_PAUSED != 'true' &&
|
|
127
|
+
if: vars.CI_BILLING_PAUSED != 'true' && inputs.mode == 'audit'
|
|
117
128
|
uses: ./.github/workflows/security.yml
|
|
118
129
|
with:
|
|
119
130
|
runtime-repository: ${{ inputs.runtime-repository }}
|
|
120
131
|
runtime-ref: ${{ inputs.runtime-ref }}
|
|
121
132
|
runner: ${{ inputs.security-runner }}
|
|
122
133
|
|
|
134
|
+
eval:
|
|
135
|
+
name: Eval
|
|
136
|
+
# Behavior evals cover content changes; the lean release lane skips them
|
|
137
|
+
# because its diff is version metadata only. The reusable workflow skips
|
|
138
|
+
# cleanly when the repository has no eval harness.
|
|
139
|
+
if: vars.CI_BILLING_PAUSED != 'true' && inputs.mode == 'audit'
|
|
140
|
+
uses: ./.github/workflows/eval.yml
|
|
141
|
+
with:
|
|
142
|
+
runtime-repository: ${{ inputs.runtime-repository }}
|
|
143
|
+
runtime-ref: ${{ inputs.runtime-ref }}
|
|
144
|
+
runner: ${{ inputs.eval-runner }}
|
|
145
|
+
artifact-prefix: ${{ inputs.artifact-prefix }}
|
|
146
|
+
|
|
123
147
|
# Aggregate check: "Validation / Gate" (caller job name + this job name).
|
|
124
148
|
# Always evaluates; only jobs required for the mode must have succeeded.
|
|
125
|
-
# The release tier
|
|
126
|
-
#
|
|
127
|
-
#
|
|
149
|
+
# The release tier requires the available fast suite; its diff is version
|
|
150
|
+
# metadata (validated against the release policy below), its content was
|
|
151
|
+
# audited on the pull requests that merged into main, and the scheduled
|
|
152
|
+
# audit lane re-covers drift. Its generated release diff policy executes
|
|
153
|
+
# as a conditional step below, so no separate release-policy job renders
|
|
154
|
+
# skipped checks on ordinary pull requests.
|
|
128
155
|
gate:
|
|
129
156
|
name: Gate
|
|
130
|
-
needs: [ci, test, security]
|
|
157
|
+
needs: [ci, test, security, eval]
|
|
131
158
|
if: vars.CI_BILLING_PAUSED != 'true' && always()
|
|
132
159
|
runs-on: ubuntu-slim
|
|
133
160
|
timeout-minutes: 10
|
|
@@ -136,6 +163,7 @@ jobs:
|
|
|
136
163
|
FOUNDRY_CI: ${{ needs.ci.result }}
|
|
137
164
|
FOUNDRY_TEST: ${{ needs.test.result }}
|
|
138
165
|
FOUNDRY_SECURITY: ${{ needs.security.result }}
|
|
166
|
+
FOUNDRY_EVAL: ${{ needs.eval.result }}
|
|
139
167
|
# This workflow is selected only for repositories whose canonical
|
|
140
168
|
# configuration explicitly disables unavailable CodeQL. Treat that
|
|
141
169
|
# policy decision as satisfied without registering a skipped PR check.
|
|
@@ -42,6 +42,14 @@ on:
|
|
|
42
42
|
required: false
|
|
43
43
|
type: string
|
|
44
44
|
default: ubuntu-slim
|
|
45
|
+
eval-runner:
|
|
46
|
+
description: >-
|
|
47
|
+
Runner used by the eval harness. Browser evals need a
|
|
48
|
+
Chrome-capable runner such as ubuntu-latest; ubuntu-slim does not
|
|
49
|
+
ship a browser.
|
|
50
|
+
required: false
|
|
51
|
+
type: string
|
|
52
|
+
default: ubuntu-latest
|
|
45
53
|
codeql-runner:
|
|
46
54
|
description: Runner used by CodeQL jobs.
|
|
47
55
|
required: false
|
|
@@ -106,14 +114,17 @@ jobs:
|
|
|
106
114
|
runner: ${{ inputs.test-runner }}
|
|
107
115
|
unit-runner: ${{ inputs.unit-runner }}
|
|
108
116
|
performance-runner: ${{ inputs.performance-runner }}
|
|
109
|
-
|
|
117
|
+
# Release pull requests change version metadata only; their content was
|
|
118
|
+
# audited on the pull requests that merged into main. Integration, E2E,
|
|
119
|
+
# smoke, and performance lanes stay in the audit and scheduled lanes.
|
|
120
|
+
unit-only: ${{ inputs.mode == 'fast' || inputs.mode == 'release' }}
|
|
110
121
|
artifact-prefix: ${{ inputs.artifact-prefix }}
|
|
111
122
|
secrets:
|
|
112
123
|
TURBO_TOKEN: ${{ secrets.TURBO_TOKEN }}
|
|
113
124
|
|
|
114
125
|
security:
|
|
115
126
|
name: Security
|
|
116
|
-
if: vars.CI_BILLING_PAUSED != 'true' &&
|
|
127
|
+
if: vars.CI_BILLING_PAUSED != 'true' && inputs.mode == 'audit'
|
|
117
128
|
uses: ./.github/workflows/security.yml
|
|
118
129
|
with:
|
|
119
130
|
runtime-repository: ${{ inputs.runtime-repository }}
|
|
@@ -134,15 +145,30 @@ jobs:
|
|
|
134
145
|
rust-threads: ${{ inputs.rust-threads }}
|
|
135
146
|
rust-max-parallel: ${{ inputs.rust-max-parallel }}
|
|
136
147
|
|
|
148
|
+
eval:
|
|
149
|
+
name: Eval
|
|
150
|
+
# Behavior evals cover content changes; the lean release lane skips them
|
|
151
|
+
# because its diff is version metadata only. The reusable workflow skips
|
|
152
|
+
# cleanly when the repository has no eval harness.
|
|
153
|
+
if: vars.CI_BILLING_PAUSED != 'true' && inputs.mode == 'audit'
|
|
154
|
+
uses: ./.github/workflows/eval.yml
|
|
155
|
+
with:
|
|
156
|
+
runtime-repository: ${{ inputs.runtime-repository }}
|
|
157
|
+
runtime-ref: ${{ inputs.runtime-ref }}
|
|
158
|
+
runner: ${{ inputs.eval-runner }}
|
|
159
|
+
artifact-prefix: ${{ inputs.artifact-prefix }}
|
|
160
|
+
|
|
137
161
|
# Aggregate check: "Validation / Gate" (caller job name + this job name).
|
|
138
162
|
# Always evaluates; only jobs required for the mode must have succeeded.
|
|
139
|
-
# The release tier
|
|
140
|
-
#
|
|
163
|
+
# The release tier requires the fast suite plus CodeQL: its diff is version
|
|
164
|
+
# metadata (validated against the release policy below), its content was
|
|
165
|
+
# audited on the pull requests that merged into main, and the scheduled
|
|
166
|
+
# audit lane re-covers drift. Its generated release diff policy executes
|
|
141
167
|
# as a conditional step below, so ordinary pull requests still render no
|
|
142
168
|
# skipped release-policy row.
|
|
143
169
|
gate:
|
|
144
170
|
name: Gate
|
|
145
|
-
needs: [ci, test, security, codeql]
|
|
171
|
+
needs: [ci, test, security, codeql, eval]
|
|
146
172
|
if: vars.CI_BILLING_PAUSED != 'true' && always()
|
|
147
173
|
runs-on: ubuntu-slim
|
|
148
174
|
timeout-minutes: 10
|
|
@@ -152,6 +178,7 @@ jobs:
|
|
|
152
178
|
FOUNDRY_TEST: ${{ needs.test.result }}
|
|
153
179
|
FOUNDRY_SECURITY: ${{ needs.security.result }}
|
|
154
180
|
FOUNDRY_CODEQL: ${{ needs.codeql.result }}
|
|
181
|
+
FOUNDRY_EVAL: ${{ needs.eval.result }}
|
|
155
182
|
steps:
|
|
156
183
|
- name: Checkout runtime
|
|
157
184
|
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
|
@@ -64,7 +64,12 @@ jobs:
|
|
|
64
64
|
validation:
|
|
65
65
|
name: Validation
|
|
66
66
|
needs: mode
|
|
67
|
-
|
|
67
|
+
# Bot-authored pull requests (Dependabot and friends) validate only on an
|
|
68
|
+
# explicit ready transition: dependency pushes never allocate validation
|
|
69
|
+
# runners by themselves. A human pushing to a bot branch also runs, since
|
|
70
|
+
# the sender is no longer a bot. Release Please heads are managed by the
|
|
71
|
+
# release workflow and always validate.
|
|
72
|
+
if: vars.CI_BILLING_PAUSED != 'true' && github.event_name == 'pull_request' && github.event.pull_request.draft == false && (github.event.pull_request.user.type != 'Bot' || github.event.sender.type != 'Bot' || startsWith(github.event.pull_request.head.ref, 'release-please--branches--main'))
|
|
68
73
|
# Reusable workflows can only maintain or reduce the caller job's scopes.
|
|
69
74
|
# The audit tier needs security-events: write for CodeQL uploads.
|
|
70
75
|
permissions:
|
|
@@ -85,6 +90,7 @@ jobs:
|
|
|
85
90
|
unit-runner: ubuntu-slim
|
|
86
91
|
performance-runner: ubuntu-latest
|
|
87
92
|
security-runner: ubuntu-slim
|
|
93
|
+
eval-runner: ubuntu-latest
|
|
88
94
|
codeql-runner: ubuntu-latest
|
|
89
95
|
rust-shards: '["all"]'
|
|
90
96
|
rust-threads: '1'
|
package/AGENTS.md
CHANGED
|
@@ -17,6 +17,15 @@ They complement `CONTRIBUTING.md`. More specific instructions in nested `AGENTS.
|
|
|
17
17
|
|
|
18
18
|
This repository may contain TypeScript, Rust, Python, or any combination of them. Detect the active stack from the files present; do not assume every check applies.
|
|
19
19
|
|
|
20
|
+
## Contributing back
|
|
21
|
+
|
|
22
|
+
Consumers are encouraged to help improve this open-source project. Open a small,
|
|
23
|
+
focused pull request for bug fixes, performance improvements, documentation, or
|
|
24
|
+
other narrowly scoped changes. For larger feature requests or architectural
|
|
25
|
+
changes, create an issue first so the proposal can be discussed and scoped.
|
|
26
|
+
Contributions should help make the tool as performant, reliable, and helpful as
|
|
27
|
+
possible for everyone.
|
|
28
|
+
|
|
20
29
|
## Read before acting
|
|
21
30
|
|
|
22
31
|
Before editing:
|
|
@@ -149,6 +158,7 @@ node src/runtime.mjs ci unit
|
|
|
149
158
|
node src/runtime.mjs ci integration
|
|
150
159
|
node src/runtime.mjs ci e2e
|
|
151
160
|
node src/runtime.mjs ci smoke
|
|
161
|
+
node src/runtime.mjs ci eval
|
|
152
162
|
node src/runtime.mjs ci performance
|
|
153
163
|
Security and dependency audits run through the GitHub Security workflow.
|
|
154
164
|
```
|
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,82 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## [1.28.2](https://github.com/0xPlayerOne/code-foundry/compare/v1.28.1...v1.28.2) (2026-09-10)
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
### Bug Fixes
|
|
7
|
+
|
|
8
|
+
* **release:** pass one qualification report per required node ([#597](https://github.com/0xPlayerOne/code-foundry/issues/597)) ([5ceb92c](https://github.com/0xPlayerOne/code-foundry/commit/5ceb92cab3afc97145ef068e6b5e33399d99b9e5))
|
|
9
|
+
|
|
10
|
+
## [1.28.1](https://github.com/0xPlayerOne/code-foundry/compare/v1.28.0...v1.28.1) (2026-09-10)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
### Bug Fixes
|
|
14
|
+
|
|
15
|
+
* **cloudflare:** remove duplicate deployment records ([#595](https://github.com/0xPlayerOne/code-foundry/issues/595)) ([d9a2a4b](https://github.com/0xPlayerOne/code-foundry/commit/d9a2a4b0363bb9a9ded91d3368bdb7b9b580330f))
|
|
16
|
+
|
|
17
|
+
## [1.28.0](https://github.com/0xPlayerOne/code-foundry/compare/v1.27.0...v1.28.0) (2026-09-10)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
### Features
|
|
21
|
+
|
|
22
|
+
* **release:** qualify only when a release or a stuck draft needs it ([#591](https://github.com/0xPlayerOne/code-foundry/issues/591)) ([4cce045](https://github.com/0xPlayerOne/code-foundry/commit/4cce045228648e6e705ad1e89c59eaf20a72a538))
|
|
23
|
+
|
|
24
|
+
## [1.27.0](https://github.com/0xPlayerOne/code-foundry/compare/v1.26.0...v1.27.0) (2026-09-10)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
### Features
|
|
28
|
+
|
|
29
|
+
* **validation:** downgrade docs-only PRs to fast and gate bot pushes ([#590](https://github.com/0xPlayerOne/code-foundry/issues/590)) ([df774f5](https://github.com/0xPlayerOne/code-foundry/commit/df774f5d1e4a8a6c5f174219eb2b24ef012f6fd3))
|
|
30
|
+
|
|
31
|
+
## [1.26.0](https://github.com/0xPlayerOne/code-foundry/compare/v1.25.3...v1.26.0) (2026-09-10)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
### Features
|
|
35
|
+
|
|
36
|
+
* **node:** support Node 24 and 26, drop 20 and 22 ([#589](https://github.com/0xPlayerOne/code-foundry/issues/589)) ([ee3e294](https://github.com/0xPlayerOne/code-foundry/commit/ee3e2943889ffd06a8d4cd4ccb75a3f36c46c0fd))
|
|
37
|
+
|
|
38
|
+
## [1.25.3](https://github.com/0xPlayerOne/code-foundry/compare/v1.25.2...v1.25.3) (2026-09-10)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
### Bug Fixes
|
|
42
|
+
|
|
43
|
+
* **release:** cover the attestation lag in publication retries ([#587](https://github.com/0xPlayerOne/code-foundry/issues/587)) ([7ae5e4b](https://github.com/0xPlayerOne/code-foundry/commit/7ae5e4bb95f68cbf4e2b48778776a0396c877271))
|
|
44
|
+
|
|
45
|
+
## [1.25.2](https://github.com/0xPlayerOne/code-foundry/compare/v1.25.1...v1.25.2) (2026-09-10)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
### Bug Fixes
|
|
49
|
+
|
|
50
|
+
* **release:** retry integrity verification through the publication consistency window ([#585](https://github.com/0xPlayerOne/code-foundry/issues/585)) ([5b997dd](https://github.com/0xPlayerOne/code-foundry/commit/5b997dda7e272b59a386eeeda115e60ad284f6ce))
|
|
51
|
+
|
|
52
|
+
## [1.25.1](https://github.com/0xPlayerOne/code-foundry/compare/v1.25.0...v1.25.1) (2026-09-10)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
### Bug Fixes
|
|
56
|
+
|
|
57
|
+
* **release:** wait for the release index before verifying a published release ([#583](https://github.com/0xPlayerOne/code-foundry/issues/583)) ([7823c14](https://github.com/0xPlayerOne/code-foundry/commit/7823c1473d8c5e7c17b039dc755ff5c8d1dba39c))
|
|
58
|
+
|
|
59
|
+
## [1.25.0](https://github.com/0xPlayerOne/code-foundry/compare/v1.24.0...v1.25.0) (2026-09-10)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
### Features
|
|
63
|
+
|
|
64
|
+
* **eval:** run the eval tier as a managed Validation / Eval lane ([#580](https://github.com/0xPlayerOne/code-foundry/issues/580)) ([9414889](https://github.com/0xPlayerOne/code-foundry/commit/94148894da9a627a6bd7c71c846848c198c1dfec))
|
|
65
|
+
|
|
66
|
+
## [1.24.0](https://github.com/0xPlayerOne/code-foundry/compare/v1.23.0...v1.24.0) (2026-09-10)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
### Features
|
|
70
|
+
|
|
71
|
+
* **validation:** run a lean release lane for Release Please pull requests ([#579](https://github.com/0xPlayerOne/code-foundry/issues/579)) ([a38ca75](https://github.com/0xPlayerOne/code-foundry/commit/a38ca75c7da577af8368467b8b026d0ec563fc41))
|
|
72
|
+
|
|
73
|
+
## [1.23.0](https://github.com/0xPlayerOne/code-foundry/compare/v1.22.1...v1.23.0) (2026-09-09)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
### Features
|
|
77
|
+
|
|
78
|
+
* add ci eval tier with shared report contract and budget gate ([#576](https://github.com/0xPlayerOne/code-foundry/issues/576)) ([7d881dd](https://github.com/0xPlayerOne/code-foundry/commit/7d881dd1bc6ace8e48dffff45b1a22a12bdae1c9))
|
|
79
|
+
|
|
3
80
|
## [1.22.1](https://github.com/0xPlayerOne/code-foundry/compare/v1.22.0...v1.22.1) (2026-09-09)
|
|
4
81
|
|
|
5
82
|
|
package/README.md
CHANGED
|
@@ -39,6 +39,9 @@ contract. For normal updates, edit that file and run `npx code-foundry sync`.
|
|
|
39
39
|
audits, Draft Guard, Draft PR, Release PR, and Release.
|
|
40
40
|
- A deterministic performance lane with ordered command support, stable result
|
|
41
41
|
artifacts, and an optional shared Node package budget profile.
|
|
42
|
+
- A deterministic eval tier (`ci eval`) that runs a repository's behavior-eval
|
|
43
|
+
harness against a shared report contract with optional budgets, keeping task
|
|
44
|
+
outcomes comparable across revisions and executors.
|
|
42
45
|
- An opt-in product-quality runner for static sites, web apps, Workers, and
|
|
43
46
|
published packages.
|
|
44
47
|
- A small `.githooks/pre-commit` launcher with language-aware formatting and
|
|
@@ -132,7 +135,7 @@ is merged; npm publication is opt-in through `npm_publish: true` and supports
|
|
|
132
135
|
npm trusted publishing or an `NPM_TOKEN` fallback.
|
|
133
136
|
|
|
134
137
|
Code Foundry's own release caller is stricter: it qualifies the package across
|
|
135
|
-
Node
|
|
138
|
+
Node 24 and 26, stages the exact qualified archive, publishes the
|
|
136
139
|
immutable GitHub Release, and publishes that archive through the verified
|
|
137
140
|
publisher. See
|
|
138
141
|
[Consumer qualification](docs/consumer-qualification.md) and [Qualified
|
package/docs/CONFIGURATION.md
CHANGED
|
@@ -89,22 +89,29 @@ See [Merge queue validation](merge-queues.md) before enabling it.
|
|
|
89
89
|
|
|
90
90
|
## Validation and quality
|
|
91
91
|
|
|
92
|
-
| Key | Values | Purpose
|
|
93
|
-
| ------------------------- | ---------------------------------------------- |
|
|
94
|
-
| `performance` | `auto`, `true`, `false` | Discover, require, or disable performance checks.
|
|
95
|
-
| `performance_command` | JSON argv array or array of argv arrays | Ordered commands for a non-package performance harness.
|
|
96
|
-
| `performance_profile` | empty or `node-package` | Shared package import, memory, archive, and dependency audit.
|
|
97
|
-
| `performance_budget_file` | repository-relative path | Budget file for `node-package`; defaults to `performance-package-budgets.json`.
|
|
98
|
-
| `
|
|
99
|
-
| `
|
|
100
|
-
| `
|
|
101
|
-
| `
|
|
102
|
-
| `
|
|
92
|
+
| Key | Values | Purpose |
|
|
93
|
+
| ------------------------- | ---------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
94
|
+
| `performance` | `auto`, `true`, `false` | Discover, require, or disable performance checks. |
|
|
95
|
+
| `performance_command` | JSON argv array or array of argv arrays | Ordered commands for a non-package performance harness. |
|
|
96
|
+
| `performance_profile` | empty or `node-package` | Shared package import, memory, archive, and dependency audit. |
|
|
97
|
+
| `performance_budget_file` | repository-relative path | Budget file for `node-package`; defaults to `performance-package-budgets.json`. |
|
|
98
|
+
| `eval` | `auto`, `true`, `false` | Discover, require, or disable the deterministic eval tier. |
|
|
99
|
+
| `eval_command` | JSON argv array | Explicit harness command when there is no `eval` package script. |
|
|
100
|
+
| `eval_report_file` | repository-relative path | Report validated against the eval contract; defaults to `eval-results/result.json`. |
|
|
101
|
+
| `eval_budget_file` | repository-relative path | Optional budget file; defaults to `eval-budgets.json`. |
|
|
102
|
+
| `eval_runner` | runner label | Runner for the `Validation / Eval` lane; defaults to the repository runner. Browser evals need a Chrome-capable runner such as `ubuntu-latest`. |
|
|
103
|
+
| `required_capabilities` | comma-separated task names | Fail closed when a required task or coverage evidence is unavailable. |
|
|
104
|
+
| `coverage_enforcement` | `auto`, `required`, `off` | Shared coverage-report policy. |
|
|
105
|
+
| `coverage_minimum` | `0`–`100` | Minimum percentage; defaults to `80`. |
|
|
106
|
+
| `coverage_metrics` | `lines`, `functions`, `branches`, `statements` | Metrics checked by the coverage gate. |
|
|
107
|
+
| `coverage_report` | comma-separated repository paths | Istanbul JSON summary or LCOV evidence files. |
|
|
103
108
|
|
|
104
109
|
Supported task capabilities are `format`, `lint`, `type_check`, `build`, `unit`,
|
|
105
|
-
`integration`, `e2e`, `smoke`, and `performance`. `coverage` is a policy
|
|
110
|
+
`integration`, `e2e`, `smoke`, `eval`, and `performance`. `coverage` is a policy
|
|
106
111
|
capability that also requires unit tests. See [Required capabilities and task
|
|
107
|
-
evidence](required-capabilities.md).
|
|
112
|
+
evidence](required-capabilities.md). The eval tier runs the repository's own
|
|
113
|
+
harness against the shared report contract and optional budgets; see
|
|
114
|
+
[Evals](EVALS.md).
|
|
108
115
|
|
|
109
116
|
The shared performance job discovers `performance:check`, then `perf:check`,
|
|
110
117
|
in JavaScript repositories. Other repositories can provide one command or an
|
package/docs/EVALS.md
ADDED
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
# Evals
|
|
2
|
+
|
|
3
|
+
`ci eval` is the repository's deterministic behavior-evaluation tier. It runs the
|
|
4
|
+
repository's own eval harness and consumes its result against a shared contract,
|
|
5
|
+
so task outcomes stay comparable across revisions and — later — across
|
|
6
|
+
executors (deterministic today, model-agent when a repository adopts one).
|
|
7
|
+
|
|
8
|
+
Evals are tests' measurement-oriented sibling: tests verify a specified
|
|
9
|
+
behavior (pass/fail); evals record how well the system does at repeated probes
|
|
10
|
+
(success rates, timing percentiles, bounded evidence) and compare those numbers
|
|
11
|
+
against pinned baselines. The `performance` tier is the resource-metric
|
|
12
|
+
special case of the same pattern.
|
|
13
|
+
|
|
14
|
+
## How the runtime finds your harness
|
|
15
|
+
|
|
16
|
+
`ci eval` discovers its subject in this order and skips when neither exists:
|
|
17
|
+
|
|
18
|
+
1. A package script named `eval`.
|
|
19
|
+
2. An explicit `eval_command` (JSON argv array) in `.github/code-foundry.yml`.
|
|
20
|
+
|
|
21
|
+
Both mechanisms use `eval: auto` by default: the tier runs when a subject is
|
|
22
|
+
present and skips cleanly otherwise. Set `eval: true` to require it (the
|
|
23
|
+
validation policy then treats a missing subject as an error), or `eval: false`
|
|
24
|
+
to disable discovery.
|
|
25
|
+
|
|
26
|
+
In managed validation the eval tier runs as its own `Validation / Eval` lane
|
|
27
|
+
during audit-mode runs, with the receipt retained as a task artifact. Browser
|
|
28
|
+
eval harnesses need a Chrome-capable runner: set `eval_runner: ubuntu-latest`
|
|
29
|
+
in `.github/code-foundry.yml` (the default when unset is the repository's
|
|
30
|
+
default runner, which may not ship a browser). The lane uploads the eval
|
|
31
|
+
report alongside the task receipt whenever a run fails, so reviewers get the
|
|
32
|
+
measured numbers with the red check. Performance's task profile and gating
|
|
33
|
+
rules apply unchanged; evals are an optional, non-gating tier like
|
|
34
|
+
performance.
|
|
35
|
+
|
|
36
|
+
| Config key | Values | Meaning |
|
|
37
|
+
| ------------------ | --------------------------------- | ---------------------------------------------------------------- |
|
|
38
|
+
| `eval` | `auto` (default), `true`, `false` | Enable, require, or disable the eval tier. |
|
|
39
|
+
| `eval_command` | JSON argv array | Explicit harness command when there is no `eval` package script. |
|
|
40
|
+
| `eval_report_file` | repository-relative path | The report to validate; defaults to `eval-results/result.json`. |
|
|
41
|
+
| `eval_budget_file` | repository-relative path | Optional budget file; defaults to `eval-budgets.json`. |
|
|
42
|
+
|
|
43
|
+
## The report contract
|
|
44
|
+
|
|
45
|
+
Your harness writes `eval-results/result.json`. The runtime validates the
|
|
46
|
+
envelope before budgets; a report that violates the contract fails the tier
|
|
47
|
+
with every violation listed (fail closed, typos can never pass silently):
|
|
48
|
+
|
|
49
|
+
```jsonc
|
|
50
|
+
{
|
|
51
|
+
"schemaVersion": 1, // contract version, currently 1
|
|
52
|
+
"revision": "<git sha>", // optional but strongly recommended
|
|
53
|
+
"dependencyHash": "<sha256>", // optional; pins the dependency set
|
|
54
|
+
"summary": {
|
|
55
|
+
"taskCount": 7,
|
|
56
|
+
"attempts": 7,
|
|
57
|
+
"passed": 7,
|
|
58
|
+
"failed": 0,
|
|
59
|
+
"harnessFailures": 0, // environment broke; not a task regression
|
|
60
|
+
"successRate": 1.0,
|
|
61
|
+
"toolCalls": 31,
|
|
62
|
+
"evidenceErrors": 0,
|
|
63
|
+
"taskDurationMs": { "count": 7, "mean": 2.1, "p50": 1.8, "p95": 3.0, "max": 3.4 },
|
|
64
|
+
"startupMs": { "count": 7, "mean": 0.3, "p50": 0.3, "p95": 0.4, "max": 0.4 },
|
|
65
|
+
"stepDurationMs": { "count": 31, "mean": 0.2, "p50": 0.1, "p95": 0.6, "max": 0.6 },
|
|
66
|
+
},
|
|
67
|
+
"tasks": [
|
|
68
|
+
{
|
|
69
|
+
"id": "form-submit",
|
|
70
|
+
"attempts": [
|
|
71
|
+
{
|
|
72
|
+
"iteration": 1,
|
|
73
|
+
"status": "passed",
|
|
74
|
+
"durationMs": 2.1,
|
|
75
|
+
"startupMs": 0.3,
|
|
76
|
+
"steps": [{ "tool": "fill_form", "status": "passed", "durationMs": 0.4 }],
|
|
77
|
+
"checks": ["fill_form completed"],
|
|
78
|
+
"metrics": { "formFields": 2 },
|
|
79
|
+
},
|
|
80
|
+
],
|
|
81
|
+
},
|
|
82
|
+
],
|
|
83
|
+
}
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
Rules the validator enforces:
|
|
87
|
+
|
|
88
|
+
- `schemaVersion` must match the contract version; `revision`/`dependencyHash`
|
|
89
|
+
must be strings when present.
|
|
90
|
+
- Summary counts are non-negative integers; `passed + failed` cannot exceed
|
|
91
|
+
`attempts`; `successRate` is between 0 and 1.
|
|
92
|
+
- `harnessFailures` counts attempts where the environment itself broke (browser
|
|
93
|
+
never started, zero steps succeeded, cleanup failed). It must not exceed
|
|
94
|
+
`attempts`. Fix the environment; do not treat these as task regressions.
|
|
95
|
+
- Failed attempts carry a bounded `failure` (non-empty `message`) plus a
|
|
96
|
+
`failureClass` of `harness` or `task`.
|
|
97
|
+
- `taskDurationMs`, `startupMs`, and `stepDurationMs` are stats objects with
|
|
98
|
+
`count`, `mean`, `p50`, `p95`, `max` (or bare `{ "count": 0 }`).
|
|
99
|
+
|
|
100
|
+
A reference implementation ships in
|
|
101
|
+
[`pi-browser-use`](https://github.com/0xPlayerOne/pi-browser-use)
|
|
102
|
+
(`scripts/eval.mjs` + `docs/eval-results.md`): deterministic browser-tool tasks
|
|
103
|
+
that emit this exact envelope.
|
|
104
|
+
|
|
105
|
+
## Budgets
|
|
106
|
+
|
|
107
|
+
Create `eval-budgets.json` (or point `eval_budget_file` at another file) to
|
|
108
|
+
gate on the measured numbers:
|
|
109
|
+
|
|
110
|
+
```json
|
|
111
|
+
{
|
|
112
|
+
"successRate": 1.0,
|
|
113
|
+
"stepP95Ms": 1000,
|
|
114
|
+
"taskP95Ms": 45000,
|
|
115
|
+
"maxHarnessFailures": 0
|
|
116
|
+
}
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
Supported budgets: `successRate`, `taskP95Ms`, `startupP95Ms`, `stepP95Ms`,
|
|
120
|
+
`maxHarnessFailures`, `maxEvidenceErrors`, `maxToolCalls`. Unknown keys fail
|
|
121
|
+
closed. When the budget file is absent the tier validates the contract only.
|
|
122
|
+
Budget files are committed configuration; only `eval-results/` is a local
|
|
123
|
+
artifact.
|
|
124
|
+
|
|
125
|
+
The runtime writes `eval-results/summary.json` with the executed commands, the
|
|
126
|
+
budget outcome, and the artifact list, mirroring the performance summary.
|
|
127
|
+
|
|
128
|
+
## Determinism rules for eval tasks
|
|
129
|
+
|
|
130
|
+
- Fixed fixtures and no network. A task's outcome must depend only on the code
|
|
131
|
+
under test.
|
|
132
|
+
- Bounded evidence: failure text and screenshots are capped; a failing task
|
|
133
|
+
never produces unbounded output.
|
|
134
|
+
- Cleanup failures are surfaced, never swallowed silently.
|
|
135
|
+
- Every attempt records the revision and dependency hash it ran against.
|
|
136
|
+
- Keep model-in-the-loop runs out of this tier. Reuse the same task IDs in a
|
|
137
|
+
separate scheduled harness with frozen model/reasoning/judge baselines when
|
|
138
|
+
you need agent-capability measurement; its variance is why it must never
|
|
139
|
+
gate pull requests.
|
package/docs/PUBLISHING.md
CHANGED
|
@@ -37,7 +37,7 @@ registry version and its provenance link before treating the setup as complete.
|
|
|
37
37
|
The self-hosted package follows a stricter contract than generated consumer
|
|
38
38
|
callers:
|
|
39
39
|
|
|
40
|
-
1. Pack the candidate once and qualify that archive across Node
|
|
40
|
+
1. Pack the candidate once and qualify that archive across Node 24 and 26.
|
|
41
41
|
2. Create a draft GitHub Release and attach the exact qualified archive plus its
|
|
42
42
|
qualification receipt.
|
|
43
43
|
3. Publish the immutable GitHub Release after verifying the tag, source commit,
|
|
@@ -50,6 +50,11 @@ the bytes selected from the same workflow run and attempt. See [Consumer
|
|
|
50
50
|
qualification](consumer-qualification.md) and [Qualified publication](qualified-publication.md)
|
|
51
51
|
for the complete contract, retries, and recovery rules.
|
|
52
52
|
|
|
53
|
+
The main-push pipeline runs Release Please first, then qualifies only when the
|
|
54
|
+
push created a release or recovery found a stuck draft. Feature merges pay for
|
|
55
|
+
the cheap release-please and recovery probes; the runner-heavy matrix runs on
|
|
56
|
+
release merges, which re-qualify the exact tree they publish.
|
|
57
|
+
|
|
53
58
|
## GitHub Releases and GitHub Packages
|
|
54
59
|
|
|
55
60
|
A GitHub Release is metadata attached to a Git tag. It is independent of npm and
|
package/docs/WORKFLOWS.md
CHANGED
|
@@ -27,7 +27,24 @@ ready transition. Release Please version heads are excluded because the
|
|
|
27
27
|
release workflow owns their state. A separate draft-control caller listens for
|
|
28
28
|
`converted_to_draft` and cancels queued or running pull-request workflows.
|
|
29
29
|
Marking a pull request ready starts validation, and each new commit on a ready
|
|
30
|
-
pull request starts it again for the current head.
|
|
30
|
+
pull request starts it again for the current head. Audit-mode runs additionally
|
|
31
|
+
execute the eval lane (`Validation / Eval`) for repositories that ship an eval
|
|
32
|
+
harness; see [Evals](EVALS.md).
|
|
33
|
+
|
|
34
|
+
Two waste-avoidance rules keep validation minutes honest without weakening
|
|
35
|
+
confidence:
|
|
36
|
+
|
|
37
|
+
- **Docs-only pull requests run the fast tier.** When an audit-classified pull
|
|
38
|
+
request changes only markdown, `docs/`, and license roots, the mode
|
|
39
|
+
classifier downgrades it to fast. Any code, lockfile, workflow,
|
|
40
|
+
configuration, or packaging change — or any diff that cannot be computed
|
|
41
|
+
deterministically — keeps the audit tier. Scheduled and manual audits always
|
|
42
|
+
run the full tier.
|
|
43
|
+
- **Bot-authored pull requests validate only on ready transitions.**
|
|
44
|
+
Dependabot and other bot pushes never allocate validation runners; a
|
|
45
|
+
maintainer marks the (Guard-drafted) pull request ready — or pushes to its
|
|
46
|
+
branch, which changes the sender — to run validation. Release Please heads
|
|
47
|
+
are exempt because the release workflow merges them through its own lane.
|
|
31
48
|
|
|
32
49
|
The separate `validation-audit.yml` caller is pinned to the configured released
|
|
33
50
|
runtime and handles scheduled and manual audits:
|
|
@@ -40,9 +57,15 @@ workflow_dispatch:
|
|
|
40
57
|
|
|
41
58
|
In the `staging-release` topology, pull requests into `staging` run the fast
|
|
42
59
|
tier, ordinary pull requests into `main` run the full audit tier, and exact
|
|
43
|
-
Release Please pull requests into `main` run
|
|
44
|
-
release-diff policy.
|
|
45
|
-
|
|
60
|
+
Release Please pull requests into `main` run a lean release lane — CI, unit
|
|
61
|
+
tests, and CodeQL plus the release-diff policy. Their diff is version
|
|
62
|
+
metadata only: the content was fully audited on the pull requests that
|
|
63
|
+
merged into `main`, and the scheduled audit lane re-covers drift, so the
|
|
64
|
+
release lane keeps repository rulesets satisfiable without re-running the
|
|
65
|
+
runner-heavy suites. The managed branch rulesets require only the aggregate
|
|
66
|
+
`Validation / Gate` check, so skipped non-required jobs never deadlock the
|
|
67
|
+
release; do not hand-require individual job contexts on release branches.
|
|
68
|
+
In the
|
|
46
69
|
`direct` topology (the default) every pull request targets `main` and runs the
|
|
47
70
|
full audit tier, because there is no integration branch for a fast pass.
|
|
48
71
|
Scheduled and manual runs select the audit tier in both topologies. Draft PR
|