code-foundry 1.22.1 → 1.25.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/workflows/eval.yml +116 -0
- package/.github/workflows/validation-no-codeql.yml +34 -6
- package/.github/workflows/validation.yml +32 -5
- package/.github/workflows/validation_audit_self-ci.yml +1 -0
- package/.github/workflows/validation_self-ci.yml +1 -0
- package/AGENTS.md +1 -0
- package/CHANGELOG.md +42 -0
- package/README.md +3 -0
- package/docs/CONFIGURATION.md +20 -13
- package/docs/EVALS.md +139 -0
- package/docs/WORKFLOWS.md +12 -4
- package/docs/required-capabilities.md +6 -4
- package/package.json +1 -1
- package/src/commands/qualified-publication.mjs +80 -2
- package/src/commands/release-integrity.mjs +37 -14
- package/src/commands/sync.mjs +1 -0
- package/src/lib/eval-envelope.mjs +234 -0
- package/src/lib/merge-queue.mjs +1 -0
- package/src/lib/task-policy.mjs +7 -0
- package/src/lib/validation-policy.mjs +10 -6
- package/src/runtime-core.mjs +154 -0
- package/src/runtime.mjs +2 -0
- package/src/templates/gitignore +2 -0
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
name: Code Foundry Eval
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
workflow_call:
|
|
5
|
+
inputs:
|
|
6
|
+
runtime-repository:
|
|
7
|
+
description: Repository containing the Code Foundry runtime.
|
|
8
|
+
required: false
|
|
9
|
+
type: string
|
|
10
|
+
default: 0xPlayerOne/code-foundry
|
|
11
|
+
runtime-ref:
|
|
12
|
+
description: Code Foundry runtime tag or ref.
|
|
13
|
+
required: false
|
|
14
|
+
type: string
|
|
15
|
+
default: v1.0.5
|
|
16
|
+
runner:
|
|
17
|
+
description: >-
|
|
18
|
+
Runner used by the eval harness. Browser evals need a
|
|
19
|
+
Chrome-capable runner such as ubuntu-latest; ubuntu-slim does not
|
|
20
|
+
ship a browser.
|
|
21
|
+
required: false
|
|
22
|
+
type: string
|
|
23
|
+
default: ubuntu-latest
|
|
24
|
+
artifact-prefix:
|
|
25
|
+
description: Prefix for task receipts; use a unique value when calling this workflow more than once.
|
|
26
|
+
required: false
|
|
27
|
+
type: string
|
|
28
|
+
default: task-result
|
|
29
|
+
|
|
30
|
+
permissions:
|
|
31
|
+
contents: read
|
|
32
|
+
|
|
33
|
+
env:
|
|
34
|
+
TURBO_TOKEN: ${{ secrets.TURBO_TOKEN }}
|
|
35
|
+
TURBO_TEAM: ${{ vars.TURBO_TEAM }}
|
|
36
|
+
REPO_FOUNDRY_PROFILE: ${{ vars.REPO_FOUNDRY_PROFILE }}
|
|
37
|
+
REPO_FOUNDRY_LANGUAGES: ${{ vars.REPO_FOUNDRY_LANGUAGES }}
|
|
38
|
+
REPO_FOUNDRY_FEATURES: ${{ vars.REPO_FOUNDRY_FEATURES }}
|
|
39
|
+
REPO_FOUNDRY_PACKAGE_MANAGER: ${{ vars.REPO_FOUNDRY_PACKAGE_MANAGER }}
|
|
40
|
+
REPO_FOUNDRY_RUNNER: ${{ vars.REPO_FOUNDRY_RUNNER }}
|
|
41
|
+
REPO_FOUNDRY_CACHE_PACKAGES: ${{ vars.REPO_FOUNDRY_CACHE_PACKAGES || 'auto' }}
|
|
42
|
+
REPO_FOUNDRY_CACHE_BUILD: ${{ vars.REPO_FOUNDRY_CACHE_BUILD || 'auto' }}
|
|
43
|
+
|
|
44
|
+
concurrency:
|
|
45
|
+
group: code-foundry-eval-${{ github.event_name }}-${{ github.event.pull_request.head.repo.full_name || github.repository }}-${{ github.event.pull_request.head.ref || github.ref_name }}
|
|
46
|
+
cancel-in-progress: true
|
|
47
|
+
|
|
48
|
+
jobs:
|
|
49
|
+
eval:
|
|
50
|
+
name: Eval
|
|
51
|
+
if: vars.CI_BILLING_PAUSED != 'true'
|
|
52
|
+
runs-on: ${{ inputs.runner }}
|
|
53
|
+
steps:
|
|
54
|
+
- name: Checkout
|
|
55
|
+
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
|
56
|
+
with:
|
|
57
|
+
persist-credentials: false
|
|
58
|
+
- name: Runtime
|
|
59
|
+
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
|
60
|
+
with:
|
|
61
|
+
persist-credentials: false
|
|
62
|
+
repository: ${{ inputs.runtime-repository }}
|
|
63
|
+
ref: ${{ inputs.runtime-ref }}
|
|
64
|
+
path: .github/.code-foundry
|
|
65
|
+
sparse-checkout: |
|
|
66
|
+
.github/actions
|
|
67
|
+
src/lib
|
|
68
|
+
src/runtime-core.mjs
|
|
69
|
+
src/runtime.mjs
|
|
70
|
+
- name: Install runtime
|
|
71
|
+
run: |
|
|
72
|
+
mkdir -p .github/actions
|
|
73
|
+
mv .github/.code-foundry "$RUNNER_TEMP/code-foundry"
|
|
74
|
+
cp -R "$RUNNER_TEMP/code-foundry/.github/actions/." .github/actions/
|
|
75
|
+
- name: Detect
|
|
76
|
+
id: applicability
|
|
77
|
+
run: node "$RUNNER_TEMP/code-foundry/src/runtime.mjs" ci should_run eval >> "$GITHUB_OUTPUT"
|
|
78
|
+
- name: Setup
|
|
79
|
+
if: steps.applicability.outputs.applicable == 'true'
|
|
80
|
+
uses: ./.github/actions/setup
|
|
81
|
+
with:
|
|
82
|
+
# Eval harnesses run through the repository's own package scripts and
|
|
83
|
+
# need their dependencies installed before the task executes.
|
|
84
|
+
install: 'true'
|
|
85
|
+
cache-build: 'true'
|
|
86
|
+
task: eval
|
|
87
|
+
- name: Eval
|
|
88
|
+
id: execute
|
|
89
|
+
if: steps.applicability.outputs.applicable == 'true'
|
|
90
|
+
run: node "$RUNNER_TEMP/code-foundry/src/runtime.mjs" ci eval
|
|
91
|
+
- name: Retain task receipt
|
|
92
|
+
if: >-
|
|
93
|
+
always() &&
|
|
94
|
+
(steps.execute.outcome == 'success' || steps.execute.outcome == 'failure' ||
|
|
95
|
+
steps.applicability.outputs.applicable == 'false') &&
|
|
96
|
+
hashFiles('.code-foundry/results/eval.json') != ''
|
|
97
|
+
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
|
98
|
+
with:
|
|
99
|
+
name: ${{ inputs.artifact-prefix }}-${{ github.run_id }}-${{ github.run_attempt }}-eval
|
|
100
|
+
path: .code-foundry/results/eval.json
|
|
101
|
+
include-hidden-files: true
|
|
102
|
+
if-no-files-found: error
|
|
103
|
+
retention-days: 14
|
|
104
|
+
- name: Retain eval report
|
|
105
|
+
if: >-
|
|
106
|
+
always() &&
|
|
107
|
+
steps.execute.outcome == 'failure' &&
|
|
108
|
+
hashFiles('eval-results/result.json') != ''
|
|
109
|
+
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
|
110
|
+
with:
|
|
111
|
+
name: ${{ inputs.artifact-prefix }}-${{ github.run_id }}-${{ github.run_attempt }}-eval-report
|
|
112
|
+
path: |
|
|
113
|
+
eval-results/result.json
|
|
114
|
+
eval-results/summary.json
|
|
115
|
+
if-no-files-found: error
|
|
116
|
+
retention-days: 14
|
|
@@ -42,6 +42,14 @@ on:
|
|
|
42
42
|
required: false
|
|
43
43
|
type: string
|
|
44
44
|
default: ubuntu-slim
|
|
45
|
+
eval-runner:
|
|
46
|
+
description: >-
|
|
47
|
+
Runner used by the eval harness. Browser evals need a
|
|
48
|
+
Chrome-capable runner such as ubuntu-latest; ubuntu-slim does not
|
|
49
|
+
ship a browser.
|
|
50
|
+
required: false
|
|
51
|
+
type: string
|
|
52
|
+
default: ubuntu-latest
|
|
45
53
|
codeql-runner:
|
|
46
54
|
description: Runner used by CodeQL jobs.
|
|
47
55
|
required: false
|
|
@@ -106,28 +114,47 @@ jobs:
|
|
|
106
114
|
runner: ${{ inputs.test-runner }}
|
|
107
115
|
unit-runner: ${{ inputs.unit-runner }}
|
|
108
116
|
performance-runner: ${{ inputs.performance-runner }}
|
|
109
|
-
|
|
117
|
+
# Release pull requests change version metadata only; their content was
|
|
118
|
+
# audited on the pull requests that merged into main. Integration, E2E,
|
|
119
|
+
# smoke, and performance lanes stay in the audit and scheduled lanes.
|
|
120
|
+
unit-only: ${{ inputs.mode == 'fast' || inputs.mode == 'release' }}
|
|
110
121
|
artifact-prefix: ${{ inputs.artifact-prefix }}
|
|
111
122
|
secrets:
|
|
112
123
|
TURBO_TOKEN: ${{ secrets.TURBO_TOKEN }}
|
|
113
124
|
|
|
114
125
|
security:
|
|
115
126
|
name: Security
|
|
116
|
-
if: vars.CI_BILLING_PAUSED != 'true' &&
|
|
127
|
+
if: vars.CI_BILLING_PAUSED != 'true' && inputs.mode == 'audit'
|
|
117
128
|
uses: ./.github/workflows/security.yml
|
|
118
129
|
with:
|
|
119
130
|
runtime-repository: ${{ inputs.runtime-repository }}
|
|
120
131
|
runtime-ref: ${{ inputs.runtime-ref }}
|
|
121
132
|
runner: ${{ inputs.security-runner }}
|
|
122
133
|
|
|
134
|
+
eval:
|
|
135
|
+
name: Eval
|
|
136
|
+
# Behavior evals cover content changes; the lean release lane skips them
|
|
137
|
+
# because its diff is version metadata only. The reusable workflow skips
|
|
138
|
+
# cleanly when the repository has no eval harness.
|
|
139
|
+
if: vars.CI_BILLING_PAUSED != 'true' && inputs.mode == 'audit'
|
|
140
|
+
uses: ./.github/workflows/eval.yml
|
|
141
|
+
with:
|
|
142
|
+
runtime-repository: ${{ inputs.runtime-repository }}
|
|
143
|
+
runtime-ref: ${{ inputs.runtime-ref }}
|
|
144
|
+
runner: ${{ inputs.eval-runner }}
|
|
145
|
+
artifact-prefix: ${{ inputs.artifact-prefix }}
|
|
146
|
+
|
|
123
147
|
# Aggregate check: "Validation / Gate" (caller job name + this job name).
|
|
124
148
|
# Always evaluates; only jobs required for the mode must have succeeded.
|
|
125
|
-
# The release tier
|
|
126
|
-
#
|
|
127
|
-
#
|
|
149
|
+
# The release tier requires the available fast suite; its diff is version
|
|
150
|
+
# metadata (validated against the release policy below), its content was
|
|
151
|
+
# audited on the pull requests that merged into main, and the scheduled
|
|
152
|
+
# audit lane re-covers drift. Its generated release diff policy executes
|
|
153
|
+
# as a conditional step below, so no separate release-policy job renders
|
|
154
|
+
# skipped checks on ordinary pull requests.
|
|
128
155
|
gate:
|
|
129
156
|
name: Gate
|
|
130
|
-
needs: [ci, test, security]
|
|
157
|
+
needs: [ci, test, security, eval]
|
|
131
158
|
if: vars.CI_BILLING_PAUSED != 'true' && always()
|
|
132
159
|
runs-on: ubuntu-slim
|
|
133
160
|
timeout-minutes: 10
|
|
@@ -136,6 +163,7 @@ jobs:
|
|
|
136
163
|
FOUNDRY_CI: ${{ needs.ci.result }}
|
|
137
164
|
FOUNDRY_TEST: ${{ needs.test.result }}
|
|
138
165
|
FOUNDRY_SECURITY: ${{ needs.security.result }}
|
|
166
|
+
FOUNDRY_EVAL: ${{ needs.eval.result }}
|
|
139
167
|
# This workflow is selected only for repositories whose canonical
|
|
140
168
|
# configuration explicitly disables unavailable CodeQL. Treat that
|
|
141
169
|
# policy decision as satisfied without registering a skipped PR check.
|
|
@@ -42,6 +42,14 @@ on:
|
|
|
42
42
|
required: false
|
|
43
43
|
type: string
|
|
44
44
|
default: ubuntu-slim
|
|
45
|
+
eval-runner:
|
|
46
|
+
description: >-
|
|
47
|
+
Runner used by the eval harness. Browser evals need a
|
|
48
|
+
Chrome-capable runner such as ubuntu-latest; ubuntu-slim does not
|
|
49
|
+
ship a browser.
|
|
50
|
+
required: false
|
|
51
|
+
type: string
|
|
52
|
+
default: ubuntu-latest
|
|
45
53
|
codeql-runner:
|
|
46
54
|
description: Runner used by CodeQL jobs.
|
|
47
55
|
required: false
|
|
@@ -106,14 +114,17 @@ jobs:
|
|
|
106
114
|
runner: ${{ inputs.test-runner }}
|
|
107
115
|
unit-runner: ${{ inputs.unit-runner }}
|
|
108
116
|
performance-runner: ${{ inputs.performance-runner }}
|
|
109
|
-
|
|
117
|
+
# Release pull requests change version metadata only; their content was
|
|
118
|
+
# audited on the pull requests that merged into main. Integration, E2E,
|
|
119
|
+
# smoke, and performance lanes stay in the audit and scheduled lanes.
|
|
120
|
+
unit-only: ${{ inputs.mode == 'fast' || inputs.mode == 'release' }}
|
|
110
121
|
artifact-prefix: ${{ inputs.artifact-prefix }}
|
|
111
122
|
secrets:
|
|
112
123
|
TURBO_TOKEN: ${{ secrets.TURBO_TOKEN }}
|
|
113
124
|
|
|
114
125
|
security:
|
|
115
126
|
name: Security
|
|
116
|
-
if: vars.CI_BILLING_PAUSED != 'true' &&
|
|
127
|
+
if: vars.CI_BILLING_PAUSED != 'true' && inputs.mode == 'audit'
|
|
117
128
|
uses: ./.github/workflows/security.yml
|
|
118
129
|
with:
|
|
119
130
|
runtime-repository: ${{ inputs.runtime-repository }}
|
|
@@ -134,15 +145,30 @@ jobs:
|
|
|
134
145
|
rust-threads: ${{ inputs.rust-threads }}
|
|
135
146
|
rust-max-parallel: ${{ inputs.rust-max-parallel }}
|
|
136
147
|
|
|
148
|
+
eval:
|
|
149
|
+
name: Eval
|
|
150
|
+
# Behavior evals cover content changes; the lean release lane skips them
|
|
151
|
+
# because its diff is version metadata only. The reusable workflow skips
|
|
152
|
+
# cleanly when the repository has no eval harness.
|
|
153
|
+
if: vars.CI_BILLING_PAUSED != 'true' && inputs.mode == 'audit'
|
|
154
|
+
uses: ./.github/workflows/eval.yml
|
|
155
|
+
with:
|
|
156
|
+
runtime-repository: ${{ inputs.runtime-repository }}
|
|
157
|
+
runtime-ref: ${{ inputs.runtime-ref }}
|
|
158
|
+
runner: ${{ inputs.eval-runner }}
|
|
159
|
+
artifact-prefix: ${{ inputs.artifact-prefix }}
|
|
160
|
+
|
|
137
161
|
# Aggregate check: "Validation / Gate" (caller job name + this job name).
|
|
138
162
|
# Always evaluates; only jobs required for the mode must have succeeded.
|
|
139
|
-
# The release tier
|
|
140
|
-
#
|
|
163
|
+
# The release tier requires the fast suite plus CodeQL: its diff is version
|
|
164
|
+
# metadata (validated against the release policy below), its content was
|
|
165
|
+
# audited on the pull requests that merged into main, and the scheduled
|
|
166
|
+
# audit lane re-covers drift. Its generated release diff policy executes
|
|
141
167
|
# as a conditional step below, so ordinary pull requests still render no
|
|
142
168
|
# skipped release-policy row.
|
|
143
169
|
gate:
|
|
144
170
|
name: Gate
|
|
145
|
-
needs: [ci, test, security, codeql]
|
|
171
|
+
needs: [ci, test, security, codeql, eval]
|
|
146
172
|
if: vars.CI_BILLING_PAUSED != 'true' && always()
|
|
147
173
|
runs-on: ubuntu-slim
|
|
148
174
|
timeout-minutes: 10
|
|
@@ -152,6 +178,7 @@ jobs:
|
|
|
152
178
|
FOUNDRY_TEST: ${{ needs.test.result }}
|
|
153
179
|
FOUNDRY_SECURITY: ${{ needs.security.result }}
|
|
154
180
|
FOUNDRY_CODEQL: ${{ needs.codeql.result }}
|
|
181
|
+
FOUNDRY_EVAL: ${{ needs.eval.result }}
|
|
155
182
|
steps:
|
|
156
183
|
- name: Checkout runtime
|
|
157
184
|
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
package/AGENTS.md
CHANGED
|
@@ -149,6 +149,7 @@ node src/runtime.mjs ci unit
|
|
|
149
149
|
node src/runtime.mjs ci integration
|
|
150
150
|
node src/runtime.mjs ci e2e
|
|
151
151
|
node src/runtime.mjs ci smoke
|
|
152
|
+
node src/runtime.mjs ci eval
|
|
152
153
|
node src/runtime.mjs ci performance
|
|
153
154
|
Security and dependency audits run through the GitHub Security workflow.
|
|
154
155
|
```
|
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,47 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## [1.25.3](https://github.com/0xPlayerOne/code-foundry/compare/v1.25.2...v1.25.3) (2026-09-10)
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
### Bug Fixes
|
|
7
|
+
|
|
8
|
+
* **release:** cover the attestation lag in publication retries ([#587](https://github.com/0xPlayerOne/code-foundry/issues/587)) ([7ae5e4b](https://github.com/0xPlayerOne/code-foundry/commit/7ae5e4bb95f68cbf4e2b48778776a0396c877271))
|
|
9
|
+
|
|
10
|
+
## [1.25.2](https://github.com/0xPlayerOne/code-foundry/compare/v1.25.1...v1.25.2) (2026-09-10)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
### Bug Fixes
|
|
14
|
+
|
|
15
|
+
* **release:** retry integrity verification through the publication consistency window ([#585](https://github.com/0xPlayerOne/code-foundry/issues/585)) ([5b997dd](https://github.com/0xPlayerOne/code-foundry/commit/5b997dda7e272b59a386eeeda115e60ad284f6ce))
|
|
16
|
+
|
|
17
|
+
## [1.25.1](https://github.com/0xPlayerOne/code-foundry/compare/v1.25.0...v1.25.1) (2026-09-10)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
### Bug Fixes
|
|
21
|
+
|
|
22
|
+
* **release:** wait for the release index before verifying a published release ([#583](https://github.com/0xPlayerOne/code-foundry/issues/583)) ([7823c14](https://github.com/0xPlayerOne/code-foundry/commit/7823c1473d8c5e7c17b039dc755ff5c8d1dba39c))
|
|
23
|
+
|
|
24
|
+
## [1.25.0](https://github.com/0xPlayerOne/code-foundry/compare/v1.24.0...v1.25.0) (2026-09-10)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
### Features
|
|
28
|
+
|
|
29
|
+
* **eval:** run the eval tier as a managed Validation / Eval lane ([#580](https://github.com/0xPlayerOne/code-foundry/issues/580)) ([9414889](https://github.com/0xPlayerOne/code-foundry/commit/94148894da9a627a6bd7c71c846848c198c1dfec))
|
|
30
|
+
|
|
31
|
+
## [1.24.0](https://github.com/0xPlayerOne/code-foundry/compare/v1.23.0...v1.24.0) (2026-09-10)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
### Features
|
|
35
|
+
|
|
36
|
+
* **validation:** run a lean release lane for Release Please pull requests ([#579](https://github.com/0xPlayerOne/code-foundry/issues/579)) ([a38ca75](https://github.com/0xPlayerOne/code-foundry/commit/a38ca75c7da577af8368467b8b026d0ec563fc41))
|
|
37
|
+
|
|
38
|
+
## [1.23.0](https://github.com/0xPlayerOne/code-foundry/compare/v1.22.1...v1.23.0) (2026-09-09)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
### Features
|
|
42
|
+
|
|
43
|
+
* add ci eval tier with shared report contract and budget gate ([#576](https://github.com/0xPlayerOne/code-foundry/issues/576)) ([7d881dd](https://github.com/0xPlayerOne/code-foundry/commit/7d881dd1bc6ace8e48dffff45b1a22a12bdae1c9))
|
|
44
|
+
|
|
3
45
|
## [1.22.1](https://github.com/0xPlayerOne/code-foundry/compare/v1.22.0...v1.22.1) (2026-09-09)
|
|
4
46
|
|
|
5
47
|
|
package/README.md
CHANGED
|
@@ -39,6 +39,9 @@ contract. For normal updates, edit that file and run `npx code-foundry sync`.
|
|
|
39
39
|
audits, Draft Guard, Draft PR, Release PR, and Release.
|
|
40
40
|
- A deterministic performance lane with ordered command support, stable result
|
|
41
41
|
artifacts, and an optional shared Node package budget profile.
|
|
42
|
+
- A deterministic eval tier (`ci eval`) that runs a repository's behavior-eval
|
|
43
|
+
harness against a shared report contract with optional budgets, keeping task
|
|
44
|
+
outcomes comparable across revisions and executors.
|
|
42
45
|
- An opt-in product-quality runner for static sites, web apps, Workers, and
|
|
43
46
|
published packages.
|
|
44
47
|
- A small `.githooks/pre-commit` launcher with language-aware formatting and
|
package/docs/CONFIGURATION.md
CHANGED
|
@@ -89,22 +89,29 @@ See [Merge queue validation](merge-queues.md) before enabling it.
|
|
|
89
89
|
|
|
90
90
|
## Validation and quality
|
|
91
91
|
|
|
92
|
-
| Key | Values | Purpose
|
|
93
|
-
| ------------------------- | ---------------------------------------------- |
|
|
94
|
-
| `performance` | `auto`, `true`, `false` | Discover, require, or disable performance checks.
|
|
95
|
-
| `performance_command` | JSON argv array or array of argv arrays | Ordered commands for a non-package performance harness.
|
|
96
|
-
| `performance_profile` | empty or `node-package` | Shared package import, memory, archive, and dependency audit.
|
|
97
|
-
| `performance_budget_file` | repository-relative path | Budget file for `node-package`; defaults to `performance-package-budgets.json`.
|
|
98
|
-
| `
|
|
99
|
-
| `
|
|
100
|
-
| `
|
|
101
|
-
| `
|
|
102
|
-
| `
|
|
92
|
+
| Key | Values | Purpose |
|
|
93
|
+
| ------------------------- | ---------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
94
|
+
| `performance` | `auto`, `true`, `false` | Discover, require, or disable performance checks. |
|
|
95
|
+
| `performance_command` | JSON argv array or array of argv arrays | Ordered commands for a non-package performance harness. |
|
|
96
|
+
| `performance_profile` | empty or `node-package` | Shared package import, memory, archive, and dependency audit. |
|
|
97
|
+
| `performance_budget_file` | repository-relative path | Budget file for `node-package`; defaults to `performance-package-budgets.json`. |
|
|
98
|
+
| `eval` | `auto`, `true`, `false` | Discover, require, or disable the deterministic eval tier. |
|
|
99
|
+
| `eval_command` | JSON argv array | Explicit harness command when there is no `eval` package script. |
|
|
100
|
+
| `eval_report_file` | repository-relative path | Report validated against the eval contract; defaults to `eval-results/result.json`. |
|
|
101
|
+
| `eval_budget_file` | repository-relative path | Optional budget file; defaults to `eval-budgets.json`. |
|
|
102
|
+
| `eval_runner` | runner label | Runner for the `Validation / Eval` lane; defaults to the repository runner. Browser evals need a Chrome-capable runner such as `ubuntu-latest`. |
|
|
103
|
+
| `required_capabilities` | comma-separated task names | Fail closed when a required task or coverage evidence is unavailable. |
|
|
104
|
+
| `coverage_enforcement` | `auto`, `required`, `off` | Shared coverage-report policy. |
|
|
105
|
+
| `coverage_minimum` | `0`–`100` | Minimum percentage; defaults to `80`. |
|
|
106
|
+
| `coverage_metrics` | `lines`, `functions`, `branches`, `statements` | Metrics checked by the coverage gate. |
|
|
107
|
+
| `coverage_report` | comma-separated repository paths | Istanbul JSON summary or LCOV evidence files. |
|
|
103
108
|
|
|
104
109
|
Supported task capabilities are `format`, `lint`, `type_check`, `build`, `unit`,
|
|
105
|
-
`integration`, `e2e`, `smoke`, and `performance`. `coverage` is a policy
|
|
110
|
+
`integration`, `e2e`, `smoke`, `eval`, and `performance`. `coverage` is a policy
|
|
106
111
|
capability that also requires unit tests. See [Required capabilities and task
|
|
107
|
-
evidence](required-capabilities.md).
|
|
112
|
+
evidence](required-capabilities.md). The eval tier runs the repository's own
|
|
113
|
+
harness against the shared report contract and optional budgets; see
|
|
114
|
+
[Evals](EVALS.md).
|
|
108
115
|
|
|
109
116
|
The shared performance job discovers `performance:check`, then `perf:check`,
|
|
110
117
|
in JavaScript repositories. Other repositories can provide one command or an
|
package/docs/EVALS.md
ADDED
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
# Evals
|
|
2
|
+
|
|
3
|
+
`ci eval` is the repository's deterministic behavior-evaluation tier. It runs the
|
|
4
|
+
repository's own eval harness and consumes its result against a shared contract,
|
|
5
|
+
so task outcomes stay comparable across revisions and — later — across
|
|
6
|
+
executors (deterministic today, model-agent when a repository adopts one).
|
|
7
|
+
|
|
8
|
+
Evals are tests' measurement-oriented sibling: tests verify a specified
|
|
9
|
+
behavior (pass/fail); evals record how well the system does at repeated probes
|
|
10
|
+
(success rates, timing percentiles, bounded evidence) and compare those numbers
|
|
11
|
+
against pinned baselines. The `performance` tier is the resource-metric
|
|
12
|
+
special case of the same pattern.
|
|
13
|
+
|
|
14
|
+
## How the runtime finds your harness
|
|
15
|
+
|
|
16
|
+
`ci eval` discovers its subject in this order and skips when neither exists:
|
|
17
|
+
|
|
18
|
+
1. A package script named `eval`.
|
|
19
|
+
2. An explicit `eval_command` (JSON argv array) in `.github/code-foundry.yml`.
|
|
20
|
+
|
|
21
|
+
Both mechanisms use `eval: auto` by default: the tier runs when a subject is
|
|
22
|
+
present and skips cleanly otherwise. Set `eval: true` to require it (the
|
|
23
|
+
validation policy then treats a missing subject as an error), or `eval: false`
|
|
24
|
+
to disable discovery.
|
|
25
|
+
|
|
26
|
+
In managed validation the eval tier runs as its own `Validation / Eval` lane
|
|
27
|
+
during audit-mode runs, with the receipt retained as a task artifact. Browser
|
|
28
|
+
eval harnesses need a Chrome-capable runner: set `eval_runner: ubuntu-latest`
|
|
29
|
+
in `.github/code-foundry.yml` (the default when unset is the repository's
|
|
30
|
+
default runner, which may not ship a browser). The lane uploads the eval
|
|
31
|
+
report alongside the task receipt whenever a run fails, so reviewers get the
|
|
32
|
+
measured numbers with the red check. Performance's task profile and gating
|
|
33
|
+
rules apply unchanged; evals are an optional, non-gating tier like
|
|
34
|
+
performance.
|
|
35
|
+
|
|
36
|
+
| Config key | Values | Meaning |
|
|
37
|
+
| ------------------ | --------------------------------- | ---------------------------------------------------------------- |
|
|
38
|
+
| `eval` | `auto` (default), `true`, `false` | Enable, require, or disable the eval tier. |
|
|
39
|
+
| `eval_command` | JSON argv array | Explicit harness command when there is no `eval` package script. |
|
|
40
|
+
| `eval_report_file` | repository-relative path | The report to validate; defaults to `eval-results/result.json`. |
|
|
41
|
+
| `eval_budget_file` | repository-relative path | Optional budget file; defaults to `eval-budgets.json`. |
|
|
42
|
+
|
|
43
|
+
## The report contract
|
|
44
|
+
|
|
45
|
+
Your harness writes `eval-results/result.json`. The runtime validates the
|
|
46
|
+
envelope before budgets; a report that violates the contract fails the tier
|
|
47
|
+
with every violation listed (fail closed, typos can never pass silently):
|
|
48
|
+
|
|
49
|
+
```jsonc
|
|
50
|
+
{
|
|
51
|
+
"schemaVersion": 1, // contract version, currently 1
|
|
52
|
+
"revision": "<git sha>", // optional but strongly recommended
|
|
53
|
+
"dependencyHash": "<sha256>", // optional; pins the dependency set
|
|
54
|
+
"summary": {
|
|
55
|
+
"taskCount": 7,
|
|
56
|
+
"attempts": 7,
|
|
57
|
+
"passed": 7,
|
|
58
|
+
"failed": 0,
|
|
59
|
+
"harnessFailures": 0, // environment broke; not a task regression
|
|
60
|
+
"successRate": 1.0,
|
|
61
|
+
"toolCalls": 31,
|
|
62
|
+
"evidenceErrors": 0,
|
|
63
|
+
"taskDurationMs": { "count": 7, "mean": 2.1, "p50": 1.8, "p95": 3.0, "max": 3.4 },
|
|
64
|
+
"startupMs": { "count": 7, "mean": 0.3, "p50": 0.3, "p95": 0.4, "max": 0.4 },
|
|
65
|
+
"stepDurationMs": { "count": 31, "mean": 0.2, "p50": 0.1, "p95": 0.6, "max": 0.6 },
|
|
66
|
+
},
|
|
67
|
+
"tasks": [
|
|
68
|
+
{
|
|
69
|
+
"id": "form-submit",
|
|
70
|
+
"attempts": [
|
|
71
|
+
{
|
|
72
|
+
"iteration": 1,
|
|
73
|
+
"status": "passed",
|
|
74
|
+
"durationMs": 2.1,
|
|
75
|
+
"startupMs": 0.3,
|
|
76
|
+
"steps": [{ "tool": "fill_form", "status": "passed", "durationMs": 0.4 }],
|
|
77
|
+
"checks": ["fill_form completed"],
|
|
78
|
+
"metrics": { "formFields": 2 },
|
|
79
|
+
},
|
|
80
|
+
],
|
|
81
|
+
},
|
|
82
|
+
],
|
|
83
|
+
}
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
Rules the validator enforces:
|
|
87
|
+
|
|
88
|
+
- `schemaVersion` must match the contract version; `revision`/`dependencyHash`
|
|
89
|
+
must be strings when present.
|
|
90
|
+
- Summary counts are non-negative integers; `passed + failed` cannot exceed
|
|
91
|
+
`attempts`; `successRate` is between 0 and 1.
|
|
92
|
+
- `harnessFailures` counts attempts where the environment itself broke (browser
|
|
93
|
+
never started, zero steps succeeded, cleanup failed). It must not exceed
|
|
94
|
+
`attempts`. Fix the environment; do not treat these as task regressions.
|
|
95
|
+
- Failed attempts carry a bounded `failure` (non-empty `message`) plus a
|
|
96
|
+
`failureClass` of `harness` or `task`.
|
|
97
|
+
- `taskDurationMs`, `startupMs`, and `stepDurationMs` are stats objects with
|
|
98
|
+
`count`, `mean`, `p50`, `p95`, `max` (or bare `{ "count": 0 }`).
|
|
99
|
+
|
|
100
|
+
A reference implementation ships in
|
|
101
|
+
[`pi-browser-use`](https://github.com/0xPlayerOne/pi-browser-use)
|
|
102
|
+
(`scripts/eval.mjs` + `docs/eval-results.md`): deterministic browser-tool tasks
|
|
103
|
+
that emit this exact envelope.
|
|
104
|
+
|
|
105
|
+
## Budgets
|
|
106
|
+
|
|
107
|
+
Create `eval-budgets.json` (or point `eval_budget_file` at another file) to
|
|
108
|
+
gate on the measured numbers:
|
|
109
|
+
|
|
110
|
+
```json
|
|
111
|
+
{
|
|
112
|
+
"successRate": 1.0,
|
|
113
|
+
"stepP95Ms": 1000,
|
|
114
|
+
"taskP95Ms": 45000,
|
|
115
|
+
"maxHarnessFailures": 0
|
|
116
|
+
}
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
Supported budgets: `successRate`, `taskP95Ms`, `startupP95Ms`, `stepP95Ms`,
|
|
120
|
+
`maxHarnessFailures`, `maxEvidenceErrors`, `maxToolCalls`. Unknown keys fail
|
|
121
|
+
closed. When the budget file is absent the tier validates the contract only.
|
|
122
|
+
Budget files are committed configuration; only `eval-results/` is a local
|
|
123
|
+
artifact.
|
|
124
|
+
|
|
125
|
+
The runtime writes `eval-results/summary.json` with the executed commands, the
|
|
126
|
+
budget outcome, and the artifact list, mirroring the performance summary.
|
|
127
|
+
|
|
128
|
+
## Determinism rules for eval tasks
|
|
129
|
+
|
|
130
|
+
- Fixed fixtures and no network. A task's outcome must depend only on the code
|
|
131
|
+
under test.
|
|
132
|
+
- Bounded evidence: failure text and screenshots are capped; a failing task
|
|
133
|
+
never produces unbounded output.
|
|
134
|
+
- Cleanup failures are surfaced, never swallowed silently.
|
|
135
|
+
- Every attempt records the revision and dependency hash it ran against.
|
|
136
|
+
- Keep model-in-the-loop runs out of this tier. Reuse the same task IDs in a
|
|
137
|
+
separate scheduled harness with frozen model/reasoning/judge baselines when
|
|
138
|
+
you need agent-capability measurement; its variance is why it must never
|
|
139
|
+
gate pull requests.
|
package/docs/WORKFLOWS.md
CHANGED
|
@@ -27,7 +27,9 @@ ready transition. Release Please version heads are excluded because the
|
|
|
27
27
|
release workflow owns their state. A separate draft-control caller listens for
|
|
28
28
|
`converted_to_draft` and cancels queued or running pull-request workflows.
|
|
29
29
|
Marking a pull request ready starts validation, and each new commit on a ready
|
|
30
|
-
pull request starts it again for the current head.
|
|
30
|
+
pull request starts it again for the current head. Audit-mode runs additionally
|
|
31
|
+
execute the eval lane (`Validation / Eval`) for repositories that ship an eval
|
|
32
|
+
harness; see [Evals](EVALS.md).
|
|
31
33
|
|
|
32
34
|
The separate `validation-audit.yml` caller is pinned to the configured released
|
|
33
35
|
runtime and handles scheduled and manual audits:
|
|
@@ -40,9 +42,15 @@ workflow_dispatch:
|
|
|
40
42
|
|
|
41
43
|
In the `staging-release` topology, pull requests into `staging` run the fast
|
|
42
44
|
tier, ordinary pull requests into `main` run the full audit tier, and exact
|
|
43
|
-
Release Please pull requests into `main` run
|
|
44
|
-
release-diff policy.
|
|
45
|
-
|
|
45
|
+
Release Please pull requests into `main` run a lean release lane — CI, unit
|
|
46
|
+
tests, and CodeQL plus the release-diff policy. Their diff is version
|
|
47
|
+
metadata only: the content was fully audited on the pull requests that
|
|
48
|
+
merged into `main`, and the scheduled audit lane re-covers drift, so the
|
|
49
|
+
release lane keeps repository rulesets satisfiable without re-running the
|
|
50
|
+
runner-heavy suites. The managed branch rulesets require only the aggregate
|
|
51
|
+
`Validation / Gate` check, so skipped non-required jobs never deadlock the
|
|
52
|
+
release; do not hand-require individual job contexts on release branches.
|
|
53
|
+
In the
|
|
46
54
|
`direct` topology (the default) every pull request targets `main` and runs the
|
|
47
55
|
full audit tier, because there is no integration branch for a fast pass.
|
|
48
56
|
Scheduled and manual runs select the audit tier in both topologies. Draft PR
|
|
@@ -5,8 +5,9 @@ Optional task discovery remains available. Configure scalar values in
|
|
|
5
5
|
`.github/code-foundry.yml`:
|
|
6
6
|
|
|
7
7
|
```yaml
|
|
8
|
-
required_capabilities: type_check,unit,e2e,performance,coverage
|
|
8
|
+
required_capabilities: type_check,unit,e2e,eval,performance,coverage
|
|
9
9
|
performance: true
|
|
10
|
+
eval: true
|
|
10
11
|
coverage_enforcement: required
|
|
11
12
|
coverage_minimum: 80
|
|
12
13
|
coverage_metrics: lines,branches
|
|
@@ -14,10 +15,11 @@ coverage_report: coverage/coverage-summary.json
|
|
|
14
15
|
```
|
|
15
16
|
|
|
16
17
|
Supported task capabilities are `format`, `lint`, `type_check`, `build`, `unit`,
|
|
17
|
-
`integration`, `e2e`, `smoke`, and `performance`. `coverage` additionally requires
|
|
18
|
+
`integration`, `e2e`, `smoke`, `eval`, and `performance`. `coverage` additionally requires
|
|
18
19
|
unit tests. Unknown names, contradictory requirements, and invalid thresholds
|
|
19
|
-
are errors. `performance: true`
|
|
20
|
-
script happens to exist. Use `performance: auto`
|
|
20
|
+
are errors. `performance: true` and `eval: true` mean required, not merely
|
|
21
|
+
enabled when a script happens to exist. Use `performance: auto` or
|
|
22
|
+
`eval: auto` to retain optional discovery.
|
|
21
23
|
|
|
22
24
|
The public `src/runtime.mjs` entrypoint delegates ecosystem execution to the
|
|
23
25
|
private `src/runtime-core.mjs`. Keep both files and `src/lib` when vendoring the
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "code-foundry",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.25.3",
|
|
4
4
|
"description": "A fast, language-aware repository factory for agent-ready workflows, testing, security, and release automation.",
|
|
5
5
|
"homepage": "https://github.com/0xPlayerOne/code-foundry#readme",
|
|
6
6
|
"bugs": {
|