code-foundry 1.22.1 → 1.25.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,116 @@
1
+ name: Code Foundry Eval
2
+
3
+ on:
4
+ workflow_call:
5
+ inputs:
6
+ runtime-repository:
7
+ description: Repository containing the Code Foundry runtime.
8
+ required: false
9
+ type: string
10
+ default: 0xPlayerOne/code-foundry
11
+ runtime-ref:
12
+ description: Code Foundry runtime tag or ref.
13
+ required: false
14
+ type: string
15
+ default: v1.0.5
16
+ runner:
17
+ description: >-
18
+ Runner used by the eval harness. Browser evals need a
19
+ Chrome-capable runner such as ubuntu-latest; ubuntu-slim does not
20
+ ship a browser.
21
+ required: false
22
+ type: string
23
+ default: ubuntu-latest
24
+ artifact-prefix:
25
+ description: Prefix for task receipts; use a unique value when calling this workflow more than once.
26
+ required: false
27
+ type: string
28
+ default: task-result
29
+
30
+ permissions:
31
+ contents: read
32
+
33
+ env:
34
+ TURBO_TOKEN: ${{ secrets.TURBO_TOKEN }}
35
+ TURBO_TEAM: ${{ vars.TURBO_TEAM }}
36
+ REPO_FOUNDRY_PROFILE: ${{ vars.REPO_FOUNDRY_PROFILE }}
37
+ REPO_FOUNDRY_LANGUAGES: ${{ vars.REPO_FOUNDRY_LANGUAGES }}
38
+ REPO_FOUNDRY_FEATURES: ${{ vars.REPO_FOUNDRY_FEATURES }}
39
+ REPO_FOUNDRY_PACKAGE_MANAGER: ${{ vars.REPO_FOUNDRY_PACKAGE_MANAGER }}
40
+ REPO_FOUNDRY_RUNNER: ${{ vars.REPO_FOUNDRY_RUNNER }}
41
+ REPO_FOUNDRY_CACHE_PACKAGES: ${{ vars.REPO_FOUNDRY_CACHE_PACKAGES || 'auto' }}
42
+ REPO_FOUNDRY_CACHE_BUILD: ${{ vars.REPO_FOUNDRY_CACHE_BUILD || 'auto' }}
43
+
44
+ concurrency:
45
+ group: code-foundry-eval-${{ github.event_name }}-${{ github.event.pull_request.head.repo.full_name || github.repository }}-${{ github.event.pull_request.head.ref || github.ref_name }}
46
+ cancel-in-progress: true
47
+
48
+ jobs:
49
+ eval:
50
+ name: Eval
51
+ if: vars.CI_BILLING_PAUSED != 'true'
52
+ runs-on: ${{ inputs.runner }}
53
+ steps:
54
+ - name: Checkout
55
+ uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
56
+ with:
57
+ persist-credentials: false
58
+ - name: Runtime
59
+ uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
60
+ with:
61
+ persist-credentials: false
62
+ repository: ${{ inputs.runtime-repository }}
63
+ ref: ${{ inputs.runtime-ref }}
64
+ path: .github/.code-foundry
65
+ sparse-checkout: |
66
+ .github/actions
67
+ src/lib
68
+ src/runtime-core.mjs
69
+ src/runtime.mjs
70
+ - name: Install runtime
71
+ run: |
72
+ mkdir -p .github/actions
73
+ mv .github/.code-foundry "$RUNNER_TEMP/code-foundry"
74
+ cp -R "$RUNNER_TEMP/code-foundry/.github/actions/." .github/actions/
75
+ - name: Detect
76
+ id: applicability
77
+ run: node "$RUNNER_TEMP/code-foundry/src/runtime.mjs" ci should_run eval >> "$GITHUB_OUTPUT"
78
+ - name: Setup
79
+ if: steps.applicability.outputs.applicable == 'true'
80
+ uses: ./.github/actions/setup
81
+ with:
82
+ # Eval harnesses run through the repository's own package scripts and
83
+ # need their dependencies installed before the task executes.
84
+ install: 'true'
85
+ cache-build: 'true'
86
+ task: eval
87
+ - name: Eval
88
+ id: execute
89
+ if: steps.applicability.outputs.applicable == 'true'
90
+ run: node "$RUNNER_TEMP/code-foundry/src/runtime.mjs" ci eval
91
+ - name: Retain task receipt
92
+ if: >-
93
+ always() &&
94
+ (steps.execute.outcome == 'success' || steps.execute.outcome == 'failure' ||
95
+ steps.applicability.outputs.applicable == 'false') &&
96
+ hashFiles('.code-foundry/results/eval.json') != ''
97
+ uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
98
+ with:
99
+ name: ${{ inputs.artifact-prefix }}-${{ github.run_id }}-${{ github.run_attempt }}-eval
100
+ path: .code-foundry/results/eval.json
101
+ include-hidden-files: true
102
+ if-no-files-found: error
103
+ retention-days: 14
104
+ - name: Retain eval report
105
+ if: >-
106
+ always() &&
107
+ steps.execute.outcome == 'failure' &&
108
+ hashFiles('eval-results/result.json') != ''
109
+ uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
110
+ with:
111
+ name: ${{ inputs.artifact-prefix }}-${{ github.run_id }}-${{ github.run_attempt }}-eval-report
112
+ path: |
113
+ eval-results/result.json
114
+ eval-results/summary.json
115
+ if-no-files-found: error
116
+ retention-days: 14
@@ -42,6 +42,14 @@ on:
42
42
  required: false
43
43
  type: string
44
44
  default: ubuntu-slim
45
+ eval-runner:
46
+ description: >-
47
+ Runner used by the eval harness. Browser evals need a
48
+ Chrome-capable runner such as ubuntu-latest; ubuntu-slim does not
49
+ ship a browser.
50
+ required: false
51
+ type: string
52
+ default: ubuntu-latest
45
53
  codeql-runner:
46
54
  description: Runner used by CodeQL jobs.
47
55
  required: false
@@ -106,28 +114,47 @@ jobs:
106
114
  runner: ${{ inputs.test-runner }}
107
115
  unit-runner: ${{ inputs.unit-runner }}
108
116
  performance-runner: ${{ inputs.performance-runner }}
109
- unit-only: ${{ inputs.mode == 'fast' }}
117
+ # Release pull requests change version metadata only; their content was
118
+ # audited on the pull requests that merged into main. Integration, E2E,
119
+ # smoke, and performance lanes stay in the audit and scheduled lanes.
120
+ unit-only: ${{ inputs.mode == 'fast' || inputs.mode == 'release' }}
110
121
  artifact-prefix: ${{ inputs.artifact-prefix }}
111
122
  secrets:
112
123
  TURBO_TOKEN: ${{ secrets.TURBO_TOKEN }}
113
124
 
114
125
  security:
115
126
  name: Security
116
- if: vars.CI_BILLING_PAUSED != 'true' && (inputs.mode == 'audit' || inputs.mode == 'release')
127
+ if: vars.CI_BILLING_PAUSED != 'true' && inputs.mode == 'audit'
117
128
  uses: ./.github/workflows/security.yml
118
129
  with:
119
130
  runtime-repository: ${{ inputs.runtime-repository }}
120
131
  runtime-ref: ${{ inputs.runtime-ref }}
121
132
  runner: ${{ inputs.security-runner }}
122
133
 
134
+ eval:
135
+ name: Eval
136
+ # Behavior evals cover content changes; the lean release lane skips them
137
+ # because its diff is version metadata only. The reusable workflow skips
138
+ # cleanly when the repository has no eval harness.
139
+ if: vars.CI_BILLING_PAUSED != 'true' && inputs.mode == 'audit'
140
+ uses: ./.github/workflows/eval.yml
141
+ with:
142
+ runtime-repository: ${{ inputs.runtime-repository }}
143
+ runtime-ref: ${{ inputs.runtime-ref }}
144
+ runner: ${{ inputs.eval-runner }}
145
+ artifact-prefix: ${{ inputs.artifact-prefix }}
146
+
123
147
  # Aggregate check: "Validation / Gate" (caller job name + this job name).
124
148
  # Always evaluates; only jobs required for the mode must have succeeded.
125
- # The release tier runs the available CI/test/security suite. Its generated
126
- # release diff policy executes as a conditional step below, so no separate
127
- # release-policy job renders skipped checks on ordinary pull requests.
149
+ # The release tier requires the available fast suite; its diff is version
150
+ # metadata (validated against the release policy below), its content was
151
+ # audited on the pull requests that merged into main, and the scheduled
152
+ # audit lane re-covers drift. Its generated release diff policy executes
153
+ # as a conditional step below, so no separate release-policy job renders
154
+ # skipped checks on ordinary pull requests.
128
155
  gate:
129
156
  name: Gate
130
- needs: [ci, test, security]
157
+ needs: [ci, test, security, eval]
131
158
  if: vars.CI_BILLING_PAUSED != 'true' && always()
132
159
  runs-on: ubuntu-slim
133
160
  timeout-minutes: 10
@@ -136,6 +163,7 @@ jobs:
136
163
  FOUNDRY_CI: ${{ needs.ci.result }}
137
164
  FOUNDRY_TEST: ${{ needs.test.result }}
138
165
  FOUNDRY_SECURITY: ${{ needs.security.result }}
166
+ FOUNDRY_EVAL: ${{ needs.eval.result }}
139
167
  # This workflow is selected only for repositories whose canonical
140
168
  # configuration explicitly disables unavailable CodeQL. Treat that
141
169
  # policy decision as satisfied without registering a skipped PR check.
@@ -42,6 +42,14 @@ on:
42
42
  required: false
43
43
  type: string
44
44
  default: ubuntu-slim
45
+ eval-runner:
46
+ description: >-
47
+ Runner used by the eval harness. Browser evals need a
48
+ Chrome-capable runner such as ubuntu-latest; ubuntu-slim does not
49
+ ship a browser.
50
+ required: false
51
+ type: string
52
+ default: ubuntu-latest
45
53
  codeql-runner:
46
54
  description: Runner used by CodeQL jobs.
47
55
  required: false
@@ -106,14 +114,17 @@ jobs:
106
114
  runner: ${{ inputs.test-runner }}
107
115
  unit-runner: ${{ inputs.unit-runner }}
108
116
  performance-runner: ${{ inputs.performance-runner }}
109
- unit-only: ${{ inputs.mode == 'fast' }}
117
+ # Release pull requests change version metadata only; their content was
118
+ # audited on the pull requests that merged into main. Integration, E2E,
119
+ # smoke, and performance lanes stay in the audit and scheduled lanes.
120
+ unit-only: ${{ inputs.mode == 'fast' || inputs.mode == 'release' }}
110
121
  artifact-prefix: ${{ inputs.artifact-prefix }}
111
122
  secrets:
112
123
  TURBO_TOKEN: ${{ secrets.TURBO_TOKEN }}
113
124
 
114
125
  security:
115
126
  name: Security
116
- if: vars.CI_BILLING_PAUSED != 'true' && (inputs.mode == 'audit' || inputs.mode == 'release')
127
+ if: vars.CI_BILLING_PAUSED != 'true' && inputs.mode == 'audit'
117
128
  uses: ./.github/workflows/security.yml
118
129
  with:
119
130
  runtime-repository: ${{ inputs.runtime-repository }}
@@ -134,15 +145,30 @@ jobs:
134
145
  rust-threads: ${{ inputs.rust-threads }}
135
146
  rust-max-parallel: ${{ inputs.rust-max-parallel }}
136
147
 
148
+ eval:
149
+ name: Eval
150
+ # Behavior evals cover content changes; the lean release lane skips them
151
+ # because its diff is version metadata only. The reusable workflow skips
152
+ # cleanly when the repository has no eval harness.
153
+ if: vars.CI_BILLING_PAUSED != 'true' && inputs.mode == 'audit'
154
+ uses: ./.github/workflows/eval.yml
155
+ with:
156
+ runtime-repository: ${{ inputs.runtime-repository }}
157
+ runtime-ref: ${{ inputs.runtime-ref }}
158
+ runner: ${{ inputs.eval-runner }}
159
+ artifact-prefix: ${{ inputs.artifact-prefix }}
160
+
137
161
  # Aggregate check: "Validation / Gate" (caller job name + this job name).
138
162
  # Always evaluates; only jobs required for the mode must have succeeded.
139
- # The release tier runs the full audit suite so Release Please pull requests
140
- # expose no neutral suite checks. Its generated release diff policy executes
163
+ # The release tier requires the fast suite plus CodeQL: its diff is version
164
+ # metadata (validated against the release policy below), its content was
165
+ # audited on the pull requests that merged into main, and the scheduled
166
+ # audit lane re-covers drift. Its generated release diff policy executes
141
167
  # as a conditional step below, so ordinary pull requests still render no
142
168
  # skipped release-policy row.
143
169
  gate:
144
170
  name: Gate
145
- needs: [ci, test, security, codeql]
171
+ needs: [ci, test, security, codeql, eval]
146
172
  if: vars.CI_BILLING_PAUSED != 'true' && always()
147
173
  runs-on: ubuntu-slim
148
174
  timeout-minutes: 10
@@ -152,6 +178,7 @@ jobs:
152
178
  FOUNDRY_TEST: ${{ needs.test.result }}
153
179
  FOUNDRY_SECURITY: ${{ needs.security.result }}
154
180
  FOUNDRY_CODEQL: ${{ needs.codeql.result }}
181
+ FOUNDRY_EVAL: ${{ needs.eval.result }}
155
182
  steps:
156
183
  - name: Checkout runtime
157
184
  uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
@@ -34,6 +34,7 @@ jobs:
34
34
  unit-runner: ubuntu-slim
35
35
  performance-runner: ubuntu-latest
36
36
  security-runner: ubuntu-slim
37
+ eval-runner: ubuntu-latest
37
38
  codeql-runner: ubuntu-latest
38
39
  rust-shards: '["all"]'
39
40
  rust-threads: '1'
@@ -85,6 +85,7 @@ jobs:
85
85
  unit-runner: ubuntu-slim
86
86
  performance-runner: ubuntu-latest
87
87
  security-runner: ubuntu-slim
88
+ eval-runner: ubuntu-latest
88
89
  codeql-runner: ubuntu-latest
89
90
  rust-shards: '["all"]'
90
91
  rust-threads: '1'
package/AGENTS.md CHANGED
@@ -149,6 +149,7 @@ node src/runtime.mjs ci unit
149
149
  node src/runtime.mjs ci integration
150
150
  node src/runtime.mjs ci e2e
151
151
  node src/runtime.mjs ci smoke
152
+ node src/runtime.mjs ci eval
152
153
  node src/runtime.mjs ci performance
153
154
  Security and dependency audits run through the GitHub Security workflow.
154
155
  ```
package/CHANGELOG.md CHANGED
@@ -1,5 +1,47 @@
1
1
  # Changelog
2
2
 
3
+ ## [1.25.3](https://github.com/0xPlayerOne/code-foundry/compare/v1.25.2...v1.25.3) (2026-09-10)
4
+
5
+
6
+ ### Bug Fixes
7
+
8
+ * **release:** cover the attestation lag in publication retries ([#587](https://github.com/0xPlayerOne/code-foundry/issues/587)) ([7ae5e4b](https://github.com/0xPlayerOne/code-foundry/commit/7ae5e4bb95f68cbf4e2b48778776a0396c877271))
9
+
10
+ ## [1.25.2](https://github.com/0xPlayerOne/code-foundry/compare/v1.25.1...v1.25.2) (2026-09-10)
11
+
12
+
13
+ ### Bug Fixes
14
+
15
+ * **release:** retry integrity verification through the publication consistency window ([#585](https://github.com/0xPlayerOne/code-foundry/issues/585)) ([5b997dd](https://github.com/0xPlayerOne/code-foundry/commit/5b997dda7e272b59a386eeeda115e60ad284f6ce))
16
+
17
+ ## [1.25.1](https://github.com/0xPlayerOne/code-foundry/compare/v1.25.0...v1.25.1) (2026-09-10)
18
+
19
+
20
+ ### Bug Fixes
21
+
22
+ * **release:** wait for the release index before verifying a published release ([#583](https://github.com/0xPlayerOne/code-foundry/issues/583)) ([7823c14](https://github.com/0xPlayerOne/code-foundry/commit/7823c1473d8c5e7c17b039dc755ff5c8d1dba39c))
23
+
24
+ ## [1.25.0](https://github.com/0xPlayerOne/code-foundry/compare/v1.24.0...v1.25.0) (2026-09-10)
25
+
26
+
27
+ ### Features
28
+
29
+ * **eval:** run the eval tier as a managed Validation / Eval lane ([#580](https://github.com/0xPlayerOne/code-foundry/issues/580)) ([9414889](https://github.com/0xPlayerOne/code-foundry/commit/94148894da9a627a6bd7c71c846848c198c1dfec))
30
+
31
+ ## [1.24.0](https://github.com/0xPlayerOne/code-foundry/compare/v1.23.0...v1.24.0) (2026-09-10)
32
+
33
+
34
+ ### Features
35
+
36
+ * **validation:** run a lean release lane for Release Please pull requests ([#579](https://github.com/0xPlayerOne/code-foundry/issues/579)) ([a38ca75](https://github.com/0xPlayerOne/code-foundry/commit/a38ca75c7da577af8368467b8b026d0ec563fc41))
37
+
38
+ ## [1.23.0](https://github.com/0xPlayerOne/code-foundry/compare/v1.22.1...v1.23.0) (2026-09-09)
39
+
40
+
41
+ ### Features
42
+
43
+ * add ci eval tier with shared report contract and budget gate ([#576](https://github.com/0xPlayerOne/code-foundry/issues/576)) ([7d881dd](https://github.com/0xPlayerOne/code-foundry/commit/7d881dd1bc6ace8e48dffff45b1a22a12bdae1c9))
44
+
3
45
  ## [1.22.1](https://github.com/0xPlayerOne/code-foundry/compare/v1.22.0...v1.22.1) (2026-09-09)
4
46
 
5
47
 
package/README.md CHANGED
@@ -39,6 +39,9 @@ contract. For normal updates, edit that file and run `npx code-foundry sync`.
39
39
  audits, Draft Guard, Draft PR, Release PR, and Release.
40
40
  - A deterministic performance lane with ordered command support, stable result
41
41
  artifacts, and an optional shared Node package budget profile.
42
+ - A deterministic eval tier (`ci eval`) that runs a repository's behavior-eval
43
+ harness against a shared report contract with optional budgets, keeping task
44
+ outcomes comparable across revisions and executors.
42
45
  - An opt-in product-quality runner for static sites, web apps, Workers, and
43
46
  published packages.
44
47
  - A small `.githooks/pre-commit` launcher with language-aware formatting and
@@ -89,22 +89,29 @@ See [Merge queue validation](merge-queues.md) before enabling it.
89
89
 
90
90
  ## Validation and quality
91
91
 
92
- | Key | Values | Purpose |
93
- | ------------------------- | ---------------------------------------------- | ------------------------------------------------------------------------------- |
94
- | `performance` | `auto`, `true`, `false` | Discover, require, or disable performance checks. |
95
- | `performance_command` | JSON argv array or array of argv arrays | Ordered commands for a non-package performance harness. |
96
- | `performance_profile` | empty or `node-package` | Shared package import, memory, archive, and dependency audit. |
97
- | `performance_budget_file` | repository-relative path | Budget file for `node-package`; defaults to `performance-package-budgets.json`. |
98
- | `required_capabilities` | comma-separated task names | Fail closed when a required task or coverage evidence is unavailable. |
99
- | `coverage_enforcement` | `auto`, `required`, `off` | Shared coverage-report policy. |
100
- | `coverage_minimum` | `0`–`100` | Minimum percentage; defaults to `80`. |
101
- | `coverage_metrics` | `lines`, `functions`, `branches`, `statements` | Metrics checked by the coverage gate. |
102
- | `coverage_report` | comma-separated repository paths | Istanbul JSON summary or LCOV evidence files. |
92
+ | Key | Values | Purpose |
93
+ | ------------------------- | ---------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------- |
94
+ | `performance` | `auto`, `true`, `false` | Discover, require, or disable performance checks. |
95
+ | `performance_command` | JSON argv array or array of argv arrays | Ordered commands for a non-package performance harness. |
96
+ | `performance_profile` | empty or `node-package` | Shared package import, memory, archive, and dependency audit. |
97
+ | `performance_budget_file` | repository-relative path | Budget file for `node-package`; defaults to `performance-package-budgets.json`. |
98
+ | `eval` | `auto`, `true`, `false` | Discover, require, or disable the deterministic eval tier. |
99
+ | `eval_command` | JSON argv array | Explicit harness command when there is no `eval` package script. |
100
+ | `eval_report_file` | repository-relative path | Report validated against the eval contract; defaults to `eval-results/result.json`. |
101
+ | `eval_budget_file` | repository-relative path | Optional budget file; defaults to `eval-budgets.json`. |
102
+ | `eval_runner` | runner label | Runner for the `Validation / Eval` lane; defaults to the repository runner. Browser evals need a Chrome-capable runner such as `ubuntu-latest`. |
103
+ | `required_capabilities` | comma-separated task names | Fail closed when a required task or coverage evidence is unavailable. |
104
+ | `coverage_enforcement` | `auto`, `required`, `off` | Shared coverage-report policy. |
105
+ | `coverage_minimum` | `0`–`100` | Minimum percentage; defaults to `80`. |
106
+ | `coverage_metrics` | `lines`, `functions`, `branches`, `statements` | Metrics checked by the coverage gate. |
107
+ | `coverage_report` | comma-separated repository paths | Istanbul JSON summary or LCOV evidence files. |
103
108
 
104
109
  Supported task capabilities are `format`, `lint`, `type_check`, `build`, `unit`,
105
- `integration`, `e2e`, `smoke`, and `performance`. `coverage` is a policy
110
+ `integration`, `e2e`, `smoke`, `eval`, and `performance`. `coverage` is a policy
106
111
  capability that also requires unit tests. See [Required capabilities and task
107
- evidence](required-capabilities.md).
112
+ evidence](required-capabilities.md). The eval tier runs the repository's own
113
+ harness against the shared report contract and optional budgets; see
114
+ [Evals](EVALS.md).
108
115
 
109
116
  The shared performance job discovers `performance:check`, then `perf:check`,
110
117
  in JavaScript repositories. Other repositories can provide one command or an
package/docs/EVALS.md ADDED
@@ -0,0 +1,139 @@
1
+ # Evals
2
+
3
+ `ci eval` is the repository's deterministic behavior-evaluation tier. It runs the
4
+ repository's own eval harness and consumes its result against a shared contract,
5
+ so task outcomes stay comparable across revisions and — later — across
6
+ executors (deterministic today, model-agent when a repository adopts one).
7
+
8
+ Evals are tests' measurement-oriented sibling: tests verify a specified
9
+ behavior (pass/fail); evals record how well the system does at repeated probes
10
+ (success rates, timing percentiles, bounded evidence) and compare those numbers
11
+ against pinned baselines. The `performance` tier is the resource-metric
12
+ special case of the same pattern.
13
+
14
+ ## How the runtime finds your harness
15
+
16
+ `ci eval` discovers its subject in this order and skips when neither exists:
17
+
18
+ 1. A package script named `eval`.
19
+ 2. An explicit `eval_command` (JSON argv array) in `.github/code-foundry.yml`.
20
+
21
+ Both mechanisms use `eval: auto` by default: the tier runs when a subject is
22
+ present and skips cleanly otherwise. Set `eval: true` to require it (the
23
+ validation policy then treats a missing subject as an error), or `eval: false`
24
+ to disable discovery.
25
+
26
+ In managed validation the eval tier runs as its own `Validation / Eval` lane
27
+ during audit-mode runs, with the receipt retained as a task artifact. Browser
28
+ eval harnesses need a Chrome-capable runner: set `eval_runner: ubuntu-latest`
29
+ in `.github/code-foundry.yml` (the default when unset is the repository's
30
+ default runner, which may not ship a browser). The lane uploads the eval
31
+ report alongside the task receipt whenever a run fails, so reviewers get the
32
+ measured numbers with the red check. Performance's task profile and gating
33
+ rules apply unchanged; evals are an optional, non-gating tier like
34
+ performance.
35
+
36
+ | Config key | Values | Meaning |
37
+ | ------------------ | --------------------------------- | ---------------------------------------------------------------- |
38
+ | `eval` | `auto` (default), `true`, `false` | Enable, require, or disable the eval tier. |
39
+ | `eval_command` | JSON argv array | Explicit harness command when there is no `eval` package script. |
40
+ | `eval_report_file` | repository-relative path | The report to validate; defaults to `eval-results/result.json`. |
41
+ | `eval_budget_file` | repository-relative path | Optional budget file; defaults to `eval-budgets.json`. |
42
+
43
+ ## The report contract
44
+
45
+ Your harness writes `eval-results/result.json`. The runtime validates the
46
+ envelope before budgets; a report that violates the contract fails the tier
47
+ with every violation listed (fail closed, typos can never pass silently):
48
+
49
+ ```jsonc
50
+ {
51
+ "schemaVersion": 1, // contract version, currently 1
52
+ "revision": "<git sha>", // optional but strongly recommended
53
+ "dependencyHash": "<sha256>", // optional; pins the dependency set
54
+ "summary": {
55
+ "taskCount": 7,
56
+ "attempts": 7,
57
+ "passed": 7,
58
+ "failed": 0,
59
+ "harnessFailures": 0, // environment broke; not a task regression
60
+ "successRate": 1.0,
61
+ "toolCalls": 31,
62
+ "evidenceErrors": 0,
63
+ "taskDurationMs": { "count": 7, "mean": 2.1, "p50": 1.8, "p95": 3.0, "max": 3.4 },
64
+ "startupMs": { "count": 7, "mean": 0.3, "p50": 0.3, "p95": 0.4, "max": 0.4 },
65
+ "stepDurationMs": { "count": 31, "mean": 0.2, "p50": 0.1, "p95": 0.6, "max": 0.6 },
66
+ },
67
+ "tasks": [
68
+ {
69
+ "id": "form-submit",
70
+ "attempts": [
71
+ {
72
+ "iteration": 1,
73
+ "status": "passed",
74
+ "durationMs": 2.1,
75
+ "startupMs": 0.3,
76
+ "steps": [{ "tool": "fill_form", "status": "passed", "durationMs": 0.4 }],
77
+ "checks": ["fill_form completed"],
78
+ "metrics": { "formFields": 2 },
79
+ },
80
+ ],
81
+ },
82
+ ],
83
+ }
84
+ ```
85
+
86
+ Rules the validator enforces:
87
+
88
+ - `schemaVersion` must match the contract version; `revision`/`dependencyHash`
89
+ must be strings when present.
90
+ - Summary counts are non-negative integers; `passed + failed` cannot exceed
91
+ `attempts`; `successRate` is between 0 and 1.
92
+ - `harnessFailures` counts attempts where the environment itself broke (browser
93
+ never started, zero steps succeeded, cleanup failed). It must not exceed
94
+ `attempts`. Fix the environment; do not treat these as task regressions.
95
+ - Failed attempts carry a bounded `failure` (non-empty `message`) plus a
96
+ `failureClass` of `harness` or `task`.
97
+ - `taskDurationMs`, `startupMs`, and `stepDurationMs` are stats objects with
98
+ `count`, `mean`, `p50`, `p95`, `max` (or bare `{ "count": 0 }`).
99
+
100
+ A reference implementation ships in
101
+ [`pi-browser-use`](https://github.com/0xPlayerOne/pi-browser-use)
102
+ (`scripts/eval.mjs` + `docs/eval-results.md`): deterministic browser-tool tasks
103
+ that emit this exact envelope.
104
+
105
+ ## Budgets
106
+
107
+ Create `eval-budgets.json` (or point `eval_budget_file` at another file) to
108
+ gate on the measured numbers:
109
+
110
+ ```json
111
+ {
112
+ "successRate": 1.0,
113
+ "stepP95Ms": 1000,
114
+ "taskP95Ms": 45000,
115
+ "maxHarnessFailures": 0
116
+ }
117
+ ```
118
+
119
+ Supported budgets: `successRate`, `taskP95Ms`, `startupP95Ms`, `stepP95Ms`,
120
+ `maxHarnessFailures`, `maxEvidenceErrors`, `maxToolCalls`. Unknown keys fail
121
+ closed. When the budget file is absent the tier validates the contract only.
122
+ Budget files are committed configuration; only `eval-results/` is a local
123
+ artifact.
124
+
125
+ The runtime writes `eval-results/summary.json` with the executed commands, the
126
+ budget outcome, and the artifact list, mirroring the performance summary.
127
+
128
+ ## Determinism rules for eval tasks
129
+
130
+ - Fixed fixtures and no network. A task's outcome must depend only on the code
131
+ under test.
132
+ - Bounded evidence: failure text and screenshots are capped; a failing task
133
+ never produces unbounded output.
134
+ - Cleanup failures are surfaced, never swallowed silently.
135
+ - Every attempt records the revision and dependency hash it ran against.
136
+ - Keep model-in-the-loop runs out of this tier. Reuse the same task IDs in a
137
+ separate scheduled harness with frozen model/reasoning/judge baselines when
138
+ you need agent-capability measurement; its variance is why it must never
139
+ gate pull requests.
package/docs/WORKFLOWS.md CHANGED
@@ -27,7 +27,9 @@ ready transition. Release Please version heads are excluded because the
27
27
  release workflow owns their state. A separate draft-control caller listens for
28
28
  `converted_to_draft` and cancels queued or running pull-request workflows.
29
29
  Marking a pull request ready starts validation, and each new commit on a ready
30
- pull request starts it again for the current head.
30
+ pull request starts it again for the current head. Audit-mode runs additionally
31
+ execute the eval lane (`Validation / Eval`) for repositories that ship an eval
32
+ harness; see [Evals](EVALS.md).
31
33
 
32
34
  The separate `validation-audit.yml` caller is pinned to the configured released
33
35
  runtime and handles scheduled and manual audits:
@@ -40,9 +42,15 @@ workflow_dispatch:
40
42
 
41
43
  In the `staging-release` topology, pull requests into `staging` run the fast
42
44
  tier, ordinary pull requests into `main` run the full audit tier, and exact
43
- Release Please pull requests into `main` run the full audit tier plus the
44
- release-diff policy. This keeps repository rulesets satisfiable for the release
45
- commit without exposing neutral or skipped suite checks. In the
45
+ Release Please pull requests into `main` run a lean release lane — CI, unit
46
+ tests, and CodeQL plus the release-diff policy. Their diff is version
47
+ metadata only: the content was fully audited on the pull requests that
48
+ merged into `main`, and the scheduled audit lane re-covers drift, so the
49
+ release lane keeps repository rulesets satisfiable without re-running the
50
+ runner-heavy suites. The managed branch rulesets require only the aggregate
51
+ `Validation / Gate` check, so skipped non-required jobs never deadlock the
52
+ release; do not hand-require individual job contexts on release branches.
53
+ In the
46
54
  `direct` topology (the default) every pull request targets `main` and runs the
47
55
  full audit tier, because there is no integration branch for a fast pass.
48
56
  Scheduled and manual runs select the audit tier in both topologies. Draft PR
@@ -5,8 +5,9 @@ Optional task discovery remains available. Configure scalar values in
5
5
  `.github/code-foundry.yml`:
6
6
 
7
7
  ```yaml
8
- required_capabilities: type_check,unit,e2e,performance,coverage
8
+ required_capabilities: type_check,unit,e2e,eval,performance,coverage
9
9
  performance: true
10
+ eval: true
10
11
  coverage_enforcement: required
11
12
  coverage_minimum: 80
12
13
  coverage_metrics: lines,branches
@@ -14,10 +15,11 @@ coverage_report: coverage/coverage-summary.json
14
15
  ```
15
16
 
16
17
  Supported task capabilities are `format`, `lint`, `type_check`, `build`, `unit`,
17
- `integration`, `e2e`, `smoke`, and `performance`. `coverage` additionally requires
18
+ `integration`, `e2e`, `smoke`, `eval`, and `performance`. `coverage` additionally requires
18
19
  unit tests. Unknown names, contradictory requirements, and invalid thresholds
19
- are errors. `performance: true` means required, not merely enabled when a
20
- script happens to exist. Use `performance: auto` to retain optional discovery.
20
+ are errors. `performance: true` and `eval: true` mean required, not merely
21
+ enabled when a script happens to exist. Use `performance: auto` or
22
+ `eval: auto` to retain optional discovery.
21
23
 
22
24
  The public `src/runtime.mjs` entrypoint delegates ecosystem execution to the
23
25
  private `src/runtime-core.mjs`. Keep both files and `src/lib` when vendoring the
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "code-foundry",
3
- "version": "1.22.1",
3
+ "version": "1.25.3",
4
4
  "description": "A fast, language-aware repository factory for agent-ready workflows, testing, security, and release automation.",
5
5
  "homepage": "https://github.com/0xPlayerOne/code-foundry#readme",
6
6
  "bugs": {