assertledger 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONTRIBUTING.md +31 -0
- package/LICENSE +21 -0
- package/README.fr.md +236 -0
- package/README.md +224 -0
- package/SECURITY.md +51 -0
- package/benchmarks/agentic-profile/README.md +15 -0
- package/benchmarks/agentic-profile/public/README.md +5 -0
- package/benchmarks/self-hosted-core/README.md +113 -0
- package/benchmarks/self-hosted-core/adapter.mjs +293 -0
- package/benchmarks/self-hosted-core/builder.ts +193 -0
- package/benchmarks/self-hosted-core/campaign.ts +233 -0
- package/benchmarks/self-hosted-core/existing-tests-builder.ts +217 -0
- package/benchmarks/self-hosted-core/existing-tests.ts +146 -0
- package/benchmarks/self-hosted-core/liveness.test.mjs +8 -0
- package/conformance/v1/bundle.json +104 -0
- package/conformance/v1/expected/canonical-order-a.json +4 -0
- package/conformance/v1/expected/canonical-order-b.json +4 -0
- package/conformance/v1/expected/create-benchmark-v1-measured.json +575 -0
- package/conformance/v1/expected/create-profile-v1-qualified.json +280 -0
- package/conformance/v1/expected/decide-collection-failure-non-kill.json +192 -0
- package/conformance/v1/expected/decide-compile-failure-non-kill.json +192 -0
- package/conformance/v1/expected/decide-infra-error-non-kill.json +192 -0
- package/conformance/v1/expected/decide-no-test-discovered-non-kill.json +192 -0
- package/conformance/v1/expected/decide-process-crash-non-kill.json +192 -0
- package/conformance/v1/expected/decide-timeout-non-kill.json +192 -0
- package/conformance/v1/expected/decide-verified.json +192 -0
- package/conformance/v1/expected/replay-benchmark-v1-resealed-summary-forgery.json +12 -0
- package/conformance/v1/expected/replay-evidence-raw-tamper.json +6 -0
- package/conformance/v1/expected/replay-evidence-resealed-semantic-forgery.json +6 -0
- package/conformance/v1/inputs/canonical-order-a.json +8 -0
- package/conformance/v1/inputs/canonical-order-b.json +8 -0
- package/conformance/v1/inputs/create-benchmark-v1-measured.json +459 -0
- package/conformance/v1/inputs/create-profile-v1-qualified.json +228 -0
- package/conformance/v1/inputs/decide-collection-failure-non-kill.json +143 -0
- package/conformance/v1/inputs/decide-compile-failure-non-kill.json +143 -0
- package/conformance/v1/inputs/decide-infra-error-non-kill.json +143 -0
- package/conformance/v1/inputs/decide-no-test-discovered-non-kill.json +143 -0
- package/conformance/v1/inputs/decide-process-crash-non-kill.json +143 -0
- package/conformance/v1/inputs/decide-timeout-non-kill.json +143 -0
- package/conformance/v1/inputs/decide-verified.json +143 -0
- package/conformance/v1/inputs/replay-benchmark-v1-resealed-summary-forgery.json +575 -0
- package/conformance/v1/inputs/replay-evidence-raw-tamper.json +201 -0
- package/conformance/v1/inputs/replay-evidence-resealed-semantic-forgery.json +192 -0
- package/conformance/v1/schemas/expected-digests.json +175 -0
- package/dist/cli.d.ts +9 -0
- package/dist/cli.d.ts.map +1 -0
- package/dist/cli.js +951 -0
- package/dist/cli.js.map +1 -0
- package/dist/contracts/diagnostics.d.ts +18 -0
- package/dist/contracts/diagnostics.d.ts.map +1 -0
- package/dist/contracts/diagnostics.js +13 -0
- package/dist/contracts/diagnostics.js.map +1 -0
- package/dist/contracts/index.d.ts +3908 -0
- package/dist/contracts/index.d.ts.map +1 -0
- package/dist/contracts/index.js +2569 -0
- package/dist/contracts/index.js.map +1 -0
- package/dist/contracts/runtime-doctor.d.ts +107 -0
- package/dist/contracts/runtime-doctor.d.ts.map +1 -0
- package/dist/contracts/runtime-doctor.js +91 -0
- package/dist/contracts/runtime-doctor.js.map +1 -0
- package/dist/core/index.d.ts +200 -0
- package/dist/core/index.d.ts.map +1 -0
- package/dist/core/index.js +2587 -0
- package/dist/core/index.js.map +1 -0
- package/dist/diagnostics.d.ts +7 -0
- package/dist/diagnostics.d.ts.map +1 -0
- package/dist/diagnostics.js +252 -0
- package/dist/diagnostics.js.map +1 -0
- package/dist/engine/adapters/node-test-profile.d.ts +14 -0
- package/dist/engine/adapters/node-test-profile.d.ts.map +1 -0
- package/dist/engine/adapters/node-test-profile.js +14 -0
- package/dist/engine/adapters/node-test-profile.js.map +1 -0
- package/dist/engine/adapters/node-test-runtime.d.ts +39 -0
- package/dist/engine/adapters/node-test-runtime.d.ts.map +1 -0
- package/dist/engine/adapters/node-test-runtime.js +173 -0
- package/dist/engine/adapters/node-test-runtime.js.map +1 -0
- package/dist/engine/adapters/runtime-facts.d.ts +26 -0
- package/dist/engine/adapters/runtime-facts.d.ts.map +1 -0
- package/dist/engine/adapters/runtime-facts.js +73 -0
- package/dist/engine/adapters/runtime-facts.js.map +1 -0
- package/dist/engine/connection.d.ts +22 -0
- package/dist/engine/connection.d.ts.map +1 -0
- package/dist/engine/connection.js +343 -0
- package/dist/engine/connection.js.map +1 -0
- package/dist/engine/git-regression.d.ts +25 -0
- package/dist/engine/git-regression.d.ts.map +1 -0
- package/dist/engine/git-regression.js +803 -0
- package/dist/engine/git-regression.js.map +1 -0
- package/dist/engine/index.d.ts +55 -0
- package/dist/engine/index.d.ts.map +1 -0
- package/dist/engine/index.js +2782 -0
- package/dist/engine/index.js.map +1 -0
- package/dist/engine/node-test-reporter.d.ts +2 -0
- package/dist/engine/node-test-reporter.d.ts.map +1 -0
- package/dist/engine/node-test-reporter.js +70 -0
- package/dist/engine/node-test-reporter.js.map +1 -0
- package/dist/engine/runtime-doctor.d.ts +16 -0
- package/dist/engine/runtime-doctor.d.ts.map +1 -0
- package/dist/engine/runtime-doctor.js +100 -0
- package/dist/engine/runtime-doctor.js.map +1 -0
- package/dist/evaluation/agentic-corpus.d.ts +161 -0
- package/dist/evaluation/agentic-corpus.d.ts.map +1 -0
- package/dist/evaluation/agentic-corpus.js +710 -0
- package/dist/evaluation/agentic-corpus.js.map +1 -0
- package/dist/index.d.ts +8 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +8 -0
- package/dist/index.js.map +1 -0
- package/dist/mcp/index.d.ts +13 -0
- package/dist/mcp/index.d.ts.map +1 -0
- package/dist/mcp/index.js +391 -0
- package/dist/mcp/index.js.map +1 -0
- package/dist/mcp/stdio.d.ts +3 -0
- package/dist/mcp/stdio.d.ts.map +1 -0
- package/dist/mcp/stdio.js +13 -0
- package/dist/mcp/stdio.js.map +1 -0
- package/dist/sdk/index.d.ts +52 -0
- package/dist/sdk/index.d.ts.map +1 -0
- package/dist/sdk/index.js +224 -0
- package/dist/sdk/index.js.map +1 -0
- package/dist/version.d.ts +2 -0
- package/dist/version.d.ts.map +1 -0
- package/dist/version.js +10 -0
- package/dist/version.js.map +1 -0
- package/docs/adapter-protocol.md +196 -0
- package/docs/agentic-benchmark.md +118 -0
- package/docs/agentic-corpus-experiment-h3.md +89 -0
- package/docs/agentic-corpus-plan.md +105 -0
- package/docs/agentic-corpus-provenance.md +59 -0
- package/docs/agentic-test-profile-pilot.md +57 -0
- package/docs/agentic-test-profile-v2.md +116 -0
- package/docs/agentic-test-profile.md +274 -0
- package/docs/architecture.md +157 -0
- package/docs/ci.md +37 -0
- package/docs/client-connections.md +61 -0
- package/docs/conformance-v1.md +72 -0
- package/docs/decisions/0001-typescript-runtime.md +24 -0
- package/docs/developer-experience.md +55 -0
- package/docs/diagnostics.md +35 -0
- package/docs/distribution.md +40 -0
- package/docs/git-regression.md +39 -0
- package/docs/migration-repository-validation-order.md +35 -0
- package/docs/migration-testforge-to-assertledger.md +64 -0
- package/docs/project-intent.md +173 -0
- package/docs/proof-model.md +116 -0
- package/docs/reference.md +336 -0
- package/docs/release-1.0.md +63 -0
- package/docs/repository-audit.md +52 -0
- package/docs/repository-init.md +60 -0
- package/docs/research-basis.md +27 -0
- package/docs/roadmap.md +74 -0
- package/docs/runtime-doctor.md +65 -0
- package/docs/testexplora-calibration.md +71 -0
- package/examples/agentic-benchmark/benchmark-request.mjs +19 -0
- package/examples/agentic-benchmark/structured-phase-adapter-fixture.mjs +35 -0
- package/examples/agentic-profile/profile-benchmark.mjs +34 -0
- package/examples/agentic-profile/profile-manifest.mjs +28 -0
- package/examples/git-history/README.md +44 -0
- package/examples/git-history/create-demo.mjs +128 -0
- package/examples/git-history/escape-string-regexp/LICENSE +9 -0
- package/examples/git-history/escape-string-regexp/before.cjs.txt +11 -0
- package/examples/git-history/escape-string-regexp/fixed.cjs.txt +13 -0
- package/examples/git-history/escape-string-regexp/provenance.json +28 -0
- package/examples/node-test/repository/package.json +5 -0
- package/examples/node-test/repository/src/is-even.js +3 -0
- package/examples/node-test/repository/tests/base.test.js +6 -0
- package/examples/node-test/request.json +93 -0
- package/integrations/skill/SKILL.md +51 -0
- package/package.json +88 -0
- package/schemas/agentic-benchmark-acquisition-replay-result.v1.json +70 -0
- package/schemas/agentic-benchmark-acquisition-request.v1.json +564 -0
- package/schemas/agentic-benchmark-acquisition-result.v1.json +1409 -0
- package/schemas/agentic-benchmark-artifact.v1.json +1251 -0
- package/schemas/agentic-benchmark-replay-result.v1.json +84 -0
- package/schemas/agentic-benchmark-request.v1.json +1034 -0
- package/schemas/agentic-corpus-allocation-commitment-replay-result.v1.json +58 -0
- package/schemas/agentic-corpus-allocation-commitment.v1.json +141 -0
- package/schemas/agentic-corpus-allocation-replay-result.v1.json +34 -0
- package/schemas/agentic-corpus-allocation-request.v1.json +65 -0
- package/schemas/agentic-corpus-allocation-reveal.v1.json +66 -0
- package/schemas/agentic-corpus-allocation.v1.json +167 -0
- package/schemas/agentic-corpus-experiment-artifact.v1.json +329 -0
- package/schemas/agentic-corpus-experiment-plan-replay-result.v1.json +50 -0
- package/schemas/agentic-corpus-experiment-plan.v1.json +424 -0
- package/schemas/agentic-corpus-experiment-replay-request.v1.json +336 -0
- package/schemas/agentic-corpus-experiment-replay-result.v1.json +106 -0
- package/schemas/agentic-corpus-experiment-request.v1.json +204 -0
- package/schemas/agentic-corpus-provenance.v1.json +143 -0
- package/schemas/agentic-corpus-trust-policy.v1.json +133 -0
- package/schemas/agentic-profile-replay-result.v1.json +56 -0
- package/schemas/agentic-profile-replay-result.v2.json +63 -0
- package/schemas/agentic-profile-report.v1.json +961 -0
- package/schemas/agentic-profile-report.v2.json +1674 -0
- package/schemas/agentic-profile-request.v1.json +671 -0
- package/schemas/agentic-profile-request.v2.json +1338 -0
- package/schemas/evidence-manifest.v1.json +636 -0
- package/schemas/replay-result.v1.json +49 -0
- package/schemas/repository-analysis.v1.json +119 -0
- package/schemas/repository-audit.v1.json +811 -0
- package/schemas/repository-init-config.v1.json +183 -0
- package/schemas/repository-init-lock.v1.json +162 -0
- package/schemas/repository-init-result.v1.json +212 -0
- package/schemas/verification-request.v1.json +389 -0
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
# Developer entry points
|
|
2
|
+
|
|
3
|
+
AssertLedger can inspect repository readiness without running candidate code:
|
|
4
|
+
|
|
5
|
+
```text
|
|
6
|
+
assertledger doctor .
|
|
7
|
+
assertledger doctor . --json
|
|
8
|
+
```
|
|
9
|
+
|
|
10
|
+
The JSON form is the existing repository initialization result. `WOULD_CREATE` means the static
|
|
11
|
+
configuration can be planned; it does not mean worlds, candidates, campaign evidence, or an MCP
|
|
12
|
+
client connection are ready. `BLOCKED` exits with code 3 and `CONFLICT` with code 4.
|
|
13
|
+
|
|
14
|
+
After installing and building AssertLedger, generate a project-local Codex MCP descriptor:
|
|
15
|
+
|
|
16
|
+
```text
|
|
17
|
+
assertledger connect . --client codex
|
|
18
|
+
assertledger connect . --client codex --write
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
The first command previews `.codex/config.toml` and the packaged project skill. `--write` creates
|
|
22
|
+
absent files, accepts byte-identical content as unchanged, and refuses to overwrite different
|
|
23
|
+
operator-owned content. It never changes global configuration, authentication, hooks, or trust.
|
|
24
|
+
Codex project trust and client restart or reload remain explicit user actions.
|
|
25
|
+
|
|
26
|
+
Use `--write` in a trusted local repository whose directory tree stays stable during the operation.
|
|
27
|
+
Existing symlinks and junctions at the configuration path are refused. This check does not provide
|
|
28
|
+
filesystem isolation against another process swapping directories concurrently; the default
|
|
29
|
+
print-only command creates no files.
|
|
30
|
+
|
|
31
|
+
The [client guide](client-connections.md) also covers Claude Code, a generic MCP descriptor and
|
|
32
|
+
`disconnect`. Removal affects only byte-identical managed files; modified files cause a conflict.
|
|
33
|
+
|
|
34
|
+
The generated server command uses the current Node executable, the installed compiled CLI entry,
|
|
35
|
+
and `mcp --root` with the repository's real path. The server is read-only by default. Candidate
|
|
36
|
+
execution remains unavailable unless an operator separately starts it with
|
|
37
|
+
`--allow-unsafe-execution`; that mode is explicitly UNSANDBOXED trusted-local.
|
|
38
|
+
|
|
39
|
+
On a connected read-only server, agents may call `assertledger_doctor` (or the legacy
|
|
40
|
+
`testforge_doctor` alias) with `{ "root": "..." }`. It returns the same static repository
|
|
41
|
+
initialization result as `assertledger doctor . --json`, after enforcing the server's allowed-root
|
|
42
|
+
boundary. The tool writes no configuration and does not execute repository code. It reports
|
|
43
|
+
configuration readiness only; dynamic dependency, reporter, permission and liveness diagnostics
|
|
44
|
+
remain outside this static check.
|
|
45
|
+
|
|
46
|
+
Use the separate [runtime doctor](runtime-doctor.md) after initialization to run controlled probes
|
|
47
|
+
with explicit authorization. [Reason-code explanations](diagnostics.md) remain available without
|
|
48
|
+
execution permission through the CLI, SDK and MCP.
|
|
49
|
+
|
|
50
|
+
`init` and every static doctor entry point require `assertledger.config.json` and
|
|
51
|
+
`assertledger.lock.json` to be regular files when they already exist. A symlink, dangling symlink,
|
|
52
|
+
directory or other file type returns `CONFLICT` with `INIT_MANAGED_PATH_UNSAFE` before its content is
|
|
53
|
+
read. Repositories that previously linked either managed file must replace the link with an
|
|
54
|
+
operator-owned regular file. This check assumes the trusted repository tree remains stable during
|
|
55
|
+
the operation; it is not a defense against a hostile concurrent path swap.
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
# Understand a refusal and choose the next action
|
|
2
|
+
|
|
3
|
+
Run `assertledger explain TARGET_STRENGTH_INSUFFICIENT` for an explanation. Add `--json` for
|
|
4
|
+
the strict `catalogueVersion: "1.0.0"` report. Several codes can be passed together; duplicates
|
|
5
|
+
are removed and results are sorted by code.
|
|
6
|
+
|
|
7
|
+
The same report is available through `new AssertLedger().explain(codes)` and the read-only
|
|
8
|
+
MCP tool `assertledger_explain` with `{ "codes": ["TARGET_STRENGTH_INSUFFICIENT"] }`.
|
|
9
|
+
`DiagnosticReportSchema` and `explainReasonCodes` are public package exports. Git qualification
|
|
10
|
+
includes this guidance in the terminal and its saved summary.
|
|
11
|
+
|
|
12
|
+
| Symptom | Meaning | Next step |
|
|
13
|
+
| --- | --- | --- |
|
|
14
|
+
| `TARGET_STRENGTH_INSUFFICIENT` | The test did not detect enough declared bugs by assertion. | Assert the corrected behavior and rerun against the same worlds. |
|
|
15
|
+
| `REFERENCE_NOT_GREEN` | The test fails on corrected code. | Fix the test or the reference before interpreting bug detection. |
|
|
16
|
+
| `OBSERVATIONS_DIVERGE` | Repeated runs disagree. | Remove nondeterministic inputs and repeat all attempts. |
|
|
17
|
+
| `CANDIDATE_DISCOVERY_INVALID` | The candidate was not identified as expected. | Check the committed path and runner selection. |
|
|
18
|
+
| `TIMEOUT` / `PROCESS_CRASH` | Execution did not yield acceptable assertion evidence. | Resolve the operational failure; it cannot count as a detected bug. |
|
|
19
|
+
| `GIT_REGRESSION_OUTPUT_EXISTS` | The chosen directory already contains something. | Choose a new evidence directory. |
|
|
20
|
+
| `MCP_REPOSITORY_ROOT_FORBIDDEN` | The server does not permit the requested repository. | Use the intended project root or a server configured by its operator. |
|
|
21
|
+
|
|
22
|
+
Guidance is derived presentation data. It does not change the verdict, evidence manifest,
|
|
23
|
+
decision digest, artifact digest or existing JSON outputs. Unknown codes remain unknown and
|
|
24
|
+
advise consulting the matching package version. Inputs accept only bounded uppercase reason
|
|
25
|
+
codes; arbitrary exception messages, source code, environment values and logs are not inputs.
|
|
26
|
+
|
|
27
|
+
## Français
|
|
28
|
+
|
|
29
|
+
La commande `assertledger explain CODE --json` rend les mêmes explications que le SDK et MCP.
|
|
30
|
+
Les explications du catalogue sont en anglais pour conserver un vocabulaire commun aux outils.
|
|
31
|
+
Un code inconnu reste signalé comme inconnu ; il ne devient jamais un résultat favorable.
|
|
32
|
+
|
|
33
|
+
Commencez par le premier contrôle en échec. Corrigez les erreurs d’exécution avant d’évaluer la
|
|
34
|
+
force du test. Un délai dépassé, une compilation impossible ou une exception générique ne
|
|
35
|
+
prouvent pas que le test détecte le défaut déclaré.
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
# Verify the installed package
|
|
2
|
+
|
|
3
|
+
Run `pnpm run smoke:package` from a checkout with its locked dependencies installed.
|
|
4
|
+
The command builds once, checks the schemas and conformance bundle, packs the distribution and
|
|
5
|
+
installs that tarball into a fresh temporary npm consumer whose path contains spaces.
|
|
6
|
+
Consumer dependencies are fetched from the public npm registry; lifecycle scripts
|
|
7
|
+
and the operator's npm configuration are disabled.
|
|
8
|
+
|
|
9
|
+
The smoke exercises both installed command aliases (`assertledger`, `testforge`),
|
|
10
|
+
both SDK class aliases, the public root and core exports, static init and audit,
|
|
11
|
+
and the bundled trusted `node:test` example. The strong test must be selected,
|
|
12
|
+
the weak test must be rejected as `WEAK_ORACLE`, and every replay rail must pass.
|
|
13
|
+
It also checks the advertised schemas, conformance bundle, integration skill and
|
|
14
|
+
runtime files in the tarball and installed package.
|
|
15
|
+
It also compiles a TypeScript consumer against the installed root/core exports
|
|
16
|
+
using the checkout's pinned compiler and Node type definitions. Consumer imports
|
|
17
|
+
resolve from the temporary project; no source-package alias is configured.
|
|
18
|
+
Runtime module resolution is confined to that consumer, so a dependency installed
|
|
19
|
+
elsewhere on the developer's machine cannot hide an incomplete package manifest.
|
|
20
|
+
The existing pinned TypeScript scanner is a production dependency of static repository analysis.
|
|
21
|
+
|
|
22
|
+
Three fault witnesses remove a runtime file and break runtime and TypeScript
|
|
23
|
+
exports. Each must fail for the expected module or type error, then pass after byte-exact
|
|
24
|
+
restoration in a fresh process. Raw command results, witnesses, manifest, replay,
|
|
25
|
+
the tarball and its SHA-256 are retained under `.testforge/package-smoke/<run-id>`.
|
|
26
|
+
Temporary consumers are removed after the run. The retained manifest is replayable;
|
|
27
|
+
its original execution paths refer to the removed consumer.
|
|
28
|
+
|
|
29
|
+
CI runs this smoke on Windows and Linux with Node 22.15.0 and 24 alongside `pnpm check`.
|
|
30
|
+
A successful local smoke proves that local tarball on the reported Node version
|
|
31
|
+
and operating system. It does not establish npm publication, cross-platform CI
|
|
32
|
+
success, hostile-code isolation, or adoption by external users. The example uses
|
|
33
|
+
explicitly unsandboxed trusted-local execution.
|
|
34
|
+
|
|
35
|
+
The installed `check` journey also evaluates a
|
|
36
|
+
[real historical correction projected into a node:test fixture](../examples/git-history/README.md).
|
|
37
|
+
It requires `VERIFIED` for the strong candidate, `REJECTED` for weak and generic-crash candidates,
|
|
38
|
+
two correct target observations each, valid replay, and unchanged source bytes, index and HEAD.
|
|
39
|
+
Each case retains its request, manifest and human summary. The upstream source snapshots,
|
|
40
|
+
license, commit IDs, Git blob IDs and SHA-256 hashes are bundled and checked before use.
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
# Qualifier un test de régression entre des révisions Git
|
|
2
|
+
|
|
3
|
+
La commande `check` vérifie qu’un test JavaScript `node:test` commité détecte une correction précise. Elle exécute le même test sur trois projections déclarées : la révision corrigée, la révision qui contient le bug et une révision neutre choisie par l’opérateur.
|
|
4
|
+
|
|
5
|
+
Remplacez `BEFORE`, `AFTER` et `NEUTRAL` par vos révisions. Cette commande s’utilise sur une ligne
|
|
6
|
+
sous PowerShell comme dans un shell POSIX :
|
|
7
|
+
|
|
8
|
+
```sh
|
|
9
|
+
assertledger check . --before BEFORE --after AFTER --neutral NEUTRAL --neutral-reason "raison explicite du contrôle" --test tests/regression.test.js --base-test tests/base.test.js --out .assertledger/evidence --allow-unsafe-execution
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
`--after` utilise `HEAD` par défaut. `--base-test` peut être répété. Le dossier donné à `--out` doit être un nouveau chemin relatif au dépôt. AssertLedger le réserve de manière exclusive, écrit `executed-request.json` et `summary.md`, puis publie `manifest.json` en dernier par renommage atomique. La présence de `manifest.json` est le marqueur de complétion ; un lecteur doit ignorer un dossier qui ne le contient pas.
|
|
13
|
+
|
|
14
|
+
L’exécution `trusted-local` est volontairement **UNSANDBOXED**. Le drapeau `--allow-unsafe-execution` constitue l’autorisation distincte de l’opérateur. L’API équivalente est `new AssertLedger().checkGitRegression(options)` et exige `allowUnsafeExecution: true`.
|
|
15
|
+
|
|
16
|
+
MCP expose le même parcours avec `assertledger_check` (alias `testforge_check`) seulement si
|
|
17
|
+
l’opérateur a démarré le serveur avec `--allow-unsafe-execution`. L’entrée reprend les options
|
|
18
|
+
du SDK, sans le champ de permission : `repository`, `before`, `after` facultatif, `neutral`,
|
|
19
|
+
`neutralReason`, `test`, `baseTests` et `out`. La racine est confinée aux dépôts autorisés ; le
|
|
20
|
+
dossier de sortie suit les mêmes contrôles que la CLI. Un client ne peut pas s’accorder cette
|
|
21
|
+
permission dans son message.
|
|
22
|
+
|
|
23
|
+
## Limites de cette première tranche
|
|
24
|
+
|
|
25
|
+
- seules les révisions commitées sont lues ; les changements locaux sont ignorés ;
|
|
26
|
+
- le test candidat doit être un fichier `.js`, `.mjs` ou `.cjs` utilisant les modules Node intégrés et des imports relatifs ;
|
|
27
|
+
- les dépendances d’exécution déclarées ne sont pas transportées ;
|
|
28
|
+
- le test de base doit être explicitement indiqué et identique dans les trois révisions ; les contrôles doivent découvrir au moins un vrai test ;
|
|
29
|
+
- les ensembles de chemins doivent rester identiques après retrait du test candidat ; les ajouts, suppressions, renommages, liens, sous-modules et modes exécutables sont refusés ;
|
|
30
|
+
- une révision neutre identique à la révision corrigée est acceptée comme contrôle répété, sans preuve indépendante de robustesse ;
|
|
31
|
+
- le rejeu vérifie l’intégrité du manifeste, mais ne réauthentifie pas les objets Git ni l’exécution passée.
|
|
32
|
+
|
|
33
|
+
Le verdict porte uniquement sur la faute déclarée. Un résultat `REJECTED` ne démontre pas que le test est inutile dans un autre contexte.
|
|
34
|
+
|
|
35
|
+
La sortie humaine et `summary.md` indiquent les SHA résolus, les observations sur le défaut,
|
|
36
|
+
les motifs du verdict et du candidat, les limites et une commande de rejeu adaptée au shell de
|
|
37
|
+
la machine. Le dossier de sortie ne peut pas se trouver dans les métadonnées Git. Les lectures
|
|
38
|
+
Git sont bornées ; le contenu est vérifié puis conservé en mémoire, dans la limite de 107 003 904
|
|
39
|
+
octets de données brutes, auxquels s’ajoutent des en-têtes bornés.
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
# Repository validation precedes runtime probing
|
|
2
|
+
|
|
3
|
+
The engine now checks the source repository's file and byte budgets before probing an adapter
|
|
4
|
+
runtime, running the `node:test` preflight, or creating the campaign directory. It still checks
|
|
5
|
+
the copied snapshot against the same budgets before executing a campaign.
|
|
6
|
+
|
|
7
|
+
Previously, an over-budget repository could return a runtime error first. A slow runtime preflight
|
|
8
|
+
could therefore mask `REPOSITORY_BYTES_BUDGET_EXCEEDED` with an engine error. The CLI now reliably
|
|
9
|
+
returns the existing repository-budget reason code and validation exit code `4` when that source
|
|
10
|
+
budget fails, including when the configured Node executable is unavailable.
|
|
11
|
+
|
|
12
|
+
For callers that submit several invalid inputs, error precedence changes: after request parsing
|
|
13
|
+
and repository-root resolution, source inventory and repository-budget errors precede runtime
|
|
14
|
+
probe errors. Callers should correct the reported repository problem before retrying runtime
|
|
15
|
+
qualification. With valid repository budgets, the existing runtime checks still apply.
|
|
16
|
+
|
|
17
|
+
No schema versions, error codes, manifest fields, gate semantics, digest projections, or unsafe
|
|
18
|
+
execution permissions change. Existing manifests retain their replay contract.
|
|
19
|
+
|
|
20
|
+
Compatibility witnesses in `tests/integration.test.ts` combine each invalid repository budget with
|
|
21
|
+
an unavailable Node executable. Both fail against the old ordering and pass with repository
|
|
22
|
+
validation first. The ordinary runtime and campaign tests cover valid-budget execution.
|
|
23
|
+
|
|
24
|
+
## Qualification output budgets
|
|
25
|
+
|
|
26
|
+
Both executable identity probes and both runtime preflight probes now propagate the caller's
|
|
27
|
+
`maximumOutputBytes`, capped by the existing internal 64 KiB ceiling. Previously these processes
|
|
28
|
+
always used 64 KiB, even when the caller requested less. Controlled reporter files retain their
|
|
29
|
+
independent fixed bound; this correction concerns captured stdout and stderr.
|
|
30
|
+
|
|
31
|
+
A runtime installed at a long path can exceed a small capture budget when reporting its identity.
|
|
32
|
+
That case now returns the existing `NODE_TEST_EXECUTABLE_PROBE_FAILED` validation error instead of
|
|
33
|
+
continuing with output captured beyond the requested limit. Increase the request budget explicitly
|
|
34
|
+
when needed. No schema, digest, or decision semantics change. Runtime tests include a real copied
|
|
35
|
+
Node executable at a long path and a runner witness for both the lower caller cap and upper safety cap.
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
# Migrating from TestForge to AssertLedger
|
|
2
|
+
|
|
3
|
+
AssertLedger is a rename, not a protocol break. The evidence model, gates, decision semantics, and
|
|
4
|
+
every wire-level identifier described in [proof-model.md](proof-model.md) and
|
|
5
|
+
[conformance-v1.md](conformance-v1.md) are unchanged. Existing integrations continue to work without
|
|
6
|
+
modification during the compatibility window; there is no forced migration date.
|
|
7
|
+
|
|
8
|
+
## What changed
|
|
9
|
+
|
|
10
|
+
- The preferred public name is **AssertLedger**; the npm package is `assertledger`.
|
|
11
|
+
- The preferred SDK entry point is the `AssertLedger` class.
|
|
12
|
+
- The preferred CLI binary is `assertledger`.
|
|
13
|
+
- The preferred MCP server name and tool prefix is `assertledger`.
|
|
14
|
+
|
|
15
|
+
## What did not change
|
|
16
|
+
|
|
17
|
+
Nothing at the wire level. The following identifiers are frozen for replay and interoperability
|
|
18
|
+
compatibility and are **not** renamed by this migration:
|
|
19
|
+
|
|
20
|
+
| Identifier | Value | Where |
|
|
21
|
+
| --- | --- | --- |
|
|
22
|
+
| Schema `$id` domain | `https://testforge.dev/schemas/...` | every published JSON Schema |
|
|
23
|
+
| H3 payload schema domain | `https://testforge.dev/payloads/...` | `src/contracts/index.ts` |
|
|
24
|
+
| Structured-command adapter kind | `"testforge-command"` | verification request/manifest `adapter.kind` |
|
|
25
|
+
| Reserved environment prefix | `TESTFORGE_*` | `RESERVED_ENVIRONMENT_VARIABLE` allowlist rejection |
|
|
26
|
+
| Corpus allocation canonicalization domain | `"TESTFORGE_AGENTIC_CORPUS_CASE_V1"` | `src/contracts/index.ts` |
|
|
27
|
+
| Observation taxonomy domain | `"TESTFORGE_OBSERVATION_V1"` | `src/contracts/index.ts` |
|
|
28
|
+
| H3 structured-result protocol | `"TESTFORGE_H3_STRUCTURED_RESULT_V1"` | `src/contracts/index.ts` |
|
|
29
|
+
| Evidence context engine name | `"testforge"` | `evidenceContext.engine.name` in every manifest |
|
|
30
|
+
| Repository-digest exclusion directory | `.testforge` | repository analyzer default excludes |
|
|
31
|
+
| All published schema bytes and conformance bundle digests | unchanged | `schemas/`, `conformance/v1/` |
|
|
32
|
+
|
|
33
|
+
Changing any of these would be a public contract change requiring a major version and a documented
|
|
34
|
+
compatibility break under [AGENTS.md](../AGENTS.md). This migration deliberately does not do that.
|
|
35
|
+
|
|
36
|
+
## Compatibility surface
|
|
37
|
+
|
|
38
|
+
| Surface | Preferred | Legacy (still supported) |
|
|
39
|
+
| --- | --- | --- |
|
|
40
|
+
| npm package | `assertledger` | — (package was renamed; `testforge` was never published) |
|
|
41
|
+
| CLI binary | `assertledger` | `testforge` — installed alongside, identical `dist/cli.js` entrypoint |
|
|
42
|
+
| SDK class | `AssertLedger` | `TestForge` — a `class TestForge extends AssertLedger {}` subclass with an identical surface |
|
|
43
|
+
| MCP server name | `assertledger` | reported server name; no separate legacy server identity |
|
|
44
|
+
| MCP tool names | `assertledger_*` | `testforge_*` — every tool is registered twice with the exact same handler and tool configuration; every pair except the variable schema-document lookup also shares one output-schema object |
|
|
45
|
+
| MCP factory function | `createAssertLedgerServer` | `createTestForgeServer` — a direct alias (`export const createTestForgeServer = createAssertLedgerServer`), not a reimplementation |
|
|
46
|
+
|
|
47
|
+
Every alias pair calls the identical underlying function. None of the facades (`src/sdk`, `src/cli.ts`,
|
|
48
|
+
`src/mcp`) reimplement a gate, a schema, or a digest computation for either name.
|
|
49
|
+
|
|
50
|
+
## What operators and integrators should do
|
|
51
|
+
|
|
52
|
+
- New integrations: use the `assertledger` package/binary, the `AssertLedger` SDK class, and the
|
|
53
|
+
`assertledger_*` MCP tool names.
|
|
54
|
+
- Existing integrations: no action is required. `testforge`, `TestForge`, and `testforge_*` continue
|
|
55
|
+
to resolve to the same code.
|
|
56
|
+
- Do not expect a new schema version, a new adapter kind, or a new reserved environment prefix. None
|
|
57
|
+
was introduced by this rename.
|
|
58
|
+
|
|
59
|
+
## What is deliberately not covered by this migration
|
|
60
|
+
|
|
61
|
+
- No npm publish, GitHub repository rename, domain change, or release under the new name — those are
|
|
62
|
+
external delivery actions outside this compatibility slice.
|
|
63
|
+
- No change to the `conformance/v1/` bundle or `schemas/*.json` bytes.
|
|
64
|
+
- No removal date for the `testforge`/`TestForge`/`testforge_*` aliases has been set.
|
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
# Intention et cap d’AssertLedger
|
|
2
|
+
|
|
3
|
+
État des lieux et cap ratifié le 8 septembre 2026. Le propriétaire a demandé d’aligner Linear et
|
|
4
|
+
les issues sur la qualification de tests de correction, puis de poursuivre jusqu’à une release
|
|
5
|
+
1.0 avec une prise en main soignée. Les contrats de preuve existants restent applicables.
|
|
6
|
+
|
|
7
|
+
## Cap 1.0 décidé
|
|
8
|
+
|
|
9
|
+
**Vérifiez que le test détecte réellement le bug corrigé, avant de l’accepter.** La cible initiale
|
|
10
|
+
est une équipe SaaS Node.js/TypeScript qui utilise des agents. Le développeur désigne le test et
|
|
11
|
+
la correction ; le tech lead obtient un verdict, les raisons, les révisions et les limites.
|
|
12
|
+
|
|
13
|
+
Le premier parcours s’appuie sur `node:test`, une version fautive, une référence corrigée et un
|
|
14
|
+
contrôle neutre justifié. Les interfaces préparent les entrées du vérificateur sans redéfinir ses
|
|
15
|
+
gates. La première exécution reste réservée au code de confiance avec opt-in explicite
|
|
16
|
+
`UNSANDBOXED`. L’ouverture aux contributions hostiles exige une isolation qualifiée.
|
|
17
|
+
|
|
18
|
+
Le projet Linear est désormais [AssertLedger 1.0 — Tests de régression vérifiés](https://linear.app/hoklims/project/assertledger-10-tests-de-regression-verifies-acf768519152).
|
|
19
|
+
Le dépôt public dédié est [hoklims/assertledger](https://github.com/hoklims/assertledger).
|
|
20
|
+
Le [contrat de livraison 1.0](release-1.0.md) décrit le parcours, les critères et les travaux différés.
|
|
21
|
+
Les sections ci-dessous conservent les constats de départ ; leur séquencement est précisé par ce cap.
|
|
22
|
+
|
|
23
|
+
## Le résultat recherché
|
|
24
|
+
|
|
25
|
+
Donner au développeur qui travaille avec des agents une réponse exploitable à cette question :
|
|
26
|
+
**quels tests ajoutent une protection démontrée contre les fautes visées, et combien coûte leur
|
|
27
|
+
exécution ?**
|
|
28
|
+
|
|
29
|
+
L’agent propose des tests. AssertLedger les exécute sur une référence, des variantes fautives et
|
|
30
|
+
des variantes censées préserver le comportement. Il explique ensuite ce que les observations
|
|
31
|
+
permettent de conclure. Le développeur conserve l’autorité sur le comportement attendu, les
|
|
32
|
+
variantes pertinentes et l’acceptation du changement.
|
|
33
|
+
|
|
34
|
+
L’usage initial proposé est le durcissement de dépôts de confiance utilisés par Guillaume et ses
|
|
35
|
+
agents. Un premier parcours doit aboutir à une preuve utile sur un dépôt consommateur, puis être
|
|
36
|
+
reproduit depuis un autre harness. L’ouverture à du code non fiable exige une isolation distincte.
|
|
37
|
+
|
|
38
|
+
## Trois résultats à distinguer
|
|
39
|
+
|
|
40
|
+
| Résultat | Question posée | Limite |
|
|
41
|
+
| --- | --- | --- |
|
|
42
|
+
| Qualification d’un candidat | Ce test passe-t-il sur la référence et les neutres, et détecte-t-il les fautes exigées ? | La conclusion porte sur les mondes et essais déclarés. |
|
|
43
|
+
| Choix d’un ensemble de tests | Quelle protection supplémentaire apporte chaque test, pour quel coût mesuré ? | La sélection actuelle porte sur les candidats déjà éligibles et sélectionnés par la source. |
|
|
44
|
+
| Validation empirique du produit | Cette méthode conserve-t-elle la détection de vrais défauts hors calibration ? | Il faut des expériences pré-engagées et un holdout indépendant. |
|
|
45
|
+
|
|
46
|
+
La rapidité intervient après les exigences de preuve. Elle ne compense jamais un échec de
|
|
47
|
+
référence, un monde neutre rouge, une observation instable ou une attribution douteuse.
|
|
48
|
+
|
|
49
|
+
## Ce qui existe réellement
|
|
50
|
+
|
|
51
|
+
Le socle local comprend les contrats versionnés, le noyau déterministe, les digests, le replay
|
|
52
|
+
sémantique, le moteur d’exécution, les façades CLI/SDK/MCP, l’audit statique et l’initialisation.
|
|
53
|
+
Le registre contient 34 schémas JSON, tous présents dans le verrou de conformance. L’ancien
|
|
54
|
+
décompte de 30 dans la roadmap était périmé.
|
|
55
|
+
|
|
56
|
+
L’adaptateur officiel intégré est `node:test`. Le protocole `testforge-command` permet d’intégrer
|
|
57
|
+
un autre framework avec un adaptateur fourni par l’opérateur. Détecter Vitest, Jest, Bun ou pytest
|
|
58
|
+
pendant `init` ne fournit pas leur adaptateur officiel.
|
|
59
|
+
|
|
60
|
+
Les profils v1 et v2, les artefacts de benchmark, les contrats de provenance et le protocole
|
|
61
|
+
expérimental H3 sont présents. Le benchmark par phases refuse encore l’adaptateur `node:test` :
|
|
62
|
+
ses événements ne suffisent pas à mesurer fidèlement les quatre phases demandées.
|
|
63
|
+
|
|
64
|
+
La documentation rapporte huit cas TestExplora admis pour calibration curatée. Elle ne fournit
|
|
65
|
+
pas de holdout réel qualifié. Cette reprise ne réexécute pas ces campagnes historiques et n’accède
|
|
66
|
+
pas aux données privées du holdout.
|
|
67
|
+
|
|
68
|
+
Le dépôt est sur `feat/testforge-core`, HEAD `cb1dde2c2fe0f66e46652e7223086dd04b460c67`.
|
|
69
|
+
À la reprise, 29 fichiers suivis sont modifiés, de nombreux nouveaux fichiers restent non suivis
|
|
70
|
+
et aucun remote Git n’est configuré. Du code local, même testé, ne constitue donc pas une release
|
|
71
|
+
publiée. La qualification `node:test` HOK-570 reste `In Progress` dans Linear.
|
|
72
|
+
|
|
73
|
+
## Les écarts qui empêchent l’usage visé
|
|
74
|
+
|
|
75
|
+
Le parcours Linear annoncé est `install → init → connect → doctor → check --changed → land → replay CI`.
|
|
76
|
+
`connect` est maintenant utilisable en CLI, et le diagnostic statique `doctor` en CLI, SDK et MCP ;
|
|
77
|
+
`check --changed` et `land` restent des travaux de backlog. `init` produit une configuration et un
|
|
78
|
+
lock ; il laisse explicitement les mondes et les candidats à fournir par l’opérateur ou le harness.
|
|
79
|
+
|
|
80
|
+
Le premier obstacle d’adoption est donc le passage entre « dépôt initialisé » et « campagne
|
|
81
|
+
pertinente ». Ajouter des adaptateurs aide à exécuter les tests, mais ne décide pas quelles fautes
|
|
82
|
+
et quels comportements les tests doivent représenter.
|
|
83
|
+
|
|
84
|
+
Un résultat historique illustre une seconde difficulté : la campagne sur les 36 tests du noyau
|
|
85
|
+
exigeait que chaque candidat tue les quatre cibles obligatoires. Tous ont été rejetés, alors que
|
|
86
|
+
six détectaient une cible. C’est le résultat attendu de la politique choisie. Il ne permet pas
|
|
87
|
+
de conclure que ces tests n’ont aucune valeur sur leurs responsabilités respectives.
|
|
88
|
+
|
|
89
|
+
Pour le pilote, les mondes obligatoires doivent donc correspondre à la responsabilité du candidat.
|
|
90
|
+
Si l’on veut accepter un ensemble dont les membres se complètent sans être individuellement
|
|
91
|
+
éligibles, il faudra un contrat distinct, des contre-exemples et une décision explicite. Cette
|
|
92
|
+
proposition ne modifie pas les gates v1 pour obtenir un résultat plus flatteur.
|
|
93
|
+
|
|
94
|
+
Enfin, l’intégrité d’un manifeste et l’authenticité de ses observations sont deux propriétés
|
|
95
|
+
différentes. Un replay valide ne prouve ni que les octets ont été exécutés, ni que le producteur est
|
|
96
|
+
indépendant. La politique externe de preuve et ses reçus restent une responsabilité séparée.
|
|
97
|
+
|
|
98
|
+
## Ordre de reprise proposé
|
|
99
|
+
|
|
100
|
+
### 1. Fiabiliser et conserver le socle existant
|
|
101
|
+
|
|
102
|
+
Terminer la vérification de la qualification `node:test`, corriger les échecs reproductibles,
|
|
103
|
+
conserver les preuves exactes et inventorier le diff hérité avant tout checkpoint Git.
|
|
104
|
+
|
|
105
|
+
Critère de sortie : `pnpm check` vert, témoins adversariaux pertinents observés, campagne minimale
|
|
106
|
+
rejouable et limites d’acceptation explicites. La revue du système de preuve porte sur l’agrégat
|
|
107
|
+
qui sera réellement accepté. Le durcissement global HOK-406 n’autorise pas implicitement une
|
|
108
|
+
modification de la politique active et ne remplace pas le reçu attendu.
|
|
109
|
+
|
|
110
|
+
### 2. Démontrer un usage complet dans un dépôt consommateur
|
|
111
|
+
|
|
112
|
+
Choisir une responsabilité métier bornée, une référence, une faute connue et un monde neutre
|
|
113
|
+
justifié. Comparer au moins un test utile à un test qui passe sans détecter la faute. Présenter le
|
|
114
|
+
verdict, ses raisons, les limites et les octets concernés depuis un harness, puis rejouer le même
|
|
115
|
+
artefact depuis un second point d’entrée.
|
|
116
|
+
|
|
117
|
+
Critères de sortie proposés : installation et parcours documentés exécutés depuis un consommateur
|
|
118
|
+
neuf ; premier résultat compréhensible ; échec générique et timeout refusés comme preuves ; mêmes
|
|
119
|
+
décisions au replay ; durée jusqu’à la première preuve mesurée. L’objectif historique de cinq
|
|
120
|
+
minutes reste un objectif à mesurer, pas une performance acquise.
|
|
121
|
+
|
|
122
|
+
L’adaptateur suivant dépend du dépôt choisi. HOK-571 prépare Vitest ; HOK-574 prépare pytest.
|
|
123
|
+
Les cinq adaptateurs de HOK-422 restent prévus, mais ne doivent pas tous précéder le premier
|
|
124
|
+
apprentissage d’usage. Le branchement au harness et les diagnostics se construisent à partir des
|
|
125
|
+
obstacles observés dans ce parcours.
|
|
126
|
+
|
|
127
|
+
### 3. Mesurer le gain, puis élargir
|
|
128
|
+
|
|
129
|
+
Mesurer d’abord le coût complet de la qualification, le temps de feedback et la protection
|
|
130
|
+
supplémentaire apportée dans un contexte fixé. Une somme de p95 individuels ne constitue pas une
|
|
131
|
+
mesure du p95 du portefeuille réellement exécuté.
|
|
132
|
+
|
|
133
|
+
Reprendre ensuite le corpus et les expériences H1–H4 avec les autorités, sources et engagements
|
|
134
|
+
prévus. La calibration peut produire un résultat nul ou défavorable : ce résultat doit rester
|
|
135
|
+
publiable. Étendre les adaptateurs, le cache et l’isolation selon les usages vérifiés.
|
|
136
|
+
|
|
137
|
+
Critère de sortie pour une affirmation empirique : hypothèses et seuils pré-engagés, provenance
|
|
138
|
+
complète, holdout sous garde indépendante et critères satisfaits. Un corpus mécaniquement `READY`
|
|
139
|
+
ne signifie pas que les hypothèses sont confirmées.
|
|
140
|
+
|
|
141
|
+
## Mesurer la valeur sans inventer un score universel
|
|
142
|
+
|
|
143
|
+
| Mesure proposée | Ce qu’elle aide à décider |
|
|
144
|
+
| --- | --- |
|
|
145
|
+
| Temps jusqu’à la première campagne comprise et rejouée | Le parcours est-il réellement accessible ? |
|
|
146
|
+
| Fautes pertinentes détectées et contrôles conservés verts | Le test protège-t-il la responsabilité visée ? |
|
|
147
|
+
| Faux positifs sur les erreurs opérationnelles | Le verdict accorde-t-il indûment de la confiance ? |
|
|
148
|
+
| Coût total de qualification et coût du feedback courant | Le bénéfice justifie-t-il la dépense ? |
|
|
149
|
+
| Protection marginale d’un test ou d’un ensemble | L’ajout apporte-t-il quelque chose de démontrable ? |
|
|
150
|
+
|
|
151
|
+
Ces mesures nécessitent un contexte et un protocole. Aucun gain, taux d’adoption ou classement
|
|
152
|
+
entre machines n’est établi par ce document.
|
|
153
|
+
|
|
154
|
+
## Frontières conservées
|
|
155
|
+
|
|
156
|
+
- Le noyau décide sans I/O, horloge, réseau, hasard ou jugement de modèle.
|
|
157
|
+
- Seul un échec d’assertion attribué peut compter comme détection d’une cible en v1.
|
|
158
|
+
- `trusted-local` reste explicitement non isolé ; ses ressources bornées ne forment pas une sandbox.
|
|
159
|
+
- Les mondes, politiques et pins de confiance ne deviennent pas valides parce qu’un agent les fournit.
|
|
160
|
+
- Les catching tests, qui doivent échouer sur une révision proposée, restent hors du mode hardening v1.
|
|
161
|
+
- Une évolution de schéma, de digest, d’éligibilité ou de permission exige son contrat et ses preuves.
|
|
162
|
+
- La publication, l’acceptation externe et la validation scientifique conservent leurs critères propres.
|
|
163
|
+
|
|
164
|
+
## Sources de reprise
|
|
165
|
+
|
|
166
|
+
- [Contrat de preuve](proof-model.md), [architecture](architecture.md),
|
|
167
|
+
[initialisation](repository-init.md) et [adaptateurs](adapter-protocol.md).
|
|
168
|
+
- [Profil v1](agentic-test-profile.md), [profil v2](agentic-test-profile-v2.md),
|
|
169
|
+
[plan de corpus](agentic-corpus-plan.md) et [résultat historique sur les tests du noyau](../benchmarks/self-hosted-core/RESULT-existing-tests-v1.md).
|
|
170
|
+
- [Projet Linear](https://linear.app/hoklims/project/assertledger-agentic-test-profile-acf768519152),
|
|
171
|
+
[parcours HOK-419](https://linear.app/hoklims/issue/HOK-419),
|
|
172
|
+
[qualification HOK-570](https://linear.app/hoklims/issue/HOK-570),
|
|
173
|
+
[politique externe HOK-406](https://linear.app/hoklims/issue/HOK-406).
|
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
# Evidence and decision model
|
|
2
|
+
|
|
3
|
+
AssertLedger uses “proof” to mean recorded operational evidence, not mathematical verification of a
|
|
4
|
+
program.
|
|
5
|
+
|
|
6
|
+
## What `VERIFIED` means
|
|
7
|
+
|
|
8
|
+
For the repository digest scope, worlds, adapter configuration, policy, attempts, and execution
|
|
9
|
+
metadata recorded in the manifest, every selected candidate:
|
|
10
|
+
|
|
11
|
+
1. stayed within configured candidate roots and size budgets;
|
|
12
|
+
2. ran from the campaign snapshot in a fresh execution workspace;
|
|
13
|
+
3. produced complete, stable, attributed candidate observations;
|
|
14
|
+
4. passed every reference and neutral world;
|
|
15
|
+
5. produced attributed assertion failures for every required target and met the weighted threshold;
|
|
16
|
+
6. was selected by the published deterministic ordering.
|
|
17
|
+
|
|
18
|
+
This statement applies only to the recorded attempts and worlds. It does not generalize to
|
|
19
|
+
unexecuted states, future environments, or faults outside the supplied targets.
|
|
20
|
+
|
|
21
|
+
## Outcome taxonomy
|
|
22
|
+
|
|
23
|
+
| Outcome | Meaning | Can kill a target in v1? |
|
|
24
|
+
| --- | --- | --- |
|
|
25
|
+
| `PASS` | Runner completed successfully | No |
|
|
26
|
+
| `ASSERTION_FAILURE` | Adapter attributed an assertion failure to the candidate | Yes |
|
|
27
|
+
| `COLLECTION_FAILURE` | Runner could not collect tests | No |
|
|
28
|
+
| `COMPILE_FAILURE` | Code could not compile or load | No |
|
|
29
|
+
| `PROCESS_CRASH` | Runner failed without an admitted assertion | No |
|
|
30
|
+
| `TIMEOUT` | Execution exceeded its time budget | No |
|
|
31
|
+
| `INFRA_ERROR` | Runner, adapter, report, or host failed | No |
|
|
32
|
+
| `NO_TEST_DISCOVERED` | Adapter found no candidate test | No |
|
|
33
|
+
|
|
34
|
+
Only an attributed `ASSERTION_FAILURE` kills a target in policy v1. A caller cannot configure
|
|
35
|
+
timeouts, crashes, compilation failures, or collection failures as accepted target outcomes.
|
|
36
|
+
|
|
37
|
+
## Candidate gates
|
|
38
|
+
|
|
39
|
+
The core evaluates gates in this order and records a status, evidence run IDs, and reason codes for
|
|
40
|
+
each gate:
|
|
41
|
+
|
|
42
|
+
| Gate | Condition |
|
|
43
|
+
| --- | --- |
|
|
44
|
+
| `COMPLETENESS` | Every world has exactly the required attempts |
|
|
45
|
+
| `STABILITY` | Normalized outcomes agree across attempts |
|
|
46
|
+
| `DISCOVERY` | Every candidate run reports at least one attributed candidate test |
|
|
47
|
+
| `REFERENCE` | All reference observations pass |
|
|
48
|
+
| `NEUTRAL` | All neutral observations pass |
|
|
49
|
+
| `TARGET_STRENGTH` | Required targets are killed and the weighted threshold is met |
|
|
50
|
+
|
|
51
|
+
A failed prerequisite leaves later dependent gates as `NOT_RUN`. Repetition measures observed
|
|
52
|
+
stability; a successful retry never erases a contradictory attempt.
|
|
53
|
+
|
|
54
|
+
Candidate statuses are:
|
|
55
|
+
|
|
56
|
+
- `ELIGIBLE`: every gate needed by policy passed;
|
|
57
|
+
- `WEAK_ORACLE`: reference and neutral evidence passed, but target strength did not;
|
|
58
|
+
- `INVALID`: discovery, reference, or neutral evidence failed;
|
|
59
|
+
- `UNSTABLE`: complete attempts produced different normalized outcomes;
|
|
60
|
+
- `INCONCLUSIVE`: required observations were missing or execution produced `TIMEOUT` or
|
|
61
|
+
`INFRA_ERROR`.
|
|
62
|
+
|
|
63
|
+
## Campaign statuses
|
|
64
|
+
|
|
65
|
+
- `VERIFIED`: at least one eligible candidate was selected.
|
|
66
|
+
- `REJECTED`: evidence was complete and conclusive, but no candidate was eligible.
|
|
67
|
+
- `INCONCLUSIVE`: controls were invalid, or at least one candidate was unstable or inconclusive.
|
|
68
|
+
- `ENGINE_ERROR`: the core could not normalize the supplied evidence safely.
|
|
69
|
+
|
|
70
|
+
## Controls and attribution
|
|
71
|
+
|
|
72
|
+
Every world runs without a candidate before candidate runs. A valid control is complete, stable,
|
|
73
|
+
`PASS`, reports zero candidate tests, and reports no candidate attribution. One invalid control makes
|
|
74
|
+
the campaign `INCONCLUSIVE`.
|
|
75
|
+
|
|
76
|
+
Attribution comes from the adapter and remains part of the trusted computing base. The built-in
|
|
77
|
+
`node:test` adapter consumes runner events and associates them with resolved candidate file paths;
|
|
78
|
+
the structured-command adapter trusts its strict runtime report. See
|
|
79
|
+
[adapter-protocol.md](adapter-protocol.md).
|
|
80
|
+
|
|
81
|
+
## Decision and artifact digests
|
|
82
|
+
|
|
83
|
+
The manifest carries two SHA-256 integrity values with different scopes:
|
|
84
|
+
|
|
85
|
+
- `decisionDigest` binds decision-relevant normalized evidence: schema and policy versions,
|
|
86
|
+
repository digest, evidence context and world provenance, candidate assessments, gates,
|
|
87
|
+
decision-relevant observations, and the final selection. It excludes observation duration, process
|
|
88
|
+
exit code, and stdout/stderr digests.
|
|
89
|
+
- `artifactDigest` binds the complete emitted manifest except `artifactDigest` itself. It therefore
|
|
90
|
+
detects changes to operational observation fields and disclosure metadata that do not alter the
|
|
91
|
+
decision digest.
|
|
92
|
+
|
|
93
|
+
`AssertLedger.replay()` validates the manifest contract, recomputes both digests, and recomputes
|
|
94
|
+
candidate assessments and the final decision from the recorded normalized evidence. Its result
|
|
95
|
+
includes `schemaValid`, `decisionDigestValid`, `artifactDigestValid`, and
|
|
96
|
+
`decisionSemanticsValid`; `valid` requires all four rails. Replay does not rerun tests, prove that
|
|
97
|
+
observations were truthful, authenticate the producer, or replace a signed external attestation.
|
|
98
|
+
|
|
99
|
+
The manifest alone is not a self-contained reproduction bundle. It stores candidate and world
|
|
100
|
+
digests rather than their file bodies, stdout/stderr digests rather than raw logs, and environment
|
|
101
|
+
allowlist names rather than effective values. The built-in `node:test` adapter records its resolved
|
|
102
|
+
executable real path, version, and SHA-256 digest; the structured-command adapter records only its
|
|
103
|
+
configured command. Preserve the original request, repository snapshot, dependencies, missing
|
|
104
|
+
executable identities, and raw logs separately when independent audit matters.
|
|
105
|
+
|
|
106
|
+
## Explicit non-claims
|
|
107
|
+
|
|
108
|
+
AssertLedger does not prove:
|
|
109
|
+
|
|
110
|
+
- absence of bugs or complete fault detection;
|
|
111
|
+
- universal oracle quality;
|
|
112
|
+
- absence of flakiness beyond observed attempts;
|
|
113
|
+
- semantic relevance or correctness of operator-supplied worlds;
|
|
114
|
+
- resistance to a candidate designed to recognize the worlds;
|
|
115
|
+
- containment of hostile code under `trusted-local`;
|
|
116
|
+
- provenance authenticity without a separate signed attestation.
|