@agentskit/doc-bridge 1.6.4 → 1.7.44
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +243 -0
- package/action.yml +1 -1
- package/dist/cli/program.js +793 -137
- package/dist/cli/program.js.map +1 -1
- package/dist/config/index.d.ts +1 -1
- package/dist/config/index.js +38 -2
- package/dist/config/index.js.map +1 -1
- package/dist/{index-DudNuwI5.d.ts → index-C2PCQSrB.d.ts} +216 -25
- package/dist/index.d.ts +837 -67
- package/dist/index.js +858 -127
- package/dist/index.js.map +1 -1
- package/docs/PRD-enterprise-hardening.md +288 -0
- package/docs/adr/0001-enterprise-verification-contract.md +35 -0
- package/docs/knowledge-engine-runbook.md +18 -2
- package/docs/spec/analyzer-plugin-v1.md +24 -0
- package/docs/spec/benchmark-v1.md +30 -0
- package/docs/spec/config-v1.md +111 -0
- package/docs/validation-cycle-plan.md +236 -0
- package/docs/verification-harness.md +33 -4
- package/mcpb/manifest.json +1 -1
- package/package.json +1 -1
- package/scripts/report-visual-check.mjs +45 -10
- package/scripts/verification-harness.mjs +216 -13
- package/skills/doc-bridge-handoff/scripts/resolve-handoff.mjs +1 -1
- package/src/agents/registry-adapter.ts +31 -7
- package/src/cli/program.ts +44 -11
- package/src/config/index.ts +2 -0
- package/src/config/schema.ts +56 -0
- package/src/discovery/documentation.ts +46 -5
- package/src/discovery/repository.ts +95 -16
- package/src/index.ts +29 -0
- package/src/metrics/benchmark.ts +176 -0
- package/src/plugins/contract.ts +89 -0
- package/src/reconciliation/reconcile.ts +137 -3
- package/src/report/html.ts +302 -78
- package/src/schemas/knowledge.ts +16 -1
- package/src/version.ts +1 -1
- package/src/workflow/engine.ts +65 -9
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
---
|
|
2
|
+
title: Doc Bridge validation cycle plan
|
|
3
|
+
description: Evidence-driven validation of Doc Bridge against the AgentsKit OS repository.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Doc Bridge validation cycle plan
|
|
7
|
+
|
|
8
|
+
## Objective
|
|
9
|
+
|
|
10
|
+
Validate the complete Doc Bridge change against `agentskit-os`, not only the published package or the HTML report. The validation must prove the bridge between repository structure, documentation, agents, and humans:
|
|
11
|
+
|
|
12
|
+
1. repository architecture is discovered at package, module, and file levels;
|
|
13
|
+
2. documentation is classified, indexed, and checked for freshness;
|
|
14
|
+
3. documentation claims are reconciled with observed code relationships;
|
|
15
|
+
4. disconnected, stale, conflicting, unresolved, and not-analyzed areas are visible;
|
|
16
|
+
5. the Registry agent proposes grounded follow-up work without becoming an unverified authority;
|
|
17
|
+
6. the report makes those results useful and navigable;
|
|
18
|
+
7. runs are reproducible, resumable, versioned, idempotent, and cost-bounded.
|
|
19
|
+
|
|
20
|
+
This plan is a validation contract. A green result from one cycle never substitutes for a missing cycle.
|
|
21
|
+
|
|
22
|
+
## Evidence rules
|
|
23
|
+
|
|
24
|
+
- Every result is tied to the source revision, configuration hash, snapshot hash, report hash, and verification run ID.
|
|
25
|
+
- `not-analyzed` is visible and counts against coverage unless an explicit, human-approved exemption exists.
|
|
26
|
+
- Unit tests and compilation are supporting evidence. They do not prove repository behavior, documentation agreement, endpoint/database behavior, or UI behavior.
|
|
27
|
+
- Deterministic agent runs and assisted agent runs are separate evidence classes. Deterministic output proves replayability; it does not prove semantic quality.
|
|
28
|
+
- No agent proposal is applied without human approval and a fresh post-apply verification run.
|
|
29
|
+
|
|
30
|
+
## Cycle gates and success metrics
|
|
31
|
+
|
|
32
|
+
### Cycle 0 — Contract and scope gate
|
|
33
|
+
|
|
34
|
+
**Purpose:** map the human request to verifiable outcomes before changing the product or narrowing the target.
|
|
35
|
+
|
|
36
|
+
**Required evidence:** task contract, acceptance matrix, repository/surface matrix, explicit exemptions, validation budget, and the decision for architecture declaration granularity.
|
|
37
|
+
|
|
38
|
+
**Success metrics:**
|
|
39
|
+
|
|
40
|
+
- 100% of requested outcomes mapped to one or more executable checks;
|
|
41
|
+
- 0 unresolved scope, authorization, or behavior ambiguities;
|
|
42
|
+
- 0 unapproved surface exclusions;
|
|
43
|
+
- 100% of required checks have an evidence location and an expected result;
|
|
44
|
+
- the contract names what the report must show for architecture, stale docs, doc/code drift, disconnected areas, Registry proposals, and UI.
|
|
45
|
+
|
|
46
|
+
**Stop condition:** any unmapped outcome returns the task to `CLARIFYING`.
|
|
47
|
+
|
|
48
|
+
### Cycle 1 — Real consumer and reproducibility
|
|
49
|
+
|
|
50
|
+
**Purpose:** prove the published package is the artifact used by `agentskit-os` and that the pipeline is repeatable.
|
|
51
|
+
|
|
52
|
+
**Checks:** install the declared package version, run discovery, index, reconcile, check, report, and Registry agent using the target repository.
|
|
53
|
+
|
|
54
|
+
**Success metrics:**
|
|
55
|
+
|
|
56
|
+
- installed version equals the declared and published version;
|
|
57
|
+
- two unchanged runs produce identical snapshot, report, and deterministic proposal hashes;
|
|
58
|
+
- every generated artifact is linked from the workflow run;
|
|
59
|
+
- resume after an interrupted stage completes without duplicating or corrupting artifacts;
|
|
60
|
+
- local execution is used unless a contract explicitly requires network or CI.
|
|
61
|
+
|
|
62
|
+
### Cycle 2 — Architecture discovery
|
|
63
|
+
|
|
64
|
+
**Purpose:** validate the project topology rather than merely listing dependencies.
|
|
65
|
+
|
|
66
|
+
**Checks:** compare discovered packages/apps/modules/files and relations with the monorepo manifests, workspace configuration, representative source imports/exports, and known application boundaries.
|
|
67
|
+
|
|
68
|
+
**Success metrics:**
|
|
69
|
+
|
|
70
|
+
- 100% of in-scope workspace packages and apps are classified;
|
|
71
|
+
- package, module, and file graph levels are independently navigable;
|
|
72
|
+
- relation kinds and direction are preserved;
|
|
73
|
+
- disconnected nodes and high-connectivity/SPOF candidates are reported with code evidence;
|
|
74
|
+
- dynamic imports, runtime wiring, and generated code are either analyzed or explicitly reported as `not-analyzed`;
|
|
75
|
+
- no architecture claim is inferred from `package.json` dependencies alone.
|
|
76
|
+
|
|
77
|
+
### Cycle 3 — Documentation inventory and freshness
|
|
78
|
+
|
|
79
|
+
**Purpose:** distinguish document presence from document usefulness and freshness.
|
|
80
|
+
|
|
81
|
+
**Checks:** inventory agent and human docs, resolve links, compare documentation ownership and referenced paths with the current repository, and run generated-doc freshness checks.
|
|
82
|
+
|
|
83
|
+
**Success metrics:**
|
|
84
|
+
|
|
85
|
+
- 100% of in-scope docs have a classification and source evidence;
|
|
86
|
+
- 100% of in-scope packages/apps have an explicit documentation status: `fresh`, `stale`, `missing`, or `unverified`;
|
|
87
|
+
- every `stale` result has at least one content/path/code evidence item;
|
|
88
|
+
- 0 stale decisions based only on filename keywords, dates, or the presence of words such as “old” or “deprecated”;
|
|
89
|
+
- generated and internal documentation checks pass, or each failure is surfaced as a finding.
|
|
90
|
+
|
|
91
|
+
### Cycle 4 — Documentation/code reconciliation
|
|
92
|
+
|
|
93
|
+
**Purpose:** prove that documented architecture and observed architecture agree at the declared semantic level.
|
|
94
|
+
|
|
95
|
+
**Checks:** use representative package/module declarations and known-positive, known-negative, stale, conflicting, unresolved, and dynamic relation cases; run the same policy against `agentskit-os`.
|
|
96
|
+
|
|
97
|
+
**Success metrics:**
|
|
98
|
+
|
|
99
|
+
- 100% of declared static relations are classified as confirmed, stale, conflicting, unresolved, or not-analyzed;
|
|
100
|
+
- undocumented observed relations are grouped at a useful package/module scope instead of producing unbounded raw noise;
|
|
101
|
+
- intentional exclusions are explicit and counted, never hidden by an empty required-kind list;
|
|
102
|
+
- the known-case fixture matrix has 100% detection of expected findings and 0 unsupported findings;
|
|
103
|
+
- the real target report does not show zero findings merely because reconciliation was disabled.
|
|
104
|
+
|
|
105
|
+
### Cycle 5 — Registry agent quality
|
|
106
|
+
|
|
107
|
+
**Purpose:** validate the configured agent from the AgentsKit Registry as an evidence-grounded assistant to discovery and classification.
|
|
108
|
+
|
|
109
|
+
**Checks:** run deterministic replay, inspect proposal origin/version/hashes/evidence, and run an assisted mode only when authorized credentials and budget are available.
|
|
110
|
+
|
|
111
|
+
**Success metrics:**
|
|
112
|
+
|
|
113
|
+
- proposal origin is the configured Registry agent ID and provider;
|
|
114
|
+
- base snapshot/report hashes match the run that produced the proposal;
|
|
115
|
+
- every intended change maps to an observed diagnostic or evidence item;
|
|
116
|
+
- deterministic replay is hash-stable;
|
|
117
|
+
- no proposal invents a path, claim, or remediation unsupported by the snapshot/report;
|
|
118
|
+
- human approval is required before any proposal application.
|
|
119
|
+
|
|
120
|
+
### Cycle 6 — Report and interaction validation
|
|
121
|
+
|
|
122
|
+
**Purpose:** prove that the evidence can be understood and explored by a human.
|
|
123
|
+
|
|
124
|
+
**Checks:** real-browser interaction, responsive layouts, keyboard access, contrast, text overflow, loading/latency, console/network errors, level navigation, selection, filters, zoom/pan, and evidence drilldown.
|
|
125
|
+
|
|
126
|
+
**Success metrics:**
|
|
127
|
+
|
|
128
|
+
- 0 automated failures across all configured viewport/theme scenarios;
|
|
129
|
+
- 0 console errors, failed requests, horizontal page overflow, clipped controls, or contrast violations;
|
|
130
|
+
- package → module → file navigation preserves context and breadcrumbs;
|
|
131
|
+
- every visible finding has evidence and an actionable explanation;
|
|
132
|
+
- human visual approval is recorded for the current report hash.
|
|
133
|
+
|
|
134
|
+
### Cycle 7 — Enterprise completion gate
|
|
135
|
+
|
|
136
|
+
**Purpose:** close the loop without overstating readiness.
|
|
137
|
+
|
|
138
|
+
**Success metrics:**
|
|
139
|
+
|
|
140
|
+
- all required cycles are `COMPLETE` for the same source revision and contract;
|
|
141
|
+
- no required surface remains `not-analyzed` without an approved exemption;
|
|
142
|
+
- source, target, config, docs, and generated artifacts have been reconciled;
|
|
143
|
+
- issue/PR tracking is updated only after authorization and includes the exact run ID;
|
|
144
|
+
- task-owned temporary artifacts are cleaned while user-owned or ambiguous artifacts remain untouched;
|
|
145
|
+
- structural decisions are documented in the relevant ADR/RFC;
|
|
146
|
+
- final report states residual risks and the next human action; enterprise readiness is not claimed while any required gate is pending.
|
|
147
|
+
|
|
148
|
+
## Initial baseline to collect
|
|
149
|
+
|
|
150
|
+
The first cycle records counts and timings without treating them as success: entities by kind, relations by kind, declared relations, findings by code/severity/status, coverage states, document classifications, proposal evidence ratio, report load time, interaction latency, generated artifact size, and agent-output token/byte counts. Later cycles ratchet these metrics against the baseline instead of hiding regressions behind aggregate pass/fail status.
|
|
151
|
+
|
|
152
|
+
The executable benchmark stores an anonymization-safe baseline at `.doc-bridge/benchmarks/baseline.json`. The baseline contains numeric metrics and rule definitions only; it must not contain repository paths, document contents, package names, credentials, or agent prompts. It is created or replaced only by an explicit human-authorized command (`--write-baseline` or `--replace-baseline`).
|
|
153
|
+
|
|
154
|
+
### Efficiency metric contract
|
|
155
|
+
|
|
156
|
+
| Metric family | Measurements | Initial gate |
|
|
157
|
+
|---|---|---|
|
|
158
|
+
| Pipeline | total, discovery, comparison, and proposal milliseconds | no more than 10% regression |
|
|
159
|
+
| Artifact | total bytes, initial HTML bytes, overview/levels/findings chunk bytes | initial HTML and total bytes no more than 10% regression |
|
|
160
|
+
| Agent context | hit rate, p50/p95 latency, p50/p95 response bytes, estimated p95 tokens, corpus-to-context reduction | 100% fixture hit rate; p95 bytes/tokens no more than 10% regression |
|
|
161
|
+
| Report UX | p95 first render, application response after real browser events, gesture duration, render-phase p95, and lazy-chunk bytes | first render ≤ 2s; application response ≤ 200ms; gesture is reported separately; phase/chunk data explains regressions |
|
|
162
|
+
| Analysis quality | analyzed ratio, documented/evidence ratios, overview node/edge counts, raw/compared relation ratio, finding density | tracked as evidence; fixture precision/recall gates semantic correctness |
|
|
163
|
+
| Reliability | UI failures, console errors, failed requests, incomplete stages | zero |
|
|
164
|
+
|
|
165
|
+
The report exposes aggregate insights suitable for case studies, but publication requires a separate redaction review. A benchmark passing means the measured contract passed; it does not prove semantic quality without the fixture set or prove enterprise readiness when surfaces are exempted.
|
|
166
|
+
|
|
167
|
+
## Latest measured cycle
|
|
168
|
+
|
|
169
|
+
Cycle 2 shipped Doc Bridge `1.7.20` to `agentskit-os` and added deterministic resolution of literal dynamic imports. The official workflow run is `1787934948972-14360`; the verification run is `1787935063723-14641`.
|
|
170
|
+
|
|
171
|
+
- discovered entities: `12,930 → 12,937`;
|
|
172
|
+
- raw relations: `40,921 → 41,300`;
|
|
173
|
+
- package-level compared relations: `2,663 → 2,698`;
|
|
174
|
+
- findings: `2,580 → 2,615`, all with evidence;
|
|
175
|
+
- benchmark: passed against baseline with pipeline `6,052ms`, initial HTML `68,069B`, report artifact `26,338,177B`, first render p95 `113ms`, application response p95 `92ms`, gesture p95 `76ms`, and zero UI failures;
|
|
176
|
+
- agent search remained at `100%` hit rate and `1,067` estimated p95 tokens with `99%` context reduction;
|
|
177
|
+
- all 10 automated viewport/theme scenarios passed; human approval is still required for the current report hash.
|
|
178
|
+
|
|
179
|
+
The result is an improvement in semantic discovery, not proof that runtime wiring or non-literal loading is resolved. The next cycle should measure and improve those boundaries or add an explicit configuration/adapter path for projects that can provide runtime architecture evidence.
|
|
180
|
+
|
|
181
|
+
Cycle 3 shipped Doc Bridge `1.7.21` to `agentskit-os` and added explicit `dynamic-literal` relation metadata plus benchmark counters for unresolved dynamic loading and runtime-wiring candidates. The official workflow run is `1787937333636-16849`; the verification run is `1787937404198-17113`.
|
|
182
|
+
|
|
183
|
+
- `379` literal dynamic-import relations are now identifiable in the snapshot;
|
|
184
|
+
- `56` files contain non-literal loading that remains unresolved;
|
|
185
|
+
- `46` files contain runtime-wiring candidates that remain explicitly not analyzed;
|
|
186
|
+
- benchmark passed with first render p95 `59ms`, application response p95 `79ms`, gesture p95 `76ms`, and zero UI failures;
|
|
187
|
+
- entity/relation/documentation and agent-search metrics remained grounded, with agent search at `100%` hit rate and `1,067` estimated p95 tokens;
|
|
188
|
+
- automated UI evidence passed across all 10 viewport/theme scenarios; human approval is required for the current report hash.
|
|
189
|
+
|
|
190
|
+
Cycle 4 shipped Doc Bridge `1.7.24` to `agentskit-os` and added configurable JS/TS runtime-wiring detection for statically imported targets, explicit unresolved-wiring coverage, a runtime-wiring benchmark counter, and browser-runtime warmup for stable visual timing. The official workflow run is `1787938291045-21432`; the verification run is `1787938325493-21506`.
|
|
191
|
+
|
|
192
|
+
- the real repository produced `12,937` entities and `41,300` raw relations;
|
|
193
|
+
- package-level reconciliation compared `2,774` relations and produced `2,691` findings, all with evidence;
|
|
194
|
+
- `379` literal dynamic-import relations were identified and `56` files still contain unresolved non-literal loading;
|
|
195
|
+
- `116` files contain unresolved runtime-wiring candidates; no real target wiring was resolved automatically, confirming the conservative boundary rather than overstating coverage;
|
|
196
|
+
- the benchmark passed without replacing the anonymization-safe baseline: pipeline `6,094ms`, artifact `26,441,496B`, initial HTML `68,069B`, first render p95 `110ms`, application response p95 `71ms`, gesture p95 `76ms`, and zero UI failures;
|
|
197
|
+
- agent search remained at `100%` hit rate and `1,067` estimated p95 tokens with `99%` context reduction;
|
|
198
|
+
- all 10 automated viewport/theme scenarios passed with no console errors, failed requests, overflow, clipping, or contrast violations; the screenshots were reviewed locally and the harness is awaiting explicit human approval.
|
|
199
|
+
|
|
200
|
+
Cycle 5 shipped Doc Bridge `1.7.25` to `agentskit-os` and tightened the default runtime-wiring heuristic by making generic `bind` and `listen` calls opt-in while preserving explicit configuration. The official workflow run is `1787938899030-22694`; the verification run is `1787938967570-22837`.
|
|
201
|
+
|
|
202
|
+
- the real repository remained stable at `12,937` entities, `41,300` raw relations, `2,774` compared relations, and `2,691` evidence-backed findings;
|
|
203
|
+
- unresolved runtime-wiring candidates fell from `116` to `54` (`53.4%` reduction) by removing generic API false positives;
|
|
204
|
+
- the benchmark passed against the unchanged baseline: pipeline `6,042ms`, first render p95 `58ms`, application response p95 `71ms`, gesture p95 `75ms`, report artifact `26,422,816B`, and zero UI failures;
|
|
205
|
+
- agent search remained at `100%` hit rate and `1,067` estimated p95 tokens with `99%` context reduction;
|
|
206
|
+
- the explicit configuration path was covered by a fixture: `listen` is ignored by default and produces a relation when configured;
|
|
207
|
+
- all 10 automated viewport/theme scenarios passed; the verification run is awaiting human visual approval.
|
|
208
|
+
|
|
209
|
+
Cycle 6 shipped Doc Bridge `1.7.26` to `agentskit-os` and tightened unresolved-wiring coverage to require a potential target argument, excluding inline registrations and no-argument calls from architectural gaps while preserving identifiers, property accesses, and factory calls. The official workflow run is `1787939273593-23788`; the verification run is `1787939317943-23884`.
|
|
210
|
+
|
|
211
|
+
- the real repository remained stable at `12,937` entities, `41,300` raw relations, `2,774` compared relations, and `2,691` evidence-backed findings;
|
|
212
|
+
- unresolved runtime-wiring candidates fell from `54` to `22` (`59.3%` reduction; `81.0%` reduction from the `116`-file starting point);
|
|
213
|
+
- the benchmark passed against the unchanged baseline: pipeline `5,903ms`, first render p95 `106ms`, application response p95 `71ms`, gesture p95 `77ms`, report artifact `26,412,971B`, and zero UI failures;
|
|
214
|
+
- agent search remained at `100%` hit rate and `1,067` estimated p95 tokens with `99%` context reduction;
|
|
215
|
+
- the known-case fixture retained both unresolved identifier detection and the explicit custom-method configuration path;
|
|
216
|
+
- all 10 automated viewport/theme scenarios passed; the verification run is awaiting human visual approval.
|
|
217
|
+
|
|
218
|
+
Cycle 7 shipped Doc Bridge `1.7.28` to `agentskit-os` and made the large-report overview payload package-scoped. The first load now carries package topology plus compact diagnostic indexes; module/file detail remains lazy. The official workflow run is `1787940069560-26699`; the verification run is `1787940112369-26783`.
|
|
219
|
+
|
|
220
|
+
- the real repository remained stable at `12,937` entities, `41,300` raw relations, `2,774` compared relations, and `2,691` evidence-backed findings;
|
|
221
|
+
- unresolved runtime-wiring candidates fell from `22` to `9` (`59.1%` additional reduction; `92.2%` reduction from the `116`-file starting point); test/spec runtime wiring is excluded by default and covered by an explicit opt-in fixture;
|
|
222
|
+
- the overview payload contains `83` package entities and `627` package relations instead of the full canonical entity/relation set, while retaining `2,691` diagnostic group indexes and `2,691` relation-finding indexes;
|
|
223
|
+
- the benchmark passed against the unchanged baseline: pipeline `5,910ms`, first render p95 `113ms`, application response p95 `71ms`, gesture p95 `78ms`, report artifact `26,370,691B`, initial HTML `68,170B`, overview chunk `843,002B`, and zero UI failures;
|
|
224
|
+
- agent search remained at `100%` hit rate and `1,067` estimated p95 tokens with `99%` context reduction; reconciliation evidence coverage remained `100%`;
|
|
225
|
+
- all 10 automated viewport/theme scenarios passed with no console errors, failed requests, overflow, clipping, or contrast violations; screenshots were reviewed locally and the run is awaiting explicit human visual approval.
|
|
226
|
+
|
|
227
|
+
## Current known blockers after the measured dogfood cycle
|
|
228
|
+
|
|
229
|
+
- package `1.7.28` is installed and verified in `agentskit-os`; the workflow run is `1787940069560-26699` and the verification run is `1787940112369-26783`;
|
|
230
|
+
- the benchmark passed against the original anonymization-safe baseline; the detailed cycle measurements are recorded above;
|
|
231
|
+
- the agent search fixture returned a grounded match for `100%` of queries, with p95 `1,067` estimated tokens and `99%` context reduction; reconciliation evidence coverage was `100%` for the `2,691` findings;
|
|
232
|
+
- the visual check passed all automated checks across 10 viewport/theme scenarios, but the current verification run remains `AWAITING_HUMAN_APPROVAL` until a human reviews the screenshots and approves run `1787940112369-26783`;
|
|
233
|
+
- only `3%` of reported analyzer scopes are complete in this target, so non-literal dynamic loading, unresolved runtime wiring, and generated code remain explicit coverage limitations;
|
|
234
|
+
- the current verification contract covers the package dogfood target, not the complete Doc Bridge enterprise objective.
|
|
235
|
+
|
|
236
|
+
The current run is `AWAITING_HUMAN_APPROVAL`, not complete. Discovery, package-level reconciliation, Registry-agent proof, documentation cohesion, export accuracy, CLI execution, and measured efficiency passed. The benchmark baseline was not replaced. After human approval, the next cycle should use these numbers as the comparison point and focus on classifying the remaining 9 production runtime-wiring candidates and improving documentation usefulness rather than report transport performance.
|
|
@@ -5,25 +5,44 @@ description: Fail-closed, evidence-backed verification for humans and agents.
|
|
|
5
5
|
|
|
6
6
|
# Verification harness
|
|
7
7
|
|
|
8
|
-
`ak-verify` is the executable completion gate for work that must be proven, not merely compiled.
|
|
8
|
+
`ak-verify` is the executable completion gate for work that must be proven, not merely compiled. Harness 1.3 also blocks failed contract outcomes and measured regressions.
|
|
9
9
|
|
|
10
10
|
```bash
|
|
11
11
|
ak-verify run --config .codex/verification.json --json
|
|
12
12
|
ak-verify status --config .codex/verification.json --json
|
|
13
13
|
ak-verify approve <run-id> approved --by human --config .codex/verification.json
|
|
14
14
|
ak-verify authorize <run-id> approved --by human --config .codex/verification.json
|
|
15
|
+
ak-verify baseline replace benchmarks/new.json approved --by human --config .codex/verification.json
|
|
15
16
|
ak-verify clean --periodic --config .codex/verification.json
|
|
16
17
|
```
|
|
17
18
|
|
|
18
19
|
The contract is JSON so it works without adding a YAML runtime. It declares the artifact surfaces that apply to the run, executable checks, explicit non-applicable reasons, the verification profile, and tracking policy.
|
|
19
20
|
|
|
21
|
+
## Global policy and project contract
|
|
22
|
+
|
|
23
|
+
The `ak-verify` executable is portable: any repository can run it from the
|
|
24
|
+
published Doc Bridge package. The verification contract remains project-local
|
|
25
|
+
at `.codex/verification.json` because endpoints, databases, UI flows, checks,
|
|
26
|
+
acceptance criteria, and tracking targets differ by repository.
|
|
27
|
+
|
|
28
|
+
The host-level agent policy is global and requires that every repository have
|
|
29
|
+
this project contract before implementation or verification. A repository
|
|
30
|
+
without it is `CLARIFYING`; agents must ask the human to define or authorize
|
|
31
|
+
the contract instead of selecting a default profile or inventing checks. This
|
|
32
|
+
combination provides one global completion rule without pretending that one
|
|
33
|
+
set of repository-specific checks fits every project.
|
|
34
|
+
|
|
35
|
+
Before implementation, the human intent and acceptance criteria must be explicit. Every criterion must map to an executable check and its expected evidence. If a criterion is not mapped, the run is `CLARIFYING` or `BLOCKED`; a project may not silently shrink the scope to the checks that are easiest to run.
|
|
36
|
+
|
|
20
37
|
## States
|
|
21
38
|
|
|
22
|
-
`PLANNED` → `VERIFYING` → `AWAITING_HUMAN_APPROVAL`
|
|
39
|
+
`CLARIFYING` → `PLANNED` → `VERIFYING` → `AWAITING_HUMAN_APPROVAL` / `AWAITING_AUTHORIZATION` → `COMPLETE`.
|
|
40
|
+
|
|
41
|
+
Any failed required check, unavailable required surface, or unmapped acceptance criterion produces `BLOCKED` or `CLARIFYING`. The harness never promotes a run from `BLOCKED`, `AWAITING_HUMAN_APPROVAL`, or `AWAITING_AUTHORIZATION` to `COMPLETE` without the corresponding evidence and intent. A `COMPLETE` result applies only to the declared contract and must not be described as broader product validation when the broader scope was not declared.
|
|
23
42
|
|
|
24
|
-
|
|
43
|
+
`default` is the low-friction profile used when `profile` is omitted. `strict` keeps fail-closed validation for the declared contract. `poc` and `custom` require explicit exemptions, which are included in the run evidence and cannot be silently hidden. `enterprise` requires all seven surfaces to be declared, requires measurement and tracking, and does not accept silent exemptions.
|
|
25
44
|
|
|
26
|
-
|
|
45
|
+
The seven applicability surfaces are `logic`, `cli`, `mcp`, `ui`, `docs`, `endpoint`, and `database`. A surface marked not applicable must include a reason. Endpoint and database validation remains conditional on the target: if the target uses one, declare a required real check; otherwise declare why it is not applicable.
|
|
27
46
|
|
|
28
47
|
## Evidence and recovery
|
|
29
48
|
|
|
@@ -33,4 +52,14 @@ Checks may emit one final JSON line with `status` set to `passed`, `failed`, or
|
|
|
33
52
|
|
|
34
53
|
Visual checks must use a real browser or an explicitly configured equivalent. A passing build is not visual approval. Endpoint, database, CLI, and MCP checks must execute their real artifact when the contract marks that surface as required.
|
|
35
54
|
|
|
55
|
+
The final evidence ledger must distinguish `validated`, `partially validated`, `not analyzed`, `blocked`, and `not applicable`. Counts such as indexed documents, package presence, or rendered reports do not prove semantic documentation/code agreement, stale-content detection, runtime wiring, or UI behavior.
|
|
56
|
+
|
|
57
|
+
When a report is shared outside its repository, configure `report.privacy: 'anonymized'`. This is separate from `safety.redactSecrets`: secret redaction does not anonymize project names, paths, identifiers, snippets, or finding messages. The privacy check must inspect the generated HTML and all lazy chunks, not only the configuration.
|
|
58
|
+
|
|
59
|
+
When `measurement.required` is enabled, the named required check must emit structured evidence with `status: "passed"`, a numeric `metrics` object, a `baselineHash`, and an empty `regressions` array. Missing or regressed measurements block completion. Baselines are explicit, versioned artifacts and are never updated implicitly by a verification run.
|
|
60
|
+
|
|
61
|
+
Baseline replacement is an explicit, human-intended operation. `baseline replace` copies a JSON baseline into the configured target only when the source and target are inside the project root and appends a hash, actor, intent, and timestamp to `.codex/verification/baseline-audit.jsonl`.
|
|
62
|
+
|
|
63
|
+
The run JSON exposes `profile`, `profilePolicy`, `applicability`, `exemptions`, `checks`, `evidenceReferences`, `metrics`, `transitions`, `sourceRevision`, `contractHash`, `outputHash`, and the exact `runId`. Approval records are bound to the input, source, contract, and output hashes of that run.
|
|
64
|
+
|
|
36
65
|
The harness only removes paths listed as task-owned and contained by configured cleanup roots. It never performs broad workspace deletion.
|
package/mcpb/manifest.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"manifest_version": "0.3",
|
|
3
3
|
"name": "doc-bridge",
|
|
4
4
|
"display_name": "Doc Bridge",
|
|
5
|
-
"version": "1.
|
|
5
|
+
"version": "1.7.44",
|
|
6
6
|
"description": "Deterministic repository handoffs for coding agents, running locally without an LLM or API key.",
|
|
7
7
|
"long_description": "Doc Bridge turns a repository's own documentation and ownership metadata into deterministic handoffs: where an agent should start, which paths it may edit, which checks it must run, and when a human must take over. The local connector exposes the same read-only contract available through Doc Bridge CLI and CI.",
|
|
8
8
|
"author": {
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@agentskit/doc-bridge",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.7.44",
|
|
4
4
|
"mcpName": "io.github.AgentsKit-io/doc-bridge",
|
|
5
5
|
"description": "Human↔agent documentation bridge — deterministic handoffs, doc-site links, memory→docs, optional AgentsKit RAG/chat.",
|
|
6
6
|
"type": "module",
|
|
@@ -1,5 +1,6 @@
|
|
|
1
|
-
import { mkdirSync, writeFileSync } from 'node:fs'
|
|
2
|
-
import {
|
|
1
|
+
import { existsSync, mkdirSync, readFileSync, writeFileSync } from 'node:fs'
|
|
2
|
+
import { createHash } from 'node:crypto'
|
|
3
|
+
import { join, relative, resolve } from 'node:path'
|
|
3
4
|
import { pathToFileURL } from 'node:url'
|
|
4
5
|
|
|
5
6
|
import { chromium } from '@playwright/test'
|
|
@@ -15,6 +16,7 @@ const themes = ['light', 'dark']
|
|
|
15
16
|
const failures = []
|
|
16
17
|
const checks = []
|
|
17
18
|
const networkRequests = []
|
|
19
|
+
const clock = () => performance.now()
|
|
18
20
|
|
|
19
21
|
mkdirSync(outputDir, { recursive: true })
|
|
20
22
|
|
|
@@ -99,25 +101,46 @@ const inspect = async (frame, width, height) => frame.evaluate(({ width: viewpor
|
|
|
99
101
|
}, { width, height })
|
|
100
102
|
|
|
101
103
|
const poll = async (read, predicate, label, timeoutMs = interactionTimeoutMs) => {
|
|
102
|
-
const started =
|
|
103
|
-
while (
|
|
104
|
-
if (predicate(await read())) return
|
|
104
|
+
const started = clock()
|
|
105
|
+
while (clock() - started < timeoutMs) {
|
|
106
|
+
if (predicate(await read())) return Math.round(clock() - started)
|
|
105
107
|
await new Promise((resolve) => setTimeout(resolve, 50))
|
|
106
108
|
}
|
|
107
109
|
throw new Error(`${label} did not reach the expected state within ${timeoutMs}ms`)
|
|
108
110
|
}
|
|
109
111
|
|
|
110
112
|
const exercise = async (frame, result) => {
|
|
111
|
-
|
|
112
|
-
|
|
113
|
+
result.interactionDurationsMs ??= []
|
|
114
|
+
const run = async (label, action, verify, browserEvent) => {
|
|
115
|
+
const started = clock()
|
|
113
116
|
try {
|
|
117
|
+
if (browserEvent) await frame.evaluate((type) => {
|
|
118
|
+
document.documentElement.dataset.docBridgeInteractionStartedAt = ''
|
|
119
|
+
window.addEventListener(type, () => {
|
|
120
|
+
document.documentElement.dataset.docBridgeInteractionStartedAt = String(performance.now())
|
|
121
|
+
}, { capture: true, once: true })
|
|
122
|
+
}, browserEvent)
|
|
114
123
|
await Promise.race([
|
|
115
124
|
(async () => { await action(); if (verify) await verify() })(),
|
|
116
125
|
new Promise((_, reject) => setTimeout(() => reject(new Error('interaction timeout')), interactionTimeoutMs)),
|
|
117
126
|
])
|
|
118
|
-
const
|
|
127
|
+
const gestureDurationMs = Math.round(clock() - started)
|
|
128
|
+
const durationMs = browserEvent
|
|
129
|
+
? await frame.evaluate(() => {
|
|
130
|
+
const startedAt = Number(document.documentElement.dataset.docBridgeInteractionStartedAt)
|
|
131
|
+
return Number.isFinite(startedAt) ? Math.round(performance.now() - startedAt) : null
|
|
132
|
+
})
|
|
133
|
+
: gestureDurationMs
|
|
134
|
+
if (durationMs == null) throw new Error(`${browserEvent} event was not observed`)
|
|
135
|
+
result.interactionDurationsMs.push({ label, durationMs, ...(browserEvent ? { gestureDurationMs } : {}) })
|
|
136
|
+
if (browserEvent) {
|
|
137
|
+
const renderTimings = await frame.evaluate(() => JSON.parse(document.documentElement.dataset.docBridgeRenderTimings || '{}'))
|
|
138
|
+
result.renderTimings ??= []
|
|
139
|
+
result.renderTimings.push({ label, ...renderTimings })
|
|
140
|
+
}
|
|
119
141
|
if (durationMs > 2000) result.failures.push(`${label} took ${durationMs}ms`)
|
|
120
142
|
} catch (error) {
|
|
143
|
+
result.interactionDurationsMs.push({ label, durationMs: Math.round(clock() - started), failed: true })
|
|
121
144
|
result.failures.push(`${label}: ${error instanceof Error ? error.message : String(error)}`)
|
|
122
145
|
}
|
|
123
146
|
}
|
|
@@ -165,7 +188,7 @@ const exercise = async (frame, result) => {
|
|
|
165
188
|
await poll(() => frame.locator('[data-level][aria-pressed="true"]').getAttribute('data-level'), (actual) => actual === expectedLevel, `drill-down ${expectedLevel}`)
|
|
166
189
|
const details = await frame.locator('#details').innerText()
|
|
167
190
|
if (/Select a node in the map/i.test(details)) throw new Error('details panel did not update')
|
|
168
|
-
})
|
|
191
|
+
}, 'dblclick')
|
|
169
192
|
}
|
|
170
193
|
if (await frame.locator('#breadcrumbs [data-breadcrumb-level]').count() < 1) result.failures.push('drill-down did not create usable breadcrumbs')
|
|
171
194
|
await run('breadcrumb back to repository', () => frame.locator('#breadcrumbs [data-breadcrumb-level="overview"]').click(), async () => {
|
|
@@ -189,6 +212,9 @@ const exercise = async (frame, result) => {
|
|
|
189
212
|
let browser
|
|
190
213
|
try {
|
|
191
214
|
browser = await chromium.launch()
|
|
215
|
+
const warmup = await browser.newPage()
|
|
216
|
+
await warmup.goto('about:blank')
|
|
217
|
+
await warmup.close()
|
|
192
218
|
for (const theme of themes) {
|
|
193
219
|
for (const viewport of viewports) {
|
|
194
220
|
const page = await browser.newPage()
|
|
@@ -203,10 +229,12 @@ try {
|
|
|
203
229
|
try {
|
|
204
230
|
await page.setViewportSize({ width: viewport[0], height: viewport[1] })
|
|
205
231
|
await page.emulateMedia({ colorScheme: theme })
|
|
232
|
+
const renderStarted = clock()
|
|
206
233
|
await page.goto(pathToFileURL(reportPath).href, { waitUntil: 'load' })
|
|
207
234
|
const frame = page.frames().find((candidate) => candidate !== page.mainFrame() && candidate.url().endsWith('/report/index.html')) ?? page.mainFrame()
|
|
208
235
|
await frame.locator('#graph').waitFor({ state: 'visible', timeout: interactionTimeoutMs })
|
|
209
236
|
result = await inspect(frame, viewport[0], viewport[1])
|
|
237
|
+
result.firstRenderMs = Math.round(clock() - renderStarted)
|
|
210
238
|
await exercise(frame, result)
|
|
211
239
|
result.pageErrors = pageErrors
|
|
212
240
|
result.consoleErrors = consoleErrors
|
|
@@ -229,8 +257,15 @@ try {
|
|
|
229
257
|
}
|
|
230
258
|
|
|
231
259
|
const status = failures.length || checks.some((check) => check.failures.length) ? 'failed' : humanApproved ? 'passed' : 'pending-human-review'
|
|
260
|
+
const artifacts = checks.map((check) => {
|
|
261
|
+
const path = join(outputDir, `${check.viewport}-${check.theme}.png`)
|
|
262
|
+
return existsSync(path) ? { type: 'screenshot', path: relative(process.cwd(), path), sha256: createHash('sha256').update(readFileSync(path)).digest('hex'), viewport: check.viewport, theme: check.theme } : null
|
|
263
|
+
}).filter(Boolean)
|
|
232
264
|
const result = {
|
|
233
265
|
status,
|
|
266
|
+
capability: 'real-browser',
|
|
267
|
+
artifacts,
|
|
268
|
+
criteria: { 'report-ui': { status: status === 'failed' ? 'failed' : 'passed' } },
|
|
234
269
|
reportPath,
|
|
235
270
|
outputDir,
|
|
236
271
|
viewports,
|
|
@@ -246,5 +281,5 @@ const result = {
|
|
|
246
281
|
}
|
|
247
282
|
writeFileSync(join(outputDir, 'result.json'), `${JSON.stringify(result, null, 2)}\n`, 'utf8')
|
|
248
283
|
console.log(JSON.stringify(result, null, 2))
|
|
249
|
-
console.log(JSON.stringify(
|
|
284
|
+
console.log(JSON.stringify(result))
|
|
250
285
|
if (status === 'failed') process.exitCode = 1
|