@agentskit/doc-bridge 1.6.4 → 1.7.45

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/CHANGELOG.md +249 -0
  2. package/CONTRIBUTING.md +6 -4
  3. package/action.yml +1 -1
  4. package/dist/cli/program.js +1139 -294
  5. package/dist/cli/program.js.map +1 -1
  6. package/dist/config/index.d.ts +1 -1
  7. package/dist/config/index.js +43 -5
  8. package/dist/config/index.js.map +1 -1
  9. package/dist/index-BUL0q7s8.d.ts +660 -0
  10. package/dist/index.d.ts +817 -2134
  11. package/dist/index.js +1154 -244
  12. package/dist/index.js.map +1 -1
  13. package/docs/PRD-enterprise-hardening.md +288 -0
  14. package/docs/RELEASE.md +22 -8
  15. package/docs/adr/0001-enterprise-verification-contract.md +35 -0
  16. package/docs/agent-corpus/INDEX.md +2 -2
  17. package/docs/agent-corpus/chat.md +2 -2
  18. package/docs/agent-corpus/cli.md +2 -2
  19. package/docs/agent-corpus/conformance.md +2 -2
  20. package/docs/agent-corpus/doc-bridge.md +1 -1
  21. package/docs/agent-corpus/doctor.md +2 -2
  22. package/docs/agent-corpus/gates.md +2 -2
  23. package/docs/agent-corpus/mcp.md +2 -2
  24. package/docs/agent-corpus/memory.md +2 -2
  25. package/docs/agent-corpus/query.md +2 -2
  26. package/docs/knowledge-engine-runbook.md +30 -2
  27. package/docs/spec/analyzer-plugin-v1.md +24 -0
  28. package/docs/spec/benchmark-v1.md +36 -0
  29. package/docs/spec/config-v1.md +156 -0
  30. package/docs/validation-cycle-plan.md +255 -0
  31. package/docs/verification-harness.md +37 -4
  32. package/mcpb/manifest.json +1 -1
  33. package/package.json +68 -70
  34. package/scripts/check-ecosystem-upstream.mjs +3 -2
  35. package/scripts/report-visual-check.mjs +64 -12
  36. package/scripts/verification-harness.mjs +216 -14
  37. package/skills/doc-bridge-handoff/scripts/resolve-handoff.mjs +1 -1
  38. package/src/agents/registry-adapter.ts +31 -7
  39. package/src/cli/demo.ts +2 -2
  40. package/src/cli/program.ts +59 -16
  41. package/src/config/index.ts +2 -0
  42. package/src/config/load-config.ts +7 -1
  43. package/src/config/schema.ts +60 -2
  44. package/src/conformance/documentation-standard-v1.ts +14 -8
  45. package/src/discovery/documentation.ts +90 -23
  46. package/src/discovery/repository.ts +147 -19
  47. package/src/doctor/run-doctor.ts +2 -15
  48. package/src/federation/llms.ts +72 -20
  49. package/src/fixes/proposals.ts +4 -3
  50. package/src/index-builder/human-adapters/fumadocs.ts +1 -1
  51. package/src/index-builder/watch-index.ts +1 -1
  52. package/src/index.ts +29 -0
  53. package/src/lib/bounded-text.ts +15 -10
  54. package/src/metrics/benchmark.ts +176 -0
  55. package/src/plugins/contract.ts +89 -0
  56. package/src/reconciliation/reconcile.ts +181 -5
  57. package/src/report/html.ts +318 -88
  58. package/src/rules/engine.ts +15 -2
  59. package/src/safety/repository.ts +1 -1
  60. package/src/schemas/knowledge.ts +21 -3
  61. package/src/validate.ts +7 -1
  62. package/src/version.ts +1 -1
  63. package/src/workflow/engine.ts +65 -9
  64. package/dist/index-DudNuwI5.d.ts +0 -2060
@@ -7,6 +7,10 @@ description: Configure documentation corpora, ownership routing, conformance, an
7
7
 
8
8
  `doc-bridge.config.ts` (or `.js`, `.mjs`, `.json`, or `package.json` → `docBridge`) is the alpha integration point for any project. Layer 0 fields are sufficient to run `index`, `query`, and MCP without an LLM.
9
9
 
10
+ Reconciliation summaries also expose deterministic `diagnosticsByCode` and
11
+ `diagnosticsByStatus` maps. They are additive rollups for agents and dashboards;
12
+ the canonical `diagnostics` array remains the source of evidence.
13
+
10
14
  | npm package | `@agentskit/doc-bridge` |
11
15
  | CLI binary | `ak-docs` |
12
16
  | Config file | `doc-bridge.config.ts` |
@@ -34,6 +38,12 @@ package.json → "docBridge" field (subset, JSON only)
34
38
 
35
39
  TypeScript/JavaScript configs are static in v0.1 alpha: `defineConfig` imports are supported, but arbitrary imports are not. YAML config files are planned.
36
40
 
41
+ Dynamic-loading coverage is evidence-backed. Literal strings, constant aliases,
42
+ parenthesized strings, and string concatenations are resolved without executing
43
+ repository code. Runtime-dependent `import()` and `require()` expressions remain
44
+ `partial`/`not-analyzed`, and the coverage entry records representative
45
+ file-and-line evidence for each observed loading site (up to the schema limit).
46
+
37
47
  ## TypeScript shape (authoritative)
38
48
 
39
49
  ```ts
@@ -71,6 +81,18 @@ export default {
71
81
 
72
82
  /** Optional deterministic documentation conformance profiles */
73
83
  conformance?: ConformanceConfig
84
+
85
+ /** Optional language analyzers and runtime-wiring coverage */
86
+ analysis?: AnalysisConfig
87
+
88
+ /** Optional reconciliation scope and orphan-document policy */
89
+ reconciliation?: ReconciliationConfig
90
+
91
+ /** Optional resumable workflow state */
92
+ workflow?: WorkflowConfig
93
+
94
+ /** Optional report publication privacy; private is the default */
95
+ report?: { privacy?: 'private' | 'anonymized' }
74
96
  } satisfies DocBridgeConfigV1
75
97
  ```
76
98
 
@@ -386,6 +408,122 @@ report status, commands, and the recorded stable-publication HITL decision.
386
408
 
387
409
  ---
388
410
 
411
+ ## `reconciliation` (optional)
412
+
413
+ ```ts
414
+ type ReconciliationConfig = {
415
+ /** Semantic comparison level; discovery still preserves raw file relations. */
416
+ scope?: 'file' | 'module' | 'package'
417
+ /** Observed relation kinds that require documentation declarations. */
418
+ requiredRelationKinds?: string[]
419
+ /** Limit missing-declaration findings to relations between internal project entities. */
420
+ requiredRelationTargets?: 'all' | 'internal'
421
+ /** Emit info findings for documentation with no observed package/module join. */
422
+ includeOrphanedDocuments?: boolean
423
+ }
424
+ ```
425
+
426
+ Use `scope: 'package'` for monorepos where file imports should be compared as package-level architecture evidence. Omit `requiredRelationKinds` to require all observed kinds; an empty array intentionally disables undocumented-relation findings and must be treated as an explicit exemption.
427
+
428
+ Use `requiredRelationTargets: 'internal'` when the repository wants package or module architecture declarations without requiring Markdown to enumerate every external library import. External relations remain in the raw snapshot and report as evidence; they simply do not generate missing-declaration findings.
429
+
430
+ The reconciliation documentation summary reports package health separately from
431
+ relation findings: `fresh` means the package has coverage documentation and no
432
+ known discrepancy; `stale` means a declared relation conflicts with observed
433
+ architecture; `missing` means no coverage document was found; and `unverified`
434
+ means coverage exists but at least one relevant relation or analyzer boundary
435
+ could not be verified. Package-level aggregation preserves the relation
436
+ endpoints used for this classification, so an undocumented relation cannot be
437
+ reported alongside a falsely `fresh` package.
438
+
439
+ The same summary keeps document inventory separate from coverage claims:
440
+ `documentCount` and `documentClassificationCounts` describe every discovered
441
+ Markdown document, while `documentedDocumentCount` and
442
+ `documentedDocumentClassificationCounts` describe documents that declare a
443
+ knowledge relation. The classification keys are analyzer output (for example
444
+ `agent`, `human`, `project`, `archive`, and `unclassified`); consumers must not
445
+ interpret total repository Markdown coverage as agent-corpus coverage. Use the
446
+ `agent` pair when measuring the configured agent documentation surface.
447
+
448
+ ## `report` (optional)
449
+
450
+ ```ts
451
+ report?: {
452
+ /** Replace project identity, names, paths, snippets, and finding text in HTML output. */
453
+ privacy?: 'private' | 'anonymized'
454
+ }
455
+ ```
456
+
457
+ `private` is the default and keeps local evidence useful for debugging. `anonymized` is intended for reports shared outside the repository: it preserves counts, relation kinds, topology, and coverage status while removing project-specific identity and evidence content. The generated HTML and every lazy chunk use the same mode.
458
+
459
+ ## `safety` (optional)
460
+
461
+ ```ts
462
+ type RepositorySafetyConfig = {
463
+ /** Additional project-relative glob patterns excluded from repository discovery. */
464
+ exclude?: string[]
465
+ maxFiles?: number
466
+ maxBytes?: number
467
+ maxTimeMs?: number
468
+ maxMemoryMb?: number
469
+ redactSecrets?: boolean
470
+ }
471
+ ```
472
+
473
+ Discovery always excludes unsafe or generated trees by default, including
474
+ `.git`, `node_modules`, `dist`, `build`, `coverage`, `.doc-bridge`, `.next`,
475
+ `out`, `.turbo`, `.svelte-kit`, `.mcpb-build`, and `.mcpb-output`, plus common
476
+ secret files. `safety.exclude` adds project-specific patterns; it does not
477
+ replace the built-in safety boundary. Excluded files remain outside the
478
+ snapshot and are represented by analyzer coverage when relevant.
479
+
480
+ ## `analysis` (optional)
481
+
482
+ ```ts
483
+ type AnalysisConfig = {
484
+ /** Ordered language/framework analyzer plugins. */
485
+ plugins?: Array<{
486
+ id: string
487
+ enabled?: boolean
488
+ order?: number
489
+ options?: Record<string, unknown>
490
+ reason?: string
491
+ }>
492
+ jsTs?: {
493
+ /** Property-access methods considered runtime wiring entry points. */
494
+ runtimeWiringMethods?: string[]
495
+ /** Additional adapter methods that represent runtime wiring. */
496
+ runtimeWiringAdapters?: Array<{ id: string; methods: string[] }>
497
+ /** Include test/spec modules in runtime-wiring coverage. Default: false. */
498
+ includeTestRuntimeWiring?: boolean
499
+ }
500
+ }
501
+ ```
502
+
503
+ The JS/TS analyzer defaults to `register`, `use`, `mount`, and `attach`.
504
+ Generic APIs such as `bind` and `listen` are intentionally opt-in to avoid
505
+ classifying ordinary function binding and server startup as architecture.
506
+ When a configured method receives an identifier bound to a
507
+ static import, Doc Bridge records a `runtime-wiring` relation with
508
+ `metadata.detection: 'runtime-wiring-static'`. Reflective, computed, or
509
+ otherwise unbound targets remain explicit `not-analyzed` coverage entries.
510
+ Dynamic `import()` and `require()` targets are resolved when their specifier is
511
+ a literal, a `const` string binding, a parenthesized static expression, or a
512
+ concatenation of other statically known strings. Expressions that depend on
513
+ runtime values remain explicit `not-analyzed` coverage entries; Doc Bridge does
514
+ not execute repository code to guess their targets.
515
+ Test/spec modules are excluded from this signal by default because their
516
+ registrations usually construct fixtures rather than production architecture;
517
+ set `includeTestRuntimeWiring: true` when test wiring is part of the contract.
518
+ This keeps runtime behavior useful without claiming that arbitrary dependency
519
+ injection or reflection was resolved. Analyzer plugins must implement the
520
+ versioned contract in `docs/spec/analyzer-plugin-v1.md`; malformed or
521
+ unsupported output remains explicit `not-analyzed` evidence.
522
+
523
+ `workflow.stateDir` optionally relocates the content-addressed workflow
524
+ artifacts. Runs are resumable and idempotent; pipeline and analyzer versions
525
+ are part of the run identity.
526
+
389
527
  ## `surfaces` (optional)
390
528
 
391
529
  ```ts
@@ -466,6 +604,18 @@ type IntelligenceConfig = {
466
604
  /** Reference runtime; custom path for non-AgentsKit engines later */
467
605
  runtime?: 'agentskit' | 'custom'
468
606
  runtimeModule?: string
607
+
608
+ registry?: {
609
+ enabled?: boolean
610
+ agentId?: string
611
+ agentRoot?: string
612
+ runnerModule?: string
613
+ deterministic?: boolean
614
+ timeoutMs?: number
615
+ maxTokens?: number
616
+ maxResponseBytes?: number
617
+ maxConcurrency?: number
618
+ }
469
619
  }
470
620
 
471
621
  type MemoryAdapterId =
@@ -475,6 +625,12 @@ type MemoryAdapterId =
475
625
  | 'bootstrap-delta' // git diff on AGENTS.md
476
626
  ```
477
627
 
628
+ Registry agents are advisory and must come from the AgentsKit Registry. The
629
+ adapter redacts secrets, validates the proposal contract, bounds execution and
630
+ response size, and requires human approval before any change is applied. The
631
+ default agent is `ecosystem-doc-bridge-corpus-scanner`; compatible Registry
632
+ agents may be selected explicitly.
633
+
478
634
  ---
479
635
 
480
636
  ## `federation` (optional — ecosystem)
@@ -0,0 +1,255 @@
1
+ ---
2
+ title: Doc Bridge validation cycle plan
3
+ description: Evidence-driven validation of Doc Bridge against the AgentsKit OS repository.
4
+ ---
5
+
6
+ # Doc Bridge validation cycle plan
7
+
8
+ ## Objective
9
+
10
+ Validate the complete Doc Bridge change against `agentskit-os`, not only the published package or the HTML report. The validation must prove the bridge between repository structure, documentation, agents, and humans:
11
+
12
+ 1. repository architecture is discovered at package, module, and file levels;
13
+ 2. documentation is classified, indexed, and checked for freshness;
14
+ 3. documentation claims are reconciled with observed code relationships;
15
+ 4. disconnected, stale, conflicting, unresolved, and not-analyzed areas are visible;
16
+ 5. the Registry agent proposes grounded follow-up work without becoming an unverified authority;
17
+ 6. the report makes those results useful and navigable;
18
+ 7. runs are reproducible, resumable, versioned, idempotent, and cost-bounded.
19
+
20
+ This plan is a validation contract. A green result from one cycle never substitutes for a missing cycle.
21
+
22
+ ## Evidence rules
23
+
24
+ - Every result is tied to the source revision, configuration hash, snapshot hash, report hash, and verification run ID.
25
+ - `not-analyzed` is visible and counts against coverage unless an explicit, human-approved exemption exists.
26
+ - Unit tests and compilation are supporting evidence. They do not prove repository behavior, documentation agreement, endpoint/database behavior, or UI behavior.
27
+ - Deterministic agent runs and assisted agent runs are separate evidence classes. Deterministic output proves replayability; it does not prove semantic quality.
28
+ - No agent proposal is applied without human approval and a fresh post-apply verification run.
29
+
30
+ ## Cycle gates and success metrics
31
+
32
+ ### Cycle 0 — Contract and scope gate
33
+
34
+ **Purpose:** map the human request to verifiable outcomes before changing the product or narrowing the target.
35
+
36
+ **Required evidence:** task contract, acceptance matrix, repository/surface matrix, explicit exemptions, validation budget, and the decision for architecture declaration granularity.
37
+
38
+ **Success metrics:**
39
+
40
+ - 100% of requested outcomes mapped to one or more executable checks;
41
+ - 0 unresolved scope, authorization, or behavior ambiguities;
42
+ - 0 unapproved surface exclusions;
43
+ - 100% of required checks have an evidence location and an expected result;
44
+ - the contract names what the report must show for architecture, stale docs, doc/code drift, disconnected areas, Registry proposals, and UI.
45
+
46
+ **Stop condition:** any unmapped outcome returns the task to `CLARIFYING`.
47
+
48
+ ### Cycle 1 — Real consumer and reproducibility
49
+
50
+ **Purpose:** prove the published package is the artifact used by `agentskit-os` and that the pipeline is repeatable.
51
+
52
+ **Checks:** install the declared package version, run discovery, index, reconcile, check, report, and Registry agent using the target repository.
53
+
54
+ **Success metrics:**
55
+
56
+ - installed version equals the declared and published version;
57
+ - two unchanged runs produce identical snapshot, report, and deterministic proposal hashes;
58
+ - every generated artifact is linked from the workflow run;
59
+ - resume after an interrupted stage completes without duplicating or corrupting artifacts;
60
+ - local execution is used unless a contract explicitly requires network or CI.
61
+
62
+ ### Cycle 2 — Architecture discovery
63
+
64
+ **Purpose:** validate the project topology rather than merely listing dependencies.
65
+
66
+ **Checks:** compare discovered packages/apps/modules/files and relations with the monorepo manifests, workspace configuration, representative source imports/exports, and known application boundaries.
67
+
68
+ **Success metrics:**
69
+
70
+ - 100% of in-scope workspace packages and apps are classified;
71
+ - package, module, and file graph levels are independently navigable;
72
+ - relation kinds and direction are preserved;
73
+ - disconnected nodes and high-connectivity/SPOF candidates are reported with code evidence;
74
+ - dynamic imports, runtime wiring, and generated code are either analyzed or explicitly reported as `not-analyzed`;
75
+ - no architecture claim is inferred from `package.json` dependencies alone.
76
+
77
+ ### Cycle 3 — Documentation inventory and freshness
78
+
79
+ **Purpose:** distinguish document presence from document usefulness and freshness.
80
+
81
+ **Checks:** inventory agent and human docs, resolve links, compare documentation ownership and referenced paths with the current repository, and run generated-doc freshness checks.
82
+
83
+ **Success metrics:**
84
+
85
+ - 100% of in-scope docs have a classification and source evidence;
86
+ - 100% of in-scope packages/apps have an explicit documentation status: `fresh`, `stale`, `missing`, or `unverified`;
87
+ - every `stale` result has at least one content/path/code evidence item;
88
+ - 0 stale decisions based only on filename keywords, dates, or the presence of words such as “old” or “deprecated”;
89
+ - generated and internal documentation checks pass, or each failure is surfaced as a finding.
90
+
91
+ ### Cycle 4 — Documentation/code reconciliation
92
+
93
+ **Purpose:** prove that documented architecture and observed architecture agree at the declared semantic level.
94
+
95
+ **Checks:** use representative package/module declarations and known-positive, known-negative, stale, conflicting, unresolved, and dynamic relation cases; run the same policy against `agentskit-os`.
96
+
97
+ **Success metrics:**
98
+
99
+ - 100% of declared static relations are classified as confirmed, stale, conflicting, unresolved, or not-analyzed;
100
+ - undocumented observed relations are grouped at a useful package/module scope instead of producing unbounded raw noise;
101
+ - intentional exclusions are explicit and counted, never hidden by an empty required-kind list;
102
+ - the known-case fixture matrix has 100% detection of expected findings and 0 unsupported findings;
103
+ - the real target report does not show zero findings merely because reconciliation was disabled.
104
+
105
+ ### Phase 3 — Real-artifact documentation inventory
106
+
107
+ The first AKOS audit exposed that the repository contains multiple documentation
108
+ surfaces. A single `documentedDocumentCount / documentCount` ratio mixed the 27
109
+ agent-corpus documents with human guides, project files, archives, and
110
+ unclassified Markdown. Doc Bridge now reports deterministic document counts by
111
+ classification and marks `docs-archive` as `archive`; the report highlights the
112
+ agent-corpus ratio separately. This keeps the metric useful without hiding the
113
+ full inventory.
114
+
115
+ The phase gate is satisfied only when the real AKOS artifact reports the
116
+ classification totals, the agent-corpus numerator/denominator, and evidence for
117
+ the classification rule. A changed source revision or configuration invalidates
118
+ the evidence and requires a new workflow and verification run.
119
+
120
+ ### Cycle 5 — Registry agent quality
121
+
122
+ **Purpose:** validate the configured agent from the AgentsKit Registry as an evidence-grounded assistant to discovery and classification.
123
+
124
+ **Checks:** run deterministic replay, inspect proposal origin/version/hashes/evidence, and run an assisted mode only when authorized credentials and budget are available.
125
+
126
+ **Success metrics:**
127
+
128
+ - proposal origin is the configured Registry agent ID and provider;
129
+ - base snapshot/report hashes match the run that produced the proposal;
130
+ - every intended change maps to an observed diagnostic or evidence item;
131
+ - deterministic replay is hash-stable;
132
+ - no proposal invents a path, claim, or remediation unsupported by the snapshot/report;
133
+ - human approval is required before any proposal application.
134
+
135
+ ### Cycle 6 — Report and interaction validation
136
+
137
+ **Purpose:** prove that the evidence can be understood and explored by a human.
138
+
139
+ **Checks:** real-browser interaction, responsive layouts, keyboard access, contrast, text overflow, loading/latency, console/network errors, level navigation, selection, filters, zoom/pan, and evidence drilldown.
140
+
141
+ **Success metrics:**
142
+
143
+ - 0 automated failures across all configured viewport/theme scenarios;
144
+ - 0 console errors, failed requests, horizontal page overflow, clipped controls, or contrast violations;
145
+ - package → module → file navigation preserves context and breadcrumbs;
146
+ - every visible finding has evidence and an actionable explanation;
147
+ - human visual approval is recorded for the current report hash.
148
+
149
+ ### Cycle 7 — Enterprise completion gate
150
+
151
+ **Purpose:** close the loop without overstating readiness.
152
+
153
+ **Success metrics:**
154
+
155
+ - all required cycles are `COMPLETE` for the same source revision and contract;
156
+ - no required surface remains `not-analyzed` without an approved exemption;
157
+ - source, target, config, docs, and generated artifacts have been reconciled;
158
+ - issue/PR tracking is updated only after authorization and includes the exact run ID;
159
+ - task-owned temporary artifacts are cleaned while user-owned or ambiguous artifacts remain untouched;
160
+ - structural decisions are documented in the relevant ADR/RFC;
161
+ - final report states residual risks and the next human action; enterprise readiness is not claimed while any required gate is pending.
162
+
163
+ ## Initial baseline to collect
164
+
165
+ The first cycle records counts and timings without treating them as success: entities by kind, relations by kind, declared relations, findings by code/severity/status, coverage states, document classifications, proposal evidence ratio, report load time, interaction latency, generated artifact size, and agent-output token/byte counts. Later cycles ratchet these metrics against the baseline instead of hiding regressions behind aggregate pass/fail status.
166
+
167
+ The executable benchmark stores an anonymization-safe baseline at `.doc-bridge/benchmarks/baseline.json`. The baseline contains numeric metrics and rule definitions only; it must not contain repository paths, document contents, package names, credentials, or agent prompts. It is created or replaced only by an explicit human-authorized command (`--write-baseline` or `--replace-baseline`).
168
+
169
+ ### Efficiency metric contract
170
+
171
+ | Metric family | Measurements | Initial gate |
172
+ |---|---|---|
173
+ | Pipeline | total, discovery, comparison, and proposal milliseconds | no more than 10% regression |
174
+ | Artifact | total bytes, initial HTML bytes, overview/levels/findings chunk bytes | initial HTML and total bytes no more than 10% regression |
175
+ | Agent context | hit rate, p50/p95 latency, p50/p95 response bytes, estimated p95 tokens, corpus-to-context reduction | 100% fixture hit rate; p95 bytes/tokens no more than 10% regression |
176
+ | Report UX | p95 first render, application response after real browser events, gesture duration, render-phase p95, and lazy-chunk bytes | first render ≤ 2s; application response ≤ 200ms; gesture is reported separately; phase/chunk data explains regressions |
177
+ | Analysis quality | analyzed ratio, documented/evidence ratios, overview node/edge counts, raw/compared relation ratio, finding density | tracked as evidence; fixture precision/recall gates semantic correctness |
178
+ | Reliability | UI failures, console errors, failed requests, incomplete stages | zero |
179
+
180
+ The report exposes aggregate insights suitable for case studies, but publication requires a separate redaction review. A benchmark passing means the measured contract passed; it does not prove semantic quality without the fixture set or prove enterprise readiness when surfaces are exempted.
181
+
182
+ ## Latest measured cycle
183
+
184
+ Cycle 2 shipped Doc Bridge `1.7.20` to `agentskit-os` and added deterministic resolution of literal dynamic imports. The official workflow run is `1787934948972-14360`; the verification run is `1787935063723-14641`.
185
+
186
+ - discovered entities: `12,930 → 12,937`;
187
+ - raw relations: `40,921 → 41,300`;
188
+ - package-level compared relations: `2,663 → 2,698`;
189
+ - findings: `2,580 → 2,615`, all with evidence;
190
+ - benchmark: passed against baseline with pipeline `6,052ms`, initial HTML `68,069B`, report artifact `26,338,177B`, first render p95 `113ms`, application response p95 `92ms`, gesture p95 `76ms`, and zero UI failures;
191
+ - agent search remained at `100%` hit rate and `1,067` estimated p95 tokens with `99%` context reduction;
192
+ - all 10 automated viewport/theme scenarios passed; human approval is still required for the current report hash.
193
+
194
+ The result is an improvement in semantic discovery, not proof that runtime wiring or non-literal loading is resolved. The next cycle should measure and improve those boundaries or add an explicit configuration/adapter path for projects that can provide runtime architecture evidence.
195
+
196
+ Cycle 3 shipped Doc Bridge `1.7.21` to `agentskit-os` and added explicit `dynamic-literal` relation metadata plus benchmark counters for unresolved dynamic loading and runtime-wiring candidates. The official workflow run is `1787937333636-16849`; the verification run is `1787937404198-17113`.
197
+
198
+ - `379` literal dynamic-import relations are now identifiable in the snapshot;
199
+ - `56` files contain non-literal loading that remains unresolved;
200
+ - `46` files contain runtime-wiring candidates that remain explicitly not analyzed;
201
+ - benchmark passed with first render p95 `59ms`, application response p95 `79ms`, gesture p95 `76ms`, and zero UI failures;
202
+ - entity/relation/documentation and agent-search metrics remained grounded, with agent search at `100%` hit rate and `1,067` estimated p95 tokens;
203
+ - automated UI evidence passed across all 10 viewport/theme scenarios; human approval is required for the current report hash.
204
+
205
+ Cycle 4 shipped Doc Bridge `1.7.24` to `agentskit-os` and added configurable JS/TS runtime-wiring detection for statically imported targets, explicit unresolved-wiring coverage, a runtime-wiring benchmark counter, and browser-runtime warmup for stable visual timing. The official workflow run is `1787938291045-21432`; the verification run is `1787938325493-21506`.
206
+
207
+ - the real repository produced `12,937` entities and `41,300` raw relations;
208
+ - package-level reconciliation compared `2,774` relations and produced `2,691` findings, all with evidence;
209
+ - `379` literal dynamic-import relations were identified and `56` files still contain unresolved non-literal loading;
210
+ - `116` files contain unresolved runtime-wiring candidates; no real target wiring was resolved automatically, confirming the conservative boundary rather than overstating coverage;
211
+ - the benchmark passed without replacing the anonymization-safe baseline: pipeline `6,094ms`, artifact `26,441,496B`, initial HTML `68,069B`, first render p95 `110ms`, application response p95 `71ms`, gesture p95 `76ms`, and zero UI failures;
212
+ - agent search remained at `100%` hit rate and `1,067` estimated p95 tokens with `99%` context reduction;
213
+ - all 10 automated viewport/theme scenarios passed with no console errors, failed requests, overflow, clipping, or contrast violations; the screenshots were reviewed locally and the harness is awaiting explicit human approval.
214
+
215
+ Cycle 5 shipped Doc Bridge `1.7.25` to `agentskit-os` and tightened the default runtime-wiring heuristic by making generic `bind` and `listen` calls opt-in while preserving explicit configuration. The official workflow run is `1787938899030-22694`; the verification run is `1787938967570-22837`.
216
+
217
+ - the real repository remained stable at `12,937` entities, `41,300` raw relations, `2,774` compared relations, and `2,691` evidence-backed findings;
218
+ - unresolved runtime-wiring candidates fell from `116` to `54` (`53.4%` reduction) by removing generic API false positives;
219
+ - the benchmark passed against the unchanged baseline: pipeline `6,042ms`, first render p95 `58ms`, application response p95 `71ms`, gesture p95 `75ms`, report artifact `26,422,816B`, and zero UI failures;
220
+ - agent search remained at `100%` hit rate and `1,067` estimated p95 tokens with `99%` context reduction;
221
+ - the explicit configuration path was covered by a fixture: `listen` is ignored by default and produces a relation when configured;
222
+ - all 10 automated viewport/theme scenarios passed; the verification run is awaiting human visual approval.
223
+
224
+ Cycle 6 shipped Doc Bridge `1.7.26` to `agentskit-os` and tightened unresolved-wiring coverage to require a potential target argument, excluding inline registrations and no-argument calls from architectural gaps while preserving identifiers, property accesses, and factory calls. The official workflow run is `1787939273593-23788`; the verification run is `1787939317943-23884`.
225
+
226
+ - the real repository remained stable at `12,937` entities, `41,300` raw relations, `2,774` compared relations, and `2,691` evidence-backed findings;
227
+ - unresolved runtime-wiring candidates fell from `54` to `22` (`59.3%` reduction; `81.0%` reduction from the `116`-file starting point);
228
+ - the benchmark passed against the unchanged baseline: pipeline `5,903ms`, first render p95 `106ms`, application response p95 `71ms`, gesture p95 `77ms`, report artifact `26,412,971B`, and zero UI failures;
229
+ - agent search remained at `100%` hit rate and `1,067` estimated p95 tokens with `99%` context reduction;
230
+ - the known-case fixture retained both unresolved identifier detection and the explicit custom-method configuration path;
231
+ - all 10 automated viewport/theme scenarios passed; the verification run is awaiting human visual approval.
232
+
233
+ Cycle 7 shipped Doc Bridge `1.7.28` to `agentskit-os` and made the large-report overview payload package-scoped. The first load now carries package topology plus compact diagnostic indexes; module/file detail remains lazy. The official workflow run is `1787940069560-26699`; the verification run is `1787940112369-26783`.
234
+
235
+ - the real repository remained stable at `12,937` entities, `41,300` raw relations, `2,774` compared relations, and `2,691` evidence-backed findings;
236
+ - unresolved runtime-wiring candidates fell from `22` to `9` (`59.1%` additional reduction; `92.2%` reduction from the `116`-file starting point); test/spec runtime wiring is excluded by default and covered by an explicit opt-in fixture;
237
+ - the overview payload contains `83` package entities and `627` package relations instead of the full canonical entity/relation set, while retaining `2,691` diagnostic group indexes and `2,691` relation-finding indexes;
238
+ - the benchmark passed against the unchanged baseline: pipeline `5,910ms`, first render p95 `113ms`, application response p95 `71ms`, gesture p95 `78ms`, report artifact `26,370,691B`, initial HTML `68,170B`, overview chunk `843,002B`, and zero UI failures;
239
+ - agent search remained at `100%` hit rate and `1,067` estimated p95 tokens with `99%` context reduction; reconciliation evidence coverage remained `100%`;
240
+ - all 10 automated viewport/theme scenarios passed with no console errors, failed requests, overflow, clipping, or contrast violations; screenshots were reviewed locally and the run is awaiting explicit human visual approval.
241
+
242
+ ## Current known blockers after the measured dogfood cycle
243
+
244
+ - package `1.7.28` is installed and verified in `agentskit-os`; the workflow run is `1787940069560-26699` and the verification run is `1787940112369-26783`;
245
+ - the benchmark passed against the original anonymization-safe baseline; the detailed cycle measurements are recorded above;
246
+ - the agent search fixture returned a grounded match for `100%` of queries, with p95 `1,067` estimated tokens and `99%` context reduction; reconciliation evidence coverage was `100%` for the `2,691` findings;
247
+ - the visual check passed all automated checks across 10 viewport/theme scenarios, but the current verification run remains `AWAITING_HUMAN_APPROVAL` until a human reviews the screenshots and approves run `1787940112369-26783`;
248
+ - only `3%` of reported analyzer scopes are complete in this target, so non-literal dynamic loading, unresolved runtime wiring, and generated code remain explicit coverage limitations;
249
+ - the current verification contract covers the package dogfood target, not the complete Doc Bridge enterprise objective.
250
+
251
+ The current run is `AWAITING_HUMAN_APPROVAL`, not complete. Discovery, package-level reconciliation, Registry-agent proof, documentation cohesion, export accuracy, CLI execution, and measured efficiency passed. The benchmark baseline was not replaced. After human approval, the next cycle should use these numbers as the comparison point and focus on classifying the remaining 9 production runtime-wiring candidates and improving documentation usefulness rather than report transport performance.
252
+
253
+ ### Phase 2 — Semantic classification measurement
254
+
255
+ The next cycle adds a small, deterministic labeled benchmark around the real reconciliation function. It covers confirmed, undocumented, stale, not-analyzed, conflicting, and unresolved declarations. The acceptance threshold is exact per-case diagnostic classification with non-empty evidence, plus `1.000` finding precision, `1.000` finding recall, and `1.000` evidence ratio. The AKOS verification contract runs this gate directly against the checked-out Doc Bridge source so a later change cannot silently preserve only the aggregate report counts.
@@ -5,25 +5,44 @@ description: Fail-closed, evidence-backed verification for humans and agents.
5
5
 
6
6
  # Verification harness
7
7
 
8
- `ak-verify` is the executable completion gate for work that must be proven, not merely compiled.
8
+ `ak-verify` is the executable completion gate for work that must be proven, not merely compiled. Harness 1.3 also blocks failed contract outcomes and measured regressions.
9
9
 
10
10
  ```bash
11
11
  ak-verify run --config .codex/verification.json --json
12
12
  ak-verify status --config .codex/verification.json --json
13
13
  ak-verify approve <run-id> approved --by human --config .codex/verification.json
14
14
  ak-verify authorize <run-id> approved --by human --config .codex/verification.json
15
+ ak-verify baseline replace benchmarks/new.json approved --by human --config .codex/verification.json
15
16
  ak-verify clean --periodic --config .codex/verification.json
16
17
  ```
17
18
 
18
19
  The contract is JSON so it works without adding a YAML runtime. It declares the artifact surfaces that apply to the run, executable checks, explicit non-applicable reasons, the verification profile, and tracking policy.
19
20
 
21
+ ## Global policy and project contract
22
+
23
+ The `ak-verify` executable is portable: any repository can run it from the
24
+ published Doc Bridge package. The verification contract remains project-local
25
+ at `.codex/verification.json` because endpoints, databases, UI flows, checks,
26
+ acceptance criteria, and tracking targets differ by repository.
27
+
28
+ The host-level agent policy is global and requires that every repository have
29
+ this project contract before implementation or verification. A repository
30
+ without it is `CLARIFYING`; agents must ask the human to define or authorize
31
+ the contract instead of selecting a default profile or inventing checks. This
32
+ combination provides one global completion rule without pretending that one
33
+ set of repository-specific checks fits every project.
34
+
35
+ Before implementation, the human intent and acceptance criteria must be explicit. Every criterion must map to an executable check and its expected evidence. If a criterion is not mapped, the run is `CLARIFYING` or `BLOCKED`; a project may not silently shrink the scope to the checks that are easiest to run.
36
+
20
37
  ## States
21
38
 
22
- `PLANNED` → `VERIFYING` → `AWAITING_HUMAN_APPROVAL` → `AWAITING_AUTHORIZATION` → `COMPLETE`.
39
+ `CLARIFYING` → `PLANNED` → `VERIFYING` → `AWAITING_HUMAN_APPROVAL` / `AWAITING_AUTHORIZATION` → `COMPLETE`.
40
+
41
+ Any failed required check, unavailable required surface, or unmapped acceptance criterion produces `BLOCKED` or `CLARIFYING`. The harness never promotes a run from `BLOCKED`, `AWAITING_HUMAN_APPROVAL`, or `AWAITING_AUTHORIZATION` to `COMPLETE` without the corresponding evidence and intent. A `COMPLETE` result applies only to the declared contract and must not be described as broader product validation when the broader scope was not declared.
23
42
 
24
- Any failed required check or unavailable required surface produces `BLOCKED`. The harness never promotes a run from `BLOCKED`, `AWAITING_HUMAN_APPROVAL`, or `AWAITING_AUTHORIZATION` to `COMPLETE` without the corresponding evidence and intent.
43
+ `default` is the low-friction profile used when `profile` is omitted. `strict` keeps fail-closed validation for the declared contract. `poc` and `custom` require explicit exemptions, which are included in the run evidence and cannot be silently hidden. `enterprise` requires all seven surfaces to be declared, requires measurement and tracking, and does not accept silent exemptions.
25
44
 
26
- `strict` is the default profile. `poc` and `custom` require explicit exemptions, which are included in the run evidence and cannot be silently hidden.
45
+ The seven applicability surfaces are `logic`, `cli`, `mcp`, `ui`, `docs`, `endpoint`, and `database`. A surface marked not applicable must include a reason. Endpoint and database validation remains conditional on the target: if the target uses one, declare a required real check; otherwise declare why it is not applicable.
27
46
 
28
47
  ## Evidence and recovery
29
48
 
@@ -33,4 +52,18 @@ Checks may emit one final JSON line with `status` set to `passed`, `failed`, or
33
52
 
34
53
  Visual checks must use a real browser or an explicitly configured equivalent. A passing build is not visual approval. Endpoint, database, CLI, and MCP checks must execute their real artifact when the contract marks that surface as required.
35
54
 
55
+ UI checks fail closed unless the check declares the `real-browser` and `screenshot` capabilities and its final structured result contains `capability: "real-browser"`, screenshot artifacts with project-relative paths, SHA-256 hashes and viewports, plus a passing result for every contract outcome mapped to that check. Missing files, stale hashes, placeholder pending results, and unmapped criteria block the run before human approval is available.
56
+
57
+ Delegated work does not weaken the gate. Subagents receive the parent contract hash, assigned criterion IDs, allowed scope, required capabilities, and expected evidence. Their output remains provisional until the orchestrator reruns this harness against the combined current source revision.
58
+
59
+ The final evidence ledger must distinguish `validated`, `partially validated`, `not analyzed`, `blocked`, and `not applicable`. Counts such as indexed documents, package presence, or rendered reports do not prove semantic documentation/code agreement, stale-content detection, runtime wiring, or UI behavior.
60
+
61
+ When a report is shared outside its repository, configure `report.privacy: 'anonymized'`. This is separate from `safety.redactSecrets`: secret redaction does not anonymize project names, paths, identifiers, snippets, or finding messages. The privacy check must inspect the generated HTML and all lazy chunks, not only the configuration.
62
+
63
+ When `measurement.required` is enabled, the named required check must emit structured evidence with `status: "passed"`, a numeric `metrics` object, a `baselineHash`, and an empty `regressions` array. Missing or regressed measurements block completion. Baselines are explicit, versioned artifacts and are never updated implicitly by a verification run.
64
+
65
+ Baseline replacement is an explicit, human-intended operation. `baseline replace` copies a JSON baseline into the configured target only when the source and target are inside the project root and appends a hash, actor, intent, and timestamp to `.codex/verification/baseline-audit.jsonl`.
66
+
67
+ The run JSON exposes `profile`, `profilePolicy`, `applicability`, `exemptions`, `checks`, `evidenceReferences`, `metrics`, `transitions`, `sourceRevision`, `contractHash`, `outputHash`, and the exact `runId`. Approval records are bound to the input, source, contract, and output hashes of that run.
68
+
36
69
  The harness only removes paths listed as task-owned and contained by configured cleanup roots. It never performs broad workspace deletion.
@@ -2,7 +2,7 @@
2
2
  "manifest_version": "0.3",
3
3
  "name": "doc-bridge",
4
4
  "display_name": "Doc Bridge",
5
- "version": "1.6.4",
5
+ "version": "1.7.45",
6
6
  "description": "Deterministic repository handoffs for coding agents, running locally without an LLM or API key.",
7
7
  "long_description": "Doc Bridge turns a repository's own documentation and ownership metadata into deterministic handoffs: where an agent should start, which paths it may edit, which checks it must run, and when a human must take over. The local connector exposes the same read-only contract available through Doc Bridge CLI and CI.",
8
8
  "author": {