@sema-agent/core 5.57.0 → 5.58.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (183) hide show
  1. package/CHANGELOG.md +48 -0
  2. package/dist/agents/cascade.d.ts +1 -1
  3. package/dist/agents/cumulative-stats.d.ts +1 -1
  4. package/dist/agents/observer.d.ts +2 -2
  5. package/dist/agents/peer-admission.d.ts +1 -1
  6. package/dist/agents/retain-ledger.d.ts +2 -2
  7. package/dist/agents/roster-store.d.ts +8 -8
  8. package/dist/agents/send-message-tool.d.ts +2 -2
  9. package/dist/agents/subagent-steps.d.ts +1 -1
  10. package/dist/agents/subagent.d.ts +13 -13
  11. package/dist/agents/team.d.ts +5 -5
  12. package/dist/agents/tool-filter.d.ts +2 -2
  13. package/dist/agents/verify.d.ts +1 -1
  14. package/dist/bench/metrics.d.ts +35 -35
  15. package/dist/brain/degrading.d.ts +1 -1
  16. package/dist/brain/errors.d.ts +3 -3
  17. package/dist/brain/reasoning.d.ts +2 -2
  18. package/dist/brain/repetition.d.ts +1 -1
  19. package/dist/brain/status-sink.d.ts +2 -2
  20. package/dist/brain/stream-shared.d.ts +1 -1
  21. package/dist/config/catalog.d.ts +5 -5
  22. package/dist/core/arg-summary.d.ts +4 -4
  23. package/dist/core/ask-class.d.ts +2 -2
  24. package/dist/core/ask-question.d.ts +1 -1
  25. package/dist/core/auto-compaction.d.ts +15 -15
  26. package/dist/core/auto-mode.d.ts +5 -5
  27. package/dist/core/background-agent-store.d.ts +20 -20
  28. package/dist/core/background-shell.d.ts +4 -4
  29. package/dist/core/checkpoint-store.d.ts +35 -27
  30. package/dist/core/context-edit.d.ts +1 -1
  31. package/dist/core/context-guard.d.ts +1 -1
  32. package/dist/core/exec-output-tail.d.ts +6 -6
  33. package/dist/core/file-snapshot-store.d.ts +8 -8
  34. package/dist/core/git-worktree-env.d.ts +3 -3
  35. package/dist/core/governance-codes.js +2 -0
  36. package/dist/core/hooks.d.ts +73 -33
  37. package/dist/core/hooks.js +87 -25
  38. package/dist/core/image-downsample.d.ts +1 -1
  39. package/dist/core/locked-config.d.ts +1 -1
  40. package/dist/core/lsp.d.ts +1 -1
  41. package/dist/core/mailbox-store.d.ts +1 -1
  42. package/dist/core/mcp.d.ts +3 -3
  43. package/dist/core/memory-engine/consolidation-driver.d.ts +207 -0
  44. package/dist/core/memory-engine/consolidation-driver.js +378 -0
  45. package/dist/core/memory-engine/consolidation.d.ts +46 -2
  46. package/dist/core/memory-engine/consolidation.js +1 -0
  47. package/dist/core/memory-engine/data-plane.d.ts +1 -1
  48. package/dist/core/memory-engine/distiller.d.ts +550 -0
  49. package/dist/core/memory-engine/distiller.js +598 -0
  50. package/dist/core/memory-engine/dual-root.d.ts +1 -1
  51. package/dist/core/memory-engine/engine.d.ts +47 -3
  52. package/dist/core/memory-engine/engine.js +37 -3
  53. package/dist/core/memory-engine/file-backend.d.ts +1 -1
  54. package/dist/core/memory-engine/index.d.ts +4 -2
  55. package/dist/core/memory-engine/index.js +4 -2
  56. package/dist/core/memory-engine/origin-clearance.d.ts +1 -1
  57. package/dist/core/memory-engine/scope-contract.d.ts +4 -4
  58. package/dist/core/memory-engine/sync-client.d.ts +16 -16
  59. package/dist/core/memory-engine/sync.d.ts +4 -4
  60. package/dist/core/memory-recall.d.ts +1 -1
  61. package/dist/core/memory.d.ts +2 -2
  62. package/dist/core/permission-rule-consent.d.ts +185 -36
  63. package/dist/core/permission-rule-consent.js +219 -44
  64. package/dist/core/permission-rule-model.d.ts +194 -31
  65. package/dist/core/permission-rule-model.js +93 -35
  66. package/dist/core/permission-rules.d.ts +9 -9
  67. package/dist/core/remote-env.d.ts +8 -8
  68. package/dist/core/roles.d.ts +3 -3
  69. package/dist/core/roles.js +1 -0
  70. package/dist/core/runner/assemble-result.d.ts +2 -2
  71. package/dist/core/runner/compaction-call-options.d.ts +3 -3
  72. package/dist/core/runner/memory-consolidation-driver.d.ts +49 -0
  73. package/dist/core/runner/memory-consolidation-driver.js +60 -0
  74. package/dist/core/runner/memory-consolidation.d.ts +1 -1
  75. package/dist/core/runner/prepare-config-doors.d.ts +3 -3
  76. package/dist/core/runner/prepare-task.d.ts +21 -21
  77. package/dist/core/runner/prepare-task.js +21 -14
  78. package/dist/core/runner/prepare-workspace-restore.d.ts +2 -2
  79. package/dist/core/runner/runtask.d.ts +11 -11
  80. package/dist/core/runner/session-rule-policy.d.ts +1 -1
  81. package/dist/core/runner/teardown-bounded.d.ts +1 -1
  82. package/dist/core/runner/tool-disclosure.d.ts +2 -2
  83. package/dist/core/runner/turn-attachments.d.ts +11 -11
  84. package/dist/core/scheduler.d.ts +5 -5
  85. package/dist/core/secret-env.d.ts +1 -1
  86. package/dist/core/sensitive-path-policy.d.ts +1 -1
  87. package/dist/core/session-policy-store.d.ts +2 -2
  88. package/dist/core/session-reconcile.d.ts +2 -2
  89. package/dist/core/session-store.d.ts +3 -3
  90. package/dist/core/session.d.ts +1 -1
  91. package/dist/core/shutdown-debug.d.ts +2 -2
  92. package/dist/core/side-query.d.ts +2 -2
  93. package/dist/core/spec-contract.d.ts +1 -1
  94. package/dist/core/store-contracts/contract-harness.d.ts +2 -2
  95. package/dist/core/store-contracts/contract-kit-version.d.ts +2 -2
  96. package/dist/core/store-contracts/mailbox-store-contract.d.ts +1 -1
  97. package/dist/core/store-contracts/mailbox-store-contract.js +1 -1
  98. package/dist/core/task-notification.d.ts +5 -5
  99. package/dist/core/task-registry-agent.d.ts +12 -12
  100. package/dist/core/task-registry-monitor.d.ts +1 -1
  101. package/dist/core/task-registry-shared.d.ts +41 -41
  102. package/dist/core/task-registry.d.ts +12 -12
  103. package/dist/core/tool-detach.d.ts +2 -2
  104. package/dist/core/tool-errors.d.ts +3 -3
  105. package/dist/core/tool-policy.d.ts +55 -28
  106. package/dist/core/tool-result-budget.d.ts +1 -1
  107. package/dist/core/tool-result-store.d.ts +2 -2
  108. package/dist/core/tools.d.ts +1 -1
  109. package/dist/core/trace.d.ts +26 -23
  110. package/dist/core/types.d.ts +123 -70
  111. package/dist/core/untrusted-egress.d.ts +1 -1
  112. package/dist/core/untrusted-text.d.ts +7 -7
  113. package/dist/core/wiring-manifest.d.ts +5 -5
  114. package/dist/core/workflow-journal-store.d.ts +14 -14
  115. package/dist/core/workflow-run-store-contract.d.ts +1 -1
  116. package/dist/core/workflow-run-store-contract.js +1 -1
  117. package/dist/core/workflow-run-store.d.ts +4 -4
  118. package/dist/engine/compaction/compaction.d.ts +3 -3
  119. package/dist/engine/compaction/utils.d.ts +2 -2
  120. package/dist/engine/execution-env/kill-tree.d.ts +1 -1
  121. package/dist/engine/execution-env/node-execution-env.d.ts +8 -8
  122. package/dist/engine/harness/agent-harness.d.ts +6 -6
  123. package/dist/engine/harness/messages.d.ts +1 -1
  124. package/dist/engine/harness/types.d.ts +10 -10
  125. package/dist/engine/llm/types.d.ts +14 -14
  126. package/dist/engine/loop/agent-loop.d.ts +3 -3
  127. package/dist/engine/loop/types.d.ts +4 -4
  128. package/dist/engine/lsp/node-lsp-manager.d.ts +2 -2
  129. package/dist/engine/session/import-validate.d.ts +1 -1
  130. package/dist/engine/session/log-digest.d.ts +1 -1
  131. package/dist/engine/session/memory-repo.d.ts +2 -2
  132. package/dist/engine/session/session.d.ts +4 -4
  133. package/dist/fixtures/index.d.ts +4 -4
  134. package/dist/index.d.ts +5 -4
  135. package/dist/index.js +3 -2
  136. package/dist/orchestration/goal.d.ts +1 -1
  137. package/dist/orchestration/run-spec.d.ts +1 -1
  138. package/dist/orchestration/run-workflow-tool.d.ts +12 -12
  139. package/dist/orchestration/workflow-governance.d.ts +4 -4
  140. package/dist/orchestration/workflow-observe.d.ts +1 -1
  141. package/dist/orchestration/workflow-script-runner.d.ts +1 -1
  142. package/dist/orchestration/workflow-script-store.d.ts +9 -9
  143. package/dist/orchestration/workflow-size-guideline.d.ts +1 -1
  144. package/dist/orchestration/workflow-types.d.ts +5 -5
  145. package/dist/orchestration/workflow.d.ts +10 -10
  146. package/dist/prompt-assembly/artifact-store.d.ts +1 -1
  147. package/dist/prompt-assembly/artifact.d.ts +1 -1
  148. package/dist/prompt-assembly/assemble.d.ts +1 -1
  149. package/dist/prompt-assembly/composer.d.ts +2 -2
  150. package/dist/prompt-assembly/epoch.d.ts +2 -2
  151. package/dist/prompt-assembly/event-registry.d.ts +1 -1
  152. package/dist/prompt-assembly/explain.d.ts +3 -3
  153. package/dist/prompt-assembly/tool-catalog.d.ts +1 -1
  154. package/dist/prompt-assembly/turn-snapshot.d.ts +4 -4
  155. package/dist/prompt-assembly/types.d.ts +12 -12
  156. package/dist/prompts/coordinator.d.ts +1 -1
  157. package/dist/prompts/default.d.ts +8 -8
  158. package/dist/prompts/simple-sections.d.ts +3 -3
  159. package/dist/prompts/supervisor.d.ts +2 -2
  160. package/dist/scenarios/full-body.d.ts +3 -3
  161. package/dist/scenarios/scenario-registry.d.ts +1 -1
  162. package/dist/stores/cc/sidecar-transcript.d.ts +3 -3
  163. package/dist/stores/file/fs-atomic.d.ts +2 -2
  164. package/dist/stores/file/index.d.ts +1 -1
  165. package/dist/stores/file/session-store.d.ts +2 -2
  166. package/dist/stores/file/workflow-journal-store.d.ts +4 -4
  167. package/dist/tools/fs/bash-readonly-classifier.d.ts +1 -1
  168. package/dist/tools/fs/encoding.d.ts +4 -4
  169. package/dist/tools/fs/fs-bash.d.ts +3 -3
  170. package/dist/tools/fs/fs-pdf.d.ts +1 -1
  171. package/dist/tools/fs/fs-shared.d.ts +6 -6
  172. package/dist/tools/fs/index.d.ts +2 -2
  173. package/dist/tools/fs/notebook.d.ts +1 -1
  174. package/dist/tools/fs/pdf.d.ts +1 -1
  175. package/dist/tools/fs/read-deny.d.ts +1 -1
  176. package/dist/tools/fs/safety.d.ts +9 -9
  177. package/dist/tools/fs/search.d.ts +2 -2
  178. package/dist/tools/monitor.d.ts +3 -3
  179. package/dist/tools/task-list.d.ts +2 -2
  180. package/dist/tools/web.d.ts +4 -4
  181. package/dist/tools/worktree.d.ts +5 -5
  182. package/package.json +1 -1
  183. package/test/export-surface.snapshot.json +56 -3
package/CHANGELOG.md CHANGED
@@ -1,5 +1,53 @@
1
1
  # Changelog
2
2
 
3
+ ## 5.58.0 — 2026-08-24
4
+
5
+ ### BREAKING (design/375 case A — per-segment batch consent; ships as one window, no compatibility arms)
6
+ - **`AskRequest.ruleSuggestions` → `ruleOffers`**: a closed discriminated union
7
+ (`single{rule,match,command}` | `batch{rules(1..5), uncoveredSegments}`); at most 2 offers, the
8
+ whole-string exact single always at index 0, the batch always last; choosing a batch is ONE yes to
9
+ all its members. `PendingAction.tool_approval` carries the same seat. An unknown offer kind
10
+ degrades to a single opaque row preserving the original wire index (fail toward asking).
11
+ - **`RuleApprovalRecord` schema:2**: `schema`+`offers` required; `selectedCandidate` → `selectedOffer`.
12
+ Pre-v2 rows refuse loudly (`record_schema_stale`, envelope read, ownership concealment keeps
13
+ precedence) — pending approvals from before the upgrade are re-asked, never migrated. Supplying the
14
+ retired `selectedCandidate` key (or `selectedOffer: null`) refuses `config.invalid_argument`.
15
+ - **`redeemRuleBatch`** returns a per-member discriminated table `{members, rev}` keyed by
16
+ `candidateIndex` (success rows carry `alreadyRedeemed`+`dot`, refused rows carry `reason`;
17
+ exactly one row per selected-offer member, projection map fixed); the three-array `ImportResult`
18
+ shape is deleted. `RuleApprovalRecordStore.get` may return `StaleRuleApprovalRecord` (envelope).
19
+ - **Engine conjunction arm** (§5.1): a compound command whose EVERY segment is admitted by some
20
+ eligible persisted rule is allowed — the segment deny/ask fence stays strictly ahead, whole-string
21
+ exact and the compound-prefix form keep their precedence, and a single-command prefix still never
22
+ admits a compound. `PersistedRuleHit` becomes `{ rules }` (covering set; evidence dots = union).
23
+ Eligibility is two-layered and single-sourced (`eligiblePersisted`/`eligibleContext`;
24
+ tombstoned/foreign-scope rules can never join a conjunction).
25
+ - The suggester mints offers (`suggestRulesForCommand` returns `RuleOffer[]`; only uncovered
26
+ segments enter a batch; the compound-prefix candidate is no longer offered — existing
27
+ compound-prefix rules keep matching forever); the card-edit face accepts one edited rule judged by
28
+ the single-rule coverage predicate.
29
+
30
+ ### Added
31
+ - **LLM-driven memory consolidation, slices 1+2 (design/376; default OFF, additive)**: the fold-plan
32
+ distiller becomes a product module (frozen prompt contract byte-equal to the benchmark's, canonical
33
+ hash pinned; structural repairs counted, repair-budget abort); `runMemoryConsolidationDriver` host
34
+ verb with a first-class run row (claim/attempt single-writer, stale-rev replay guard, write-failure
35
+ groups end the run `driver_failed`, `memory.consolidation_incomplete` notice, FORCED announcement
36
+ once per run); `consolidate` model role resolving consolidate → summarize → loud refusal (never the
37
+ default model). The bench rig now consumes the product module (scripted-arm projection digest
38
+ unchanged across the port). Auto-trigger wiring is a later slice; nothing runs without an explicit
39
+ driver seat.
40
+ - **Dependency-direction gate**: the engine's bottom-of-graph position is frozen mechanically (no
41
+ sibling `@sema-agent/*` in manifest or src imports).
42
+
43
+ ### Fixed
44
+ - **Internal-wording sweep over the shipped surface** (S10 class): board-coordinate and campaign-name
45
+ vocabulary cleared to zero across shipped `.d.ts` JSDoc and runtime strings (744 comment lines
46
+ rewritten 1:1, two runtime string leaks fixed); the wording gate's lexicon now also catches
47
+ 3-digit coordinate forms.
48
+ - Bench provenance axis (P1) landed with two harness fixes (seed origin carriage, read/write
49
+ provenance mode mismatch) — measurement only, no engine behavior change.
50
+
3
51
  ## 5.57.0 — 2026-08-23
4
52
 
5
53
  ### Added
@@ -105,7 +105,7 @@ export interface CascadeAttempt {
105
105
  index: number;
106
106
  model: ModelRef;
107
107
  passed: boolean;
108
- /** This rung's OWN+nested spend. ABSENT when the rung ran unpriced (RB-368 knownness, codex batch-4
108
+ /** This rung's OWN+nested spend. ABSENT when the rung ran unpriced (RB-368 knownness batch-4
109
109
  * F4) — a fabricated 0 here both misreported the attempt and let escalation ride under a finite
110
110
  * ceiling the engine could not actually enforce. */
111
111
  costMicroUsd?: number;
@@ -58,7 +58,7 @@ export interface CumulativeStatsAccumulator {
58
58
  /**
59
59
  * TRUE when ANY contributing leg published no `costMicroUsd`, i.e. it ran on a model with neither a
60
60
  * `RunnerDeps.pricing` entry nor a `Model.cost` declaration (`stats.costMicroUsd` is contractually
61
- * ABSENT there — types.ts / RB-368 [2076]).
61
+ * ABSENT there — types.ts / RB-368).
62
62
  *
63
63
  * Accumulating such a leg as `?? 0` publishes a total that is not one: the operation's real cost is
64
64
  * unknown, and reporting the priced legs' partial sum makes "no price table" indistinguishable from
@@ -257,12 +257,12 @@ export declare class ObserverPairing {
257
257
  retire(state: Exclude<ObserverPairingState, "armed">): void;
258
258
  }
259
259
  export declare function markObserverTaskId(taskId: string): void;
260
- /** codex OBS-2 F3 — lifecycle revocation: the wiring unmarks at observed-run settle (after the final
260
+ /** lifecycle revocation: the wiring unmarks at observed-run settle (after the final
261
261
  * drain + session release), so the set tracks LIVE observers only instead of growing per delegation
262
262
  * forever (and a long-dead observer id no longer trips the SendMessage target refusal). */
263
263
  export declare function unmarkObserverTaskId(taskId: string): void;
264
264
  export declare function isObserverTaskId(taskId: string): boolean;
265
- /** Diagnostic face (codex OBS-2b F-08): how many observer identities are currently LIVE — a test's
265
+ /** Diagnostic face: how many observer identities are currently LIVE — a test's
266
266
  * lifecycle assertion ("armed here, revoked after settle") without exposing the ids themselves. */
267
267
  export declare function observerTaskIdCount(): number;
268
268
  /** CC @18371202 — SendMessage refusal when the SENDER is an observer run. */
@@ -127,7 +127,7 @@ export interface PeerAdmissionOptions {
127
127
  export declare function createPeerAdmission(options?: PeerAdmissionOptions): PeerAdmission;
128
128
  export declare function peerAdmissionFor(scope: string | undefined, recipientKey: string, config: PeerAdmissionConfig, options?: PeerAdmissionOptions): PeerAdmission;
129
129
  /**
130
- * The delivery legs' ONE admission entry (codex 176-r2): registry seat-commit follows the SAME
130
+ * The delivery legs' ONE admission entry: registry seat-commit follows the SAME
131
131
  * refusal-is-side-effect-free rule as the sender table inside `admit` — the instance is looked up
132
132
  * WITHOUT an LRU touch (a detached fresh one serves a first-contact recipient), judged, and only an
133
133
  * ADMITTED message commits the seat (insert + touch + bounded eviction). A refusal to a
@@ -96,7 +96,7 @@ export interface RetainLedgerHooks {
96
96
  * `prepareTask` when the spec opts in, threaded to delegation tools as the TRUSTED `ctx.subagentRetain`,
97
97
  * and disposed by the Runner in the task's terminal `finally` (D4: abort in-flight resumes + unpin +
98
98
  * release every retained session — retain is NOT durable; resume is reachable only while the parent run
99
- * lives, codex-B3). Leaks are double-bounded by `max` + `ttlMs` (codex-B2).
99
+ * lives). Leaks are double-bounded by `max` + `ttlMs`.
100
100
  */
101
101
  export declare class SubagentRetainLedger {
102
102
  readonly ttlMs: number;
@@ -159,7 +159,7 @@ export declare class SubagentRetainLedger {
159
159
  get size(): number;
160
160
  get(parentToolCallId: string): SubagentRetainEntry | undefined;
161
161
  wasEvicted(parentToolCallId: string): boolean;
162
- /** codex impl-review MAJOR-1 — LAZY TTL sweep: evict every settled idle entry whose retain TTL
162
+ /** LAZY TTL sweep: evict every settled idle entry whose retain TTL
163
163
  * elapsed (unpin + release + tombstone), so an expired session is freed at the NEXT ledger touch
164
164
  * (every spawn-registration and resume entry call this) rather than only when its own resume is tried.
165
165
  * fidelity R4-1: no longer the only reaper — `markSettled` also arms a per-entry ACTIVE timer, so
@@ -8,7 +8,7 @@ export interface RosterEntry {
8
8
  sessionId?: string;
9
9
  /** The spawn's parent tool-call id (retain-ledger key), when known. */
10
10
  toolUseId?: string;
11
- /** δ 批 [1498]⑦ — the ROOT host session of the delegation tree at spawn (recovery-face grouping
11
+ /** the ROOT host session of the delegation tree at spawn (recovery-face grouping
12
12
  * key; equals the spawner's session at depth 1). Stored verbatim; no predicate arm consumes it
13
13
  * yet (enumeration/recovery is the reader). */
14
14
  rootSessionId?: string;
@@ -19,11 +19,11 @@ export interface RosterEntry {
19
19
  * inherited default. Closed set, single member today; absent = bound normally (or no word). */
20
20
  modelFallback?: "inherit_no_tier_binding";
21
21
  /** Spawn owner (task/session id) — consumers re-apply access checks against these two axes.
22
- * β 批 A-2: REQUIRED (default-deny predicate; an axis-less row would be unreachable). */
22
+ * A-2: REQUIRED (default-deny predicate; an axis-less row would be unreachable). */
23
23
  owner: string;
24
- /** Spawn principal scope. β 批 A-2: REQUIRED (`"default"` is the single-tenant spelling). */
24
+ /** Spawn principal scope. A-2: REQUIRED (`"default"` is the single-tenant spelling). */
25
25
  scope: string;
26
- /** codex R3 — mirrors the live registry's explicit session-scoping flag: session-ID owner matching
26
+ /** mirrors the live registry's explicit session-scoping flag: session-ID owner matching
27
27
  * is permitted ONLY when the spawn was session-scoped (TaskRegistry.canAccess parity — without
28
28
  * the flag, a task-scoped agent that fell out of the live registry would become visible through
29
29
  * the durable fallback to same-session callers the registry itself rejects). */
@@ -33,7 +33,7 @@ export interface RosterEntry {
33
33
  }
34
34
  /**
35
35
  * The caller's access axes for a roster read — the SAME predicate the live task registry applies
36
- * (`canAccess`): scope mismatch = invisible; then owner match, or session match. codex F3 (S1
36
+ * (`canAccess`): scope mismatch = invisible; then owner match, or session match. (S1
37
37
  * review): resolution MUST filter the candidate pool by access BEFORE the latest-wins reduce —
38
38
  * reduce-then-check lets one tenant's newer same-name entry shadow (and thereby suppress) another
39
39
  * tenant's older authorized binding. Implementations (incl. pg/tidb) apply this in the query.
@@ -43,7 +43,7 @@ export interface RosterAccess {
43
43
  scope?: string;
44
44
  sessionId?: string;
45
45
  }
46
- /** β 批 A-2 (clay 裁定 2026-07-22): DEFAULT-DENY both axes (TaskRegistry.canAccess byte-parity —
46
+ /** Ruled 2026-07-22: DEFAULT-DENY both axes (TaskRegistry.canAccess byte-parity —
47
47
  * the polarity split was the defect). Entries are written with both axes (spawn chain guarantees
48
48
  * it; {@link RosterEntry} requires them), so a missing axis is a broken row, answered with a miss. */
49
49
  export declare function entryAccessible(e: RosterEntry, access: RosterAccess): boolean;
@@ -77,7 +77,7 @@ export interface RosterStore {
77
77
  /** Record (or supersede — latest-wins per normalized name is applied at READ time) a binding. */
78
78
  record(entry: RosterEntry): void | Promise<void>;
79
79
  /** Resolve a name for a CALLER: filter by {@link RosterAccess} FIRST, then latest `createdAt`
80
- * wins within the authorized pool (codex F3 filter-before-reduce). Undefined = miss. */
80
+ * wins within the authorized pool (filter-before-reduce). Undefined = miss. */
81
81
  resolve(name: string, access: RosterAccess): RosterEntry | undefined | Promise<RosterEntry | undefined>;
82
82
  /** All live entries (diagnostics / deployment listing; order unspecified). */
83
83
  list(): RosterEntry[] | Promise<RosterEntry[]>;
@@ -148,7 +148,7 @@ export declare class FileRosterStore implements RosterStore {
148
148
  * miss on resolve/list — advisory layer), but it is no longer SILENT: an IO failure or corrupt
149
149
  * document was indistinguishable from an honestly empty roster. Never fires on plain ENOENT. */
150
150
  private discloseCorrupt;
151
- /** codex F4 — two read grades: MISSING file (ENOENT) is an honest empty roster, but a corrupt or
151
+ /** two read grades: MISSING file (ENOENT) is an honest empty roster, but a corrupt or
152
152
  * transiently unreadable file must THROW on the mutation path — treating it as empty would let
153
153
  * the next `record` atomically REPLACE the existing file with just one row (silent data loss).
154
154
  * Read paths (resolve/list) degrade the throw to a miss via `lenient`. */
@@ -59,7 +59,7 @@ export interface SendMessageToolOptions {
59
59
  senderName?: string;
60
60
  /** design/147 S1c — the deployment's durable roster, consulted after the live registry misses. */
61
61
  roster?: import("./roster-store.js").RosterStore;
62
- /** design/147 S3a (codex F2) — the PARENT run's retain ledger (RunInternals.parentRetainLedger):
62
+ /** design/147 S3a — the PARENT run's retain ledger (RunInternals.parentRetainLedger):
63
63
  * the sibling leg's retain entries live there. Consulted AFTER the own-run and session ledgers. */
64
64
  siblingRetain?: SubagentRetainLedger;
65
65
  /** RB-390 — the SPAWNING run's taskId when THIS run is a delegated child (RunInternals.parentTaskId,
@@ -78,7 +78,7 @@ export interface SendMessageToolOptions {
78
78
  * session-scoped sibling registers with owner = the parent's sessionId, which the parent's
79
79
  * taskId alone cannot satisfy when the two differ (same rationale as ToolExecuteContext.parentSessionId). */
80
80
  parentSessionId?: string;
81
- /** [1358] the process-level background-child observer (RunnerDeps.onBackgroundChildEvent) — the
81
+ /** the process-level background-child observer (RunnerDeps.onBackgroundChildEvent) — the
82
82
  * mount fills this so a SendMessage resume re-emits the spawn→tick→terminal family (fleet row
83
83
  * revival). Rides the OPTS closure like notify; RB-409's {@link enrichCtx} feeds the ctx twin. */
84
84
  onBackgroundChildEvent?: (event: import("../core/types.js").BackgroundChildEvent) => void;
@@ -84,7 +84,7 @@ export declare class SubagentStepRecorder {
84
84
  editedFiles(): SubagentEditedFile[] | undefined;
85
85
  /** The child's most recent tool intent as one human line ("Bash npm test", "Edit src/x.ts"); undefined if none. */
86
86
  currentAction(): string | undefined;
87
- /** [1828] (server cross-repo): {@link currentAction} pre-formatted into a fleet-view Progress-section
87
+ /** {@link currentAction} pre-formatted into a fleet-view Progress-section
88
88
  * shape — the SAME `{tool, target}` pair `currentAction` already concatenates into one line, exposed
89
89
  * separately so a consumer can look the tool up in its own registry instead of parsing prose.
90
90
  * Undefined if none observed (mirrors {@link currentAction}'s own undefined case exactly). */
@@ -335,8 +335,8 @@ export interface SubagentSteerHandle {
335
335
  /**
336
336
  * design/122 D1 — the retained child session id; present ONLY when the parent run enabled
337
337
  * {@link import("../core/types.js").TaskSpec.retainSubagentSessions} AND this child was actually
338
- * retained. ⚠️ Control-plane only (codex-B6): this is a continuation CAPABILITY — never expose it to
339
- * clients; target children by the opaque `parentToolCallId` instead (codex-m3).
338
+ * retained. ⚠️ Control-plane only: this is a continuation CAPABILITY — never expose it to
339
+ * clients; target children by the opaque `parentToolCallId` instead.
340
340
  */
341
341
  childSessionId?: string;
342
342
  /**
@@ -385,7 +385,7 @@ export declare function createSubagentResume(deps: {
385
385
  * apply to EVERY stop cycle, spawn and resume alike). Absent ⇒ honest ungated degradation (no
386
386
  * registry to count against — the pre-C1 immediate send). */
387
387
  registry?: import("../core/task-registry.js").TaskRegistry;
388
- /** design/147 S1a (codex F1) — the RESUMING caller's live injection entry: overrides the retained
388
+ /** design/147 S1a — the RESUMING caller's live injection entry: overrides the retained
389
389
  * snapshot's spawn-turn `parentNotify` (that lane is torn down with its turn — uplinks through it
390
390
  * would PARK instead of reaching the currently active parent turn, behind a success receipt). */
391
391
  currentParentNotify?: (n: TaskNotificationPayload, opts?: {
@@ -442,7 +442,7 @@ export declare function createSubagentResume(deps: {
442
442
  taskId?: string;
443
443
  /** S2b RB-27② — the resuming caller's resolved registry access (pairs with taskId). */
444
444
  taskAccess?: import("../core/task-registry.js").TaskAccess;
445
- /** [1358] the process-level background-child observer (ctx.onBackgroundChildEvent, Runner-filled).
445
+ /** the process-level background-child observer (ctx.onBackgroundChildEvent, Runner-filled).
446
446
  * When present TOGETHER with a revived registry row (taskId + successful revive), the resume cycle
447
447
  * re-emits the SAME spawn→tick→terminal event family as a first spawn — a fleet view's row revives
448
448
  * on the spawn frame (server treats spawn-after-tombstone as row revival). The spawn frame is
@@ -450,11 +450,11 @@ export declare function createSubagentResume(deps: {
450
450
  * tick — the observer's tombstone-period tick rejection depends on that order). No revived row ⇒
451
451
  * no frames (the steer-handle resume path has no a*-domain row to project). */
452
452
  bgSink?: (event: import("../core/types.js").BackgroundChildEvent) => void;
453
- /** [1358] row lifetime class echoed onto the revive frames (mirrors the registry row). */
453
+ /** row lifetime class echoed onto the revive frames (mirrors the registry row). */
454
454
  sessionScoped?: boolean;
455
- /** [1358] the row's display description, echoed onto the revive spawn frame (resume-marked). */
455
+ /** the row's display description, echoed onto the revive spawn frame (resume-marked). */
456
456
  rowDescription?: string;
457
- /** [1358]/codex 1351 F3 — the row's original observer metadata, reproduced on the revive frames so
457
+ /** the row's original observer metadata, reproduced on the revive frames so
458
458
  * a consumer rebuilding a tombstoned row gets its TYPE/name/ancestry back (never a blank row). */
459
459
  rowName?: string;
460
460
  rowAgentType?: string;
@@ -516,10 +516,10 @@ export interface SubagentToolOptions {
516
516
  */
517
517
  background?: {
518
518
  registry: import("../core/task-registry.js").TaskRegistry;
519
- /** β 批 A-2 (BREAKING 1.365.0): REQUIRED — the registry is default-deny on both axes, so a
519
+ /** A-2 (BREAKING 1.365.0): REQUIRED — the registry is default-deny on both axes, so a
520
520
  * background registration without a declared owner would be unreachable by every caller. */
521
521
  owner: string;
522
- /** β 批 A-2 (BREAKING 1.365.0): REQUIRED — `"default"` is the single-tenant spelling (explicit,
522
+ /** A-2 (BREAKING 1.365.0): REQUIRED — `"default"` is the single-tenant spelling (explicit,
523
523
  * matching the engine chain's `principal ?? "default"`), never implied by omission. */
524
524
  scope: string;
525
525
  /**
@@ -559,7 +559,7 @@ export interface SubagentToolOptions {
559
559
  * vetoes the park: the child settles failed exactly as pre-153, and the already-minted
560
560
  * checkpoint is expired (no orphans). Never called for sync children or non-suspending runs.
561
561
  *
562
- * ⚠️ Implementation obligation (blackboard [1593]/[1594], field-proven): "migrates it here"
562
+ * ⚠️ Implementation obligation (field-proven): "migrates it here"
563
563
  * means ACTIVELY PROMOTE — a check-only implementation that merely LOOKS UP the host durable
564
564
  * store rejects every child whose session lives in a split/transient tier (a subRunner's
565
565
  * private TTL store), and the veto fires on EVERY park: the checkpoint expires within
@@ -658,7 +658,7 @@ export interface SubagentToolOptions {
658
658
  * agent is selected) govern them verbatim: an explicit allowlist must NAME an injected tool for the
659
659
  * child to see it; `"*"`/absent allowlist = all. Nested delegation threads the SAME factory down
660
660
  * (`createSubagentToolNode(opts, depth+1)` carries it), so a grandchild spawn re-evaluates it — per-spawn,
661
- * at every level; each child gets its own instances bound to its own spawn moment ([797]: the
661
+ * at every level; each child gets its own instances bound to its own spawn moment (the
662
662
  * server-side factory is execute-time late-bound, so per-spawn evaluation is the confirmed shape).
663
663
  * A THROWING/rejecting factory degrades THAT spawn to zero injected tools (the child still runs)
664
664
  * with a FIXED generic `note:` disclosure on the report/async card (same lane as the model-override
@@ -793,7 +793,7 @@ export declare function asyncLaunchedReceipt(p: {
793
793
  taskId: string;
794
794
  /** The lane's own "what is running" opener — the Agent lane and the fork lane say different things. */
795
795
  workingLine: string;
796
- /** Whether a real notification sink is wired (codex R4: the receipt speaks the RUNTIME sink truth). */
796
+ /** Whether a real notification sink is wired (the receipt speaks the RUNTIME sink truth). */
797
797
  notify: boolean;
798
798
  /**
799
799
  * RB-220 — mirrors {@link import("../core/types.js").ToolExecuteContext.oneShot}: this run has no
@@ -821,7 +821,7 @@ export declare function createSubagentTool(opts: SubagentToolOptions): ToolSpec;
821
821
  * "All tools except …"; neither ⇒ "All tools". sema delta: `allowTools: ["*"]` is the documented
822
822
  * allow-everything sentinel (AgentDefinition.allowTools) — treated as NO allowlist, not a literal list.
823
823
  *
824
- * [901] deliberate: this renders the AUTHORED list raw — upstream's listing does too (gHm reads the
824
+ * deliberate: this renders the AUTHORED list raw — upstream's listing does too (gHm reads the
825
825
  * definition; the unknown-item split happens later, at spawn, in its resolveAgentTools). It is the
826
826
  * DECLARED boundary, NOT the effective child roster: an unknown entry may appear here while the
827
827
  * spawn-time filter (resolveToolSubset — the authority) drops it, and an alias-form divergence can
@@ -45,7 +45,7 @@ export type TeamEvent = {
45
45
  role: string;
46
46
  text: string;
47
47
  }
48
- /** [571]③ budget axes: the cumulative team budget was exhausted after this member's turn settled —
48
+ /** budget axes: the cumulative team budget was exhausted after this member's turn settled —
49
49
  * remaining rounds/members are skipped and the discussion goes straight to synthesis. */
50
50
  | {
51
51
  type: "budget_stop";
@@ -113,7 +113,7 @@ export interface TeamDiscussionOptions {
113
113
  /**
114
114
  * `maxWalltimeMs`/`maxTurns` are PER-RUN caps forwarded to every member/summary/synthesizer run.
115
115
  *
116
- * `maxTokens`/`maxCostUsd` ([571]③, CollabTemplate.budget mid-flight enforcement) are CUMULATIVE
116
+ * `maxTokens`/`maxCostUsd` (CollabTemplate.budget mid-flight enforcement) are CUMULATIVE
117
117
  * team budgets over member + summary + synthesizer spend (nested/delegated spend included, same
118
118
  * coordinate as the `stats` totals). Enforcement is checked after each member turn settles and is
119
119
  * booked — the crossing member is never killed in flight — and once a budget is exhausted
@@ -176,7 +176,7 @@ export interface TeamResult {
176
176
  * conclusion — so callers can tell a junk conclusion from a legitimate one. */
177
177
  conclusionValid: boolean;
178
178
  transcript: TeamTurn[];
179
- /** `costMicroUsd` ([571]③): cumulative team LLM spend in integer micro-USD (member + summary +
179
+ /** `costMicroUsd`: cumulative team LLM spend in integer micro-USD (member + summary +
180
180
  * synthesizer, nested included) — the same engine coordinate as `TaskResult.stats.costMicroUsd`.
181
181
  * Always set (0 when no run reported cost); optional only for type-level back-compat. */
182
182
  stats: {
@@ -185,7 +185,7 @@ export interface TeamResult {
185
185
  costMicroUsd?: number;
186
186
  };
187
187
  /**
188
- * [571]③ budget-stop attribution: set when a cumulative budget axis was exhausted and the
188
+ * budget-stop attribution: set when a cumulative budget axis was exhausted and the
189
189
  * discussion stopped dispatching further members/rounds early. `round`/`role`/`memberIndex`
190
190
  * identify the LAST member turn that ran (the one whose settled totals crossed the budget);
191
191
  * everything scheduled after it was skipped and the transcript went straight to synthesis.
@@ -202,7 +202,7 @@ export interface TeamResult {
202
202
  failures: number;
203
203
  /** How many times the shared transcript was summarized to stay under maxTranscriptTokens. */
204
204
  transcriptCompactions: number;
205
- /** design/80 D-B (codex review): set when a member durably PAUSED (suspended/needs_review) on a HITL gate —
205
+ /** design/80 D-B: set when a member durably PAUSED (suspended/needs_review) on a HITL gate —
206
206
  * the discussion STOPS (no synthesis on a half-done team) and surfaces the resume capability so the caller
207
207
  * can resume the paused member via the token, then re-run. `conclusionValid` is false in this case. */
208
208
  durablePause?: boolean;
@@ -13,10 +13,10 @@ import type { ToolSpec } from "../core/types.js";
13
13
  * newly-added sensitive tool to every agent. Names not present in the pool are ignored (no error) — an
14
14
  * allow/deny list is a filter over what's available, not an assertion that those tools exist.
15
15
  *
16
- * [901] anchor ruling (dual-leg verified, adversarially reviewed): this spawn-time item-level filter IS
16
+ * anchor ruling (dual-leg verified, adversarially reviewed): this spawn-time item-level filter IS
17
17
  * the upstream shape for allow entries — CC's resolveAgentTools partitions unknown names into an
18
18
  * `invalidTools` bucket nobody consumes at runtime (88 readable source), and a live 2.1.207 probe shows
19
- * the agent stays listed/delegable with the unknown item silently dropped, zero warnings. The [876]
19
+ * the agent stays listed/delegable with the unknown item silently dropped, zero warnings. The
20
20
  * "rejected at startup" posture this replaced had no verbatim anchor and did not survive verification.
21
21
  * The asymmetric prepare-time fail-loud for `TaskSpec.agents` DENY entries is a deliberate sema
22
22
  * extension (no upstream deny-list exists): an allow-typo silently narrows (safe direction), a
@@ -25,7 +25,7 @@ import type { ModelRef, TaskResult, TaskSpec, ToolSpec } from "../core/types.js"
25
25
  * orthogonal to and composable with the Stop hook (wire the verdict into a stop() hook to make it a
26
26
  * hard completion gate).
27
27
  *
28
- * DEPLOYMENT POSTURE (clay 裁定 2026-07-14): opt-in library primitive ONLY — never a scenario default,
28
+ * DEPLOYMENT POSTURE (ruled 2026-07-14): opt-in library primitive ONLY — never a scenario default,
29
29
  * never deployed implicitly. It is a **thin composition** over existing core seams — a verifier subtask
30
30
  * (own model role), a read-only tool set (via tool `effect`), {@link TaskSpec.outputSchema} for the
31
31
  * verdict, and the teacher-style fix loop — so it adds no Runner-core surface. Off by default; opt in
@@ -71,7 +71,7 @@ export interface DeliveryDecision {
71
71
  }
72
72
  /**
73
73
  * design/95 §3.2 — the lifecycle status of a run, so INFRA noise never poisons the mode signal
74
- * (codex BLOCKER B4 / design/89 §3.3 Beatsep red line). The Beatsep task#2b 9% pass-rate was infra
74
+ * (design/89 §3.3 Beatsep red line). The Beatsep task#2b 9% pass-rate was infra
75
75
  * death (OOM / K8S passthrough / nested-root), NOT a mode signal — feeding such runs into the
76
76
  * numerator turns them into spurious DELIVERED-WRONG / zero datapoints and reproduces the very noise
77
77
  * the design claims to have isolated. `buildReport` SCORES only `scored`; `infra-failed` / `excluded`
@@ -94,7 +94,7 @@ export interface RunRecord {
94
94
  /** Repeat/seed index within the cell (design/89 §3.4 N-repeat). */
95
95
  seed?: number | string;
96
96
  /**
97
- * design/95 §3.2 / codex B4 — lifecycle status. Only "scored" runs reach the numerator / cost /
97
+ * design/95 §3.2 — lifecycle status. Only "scored" runs reach the numerator / cost /
98
98
  * Pareto. Absent ⇒ "scored" (back-compat). See {@link RunStatus}.
99
99
  */
100
100
  runStatus?: RunStatus;
@@ -313,7 +313,7 @@ export interface ArmCell {
313
313
  /** Avoided-loss rate = CORRECTLY-WITHHELD / n. The supervisor's §2.2.1 value, single-listed. */
314
314
  correctlyWithheldRate: number;
315
315
  /**
316
- * codex MINOR / design/95 M10 — withhold treated as a binary detector of would-be-wrong delivery
316
+ * design/95 M10 — withhold treated as a binary detector of would-be-wrong delivery
317
317
  * (positive = withheld; ground truth = would-be-wrong, established by the §2.4 counterfactual). Reported
318
318
  * as precision/recall — NOT just the raw correctly-withheld count — so an arm cannot look good by
319
319
  * withholding indiscriminately (high count, low precision) or by rarely withholding (high precision,
@@ -420,10 +420,10 @@ export declare function riskTransferDisclosure(cells: ArmCell[]): {
420
420
  };
421
421
  };
422
422
  /**
423
- * codex B2 — a paired-binary comparison of two arms on the SAME tasks/seeds. The two arms' truly-
423
+ * a paired-binary comparison of two arms on the SAME tasks/seeds. The two arms' truly-
424
424
  * correct flags are paired row-by-row (paired-seed, design/89 §3.4). For binary paired data the
425
425
  * RIGHT tests are McNemar (discordant pairs) and a paired-bootstrap difference interval — NOT two
426
- * independent proportions with "CI non-overlap" (that ignores the pairing and is under-powered, codex B2).
426
+ * independent proportions with "CI non-overlap" (that ignores the pairing and is under-powered).
427
427
  */
428
428
  export interface PairedBinaryComparison {
429
429
  /** Arm A (e.g. "sup") vs arm B (e.g. "solo"). pDiff = P(A) − P(B). */
@@ -440,12 +440,12 @@ export interface PairedBinaryComparison {
440
440
  * b+c large). */
441
441
  mcnemarP: number;
442
442
  /** Bootstrap percentile 95% CI for the paired difference (lo, hi) — a percentile CI on the seeded
443
- * paired-bootstrap distribution, NOT Newcombe's analytic interval (codex MAJOR-B). Crosses 0 ⇒ not significant. */
443
+ * paired-bootstrap distribution, NOT Newcombe's analytic interval. Crosses 0 ⇒ not significant. */
444
444
  ci95: [number, number];
445
445
  /** TRUE iff the difference is statistically significant at α=0.05 (CI excludes 0 AND McNemar p<0.05). */
446
446
  significant: boolean;
447
447
  /**
448
- * codex B2 — pre-registered Minimum Detectable Effect at the observed nPairs (the difference this
448
+ * pre-registered Minimum Detectable Effect at the observed nPairs (the difference this
449
449
  * comparison COULD have detected at 80% power). When |pDiff| is below this AND not significant, the
450
450
  * verdict is "not powered" — NOT "no difference". This is the field that stops "N≥15 is an assertion".
451
451
  */
@@ -456,8 +456,8 @@ export interface PairedBinaryComparison {
456
456
  verdict: "A-better" | "B-better" | "no-detectable-difference" | "not-powered";
457
457
  }
458
458
  /**
459
- * codex B2 — compare two arms' delivered truly-correct as PAIRED binary. Pairs runs by the COMPOSITE
460
- * `(taskId, seed)` key (codex MAJOR-A: a bare `seed` collides across tasks — the same repeat index
459
+ * compare two arms' delivered truly-correct as PAIRED binary. Pairs runs by the COMPOSITE
460
+ * `(taskId, seed)` key (a bare `seed` collides across tasks — the same repeat index
461
461
  * recurs per task — so cross-task input would overwrite pairs and contaminate the McNemar sample);
462
462
  * only rows where BOTH arms delivered (not withheld, both scored) form a pair — a
463
463
  * withheld run has no delivered binary to pair (it is scored in the withhold/avoided-loss axis, not
@@ -469,7 +469,7 @@ export declare function pairedBinaryCompare(scoredRuns: RunRecord[], armA: Arm,
469
469
  bootstrapIters?: number;
470
470
  }): PairedBinaryComparison;
471
471
  /**
472
- * codex B3 — the pre-registered C2 exchange rate(s): how many human-review SECONDS we are willing to
472
+ * the pre-registered C2 exchange rate(s): how many human-review SECONDS we are willing to
473
473
  * pay to buy one unit of supervisor value. Without these, "higher P / more withholds ⇒ worth it" is
474
474
  * unfalsifiable (the "helpful but too expensive" counter-thesis cannot be observed). Declared BEFORE
475
475
  * the run (design/95 §9), not fit after.
@@ -482,20 +482,20 @@ export interface C2Thresholds {
482
482
  }
483
483
  export interface ParetoVerdict {
484
484
  arm: Arm;
485
- /** Non-dominated on (C1, C2.sec, P) — necessary condition to be considered at all (codex B3). */
485
+ /** Non-dominated on (C1, C2.sec, P) — necessary condition to be considered at all. */
486
486
  nonDominated: boolean;
487
487
  /** The pre-registered exchange-rate check: does the arm's extra C2 buy enough avoided-loss / saved
488
488
  * decisions to clear the declared threshold? Undefined when the arm has no extra C2 over baseline. */
489
489
  clearsC2Threshold?: boolean;
490
490
  /**
491
- * codex B3 — the label. An arm may ONLY be "worth-it" when it is non-dominated AND clears the C2
491
+ * the label. An arm may ONLY be "worth-it" when it is non-dominated AND clears the C2
492
492
  * threshold. A non-dominated but threshold-failing arm is "quality-tradeoff" (higher quality, but the
493
493
  * cost is not bought back) — NEVER "worth-it". A dominated arm is "dominated".
494
494
  */
495
495
  label: "worth-it" | "quality-tradeoff" | "dominated";
496
496
  }
497
497
  /**
498
- * codex B3 — classify each arm against the SUP/TEAM-vs-baseline value question with a pre-registered
498
+ * classify each arm against the SUP/TEAM-vs-baseline value question with a pre-registered
499
499
  * exchange rate. `baselineArm` is the cost floor to compare extra C2 against (default "solo").
500
500
  * `avoidedWrong` / `savedDecisions` per arm come from the cells / campaign report.
501
501
  *
@@ -518,31 +518,31 @@ export declare function paretoValueVerdict(cells: ArmCell[], thresholds: C2Thres
518
518
  * design/95 §7.1 — V1 (免重复劳动): campaign-level human-decision delta. The ONLY authoritative V1
519
519
  * mechanism (design/95 reconcile, MAJOR#2): cross-task reuse of a decided strategy. Semantics:
520
520
  * - `soloHumanDecisions` is the V1 BASELINE — the operator's actual up-front decision count when
521
- * running solo across the campaign (codex M7: NOT 0, NOT synthetic — measured from real operator
521
+ * running solo across the campaign (NOT 0, NOT synthetic — measured from real operator
522
522
  * prep across runs). `0` is only valid when the campaign genuinely needed no human decision.
523
523
  * - `repeatedDecisionsSaved = soloHumanDecisions − supHumanDecisions` (may be negative = SUP cost
524
524
  * MORE human decisions; reported honestly, not floored).
525
525
  */
526
526
  export interface CampaignV1Saved {
527
527
  decisionKind: string;
528
- /** V1 baseline (codex M7): operator's measured up-front decisions when running SOLO. Provenance MUST
528
+ /** V1 baseline: operator's measured up-front decisions when running SOLO. Provenance MUST
529
529
  * be recorded in the stamp (real prep, not 0/synthetic). */
530
530
  soloHumanDecisions: number;
531
531
  /** SUP arm's actual human-review decision count for this decision kind. */
532
532
  supHumanDecisions: number;
533
533
  /** = soloHumanDecisions − supHumanDecisions. Positive ⇒ SUP saved repeated labour (V1 evidence). */
534
534
  repeatedDecisionsSaved: number;
535
- /** Whether the baseline is a real measurement vs absent (codex M7 honesty gate). */
535
+ /** Whether the baseline is a real measurement vs absent (honesty gate). */
536
536
  baselineProvenance: "measured-operator-baseline" | "absent-not-claimable";
537
537
  }
538
538
  /**
539
- * design/95 §7.2 — V2 (蓝图清晰度) THREE-arm ablation (codex BLOCKER#1 + MAJOR#3 / M8). All three are
539
+ * design/95 §7.2 — V2 (蓝图清晰度) THREE-arm ablation. All three are
540
540
  * paired on the SAME seeds (paired===true ⇒ the three pX arrays/values are over the same n seeds).
541
541
  * INVARIANT: `n` is the paired-seed count common to all three sub-arms; the three deltas are computed
542
542
  * over those n pairs.
543
543
  * - pSoloRaw — only the original brief (no blueprint baseline).
544
544
  * - pSoloSupGenerated — the SUP-arm-GENERATED blueprint (attributable to supervisor; gen cost charged).
545
- * - pSoloExperimenterIdeal — the experimenter "ideal" blueprint. codex M8: this is a CONSTRAINED upper
545
+ * - pSoloExperimenterIdeal — the experimenter "ideal" blueprint. this is a CONSTRAINED upper
546
546
  * bound on "better-prompt help", NOT "supervisor's mechanistic ceiling", UNLESS the ideal blueprint
547
547
  * was generated under the {@link IdealBlueprintConstraints} (run-time, visible-brief-only, no oracle
548
548
  * access, same info budget, leak-reviewed). `idealConstraintsSatisfied` records which it is.
@@ -557,16 +557,16 @@ export interface BlueprintAblationTriple {
557
557
  pSoloExperimenterIdeal: number;
558
558
  /** = pSoloSupGenerated − pSoloRaw (supervisor blueprint's real net value). */
559
559
  deltaSupVsRaw: number;
560
- /** = pSoloExperimenterIdeal − pSoloRaw. LABELLED per `idealConstraintsSatisfied` (codex M8). */
560
+ /** = pSoloExperimenterIdeal − pSoloRaw. LABELLED per `idealConstraintsSatisfied`. */
561
561
  deltaIdealVsRaw: number;
562
- /** codex M8 — when false, deltaIdealVsRaw is "manual-prompt upper bound", NOT "supervisor potential". */
562
+ /** when false, deltaIdealVsRaw is "manual-prompt upper bound", NOT "supervisor potential". */
563
563
  idealConstraintsSatisfied: boolean;
564
- /** codex M3 — blueprint generation cost charged to the sup-generated arm (budget-match red line). */
564
+ /** blueprint generation cost charged to the sup-generated arm (budget-match red line). */
565
565
  blueprintGenC1MicroUsd: number;
566
566
  blueprintGenC2Count: number;
567
567
  blueprintGenC2Sec: number;
568
568
  }
569
- /** codex M8 — the constraints under which an "ideal" blueprint may be read as a supervisor-mechanism
569
+ /** the constraints under which an "ideal" blueprint may be read as a supervisor-mechanism
570
570
  * upper bound rather than a generic "better prompt helps" result. Recorded per ablation. */
571
571
  export interface IdealBlueprintConstraints {
572
572
  generatedBeforeRun: boolean;
@@ -575,10 +575,10 @@ export interface IdealBlueprintConstraints {
575
575
  sameInfoBudget: boolean;
576
576
  solutionLeakReviewed: boolean;
577
577
  }
578
- /** True iff ALL ideal-blueprint constraints hold (codex M8 gate). */
578
+ /** True iff ALL ideal-blueprint constraints hold. */
579
579
  export declare function idealConstraintsSatisfied(c: IdealBlueprintConstraints): boolean;
580
580
  /**
581
- * design/95 §7.3 — V3 (无关性隔离) contamination probe (canary, mechanical, zero LLM-judge). codex M9:
581
+ * design/95 §7.3 — V3 (无关性隔离) contamination probe (canary, mechanical, zero LLM-judge).
582
582
  * the isolated-vs-shared P delta CONFOUNDS execution order / context size / worker count / budget; the
583
583
  * PRIMARY signal must be per-subgoal oracle failure + explicit wrong-use of a sibling artifact (the
584
584
  * canary leak), with the P delta kept as a DIAGNOSTIC only.
@@ -590,12 +590,12 @@ export interface IsolationContamination {
590
590
  isolatedSubgoalP: number[];
591
591
  /** Per-factor P under shared context. length === factorCount. */
592
592
  sharedSubgoalP: number[];
593
- /** PRIMARY mechanical signal (codex M9): cross-factor canary leak rate (explicit token bleed). */
593
+ /** PRIMARY mechanical signal: cross-factor canary leak rate (explicit token bleed). */
594
594
  canaryLeakRate: number;
595
- /** PRIMARY mechanical signal (codex M9): count of factors that failed their oracle AND demonstrably
595
+ /** PRIMARY mechanical signal: count of factors that failed their oracle AND demonstrably
596
596
  * used a sibling factor's artifact (the controlled-ablation harm signal). */
597
597
  wrongSiblingArtifactUses: number;
598
- /** DIAGNOSTIC ONLY (codex M9 confound): mean(isolated P) − mean(shared P). Not the primary signal. */
598
+ /** DIAGNOSTIC ONLY (confound): mean(isolated P) − mean(shared P). Not the primary signal. */
599
599
  contaminationRateDiag: number;
600
600
  /** Extra C1 the isolation cost (multi-worker / repeated context load). */
601
601
  isolationOverheadC1: number;
@@ -614,7 +614,7 @@ export interface CampaignReport {
614
614
  implementedAxes: ImplementedAxes;
615
615
  }
616
616
  /**
617
- * design/95 §2 / codex B1 — one heterogeneity CELL: a `(suiteVersion, taskId, archetype, valueDimension,
617
+ * design/95 §2 — one heterogeneity CELL: a `(suiteVersion, taskId, archetype, valueDimension,
618
618
  * arm)` group with its folded {@link ArmCell}. The decision surface is the per-cell vector — NOT the
619
619
  * arm mean (the Simpson's-paradox guard). `groupKey` is the stable join key.
620
620
  */
@@ -648,7 +648,7 @@ export interface ComparisonCell {
648
648
  pareto: ParetoResult;
649
649
  }
650
650
  /**
651
- * codex MINOR11 / council Q1 — capability metadata: which of the four §7/Pareto value axes are actually
651
+ * council Q1 — capability metadata: which of the four §7/Pareto value axes are actually
652
652
  * IMPLEMENTED in this report. Consumers must read this before claiming a V1/V2/V3 result; an unset axis
653
653
  * is NOT a null result, it is "not measured here". (The T1 structures above are types + pure helpers;
654
654
  * the report-level V1/V2/V3 fields are populated by the harness, not synthesized in `buildReport`.)
@@ -660,7 +660,7 @@ export interface ImplementedAxes {
660
660
  v3: boolean;
661
661
  }
662
662
  export interface ValueJudgmentReport {
663
- /** codex B1 — per-arm cells grouped by full heterogeneity key (incl. arm). Single-arm; use for cell
663
+ /** per-arm cells grouped by full heterogeneity key (incl. arm). Single-arm; use for cell
664
664
  * inspection / `pairedBinaryCompare` inputs. NOT directly a cross-arm verdict surface (each is one arm). */
665
665
  groupedCells: GroupedCell[];
666
666
  /**
@@ -671,7 +671,7 @@ export interface ValueJudgmentReport {
671
671
  */
672
672
  comparisons: ComparisonCell[];
673
673
  /**
674
- * codex B1 — suite-level rollup: per-arm cells folded over ALL scored runs. A WEIGHTED summary for
674
+ * suite-level rollup: per-arm cells folded over ALL scored runs. A WEIGHTED summary for
675
675
  * dashboards ONLY — NOT a decision surface (folding heterogeneous tasks can flip a verdict, Simpson).
676
676
  * Consumers MUST decide on `comparisons`; this is convenience aggregation.
677
677
  */
@@ -687,7 +687,7 @@ export interface ValueJudgmentReport {
687
687
  /** design/89 §3.4: cells with n<8 are directional-only, not verdict-grade. Flagged, not dropped. */
688
688
  directionalOnly: Arm[];
689
689
  /**
690
- * codex B4 — the exclusion ledger: runs dropped from scoring (infra-failed / excluded) with reasons.
690
+ * the exclusion ledger: runs dropped from scoring (infra-failed / excluded) with reasons.
691
691
  * A regression-grade invariant: these never reach `cells` / `pareto` (no infra noise in the verdict).
692
692
  */
693
693
  excluded: Array<{
@@ -696,16 +696,16 @@ export interface ValueJudgmentReport {
696
696
  seed?: number | string;
697
697
  reason: RunStatus;
698
698
  }>;
699
- /** codex MINOR11 — which value axes this report actually measured. */
699
+ /** which value axes this report actually measured. */
700
700
  implementedAxes: ImplementedAxes;
701
701
  }
702
702
  /**
703
- * Top-level (codex B1+B4): SCORE only `scored` runs (B4 — drop infra-failed/excluded with reasons),
703
+ * Top-level: SCORE only `scored` runs (B4 — drop infra-failed/excluded with reasons),
704
704
  * group by `(suiteVersion, taskId, archetype, valueDimension, arm)` (B1 — the per-cell decision surface,
705
705
  * Simpson's-paradox guard), fold each group, AND provide a suite-level per-arm rollup for dashboards
706
706
  * (explicitly NOT a decision surface). Deterministic; no model.
707
707
  *
708
- * `opts.implementedAxes` lets the harness declare which §7 axes it populated (codex MINOR11); default
708
+ * `opts.implementedAxes` lets the harness declare which §7 axes it populated; default
709
709
  * = only Pareto (the V1/V2/V3 structures are T1 net-new, not synthesized here).
710
710
  */
711
711
  export declare function buildReport(runs: RunRecord[], opts?: {