omnius 1.0.591 → 1.0.592

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. package/.aiwg/addons/omnius-docs/README.md +15 -1
  2. package/.aiwg/addons/omnius-docs/manifest.json +28 -68
  3. package/.aiwg/addons/omnius-docs/skills/agent-failure-recovery/SKILL.md +2 -1
  4. package/.aiwg/addons/omnius-docs/skills/browser-interaction-validation/SKILL.md +2 -1
  5. package/.aiwg/addons/omnius-docs/skills/evidence-directed-delivery/SKILL.md +2 -1
  6. package/.aiwg/addons/omnius-docs/skills/hardware-evidence-audit/SKILL.md +2 -1
  7. package/.aiwg/addons/omnius-docs/skills/omnius-docs/SKILL.md +17 -7
  8. package/.aiwg/addons/omnius-docs/skills/omnius-inference-docs/SKILL.md +27 -0
  9. package/.aiwg/addons/omnius-docs/skills/omnius-integration-docs/SKILL.md +21 -0
  10. package/.aiwg/addons/omnius-docs/skills/omnius-ops-docs/SKILL.md +2 -0
  11. package/.aiwg/addons/omnius-docs/skills/omnius-realtime-docs/SKILL.md +2 -0
  12. package/.aiwg/addons/omnius-docs/skills/omnius-sponsor-docs/SKILL.md +2 -0
  13. package/.aiwg/addons/omnius-docs/skills/omnius-telegram-docs/SKILL.md +2 -0
  14. package/.aiwg/addons/omnius-docs/skills/omnius-tools-docs/SKILL.md +23 -0
  15. package/.aiwg/addons/omnius-docs/skills/omnius-version-compatibility-docs/SKILL.md +23 -0
  16. package/.aiwg/addons/omnius-docs/skills/runtime-provenance-audit/SKILL.md +2 -1
  17. package/.aiwg/addons/omnius-docs/skills/secrets-and-config-audit/SKILL.md +2 -1
  18. package/.aiwg/addons/omnius-docs/skills/test-surface-audit/SKILL.md +2 -1
  19. package/.aiwg/addons/omnius-docs/skills/workspace-reality-audit/SKILL.md +2 -1
  20. package/.aiwg/addons/omnius-rest-docs/README.md +3 -0
  21. package/.aiwg/addons/omnius-rest-docs/manifest.json +27 -20
  22. package/.aiwg/addons/omnius-rest-docs/skills/omnius-rest-docs/SKILL.md +9 -5
  23. package/README.md +36 -0
  24. package/dist/discovery.d.ts +50 -0
  25. package/dist/index.js +5975 -4021
  26. package/dist/library.d.ts +7 -0
  27. package/dist/library.js +950 -0
  28. package/dist/postinstall-daemon.cjs +18 -0
  29. package/dist/providerRegistry.d.ts +80 -0
  30. package/dist/service-version.d.ts +35 -0
  31. package/docs/.vitepress/config.mts +8 -0
  32. package/docs/DISCOVERY.json +20224 -0
  33. package/docs/DISCOVERY.md +648 -0
  34. package/docs/HANDOFF-crl-encoder-decoder-fix.md +129 -0
  35. package/docs/agent-memory/INDEX.md +9 -4
  36. package/docs/agent-memory/index.md +7 -0
  37. package/docs/concept-relational-language.md +869 -0
  38. package/docs/context-management-medium-models-proposal.md +449 -0
  39. package/docs/dedup-false-positive-meta-analysis.md +96 -0
  40. package/docs/discovery/catalog-overrides.json +724 -0
  41. package/docs/duplicate-calls-root-cause-analysis.md +91 -0
  42. package/docs/duplicate-calls-root-cause-deep.md +155 -0
  43. package/docs/ephemeral-skill-pack-small-context.md +57 -0
  44. package/docs/explorations/context-window-todo-association.md +156 -0
  45. package/docs/explorations/todo-association-verify.json +30 -0
  46. package/docs/explorations/verification-ledger.json +45 -0
  47. package/docs/explorations/verify-todo-association.sh +30 -0
  48. package/docs/flowstate.md +806 -0
  49. package/docs/getting-started/install.md +24 -0
  50. package/docs/getting-started/model-providers.md +13 -0
  51. package/docs/guides/agent-integration.md +87 -0
  52. package/docs/guides/bring-your-own-inference.md +126 -0
  53. package/docs/guides/tools-and-web-search.md +95 -0
  54. package/docs/index.md +14 -0
  55. package/docs/longhaul-35b-workorders.md +496 -0
  56. package/docs/memory-integration-analysis.md +303 -0
  57. package/docs/model-capability-awareness-and-multimodal-memory-root-fix.md +799 -0
  58. package/docs/multimodal-identity-memory-implementation.md +76 -0
  59. package/docs/omnius-self-edit-eval-2026-06-10.md +169 -0
  60. package/docs/opencode-agentic-loop-comparison.md +290 -0
  61. package/docs/operations/security-and-remote-access.md +2 -2
  62. package/docs/operations/version-compatibility.md +63 -0
  63. package/docs/proposals/git-progress-tracking-strategy.md +289 -0
  64. package/docs/proposals/opencode-modules/backendAdapter.ts +443 -0
  65. package/docs/proposals/opencode-modules/childSession.ts +288 -0
  66. package/docs/proposals/opencode-modules/compactionAgent.ts +101 -0
  67. package/docs/proposals/opencode-modules/orchestrator.ts +387 -0
  68. package/docs/proposals/opencode-modules/runner.ts +258 -0
  69. package/docs/reference/auth-map.md +87 -196
  70. package/docs/reference/configuration.md +27 -0
  71. package/docs/reference/rest-api.md +7 -0
  72. package/docs/reference/slash-commands.md +125 -2
  73. package/docs/research/_archived/README.md +18 -0
  74. package/docs/research/_archived/context_window_attention_model.py +418 -0
  75. package/docs/research/_archived/context_window_attention_spec.md +55 -0
  76. package/docs/research/_archived/context_window_attention_weights.json +68 -0
  77. package/docs/research/k-splanifolds.pdf +0 -0
  78. package/docs/research/personality-verbosity-control.md +293 -0
  79. package/docs/rest/INDEX.md +7 -0
  80. package/docs/rest/QUICKREF.md +18 -0
  81. package/docs/rest/REST-DOCS-MANIFEST.json +1 -0
  82. package/docs/rest/auth-and-scopes.md +7 -1
  83. package/docs/rest/endpoints/discovery.md +44 -0
  84. package/docs/rest/endpoints/events.md +5 -0
  85. package/docs/rest/endpoints/tools.md +9 -0
  86. package/docs/reviews/adversary-system-review.md +42 -0
  87. package/docs/sana-and-video-generation-integration-plan.md +712 -0
  88. package/docs/session-diary-llm-training-analysis.md +218 -0
  89. package/docs/telegram-dmn-curiosity-outreach-scaffold.md +91 -0
  90. package/docs/telegram-mid-horizon-download-loop-handoff.md +468 -0
  91. package/docs/telegram-reflection-corpus-integration-plan.md +306 -0
  92. package/docs/telegram-unified-tooling-architecture.md +332 -0
  93. package/docs/threat-model.md +868 -0
  94. package/docs/trajectory-grounding.md +160 -0
  95. package/docs/voice-flow-architecture.md +489 -0
  96. package/docs/work-orders/WO-AM-GAPS.md +638 -0
  97. package/docs/work-orders/daemon-hud-ui-overhaul.md +82 -0
  98. package/docs/work-orders/hermes-architecture-deltas/01-public-scrutiny-provenance-control/INDEX.md +21 -0
  99. package/docs/work-orders/hermes-architecture-deltas/01-public-scrutiny-provenance-control/WORKORDER.md +225 -0
  100. package/docs/work-orders/hermes-architecture-deltas/02-context-engine-plugin-boundary/INDEX.md +20 -0
  101. package/docs/work-orders/hermes-architecture-deltas/02-context-engine-plugin-boundary/WORKORDER.md +198 -0
  102. package/docs/work-orders/hermes-architecture-deltas/03-typed-gateway-event-stream/INDEX.md +19 -0
  103. package/docs/work-orders/hermes-architecture-deltas/03-typed-gateway-event-stream/WORKORDER.md +172 -0
  104. package/docs/work-orders/hermes-architecture-deltas/04-task-local-gateway-context/INDEX.md +19 -0
  105. package/docs/work-orders/hermes-architecture-deltas/04-task-local-gateway-context/WORKORDER.md +169 -0
  106. package/docs/work-orders/hermes-architecture-deltas/05-process-lifecycle-monitoring-notifications/INDEX.md +22 -0
  107. package/docs/work-orders/hermes-architecture-deltas/05-process-lifecycle-monitoring-notifications/WORKORDER.md +189 -0
  108. package/docs/work-orders/hermes-architecture-deltas/06-vision-evidence-routing-ladder/INDEX.md +22 -0
  109. package/docs/work-orders/hermes-architecture-deltas/06-vision-evidence-routing-ladder/WORKORDER.md +199 -0
  110. package/docs/work-orders/hermes-architecture-deltas/07-durable-multi-agent-kanban/INDEX.md +20 -0
  111. package/docs/work-orders/hermes-architecture-deltas/07-durable-multi-agent-kanban/WORKORDER.md +174 -0
  112. package/docs/work-orders/hermes-architecture-deltas/08-completion-critic-reconciliation-ledger/INDEX.md +22 -0
  113. package/docs/work-orders/hermes-architecture-deltas/08-completion-critic-reconciliation-ledger/WORKORDER.md +226 -0
  114. package/docs/work-orders/hermes-architecture-deltas/INDEX.md +38 -0
  115. package/docs/work-orders/omnius-context-engineering-behavior-fixes.md +281 -0
  116. package/docs/work-orders/telegram-dropbear-context-rca-workorder.md +202 -0
  117. package/docs/work-orders/world-class-memory-compiler/README.md +162 -0
  118. package/docs/work-orders/world-class-memory-compiler/TRACKER.md +179 -0
  119. package/docs/work-orders/world-class-memory-compiler/WO-01-exact-request-budget.md +79 -0
  120. package/docs/work-orders/world-class-memory-compiler/WO-02-typed-memory-fabric.md +65 -0
  121. package/docs/work-orders/world-class-memory-compiler/WO-03-dependency-working-set.md +55 -0
  122. package/docs/work-orders/world-class-memory-compiler/WO-04-inference-memory-compiler.md +67 -0
  123. package/docs/work-orders/world-class-memory-compiler/WO-05-artifact-fidelity-materialization.md +72 -0
  124. package/docs/work-orders/world-class-memory-compiler/WO-06-temporal-hybrid-retrieval.md +49 -0
  125. package/docs/work-orders/world-class-memory-compiler/WO-07-evaluation-harness.md +45 -0
  126. package/docs/work-orders/world-class-memory-compiler/WO-08-rollout-legacy-removal.md +45 -0
  127. package/docs/x402-remote-inference-plan.md +323 -0
  128. package/npm-shrinkwrap.json +108 -117
  129. package/package.json +7 -6
  130. package/templates/AGENTS.md +6 -0
  131. package/templates/OMNIUS.md +20 -0
@@ -0,0 +1,162 @@
1
+ # World-Class Memory Compiler Program
2
+
3
+ **Status:** active implementation program
4
+ **Owner:** Omnius orchestration and memory packages
5
+ **Last updated:** 2026-07-13
6
+
7
+ ## Mission
8
+
9
+ Replace chronological transcript compaction with a provenance-preserving memory
10
+ compiler. The compiler manages context as a materialized view of typed,
11
+ versioned task memory; it does not coerce the agent's reasoning, deny tool use,
12
+ or replace evidence with generic summaries.
13
+
14
+ The implementation has one non-negotiable invariant:
15
+
16
+ > Memory management may preserve, retrieve, organize, and defer. It must never
17
+ > block exploration, source reads, edits, verification, or a model's legitimate
18
+ > change of plan.
19
+
20
+ ## Architecture at a glance
21
+
22
+ ```text
23
+ immutable event ledger ──► typed memory records ──► dependency graph
24
+ │ │
25
+ └────► hybrid index ◄─┘
26
+ │
27
+ exact outbound request ──► budget compiler ──► working-set materializer
28
+ │
29
+ isolated inference memory compiler
30
+ │
31
+ validated MemoryDelta + audit trail
32
+ ```
33
+
34
+ The outbound request is built before any compaction decision. Its complete
35
+ budget includes stable prompts, tool schemas, active steering, task state,
36
+ selected evidence, output reservation, and safety margin. No automatic or
37
+ manual compaction may run while the final request has at least 40% of the real
38
+ model context window free.
39
+
40
+ ## Memory classes
41
+
42
+ | Class | Canonical contents | Rule |
43
+ |---|---|---|
44
+ | Authority | current user intent, active steering, trusted policy | never summarize or retrieve from untrusted content |
45
+ | Task / working set | open requirements, next action, unresolved claims, live verifier | materialize for the active task epoch |
46
+ | Artifact | exact files/chunks, content hash, range, revision | exact body is canonical; extract is derived |
47
+ | Action ledger | intent → tool call → result → mutation → verifier → outcome | preserve transaction integrity |
48
+ | Episodic | immutable event history | archive-first; not blindly model-visible |
49
+ | Semantic | validated facts and decisions with temporal validity | source-backed and supersedable |
50
+ | Procedural | validated reusable workflow lessons | explicitly promoted, never inferred from one failure |
51
+
52
+ ## Work-order index and execution todos
53
+
54
+ | ID | Work order | Depends on | Status | Completion evidence |
55
+ |---|---|---|---|---|
56
+ | WO-01 | [Exact Request and Budget Compiler](./WO-01-exact-request-budget.md) | — | in progress | exact serializer, dump ledger, threshold tests |
57
+ | WO-02 | [Typed Memory Fabric and Event Ledger](./WO-02-typed-memory-fabric.md) | WO-01 | in progress | immutable ledger, authority/progress tests |
58
+ | WO-03 | [Dependency Graph and Working-Set Materializer](./WO-03-dependency-working-set.md) | WO-02 | in progress | graph-cut, active-root, and transaction tests |
59
+ | WO-04 | [Inference-Driven Memory Compiler](./WO-04-inference-memory-compiler.md) | WO-01, WO-03 | in progress | exact-request shadow proposal tests |
60
+ | WO-05 | [Artifact Fidelity and Explicit Materialization](./WO-05-artifact-fidelity-materialization.md) | WO-02, WO-03 | in progress | full-read, stale revision, extract coverage tests |
61
+ | WO-06 | [Temporal Hybrid Retrieval](./WO-06-temporal-hybrid-retrieval.md) | WO-02, WO-03, WO-05 | in progress | retrieval/relevance/abstention tests |
62
+ | WO-07 | [Replay Evaluation and Adversarial Harness](./WO-07-evaluation-harness.md) | WO-01–WO-06 | in progress | deterministic trace reports and tiered harness results |
63
+ | WO-08 | [Shadow Rollout, Migration, and Legacy Removal](./WO-08-rollout-legacy-removal.md) | WO-07 | planned | canary report, rollback drill, legacy deletion |
64
+
65
+ `[ ]` entries in each work order are the canonical engineering todos. The
66
+ [granular tracker](./TRACKER.md) reconciles every implementation and promotion
67
+ subtask. An item cannot be marked complete from a model claim: it requires the
68
+ named code, focused tests, typecheck, and recorded acceptance evidence.
69
+
70
+ ## Global acceptance criteria
71
+
72
+ - [ ] The compaction trigger uses the exact pending request, not historical
73
+ messages alone.
74
+ - [ ] No request compacts with 40% or more real-window headroom remaining.
75
+ - [ ] A current action, exact evidence it needs, and its verifier remain an
76
+ atomic dependency transaction.
77
+ - [ ] Current user authority and unacknowledged steering are present exactly
78
+ once and cannot be overwritten by history or tool output.
79
+ - [ ] Full source reads are first-class; branch extraction is derived,
80
+ coverage-checked, and never silently substitutes for source.
81
+ - [ ] Memory compiler failure results in `hold` and an audit record, never
82
+ heuristic deletion.
83
+ - [ ] Repeated identical compiler analyses are deduplicated by a content/state
84
+ fingerprint; state change is required to re-run.
85
+ - [ ] All compaction choices are inspectable in the TUI and session logs.
86
+ - [ ] The implementation is benchmarked on real replay traces and adversarial
87
+ traces across small, medium, and large model tiers before enforcement.
88
+
89
+ ## Research basis
90
+
91
+ These work orders borrow principles, not implementations. LongMemEval motivates
92
+ separate indexing, retrieval, and reading with temporal updates and abstention;
93
+ MemGPT motivates memory tiers; RAPTOR motivates retaining raw leaves beside
94
+ derived hierarchy; graph-memory work motivates associative recall; and
95
+ instruction hierarchy motivates authority separation. See
96
+ [LongMemEval](https://arxiv.org/abs/2410.10813),
97
+ [MemGPT](https://arxiv.org/abs/2310.08560),
98
+ [RAPTOR](https://arxiv.org/abs/2401.18059),
99
+ [HippoRAG 2](https://arxiv.org/abs/2502.14802), and
100
+ [Instruction Hierarchy](https://arxiv.org/abs/2404.13208).
101
+
102
+ Research claims are hypotheses to test against Omnius traces, not a reason to
103
+ copy a paper's architecture wholesale.
104
+
105
+ ## Verified implementation evidence — 2026-07-13
106
+
107
+ - `request-budget.ts` snapshots and fingerprints the actual JSON-wire-equivalent
108
+ outgoing payload (including aliases and model-visible memory prefixes), then
109
+ computes a segmented budget from that one immutable request; every
110
+ context-window dump persists or safely recompiles that exact budget.
111
+ - `ContextMemoryLedger` persists append-only authority, task, action, tool,
112
+ artifact, mutation, verifier, dependency, transaction, and supersession
113
+ records. The runner records the user/task root before its first tool call and
114
+ records tool transactions without making telemetry a tool gate.
115
+ - `memory-compiler.ts` rejects an inference cut that would archive authority,
116
+ active dependency evidence, a partial tool transaction, an unknown record,
117
+ or an insufficiently classified proposal.
118
+ - Full source reads remain canonical in `EvidenceLedger`; a completed manual
119
+ branch extract can replace only the rendered view while canonical source is
120
+ retained. Automatic branch routing now requires real headroom exhaustion
121
+ rather than a fixed 20% file-size boundary. Capped grep/tool output is typed
122
+ partial success with recoverable provenance and deterministic narrowing.
123
+ - Active working sets are graph- and transaction-closed: a soft rendering
124
+ budget may omit optional history but cannot split a live action from its
125
+ call, result, source, mutation, or verifier. The bounded retriever filters
126
+ by epoch, authority, temporal validity, path/revision/range and active graph
127
+ roots before lexical/host-semantic ranking; it returns explicit abstention on
128
+ missing or over-budget evidence.
129
+ - The existing isolated compaction analyst now caches byte-identical state
130
+ signatures and exposes cache/fingerprint information in its audit record.
131
+ - The exact final request now drives the v2 inference memory compiler at the
132
+ shared pre-backend boundary. It projects every mutable message body into a
133
+ typed candidate, requires one inference-selected disposition per candidate,
134
+ and applies nothing unless graph, authority, artifact, and exact
135
+ post-materialization budget checks all pass. Missing or invalid inference is
136
+ a `hold`, never a heuristic truncation. `OMNIUS_MEMORY_COMPILER_MODE=shadow`
137
+ is the explicit rollback/diagnostic mode.
138
+ - Immutable `omnius-artifact://sha256/...` references are range-readable with
139
+ the first-class `artifact_read` tool. A partial representation is trusted
140
+ code materializing declared lines from that object, never model-authored
141
+ source text.
142
+ - `memory-compiler-replay.mjs` executes versioned, deterministic adversarial
143
+ fixtures against the production ledger, request budget, and graph-cut
144
+ validator. Current cases include active evidence, partial extraction,
145
+ prompt-shaped tool data, source supersession, verifier-after-change, and a
146
+ 20-artifact batch.
147
+
148
+ Focused verification completed:
149
+
150
+ ```text
151
+ @omnius/orchestrator: 66 focused tests passed
152
+ @omnius/memory: 11 SQLite ledger tests passed
153
+ @omnius/execution: 27 focused tests passed
154
+ TypeScript: all 11 workspace packages built
155
+ memory compiler replay: 14/14 deterministic adversarial scenarios passed
156
+ live large-model extraction: 20/20 approved-A100 contracts passed
157
+ ```
158
+
159
+ These are foundations, not a release claim. WO-07 and WO-08 remain incomplete
160
+ until real redacted trace comparison, controlled model-tier evaluation,
161
+ shadow-operation evidence, canary/rollback, and legacy-path deletion prove the
162
+ new pipeline is safer than the old one.
@@ -0,0 +1,179 @@
1
+ # Memory Compiler Implementation Tracker
2
+
3
+ **Last reconciled:** 2026-07-13
4
+ **Authority:** This file is the granular program tracker. The work-order
5
+ documents define the contract and acceptance criteria; neither test output nor
6
+ a model claim may check an item without the named evidence.
7
+
8
+ ## WO-01 — Exact outbound request
9
+
10
+ - [x] Capture a JSON-wire-equivalent outbound request snapshot before shadow
11
+ analysis.
12
+ - [x] Fingerprint the complete request payload, not selected message fields.
13
+ - [x] Account for messages, tools, memory prefix, output reserve, safety
14
+ reserve, `numCtx`/`num_ctx`, and `maxTokens`/`max_tokens`.
15
+ - [x] Persist/reuse a precompiled budget only after request-byte/hash match.
16
+ - [x] Keep normal and brute request envelope fields in parity.
17
+ - [x] Regression-test late prefix/schema/alias changes and snapshot isolation.
18
+ - [ ] Extract all request construction into one side-effect-free intermediate
19
+ model.
20
+ - [ ] Add a backend tokenizer adapter and estimator provenance to accounting.
21
+ - [ ] Include backend prompt-template normalization and think-mode reserve in
22
+ the final compiler budget.
23
+ - [x] Route active automatic compaction through the final request budget and
24
+ shared pre-backend boundary; the legacy `/compact` path is available only
25
+ under the explicit `OMNIUS_MEMORY_COMPILER_MODE=shadow` rollback mode.
26
+ - [ ] Remove legacy post-assembly accounting after a parity replay proves it
27
+ redundant.
28
+
29
+ ## WO-02 — Typed memory fabric
30
+
31
+ - [x] Define authority, task, claim, artifact, extract, action, tool result,
32
+ mutation, verifier, outcome, semantic, procedural, and orientation records.
33
+ - [x] Persist epoch, authority/trust class, provenance, revision/range,
34
+ fidelity, content hash, validity interval, and metadata.
35
+ - [x] Create append-only SQLite records, transactions, dependencies, and
36
+ working-set snapshots.
37
+ - [x] Add record and transaction idempotency/conflict protection.
38
+ - [x] Reject non-user/non-system `trusted_instruction` records.
39
+ - [x] Reject unsupported semantic facts.
40
+ - [x] Store raw user intent and substantive progress independently of read and
41
+ diagnostic noise.
42
+ - [x] Model source supersession without deleting historical records.
43
+ - [ ] Add a dual-read adapter for every older memory/event store.
44
+ - [ ] Migrate historical evidence/run-event ingestion through the adapter.
45
+ - [ ] Add a deterministic replay of a migrated real event stream.
46
+
47
+ ## WO-03 — Dependency graph and working set
48
+
49
+ - [x] Record action → tool call → result transactions from runner tool calls.
50
+ - [x] Attach canonical file artifacts once; store a compact tool-result receipt.
51
+ - [x] Attach mutations and shell verifier observations to the transaction.
52
+ - [x] Close materialization over required graph dependencies.
53
+ - [x] Close materialization over entire intersecting transactions.
54
+ - [x] Preserve a transaction even when its presentation cap is lower than its
55
+ atomic record count.
56
+ - [x] Persist active-root snapshots and graph explanations in request audits.
57
+ - [ ] Link todo/workboard state and every legacy evidence event into the graph.
58
+ - [ ] Rank open requirement/current action/unresolved claim/verifier state,
59
+ rather than only record kind and recency.
60
+ - [x] Render a production model-visible request projection only from a
61
+ validated v2 plan; the durable graph and transcript remain untouched.
62
+
63
+ ## WO-04 — Inference memory compiler
64
+
65
+ - [x] Send isolated compiler candidates as explicitly untrusted data.
66
+ - [x] Validate full candidate classification, IDs, authority boundary,
67
+ confidence, request fingerprint, graph cut, and transaction integrity.
68
+ - [x] Cache unchanged decisions by exact final-request fingerprint.
69
+ - [x] Make malformed/overlarge/incomplete analysis a `hold`, never deletion.
70
+ - [x] Run the compiler in non-blocking shadow mode and audit its decision.
71
+ - [x] Test prompt-like tool data, stale fingerprints, duplicate classifications,
72
+ active evidence, partial transaction, and 40%-headroom holds.
73
+ - [x] Add an exact-request compatibility projection from every mutable
74
+ model-visible string body to a typed, provenance-bound ledger candidate.
75
+ - [x] Bound and source-link model-visible orientation separately from audit
76
+ payloads; audit receipts contain locators and justifications, never bodies.
77
+ - [ ] Add an independent second opinion for ambiguous/high-risk shadow deltas.
78
+ - [x] Apply accepted v2 plans atomically only after inference, graph/artifact,
79
+ and exact post-materialization 45–52% budget validation pass.
80
+ - [x] Disable every legacy heuristic/message-summary path while active; any
81
+ failed, missing, or invalid inference plan holds the complete request.
82
+
83
+ ## WO-05 — Artifact fidelity and materialization
84
+
85
+ - [x] Keep normal/explicit full reads canonical and hash/range-addressable.
86
+ - [x] Admit full reads using actual headroom rather than a fixed percentage.
87
+ - [x] Preserve an admitted full body for its turn instead of generic output
88
+ folding.
89
+ - [x] Retain a completed extract as derived evidence while canonical full
90
+ source remains recoverable.
91
+ - [x] Keep partial extracts from hiding full source or satisfying coverage.
92
+ - [x] Honor declared requirement search terms and credit an overlapping exact
93
+ source anchor to every requirement it independently proves.
94
+ - [x] Treat output/line/capture-capped grep output as success with partial
95
+ provenance and deterministic narrowing.
96
+ - [x] Supersede changed canonical source revisions and permit a justified reread.
97
+ - [x] Enforce a closed A100-class hardware preflight in the live 20-file
98
+ harness.
99
+ - [x] Capture a successful 20-file `robit/ornith:35b` A100 report: 20/20
100
+ complete two-requirement contracts, no unresolved coverage, 20 isolated
101
+ requests, and a 6,080-character largest branch prompt.
102
+ - [x] Persist an idempotent materialization record for every rendered full
103
+ source body, naming the active action and canonical artifact revision.
104
+ - [x] Record extractor kind/model/confidence/search-round provenance as typed
105
+ metadata on every derived artifact record.
106
+ - [x] Persist canonical model-visible tool artifacts in an immutable SHA-256
107
+ store and expose a range-safe `artifact_read` resolver for retained refs.
108
+
109
+ ## WO-06 — Temporal hybrid retrieval
110
+
111
+ - [x] Enforce epoch, typed task/action/claim roots, authority boundaries, and
112
+ token/character limits.
113
+ - [x] Implement exact path/revision/range filtering before lexical scoring.
114
+ - [x] Add temporal `asOf` filtering and artifact supersession handling.
115
+ - [x] Traverse active-work dependencies bidirectionally with six-hop/128-node
116
+ bounds and truncation disclosure.
117
+ - [x] Use deterministic ranking and one-materialized-body-per-source diversity.
118
+ - [x] Surface active refutation and explicit missing/over-budget abstention.
119
+ - [ ] Integrate a versioned vector/semantic index rather than host callback
120
+ scoring only.
121
+ - [ ] Join retrieval across legacy memory stores through the WO-02 adapter.
122
+ - [ ] Measure next-action evidence recall/precision against redacted traces.
123
+
124
+ ## WO-07 — Evaluation
125
+
126
+ - [x] Create a versioned deterministic graph-memory replay report with stable
127
+ scenario/reason/report fingerprints.
128
+ - [x] Exercise active source, transaction, supersession, partial extract,
129
+ `/nothink` tool data, stale fingerprint, duplicate, confidence, and headroom
130
+ adversarial cases.
131
+ - [x] Emit candidate/retention/archive/request/graph/transaction metrics.
132
+ - [x] Add a closed-whitelist real-trace importer that emits text-free,
133
+ salted-pseudonym behavioral fixtures and rejects raw/unknown fields.
134
+ - [ ] Define a redacted real-trace ingestion schema and source-body redactor.
135
+ - [ ] Import representative MyActuator and Omnius traces with consented,
136
+ non-source-sensitive fixtures.
137
+ - [ ] Run current/no-compaction/summary/selector/compiler baselines over the
138
+ same trace corpus.
139
+ - [ ] Measure stale use, duplicate reads, false-not-found, repeated action
140
+ signatures, completion, latency, and cost.
141
+ - [ ] Add a natural mid-action no-mutation steering replay with first-affected
142
+ action assertion.
143
+ - [x] Complete the live large-model 20-file run on approved A100 hardware.
144
+ - [ ] Run approved small and medium model tiers with identical contracts.
145
+ - [ ] Publish objective promotion thresholds from the comparison report.
146
+
147
+ ## WO-08 — Rollout and removal
148
+
149
+ - [x] Make strict v2 request compilation the default. `shadow` is an explicit
150
+ rollback/diagnostic mode; active inference failure holds instead of reviving
151
+ a heuristic deletion path.
152
+ - [x] Emit exact request budget and shadow decision into request-dump audit.
153
+ - [x] Emit body-free applied/hold/rejected plan receipts with classifications,
154
+ artifact locators, rationale, and pre/post request fingerprints into dumps
155
+ and expandable TUI compaction audit boxes.
156
+ - [ ] Define explicit flags for ledger dual-write, graph materialization,
157
+ shadow, active compiler, and second opinion.
158
+ - [ ] Render the full lifecycle in TUI/session telemetry: budget → working set
159
+ → delta → hold/apply → first affected action.
160
+ - [ ] Define authority/fidelity/tool-freedom promotion thresholds.
161
+ - [ ] Canary by session and model tier with one-click rollback to untouched
162
+ history.
163
+ - [ ] Exercise and document rollback.
164
+ - [ ] Delete synthetic recap, silent rehydration, stale-controller, and
165
+ heuristic fallback paths only after canary acceptance.
166
+ - [ ] Remove obsolete flags/tests/docs in the same release.
167
+
168
+ ## Promotion status
169
+
170
+ - [x] Foundation code typechecks and focused regression suite passes.
171
+ - [ ] Deterministic replay meets published comparison thresholds.
172
+ - [x] Approved large-model live harness report is captured.
173
+ - [ ] Approved small/medium/large comparative runs are captured.
174
+ - [ ] Shadow telemetry is reviewed on real redacted traces.
175
+ - [ ] Canary and rollback drill are accepted.
176
+ - [ ] Legacy deletion is complete.
177
+
178
+ Until every promotion item is checked, this program is **not deployable** and
179
+ must not be represented as replacing the legacy compaction path.
@@ -0,0 +1,79 @@
1
+ # WO-01 — Exact Request and Budget Compiler
2
+
3
+ **Status:** in progress
4
+ **Primary modules:** `packages/orchestrator/src/agenticRunner.ts`,
5
+ `contextWindowDump.ts`, `context-compiler.ts`, `context-fabric.ts`
6
+ **Depends on:** none
7
+
8
+ ## Problem
9
+
10
+ The old compaction path estimated only durable history. The actual outgoing
11
+ request subsequently appended system text, tool schemas, dynamic controller
12
+ frames, active steering, evidence, and output reservation. A history that
13
+ appeared safe could therefore send an overfull or low-signal request; the
14
+ inverse caused needless compaction.
15
+
16
+ ## Contract
17
+
18
+ Create one pure `compileOutboundRequestBudget()` path that receives the exact
19
+ request payload and returns a reproducible ledger:
20
+
21
+ ```ts
22
+ interface OutboundRequestBudget {
23
+ requestFingerprint: string;
24
+ modelContextTokens: number;
25
+ stableSystemTokens: number;
26
+ toolSchemaTokens: number;
27
+ authorityTokens: number;
28
+ taskAndWorkingSetTokens: number;
29
+ evidenceTokens: number;
30
+ historyTokens: number;
31
+ outputReservationTokens: number;
32
+ safetyMarginTokens: number;
33
+ totalInputTokens: number;
34
+ projectedTotalTokens: number;
35
+ freeTokens: number;
36
+ freeRatio: number;
37
+ compactionEligible: boolean;
38
+ }
39
+ ```
40
+
41
+ `compactionEligible` is false while `freeRatio >= 0.40`. It must be based on
42
+ the serialized request, with a conservative estimator used only where the
43
+ backend tokenizer is unavailable. The fingerprint covers all materialized
44
+ segments and their hashes, not timestamps.
45
+
46
+ ## Todos
47
+
48
+ - [ ] Extract request assembly into a side-effect-free intermediate request
49
+ model, preserving current request behavior.
50
+ - [x] Count each rendered segment separately, including tool schemas and output
51
+ reservation.
52
+ - [ ] Add a tokenizer adapter with deterministic conservative fallback and
53
+ expose estimator provenance in dumps.
54
+ - [ ] Make automatic and manual compaction consume this budget only.
55
+ - [x] Use request fingerprints to suppress duplicate analysis while all inputs
56
+ are identical.
57
+ - [ ] Extend context-window dump and TUI compaction box with the segment ledger,
58
+ projected headroom, threshold decision, and fingerprint.
59
+ - [ ] Remove duplicated post-assembly accounting paths after parity tests pass.
60
+
61
+ ## Acceptance tests
62
+
63
+ - A large tool schema or dynamic frame changes eligibility even if `messages[]`
64
+ does not change.
65
+ - At 40.0% or more free real context, both auto and manual compaction hold.
66
+ - At 39.9% free context, the compiler can request compaction but does not alter
67
+ agent/tool permissions.
68
+ - Two byte-equivalent requests produce the same fingerprint and one analyst
69
+ invocation.
70
+ - A changed active steering slot, file hash, tool schema, or output reservation
71
+ produces a new fingerprint.
72
+ - The dump totals reconcile to the exact request estimate within documented
73
+ tokenizer rounding.
74
+
75
+ ## Definition of done
76
+
77
+ Focused unit and integration tests pass, the orchestrator typecheck passes, and
78
+ one fixture demonstrates a previous history-only false decision corrected by
79
+ final-request accounting.
@@ -0,0 +1,65 @@
1
+ # WO-02 — Typed Memory Fabric and Immutable Event Ledger
2
+
3
+ **Status:** in progress
4
+ **Primary modules:** `packages/memory/src/*`, `packages/schemas/src/memory.ts`,
5
+ `packages/orchestrator/src/evidenceLedger.ts`, `runEvents.ts`
6
+ **Depends on:** WO-01
7
+
8
+ ## Problem
9
+
10
+ Conversation messages currently carry incompatible roles: authority,
11
+ observations, evidence, plans, outcomes, and stale controller prose. A summary
12
+ can accidentally change their meaning or authority.
13
+
14
+ ## Contract
15
+
16
+ Introduce a canonical envelope shared by persisted memory classes:
17
+
18
+ ```ts
19
+ interface MemoryRecord<T> {
20
+ id: string;
21
+ kind: "authority" | "task" | "artifact" | "action" | "episode" |
22
+ "semantic" | "procedural";
23
+ taskEpoch: string;
24
+ authority: "system" | "user" | "agent" | "tool-data" | "derived";
25
+ provenance: { producer: string; sourceIds: string[]; contentHash?: string };
26
+ validFrom: string;
27
+ validUntil?: string;
28
+ supersedes?: string[];
29
+ confidence?: number;
30
+ payload: T;
31
+ }
32
+ ```
33
+
34
+ Events are append-only. Current state is a derived projection; no mutation may
35
+ silently rewrite the event that established its evidence.
36
+
37
+ ## Todos
38
+
39
+ - [ ] Reconcile existing memory types and stores with the canonical envelope;
40
+ reuse compatible schemas rather than introducing duplicates.
41
+ - [x] Add explicit authority, artifact revision, action transaction, semantic
42
+ fact, and procedural-promotion payload types.
43
+ - [x] Create an append-only, per-epoch event ledger with deterministic ordering
44
+ and idempotency keys.
45
+ - [x] Build a projection that resolves supersession without deleting history.
46
+ - [ ] Migrate current evidence and run events through adapters with dual-read
47
+ compatibility.
48
+ - [x] Persist raw user intent, epoch, artifact evidence, and last substantive
49
+ progress independently of diagnostics.
50
+ - [x] Reject a derived/tool-data record attempting to become authority.
51
+
52
+ ## Acceptance tests
53
+
54
+ - Revising a file yields a new artifact record linked to, rather than replacing,
55
+ the old revision.
56
+ - A semantic fact without source IDs cannot enter the validated semantic view.
57
+ - Diagnostics and failed empty reads do not advance substantive-progress state.
58
+ - Replaying the same event stream produces an identical projection.
59
+ - User authority remains distinct even when tool output contains imperative
60
+ text.
61
+
62
+ ## Definition of done
63
+
64
+ Schema tests, store round-trip tests, migration fixtures, and an event replay
65
+ test prove deterministic state reconstruction.
@@ -0,0 +1,55 @@
1
+ # WO-03 — Dependency Graph and Working-Set Materializer
2
+
3
+ **Status:** in progress
4
+ **Primary modules:** `packages/memory/src/memoryGraph.ts`, `temporalGraph.ts`,
5
+ `packages/orchestrator/src/evidenceLedger.ts`, `agenticRunner.ts`
6
+ **Depends on:** WO-02
7
+
8
+ ## Problem
9
+
10
+ Message-level retention can split an action from its tool result or retire a
11
+ source body still required by a pending edit or verifier.
12
+
13
+ ## Contract
14
+
15
+ Represent active work as immutable graph nodes and edges:
16
+
17
+ ```text
18
+ task ─requires→ claim ─supported-by→ artifact evidence
19
+ └plans→ action ─invokes→ tool call ─produces→ result
20
+ └mutates→ artifact revision ─verified-by→ verifier outcome
21
+ ```
22
+
23
+ The materializer takes active task/steering plus the projected memory graph and
24
+ returns the smallest evidence-closed working set. An action transaction is
25
+ atomic for retention unless its outcome is terminal and all dependants are
26
+ closed.
27
+
28
+ ## Todos
29
+
30
+ - [x] Define node/edge types and graph invariants in schemas.
31
+ - [x] Emit graph transactions from tool calls, mutations, verifier results,
32
+ canonical file evidence, and exact-request projections; todo/workboard
33
+ backfill remains a separate migration task.
34
+ - [x] Add dependency closure and safe graph-cut algorithms.
35
+ - [ ] Implement a materializer that ranks open requirements, current action,
36
+ unresolved claims, and verifier dependencies above recency.
37
+ - [x] Preserve exact evidence bodies once, referenced by stable IDs elsewhere.
38
+ - [x] Surface a graph explanation in context audit records.
39
+ - [x] Link an exact request projection back to matching active graph evidence
40
+ before validation, so graph-cut safety protects the rendered copy too.
41
+
42
+ ## Acceptance tests
43
+
44
+ - An assistant tool call and each matching tool result cannot be split.
45
+ - An exact file range remains materialized while an open edit or verifier cites
46
+ it.
47
+ - A completed, unreferenced discovery transaction can archive without changing
48
+ the active set.
49
+ - A failed verifier remains eligible for rerun after a dependent source revision
50
+ changes; no verifier lock is introduced.
51
+
52
+ ## Definition of done
53
+
54
+ Graph invariants, closure tests, and a MyActuator-derived fixture pass without
55
+ reintroducing historical controller recaps.
@@ -0,0 +1,67 @@
1
+ # WO-04 — Inference-Driven Memory Compiler
2
+
3
+ **Status:** in progress
4
+ **Primary modules:** `packages/orchestrator/src/compaction-analyst.ts`,
5
+ `agenticRunner.ts`, `contextWindowDump.ts`
6
+ **Depends on:** WO-01, WO-03
7
+
8
+ ## Problem
9
+
10
+ An inference selector that chooses individual transcript messages is safer than
11
+ heuristics but is still not a memory compiler: it lacks graph constraints,
12
+ state-signature deduplication, version awareness, and a meaningful `hold`
13
+ contract.
14
+
15
+ ## Contract
16
+
17
+ The isolated worker sees the real candidate context, typed as untrusted data,
18
+ and returns a schema-validated `MemoryDelta`. It is read-only, cannot write
19
+ model-visible history directly, and cannot deny main-agent tools.
20
+
21
+ ```ts
22
+ interface MemoryDelta {
23
+ decision: "compact" | "hold";
24
+ requestFingerprint: string;
25
+ retain: string[];
26
+ archive: Array<{ id: string; reason: "completed" | "superseded" | "noise" }>;
27
+ supersede: Array<{ old: string; next: string }>;
28
+ unresolvedClaims: string[];
29
+ orientation?: string;
30
+ coverage: { allCandidatesClassified: boolean };
31
+ confidence: number;
32
+ }
33
+ ```
34
+
35
+ ## Todos
36
+
37
+ - [x] Project every mutable exact-request message body into a provenance-bound
38
+ graph record while retaining the legacy analyst only for explicit shadow
39
+ rollback tests.
40
+ - [x] Supply the complete outbound-request budget and working-set query.
41
+ - [x] Require coverage of all eligible candidates and validate hashes, graph
42
+ closure, authority, and budget before apply.
43
+ - [x] Cache `hold` and `compact` decisions by request fingerprint; invalidate
44
+ only on task, authority, artifact, action, or budget change.
45
+ - [x] Keep model-visible orientation bounded, source-linked, and newly derived;
46
+ retain full analyst inputs/outputs only in audit logs.
47
+ - [ ] Route ambiguous/high-risk deltas to a second opinion in shadow mode; a
48
+ disagreement holds compaction, never blocks the main agent.
49
+ - [x] In active mode, disable heuristic fallback behavior; invalid/missing
50
+ inference keeps the complete exact request unchanged. Legacy behavior exists
51
+ only behind explicit shadow rollback while canary evidence is gathered.
52
+
53
+ ## Acceptance tests
54
+
55
+ - Invalid IDs, missing classifications, broken dependencies, unknown hashes, or
56
+ over-budget output produce `hold` with no memory deletion.
57
+ - Identical calls reuse a cached result; a changed file hash invokes analysis.
58
+ - Prompt-like content in source/tool output cannot change the analyst's
59
+ authority contract.
60
+ - The worker may defer compaction but the parent still permits reads, edits,
61
+ exploration, and verification.
62
+
63
+ ## Definition of done
64
+
65
+ Contract, cache, fault-injection, prompt-injection, active materialization,
66
+ and audit tests pass. Shadow remains available for rollback/replay comparison;
67
+ active mode is strictly hold-on-failure and never blocks agent tools.