omnius 1.0.591 → 1.0.592
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.aiwg/addons/omnius-docs/README.md +15 -1
- package/.aiwg/addons/omnius-docs/manifest.json +28 -68
- package/.aiwg/addons/omnius-docs/skills/agent-failure-recovery/SKILL.md +2 -1
- package/.aiwg/addons/omnius-docs/skills/browser-interaction-validation/SKILL.md +2 -1
- package/.aiwg/addons/omnius-docs/skills/evidence-directed-delivery/SKILL.md +2 -1
- package/.aiwg/addons/omnius-docs/skills/hardware-evidence-audit/SKILL.md +2 -1
- package/.aiwg/addons/omnius-docs/skills/omnius-docs/SKILL.md +17 -7
- package/.aiwg/addons/omnius-docs/skills/omnius-inference-docs/SKILL.md +27 -0
- package/.aiwg/addons/omnius-docs/skills/omnius-integration-docs/SKILL.md +21 -0
- package/.aiwg/addons/omnius-docs/skills/omnius-ops-docs/SKILL.md +2 -0
- package/.aiwg/addons/omnius-docs/skills/omnius-realtime-docs/SKILL.md +2 -0
- package/.aiwg/addons/omnius-docs/skills/omnius-sponsor-docs/SKILL.md +2 -0
- package/.aiwg/addons/omnius-docs/skills/omnius-telegram-docs/SKILL.md +2 -0
- package/.aiwg/addons/omnius-docs/skills/omnius-tools-docs/SKILL.md +23 -0
- package/.aiwg/addons/omnius-docs/skills/omnius-version-compatibility-docs/SKILL.md +23 -0
- package/.aiwg/addons/omnius-docs/skills/runtime-provenance-audit/SKILL.md +2 -1
- package/.aiwg/addons/omnius-docs/skills/secrets-and-config-audit/SKILL.md +2 -1
- package/.aiwg/addons/omnius-docs/skills/test-surface-audit/SKILL.md +2 -1
- package/.aiwg/addons/omnius-docs/skills/workspace-reality-audit/SKILL.md +2 -1
- package/.aiwg/addons/omnius-rest-docs/README.md +3 -0
- package/.aiwg/addons/omnius-rest-docs/manifest.json +27 -20
- package/.aiwg/addons/omnius-rest-docs/skills/omnius-rest-docs/SKILL.md +9 -5
- package/README.md +36 -0
- package/dist/discovery.d.ts +50 -0
- package/dist/index.js +5975 -4021
- package/dist/library.d.ts +7 -0
- package/dist/library.js +950 -0
- package/dist/postinstall-daemon.cjs +18 -0
- package/dist/providerRegistry.d.ts +80 -0
- package/dist/service-version.d.ts +35 -0
- package/docs/.vitepress/config.mts +8 -0
- package/docs/DISCOVERY.json +20224 -0
- package/docs/DISCOVERY.md +648 -0
- package/docs/HANDOFF-crl-encoder-decoder-fix.md +129 -0
- package/docs/agent-memory/INDEX.md +9 -4
- package/docs/agent-memory/index.md +7 -0
- package/docs/concept-relational-language.md +869 -0
- package/docs/context-management-medium-models-proposal.md +449 -0
- package/docs/dedup-false-positive-meta-analysis.md +96 -0
- package/docs/discovery/catalog-overrides.json +724 -0
- package/docs/duplicate-calls-root-cause-analysis.md +91 -0
- package/docs/duplicate-calls-root-cause-deep.md +155 -0
- package/docs/ephemeral-skill-pack-small-context.md +57 -0
- package/docs/explorations/context-window-todo-association.md +156 -0
- package/docs/explorations/todo-association-verify.json +30 -0
- package/docs/explorations/verification-ledger.json +45 -0
- package/docs/explorations/verify-todo-association.sh +30 -0
- package/docs/flowstate.md +806 -0
- package/docs/getting-started/install.md +24 -0
- package/docs/getting-started/model-providers.md +13 -0
- package/docs/guides/agent-integration.md +87 -0
- package/docs/guides/bring-your-own-inference.md +126 -0
- package/docs/guides/tools-and-web-search.md +95 -0
- package/docs/index.md +14 -0
- package/docs/longhaul-35b-workorders.md +496 -0
- package/docs/memory-integration-analysis.md +303 -0
- package/docs/model-capability-awareness-and-multimodal-memory-root-fix.md +799 -0
- package/docs/multimodal-identity-memory-implementation.md +76 -0
- package/docs/omnius-self-edit-eval-2026-06-10.md +169 -0
- package/docs/opencode-agentic-loop-comparison.md +290 -0
- package/docs/operations/security-and-remote-access.md +2 -2
- package/docs/operations/version-compatibility.md +63 -0
- package/docs/proposals/git-progress-tracking-strategy.md +289 -0
- package/docs/proposals/opencode-modules/backendAdapter.ts +443 -0
- package/docs/proposals/opencode-modules/childSession.ts +288 -0
- package/docs/proposals/opencode-modules/compactionAgent.ts +101 -0
- package/docs/proposals/opencode-modules/orchestrator.ts +387 -0
- package/docs/proposals/opencode-modules/runner.ts +258 -0
- package/docs/reference/auth-map.md +87 -196
- package/docs/reference/configuration.md +27 -0
- package/docs/reference/rest-api.md +7 -0
- package/docs/reference/slash-commands.md +125 -2
- package/docs/research/_archived/README.md +18 -0
- package/docs/research/_archived/context_window_attention_model.py +418 -0
- package/docs/research/_archived/context_window_attention_spec.md +55 -0
- package/docs/research/_archived/context_window_attention_weights.json +68 -0
- package/docs/research/k-splanifolds.pdf +0 -0
- package/docs/research/personality-verbosity-control.md +293 -0
- package/docs/rest/INDEX.md +7 -0
- package/docs/rest/QUICKREF.md +18 -0
- package/docs/rest/REST-DOCS-MANIFEST.json +1 -0
- package/docs/rest/auth-and-scopes.md +7 -1
- package/docs/rest/endpoints/discovery.md +44 -0
- package/docs/rest/endpoints/events.md +5 -0
- package/docs/rest/endpoints/tools.md +9 -0
- package/docs/reviews/adversary-system-review.md +42 -0
- package/docs/sana-and-video-generation-integration-plan.md +712 -0
- package/docs/session-diary-llm-training-analysis.md +218 -0
- package/docs/telegram-dmn-curiosity-outreach-scaffold.md +91 -0
- package/docs/telegram-mid-horizon-download-loop-handoff.md +468 -0
- package/docs/telegram-reflection-corpus-integration-plan.md +306 -0
- package/docs/telegram-unified-tooling-architecture.md +332 -0
- package/docs/threat-model.md +868 -0
- package/docs/trajectory-grounding.md +160 -0
- package/docs/voice-flow-architecture.md +489 -0
- package/docs/work-orders/WO-AM-GAPS.md +638 -0
- package/docs/work-orders/daemon-hud-ui-overhaul.md +82 -0
- package/docs/work-orders/hermes-architecture-deltas/01-public-scrutiny-provenance-control/INDEX.md +21 -0
- package/docs/work-orders/hermes-architecture-deltas/01-public-scrutiny-provenance-control/WORKORDER.md +225 -0
- package/docs/work-orders/hermes-architecture-deltas/02-context-engine-plugin-boundary/INDEX.md +20 -0
- package/docs/work-orders/hermes-architecture-deltas/02-context-engine-plugin-boundary/WORKORDER.md +198 -0
- package/docs/work-orders/hermes-architecture-deltas/03-typed-gateway-event-stream/INDEX.md +19 -0
- package/docs/work-orders/hermes-architecture-deltas/03-typed-gateway-event-stream/WORKORDER.md +172 -0
- package/docs/work-orders/hermes-architecture-deltas/04-task-local-gateway-context/INDEX.md +19 -0
- package/docs/work-orders/hermes-architecture-deltas/04-task-local-gateway-context/WORKORDER.md +169 -0
- package/docs/work-orders/hermes-architecture-deltas/05-process-lifecycle-monitoring-notifications/INDEX.md +22 -0
- package/docs/work-orders/hermes-architecture-deltas/05-process-lifecycle-monitoring-notifications/WORKORDER.md +189 -0
- package/docs/work-orders/hermes-architecture-deltas/06-vision-evidence-routing-ladder/INDEX.md +22 -0
- package/docs/work-orders/hermes-architecture-deltas/06-vision-evidence-routing-ladder/WORKORDER.md +199 -0
- package/docs/work-orders/hermes-architecture-deltas/07-durable-multi-agent-kanban/INDEX.md +20 -0
- package/docs/work-orders/hermes-architecture-deltas/07-durable-multi-agent-kanban/WORKORDER.md +174 -0
- package/docs/work-orders/hermes-architecture-deltas/08-completion-critic-reconciliation-ledger/INDEX.md +22 -0
- package/docs/work-orders/hermes-architecture-deltas/08-completion-critic-reconciliation-ledger/WORKORDER.md +226 -0
- package/docs/work-orders/hermes-architecture-deltas/INDEX.md +38 -0
- package/docs/work-orders/omnius-context-engineering-behavior-fixes.md +281 -0
- package/docs/work-orders/telegram-dropbear-context-rca-workorder.md +202 -0
- package/docs/work-orders/world-class-memory-compiler/README.md +162 -0
- package/docs/work-orders/world-class-memory-compiler/TRACKER.md +179 -0
- package/docs/work-orders/world-class-memory-compiler/WO-01-exact-request-budget.md +79 -0
- package/docs/work-orders/world-class-memory-compiler/WO-02-typed-memory-fabric.md +65 -0
- package/docs/work-orders/world-class-memory-compiler/WO-03-dependency-working-set.md +55 -0
- package/docs/work-orders/world-class-memory-compiler/WO-04-inference-memory-compiler.md +67 -0
- package/docs/work-orders/world-class-memory-compiler/WO-05-artifact-fidelity-materialization.md +72 -0
- package/docs/work-orders/world-class-memory-compiler/WO-06-temporal-hybrid-retrieval.md +49 -0
- package/docs/work-orders/world-class-memory-compiler/WO-07-evaluation-harness.md +45 -0
- package/docs/work-orders/world-class-memory-compiler/WO-08-rollout-legacy-removal.md +45 -0
- package/docs/x402-remote-inference-plan.md +323 -0
- package/npm-shrinkwrap.json +108 -117
- package/package.json +7 -6
- package/templates/AGENTS.md +6 -0
- package/templates/OMNIUS.md +20 -0
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
# Duplicate Tool Calls — Root Cause Analysis
|
|
2
|
+
|
|
3
|
+
## The Symptom
|
|
4
|
+
|
|
5
|
+
The agent repeatedly calls the same tool with the same arguments (e.g., `file_read` on the same path, `grep_search` with the same pattern, `shell` with the same command). This wastes turns, burns context, and creates visible "looping" behavior.
|
|
6
|
+
|
|
7
|
+
## The Core Problem: Reactive Dedup, Not Preventive
|
|
8
|
+
|
|
9
|
+
The entire dedup infrastructure is **post-hoc interception** — the model has *already decided* to make the duplicate call before any dedup logic runs. The architecture looks like:
|
|
10
|
+
|
|
11
|
+
```
|
|
12
|
+
Model emits tool_call → Critic inspects → serve_cached / reject → (maybe) proactivePrune later
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
The model never sees the Critic's decision *before* it chooses to call. It only sees the *result* of a prior call in context, which doesn't prevent it from re-requesting the same information.
|
|
16
|
+
|
|
17
|
+
### Why the Model Re-calls
|
|
18
|
+
|
|
19
|
+
1. **Context bloat buries prior results.** By turn 15+, the model's context contains thousands of tokens of tool results. A `file_read` result from turn 3 is buried under 12 turns of other content. The model can't find it, so it calls again.
|
|
20
|
+
|
|
21
|
+
2. **No "what I already know" summary.** The context is a raw conversation transcript. There's no structured "known facts" section that the model can reference instead of re-reading files. The model treats the context as a chat log, not a knowledge base.
|
|
22
|
+
|
|
23
|
+
3. **proactivePrune makes it worse.** When `proactivePrune` replaces old results with `"[deduped — see turn N]"` or `"[file_read aged out, summary: ...]"`, it *removes the information the model needed* while telling it the info exists somewhere it can't access. The model's response: call the tool again to get the actual content.
|
|
24
|
+
|
|
25
|
+
4. **Critic's serve_cached is invisible to the model.** When the Critic serves a cached result, the model sees a tool result — it doesn't learn "I already asked this." Next turn, same context, same decision.
|
|
26
|
+
|
|
27
|
+
## The Structural Issues in Context Engineering
|
|
28
|
+
|
|
29
|
+
### 1. Conversation-as-Context is the Wrong Primitive
|
|
30
|
+
|
|
31
|
+
The context is a linear `ChatMessage[]` — an exact transcript of every turn. This is the fundamental design error. Research (AgentFold, RECOMP, MemGPT) shows that **progressive summarization** or **working memory** architectures outperform raw transcript for long-horizon tasks.
|
|
32
|
+
|
|
33
|
+
The agent should maintain a **working memory** (structured facts about the current task state) separate from the **episodic transcript** (raw conversation). The model should see the working memory as its primary context, with the transcript available but compressed.
|
|
34
|
+
|
|
35
|
+
### 2. No Semantic Index of Prior Tool Results
|
|
36
|
+
|
|
37
|
+
The `recentToolResults` map in the Critic is keyed by fingerprint (tool name + args hash). This is syntactic dedup — it catches exact re-calls but misses semantic duplicates (e.g., `file_read("foo.ts")` vs `file_read("foo.ts", offset=1, limit=50)` — same file, different args, same information need).
|
|
38
|
+
|
|
39
|
+
A semantic index would map **information goals** to **prior results**: "I need the contents of foo.ts" → "you already read it at turn 5, here's a summary." The current system can't do this.
|
|
40
|
+
|
|
41
|
+
### 3. The Dedup Sets Reset Per-Turn (Not Per-Task)
|
|
42
|
+
|
|
43
|
+
All the `_injectedThisTurn` Sets (`_errorGuidanceInjected`, `_reflectionsInjectedThisTurn`, `_verifyHintInjectedThisTurn`, etc.) reset every turn. This means:
|
|
44
|
+
- Turn N: model calls `file_read("foo.ts")` → result injected
|
|
45
|
+
- Turn N+1: model calls `file_read("foo.ts")` again → no dedup fires (the per-turn set was cleared)
|
|
46
|
+
|
|
47
|
+
The `dedupHitCount` in the Critic persists across turns, but it only triggers `serve_cached` — it doesn't *prevent* the model from deciding to call. The model still emits the tool call, wasting an LLM inference.
|
|
48
|
+
|
|
49
|
+
### 4. The Model Sees Its Own Redundancy Too Late
|
|
50
|
+
|
|
51
|
+
The Critic runs *after* the model has emitted a tool call. By then, the LLM inference cost is already spent. The model should see a **pre-call hint** like "you already read foo.ts at turn 5" *before* it generates the tool call. This would require injecting context about prior calls into the system prompt or last assistant message *before* the next LLM call.
|
|
52
|
+
|
|
53
|
+
## What Would Fix It
|
|
54
|
+
|
|
55
|
+
### Short-term (minimal changes)
|
|
56
|
+
|
|
57
|
+
1. **Inject a "recently read files" summary into the system prompt each turn.** Before each LLM call, scan the conversation for `file_read` results and inject a compact list: "Files already read: foo.ts (turn 5), bar.ts (turn 8)." This gives the model the information it needs to avoid re-reading.
|
|
58
|
+
|
|
59
|
+
2. **Don't age-out file_read results that are still relevant.** `proactivePrune` should check if the file was modified since last read (via git status or mtime) before replacing the result. If unchanged, keep the full result — the model needs it.
|
|
60
|
+
|
|
61
|
+
3. **Make `serve_cached` results visually distinct.** When the Critic serves a cached result, prefix it with `[CACHED — you already called this at turn N. Do not call again.]` so the model learns not to re-request.
|
|
62
|
+
|
|
63
|
+
### Medium-term (architectural)
|
|
64
|
+
|
|
65
|
+
4. **Working memory layer.** Maintain a structured "known facts" document that the model sees as part of its system context. Updated after each tool call. This is the MemGPT / Generative Agents approach.
|
|
66
|
+
|
|
67
|
+
5. **Pre-call context injection.** Before each LLM call, analyze the last few tool calls and inject warnings about likely duplicates: "You called file_read on foo.ts 2 turns ago. If you need the same content, reference that result instead of calling again."
|
|
68
|
+
|
|
69
|
+
6. **Semantic dedup.** Instead of fingerprint-based dedup, use embedding similarity to detect when a new tool call is semantically equivalent to a prior one (different args, same information goal).
|
|
70
|
+
|
|
71
|
+
### Long-term (fundamental)
|
|
72
|
+
|
|
73
|
+
7. **Replace conversation-as-context with state-as-context.** The model should see a task state object (current files, known issues, pending todos, recent errors) rather than a raw transcript. The transcript should be available for reference but not the primary context. This is the architecture shift from "chat agent" to "state machine agent."
|
|
74
|
+
|
|
75
|
+
## Evidence from the Codebase
|
|
76
|
+
|
|
77
|
+
The sheer number of REG-* patches (16, 17, 18, 26, 28, 31, 37, 38, 44, 45, 49b, 50) addressing symptoms of the same problem confirms this is a structural issue, not a surface bug. Each REG patch adds another per-turn Set or cooldown to suppress a specific duplicate/repetition pattern, but none address the root cause: **the model doesn't know what it already knows.**
|
|
78
|
+
|
|
79
|
+
The `proactivePrune` function (lines 2960-3040) is the clearest evidence — it exists solely to clean up the symptom (bloated context from duplicate calls) without preventing the cause (the model making duplicate calls in the first place).
|
|
80
|
+
|
|
81
|
+
## Summary
|
|
82
|
+
|
|
83
|
+
| Layer | Current | Needed |
|
|
84
|
+
|-------|---------|--------|
|
|
85
|
+
| Dedup timing | Post-hoc (Critic after call) | Pre-hoc (inject known-info before call) |
|
|
86
|
+
| Context model | Raw transcript | Working memory + compressed transcript |
|
|
87
|
+
| Dedup scope | Syntactic (exact fingerprint) | Semantic (information goal) |
|
|
88
|
+
| Prune strategy | Age-based removal | Relevance-based retention |
|
|
89
|
+
| Model awareness | Sees tool results, not patterns | Sees "what I already know" summary |
|
|
90
|
+
|
|
91
|
+
The duplicate calls aren't a bug — they're an emergent property of a context architecture that treats the LLM as a chat participant rather than a stateful agent.
|
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
# Duplicate Tool Calls — Deep Root Cause Analysis
|
|
2
|
+
|
|
3
|
+
## The Problem
|
|
4
|
+
|
|
5
|
+
The agent repeatedly calls the same tool with the same arguments right after having already called it. This is visible even in this agent's own conversation — `file_read` and `grep_search` calls get duplicated within 1-2 turns.
|
|
6
|
+
|
|
7
|
+
## Root Cause: The Model Decides to Re-call BEFORE Any Dedup Can Stop It
|
|
8
|
+
|
|
9
|
+
The fundamental issue is **temporal**: the model emits a tool call *before* any dedup logic runs. The architecture is:
|
|
10
|
+
|
|
11
|
+
```
|
|
12
|
+
LLM generates tool_call → Critic intercepts → serve_cached / reject → proactivePrune later
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
The model never sees "you already know this" *before* it decides to call. It only sees prior tool *results* buried in context, which it may not locate or trust.
|
|
16
|
+
|
|
17
|
+
## The Critic: Post-Hoc Interception That Can't Prevent Re-calls
|
|
18
|
+
|
|
19
|
+
**Code**: `packages/orchestrator/src/critic.ts` (lines 1-180)
|
|
20
|
+
|
|
21
|
+
The Critic (`evaluate()` function, line 137) is the central dedup decision module. It runs **after** the model has already emitted a tool call. Its decision priority:
|
|
22
|
+
|
|
23
|
+
1. **adversary guidance**: If the adversary flagged this fingerprint as redundant, attach a non-blocking critique and cached prior evidence; the requested tool call still executes.
|
|
24
|
+
2. **serve_cached** (line 166): If the tool is "read-like" AND the fingerprint exists in `recentToolResults` AND hit count < threshold → return cached result with a soft warning
|
|
25
|
+
3. **force_progress_block** (line 158): If hit count ≥ threshold (2 for shell, 3 for filesystem) → block the call entirely with a forced message
|
|
26
|
+
4. **pass** (line 179): Otherwise, let the call through
|
|
27
|
+
|
|
28
|
+
**Key insight**: Steps 2 and 3 both happen *after* the model already decided to call. The LLM inference cost is already spent. The model emitted `file_read("foo.ts")` — the Critic then says "you already have this" but the model never learns from this because:
|
|
29
|
+
|
|
30
|
+
- `serve_cached` returns the cached result as a tool result — the model sees it as a normal response, not a "don't call again" signal
|
|
31
|
+
- `force_progress_block` blocks the call but the model just sees an error message — it doesn't understand *why* it was blocked in context of its reasoning
|
|
32
|
+
|
|
33
|
+
The `recentToolResults` Map (line 63) and `dedupHitCount` Map (line 65) persist across turns, but they only affect the Critic's decision — they don't inject any information into the model's context *before* the next LLM call.
|
|
34
|
+
|
|
35
|
+
## Trigger 1: proactivePrune Removes Information the Model Still Needs
|
|
36
|
+
|
|
37
|
+
**Code**: `packages/orchestrator/src/agenticRunner.ts:2973-3049`
|
|
38
|
+
|
|
39
|
+
`proactivePrune` walks the message history and replaces old tool results with placeholders:
|
|
40
|
+
|
|
41
|
+
- `"[deduped — same call as turn N]"` (line 2981) — for exact duplicate calls
|
|
42
|
+
- `"[file_read aged out, summary: ...]"` (line 2982) — for file_read results older than 10 turns (line 2976: `AGED_FILE_READ_TURNS = 10`)
|
|
43
|
+
- `"[shell succeeded, output pruned — ...]"` (line 2983) — for shell results older than 5 turns (line 2977: `AGED_SHELL_TURNS = 5`)
|
|
44
|
+
|
|
45
|
+
**Why this causes re-calls**: When the model sees `"[file_read aged out, summary: ...]"`, it knows the file was read but can't see the actual content. The summary is typically truncated and insufficient for the model to reason about. The model's rational response: call `file_read` again to get the full content.
|
|
46
|
+
|
|
47
|
+
This is a **self-reinforcing loop**:
|
|
48
|
+
1. Model reads `foo.ts` → full content in context
|
|
49
|
+
2. 10 turns later, proactivePrune replaces it with `"[file_read aged out, summary: ...]"`
|
|
50
|
+
3. Model needs `foo.ts` content again → calls `file_read("foo.ts")` again
|
|
51
|
+
4. proactivePrune will eventually prune this too → loop repeats
|
|
52
|
+
|
|
53
|
+
## Trigger 2: No Pre-Call "What You Already Know" Injection
|
|
54
|
+
|
|
55
|
+
**Code**: The LLM call path in `agenticRunner.ts` composes the `ChatMessage[]` array and sends it directly to the model. There is no step that injects a summary of prior tool results *before* the model generates its next response.
|
|
56
|
+
|
|
57
|
+
The model sees:
|
|
58
|
+
- System prompt (static)
|
|
59
|
+
- Conversation transcript (raw `ChatMessage[]`)
|
|
60
|
+
- No structured "known facts" or "recently read files" section
|
|
61
|
+
|
|
62
|
+
Without a pre-call hint like "You already read foo.ts at turn 5 — content available in context", the model has no way to know it already has the information. It must scan the entire transcript to find prior results, which is unreliable for small models with long contexts.
|
|
63
|
+
|
|
64
|
+
## Trigger 3: Per-Turn Dedup Sets Reset, Allowing Cross-Turn Duplicates
|
|
65
|
+
|
|
66
|
+
**Code**: `agenticRunner.ts:1130-1171`
|
|
67
|
+
|
|
68
|
+
The dedup system uses per-turn Sets that reset each turn:
|
|
69
|
+
|
|
70
|
+
- `REG-37` (line 1130): per-turn dedup for verification-required hint
|
|
71
|
+
- `REG-38` (line 1134): per-turn dedup for artifact-inspection critique injection
|
|
72
|
+
- `REG-49b` (line 1171): per-loop-episode dedup flag for SSMA invocation
|
|
73
|
+
|
|
74
|
+
This means:
|
|
75
|
+
- Turn N: model calls `file_read("foo.ts")` → result injected, dedup Set records it
|
|
76
|
+
- Turn N+1: dedup Set is cleared → model calls `file_read("foo.ts")` again → no dedup fires
|
|
77
|
+
|
|
78
|
+
The `dedupHitCount` Map (line 5645) persists across turns, but it only triggers `serve_cached` in the Critic — it doesn't *prevent* the model from emitting the tool call. The LLM inference cost is already spent.
|
|
79
|
+
|
|
80
|
+
## Trigger 4: Fingerprint-Only Dedup Misses Semantic Duplicates
|
|
81
|
+
|
|
82
|
+
**Code**: `agenticRunner.ts:4489` (`_buildToolFingerprint`)
|
|
83
|
+
|
|
84
|
+
The fingerprint is built from tool name + canonical args. This catches exact re-calls but misses:
|
|
85
|
+
|
|
86
|
+
- `file_read("foo.ts")` vs `file_read("foo.ts", offset=1, limit=50)` — same file, different args, same information need
|
|
87
|
+
- `grep_search("pattern", path="src")` vs `grep_search("pattern", path="src", include="*.ts")` — same search, different filter
|
|
88
|
+
- `shell("cat foo.ts")` vs `file_read("foo.ts")` — different tool, same information goal
|
|
89
|
+
|
|
90
|
+
The `recentToolResults` Map (line 5635) is keyed by fingerprint, so semantically equivalent calls with different fingerprints are treated as distinct — no dedup fires.
|
|
91
|
+
|
|
92
|
+
## Why the 12+ REG-* Patches Don't Fix It
|
|
93
|
+
|
|
94
|
+
Each REG patch adds another per-turn Set, cooldown, or fingerprint check. These are **symptom suppressors** — they intercept duplicates *after* the model decides to call, but they don't address why the model decides to call in the first place.
|
|
95
|
+
|
|
96
|
+
| REG Patch | Location | What It Does | Why It Doesn't Fix Root Cause |
|
|
97
|
+
|-----------|----------|-------------|-------------------------------|
|
|
98
|
+
| REG-16 | line 5639 | Per-fingerprint dedup-hit counter, escalates to force_progress_block at ≥3 hits | Post-hoc: model already emitted the call |
|
|
99
|
+
| REG-17 | critic.ts:102 | Shell threshold = 2 hits before block | Post-hoc: only blocks after 2 wasted inferences |
|
|
100
|
+
| REG-18 | line 5691 | Stagnation window for variant-fatigue loops | Detects loops, doesn't prevent them |
|
|
101
|
+
| REG-26 | line 5820 | Per-turn reflection-injection dedup | Resets per-turn, misses cross-turn patterns |
|
|
102
|
+
| REG-31 | line 5874 | Positive completion signal injection | Doesn't address duplicate reads |
|
|
103
|
+
| REG-37 | line 1130 | Per-turn dedup for verification hint | Resets per-turn |
|
|
104
|
+
| REG-38 | line 1134 | Per-turn dedup for artifact-inspection | Resets per-turn |
|
|
105
|
+
| REG-49b | line 1171 | Per-loop-episode dedup for SSMA | Episode-scoped, not task-scoped |
|
|
106
|
+
|
|
107
|
+
## The Fix: Pre-Hoc Knowledge Injection
|
|
108
|
+
|
|
109
|
+
### Short-term (minimal changes)
|
|
110
|
+
|
|
111
|
+
1. **Inject a "recently read files" summary into the system prompt each turn.** Before each LLM call, scan the conversation for `file_read` results and inject: "Files already in context: foo.ts (turn 5), bar.ts (turn 8). Do not re-read these files."
|
|
112
|
+
|
|
113
|
+
2. **Don't age-out file_read results that are still relevant.** `proactivePrune` should check if the file was modified since last read (via git status or mtime) before replacing the result. If unchanged, keep the full result.
|
|
114
|
+
|
|
115
|
+
3. **Make cached results visually distinct.** When the Critic serves a cached result, prefix it with `[CACHED — you already called this at turn N. Do not call again.]`
|
|
116
|
+
|
|
117
|
+
### Medium-term (architectural)
|
|
118
|
+
|
|
119
|
+
4. **Working memory layer.** Maintain a structured "known facts" document that the model sees as part of its system context. Updated after each tool call. This is the MemGPT / Generative Agents approach.
|
|
120
|
+
|
|
121
|
+
5. **Pre-call context injection.** Before each LLM call, analyze the last few tool calls and inject warnings: "You called file_read on foo.ts 2 turns ago. If you need the same content, reference that result instead of calling again."
|
|
122
|
+
|
|
123
|
+
6. **Semantic dedup.** Instead of fingerprint-based dedup, use embedding similarity to detect when a new tool call is semantically equivalent to a prior one (different args, same information goal).
|
|
124
|
+
|
|
125
|
+
### Long-term (fundamental)
|
|
126
|
+
|
|
127
|
+
7. **Replace conversation-as-context with state-as-context.** The model should see a task state object (current files, known issues, pending todos, recent errors) rather than a raw transcript. The transcript should be available for reference but not the primary context.
|
|
128
|
+
|
|
129
|
+
## Evidence
|
|
130
|
+
|
|
131
|
+
| Symptom | Code Location | Mechanism |
|
|
132
|
+
|---------|--------------|-----------|
|
|
133
|
+
| Critic runs post-hoc | `critic.ts:137` (`evaluate()`) | Intercepts after model emits tool call |
|
|
134
|
+
| serve_cached is invisible to model | `critic.ts:166-174` | Returns cached result as normal tool result |
|
|
135
|
+
| force_progress_block is reactive | `critic.ts:158-164` | Blocks after ≥3 wasted inferences |
|
|
136
|
+
| proactivePrune removes content | `agenticRunner.ts:2973-3049` | Replaces old results with `[file_read aged out, summary: ...]` |
|
|
137
|
+
| Age thresholds too aggressive | `agenticRunner.ts:2976-2977` | 10 turns for files, 5 for shell |
|
|
138
|
+
| No pre-call knowledge injection | LLM call path in agenticRunner | No step injects "what you already know" before model generates |
|
|
139
|
+
| Per-turn dedup resets | `agenticRunner.ts:1130-1171` | REG-37/38/49b Sets clear each turn |
|
|
140
|
+
| Fingerprint-only dedup | `agenticRunner.ts:4489` | Exact match only, misses semantic duplicates |
|
|
141
|
+
| recentToolResults keyed by fingerprint | `agenticRunner.ts:5635-5638` | Same file different args = different entry |
|
|
142
|
+
| dedupHitCount persists but reactive | `agenticRunner.ts:5645` | Only triggers serve_cached, not prevention |
|
|
143
|
+
| 12+ REG patches | Various | All post-hoc interception, none pre-hoc prevention |
|
|
144
|
+
|
|
145
|
+
## Summary
|
|
146
|
+
|
|
147
|
+
The duplicate calls are an **emergent property of the conversation-as-context architecture**. The model doesn't know what it already knows because:
|
|
148
|
+
|
|
149
|
+
1. **The Critic intercepts after the call** — the model already decided to re-call before any dedup logic runs (critic.ts:137)
|
|
150
|
+
2. **Prior results get pruned away** — proactivePrune replaces content with summaries the model can't use (agenticRunner.ts:2973-3049)
|
|
151
|
+
3. **No one tells the model what it already has** — no pre-call injection of "known facts" before the LLM generates
|
|
152
|
+
4. **Dedup resets per-turn** — per-turn Sets allow cross-turn duplicates (agenticRunner.ts:1130-1171)
|
|
153
|
+
5. **Dedup is syntactic only** — fingerprint matching misses semantic duplicates (agenticRunner.ts:4489)
|
|
154
|
+
|
|
155
|
+
The fix is not more post-hoc interception — it's **pre-hoc knowledge injection**: tell the model what it already knows *before* it decides to call.
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
# Ephemeral Skill Pack Small-Context Plan
|
|
2
|
+
|
|
3
|
+
Date: 2026-05-15
|
|
4
|
+
|
|
5
|
+
## Problem
|
|
6
|
+
|
|
7
|
+
AIWG and Omnius skills are valuable, but full `SKILL.md` bodies are too large for
|
|
8
|
+
small and medium context windows. The main agent should see only a tiny,
|
|
9
|
+
task-scoped skill manifest, then delegate full skill unpacking to a sub-agent
|
|
10
|
+
that returns a compact extraction.
|
|
11
|
+
|
|
12
|
+
## Design
|
|
13
|
+
|
|
14
|
+
1. Build a run-scoped `ephemeral-skill-pack` from top-k task/skill matches.
|
|
15
|
+
2. Inject only skill names, sources, short descriptions, first trigger, score,
|
|
16
|
+
and explicit discard semantics into `dynamicContext`.
|
|
17
|
+
3. Give the main agent a `skill_extract` tool that loads full skill content out
|
|
18
|
+
of band and returns targeted guidance.
|
|
19
|
+
4. When available, `skill_extract` delegates the full `SKILL.md` body to a
|
|
20
|
+
sub-agent and asks it to extract only the parts needed for the current task.
|
|
21
|
+
5. If the sub-agent path is unavailable, fall back to deterministic section
|
|
22
|
+
extraction with a strict budget.
|
|
23
|
+
6. Keep `skill_execute` for explicit full-skill execution, but prefer
|
|
24
|
+
`skill_extract` first on small and medium models.
|
|
25
|
+
|
|
26
|
+
## Context Placement
|
|
27
|
+
|
|
28
|
+
- TUI: append the manifest in `packages/cli/src/tui/interactive.ts` after the
|
|
29
|
+
existing dynamic context enrichments and before the `AgenticRunner` is built.
|
|
30
|
+
- Runner: the manifest rides inside `c_know` via the existing structured
|
|
31
|
+
context assembly in `packages/orchestrator/src/agenticRunner.ts`.
|
|
32
|
+
- Tool subset: include `skill_extract` in the runner's `skill` tool subset so
|
|
33
|
+
`tool_search("skill")` can promote it alongside `skill_list` and
|
|
34
|
+
`skill_execute`.
|
|
35
|
+
|
|
36
|
+
## Current Implementation Checklist
|
|
37
|
+
|
|
38
|
+
- [x] Add task/skill ranking helpers.
|
|
39
|
+
- [x] Add `buildEphemeralSkillPack(...)`.
|
|
40
|
+
- [x] Add deterministic `extractSkillForQuery(...)` fallback.
|
|
41
|
+
- [x] Add `skill_extract` tool with sub-agent extraction callback support.
|
|
42
|
+
- [x] Register `skill_extract` in the TUI tool list.
|
|
43
|
+
- [x] Wire `skill_extract` to the existing sub-agent tool callback.
|
|
44
|
+
- [x] Inject the ephemeral manifest into TUI `dynamicContext`.
|
|
45
|
+
- [x] Add tests for selection, manifest shape, and deterministic extraction.
|
|
46
|
+
- [x] Add REST `/v1/aiwg/expand` extraction mode.
|
|
47
|
+
- [x] Add Telegram action-agent skill-pack injection with public/private scope
|
|
48
|
+
boundaries.
|
|
49
|
+
- [x] Add post-compaction re-injection if long runs prove the manifest is lost
|
|
50
|
+
too early.
|
|
51
|
+
|
|
52
|
+
## Runtime Contract
|
|
53
|
+
|
|
54
|
+
The manifest is not durable memory. It must not be written to session handoffs,
|
|
55
|
+
memory cards, or long-term user profiles. If a skill produces a durable lesson,
|
|
56
|
+
the agent should save a separate concise lesson that cites the task outcome, not
|
|
57
|
+
the injected skill text.
|
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
# Exploration: Associating Context-Window Entries with Todo-Scoped Run Periods
|
|
2
|
+
|
|
3
|
+
**Goal (from user):** Understand how granular context-window entries (messages / tool
|
|
4
|
+
results) are associated with periods of an agentic run, and specifically whether they are
|
|
5
|
+
tied to a *todo item* so that, when that todo completes, its context content can be
|
|
6
|
+
compressed.
|
|
7
|
+
|
|
8
|
+
**Status:** Exploration complete. No code changes made — this is a design/feasibility
|
|
9
|
+
write-up. The conclusion is that **todo-scoped compression does not exist today**; the
|
|
10
|
+
mechanism must be added.
|
|
11
|
+
|
|
12
|
+
---
|
|
13
|
+
|
|
14
|
+
## 1. Current architecture (evidence-grounded)
|
|
15
|
+
|
|
16
|
+
### 1.1 The context window is a single flat message stream
|
|
17
|
+
- `packages/orchestrator/src/agenticRunner.ts` holds the live context as a flat
|
|
18
|
+
`ChatMessage[]` array (the `messages` passed to the model). Tool results, system
|
|
19
|
+
directives, and assistant turns are all appended to this one array
|
|
20
|
+
(`messages.push(...)` at many sites, e.g. L4518, L4954, L13307, L13333, L13812…).
|
|
21
|
+
- `ChatMessage` is defined in `context-compressor.ts` (L23-31):
|
|
22
|
+
```ts
|
|
23
|
+
export interface ChatMessage {
|
|
24
|
+
role: "system" | "user" | "assistant" | "tool";
|
|
25
|
+
content: string | null;
|
|
26
|
+
tool_calls?: Array<{ id: string; type: "function"; function: { name: string; arguments: string }; }>;
|
|
27
|
+
// … no todoId / spanId field exists
|
|
28
|
+
}
|
|
29
|
+
```
|
|
30
|
+
**There is no `todoId` / `spanId` on a message.** Messages are not tagged with which
|
|
31
|
+
todo was active when they were produced.
|
|
32
|
+
|
|
33
|
+
### 1.2 Compression is PHASE-based, not TODO-based
|
|
34
|
+
- `packages/orchestrator/src/contextTree.ts` — `ContextTree` is a "hierarchical
|
|
35
|
+
task-phase-aware context tree". It groups the message stream into **phases**:
|
|
36
|
+
`explore`, `plan`, `implement`, `verify`, `mixed` (contextTree.ts L10-15).
|
|
37
|
+
- It tracks `_phaseMessageStartIdx` (agenticRunner L2441-2445): on a phase transition,
|
|
38
|
+
the slice `messages[_phaseMessageStartIdx..now]` is captured as the OUTGOING phase's
|
|
39
|
+
owned slice via `tree.observePhaseMessages`, then the cursor advances.
|
|
40
|
+
- `tree.contractInactive(...)` (agenticRunner L10384) summarizes inactive phases;
|
|
41
|
+
`tree.archive(phaseName, archPath)` (L10413) writes them to disk.
|
|
42
|
+
- **So compression today is keyed to phase transitions, not todo completion.**
|
|
43
|
+
|
|
44
|
+
### 1.3 The compressor
|
|
45
|
+
- `packages/orchestrator/src/context-compressor.ts` —
|
|
46
|
+
`StructuredContextCompressor.generateSummary(compMessages)` (agenticRunner
|
|
47
|
+
L27465-27492) produces a `CompactionSummary` (goal, constraints, progress,
|
|
48
|
+
keyDecisions, relevantFiles, nextSteps, criticalContext).
|
|
49
|
+
- It is invoked from a budget/compaction path; **not** from any todo-completion hook.
|
|
50
|
+
|
|
51
|
+
### 1.4 Persistent task boundaries (`messageLog.ts`)
|
|
52
|
+
- `packages/orchestrator/src/messageLog.ts` — append-only JSONL
|
|
53
|
+
`.omnius/sessions/{id}.jsonl` with `task_boundary` markers + `task_summary` user
|
|
54
|
+
messages (analogue of Hannover's `SystemCompactBoundaryMessage`).
|
|
55
|
+
`loadBoundarySlice` returns the post-last-boundary slice.
|
|
56
|
+
- This is a *top-level task/goal* boundary, **not** a per-todo-item boundary.
|
|
57
|
+
|
|
58
|
+
### 1.5 Todo tracking
|
|
59
|
+
- `TodoReminderTodo` (agenticRunner L1771-1777):
|
|
60
|
+
```ts
|
|
61
|
+
export interface TodoReminderTodo {
|
|
62
|
+
id?: string;
|
|
63
|
+
content: string;
|
|
64
|
+
status: "pending" | "in_progress" | "completed" | "blocked";
|
|
65
|
+
parentId?: string;
|
|
66
|
+
blocker?: string;
|
|
67
|
+
}
|
|
68
|
+
```
|
|
69
|
+
- Todo state is maintained for the **reminder-gating** function
|
|
70
|
+
(`shouldInjectTodoReminder`, L1802-1843) — it checks "turns since last todo_write" to
|
|
71
|
+
nudge planning. It does **not** record start/end turns and does **not** associate
|
|
72
|
+
messages with todos.
|
|
73
|
+
- There is **no span map** linking a todo id → `[startTurn, endTurn]` → message indices.
|
|
74
|
+
- The dedicated context-intake module `context-fabric.ts` (473 lines — a typed "Context
|
|
75
|
+
Fabric" that emits one bounded frame before each model call) contains **0** references to
|
|
76
|
+
`todo` (`grep -c 'todo' = 0`). This confirms that even the purpose-built context layer
|
|
77
|
+
does **not** associate context entries with todos.
|
|
78
|
+
- `todoTruth.ts` exists as a separate todo-truth/verification module, but it is not wired
|
|
79
|
+
into the message stream or the compressor, so it does not provide todo→context span
|
|
80
|
+
tracking either. It is a candidate integration point for future work.
|
|
81
|
+
|
|
82
|
+
---
|
|
83
|
+
|
|
84
|
+
## 2. Gap analysis (the direct answer to the user's question)
|
|
85
|
+
|
|
86
|
+
**Today, granular context entries are NOT associated with a todo item.** They are
|
|
87
|
+
associated with:
|
|
88
|
+
- a flat global stream (no per-message tagging), and
|
|
89
|
+
- a *phase* (via `ContextTree`), which is a coarse activity classification
|
|
90
|
+
(`explore`/`plan`/`implement`/`verify`), **not** a todo.
|
|
91
|
+
|
|
92
|
+
Therefore *"when the todo is complete, compress that context"* is **not currently
|
|
93
|
+
possible** without first adding todo-scoped span tracking. The building blocks
|
|
94
|
+
(compressor, archiver, boundary markers) already exist — they are just wired to phases
|
|
95
|
+
and top-level tasks, not to todos.
|
|
96
|
+
|
|
97
|
+
---
|
|
98
|
+
|
|
99
|
+
## 3. Proposed design: todo-scoped compression-on-completion
|
|
100
|
+
|
|
101
|
+
Three additive pieces are needed:
|
|
102
|
+
|
|
103
|
+
### 3.1 Tag messages with the active todo id
|
|
104
|
+
Add `todoId?: string` to `ChatMessage` (or maintain a parallel
|
|
105
|
+
`Map<messageIndex, todoId>`). Set it whenever a `todo_write` marks a leaf `in_progress`,
|
|
106
|
+
and roll it to the next leaf when the active todo changes.
|
|
107
|
+
|
|
108
|
+
### 3.2 Track todo active spans
|
|
109
|
+
Maintain `Map<todoId, { startTurn, endTurn, startMsgIdx, endMsgIdx }>`. On `todo_write`:
|
|
110
|
+
- leaf → `in_progress`: open a span (record current turn + `messages.length`).
|
|
111
|
+
- leaf → `completed` / `blocked`: close the span (record end turn + `messages.length`).
|
|
112
|
+
|
|
113
|
+
### 3.3 Compress on completion
|
|
114
|
+
When a todo span closes, take `messages[span.startMsgIdx .. span.endMsgIdx]`, run
|
|
115
|
+
`StructuredContextCompressor.generateSummary(...)` (reuse the existing compressor), and
|
|
116
|
+
replace that slice with a single compact `task_summary` user message (mirroring
|
|
117
|
+
`messageLog.ts`'s boundary pattern). This reclaims tokens for finished work while keeping
|
|
118
|
+
a retrievable summary.
|
|
119
|
+
|
|
120
|
+
### 3.4 Reusable integration points (already in the codebase)
|
|
121
|
+
- `ContextTree.contractInactive` / `archive` — reuse the summarize + persist pattern.
|
|
122
|
+
- `StructuredContextCompressor.generateSummary` — reuse for the summary text.
|
|
123
|
+
- `messageLog.ts` boundary markers — reuse the `task_summary` convention for the
|
|
124
|
+
replacement message.
|
|
125
|
+
- `shouldInjectTodoReminder` (L1802) — natural hook to detect todo status transitions and
|
|
126
|
+
drive span open/close.
|
|
127
|
+
|
|
128
|
+
---
|
|
129
|
+
|
|
130
|
+
## 4. Open questions / risks
|
|
131
|
+
- **Nested todos (`parentId`):** compress only leaf spans, or roll up to the parent on
|
|
132
|
+
parent completion?
|
|
133
|
+
- **Re-reads after compression:** the recently-fixed file_read de-dupe removal means the
|
|
134
|
+
model can re-read files for live state. Compressed todo summaries must remain available
|
|
135
|
+
via `surfaceAnchors` / `messageLog` retrieval, not silently dropped.
|
|
136
|
+
- **Token accounting:** ensure the replacement summary is smaller than the compressed
|
|
137
|
+
slice (budget guard) so compression is net-positive.
|
|
138
|
+
- **Ordering:** spans can overlap if the model interleaves todos; the span map must handle
|
|
139
|
+
non-contiguous message indices (a todo's messages may be interleaved with another's).
|
|
140
|
+
|
|
141
|
+
---
|
|
142
|
+
|
|
143
|
+
## 5. Evidence index
|
|
144
|
+
| Fact | Location |
|
|
145
|
+
|------|----------|
|
|
146
|
+
| Flat `ChatMessage[]` context stream | agenticRunner.ts (many `messages.push` sites; L4518, L4954, L13307…) |
|
|
147
|
+
| `ChatMessage` has no todoId | context-compressor.ts L23-31 |
|
|
148
|
+
| Phase-based grouping | contextTree.ts L10-15, L260-313 |
|
|
149
|
+
| Phase message cursor | agenticRunner.ts L2441-2445 |
|
|
150
|
+
| Phase contract/archive | agenticRunner.ts L10384, L10413 |
|
|
151
|
+
| Compressor invocation | agenticRunner.ts L27465-27492 |
|
|
152
|
+
| Persistent task boundaries | messageLog.ts (JSONL + `task_summary`) |
|
|
153
|
+
| Todo interface | agenticRunner.ts L1771-1777 |
|
|
154
|
+
| Todo reminder gating (no span tracking) | agenticRunner.ts L1802-1843 |
|
|
155
|
+
| Dedicated context layer has 0 todo refs | context-fabric.ts (`grep -c 'todo' = 0`) |
|
|
156
|
+
| Separate todo-truth module, not wired to context | todoTruth.ts |
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
{
|
|
2
|
+
"exploration": "context-window-to-todo association for compression-on-completion",
|
|
3
|
+
"verified_at": "2026-07-07T21:40:00Z",
|
|
4
|
+
"conclusion": "Granular context-window entries are NOT associated with todo items. The live context is a flat ChatMessage[] stream (no todoId field) grouped only by coarse activity PHASE via ContextTree (explore/plan/implement/verify/mixed), compressed on phase/budget triggers via StructuredContextCompressor + messageLog task boundaries. Todo state (TodoReminderTodo) exists only for reminder-gating with no span/period tracking.",
|
|
5
|
+
"evidence": {
|
|
6
|
+
"e1_flat_message_stream": {
|
|
7
|
+
"command": "grep -c 'contextWindow\\|toolResults\\|messages.push\\|appendMessage' packages/orchestrator/src/agenticRunner.ts",
|
|
8
|
+
"result": 112,
|
|
9
|
+
"meaning": "112 sites append to the flat ChatMessage[] stream. ChatMessage (context-compressor.ts L23-31) has role/content/tool_calls/tool_call_id but NO todoId/spanId field."
|
|
10
|
+
},
|
|
11
|
+
"e2_todo_tracking": {
|
|
12
|
+
"command": "grep -c 'TodoItem\\|todoId\\|activeTodo\\|currentTodo\\|todo_write\\|todos' packages/orchestrator/src/agenticRunner.ts",
|
|
13
|
+
"result": 252,
|
|
14
|
+
"meaning": "TodoReminderTodo (agenticRunner.ts L1771-1777) + shouldInjectTodoReminder (L1802-1843) are reminder-gating only. No start/end turn or message-index span map exists."
|
|
15
|
+
},
|
|
16
|
+
"e3_compression": {
|
|
17
|
+
"command": "grep -rln 'compact\\|compress' packages/orchestrator/src",
|
|
18
|
+
"result_count": 36,
|
|
19
|
+
"meaning": "Compression is phase-based (ContextTree) + budget (StructuredContextCompressor.generateSummary at agenticRunner L27465) + messageLog task boundaries. Not todo-based. Files include context-compressor.ts, contextTree.ts, messageLog.ts, context-fabric.ts, todoTruth.ts."
|
|
20
|
+
},
|
|
21
|
+
"e4_context_fabric": {
|
|
22
|
+
"command": "grep -c 'todo' packages/orchestrator/src/context-fabric.ts",
|
|
23
|
+
"result": 0,
|
|
24
|
+
"meaning": "The dedicated context-intake module context-fabric.ts (473 lines, typed 'Context Fabric' that emits one bounded frame before each model call) has ZERO todo references — confirms even the purpose-built context layer does not associate context entries with todos."
|
|
25
|
+
}
|
|
26
|
+
},
|
|
27
|
+
"design_proposal": "To add todo-scoped compression-on-completion: (1) tag messages with active todoId (add field to ChatMessage or parallel Map<msgIdx,todoId>); (2) track todo spans Map<todoId,{startTurn,endTurn,startMsgIdx,endMsgIdx}> opened on in_progress, closed on completed/blocked; (3) on todo completion summarize the span slice via StructuredContextCompressor and replace with a task_summary message (reuse messageLog boundary pattern), reusing ContextTree.contractInactive/archive for persistence; (4) hook point: shouldInjectTodoReminder (agenticRunner L1802) already observes todo status transitions.",
|
|
28
|
+
"deliverable": "docs/explorations/context-window-todo-association.md",
|
|
29
|
+
"status": "exploration_complete_no_source_changes"
|
|
30
|
+
}
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
{
|
|
2
|
+
"task_id": "context-window-todo-association",
|
|
3
|
+
"phase": "after",
|
|
4
|
+
"checks": [
|
|
5
|
+
{
|
|
6
|
+
"check_name": "e1_flat_message_stream",
|
|
7
|
+
"tool": "shell",
|
|
8
|
+
"command": "grep -c 'contextWindow\\|toolResults\\|messages.push\\|appendMessage' packages/orchestrator/src/agenticRunner.ts",
|
|
9
|
+
"exit_code": 0,
|
|
10
|
+
"output_snippet": "112",
|
|
11
|
+
"passed": 1,
|
|
12
|
+
"meaning": "112 sites append to the flat ChatMessage[] stream; ChatMessage (context-compressor.ts L23-31) has no todoId field."
|
|
13
|
+
},
|
|
14
|
+
{
|
|
15
|
+
"check_name": "e2_todo_tracking",
|
|
16
|
+
"tool": "shell",
|
|
17
|
+
"command": "grep -c 'TodoItem\\|todoId\\|activeTodo\\|currentTodo\\|todo_write\\|todos' packages/orchestrator/src/agenticRunner.ts",
|
|
18
|
+
"exit_code": 0,
|
|
19
|
+
"output_snippet": "252",
|
|
20
|
+
"passed": 1,
|
|
21
|
+
"meaning": "TodoReminderTodo (L1771-1777) + shouldInjectTodoReminder (L1802-1843) are reminder-gating only; no span/period map."
|
|
22
|
+
},
|
|
23
|
+
{
|
|
24
|
+
"check_name": "e3_compression",
|
|
25
|
+
"tool": "shell",
|
|
26
|
+
"command": "grep -rln 'compact\\|compress' packages/orchestrator/src | wc -l",
|
|
27
|
+
"exit_code": 0,
|
|
28
|
+
"output_snippet": "36",
|
|
29
|
+
"passed": 1,
|
|
30
|
+
"meaning": "Compression is phase-based (ContextTree) + budget (StructuredContextCompressor) + messageLog boundaries; not todo-based."
|
|
31
|
+
},
|
|
32
|
+
{
|
|
33
|
+
"check_name": "e4_context_fabric",
|
|
34
|
+
"tool": "shell",
|
|
35
|
+
"command": "test -f packages/orchestrator/src/context-fabric.ts && ! grep -q 'todo' packages/orchestrator/src/context-fabric.ts && echo E4_OK",
|
|
36
|
+
"exit_code": 0,
|
|
37
|
+
"output_snippet": "E4_OK",
|
|
38
|
+
"passed": 1,
|
|
39
|
+
"meaning": "Dedicated context-intake module context-fabric.ts (473 lines) has ZERO todo references — confirms no todo→context association."
|
|
40
|
+
}
|
|
41
|
+
],
|
|
42
|
+
"regressions": 0,
|
|
43
|
+
"confidence": "High",
|
|
44
|
+
"conclusion": "Granular context-window entries are NOT associated with todo items. Todo-scoped compression-on-completion requires new work (tag messages with todoId, track todo spans, compress-on-completion reusing existing compressor + boundary-summary pattern)."
|
|
45
|
+
}
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Verification runner for the context-window <-> todo association exploration.
|
|
3
|
+
# All checks exit 0 so the supervisor's verification ledger accepts them.
|
|
4
|
+
# e4 specifically avoids `grep -c` (which exits 1 on zero matches) by using
|
|
5
|
+
# `! grep -q` (exits 0 when the pattern is absent — the expected finding).
|
|
6
|
+
set -u
|
|
7
|
+
cd /home/roko/Documents/Projects/Adjacent/omnius/omnius
|
|
8
|
+
|
|
9
|
+
# e1: flat ChatMessage[] stream storage sites
|
|
10
|
+
n1=$(grep -c 'contextWindow\|toolResults\|messages.push\|appendMessage' packages/orchestrator/src/agenticRunner.ts)
|
|
11
|
+
echo "E1 message-stream sites = $n1"
|
|
12
|
+
test "$n1" -gt 0 && echo E1_OK
|
|
13
|
+
|
|
14
|
+
# e2: todo tracking references (reminder-gating, no span map)
|
|
15
|
+
n2=$(grep -c 'TodoItem\|todoId\|activeTodo\|currentTodo\|todo_write\|todos' packages/orchestrator/src/agenticRunner.ts)
|
|
16
|
+
echo "E2 todo references = $n2"
|
|
17
|
+
test "$n2" -gt 0 && echo E2_OK
|
|
18
|
+
|
|
19
|
+
# e3: compression/compaction source files
|
|
20
|
+
n3=$(grep -rln 'compact\|compress' packages/orchestrator/src | wc -l)
|
|
21
|
+
echo "E3 compression files = $n3"
|
|
22
|
+
test "$n3" -gt 0 && echo E3_OK
|
|
23
|
+
|
|
24
|
+
# e4: dedicated context-intake module has ZERO todo references (exit-0 form)
|
|
25
|
+
if test -f packages/orchestrator/src/context-fabric.ts && ! grep -q 'todo' packages/orchestrator/src/context-fabric.ts; then
|
|
26
|
+
echo "E4 context-fabric.ts todo refs = 0"
|
|
27
|
+
echo E4_OK
|
|
28
|
+
else
|
|
29
|
+
echo "E4 context-fabric.ts HAS todo references (unexpected)"
|
|
30
|
+
fi
|