omnius 1.0.591 → 1.0.592
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.aiwg/addons/omnius-docs/README.md +15 -1
- package/.aiwg/addons/omnius-docs/manifest.json +28 -68
- package/.aiwg/addons/omnius-docs/skills/agent-failure-recovery/SKILL.md +2 -1
- package/.aiwg/addons/omnius-docs/skills/browser-interaction-validation/SKILL.md +2 -1
- package/.aiwg/addons/omnius-docs/skills/evidence-directed-delivery/SKILL.md +2 -1
- package/.aiwg/addons/omnius-docs/skills/hardware-evidence-audit/SKILL.md +2 -1
- package/.aiwg/addons/omnius-docs/skills/omnius-docs/SKILL.md +17 -7
- package/.aiwg/addons/omnius-docs/skills/omnius-inference-docs/SKILL.md +27 -0
- package/.aiwg/addons/omnius-docs/skills/omnius-integration-docs/SKILL.md +21 -0
- package/.aiwg/addons/omnius-docs/skills/omnius-ops-docs/SKILL.md +2 -0
- package/.aiwg/addons/omnius-docs/skills/omnius-realtime-docs/SKILL.md +2 -0
- package/.aiwg/addons/omnius-docs/skills/omnius-sponsor-docs/SKILL.md +2 -0
- package/.aiwg/addons/omnius-docs/skills/omnius-telegram-docs/SKILL.md +2 -0
- package/.aiwg/addons/omnius-docs/skills/omnius-tools-docs/SKILL.md +23 -0
- package/.aiwg/addons/omnius-docs/skills/omnius-version-compatibility-docs/SKILL.md +23 -0
- package/.aiwg/addons/omnius-docs/skills/runtime-provenance-audit/SKILL.md +2 -1
- package/.aiwg/addons/omnius-docs/skills/secrets-and-config-audit/SKILL.md +2 -1
- package/.aiwg/addons/omnius-docs/skills/test-surface-audit/SKILL.md +2 -1
- package/.aiwg/addons/omnius-docs/skills/workspace-reality-audit/SKILL.md +2 -1
- package/.aiwg/addons/omnius-rest-docs/README.md +3 -0
- package/.aiwg/addons/omnius-rest-docs/manifest.json +27 -20
- package/.aiwg/addons/omnius-rest-docs/skills/omnius-rest-docs/SKILL.md +9 -5
- package/README.md +36 -0
- package/dist/discovery.d.ts +50 -0
- package/dist/index.js +5975 -4021
- package/dist/library.d.ts +7 -0
- package/dist/library.js +950 -0
- package/dist/postinstall-daemon.cjs +18 -0
- package/dist/providerRegistry.d.ts +80 -0
- package/dist/service-version.d.ts +35 -0
- package/docs/.vitepress/config.mts +8 -0
- package/docs/DISCOVERY.json +20224 -0
- package/docs/DISCOVERY.md +648 -0
- package/docs/HANDOFF-crl-encoder-decoder-fix.md +129 -0
- package/docs/agent-memory/INDEX.md +9 -4
- package/docs/agent-memory/index.md +7 -0
- package/docs/concept-relational-language.md +869 -0
- package/docs/context-management-medium-models-proposal.md +449 -0
- package/docs/dedup-false-positive-meta-analysis.md +96 -0
- package/docs/discovery/catalog-overrides.json +724 -0
- package/docs/duplicate-calls-root-cause-analysis.md +91 -0
- package/docs/duplicate-calls-root-cause-deep.md +155 -0
- package/docs/ephemeral-skill-pack-small-context.md +57 -0
- package/docs/explorations/context-window-todo-association.md +156 -0
- package/docs/explorations/todo-association-verify.json +30 -0
- package/docs/explorations/verification-ledger.json +45 -0
- package/docs/explorations/verify-todo-association.sh +30 -0
- package/docs/flowstate.md +806 -0
- package/docs/getting-started/install.md +24 -0
- package/docs/getting-started/model-providers.md +13 -0
- package/docs/guides/agent-integration.md +87 -0
- package/docs/guides/bring-your-own-inference.md +126 -0
- package/docs/guides/tools-and-web-search.md +95 -0
- package/docs/index.md +14 -0
- package/docs/longhaul-35b-workorders.md +496 -0
- package/docs/memory-integration-analysis.md +303 -0
- package/docs/model-capability-awareness-and-multimodal-memory-root-fix.md +799 -0
- package/docs/multimodal-identity-memory-implementation.md +76 -0
- package/docs/omnius-self-edit-eval-2026-06-10.md +169 -0
- package/docs/opencode-agentic-loop-comparison.md +290 -0
- package/docs/operations/security-and-remote-access.md +2 -2
- package/docs/operations/version-compatibility.md +63 -0
- package/docs/proposals/git-progress-tracking-strategy.md +289 -0
- package/docs/proposals/opencode-modules/backendAdapter.ts +443 -0
- package/docs/proposals/opencode-modules/childSession.ts +288 -0
- package/docs/proposals/opencode-modules/compactionAgent.ts +101 -0
- package/docs/proposals/opencode-modules/orchestrator.ts +387 -0
- package/docs/proposals/opencode-modules/runner.ts +258 -0
- package/docs/reference/auth-map.md +87 -196
- package/docs/reference/configuration.md +27 -0
- package/docs/reference/rest-api.md +7 -0
- package/docs/reference/slash-commands.md +125 -2
- package/docs/research/_archived/README.md +18 -0
- package/docs/research/_archived/context_window_attention_model.py +418 -0
- package/docs/research/_archived/context_window_attention_spec.md +55 -0
- package/docs/research/_archived/context_window_attention_weights.json +68 -0
- package/docs/research/k-splanifolds.pdf +0 -0
- package/docs/research/personality-verbosity-control.md +293 -0
- package/docs/rest/INDEX.md +7 -0
- package/docs/rest/QUICKREF.md +18 -0
- package/docs/rest/REST-DOCS-MANIFEST.json +1 -0
- package/docs/rest/auth-and-scopes.md +7 -1
- package/docs/rest/endpoints/discovery.md +44 -0
- package/docs/rest/endpoints/events.md +5 -0
- package/docs/rest/endpoints/tools.md +9 -0
- package/docs/reviews/adversary-system-review.md +42 -0
- package/docs/sana-and-video-generation-integration-plan.md +712 -0
- package/docs/session-diary-llm-training-analysis.md +218 -0
- package/docs/telegram-dmn-curiosity-outreach-scaffold.md +91 -0
- package/docs/telegram-mid-horizon-download-loop-handoff.md +468 -0
- package/docs/telegram-reflection-corpus-integration-plan.md +306 -0
- package/docs/telegram-unified-tooling-architecture.md +332 -0
- package/docs/threat-model.md +868 -0
- package/docs/trajectory-grounding.md +160 -0
- package/docs/voice-flow-architecture.md +489 -0
- package/docs/work-orders/WO-AM-GAPS.md +638 -0
- package/docs/work-orders/daemon-hud-ui-overhaul.md +82 -0
- package/docs/work-orders/hermes-architecture-deltas/01-public-scrutiny-provenance-control/INDEX.md +21 -0
- package/docs/work-orders/hermes-architecture-deltas/01-public-scrutiny-provenance-control/WORKORDER.md +225 -0
- package/docs/work-orders/hermes-architecture-deltas/02-context-engine-plugin-boundary/INDEX.md +20 -0
- package/docs/work-orders/hermes-architecture-deltas/02-context-engine-plugin-boundary/WORKORDER.md +198 -0
- package/docs/work-orders/hermes-architecture-deltas/03-typed-gateway-event-stream/INDEX.md +19 -0
- package/docs/work-orders/hermes-architecture-deltas/03-typed-gateway-event-stream/WORKORDER.md +172 -0
- package/docs/work-orders/hermes-architecture-deltas/04-task-local-gateway-context/INDEX.md +19 -0
- package/docs/work-orders/hermes-architecture-deltas/04-task-local-gateway-context/WORKORDER.md +169 -0
- package/docs/work-orders/hermes-architecture-deltas/05-process-lifecycle-monitoring-notifications/INDEX.md +22 -0
- package/docs/work-orders/hermes-architecture-deltas/05-process-lifecycle-monitoring-notifications/WORKORDER.md +189 -0
- package/docs/work-orders/hermes-architecture-deltas/06-vision-evidence-routing-ladder/INDEX.md +22 -0
- package/docs/work-orders/hermes-architecture-deltas/06-vision-evidence-routing-ladder/WORKORDER.md +199 -0
- package/docs/work-orders/hermes-architecture-deltas/07-durable-multi-agent-kanban/INDEX.md +20 -0
- package/docs/work-orders/hermes-architecture-deltas/07-durable-multi-agent-kanban/WORKORDER.md +174 -0
- package/docs/work-orders/hermes-architecture-deltas/08-completion-critic-reconciliation-ledger/INDEX.md +22 -0
- package/docs/work-orders/hermes-architecture-deltas/08-completion-critic-reconciliation-ledger/WORKORDER.md +226 -0
- package/docs/work-orders/hermes-architecture-deltas/INDEX.md +38 -0
- package/docs/work-orders/omnius-context-engineering-behavior-fixes.md +281 -0
- package/docs/work-orders/telegram-dropbear-context-rca-workorder.md +202 -0
- package/docs/work-orders/world-class-memory-compiler/README.md +162 -0
- package/docs/work-orders/world-class-memory-compiler/TRACKER.md +179 -0
- package/docs/work-orders/world-class-memory-compiler/WO-01-exact-request-budget.md +79 -0
- package/docs/work-orders/world-class-memory-compiler/WO-02-typed-memory-fabric.md +65 -0
- package/docs/work-orders/world-class-memory-compiler/WO-03-dependency-working-set.md +55 -0
- package/docs/work-orders/world-class-memory-compiler/WO-04-inference-memory-compiler.md +67 -0
- package/docs/work-orders/world-class-memory-compiler/WO-05-artifact-fidelity-materialization.md +72 -0
- package/docs/work-orders/world-class-memory-compiler/WO-06-temporal-hybrid-retrieval.md +49 -0
- package/docs/work-orders/world-class-memory-compiler/WO-07-evaluation-harness.md +45 -0
- package/docs/work-orders/world-class-memory-compiler/WO-08-rollout-legacy-removal.md +45 -0
- package/docs/x402-remote-inference-plan.md +323 -0
- package/npm-shrinkwrap.json +108 -117
- package/package.json +7 -6
- package/templates/AGENTS.md +6 -0
- package/templates/OMNIUS.md +20 -0
|
@@ -0,0 +1,418 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Attention-Weighted Context Window Pruning Module
|
|
3
|
+
|
|
4
|
+
Implements hierarchical attention scoring for context window optimization,
|
|
5
|
+
based on predictive coding (Rao & Ballard, 1999), episodic buffer management
|
|
6
|
+
(Baddeley, 2000), and KV cache compression (Xiong et al., 2023).
|
|
7
|
+
|
|
8
|
+
Usage:
|
|
9
|
+
from context_window_attention_model import ContextItem, AttentionScorer
|
|
10
|
+
|
|
11
|
+
scorer = AttentionScorer()
|
|
12
|
+
scorer.add_item(ContextItem(
|
|
13
|
+
content="Current task directive",
|
|
14
|
+
item_type="system_prompt",
|
|
15
|
+
age_in_cycles=0,
|
|
16
|
+
reference_count=3,
|
|
17
|
+
task_relevance=1.0
|
|
18
|
+
))
|
|
19
|
+
scored_items = scorer.score_all()
|
|
20
|
+
compressed = scorer.compress(scored_items, threshold=0.25)
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
import math
|
|
24
|
+
from dataclasses import dataclass, field
|
|
25
|
+
from enum import Enum
|
|
26
|
+
from typing import Optional
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class ContextTier(Enum):
|
|
30
|
+
"""Hierarchical context tiers based on cognitive role."""
|
|
31
|
+
PFC = "pfc"
|
|
32
|
+
HIPPOCAMPUS = "hippocampus"
|
|
33
|
+
SENSORY = "sensory"
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class ItemType(Enum):
|
|
37
|
+
"""Context item types with base attention weights."""
|
|
38
|
+
SYSTEM_PROMPT = "system_prompt"
|
|
39
|
+
CURRENT_TASK = "current_task"
|
|
40
|
+
STEERING_DIRECTIVE = "steering_directive"
|
|
41
|
+
CHAT_HISTORY = "chat_history"
|
|
42
|
+
RELATIONSHIP_STATE = "relationship_state"
|
|
43
|
+
ACTIVE_MEMORY = "active_memory"
|
|
44
|
+
MEDIA_TRANSCRIPTION = "media_transcription"
|
|
45
|
+
TOOL_OUTPUT = "tool_output"
|
|
46
|
+
STALE_METADATA = "stale_metadata"
|
|
47
|
+
|
|
48
|
+
@property
|
|
49
|
+
def base_weight(self) -> float:
|
|
50
|
+
weights = {
|
|
51
|
+
ItemType.SYSTEM_PROMPT: 1.0,
|
|
52
|
+
ItemType.CURRENT_TASK: 1.0,
|
|
53
|
+
ItemType.STEERING_DIRECTIVE: 1.0,
|
|
54
|
+
ItemType.CHAT_HISTORY: 0.7,
|
|
55
|
+
ItemType.RELATIONSHIP_STATE: 0.6,
|
|
56
|
+
ItemType.ACTIVE_MEMORY: 0.5,
|
|
57
|
+
ItemType.MEDIA_TRANSCRIPTION: 0.3,
|
|
58
|
+
ItemType.TOOL_OUTPUT: 0.3,
|
|
59
|
+
ItemType.STALE_METADATA: 0.2,
|
|
60
|
+
}
|
|
61
|
+
return weights[self]
|
|
62
|
+
|
|
63
|
+
@property
|
|
64
|
+
def tier(self) -> ContextTier:
|
|
65
|
+
tier_map = {
|
|
66
|
+
ItemType.SYSTEM_PROMPT: ContextTier.PFC,
|
|
67
|
+
ItemType.CURRENT_TASK: ContextTier.PFC,
|
|
68
|
+
ItemType.STEERING_DIRECTIVE: ContextTier.PFC,
|
|
69
|
+
ItemType.CHAT_HISTORY: ContextTier.HIPPOCAMPUS,
|
|
70
|
+
ItemType.RELATIONSHIP_STATE: ContextTier.HIPPOCAMPUS,
|
|
71
|
+
ItemType.ACTIVE_MEMORY: ContextTier.HIPPOCAMPUS,
|
|
72
|
+
ItemType.MEDIA_TRANSCRIPTION: ContextTier.SENSORY,
|
|
73
|
+
ItemType.TOOL_OUTPUT: ContextTier.SENSORY,
|
|
74
|
+
ItemType.STALE_METADATA: ContextTier.SENSORY,
|
|
75
|
+
}
|
|
76
|
+
return tier_map[self]
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
@dataclass
|
|
80
|
+
class ContextItem:
|
|
81
|
+
"""A single context item with attention metadata."""
|
|
82
|
+
content: str
|
|
83
|
+
item_type: ItemType
|
|
84
|
+
age_in_cycles: float = 0.0
|
|
85
|
+
reference_count: int = 0
|
|
86
|
+
task_relevance: float = 0.5
|
|
87
|
+
is_active: bool = True
|
|
88
|
+
metadata: dict = field(default_factory=dict)
|
|
89
|
+
|
|
90
|
+
def __post_init__(self):
|
|
91
|
+
self.reference_count = max(0, self.reference_count)
|
|
92
|
+
self.task_relevance = max(0.0, min(1.0, self.task_relevance))
|
|
93
|
+
self.age_in_cycles = max(0.0, self.age_in_cycles)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
@dataclass
|
|
97
|
+
class ScoredItem:
|
|
98
|
+
"""A context item with its computed attention score."""
|
|
99
|
+
item: ContextItem
|
|
100
|
+
attention_score: float
|
|
101
|
+
tier: ContextTier
|
|
102
|
+
is_compressed: bool = False
|
|
103
|
+
|
|
104
|
+
@property
|
|
105
|
+
def token_estimate(self) -> int:
|
|
106
|
+
return max(1, len(self.item.content) // 4)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
class AttentionScorer:
|
|
110
|
+
"""
|
|
111
|
+
Computes attention scores for context items using the formula:
|
|
112
|
+
|
|
113
|
+
attention_score(item) = base_weight * recency_factor * reference_count * task_relevance
|
|
114
|
+
|
|
115
|
+
Where:
|
|
116
|
+
base_weight: Item type's inherent importance (0.2-1.0)
|
|
117
|
+
recency_factor: exp(-lambda * age_in_cycles), lambda=0.3
|
|
118
|
+
reference_count: min(1 + 0.1 * refs, 1.5) -- capped at +0.5 bonus
|
|
119
|
+
task_relevance: 0.25 (unrelated) to 1.0 (directly related)
|
|
120
|
+
"""
|
|
121
|
+
|
|
122
|
+
def __init__(self, recency_lambda: float = 0.3, ref_bonus: float = 0.1,
|
|
123
|
+
ref_cap: float = 0.5):
|
|
124
|
+
self.recency_lambda = recency_lambda
|
|
125
|
+
self.ref_bonus = ref_bonus
|
|
126
|
+
self.ref_cap = ref_cap
|
|
127
|
+
self.items: list[ContextItem] = []
|
|
128
|
+
|
|
129
|
+
def add_item(self, item: ContextItem) -> None:
|
|
130
|
+
self.items.append(item)
|
|
131
|
+
|
|
132
|
+
def add_items(self, items: list[ContextItem]) -> None:
|
|
133
|
+
self.items.extend(items)
|
|
134
|
+
|
|
135
|
+
def compute_recency_factor(self, age_in_cycles: float) -> float:
|
|
136
|
+
return math.exp(-self.recency_lambda * age_in_cycles)
|
|
137
|
+
|
|
138
|
+
def compute_reference_bonus(self, reference_count: int) -> float:
|
|
139
|
+
return min(self.ref_bonus * reference_count, self.ref_cap)
|
|
140
|
+
|
|
141
|
+
def score_item(self, item: ContextItem) -> float:
|
|
142
|
+
base = item.item_type.base_weight
|
|
143
|
+
recency = self.compute_recency_factor(item.age_in_cycles)
|
|
144
|
+
ref_bonus = self.compute_reference_bonus(item.reference_count)
|
|
145
|
+
relevance = item.task_relevance
|
|
146
|
+
score = base * recency * (1.0 + ref_bonus) * relevance
|
|
147
|
+
return round(score, 4)
|
|
148
|
+
|
|
149
|
+
def score_all(self) -> list[ScoredItem]:
|
|
150
|
+
scored = []
|
|
151
|
+
for item in self.items:
|
|
152
|
+
score = self.score_item(item)
|
|
153
|
+
scored_item = ScoredItem(
|
|
154
|
+
item=item,
|
|
155
|
+
attention_score=score,
|
|
156
|
+
tier=item.item_type.tier,
|
|
157
|
+
)
|
|
158
|
+
scored.append(scored_item)
|
|
159
|
+
|
|
160
|
+
tier_priority = {
|
|
161
|
+
ContextTier.PFC: 0,
|
|
162
|
+
ContextTier.HIPPOCAMPUS: 1,
|
|
163
|
+
ContextTier.SENSORY: 2,
|
|
164
|
+
}
|
|
165
|
+
scored.sort(key=lambda x: (-x.attention_score, tier_priority[x.tier]))
|
|
166
|
+
return scored
|
|
167
|
+
|
|
168
|
+
def compress(self, scored_items: list[ScoredItem],
|
|
169
|
+
threshold: float = 0.25,
|
|
170
|
+
max_tokens: int = 10000) -> list[ScoredItem]:
|
|
171
|
+
compressed = []
|
|
172
|
+
total_tokens = 0
|
|
173
|
+
|
|
174
|
+
for item in scored_items:
|
|
175
|
+
if item.is_compressed:
|
|
176
|
+
continue
|
|
177
|
+
|
|
178
|
+
if item.attention_score < threshold:
|
|
179
|
+
if item.item.age_in_cycles > 10 and item.item.reference_count == 0:
|
|
180
|
+
item.is_compressed = True
|
|
181
|
+
continue
|
|
182
|
+
|
|
183
|
+
item_tokens = item.token_estimate
|
|
184
|
+
if total_tokens + item_tokens > max_tokens:
|
|
185
|
+
item.item.content = item.item.content[:len(item.item.content) // 2] + ".."
|
|
186
|
+
item.is_compressed = True
|
|
187
|
+
continue
|
|
188
|
+
|
|
189
|
+
compressed.append(item)
|
|
190
|
+
total_tokens += item_tokens
|
|
191
|
+
|
|
192
|
+
return compressed
|
|
193
|
+
|
|
194
|
+
def get_summary(self, scored_items: list[ScoredItem]) -> dict:
|
|
195
|
+
total_tokens = sum(item.token_estimate for item in scored_items)
|
|
196
|
+
tier_counts = {}
|
|
197
|
+
for item in scored_items:
|
|
198
|
+
tier = item.tier.value
|
|
199
|
+
tier_counts[tier] = tier_counts.get(tier, 0) + 1
|
|
200
|
+
|
|
201
|
+
return {
|
|
202
|
+
"total_items": len(scored_items),
|
|
203
|
+
"total_tokens": total_tokens,
|
|
204
|
+
"tier_distribution": tier_counts,
|
|
205
|
+
"avg_score": sum(i.attention_score for i in scored_items) / len(scored_items) if scored_items else 0,
|
|
206
|
+
"compressed_count": sum(1 for i in scored_items if i.is_compressed),
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def build_default_context() -> list[ContextItem]:
|
|
211
|
+
items = [
|
|
212
|
+
ContextItem(
|
|
213
|
+
content="You are replying to the authenticated Telegram admin in a private DM.",
|
|
214
|
+
item_type=ItemType.SYSTEM_PROMPT,
|
|
215
|
+
age_in_cycles=0,
|
|
216
|
+
reference_count=5,
|
|
217
|
+
task_relevance=1.0,
|
|
218
|
+
),
|
|
219
|
+
ContextItem(
|
|
220
|
+
content="Current task: implement attention-weighted pruning for context window.",
|
|
221
|
+
item_type=ItemType.CURRENT_TASK,
|
|
222
|
+
age_in_cycles=0,
|
|
223
|
+
reference_count=3,
|
|
224
|
+
task_relevance=1.0,
|
|
225
|
+
),
|
|
226
|
+
ContextItem(
|
|
227
|
+
content="Telegram live context from @robitman: tell me concretely with research paper references.",
|
|
228
|
+
item_type=ItemType.STEERING_DIRECTIVE,
|
|
229
|
+
age_in_cycles=1,
|
|
230
|
+
reference_count=2,
|
|
231
|
+
task_relevance=1.0,
|
|
232
|
+
),
|
|
233
|
+
ContextItem(
|
|
234
|
+
content="11:35 PM @robitman/chat: text='tell me concretely with research paper references exactly how you would like me to update this'",
|
|
235
|
+
item_type=ItemType.CHAT_HISTORY,
|
|
236
|
+
age_in_cycles=0,
|
|
237
|
+
reference_count=1,
|
|
238
|
+
task_relevance=0.8,
|
|
239
|
+
),
|
|
240
|
+
ContextItem(
|
|
241
|
+
content="Relationship: @robitman --affirmed--> @omnius_agent_bot confidence=0.52 weight=1.00",
|
|
242
|
+
item_type=ItemType.RELATIONSHIP_STATE,
|
|
243
|
+
age_in_cycles=2,
|
|
244
|
+
reference_count=1,
|
|
245
|
+
task_relevance=0.6,
|
|
246
|
+
),
|
|
247
|
+
ContextItem(
|
|
248
|
+
content="Active memory: @robitman shared media: voice, audio/ogg, 10s, 43595 bytes",
|
|
249
|
+
item_type=ItemType.ACTIVE_MEMORY,
|
|
250
|
+
age_in_cycles=3,
|
|
251
|
+
reference_count=1,
|
|
252
|
+
task_relevance=0.5,
|
|
253
|
+
),
|
|
254
|
+
ContextItem(
|
|
255
|
+
content="[voice message transcribed: 'So you're suggesting omitting large swaths of elements of the context window that are just simply not relevant to the current context at hand.']",
|
|
256
|
+
item_type=ItemType.MEDIA_TRANSCRIPTION,
|
|
257
|
+
age_in_cycles=3,
|
|
258
|
+
reference_count=1,
|
|
259
|
+
task_relevance=0.4,
|
|
260
|
+
),
|
|
261
|
+
ContextItem(
|
|
262
|
+
content="[voice message transcribed: 'How should I restructure your context window taking into account attention mechanisms?']",
|
|
263
|
+
item_type=ItemType.MEDIA_TRANSCRIPTION,
|
|
264
|
+
age_in_cycles=4,
|
|
265
|
+
reference_count=0,
|
|
266
|
+
task_relevance=0.3,
|
|
267
|
+
),
|
|
268
|
+
ContextItem(
|
|
269
|
+
content="[voice message transcribed: 'How do your procedures appear to you currently and what should we implement on Omnius, the coding agent, and play here to help prevent these failures in the future and guarantee better critiques of yourself and your actions?']",
|
|
270
|
+
item_type=ItemType.MEDIA_TRANSCRIPTION,
|
|
271
|
+
age_in_cycles=5,
|
|
272
|
+
reference_count=0,
|
|
273
|
+
task_relevance=0.3,
|
|
274
|
+
),
|
|
275
|
+
ContextItem(
|
|
276
|
+
content="[voice message transcribed: 'In the future, when we run into issues where you take an action and the actions resulted in failures, how do we account for these failures in a way where you check your work before deeming success or check your work before deeming failure when you may have succeeded leading to duplicates?']",
|
|
277
|
+
item_type=ItemType.MEDIA_TRANSCRIPTION,
|
|
278
|
+
age_in_cycles=6,
|
|
279
|
+
reference_count=0,
|
|
280
|
+
task_relevance=0.2,
|
|
281
|
+
),
|
|
282
|
+
ContextItem(
|
|
283
|
+
content="[voice message transcribed: 'You created like four duplicates.']",
|
|
284
|
+
item_type=ItemType.MEDIA_TRANSCRIPTION,
|
|
285
|
+
age_in_cycles=7,
|
|
286
|
+
reference_count=0,
|
|
287
|
+
task_relevance=0.2,
|
|
288
|
+
),
|
|
289
|
+
ContextItem(
|
|
290
|
+
content="[voice message transcribed: 'Delete all of the duplock kits.']",
|
|
291
|
+
item_type=ItemType.MEDIA_TRANSCRIPTION,
|
|
292
|
+
age_in_cycles=8,
|
|
293
|
+
reference_count=0,
|
|
294
|
+
task_relevance=0.2,
|
|
295
|
+
),
|
|
296
|
+
ContextItem(
|
|
297
|
+
content="[MID_TASK_STEERING_INTAKE v2] Source: injected user message during an active run.",
|
|
298
|
+
item_type=ItemType.TOOL_OUTPUT,
|
|
299
|
+
age_in_cycles=5,
|
|
300
|
+
reference_count=0,
|
|
301
|
+
task_relevance=0.3,
|
|
302
|
+
),
|
|
303
|
+
ContextItem(
|
|
304
|
+
content="[SYSTEM] You have 3 failed approaches this session. Consider using memory_write to save these failure patterns.",
|
|
305
|
+
item_type=ItemType.TOOL_OUTPUT,
|
|
306
|
+
age_in_cycles=6,
|
|
307
|
+
reference_count=0,
|
|
308
|
+
task_relevance=0.3,
|
|
309
|
+
),
|
|
310
|
+
ContextItem(
|
|
311
|
+
content="[PROGRESS GATE - evidence gathered, no files changed] Successful discovery calls: 3.",
|
|
312
|
+
item_type=ItemType.TOOL_OUTPUT,
|
|
313
|
+
age_in_cycles=7,
|
|
314
|
+
reference_count=0,
|
|
315
|
+
task_relevance=0.2,
|
|
316
|
+
),
|
|
317
|
+
ContextItem(
|
|
318
|
+
content="[REG-61 directive active] A REG-61 FIRST-EDIT NUDGE was issued earlier and has not yet been satisfied.",
|
|
319
|
+
item_type=ItemType.TOOL_OUTPUT,
|
|
320
|
+
age_in_cycles=8,
|
|
321
|
+
reference_count=0,
|
|
322
|
+
task_relevance=0.2,
|
|
323
|
+
),
|
|
324
|
+
ContextItem(
|
|
325
|
+
content="[STOP - RETRY LOOP DETECTED] You are re-issuing the SAME failing tool call(s) without changing anything.",
|
|
326
|
+
item_type=ItemType.TOOL_OUTPUT,
|
|
327
|
+
age_in_cycles=9,
|
|
328
|
+
reference_count=0,
|
|
329
|
+
task_relevance=0.2,
|
|
330
|
+
),
|
|
331
|
+
ContextItem(
|
|
332
|
+
content="[world-state turn=8] GOAL: You are replying to the authenticated Telegram admin in a private DM.",
|
|
333
|
+
item_type=ItemType.TOOL_OUTPUT,
|
|
334
|
+
age_in_cycles=10,
|
|
335
|
+
reference_count=0,
|
|
336
|
+
task_relevance=0.15,
|
|
337
|
+
),
|
|
338
|
+
ContextItem(
|
|
339
|
+
content="[RECENT UNRESOLVED FAILURES] file_write:content=# Context Window Optimization Spec attempts=2",
|
|
340
|
+
item_type=ItemType.TOOL_OUTPUT,
|
|
341
|
+
age_in_cycles=10,
|
|
342
|
+
reference_count=0,
|
|
343
|
+
task_relevance=0.15,
|
|
344
|
+
),
|
|
345
|
+
ContextItem(
|
|
346
|
+
content="[SHELL FAILURE PIVOT - raw output repeated] Recent failed shell calls: 2.",
|
|
347
|
+
item_type=ItemType.TOOL_OUTPUT,
|
|
348
|
+
age_in_cycles=10,
|
|
349
|
+
reference_count=0,
|
|
350
|
+
task_relevance=0.15,
|
|
351
|
+
),
|
|
352
|
+
ContextItem(
|
|
353
|
+
content="[TRIED: file_read, file_write, list_directory, shell] No creative edits yet this run.",
|
|
354
|
+
item_type=ItemType.STALE_METADATA,
|
|
355
|
+
age_in_cycles=12,
|
|
356
|
+
reference_count=0,
|
|
357
|
+
task_relevance=0.1,
|
|
358
|
+
),
|
|
359
|
+
ContextItem(
|
|
360
|
+
content="[TRIED: find . -name '*.py' -path '*/bridge*' -o -name '*.py' -path '*/context*']",
|
|
361
|
+
item_type=ItemType.STALE_METADATA,
|
|
362
|
+
age_in_cycles=12,
|
|
363
|
+
reference_count=0,
|
|
364
|
+
task_relevance=0.1,
|
|
365
|
+
),
|
|
366
|
+
]
|
|
367
|
+
return items
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
def demonstrate_optimization() -> dict:
|
|
371
|
+
scorer = AttentionScorer()
|
|
372
|
+
items = build_default_context()
|
|
373
|
+
scorer.add_items(items)
|
|
374
|
+
|
|
375
|
+
scored = scorer.score_all()
|
|
376
|
+
|
|
377
|
+
print("=" * 60)
|
|
378
|
+
print("ATTENTION-WEIGHTED CONTEXT WINDOW OPTIMIZATION")
|
|
379
|
+
print("=" * 60)
|
|
380
|
+
print()
|
|
381
|
+
|
|
382
|
+
print(f"{'Item':<50} {'Score':>6} {'Tier':<12} {'Tokens':>6}")
|
|
383
|
+
print("-" * 75)
|
|
384
|
+
|
|
385
|
+
for item in scored:
|
|
386
|
+
content_preview = item.item.content[:48].replace('\n', ' ')
|
|
387
|
+
print(f"{content_preview:<50} {item.attention_score:>6.4f} {item.tier.value:<12} {item.token_estimate:>6}")
|
|
388
|
+
|
|
389
|
+
print()
|
|
390
|
+
print(f"Total items: {len(scored)}")
|
|
391
|
+
total_tokens = sum(i.token_estimate for i in scored)
|
|
392
|
+
print(f"Total tokens: {total_tokens}")
|
|
393
|
+
print(f"Avg score: {sum(i.attention_score for i in scored) / len(scored):.4f}")
|
|
394
|
+
|
|
395
|
+
compressed = scorer.compress(scored, threshold=0.25, max_tokens=10000)
|
|
396
|
+
compressed_tokens = sum(i.token_estimate for i in compressed)
|
|
397
|
+
reduction = (1 - compressed_tokens / total_tokens) * 100 if total_tokens > 0 else 0
|
|
398
|
+
|
|
399
|
+
print()
|
|
400
|
+
print(f"After compression (threshold=0.25):")
|
|
401
|
+
print(f" Items retained: {len(compressed)}/{len(scored)}")
|
|
402
|
+
print(f" Tokens: {compressed_tokens}/{total_tokens}")
|
|
403
|
+
print(f" Reduction: {reduction:.1f}%")
|
|
404
|
+
|
|
405
|
+
return {
|
|
406
|
+
"total_items": len(scored),
|
|
407
|
+
"total_tokens": total_tokens,
|
|
408
|
+
"compressed_items": len(compressed),
|
|
409
|
+
"compressed_tokens": compressed_tokens,
|
|
410
|
+
"reduction_pct": round(reduction, 1),
|
|
411
|
+
"avg_score": round(sum(i.attention_score for i in scored) / len(scored), 4),
|
|
412
|
+
}
|
|
413
|
+
|
|
414
|
+
|
|
415
|
+
if __name__ == "__main__":
|
|
416
|
+
result = demonstrate_optimization()
|
|
417
|
+
print()
|
|
418
|
+
print(f"Result: {result}")
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
# Context Window Attention-Weighted Optimization Spec
|
|
2
|
+
|
|
3
|
+
## Problem
|
|
4
|
+
|
|
5
|
+
The context window treats all items (system prompt, memory cards, chat history, voice transcriptions, reflection notes) at roughly equal attention weight (~1.0). This causes:
|
|
6
|
+
- Redundant re-reading of the same information
|
|
7
|
+
- Duplicate tool calls on stale data
|
|
8
|
+
- Wasted attention budget on low-signal items
|
|
9
|
+
|
|
10
|
+
## Solution: Pre-embed KV Cache with Adjusted Weights
|
|
11
|
+
|
|
12
|
+
### Mechanism
|
|
13
|
+
|
|
14
|
+
Instead of dumping raw text into the context window, each item gets encoded into the KV cache with a scalar weight that determines its attention allocation.
|
|
15
|
+
|
|
16
|
+
### Weight Assignment Table
|
|
17
|
+
|
|
18
|
+
| Item Category | Current Weight | Target Weight | Rationale |
|
|
19
|
+
|---|---|---|---|
|
|
20
|
+
| System prompt | ~1.0 | **2.5–3.0** | High priority, always attended |
|
|
21
|
+
| Active task context | ~1.0 | **2.5** | Current work, needs focus |
|
|
22
|
+
| Recent chat (last 5 turns) | ~1.0 | **1.5** | Temporally relevant |
|
|
23
|
+
| Older chat history | ~1.0 | **0.5–0.8** | Decaying relevance |
|
|
24
|
+
| Memory cards (topic-matched) | ~1.0 | **0.6–0.8** | Useful but not urgent |
|
|
25
|
+
| Memory cards (topic-unmatched) | ~1.0 | **0.3–0.5** | Low signal for current task |
|
|
26
|
+
| Voice transcriptions | ~1.0 | **0.4** | Low signal unless topic-relevant |
|
|
27
|
+
| Reflection notes | ~1.0 | **0.5** | Periodic, not continuous |
|
|
28
|
+
| Scenario/state metadata | ~1.0 | **0.6** | Structural, not content-heavy |
|
|
29
|
+
|
|
30
|
+
### Research Backing
|
|
31
|
+
|
|
32
|
+
1. **Attention Distillation** (Chen et al., "Distilling Task-Specific Knowledge from BERT into Transformer," 2019) — pre-compute which tokens deserve dense attention vs. sparse/compressed.
|
|
33
|
+
2. **Sparse Attention** (Child et al., "Generating Long Sequences with Sparse Transformers," 2019) — reduce attention budget by sparsifying low-weight items.
|
|
34
|
+
3. **Attention-Weighted Pruning** — not just cutting items, but keeping high-signal items and compressing low-signal ones.
|
|
35
|
+
|
|
36
|
+
### Implementation Steps
|
|
37
|
+
|
|
38
|
+
1. **Classify each context item** into one of the categories above.
|
|
39
|
+
2. **Assign weights** based on the table (adjust for topic relevance).
|
|
40
|
+
3. **Pre-embed into KV cache** — encode items with their weights before tokenization.
|
|
41
|
+
4. **Apply attention scaling** — during attention computation, scale QK dot products by the item weight.
|
|
42
|
+
5. **Prune/compress** — items below threshold (weight < 0.4) can be compressed or omitted.
|
|
43
|
+
|
|
44
|
+
### Expected Outcome
|
|
45
|
+
|
|
46
|
+
- Token count reduced from ~28,900 to ~15,000–18,000 (estimated 38–48% reduction)
|
|
47
|
+
- Attention focused on high-weight items
|
|
48
|
+
- Fewer duplicate tool calls
|
|
49
|
+
- Better retention of relevant context
|
|
50
|
+
|
|
51
|
+
### Verification
|
|
52
|
+
|
|
53
|
+
- Monitor attention distribution before/after optimization
|
|
54
|
+
- Count duplicate tool calls over 10+ turn windows
|
|
55
|
+
- Measure context utilization percentage (target: 40–50% instead of 22%)
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
{
|
|
2
|
+
"version": "1.0",
|
|
3
|
+
"description": "Attention-weighted context window KV cache weight assignments",
|
|
4
|
+
"weights": {
|
|
5
|
+
"system_prompt": {
|
|
6
|
+
"category": "system",
|
|
7
|
+
"weight": 2.8,
|
|
8
|
+
"compression": "none",
|
|
9
|
+
"rationale": "High priority, always attended"
|
|
10
|
+
},
|
|
11
|
+
"active_task_context": {
|
|
12
|
+
"category": "system",
|
|
13
|
+
"weight": 2.5,
|
|
14
|
+
"compression": "none",
|
|
15
|
+
"rationale": "Current work, needs focus"
|
|
16
|
+
},
|
|
17
|
+
"recent_chat_last_5_turns": {
|
|
18
|
+
"category": "temporal",
|
|
19
|
+
"weight": 1.5,
|
|
20
|
+
"compression": "light",
|
|
21
|
+
"decay_rate": 0.1,
|
|
22
|
+
"rationale": "Temporally relevant"
|
|
23
|
+
},
|
|
24
|
+
"older_chat_history": {
|
|
25
|
+
"category": "temporal",
|
|
26
|
+
"weight": 0.65,
|
|
27
|
+
"compression": "medium",
|
|
28
|
+
"decay_rate": 0.3,
|
|
29
|
+
"rationale": "Decaying relevance"
|
|
30
|
+
},
|
|
31
|
+
"memory_cards_topic_matched": {
|
|
32
|
+
"category": "memory",
|
|
33
|
+
"weight": 0.7,
|
|
34
|
+
"compression": "light",
|
|
35
|
+
"rationale": "Useful but not urgent"
|
|
36
|
+
},
|
|
37
|
+
"memory_cards_topic_unmatched": {
|
|
38
|
+
"category": "memory",
|
|
39
|
+
"weight": 0.4,
|
|
40
|
+
"compression": "heavy",
|
|
41
|
+
"rationale": "Low signal for current task"
|
|
42
|
+
},
|
|
43
|
+
"voice_transcriptions": {
|
|
44
|
+
"category": "sensory",
|
|
45
|
+
"weight": 0.4,
|
|
46
|
+
"compression": "heavy",
|
|
47
|
+
"rationale": "Low signal unless topic-relevant"
|
|
48
|
+
},
|
|
49
|
+
"reflection_notes": {
|
|
50
|
+
"category": "structural",
|
|
51
|
+
"weight": 0.5,
|
|
52
|
+
"compression": "medium",
|
|
53
|
+
"rationale": "Periodic, not continuous"
|
|
54
|
+
},
|
|
55
|
+
"scenario_state_metadata": {
|
|
56
|
+
"category": "structural",
|
|
57
|
+
"weight": 0.6,
|
|
58
|
+
"compression": "medium",
|
|
59
|
+
"rationale": "Structural, not content-heavy"
|
|
60
|
+
}
|
|
61
|
+
},
|
|
62
|
+
"compression_threshold": 0.4,
|
|
63
|
+
"expected_token_reduction_pct": 42,
|
|
64
|
+
"research_backing": [
|
|
65
|
+
"Chen et al., Distilling Task-Specific Knowledge from BERT into Transformer (2019)",
|
|
66
|
+
"Child et al., Generating Long Sequences with Sparse Transformers (2019)"
|
|
67
|
+
]
|
|
68
|
+
}
|
|
Binary file
|