switchroom 0.19.17 → 0.19.19
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/run-hook.sh +148 -0
- package/bin/workspace-dynamic-hook.sh +147 -38
- package/dist/agent-scheduler/index.js +13 -4
- package/dist/auth-broker/index.js +32 -5
- package/dist/cli/drive-write-pretool.mjs +48 -5
- package/dist/cli/ms-365-write-pretool.mjs +40 -2
- package/dist/cli/notion-write-pretool.mjs +13 -4
- package/dist/cli/switchroom.js +10614 -8104
- package/dist/host-control/main.js +12849 -11446
- package/dist/vault/approvals/kernel-server.js +90 -12
- package/dist/vault/broker/server.js +277 -94
- package/package.json +5 -3
- package/profiles/_base/start.sh.hbs +69 -5
- package/profiles/coding/CLAUDE.md.hbs +1 -1
- package/profiles/default/CLAUDE.md.hbs +3 -3
- package/profiles/executive-assistant/CLAUDE.md.hbs +1 -1
- package/profiles/health-coach/CLAUDE.md.hbs +1 -1
- package/skills/mental-model-curator/SKILL.md +8 -6
- package/telegram-plugin/bridge/bridge.ts +25 -19
- package/telegram-plugin/bridge/mcp-instructions.ts +87 -0
- package/telegram-plugin/dist/bridge/bridge.js +28 -20
- package/telegram-plugin/dist/gateway/gateway.js +2077 -1087
- package/telegram-plugin/dist/server.js +32 -20
- package/telegram-plugin/gateway/always-allow-persist-queue.ts +97 -11
- package/telegram-plugin/gateway/boot-card.ts +5 -1
- package/telegram-plugin/gateway/boot-probes.ts +113 -0
- package/telegram-plugin/gateway/config-approval-handler.test.ts +54 -0
- package/telegram-plugin/gateway/config-approval-handler.ts +16 -1
- package/telegram-plugin/gateway/disconnect-flush.ts +17 -0
- package/telegram-plugin/gateway/gateway.ts +43 -1
- package/telegram-plugin/gateway/handback-preturn-signal.ts +61 -7
- package/telegram-plugin/gateway/ipc-protocol.ts +5 -0
- package/telegram-plugin/gateway/ipc-server.ts +13 -0
- package/telegram-plugin/gateway/liveness-wiring.ts +125 -5
- package/telegram-plugin/gateway/missed-approvals-store.ts +66 -17
- package/telegram-plugin/gateway/obligation-ledger.ts +84 -4
- package/telegram-plugin/gateway/pending-card-store.ts +46 -16
- package/telegram-plugin/gateway/resume-inbound-builder.ts +13 -4
- package/telegram-plugin/gateway/scoped-grant-store.ts +39 -14
- package/telegram-plugin/gateway/store-file.ts +244 -0
- package/telegram-plugin/gateway/stream-render.ts +24 -5
- package/telegram-plugin/hooks/secret-guard-pretool.mjs +249 -76
- package/telegram-plugin/hooks/tool-label-pretool.mjs +88 -2
- package/telegram-plugin/registry/turns-schema.test.ts +8 -3
- package/telegram-plugin/registry/turns-schema.ts +40 -12
- package/telegram-plugin/runtime-metrics.ts +14 -0
- package/telegram-plugin/silence-poke.ts +138 -0
- package/telegram-plugin/tests/boot-probe-drift.test.ts +152 -0
- package/telegram-plugin/tests/bridge-tool-parity.test.ts +95 -0
- package/telegram-plugin/tests/gateway-disconnect-flush.test.ts +32 -0
- package/telegram-plugin/tests/handback-preturn-signal.test.ts +62 -0
- package/telegram-plugin/tests/helpers/liveness-wiring-fixture.ts +178 -0
- package/telegram-plugin/tests/ipc-server-validate-config-approval.test.ts +95 -0
- package/telegram-plugin/tests/mcp-instructions-budget.test.ts +184 -0
- package/telegram-plugin/tests/multitopic-routing-wiring.test.ts +22 -2
- package/telegram-plugin/tests/obligation-determinism.test.ts +114 -3
- package/telegram-plugin/tests/obligation-ledger.test.ts +310 -0
- package/telegram-plugin/tests/registry-turns.test.ts +13 -0
- package/telegram-plugin/tests/resume-inbound-builder.test.ts +15 -0
- package/telegram-plugin/tests/secret-guard-pretool.test.ts +347 -16
- package/telegram-plugin/tests/silence-poke-orphan-reap.test.ts +392 -0
- package/telegram-plugin/tests/silence-poke-teardown-notice.test.ts +301 -0
- package/telegram-plugin/tests/store-atomic-write.test.ts +411 -0
- package/telegram-plugin/tests/stream-render-golden.test.ts +103 -1
- package/telegram-plugin/tests/tool-activity-summary.test.ts +9 -2
- package/telegram-plugin/tests/tool-label-pretool.test.ts +94 -0
- package/telegram-plugin/tests/tts-normalize.test.ts +43 -0
- package/telegram-plugin/tests/voice-normalize-text.test.ts +212 -3
- package/telegram-plugin/tests/worker-feed-repeat-steps.test.ts +147 -0
- package/telegram-plugin/tts-normalize.ts +6 -4
- package/telegram-plugin/voice-normalize-text.ts +168 -11
- package/telegram-plugin/worker-activity-feed.ts +51 -1
- package/vendor/hindsight-memory/CHANGELOG.md +73 -0
- package/vendor/hindsight-memory/scripts/drain_pending.py +668 -56
- package/vendor/hindsight-memory/scripts/lib/client.py +124 -0
- package/vendor/hindsight-memory/scripts/lib/config.py +8 -3
- package/vendor/hindsight-memory/scripts/lib/directives.py +62 -4
- package/vendor/hindsight-memory/scripts/lib/pending.py +865 -33
- package/vendor/hindsight-memory/scripts/lib/retain_split.py +449 -0
- package/vendor/hindsight-memory/scripts/recall.py +257 -12
- package/vendor/hindsight-memory/scripts/retain.py +12 -6
- package/vendor/hindsight-memory/scripts/session_start.py +48 -0
- package/vendor/hindsight-memory/scripts/tests/test_client_document_exists.py +470 -0
- package/vendor/hindsight-memory/scripts/tests/test_directives.py +80 -9
- package/vendor/hindsight-memory/scripts/tests/test_pending_drops.py +2121 -0
- package/vendor/hindsight-memory/scripts/tests/test_recall_integration.py +362 -18
- package/vendor/hindsight-memory/scripts/tests/test_retain_split.py +430 -0
- package/vendor/hindsight-memory/scripts/tests/test_session_start_version_skew.py +204 -0
- package/vendor/hindsight-memory/settings.json +1 -1
- package/vendor/hindsight-memory/tests/test_drain_pending.py +102 -6
- package/vendor/hindsight-memory/tests/test_pending.py +32 -7
|
@@ -0,0 +1,449 @@
|
|
|
1
|
+
"""Deterministic size bound for retain content, with structure-preserving split.
|
|
2
|
+
|
|
3
|
+
Why this exists
|
|
4
|
+
---------------
|
|
5
|
+
A retain POST is *not* one model call. The Hindsight daemon chunks the
|
|
6
|
+
submitted content at ``retain_chunk_size`` chars and runs fact extraction as
|
|
7
|
+
one **sequential** LLM call per chunk
|
|
8
|
+
(``hindsight_api/config.py`` ``DEFAULT_RETAIN_CHUNK_SIZE = 3000``). So the
|
|
9
|
+
server-side wall time of a retain is linear in ``len(content)``, while the
|
|
10
|
+
client only ever gets ONE deadline for the whole POST. Past some content
|
|
11
|
+
length, no client timeout can cover the work and the retain can never succeed
|
|
12
|
+
— not on the live Stop hook, not on the SessionStart drain, not ever. Those
|
|
13
|
+
memories are permanently unsaveable.
|
|
14
|
+
|
|
15
|
+
Measured on the 2026-07-25 fleet backlog (629 queued entries): median content
|
|
16
|
+
26,112 chars, p90 150,224, max 744,546. 154 entries exceeded 60,000 chars and
|
|
17
|
+
burned the full client deadline on every attempt. A 744,546-char entry is
|
|
18
|
+
~249 sequential extraction calls.
|
|
19
|
+
|
|
20
|
+
Why the bound is enforced here and not server-side
|
|
21
|
+
--------------------------------------------------
|
|
22
|
+
The daemon *does* have an auto-splitter — ``retain_batch_tokens``
|
|
23
|
+
(``DEFAULT_RETAIN_BATCH_TOKENS = 10_000``, "Max chars per sub-batch for async
|
|
24
|
+
retain auto-splitting") — but it only applies on the ``async=True`` path.
|
|
25
|
+
Every durability retain in this plugin posts ``async=False`` on purpose
|
|
26
|
+
(commit-before-ack, switchroom #3244 §1.1): the 200 must prove durable
|
|
27
|
+
persistence because a 200 is what lets the caller delete the queue entry and
|
|
28
|
+
advance the watermark. So the one server-side mechanism that would help is
|
|
29
|
+
structurally unavailable to us. It is also the wrong layer: the server
|
|
30
|
+
receives an opaque string and could only cut it at a byte offset, whereas the
|
|
31
|
+
plugin still knows the transcript's message structure and can cut on a
|
|
32
|
+
message boundary.
|
|
33
|
+
|
|
34
|
+
The bound is applied at ``HindsightClient.retain()`` — the single function
|
|
35
|
+
every retain POST in this plugin goes through (Stop hook, SessionEnd,
|
|
36
|
+
drain_pending, reconcile_tail, backfill_transcripts, subagent_retain) — so it
|
|
37
|
+
is code-enforced for every present and future caller rather than something a
|
|
38
|
+
producer has to remember.
|
|
39
|
+
|
|
40
|
+
Splitting rules
|
|
41
|
+
---------------
|
|
42
|
+
Splits are taken on transcript structure, never on a raw byte offset, so a
|
|
43
|
+
part is always a syntactically complete transcript with role attribution
|
|
44
|
+
intact:
|
|
45
|
+
|
|
46
|
+
* JSON transcript (``retainToolCalls`` default, a JSON array of
|
|
47
|
+
``{role, content:[blocks]}``): split the array into groups; each part
|
|
48
|
+
re-serializes as a valid JSON array.
|
|
49
|
+
* Text transcript (``[role: x]\\n…\\n[x:end]`` blocks joined by a blank
|
|
50
|
+
line): split on block boundaries; each part is a whole number of blocks.
|
|
51
|
+
* An atom (one message / one block) larger than the bound is split at line
|
|
52
|
+
boundaries and each fragment is re-wrapped in the SAME role envelope, so
|
|
53
|
+
context is still attributed rather than stranded.
|
|
54
|
+
* Last resort, only if the structural passes cannot get a part under the
|
|
55
|
+
bound (e.g. one unbroken 100k-char line): a hard character slice. The size
|
|
56
|
+
invariant wins over structure at that extreme, because a part over the
|
|
57
|
+
bound is a part that never persists at all.
|
|
58
|
+
"""
|
|
59
|
+
|
|
60
|
+
from __future__ import annotations
|
|
61
|
+
|
|
62
|
+
import json
|
|
63
|
+
import os
|
|
64
|
+
from typing import Optional
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
# ---------------------------------------------------------------------------
|
|
68
|
+
# The bound — derived, not chosen
|
|
69
|
+
# ---------------------------------------------------------------------------
|
|
70
|
+
#
|
|
71
|
+
# max_content_chars = retain_chunk_size × floor(client_deadline / chunk_latency)
|
|
72
|
+
#
|
|
73
|
+
# Each input is a measured or read property of the deployment, not a taste
|
|
74
|
+
# call, and each is env-overridable so the bound tracks the deployment:
|
|
75
|
+
#
|
|
76
|
+
# retain_chunk_size 3000 s — server-side chars per extraction chunk.
|
|
77
|
+
# `hindsight_api/config.py`
|
|
78
|
+
# DEFAULT_RETAIN_CHUNK_SIZE = 3000. One chunk =
|
|
79
|
+
# one sequential LLM call.
|
|
80
|
+
# chunk_latency 18.4 s — measured mean wall time of a healthy retain
|
|
81
|
+
# extraction call, n=752, LiteLLM SpendLogs
|
|
82
|
+
# 2026-07-25 07:00–08:05 UTC (the 0–7,941
|
|
83
|
+
# completion-token bucket; the runaway bucket is
|
|
84
|
+
# a separate defect, fixed by capping
|
|
85
|
+
# HINDSIGHT_API_RETAIN_MAX_COMPLETION_TOKENS).
|
|
86
|
+
# client_deadline 280 s — the deadline of the DURABILITY path (the
|
|
87
|
+
# out-of-hook backlog drain, #3599), deliberately
|
|
88
|
+
# not the live Stop hook's 15s. The live path is
|
|
89
|
+
# allowed to miss its deadline: it enqueues to
|
|
90
|
+
# pending-retains and the drain retries. Sizing
|
|
91
|
+
# to 15s would cut content to a single chunk and
|
|
92
|
+
# shred every transcript for no gain.
|
|
93
|
+
# This is the ONE definition of that deadline:
|
|
94
|
+
# `drain_pending._backlog_timeout()` defaults to
|
|
95
|
+
# `retain_client_deadline()` rather than a second
|
|
96
|
+
# literal, and `src/setup/hindsight.ts`
|
|
97
|
+
# (`HINDSIGHT_RETAIN_CLIENT_DEADLINE_S`, #3611)
|
|
98
|
+
# mirrors it as the client half of that PR's
|
|
99
|
+
# `server per-call timeout < client deadline`
|
|
100
|
+
# assertion — which its derived 204s server
|
|
101
|
+
# timeout satisfies against 280 and would NOT
|
|
102
|
+
# against #3599's original 180s literal.
|
|
103
|
+
#
|
|
104
|
+
# floor(280 / 18.4) = 15 chunks → 15 × 3000 = 45,000 chars
|
|
105
|
+
#
|
|
106
|
+
# Sanity check against the same backlog: every entry at or below 60,000 chars
|
|
107
|
+
# drained successfully inside the 280s deadline (observed per-entry times
|
|
108
|
+
# 0.3s–153.4s at concurrency 3), so 45,000 sits inside demonstrated-good
|
|
109
|
+
# territory with margin for a slower model or a busier box.
|
|
110
|
+
DEFAULT_RETAIN_CHUNK_SIZE = 3000
|
|
111
|
+
DEFAULT_RETAIN_CHUNK_LATENCY_S = 18.4
|
|
112
|
+
DEFAULT_RETAIN_CLIENT_DEADLINE_S = 280.0
|
|
113
|
+
|
|
114
|
+
# Absolute floor: one chunk. A bound below one chunk would split every
|
|
115
|
+
# transcript into extraction-sized confetti and is never the right answer.
|
|
116
|
+
MIN_RETAIN_CONTENT_CHARS = DEFAULT_RETAIN_CHUNK_SIZE
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _env_number(name: str, default: float) -> float:
|
|
120
|
+
raw = os.environ.get(name)
|
|
121
|
+
if not raw:
|
|
122
|
+
return default
|
|
123
|
+
try:
|
|
124
|
+
value = float(raw)
|
|
125
|
+
except (TypeError, ValueError):
|
|
126
|
+
return default
|
|
127
|
+
return value if value > 0 else default
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def retain_client_deadline() -> float:
|
|
131
|
+
"""Seconds a DURABILITY caller waits for one retain POST.
|
|
132
|
+
|
|
133
|
+
The single definition of that deadline. ``retain_content_limit()`` sizes
|
|
134
|
+
content so a whole POST fits inside it, and
|
|
135
|
+
``drain_pending._backlog_timeout()`` takes it as its default so the drainer
|
|
136
|
+
actually waits that long — a shorter drain deadline would abandon a
|
|
137
|
+
correctly-sized part mid-extraction and rebuild the very re-post loop #3599
|
|
138
|
+
fixed, one size class up.
|
|
139
|
+
"""
|
|
140
|
+
return _env_number("HINDSIGHT_RETAIN_CLIENT_DEADLINE_S", DEFAULT_RETAIN_CLIENT_DEADLINE_S)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def retain_content_limit() -> int:
|
|
144
|
+
"""Max chars of retain content that can complete inside the client deadline.
|
|
145
|
+
|
|
146
|
+
Recomputed per call (cheap) so a test or an operator can move an input via
|
|
147
|
+
the environment without reimporting the module.
|
|
148
|
+
"""
|
|
149
|
+
override = os.environ.get("HINDSIGHT_RETAIN_MAX_CONTENT_CHARS")
|
|
150
|
+
if override:
|
|
151
|
+
try:
|
|
152
|
+
forced = int(override)
|
|
153
|
+
if forced > 0:
|
|
154
|
+
return max(MIN_RETAIN_CONTENT_CHARS, forced)
|
|
155
|
+
except (TypeError, ValueError):
|
|
156
|
+
pass
|
|
157
|
+
|
|
158
|
+
chunk_size = int(_env_number("HINDSIGHT_RETAIN_CHUNK_SIZE", DEFAULT_RETAIN_CHUNK_SIZE))
|
|
159
|
+
latency = _env_number("HINDSIGHT_RETAIN_CHUNK_LATENCY_S", DEFAULT_RETAIN_CHUNK_LATENCY_S)
|
|
160
|
+
deadline = retain_client_deadline()
|
|
161
|
+
|
|
162
|
+
chunks = int(deadline // latency)
|
|
163
|
+
if chunks < 1:
|
|
164
|
+
chunks = 1
|
|
165
|
+
return max(MIN_RETAIN_CONTENT_CHARS, chunk_size * chunks)
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
# ---------------------------------------------------------------------------
|
|
169
|
+
# Part identity
|
|
170
|
+
# ---------------------------------------------------------------------------
|
|
171
|
+
|
|
172
|
+
def part_document_id(document_id: str, index: int, total: int) -> str:
|
|
173
|
+
"""Document id for part ``index`` (0-based) of ``total``.
|
|
174
|
+
|
|
175
|
+
``total <= 1`` returns the id unchanged, so an unsplit retain keeps EXACTLY
|
|
176
|
+
the id it has today and its upsert convergence (retain.py
|
|
177
|
+
``slice_document_id``) is untouched.
|
|
178
|
+
|
|
179
|
+
For a split, ``{base}-p{i}of{n}``: deterministic (same content + same bound
|
|
180
|
+
⇒ same parts ⇒ same ids, so a retry upserts rather than duplicates), unique
|
|
181
|
+
per part (parts must never overwrite each other — that would be silent
|
|
182
|
+
loss), and prefix-preserving, so the ``{session_id}`` prefix probe in
|
|
183
|
+
``client.list_session_document_ids`` still finds them.
|
|
184
|
+
"""
|
|
185
|
+
if total <= 1:
|
|
186
|
+
return document_id
|
|
187
|
+
return f"{document_id}-p{index + 1}of{total}"
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def part_metadata(metadata: Optional[dict], index: int, total: int) -> dict:
|
|
191
|
+
"""Metadata for part ``index`` of ``total``, carrying provenance.
|
|
192
|
+
|
|
193
|
+
Unsplit retains are returned untouched (a copy). A split stamps the part
|
|
194
|
+
position, so any one part says where it sits in the logical retain and the
|
|
195
|
+
base document id is recoverable from the part id's ``-p{i}of{n}`` suffix.
|
|
196
|
+
Values are strings to match the existing metadata convention
|
|
197
|
+
(``retain.py`` writes ``message_count`` as a string).
|
|
198
|
+
"""
|
|
199
|
+
base = dict(metadata or {})
|
|
200
|
+
if total <= 1:
|
|
201
|
+
return base
|
|
202
|
+
base["retain_part_index"] = str(index + 1)
|
|
203
|
+
base["retain_part_count"] = str(total)
|
|
204
|
+
return base
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
# ---------------------------------------------------------------------------
|
|
208
|
+
# Splitting
|
|
209
|
+
# ---------------------------------------------------------------------------
|
|
210
|
+
|
|
211
|
+
_TEXT_BLOCK_SEP = "\n\n"
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def split_retain_content(content: str, max_chars: Optional[int] = None) -> list:
|
|
215
|
+
"""Split ``content`` into parts each at most ``max_chars`` long.
|
|
216
|
+
|
|
217
|
+
Returns ``[content]`` unchanged when it already fits — the overwhelmingly
|
|
218
|
+
common case, and the one where behaviour must not change at all.
|
|
219
|
+
|
|
220
|
+
Guarantees, all asserted by the test suite:
|
|
221
|
+
* every part is non-empty and ``<= max_chars``
|
|
222
|
+
* concatenating the parts loses no *message*; only the pathological
|
|
223
|
+
single-unbreakable-line case cuts inside text
|
|
224
|
+
* the split is a pure function of (content, max_chars) — no clock, no
|
|
225
|
+
randomness — so retries converge on identical parts and identical ids
|
|
226
|
+
"""
|
|
227
|
+
limit = max_chars if (max_chars and max_chars > 0) else retain_content_limit()
|
|
228
|
+
# Non-string / empty content is passed straight through: the retain path's
|
|
229
|
+
# existing behaviour for it is not this module's business to change.
|
|
230
|
+
if not isinstance(content, str) or len(content) <= limit:
|
|
231
|
+
return [content]
|
|
232
|
+
|
|
233
|
+
parts = _split_json(content, limit)
|
|
234
|
+
if parts is None:
|
|
235
|
+
parts = _split_text(content, limit)
|
|
236
|
+
|
|
237
|
+
# Normalisation: structure-preserving passes are best-effort; the size
|
|
238
|
+
# bound is not. Anything still over the limit gets hard-sliced, because a
|
|
239
|
+
# part over the limit is a part that never persists.
|
|
240
|
+
normalised: list = []
|
|
241
|
+
for part in parts:
|
|
242
|
+
if len(part) <= limit:
|
|
243
|
+
if part:
|
|
244
|
+
normalised.append(part)
|
|
245
|
+
else:
|
|
246
|
+
normalised.extend(_hard_split(part, limit))
|
|
247
|
+
return normalised or [content[:limit]]
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def _hard_split(text: str, limit: int) -> list:
|
|
251
|
+
"""Last-resort fixed-width slice. Always satisfies the bound."""
|
|
252
|
+
return [text[i : i + limit] for i in range(0, len(text), limit)] or [""]
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def _json_dump(value) -> str:
|
|
256
|
+
# Must match _prepare_json_transcript in lib/content.py exactly, or a part
|
|
257
|
+
# would measure differently here than it serialises there.
|
|
258
|
+
return json.dumps(value, indent=None, ensure_ascii=False)
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def _split_json(content: str, limit: int) -> Optional[list]:
|
|
262
|
+
"""Split a JSON-array transcript on message boundaries.
|
|
263
|
+
|
|
264
|
+
Returns ``None`` (not a partial result) when ``content`` is not the JSON
|
|
265
|
+
transcript shape, so the caller falls through to the text splitter.
|
|
266
|
+
"""
|
|
267
|
+
stripped = content.lstrip()
|
|
268
|
+
if not stripped.startswith("["):
|
|
269
|
+
return None
|
|
270
|
+
try:
|
|
271
|
+
messages = json.loads(content)
|
|
272
|
+
except (ValueError, TypeError):
|
|
273
|
+
return None
|
|
274
|
+
if not isinstance(messages, list) or not messages:
|
|
275
|
+
return None
|
|
276
|
+
|
|
277
|
+
# Expand any single message that cannot fit on its own, so the grouping
|
|
278
|
+
# pass below only ever sees messages that individually fit.
|
|
279
|
+
expanded: list = []
|
|
280
|
+
for message in messages:
|
|
281
|
+
serialised = _json_dump([message])
|
|
282
|
+
if len(serialised) <= limit:
|
|
283
|
+
expanded.append(message)
|
|
284
|
+
else:
|
|
285
|
+
expanded.extend(_split_json_message(message, limit))
|
|
286
|
+
|
|
287
|
+
parts: list = []
|
|
288
|
+
group: list = []
|
|
289
|
+
for message in expanded:
|
|
290
|
+
candidate = group + [message]
|
|
291
|
+
if group and len(_json_dump(candidate)) > limit:
|
|
292
|
+
parts.append(_json_dump(group))
|
|
293
|
+
group = [message]
|
|
294
|
+
else:
|
|
295
|
+
group = candidate
|
|
296
|
+
if group:
|
|
297
|
+
parts.append(_json_dump(group))
|
|
298
|
+
return parts
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
def _split_json_message(message, limit: int) -> list:
|
|
302
|
+
"""Split ONE oversized JSON message into several same-role messages.
|
|
303
|
+
|
|
304
|
+
Preserves the role envelope on every fragment: the model still sees who
|
|
305
|
+
said what, which is the context that a naive byte cut strands.
|
|
306
|
+
"""
|
|
307
|
+
if not isinstance(message, dict):
|
|
308
|
+
return [message]
|
|
309
|
+
role = message.get("role", "unknown")
|
|
310
|
+
blocks = message.get("content")
|
|
311
|
+
if not isinstance(blocks, list) or not blocks:
|
|
312
|
+
return [message]
|
|
313
|
+
|
|
314
|
+
# Envelope cost of a one-message array with no blocks — the budget that
|
|
315
|
+
# blocks have to fit inside.
|
|
316
|
+
envelope = len(_json_dump([{"role": role, "content": []}]))
|
|
317
|
+
budget = limit - envelope
|
|
318
|
+
if budget <= 0:
|
|
319
|
+
return [message]
|
|
320
|
+
|
|
321
|
+
# Expand oversized individual blocks first (a single giant tool_result).
|
|
322
|
+
atoms: list = []
|
|
323
|
+
for block in blocks:
|
|
324
|
+
if len(_json_dump(block)) <= budget:
|
|
325
|
+
atoms.append(block)
|
|
326
|
+
else:
|
|
327
|
+
atoms.extend(_split_json_block(block, budget))
|
|
328
|
+
|
|
329
|
+
out: list = []
|
|
330
|
+
group: list = []
|
|
331
|
+
for atom in atoms:
|
|
332
|
+
candidate = group + [atom]
|
|
333
|
+
if group and len(_json_dump(candidate)) > budget:
|
|
334
|
+
out.append({"role": role, "content": group})
|
|
335
|
+
group = [atom]
|
|
336
|
+
else:
|
|
337
|
+
group = candidate
|
|
338
|
+
if group:
|
|
339
|
+
out.append({"role": role, "content": group})
|
|
340
|
+
return out or [message]
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
def _split_json_block(block, budget: int) -> list:
|
|
344
|
+
"""Split one oversized content block by slicing its longest text field."""
|
|
345
|
+
if not isinstance(block, dict):
|
|
346
|
+
return [block]
|
|
347
|
+
field = None
|
|
348
|
+
for candidate in ("text", "content", "input", "output"):
|
|
349
|
+
if isinstance(block.get(candidate), str) and block[candidate]:
|
|
350
|
+
field = candidate
|
|
351
|
+
break
|
|
352
|
+
if field is None:
|
|
353
|
+
return [block]
|
|
354
|
+
|
|
355
|
+
# Room the text has once the rest of the block is serialised.
|
|
356
|
+
skeleton = dict(block)
|
|
357
|
+
skeleton[field] = ""
|
|
358
|
+
room = budget - len(_json_dump(skeleton))
|
|
359
|
+
if room <= 0:
|
|
360
|
+
return [block]
|
|
361
|
+
|
|
362
|
+
out = []
|
|
363
|
+
for fragment in _split_text_by_lines(block[field], room):
|
|
364
|
+
clone = dict(block)
|
|
365
|
+
clone[field] = fragment
|
|
366
|
+
out.append(clone)
|
|
367
|
+
return out or [block]
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
def _split_text(content: str, limit: int) -> list:
|
|
371
|
+
"""Split a ``[role: x]…[x:end]`` transcript on block boundaries."""
|
|
372
|
+
blocks = content.split(_TEXT_BLOCK_SEP)
|
|
373
|
+
|
|
374
|
+
expanded: list = []
|
|
375
|
+
for block in blocks:
|
|
376
|
+
if len(block) <= limit:
|
|
377
|
+
expanded.append(block)
|
|
378
|
+
else:
|
|
379
|
+
expanded.extend(_split_text_block(block, limit))
|
|
380
|
+
|
|
381
|
+
parts: list = []
|
|
382
|
+
group: list = []
|
|
383
|
+
group_len = 0
|
|
384
|
+
sep_len = len(_TEXT_BLOCK_SEP)
|
|
385
|
+
for block in expanded:
|
|
386
|
+
added = len(block) + (sep_len if group else 0)
|
|
387
|
+
if group and group_len + added > limit:
|
|
388
|
+
parts.append(_TEXT_BLOCK_SEP.join(group))
|
|
389
|
+
group, group_len = [block], len(block)
|
|
390
|
+
else:
|
|
391
|
+
group.append(block)
|
|
392
|
+
group_len += added
|
|
393
|
+
if group:
|
|
394
|
+
parts.append(_TEXT_BLOCK_SEP.join(group))
|
|
395
|
+
return parts
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
def _split_text_block(block: str, limit: int) -> list:
|
|
399
|
+
"""Split one oversized ``[role: x]…[x:end]`` block, re-wrapping each part.
|
|
400
|
+
|
|
401
|
+
Falls back to a plain line split when the block is not in the marker form
|
|
402
|
+
(e.g. a legacy flat transcript), which the caller's normalisation pass
|
|
403
|
+
then hard-slices if any fragment is still over the bound.
|
|
404
|
+
"""
|
|
405
|
+
lines = block.split("\n")
|
|
406
|
+
if len(lines) < 3 or not lines[0].startswith("[role: ") or not lines[-1].endswith(":end]"):
|
|
407
|
+
return _split_text_by_lines(block, limit)
|
|
408
|
+
|
|
409
|
+
header, footer = lines[0], lines[-1]
|
|
410
|
+
body = "\n".join(lines[1:-1])
|
|
411
|
+
# +2 for the two newlines that rejoin header/body/footer.
|
|
412
|
+
room = limit - len(header) - len(footer) - 2
|
|
413
|
+
if room <= 0:
|
|
414
|
+
return _split_text_by_lines(block, limit)
|
|
415
|
+
return [f"{header}\n{fragment}\n{footer}" for fragment in _split_text_by_lines(body, room)]
|
|
416
|
+
|
|
417
|
+
|
|
418
|
+
def _split_text_by_lines(text: str, limit: int) -> list:
|
|
419
|
+
"""Group whole lines into fragments of at most ``limit`` chars.
|
|
420
|
+
|
|
421
|
+
A single line longer than ``limit`` is hard-sliced — the only place this
|
|
422
|
+
module cuts inside a line, and unavoidable: an unbroken 100k-char line has
|
|
423
|
+
no structural boundary to cut on.
|
|
424
|
+
"""
|
|
425
|
+
if limit <= 0:
|
|
426
|
+
return [text]
|
|
427
|
+
if len(text) <= limit:
|
|
428
|
+
return [text]
|
|
429
|
+
|
|
430
|
+
out: list = []
|
|
431
|
+
buf: list = []
|
|
432
|
+
buf_len = 0
|
|
433
|
+
for line in text.split("\n"):
|
|
434
|
+
if len(line) > limit:
|
|
435
|
+
if buf:
|
|
436
|
+
out.append("\n".join(buf))
|
|
437
|
+
buf, buf_len = [], 0
|
|
438
|
+
out.extend(_hard_split(line, limit))
|
|
439
|
+
continue
|
|
440
|
+
added = len(line) + (1 if buf else 0)
|
|
441
|
+
if buf and buf_len + added > limit:
|
|
442
|
+
out.append("\n".join(buf))
|
|
443
|
+
buf, buf_len = [line], len(line)
|
|
444
|
+
else:
|
|
445
|
+
buf.append(line)
|
|
446
|
+
buf_len += added
|
|
447
|
+
if buf:
|
|
448
|
+
out.append("\n".join(buf))
|
|
449
|
+
return out or [text]
|