pi-zip 0.0.0-stage → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +139 -2
- package/package.json +38 -4
- package/src/cache.ts +116 -0
- package/src/classify.ts +77 -0
- package/src/guard.ts +192 -0
- package/src/index.ts +20 -0
- package/src/learn.ts +141 -0
- package/src/notice.ts +108 -0
- package/src/placeholder.ts +135 -0
- package/src/plan.ts +588 -0
- package/src/recall.ts +199 -0
- package/src/run.ts +649 -0
- package/src/summary.ts +500 -0
- package/src/util.ts +50 -0
package/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 pi-zip contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
package/README.md
CHANGED
|
@@ -1,3 +1,140 @@
|
|
|
1
|
-
#
|
|
1
|
+
# pi-zip
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
Keeps long [Pi](https://github.com/earendil-works/pi) sessions cheap without losing anything.
|
|
4
|
+
|
|
5
|
+
pi-zip folds old tool output out of the context **only when the prompt cache has already gone cold**, so the edit costs nothing extra, and every folded output can be fetched back byte for byte. No configuration, no proxy process, no footer.
|
|
6
|
+
|
|
7
|
+
## Install
|
|
8
|
+
|
|
9
|
+
```bash
|
|
10
|
+
pi install npm:pi-zip
|
|
11
|
+
# or
|
|
12
|
+
pi install git:github.com/purboo/pi-zip
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
Try it for one run without installing: `pi -e npm:pi-zip`.
|
|
16
|
+
|
|
17
|
+
## Measured results (v0.2.0)
|
|
18
|
+
|
|
19
|
+
Live A/B runs on the same coding tasks (4 task templates × 4 seeds, the user away 6 minutes between prompts), paired by task, 95% bootstrap intervals. BC = [billion-context](https://github.com/ranxianglei/billion-context) with its defaults.
|
|
20
|
+
|
|
21
|
+
| Model | Cost vs BC | Speed | Quality (task done / planted facts recalled) |
|
|
22
|
+
|---|---|---|---|
|
|
23
|
+
| Claude Sonnet 5.5 | **0.79×** [0.72, 0.86] | p90 wait per prompt 53 s vs 85 s; same as plain Pi | 16/16 and 1.00, same as BC |
|
|
24
|
+
| GPT-6.1-sol (14 tasks so far) | 1.00× [0.91, 1.09] | median task 197 s vs 293 s | 14/14 and 1.00 vs 13/14 and 0.96 |
|
|
25
|
+
|
|
26
|
+
Against plain Pi on the same Claude runs: same speed, 0.45× the cost. Hidden-question probes on real long sessions (questions whose answer had been folded): 45–52% answered from the original via `zip_recall` vs 5% for BC's reconstruction, same number of wrong answers.
|
|
27
|
+
|
|
28
|
+
Known limits:
|
|
29
|
+
|
|
30
|
+
- **GLM** (automatic prefix cache that outlives its declared 5 minutes): v0.1 cost about 1.2× BC live. v0.2 learns the real cache lifetime and folds the previous turn after the declared TTL (offline 0.95–0.97× v0.1), but this was not verified live.
|
|
31
|
+
- **Long autonomous runs** (one prompt, hundreds of tool calls, e.g. sub-agents): roughly on par with BC, not better. Outputs are only folded inside a running turn once they are 60 requests old.
|
|
32
|
+
- If you always answer within the cache lifetime, there is little to save, by design.
|
|
33
|
+
|
|
34
|
+
## What you see
|
|
35
|
+
|
|
36
|
+
At most one line per turn, only when something was folded:
|
|
37
|
+
|
|
38
|
+
```
|
|
39
|
+
pi-zip · folded 12 old outputs · 84.2K → 31.5K tokens · 0.9 ms · originals recallable
|
|
40
|
+
pi-zip · summarized 64 requests · 182K → 41K tokens · 8.4 s (done while you were away)
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
The line goes to the UI only; it never enters the model's context. Nothing else is added to the screen.
|
|
44
|
+
|
|
45
|
+
## The three rules
|
|
46
|
+
|
|
47
|
+
1. **Nothing is lost.** The session file stays the single source of truth. pi-zip only changes the view sent to the model, never deletes anything. User messages, tool-call arguments, the system prompt, tool definitions and thinking are never rewritten. Every folded block carries a handle and shows its key lines (errors, ids, first and last line).
|
|
48
|
+
2. **Edit only when the cache is already gone.** Provider prompt caches expire (the model's declared TTL, usually minutes). Changing the context while the cache is warm means paying to rewrite it; changing it after it expired is free, because the whole context is rewritten anyway, and a smaller context makes that rewrite cheaper. So when you come back after the TTL (or after switching model, or when the session was last touched longer ago than the TTL), pi-zip folds old outputs down to about 40K real tokens in one step, and every request of that turn sends the same bytes. While the cache is warm it does nothing, with one exception, the warm valve: above the 40K target it applies that plan, minus the previous user turn (a warm edit never folds the turn you just finished, except when you come back after the declared TTL: there the previous turn is eligible exactly as at a cold return, so a provider whose cache outlives its TTL does not keep it at every return), when the edit pays for the rewrite it causes, and never on the request right after an edited one (no back-to-back warm rewrites). That is one inequality, r Δ²/(2g) + η Δ ≥ K with K = (w − r)(P T − (1 − P) Δ): the reads the removed Δ tokens would cost while the context grows back at g tokens per request (measured in the session), plus, near Pi's compaction trigger, what Pi would charge for the same room (η), against the rewrite of the T = A tokens left after the edit (pricing only the suffix after the earliest edit fires warm edits earlier and lost quality in the offline evaluation; the suffix is logged as `Tsuf` for measurement). r and w are the read and rewrite price ratios of the cache class, never the model's price table: explicit write premium 0.1 / 1.25 x input (2 x on the 1-hour tier), automatic prefix cache 0.2 / 1 x input; the class is read from the provider's usage reports, and until the first response the old fixed rule applies. P is the probability that the cache is still warm. A cold return is P = 0, so K < 0 and it always fires; a single small fold never pays at a warm cache, a large one does.
|
|
49
|
+
3. **Never in the way.** Planning is local and takes milliseconds. Anything that needs a model call (a summary, only when folding is not enough and the same inequality prices the extra model call in) is prepared while you are away: if the cache is about to expire (0.8 x its lifetime after your last request) and you have not come back, a background timer writes the summary with a separate, uncached call. The timer is cancelled the moment you send a prompt. If you return before the summary finishes, only the remaining time is waited, Esc stops the waiting, and the notice says so. If you return while the cache is still warm and the valve does not fire, the prepared summary is discarded (its cost is still counted). In non-interactive modes (`-p`, `--mode json`) nothing is ever started in the background: a cold return that needs a summary computes it right then.
|
|
50
|
+
|
|
51
|
+
**The cache lifetime is learned, not configured.** Every response says how much of the prompt came from the cache. pi-zip compares that read with what the request re-sent unchanged (the previous prompt, or the untouched prefix before one of its own edits: on an automatic prefix cache every edit leaves the first 8K tokens alone, so even the response right after a fold says whether the cache survived) and so learns, per provider and model, whether the cache survived a gap of that length: a few counts per gap bin (30 s to 90 min, with bin edges on the 5-minute and 1-hour tiers), monotone in the gap, older evidence halved after 16 newer observations of the same bin, stored without any content in `~/.pi/agent/pi-zip/cache-survival.json`. Before any evidence the model's declared TTL decides, exactly as before (300 s when it declares none; too short a guess is cheaper than too long); beyond it one clean read overrides it, inside it a lone miss counts as noise (warm caches do miss now and then) and only repeated misses do. A GLM cache read in full after 365 s makes the next 365 s return warm; a Claude 5-minute cache that read nothing after 360 s stays dead. Whether the provider bills cache writes (explicit cache) or not (automatic prefix cache) is read from the first response too. `/zip status` shows the class, its price ratios, and the learned survival per bin with its sample count.
|
|
52
|
+
|
|
53
|
+
Protected from folding: the current user turn and the previous one. When the context is above the target, re-readable outputs of the previous turn (an unchanged file, a read-only command) can still be folded at a cold return or at any return after the declared TTL (never on a warm request inside it), and any output in either turn can be folded once it is 60 assistant requests old (so a long agent run that is a single user turn with hundreds of tool calls is not exempt from folding; the newest 59 requests' outputs always stay, and every fold stays recallable). Messages you type while the agent is running (steering, follow-up) belong to that turn and do not start a new one. Outputs you have already recalled, and `zip_recall` results themselves, are never folded again. "Read-only" is a conservative whitelist: `find -delete` or `-exec`, command substitution, redirects, background jobs, `git diff --output` and the like are not.
|
|
54
|
+
|
|
55
|
+
**Token counts are calibrated, not guessed.** Sizes are estimated as chars/4, which undercounts real tokens (typically by about 1.7x in coding sessions). So the cold cap, the compaction room and the law's token counts are all compared against `k` x the estimate, where `k` = real tokens / estimated tokens for the newest assistant message that reports usage (input + cache read + cache write, over the estimate of the context that request carried; clamped to 1 to 2.5). `k` is read from the session itself on every decision, so a restart, `pi -p` or a resumed session calibrates exactly like a long-lived one, and nothing extra is stored. With no usage to read (a brand-new session, or a provider that reports none) `k` is 1.7. Sizes in the notices, `/zip stats` and the ledger use the same scale.
|
|
56
|
+
|
|
57
|
+
The cold cap is kept below Pi's own compaction trigger (window minus `compaction.reserveTokens`), so on small windows Pi's lossy compaction does not get there first.
|
|
58
|
+
|
|
59
|
+
## Commands
|
|
60
|
+
|
|
61
|
+
| Command | Effect |
|
|
62
|
+
|---|---|
|
|
63
|
+
| `/zip status` | on / off / paused, cache TTL, totals |
|
|
64
|
+
| `/zip stats` | tokens folded, estimated $ saved versus doing nothing, summaries, recalls |
|
|
65
|
+
| `/zip off` | strict no-op: no folds, no summaries, requests left untouched (earlier folds stay recallable) |
|
|
66
|
+
| `/zip on` | resume |
|
|
67
|
+
| `/zip quiet` | toggle the per-turn notice (folding continues) |
|
|
68
|
+
|
|
69
|
+
`off` and `quiet` are remembered per session.
|
|
70
|
+
|
|
71
|
+
## Recall
|
|
72
|
+
|
|
73
|
+
A folded block says what produced it (tool, command or path with offset/limit, user turn, exit status and test counts for commands), its size, a handle and its key lines, so look-alike runs stay apart. It also tells the model to prefer `zip_recall` (original bytes, instant, free, no side effects) over re-running or re-reading, because a re-run may give a different result:
|
|
74
|
+
|
|
75
|
+
```
|
|
76
|
+
[folded by pi-zip · bash bun test test/api.test.ts · turn 12 · exit 1, 79 passed, 2 failed · 5637 chars, 109 lines · handle k3f9a0x1qz]
|
|
77
|
+
key lines kept (original line numbers; up to 8):
|
|
78
|
+
1: (pass) suite 0 > case parse config ...
|
|
79
|
+
106: 2 fail
|
|
80
|
+
109: Command exited with code 1
|
|
81
|
+
Original kept byte for byte, recallable even after summaries or compaction: zip_recall("k3f9a0x1qz") (optional grep/range) is instant, free, no side effects; prefer it to re-running or re-reading (output may differ). Do not guess its content.
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
Placeholders are written once, when the fold is saved: sessions folded by an earlier version keep their old placeholder text unchanged. Summaries list each folded output with its handle, turn and outcome in the same way. A summary of an earlier summary merges it section by section (requests, files, commands, other calls, errors, handle table, handle index) instead of clipping its text: items are deduplicated, the oldest are dropped first under a per-section budget with a count of what was left out, and the previous narrative survives as a short tail excerpt. Pi's own free-text compaction summaries are carried as a head-and-tail excerpt.
|
|
85
|
+
|
|
86
|
+
The model recalls by itself when it needs the content:
|
|
87
|
+
|
|
88
|
+
```
|
|
89
|
+
zip_recall({ handles: ["k3f9a0x1qz", "m2b7c4d8ww"] })
|
|
90
|
+
zip_recall({ handle: "k3f9a0x1qz", grep: "ERROR|FAIL" })
|
|
91
|
+
zip_recall({ handle: "k3f9a0x1qz", range: "120-240" })
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
Recall is batched, exact, and also works for outputs from before a compaction or a pi-zip summary (summaries carry a handle table and a budgeted index of older handles). Output comes back in pages of 20,000 characters; the page says how to continue:
|
|
95
|
+
|
|
96
|
+
```
|
|
97
|
+
zip_recall({ handle: "k3f9a0x1qz", offset: 20000 }) // next page, by characters
|
|
98
|
+
zip_recall({ handle: "k3f9a0x1qz", offset: 20000, limit: 50000 }) // up to 50,000 characters per page
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
Paging is by characters, so even one 45,000-character line can be read in full, and the pages add up to the original byte for byte. `offset` and `limit` also page a `grep` or `range` selection. `grep` is a case-insensitive regular expression; patterns that could backtrack badly (nested quantifiers, quantified alternation, back-references, very long patterns) are searched as literal text instead.
|
|
102
|
+
|
|
103
|
+
## Coexistence
|
|
104
|
+
|
|
105
|
+
If another extension that manages context is loaded (for example billion-context, magic-context, pi-smart-compact, pi-hot-compact, pi-context-prune), pi-zip pauses folding, says so once, and keeps only its request guard. Two writers on the same view give unpredictable results. `/zip status` shows `paused`.
|
|
106
|
+
|
|
107
|
+
The guard has two parts. Before saving a fold or a summary it checks that the edit itself would not leave a tool result without its tool call (failed or aborted assistant turns, which the provider layer drops anyway, are ignored, as is any oddity the session already had). And before a request is sent it repairs a tool result that lost its tool call, instead of letting the provider reject the request and brick the session. The repair understands the request shapes of Pi's Anthropic, OpenAI chat completions (and Mistral), OpenAI Responses (Azure, Codex), Google Gemini/Vertex and Bedrock Converse providers; a request of any other shape is passed through untouched.
|
|
108
|
+
|
|
109
|
+
## FAQ
|
|
110
|
+
|
|
111
|
+
**Will it save money?** Mostly on cold returns, which is where a long session pays for a full cache rewrite. While the cache is warm it edits only when the inequality above says the rewrite pays back (large contexts, near Pi's compaction trigger, outputs 60+ requests old). `/zip stats` shows an estimate based on the model's declared prices: avoided cache writes and reads, minus summary calls, minus the cost of content you recalled. It is an estimate (token counts are chars/4, scaled by the calibration above), and it can be negative.
|
|
112
|
+
|
|
113
|
+
**Does it cost extra?** Planning is free. A summary is one extra model call (the current model, no tools, no prompt cache), shown in `/zip stats`. A background summary you never use (you came back while the cache was warm) is counted too. Recalled content re-enters the context at normal prices.
|
|
114
|
+
|
|
115
|
+
**Can the model lose information?** Folded outputs are replaced by a placeholder with key lines and a handle, and the placeholder tells the model not to guess. Summaries quote your requests verbatim and never paraphrase the model's reasoning. If the model ignores the handle and guesses, that is a model failure pi-zip cannot catch; this is why only re-readable outputs of the previous turn are folded at a cold return; other outputs of the protected turns wait until they are 60 requests old.
|
|
116
|
+
|
|
117
|
+
**Does it work with `/tree`, fork, resume and model switches?** Yes. Folds are ordinary `context_edit` entries, projected per branch by Pi; the originals stay in the session file.
|
|
118
|
+
|
|
119
|
+
**Which providers?** Anything Pi supports. The cache TTL comes from the model's `promptCache` declaration (seconds) for the tier Pi uses (`short`, or `long` when `PI_CACHE_RETENTION=long`), falling back to 5 minutes. Pi's own idle cache refreshes count as a cache touch, and a different provider or model than the last request counts as cold. Providers that do not report cache usage still work; the stats are then estimates only.
|
|
120
|
+
|
|
121
|
+
Some relays rewrite `cache_control`, so the TTL actually written can differ from the one requested. pi-zip checks `usage.cacheWrite1h` (reported by the Anthropic messages API and Bedrock) on the newest few assistant messages of the current model: if you requested 1h but the provider wrote 5m, it uses the 5m TTL (and the reverse, when the model declares a 1h tier), and says so once per session unless `/zip quiet` is on. Without that field it keeps the declared TTL. `PI_ZIP_TTL_SECS` still wins.
|
|
122
|
+
|
|
123
|
+
Some relays reject `PI_CACHE_RETENTION=long` with 400 `a ttl='1h' cache_control block must not come after a ttl='5m' cache_control block`, because they inject their own 5m `cache_control`. That is a provider issue: unset the variable.
|
|
124
|
+
|
|
125
|
+
## Testing
|
|
126
|
+
|
|
127
|
+
```bash
|
|
128
|
+
bun test # unit + invariant tests + integration with Pi itself
|
|
129
|
+
bun run build-check # bundles src/index.ts with Pi's packages external
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
`test/pi-integration.test.ts` runs the extension through Pi's real session manager, extension loader and `emitContext`, with a mid-conversation system update and an aborted tool-call turn in the session, and checks that the request is byte-identical before and after turn_end persists the edits.
|
|
133
|
+
|
|
134
|
+
Environment overrides exist for tests only and are not part of the product surface: `PI_ZIP_TTL_SECS` (cache TTL), `PI_ZIP_COLD_CAP` (fold target in tokens, default 40000), `PI_ZIP_FOLD_MIN` (smallest output worth folding, default 500 tokens), `PI_ZIP_KEEP_LINES`, `PI_ZIP_MIN_GAIN` (summary gain floor of the legacy rule used only when nothing is known about prices, default 10000), `PI_ZIP_INTURN_AGE` (age in assistant requests from which any output of the protected turns may fold, default 60; 0 = never), `PI_ZIP_CACHE_STATS=<path>` (where the learned cache survival lives; benches isolate it), `PI_ZIP_OFF=1` (register nothing), `PI_ZIP_LEDGER=<path>` (append a JSON line per decision; `prompt` records `ttlMs` and `ttlSource`, `declared` or `observed`; including the summary call's token usage; `cold_plan` records the calibration as `k`, `calReal`, `calEst`, `calSource`, next to the scaled `ctxBefore` and `ctxAfter`; `prompt` also records `gapS`, `pWarm`, `survSrc` and `cls`; `law` records every evaluation with `g`, `wr` (w/r), `pWarm`, `survSrc`, `prSrc` and per step `B`, `A`, `T` (= A), `Tsuf` (the suffix after the earliest edit, measurement only), `phi`, `K`, `eta`; `b2b_skip` marks a warm edit skipped right after an edited request; `cache_sample` records each learned survival observation).
|
|
135
|
+
|
|
136
|
+
Internally, `RELAX_PREV_TURN` in `src/plan.ts` selects whether re-readable outputs of the previous user turn may be folded on a cold return or a return after the declared TTL (default `true`; other warm plans never fold them).
|
|
137
|
+
|
|
138
|
+
## License
|
|
139
|
+
|
|
140
|
+
MIT
|
package/package.json
CHANGED
|
@@ -1,6 +1,40 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-zip",
|
|
3
|
-
"version": "0.
|
|
4
|
-
"
|
|
5
|
-
"
|
|
6
|
-
|
|
3
|
+
"version": "0.2.0",
|
|
4
|
+
"description": "Keeps long Pi sessions cheap without losing anything: folds old tool output only when the prompt cache has already gone cold, and every fold can be recalled byte for byte. Zero config.",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"license": "MIT",
|
|
7
|
+
"author": "purboo",
|
|
8
|
+
"repository": {
|
|
9
|
+
"type": "git",
|
|
10
|
+
"url": "git+https://github.com/purboo/pi-zip.git"
|
|
11
|
+
},
|
|
12
|
+
"homepage": "https://github.com/purboo/pi-zip#readme",
|
|
13
|
+
"bugs": {
|
|
14
|
+
"url": "https://github.com/purboo/pi-zip/issues"
|
|
15
|
+
},
|
|
16
|
+
"keywords": [
|
|
17
|
+
"pi-package",
|
|
18
|
+
"pi-extension",
|
|
19
|
+
"context",
|
|
20
|
+
"cache",
|
|
21
|
+
"compaction"
|
|
22
|
+
],
|
|
23
|
+
"files": [
|
|
24
|
+
"src",
|
|
25
|
+
"README.md",
|
|
26
|
+
"LICENSE"
|
|
27
|
+
],
|
|
28
|
+
"pi": {
|
|
29
|
+
"extensions": [
|
|
30
|
+
"./src/index.ts"
|
|
31
|
+
]
|
|
32
|
+
},
|
|
33
|
+
"scripts": {
|
|
34
|
+
"test": "bun test",
|
|
35
|
+
"build-check": "bun build src/index.ts --target=node --external '@earendil-works/*' --outfile /dev/null"
|
|
36
|
+
},
|
|
37
|
+
"peerDependencies": {
|
|
38
|
+
"@earendil-works/pi-coding-agent": "*"
|
|
39
|
+
}
|
|
40
|
+
}
|
package/src/cache.ts
ADDED
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
// Cache temperature (F7): "cold" is a fact about real time, never a guess.
|
|
2
|
+
import { type Any, PRODUCT, envInt } from "./util.ts";
|
|
3
|
+
|
|
4
|
+
export const DEFAULT_TTL_MS = 300_000;
|
|
5
|
+
|
|
6
|
+
export type CacheRetention = "none" | "short" | "long";
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* Lifetime of the prompt cache entry a request writes, mirroring Pi's getPromptCacheTtlMs (core/cache-warmer.js): the tier is the
|
|
10
|
+
* request's cacheRetention, else "long" when PI_CACHE_RETENTION=long, else "short"; the model's promptCache holds SECONDS for
|
|
11
|
+
* that tier only. Pi sends no explicit cacheRetention, so in practice the environment decides. "none" = no cache at all = 0.
|
|
12
|
+
* A model that declares no lifetime for the tier gets fallbackMs. An explicit PI_ZIP_TTL_SECS (tests) wins in ttlFor.
|
|
13
|
+
*/
|
|
14
|
+
export function cacheTtlMs(
|
|
15
|
+
pc: { short?: number; long?: number } | undefined,
|
|
16
|
+
fallbackMs: number = DEFAULT_TTL_MS,
|
|
17
|
+
opts: { cacheRetention?: CacheRetention; env?: Record<string, string | undefined> } = {},
|
|
18
|
+
): number {
|
|
19
|
+
const env = opts.env ?? process.env;
|
|
20
|
+
const retention: CacheRetention = opts.cacheRetention ?? (env.PI_CACHE_RETENTION === "long" ? "long" : "short");
|
|
21
|
+
if (retention === "none") return 0;
|
|
22
|
+
const secs = pc?.[retention];
|
|
23
|
+
return typeof secs === "number" && secs > 0 ? secs * 1000 : fallbackMs;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* TTL tier the provider really wrote, from the newest <=3 usable assistant messages of this model (real run, cacheWrite >= MIN_WRITE):
|
|
28
|
+
* `usage.cacheWrite1h` (Anthropic, Bedrock) is the 1h share of `cacheWrite`. >= 50% -> "1h", ~0 -> "5m". The field absent (the API
|
|
29
|
+
* does not report it) or a split in between = no evidence (undefined): the declared TTL stands. Relays may rewrite cache_control.
|
|
30
|
+
*/
|
|
31
|
+
const MIN_WRITE = 1024;
|
|
32
|
+
export function observedTier(branch: Any[], key: string): "1h" | "5m" | undefined {
|
|
33
|
+
let w = 0, w1h = 0, seen = 0;
|
|
34
|
+
for (let i = branch.length - 1; i >= 0 && seen < 3; i--) {
|
|
35
|
+
const m = branch[i]?.type === "message" ? branch[i].message : null;
|
|
36
|
+
if (m?.role !== "assistant" || m.stopReason === "error" || m.stopReason === "aborted" || `${m.provider}/${m.model ?? ""}` !== key) continue;
|
|
37
|
+
const u = m.usage;
|
|
38
|
+
if (!(u?.cacheWrite >= MIN_WRITE)) continue;
|
|
39
|
+
seen++;
|
|
40
|
+
if (typeof u.cacheWrite1h === "number") (w += u.cacheWrite), (w1h += u.cacheWrite1h);
|
|
41
|
+
}
|
|
42
|
+
return !w ? undefined : w1h >= w / 2 ? "1h" : w1h < w * 0.05 ? "5m" : undefined;
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
export interface TtlInfo { ms: number; source: "declared" | "observed"; note?: string }
|
|
46
|
+
|
|
47
|
+
/** Declared TTL, corrected by what the provider was seen to write. PI_ZIP_TTL_SECS wins. `note` = the one-time mismatch notice text. */
|
|
48
|
+
export function resolveTtl(model: Any, branch: Any[] = []): TtlInfo {
|
|
49
|
+
if (process.env.PI_ZIP_TTL_SECS) return { ms: envInt("TTL_SECS", 300) * 1000, source: "declared" };
|
|
50
|
+
const pc = model?.promptCache;
|
|
51
|
+
const declared = cacheTtlMs(pc);
|
|
52
|
+
const long = process.env.PI_CACHE_RETENTION === "long";
|
|
53
|
+
const seen = declared > 0 ? observedTier(branch, modelKey(model)) : undefined;
|
|
54
|
+
const mismatch = (req: string, got: string, ms: number): TtlInfo => ({ ms, source: "observed", note: `${PRODUCT} · requested ${req} prompt cache, provider wrote ${got} · using ${got}` });
|
|
55
|
+
if (long && seen === "5m") return mismatch("1h", "5m", cacheTtlMs(pc, DEFAULT_TTL_MS, { cacheRetention: "short" }));
|
|
56
|
+
if (!long && seen === "1h" && pc?.long > 0) return mismatch("5m", "1h", pc.long * 1000);
|
|
57
|
+
return { ms: declared, source: "declared" };
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
export const ttlFor = (model: Any, branch: Any[] = []): number => resolveTtl(model, branch).ms;
|
|
61
|
+
|
|
62
|
+
/** Cold = a prior request exists and nothing touched the cache for longer than the TTL. No prior request counts as warm. */
|
|
63
|
+
export const isColdByTtl = (lastActivityMs: number, nowMs: number, ttlMs: number): boolean => lastActivityMs > 0 && nowMs - lastActivityMs > ttlMs;
|
|
64
|
+
|
|
65
|
+
export const modelKey = (m: Any): string => (m ? `${m.provider ?? ""}/${m.id ?? m.model ?? ""}` : "");
|
|
66
|
+
|
|
67
|
+
const tsOf = (en: Any, inner?: Any): number => {
|
|
68
|
+
const t = typeof inner?.timestamp === "number" ? inner.timestamp : Date.parse(en?.timestamp ?? "");
|
|
69
|
+
return Number.isFinite(t) ? t : 0;
|
|
70
|
+
};
|
|
71
|
+
|
|
72
|
+
/**
|
|
73
|
+
* Last time anything touched the cache, according to the branch: the newest message, or the newest cache-warm usage entry
|
|
74
|
+
* (Pi's own idle refresh persists `{ type: "usage", kind: "cache_warm" }`, and a refresh keeps the entry alive for another TTL).
|
|
75
|
+
* Fresh process: err on the warm side (0 = unknown).
|
|
76
|
+
*/
|
|
77
|
+
export function lastMessageMs(branch: Any[]): number {
|
|
78
|
+
let last = 0;
|
|
79
|
+
for (const en of branch) {
|
|
80
|
+
if (en?.type === "message" && en.message) last = Math.max(last, tsOf(en, en.message));
|
|
81
|
+
else if (en?.type === "usage" && en.kind === "cache_warm") last = Math.max(last, tsOf(en));
|
|
82
|
+
}
|
|
83
|
+
return last;
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/** The newest assistant message that really ran (errors and aborts may never have reached a cache): its provider/model and prompt size. */
|
|
87
|
+
export function lastPrompt(branch: Any[]): { key: string; total: number } {
|
|
88
|
+
for (let i = branch.length - 1; i >= 0; i--) {
|
|
89
|
+
const m = branch[i]?.type === "message" ? branch[i].message : null;
|
|
90
|
+
if (m?.role === "assistant" && m.stopReason !== "error" && m.stopReason !== "aborted" && m.provider) {
|
|
91
|
+
const u = m.usage;
|
|
92
|
+
return { key: `${m.provider}/${m.model ?? ""}`, total: u ? (u.input ?? 0) + (u.cacheRead ?? 0) + (u.cacheWrite ?? 0) : 0 };
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
return { key: "", total: 0 };
|
|
96
|
+
}
|
|
97
|
+
export const lastModelInBranch = (branch: Any[]): string => lastPrompt(branch).key;
|
|
98
|
+
|
|
99
|
+
export type Survival = (gapS: number, priorS: number) => { p: number; src: string };
|
|
100
|
+
/** pastTtl: the idle gap exceeds the declared TTL (the shipped rule's "cold"), whatever the learned P(warm) says. */
|
|
101
|
+
export interface ColdInfo { cold: boolean; reason: string; ttl: TtlInfo; pWarm: number; src: string; gapS: number | null; pastTtl: boolean }
|
|
102
|
+
|
|
103
|
+
/** Cold by time, or because the model changed: a cache entry belongs to one provider and model. `learned` = P(warm) after a gap from the
|
|
104
|
+
* survival the provider was seen to have (learn.ts); without it, or under PI_ZIP_TTL_SECS (tests), the TTL decides. cold = P(warm) < 0.5. */
|
|
105
|
+
export function detectCold(model: Any, lastReqMs: number, branch: Any[], nowMs = Date.now(), lastModel = "", learned?: Survival): ColdInfo {
|
|
106
|
+
const ttl = resolveTtl(model, branch);
|
|
107
|
+
const last = Math.max(lastReqMs, lastMessageMs(branch));
|
|
108
|
+
const prev = lastModel || lastModelInBranch(branch);
|
|
109
|
+
if (last && prev && modelKey(model) && prev !== modelKey(model)) return { cold: true, reason: `model switch ${prev} -> ${modelKey(model)}`, ttl, pWarm: 0, src: "model switch", gapS: null, pastTtl: false };
|
|
110
|
+
if (!last) return { cold: false, reason: "no prior request", ttl, pWarm: 1, src: "no prior request", gapS: null, pastTtl: false };
|
|
111
|
+
const gapS = (nowMs - last) / 1000;
|
|
112
|
+
const byTtl = isColdByTtl(last, nowMs, ttl.ms);
|
|
113
|
+
const s = learned && !process.env.PI_ZIP_TTL_SECS ? learned(gapS, ttl.ms / 1000) : { p: byTtl ? 0 : 1, src: "prior" };
|
|
114
|
+
const reason = `ttl gap ${Math.round(gapS)}s ${byTtl ? ">" : "<="} ${Math.round(ttl.ms / 1000)}s${s.src === "prior" ? "" : `, learned P(warm) ${s.p.toFixed(2)}`}`;
|
|
115
|
+
return { cold: s.p < 0.5, reason, ttl, pWarm: s.p, src: s.src, gapS, pastTtl: byTtl };
|
|
116
|
+
}
|
package/src/classify.ts
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
// Retrievability classification (F5): can the model get this output back by simply re-running the call?
|
|
2
|
+
import { createHash } from "node:crypto";
|
|
3
|
+
import { readFileSync, statSync } from "node:fs";
|
|
4
|
+
import { isAbsolute, join } from "node:path";
|
|
5
|
+
import type { Any } from "./util.ts";
|
|
6
|
+
|
|
7
|
+
export type Recover = "rereadable" | "nonrereadable";
|
|
8
|
+
|
|
9
|
+
const READ_ONLY_CMDS = new Set(["ls", "cat", "head", "tail", "wc", "grep", "rg", "find", "file", "stat", "du", "pwd", "echo"]);
|
|
10
|
+
const READ_ONLY_GIT = new Set(["log", "show", "diff", "status", "blame", "shortlog"]);
|
|
11
|
+
|
|
12
|
+
// Options that make an otherwise read-only command write files or run other programs.
|
|
13
|
+
const FIND_UNSAFE = /^-(delete|exec|execdir|ok|okdir|fprint|fprint0|fprintf|fls)$/;
|
|
14
|
+
const GIT_UNSAFE = /^--(output(=.*)?|ext-diff|textconv|open-files-in-pager.*|exec-path.*)$/;
|
|
15
|
+
const RG_UNSAFE = /^--(pre|pre-glob)(=.*)?$/;
|
|
16
|
+
|
|
17
|
+
/** Read-only bash whitelist: every pipe/;/&& segment must be a whitelisted command with no write or exec option; substitutions, redirections and background jobs are never read-only. */
|
|
18
|
+
export function isReadOnlyBash(cmd: string): boolean {
|
|
19
|
+
const c = String(cmd ?? "").replace(/2>(\/dev\/null|&1)/g, " ").trim();
|
|
20
|
+
if (!c || /[<>`]|\$\(/.test(c)) return false; // any remaining redirection can write; $(...) and backticks run arbitrary commands
|
|
21
|
+
if (/&/.test(c.replace(/&&/g, ";"))) return false; // a lone & backgrounds a job (and `&>` redirects)
|
|
22
|
+
const segments = c.split(/[|;\n]|&&/).map((s) => s.trim()).filter(Boolean);
|
|
23
|
+
if (!segments.length) return false;
|
|
24
|
+
for (const seg of segments) {
|
|
25
|
+
let toks = seg.split(/\s+/).filter(Boolean);
|
|
26
|
+
while (toks.length > 1 && /^[A-Za-z_][A-Za-z0-9_]*=/.test(toks[0])) toks = toks.slice(1);
|
|
27
|
+
const head = (toks[0] ?? "").toLowerCase();
|
|
28
|
+
const rest = toks.slice(1);
|
|
29
|
+
if (head === "find") {
|
|
30
|
+
if (rest.some((t) => FIND_UNSAFE.test(t))) return false;
|
|
31
|
+
continue;
|
|
32
|
+
}
|
|
33
|
+
if (head === "rg") {
|
|
34
|
+
if (rest.some((t) => RG_UNSAFE.test(t))) return false;
|
|
35
|
+
continue;
|
|
36
|
+
}
|
|
37
|
+
if (head === "file") {
|
|
38
|
+
if (rest.some((t) => /^-[A-Za-z]*C/.test(t) || t === "--compile")) return false; // file -C writes a magic database
|
|
39
|
+
continue;
|
|
40
|
+
}
|
|
41
|
+
if (READ_ONLY_CMDS.has(head)) continue;
|
|
42
|
+
if (head === "git" && READ_ONLY_GIT.has((toks[1] ?? "").toLowerCase())) {
|
|
43
|
+
if (rest.some((t) => GIT_UNSAFE.test(t))) return false;
|
|
44
|
+
continue;
|
|
45
|
+
}
|
|
46
|
+
return false;
|
|
47
|
+
}
|
|
48
|
+
return true;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* read: rereadable iff the file still exists and its bytes equal what the tool returned (unchanged since).
|
|
53
|
+
* bash: rereadable iff read-only. grep/find/ls/glob: re-runnable queries. Everything else: not rereadable.
|
|
54
|
+
*/
|
|
55
|
+
export function classifyRecoverability(tool: string, args: Any, rawText: string, cwd: string): Recover {
|
|
56
|
+
tool = String(tool ?? "").toLowerCase();
|
|
57
|
+
const a: Any = args ?? {};
|
|
58
|
+
if (tool === "read") {
|
|
59
|
+
const p = typeof a.path === "string" ? a.path : typeof a.file_path === "string" ? a.file_path : null;
|
|
60
|
+
if (!p) return "nonrereadable";
|
|
61
|
+
try {
|
|
62
|
+
const full = isAbsolute(p) ? p : join(cwd, p);
|
|
63
|
+
const st = statSync(full);
|
|
64
|
+
if (!st.isFile()) return "nonrereadable";
|
|
65
|
+
if (st.size > 2_000_000) return "rereadable"; // too big to hash cheaply; existence is the test
|
|
66
|
+
const onDisk = readFileSync(full);
|
|
67
|
+
const out = rawText.replace(/\n*\[Showing lines [^\]]*\]\s*$/, ""); // strip the continuation notice before comparing
|
|
68
|
+
const h = (t: string) => createHash("sha1").update(t.trimEnd()).digest("hex"); // trailing whitespace may differ around the notice
|
|
69
|
+
return h(onDisk.toString("utf8")) === h(out) ? "rereadable" : "nonrereadable";
|
|
70
|
+
} catch {
|
|
71
|
+
return "nonrereadable";
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
if (tool === "bash") return typeof a.command === "string" && isReadOnlyBash(a.command) ? "rereadable" : "nonrereadable";
|
|
75
|
+
if (tool === "grep" || tool === "find" || tool === "ls" || tool === "glob") return "rereadable";
|
|
76
|
+
return "nonrereadable";
|
|
77
|
+
}
|
package/src/guard.ts
ADDED
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
// Guard (F13, I5): validate an edit set before committing it, and repair orphan tool results in the outgoing payload.
|
|
2
|
+
import { RELAX_PREV_TURN, settings, type Block } from "./plan.ts";
|
|
3
|
+
import type { Any } from "./util.ts";
|
|
4
|
+
|
|
5
|
+
const PROTECT_USER_TURNS = 2;
|
|
6
|
+
|
|
7
|
+
export interface EditSet {
|
|
8
|
+
folds: Map<number, string>; // block idx -> placeholder text
|
|
9
|
+
cut: number | null; // summarise blocks[0..cut-1]; blocks[cut] becomes firstKeptEntryId
|
|
10
|
+
recover?: Map<number, string | undefined>; // block idx -> "rereadable" | "nonrereadable" (needed for folds in the previous user turn)
|
|
11
|
+
relax?: boolean; // default RELAX_PREV_TURN
|
|
12
|
+
inturnAge?: number; // default settings().inturnAge: an output (any class) at least this many assistant requests old may fold in a protected turn (0 = never)
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* Tool-call pairing problems of a block list, the way the provider layer sees it (pi-ai transform-messages): errored or
|
|
17
|
+
* aborted assistant messages are dropped, so their calls never count; a toolResult must follow a call of the nearest
|
|
18
|
+
* assistant message; calls without results are synthesised by the provider but still reported (the caller decides).
|
|
19
|
+
*/
|
|
20
|
+
function pairingIssues(blocks: Block[], from: number): Set<string> {
|
|
21
|
+
const issues = new Set<string>();
|
|
22
|
+
let pending: Set<string> | null = null;
|
|
23
|
+
const closePending = (at: number) => {
|
|
24
|
+
if (pending) for (const id of pending) issues.add(`noResult:${id}@${at}`);
|
|
25
|
+
pending = null;
|
|
26
|
+
};
|
|
27
|
+
for (let i = from; i < blocks.length; i++) {
|
|
28
|
+
const b = blocks[i];
|
|
29
|
+
if (b.kind === "summary" && from > 0) continue; // replaced by our summary
|
|
30
|
+
const m = b.msg;
|
|
31
|
+
if (m.role === "toolResult") {
|
|
32
|
+
if (!pending || !pending.has(m.toolCallId)) issues.add(`orphanResult:${m.toolCallId}@${b.entryId ?? i}`);
|
|
33
|
+
else pending.delete(m.toolCallId);
|
|
34
|
+
continue;
|
|
35
|
+
}
|
|
36
|
+
if (m.role === "system") continue; // transparent: the provider layer holds it back
|
|
37
|
+
closePending(i);
|
|
38
|
+
if (m.role === "assistant") {
|
|
39
|
+
if (m.stopReason === "error" || m.stopReason === "aborted") continue; // dropped by the provider layer
|
|
40
|
+
const ids = (m.content ?? []).filter((c: Any) => c?.type === "toolCall").map((c: Any) => c.id);
|
|
41
|
+
if (ids.length) pending = new Set(ids);
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
closePending(blocks.length);
|
|
45
|
+
return issues;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
/** null = legal. Otherwise the reason: bad targets, edits inside the protected user turns, or a tool_use/tool_result pairing the edits themselves would break. */
|
|
49
|
+
export function validateEdits(blocks: Block[], plan: EditSet, userTurns: number): string | null {
|
|
50
|
+
const relax = plan.relax ?? RELAX_PREV_TURN;
|
|
51
|
+
const inturnAge = plan.inturnAge ?? settings().inturnAge;
|
|
52
|
+
const limit = blocks.findIndex((b) => b.userTurn >= userTurns - PROTECT_USER_TURNS + 1);
|
|
53
|
+
for (const [i, text] of plan.folds) {
|
|
54
|
+
const b = blocks[i];
|
|
55
|
+
if (!b || b.kind !== "toolResult") return `fold target ${i} is not a toolResult`;
|
|
56
|
+
if (b.edited) return `fold target ${i} already edited`;
|
|
57
|
+
const rereadable = plan.recover?.get(i) === "rereadable";
|
|
58
|
+
const aged = inturnAge > 0 && b.age >= inturnAge; // old enough: allowed anywhere, any class (same rule as the planner)
|
|
59
|
+
if (b.userTurn >= userTurns && !aged) return `fold target ${i} is inside the latest user turn`;
|
|
60
|
+
if (b.userTurn === userTurns - 1 && !aged && (!relax || !rereadable)) return `fold target ${i} is in the previous user turn and not re-readable`;
|
|
61
|
+
if (!b.entryId) return `fold target ${i} has no entry id`;
|
|
62
|
+
if (typeof text !== "string" || !text.trim()) return `empty placeholder for ${i}`;
|
|
63
|
+
}
|
|
64
|
+
if (plan.cut !== null) {
|
|
65
|
+
if (plan.cut < 1 || plan.cut >= blocks.length) return `bad cut ${plan.cut}`;
|
|
66
|
+
if (limit >= 0 && plan.cut > limit) return `cut ${plan.cut} inside the last ${PROTECT_USER_TURNS} user turns`;
|
|
67
|
+
const kb = blocks[plan.cut];
|
|
68
|
+
if (!kb.entryId || (kb.kind !== "user" && kb.kind !== "assistant")) return `cut ${plan.cut} is not at a user/assistant message`;
|
|
69
|
+
// only what the cut itself would create is our problem; a session that was already odd stays as odd as it was
|
|
70
|
+
const before = pairingIssues(blocks, 0);
|
|
71
|
+
for (const issue of pairingIssues(blocks, plan.cut)) if (!before.has(issue) && issue.startsWith("orphanResult")) return `the cut would orphan a tool result: ${issue}`;
|
|
72
|
+
for (const issue of pairingIssues(blocks, plan.cut)) if (!before.has(issue)) return `the cut would break tool-call pairing: ${issue}`;
|
|
73
|
+
}
|
|
74
|
+
const kept = plan.cut ?? 0;
|
|
75
|
+
return blocks.slice(kept).some((b) => b.kind !== "summary") ? null : "empty context after edits";
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
const toText = (c: Any): string => (typeof c === "string" ? c : Array.isArray(c) ? c.map((x: Any) => (typeof x === "string" ? x : (x?.text ?? ""))).join("\n") : c === undefined || c === null ? "" : JSON.stringify(c));
|
|
79
|
+
const FOLDED = "[tool result, call folded]\n";
|
|
80
|
+
const hasType = (b: Any, t: string) => b?.type === t;
|
|
81
|
+
|
|
82
|
+
/** Index of the nearest earlier message that is not one of `skip` roles. */
|
|
83
|
+
function prevIndex(list: Any[], i: number, skip: (m: Any) => boolean): number {
|
|
84
|
+
let j = i - 1;
|
|
85
|
+
while (j >= 0 && skip(list[j])) j--;
|
|
86
|
+
return j;
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/** Anthropic messages: tool_result blocks inside a user message must answer tool_use blocks of the assistant message right before it. */
|
|
90
|
+
function repairAnthropic(list: Any[], edit: () => Any): number {
|
|
91
|
+
let repaired = 0;
|
|
92
|
+
for (let i = 0; i < list.length; i++) {
|
|
93
|
+
const m = list[i];
|
|
94
|
+
if (m?.role !== "user" || !Array.isArray(m.content) || !m.content.some((b: Any) => hasType(b, "tool_result"))) continue;
|
|
95
|
+
const prev = list[prevIndex(list, i, (x) => x?.role === "system")];
|
|
96
|
+
const ids = new Set((prev?.role === "assistant" && Array.isArray(prev.content) ? prev.content : []).filter((b: Any) => hasType(b, "tool_use")).map((b: Any) => b.id));
|
|
97
|
+
const content = m.content.map((b: Any) => {
|
|
98
|
+
if (!hasType(b, "tool_result") || ids.has(b.tool_use_id)) return b;
|
|
99
|
+
repaired++;
|
|
100
|
+
return { type: "text", text: FOLDED + toText(b.content) };
|
|
101
|
+
});
|
|
102
|
+
if (content.some((b: Any, k: number) => b !== m.content[k])) edit().messages[i].content = content;
|
|
103
|
+
}
|
|
104
|
+
return repaired;
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
/** Bedrock Converse: { toolResult: { toolUseId } } blocks answer { toolUse: { toolUseId } } blocks of the assistant message before. */
|
|
108
|
+
function repairBedrock(list: Any[], edit: () => Any): number {
|
|
109
|
+
let repaired = 0;
|
|
110
|
+
for (let i = 0; i < list.length; i++) {
|
|
111
|
+
const m = list[i];
|
|
112
|
+
if (m?.role !== "user" || !Array.isArray(m.content) || !m.content.some((b: Any) => b?.toolResult)) continue;
|
|
113
|
+
const prev = list[i - 1];
|
|
114
|
+
const ids = new Set((prev?.role === "assistant" && Array.isArray(prev.content) ? prev.content : []).filter((b: Any) => b?.toolUse).map((b: Any) => b.toolUse.toolUseId));
|
|
115
|
+
const content = m.content.map((b: Any) => {
|
|
116
|
+
if (!b?.toolResult || ids.has(b.toolResult.toolUseId)) return b;
|
|
117
|
+
repaired++;
|
|
118
|
+
return { text: FOLDED + toText(b.toolResult.content) };
|
|
119
|
+
});
|
|
120
|
+
if (content.some((b: Any, k: number) => b !== m.content[k])) edit().messages[i].content = content;
|
|
121
|
+
}
|
|
122
|
+
return repaired;
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
/** OpenAI chat completions and Mistral: a `tool` message must follow the assistant message whose tool_calls it answers. */
|
|
126
|
+
function repairChat(list: Any[], edit: () => Any): number {
|
|
127
|
+
let repaired = 0;
|
|
128
|
+
for (let i = 0; i < list.length; i++) {
|
|
129
|
+
const m = list[i];
|
|
130
|
+
if (m?.role !== "tool" || m.tool_call_id === undefined) continue;
|
|
131
|
+
const prev = list[prevIndex(list, i, (x) => x?.role === "tool")];
|
|
132
|
+
if (!(prev?.role === "assistant" && (prev.tool_calls ?? []).some((c: Any) => c.id === m.tool_call_id))) {
|
|
133
|
+
repaired++;
|
|
134
|
+
edit().messages[i] = { role: "user", content: FOLDED + toText(m.content) };
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
return repaired;
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
/** OpenAI Responses (also Azure and Codex): a function_call_output needs a function_call with the same call_id earlier in the input. */
|
|
141
|
+
function repairResponses(list: Any[], edit: () => Any): number {
|
|
142
|
+
let repaired = 0;
|
|
143
|
+
const calls = new Set<string>();
|
|
144
|
+
for (let i = 0; i < list.length; i++) {
|
|
145
|
+
const it = list[i];
|
|
146
|
+
if (it?.type === "function_call" || it?.type === "custom_tool_call") calls.add(it.call_id);
|
|
147
|
+
else if ((it?.type === "function_call_output" || it?.type === "custom_tool_call_output") && !calls.has(it.call_id)) {
|
|
148
|
+
repaired++;
|
|
149
|
+
edit().input[i] = { role: "user", content: [{ type: "input_text", text: FOLDED + toText(it.output) }] };
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
return repaired;
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
/** Gemini / Vertex: a functionResponse part answers a functionCall part of the model turn right before (by id when present, else by name). */
|
|
156
|
+
function repairGoogle(list: Any[], edit: () => Any, path: (q: Any) => Any[]): number {
|
|
157
|
+
let repaired = 0;
|
|
158
|
+
for (let i = 0; i < list.length; i++) {
|
|
159
|
+
const c = list[i];
|
|
160
|
+
if (c?.role !== "user" || !Array.isArray(c.parts) || !c.parts.some((x: Any) => x?.functionResponse)) continue;
|
|
161
|
+
const prev = list[i - 1];
|
|
162
|
+
const open: Any[] = prev?.role === "model" && Array.isArray(prev.parts) ? prev.parts.filter((x: Any) => x?.functionCall).map((x: Any) => x.functionCall) : [];
|
|
163
|
+
const parts = c.parts.map((x: Any) => {
|
|
164
|
+
const r = x?.functionResponse;
|
|
165
|
+
if (!r) return x;
|
|
166
|
+
const k = open.findIndex((f) => (r.id !== undefined && f.id !== undefined ? f.id === r.id : f.name === r.name));
|
|
167
|
+
if (k >= 0) { open.splice(k, 1); return x; }
|
|
168
|
+
repaired++;
|
|
169
|
+
const out = r.response?.output ?? r.response?.error ?? r.response;
|
|
170
|
+
return { text: FOLDED + toText(out) };
|
|
171
|
+
});
|
|
172
|
+
if (parts.some((x: Any, k: number) => x !== c.parts[k])) path(edit())[i].parts = parts;
|
|
173
|
+
}
|
|
174
|
+
return repaired;
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
/**
|
|
178
|
+
* Turn tool results that no longer follow their tool call into plain text, in the request shapes Pi's providers produce:
|
|
179
|
+
* Anthropic messages, OpenAI chat completions (and Mistral), OpenAI Responses (Azure, Codex), Google Gemini/Vertex and
|
|
180
|
+
* Bedrock Converse. Any other shape is left alone. Returns the repaired payload, or null if nothing was wrong.
|
|
181
|
+
*/
|
|
182
|
+
export function repairPayload(p: Any): { payload: Any; repaired: number } | null {
|
|
183
|
+
if (!p || typeof p !== "object") return null;
|
|
184
|
+
let q: Any = null;
|
|
185
|
+
const edit = () => (q ??= structuredClone(p));
|
|
186
|
+
let repaired = 0;
|
|
187
|
+
if (Array.isArray(p.messages)) repaired += repairAnthropic(p.messages, edit) + repairBedrock(p.messages, edit) + repairChat(p.messages, edit);
|
|
188
|
+
if (Array.isArray(p.input)) repaired += repairResponses(p.input, edit);
|
|
189
|
+
if (Array.isArray(p.contents)) repaired += repairGoogle(p.contents, edit, (x) => x.contents);
|
|
190
|
+
else if (Array.isArray(p.request?.contents)) repaired += repairGoogle(p.request.contents, edit, (x) => x.request.contents);
|
|
191
|
+
return repaired ? { payload: q, repaired } : null;
|
|
192
|
+
}
|
package/src/index.ts
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
// pi-zip: keeps long Pi sessions cheap without losing anything. Wiring only; the logic lives in the sibling modules.
|
|
2
|
+
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
|
3
|
+
import { registerZipCommand } from "./notice.ts";
|
|
4
|
+
import { registerRecallTool } from "./recall.ts";
|
|
5
|
+
import { Zip } from "./run.ts";
|
|
6
|
+
|
|
7
|
+
export default function piZip(pi: ExtensionAPI) {
|
|
8
|
+
if (process.env.PI_ZIP_OFF === "1") return; // test only: behave exactly as if not installed (registers nothing)
|
|
9
|
+
const zip = new Zip(pi);
|
|
10
|
+
registerRecallTool(pi, zip); // stays registered even when off: earlier folds must stay recallable, and a tool-list change would bust the cache
|
|
11
|
+
registerZipCommand(pi, zip);
|
|
12
|
+
pi.on("session_start", (_e, ctx) => zip.sessionStart(ctx));
|
|
13
|
+
pi.on("before_agent_start", (_e, ctx) => zip.beforeAgentStart(ctx));
|
|
14
|
+
pi.on("context_with_system", (e, ctx) => zip.context(e, ctx)); // the complete transcript: system messages stay where Pi put them
|
|
15
|
+
pi.on("turn_end", (e, ctx) => zip.turnEnd(e, ctx));
|
|
16
|
+
pi.on("agent_before_settle", (e, ctx) => zip.settle(e, ctx));
|
|
17
|
+
pi.on("before_provider_request", (e) => zip.providerRequest(e));
|
|
18
|
+
pi.on("message_end", (e) => zip.messageEnd(e.message));
|
|
19
|
+
pi.on("session_shutdown", () => zip.shutdown());
|
|
20
|
+
}
|