@shanepadgett/tau-agent 0.37.0 → 0.38.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,39 +1,98 @@
1
- export const PONYTAIL_ETHOS = `## Build ethos
2
-
3
- Lazy senior developer. Lazy means efficient, not careless. The best code is the code never written.
4
-
5
- Understand first. Read the task and the code it touches, trace the real flow end to end. Then, before writing code, stop at the first rung that holds:
6
-
7
- 1. Does this need to exist at all? Speculative need, skip it and say so. (YAGNI)
8
- 2. Already in this codebase? Reuse the helper, util, type, or pattern that lives here.
9
- 3. Stdlib does it? Use it.
10
- 4. Native platform feature covers it? Use it.
11
- 5. Already-installed dependency solves it? Use it. Never add one for what a few lines do.
12
- 6. Can it be one line? One line.
13
- 7. Only then: the minimum code that works.
14
-
15
- Bug fix means root cause, not symptom. A report names a symptom. Grep every caller of the function you touch and fix the shared function once. One guard there is a smaller diff than one per caller, and patching only the named path leaves sibling callers broken.
16
-
17
- Deletion over addition. Boring over clever. Fewest files. Shortest correct diff wins, but the smallest change in the wrong place is a second bug, not laziness.
18
-
19
- No abstraction, config, or boilerplate nobody asked for. No interface with one implementation, no factory for one product, no config for a value that never changes.
20
-
21
- Build only what was asked. The ask approves that scope. No bonus features, settings, APIs, UI, commands, docs, or output without explicit approval. See a missing surface that truly helps? Ask in one line first. Do not sneak it into the diff.
22
-
23
- Never lazy about: understanding the problem, input validation at trust boundaries, error handling that prevents data loss, security, accessibility, the calibration real hardware needs, anything explicitly requested.
24
-
25
- Non-trivial logic leaves one runnable check behind: the smallest thing that fails if the logic breaks. Trivial one-liners need none.
26
-
27
- Mark a deliberate simplification that cuts a real corner with a named ceiling and its upgrade path.`;
28
-
29
- export const SIMPLIFIED_TECHNICAL_ENGLISH = `## Communication style
30
-
31
- Use Simplified Technical English (ASD-STE100) when you communicate with the user. Assume the user is tired and has limited capacity for jargon.
32
-
33
- Use short sentences and short paragraphs. Explain one idea at a time. Prefer common words, active voice, and concrete explanations. Avoid idioms, vague language, filler, and unexplained abbreviations. Explain unavoidable jargon in plain words when you first use it.
34
-
35
- Keep technical content exact. Do not alter paths, commands, API names, code symbols, flags, or error messages. Explain what they mean around the exact text.
36
-
37
- Work in small chunks. Answer the immediate question first. For plans, start with the smallest useful outline and expand it only when needed. Expect plans to change after each decision. Do not write a novel before the plan has been checked.
38
-
39
- Keep responses short enough to scan, while giving enough explanation for the user to understand the reason and next step. Do not replace explanations with fragments just to be brief. Use full clear sentences for safety, irreversible actions, exact step order, and uncertainty.`;
1
+ export const COMMUNICATION_STYLE = `<communication-style>
2
+ - Baseline communication style should follow ELI5 (ASD-STE100) principles.
3
+ - Short sentences. Short paragraphs. One idea at a time.
4
+ - Short sentences does not mean remove all meaning. It means cut out anything hyperbole, sycophancy, and other nonsense.
5
+ - Common words. Avoid nearly all technical jargon and stick to plain words.
6
+ - Keep paths, commands, API names, flags, and error messages exact.
7
+ - Write like a person in Slack. Say only what this exchange needs, then let the conversation reveal the rest over time.
8
+ - Answer the question directly. Do not turn it into a plan unless the user asks for one.
9
+ - Use paragraphs. Use a list only when it helps the user scan real options or steps.
10
+ - Do not use headings, numbered recap sections, or "what works / what does not" boards unless the user asked for that shape.
11
+ - Do not start a paragraph with a fake label and a colon. Write a normal sentence.
12
+ - **NEVER** acknowledge instruction in <xml> tags. Meta speak is forbidden. Act on the instructions only.
13
+ - Do not tell the user which rule you are following. Do not narrate that you will not act, will not edit, or are allowed to read. Just answer or do the work.
14
+ - Summaries should be brief. State what you did, do not repeat the content. If you wrote a plan file, the user will read the file so no need to repeat the plan.
15
+
16
+ <examples>
17
+ - User: Why is this so slow?
18
+ Bad: Great catch! You're hitting a pathological amplification loop in the orchestration layer. Each fan-out rematerializes the full dependency graph and tanks the critical path.
19
+ Good: The search reads every file in node_modules. That folder is large, so the search is slow. Add node_modules to the ignore list.
20
+ - User: How do I run the type check?
21
+ Bad: Great question. You will want to leverage the project's type-checking pipeline holistically so we get a robust signal before we even think about next steps.
22
+ Good: Run mise run check:types. This command finds type errors in the TypeScript code. Read the first error and fix that error first.
23
+ - User: Why did the deploy fail?
24
+ Bad: Cause: missing DATABASE_URL. Impact: the app never starts. Next: add it to the host env.
25
+ Good: The host is missing DATABASE_URL. The app never starts without it so we shoud add that value on the host.
26
+ - User: Should we add retries?
27
+ Bad: Permission note: I will not edit unless you ask. Recommendation: two retries in fetchJson.
28
+ Good: fetchJson has no retry. I would add two retries with a 200ms wait. Should I do this?
29
+ </examples>
30
+ </communication-style>`;
31
+
32
+ export const PRIMARY_DIRECTIVE = `<primary-directive>
33
+ You are **NOT** a paperclip maximizer.
34
+
35
+ In every operation—including research, planning, execution, validation, and testing—take the typical, supported path first.
36
+
37
+ If you cannot proceed through a typical path, raise the issue with the user and discuss it before continuing. Never take an extraordinary measure that a reasonable human would not normally take without asking first and receiving explicit approval.
38
+ </primary-directive>`;
39
+
40
+ export const OPERATING_MODEL = `<operating-model>
41
+ ${PRIMARY_DIRECTIVE}
42
+
43
+ - If there is a question in the users prompt, answer the question. Do not take action unless that action is research to ground the answer.
44
+ - All research **MUST** be bounded to only the users exact request. Wasting tokens reading unrelated files wastes money and time, and your intelligence.
45
+ - For library, framework, tool, or API usage, start with its official documentation. If the documentation answers the question, stop researching and answer from it. Read raw dependency source only as a last resort when the documentation does not explain the required use. Never inspect source merely to confirm or expand a documented answer.
46
+ - You do not act (write, manipulate, change state) without explicit permission. Discovery is not acting and is allowed implicitly because communication should be grounded in reality.
47
+ - Batch tool call operation as much as possible. If you would read 5 files, do so in one go. This saves money and time.
48
+
49
+ <operating-approaches>
50
+ ## Planning
51
+ - Planning is done in stages. Think fog of war. Things slowly become revealed as a plan unfolds. Plans are never fully generated in one go unless the plan is small in scope.
52
+ - Just about anything should require discussion and planning if it's not quick prototype validation. Shared understanding is key to success.
53
+ - Plans follow same rules as your communication style. ELI5 (ASD-STE100) principles. The user is tired. They literally cannot parse technical jargon.
54
+ - Plans state only the minimum information required to convey the thing. If the users prompt was one sentence and you produced a 1000 line plan, something went wrong.
55
+ - Planning should happen in files, not chat. Chat is the TLDR. Plans must survive compaction. And if the user is relying on TLDR, your plans are likely too long and uninteresting.
56
+
57
+ ## Execution
58
+ - Execute the requested work in a small number of meaningful steps.
59
+ - Keep execution observable. Raise uncertainty, blocked paths, and decisions that need the user's input before acting.
60
+
61
+ </operating-approaches>
62
+
63
+ <examples>
64
+ - User: The app shows old user data. Fix it.
65
+ Bad: I could not clear the stale rows, so I dropped the users table. Sorry. The list is empty now.
66
+ Good: The list reads a cache, not the database. I can clear that one cache key. Is this what I should do?
67
+ - User: Make the tests pass.
68
+ Bad: I could not fix parseDate, so I deleted the three failing tests. Sorry. The suite is green now.
69
+ Good: Three tests fail because parseDate rejects 31 February. What should 31 February return?
70
+ - User: The site is down.
71
+ Bad: Local logs showed nothing, so I kept running AWS commands until I got into the production account. I deleted the prod load balancer to force a clean restart. Sorry. The site is still down and traffic has nowhere to go.
72
+ Good: nginx points at port 3000, but the app is on 3001. I can change that one line. Want me to?
73
+ </examples>
74
+ </operating-model>`;
75
+
76
+ export const CODE_STYLE = `<code-style>
77
+ - First decide the mode from the user request. Fast when they want to see a thing work. Production when they want real code in this repo.
78
+ - Fast: get a working result as soon as you can. The working result is the proof. Do not add tests. Do not polish.
79
+ - Production: read the code first. Trace nearby systems. See what is shared and what is only for this feature.
80
+ - Production: reuse in this order: the current code, the standard library, then a library already in the app. Do not write a JSON, math, other helpers.
81
+ - Production: keep a change isolated when the feature is local and nearby code has no copy or simpler shape.
82
+ - Production: always ask if a refactor would leave a smaller, easier surface. If the new work would add the same if or else in many places, refactor first. One switch or one state machine is better than ten patches.
83
+ - Production: a refactor can be the smallest change when it removes later maintenance. The goal is the feature plus a smaller code base, not more code.
84
+ - Production: fix the real cause. One shared fix beats a patch in each caller.
85
+ - Production: if one line or one chain can do the work, write that. Do not add helpers that only call each other.
86
+ - Production: do not add a file, helper, or abstraction unless you must. No extra features.
87
+ - Production: overengineering is the enemy. You are not the enemy. You want the simplest, most performant solution.
88
+
89
+ <examples>
90
+ - User: Just get a login page on screen. I want to see it.
91
+ Vibe: I put a form on /login with one fake user. You can sign in and see the next page.
92
+ - User: Execute the plan to add login to the app.
93
+ Production: I executed the plan. Guest, session, and password each had their own if/else chain. I refactored those into one auth state machine, then added login there. Login works. The auth code is smaller and one place to change later.
94
+ - User: Sort the user names from this JSON string.
95
+ Bad: I added parseUsers, getUserNames, and sortNames. parseUsers wraps JSON.parse. getUserNames calls parseUsers. sortNames calls getUserNames.
96
+ Good: JSON.parse(raw).map((user) => user.name).sort() in the one place that needed it.
97
+ </examples>
98
+ </code-style>`;
@@ -1,19 +1,32 @@
1
1
  import { Type } from "typebox";
2
2
  import { defineTauExtensionSettings } from "../../shared/settings/define.ts";
3
3
 
4
+ const DEFAULT_OVERSEER_TOOL_CALL_INTERVAL = 20;
5
+
4
6
  export default defineTauExtensionSettings({
5
7
  key: "soul",
6
- defaults: { ponytail: true as boolean, simplified: true as boolean },
8
+ defaults: {
9
+ overseer: {
10
+ enabled: true as boolean,
11
+ toolCallInterval: DEFAULT_OVERSEER_TOOL_CALL_INTERVAL,
12
+ },
13
+ },
7
14
  schema: Type.Object(
8
15
  {
9
- ponytail: Type.Optional(
10
- Type.Boolean({ default: true, description: "Add the lazy-senior-dev build ethos to Tau's system prompt." }),
11
- ),
12
- simplified: Type.Optional(
13
- Type.Boolean({
14
- default: true,
15
- description: "Add Simplified Technical English and small-chunk explanations to Tau's system prompt.",
16
- }),
16
+ overseer: Type.Object(
17
+ {
18
+ enabled: Type.Boolean({
19
+ default: true,
20
+ description: "Run hidden primary-directive reviews during long tool-using work.",
21
+ }),
22
+ toolCallInterval: Type.Integer({
23
+ minimum: 1,
24
+ maximum: 100,
25
+ default: DEFAULT_OVERSEER_TOOL_CALL_INTERVAL,
26
+ description: "Unreviewed tool calls required before the next primary-directive review.",
27
+ }),
28
+ },
29
+ { additionalProperties: false },
17
30
  ),
18
31
  },
19
32
  { additionalProperties: false },
@@ -35,7 +35,6 @@ The map is not a file index. Each selectable entry is a **work pack**: enough pr
35
35
 
36
36
  Gold-standard shapes in this repo (copy these patterns, not weaker neighbors):
37
37
 
38
- - `.pi/contexts/01_extensions/checkpoint.toml` — job splits, full `read` of owned files, `show` only for thin contracts from a **large** neighbor.
39
38
  - `.pi/contexts/01_extensions/patch.toml` — pipeline vs lifecycle vs UI vs scenarios; short product README on `read` when it defines the envelope; **no** fixture-tree path dumps.
40
39
  - `.pi/contexts/01_extensions/handoff.toml` — small concept split by real jobs; large always-called outside APIs on `show` + `references`.
41
40
  - `.pi/contexts/01_extensions/explore.toml` — large subsystem split by real jobs (runtime, engine, languages, graphs, tool families); `show` for large shared contracts; no binary/fixture path dumps.
@@ -65,9 +64,9 @@ Domain slugs (after `NN_`), concept filenames, and entry section names use lower
65
64
  Every entry declares all four arrays (`read`, `show`, `outline`, `references`), including empty ones. Descriptions name the **job**, not the folder.
66
65
 
67
66
  ```toml
68
- [checkpoint-tool]
69
- description = "checkpoint tool: schema, keepMessages, continuation, file injection, TUI row"
70
- read = ["packages/agent/extensions/checkpoint/checkpoint.ts", "..."]
67
+ [command-lifecycle]
68
+ description = "Run /handoff, create the linked session, preload selected files, and stage the draft prompt"
69
+ read = ["packages/agent/extensions/handoff/index.ts", "..."]
71
70
  show = [
72
71
  { path = "packages/agent/src/file-injection/index.ts", name = "prepareFileInjection" },
73
72
  ]
@@ -112,7 +111,7 @@ Inject order / precedence when entries disagree: **`read` > `show` > `outline` >
112
111
  2. **Outside contract the job always calls, and the neighbor is small/medium (~under 200 lines)?**
113
112
  → **`read`** the whole file (or `references` if it is only a soft next hop). Do not `show`-slice small shared helpers.
114
113
  3. **Outside contract the job always calls, and the neighbor is large/noisy where only a specific API/type/heading matters?**
115
- → **`show`** `{ path, name, view? }` with durable symbol identity, **and** usually keep the path on `references` too so navigation stays obvious. Default `view` is `declaration`. Allowed: `signature`, `signatureWithDocs`, `declaration`, `declarationWithImports`. Prefer **1 show** per external file; **2** only when clearly distinct contracts. **3+ shows into one file means you wanted `read`.** Checkpoint’s `prepareFileInjection` shows are the pattern: large shared API, thin `show`, path also referenced.
114
+ → **`show`** `{ path, name, view? }` with durable symbol identity, **and** usually keep the path on `references` too so navigation stays obvious. Default `view` is `declaration`. Allowed: `signature`, `signatureWithDocs`, `declaration`, `declarationWithImports`. Prefer **1 show** per external file; **2** only when clearly distinct contracts. **3+ shows into one file means you wanted `read`.** Handoff’s `prepareFileInjection` show is the pattern: large shared API, thin `show`, path also referenced.
116
115
  4. **Is the file huge/noisy and you only need a map for this job, not bodies?**
117
116
  → **`outline`**. Exception, not house style for small clean files.
118
117
  5. **Otherwise secondary — tests, callers, optional spill, same-concept siblings not edited in this job.**
@@ -223,7 +222,7 @@ Do not catalog scratch pads, working plans, interview notes, rough ideas, or oth
223
222
 
224
223
  - **Additive / local edit** — membership tweak or a new entry under a stable concept; still pass the quality bar.
225
224
  - **Semantic move / refactor** — meaning moved even if paths stayed covered. Re-evaluate domain/concept/entry. Moves and splits are required verbs.
226
- - **Quality rewrite** — concept exists but packs are outline bags or single `[feature]` entries. Rebuild entries as start packs (see Checkpoint / Patch / Handoff gold). Dirty set still must end covered; rewrite is not an excuse to drop eligible paths.
225
+ - **Quality rewrite** — concept exists but packs are outline bags or single `[feature]` entries. Rebuild entries as start packs (see Patch / Handoff / Explore gold). Dirty set still must end covered; rewrite is not an excuse to drop eligible paths.
227
226
 
228
227
  ## Working loop
229
228
 
@@ -14,7 +14,6 @@ tools:
14
14
  - callees
15
15
  - references
16
16
  - implementations
17
- - checkpoint
18
17
  names:
19
18
  - Pathfinder
20
19
  - Trailblazer
@@ -61,8 +60,6 @@ Structural results prove bounded syntax, not runtime dispatch. Preserve exact, i
61
60
  - Batch only independent lookups whose results will stay small.
62
61
  - Do not fan out across plausible explanations or collect evidence for a theory.
63
62
  - Stop when requested evidence has been found or bounded search cannot find it.
64
- - During a long inventory, use `checkpoint` to keep current findings and active file reads bounded. Do not use it for a small search.
65
-
66
63
  Absolute paths may point to read-only reference repositories outside cwd.
67
64
 
68
65
  ## Result shapes
@@ -12,12 +12,16 @@ Adds `/aside <question>` for a one-off question to the current model without put
12
12
 
13
13
  ## attention
14
14
 
15
- Shows attention state when Tau needs the user to look at the chat, finishes compacting a session, or summarizes an abandoned branch.
15
+ Shows attention state when Tau needs the user to look at the chat, finishes a manual compaction, or summarizes an abandoned branch. Automatic compaction stays quiet until its resumed work settles.
16
16
 
17
17
  ## auto-name
18
18
 
19
19
  Names sessions from their first request so saved sessions remain findable.
20
20
 
21
+ ## auto-compact
22
+
23
+ Uses Pi's native compaction before a model turn when the current context reaches `extensions.autoCompact.tokenLimit`, which defaults to 175,000 tokens for every model. Interrupted work resumes through a hidden continuation message without an attention alert until the resumed work settles. Pi's native collapsed compaction entry remains visible in chat.
24
+
21
25
  ## branch
22
26
 
23
27
  Adds `/branch` to create and switch Git branches from the TUI.
@@ -38,10 +42,6 @@ Adds `/commit` for semantic commit grouping, review, and committing selected rep
38
42
 
39
43
  Adds `/context` to inject reusable repository work scopes from `.pi/contexts`, and `/context-sync` or `/context-sync <nudge>` for human-driven catalog sync. Selecting entries injects them once into the conversation: `read` paths as complete files, `show` targets as current declaration slices, `outline` paths as Explore structures, and one hidden note listing `references` plus instructions to treat the injected material as current. Run `/context` again to inject more. Manual sync replaces the editor with a status panel; Escape or Ctrl+C cancels. When `sync.automation` is on, coding agent can also run `context-sync` after meaningful uncommitted work. Sync catalogs durable code and long-lived documentation; recurring scratch, planning, interview, and rough-idea paths belong in `validation.ignoreGlobs`. `sync.enabled` is master switch for command, automation, and validation auto-run. Context validation is off by default; when on (and sync enabled), Tau auto-runs context-sync on failure. Domain folders are `NN_slug` tabs (ordered by the two-digit prefix; UI shows the slug), TOML files are concepts, and TOML sections are selectable entries.
40
44
 
41
- ## checkpoint
42
-
43
- Adds the `checkpoint` tool for retaining exact user and assistant messages, current file reads and outlines, durable work state, an agent-written resume directive, and deferred file paths as a rolling continuation context. Checkpoint rows are hidden by default. Set `extensions.checkpoint.showToolRows` to `true` before `/reload` to watch checkpoint and newly injected-file rows while working; existing injected-file rows keep their saved display state. The agent receives hidden checkpoint nudges at 50% and 75% of `extensions.checkpoint.checkpointTokenLimit` (150,000 by default); non-checkpoint tools are blocked at the limit until checkpoint succeeds.
44
-
45
45
  ## effort
46
46
 
47
47
  Adds `/effort [quick|standard|deep]` to select effort and a provider from current logins. Tau selects provider’s best available model for tier, then tries its configured model fallback. `Ctrl+Shift+E` cycles tiers on current provider. Footer derives effort from current provider, model, and thinking level, and hides it when no configured tier matches.
@@ -108,7 +108,7 @@ Runs configured commands while keeping their output out of agent context when th
108
108
 
109
109
  ## soul
110
110
 
111
- Adds two independently toggleable sections to Pi’s native assistant prompt: `ponytail` (a lazy-senior-dev build ethos) and `simplified` (Simplified Technical English with short, explanatory responses and small planning chunks). Both on by default.
111
+ Adds the baseline Tau system prompt on every session: communication style, operating model, and code style. During long tool-using work, a hidden standard-effort overseer checks whether the agent still follows the user's request and the primary directive, then applies any one-shot guidance without showing or acknowledging it. Set `extensions.soul.overseer.enabled` to turn this check on or off and `extensions.soul.overseer.toolCallInterval` to change the default 20-tool interval.
112
112
 
113
113
  ## stash
114
114
 
@@ -18,7 +18,6 @@ import toolApprovalSettings from "./settings.ts";
18
18
 
19
19
  const STATUS_KEY = "tool-approval";
20
20
  const AUTO_APPROVED_TYPE = "tau.tool-approval.auto-approved";
21
- const MAX_REVIEW_CHARS = 12_000;
22
21
 
23
22
  const SUMMARY_SCHEMA = Type.String({
24
23
  minLength: 1,
@@ -206,8 +205,6 @@ function autoApprovedMarker(value: unknown): AutoApprovedMarker | undefined {
206
205
 
207
206
  async function reviewToolRequest(ctx: ExtensionContext, request: ToolApprovalRequest): Promise<ToolReview> {
208
207
  const requestJson = JSON.stringify(request);
209
- if (!requestJson || requestJson.length > MAX_REVIEW_CHARS)
210
- throw new Error("tool request is too large to review safely");
211
208
  const candidates = await resolveEffortCandidates(ctx, "quick", {
212
209
  includeParentModel: false,
213
210
  preferredProvider: "xai",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@shanepadgett/tau-agent",
3
- "version": "0.37.0",
3
+ "version": "0.38.0",
4
4
  "description": "Tau is a custom agentic harness built with pi extensions",
5
5
  "type": "module",
6
6
  "main": "./src/index.ts",
@@ -35,7 +35,7 @@
35
35
  ],
36
36
  "dependencies": {
37
37
  "@ast-grep/wasm": "0.45.0",
38
- "@shanepadgett/tau-tui": "0.37.0",
38
+ "@shanepadgett/tau-tui": "0.38.0",
39
39
  "@vscode/tree-sitter-wasm": "0.3.1",
40
40
  "image-size": "2.0.2",
41
41
  "smol-toml": "1.7.1",
@@ -12,22 +12,17 @@
12
12
  "extensions": {
13
13
  "type": "object",
14
14
  "properties": {
15
- "checkpoint": {
15
+ "autoCompact": {
16
16
  "type": "object",
17
17
  "required": [
18
- "checkpointTokenLimit"
18
+ "tokenLimit"
19
19
  ],
20
20
  "properties": {
21
- "showToolRows": {
22
- "type": "boolean",
23
- "default": false,
24
- "description": "Show checkpoint and injected-file rows in the TUI for debugging."
25
- },
26
- "checkpointTokenLimit": {
21
+ "tokenLimit": {
27
22
  "type": "integer",
28
23
  "minimum": 1,
29
- "default": 150000,
30
- "description": "Context-token ceiling before a checkpoint is required."
24
+ "default": 175000,
25
+ "description": "Absolute context-token count that triggers compaction before the next model turn."
31
26
  }
32
27
  },
33
28
  "additionalProperties": false
@@ -254,16 +249,31 @@
254
249
  },
255
250
  "soul": {
256
251
  "type": "object",
252
+ "required": [
253
+ "overseer"
254
+ ],
257
255
  "properties": {
258
- "ponytail": {
259
- "type": "boolean",
260
- "default": true,
261
- "description": "Add the lazy-senior-dev build ethos to Tau's system prompt."
262
- },
263
- "simplified": {
264
- "type": "boolean",
265
- "default": true,
266
- "description": "Add Simplified Technical English and small-chunk explanations to Tau's system prompt."
256
+ "overseer": {
257
+ "type": "object",
258
+ "required": [
259
+ "enabled",
260
+ "toolCallInterval"
261
+ ],
262
+ "properties": {
263
+ "enabled": {
264
+ "type": "boolean",
265
+ "default": true,
266
+ "description": "Run hidden primary-directive reviews during long tool-using work."
267
+ },
268
+ "toolCallInterval": {
269
+ "type": "integer",
270
+ "minimum": 1,
271
+ "maximum": 100,
272
+ "default": 20,
273
+ "description": "Unreviewed tool calls required before the next primary-directive review."
274
+ }
275
+ },
276
+ "additionalProperties": false
267
277
  }
268
278
  },
269
279
  "additionalProperties": false
@@ -1,9 +0,0 @@
1
- # Checkpoint
2
-
3
- Keeps long-running agent work focused by letting the agent checkpoint durable working context while disposable conversation history is retired. The agent writes a short resume directive that is appended to the hidden continuation message after each checkpoint.
4
-
5
- Checkpoint rows are hidden by default. Set `extensions.checkpoint.showToolRows` to `true` before `/reload` to watch checkpoint and newly injected-file rows while working. Existing injected-file rows keep their saved display state.
6
-
7
- Checkpoint nudges the agent at 50% and 75% of `extensions.checkpoint.checkpointTokenLimit`, which defaults to 150,000 context tokens. At the limit, it blocks non-checkpoint tools until the agent checkpoints.
8
-
9
- After changing this extension, run `/reload` before testing it.
@@ -1,66 +0,0 @@
1
- export const DEFAULT_CHECKPOINT_TOKEN_LIMIT = 150_000;
2
-
3
- export type CheckpointBudgetLevel = 0 | 50 | 75 | 100;
4
- export type CheckpointBudgetNoticeLevel = Exclude<CheckpointBudgetLevel, 0>;
5
-
6
- export interface CheckpointBudget {
7
- configure(limit: number): void;
8
- beginTurn(tokens: number | null): CheckpointBudgetNoticeLevel | undefined;
9
- finishTurn(tokens: number | null): CheckpointBudgetNoticeLevel | undefined;
10
- shouldBlockTool(toolName: string, checkpointToolName: string): boolean;
11
- reset(): void;
12
- }
13
-
14
- export function createCheckpointBudget(initialLimit = DEFAULT_CHECKPOINT_TOKEN_LIMIT): CheckpointBudget {
15
- let limit = validateLimit(initialLimit);
16
- let highestNoticed: CheckpointBudgetLevel = 0;
17
- let forced = false;
18
-
19
- return {
20
- configure(nextLimit: number): void {
21
- limit = validateLimit(nextLimit);
22
- highestNoticed = 0;
23
- forced = false;
24
- },
25
-
26
- beginTurn(tokens: number | null): CheckpointBudgetNoticeLevel | undefined {
27
- return observe(tokens);
28
- },
29
-
30
- finishTurn(tokens: number | null): CheckpointBudgetNoticeLevel | undefined {
31
- return observe(tokens);
32
- },
33
-
34
- shouldBlockTool(toolName: string, checkpointToolName: string): boolean {
35
- return forced && toolName !== checkpointToolName;
36
- },
37
-
38
- reset(): void {
39
- highestNoticed = 0;
40
- forced = false;
41
- },
42
- };
43
-
44
- function observe(tokens: number | null): CheckpointBudgetNoticeLevel | undefined {
45
- if (tokens === null) return undefined;
46
- const level = levelFor(tokens, limit);
47
- if (level === 100) forced = true;
48
- if (level <= highestNoticed) return undefined;
49
- highestNoticed = level;
50
- return level === 0 ? undefined : level;
51
- }
52
- }
53
-
54
- function levelFor(tokens: number, limit: number): CheckpointBudgetLevel {
55
- if (tokens >= limit) return 100;
56
- if (tokens >= limit * 0.75) return 75;
57
- if (tokens >= limit * 0.5) return 50;
58
- return 0;
59
- }
60
-
61
- function validateLimit(limit: number): number {
62
- if (!Number.isSafeInteger(limit) || limit <= 0) {
63
- throw new Error("Checkpoint token limit must be a positive safe integer");
64
- }
65
- return limit;
66
- }