pi-extended-teams 2.2.10 → 2.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. package/README.md +182 -2
  2. package/extensions/agents/read-agent-report.ts +11 -2
  3. package/extensions/agents/read-agent-session-lifecycle.ts +25 -10
  4. package/extensions/agents/read-agent.ts +387 -65
  5. package/extensions/agents/write-agent.ts +3 -0
  6. package/extensions/events/register-events.ts +22 -10
  7. package/extensions/index.ts +132 -38
  8. package/extensions/internal/pi-check-operations.ts +24 -0
  9. package/extensions/internal/session-files.ts +10 -2
  10. package/extensions/runtime/teammate-interrupt.ts +11 -5
  11. package/extensions/runtime/types.ts +4 -0
  12. package/extensions/team/lifecycle.ts +26 -0
  13. package/extensions/team/session-cost.ts +155 -0
  14. package/extensions/tools/agent-communication-tools.ts +24 -8
  15. package/extensions/tools/agent-status-tool.ts +50 -7
  16. package/extensions/tools/coordination-tools.ts +78 -8
  17. package/extensions/tools/task-runtime-tools.ts +23 -4
  18. package/extensions/tools/team-tools.ts +287 -40
  19. package/extensions/ui/activity-colors.ts +59 -0
  20. package/extensions/ui/agent-follow-view.ts +15 -11
  21. package/extensions/ui/agent-navigation.ts +35 -1
  22. package/extensions/ui/checkpoints-command.ts +28 -0
  23. package/extensions/ui/onboarding-command.ts +218 -0
  24. package/extensions/ui/status-widget.ts +22 -14
  25. package/package.json +1 -1
  26. package/skills/teams.md +16 -0
  27. package/src/orchestration/index.ts +19 -1
  28. package/src/orchestration/status-projection.ts +2 -3
  29. package/src/orchestration/types.ts +9 -1
  30. package/src/results/check-journal.ts +149 -0
  31. package/src/results/check-policy.ts +86 -0
  32. package/src/results/check-runner.ts +110 -0
  33. package/src/results/checkpoint-assignment.ts +60 -0
  34. package/src/results/checkpoint-policy.ts +20 -0
  35. package/src/results/checkpoint-report.ts +109 -0
  36. package/src/results/checkpoint-retention.ts +15 -0
  37. package/src/results/completion-group-delivery.ts +66 -0
  38. package/src/results/completion-group-recovery.ts +61 -0
  39. package/src/results/completion-group-wake.ts +62 -0
  40. package/src/results/completion-group.ts +290 -0
  41. package/src/results/continuation-recipient.ts +37 -0
  42. package/src/results/durable-json.ts +29 -0
  43. package/src/results/repair-policy.ts +24 -0
  44. package/src/results/report-result.ts +77 -0
  45. package/src/results/source-identity.ts +115 -0
  46. package/src/results/specialist-checkpoint.ts +227 -0
  47. package/src/results/verification-controller.ts +308 -0
  48. package/src/utils/messaging.ts +47 -10
  49. package/src/utils/models.ts +13 -0
  50. package/src/utils/paths.ts +4 -0
  51. package/src/utils/report-events.ts +95 -10
  52. package/src/utils/teams.ts +3 -1
  53. package/src/utils/write-queue.ts +3 -0
package/README.md CHANGED
@@ -30,13 +30,21 @@ Or install directly from GitHub:
30
30
  pi install git:github.com/dantetekanem/pi-extended-teams
31
31
  ```
32
32
 
33
+ For the first run, start with:
34
+
35
+ ```text
36
+ /pi-extended-teams-onboard
37
+ ```
38
+
39
+ On a new installation with no pi-extended-teams settings, a startup notice points to this command. It gives the current agent a read-only snapshot of the models, favorite tiers, shared extensions, and package source available in that Pi session. The agent recommends a setup, shows the exact changes it wants to make, and explains how to update the installed package. Nothing changes until you approve it. Run the command again whenever your models or extensions change.
40
+
33
41
  Then ask for help naturally:
34
42
 
35
43
  ```text
36
44
  Review the current changes with separate agents for correctness, test gaps, and security. Give me the evidence so I can make the final call.
37
45
  ```
38
46
 
39
- The current Pi session becomes the agent group automatically. There is no separate setup step. Until you configure a tier, it uses the current lead-session model and thinking level. Run `/agents-favorite-models` when you want particular tiers to use different models.
47
+ The current Pi session becomes the agent group automatically. Setup remains optional: an unset tier inherits the current lead-session model and thinking level. Use `/agents-favorite-models` to configure tiers directly and `/agents-extensions` to choose which observable loaded extensions spawned agents receive.
40
48
 
41
49
  ## How it works
42
50
 
@@ -67,6 +75,156 @@ Completed reports wake the lead automatically. End the current turn to wait. One
67
75
 
68
76
  The lead owns decomposition, integration, and acceptance. Pi packages and spawned agents run with your system permissions, so review project-local instructions and configuration through Pi's normal trust flow.
69
77
 
78
+ ## Task outcomes and full reports
79
+
80
+ A final report or clean exit does not mean the task succeeded. Agents can include an explicit outcome with their full report:
81
+
82
+ ```text
83
+ report_and_exit({
84
+ content: "The implementation needs a product decision. Full findings follow…",
85
+ summary: "API decision needed",
86
+ outcome: "blocked",
87
+ questions: ["Should the endpoint require authentication?"]
88
+ })
89
+ ```
90
+
91
+ Optional outcomes are `succeeded`, `blocked`, `failed`, and `cancelled`. Reports can also include `changedPaths`, `artifacts` (`path`, optional `label`), and `findings` (`id`, `text`, `evidence`). Finding IDs must be unique within the report. These are agent-reported claims and references, not independent verification.
92
+
93
+ New reports store a versioned `result` with runtime-assigned task, run, and report IDs in `reports.json`; the full Markdown report remains unchanged. Repeating the same run's report preserves the first stored report. Plain reports remain supported, and omitted outcomes stay unspecified.
94
+
95
+ Without assigned checks, verification is `not-requested`. Lead acceptance starts as `pending`. Agent-reported claims cannot change either state. Trusted integrations can record an explicit decision through `recordReportAcceptance(teamName, reportId, "accepted" | "rejected", reason?)` in `src/utils/report-events.ts`. Final-report tools cannot assign identities, verification, or acceptance.
96
+
97
+ `get_agent_status` shows task outcome separately from lifecycle status. The lead can recover the full persisted result through `check_teammate` after the agent leaves the roster.
98
+
99
+ ## Assigned checks
100
+
101
+ The lead can attach explicit checks to `spawn_agent`, individual swarm agents, or swarm defaults:
102
+
103
+ ```text
104
+ spawn_agent({
105
+ name: "result-review",
106
+ model_slot: "read-review",
107
+ prompt: "Review task outcomes without editing files.",
108
+ checks: [{
109
+ name: "result-contract",
110
+ command: "pnpm --config.verify-deps-before-run=false exec vitest run src/results/report-result.test.ts",
111
+ timeoutSeconds: 60
112
+ }]
113
+ })
114
+ ```
115
+
116
+ Each check needs a unique name, a command, and a finite positive per-command timeout in seconds. Commands run through Pi's native local BashOperations in the agent's cwd, before its final report closes the recipient. Swarm agents inherit defaults; `checks: []` disables that inheritance. Nested helpers cannot assign checks, and report fields or metadata cannot authorize commands. Trusted spawn integrations receive `checks` on the orchestration request and must bind them as the admitted member's `assignedChecks`.
117
+
118
+ Verification records the actual exit, full output, and source before and after execution. Fingerprints include Git HEAD/index metadata, tracked working-file bytes, and nonignored untracked files. Optional `inputs` are literal paths relative to the agent's cwd, contained within its repository; omitting them covers the repository. Unsupported source inputs or Pi runtimes fail visibly rather than falling back to another command runner.
119
+
120
+ Check records and private full logs live under `~/.pi/teams/<team>/checks/`. Report IDs reference their check IDs. `listTeamReportEvents` and orchestration/status reads recheck current source and expose stale verification without rewriting historical evidence. After roster removal, `check_teammate` retrieves the report, observed check records, and full-log paths. Ordinary lead notifications include verification state, not full logs.
121
+
122
+ Duplicate submissions do not rerun commands. An unresolved execution claim is not replayed and prevents clean finalization. In-process interruption and shutdown cancel assigned checks and wait for raw settlement; nonsettling operations remain quarantined without releasing claims. Failed checks do not imply a failed lifecycle, overwrite the reported task outcome, or grant lead acceptance. Automatic repair is disabled unless explicitly authorized.
123
+
124
+ These fingerprints are observations, not snapshots or workspace isolation. They do not capture ignored/external inputs, external services, or edits restored between observations. Checks and agents share the host's permissions; this is not a sandbox against other same-user processes.
125
+
126
+ ## Optional bounded repair
127
+
128
+ Add `repair: { maxAttempts: 1 }` to a spawn with assigned `checks` to allow one additional repair attempt after initial verification. The limit is zero to five; zero disables repair. Swarm defaults are inherited, and an agent can override them with `{ maxAttempts: 0 }`. Only the lead or a trusted integration can authorize repair. Metadata, report parameters, and nested helpers cannot enable it. Trusted integrations bind the request's `repair` as the member's `repairPolicy`.
129
+
130
+ A failed check returns its observed exit and full-log reference to the responsible agent. A `report_and_exit` repair receipt has `accepted: false` and a `repairRequest`; the agent remains active, repairs only its assigned scope, and resubmits. Repair does not grant edit permissions: read agents must report a blocker when edits are needed, and edit agents must keep or reacquire claims before changing files. Plaintext reporting uses a separate repair turn after current work settles. An accepted final-report receipt is still separate from lead acceptance.
131
+
132
+ The harness reruns failed checks and checks whose scoped inputs changed. It reuses passed evidence only when the current scoped source matches. Runtime submission IDs prevent duplicate reports from consuming another attempt. Reservations and decisions are persisted before check execution or repair feedback; cancellation prevents further automatic attempts. A pending native claim or controller reservation keeps cleanup fenced even after in-process state is lost. A stop may therefore report blocked cleanup rather than claim the agent stopped.
133
+
134
+ Repair history lives under `~/.pi/teams/<team>/repairs/`. The report's `repair` includes its controller ID, ledger path, state, and available attempt counts/request IDs. Missing or corrupt records remain pending with an error; unknown counts are omitted. `listTeamReportEvents`, status, and report recovery read authoritative repair/check evidence without rewriting history. `readStoredTeamReportEvent` retrieves an exact historical record without observing source.
135
+
136
+ Exhausted, declined, cancelled, or otherwise blocked repair produces an effective blocker while preserving the agent's reported `outcome`. `effectiveTaskOutcome(result)` exposes that distinction to trusted consumers. Passing checks never invent a success claim or grant lead acceptance. Existing reports and spawns without a repair policy retain their behavior.
137
+
138
+ ## Optional grouped reports
139
+
140
+ The lead can request compact reports for a batch:
141
+
142
+ ```text
143
+ spawn_swarm_agents({
144
+ completion_group: { delivery: "all-settled" },
145
+ defaults: { model_slot: "read-review" },
146
+ agents: [
147
+ { name: "correctness", prompt: "Review correctness without editing." },
148
+ { name: "tests", prompt: "Review test coverage without editing." }
149
+ ]
150
+ })
151
+ ```
152
+
153
+ Omitting `completion_group` keeps immediate full-report delivery. `delivery: "immediate"` sends compact member indexes; `"all-settled"` waits for the batch, including queued members. Rejected admissions, cancellations and interruptions have explicit states. Blockers, runtime failures and failed/stale verification can request an early wake. An urgent last result does not require a redundant final wake.
154
+
155
+ Reported members' indexes include task/run/slot identity, reported and effective outcomes, verification, acceptance, findings/questions, and a full-report ID/path. Unadmitted assignments retain their slot/state/reason without invented run IDs or results. Read the referenced Markdown when more evidence is needed; it survives private transcript deletion. Index verification is a stored snapshot, so recheck current source-bound evidence before accepting work. Group settlement never grants task success or lead acceptance. Nested reports still target their exact parent, and workflow/pi-prompt suppression excludes lead-facing evidence and wakes.
156
+
157
+ Journals live under `~/.pi/teams/<team>/completion-groups/`. A sibling `<inbox>.json.durable` marker keeps an opted-in inbox's later writes synchronized even after its last grouped index is removed; never-grouped inboxes retain ordinary atomic writes. The harness binds membership before admission and actual run IDs before launch; agent metadata cannot authorize grouping. Reload preserves explicit terminal states, records lost unadmitted work as interrupted, and leaves uncertain running ownership unresolved. It does not restart work, revive recipients or release claims.
158
+
159
+ A durable inbox index and a wake request have separate receipts. Pi's custom-message call does not return an admission acknowledgment. A reserved request remains `pending` until exact session history is `observed`; that observation is not provider success or a power-loss guarantee. Ambiguous requests are not automatically repeated after errors or reload. Use `read_inbox` for a saved index when a warning reports an unconfirmed wake. Grouped correlation requires Pi's custom-message/history APIs; ordinary delivery keeps its existing fallback.
160
+
161
+ ### Measured delivery replay
162
+
163
+ One September 8, 2026 comparison used Pi 0.85.1 and configured `read-review` model `openai-codex/gpt-5.6-terra`, high thinking. Two fresh SDK sessions synthesized the same ten supplied source-backed reports, first immediate/full, then all-settled/compact. SSE transport, retries and compaction settings were identical; retries and compaction were disabled.
164
+
165
+ | Observation | Immediate/full | All-settled/compact |
166
+ | --- | ---: | ---: |
167
+ | Wake requests | 10 | 1 |
168
+ | Model requests / assistant responses | 11 | 3 |
169
+ | Provider input tokens, excluding cache | 17,796 | 5,149 |
170
+ | Cache-read tokens | 17,408 | 5,120 |
171
+ | Cache-write tokens | 0 | 0 |
172
+ | Output tokens | 293 | 237 |
173
+ | Full-report retrievals | 0 | 1 |
174
+ | Elapsed time | 26.239s | 10.748s |
175
+
176
+ Both syntheses preserved all ten required finding/control pairs, pending acceptance and the exact detail retrieved from full report F9. Each session emitted one `agent_settled` event; wake requests are not equivalent to completed agent runs. These are actual provider usage figures from one controlled replay, not estimates, randomized statistics, new specialist investigations or a general review-speed guarantee. The [measurement record](docs/grouped-report-measurement.json) retains the assignment, complete corpus, tested source identity, method and observations without session histories or credentials.
177
+
178
+ ## Optional specialist continuation
179
+
180
+ Save a specialist's findings when you expect a later follow-up:
181
+
182
+ ```text
183
+ spawn_agent({
184
+ name: "auth-review",
185
+ model_slot: "read-review",
186
+ prompt: "Review auth and config. Report stable finding IDs, evidence and inspected source references.",
187
+ checkpoint: { inputs: ["auth", "config"], retentionDays: 30 }
188
+ })
189
+ ```
190
+
191
+ The checkpoint preserves the original assignment, scoped source observations, author/run/tier, findings, questions, reported `inspectedEvidence`, lead-supplied `decisions`, and independent report references. It keeps the original report plus bounded recent history. Records are limited to 64 KiB and live in `~/.pi/agent/checkpoints/`, outside private transcripts and Pi's resume picker. Omit both `checkpoint` and `continue_from` to keep ordinary spawning unchanged; metadata and nested helpers cannot enable them. Swarm defaults and individual agents also accept `checkpoint`; put `continue_from` on each selected agent.
192
+
193
+ Copy the exact checkpoint ID from the saved report into a new assignment:
194
+
195
+ ```text
196
+ spawn_agent({
197
+ name: "auth-followup",
198
+ model_slot: "read-review",
199
+ continue_from: "checkpoint:<saved hash>",
200
+ prompt: "Recheck F1 and F2 after the tenant fix. Revalidate token-policy dependencies, including unchanged callers."
201
+ })
202
+ ```
203
+
204
+ `name` becomes a prefix for a fresh recipient. Continuation creates a new run and SDK session; it does not reopen the old mailbox or transcript, resume a process, transfer claims, or inherit execution permissions or repair budgets. The current prompt, cwd, tier, instructions, checks and repair policy govern. Old findings, verification and acceptance remain historical claims. Edit agents must acquire their own claims normally.
205
+
206
+ Continuation inherits the checkpoint's input scope and retention unless you supply a current `checkpoint` override; old lead decisions are not reauthorized. Scope entries are literal paths relative to the current cwd, contained within its repository. At actual launch, including after queue delays, the harness reloads the checkpoint and compares each retained report's dependency scope. Changed, uncertain or changed-during-investigation evidence requires revalidation beyond the diff. An unchanged fingerprint is not proof that a finding still holds, a dependency graph or a snapshot; ignored/external inputs and restored edits remain outside its evidence.
207
+
208
+ Full reports are synchronized independently before checkpoint publication and cleanup. Missing known report provenance or uncertain publication prevents destructive cleanup. In-process read and edit admission supports checkpoints; legacy terminal startup/queue requests reject checkpoint and continuation options. Existing legacy report producers can preserve checkpoint-bound reports. Trusted orchestration requests use `continueFrom`; a custom start callback still owns actual admission and execution.
209
+
210
+ Retention defaults to 30 days, with an integer range of 1 to 365. Use the lead-only `/agents-checkpoints list` and `/agents-checkpoints delete <checkpoint ID>` commands. Expired records cannot be loaded; one lead startup sweep retires due records and reports corrupt entries without discarding healthy siblings. Missing, incompatible, corrupt, deleted or expired selections fail with an actionable error. Deletion retires that exact root record permanently, preventing future continuation and replay. It does not erase independent report history or context already delivered to a successor.
211
+
212
+ ### Measured follow-up
213
+
214
+ One September 8, 2026 pair used Pi 0.85.1 and `openai-codex/gpt-5.6-terra`, high thinking, with identical current source, task and tools. A scripted original investigation saved F1/F2/F3, then its real SDK session was disposed and its private transcript removed before either fresh follow-up session. The fixture fixed tenant isolation, left expiration broken and reopened unsigned-token acceptance through a changed configuration dependency. Both follow-ups correctly classified all three controls.
215
+
216
+ | Observation | Fresh | Checkpoint |
217
+ | --- | ---: | ---: |
218
+ | Model / HTTP requests | 3 / 3 | 3 / 3 |
219
+ | Provider input tokens, excluding cache | 1,841 | 3,789 |
220
+ | Cache-read / cache-write tokens | 0 / 0 | 1,536 / 0 |
221
+ | Output tokens | 267 | 233 |
222
+ | File reads / unchanged-file rereads | 3 / 1 | 3 / 1 |
223
+ | First useful persisted result | 8.682s | 8.344s |
224
+ | Elapsed through final response | 10.296s | 10.156s |
225
+
226
+ This fixture showed no read or input-token savings: input including cache was 1,841 versus 5,325 tokens. One fixed fresh-then-checkpoint pair cannot establish a speed improvement. Each session emitted one `agent_settled`; lead acceptance stayed pending. The controlled starter/report submission exercised real SDK and checkpoint APIs, not full production admission-to-teardown or an OS-process restart. The pair predates the replay, missing-report cleanup and native cancellation repairs, so it does not measure those fixes. The [public record](docs/continuation-measurement.json) preserves the fixed corpus, assignments, method, historical source observation, aggregate results and limits; scripted usage remains null.
227
+
70
228
  ## Intent tiers
71
229
 
72
230
  Every spawn names a `model_slot`. Configured favorites take priority; otherwise the tier uses the current lead-session model and thinking level:
@@ -100,9 +258,31 @@ spawn_swarm_agents({
100
258
 
101
259
  For an edit, choose a write tier and name the files it may claim. Never run overlapping writers against the same paths.
102
260
 
261
+ ### Programmatic event launch
262
+
263
+ Another loaded extension can ask the lead session to launch one public agent through the orchestration event. Register the response listener and correlate it by `requestId` before emitting the request:
264
+
265
+ ```ts
266
+ pi.events.emit("pi-extended-teams:orchestration-request", {
267
+ requestId,
268
+ type: "spawn_agent",
269
+ ctx, // pass the current Pi command context when needed
270
+ params: {
271
+ name: "implementation",
272
+ prompt: "Implement the claimed change and report the evidence.",
273
+ cwd,
274
+ model_slot: "write-critical",
275
+ allow_nested_read_agents: true,
276
+ metadata: { operationId },
277
+ },
278
+ });
279
+ ```
280
+
281
+ The correlated response is `{ requestId, type, ok: true, details, content }` on success or `{ requestId, type, ok: false, error }` on failure. `prompt` is always a direct string; it may tell the child where a packaged prompt file lives, but there is no `prompt_file` API. The configured extension allowlist still determines which tools are available inside the child session. Teammate sessions cannot satisfy these requests.
282
+
103
283
  ## Configuration
104
284
 
105
- Global settings live at `~/.pi/agent/pi-extended-teams/settings.json`. Project overrides live at `.pi/pi-extended-teams.json`. Favorite intent tiers are global so `/agents-favorite-models` and spawning use the same choices. Configuring favorites is optional; an unset tier falls back to the current lead-session model and thinking level.
285
+ Global settings live at `~/.pi/agent/pi-extended-teams/settings.json`. Project overrides live at `.pi/pi-extended-teams.json`. Favorite intent tiers are global so `/agents-favorite-models` and spawning use the same choices. Configuring favorites is optional; an unset tier falls back to the current lead-session model and thinking level. `/pi-extended-teams-onboard` inspects both settings layers, recommends a complete model and extension policy, and gives source-specific package update instructions without changing either file on its first pass.
106
286
 
107
287
  Public read and edit spawns respect their role's concurrency limit and overflow setting. Enabled overflow queues accepted work; disabled overflow returns a capacity error. Quarantined requests stay fenced without blocking unrelated eligible work. `stop_teammate` can cancel a queued request before launch. Failed admissions trigger an attempted recipient notification and remain visible in status (up to 20 recent failures). The public queue and recent failure index are session-local, not restart-durable.
108
288
 
@@ -2,6 +2,7 @@ import fs from "node:fs";
2
2
  import type { AgentReportSource } from "../runtime/types";
3
3
  import { getLastAssistantText } from "../ui/renderers";
4
4
  import type { SubmittedAgentReport } from "../tools/agent-communication-tools";
5
+ import { normalizeReportedTaskDetails, type ReportedTaskDetails } from "../../src/results/report-result";
5
6
 
6
7
  export const EMPTY_REPORT_RECOVERY_PROMPT = [
7
8
  "Your previous turn ended without a usable final report.",
@@ -10,7 +11,7 @@ export const EMPTY_REPORT_RECOVERY_PROMPT = [
10
11
  "If the assignment could not be completed, call report_and_exit with a non-empty blocker or failure report explaining why.",
11
12
  ].join(" ");
12
13
 
13
- export interface ResolvedReadAgentReport {
14
+ export interface ResolvedReadAgentReport extends ReportedTaskDetails {
14
15
  report?: string;
15
16
  summary?: string;
16
17
  source?: AgentReportSource;
@@ -43,6 +44,11 @@ export function nonEmptyReportText(value: unknown): string | undefined {
43
44
  return text || undefined;
44
45
  }
45
46
 
47
+ function compatiblePersistedTaskDetails(value: unknown): ReportedTaskDetails {
48
+ try { return normalizeReportedTaskDetails(value); }
49
+ catch { return {}; }
50
+ }
51
+
46
52
  function getAcceptedReportToolSubmission(messages: any[]): SubmittedAgentReport | undefined {
47
53
  const acceptedToolCallIds = new Set<string>();
48
54
  for (const message of messages || []) {
@@ -70,6 +76,7 @@ function getAcceptedReportToolSubmission(messages: any[]): SubmittedAgentReport
70
76
  const content = nonEmptyReportText(part.arguments?.content);
71
77
  if (!content) continue;
72
78
  return {
79
+ ...compatiblePersistedTaskDetails(part.arguments),
73
80
  content,
74
81
  summary: nonEmptyReportText(part.arguments?.summary),
75
82
  };
@@ -126,6 +133,7 @@ export function resolveReadAgentReport(
126
133
  const submittedContent = nonEmptyReportText(submittedFinalReport?.content);
127
134
  if (submittedContent) {
128
135
  return {
136
+ ...normalizeReportedTaskDetails(submittedFinalReport),
129
137
  report: submittedContent,
130
138
  summary: nonEmptyReportText(submittedFinalReport?.summary),
131
139
  source: "report_and_exit",
@@ -134,11 +142,12 @@ export function resolveReadAgentReport(
134
142
 
135
143
  const immediateToolReport = getAcceptedReportToolSubmission(immediateMessages);
136
144
  if (immediateToolReport) {
137
- return { report: immediateToolReport.content, summary: immediateToolReport.summary, source: "report_and_exit" };
145
+ return { ...normalizeReportedTaskDetails(immediateToolReport), report: immediateToolReport.content, summary: immediateToolReport.summary, source: "report_and_exit" };
138
146
  }
139
147
  const persistedToolReport = getAcceptedReportToolSubmission(persistedMessages);
140
148
  if (persistedToolReport) {
141
149
  return {
150
+ ...normalizeReportedTaskDetails(persistedToolReport),
142
151
  report: persistedToolReport.content,
143
152
  summary: persistedToolReport.summary,
144
153
  source: "persisted-report_and_exit",
@@ -25,6 +25,9 @@ export interface ReadAgentDeliveryState {
25
25
 
26
26
  export interface ManagedReadAgentLifecycleState extends ReadAgentDeliveryState {
27
27
  session?: AgentSession;
28
+ checkOperation?: { controller: AbortController; settled: Promise<void> };
29
+ checkpointOperation?: { controller: AbortController; settled: Promise<void> };
30
+ activeOperationSettlementPromise?: Promise<void>;
28
31
  startupState?: ReadAgentStartupState;
29
32
  sessionCreation?: Promise<AgentSession | undefined>;
30
33
  stopRequested?: boolean;
@@ -77,7 +80,8 @@ interface NestedSessionLifecycle {
77
80
  requestShutdown(
78
81
  reason: unknown,
79
82
  rawDeliverySettlement: Promise<void>,
80
- timeoutMs?: number
83
+ timeoutMs?: number,
84
+ operationSettlement?: Promise<void>
81
85
  ): Promise<NestedSessionTeardownResult>;
82
86
  finalized: Promise<void>;
83
87
  }
@@ -220,7 +224,7 @@ async function operationTimedOut(operation: Promise<void>, timeoutMs: number): P
220
224
  return timedOut;
221
225
  }
222
226
 
223
- function installNestedSessionLifecycle(session: AgentSession): NestedSessionLifecycle {
227
+ function installNestedSessionLifecycle(session: AgentSession, beforeDispose?: () => void): NestedSessionLifecycle {
224
228
  const existing = sessionLifecycles.get(session);
225
229
  if (existing) return existing;
226
230
 
@@ -232,6 +236,7 @@ function installNestedSessionLifecycle(session: AgentSession): NestedSessionLife
232
236
  const disposeOnce = (): void => {
233
237
  if (disposed) return;
234
238
  disposed = true;
239
+ try { beforeDispose?.(); } catch { /* Accounting must not prevent disposal. */ }
235
240
  try {
236
241
  session.dispose();
237
242
  finalized.resolve();
@@ -243,7 +248,7 @@ function installNestedSessionLifecycle(session: AgentSession): NestedSessionLife
243
248
 
244
249
  const lifecycle: NestedSessionLifecycle = {
245
250
  finalized: finalized.promise,
246
- requestShutdown(reasonInput, rawDeliverySettlement, timeoutMs = NESTED_SESSION_TEARDOWN_TIMEOUT_MS) {
251
+ requestShutdown(reasonInput, rawDeliverySettlement, timeoutMs = NESTED_SESSION_TEARDOWN_TIMEOUT_MS, operationSettlement = Promise.resolve()) {
247
252
  if (shutdownPromise) return shutdownPromise;
248
253
  const reason = normalizeShutdownReason(reasonInput);
249
254
 
@@ -309,6 +314,7 @@ function installNestedSessionLifecycle(session: AgentSession): NestedSessionLife
309
314
  observedExtensionShutdown,
310
315
  observedDelivery,
311
316
  observedAbort,
317
+ operationSettlement.catch(() => {}),
312
318
  ]).then(() => {});
313
319
  void rawOperations.then(disposeOnce);
314
320
 
@@ -348,8 +354,8 @@ function installNestedSessionLifecycle(session: AgentSession): NestedSessionLife
348
354
  return lifecycle;
349
355
  }
350
356
 
351
- export function installReadAgentSessionLifecycle(session: AgentSession): NestedSessionLifecycle {
352
- return installNestedSessionLifecycle(session);
357
+ export function installReadAgentSessionLifecycle(session: AgentSession, beforeDispose?: () => void): NestedSessionLifecycle {
358
+ return installNestedSessionLifecycle(session, beforeDispose);
353
359
  }
354
360
 
355
361
  export function requestReadAgentTeardown(
@@ -357,6 +363,13 @@ export function requestReadAgentTeardown(
357
363
  options: ReadAgentTeardownOptions
358
364
  ): Promise<ReadAgentTeardownResult> {
359
365
  state.stopRequested = true;
366
+ const checkOperation = state.checkOperation;
367
+ const checkpointOperation = state.checkpointOperation;
368
+ checkOperation?.controller.abort();
369
+ checkpointOperation?.controller.abort();
370
+ const operationSettlement = Promise.all([
371
+ checkOperation?.settled.catch(() => {}), checkpointOperation?.settled.catch(() => {}), state.activeOperationSettlementPromise?.catch(() => {}),
372
+ ]).then(() => {});
360
373
  if (state.heartbeatTimer) clearInterval(state.heartbeatTimer);
361
374
  state.heartbeatTimer = undefined;
362
375
  const deliveryClose = closeReadAgentMessageDelivery(state);
@@ -474,6 +487,7 @@ export function requestReadAgentTeardown(
474
487
  }
475
488
  );
476
489
 
490
+ const observedOperations = Promise.all([observedDelivery, operationSettlement]).then(() => {});
477
491
  let session = state.session;
478
492
  if (state.startupState === "pending" && state.sessionCreation) {
479
493
  let startupSettled = false;
@@ -497,10 +511,10 @@ export function requestReadAgentTeardown(
497
511
  const lateRawOperations = observedStartup.then(async () => {
498
512
  if (session) {
499
513
  const sessionLifecycle = installNestedSessionLifecycle(session);
500
- await sessionLifecycle.requestShutdown(reason, deliveryClose.rawDeliverySettlement, 0);
514
+ await sessionLifecycle.requestShutdown(reason, deliveryClose.rawDeliverySettlement, 0, operationSettlement);
501
515
  await sessionLifecycle.finalized;
502
516
  } else {
503
- await observedDelivery;
517
+ await observedOperations;
504
518
  }
505
519
  });
506
520
  if (!deliverySettled) deliveryOutcome = "timed_out";
@@ -518,7 +532,7 @@ export function requestReadAgentTeardown(
518
532
  session ??= state.session;
519
533
  if (!session) {
520
534
  const deliveryTimedOut = await operationTimedOut(
521
- observedDelivery,
535
+ observedOperations,
522
536
  Math.max(0, deadline - Date.now())
523
537
  );
524
538
  if (deliveryTimedOut) {
@@ -529,7 +543,7 @@ export function requestReadAgentTeardown(
529
543
  delivery: deliveryOutcome,
530
544
  dispose: "deferred",
531
545
  });
532
- deferFinalization(observedDelivery, timedOutResult);
546
+ deferFinalization(observedOperations, timedOutResult);
533
547
  return publishBoundedResult(timedOutResult);
534
548
  }
535
549
 
@@ -548,7 +562,8 @@ export function requestReadAgentTeardown(
548
562
  const sessionResult = await sessionLifecycle.requestShutdown(
549
563
  reason,
550
564
  deliveryClose.rawDeliverySettlement,
551
- Math.max(0, deadline - Date.now())
565
+ Math.max(0, deadline - Date.now()),
566
+ operationSettlement
552
567
  );
553
568
  if (sessionResult.status === "timed_out") {
554
569
  const timedOutResult = unfinishedResult("timed_out", sessionResult.reason, {