@dorokuma/herdsman-pi 0.13.6 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/package.json +1 -1
  2. package/src/index.ts +558 -62
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@dorokuma/herdsman-pi",
3
- "version": "0.13.6",
3
+ "version": "0.14.0",
4
4
  "description": "Pi extension bridge for Herdsman agent history.",
5
5
  "type": "module",
6
6
  "keywords": [
package/src/index.ts CHANGED
@@ -130,6 +130,85 @@ type HerdsmanState = {
130
130
  * from the current state instead of deferring again.
131
131
  */
132
132
  wakeForcedRelease: boolean;
133
+ /**
134
+ * Event ids handed to Pi as a *queued* (non-triggering) follow-up whose
135
+ * content has not been seen entering the transcript yet.
136
+ *
137
+ * A wake injected while the orchestrator streams is parked in the agent's
138
+ * follow-up queue, and Pi only drains that queue when a run reaches its stop
139
+ * point. When the run it rode on already passed that point, the update sits in
140
+ * the queue until the next user message. This set is what keeps such a
141
+ * delivery from being written off, and it carries three guarantees at once:
142
+ *
143
+ * - never lost: while it is non-empty the settlement drives a continuation
144
+ * (bounded by MAX_WAKE_CONTINUATION_ATTEMPTS) so a later run drains the queue
145
+ * and carries the update out;
146
+ * - never acknowledged unseen: these ids are excluded from the acknowledgement
147
+ * path, so the daemon keeps them pending and redelivers them when this
148
+ * session never consumes them (the only way a delivery Pi itself dropped —
149
+ * clearQueue / restoreQueuedMessagesToEditor — can still be recovered);
150
+ * - never duplicated: they are excluded from every later injection, so a
151
+ * redelivery of the same id cannot put the same content into the transcript
152
+ * twice.
153
+ *
154
+ * Ids leave the set when their content reaches the transcript (consumption
155
+ * evidence: the hidden wake message's `message_end`, which then moves them to
156
+ * `wakeConsumptionObserved`) or when the event leaves the delivery queue for
157
+ * good (acknowledged, covered by the acknowledgement watermark,
158
+ * dead-lettered), and with the delivery queue on a role/scope reset.
159
+ */
160
+ wakeAwaitingConsumption: Set<number>;
161
+ /**
162
+ * Ids whose content was observed in the transcript and which are therefore
163
+ * confirmed on the evidence alone, whatever the turn that carried them did.
164
+ *
165
+ * The daemon confirms by a monotonic watermark (`where id <= ?`), so an id that
166
+ * stays unacknowledged blocks every later one: a turn that ends in error after
167
+ * the content already reached the orchestrator must not pin that watermark, so
168
+ * the evidence — not the turn outcome — authorises these ids.
169
+ */
170
+ wakeConsumptionObserved: Set<number>;
171
+ /**
172
+ * Ids already handed to the orchestrator in this session that must never be
173
+ * injected again, because the copy that carries them can outlive the bookkeeping
174
+ * that knew about it.
175
+ *
176
+ * Pi's follow-up queue is process-wide and the extension has no API to query or
177
+ * clear it, so an id can become "un-presented" again while its content is still
178
+ * on its way: a role/scope reset clears the presentation guard (and the delivery
179
+ * queue), and a consumed id leaves the queue without ever being acknowledged (its
180
+ * batch is gone), so the daemon keeps redelivering it. This set is the guard that
181
+ * survives all of that:
182
+ *
183
+ * - an unconsumed delivery carried over a reset lands here (see
184
+ * `clearDeliveryBookkeeping`), which deliberately trades away the redelivery
185
+ * remedy for it (logged there) in exchange for never presenting a duplicate;
186
+ * - a consumed id lands here too, so a redelivery after the scope that consumed
187
+ * it is gone still cannot inject it a second time (the id also re-enters the
188
+ * current scope's `presentedEventIds`, see the consumption evidence handler).
189
+ *
190
+ * Ids leave the set when the daemon has confirmed them (their own
191
+ * acknowledgement, the acknowledgement watermark, dead-lettering) or when they
192
+ * leave the delivery queue for good — the daemon then holds nothing that could be
193
+ * redelivered, so there is nothing left to block.
194
+ */
195
+ wakeSuppressedEventIds: Set<number>;
196
+ /**
197
+ * Last reason an event id was skipped for injection, for the rate-limited
198
+ * diagnostic in `noteSkippedWakeInjection`: one line per id and reason.
199
+ *
200
+ * Skipping a redelivery is intentional but invisible, so without this the log
201
+ * could not tell "no update arrived" from "updates were suppressed"; with it a
202
+ * daemon that redelivers the same event in a loop still cannot flood the file.
203
+ */
204
+ wakeSkipLogReasons: Map<number, string>;
205
+ /**
206
+ * Continuation drives already spent on the current unconsumed delivery.
207
+ *
208
+ * Reset when a delivery is handed over and when the unconsumed set empties, so
209
+ * the bound applies per delivery instead of accumulating for the session.
210
+ */
211
+ wakeContinuationAttempts: number;
133
212
  /**
134
213
  * Event content queued for a busy orchestrator. Injected through the
135
214
  * `context` hook so the running turn sees the update without being
@@ -201,6 +280,30 @@ export const WAKE_BUSY_SPIN_MS = 100;
201
280
  * parked forever.
202
281
  */
203
282
  export const WAKE_DEFERRED_TIMEOUT_MS = 5_000;
283
+ /**
284
+ * Upper bound on the continuation drives spent on one unconsumed delivery.
285
+ *
286
+ * Every drive costs a full agent run, so three attempts cover the drive that
287
+ * follows the settlement which missed Pi's follow-up queue plus two retries
288
+ * after intervening runs that also ended before their stop point. Beyond that
289
+ * the condition is systemic (Pi dropped the queue, the runs keep ending early)
290
+ * and further attempts could only start runs without delivering anything: the
291
+ * delivery stays unacknowledged, hence pending on the daemon, which redelivers
292
+ * it to a later run or session.
293
+ */
294
+ export const MAX_WAKE_CONTINUATION_ATTEMPTS = 3;
295
+ /** `customType` of the hidden wake context this extension injects. */
296
+ const WAKE_CONTEXT_CUSTOM_TYPE = "herdsman-wake-context";
297
+ /** `customType` of the hidden marker that drives a missed wake continuation. */
298
+ const WAKE_CONTINUATION_CUSTOM_TYPE = "herdsman-wake-continuation";
299
+ /**
300
+ * Content of the continuation marker. Its only job is to start a run
301
+ * (`triggerTurn: true`) so the run's loop drains the queued follow-up that was
302
+ * never delivered; the wake content itself is not repeated here, so the
303
+ * evidence is not presented twice.
304
+ */
305
+ const WAKE_CONTINUATION_CONTENT =
306
+ "[HERDSMAN WAKE CONTINUATION]\nA queued Herdsman agent update was not delivered by the previous turn; it follows this message. Handle it, and do not start unrelated work.";
204
307
 
205
308
  type AckFailureClass = "terminal" | "resync" | "transient";
206
309
 
@@ -287,6 +390,11 @@ export function createHerdsmanPiExtension(options: ExtensionOptions = {}) {
287
390
  wakeDeferredUntilSettled: false,
288
391
  wakeDeferredSince: undefined,
289
392
  wakeForcedRelease: false,
393
+ wakeAwaitingConsumption: new Set(),
394
+ wakeConsumptionObserved: new Set(),
395
+ wakeSuppressedEventIds: new Set(),
396
+ wakeSkipLogReasons: new Map(),
397
+ wakeContinuationAttempts: 0,
290
398
  wakeContext: undefined,
291
399
  wakeRequested: false,
292
400
  wakeRequestedThroughEventId: 0,
@@ -376,13 +484,67 @@ export function createHerdsmanPiExtension(options: ExtensionOptions = {}) {
376
484
  };
377
485
 
378
486
  /**
379
- * Drops one event from the delivery queue. The only callers are the two
380
- * acknowledgement outcomes (accepted, or terminally refused by the daemon)
381
- * and the role/scope reset, which clears the whole stream together with
382
- * `presentedEventIds`.
487
+ * Drops one event from the delivery queue. The callers are the two
488
+ * acknowledgement outcomes (accepted, or terminally refused by the daemon),
489
+ * the acknowledgement watermark that covers a whole range of ids, and the
490
+ * role/scope reset, which clears the entire stream through
491
+ * `clearDeliveryBookkeeping`.
492
+ *
493
+ * An event that leaves the queue can no longer be awaiting consumption: its
494
+ * acknowledgement cursor covered it, or it is dead-lettered and will never be
495
+ * redelivered. The awaiting set therefore follows the queue here — without
496
+ * that, a dead-lettered id would keep the settlement driving continuations
497
+ * (and holding its batch open) for content that can never arrive. The other
498
+ * two sets describe the same ids and follow the queue just as well: a
499
+ * confirmed id needs no evidence flag, and an id that is gone from the queue
500
+ * cannot be re-injected, so it needs no suppression either.
383
501
  */
384
502
  const dropUnackedDelivered = (eventId: number): void => {
385
503
  state.unackedDelivered.delete(eventId);
504
+ state.wakeAwaitingConsumption.delete(eventId);
505
+ state.wakeConsumptionObserved.delete(eventId);
506
+ state.wakeSuppressedEventIds.delete(eventId);
507
+ if (state.wakeAwaitingConsumption.size === 0) state.wakeContinuationAttempts = 0;
508
+ };
509
+
510
+ /**
511
+ * Drops the presentation guard, the delivery queue, and the unconsumed
512
+ * bookkeeping together.
513
+ *
514
+ * These describe the same events — handed to Pi, not yet confirmed by the
515
+ * daemon — so they are only ever cleared together, and only by a reset that
516
+ * also discards the pending projection: a genuine role/scope loss and
517
+ * shutdown. A transient disconnect keeps all of them (`preservePresented`) so
518
+ * an in-flight batch can still be settled and acknowledged after the
519
+ * reconnect, and so no already-presented event is presented again.
520
+ *
521
+ * One thing survives the reset (`wakeSuppressedEventIds`), because nothing here
522
+ * can invalidate it: an id handed to Pi may still sit in Pi's process-wide
523
+ * follow-up queue (no API to query or clear it), so dropping it together with
524
+ * the queue would let a daemon redelivery present the same update a second
525
+ * time in the scope that takes over. Unconsumed ids are carried over for that
526
+ * reason; ids consumed *after* their scope was reset join the same set from the
527
+ * consumption handler. The trade-off is explicit: for a carried-over id the
528
+ * redelivery remedy is given up on purpose (see the log line) until the daemon
529
+ * confirms it some other way.
530
+ */
531
+ const clearDeliveryBookkeeping = () => {
532
+ const carriedOver = [...state.wakeAwaitingConsumption].sort((left, right) => left - right);
533
+ for (const eventId of carriedOver) state.wakeSuppressedEventIds.add(eventId);
534
+ if (carriedOver.length > 0) {
535
+ logHerdsmanPi(
536
+ "info",
537
+ `[herdsman-pi] keeping ${carriedOver.length} unconsumed wake event id(s) suppressed across the scope change eventIds=${carriedOver.join(",")} · daemon redelivery for them is traded away to keep the transcript single-copy`,
538
+ );
539
+ }
540
+ state.presentedEventIds.clear();
541
+ // Once the guard is gone the daemon's pending events can be presented (and
542
+ // acknowledged) again, so keeping the old queue would only risk a stale id.
543
+ state.unackedDelivered.clear();
544
+ state.wakeAwaitingConsumption.clear();
545
+ state.wakeConsumptionObserved.clear();
546
+ state.wakeSkipLogReasons.clear();
547
+ state.wakeContinuationAttempts = 0;
386
548
  };
387
549
 
388
550
  const pruneAcknowledgedEvents = (ackedEventId: number | undefined) => {
@@ -401,10 +563,71 @@ export function createHerdsmanPiExtension(options: ExtensionOptions = {}) {
401
563
  // superseded (the daemon acknowledges by watermark), so it leaves the
402
564
  // queue and is never re-acknowledged.
403
565
  for (const eventId of [...state.unackedDelivered.keys()]) {
404
- if (eventId <= ackedEventId) state.unackedDelivered.delete(eventId);
566
+ if (eventId <= ackedEventId) dropUnackedDelivered(eventId);
567
+ }
568
+ // The watermark also covers ids that survived a reset as suppression
569
+ // entries: the daemon considers them confirmed, so they will not be
570
+ // redelivered and the suppression has nothing left to block.
571
+ for (const eventId of [...state.wakeSuppressedEventIds]) {
572
+ if (eventId <= ackedEventId) state.wakeSuppressedEventIds.delete(eventId);
573
+ }
574
+ for (const eventId of [...state.wakeSkipLogReasons.keys()]) {
575
+ if (eventId <= ackedEventId) state.wakeSkipLogReasons.delete(eventId);
405
576
  }
406
577
  };
407
578
 
579
+ /**
580
+ * Records, once per id and reason, that an event which is already in the
581
+ * orchestrator's hands was not injected again.
582
+ *
583
+ * Skipping a redelivery is deliberate — the content must not enter the
584
+ * transcript twice — but it is also invisible: a daemon that keeps redelivering
585
+ * one event would otherwise leave no trace at all, and "an update never
586
+ * arrived" could not be told apart from "an update was suppressed" in the log.
587
+ * The line is emitted once per id and reason (a change of reason is logged
588
+ * again: `presented`, `awaiting` and `suppressed` describe different ownership
589
+ * of the same id), so a redelivery loop cannot flood the file.
590
+ */
591
+ const noteSkippedWakeInjection = (eventId: number, reason: string): false => {
592
+ if (state.wakeSkipLogReasons.get(eventId) !== reason) {
593
+ state.wakeSkipLogReasons.set(eventId, reason);
594
+ logHerdsmanPi(
595
+ "info",
596
+ `[herdsman-pi] wake injection skipped eventId=${eventId} reason=${reason} · update already presented (or still in flight) in this session, so a daemon redelivery is not shown twice`,
597
+ );
598
+ }
599
+ return false;
600
+ };
601
+
602
+ /**
603
+ * Whether an event was already handed to the orchestrator in this session and
604
+ * therefore must not be injected again (logging why, at most once per reason).
605
+ *
606
+ * - `awaiting`: a copy of it was handed over as a queued follow-up that no run
607
+ * has drained yet, and the continuation is what carries it out (checked
608
+ * first: for a queued copy this is the state that explains the redelivery);
609
+ * - `presented`: it was injected before (or its content has since been observed
610
+ * in the transcript), so a redelivery would duplicate it;
611
+ * - `suppressed`: its copy may still sit in Pi's process-wide follow-up queue
612
+ * after a role/scope reset cleared the local guards, so the redelivery is the
613
+ * only one that must not be shown (see `wakeSuppressedEventIds`).
614
+ */
615
+ const alreadyPresented = (eventId: number): boolean => {
616
+ if (state.wakeAwaitingConsumption.has(eventId)) {
617
+ noteSkippedWakeInjection(eventId, "awaiting");
618
+ return true;
619
+ }
620
+ if (state.presentedEventIds.has(eventId)) {
621
+ noteSkippedWakeInjection(eventId, "presented");
622
+ return true;
623
+ }
624
+ if (state.wakeSuppressedEventIds.has(eventId)) {
625
+ noteSkippedWakeInjection(eventId, "suppressed");
626
+ return true;
627
+ }
628
+ return false;
629
+ };
630
+
408
631
  const isWakeableEvent = (event: AgentEventWireRecord | undefined) =>
409
632
  !event?.nextAttemptAt || event.nextAttemptAt <= Date.now();
410
633
 
@@ -607,6 +830,174 @@ export function createHerdsmanPiExtension(options: ExtensionOptions = {}) {
607
830
  }, WAKE_SETTLE_MS);
608
831
  };
609
832
 
833
+ /**
834
+ * Ends the current wake-deferral episode without injecting.
835
+ *
836
+ * `wakeForcedRelease` and `wakeDeferredSince` describe *one* bounded
837
+ * deferral: the released deadline is re-derived from `wakeDeferredSince`
838
+ * every time `scheduleDeferredWake` runs, so a stale pair would let an
839
+ * unrelated later wake bypass the busy gate. Every path that ends a wake
840
+ * pass without injecting clears both, which gives the next deferral its own
841
+ * full `WAKE_DEFERRED_TIMEOUT_MS` budget. The 5s hard deadline itself is
842
+ * unchanged: it is measured inside a single episode, and an episode that
843
+ * reaches it still force-releases the batch on its next pass.
844
+ */
845
+ const endWakeDeferral = () => {
846
+ state.wakeForcedRelease = false;
847
+ state.wakeDeferredSince = undefined;
848
+ };
849
+
850
+ /**
851
+ * Event ids a hidden wake message proves to have reached the transcript.
852
+ *
853
+ * A wake injection names the outcomes it presents in
854
+ * `details.presentedEventIds`, and Pi writes those details onto the session
855
+ * entry it emits on `message_end` (both for a triggered turn and when a run
856
+ * finally drains the queued follow-up). That emission is the only consumption
857
+ * evidence there is: nothing else tells us that the content — not merely the
858
+ * request that carried it — reached the orchestrator.
859
+ *
860
+ * The wider `details.eventIds` list is deliberately not evidence: it names
861
+ * every id that was pending at injection time, including copies that an
862
+ * earlier injection presented (and that may have been dropped by Pi), so using
863
+ * it would confirm content nobody ever saw. Only real, numeric ids count, and
864
+ * a message without readable ones proves nothing: the caller logs that instead
865
+ * of confirming anything.
866
+ */
867
+ const wakeConsumedEventIds = (message: Record<string, unknown>): number[] => {
868
+ const presented = record(message.details).presentedEventIds;
869
+ if (!Array.isArray(presented)) return [];
870
+ return presented.filter((eventId): eventId is number => typeof eventId === "number");
871
+ };
872
+
873
+ /**
874
+ * The contiguous prefix of the delivery queue that may be acknowledged now.
875
+ *
876
+ * The delivery queue — not the injection snapshot — is what gets acknowledged:
877
+ * it holds every event handed to Pi that is still unconfirmed (a release merges
878
+ * batches instead of replacing them), is id-ascending, and an id leaves it only
879
+ * when the daemon accepts it.
880
+ *
881
+ * An acknowledgement tells the daemon "the orchestrator has this", and the
882
+ * daemon confirms by a monotonic watermark (`update agent_events set status =
883
+ * 'acked' where id <= ?`), so acking a larger id confirms every smaller one with
884
+ * it. The queue is therefore walked in ascending id order and only its
885
+ * *contiguous* confirmable prefix is returned: the walk stops at the first id
886
+ * that is not confirmable. Skipping an unconfirmed id would hand it to the
887
+ * watermark, which swallows it for good.
888
+ *
889
+ * An id is confirmable when its content is known to have reached the transcript
890
+ * (`wakeConsumptionObserved`), or — for a delivery that was never queued, i.e. a
891
+ * triggered prompt — when the turn that received it produced a final response or
892
+ * was aborted by the user. An id still awaiting consumption has no evidence and
893
+ * blocks the prefix: leaving it unacknowledged keeps it pending, which is what
894
+ * lets the daemon redeliver the one copy that never arrived.
895
+ *
896
+ * An event whose own acknowledgement already failed is left out (the daemon's
897
+ * cursor advance sweeps it, and a retry would reset its attempt/backoff
898
+ * accounting), but it must not hold the prefix back: it stays in the queue until
899
+ * that cursor or a scope reset confirms it, so a failed or dead-lettered
900
+ * acknowledgement never depends on the timing of the release to stay
901
+ * recoverable. Restricting to a *live* row with `attempts === 0` has two holes
902
+ * on purpose: `?? 0` covers an event the live projection no longer holds at all
903
+ * (the server stopped listing it, or `failedWakeThroughEventId` filters it out),
904
+ * and the projection cannot tell us it already failed — so the event gets one
905
+ * more attempt, a deliberate self-healing opportunity that then accumulates on
906
+ * the queue copy's counter and can reach MAX_ACK_ATTEMPTS instead of restarting
907
+ * at 1 every round.
908
+ */
909
+ const confirmableDeliveryPrefix = (turnProducedFinalResponse: boolean) => {
910
+ const deliveryQueue = unackedDeliveredAscending();
911
+ const confirmablePrefix: AgentEventWireRecord[] = [];
912
+ let blockedByMissingTurn = false;
913
+ for (const event of deliveryQueue) {
914
+ if (state.wakeAwaitingConsumption.has(event.id)) break;
915
+ if (!state.wakeConsumptionObserved.has(event.id) && !turnProducedFinalResponse) {
916
+ blockedByMissingTurn = true;
917
+ break;
918
+ }
919
+ // An event whose own acknowledgement already failed is left to the
920
+ // daemon's cursor sweep (see above); it must not hold the prefix back.
921
+ if ((state.pendingEvents.find((pending) => pending.id === event.id)?.attempts ?? 0) > 0) {
922
+ continue;
923
+ }
924
+ confirmablePrefix.push(event);
925
+ }
926
+ // Any id still awaiting consumption keeps the batch in flight: it owns the
927
+ // acknowledgement cursor, so the settlement of the run that finally drains
928
+ // the queued copy (the continuation driven by the settle handler) confirms it
929
+ // then. Reading the queue directly — instead of comparing two filtered lists —
930
+ // keeps that decision independent of the `attempts` exclusion above, which
931
+ // would otherwise hide an unconsumed id and drop the batch too early.
932
+ const stillAwaitingConsumption = deliveryQueue.some((event) =>
933
+ state.wakeAwaitingConsumption.has(event.id),
934
+ );
935
+ return { deliveryQueue, confirmablePrefix, blockedByMissingTurn, stillAwaitingConsumption };
936
+ };
937
+
938
+ /**
939
+ * Drives one continuation that carries out a wake Pi has not drained.
940
+ *
941
+ * A wake injected while the orchestrator streams is delivered as a queued
942
+ * follow-up (`triggerTurn: false`), and Pi only drains that queue when a run
943
+ * reaches its stop point (`agent-loop` "Agent would stop here. Check for
944
+ * follow-up messages."). When the run it rode on already passed that point,
945
+ * the update sits in the queue until the next user message. At settlement the
946
+ * orchestrator is no longer streaming, so `triggerTurn: true` starts a real
947
+ * run (`_runAgentPrompt`) and that run's loop drains the queued follow-up.
948
+ * The marker carries no wake content: the queued follow-up is what delivers
949
+ * the evidence, exactly once.
950
+ *
951
+ * Only consumption evidence writes a delivery off, so an intervening run that
952
+ * ends before its stop point (error, user abort, a refused tool) leaves the
953
+ * next settlement driving again. That is bounded by
954
+ * MAX_WAKE_CONTINUATION_ATTEMPTS: starting runs cannot fix a cause that is
955
+ * not about the queue, and once the budget is spent the delivery stays
956
+ * unacknowledged and therefore pending on the daemon.
957
+ */
958
+ const driveWakeContinuation = (ctx: PiContext) => {
959
+ if (state.wakeAwaitingConsumption.size === 0) return;
960
+ if (state.wakeContinuationAttempts >= MAX_WAKE_CONTINUATION_ATTEMPTS) return;
961
+ if (!pi.sendMessage) return;
962
+ const eventIds = [...state.wakeAwaitingConsumption].sort((left, right) => left - right);
963
+ state.wakeContinuationAttempts += 1;
964
+ try {
965
+ pi.sendMessage(
966
+ {
967
+ content: WAKE_CONTINUATION_CONTENT,
968
+ customType: WAKE_CONTINUATION_CUSTOM_TYPE,
969
+ display: false,
970
+ },
971
+ { deliverAs: "followUp", triggerTurn: true },
972
+ );
973
+ logHerdsmanPi(
974
+ "info",
975
+ `[herdsman-pi] wake continuation driven eventIds=${eventIds[0] ?? 0}-${eventIds.at(-1) ?? 0} count=${eventIds.length} attempt=${state.wakeContinuationAttempts}`,
976
+ );
977
+ } catch {
978
+ logHerdsmanPi("warn", "[herdsman-pi] wake continuation refused by pi");
979
+ }
980
+ if (state.wakeContinuationAttempts >= MAX_WAKE_CONTINUATION_ATTEMPTS) {
981
+ logHerdsmanPi(
982
+ "warn",
983
+ `[herdsman-pi] wake continuation limit reached awaiting=${eventIds.join(",")} attempts=${state.wakeContinuationAttempts} · delivery stays unacknowledged for daemon redelivery`,
984
+ );
985
+ // Bounded retries cannot deliver this update, and the daemon does not
986
+ // redeliver while this terminal still holds the scope
987
+ // (`nextDeliverableAfter` skips events already delivered to the same
988
+ // owner), so the session is where it has to be visible. The notice must
989
+ // not promise what the extension cannot do: a copy that Pi dropped cannot
990
+ // be drained by "the next turn", it is simply gone here — so it says what
991
+ // is true (unconfirmed, possibly dropped) and what the user can actually
992
+ // do about it (a later turn still drains a queued copy, and handing the
993
+ // workspace to another terminal makes the daemon redeliver the update).
994
+ ctx.ui.notify?.(
995
+ `Herdsman · ${eventIds.length} agent update${eventIds.length === 1 ? "" : "s"} could not be delivered by a wake turn · unconfirmed and possibly dropped: keep working here so a later turn drains a queued copy, or hand this workspace to another terminal so the daemon redelivers it`,
996
+ "warning",
997
+ );
998
+ }
999
+ };
1000
+
610
1001
  /**
611
1002
  * Arms the bounded deferral for a wake that cannot be injected right now.
612
1003
  *
@@ -639,12 +1030,19 @@ export function createHerdsmanPiExtension(options: ExtensionOptions = {}) {
639
1030
 
640
1031
  const scheduleWake = (ctx: PiContext | undefined) => {
641
1032
  if (!ctx || !state.isOrchestrator || !state.currentScope || !pi.sendMessage) return;
642
- if (state.wakeTimer || state.wakeRequested) return;
1033
+ if (state.wakeTimer || state.wakeRequested) {
1034
+ // A pending wake owns the release; this branch is also hit re-entrantly by
1035
+ // a running pass (its fired settle timer is still set), where clearing
1036
+ // `wakeForcedRelease` would re-defer a batch the deadline just released.
1037
+ return;
1038
+ }
643
1039
  const projection = projectAgentOutcomes(state.pendingEvents, wakeFilter);
644
1040
  const outcomes = projection.outcomes.filter(
645
1041
  (outcome) =>
646
1042
  outcome.eventId > state.failedWakeThroughEventId &&
647
- !state.presentedEventIds.has(outcome.eventId),
1043
+ // Already in the orchestrator's hands: injecting it again would put the
1044
+ // same content into the transcript twice. See `alreadyPresented`.
1045
+ !alreadyPresented(outcome.eventId),
648
1046
  );
649
1047
  const suppressedEvents = projection.suppressedUpstreamErrorEventIds
650
1048
  .filter(
@@ -659,6 +1057,10 @@ export function createHerdsmanPiExtension(options: ExtensionOptions = {}) {
659
1057
  isWakeableEvent(state.pendingEvents.find((pending) => pending.id === outcome.eventId)),
660
1058
  );
661
1059
  if (wakeable.length === 0) {
1060
+ // Nothing is wakeable right now, so the deferral that led here is over:
1061
+ // its released deadline must not let a later, unrelated wake bypass the
1062
+ // busy gate.
1063
+ endWakeDeferral();
662
1064
  // A suppressed upstream error that is now due must be silently
663
1065
  // acknowledged before any backoff timer is planted: planting the timer
664
1066
  // first would make scheduleSilentUpstreamErrorAck's entry guard
@@ -712,10 +1114,16 @@ export function createHerdsmanPiExtension(options: ExtensionOptions = {}) {
712
1114
  state.currentScope?.workspaceId !== ownerWorkspaceId
713
1115
  ) {
714
1116
  state.wakeTimer = undefined;
1117
+ // The pass belongs to a stale generation or scope, so its deferral
1118
+ // episode ends here (a scope reset usually got there first through
1119
+ // `cancelWakeTimer`).
1120
+ endWakeDeferral();
715
1121
  return;
716
1122
  }
717
1123
  if (ctx.isIdle?.() === false && !state.wakeForcedRelease) {
718
1124
  state.wakeTimer = undefined;
1125
+ // Still busy: this pass re-defers, so the deferral episode (and with
1126
+ // it the 5s deadline) must keep running instead of restarting.
719
1127
  scheduleDeferredWake(ctx);
720
1128
  return;
721
1129
  }
@@ -727,11 +1135,13 @@ export function createHerdsmanPiExtension(options: ExtensionOptions = {}) {
727
1135
  )) as ConnectionStateResponse | undefined;
728
1136
  if (!response) {
729
1137
  state.wakeTimer = undefined;
1138
+ endWakeDeferral();
730
1139
  return;
731
1140
  }
732
1141
  applyConnectionStateResponse(response, ctx);
733
1142
  } catch {
734
1143
  state.wakeTimer = undefined;
1144
+ endWakeDeferral();
735
1145
  // A failed load is only temporary: the batch stays pending and is
736
1146
  // retried on the next wake instead of being permanently suppressed.
737
1147
  ctx.ui.notify?.(
@@ -749,10 +1159,14 @@ export function createHerdsmanPiExtension(options: ExtensionOptions = {}) {
749
1159
  state.currentScope?.workspaceId !== ownerWorkspaceId
750
1160
  ) {
751
1161
  state.wakeTimer = undefined;
1162
+ // See the earlier generation re-check: the episode ends with it.
1163
+ endWakeDeferral();
752
1164
  return;
753
1165
  }
754
1166
  if (ctx.isIdle?.() === false && !state.wakeForcedRelease) {
755
1167
  state.wakeTimer = undefined;
1168
+ // See the earlier re-check: re-deferring keeps this episode's 5s
1169
+ // deadline intact.
756
1170
  scheduleDeferredWake(ctx);
757
1171
  return;
758
1172
  }
@@ -763,11 +1177,12 @@ export function createHerdsmanPiExtension(options: ExtensionOptions = {}) {
763
1177
  const batchOutcomes = batchProjection.outcomes.filter(
764
1178
  (outcome) =>
765
1179
  outcome.eventId > state.failedWakeThroughEventId &&
766
- !state.presentedEventIds.has(outcome.eventId) &&
1180
+ !alreadyPresented(outcome.eventId) &&
767
1181
  isWakeableEvent(batchEvents.find((event) => event.id === outcome.eventId)),
768
1182
  );
769
1183
  if (batchOutcomes.length === 0) {
770
1184
  state.wakeTimer = undefined;
1185
+ endWakeDeferral();
771
1186
  return;
772
1187
  }
773
1188
  const current = batchOutcomes;
@@ -785,11 +1200,18 @@ export function createHerdsmanPiExtension(options: ExtensionOptions = {}) {
785
1200
  // running turn can already see it.
786
1201
  const orchestratorBusy = ctx.isIdle?.() === false;
787
1202
  const wakeContent = formatAgentOutcomeUpdates(batchOutcomes);
1203
+ // Single line, injection path only: the decision that produced this
1204
+ // batch plus the signals it came from, so a wake that still arrives
1205
+ // late can be told apart from one parked by a stale gate.
1206
+ logHerdsmanPi(
1207
+ "info",
1208
+ `[herdsman-pi] wake inject deliverAs=followUp triggerTurn=${String(!orchestratorBusy)} forced=${state.wakeForcedRelease} runActive=${String(state.runActive)} isIdle=${ctx.isIdle === undefined ? "unknown" : String(ctx.isIdle())} eventIds=${batchOutcomes[0]?.eventId ?? 0}-${batchOutcomes.at(-1)?.eventId ?? 0} count=${batchOutcomes.length}`,
1209
+ );
788
1210
  try {
789
1211
  pi.sendMessage?.(
790
1212
  {
791
1213
  content: wakeContent,
792
- customType: "herdsman-wake-context",
1214
+ customType: WAKE_CONTEXT_CUSTOM_TYPE,
793
1215
  // Suppressed upstream errors are dropped from the injected
794
1216
  // context, but every other pending id stays listed so the
795
1217
  // evidence trail for the decision still names what was pending.
@@ -797,6 +1219,11 @@ export function createHerdsmanPiExtension(options: ExtensionOptions = {}) {
797
1219
  eventIds: batchEvents
798
1220
  .filter((event) => !batchSuppressedIds.has(event.id))
799
1221
  .map((event) => event.id),
1222
+ // The evidence channel: the ids whose content this message
1223
+ // actually carries (`eventIds` above lists everything that was
1224
+ // pending, including copies an earlier injection presented).
1225
+ // Only these prove consumption when a run drains this message.
1226
+ presentedEventIds: batchOutcomes.map((outcome) => outcome.eventId),
800
1227
  },
801
1228
  display: false,
802
1229
  },
@@ -804,6 +1231,18 @@ export function createHerdsmanPiExtension(options: ExtensionOptions = {}) {
804
1231
  ? { deliverAs: "followUp", triggerTurn: false }
805
1232
  : { deliverAs: "followUp", triggerTurn: true },
806
1233
  );
1234
+ // A queued (non-triggering) delivery rides the running turn: Pi parks it
1235
+ // in the agent's follow-up queue, which a run only drains when it reaches
1236
+ // its stop point. If the run it rode on already passed that point, nothing
1237
+ // drains it — the settlement drives a continuation instead (see
1238
+ // `driveWakeContinuation`), and until the content is seen in the
1239
+ // transcript the delivery is neither acknowledged nor injected again.
1240
+ if (orchestratorBusy) {
1241
+ for (const outcome of batchOutcomes) {
1242
+ state.wakeAwaitingConsumption.add(outcome.eventId);
1243
+ }
1244
+ state.wakeContinuationAttempts = 0;
1245
+ }
807
1246
  // A queued (non-triggering) delivery keeps the content available to
808
1247
  // the current turn through the context hook until it is settled or
809
1248
  // superseded by the next injection.
@@ -851,6 +1290,13 @@ export function createHerdsmanPiExtension(options: ExtensionOptions = {}) {
851
1290
  } catch {
852
1291
  state.deliveredBatch = undefined;
853
1292
  state.wakeRequested = false;
1293
+ // Nothing was queued by the refused injection, so nothing is awaiting
1294
+ // consumption on its account either: the ids only enter the awaiting set
1295
+ // once Pi accepted the message (see the queued branch above). The
1296
+ // refused injection consumed this deferral, so clearing it keeps the
1297
+ // elapsed deadline from letting a later, unrelated wake bypass the busy
1298
+ // gate.
1299
+ endWakeDeferral();
854
1300
  }
855
1301
  };
856
1302
  void startWake();
@@ -894,13 +1340,7 @@ export function createHerdsmanPiExtension(options: ExtensionOptions = {}) {
894
1340
  // A transient disconnect (reconnect) keeps the presentation guard so an
895
1341
  // event already presented in this scope session is not presented again;
896
1342
  // only a genuine role/scope loss or shutdown resets it.
897
- if (!options.preservePresented) {
898
- state.presentedEventIds.clear();
899
- // The delivery queue dies with the presentation guard: once the guard is
900
- // gone the daemon's pending events can be presented (and acknowledged)
901
- // again, so keeping the old queue would only risk a stale id.
902
- state.unackedDelivered.clear();
903
- }
1343
+ if (!options.preservePresented) clearDeliveryBookkeeping();
904
1344
  state.reconnectingFromOn = false;
905
1345
  setHerdsmanUi(ctx);
906
1346
  };
@@ -930,8 +1370,7 @@ export function createHerdsmanPiExtension(options: ExtensionOptions = {}) {
930
1370
  cancelWake();
931
1371
  state.failedWakeThroughEventId = 0;
932
1372
  state.pendingEvents = [];
933
- state.presentedEventIds.clear();
934
- state.unackedDelivered.clear();
1373
+ clearDeliveryBookkeeping();
935
1374
  setHerdsmanUi(ctx);
936
1375
  };
937
1376
 
@@ -1337,6 +1776,39 @@ export function createHerdsmanPiExtension(options: ExtensionOptions = {}) {
1337
1776
 
1338
1777
  pi.on("message_end", (event: Record<string, unknown>) => {
1339
1778
  const message = record(event.message);
1779
+ if (message.role === "custom" && message.customType === WAKE_CONTEXT_CUSTOM_TYPE) {
1780
+ // Consumption evidence: the wake content reached the transcript, so some
1781
+ // run drained the queued follow-up and carried the update out. The ids the
1782
+ // message presented are confirmed — no further continuation is owed for
1783
+ // them, and they are now authorised for acknowledgement on their own,
1784
+ // whatever conclusion the turn reached.
1785
+ const consumedEventIds = wakeConsumedEventIds(message);
1786
+ if (consumedEventIds.length === 0) {
1787
+ // Nothing may be confirmed from a message whose evidence cannot be read
1788
+ // (a shape this extension never emits), and that must not pass silently:
1789
+ // the ids that are still waiting for their evidence must be named so the
1790
+ // stuck delivery is traceable.
1791
+ const awaiting = [...state.wakeAwaitingConsumption].sort((left, right) => left - right);
1792
+ logHerdsmanPi(
1793
+ "warn",
1794
+ `[herdsman-pi] wake consumption evidence unusable customType=${String(message.customType)} awaiting=${awaiting.length === 0 ? "none" : awaiting.join(",")} detailsKeys=${Object.keys(record(message.details)).join(",") || "none"} · no event confirmed by this message`,
1795
+ );
1796
+ }
1797
+ for (const eventId of consumedEventIds) {
1798
+ state.wakeAwaitingConsumption.delete(eventId);
1799
+ // The content reached the transcript, so this id counts as presented from
1800
+ // now on: a daemon redelivery of it (it is still unacknowledged whenever
1801
+ // its batch was already dropped) must not inject the same update again.
1802
+ // The scope's own guard may be gone — the copy can be drained after a
1803
+ // role/scope reset cleared it — so the session-wide suppression set keeps
1804
+ // the id as well; it is released when the daemon confirms the id.
1805
+ state.presentedEventIds.add(eventId);
1806
+ state.wakeSuppressedEventIds.add(eventId);
1807
+ // Only an id the daemon can still be told about is worth remembering.
1808
+ if (state.unackedDelivered.has(eventId)) state.wakeConsumptionObserved.add(eventId);
1809
+ }
1810
+ if (state.wakeAwaitingConsumption.size === 0) state.wakeContinuationAttempts = 0;
1811
+ }
1340
1812
  if (message.role !== "assistant") return;
1341
1813
  const stopReason = stringValue(message.stopReason);
1342
1814
  if (state.deliveredBatch) {
@@ -1365,6 +1837,13 @@ export function createHerdsmanPiExtension(options: ExtensionOptions = {}) {
1365
1837
  });
1366
1838
 
1367
1839
  pi.on("agent_start", () => {
1840
+ // Deliberately no clearing here: a starting run does drain Pi's follow-up
1841
+ // queue, but that is not evidence that the content reached the transcript —
1842
+ // a run can end in error or be aborted before its stop point and leave the
1843
+ // queue untouched. Only the consumption signal (the hidden wake message's
1844
+ // `message_end`) writes a delivery off, so an update a run did not take
1845
+ // along is still driven out at the next settlement instead of waiting for
1846
+ // the next user message.
1368
1847
  if (state.runActive) return;
1369
1848
  state.runActive = true;
1370
1849
  state.pinnedContext =
@@ -1400,6 +1879,13 @@ export function createHerdsmanPiExtension(options: ExtensionOptions = {}) {
1400
1879
  // to the orchestrator in the turn that presented them, so they are not
1401
1880
  // re-listed here. The queued follow-up message itself keeps the wider
1402
1881
  // `details.eventIds` set (everything still unconfirmed).
1882
+ //
1883
+ // The pin is not a second delivery of the update: it makes the same queued
1884
+ // copy visible early to the turn that is running, it is not written to the
1885
+ // transcript, it is never acknowledged on its own, and it is not consumption
1886
+ // evidence — only the hidden wake message's `message_end` is. So it may not be
1887
+ // dropped in the name of "one copy only" either: it is the only way this turn
1888
+ // ever sees the update.
1403
1889
  const queuedWake = state.wakeContext;
1404
1890
  if (queuedWake) {
1405
1891
  additions.push({
@@ -1417,16 +1903,12 @@ export function createHerdsmanPiExtension(options: ExtensionOptions = {}) {
1417
1903
  pi.on("agent_settled", async (_event: unknown, ctx: PiContext) => {
1418
1904
  state.runActive = false;
1419
1905
  state.pinnedContext = undefined;
1420
- const batch = state.deliveredBatch;
1421
- if (!batch) {
1422
- state.wakeDeferredUntilSettled = false;
1423
- scheduleWake(ctx);
1424
- return;
1425
- }
1426
- state.deliveredBatch = undefined;
1427
- state.ackInFlight = true;
1428
- const stillOwner =
1429
- state.isOrchestrator && state.currentScope?.terminalId === batch.ownerTerminalId;
1906
+ // A wake delivered as a queued follow-up is drained only when a run reaches
1907
+ // its stop point. If the run that received it settled without draining it,
1908
+ // no further run exists to carry the update out: drive a continuation (up to
1909
+ // the per-delivery bound, see `driveWakeContinuation`), which is what
1910
+ // surfaces the queued update.
1911
+ driveWakeContinuation(ctx);
1430
1912
  const failBatch = () => {
1431
1913
  ctx.ui.notify?.(
1432
1914
  "Herdsman couldn’t acknowledge agent updates · updates remain pending",
@@ -1443,47 +1925,61 @@ export function createHerdsmanPiExtension(options: ExtensionOptions = {}) {
1443
1925
  scheduleWake(ctx);
1444
1926
  };
1445
1927
 
1446
- // The delivery queue — not the injection snapshot — is what gets
1447
- // acknowledged: it holds every event handed to Pi that is still
1448
- // unconfirmed (a release merges batches instead of replacing them), is
1449
- // id-ascending, and an id leaves it only when the daemon accepts it.
1450
- //
1451
- // An event whose own acknowledgement already failed is left out here: it is
1452
- // never retried by this path (the daemon's cursor advance sweeps it, and a
1453
- // retry would reset its attempt/backoff accounting), but it stays in the
1454
- // queue until that cursor or a scope reset confirms it. A failed or
1455
- // dead-lettered acknowledgement thus never depends on the timing of the
1456
- // release to stay recoverable.
1457
- //
1458
- // Restricting to a *live* row with `attempts === 0` has two holes on
1459
- // purpose. `?? 0` covers an event the live projection no longer holds at
1460
- // all (the server stopped listing it, or `failedWakeThroughEventId`
1461
- // filters it out): the projection cannot tell us it already failed, so the
1462
- // event gets one more attempt — a deliberate self-healing opportunity that
1463
- // then accumulates on the queue copy's counter and can reach
1464
- // MAX_ACK_ATTEMPTS instead of restarting at 1 every round.
1465
- const ackable = unackedDeliveredAscending().filter(
1466
- (event) =>
1467
- (state.pendingEvents.find((pending) => pending.id === event.id)?.attempts ?? 0) === 0,
1468
- );
1469
- if (ackable.length === 0) {
1470
- finishBatch();
1928
+ const batch = state.deliveredBatch;
1929
+ if (!batch) {
1930
+ state.wakeDeferredUntilSettled = false;
1931
+ // A queue can outlive its batch: the batch record is dropped as soon as
1932
+ // nothing awaits consumption, while an acknowledgement it still owed can be
1933
+ // missing — most visibly when the daemon was unreachable while the content
1934
+ // was consumed. Without this pass those ids would pin the watermark until an
1935
+ // unrelated event happened to form a new batch, and a session that receives
1936
+ // no further update would never confirm what its transcript already holds.
1937
+ // With no batch left, no turn can vouch for an untracked delivery, so only
1938
+ // consumption evidence confirms an id.
1939
+ const stranded = confirmableDeliveryPrefix(false);
1940
+ if (
1941
+ stranded.confirmablePrefix.length > 0 &&
1942
+ state.isOrchestrator &&
1943
+ state.client !== undefined &&
1944
+ state.connected
1945
+ ) {
1946
+ const resumeAckInFlight = state.ackInFlight;
1947
+ state.ackInFlight = true;
1948
+ try {
1949
+ await acknowledgeEventIds(stranded.confirmablePrefix, { notify: true }, ctx);
1950
+ } finally {
1951
+ state.ackInFlight = resumeAckInFlight;
1952
+ }
1953
+ }
1954
+ scheduleWake(ctx);
1471
1955
  return;
1472
1956
  }
1473
1957
 
1474
- if (
1475
- (!batch.assistantFinalSucceeded && !batch.abortedByUser) ||
1476
- batch.invalidated ||
1477
- !stillOwner ||
1478
- !state.client ||
1479
- !state.connected
1480
- ) {
1481
- failBatch();
1958
+ const stillOwner =
1959
+ state.isOrchestrator && state.currentScope?.terminalId === batch.ownerTerminalId;
1960
+ const { deliveryQueue, confirmablePrefix, blockedByMissingTurn, stillAwaitingConsumption } =
1961
+ confirmableDeliveryPrefix(batch.assistantFinalSucceeded || batch.abortedByUser);
1962
+ const reachable = stillOwner && state.client !== undefined && state.connected;
1963
+ if (!stillAwaitingConsumption) state.deliveredBatch = undefined;
1964
+ state.ackInFlight = true;
1965
+ if (!reachable) {
1966
+ // Unreachable (ownership gone, or a disconnect): attempting an
1967
+ // acknowledgement now would only be refused and would burn the event's
1968
+ // attempt budget, while the queue keeps every event for the next
1969
+ // settlement on a live connection.
1970
+ if (deliveryQueue.length > 0) failBatch();
1971
+ finishBatch();
1972
+ return;
1973
+ }
1974
+ if (blockedByMissingTurn) failBatch();
1975
+ if (confirmablePrefix.length === 0) {
1976
+ // Nothing to confirm: an empty queue (the batch is settled) or a prefix
1977
+ // blocked by unconsumed content, which must not be confirmed yet.
1482
1978
  finishBatch();
1483
1979
  return;
1484
1980
  }
1485
1981
 
1486
- await acknowledgeEventIds(ackable, { notify: true }, ctx);
1982
+ await acknowledgeEventIds(confirmablePrefix, { notify: true }, ctx);
1487
1983
  finishBatch();
1488
1984
  });
1489
1985