terret-core 0.1.0 → 0.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
data/lib/terret/loop.rb CHANGED
@@ -1,6 +1,34 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Terret
4
+ # Raised when run_turn is called on an agent already mid-turn: concurrent
5
+ # turns would interleave the durable log, so this refuses loudly instead.
6
+ class TurnAlreadyRunning < StandardError; end
7
+
8
+ # Raised when run_turn is called on a session whose log still holds an open
9
+ # turn. A second turn/start would strand whatever the open turn owes —
10
+ # resumable? scans from the LAST turn/start — and leave the projection
11
+ # carrying an assistant tool call with no result, which a real adapter
12
+ # rejects outright, forever. Resume the open turn instead.
13
+ class TurnOpenInLog < StandardError; end
14
+
15
+ # Raised on id or session collision in spawn_agent: silent replacement
16
+ # leaked the old agent's forked context and orphaned its mid-turn state
17
+ # (plan §14). Dispose the old agent first, explicitly.
18
+ class AgentExists < StandardError; end
19
+
20
+ # The hard cap on agents per process (plan §14, blast-radius): shard by
21
+ # process rather than raising the cap when this bites.
22
+ class AgentCapExceeded < StandardError; end
23
+
24
+ # Raised when a turn is asked of an agent dispose_agent already tore down.
25
+ # Its forked context is gone along with every effect it owned, so the honest
26
+ # answer is a refusal rather than a turn half-working against a dead fork.
27
+ class AgentDisposed < StandardError; end
28
+
29
+ # Claimed messages are [type, text] pairs — "user/message" for the turn's
30
+ # input, "context/injected" for steers drained from the inbox — so the log
31
+ # records provenance even after a pre_step listener rewrites the claim.
4
32
  Claim = Data.define(:messages, :rejected, :reason) do
5
33
  def self.of(messages) = new(messages:, rejected: false, reason: nil)
6
34
  def self.reject(reason:) = new(messages: [], rejected: true, reason:)
@@ -10,12 +38,27 @@ module Terret
10
38
  attr_reader :id, :session_id, :ctx
11
39
  attr_accessor :status
12
40
 
41
+ # True when no human can be asked about this agent's tool calls. Nothing
42
+ # routes an approval request for a subagent's session to an operator — the
43
+ # parent's log does not even name it (docs/subagents.md §2) — so the
44
+ # approvals gate denies rather than parking on a verdict that can never
45
+ # arrive. Set by the subagent provider on the children it spawns; a
46
+ # top-level agent is attended and parks exactly as it always did.
47
+ attr_accessor :unattended
48
+
13
49
  def initialize(id:, session_id:, ctx:)
14
50
  @id = id
15
51
  @session_id = session_id
16
52
  @ctx = ctx # a forked, agent-scoped context
17
53
  @inbox = [] # injected context waits here until a waking message
54
+ @unattended = false
55
+ # :idle | :running | :waiting_approval (parked in the tools pipeline) |
56
+ # :stopping (cancelled, still finishing) | :done (disposed, terminal).
57
+ # docs/subagents.md §8: :failed is a TURN status, not an agent one, and
58
+ # :waiting_input stays vocabulary until something parks a turn on it.
18
59
  @status = :idle
60
+ @cancelled = false
61
+ @cancel_reason = nil
19
62
  end
20
63
 
21
64
  def inject(text)
@@ -23,6 +66,45 @@ module Terret
23
66
  end
24
67
 
25
68
  def drain_inbox = @inbox.slice!(0..)
69
+
70
+ def inbox_empty? = @inbox.empty?
71
+
72
+ # Steers drained for a step that never happened go back to the front of
73
+ # the queue, so a rejected claim cannot silently eat an inject.
74
+ def requeue(items) = @inbox.unshift(*items)
75
+
76
+ attr_reader :cancel_reason
77
+
78
+ # Cooperative stop: the loop honors it at step boundaries. Mid-stream
79
+ # abort arrives with the async task-tree work (plan §8); until then this
80
+ # is the honest synchronous form.
81
+ #
82
+ # A cancel raised DURING a turn is per-turn and best-effort: whether that
83
+ # turn rejects, fails, or completes before a boundary honors it, turning's
84
+ # ensure clears the flag, so it never haunts the next one. A cancel on an
85
+ # IDLE agent has no turn to clear it and so it persists — the next turn
86
+ # honors it at once and closes cancelled having spent no step, which is
87
+ # what a stop pressed just before a message lands should do.
88
+ #
89
+ # The status moves only from :running: :stopping is a sub-state of a turn
90
+ # that is still working and is no longer going to finish, so there is
91
+ # nothing for it to mean on an idle agent — and an idle agent left
92
+ # non-idle by a cancel could never start the turn that would clear it.
93
+ # A parked agent keeps saying :waiting_approval, which is still true; the
94
+ # approvals gate's restore is what reads the standing cancel and returns
95
+ # it to :stopping rather than to :running.
96
+ def cancel(reason = nil)
97
+ @cancel_reason = reason
98
+ @cancelled = true
99
+ @status = :stopping if @status == :running
100
+ end
101
+
102
+ def cancelled? = !!@cancelled
103
+
104
+ def clear_cancel!
105
+ @cancelled = false
106
+ @cancel_reason = nil
107
+ end
26
108
  end
27
109
 
28
110
  # ctx.loop — the default driver. A step is one model request plus the tool
@@ -32,45 +114,218 @@ module Terret
32
114
  class Loop < Hames::Service
33
115
  service_key :loop
34
116
  inject :sessions, :tools, :llm, :prompt
117
+ config_schema max_agents: { type: Integer, default: 128,
118
+ doc: "cap on concurrently live agents in this loop" }
35
119
 
36
120
  MAX_STEPS = 25
37
121
 
38
122
  def start(ctx)
39
123
  @ctx = ctx
124
+ @agents = {}
125
+ @by_session = {}
126
+ @max_agents = config[:max_agents] || 128
40
127
  end
41
128
 
42
- def spawn_agent(session_id:, id: "agent-#{session_id}")
43
- Agent.new(id:, session_id:, ctx: @ctx.fork)
129
+ def reconfigure(config)
130
+ @max_agents = config[:max_agents] || 128
44
131
  end
45
132
 
133
+ # `parent:` is the context the agent's own scope forks from. It defaults to
134
+ # this service's root exactly as it always did, so every interface spawning
135
+ # a top-level agent keeps its call site; the subagent provider is the one
136
+ # caller that passes something else — the CALLING agent's fork, which is
137
+ # what makes a child inherit that agent's roster and policy floor instead
138
+ # of the root's (docs/subagents.md §3).
139
+ def spawn_agent(session_id:, id: "agent-#{session_id}", parent: @ctx)
140
+ raise AgentExists, "agent #{id} already exists" if @agents.key?(id)
141
+ if (live = @by_session[session_id])
142
+ raise AgentExists, "session #{session_id} already has agent #{live.id}"
143
+ end
144
+ if @agents.size >= @max_agents
145
+ raise AgentCapExceeded,
146
+ "#{@agents.size} agents live; max_agents is #{@max_agents}"
147
+ end
148
+
149
+ agent = Agent.new(id:, session_id:, ctx: parent.fork)
150
+ @agents[id] = agent
151
+ @by_session[session_id] = agent
152
+ agent
153
+ end
154
+
155
+ # Live-agent lookup for interfaces (§9.2); nil when never spawned.
156
+ def agent(id) = @agents[id]
157
+
158
+ # Session -> live agent, for services that learn a session id from an
159
+ # event and need the agent (approvals flips status through this).
160
+ def agent_for_session(session_id) = @by_session[session_id]
161
+
162
+ # Tear an idle agent down: its forked context disposes (listeners and
163
+ # effects die with it) and both registry slots free. Mid-turn agents
164
+ # refuse — cancel or resolve first.
165
+ def dispose_agent(id)
166
+ agent = @agents.fetch(id)
167
+ unless agent.status == :idle
168
+ raise TurnAlreadyRunning, "agent #{id} is #{agent.status}; dispose only idle agents"
169
+ end
170
+
171
+ tear_down_agent(agent)
172
+ agent
173
+ end
174
+
175
+ # Lifecycle teardown: the loader calls this when the loop row unloads
176
+ # (Boot.shutdown unloads every row through unload!, which calls stop). The
177
+ # loop is a plugin like any other, and its agents are state it holds — an
178
+ # idle fork left mounted keeps its per-agent policy listeners and forked
179
+ # tool registrations alive past the shutdown meant to end them. Every agent
180
+ # goes, whatever its status, because the process is coming down; best-effort
181
+ # and idempotent, so one fork whose disposal raises cannot strand the rest
182
+ # and a second call finds nothing left to do.
183
+ def stop(_ctx)
184
+ @agents.values.each do |agent|
185
+ tear_down_agent(agent)
186
+ rescue StandardError => e
187
+ warn "terret: loop shutdown: agent #{agent.id} would not dispose: #{e.class}: #{e.message}"
188
+ end
189
+ end
190
+
191
+ TurnState = Struct.new(:status, :steered)
192
+
46
193
  # Runs one turn for `input`. Returns the turn status symbol.
47
194
  def run_turn(agent, input)
195
+ # Only an idle agent is asked this: while it is mid-turn the log's open
196
+ # turn is its own, TurnAlreadyRunning is the accurate answer, and the
197
+ # socket's raced-wake requeue is written against it.
198
+ if agent.status == :idle && resumable?(agent.session_id)
199
+ raise TurnOpenInLog,
200
+ "session #{agent.session_id} has an open turn; resume_turn it"
201
+ end
202
+
203
+ turning(agent) do |state, sessions, sid|
204
+ sessions.append(sid, "turn/start", { agent: agent.id })
205
+ step_loop(agent, state, pending: input.nil? ? [] : [["user/message", input]], steps: 0)
206
+ end
207
+ end
208
+
209
+ # Continue a turn the log left open (a process death mid-park, plan
210
+ # §6.3/§12 M6). No second turn/start — the open one is already durable.
211
+ # The open step completes first: tool calls owed by the last assistant
212
+ # message that lack a tool/result re-execute through the pipeline, where
213
+ # the approvals gate reads verdicts from the log — an approved call runs,
214
+ # an unresolved one parks again on its standing request. Then stepping
215
+ # continues as normal.
216
+ def resume_turn(agent)
217
+ raise ArgumentError, "session #{agent.session_id} has no open turn" unless resumable?(agent.session_id)
218
+
219
+ # A transient failure here (an LLM outage) must leave the turn open for
220
+ # the next stimulus: closing it would strand the owed tool call for good.
221
+ turning(agent, close_on_failure: false) do |state, _sessions, _sid|
222
+ step_loop(agent, state, pending: [], steps: complete_dangling(agent))
223
+ end
224
+ end
225
+
226
+ # The log has a turn/start after its last turn/end.
227
+ def resumable?(session_id)
228
+ events = @ctx[:sessions].fetch(session_id).events
229
+ opened = events.rindex { |e| e.type == "turn/start" }
230
+ return false unless opened
231
+
232
+ events[opened..].none? { |e| e.type == "turn/end" }
233
+ end
234
+
235
+ private
236
+
237
+ # The disposal both dispose_agent and stop share. The registry slots are
238
+ # freed in an ensure, so a fork disposer that raises still returns the cap
239
+ # slot the agent held rather than leaving the child registered forever while
240
+ # Subagents#dispose only warns — a run of those would exhaust max_agents.
241
+ # The raise still propagates: the caller decides what to make of it.
242
+ #
243
+ # Fork disposal reaps the agent's registrations, but the process state a
244
+ # tool call created — ctx[:shell]'s bash, ctx[:terminals]' PTYs, ctx[:jobs]'
245
+ # subprocesses — is root-mounted and keyed by session, so it survives the
246
+ # fork. The emit is that reaping signal; core stays decoupled from the
247
+ # optional exec gem by only emitting, and emit isolates a listener fault so
248
+ # a reaping bug cannot strand the disposal that already happened. It fires
249
+ # from the ensure too, so a raised fork disposal still sweeps the session's
250
+ # processes.
251
+ def tear_down_agent(agent)
252
+ agent.ctx.dispose!
253
+ ensure
254
+ agent.status = :done # terminal: the handle now refuses a turn outright
255
+ @agents.delete(agent.id)
256
+ @by_session.delete(agent.session_id)
257
+ @ctx.emit("agent/disposed", agent.session_id)
258
+ end
259
+
260
+ # Shared turn envelope: the status guard, the failure rescue, and the
261
+ # turn/end ensure. Both entry points run their body inside it.
262
+ def turning(agent, close_on_failure: true)
263
+ raise AgentDisposed, "agent #{agent.id} was disposed" if agent.status == :done
264
+ unless agent.status == :idle
265
+ raise TurnAlreadyRunning, "agent #{agent.id} is #{agent.status}"
266
+ end
267
+
48
268
  ctx = agent.ctx
49
269
  sessions = ctx[:sessions]
50
270
  sid = agent.session_id
51
271
  agent.status = :running
272
+ state = TurnState.new(:completed, [])
52
273
 
53
- sessions.append(sid, "turn/start", { agent: agent.id })
54
- status = :completed
55
- steps = 0
274
+ begin
275
+ yield state, sessions, sid
276
+ rescue Exception
277
+ state.status = :failed
278
+ agent.requeue(state.steered) unless state.steered.empty?
279
+ raise
280
+ ensure
281
+ begin
282
+ ctx.serial("agent/turn_stopping", agent)
283
+ if state.status != :failed || close_on_failure
284
+ payload = { status: state.status }
285
+ payload[:reason] = agent.cancel_reason if state.status == :cancelled && agent.cancel_reason
286
+ sessions.append(sid, "turn/end", payload)
287
+ end
288
+ ensure
289
+ agent.clear_cancel!
290
+ agent.status = :idle
291
+ end
292
+ end
293
+ state.status
294
+ end
56
295
 
57
- # claim next-step input plus anything waiting in the inbox
58
- pending = [input, *agent.drain_inbox].compact
296
+ # The step cycle run_turn always had, extracted so resume_turn can enter
297
+ # it mid-turn. `pending` holds [type, text] pairs (Task 5); `steps` is
298
+ # how many step/starts the turn has already logged.
299
+ def step_loop(agent, state, pending:, steps:)
300
+ ctx = agent.ctx
301
+ sessions = ctx[:sessions]
302
+ sid = agent.session_id
59
303
 
60
304
  loop do
305
+ if agent.cancelled?
306
+ state.status = :cancelled
307
+ return
308
+ end
309
+
310
+ # anything injected since the last step rides along with this one
311
+ state.steered = agent.drain_inbox
312
+ pending.concat(state.steered.map { |t| ["context/injected", t] })
313
+
61
314
  claim = ctx.waterfall("agent/pre_step", Claim.of(pending)) { |c| c }
62
315
  if claim.rejected || (steps.zero? && claim.messages.empty? && pending.empty?)
63
316
  # a rejected or empty first claim still closes a durable turn that
64
317
  # spent no step, so the log records the attempt
65
- status = claim.rejected ? :rejected : :empty
66
- break
318
+ agent.requeue(state.steered) if claim.rejected
319
+ state.status = claim.rejected ? :rejected : :empty
320
+ return
67
321
  end
68
322
 
69
323
  steps += 1
70
324
  raise "runaway turn" if steps > MAX_STEPS
71
325
 
72
326
  sessions.append(sid, "step/start", { n: steps })
73
- claim.messages.each { |t| sessions.append(sid, "user/message", { text: t }) }
327
+ claim.messages.each { |(type, text)| sessions.append(sid, type, { text: text }) }
328
+ state.steered = [] # once logged, these must never requeue
74
329
  pending = []
75
330
 
76
331
  history = sessions.derive_messages(sid)
@@ -79,36 +334,292 @@ module Terret
79
334
  request = ctx.waterfall("agent/request", request)
80
335
  sessions.assert_log_invariant!(sid, request.messages)
81
336
 
337
+ usage = nil
338
+ # A provider's deltas break at token boundaries, so a credential can
339
+ # straddle two of them: "...is sk-a" + "bc123def" defeats a pattern
340
+ # that matches the whole secret perfectly, and the tail lands in the
341
+ # log verbatim. A scrubber can only be trusted with text it sees
342
+ # WHOLE, so while anything is scrubbing the entire run is held and
343
+ # appended as ONE chunk at the run's end. That trades live token
344
+ # streaming in the chunk log for the §13 guarantee, and the trade is
345
+ # only affordable because chunks are replay/UI fidelity: they are not
346
+ # projected by derive_messages, so nothing here can move the digest,
347
+ # and only their concatenation is contractual (docs/exec.md §6).
348
+ #
349
+ # Captured once rather than per delta, so one run is governed by one
350
+ # policy even if a scrubber is registered while it streams.
351
+ scrubbing = sessions.scrubbing?
352
+ run = +""
82
353
  message = ctx[:llm].stream(ctx, role: :main, request: request) do |ev|
83
354
  case ev
84
355
  when LLM::TextDelta
85
- sessions.append(sid, "assistant/chunk", { text: ev.text })
356
+ if scrubbing
357
+ run << ev.text
358
+ else
359
+ sessions.append(sid, "assistant/chunk", { text: ev.text })
360
+ end
361
+ when LLM::ToolCallEnd, LLM::MessageStop
362
+ flush_run(sessions, sid, run) # this run of text ended here
363
+ when LLM::Usage
364
+ usage = ev
86
365
  end
87
366
  end
88
- sessions.append(sid, "assistant/message", { parts: message.parts })
367
+ # Belt and braces for an adapter that ends without a MessageStop. A
368
+ # stream that RAISES loses the whole held run: run_turn closes the
369
+ # failed turn (close_on_failure), so the next turn starts fresh with
370
+ # no assistant/message for this step either — the chunk log and the
371
+ # authoritative log agree about a step that never completed.
372
+ flush_run(sessions, sid, run)
373
+ sessions.append(sid, "assistant/message",
374
+ { parts: message.parts.map { |p| LLM.encode_part(p) } })
375
+ step_end = usage ? { n: steps, usage: usage.to_h } : { n: steps }
89
376
 
90
377
  calls = message.tool_calls
91
378
  if calls.empty?
92
- sessions.append(sid, "step/end", { n: steps })
93
- break # nothing owed
379
+ sessions.append(sid, "step/end", step_end)
380
+ state.status = :cancelled if agent.cancelled?
381
+ return # nothing owed
94
382
  end
95
383
 
96
- calls.each do |tc|
97
- sessions.append(sid, "tool/call", { id: tc.id, name: tc.name, args: tc.args })
98
- result = ctx[:tools].execute(
99
- Tools::Call.new(id: tc.id, name: tc.name, args: tc.args, session_id: sid)
100
- )
101
- sessions.append(sid, "tool/result",
102
- { id: result.id, content: result.content, error: result.error })
384
+ execute_batch(agent, ctx, sessions, sid, calls)
385
+ sessions.append(sid, "step/end", step_end)
386
+ # redundant under the sync driver (the next iteration's top check
387
+ # would catch it); becomes load-bearing once tools can yield (§8)
388
+ if agent.cancelled?
389
+ state.status = :cancelled
390
+ return
103
391
  end
104
- sessions.append(sid, "step/end", { n: steps })
105
392
  # tools owe another request -> next step
106
393
  end
394
+ end
395
+
396
+ # Resume replays a tool call decoded from the LOG, and the log is
397
+ # scrubbed: whatever carried a credential carries the replacement token
398
+ # instead. Re-running `deploy --key [REDACTED]` is not a retry of what the
399
+ # model asked for, it is a DIFFERENT command with the same name — and for
400
+ # Bash or Write that difference is a real, irreversible side effect. The M6
401
+ # at-least-once contract (docs/lifecycle.md) yields to honesty here: the
402
+ # call is refused with a result that says why, and the model's next step
403
+ # decides what to do about it.
404
+ #
405
+ # The NAME is checked as well as the args, though a redacted name could
406
+ # never have resolved anyway: "no such tool" would tell the model its
407
+ # roster is broken, when what actually happened is that the log rewrote
408
+ # its own record of the call. One refusal, one accurate reason, wherever
409
+ # the token landed.
410
+ #
411
+ # Only the redactor's own token is known. A scrubber registered directly
412
+ # with some other replacement is not detectable from here, and a call it
413
+ # rewrote still replays — stated in docs/exec.md §6 rather than guessed at.
414
+ def redaction_token(ctx) = ctx.service?(:redactor) ? ctx[:redactor].replacement : nil
415
+
416
+ def redacted?(value, token)
417
+ case value
418
+ when String then value.include?(token)
419
+ when Array then value.any? { |v| redacted?(v, token) }
420
+ when Hash then value.any? { |_k, v| redacted?(v, token) }
421
+ else false
422
+ end
423
+ end
424
+
425
+ # Append a held run of text as one chunk. No cuts anywhere: a partial run
426
+ # is exactly what a scrubber cannot be shown, and a whole one never splits
427
+ # a multibyte character either.
428
+ def flush_run(sessions, sid, buffer)
429
+ return if buffer.empty?
430
+
431
+ # Emptied first: the buffer records what has NOT been logged, and it
432
+ # must not still claim text an append is already carrying.
433
+ text = buffer.dup
434
+ buffer.clear
435
+ sessions.append(sid, "assistant/chunk", { text: text })
436
+ end
437
+
438
+ # One assistant message's tool calls, run as maximal runs of the
439
+ # concurrency their definitions declare (docs/subagents.md §5).
440
+ #
441
+ # MAXIMAL runs, rather than every parallel call in the message gathered
442
+ # together, is what preserves a serial call's meaning: it is a barrier of
443
+ # one, and nothing may reorder across it.
444
+ #
445
+ # Cancellation lands BETWEEN runs, because a barrier cannot be interrupted
446
+ # from outside once it starts. Every call still ends with a result either
447
+ # way — the projection may never hold a call without one.
448
+ def execute_batch(agent, ctx, sessions, sid, calls)
449
+ maximal_runs(ctx, calls).each do |concurrent, run|
450
+ if agent.cancelled?
451
+ run.each do |tc|
452
+ log_call(sessions, sid, tc)
453
+ sessions.append(sid, "tool/result",
454
+ { id: tc.id, content: nil, error: "cancelled before execution" })
455
+ end
456
+ next
457
+ end
458
+
459
+ # The whole run is logged before any of it executes: it is launched as
460
+ # a group, so there is no per-call moment to interleave a call event
461
+ # into.
462
+ run.each { |tc| log_call(sessions, sid, tc) }
463
+ results = if concurrent
464
+ execute_together(ctx, sid, run)
465
+ else
466
+ # A barrier of one gets the same guard the parallel barrier
467
+ # applies (execute_together, via guarded_call): a listener
468
+ # that raises AROUND the handler — the one crash
469
+ # Registry#execute does not itself render — is this call's
470
+ # error result, not an exception that fails the turn with the
471
+ # tool/call already logged and no tool/result under it. That
472
+ # dangling call cannot be repaired: run_turn closes a failed
473
+ # turn, resumable? goes false, and every later turn's
474
+ # projection carries an assistant tool_call a real provider
475
+ # rejects. One call's failure is one call's result, serial or
476
+ # parallel.
477
+ run.map { |tc| guarded_call(ctx, sid, tc) }
478
+ end
479
+ # In CALL order, always. Concurrency may change when work happens; it
480
+ # may not change what the log says happened, because derive_messages
481
+ # projects the model's history from this order and resume rebuilds it.
482
+ results.each do |r|
483
+ sessions.append(sid, "tool/result", { id: r.id, content: r.content, error: r.error })
484
+ end
485
+ end
486
+ end
487
+
488
+ # [[concurrent?, [call, ...]], ...]. A lone :parallel call is executed as
489
+ # a run of one — there is nothing to overlap it with, and a fiber for it
490
+ # would buy latency rather than spend it.
491
+ def maximal_runs(ctx, calls)
492
+ calls.chunk_while { |a, b| parallel?(ctx, a) && parallel?(ctx, b) }
493
+ .map { |run| [run.length > 1, run] }
494
+ end
495
+
496
+ # An unknown tool is a barrier of one: it cannot be dispatched, and the
497
+ # pipeline renders its "no such tool" error as an ordinary result.
498
+ def parallel?(ctx, call)
499
+ ctx[:tools].fetch(call.name).concurrency == :parallel
500
+ rescue KeyError
501
+ false
502
+ end
107
503
 
108
- ctx.serial("agent/turn_stopping", agent)
109
- sessions.append(sid, "turn/end", { status: status })
110
- agent.status = :idle
111
- status
504
+ # The barrier: one Async task per call on the one reactor, and dispatch
505
+ # completes only when every call has. Async is not a dependency of this
506
+ # gem — without a reactor the run still completes as a group, one call at
507
+ # a time, exactly the contract Hames' own :parallel dispatch keeps.
508
+ #
509
+ # Nothing a call does escapes its own fiber. Registry#execute already
510
+ # renders a handler's crash as an error Result; a listener that raises
511
+ # AROUND it escapes that rendering, and letting it out here would abandon
512
+ # the whole run — siblings that had already done their work would lose
513
+ # their results, and the projection would be left owing calls it can never
514
+ # be given results for, because the turn closes and `resumable?` goes
515
+ # false. So the same shape is applied one level out: one call's failure is
516
+ # one call's error result, and every other result still appends.
517
+ def execute_together(ctx, sid, run)
518
+ task = defined?(Async::Task) ? Async::Task.current? : nil
519
+ # Guarded on both paths. Without a reactor the run is still a run, and a
520
+ # host that never loaded async must not be the one deployment where a
521
+ # raising listener eats its siblings' results.
522
+ return run.map { |tc| guarded_call(ctx, sid, tc) } unless task
523
+
524
+ results = Array.new(run.length)
525
+ children = run.each_with_index.map do |tc, i|
526
+ task.async { results[i] = guarded_call(ctx, sid, tc) }
527
+ end
528
+ # Every sibling is awaited whatever the first wait does, so the batch's
529
+ # bookkeeping finishes even while the task tree is being torn down —
530
+ # Async::Stop is not a StandardError, and a half-awaited run would leave
531
+ # fibers writing into an array nobody is watching. The first exception
532
+ # is re-raised once there is nothing left in flight.
533
+ stopped = nil
534
+ children.each do |child|
535
+ child.wait
536
+ rescue Exception => e # rubocop:disable Lint/RescueException
537
+ stopped ||= e
538
+ end
539
+ raise stopped if stopped
540
+
541
+ results
542
+ end
543
+
544
+ # The same split Registry#execute makes one level in, so a plugin cannot
545
+ # tell which layer caught it: a Failure's message is the whole story and
546
+ # renders alone, while any other exception keeps its class name, because
547
+ # a crash's class is diagnostics rather than noise.
548
+ def guarded_call(ctx, sid, tc)
549
+ execute_call(ctx, sid, tc)
550
+ rescue Tools::Failure => e
551
+ Tools::Result.new(id: tc.id, content: nil, error: e.message)
552
+ rescue StandardError => e
553
+ Tools::Result.new(id: tc.id, content: nil, error: "#{e.class}: #{e.message}")
554
+ end
555
+
556
+ def log_call(sessions, sid, tc)
557
+ sessions.append(sid, "tool/call", { id: tc.id, name: tc.name, args: tc.args })
558
+ end
559
+
560
+ def execute_call(ctx, sid, tc)
561
+ ctx[:tools].execute(
562
+ Tools::Call.new(id: tc.id, name: tc.name, args: tc.args, session_id: sid),
563
+ ctx: ctx
564
+ )
565
+ end
566
+
567
+ def execute_and_record(ctx, sessions, sid, tc)
568
+ result = execute_call(ctx, sid, tc)
569
+ sessions.append(sid, "tool/result",
570
+ { id: result.id, content: result.content, error: result.error })
571
+ end
572
+
573
+ # Close the crash-opened step: execute tool calls the open turn's own last
574
+ # assistant message owes that have no tool/result yet (that event is the
575
+ # truth — a tool/call event may itself have died unwritten; and scoping to
576
+ # the open turn is what keeps a mutation an earlier, closed turn already
577
+ # resolved from running twice), append the missing
578
+ # tool/call events, results, and the step's step/end (without usage: the
579
+ # original step's usage died with the process). Returns the step count so
580
+ # step_loop numbers onward from it. Honest edges: an unclosed step/start
581
+ # with no owed calls stays unclosed and stepping just continues; a turn
582
+ # that crashed after a final no-tool assistant message resumes with one
583
+ # extra model request (the model sees its history and wraps up).
584
+ def complete_dangling(agent)
585
+ ctx = agent.ctx
586
+ sessions = ctx[:sessions]
587
+ sid = agent.session_id
588
+ events = sessions.fetch(sid).events
589
+ turn = events[events.rindex { |e| e.type == "turn/start" }..]
590
+ steps = turn.count { |e| e.type == "step/start" }
591
+
592
+ resolved = turn.filter_map { |e| e.payload[:id] if e.type == "tool/result" }
593
+ last_assistant = turn.reverse_each.find { |e| e.type == "assistant/message" }
594
+ owed = if last_assistant
595
+ last_assistant.payload[:parts]
596
+ .map { |p| LLM.decode_part(p) }
597
+ .grep(LLM::ToolCall)
598
+ .reject { |tc| resolved.include?(tc.id) }
599
+ else
600
+ [] # the open turn never got a model reply; nothing is owed
601
+ end
602
+ return steps if owed.empty?
603
+
604
+ logged = turn.filter_map { |e| e.payload[:id] if e.type == "tool/call" }
605
+ token = redaction_token(ctx)
606
+ owed.each do |tc|
607
+ unless logged.include?(tc.id)
608
+ sessions.append(sid, "tool/call", { id: tc.id, name: tc.name, args: tc.args })
609
+ end
610
+ if token && (redacted?(tc.name, token) || redacted?(tc.args, token))
611
+ sessions.append(sid, "tool/result",
612
+ { id: tc.id, content: nil,
613
+ error: "#{tc.name} was not replayed on resume: the session log " \
614
+ "redacted part of this call, so the recorded call is not " \
615
+ "the call that was made" })
616
+ next
617
+ end
618
+
619
+ execute_and_record(ctx, sessions, sid, tc)
620
+ end
621
+ sessions.append(sid, "step/end", { n: steps })
622
+ steps
112
623
  end
113
624
  end
114
625
 
@@ -116,6 +627,7 @@ module Terret
116
627
  # per step; registration is an effect.
117
628
  class Prompt < Hames::Service
118
629
  service_key :prompt
630
+ config_schema({}) # the prompt assembler takes no config
119
631
 
120
632
  def start(ctx)
121
633
  @ctx = ctx