insika 0.0.1 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (277) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +361 -0
  3. data/LICENSE +21 -0
  4. data/README.md +136 -2
  5. data/bin/insika +366 -0
  6. data/docs/AGENTS.md +618 -0
  7. data/docs/ARCHITECTURE.md +333 -0
  8. data/docs/BENCHMARK.md +114 -0
  9. data/docs/CHANNELS.md +453 -0
  10. data/docs/CONTEXT.md +117 -0
  11. data/docs/DEPLOY.md +354 -0
  12. data/docs/EMBEDDING.md +198 -0
  13. data/docs/EVALS.md +273 -0
  14. data/docs/LOADTEST.md +232 -0
  15. data/docs/OBSERVABILITY.md +374 -0
  16. data/docs/PLUGINS.md +211 -0
  17. data/docs/REFINEMENT.md +477 -0
  18. data/docs/RELEASING.md +70 -0
  19. data/docs/RUNNING-LOCAL.md +153 -0
  20. data/docs/SANDBOX.md +114 -0
  21. data/docs/SECURITY.md +375 -0
  22. data/docs/SKILLS.md +284 -0
  23. data/docs/TOOLS.md +302 -0
  24. data/docs/WHY.md +137 -0
  25. data/docs/WORKFLOWS.md +225 -0
  26. data/docs/build.md +14 -0
  27. data/docs/index.md +68 -0
  28. data/docs/onboarding/start.md +126 -0
  29. data/docs/operate.md +12 -0
  30. data/docs/ship.md +10 -0
  31. data/docs/understand.md +10 -0
  32. data/lib/insika/agent_file_store.rb +125 -0
  33. data/lib/insika/agent_profile.rb +255 -0
  34. data/lib/insika/alert_dispatcher.rb +139 -0
  35. data/lib/insika/allowlist.rb +28 -0
  36. data/lib/insika/baseline_store.rb +74 -0
  37. data/lib/insika/budget_ledger.rb +135 -0
  38. data/lib/insika/capability/resolved_tool.rb +34 -0
  39. data/lib/insika/capability_registry.rb +112 -0
  40. data/lib/insika/channel_delivery.rb +153 -0
  41. data/lib/insika/channel_registry.rb +30 -0
  42. data/lib/insika/channels/relay.rb +178 -0
  43. data/lib/insika/channels/web/widget.js +283 -0
  44. data/lib/insika/channels/web.rb +211 -0
  45. data/lib/insika/channels/webhook.rb +58 -0
  46. data/lib/insika/chat_builder.rb +303 -0
  47. data/lib/insika/checkpoint.rb +13 -0
  48. data/lib/insika/checkpoint_store.rb +153 -0
  49. data/lib/insika/circuit_state.rb +114 -0
  50. data/lib/insika/coercion.rb +58 -0
  51. data/lib/insika/command.rb +32 -0
  52. data/lib/insika/command_bus.rb +39 -0
  53. data/lib/insika/commands/agent_payload.rb +43 -0
  54. data/lib/insika/commands/approve_action.rb +46 -0
  55. data/lib/insika/commands/cancel_task.rb +33 -0
  56. data/lib/insika/commands/create_agent.rb +54 -0
  57. data/lib/insika/commands/create_session.rb +67 -0
  58. data/lib/insika/commands/delete_agent.rb +33 -0
  59. data/lib/insika/commands/delete_agent_file.rb +50 -0
  60. data/lib/insika/commands/delete_data_tool.rb +33 -0
  61. data/lib/insika/commands/delete_llm_provider.rb +36 -0
  62. data/lib/insika/commands/delete_mcp.rb +30 -0
  63. data/lib/insika/commands/delete_skill.rb +43 -0
  64. data/lib/insika/commands/delete_system_file.rb +29 -0
  65. data/lib/insika/commands/gate_refinement.rb +245 -0
  66. data/lib/insika/commands/import_mcp_tools.rb +48 -0
  67. data/lib/insika/commands/import_tools.rb +81 -0
  68. data/lib/insika/commands/issue_tenant_token.rb +41 -0
  69. data/lib/insika/commands/memory_add_note.rb +32 -0
  70. data/lib/insika/commands/memory_forget_fact.rb +32 -0
  71. data/lib/insika/commands/memory_put_fact.rb +35 -0
  72. data/lib/insika/commands/pause_task.rb +29 -0
  73. data/lib/insika/commands/resolve_refinement.rb +126 -0
  74. data/lib/insika/commands/restore_agent_file.rb +36 -0
  75. data/lib/insika/commands/restore_data_tool.rb +34 -0
  76. data/lib/insika/commands/restore_system_file.rb +31 -0
  77. data/lib/insika/commands/resume_task.rb +85 -0
  78. data/lib/insika/commands/revoke_token.rb +39 -0
  79. data/lib/insika/commands/rotate_tenant_token.rb +43 -0
  80. data/lib/insika/commands/run_refinement.rb +133 -0
  81. data/lib/insika/commands/send_message.rb +150 -0
  82. data/lib/insika/commands/set_agent_tools.rb +39 -0
  83. data/lib/insika/commands/set_skill_agents.rb +112 -0
  84. data/lib/insika/commands/trigger_workflow.rb +80 -0
  85. data/lib/insika/commands/update_agent.rb +49 -0
  86. data/lib/insika/commands/update_settings.rb +33 -0
  87. data/lib/insika/commands/upsert_llm_provider.rb +34 -0
  88. data/lib/insika/commands/upsert_mcp.rb +32 -0
  89. data/lib/insika/commands/write_agent_file.rb +57 -0
  90. data/lib/insika/commands/write_data_tool.rb +43 -0
  91. data/lib/insika/commands/write_golden.rb +58 -0
  92. data/lib/insika/commands/write_skill.rb +60 -0
  93. data/lib/insika/commands/write_system_file.rb +31 -0
  94. data/lib/insika/config_store.rb +89 -0
  95. data/lib/insika/context/builder.rb +166 -0
  96. data/lib/insika/context/catalog_provider.rb +23 -0
  97. data/lib/insika/context/fragment.rb +43 -0
  98. data/lib/insika/context/priority.rb +30 -0
  99. data/lib/insika/context/provider.rb +19 -0
  100. data/lib/insika/context/providers/memory.rb +60 -0
  101. data/lib/insika/context/providers/prompt.rb +105 -0
  102. data/lib/insika/context/providers/request.rb +32 -0
  103. data/lib/insika/context/providers/session.rb +123 -0
  104. data/lib/insika/context/providers/skill.rb +24 -0
  105. data/lib/insika/context/providers/skill_trigger.rb +128 -0
  106. data/lib/insika/context/providers/tool_search.rb +20 -0
  107. data/lib/insika/context_trace_store.rb +92 -0
  108. data/lib/insika/delegation_store.rb +153 -0
  109. data/lib/insika/doctor.rb +539 -0
  110. data/lib/insika/dsl/definition.rb +55 -0
  111. data/lib/insika/dsl/runtime.rb +382 -0
  112. data/lib/insika/dsl/server_boot.rb +98 -0
  113. data/lib/insika/dsl/system.rb +93 -0
  114. data/lib/insika/dsl/workflow_adapter.rb +59 -0
  115. data/lib/insika/dsl.rb +364 -0
  116. data/lib/insika/edge_limiter.rb +268 -0
  117. data/lib/insika/egress_guard.rb +75 -0
  118. data/lib/insika/env_schema.rb +249 -0
  119. data/lib/insika/errors.rb +201 -0
  120. data/lib/insika/evals/assertions.rb +247 -0
  121. data/lib/insika/evals/baseline.rb +69 -0
  122. data/lib/insika/evals/golden.rb +172 -0
  123. data/lib/insika/evals/judge.rb +225 -0
  124. data/lib/insika/evals/pairwise.rb +178 -0
  125. data/lib/insika/evals/report.rb +115 -0
  126. data/lib/insika/evals/runner.rb +141 -0
  127. data/lib/insika/evals/transport.rb +178 -0
  128. data/lib/insika/event.rb +18 -0
  129. data/lib/insika/event_stream.rb +132 -0
  130. data/lib/insika/executor.rb +1995 -0
  131. data/lib/insika/frontmatter.rb +42 -0
  132. data/lib/insika/golden_store.rb +145 -0
  133. data/lib/insika/hooks.rb +48 -0
  134. data/lib/insika/http_client.rb +63 -0
  135. data/lib/insika/inbound_log.rb +84 -0
  136. data/lib/insika/llm_configurator.rb +99 -0
  137. data/lib/insika/llm_provider_store.rb +83 -0
  138. data/lib/insika/loop_detector.rb +143 -0
  139. data/lib/insika/mcp_http_client.rb +67 -0
  140. data/lib/insika/mcp_store.rb +115 -0
  141. data/lib/insika/mcp_tool_ingestor.rb +143 -0
  142. data/lib/insika/memory_store.rb +93 -0
  143. data/lib/insika/message_origin.rb +76 -0
  144. data/lib/insika/middleware.rb +36 -0
  145. data/lib/insika/model_policy.rb +52 -0
  146. data/lib/insika/model_resolver.rb +176 -0
  147. data/lib/insika/model_selection.rb +115 -0
  148. data/lib/insika/onboarding.rb +208 -0
  149. data/lib/insika/outbox_store.rb +166 -0
  150. data/lib/insika/overlay_tool_registry.rb +102 -0
  151. data/lib/insika/pack.rb +102 -0
  152. data/lib/insika/pack_importer.rb +123 -0
  153. data/lib/insika/pending_action_store.rb +120 -0
  154. data/lib/insika/plugin/loader.rb +356 -0
  155. data/lib/insika/plugin.rb +35 -0
  156. data/lib/insika/policy/engine.rb +83 -0
  157. data/lib/insika/policy/policy.rb +120 -0
  158. data/lib/insika/policy_registry.rb +23 -0
  159. data/lib/insika/profile_source.rb +143 -0
  160. data/lib/insika/prompt_catalog.rb +61 -0
  161. data/lib/insika/provider_error_classifier.rb +160 -0
  162. data/lib/insika/queue_policy.rb +167 -0
  163. data/lib/insika/recovery.rb +168 -0
  164. data/lib/insika/refinement/candidate.rb +159 -0
  165. data/lib/insika/refinement/evidence_collector.rb +371 -0
  166. data/lib/insika/refinement/gate.rb +234 -0
  167. data/lib/insika/refinement/panel.rb +222 -0
  168. data/lib/insika/refinement/proposer.rb +262 -0
  169. data/lib/insika/refinement_store.rb +295 -0
  170. data/lib/insika/registry.rb +59 -0
  171. data/lib/insika/reliability.rb +185 -0
  172. data/lib/insika/safety/config.rb +109 -0
  173. data/lib/insika/safety/detectors.rb +176 -0
  174. data/lib/insika/safety/factory.rb +102 -0
  175. data/lib/insika/safety/input_guardrail.rb +102 -0
  176. data/lib/insika/safety/moderator.rb +94 -0
  177. data/lib/insika/safety/output_filter.rb +79 -0
  178. data/lib/insika/safety/output_validator.rb +101 -0
  179. data/lib/insika/safety/safe_responses.rb +47 -0
  180. data/lib/insika/sandbox/boundary.rb +93 -0
  181. data/lib/insika/sandbox/docker.rb +74 -0
  182. data/lib/insika/sandbox/local.rb +33 -0
  183. data/lib/insika/sandbox/runner.rb +80 -0
  184. data/lib/insika/sandbox.rb +85 -0
  185. data/lib/insika/schema_guard.rb +147 -0
  186. data/lib/insika/secret_masking.rb +34 -0
  187. data/lib/insika/server/a2a/agent_card.rb +27 -0
  188. data/lib/insika/server/a2a/app.rb +112 -0
  189. data/lib/insika/server/a2a/client.rb +101 -0
  190. data/lib/insika/server/a2a/errors.rb +32 -0
  191. data/lib/insika/server/a2a/http.rb +42 -0
  192. data/lib/insika/server/a2a/message.rb +27 -0
  193. data/lib/insika/server/a2a/protocol.rb +45 -0
  194. data/lib/insika/server/a2a/remotes.rb +25 -0
  195. data/lib/insika/server/a2a/task_projection.rb +40 -0
  196. data/lib/insika/server/app.rb +1022 -0
  197. data/lib/insika/server/boot.rb +119 -0
  198. data/lib/insika/server/rack_app.rb +118 -0
  199. data/lib/insika/server/responses.rb +165 -0
  200. data/lib/insika/server/sse_body.rb +96 -0
  201. data/lib/insika/server/tenant_auth.rb +61 -0
  202. data/lib/insika/session_actor.rb +162 -0
  203. data/lib/insika/session_store.rb +143 -0
  204. data/lib/insika/settings_store.rb +154 -0
  205. data/lib/insika/shutdown.rb +125 -0
  206. data/lib/insika/skill_catalog.rb +220 -0
  207. data/lib/insika/skill_store.rb +127 -0
  208. data/lib/insika/steer_injector.rb +110 -0
  209. data/lib/insika/store.rb +52 -0
  210. data/lib/insika/stores/memory.rb +123 -0
  211. data/lib/insika/stores/sqlite.rb +183 -0
  212. data/lib/insika/studio/app.rb +1693 -0
  213. data/lib/insika/studio/assets/dist/application.css +1 -0
  214. data/lib/insika/studio/assets/dist/application.js +70 -0
  215. data/lib/insika/studio/forms.rb +335 -0
  216. data/lib/insika/studio/nav_icons.rb +31 -0
  217. data/lib/insika/studio/views/_message.erb +44 -0
  218. data/lib/insika/studio/views/agent_detail.erb +285 -0
  219. data/lib/insika/studio/views/agents.erb +63 -0
  220. data/lib/insika/studio/views/approvals.erb +41 -0
  221. data/lib/insika/studio/views/chats.erb +34 -0
  222. data/lib/insika/studio/views/evals.erb +83 -0
  223. data/lib/insika/studio/views/home.erb +72 -0
  224. data/lib/insika/studio/views/layout.erb +94 -0
  225. data/lib/insika/studio/views/login.erb +17 -0
  226. data/lib/insika/studio/views/mcp.erb +91 -0
  227. data/lib/insika/studio/views/not_found.erb +5 -0
  228. data/lib/insika/studio/views/playground.erb +47 -0
  229. data/lib/insika/studio/views/refinement.erb +234 -0
  230. data/lib/insika/studio/views/session.erb +137 -0
  231. data/lib/insika/studio/views/settings.erb +168 -0
  232. data/lib/insika/studio/views/skills.erb +141 -0
  233. data/lib/insika/studio/views/system_files.erb +65 -0
  234. data/lib/insika/studio/views/task.erb +105 -0
  235. data/lib/insika/studio/views/tasks.erb +33 -0
  236. data/lib/insika/studio/views/tool_edit.erb +107 -0
  237. data/lib/insika/studio/views/tools.erb +89 -0
  238. data/lib/insika/subagent_graph.rb +96 -0
  239. data/lib/insika/system_file_store.rb +96 -0
  240. data/lib/insika/task_actor.rb +128 -0
  241. data/lib/insika/task_store.rb +250 -0
  242. data/lib/insika/telemetry/pricing.rb +104 -0
  243. data/lib/insika/telemetry/recorder.rb +228 -0
  244. data/lib/insika/telemetry.rb +127 -0
  245. data/lib/insika/testing/store_contract.rb +270 -0
  246. data/lib/insika/tick.rb +122 -0
  247. data/lib/insika/token_estimator.rb +16 -0
  248. data/lib/insika/token_store.rb +168 -0
  249. data/lib/insika/tool_assembly.rb +140 -0
  250. data/lib/insika/tool_catalog.rb +89 -0
  251. data/lib/insika/tool_definition.rb +518 -0
  252. data/lib/insika/tool_envelope.rb +140 -0
  253. data/lib/insika/tool_manifest.rb +218 -0
  254. data/lib/insika/tool_output_compressor.rb +100 -0
  255. data/lib/insika/tool_registry.rb +21 -0
  256. data/lib/insika/tool_store.rb +135 -0
  257. data/lib/insika/tool_trace_store.rb +92 -0
  258. data/lib/insika/tools/a2a_remote.rb +48 -0
  259. data/lib/insika/tools/agent_enum.rb +68 -0
  260. data/lib/insika/tools/concurrency.rb +54 -0
  261. data/lib/insika/tools/data_defined_tool.rb +219 -0
  262. data/lib/insika/tools/load_skill.rb +99 -0
  263. data/lib/insika/tools/remember.rb +53 -0
  264. data/lib/insika/tools/stuck_signal.rb +44 -0
  265. data/lib/insika/tools/subagent.rb +75 -0
  266. data/lib/insika/tools/subagents.rb +77 -0
  267. data/lib/insika/tools/tool_search.rb +94 -0
  268. data/lib/insika/turn_output.rb +139 -0
  269. data/lib/insika/turn_state.rb +162 -0
  270. data/lib/insika/turn_timing.rb +56 -0
  271. data/lib/insika/usage_ledger.rb +47 -0
  272. data/lib/insika/version.rb +3 -1
  273. data/lib/insika/wiring/graph.rb +249 -0
  274. data/lib/insika/workflow.rb +185 -0
  275. data/lib/insika/workflow_registry.rb +33 -0
  276. data/lib/insika.rb +220 -4
  277. metadata +412 -8
@@ -0,0 +1,1995 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "time"
4
+ require "securerandom"
5
+ require "async/queue"
6
+
7
+ module Insika
8
+ # Coordinates execution. It does not build context, decide policy, or talk to
9
+ # the provider. This file does NOT require ruby_llm at load-time —
10
+ # the require is lazy inside the chat methods.
11
+ #
12
+ # Surrounds the stages: spawn, state lifecycle
13
+ # (always via TaskStore), mailbox drain at the boundaries, in-process
14
+ # registration of live fibers (running?), and the emitter with meta + seq.
15
+ class Executor
16
+ def initialize(context_builder:, policy_engine:, middleware:, hooks:,
17
+ tool_registry:, skill_catalog:, profiles:,
18
+ session_store:, task_store:, checkpoint_store:,
19
+ event_stream:, workflow_registry: nil, pending_action_store: nil,
20
+ capability_registry: nil, tool_catalog: nil, memory_store: nil,
21
+ tool_trace_store: nil, settings_store: nil, content_filter_factory: nil,
22
+ delegation_store: nil, channel_delivery: nil, llm: nil,
23
+ context_trace_store: nil, reliability: nil)
24
+ @context_builder = context_builder
25
+ @policy_engine = policy_engine
26
+ @middleware = middleware
27
+ @hooks = hooks
28
+ @tool_registry = tool_registry
29
+ @skill_catalog = skill_catalog
30
+ # Legacy Hash -> StaticProfileSource; a ProfileSource passes through.
31
+ @profiles = ProfileSource.coerce(profiles)
32
+ @session_store = session_store
33
+ @task_store = task_store
34
+ @checkpoint_store = checkpoint_store
35
+ @event_stream = event_stream
36
+ @workflow_registry = workflow_registry # stage 6 of trigger_workflow
37
+ @pending_action_store = pending_action_store # approval gate
38
+ @capability_registry = capability_registry # capability resolution (nil = off)
39
+ @tool_trace_store = tool_trace_store # tool-call trace for Studio debugging (nil = off)
40
+ # per-turn context breakdown (tokens by category + budget) for the
41
+ # Studio session card. nil = off (no record, zero overhead — parity).
42
+ @context_trace_store = context_trace_store
43
+ # Guardrails output filter: ->(state) { OutputFilter | nil }.
44
+ # Injected by the Safety::Factory; nil = off (parity — the stream is untouched).
45
+ # The INPUT guardrail is a Middleware (in the stack, not here); this is the seam
46
+ # for the stream-side redaction the Executor owns.
47
+ @content_filter_factory = content_filter_factory
48
+ # durable record of ASYNC delegations. nil = async
49
+ # delegation OFF (only the synchronous spawn_subagent works — parity). When
50
+ # present, run_subagent(async: true) dispatches + returns immediately and the
51
+ # child's result is delivered to the parent session as a NEW turn on completion.
52
+ @delegation_store = delegation_store
53
+ # outbound delivery for Shape B channels. nil = no channel
54
+ # delivers out of band (parity — every surface today answers on the request's
55
+ # own connection). When present, a turn that CAME IN through a channel writes
56
+ # its answer to the outbox at the terminal and the dispatcher POSTs it.
57
+ @channel_delivery = channel_delivery
58
+ # the chat FACTORY this executor asks — a RubyLLM::Context (an
59
+ # isolated config dup) when the graph owns its credentials, nil = the
60
+ # process-wide RubyLLM constant (the historic single-graph deployment).
61
+ # Duck-typed: Context#chat and RubyLLM.chat take the same keywords.
62
+ @llm = llm
63
+ # stability for the turn's single agent interaction (WS3): retries /
64
+ # fallback / circuit breaker, all DATA on AgentProfile#reliability.
65
+ # nil = the plain single ask (parity).
66
+ @reliability = reliability
67
+ # LLM config v2: resolves the model at turn start (Chat > Agent >
68
+ # platform default) + model_policy + fallback chain. settings_store nil =
69
+ # no platform layer (pre-v2 behavior: the agent's own model is used as-is).
70
+ @model_resolver = ModelResolver.new(settings_store: settings_store)
71
+ # the platform layer of the queue policy (nil = per-agent and
72
+ # defaults only, which is `followup` with no window — today's behavior).
73
+ @settings_store = settings_store
74
+ # RubyLLM glue (stages 5-7): chat assembly delegated to ChatBuilder. Its
75
+ # optional deps tool_catalog (Tool Search) and memory_store (cross-session
76
+ # memory) matter only to it — nil = parity (deferred
77
+ # not partitioned; no system remember).
78
+ @chat_builder = ChatBuilder.new(
79
+ tool_registry: tool_registry, skill_catalog: skill_catalog,
80
+ checkpoint_store: checkpoint_store, event_stream: event_stream, hooks: hooks,
81
+ tool_catalog: tool_catalog, memory_store: memory_store,
82
+ # load_skill is not enveloped, so it records its own trace entry.
83
+ tool_trace_store: tool_trace_store,
84
+ # the ChatBuilder wires the spawn_subagent system tool (gated by
85
+ # profile.subagents) and hands it this Executor as the runner. `self` is not
86
+ # yet fully built here, but the ChatBuilder only STORES it (used per-turn).
87
+ subagent_runner: self
88
+ )
89
+ # Stage-3-tail tool assembly (capability resolution, instantiation,
90
+ # injection, dedup join, ToolEnvelope wrap) — extracted collaborator.
91
+ @tool_assembly = ToolAssembly.new(
92
+ tool_registry: tool_registry, capability_registry: capability_registry,
93
+ event_stream: event_stream, checkpoint_store: checkpoint_store,
94
+ tool_trace_store: tool_trace_store
95
+ )
96
+ @running = {} # task_id => TaskActor (live fibers in this process)
97
+ @seqs = Hash.new(0) # monotonic counter per task
98
+ @supervised = false # serving mode? — see #turn_parent
99
+ @supervisor = nil # lazy long-lived supervisor (created when serving)
100
+ @session_actors = {} # session_id => SessionActor (FIFO queue)
101
+ @draining = false # shutdown drain — see #begin_drain!
102
+ end
103
+
104
+ # Turns on SERVING mode: the composition root's serving arm (serve.rb /
105
+ # config.ru) sets true AFTER recovery. Under HTTP `spawn` runs on the request's
106
+ # EPHEMERAL fiber; without this the turn would be its child and the runtime
107
+ # would CANCEL it on disconnect (violates the contract: "execution belongs to
108
+ # the runtime, not the connection"). When on, the turn is born a child of a
109
+ # long-lived supervisor (sibling of the accept loop) and outlives the
110
+ # connection. false (default) = parents on the current fiber: at recovery/boot
111
+ # and in tests the owner WANTS to wait for the turn to finish (structured
112
+ # concurrency).
113
+ attr_accessor :supervised
114
+
115
+ # the periodic tick (outbox drain + stale recovery sweep), wired
116
+ # by the graph AFTER the bus exists (the tick's recovery half dispatches
117
+ # through it). nil = no tick (parity — recovery stays boot-only). When
118
+ # present and serving, it starts as a child of the turn supervisor (see
119
+ # #turn_parent).
120
+ attr_accessor :tick
121
+
122
+ # The WS6 alert dispatcher: answers budget_warning / breaker_open /
123
+ # delivery_failed with a durable webhook delivery. Started as a child of
124
+ # the turn supervisor in serving mode (like the tick); nil = no alerts.
125
+ attr_accessor :alert_dispatcher
126
+
127
+ # closes the TURN intake for shutdown. Armed by Insika::Shutdown
128
+ # when the process is asked to stop: from here on a new top-level turn is left
129
+ # `:queued` (durable — the next boot's recovery replays it) instead of
130
+ # spawning, while the in-flight turns run to their natural end. One-way by
131
+ # design: a draining process never takes work again.
132
+ def begin_drain!
133
+ @draining = true
134
+ end
135
+
136
+ def draining? = @draining
137
+
138
+ # The turns still running in THIS process — what a drain waits on.
139
+ def in_flight = @running.keys
140
+
141
+ # In-process registry of live fibers (ResumeTask's criterion).
142
+ def running?(task_id) = @running.key?(task_id)
143
+
144
+ # CancelTask's access point: posts :cancel if there is a live fiber.
145
+ # Idempotent no-op if there is none (terminal/orphan). Returns whether there
146
+ # was a fiber.
147
+ def cancel(task_id)
148
+ actor = @running[task_id]
149
+ actor&.post(:cancel)
150
+ !actor.nil?
151
+ end
152
+
153
+ # PauseTask's access point: posts :pause if there is a live fiber; the turn
154
+ # suspends at the next boundary (drain_and_maybe_suspend). Idempotent no-op
155
+ # if there is none. Returns whether there was a fiber.
156
+ def pause(task_id)
157
+ actor = @running[task_id]
158
+ actor&.post(:pause)
159
+ !actor.nil?
160
+ end
161
+
162
+ # ResumeTask's access point for a paused IN-PROCESS task:
163
+ # posts :resume on the live fiber (which is blocked in await). Returns whether
164
+ # there was a live fiber — the handler decides between in-process resume and
165
+ # crash re-dispatch based on that.
166
+ def resume_live(task_id)
167
+ actor = @running[task_id]
168
+ actor&.post(:resume)
169
+ !actor.nil?
170
+ end
171
+
172
+ # ApproveAction's access point: WAKES the turn suspended in
173
+ # :waiting by posting :approval on the live fiber. The decision was already
174
+ # written to the store by the handler BEFORE this post (request_approval
175
+ # re-reads it from the store). No-op if there is no live fiber (process
176
+ # crashed) — recovery re-executes and uses the durable decision. Returns
177
+ # whether there was a fiber.
178
+ def approve(task_id)
179
+ actor = @running[task_id]
180
+ actor&.post(:approval)
181
+ !actor.nil?
182
+ end
183
+
184
+ # Approval gate, called by the ToolEnvelope at stage 6 when the
185
+ # tool requires approval. Creates/queries the PendingAction (deterministic id
186
+ # by task+turn+tool — per-tool correlation as with the side-effect),
187
+ # suspends the turn in :waiting and BLOCKS (await(:approval)) until the
188
+ # operator resolves it via ApproveAction. The AUTHORITATIVE decision comes
189
+ # from the durable store (crash-safe): on a post-crash re-execution, an
190
+ # already-resolved PendingAction is reused without re-suspending; a :pending
191
+ # one re-suspends.
192
+ # -> "approved" | "rejected".
193
+ def request_approval(task:, turn:, tool:, args:, actor:)
194
+ # Fail-closed: requiring approval with nowhere to persist/query the decision
195
+ # is a misconfiguration — fail LOUD, never hang nor auto-approve.
196
+ if @pending_action_store.nil?
197
+ raise Insika::Error, "tool '#{tool}' requires approval but PendingActionStore is not configured"
198
+ end
199
+
200
+ id = pending_id(task.id, turn, tool)
201
+ existing = @pending_action_store.find(id)
202
+ return existing.status.to_s if existing && existing.status != :pending # re-execution: already resolved
203
+
204
+ unless existing
205
+ @pending_action_store.create(id: id, task_id: task.id, turn: turn, tool: tool, args: args || {})
206
+ emit(:approval_requested, { pending_id: id, tool: tool.to_s, args: args }, task: task)
207
+ end
208
+
209
+ @task_store.transition(task.id, to: :waiting) if @task_store.find(task.id).status == :running
210
+
211
+ # Awaits the resolution of THIS pending. A spurious :approval (duplicate or
212
+ # from another pending of the same actor) that wakes up before resolution is
213
+ # ignored (re-await) — fail-closed: only the real resolution of this id
214
+ # unblocks.
215
+ status = nil
216
+ loop do
217
+ actor.await(reason: :approval) # blocks (or raises on :cancel/:timeout)
218
+ status = @pending_action_store.find(id)&.status
219
+ break unless status == :pending
220
+ end
221
+
222
+ @task_store.transition(task.id, to: :running)
223
+ (status || :rejected).to_s # fail-closed: missing record -> reject (defensive)
224
+ end
225
+
226
+ # Stage 1 (async part): creates the actor, registers it and fires the fiber.
227
+ # Called by the turn handlers (SendMessage/ResumeTask/TriggerWorkflow).
228
+ def spawn(task, profile:, resume_from: nil)
229
+ raise Insika::ValidationError, "task already running: #{task.id}" if running?(task.id)
230
+
231
+ actor = TaskActor.new(task_id: task.id, parent: turn_parent)
232
+ @running[task.id] = actor
233
+ actor.run { execute(task, profile: profile, resume_from: resume_from, actor: actor) }
234
+ task.id
235
+ end
236
+
237
+ # Turn entry point that RESPECTS the session: a turn with a
238
+ # session_id is SERIALIZED in that session's SessionActor queue (one at a
239
+ # time); without a session_id (one-shot/history) it goes straight to spawn
240
+ # (standalone).
241
+ def spawn_in_session(task, profile:, resume_from: nil)
242
+ # the intake is closed. The task is already durable (queued);
243
+ # answering with its id and spawning NOTHING is what "stops accepting new
244
+ # turns" means — the next boot's recovery replays it. Subagent turns are NOT
245
+ # gated (they spawn directly): a child of an in-flight parent is part of the
246
+ # work the drain is waiting FOR, and refusing it would wedge the parent.
247
+ return defer_turn(task) if @draining
248
+
249
+ # SessionActor only in SERVING mode (@supervised): serializes concurrent
250
+ # REQUESTS. At boot/recovery (non-supervised) the replay is sequential and
251
+ # the long-lived loop would hang the Boot's Sync — use direct spawn (the
252
+ # owner awaits the turn). One-shot/history (no session_id)
253
+ # never serialize.
254
+ unless @supervised && task.session_id
255
+ return spawn(task, profile: profile, resume_from: resume_from)
256
+ end
257
+
258
+ session_actor(task.session_id).enqueue(task, profile: profile, resume_from: resume_from,
259
+ policy: queue_policy(profile, task.session_id))
260
+ end
261
+
262
+ # the `collect` door, asked BEFORE a task is created.
263
+ # -> the task id the fragment joined, or nil (create a task and spawn as usual).
264
+ #
265
+ # Asking first is what keeps the store clean: creating a task and then
266
+ # discarding it would leave an orphan :queued record for every fragment, and
267
+ # `queued` is what Recovery replays at boot.
268
+ def collect_into_pending(session_id, text, profile:)
269
+ return nil unless @supervised && session_id
270
+
271
+ policy = queue_policy(profile, session_id)
272
+ return nil unless policy.collect? && policy.debounce?
273
+
274
+ actor = @session_actors[session_id]
275
+ return nil unless actor&.alive?
276
+
277
+ actor.collect(text)
278
+ end
279
+
280
+ # the `steer` door: a message for a session whose turn is ALREADY
281
+ # running is appended to that run instead of becoming a turn of its own.
282
+ # -> the RUNNING task's id (the turn that will answer it), or nil (create a task and
283
+ # spawn as usual, which is `followup`).
284
+ #
285
+ # Asked BEFORE a task is created, like the `collect` door, and for the same reason.
286
+ # The post lands in the turn's mailbox; `SteerInjector` reads it at the next
287
+ # tool-batch boundary. Nothing here touches the run in flight — a message that no
288
+ # boundary ever arrives for is released as a follow-up turn (#release_steered).
289
+ def steer_into_running(session_id, text, profile:)
290
+ return nil unless @supervised && session_id
291
+
292
+ policy = queue_policy(profile, session_id)
293
+ return nil unless policy.steer?
294
+
295
+ session_actor = @session_actors[session_id]
296
+ return nil unless session_actor&.alive?
297
+
298
+ # No turn running (or one still at the door): there is nothing to steer INTO.
299
+ # A turn at the door belongs to `collect`, which is a different mode.
300
+ task = session_actor.current_task
301
+ return nil if task.nil?
302
+ # A workflow turn orchestrates RubyLLM itself and has no Insika chat to append to
303
+ # (docs/WORKFLOWS.md says so). Refused at the door so the message becomes an
304
+ # ordinary next turn instead of a phantom one.
305
+ return nil if workflow_turn?(task)
306
+
307
+ actor = @running[task.id]
308
+ return nil if actor.nil?
309
+ # The bound is on what ONE run may absorb, so it counts posts and not the buffer:
310
+ # a message already injected still spent its slot. Overflow degrades to
311
+ # `followup` rather than growing an unbounded tail.
312
+ return nil if actor.user_messages_posted >= policy.steer_max_messages
313
+
314
+ actor.post(:user_message, text)
315
+ task.id
316
+ end
317
+
318
+ # the `interrupt` door: the turn in flight is answering a question
319
+ # the customer has already replaced, so it is abandoned and the new message becomes an
320
+ # ordinary turn. -> the abandoned task's id, or nil (nothing was running).
321
+ #
322
+ # Unlike `collect`/`steer` this one JOINS nothing: `replaced_by` already has its own
323
+ # task and its own reply, so there is no verdict to report and every surface can use
324
+ # it, `/v1/responses` included.
325
+ #
326
+ # What "abandon" means here is Insika's existing cancellation semantics, unchanged:
327
+ # `:cancel` is observed only at a stage boundary, so a tool call in flight runs to
328
+ # completion and is recorded. Cancelling the not-yet-started calls of a batch would
329
+ # leave it half applied, and fabricating failure results would teach the model that
330
+ # tools failed when they did not (records the same boundary for `turn_timeout`).
331
+ def interrupt_running(session_id, profile:, replaced_by: nil)
332
+ return nil unless @supervised && session_id
333
+
334
+ return nil unless queue_policy(profile, session_id).interrupt?
335
+
336
+ session_actor = @session_actors[session_id]
337
+ return nil unless session_actor&.alive?
338
+
339
+ task = session_actor.current_task
340
+ return nil if task.nil?
341
+
342
+ actor = @running[task.id]
343
+ return nil if actor.nil?
344
+
345
+ actor.post(:cancel)
346
+ emit(:turn_interrupted, { task_id: task.id, replaced_by: replaced_by }, task: task)
347
+ task.id
348
+ end
349
+
350
+ # The SessionActor writes the merged fragment through this (it has no store of
351
+ # its own, by design — it owns scheduling, not persistence).
352
+ attr_reader :task_store
353
+
354
+ # Emitted by the SessionActor when a window closes having merged
355
+ # more than one fragment. `arrivals` are the ISO8601 times each fragment landed
356
+ # — the ONLY record that they were separate messages, since a merged fragment
357
+ # creates no task of its own. Ids and times, never content.
358
+ def emit_coalesced(task, merged:, arrivals: [])
359
+ emit(:turn_coalesced, { task_id: task.id, merged: merged, arrivals: arrivals }, task: task)
360
+ end
361
+
362
+ # a steered message the run could NOT absorb: no tool batch ever
363
+ # closed (a text-only turn), the batch ended in `halt_when`, the turn failed, or it
364
+ # was cancelled. The message is a person's and must not evaporate, so it is released
365
+ # as the next turn on this session — which is `followup`, arrived at late.
366
+ #
367
+ # Runs in `execute`'s ensure, on the dying turn's fiber: `spawn_in_session` only
368
+ # enqueues, and the session loop is still awaiting THIS turn, so the follow-up runs
369
+ # after it, in order.
370
+ def release_steered(task, profile, actor)
371
+ # Cheap gate first: a turn nobody steered pays one integer read and no store round
372
+ # trip, which is every turn on every agent that never turned the mode on.
373
+ return unless actor.user_messages_posted.positive?
374
+
375
+ # Only a turn that actually FINISHED releases. A process going down (Async::Stop
376
+ # through this ensure) leaves the task `:running` for Recovery to replay, and
377
+ # spawning a follow-up during shutdown would parent a turn on a supervisor that is
378
+ # already stopping. The honest limitation: a steered message lives in memory until a
379
+ # boundary writes it to the transcript, so a hard stop in that window loses it.
380
+ current = @task_store.find(task.id)
381
+ return unless current && TERMINAL_STATUSES.include?(current.status)
382
+
383
+ texts = actor.take_user_messages!
384
+ return if texts.empty?
385
+
386
+ command = Insika::Command.build(
387
+ :send_message,
388
+ # No `origin`: a person typed this, which is exactly what an absent origin means.
389
+ { "agent" => profile.id, "message" => texts.join("\n"), "session_id" => task.session_id }
390
+ ).to_h
391
+ follow_up = @task_store.create(command: command, session_id: task.session_id)
392
+ emit(:turn_steer_released, { task_id: task.id, released_as: follow_up.id, count: texts.size },
393
+ task: task)
394
+ spawn_in_session(follow_up, profile: profile)
395
+ rescue Insika::Error
396
+ # Best-effort: the turn is already terminal and durable, and a store that refuses
397
+ # here must not turn a committed turn into a failed one. The message is then lost,
398
+ # and the ABSENCE of :turn_steer_released is what says so — there is no half state.
399
+ nil
400
+ end
401
+
402
+ # Runs ONE turn serially (called by the SessionActor loop):
403
+ # spawns (the turn is born a child of the supervisor, non-blocking) and AWAITS
404
+ # its completion before returning — that is what serializes the session. A
405
+ # turn error is already mapped to a terminal state inside its own fiber (single
406
+ # capture); here we only ensure the session loop does not die.
407
+ def run_serial(task, profile:, resume_from: nil)
408
+ # a drain that started with turns already queued behind this
409
+ # session's current one must not keep feeding the loop — without this gate
410
+ # the drain would only converge when the whole backlog ran out.
411
+ return defer_turn(task) if @draining
412
+
413
+ spawn(task, profile: profile, resume_from: resume_from)
414
+ @running[task.id]&.wait
415
+ rescue Async::Stop
416
+ raise # shutdown: propagate (ends the session loop)
417
+ rescue StandardError => e
418
+ # A SYNCHRONOUS spawn error (before the fiber): execute's single capture
419
+ # does not act (the turn never ran). Mark :failed here so as NOT to orphan
420
+ # the task as :queued with no terminal state nor event (the client would
421
+ # hang).
422
+ fail_spawn(task, e)
423
+ end
424
+
425
+ # Shuts down all SessionActors (server shutdown / tests — the loop blocks
426
+ # forever on dequeue when idle).
427
+ def stop_session_actors
428
+ @session_actors.each_value(&:stop)
429
+ @session_actors.clear
430
+ nil
431
+ end
432
+
433
+ # runs a CHILD agent turn (called by Tools::Subagent during
434
+ # stage 6). Isolated context (fresh child session), capability NON-inheritance
435
+ # (child profile resolved fresh), environment inheritance (model/thinking seeded
436
+ # from the parent). NEVER raises: a bad agent/depth/child failure is a message
437
+ # to the model, not a turn-killer.
438
+ #
439
+ # async:false (default) — SYNCHRONOUS: runs the child inside the parent's
440
+ # fiber and returns { text:, session_id: } (the child result is the tool result).
441
+ # async:true — DURABLE dispatch: spawns the child NON-blocking, persists
442
+ # a Delegation, returns { dispatched:, agent:, session_id: } immediately; the
443
+ # parent turn ends and the child's result is later delivered as a NEW turn on
444
+ # the parent session (needs a delegation_store — else falls back to sync).
445
+ def run_subagent(agent:, message:, parent_state:, async: false)
446
+ plan = plan_subagent({ "agent" => agent, "message" => message }, parent_state)
447
+ return { error: plan[:error] } if plan[:error]
448
+
449
+ if async && @delegation_store
450
+ dispatch_async_child(plan[:profile], plan[:message], plan[:depth], parent_state)
451
+ else
452
+ spawn_and_await_child(plan[:profile], plan[:message], plan[:depth], parent_state)
453
+ end
454
+ end
455
+
456
+ # (fan-out): runs SEVERAL child turns IN PARALLEL and returns all
457
+ # results together, in the requested order. This is the real latency win — the
458
+ # children overlap their provider waits on the reactor, so wall-clock ≈ the
459
+ # slowest child, not the sum. Always sync-join (a combined result in the parent's
460
+ # turn); the async deliver-as-new-turn mode is single-child only, by design.
461
+ # NEVER raises: per-task errors keep their slot; a bad envelope returns { error: }.
462
+ # -> { results: [{agent:, text:, session_id:} | {agent:, error:}, ...] } | { error: }
463
+ def run_subagents(tasks:, parent_state:)
464
+ list = Array(tasks)
465
+ return { error: "tasks must be a non-empty list of {agent, message}" } if list.empty?
466
+
467
+ cap = SubagentGraph.fan_out_cap
468
+ return { error: "too many subagents in one call: #{list.size} (max #{cap})" } if list.size > cap
469
+
470
+ plans = list.map { |t| plan_subagent(t, parent_state) }
471
+ { results: spawn_all_and_project(plans, parent_state) }
472
+ end
473
+
474
+ # (boot): reconciles ASYNC delegations after a crash so a
475
+ # completed child's result is never lost. For each undelivered Delegation:
476
+ # · child TERMINAL, not captured -> capture + deliver.
477
+ # · child TERMINAL, captured (completed) -> deliver (crash before delivery).
478
+ # · child NOT terminal -> left as-is; the normal task Recovery resumes the
479
+ # child and its terminal hook (finalize_delegation) delivers when it finishes.
480
+ # Called once at boot AFTER the task Recovery. No-op without a delegation_store.
481
+ # -> { delivered: [ids] }
482
+ def recover_delegations
483
+ return { delivered: [] } unless @delegation_store
484
+
485
+ delivered = []
486
+ @delegation_store.undelivered.each do |deleg|
487
+ child = @task_store.find(deleg.child_task_id)
488
+ next unless child && TERMINAL_STATUSES.include?(child.status)
489
+
490
+ finalize_delegation(child) # capture (if needed) + claim + deliver
491
+ delivered << deleg.id
492
+ end
493
+ { delivered: delivered }
494
+ end
495
+
496
+ # (boot): re-drives the outbound replies a previous process
497
+ # recorded and never claimed. Records left `delivering` are NOT swept — that
498
+ # process may have POSTed before it died, and re-sending is the duplicate the
499
+ # claim exists to prevent. No-op without a channel_delivery.
500
+ # -> { dispatched: [ids] }
501
+ def recover_channel_deliveries
502
+ return { dispatched: [] } unless @channel_delivery
503
+
504
+ @channel_delivery.sweep
505
+ end
506
+
507
+ # Stages 2..9. Runs INSIDE the task's fiber.
508
+ def execute(task, profile:, actor:, resume_from: nil)
509
+ # Resume of a crash orphan: the interrupted attempt's Execution was left OPEN
510
+ # (the fiber died). The TaskStore forbids opening a second one while one is
511
+ # open -> close the orphan as :interrupted before opening the N+1 (a new
512
+ # entry, never overwrites).
513
+ close_orphan_execution(task) if resume_from
514
+ @task_store.begin_execution(task.id) # attempt N+1
515
+ # queued (normal spawn) and paused/waiting (resume) -> running. An orphan is
516
+ # already :running (running->running is invalid) -> no transition.
517
+ status = @task_store.find(task.id).status
518
+ @task_store.transition(task.id, to: :running) if %i[queued paused waiting].include?(status)
519
+ emit(:task_started, started_data(task, profile), task: task)
520
+
521
+ actor.drain!
522
+ run_pipeline(task, profile, actor, resume_from)
523
+ # SINGLE capture at the top of the fiber: a single place maps
524
+ # error -> terminal state -> events. Stages do no rescue of their own
525
+ # (except tool, RubyLLM semantics). The fiber NEVER re-raises.
526
+ rescue CancelledError
527
+ # cancel is not an error: transition WITHOUT error: (does not close the
528
+ # Execution), then finish_execution closes it with outcome :cancelled.
529
+ @task_store.transition(task.id, to: :cancelled)
530
+ @task_store.finish_execution(task.id, outcome: :cancelled)
531
+ emit(:task_cancelled, { task_id: task.id }, task: task)
532
+ rescue PolicyDenied => e
533
+ emit(:policy_denied, { policy: e.policy, reason: e.reason }, task: task)
534
+ fail_task(task, e, stage: :policy)
535
+ rescue BudgetExceeded => e
536
+ # WS2 hard budget: a typed, retryable failure — the envelope reads
537
+ # budget_exceeded + retry_after (window roll), never a silent drop.
538
+ fail_task(task, e, stage: :budget)
539
+ rescue CircuitOpenError => e
540
+ # WS3 breaker: the turn died BEFORE the provider call — the envelope
541
+ # reads circuit_open + retry_after (cooldown remaining). Worth its own
542
+ # stage: an open breaker is a reliability decision, not an error bug.
543
+ fail_task(task, e, stage: :reliability)
544
+ rescue Insika::WorkflowSchemaError => e
545
+ # a workflow OUTPUT that violates its output_schema. Distinct
546
+ # stage so a contract breach is not conflated with an :unknown failure. (INPUT
547
+ # is validated synchronously in TriggerWorkflow -> 422, never reaches here.)
548
+ fail_task(task, e, stage: :workflow_schema)
549
+ rescue ContextError => e
550
+ fail_task(task, e, stage: :context)
551
+ rescue CapabilityError => e
552
+ fail_task(task, e, stage: :capability)
553
+ rescue ProviderError => e
554
+ fail_task(task, e, stage: :ruby_llm)
555
+ rescue StoreError => e
556
+ fail_task(task, e, stage: :persistence)
557
+ rescue TimeoutError => e
558
+ fail_task(task, e, stage: e.stage)
559
+ rescue StandardError => e
560
+ # A provider/transport failure is NOT an :unknown bug: wrap it with its
561
+ # action classification (B9) so the envelope can quote retryable and the
562
+ # provider's own retry_after (A8). The classifier is class-name based —
563
+ # the :ruby_llm stage stays reachable even under the smoke-shim's fake.
564
+ if ProviderErrorClassifier.provider_error?(e)
565
+ fail_task(task, ProviderErrorClassifier.wrap(e), stage: :ruby_llm)
566
+ else
567
+ fail_task(task, e, stage: :unknown)
568
+ end
569
+ ensure
570
+ @running.delete(task.id) # ALWAYS deregister (a false-positive running? would break the resume)
571
+ # Deregistered FIRST on purpose: from here on the `steer` door finds no actor for
572
+ # this session and answers nil, so a message arriving during the release becomes
573
+ # its own turn instead of a post into a mailbox nobody reads again.
574
+ release_steered(task, profile, actor)
575
+ end
576
+
577
+ private
578
+
579
+ # resolves session vars > profile.limits > settings["queue"] >
580
+ # defaults. One session read plus one settings read per message, the same
581
+ # order of cost the EdgeLimiter already pays per turn. A missing session (or
582
+ # no session store) simply drops the vars layer.
583
+ def queue_policy(profile, session_id)
584
+ vars = session_id && @session_store&.find(session_id)&.vars
585
+ QueuePolicy.resolve(profile, settings_store: @settings_store, vars: vars)
586
+ end
587
+
588
+ # Lazy SessionActor per session, parented in the turn scope:
589
+ # the session loop outlives the connection, like the supervisor. Revalidates
590
+ # liveness: if the cached loop died (supervisor recreated/stopped), create a
591
+ # new one — otherwise the queued turns would be black-holed in a dead loop.
592
+ def session_actor(session_id)
593
+ existing = @session_actors[session_id]
594
+ return existing if existing&.alive?
595
+
596
+ @session_actors[session_id] =
597
+ SessionActor.new(session_id: session_id, executor: self, parent: turn_parent)
598
+ end
599
+
600
+ # a turn that arrived while the process is draining: left
601
+ # `:queued` on purpose (recovery replays :queued at the next boot). The event
602
+ # is the deferral's only trace — without it, "why was my message answered
603
+ # only after the deploy" is unanswerable.
604
+ def defer_turn(task)
605
+ emit(:turn_deferred, { task_id: task.id }, task: task)
606
+ task.id
607
+ end
608
+
609
+ # Marks :failed a task whose turn FAILED at spawn (before the fiber).
610
+ # Idempotent against a terminal state; StoreError re-raises (the session loop
611
+ # handles it).
612
+ def fail_spawn(task, error)
613
+ current = @task_store.find(task.id)
614
+ return if current.nil? || TERMINAL_STATUSES.include?(current.status)
615
+
616
+ @task_store.transition(task.id, to: :failed,
617
+ error: { class: error.class.name, message: error.message, stage: :spawn })
618
+ emit(:task_failed, { task_id: task.id, error: error.class.name, message: error.message }, task: task)
619
+ rescue Insika::Error
620
+ nil
621
+ end
622
+
623
+ # Parent of the turn's fiber. Non-serving: the current fiber (the owner waits
624
+ # for the turn). Serving: a long-lived supervisor created lazily at the reactor
625
+ # BOUNDARY — the task whose parent is the reactor itself (sibling of the accept
626
+ # loop), outside the subtree of any request. This way the turn outlives the
627
+ # request stop (disconnect). One supervisor per Executor, reused while alive.
628
+ def turn_parent
629
+ return Async::Task.current unless @supervised
630
+ return @supervisor if @supervisor&.running?
631
+
632
+ reactor = Async::Task.current.reactor
633
+ node = Async::Task.current
634
+ node = node.parent while node.parent && node.parent != reactor
635
+ # Idle with no spin nor deprecated API (Async::Task#sleep is deprecated in
636
+ # 2.42): blocks on a dequeue that never arrives. Ends only when the scope is
637
+ # stopped (server shutdown) — then any child turns still running go with it.
638
+ # In a deployment that is the LAST resort, not the plan: Insika::Shutdown
639
+ # drains first, so only what outlives the drain deadline dies
640
+ # here, `:running`, for the next boot's recovery to replay.
641
+ @supervisor = node.async { |t| t.annotate("harness-turn-supervisor"); Async::Queue.new.dequeue }
642
+ # the periodic tick is a child of the supervisor — it binds to
643
+ # the serving reactor in every arm with no arm edits, and dies with the
644
+ # supervisor at shutdown (after Shutdown's drain, like any turn).
645
+ @tick&.start(parent: @supervisor)
646
+ # the alert dispatcher (WS6) lives on the same supervisor: its consumer
647
+ # answers every alert event for as long as the process serves.
648
+ @alert_dispatcher&.start(parent: @supervisor)
649
+ @supervisor
650
+ end
651
+
652
+ # Deterministic PendingAction id: correlation by task+turn+tool.
653
+ # Limitation inherited from the side-effect: the SAME tool with approval more
654
+ # than once in a turn collides (the 2nd reuses the 1st's decision) — per-step
655
+ # checkpointing is a future slice. One call per tool is safe.
656
+ def pending_id(task_id, turn, tool) = "#{task_id}:#{turn}:#{tool}"
657
+
658
+ # all readings of the persisted command go through rebuild_command
659
+ # (the single normalizer). command_type stays a STRING (the :task_started
660
+ # event and the Telemetry attribute are string-typed).
661
+ def command_type(task)
662
+ rebuild_command(task).type.to_s
663
+ end
664
+
665
+ # Closes the orphan Execution (open) of an attempt interrupted by a crash,
666
+ # marking it :interrupted — the record of what happened is preserved; the
667
+ # N+1 attempt is opened right after by begin_execution.
668
+ def close_orphan_execution(task)
669
+ open = @task_store.find(task.id).executions.last
670
+ @task_store.finish_execution(task.id, outcome: :interrupted) if open && open.finished_at.nil?
671
+ end
672
+
673
+ # Maps error -> task :failed + events. `transition` with error: ALREADY closes
674
+ # the open Execution (real TaskStore) — do NOT call finish_execution
675
+ # (double-close). The previous checkpoint is NEVER touched on failure. Never
676
+ # re-raises: the fiber dies clean.
677
+ def fail_task(task, error, stage:)
678
+ # Defense-in-depth: if the task is already terminal (e.g. a failure in
679
+ # cleanup AFTER transition(:completed)), completed->failed is invalid and
680
+ # would raise ArgumentError INSIDE the rescue, leaking from the fiber. In
681
+ # that case only report the error — the durability of the committed turn is
682
+ # preserved. This :error is the ONE deliberate, non-twin executor error
683
+ # signal (R2b): the terminal already fired, so a session-scoped subscriber
684
+ # is the only one that may still see it. Same family as the overflow :error.
685
+ current = @task_store.find(task.id)
686
+ if current && TERMINAL_STATUSES.include?(current.status)
687
+ emit(:error, { message: error.message }, task: task)
688
+ return nil
689
+ end
690
+
691
+ # The task's error record and the :task_failed event both carry the
692
+ # classification when the failure is a wrapped ProviderError (B9/A8):
693
+ # additive fields, absent for every other error.
694
+ classification = error.respond_to?(:classification) ? error.classification : {}
695
+ spec = { class: error.class.name, message: error.message, stage: stage }
696
+ spec = spec.merge(classification) unless classification.empty?
697
+ @task_store.transition(task.id, to: :failed, error: spec)
698
+ data = { task_id: task.id, error: error.class.name, message: error.message }
699
+ data = data.merge(classification) unless classification.empty?
700
+ emit(:task_failed, data, task: task)
701
+ # a FAILED delegation child still delivers — the parent
702
+ # receives an error note as a new turn (never left hanging).
703
+ finalize_delegation(task)
704
+ nil
705
+ end
706
+
707
+ TERMINAL_STATUSES = %i[completed failed cancelled].freeze
708
+ private_constant :TERMINAL_STATUSES
709
+
710
+ # Stages 2-9, with mailbox drain only at the boundaries and the
711
+ # turn-timeout wrapping everything via Async::Task#with_timeout — NEVER
712
+ # stdlib Timeout.timeout.
713
+ def run_pipeline(task, profile, actor, resume_from)
714
+ timing = TurnTiming.new if TurnTiming.enabled?
715
+ timing&.mark(:prep_start)
716
+ state = build_turn_state(task, profile, resume_from)
717
+ turn_timeout = turn_timeout_for(profile)
718
+
719
+ Async::Task.current.with_timeout(turn_timeout) do
720
+ # :task pair: wraps the turn's stages. before_task may
721
+ # rewrite the TurnState before stage 2; after_task runs after stage 9.
722
+ # The block-param `state` shadows the outer one (uses the TurnState
723
+ # possibly rewritten by before_task). Subject == result == TurnState
724
+ # (the Response content lives in the :done event).
725
+ @hooks.around(:task, state) do |state|
726
+ prepare_turn(task, profile, state, actor, resume_from) # stages 2-3 (mutates state)
727
+
728
+ # stage 4: Middleware wraps stages 5-9. A link that
729
+ # short-circuits does NOT call the terminal and sets state.halt_reason.
730
+ terminal_ran = false
731
+ @middleware.call(state) do |st|
732
+ raise Insika::Error, "turn halted: #{st.halt_reason}" if st.halt_reason
733
+
734
+ terminal_ran = true
735
+ run_turn_body(task, profile, st, actor, timing) # stages 5-9
736
+ end
737
+
738
+ # A Middleware short-circuited (did not call the terminal). Three cases:
739
+ # · halt_response set -> GRACEFUL halt: the turn COMPLETES
740
+ # with a safe reply, reusing stages 8-9, without ever touching the LLM.
741
+ # · halt_reason set -> halt-as-FAILURE (the pre-existing contract).
742
+ # · neither -> contract violation (short-circuit with no signal).
743
+ if !terminal_ran && state.halt_response
744
+ complete_with_halt(task, profile, state)
745
+ elsif state.halt_reason
746
+ raise Insika::Error, "turn halted: #{state.halt_reason}"
747
+ elsif !terminal_ran
748
+ raise Insika::Error, "middleware short-circuited without halt_reason"
749
+ end
750
+
751
+ state # subject of the :task pair (after_task receives it; the caller discards)
752
+ end.tap { |st| emit_guardrail_flags(task, st) }
753
+ end
754
+ rescue Async::TimeoutError
755
+ raise Insika::TimeoutError.new("turn exceeded #{turn_timeout}s", stage: :turn)
756
+ end
757
+
758
+ # Builds the turn's mutable state (turn number, message, memory tenant,
759
+ # turn context). before_task (hooks.around) may still rewrite it before stage 2.
760
+ def build_turn_state(task, profile, resume_from)
761
+ turn = resume_from ? resume_from.turn : 1
762
+ state = TurnState.new(task: task, profile: profile, turn: turn,
763
+ message: extract_message(task))
764
+ state.tenant = memory_tenant(task) # WRITE-path memory scope (`remember`); =chat
765
+ state.turn_context = build_turn_context(task, profile, state) # data-tools' ctx.*
766
+ state.resumed = !resume_from.nil? # EdgeLimiter: an admitted turn is never re-counted
767
+ # resolved for the RUN, not per message, so what the turn accepts cannot
768
+ # change under it. Same cost as the EdgeLimiter's per-turn resolution. Only for a
769
+ # SESSION turn: steering needs a session to arrive through, and resolving here for a
770
+ # one-shot would make an unrelated turn fail on a queue key it can never use.
771
+ state.queue_policy = task.session_id ? queue_policy(profile, task.session_id) : nil
772
+ state
773
+ end
774
+
775
+ # with_timeout budget for the turn. A turn that MAY require human approval
776
+ # gets approval_timeout (~1h) so the wrapper does not kill a legitimate
777
+ # operator wait. LLM runaway is already bounded by max_tool_calls + turn_timeout.
778
+ def turn_timeout_for(profile)
779
+ base = profile.limits[:turn_timeout] || 300
780
+ return base if Array(profile.approvals_required).empty?
781
+
782
+ [base, profile.limits[:approval_timeout] || 3_600].max
783
+ end
784
+
785
+ # Stages 2-3: Context -> initial checkpoint -> capability resolution -> Policy
786
+ # -> tool assembly, each followed by a mailbox drain at the boundary. Mutates
787
+ # `state` in place (the TurnState possibly rewritten by before_task).
788
+ def prepare_turn(task, profile, state, actor, resume_from)
789
+ # stage 2: Context. The :prompt hook pair is wrapped INSIDE the
790
+ # ContextBuilder#call — do NOT wrap here (a double-wrap would fire the hooks
791
+ # twice). Hooks is the SAME instance injected into the Builder and here.
792
+ request = build_context_request(task, profile, state, resume_from)
793
+ state.context = @context_builder.call(request)
794
+ drain_and_maybe_suspend(task, actor)
795
+
796
+ # INITIAL checkpoint of the turn ("the checkpoint of turn n contains the
797
+ # state AT THE START of turn n"). Without it, a crash mid stage 6 (before
798
+ # stage 8) would orphan the task WITH NO checkpoint -> unrecoverable.
799
+ # Idempotent: only writes on the 1st turn of a new task.
800
+ save_initial_checkpoint(task, profile, state)
801
+
802
+ # capability-resolution sub-step BETWEEN Context and Policy: fills
803
+ # state.capability_names for the POST-Policy join; candidate_tools stays
804
+ # tool_registry.entries (a capability does not go through the ToolAllowlist).
805
+ state.capability_names = resolve_capabilities(profile, state.context)
806
+
807
+ # stage 3: Policy (candidate_skills from the CATALOG; candidate_tools =
808
+ # tool_registry.entries, direct tools only).
809
+ resolution = @policy_engine.decide(policy_request(profile, task, state))
810
+ # on resume, tool calls already completed in the interrupted turn are
811
+ # "skipped" (standalone key ∪ turn checkpoint), propagated to promoted tools too.
812
+ skip = resume_from ? @checkpoint_store.side_effects(task.id, turn: state.turn) : []
813
+ state.skip_side_effects = skip
814
+ # wire the approval gate into the state (the ToolEnvelope reads it at stage 6).
815
+ state.actor = actor
816
+ state.approval_coordinator = self
817
+ state.requires_approval = resolution.requires_approval
818
+ state.allowed_tools = wrap_tools(assemble_tool_instances(resolution.allowed_tools, state), state, skip)
819
+ state.allowed_skills = resolution.allowed_skills
820
+ record_context_trace(task, state)
821
+ announce_context_skills(task, state)
822
+ drain_and_maybe_suspend(task, actor)
823
+ end
824
+
825
+ # one entry per turn in the ContextTraceStore — tokens per
826
+ # category (the demodulized provider id), the tools-schema estimate and the
827
+ # budget verdict. Counts and ids ONLY, never content. nil store = off; the
828
+ # store itself rescues everything (the trace never breaks the turn).
829
+ def record_context_trace(task, state)
830
+ return unless @context_trace_store && task.session_id
831
+
832
+ package = state.context
833
+ # A custom builder that does not produce the full package (fragments +
834
+ # budget) simply has no breakdown to record — never an error.
835
+ return unless package.respond_to?(:fragments) && package.respond_to?(:budget)
836
+
837
+ categories = package.fragments.each_with_object({}) do |f, acc|
838
+ c = (acc[context_category(f.source)] ||= { tokens: 0, fragments: 0, pinned: 0 })
839
+ c[:tokens] += f.tokens || 0
840
+ c[:fragments] += 1
841
+ c[:pinned] += (f.tokens || 0) if f.pinned
842
+ # WHICH skills/tools the fragment carried and WHY — ids only, still
843
+ # content-free. Without this the trace proves a turn injected N tokens of
844
+ # skill but not which ones, and deterministic activation is unauditable
845
+ # after the fact.
846
+ labels = Array(f.labels)
847
+ (c[:labels] ||= []).concat(labels) unless labels.empty?
848
+ end
849
+ @context_trace_store.record(
850
+ session_id: task.session_id,
851
+ entry: { task_id: task.id, turn: state.turn, at: Time.now.utc.iso8601,
852
+ cap: package.budget[:cap], used: package.budget[:used],
853
+ evicted: package.budget[:evicted], categories: categories,
854
+ tools: { count: state.allowed_tools.size,
855
+ tokens: estimate_tools_tokens(state.allowed_tools) } }
856
+ )
857
+ end
858
+
859
+ # Skill bodies that reached the prompt WITHOUT a tool call (`triggers:` or
860
+ # `skills_eager`) leave no trace in the transcript: there is no load_skill to
861
+ # render, so an active skill looked exactly like an absent one.
862
+ #
863
+ # Emitted HERE rather than in the provider because only the Executor holds the
864
+ # correlation the Studio's SSE filters on — an event whose meta lacks `task_id`
865
+ # never reaches a task-scoped subscriber (EventStream::Subscription#matches?).
866
+ # `skills` (plural, with reasons) marks the CONTEXT path; the load_skill tool
867
+ # emits the same type with a singular `name`, and the Studio must not conflate them.
868
+ #
869
+ # Read from `package.fragments`, which is POST-BUDGET: a body the cut evicted is
870
+ # not in the prompt, and announcing it as active would make the one surface built
871
+ # to tell the truth the one that lies. Eviction is reported by the trace's own
872
+ # `evicted` list, never as an activation.
873
+ SKILL_BODY_CATEGORY = "skilltrigger"
874
+
875
+ def announce_context_skills(task, state)
876
+ package = state.context
877
+ return unless package.respond_to?(:fragments)
878
+
879
+ skills = Array(package.fragments)
880
+ .select { |f| context_category(f.source) == SKILL_BODY_CATEGORY }
881
+ .flat_map { |f| Array(f.labels) }
882
+ .map { |l| { name: l["name"], reason: l["reason"] || "pack" } }
883
+ .uniq
884
+ return if skills.empty?
885
+
886
+ emit(:skill_activated, { skills: skills, source: "context" }, task: task)
887
+ end
888
+
889
+ # "Insika::Context::Providers::Prompt" -> "prompt" (a plugin provider keeps
890
+ # its own demodulized name — still content-free).
891
+ def context_category(source) = source.to_s.split("::").last.to_s.downcase
892
+
893
+ # Same yardstick as the fragments (TokenEstimator), so the categories are
894
+ # comparable. `parameters` is not guaranteed JSON-safe — inspect it.
895
+ def estimate_tools_tokens(tools)
896
+ TokenEstimator.estimate(tools.map { |t| "#{t.name} #{t.description} #{t.parameters.inspect}" }.join(" "))
897
+ rescue StandardError
898
+ 0
899
+ end
900
+
901
+ # Stages 5-9 (inside the Middleware wrap): assemble chat, the single agent
902
+ # interaction, persistence, terminal event. `st` is the Middleware-yielded state.
903
+ def run_turn_body(task, profile, st, actor, timing = nil)
904
+ # stage 5: assemble chat + check mailbox (send_message only; a workflow does
905
+ # not use the Insika chat — it orchestrates RubyLLM internally).
906
+ drain_and_maybe_suspend(task, actor)
907
+ unless workflow_turn?(task)
908
+ st.chat = create_chat(profile, st)
909
+ @chat_builder.assemble(st.chat, st, emit: ->(type, data) { emit(type, data, task: task) })
910
+ # R1: baseline = seeded-history size, before `ask` appends the turn.
911
+ st.chat_baseline = Array(st.chat.messages).size if st.chat.respond_to?(:messages)
912
+ end
913
+
914
+ # guardrails: per-turn stream redactor (nil = off).
915
+ st.output_filter = @content_filter_factory&.call(st)
916
+
917
+ # stage 6: the turn's single agent interaction (send_message -> chat.ask;
918
+ # trigger_workflow -> workflow.call). Returns the turn's final content.
919
+ content = run_agent_stage(task, st, timing)
920
+ st.response_content = content # after_task OutputValidator inspects this
921
+
922
+ # stage 8: Persistence (fixed order checkpoint->session->task). pure drain!
923
+ # (NEVER suspends at stage 8 — forbidden window): a :pause here arms the flag
924
+ # but is not honored (last stage); :cancel here still raises.
925
+ actor.drain!
926
+ persist_turn(task, profile, st, content)
927
+
928
+ # stage 9: Response. usage (tokens) captured at stage 6 travels in the
929
+ # terminal event -> /v1/responses usage + Telemetry (OTEL).
930
+ timing&.mark(:done)
931
+ data = { task_id: task.id, content: content, usage: st.usage }
932
+ data[:timing] = timing.to_h if timing # opt-in TTFB breakdown (INSIKA_TURN_TIMING)
933
+ # WS5: the agent declared it cannot proceed (signal_stuck). The turn still
934
+ # COMPLETES (its final message was published) — but the consumer must be able
935
+ # to act on that, so the contract carries it twice: a dedicated :turn_stuck
936
+ # event (subscribable) and an additive `outcome` sibling on the terminal event.
937
+ if (stuck = st.stuck_outcome)
938
+ emit(:turn_stuck, { task_id: task.id, agent: profile.id.to_s,
939
+ reason: stuck[:reason], message: content }, task: task)
940
+ data[:outcome] = :stuck
941
+ end
942
+ emit(:task_completed, data, task: task)
943
+ end
944
+
945
+ def workflow_turn?(task)
946
+ command_type(task).to_s == "trigger_workflow"
947
+ end
948
+
949
+ # Stage 6: the single agent interaction. Returns the turn's final content.
950
+ def run_agent_stage(task, state, timing = nil)
951
+ if workflow_turn?(task)
952
+ # workflow = a Ruby callable that orchestrates RubyLLM internally (RubyLLM
953
+ # First). tools: are the SAME instances filtered by the Resolution and
954
+ # enveloped (stage 7) — the workflow inherits timeout/side-effect/skip.
955
+ # the EXPOSED surface — the run (== task.id) is announced on
956
+ # the stream (:workflow_started), the RETURN is validated against the
957
+ # output_schema (WorkflowSchemaError -> :workflow_schema stage), and the
958
+ # typed output is published (:workflow_completed).
959
+ definition = @workflow_registry.definition(workflow_name(task))
960
+ emit(:workflow_started,
961
+ { run_id: task.id, workflow: definition.name, input: state.message || {} }, task: task)
962
+ output = @hooks.around(:agent, state) do |s|
963
+ # input omitted from the payload -> {} (the workflow expects a Hash).
964
+ definition.call(s.message || {}, context: s.context, tools: s.allowed_tools)
965
+ end
966
+ definition.validate_output!(output)
967
+ emit(:workflow_completed, { run_id: task.id, workflow: definition.name, output: output }, task: task)
968
+ output
969
+ else
970
+ filter = state.output_filter # nil = off (stream untouched)
971
+ timing&.mark(:ask)
972
+ # TurnOutput owns what the customer is allowed to read: chunks ride
973
+ # :intermediate live and only the message that ENDS the turn is published as
974
+ # :content. Registered on the chat (fresh per turn/attempt, no leak).
975
+ # With WS3 reliability the attempts build their own chats + outputs.
976
+ output = nil
977
+ asked = nil
978
+ response = @hooks.around(:agent, state) do |s|
979
+ result = @reliability ? run_reliable_ask(task, s, filter, timing)
980
+ : run_single_ask(task, s, filter, timing)
981
+ output = result[:output]
982
+ asked = result[:asked]
983
+ result[:response]
984
+ end
985
+ # release the redactor's retained tail (a value that never completed into a
986
+ # match is emitted redacted-if-needed, not lost) before reading anything back.
987
+ output.flush
988
+ state.usage = with_model_source(usage_of(response), state.model_selection) unless halted?(response)
989
+
990
+ # BOUNDARY BEFORE THE ANSWER GOES OUT. A cancel that arrived while the provider
991
+ # was working used to be observed at stage 8 — AFTER `:content` had already been
992
+ # published — so the customer read the answer of a turn that then terminated
993
+ # `:cancelled` and persisted nothing: text delivered, transcript silent about it.
994
+ # Honoring it here is what makes `interrupt` mean anything, and it
995
+ # is a safe boundary: the tool batch is finished and nothing is half applied.
996
+ # A `:pause` is deliberately NOT honored here (drain!, not the suspending form):
997
+ # holding a completed answer for an operator would strand it unpublished.
998
+ state.actor&.drain!
999
+
1000
+ output.publish(turn_answer(response, asked, output, filter))
1001
+ end
1002
+ end
1003
+
1004
+ # stage 6, plain path: the single ask on the assembled chat. Fresh TurnOutput
1005
+ # + steer wiring per interaction (registered on state.chat, which the solve
1006
+ # already assembled). -> { response:, asked:, output: }.
1007
+ def run_single_ask(task, state, filter, timing)
1008
+ output = new_turn_output(task, state, filter)
1009
+ wire_chat_output(task, state, output)
1010
+ asked = ask_on(task, state, state.chat, output, timing)
1011
+ { response: asked, asked: asked, output: output }
1012
+ end
1013
+
1014
+ # stage 6, WS3 path: the Reliability coordinator drives retries, backoff,
1015
+ # circuit breaker and the fallback rotation. Each ATTEMPT gets a fresh chat
1016
+ # + output (a failed `ask` leaves its message in the chat, so re-asking the
1017
+ # same one would double the input) and, on a fallback, state.model_selection
1018
+ # follows — the turn's usage is attributed to the model that actually spoke
1019
+ # ("contabilizado no trace"). -> { response:, asked:, output: }.
1020
+ def run_reliable_ask(task, state, filter, timing)
1021
+ policy = state.profile.respond_to?(:reliability) ? state.profile.reliability : nil
1022
+ return run_single_ask(task, state, filter, timing) if policy.nil? || @reliability.nil?
1023
+
1024
+ attempt_output = nil
1025
+ attempt_asked = nil
1026
+ primary = state.model_selection
1027
+ response = @reliability.call(
1028
+ policy: policy, tenant: task_tenant(task), agent: state.profile.id.to_s,
1029
+ selection: primary, chain: reliability_chain(state)
1030
+ ) do |selection, tries|
1031
+ first_attempt = state.chat && selection == primary && tries == 1
1032
+ if selection != primary
1033
+ state.model_selection = selection # attribution follows the fallback
1034
+ end
1035
+ unless first_attempt
1036
+ chat = build_attempt_chat(state, selection)
1037
+ state.chat = chat
1038
+ end
1039
+ attempt_output = new_turn_output(task, state, filter)
1040
+ wire_chat_output(task, state, attempt_output)
1041
+ attempt_asked = ask_on(task, state, state.chat, attempt_output, timing)
1042
+ attempt_asked
1043
+ end
1044
+ { response: response, asked: attempt_asked, output: attempt_output }
1045
+ end
1046
+
1047
+ # The fallback chain for WS3: profile's `reliability["fallback"]` refs first,
1048
+ # then the platform-resolved fallbacks (ModelSelection). Each node is a
1049
+ # ModelSelection (the SAME duck the primary is — usage attribution and
1050
+ # apply_params just work), source: :fallback, params inherited from the
1051
+ # primary. Deduped by ref, primary excluded.
1052
+ def reliability_chain(state)
1053
+ primary = state.model_selection
1054
+ refs = Array((state.profile.reliability || {})["fallback"]).map(&:to_s)
1055
+ nodes = refs.filter_map { |r| parse_model_ref(r) }.reject { |n| n[:model].to_s.empty? }
1056
+ nodes.concat(Array(primary.fallbacks).map { |f| { model: f[:model], provider: f[:provider] } })
1057
+ seen = { ref_of(primary) => true }
1058
+ nodes.filter_map do |node|
1059
+ ref = model_ref(node)
1060
+ # normalize "model" vs "provider/model": a provider-less ref IS the same
1061
+ # physical model as any known "provider/model" spelling of it — the same
1062
+ # model must never be tried twice just because one spelling omits the
1063
+ # provider (WS3: fallback ["deepseek-v4-flash"] under primary
1064
+ # deepseek/deepseek-v4-flash used to re-ask the dropped primary). A
1065
+ # qualified ref still matches exactly.
1066
+ duplicate = ref.include?("/") ? seen[ref]
1067
+ : seen.keys.any? { |known| known.split("/").last == ref }
1068
+ next if duplicate
1069
+
1070
+ seen[ref] = true
1071
+ Insika::ModelSelection.new(model: node[:model], provider: node[:provider],
1072
+ source: :fallback, params: primary.params, fallbacks: [])
1073
+ end
1074
+ end
1075
+
1076
+ def ref_of(selection) = model_ref(selection)
1077
+
1078
+ # "provider/model" for any selection duck (ModelSelection | { model:, provider: }).
1079
+ def model_ref(selection)
1080
+ model = selection.respond_to?(:model) ? selection.model.to_s : selection[:model].to_s
1081
+ provider = selection.respond_to?(:provider) ? selection.provider : selection[:provider]
1082
+ provider ? "#{provider}/#{model}" : model
1083
+ end
1084
+
1085
+ # "provider/model" -> { model:, provider: }; "model" -> { model:, provider: nil }.
1086
+ def parse_model_ref(entry)
1087
+ s = entry.to_s.strip
1088
+ return nil if s.empty?
1089
+
1090
+ if s.include?("/")
1091
+ provider, model = s.split("/", 2)
1092
+ { model: model, provider: presence_or_nil(provider)&.to_sym }
1093
+ else
1094
+ { model: s, provider: nil }
1095
+ end
1096
+ end
1097
+
1098
+ def presence_or_nil(value)
1099
+ v = value.to_s.strip
1100
+ v.empty? ? nil : v
1101
+ end
1102
+
1103
+ # A fresh chat for a retry/fallback attempt: REASSEMBLED from the same turn
1104
+ # state (the seed history is identical), baseline reset -> the transcript
1105
+ # recorded from than point is the attempt that spoke.
1106
+ def build_attempt_chat(state, selection)
1107
+ chat = build_chat(selection, state.model_selection)
1108
+ @chat_builder.assemble(chat, state, emit: ->(type, data) { emit(type, data, task: state.task) })
1109
+ state.chat_baseline = Array(chat.messages).size if chat.respond_to?(:messages)
1110
+ chat
1111
+ end
1112
+
1113
+ def new_turn_output(task, state, filter)
1114
+ TurnOutput.new(filter: filter, emit: ->(type, data) { emit(type, data, task: task) },
1115
+ public_intermediate: state.profile.stream_public?(:intermediate))
1116
+ end
1117
+
1118
+ # The message-boundary + steer wiring ON the current chat. Registered
1119
+ # AFTER TurnOutput so the publishing decision for a message is made before
1120
+ # anything is appended after it — the gem's callbacks are additive and run
1121
+ # in registration order.
1122
+ def wire_chat_output(task, state, output)
1123
+ chat = state.chat
1124
+ chat.after_message { |message| output.message_ended(message) } if chat.respond_to?(:after_message)
1125
+ install_steer_injector(task, state)
1126
+ end
1127
+
1128
+ # The ask itself, chunk-by-chunk (WS3 attempts and the plain path share it).
1129
+ # With INSIKA_TURN_TIMING the FIRST content chunk also emits the live
1130
+ # :ttft event — the streaming envelope's TTFB signal (WS6), additive. The
1131
+ # emit is gated to that first chunk: a probe proved the old code re-emitted
1132
+ # :ttft on EVERY content chunk (3 chunks = 3 insika.ttft frames); the spec
1133
+ # passed because FakeChat emits a single chunk.
1134
+ def ask_on(task, state, chat, output, timing)
1135
+ public_thinking = state.profile.stream_public?(:thinking)
1136
+ ttft_sent = false
1137
+ chat.ask(state.message) do |chunk|
1138
+ emit_thinking(chunk, task, public: public_thinking)
1139
+ next unless chunk.content
1140
+
1141
+ timing&.mark(:first_token) # first-write-wins -> the PROVIDER's TTFB
1142
+ unless ttft_sent
1143
+ emit_ttft(task, timing) if timing
1144
+ ttft_sent = true
1145
+ end
1146
+ output.push(chunk.content)
1147
+ end
1148
+ end
1149
+
1150
+ # The provider's TTFB as a live event (data: ttft_ms) — only under
1151
+ # INSIKA_TURN_TIMING, so absent by default (parity).
1152
+ def emit_ttft(task, timing)
1153
+ ttft = timing.to_h[:ttft_ms]
1154
+ return if ttft.nil?
1155
+
1156
+ @event_stream.emit(Insika::Event.new(
1157
+ type: :ttft, data: { ttft_ms: ttft },
1158
+ meta: { task_id: task.id, session_id: task.session_id,
1159
+ at: Time.now.utc.iso8601 }
1160
+ ))
1161
+ rescue StandardError
1162
+ nil
1163
+ end
1164
+
1165
+ # wires the tool-batch boundary that lets a message which arrived
1166
+ # mid-run enter the conversation. No-op unless the agent asked for `steer`: an
1167
+ # unregistered callback is the difference between a feature that is off and one that
1168
+ # is on and finds nothing.
1169
+ def install_steer_injector(task, state)
1170
+ policy = state.queue_policy
1171
+ return unless policy&.steer? && task.session_id && state.actor
1172
+ # A chat that does not answer the three callbacks cannot host the boundary (the smoke
1173
+ # shim, a minimal double). Steering is then simply off, never half-wired.
1174
+ return unless %i[after_message after_tool_result add_message].all? { |m| state.chat.respond_to?(m) }
1175
+
1176
+ injector = SteerInjector.new(
1177
+ chat: state.chat, actor: state.actor, policy: policy,
1178
+ # `task_id` in the payload as well as the meta, so `:turn_steered` reads like
1179
+ # `:turn_coalesced` for a subscriber that only looks at data.
1180
+ emit: ->(type, data) { emit(type, data.merge(task_id: task.id), task: task) }
1181
+ )
1182
+ state.chat.after_message { |message| injector.message_ended(message) }
1183
+ state.chat.after_tool_result { |result| injector.tool_result(result) }
1184
+ injector
1185
+ end
1186
+
1187
+ # WHAT THE CUSTOMER GETS, in precedence order. Published once, at the end of the
1188
+ # stage, because the `:agent` after-hook runs after the message boundary and is
1189
+ # allowed to have the last word.
1190
+ def turn_answer(response, asked, output, filter)
1191
+ # HALTED BY A TOOL RESULT (`halt_when`): RubyLLM returns the Tool::Halt itself
1192
+ # instead of a Message, and its `content` is the tool PAYLOAD — publishing it
1193
+ # would ship the envelope to the customer as the answer. The turn is worth
1194
+ # exactly the lead-in of the message that called the tool (the model's "vou te
1195
+ # inscrever agora"), and nothing after: the backend already said the rest.
1196
+ # Nothing streamed -> the tool's `halt_when.say`, when it declared one; else an
1197
+ # empty turn, which is what the consumer suppresses. The lead-in still WINS —
1198
+ # a model that introduced the escalation already said the right thing, and
1199
+ # publishing both would deliver the message twice, which is the whole reason
1200
+ # `halt_when` exists.
1201
+ return halt_answer(response, output) if halted?(response)
1202
+ # An :agent after-hook REPLACED the response. An explicit override outranks
1203
+ # what the model streamed (the OutputValidator still sees the result).
1204
+ return content_of(response) unless response.equal?(asked)
1205
+ # The normal path: the message that ended the turn, as it was streamed and
1206
+ # (when guardrails are on) redacted.
1207
+ return output.candidate if output.candidate
1208
+
1209
+ # No message boundary was reported — a transport that does not implement
1210
+ # `after_message`. Fall back to the whole turn's redacted text, else the raw
1211
+ # response: the pre-boundary behaviour, kept as the floor.
1212
+ filter ? filter.output : content_of(response)
1213
+ end
1214
+
1215
+ # The lead-in when there is one; otherwise whatever the halting tool declared it
1216
+ # wants the customer to get (`halt_when.say`). Never both.
1217
+ def halt_answer(response, output)
1218
+ lead = output.halt_text.to_s
1219
+ return lead unless lead.strip.empty?
1220
+
1221
+ Insika::ToolDefinition.halt_say_of(content_of(response)).to_s
1222
+ end
1223
+
1224
+ def halted?(response) = response.is_a?(RubyLLM::Tool::Halt)
1225
+
1226
+ def content_of(response) = response.respond_to?(:content) ? response.content : response.to_s
1227
+
1228
+ # The provider's REASONING (DeepSeek `reasoning_content`, Anthropic thinking
1229
+ # blocks): RubyLLM parks it in `chunk.thinking`, NEVER in `chunk.content`, so it
1230
+ # was being dropped on the floor — invisible in the Studio and in the trace. It
1231
+ # rides the Event Stream as its OWN type (:thinking), and `/v1/responses` does not
1232
+ # translate it BY DEFAULT: the deliberation is observability, not the answer.
1233
+ #
1234
+ # An agent may opt in (`edge_stream thinking: true`) — a product with a "thinking"
1235
+ # panel wants it — and then the event is TAGGED `public: true` and the edge gives
1236
+ # it the reasoning frame, never the answer's. The tag rides the event because
1237
+ # `Responses.frame_for` is a pure static mapper with no agent in scope.
1238
+ #
1239
+ # Three deliberate omissions:
1240
+ # · the guardrail filter is NOT applied — it accumulates the PERSISTED content
1241
+ # and pushing reasoning through it would corrupt the turn's answer;
1242
+ # · `timing.mark(:first_token)` stays on the content chunks — ttft is the
1243
+ # PROVIDER's first token ('s baselines measure that, not the first
1244
+ # thought, and not when TurnOutput publishes the answer);
1245
+ # · nothing is persisted — the reasoning is not part of the conversation.
1246
+ #
1247
+ # Duck-typed like `usage_of`: a provider/fake with no thinking -> nothing to emit.
1248
+ def emit_thinking(chunk, task, public: false)
1249
+ return unless chunk.respond_to?(:thinking)
1250
+
1251
+ thought = chunk.thinking
1252
+ text = thought.respond_to?(:text) ? thought.text : thought # RubyLLM::Thinking | String | nil
1253
+ return if text.to_s.empty?
1254
+
1255
+ data = { delta: text.to_s }
1256
+ data[:public] = true if public
1257
+ emit(:thinking, data, task: task)
1258
+ end
1259
+
1260
+ # Annotates the usage with the RESOLVED model-selection source:
1261
+ # where the model came from (:chat/:agent/:platform_default) travels alongside
1262
+ # the resolved model id (from the provider) into the terminal event/Telemetry,
1263
+ # so billing/telemetry can attribute the turn to a config layer. nil usage
1264
+ # (workflow / provider without counts) -> nothing to annotate.
1265
+ def with_model_source(usage, selection)
1266
+ return usage if usage.nil? || selection.nil?
1267
+
1268
+ usage[:model_source] = selection.source
1269
+ usage[:model] ||= selection.model # falls back to the resolved id when the provider omits it
1270
+ usage
1271
+ end
1272
+
1273
+ # Token usage of the provider's response (RubyLLM::Message exposes
1274
+ # input_tokens/output_tokens/cached_tokens/model_id). Duck-typed: a provider/
1275
+ # fake with no counting -> nil (nothing to report). Shape compatible with the
1276
+ # OpenAI Responses usage (input/output/total) + model, consumed by
1277
+ # /v1/responses and Telemetry.
1278
+ def usage_of(response)
1279
+ return nil unless response.respond_to?(:input_tokens)
1280
+
1281
+ input = response.input_tokens.to_i
1282
+ output = response.output_tokens.to_i
1283
+ usage = { input_tokens: input, output_tokens: output, total_tokens: input + output }
1284
+ if response.respond_to?(:cached_tokens) && response.cached_tokens
1285
+ usage[:cached_tokens] = response.cached_tokens.to_i # cache_read_input_tokens
1286
+ end
1287
+ # R3: prompt-cache WRITE tokens (Anthropic cache_creation_input_tokens),
1288
+ # billed at ~1.25x. Reported so the first (write) turn vs later (read) turns
1289
+ # are distinguishable in telemetry/usage.
1290
+ if response.respond_to?(:cache_creation_tokens) && response.cache_creation_tokens
1291
+ usage[:cache_creation_tokens] = response.cache_creation_tokens.to_i
1292
+ end
1293
+ usage[:model] = response.model_id.to_s if response.respond_to?(:model_id) && response.model_id
1294
+ usage
1295
+ end
1296
+
1297
+ def workflow_name(task)
1298
+ rebuild_command(task).payload["workflow"]
1299
+ end
1300
+
1301
+ # turn message: send_message -> payload.message; trigger_workflow ->
1302
+ # payload.input (the input becomes the "user" content and the workflow
1303
+ # argument).
1304
+ def extract_message(task)
1305
+ payload = rebuild_command(task).payload
1306
+ payload["message"] || payload["input"]
1307
+ end
1308
+
1309
+ def command_history(task)
1310
+ rebuild_command(task).payload["history"]
1311
+ end
1312
+
1313
+ def build_context_request(task, profile, state, resume_from)
1314
+ session = task.session_id ? @session_store.find(task.session_id) : nil
1315
+ state.session = session # create_chat reads it for the per-chat model pin
1316
+ hist = command_history(task)
1317
+ # `vars` reconciles the seam (the Request/Session provider already
1318
+ # called request.vars): session metadata + the explicit `history` in the
1319
+ # convention the Session provider consumes (vars["history"]).
1320
+ vars = (session&.vars || {}).dup
1321
+ vars["history"] = hist if hist
1322
+ # The single type is Insika::ContextRequest (Data); the explicit `history`
1323
+ # travels in vars["history"] (Session provider convention), not in a field
1324
+ # of its own.
1325
+ ContextRequest.new(profile: profile, message: state.message, session: session,
1326
+ checkpoint: resume_from, tenant: command_tenant(task), vars: vars)
1327
+ end
1328
+
1329
+ # task_started payload. Carries the EXPLICIT command tenant so
1330
+ # the observability convention can group by it — the one operator-set label that
1331
+ # is not derivable from the task itself. Omitted when absent: the terminal
1332
+ # events keep their shape and no consumer sees a null it never saw before. NOT
1333
+ # memory_tenant: that one falls back to the chat id (per-chat cardinality).
1334
+ def started_data(task, profile)
1335
+ data = { task_id: task.id, command: command_type(task), agent: profile&.id }
1336
+ tenant = command_tenant(task)
1337
+ data[:tenant] = tenant if tenant
1338
+ data
1339
+ end
1340
+
1341
+ # Command tenant (Command.build(..., tenant:) -> meta[:tenant],
1342
+ # command.rb). Absent -> nil (the MemoryStore applies DEFAULT_TENANT).
1343
+ def command_tenant(task)
1344
+ rebuild_command(task).meta["tenant"]
1345
+ end
1346
+
1347
+ # Who produced the message this turn is answering (MessageOrigin). Absent for an
1348
+ # ordinary turn — a customer typed it — and declared by the producer when not:
1349
+ # the engine delivering a delegation result, or a consumer that composed the
1350
+ # input out of context blocks. Validated where it enters (Commands::SendMessage).
1351
+ def command_origin(task)
1352
+ Coercion.presence(rebuild_command(task).payload["origin"])
1353
+ end
1354
+
1355
+ # Engine memory scope: the Command's EXPLICIT tenant wins (multi-merchant
1356
+ # override); otherwise the SESSION (=chat) — engine-owner memory is per-chat.
1357
+ # Symmetric to the READ path (Memory provider). One-shot with no tenant -> nil
1358
+ # (_default). It is NOT the <request_context> tenant (that follows
1359
+ # command_tenant, prompt parity) — only the memory read/write scope.
1360
+ def memory_tenant(task)
1361
+ command_tenant(task) || task.session_id
1362
+ end
1363
+
1364
+ # Turn context: the ids the data-tools resolve via
1365
+ # {{ctx.*}} to emit X-Chat-Id/X-Store-Id/X-Agent-Id to /api/internal/*. They
1366
+ # come from the TURN, never from the model args (R2). chat_id = the session
1367
+ # (the /v1/responses adapter creates the session with id = user = chat.id);
1368
+ # tenant = the Command tenant (memory) OR chat_id (drop-in default); agent_id =
1369
+ # profile; store_id = the profile metadata (stable per store, from the pack).
1370
+ # Absent fields -> nil (the data-tool emits an empty header; in the pilot the
1371
+ # profile carries store_id). Generic: nothing here mentions a consumer.
1372
+ def build_turn_context(task, profile, state)
1373
+ {
1374
+ chat_id: task.session_id,
1375
+ agent_id: profile.id,
1376
+ tenant: state.tenant, # already = command_tenant || session_id (memory_tenant)
1377
+ store_id: profile.store_id,
1378
+ # current delegation depth (0 for a top-level turn). Carried in
1379
+ # the child command's payload by run_subagent; read here so the child's OWN
1380
+ # spawn_subagent tool sees depth+1 and the runtime cap holds down the chain.
1381
+ delegation_depth: delegation_depth(task)
1382
+ }
1383
+ end
1384
+
1385
+ # Delegation depth of THIS turn: the value run_subagent stamped in
1386
+ # the child command, or 0 for a top-level turn. Integer-coerced (JSON round-trip
1387
+ # of the persisted command may deliver a String).
1388
+ def delegation_depth(task)
1389
+ rebuild_command(task).payload["delegation_depth"].to_i
1390
+ end
1391
+
1392
+ # R2: environment (model/thinking) inherits as DEFAULT — the child's
1393
+ # explicit value wins; when absent, seed from the parent's RESOLVED selection.
1394
+ # Capacity fields are untouched (R1: the child profile is used as-is). Returns
1395
+ # the child profile unchanged when there is nothing to inherit.
1396
+ def inherit_environment(child_profile, parent_state)
1397
+ sel = parent_state.model_selection
1398
+ return child_profile if sel.nil?
1399
+
1400
+ model = child_profile.model || sel.model
1401
+ provider = child_profile.provider || sel.provider
1402
+ params = child_profile.params.dup # build stringified it => string keys
1403
+ inherited_thinking = (sel.params || {})[:thinking]
1404
+ params["thinking"] = inherited_thinking if !params.key?("thinking") && !inherited_thinking.nil?
1405
+
1406
+ return child_profile if model == child_profile.model &&
1407
+ provider == child_profile.provider &&
1408
+ params == child_profile.params
1409
+
1410
+ child_profile.with(model: model, provider: provider, params: params)
1411
+ end
1412
+
1413
+ # Single validation path for a delegation — shared by run_subagent
1414
+ # and the fan-out run_subagents. `task` is {agent, message} (string OR symbol
1415
+ # keys — the model's args arrive string-keyed). Returns a resolved plan
1416
+ # { agent:, profile:, message:, depth: } or { agent:, error: } (the agent name is
1417
+ # kept even on error so a fan-out result keeps its slot labeled).
1418
+ def plan_subagent(task, parent_state)
1419
+ agent = (task["agent"] || task[:agent]).to_s
1420
+ message = (task["message"] || task[:message]).to_s
1421
+
1422
+ # Gate = the PARENT's subagents allowlist (a capacity field — never inherited).
1423
+ allowed = Array(parent_state.profile.subagents).map(&:to_s)
1424
+ return { agent: agent, error: "agent '#{agent}' is not in this agent's subagents allowlist" } unless allowed.include?(agent)
1425
+
1426
+ # Runtime depth guard (belt-and-suspenders; SubagentGraph is the primary,
1427
+ # definition-time check). parent depth from its turn context, +1 for the child.
1428
+ cap = SubagentGraph.depth_cap
1429
+ depth = (parent_state.turn_context&.dig(:delegation_depth) || 0) + 1
1430
+ return { agent: agent, error: "subagent depth #{depth} exceeds cap #{cap}" } if depth > cap
1431
+
1432
+ profile = @profiles[agent]
1433
+ return { agent: agent, error: "agent '#{agent}' not configured" } if profile.nil?
1434
+
1435
+ { agent: agent, profile: inherit_environment(profile, parent_state), message: message, depth: depth }
1436
+ end
1437
+
1438
+ # Fan-out barrier: spawn ALL valid children first (non-blocking), THEN await ALL.
1439
+ # Spawning before awaiting is what makes them concurrent — each child spends its
1440
+ # time on the provider HTTP wait, and those waits overlap on the reactor. Invalid
1441
+ # plans keep their ordered slot as an { agent:, error: } result.
1442
+ def spawn_all_and_project(plans, parent_state)
1443
+ spawned = plans.map do |plan|
1444
+ next plan if plan[:error]
1445
+
1446
+ session_id, child_task = create_child(plan[:profile], plan[:message], plan[:depth], parent_state, async: false)
1447
+ spawn(child_task, profile: plan[:profile])
1448
+ { agent: plan[:agent], task_id: child_task.id, session_id: session_id }
1449
+ end
1450
+
1451
+ spawned.each { |s| @running[s[:task_id]]&.wait unless s[:error] }
1452
+ spawned.map do |s|
1453
+ next { agent: s[:agent], error: s[:error] } if s[:error]
1454
+
1455
+ # label each result with its agent so the model can tell the N apart.
1456
+ project_child_result(s[:task_id], s[:session_id], s[:agent], parent_state).merge(agent: s[:agent])
1457
+ end
1458
+ end
1459
+
1460
+ # Creates the isolated child session (linked to the parent) + child task.
1461
+ # Shared by the sync and async paths. Emits :subagent_started correlated to the
1462
+ # PARENT task. -> [child_session_id, child_task].
1463
+ def create_child(child_profile, message, depth, parent_state, async:, delegation_id: nil)
1464
+ child_session_id = "sub-#{SecureRandom.uuid}"
1465
+ @session_store.create(
1466
+ id: child_session_id,
1467
+ vars: { "parent_session_id" => parent_state.task.session_id,
1468
+ "parent_task_id" => parent_state.task.id, "delegation_depth" => depth }
1469
+ )
1470
+ payload = { "agent" => child_profile.id, "message" => message,
1471
+ "session_id" => child_session_id, "delegation_depth" => depth }
1472
+ # Async children carry their delegation id in the command payload so the
1473
+ # terminal hook (finalize_delegation) is O(1) — a normal turn has no marker and
1474
+ # skips the delegation store entirely (no per-turn O(n) scan).
1475
+ payload["delegation_id"] = delegation_id if delegation_id
1476
+ command = Insika::Command.build(:send_message, payload).to_h
1477
+ child_task = @task_store.create(command: command, session_id: child_session_id)
1478
+
1479
+ emit(:subagent_started,
1480
+ { agent: child_profile.id, parent_task_id: parent_state.task.id,
1481
+ child_session_id: child_session_id, depth: depth, async: async },
1482
+ task: parent_state.task)
1483
+ [child_session_id, child_task]
1484
+ end
1485
+
1486
+ # SYNC: spawns the child and AWAITS it on the parent's fiber, then
1487
+ # projects the terminal content. Direct `spawn` (not spawn_in_session): the
1488
+ # child session is brand-new, so there is no SessionActor contention — the child
1489
+ # is parented at turn_parent and the parent yields cooperatively on `wait`.
1490
+ def spawn_and_await_child(child_profile, message, depth, parent_state)
1491
+ child_session_id, child_task = create_child(child_profile, message, depth, parent_state, async: false)
1492
+ spawn(child_task, profile: child_profile)
1493
+ @running[child_task.id]&.wait
1494
+ project_child_result(child_task.id, child_session_id, child_profile.id, parent_state)
1495
+ end
1496
+
1497
+ # ASYNC: persists a Delegation, spawns the child NON-blocking, and
1498
+ # returns a dispatch ack immediately — the parent turn ends without waiting. The
1499
+ # child's terminal hook (finalize_delegation) delivers the result later, as a
1500
+ # NEW turn on the parent session.
1501
+ def dispatch_async_child(child_profile, message, depth, parent_state)
1502
+ delegation_id = SecureRandom.uuid
1503
+ child_session_id, child_task =
1504
+ create_child(child_profile, message, depth, parent_state, async: true, delegation_id: delegation_id)
1505
+ @delegation_store.create(
1506
+ id: delegation_id,
1507
+ parent_task_id: parent_state.task.id, parent_session_id: parent_state.task.session_id,
1508
+ parent_agent: parent_state.profile.id, child_agent: child_profile.id,
1509
+ child_task_id: child_task.id, child_session_id: child_session_id, depth: depth
1510
+ )
1511
+ spawn(child_task, profile: child_profile)
1512
+ { dispatched: child_task.id, agent: child_profile.id, session_id: child_session_id }
1513
+ end
1514
+
1515
+ # Child result = last `assistant` message of the child session (R3), or the
1516
+ # terminal error if the child turn failed. Same projection as the A2A edge.
1517
+ def project_child_result(child_task_id, child_session_id, agent_id, parent_state)
1518
+ task = @task_store.find(child_task_id)
1519
+ if task && task.status.to_s == "failed"
1520
+ err = child_terminal_error(task)
1521
+ emit(:subagent_completed,
1522
+ { agent: agent_id, child_session_id: child_session_id, state: "failed" },
1523
+ task: parent_state.task)
1524
+ return { error: "subagent '#{agent_id}' failed: #{err || 'unknown error'}" }
1525
+ end
1526
+
1527
+ emit(:subagent_completed,
1528
+ { agent: agent_id, child_session_id: child_session_id, state: "completed" },
1529
+ task: parent_state.task)
1530
+ { text: child_final_text(child_session_id).to_s, session_id: child_session_id }
1531
+ end
1532
+
1533
+ def child_final_text(child_session_id)
1534
+ session = @session_store.find(child_session_id)
1535
+ msg = session&.messages&.reverse&.find { |m| (m["role"] || m[:role]).to_s == "assistant" }
1536
+ msg && (msg["content"] || msg[:content])
1537
+ end
1538
+
1539
+ def child_terminal_error(task)
1540
+ exec = task.executions.last
1541
+ return nil unless exec && exec.outcome.to_s == "failed"
1542
+
1543
+ exec.error && (exec.error["message"] || exec.error[:message])
1544
+ end
1545
+
1546
+ # terminal hook: when a turn ends (success OR failure), if the
1547
+ # task is the child of an ASYNC delegation, capture its result and deliver it to
1548
+ # the parent. Fires for both a normal completion and a resumed one (recovery),
1549
+ # so it needs no live watcher fiber. No-op without a delegation_store or when the
1550
+ # task is not a delegation child. Best-effort: a failure here must NOT re-fail an
1551
+ # already-committed child turn — swallow (recovery re-drives from the record).
1552
+ def finalize_delegation(child_task)
1553
+ return unless @delegation_store
1554
+
1555
+ # Fresh read: the snapshot the hook passes predates transition(:completed/
1556
+ # :failed); the store is the single source of truth for terminal state + the
1557
+ # command marker.
1558
+ child_id = child_task.respond_to?(:id) ? child_task.id : child_task
1559
+ child = @task_store.find(child_id)
1560
+ # Cheap gate: only async delegation children carry a delegation_id in the
1561
+ # command payload — a normal turn skips the store entirely (no O(n) scan).
1562
+ deleg_id = child && rebuild_command(child).payload["delegation_id"]
1563
+ return unless deleg_id
1564
+
1565
+ deleg = @delegation_store.find(deleg_id)
1566
+ return if deleg.nil? || deleg.status == :delivered
1567
+
1568
+ if deleg.status == :dispatched
1569
+ if child.status.to_s == "failed"
1570
+ @delegation_store.mark_completed(deleg.id, error: child_terminal_error(child) || "unknown error")
1571
+ else
1572
+ @delegation_store.mark_completed(deleg.id, result: child_final_text(deleg.child_session_id).to_s)
1573
+ end
1574
+ end
1575
+ deliver_delegation(@delegation_store.find(deleg.id))
1576
+ rescue Insika::Error
1577
+ nil # best-effort: never re-fail a committed child turn (recovery re-drives).
1578
+ end
1579
+
1580
+ # Delivers a captured delegation as a NEW turn on the PARENT session. The claim
1581
+ # (completed -> delivered) makes this AT-MOST-ONCE: only the caller that wins the
1582
+ # transition spawns the delivery turn (the hook and recovery may both fire).
1583
+ # spawn_in_session routes through the parent's SessionActor FIFO — that is what
1584
+ # makes it a new turn "when idle" (queued behind any in-flight parent turn),
1585
+ # preserving role alternation + prompt cache.
1586
+ def deliver_delegation(deleg)
1587
+ return unless @delegation_store.claim_delivery(deleg.id)
1588
+
1589
+ profile = @profiles[deleg.parent_agent]
1590
+ # The parent agent vanished (deleted mid-flight): nothing to deliver to. The
1591
+ # record stays :delivered (claimed) so recovery does not loop on it.
1592
+ return if profile.nil?
1593
+
1594
+ message = format_delegation_message(deleg)
1595
+ command = Insika::Command.build(
1596
+ :send_message,
1597
+ # `origin: engine` — this "user" message is Insika writing to itself. Without
1598
+ # it a report reads a delegation result as something the customer said.
1599
+ { "agent" => deleg.parent_agent, "message" => message,
1600
+ "session_id" => deleg.parent_session_id, "delegation_id" => deleg.id,
1601
+ "origin" => Insika::MessageOrigin::ENGINE }
1602
+ ).to_h
1603
+ task = @task_store.create(command: command, session_id: deleg.parent_session_id)
1604
+ emit(:subagent_delivered,
1605
+ { delegation_id: deleg.id, agent: deleg.child_agent,
1606
+ child_session_id: deleg.child_session_id, state: deleg.error ? "failed" : "completed" },
1607
+ task: task)
1608
+ spawn_in_session(task, profile: profile)
1609
+ end
1610
+
1611
+ # The synthetic message the parent turn receives. A clear, self-describing note
1612
+ # so the parent agent knows a delegated subtask returned (and can relay it).
1613
+ def format_delegation_message(deleg)
1614
+ if deleg.error
1615
+ "[subagent:#{deleg.child_agent}] delegated task FAILED: #{deleg.error}"
1616
+ else
1617
+ "[subagent:#{deleg.child_agent}] delegated task completed. Result:\n\n#{deleg.result}"
1618
+ end
1619
+ end
1620
+
1621
+ # The real Request: command (for the WorkflowAllowlist), context
1622
+ # (Context before Policy), candidate_tools (registry Entries, UNfiltered) and
1623
+ # candidate_skills (from the CATALOG).
1624
+ def policy_request(profile, task, state)
1625
+ Insika::Policy::PolicyRequest.new(
1626
+ profile: profile,
1627
+ command: rebuild_command(task),
1628
+ context: state.context,
1629
+ candidate_tools: @tool_registry.entries,
1630
+ # agent: so a specialized skill reaches the policy as the agent's own version
1631
+ # (same name, its body) instead of the shared one it overrides.
1632
+ candidate_skills: @skill_catalog.effective(profile.skills, agent: profile.id)
1633
+ )
1634
+ end
1635
+
1636
+ # The Task persists the Command as a Hash; the WorkflowAllowlist needs
1637
+ # a Command with #type (Symbol) and #payload.: the SINGLE point that
1638
+ # reconciles the string||symbol keys of the persisted command — payload/meta
1639
+ # keys are stringified ONCE here, so every reader (command_type/workflow_name/
1640
+ # extract_message/command_history/command_tenant) works with string keys.
1641
+ def rebuild_command(task)
1642
+ cmd = task.command
1643
+ Insika::Command.new(
1644
+ type: (cmd["type"] || cmd[:type]).to_s.to_sym,
1645
+ payload: normalize_keys(cmd["payload"] || cmd[:payload]),
1646
+ meta: normalize_keys(cmd["meta"] || cmd[:meta])
1647
+ )
1648
+ end
1649
+
1650
+ # Shallow key stringification (nil -> {}). The persisted command comes
1651
+ # string-keyed from the TaskStore; this only matters for a symbol-keyed
1652
+ # command that bypassed it (fakes/tests).
1653
+ def normalize_keys(hash)
1654
+ (hash || {}).each_with_object({}) { |(k, v), acc| acc[k.to_s] = v }
1655
+ end
1656
+
1657
+ # Stage-3-tail tool assembly — delegated to ToolAssembly. Kept as
1658
+ # thin private methods so the existing spec contract (executor.send(:...))
1659
+ # stays intact and run_pipeline reads unchanged.
1660
+ def resolve_capabilities(profile, context) = @tool_assembly.resolve_capabilities(profile, context)
1661
+ def assemble_tool_instances(allowed, state) = @tool_assembly.assemble_tool_instances(allowed, state)
1662
+
1663
+ def wrap_tools(tools, state, skip_side_effects = [])
1664
+ @tool_assembly.wrap_tools(tools, state, skip_side_effects)
1665
+ end
1666
+
1667
+ # Stage boundary: drains the mailbox and, if the operator requested a pause,
1668
+ # SUSPENDS the turn in :paused until :resume. Cooperative — never in the
1669
+ # middle of an operation. NOT called at stage 8 (forbidden window).
1670
+ # :cancel during the wait becomes CancelledError (top single capture);
1671
+ # :timeout becomes TimeoutError. The turn's initial checkpoint was already
1672
+ # written, so a kill -9 in :paused is resumable.
1673
+ def drain_and_maybe_suspend(task, actor)
1674
+ actor.drain!
1675
+ return unless actor.pause_requested?
1676
+
1677
+ @task_store.transition(task.id, to: :paused)
1678
+ emit(:task_paused, { task_id: task.id }, task: task)
1679
+ actor.await(reason: :paused) # blocks until :resume (or raises on :cancel/:timeout)
1680
+ @task_store.transition(task.id, to: :running)
1681
+ emit(:task_resumed, { task_id: task.id }, task: task)
1682
+ end
1683
+
1684
+ # Checkpoint of the turn's initial state (see the call in run_pipeline). Only
1685
+ # on the 1st turn of a new task: on resume the turn checkpoint already exists
1686
+ # (does not re-save — the CheckpointStore's monotonicity would raise). It emits
1687
+ # NO event (:checkpoint_created is stage 8's only) and touches no side-effects.
1688
+ def save_initial_checkpoint(task, profile, state)
1689
+ return unless @checkpoint_store.find(task.id, turn: state.turn).nil?
1690
+
1691
+ @checkpoint_store.save(Insika::Checkpoint.new(
1692
+ task_id: task.id, turn: state.turn, session_id: task.session_id,
1693
+ agent_id: profile.id, messages: flatten_history(state.context.history),
1694
+ completed_side_effects: [], created_at: Time.now.utc.iso8601
1695
+ ))
1696
+ end
1697
+
1698
+ # GRACEFUL halt: a Middleware short-circuited with a safe reply.
1699
+ # The turn COMPLETES — same stages 8-9 as a normal turn — but the "assistant
1700
+ # content" is the guardrail's safe response, produced with ZERO LLM calls. The
1701
+ # order mirrors a real turn so both the /v1/responses consumer (which reads the
1702
+ # text off :content deltas) and the Studio viewer render it: audit -> safe text
1703
+ # -> persist -> terminal.
1704
+ def complete_with_halt(task, profile, state)
1705
+ content = state.halt_response.to_s
1706
+ state.response_content = content
1707
+
1708
+ if (block = state.guardrail_block)
1709
+ emit(:guardrail_blocked, {
1710
+ task_id: task.id, category: block[:category], source: block[:source],
1711
+ action: block[:action], detail: block[:detail]
1712
+ }, task: task)
1713
+ end
1714
+ emit(:content, { delta: content }, task: task) unless content.empty?
1715
+ # An EDGE-blocked turn (rate limit / token ceiling) completes but
1716
+ # stays OUT of the session history: a flood at the wall must not bloat the
1717
+ # session nor evict real conversation from the context budget — the
1718
+ # :guardrail_blocked event is the audit trail. Content-guardrail blocks
1719
+ # keep persisting (the refusal is part of the conversation).
1720
+ # The reply is the guardrail's, produced with zero LLM calls — so it is NOT the
1721
+ # agent talking, and a report that counts it as the agent repeating itself is
1722
+ # reading the engine's own canned text (the `safe_reply` finding exists exactly
1723
+ # because that text is otherwise indistinguishable in the transcript).
1724
+ persist_turn(task, profile, state, content, reply_origin: MessageOrigin::ENGINE,
1725
+ session: state.guardrail_block&.[](:source) != "edge")
1726
+ emit(:task_completed, { task_id: task.id, content: content, usage: state.usage }, task: task)
1727
+ end
1728
+
1729
+ # Emits one :guardrail_flagged per flag the OutputValidator appended in
1730
+ # after_task (audit only — the turn already completed). Reads a plain Array off
1731
+ # the state, keeping the Executor decoupled from Safety.
1732
+ def emit_guardrail_flags(task, state)
1733
+ return unless state.respond_to?(:guardrail_flags)
1734
+
1735
+ Array(state.guardrail_flags).each do |flag|
1736
+ emit(:guardrail_flagged, {
1737
+ task_id: task.id, category: flag[:category], source: flag[:source], detail: flag[:detail]
1738
+ }, task: task)
1739
+ end
1740
+ end
1741
+
1742
+ # Stage 8: FIXED order checkpoint -> session -> task. If it crashes
1743
+ # between writes, the worst case is a new checkpoint with the task :running ->
1744
+ # Recovery re-executes the already-saved turn (safe thanks to the side-effect
1745
+ # recording).
1746
+ #
1747
+ # CHECKPOINT vs SESSION — the two stores DIVERGE by design (R2c):
1748
+ # · Checkpoint.messages = flatten_history(context.history) + new_messages,
1749
+ # i.e. "what the model actually SAW this turn" AFTER budget eviction
1750
+ # (context.history is the post-budget assembly). It is the deterministic
1751
+ # replay tape for Recovery: resuming from it reproduces the exact prompt.
1752
+ # It is per-task and pruned to keep: 1 (only the latest matters for resume).
1753
+ # · SessionStore = the INTEGRAL source of truth: append-only, never evicted,
1754
+ # the full human-readable transcript (viewer, audit, next turns' raw input).
1755
+ # So a long session legitimately has a Checkpoint SHORTER than the Session:
1756
+ # that is not drift to reconcile — it is the point. Do NOT "fix" the checkpoint
1757
+ # to carry the full history (it would defeat the budget) nor evict the session.
1758
+ def persist_turn(task, profile, state, content, session: true, reply_origin: nil)
1759
+ new_messages = turn_transcript(state, content, origin: command_origin(task), reply_origin: reply_origin)
1760
+ transcript = flatten_history(state.context.history) + new_messages
1761
+
1762
+ @checkpoint_store.save(Insika::Checkpoint.new(
1763
+ task_id: task.id, turn: state.turn + 1, session_id: task.session_id,
1764
+ agent_id: profile.id, messages: transcript,
1765
+ completed_side_effects: [], created_at: Time.now.utc.iso8601
1766
+ ))
1767
+
1768
+ # session only when the turn is from a persisted session; one-shot/history
1769
+ # do not persist TO THE SESSION (but always checkpoint). `session: false` is
1770
+ # the edge-blocked halt (see complete_with_halt).
1771
+ @session_store.append_messages(task.session_id, new_messages) if session && task.session_id
1772
+
1773
+ # finish_execution (closes the Execution) BEFORE transition(:completed) —
1774
+ # transition without error: does not close, so the finish is needed here.
1775
+ @task_store.finish_execution(task.id, outcome: :completed)
1776
+ @task_store.transition(task.id, to: :completed)
1777
+ # prune is best-effort cleanup: a failure here must NOT re-fail an
1778
+ # already-committed turn (the task is already :completed and durable).
1779
+ # Swallow.
1780
+ begin
1781
+ @checkpoint_store.prune(task.id, keep: 1)
1782
+ rescue Insika::StoreError
1783
+ nil
1784
+ end
1785
+
1786
+ emit(:checkpoint_created, { task_id: task.id, turn: state.turn + 1 }, task: task)
1787
+
1788
+ # if this completed turn is an ASYNC delegation child,
1789
+ # deliver its result to the parent as a NEW turn. No-op for a normal turn
1790
+ # (not a delegation child) or without a delegation_store.
1791
+ finalize_delegation(task)
1792
+
1793
+ # if this turn CAME IN through a Shape B channel, its answer
1794
+ # has to travel out of band. Same terminal hook, next door to the delegation
1795
+ # one, for the same reason: it fires for a fresh turn and a recovered one.
1796
+ finalize_channel_delivery(task, content)
1797
+ end
1798
+
1799
+ # Records the answer in the outbox and dispatches it. The discriminator is the
1800
+ # turn's TRANSPORT (`channel:<id>` on the persisted command), not the session:
1801
+ # a session belongs to the channel forever, but a message an operator types into
1802
+ # the Studio playground against that same session must not reach the customer.
1803
+ # Human handoff is not a product feature (``), and it would be a
1804
+ # surprising way to acquire one.
1805
+ #
1806
+ # The consequence, stated rather than discovered later: a turn the ENGINE
1807
+ # spawned on a channel session — an async delegation result delivered as a new
1808
+ # turn — is not delivered either. There is no consumer for that yet; when there
1809
+ # is, the fix is to give those turns the channel transport, not to widen this.
1810
+ #
1811
+ # Best-effort: the turn is already committed and durable, and a delivery problem
1812
+ # must never re-fail it.
1813
+ def finalize_channel_delivery(task, content)
1814
+ return unless @channel_delivery
1815
+
1816
+ channel_id = channel_transport(task)
1817
+ return unless channel_id
1818
+
1819
+ delivery = @channel_delivery.record(task: task, channel_id: channel_id, content: content)
1820
+ return unless delivery
1821
+
1822
+ dispatch_delivery(delivery.id)
1823
+ rescue Insika::Error
1824
+ nil
1825
+ end
1826
+
1827
+ # The POST goes out on the SUPERVISOR, never on the turn's fiber: a bounded
1828
+ # retry against a third party would otherwise hold the session's FIFO — the
1829
+ # customer's next message would wait on their previous answer's delivery.
1830
+ # Non-serving (boot sweep, specs) delivers inline, where waiting is what the
1831
+ # caller wants.
1832
+ def dispatch_delivery(delivery_id)
1833
+ return @channel_delivery.deliver(delivery_id) unless @supervised
1834
+
1835
+ turn_parent.async do |t|
1836
+ t.annotate("outbox:#{delivery_id}")
1837
+ @channel_delivery.deliver(delivery_id)
1838
+ end
1839
+ end
1840
+
1841
+ # `channel:<id>` -> "<id>"; anything else -> nil. The transport is persisted with
1842
+ # the command, so a turn resumed after a crash still knows where it came from.
1843
+ def channel_transport(task)
1844
+ transport = rebuild_command(task).meta["transport"].to_s
1845
+ transport.start_with?("channel:") ? transport.delete_prefix("channel:") : nil
1846
+ end
1847
+
1848
+ # Truncation cap for a persisted `role: tool` content (R1): the transcript
1849
+ # keeps the loop coherent; the FULL result lives in the ToolTraceStore (viewer).
1850
+ TOOL_CONTENT_CAP = 4_000
1851
+
1852
+ # The turn's messages in the ADDITIVE string-keyed format (R1). Prefers the
1853
+ # real chat transcript (`chat.messages.drop(baseline)`) so tool calls/results
1854
+ # survive between turns; falls back to the {user, assistant} pair when the chat
1855
+ # did not record the turn (workflow, graceful halt, or the specs' FakeChat).
1856
+ # The final assistant text is the REDACTED `content` (output_filter),
1857
+ # never the raw text the gem stored.
1858
+ # `origin` (MessageOrigin) travels on the turn's Command and is stamped on the
1859
+ # message it describes: the INCOMING one. It is absent for an ordinary turn, and
1860
+ # present when the engine wrote the text itself (an async delegation result
1861
+ # delivered as a new turn) or when the consumer declared that it composed it.
1862
+ # The reply's origin is not a parameter — the model wrote it, unless a guardrail
1863
+ # short-circuited, which `complete_with_halt` says explicitly.
1864
+ def turn_transcript(state, content, origin: nil, reply_origin: nil)
1865
+ recorded = recorded_turn_messages(state)
1866
+ if recorded.empty?
1867
+ return [MessageOrigin.stamp({ "role" => "user", "content" => state.message.to_s }, origin),
1868
+ MessageOrigin.stamp({ "role" => "assistant", "content" => content.to_s }, reply_origin)]
1869
+ end
1870
+
1871
+ recorded[0] = MessageOrigin.stamp(recorded[0], origin) if recorded[0]["role"] == "user"
1872
+ recorded[-1] = recorded[-1].merge("content" => content.to_s) if recorded.last["role"] == "assistant"
1873
+ recorded
1874
+ end
1875
+
1876
+ # Slices the chat's messages added DURING this turn and serializes them.
1877
+ # [] when there is no recorded transcript (baseline nil / no #messages).
1878
+ def recorded_turn_messages(state)
1879
+ chat = state.chat
1880
+ baseline = state.chat_baseline
1881
+ return [] unless chat && baseline && chat.respond_to?(:messages)
1882
+
1883
+ Array(chat.messages).drop(baseline).filter_map { |m| serialize_chat_message(m) }
1884
+ end
1885
+
1886
+ # A RubyLLM::Message (duck-typed) -> string-keyed Hash. Assistant carries
1887
+ # "tool_calls" only when present; tool carries "tool_call_id" + a clipped content.
1888
+ def serialize_chat_message(msg)
1889
+ role = msg_field(msg, :role).to_s
1890
+ content = msg_field(msg, :content).to_s
1891
+ case role
1892
+ when "assistant"
1893
+ calls = serialize_tool_calls(msg_field(msg, :tool_calls))
1894
+ h = { "role" => "assistant", "content" => content }
1895
+ h["tool_calls"] = calls unless calls.empty?
1896
+ h
1897
+ when "tool"
1898
+ { "role" => "tool", "tool_call_id" => msg_field(msg, :tool_call_id).to_s,
1899
+ "content" => clip_tool_content(content) }
1900
+ else
1901
+ { "role" => role, "content" => content }
1902
+ end
1903
+ end
1904
+
1905
+ # RubyLLM keeps tool_calls as {id => ToolCall}; tolerate an Array too. -> [{id,name,arguments}].
1906
+ def serialize_tool_calls(tool_calls)
1907
+ return [] if tool_calls.nil?
1908
+
1909
+ list = tool_calls.is_a?(Hash) ? tool_calls.values : Array(tool_calls)
1910
+ list.filter_map do |tc|
1911
+ next nil unless tc
1912
+
1913
+ { "id" => msg_field(tc, :id).to_s, "name" => msg_field(tc, :name).to_s,
1914
+ "arguments" => msg_field(tc, :arguments) || {} }
1915
+ end
1916
+ end
1917
+
1918
+ # Reads a field off a Message/ToolCall (method) OR a Hash (sym|string key).
1919
+ def msg_field(obj, key)
1920
+ return obj.public_send(key) if obj.respond_to?(key)
1921
+
1922
+ obj[key] || obj[key.to_s] if obj.respond_to?(:[])
1923
+ end
1924
+
1925
+ def clip_tool_content(str)
1926
+ return str if str.length <= TOOL_CONTENT_CAP
1927
+
1928
+ "#{str[0, TOOL_CONTENT_CAP]}…(truncated — full result in the viewer)"
1929
+ end
1930
+
1931
+ # context.history may carry "eviction units" (an assistant+tool_results cycle
1932
+ # grouped as one Array by the Session provider, R1). Checkpoints store a
1933
+ # FLAT list — the provider regroups on read. Flatten one level; message Hashes
1934
+ # are untouched.
1935
+ def flatten_history(history) = Array(history).flatten(1)
1936
+
1937
+ # Stage 6 (factory): the ONLY point that touches the gem. lazy require,
1938
+ # confined — not covered by unit (factory line). It also loads the system
1939
+ # builtins (load_skill/tool_search/remember) that the ChatBuilder assembles at
1940
+ # stage 5 — lazy, so the core installs without ruby_llm.
1941
+ def create_chat(profile, state)
1942
+ require "ruby_llm"
1943
+ require_relative "tools/load_skill"
1944
+ require_relative "tools/tool_search"
1945
+ require_relative "tools/remember"
1946
+ require_relative "tools/subagent"
1947
+ require_relative "tools/subagents"
1948
+ require_relative "tools/stuck_signal"
1949
+ # v2 resolution: Chat pin > Agent model > platform default, model_policy
1950
+ # enforced, fallback chain resolved. Kept on the state for telemetry (usage).
1951
+ selection = @model_resolver.resolve(profile: profile, session: state.session)
1952
+ state.model_selection = selection
1953
+ build_chat(selection, selection)
1954
+ end
1955
+
1956
+ # The gem boundary: one chat for a model selection (the resolved primary or
1957
+ # a WS3 fallback node). The primary's generation params apply to the whole
1958
+ # chain (params_source: ModelSelection#apply_params).
1959
+ def build_chat(selection, params_source)
1960
+ model = selection.respond_to?(:model) ? selection.model : selection[:model]
1961
+ provider = selection.respond_to?(:provider) ? selection.provider : selection[:provider]
1962
+ chat = (@llm || RubyLLM).chat(
1963
+ model: model,
1964
+ provider: provider,
1965
+ assume_model_exists: !provider.nil?
1966
+ )
1967
+ params_source.apply_params(chat) # temperature/max_tokens/thinking (per-agent)
1968
+ chat
1969
+ end
1970
+
1971
+ # Single emitter: an Event with meta and a monotonic seq per task. @seqs is not
1972
+ # cleared at the end of the task — the resume (new Execution) continues the
1973
+ # numbering (reliable replay). A task WITH a tenant (WS1) tags every event it
1974
+ # emits — the tenant-scoped /v1/events subscription filters on it (a control
1975
+ # event without a task has no tenant and never matches a tenant stream);
1976
+ # absent tenant -> the meta is byte-identical to before.
1977
+ def emit(type, data, task:)
1978
+ meta = { task_id: task.id, session_id: task.session_id,
1979
+ seq: (@seqs[task.id] += 1), at: Time.now.utc.iso8601 }
1980
+ tenant = task_tenant(task)
1981
+ meta[:tenant] = tenant unless tenant.nil?
1982
+ @event_stream.emit(Insika::Event.new(type: type, data: data, meta: meta))
1983
+ end
1984
+
1985
+ # The tenant stamped on the task's command (WS1), nil when the request was
1986
+ # operator-made. Cheap read on the persisted command hash — never rebuilds.
1987
+ def task_tenant(task)
1988
+ command = task.respond_to?(:command) ? task.command : nil
1989
+ return nil unless command.is_a?(Hash)
1990
+
1991
+ meta = command["meta"] || command[:meta] || {}
1992
+ meta["tenant"] || meta[:tenant]
1993
+ end
1994
+ end
1995
+ end