insika 0.0.1 → 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (260) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +295 -0
  3. data/LICENSE +21 -0
  4. data/README.md +136 -2
  5. data/bin/insika +351 -0
  6. data/docs/AGENTS.md +494 -0
  7. data/docs/ARCHITECTURE.md +333 -0
  8. data/docs/BENCHMARK.md +114 -0
  9. data/docs/CHANNELS.md +453 -0
  10. data/docs/CONTEXT.md +100 -0
  11. data/docs/DEPLOY.md +334 -0
  12. data/docs/EMBEDDING.md +194 -0
  13. data/docs/EVALS.md +273 -0
  14. data/docs/LOADTEST.md +231 -0
  15. data/docs/OBSERVABILITY.md +365 -0
  16. data/docs/PLUGINS.md +211 -0
  17. data/docs/REFINEMENT.md +477 -0
  18. data/docs/RELEASING.md +70 -0
  19. data/docs/RUNNING-LOCAL.md +153 -0
  20. data/docs/SANDBOX.md +114 -0
  21. data/docs/SECURITY.md +362 -0
  22. data/docs/SKILLS.md +98 -0
  23. data/docs/TOOLS.md +302 -0
  24. data/docs/WHY.md +137 -0
  25. data/docs/WORKFLOWS.md +225 -0
  26. data/docs/build.md +14 -0
  27. data/docs/index.md +68 -0
  28. data/docs/onboarding/start.md +126 -0
  29. data/docs/operate.md +12 -0
  30. data/docs/ship.md +10 -0
  31. data/docs/understand.md +10 -0
  32. data/lib/insika/agent_file_store.rb +125 -0
  33. data/lib/insika/agent_profile.rb +188 -0
  34. data/lib/insika/allowlist.rb +28 -0
  35. data/lib/insika/baseline_store.rb +74 -0
  36. data/lib/insika/capability/resolved_tool.rb +34 -0
  37. data/lib/insika/capability_registry.rb +112 -0
  38. data/lib/insika/channel_delivery.rb +150 -0
  39. data/lib/insika/channel_registry.rb +30 -0
  40. data/lib/insika/channels/relay.rb +178 -0
  41. data/lib/insika/channels/web/widget.js +283 -0
  42. data/lib/insika/channels/web.rb +211 -0
  43. data/lib/insika/chat_builder.rb +254 -0
  44. data/lib/insika/checkpoint.rb +13 -0
  45. data/lib/insika/checkpoint_store.rb +153 -0
  46. data/lib/insika/coercion.rb +50 -0
  47. data/lib/insika/command.rb +32 -0
  48. data/lib/insika/command_bus.rb +39 -0
  49. data/lib/insika/commands/agent_payload.rb +41 -0
  50. data/lib/insika/commands/approve_action.rb +46 -0
  51. data/lib/insika/commands/cancel_task.rb +33 -0
  52. data/lib/insika/commands/create_agent.rb +54 -0
  53. data/lib/insika/commands/create_session.rb +67 -0
  54. data/lib/insika/commands/delete_agent.rb +33 -0
  55. data/lib/insika/commands/delete_agent_file.rb +50 -0
  56. data/lib/insika/commands/delete_data_tool.rb +33 -0
  57. data/lib/insika/commands/delete_llm_provider.rb +36 -0
  58. data/lib/insika/commands/delete_mcp.rb +30 -0
  59. data/lib/insika/commands/delete_system_file.rb +29 -0
  60. data/lib/insika/commands/gate_refinement.rb +245 -0
  61. data/lib/insika/commands/import_mcp_tools.rb +48 -0
  62. data/lib/insika/commands/import_tools.rb +81 -0
  63. data/lib/insika/commands/memory_add_note.rb +32 -0
  64. data/lib/insika/commands/memory_forget_fact.rb +32 -0
  65. data/lib/insika/commands/memory_put_fact.rb +35 -0
  66. data/lib/insika/commands/pause_task.rb +29 -0
  67. data/lib/insika/commands/resolve_refinement.rb +126 -0
  68. data/lib/insika/commands/restore_agent_file.rb +36 -0
  69. data/lib/insika/commands/restore_data_tool.rb +34 -0
  70. data/lib/insika/commands/restore_system_file.rb +31 -0
  71. data/lib/insika/commands/resume_task.rb +85 -0
  72. data/lib/insika/commands/run_refinement.rb +133 -0
  73. data/lib/insika/commands/send_message.rb +150 -0
  74. data/lib/insika/commands/set_agent_tools.rb +39 -0
  75. data/lib/insika/commands/set_skill_agents.rb +71 -0
  76. data/lib/insika/commands/trigger_workflow.rb +80 -0
  77. data/lib/insika/commands/update_agent.rb +49 -0
  78. data/lib/insika/commands/update_settings.rb +33 -0
  79. data/lib/insika/commands/upsert_llm_provider.rb +34 -0
  80. data/lib/insika/commands/upsert_mcp.rb +32 -0
  81. data/lib/insika/commands/write_agent_file.rb +57 -0
  82. data/lib/insika/commands/write_data_tool.rb +43 -0
  83. data/lib/insika/commands/write_golden.rb +58 -0
  84. data/lib/insika/commands/write_skill.rb +50 -0
  85. data/lib/insika/commands/write_system_file.rb +31 -0
  86. data/lib/insika/config_store.rb +85 -0
  87. data/lib/insika/context/builder.rb +166 -0
  88. data/lib/insika/context/catalog_provider.rb +23 -0
  89. data/lib/insika/context/fragment.rb +19 -0
  90. data/lib/insika/context/priority.rb +29 -0
  91. data/lib/insika/context/provider.rb +19 -0
  92. data/lib/insika/context/providers/memory.rb +60 -0
  93. data/lib/insika/context/providers/prompt.rb +105 -0
  94. data/lib/insika/context/providers/request.rb +32 -0
  95. data/lib/insika/context/providers/session.rb +108 -0
  96. data/lib/insika/context/providers/skill.rb +20 -0
  97. data/lib/insika/context/providers/tool_search.rb +20 -0
  98. data/lib/insika/delegation_store.rb +153 -0
  99. data/lib/insika/doctor.rb +294 -0
  100. data/lib/insika/dsl/definition.rb +55 -0
  101. data/lib/insika/dsl/runtime.rb +379 -0
  102. data/lib/insika/dsl/server_boot.rb +97 -0
  103. data/lib/insika/dsl/system.rb +93 -0
  104. data/lib/insika/dsl/workflow_adapter.rb +59 -0
  105. data/lib/insika/dsl.rb +307 -0
  106. data/lib/insika/edge_limiter.rb +130 -0
  107. data/lib/insika/egress_guard.rb +75 -0
  108. data/lib/insika/env_schema.rb +246 -0
  109. data/lib/insika/errors.rb +145 -0
  110. data/lib/insika/evals/assertions.rb +247 -0
  111. data/lib/insika/evals/baseline.rb +69 -0
  112. data/lib/insika/evals/golden.rb +172 -0
  113. data/lib/insika/evals/judge.rb +225 -0
  114. data/lib/insika/evals/pairwise.rb +178 -0
  115. data/lib/insika/evals/report.rb +115 -0
  116. data/lib/insika/evals/runner.rb +141 -0
  117. data/lib/insika/evals/transport.rb +178 -0
  118. data/lib/insika/event.rb +18 -0
  119. data/lib/insika/event_stream.rb +114 -0
  120. data/lib/insika/executor.rb +1680 -0
  121. data/lib/insika/frontmatter.rb +42 -0
  122. data/lib/insika/golden_store.rb +145 -0
  123. data/lib/insika/hooks.rb +48 -0
  124. data/lib/insika/http_client.rb +63 -0
  125. data/lib/insika/inbound_log.rb +84 -0
  126. data/lib/insika/llm_configurator.rb +99 -0
  127. data/lib/insika/llm_provider_store.rb +83 -0
  128. data/lib/insika/mcp_http_client.rb +67 -0
  129. data/lib/insika/mcp_store.rb +115 -0
  130. data/lib/insika/mcp_tool_ingestor.rb +143 -0
  131. data/lib/insika/memory_store.rb +93 -0
  132. data/lib/insika/message_origin.rb +76 -0
  133. data/lib/insika/middleware.rb +36 -0
  134. data/lib/insika/model_policy.rb +52 -0
  135. data/lib/insika/model_resolver.rb +176 -0
  136. data/lib/insika/model_selection.rb +114 -0
  137. data/lib/insika/onboarding.rb +208 -0
  138. data/lib/insika/outbox_store.rb +166 -0
  139. data/lib/insika/overlay_tool_registry.rb +103 -0
  140. data/lib/insika/pack.rb +102 -0
  141. data/lib/insika/pack_importer.rb +121 -0
  142. data/lib/insika/pending_action_store.rb +120 -0
  143. data/lib/insika/plugin/loader.rb +356 -0
  144. data/lib/insika/plugin.rb +35 -0
  145. data/lib/insika/policy/engine.rb +83 -0
  146. data/lib/insika/policy/policy.rb +120 -0
  147. data/lib/insika/policy_registry.rb +23 -0
  148. data/lib/insika/profile_source.rb +137 -0
  149. data/lib/insika/prompt_catalog.rb +61 -0
  150. data/lib/insika/queue_policy.rb +167 -0
  151. data/lib/insika/recovery.rb +127 -0
  152. data/lib/insika/refinement/candidate.rb +159 -0
  153. data/lib/insika/refinement/evidence_collector.rb +371 -0
  154. data/lib/insika/refinement/gate.rb +234 -0
  155. data/lib/insika/refinement/panel.rb +222 -0
  156. data/lib/insika/refinement/proposer.rb +262 -0
  157. data/lib/insika/refinement_store.rb +295 -0
  158. data/lib/insika/registry.rb +59 -0
  159. data/lib/insika/safety/config.rb +109 -0
  160. data/lib/insika/safety/detectors.rb +176 -0
  161. data/lib/insika/safety/factory.rb +102 -0
  162. data/lib/insika/safety/input_guardrail.rb +87 -0
  163. data/lib/insika/safety/moderator.rb +86 -0
  164. data/lib/insika/safety/output_filter.rb +79 -0
  165. data/lib/insika/safety/output_validator.rb +101 -0
  166. data/lib/insika/safety/safe_responses.rb +47 -0
  167. data/lib/insika/sandbox/boundary.rb +93 -0
  168. data/lib/insika/sandbox/docker.rb +74 -0
  169. data/lib/insika/sandbox/local.rb +33 -0
  170. data/lib/insika/sandbox/runner.rb +80 -0
  171. data/lib/insika/sandbox.rb +85 -0
  172. data/lib/insika/schema_guard.rb +147 -0
  173. data/lib/insika/secret_masking.rb +34 -0
  174. data/lib/insika/server/a2a/agent_card.rb +27 -0
  175. data/lib/insika/server/a2a/app.rb +112 -0
  176. data/lib/insika/server/a2a/client.rb +101 -0
  177. data/lib/insika/server/a2a/errors.rb +32 -0
  178. data/lib/insika/server/a2a/http.rb +42 -0
  179. data/lib/insika/server/a2a/message.rb +27 -0
  180. data/lib/insika/server/a2a/protocol.rb +45 -0
  181. data/lib/insika/server/a2a/remotes.rb +25 -0
  182. data/lib/insika/server/a2a/task_projection.rb +40 -0
  183. data/lib/insika/server/admin_auth.rb +29 -0
  184. data/lib/insika/server/app.rb +850 -0
  185. data/lib/insika/server/boot.rb +119 -0
  186. data/lib/insika/server/rack_app.rb +110 -0
  187. data/lib/insika/server/responses.rb +155 -0
  188. data/lib/insika/server/sse_body.rb +96 -0
  189. data/lib/insika/session_actor.rb +162 -0
  190. data/lib/insika/session_store.rb +143 -0
  191. data/lib/insika/settings_store.rb +154 -0
  192. data/lib/insika/shutdown.rb +125 -0
  193. data/lib/insika/skill_catalog.rb +113 -0
  194. data/lib/insika/skill_store.rb +79 -0
  195. data/lib/insika/steer_injector.rb +110 -0
  196. data/lib/insika/store.rb +52 -0
  197. data/lib/insika/stores/memory.rb +123 -0
  198. data/lib/insika/stores/sqlite.rb +183 -0
  199. data/lib/insika/studio/app.rb +1571 -0
  200. data/lib/insika/studio/assets/dist/application.css +1 -0
  201. data/lib/insika/studio/assets/dist/application.js +69 -0
  202. data/lib/insika/studio/forms.rb +340 -0
  203. data/lib/insika/studio/nav_icons.rb +31 -0
  204. data/lib/insika/studio/views/_message.erb +44 -0
  205. data/lib/insika/studio/views/agent_detail.erb +285 -0
  206. data/lib/insika/studio/views/agents.erb +63 -0
  207. data/lib/insika/studio/views/approvals.erb +41 -0
  208. data/lib/insika/studio/views/chats.erb +34 -0
  209. data/lib/insika/studio/views/evals.erb +83 -0
  210. data/lib/insika/studio/views/home.erb +72 -0
  211. data/lib/insika/studio/views/layout.erb +94 -0
  212. data/lib/insika/studio/views/login.erb +17 -0
  213. data/lib/insika/studio/views/mcp.erb +91 -0
  214. data/lib/insika/studio/views/not_found.erb +5 -0
  215. data/lib/insika/studio/views/playground.erb +47 -0
  216. data/lib/insika/studio/views/refinement.erb +234 -0
  217. data/lib/insika/studio/views/session.erb +62 -0
  218. data/lib/insika/studio/views/settings.erb +173 -0
  219. data/lib/insika/studio/views/skills.erb +86 -0
  220. data/lib/insika/studio/views/system_files.erb +65 -0
  221. data/lib/insika/studio/views/task.erb +105 -0
  222. data/lib/insika/studio/views/tasks.erb +33 -0
  223. data/lib/insika/studio/views/tool_edit.erb +107 -0
  224. data/lib/insika/studio/views/tools.erb +89 -0
  225. data/lib/insika/subagent_graph.rb +96 -0
  226. data/lib/insika/system_file_store.rb +96 -0
  227. data/lib/insika/task_actor.rb +128 -0
  228. data/lib/insika/task_store.rb +250 -0
  229. data/lib/insika/telemetry/pricing.rb +104 -0
  230. data/lib/insika/telemetry/recorder.rb +228 -0
  231. data/lib/insika/telemetry.rb +127 -0
  232. data/lib/insika/testing/store_contract.rb +270 -0
  233. data/lib/insika/token_estimator.rb +16 -0
  234. data/lib/insika/tool_assembly.rb +140 -0
  235. data/lib/insika/tool_catalog.rb +89 -0
  236. data/lib/insika/tool_definition.rb +518 -0
  237. data/lib/insika/tool_envelope.rb +140 -0
  238. data/lib/insika/tool_manifest.rb +218 -0
  239. data/lib/insika/tool_registry.rb +21 -0
  240. data/lib/insika/tool_store.rb +135 -0
  241. data/lib/insika/tool_trace_store.rb +92 -0
  242. data/lib/insika/tools/a2a_remote.rb +48 -0
  243. data/lib/insika/tools/agent_enum.rb +68 -0
  244. data/lib/insika/tools/concurrency.rb +54 -0
  245. data/lib/insika/tools/data_defined_tool.rb +220 -0
  246. data/lib/insika/tools/load_skill.rb +41 -0
  247. data/lib/insika/tools/remember.rb +53 -0
  248. data/lib/insika/tools/subagent.rb +75 -0
  249. data/lib/insika/tools/subagents.rb +77 -0
  250. data/lib/insika/tools/tool_search.rb +94 -0
  251. data/lib/insika/turn_output.rb +139 -0
  252. data/lib/insika/turn_state.rb +158 -0
  253. data/lib/insika/turn_timing.rb +56 -0
  254. data/lib/insika/usage_ledger.rb +47 -0
  255. data/lib/insika/version.rb +3 -1
  256. data/lib/insika/wiring/graph.rb +198 -0
  257. data/lib/insika/workflow.rb +185 -0
  258. data/lib/insika/workflow_registry.rb +33 -0
  259. data/lib/insika.rb +203 -4
  260. metadata +395 -8
data/docs/AGENTS.md ADDED
@@ -0,0 +1,494 @@
1
+ ---
2
+ title: Agents
3
+ parent: Build an agent
4
+ nav_order: 1
5
+ permalink: /agents/
6
+ ---
7
+
8
+ # Agents
9
+
10
+ An **agent** is the unit you configure and address. It is an immutable
11
+ `AgentProfile` value object — an identity (the system prompt), a model, and a set
12
+ of layered access controls — stored as a row in SQLite. Everything about an agent
13
+ is **data**: created and edited at runtime through the DSL, the API, or the
14
+ Studio, and every edit is **hot** — no restart, no redeploy. An in-flight turn
15
+ keeps the profile it captured when it started; the next turn sees the new one.
16
+
17
+ > Smallest possible agent — see [`examples/hello-agent/`](https://github.com/guizaols/insika/tree/main/examples/hello-agent/):
18
+ >
19
+ > ```ruby
20
+ > agent = Insika.agent("assistant") do
21
+ > model "deepseek-chat"
22
+ > provider :deepseek
23
+ > instructions "You are a concise, friendly assistant."
24
+ > end
25
+ >
26
+ > puts agent.reply("hi") # => one turn, in-process
27
+ > ```
28
+
29
+ ## Three ways to create an agent
30
+
31
+ All three land on the **same** config-over-code import path — they differ only in
32
+ ergonomics, not in what they produce.
33
+
34
+ - **DSL** — `Insika.agent("id") { … }` builds an agent definition and imports it
35
+ into the durable store. Best for code-defined agents and examples. The DSL
36
+ auto-enables the tool- and skill-allowlist policies. `model` is optional — a nil
37
+ model resolves the platform `default_model` at turn start.
38
+ - **API** — `POST /v1/agents` with a definition (a "pack": an agent config plus its
39
+ prompt files, skills, and data-tools). The import is **idempotent** and
40
+ **authoritative** — what leaves the definition leaves the agent, so a
41
+ re-import that drops a tool or skill also removes it. `DELETE /v1/agents/:id`
42
+ removes an agent.
43
+ - **Studio** — create and edit an agent by hand in the control UI (Config /
44
+ Prompts / Skills / Memory / History tabs), backed by the same commands.
45
+
46
+ Creating an agent validates its id (required, must be unique) and its subagent
47
+ graph (cycle/depth — see [subagents](#delegation-subagents)) **before**
48
+ persisting anything.
49
+
50
+ ### Addressing an agent
51
+
52
+ Once it exists, an agent is addressable by id as the `model` on the
53
+ OpenAI-Responses-compatible endpoint:
54
+
55
+ ```jsonc
56
+ POST /v1/responses
57
+ { "model": "<agent-id>", "user": "<session-id>", "stream": true, "input": "hello" }
58
+ ```
59
+
60
+ `user` is the session id (see [Context](CONTEXT.md)); `stream: true` streams the
61
+ turn as Server-Sent Events.
62
+
63
+ An optional `"origin"` declares **who wrote the input**. Omit it and it means what
64
+ it always meant: a customer typed this. Send `"engine"` when your consumer composed
65
+ the message out of context blocks (`<memoria> …`) rather than relaying something a
66
+ person said — the transcript then records it, and a report stops counting your own
67
+ injected text as the customer repeating themselves. See
68
+ [Refinement](REFINEMENT.md#who-wrote-a-message).
69
+
70
+ ## The AgentProfile
71
+
72
+ A profile is built through one front door — `AgentProfile.build(id:, model: nil, …)`.
73
+ A prompt file is **text**. Passing a structured value where the markdown belongs —
74
+ a `{"content": …}` wrapper in a pack, or a store entry read and written back — is
75
+ rejected, not coerced: `to_s` on a Hash produces Ruby's `#inspect`, and a prompt made
76
+ of that is served on every turn while looking healthy.
77
+
78
+ Its free-form hashes (`params`, `guardrails`, `sandbox`, `metadata`, …) are
79
+ normalized to string keys once, at build time; no reader downstream does dual-key
80
+ lookups.
81
+
82
+ ### Default limits
83
+
84
+ ```ruby
85
+ DEFAULT_LIMITS = {
86
+ turn_timeout: 300, tool_timeout: 60, provider_timeout: 5,
87
+ context_budget: 8_000, max_tool_calls: 50, approval_timeout: 3_600,
88
+ tool_concurrency: 1
89
+ }
90
+ ```
91
+
92
+ `build` merges your overrides over these — you set only the deltas.
93
+
94
+ ### Why some limits are missing from that list
95
+
96
+ `chat_rate_limit`, `agent_token_ceiling`, `queue_mode`, `debounce_ms`,
97
+ `debounce_max_ms`, `steer_max_messages` and `steer_join` are real limits, and none
98
+ of them appears above. That is the rule, not an oversight:
99
+
100
+ > **A limit that has a platform-wide layer is absent from `DEFAULT_LIMITS`.**
101
+
102
+ Those limits resolve **agent → platform (Studio settings) → off**, and the agent
103
+ layer wins whenever the key is *present* — including when you set it to `nil` or
104
+ `0`, which means *off for this agent*, never *inherit the platform value*. A
105
+ default baked into every profile would make the key present on every agent, and
106
+ the platform layer would then apply to nobody.
107
+
108
+ So the two groups read differently on purpose:
109
+
110
+ | | In `DEFAULT_LIMITS` | Absent |
111
+ |---|---|---|
112
+ | Examples | `turn_timeout`, `tool_concurrency`, `context_budget` | `chat_rate_limit`, `queue_mode`, `debounce_ms` |
113
+ | Absent from your profile means | the constant above | ask the platform, then off |
114
+ | You set it to `nil`/`0` | back to the constant | **off**, platform ignored |
115
+
116
+ ### `tool_concurrency` — parallel tool calls
117
+
118
+ When the model asks for several tools in one step, they run **one at a time by
119
+ default**. Raise `tool_concurrency` and they run together, capped at that number
120
+ in flight:
121
+
122
+ ```ruby
123
+ limit :tool_concurrency, 4 # nil / 0 / 1 = serial (the default); N = at most N at once
124
+ ```
125
+
126
+ One number is both the switch and the cap. It pays off only when a turn issues
127
+ several **slow, independent** calls (data tools waiting on HTTP) — the wall-clock
128
+ becomes the slowest call instead of the sum. It buys nothing for fast in-process
129
+ tools, and it is the *model* that decides the fan-out, which is why the cap is not
130
+ optional: an uncapped batch of 15 data tools is 15 simultaneous requests to the
131
+ same backend.
132
+
133
+ > ⚠️ **It is silently disabled for any turn that has an approval-required tool.**
134
+ > The approval wait is one mailbox per task, so two tool calls suspended for an
135
+ > operator would deadlock — the turn runs serially instead. The downgrade is
136
+ > per *turn*, not per agent (an agent that lists approvals still gets parallelism
137
+ > on turns where none of the allowed tools require one), and it emits one
138
+ > `provider_warning` event so the lost speedup is never a mystery.
139
+
140
+ Two behaviours change once it is on — see [Tools](TOOLS.md#parallel-tool-calls):
141
+ `max_tool_calls` becomes approximate, and the transcript records tool results in
142
+ completion order.
143
+
144
+ ### `queue_mode` — when a message arrives while the agent is busy
145
+
146
+ A person on WhatsApp rarely writes one message. They write three:
147
+
148
+ ```
149
+ 14:02:31 "oi"
150
+ 14:02:33 "queria saber do pedido"
151
+ 14:02:36 "1234567"
152
+ ```
153
+
154
+ By default each one is a turn, and they run one at a time. So the agent answers
155
+ `"oi"` with a greeting the customer has already moved past, and may go looking for
156
+ an order before the number arrives three seconds later.
157
+
158
+ Which mode you want depends on **when** the message arrives:
159
+
160
+ | `queue_mode` | The message arrives… | What happens |
161
+ |---|---|---|
162
+ | `followup` (default) | any time | it waits its turn in the queue — today's behavior, named |
163
+ | `collect` | before the turn starts | the fragments merge into ONE turn |
164
+ | `steer` | while the turn is running tools | it is appended to the run in flight |
165
+ | `interrupt` | while a turn is running that is now **wrong** | that turn is abandoned; this message becomes its own turn |
166
+
167
+ #### `collect` — the fragments become one turn
168
+
169
+ `collect` merges the fragments that land **before the turn starts** into a single
170
+ turn:
171
+
172
+ ```ruby
173
+ limit :queue_mode, "collect" # "followup" (the default) = one turn per message
174
+ limit :debounce_ms, 2_000 # 0 (the default) = no waiting; N = the quiet window
175
+ limit :debounce_max_ms, 10_000 # ceiling on the total wait, so typing forever
176
+ # cannot postpone the answer forever
177
+ ```
178
+
179
+ With those settings the three fragments above become one turn carrying
180
+ `"oi\nqueria saber do pedido\n1234567"`, released 2 s after the last one.
181
+
182
+ All three follow the platform-layer rule above, with one extra rung on top:
183
+ **session vars → this agent's limits → the platform default (Studio, `queue.*`) →
184
+ off**. Pinning `queue_mode` in a session's vars is how an operator takes one
185
+ difficult conversation off `collect` without touching the agent.
186
+
187
+ > ⚠️ **Your caller has to know it was merged.** When the engine coalesces, only
188
+ > one of the three calls owns the reply; the other two answer
189
+ > `200 {"task_id": "…", "merged": true}` and stream nothing. A caller that
190
+ > delivers a `merged` response anyway sends the same answer to the customer three
191
+ > times.
192
+ >
193
+ > Because of that, `collect` works **only on surfaces that can report the
194
+ > verdict**: `POST /v1/messages?stream=false` and channel endpoints. On
195
+ > `/v1/responses` and on any open stream it is refused and the agent falls back to
196
+ > `followup` — the response body there is fixed by someone else's wire format and
197
+ > has nowhere to put the field.
198
+
199
+ Waiting happens inside the engine, not in your request: the POST is acked
200
+ immediately with its `task_id`. Debouncing costs one thing — a customer who sends
201
+ a *single* message still waits out the window before their turn starts, which is
202
+ why 2 s is a sane value and 10 s is not.
203
+
204
+ A merged fragment creates no task of its own, so the record that it arrived
205
+ separately lives in one event, emitted when the window closes:
206
+
207
+ ```jsonc
208
+ { "type": "turn_coalesced",
209
+ "data": { "task_id": "…", "merged": 3,
210
+ "arrivals": ["2026-08-07T14:02:31Z", "…:33Z", "…:36Z"] } }
211
+ ```
212
+
213
+ Times and counts, never content. That is what answers "the customer says they
214
+ sent the order number" without keeping a throwaway task per fragment.
215
+
216
+ #### `steer` — the message arrives while the turn is already running
217
+
218
+ `collect` only ever touches a turn that has **not started**. Once the agent is
219
+ running tools, the customer's next message has nowhere to go but the back of the
220
+ queue — so a correction that arrives three seconds into a fifteen-second run is
221
+ answered after the run that did not know about it.
222
+
223
+ `steer` appends it to the run in flight instead:
224
+
225
+ ```ruby
226
+ limit :queue_mode, "steer"
227
+ limit :steer_max_messages, 5 # how many one run may absorb; the 6th becomes its own turn
228
+ limit :steer_join, nil # nil = the raw text; a template frames it (below)
229
+ ```
230
+
231
+ Where it lands is the whole design: **at a tool-batch boundary, appended at the
232
+ tail.** After the last result of a batch and before the model's next step — never
233
+ between two tool results (Anthropic rejects that outright, OpenAI merely tolerates
234
+ it), and never rewriting a message already sent, which is what keeps the prompt
235
+ cache valid. So the model sees the correction on its very next step, with the full
236
+ context of what it has already found.
237
+
238
+ Reach for `steer` when turns are long **because they call tools**. If your turns
239
+ are one provider round-trip, `collect` is the mode that helps and `steer` has no
240
+ boundary to use.
241
+
242
+ Four cases where the run cannot absorb the message. In every one it becomes the
243
+ next turn on the session instead — `followup`, arrived at late, reported as
244
+ `turn_steer_released`:
245
+
246
+ | The run… | Why |
247
+ |---|---|
248
+ | never calls a tool | there is no batch boundary to append at |
249
+ | ends in [`halt_when`](TOOLS.md#halt_when-when-the-answer-is-already-out) | there is no next model step; the message would sit unanswered forever |
250
+ | is a [workflow](WORKFLOWS.md) | a workflow orchestrates the model itself and has no chat to append to |
251
+ | already absorbed `steer_max_messages` | the bound exists so a tail cannot grow without one |
252
+
253
+ `steer_join` is for an agent that needs the model to *know* the text arrived
254
+ mid-run. It must contain `%{message}`, or the config is refused:
255
+
256
+ ```ruby
257
+ limit :steer_join, "the customer just added: %{message}"
258
+ ```
259
+
260
+ Default `nil` appends exactly what the person typed — and a steered message is a
261
+ first-class transcript message, with no origin, because a person wrote it. The
262
+ Studio marks it `steered` in the transcript, derived from its position (a `user`
263
+ message right after a tool result); nothing else in the engine puts one there.
264
+
265
+ > ⚠️ **Same verdict rule as `collect`, different word.** The reply comes out of the
266
+ > turn the message joined, so the steered caller is told it does not own it:
267
+ > `200 {"task_id": "<the running turn>", "steered": true}`, no stream opened. Only
268
+ > surfaces that can carry that verdict may steer — `/v1/messages?stream=false` and
269
+ > channel endpoints, never `/v1/responses` or an open stream.
270
+ >
271
+ > One consequence worth knowing before you turn it on: when the run *cannot* absorb
272
+ > the message, the follow-up turn's reply belongs to no caller. It travels the event
273
+ > stream like any engine-initiated turn (an async subagent's delivery has the same
274
+ > shape). `steer` therefore fits a consumer that reads replies off the stream or off
275
+ > a channel delivery — not one that only reads its own POST response.
276
+ >
277
+ > A steered message also lives **in memory** until a boundary writes it to the
278
+ > transcript. A hard stop inside that window loses it; a merged fragment, by
279
+ > contrast, is persisted before the window opens.
280
+
281
+ #### `interrupt` — the turn in flight is answering the wrong question
282
+
283
+ `steer` assumes the run is still worth finishing. Sometimes it is not: the customer
284
+ says "não, esquece isso" while the agent is three tool calls into the wrong order.
285
+
286
+ ```ruby
287
+ limit :queue_mode, "interrupt" # no other knob: see below
288
+ ```
289
+
290
+ The running turn is abandoned and the new message becomes an **ordinary turn** — its
291
+ own `task_id`, its own reply. That is why `interrupt` needs no verdict field and works
292
+ on **every** surface, `/v1/responses` included: nothing joins anything.
293
+
294
+ What "abandoned" means, exactly:
295
+
296
+ - The turn terminates `:cancelled` and **publishes nothing**. The answer to the
297
+ question the customer already replaced never reaches them, and nothing is written to
298
+ the transcript — so what they read and what the session holds still agree.
299
+ - **A tool call in flight runs to completion** and its result is recorded on the
300
+ stream. The batch is one unit of work: cancelling the calls that had not started
301
+ would leave it half applied, and fabricating failure results would teach the model
302
+ that tools failed when they did not. The same boundary bounds `turn_timeout`.
303
+ - The next turn starts from the last **committed** state. The abandoned attempt is
304
+ visible to an *operator* (its `tool_call`/`tool_result` events and the trace), not to
305
+ the model — a half batch in the history would be an invalid prompt.
306
+
307
+ > **No grace knob.** RFC-0015 sketched an `interrupt_grace_ms`; it is not
308
+ > implemented, and would buy nothing here. The new turn is queued behind the abandoned
309
+ > one either way (one turn at a time per session is the invariant), and waiting for a
310
+ > boundary inside the request would break the ack-fast rule that put the debounce
311
+ > window on the session's fiber in the first place.
312
+
313
+ > ⚠️ **`context_budget` defaults to 8000 tokens.** A large system prompt (a rich
314
+ > persona can run tens of thousands of tokens) exceeds it, and a pinned identity
315
+ > that overflows the budget fails the turn rather than truncating the identity.
316
+ > If a freshly created agent returns empty turns, raise `context_budget` first.
317
+ > See [Context](CONTEXT.md).
318
+
319
+ ### The allowlist convention
320
+
321
+ The same three-state rule governs tools, skills, context providers, and workflows
322
+ — learn it once:
323
+
324
+ - `nil` (or absent) = **all** (opt-in capabilities aside);
325
+ - `[]` = **none**;
326
+ - `[names]` = **exactly** those.
327
+
328
+ For tools, a paired deny list (`tools_deny`) **always wins**, and
329
+ `tools_allow_groups` unions a per-group allowlist on top of `tools_allow`.
330
+
331
+ Three capabilities invert the default — `nil`/absent means **OFF**, not "all":
332
+ `subagents`, `memory`, and `guardrails` (each defaults to off or a conservative
333
+ setting, never "everything on").
334
+
335
+ ### Declaring what this deployment has
336
+
337
+ `declares "promotions", "human_handoff"` records facts about the deployment that
338
+ are not tools. It decides **nothing** at runtime — it exists so an eval case can
339
+ say what it needs and be *skipped* where it is absent instead of failing for the
340
+ wrong reason (see [Evals](EVALS.md)). A flat list you write: inferring "this store
341
+ has promotions" from data is how a test suite starts lying.
342
+
343
+ ## The five access layers
344
+
345
+ What an agent may do is layered. Each layer is independent, opt-in where it
346
+ matters, and editable hot.
347
+
348
+ ### Layer 1: Tools (what it can call)
349
+
350
+ `tools_allow` / `tools_deny` / `tools_allow_groups` decide which tools enter the
351
+ turn's tool-loop, enforced by the tool-allowlist policy. See [Tools](TOOLS.md)
352
+ for how tools are defined and registered, and [`examples/data-tool/`](https://github.com/guizaols/insika/tree/main/examples/data-tool/).
353
+
354
+ ### Layer 2: Policies and approvals
355
+
356
+ Policies are named entries evaluated before the turn runs. Builtins cover
357
+ tool-, skill-, and workflow-allowlisting, plus **`ApprovalRequired`** — which
358
+ does not allow or deny but *tags* a tool as needing human approval. Set
359
+ `approvals_required: [tool names]`; the gate then fires when the model tries to
360
+ call that tool, suspending the turn until an operator approves it in the Studio.
361
+ See [Security](SECURITY.md#human-approval).
362
+
363
+ ### Layer 3: Guardrails (content safety)
364
+
365
+ `guardrails` configures input/output content safety per agent — **opt-in**, so an
366
+ agent that says nothing gets a conservative default (deterministic detectors on,
367
+ LLM moderator off). See [Security](SECURITY.md#guardrails) and
368
+ [`examples/guardrails/`](https://github.com/guizaols/insika/tree/main/examples/guardrails/).
369
+
370
+ ### Layer 4: Edge limits (flood and spend control)
371
+
372
+ Two independent, opt-in ceilings, enforced *before* the model is ever called —
373
+ opt-in everywhere except on a public channel, where `chat_rate_limit` is
374
+ [required](CHANNELS.md#a-rate-limit-is-required-not-suggested) and the
375
+ [web widget](CHANNELS.md#the-web-widget) refuses to serve without one:
376
+
377
+ - **`chat_rate_limit`** — turn attempts per session per `chat_rate_window`.
378
+ - **`agent_token_ceiling`** — total tokens per agent per `agent_token_window`.
379
+
380
+ On breach the turn halts gracefully with a configurable `limit_response` and
381
+ **zero LLM calls**. Windows are set at the platform level; the ceilings can be set
382
+ per agent (blank inherits the platform value, `0` explicitly disables it).
383
+
384
+ > ⚠️ The token window default is **86400 (daily)**. To express "500k tokens per
385
+ > **hour**", set `agent_token_window = 3600` explicitly. A per-agent key that is
386
+ > *present but nil* reads as OFF for that agent — leave the key **absent** to
387
+ > inherit. See [Security](SECURITY.md#edge-limits).
388
+
389
+ ### Layer 5: Reasoning (thinking)
390
+
391
+ Controls the model's thinking budget, resolved by precedence
392
+ **Chat > Agent > Model > Global** (first non-blank wins):
393
+
394
+ | Scope | Where |
395
+ |-------|-------|
396
+ | Chat | session var `__llm__.thinking` |
397
+ | Agent | `profile.params["thinking"]` |
398
+ | Model | platform `model_params[<ref>].thinking` |
399
+ | Global | platform `thinking` |
400
+
401
+ Values: `off | on | low | medium | high`. `off`/`on` toggle thinking; the effort
402
+ levels map to the provider's thinking-effort parameter. This is a control
403
+ primitive, not a latency lever — turning reasoning off does not necessarily speed
404
+ up a turn, because most of a turn's latency is the provider itself, not thinking.
405
+
406
+ Whether the reasoning ever reaches the **customer** is a separate switch, off by
407
+ default:
408
+
409
+ ```ruby
410
+ edge_stream thinking: true, intermediate: false
411
+ ```
412
+
413
+ `thinking` is the provider's reasoning; `intermediate` is the model narrating its
414
+ own tool loop ("let me look that up"). Both are always on the event stream for the
415
+ Studio and the trace — this decides only whether `/v1/responses` translates them,
416
+ and each opted-in channel gets its own frame type, never the answer's. See
417
+ [Architecture](ARCHITECTURE.md#what-crosses-the-edge).
418
+
419
+ > ⚠️ Turn it on knowing your consumer. One that concatenates every text delta into
420
+ > a single message — a WhatsApp adapter — will only be affected once it learns to
421
+ > read the new frames, and when it does, the deliberation is what the customer
422
+ > reads. That is the operator's call, which is why it is neither a default nor a
423
+ > global.
424
+
425
+ ## Refinement
426
+
427
+ `refinement` configures how an agent's own traffic is read back as a report — what
428
+ broke, how often, in which conversations. Unlike the layers above it grants
429
+ nothing: a run calls no model and edits nothing, so it needs no opt-in and an
430
+ absent key still reports. See [Refinement](REFINEMENT.md).
431
+
432
+ ```ruby
433
+ refine window: { last_sessions: 200 }, max_findings: 20
434
+ ```
435
+
436
+ Editing the agent from that report is a separate, explicit `mode` — with a write
437
+ allowlist, one or more `proposers`, a token `budget`, and a gate that replays the
438
+ golden set before anything reaches a human. All of it is in
439
+ [Refinement](REFINEMENT.md); none of it is on until you name it.
440
+
441
+ ## Delegation (subagents)
442
+
443
+ An agent can delegate to **subagents**: named child agents it may invoke as a
444
+ tool, fanning work out and collecting results. Subagents are **opt-in**
445
+ (`subagents` defaults to none) and the graph is validated for cycles and depth at
446
+ create time. This is off by default because it multiplies model calls — enable it
447
+ deliberately.
448
+
449
+ Delegation only means something when the children are resolvable in the same
450
+ graph, which is what `Insika.system` is for — several agents, one runtime:
451
+
452
+ ```ruby
453
+ system = Insika.system do
454
+ agent("security") { instructions "Review code for security issues." }
455
+ agent("performance") { instructions "Review code for performance issues." }
456
+
457
+ agent "reviewer" do
458
+ instructions "Delegate to the specialists, then synthesize their reports."
459
+ subagents "security", "performance"
460
+ end
461
+ end
462
+
463
+ system.reply("reviewer", code) # one turn; the parent fans out and synthesizes
464
+ system.serve # all three on /studio + /v1 (each id is a `model`)
465
+ ```
466
+
467
+ When the *shape* of the work is known in advance — draft then edit, classify then
468
+ answer, three reviewers then a summary — put the choice in Ruby instead: see
469
+ [Workflows](WORKFLOWS.md).
470
+
471
+ The parent gets two system tools: `spawn_subagent` (one child) and
472
+ `spawn_subagents` (**N children in parallel**, one combined result — wall-clock
473
+ is the slowest child, not the sum, capped by `INSIKA_SUBAGENT_FANOUT_CAP`,
474
+ default 8). A child inherits the *environment* (model, thinking) as a default and
475
+ **never** inherits capability: its tools, skills and own subagents come from its
476
+ own profile.
477
+
478
+ ## Where agent data lives
479
+
480
+ Every agent is a row in one SQLite key-value table (WAL mode), namespaced under
481
+ `config:agents`. The database file is `INSIKA_DB`. The profile source reads
482
+ **fresh** on each dispatch, which is why Studio and API edits take effect on the
483
+ next turn with no restart. See [Deploy](DEPLOY.md) for the durable-volume setup and
484
+ [Context](CONTEXT.md#the-volume) for why editing a committed file does *not* change
485
+ a running agent.
486
+
487
+ ## See also
488
+
489
+ - [Tools](TOOLS.md) — define, register, and troubleshoot tools.
490
+ - [Skills](SKILLS.md) — progressive playbooks an agent loads on demand.
491
+ - [Context](CONTEXT.md) — what fills a turn's prompt, and memory.
492
+ - [Security](SECURITY.md) — guardrails, sandbox, approvals, edge limits.
493
+ - [Architecture](ARCHITECTURE.md) — how a turn actually runs.
494
+ - [`examples/`](https://github.com/guizaols/insika/tree/main/examples/) — one runnable project per capability.