insika 0.3.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (204) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +296 -0
  3. data/README.md +48 -12
  4. data/bin/insika +725 -0
  5. data/bin/insika-router +87 -0
  6. data/docs/AGENTS.md +116 -406
  7. data/docs/API.md +5 -5
  8. data/docs/ARCHITECTURE.md +3 -2
  9. data/docs/ARTIFACTS.md +137 -0
  10. data/docs/BENCHMARK.md +2 -2
  11. data/docs/CHANNELS.md +14 -14
  12. data/docs/CONTEXT.md +63 -19
  13. data/docs/DEMO.md +80 -0
  14. data/docs/DEPLOY.md +87 -10
  15. data/docs/EMBEDDING.md +1 -1
  16. data/docs/EVALS.md +128 -3
  17. data/docs/FACTS.md +3 -3
  18. data/docs/HARVEST.md +5 -6
  19. data/docs/KNOWLEDGE.md +290 -0
  20. data/docs/LOADTEST.md +17 -29
  21. data/docs/MEDIA.md +128 -0
  22. data/docs/OBSERVABILITY.md +46 -12
  23. data/docs/OUTCOMES.md +137 -0
  24. data/docs/PLUGINS.md +51 -6
  25. data/docs/POLICY.md +222 -0
  26. data/docs/REFINEMENT.md +14 -9
  27. data/docs/RELEASING.md +4 -4
  28. data/docs/ROUTER.md +213 -0
  29. data/docs/RUNNING-LOCAL.md +5 -5
  30. data/docs/SCHEDULING.md +121 -0
  31. data/docs/SECURITY.md +23 -7
  32. data/docs/SKILLS.md +11 -2
  33. data/docs/SOAK.md +3 -3
  34. data/docs/TEMPLATES.md +134 -0
  35. data/docs/TOOLS.md +176 -27
  36. data/docs/WHY.md +1 -1
  37. data/docs/WORKFLOWS.md +2 -2
  38. data/docs/_includes/head_custom.html +5 -0
  39. data/docs/_includes/title.html +13 -0
  40. data/docs/_sass/color_schemes/insika.scss +32 -0
  41. data/docs/_sass/custom/custom.scss +199 -0
  42. data/docs/_sass/custom/setup.scss +26 -0
  43. data/docs/assets/img/favicon.svg +7 -0
  44. data/docs/assets/img/insika-mark.svg +7 -0
  45. data/docs/core-concepts.md +21 -0
  46. data/docs/domain.md +4 -4
  47. data/docs/improve.md +20 -0
  48. data/docs/index.md +8 -5
  49. data/docs/integrate.md +20 -0
  50. data/docs/operate.md +13 -6
  51. data/docs/prompts/ADD-TOOL.md +118 -0
  52. data/docs/prompts/DIAGNOSE-TURN.md +65 -0
  53. data/docs/prompts/GO-LIVE.md +138 -0
  54. data/docs/prompts/RUN-EXAMPLES.md +70 -0
  55. data/docs/reference.md +19 -0
  56. data/docs/ship.md +10 -2
  57. data/docs/start-here.md +18 -0
  58. data/lib/insika/agent_profile.rb +99 -17
  59. data/lib/insika/artifact_signing.rb +82 -0
  60. data/lib/insika/artifact_store.rb +160 -0
  61. data/lib/insika/channel_delivery.rb +1 -1
  62. data/lib/insika/chat_builder.rb +50 -19
  63. data/lib/insika/commands/agent_payload.rb +2 -2
  64. data/lib/insika/commands/backfill_knowledge.rb +145 -0
  65. data/lib/insika/commands/delete_artifact.rb +35 -0
  66. data/lib/insika/commands/delete_concept.rb +34 -0
  67. data/lib/insika/commands/delete_mcp.rb +6 -2
  68. data/lib/insika/commands/delete_tenant_data.rb +15 -3
  69. data/lib/insika/commands/gate_refinement.rb +1 -1
  70. data/lib/insika/commands/refresh_mcp_tools.rb +47 -0
  71. data/lib/insika/commands/restore_concept.rb +34 -0
  72. data/lib/insika/commands/seed_demo_data.rb +31 -0
  73. data/lib/insika/commands/upsert_mcp.rb +6 -3
  74. data/lib/insika/commands/write_concept.rb +57 -0
  75. data/lib/insika/compaction.rb +196 -0
  76. data/lib/insika/context/builder.rb +6 -2
  77. data/lib/insika/context/fragment.rb +4 -1
  78. data/lib/insika/context/priority.rb +8 -0
  79. data/lib/insika/context/providers/briefing.rb +53 -24
  80. data/lib/insika/context/providers/knowledge.rb +108 -0
  81. data/lib/insika/context/providers/prompt.rb +30 -24
  82. data/lib/insika/context/providers/session.rb +46 -10
  83. data/lib/insika/context_trace_store.rb +11 -1
  84. data/lib/insika/cron.rb +189 -0
  85. data/lib/insika/demo/agent_attrs.rb +43 -0
  86. data/lib/insika/demo/golden_cases.rb +81 -0
  87. data/lib/insika/demo/seeder.rb +336 -0
  88. data/lib/insika/doctor.rb +280 -17
  89. data/lib/insika/dsl/definition.rb +3 -2
  90. data/lib/insika/dsl/runtime.rb +64 -79
  91. data/lib/insika/dsl/server_boot.rb +23 -1
  92. data/lib/insika/dsl/system.rb +10 -2
  93. data/lib/insika/dsl.rb +103 -2
  94. data/lib/insika/env_schema.rb +21 -7
  95. data/lib/insika/evals/golden.rb +41 -4
  96. data/lib/insika/evals/judge.rb +47 -2
  97. data/lib/insika/evals/pairwise.rb +11 -0
  98. data/lib/insika/evals/persona.rb +98 -0
  99. data/lib/insika/evals/runner.rb +9 -0
  100. data/lib/insika/evals/simulator.rb +225 -0
  101. data/lib/insika/evals/transport.rb +84 -2
  102. data/lib/insika/event_stream.rb +10 -0
  103. data/lib/insika/executor.rb +295 -55
  104. data/lib/insika/followup_policy.rb +2 -25
  105. data/lib/insika/golden_store.rb +16 -1
  106. data/lib/insika/grounding/matcher.rb +1 -1
  107. data/lib/insika/knowledge.rb +680 -0
  108. data/lib/insika/knowledge_store.rb +140 -0
  109. data/lib/insika/loop_detector.rb +5 -34
  110. data/lib/insika/mcp_client.rb +94 -0
  111. data/lib/insika/mcp_json.rb +74 -0
  112. data/lib/insika/mcp_live_tool.rb +43 -0
  113. data/lib/insika/mcp_store.rb +98 -26
  114. data/lib/insika/mcp_tool_ingestor.rb +30 -8
  115. data/lib/insika/mcp_tool_registry.rb +100 -0
  116. data/lib/insika/media.rb +115 -31
  117. data/lib/insika/message_origin.rb +1 -1
  118. data/lib/insika/middleware.rb +9 -0
  119. data/lib/insika/onboarding.rb +17 -1
  120. data/lib/insika/outcome_store.rb +1 -1
  121. data/lib/insika/overlay_tool_registry.rb +37 -17
  122. data/lib/insika/packaging.rb +2 -2
  123. data/lib/insika/profile_source.rb +15 -1
  124. data/lib/insika/prompt_catalog.rb +10 -0
  125. data/lib/insika/retention.rb +36 -1
  126. data/lib/insika/router/app.rb +157 -0
  127. data/lib/insika/router/backend_pool.rb +98 -0
  128. data/lib/insika/router/hash_ring.rb +55 -0
  129. data/lib/insika/router/proxy_body.rb +34 -0
  130. data/lib/insika/router/session_key.rb +54 -0
  131. data/lib/insika/router.rb +18 -0
  132. data/lib/insika/schedule.rb +177 -0
  133. data/lib/insika/schedule_engine.rb +314 -0
  134. data/lib/insika/schedule_store.rb +208 -0
  135. data/lib/insika/server/app.rb +105 -15
  136. data/lib/insika/server/rack_app.rb +5 -1
  137. data/lib/insika/server/responses.rb +5 -5
  138. data/lib/insika/session_store.rb +34 -4
  139. data/lib/insika/settings_store.rb +8 -1
  140. data/lib/insika/skill_catalog.rb +12 -0
  141. data/lib/insika/soak/runner.rb +4 -4
  142. data/lib/insika/steer_injector.rb +21 -10
  143. data/lib/insika/studio/app.rb +591 -47
  144. data/lib/insika/studio/assets/dist/application.css +1 -1
  145. data/lib/insika/studio/assets/dist/application.js +21 -21
  146. data/lib/insika/studio/forms.rb +57 -5
  147. data/lib/insika/studio/nav_icons.rb +14 -1
  148. data/lib/insika/studio/views/_agent_tab_cache.erb +25 -0
  149. data/lib/insika/studio/views/_agent_tab_config.erb +514 -0
  150. data/lib/insika/studio/views/_agent_tab_history.erb +24 -0
  151. data/lib/insika/studio/views/_agent_tab_loops.erb +54 -0
  152. data/lib/insika/studio/views/_agent_tab_memory.erb +51 -0
  153. data/lib/insika/studio/views/_agent_tab_outcomes.erb +31 -0
  154. data/lib/insika/studio/views/_agent_tab_prompts.erb +108 -0
  155. data/lib/insika/studio/views/_agent_tab_skills.erb +38 -0
  156. data/lib/insika/studio/views/_agents_master.erb +44 -0
  157. data/lib/insika/studio/views/_message.erb +49 -32
  158. data/lib/insika/studio/views/agent_detail.erb +61 -820
  159. data/lib/insika/studio/views/agents.erb +70 -57
  160. data/lib/insika/studio/views/artifact.erb +23 -0
  161. data/lib/insika/studio/views/artifacts.erb +59 -0
  162. data/lib/insika/studio/views/evals.erb +2 -2
  163. data/lib/insika/studio/views/facts.erb +1 -1
  164. data/lib/insika/studio/views/funnel.erb +1 -1
  165. data/lib/insika/studio/views/home.erb +106 -67
  166. data/lib/insika/studio/views/knowledge.erb +123 -0
  167. data/lib/insika/studio/views/layout.erb +14 -11
  168. data/lib/insika/studio/views/mcp.erb +174 -80
  169. data/lib/insika/studio/views/session.erb +231 -177
  170. data/lib/insika/studio/views/settings.erb +50 -1
  171. data/lib/insika/studio/views/skills.erb +1 -1
  172. data/lib/insika/studio/views/tools.erb +24 -9
  173. data/lib/insika/telemetry/recorder.rb +49 -1
  174. data/lib/insika/templates/browser-agent/README.md +36 -0
  175. data/lib/insika/templates/browser-agent/agent.rb +49 -0
  176. data/lib/insika/templates/daily-digest/README.md +47 -0
  177. data/lib/insika/templates/daily-digest/agent.rb +77 -0
  178. data/lib/insika/templates/repo-explorer/README.md +36 -0
  179. data/lib/insika/templates/repo-explorer/agent.rb +45 -0
  180. data/lib/insika/templates/research-analyst/README.md +26 -0
  181. data/lib/insika/templates/research-analyst/agent.rb +68 -0
  182. data/lib/insika/templates/review-panel/README.md +20 -0
  183. data/lib/insika/templates/review-panel/agent.rb +50 -0
  184. data/lib/insika/templates/travel-planner/README.md +35 -0
  185. data/lib/insika/templates/travel-planner/agent.rb +87 -0
  186. data/lib/insika/templates.rb +112 -0
  187. data/lib/insika/tick.rb +24 -12
  188. data/lib/insika/timezone.rb +45 -0
  189. data/lib/insika/tool_batch.rb +67 -0
  190. data/lib/insika/tool_usage_report.rb +162 -0
  191. data/lib/insika/tools/generate_image.rb +52 -7
  192. data/lib/insika/tools/load_knowledge.rb +74 -0
  193. data/lib/insika/tools/run_persona_eval.rb +328 -0
  194. data/lib/insika/tools/save_artifact.rb +95 -0
  195. data/lib/insika/turn_budget.rb +91 -0
  196. data/lib/insika/turn_output.rb +1 -1
  197. data/lib/insika/turn_state.rb +15 -4
  198. data/lib/insika/version.rb +1 -1
  199. data/lib/insika/wiring/graph.rb +184 -12
  200. data/lib/insika/wiring/graph_chat.rb +102 -0
  201. data/lib/insika.rb +64 -0
  202. metadata +109 -5
  203. data/docs/build.md +0 -14
  204. data/docs/understand.md +0 -10
data/docs/DEPLOY.md CHANGED
@@ -25,7 +25,7 @@ is durable SQLite (WAL) at `INSIKA_DB`; mount a volume and point it inside.
25
25
  docker build -t insika .
26
26
  docker run -p 9292:9292 -v insika-data:/data \
27
27
  -e DEEPSEEK_API_KEY=sk-... \
28
- -e OPENCLAW_GATEWAY_TOKEN=change-me \
28
+ -e INSIKA_GATEWAY_TOKEN=change-me \
29
29
  insika
30
30
  curl localhost:9292/up # {"status":"ok"}
31
31
  ```
@@ -57,7 +57,14 @@ contract:
57
57
  worker that holds the session's actor. The engine does **not** promise them
58
58
  across workers. A deploy that needs those semantics for a session must route
59
59
  that session's traffic to one worker (sticky routing) — or accept per-worker
60
- best-effort.
60
+ best-effort. **On Railway there is no sticky-routing option at any layer** —
61
+ confirmed against Railway's own docs, which distribute traffic randomly and
62
+ explicitly do not support sticky sessions — so N>1 there is not "per-worker
63
+ best-effort," it is a guaranteed cross-session leak the first time two
64
+ requests for the same session land on different workers. See the Railway
65
+ section below; `insika doctor` errors on `WEB_CONCURRENCY>1` there — unless
66
+ [`insika-router`](ROUTER.md), the session-sticky proxy, is in front,
67
+ which is what makes N>1 safe on Railway too (N local workers behind it).
61
68
  3. **Recovery is part of boot, in every wiring.** Every worker boots through
62
69
  `Server::Boot`, which runs recovery **before the listen**. The per-record
63
70
  sweeps (undelivered outbox records, undelivered delegation results) run in
@@ -108,7 +115,7 @@ this section is the single source of truth for what changing it means.
108
115
  | `INSIKA_DRAIN_TIMEOUT` | `20` | seconds a stopping worker waits for in-flight turns before abandoning them to the next boot's recovery (process model, item 4). The entrypoint sizes Falcon's `--graceful-stop` from it; on Railway also set `RAILWAY_DEPLOYMENT_DRAINING_SECONDS` ≥ drain + 10 |
109
116
  | `INSIKA_TICK_INTERVAL` | `60` | seconds between tick passes — outbox drain + stale recovery sweep (process model, item 5). `0` disables |
110
117
  | `INSIKA_TICK_STALE_AFTER` | `900` | seconds a `:queued`/`:running` task must sit untouched before the tick sweeps it. Must exceed the largest `turn_timeout` of the deployment |
111
- | `OPENCLAW_GATEWAY_TOKEN` | falls back to `ADMIN_TOKEN` | Bearer for `/v1/responses` and `/v1/agents` (the API contract) |
118
+ | `INSIKA_GATEWAY_TOKEN` | falls back to `ADMIN_TOKEN` | Bearer for `/v1/responses` and `/v1/agents` (the API contract) |
112
119
  | `ADMIN_TOKEN` | `local-demo` | login token for `/studio` (**change in production**) |
113
120
  | `DEEPSEEK_API_KEY` | — | provider key. **Without it the engine still boots** (`/up` green), but turns fail until it is configured (env or Studio → LLM providers) — cloud resilience |
114
121
  | `DEEPSEEK_MODEL` | `deepseek-v4-flash` | model |
@@ -118,7 +125,7 @@ this section is the single source of truth for what changing it means.
118
125
  | `INSIKA_RELAY_TOKEN` | — | **mounts the relay channel** at `POST /channels/relay/events`, and is the Bearer it requires. Empty = the route does not exist (`404`). See [Channels](CHANNELS.md) |
119
126
  | `INSIKA_RELAY_DELIVER_URL` | — | your callback; the engine POSTs each reply there. Goes through the egress guard |
120
127
  | `INSIKA_RELAY_DELIVER_TOKEN` | — | Bearer the engine sends **to** your callback (optional) |
121
- | `INSIKA_RELAY_DELIVERY` | `at_end` | how the relay flushes the outbox: `at_end` (one POST) or `progressive` (one POST per balloon — [RFC-0027](CHANNELS.md#delivery-policy)) |
128
+ | `INSIKA_RELAY_DELIVERY` | `at_end` | how the relay flushes the outbox: `at_end` (one POST) or `progressive` (one POST per balloon — [delivery policy](CHANNELS.md#delivery-policy)) |
122
129
  | `INSIKA_WIDGET_ORIGINS` | — | exact-match origins allowed to embed the [web widget](CHANNELS.md#the-web-widget), comma-separated. No wildcards. **Half the switch**: with `INSIKA_WIDGET_AGENTS` unset, nothing is mounted (`404`) |
123
130
  | `INSIKA_WIDGET_AGENTS` | — | agent ids a widget visitor may address, comma-separated. The other half of the switch. **A chat rate limit is also required** or the widget answers `503` |
124
131
  | `LITESTREAM_REPLICA_URL` | — | **enables Litestream** (backup/DR). Empty = disabled (default). See below |
@@ -150,7 +157,7 @@ values (the API token falling back to `ADMIN_TOKEN` is a dev convenience only):
150
157
  - **`ADMIN_TOKEN`** — the `/studio` login (cookie auth). This is the **operator**
151
158
  surface (just you). Rotating it is **safe and independent**: change it, redeploy,
152
159
  log in with the new value. It does not affect any API consumer.
153
- - **`OPENCLAW_GATEWAY_TOKEN`** — the Bearer for `/v1/responses` and `/v1/agents`.
160
+ - **`INSIKA_GATEWAY_TOKEN`** — the Bearer for `/v1/responses` and `/v1/agents`.
154
161
  This is the **contract with your API consumers**. Rotating it means **changing
155
162
  both sides together** (or the integration breaks): update the runtime var **and**
156
163
  each consumer's token in the same step.
@@ -180,15 +187,24 @@ insika doctor # colored report; exits != 0 on any error
180
187
  insika doctor --json # machine-readable (CI / monitoring)
181
188
  insika doctor --fix # applies the safe autofixes and re-diagnoses
182
189
  insika env # lists known keys + current values (secrets masked)
190
+ insika tools:report # tool audit over the stored traces: never-called
191
+ # allowlisted tools, error rate > 30%, stale tools —
192
+ # read-only, the operator removes ([--agent ID] [--days N] [--json])
183
193
  ```
184
194
 
185
195
  Checks: env (the schema above), settings schema version (a pending migration →
186
196
  `--fix` applies it), a missing platform `default_model` (`--fix` seeds it from
187
197
  `DEEPSEEK_MODEL`), durable vs ephemeral backend, LLM provider configured,
188
- `ADMIN_TOKEN` set, data-tool definitions still valid, **prompt files that hold
198
+ `ADMIN_TOKEN` set, data-tool definitions still valid, **a stored agent whose
199
+ declared tool allow/deny list does not name the `tool_allowlist` policy** (the
200
+ engine repairs it on read, but as stored the list is inert — see
201
+ [Agents](AGENTS.md#the-allowlist-convention)), **prompt files that hold
189
202
  text rather than a serialized object** (a file whose content is a stringified Hash
190
203
  serves a mangled prompt on every turn while looking perfectly healthy — present,
191
- non-empty, and the agent still answers), and **skill drift** — a shared skill whose
204
+ non-empty, and the agent still answers), **a prompt file that outgrew a prompt**
205
+ (WARN past ~6 000 estimated tokens or 600 lines — the LLM-generated pack shape
206
+ that costs 20%+ extra tokens per turn for no better instruction-following), and
207
+ **skill drift** — a shared skill whose
192
208
  body names one store, a prompt file routing to a skill the agent cannot load, a broken
193
209
  companion pair, a stale `eager:` key (see
194
210
  [Skills](SKILLS.md#drift-guards)). Settings-schema migrations are **explicit**
@@ -220,12 +236,65 @@ healthcheck, and a restart policy.
220
236
  2. **Volume**: mount it at `/data` (the default `INSIKA_DB` points there) —
221
237
  without a volume, SQLite is ephemeral and recovery resumes nothing after a
222
238
  redeploy.
223
- 3. **Vars**: `DEEPSEEK_API_KEY`, `OPENCLAW_GATEWAY_TOKEN`, `CONSUMER_INTERNAL_URL`,
224
- `INSIKA_EGRESS_HOSTS` (and `WEB_CONCURRENCY` to match your plan/CPU).
239
+ 3. **Vars**: `DEEPSEEK_API_KEY`, `INSIKA_GATEWAY_TOKEN`, `CONSUMER_INTERNAL_URL`,
240
+ `INSIKA_EGRESS_HOSTS`. **Leave `WEB_CONCURRENCY` at its default of 1** unless
241
+ you run [`insika-router`](ROUTER.md) in front. Railway's own docs say it
242
+ "does not support sticky sessions" and randomly distributes traffic across
243
+ replicas/workers — there is no way on this platform to satisfy the "sticky
244
+ routing per session in front" precondition item 2 of the process model
245
+ requires, at any layer (Railway replicas or Falcon's own `--count` workers
246
+ within one container). Raising `WEB_CONCURRENCY` without a router in front
247
+ is not a throughput knob, it is a guaranteed way to leak a reply across
248
+ sessions the first time two requests for the same session land on different
249
+ workers — `insika doctor` errors on this combination for exactly that
250
+ reason. `insika-router` is the fix: N local Falcon workers
251
+ behind one sticky proxy, still one Railway replica.
225
252
  4. The healthcheck hits `/up`.
226
253
  5. Point your consumer at the service's public URL, with a matching API token
227
254
  (see [RUNNING-LOCAL.md](RUNNING-LOCAL.md)).
228
255
 
256
+ ## Kubernetes
257
+
258
+ Same contract as Railway, for the same reason, verified against
259
+ ingress-nginx's own docs: **run one pod, `WEB_CONCURRENCY=1`.** Scale
260
+ vertically (bigger pod) for now.
261
+
262
+ Why a standard K8s setup does not get you out of this: the session id lives
263
+ in the **JSON body** of `POST /v1/responses` (the `user` field), not in a
264
+ header, cookie, or URL segment. That rules out every sticky-routing
265
+ mechanism K8s gives you for free:
266
+
267
+ - A vanilla `Service` (`ClusterIP`) load-balances across pod endpoints with
268
+ no notion of session at all — same failure mode as Railway's replicas,
269
+ one layer down.
270
+ - `Service.spec.sessionAffinity: ClientIP` does not help either: the caller
271
+ is the consumer's own backend (achei-b2b, a server-to-server call), not
272
+ the end customer's browser, so many different customers' sessions arrive
273
+ from the same source IP and would collide on the same pod instead of
274
+ spreading — the opposite of what you want, and it still does not restore
275
+ a real per-session guarantee.
276
+ - ingress-nginx's `nginx.ingress.kubernetes.io/upstream-hash-by` (the
277
+ standard sticky/consistent-hash annotation) hashes on nginx *variables* —
278
+ headers, cookies, `$request_uri`, client IP — not on a field parsed out of
279
+ a POST body. Reaching into the JSON body needs a custom Lua/OpenResty
280
+ snippet (or an Envoy filter) that parses the request and extracts `user`
281
+ before hashing. That is real, unbuilt engineering work, not a K8s
282
+ annotation to flip.
283
+
284
+ If real horizontal throughput is ever needed, two paths get there, neither
285
+ of them "just add replicas":
286
+ 1. **Build the body-aware sticky layer** above (Lua/Envoy consistent-hash on
287
+ the `user` field) in front of N pods, each still `WEB_CONCURRENCY=1`.
288
+ 2. **Move the session id into the URL**, the way the web widget transport
289
+ already does (`POST /api/widget/sessions/:token/messages` carries the
290
+ session token in the path) — that surface, unlike `/v1/responses`, *is*
291
+ sticky-routable today with a plain `upstream-hash-by: $request_uri`, no
292
+ custom scripting required.
293
+
294
+ Neither is urgent while a single pod's throughput is enough — this section
295
+ exists so scaling this deployment does not silently reintroduce the same
296
+ cross-session leak the Railway incident above already found once.
297
+
229
298
  ## Backup / DR — Litestream (opt-in, configurable)
230
299
 
231
300
  A single volume is the **one point of total loss** between a pilot and production
@@ -300,6 +369,12 @@ per pod + **sticky-by-agent** routing (shard by tenant), or **LiteFS**, or an
300
369
  optional **Postgres** adapter. **Litestream** (above) for backup/DR from day one —
301
370
  orthogonal to topology.
302
371
 
372
+ For the *session-routing* half specifically — as opposed to the storage
373
+ topology above — see [`insika-router`](ROUTER.md): N pods behind a
374
+ headless `Service`, one `insika-router` Deployment in front doing the
375
+ consistent-hash routing ingress-nginx's own `upstream-hash-by` cannot (it
376
+ hashes nginx variables, never a field parsed out of a POST body).
377
+
303
378
  ---
304
379
 
305
380
  ## Measuring performance / load
@@ -336,7 +411,7 @@ cache hits, P50/P95, error rate. Runs against local or a remote deployment. See
336
411
  [LOADTEST.md](LOADTEST.md).
337
412
 
338
413
  ```bash
339
- INSIKA_URL=http://localhost:9292 OPENCLAW_GATEWAY_TOKEN=xxx \
414
+ INSIKA_URL=http://localhost:9292 INSIKA_GATEWAY_TOKEN=xxx \
340
415
  bundle exec ruby scripts/loadtest.rb --agents assistant --concurrency 16 --iterations 3
341
416
  ```
342
417
 
@@ -351,6 +426,8 @@ DEEPSEEK_API_KEY=sk-... ./scripts/loadtest-local.sh 4 24
351
426
 
352
427
  ## See also
353
428
 
429
+ - [ROUTER.md](ROUTER.md) — `insika-router`, the session-sticky proxy for
430
+ scaling past one worker.
354
431
  - [RUNNING-LOCAL.md](RUNNING-LOCAL.md) — run the engine locally, single-process.
355
432
  - [Security](SECURITY.md) — tokens, egress, strict config.
356
433
  - [BENCHMARK.md](BENCHMARK.md) — the neutral, key-free engine benchmark.
data/docs/EMBEDDING.md CHANGED
@@ -1,6 +1,6 @@
1
1
  ---
2
2
  title: Embedding
3
- parent: Ship it
3
+ parent: Integrate
4
4
  nav_order: 4
5
5
  permalink: /embedding/
6
6
  ---
data/docs/EVALS.md CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  title: Evals
3
- parent: Operate & prove it
4
- nav_order: 4
3
+ parent: Improve
4
+ nav_order: 1
5
5
  permalink: /evals/
6
6
  ---
7
7
 
@@ -157,6 +157,82 @@ Three rules make the number worth quoting:
157
157
  Cost: **two provider calls per judge per case**, which is why it is opt-in and never
158
158
  part of the gate.
159
159
 
160
+ ## Simulated users — conversations the agent never had
161
+
162
+ A golden case fixes the user's turns in advance. Real customers branch — they answer
163
+ the agent's question, or ignore it, or send an order number three messages later. A
164
+ case with frozen turns can only ever test the path the author imagined.
165
+
166
+ A **simulated** case generates the conversation instead: a persona model (the cheap
167
+ platform `utility_model`) plays a customer with the persona as its **whole
168
+ instruction**, and the target agent answers through the same transport a replay uses.
169
+ They talk until the persona's `max_turns`, or until the persona emits a stop marker —
170
+ `<<goal_met>>` when its goal is served, `<<gave_up>>` when it abandons. Both are
171
+ recorded: "gave up at turn 3" is a finding.
172
+
173
+ ```yaml
174
+ id: loja-objetivo-difuso
175
+ agent: loja-chocolates
176
+ persona: # the SIMULATED customer (alternative to `turns:`)
177
+ goal: "descobrir o que comprar de presente; não sabe nomes de produto"
178
+ style: "mensagens curtas, responde o que perguntam, desiste se tiver que explicar duas vezes"
179
+ opens_with: "oi, queria um presente"
180
+ knows: # the ONLY facts the persona may assert
181
+ orcamento: "até R$ 100"
182
+ ocasião: "aniversário"
183
+ max_turns: 8
184
+ expect:
185
+ policy: investigate_first
186
+ rubric: |
187
+ Descobre o objetivo antes de recomendar (uma ou duas perguntas, não um formulário),
188
+ recomenda do catálogo real e fecha com um próximo passo claro. Reprova se despejar
189
+ catálogo antes de entender, ou se ficar perguntando sem nunca buscar.
190
+ min_score: 0.7
191
+ ```
192
+
193
+ **The anti-invention rule is the soul of the feature.** The persona's prompt contains
194
+ only the `knows` facts and the rule that it may assert exactly those and nothing else
195
+ — asked about anything it does not have, it answers with ignorance ("não sei", "não
196
+ tenho isso aqui"), like a real customer without that fact. A simulator that invents an
197
+ order number produces a conversation the agent could never have had, and a case that
198
+ tests nothing.
199
+
200
+ ```bash
201
+ insika evals:simulate --persona persona.yml --target loja-chocolates --staging
202
+ ```
203
+
204
+ - `--target <agent|url>` — an agent id over the deployment's `/v1/responses`, or a full
205
+ A2A URL (an agent that only speaks A2A rides the same Simulator through a thin A2A
206
+ transport; it polls once per second, bounded by `--timeout`, since a remote agent
207
+ takes seconds to finish a task).
208
+ - `--staging` or `--eval-profile` — **one is required**. A simulated conversation
209
+ marks its transcript `simulated: true`; it does NOT disarm the agent's tools. So the
210
+ Simulator only runs against (a) a target the operator declares is a staging
211
+ deployment (`--staging`), or (b) an eval profile where the agent's side-effect tools
212
+ are swapped for dry-runs (`--eval-profile`).
213
+ - `--eval-profile` is a **verified declaration, not a trust-me flag**. The CLI derives
214
+ the target's side-effect tools — the deployment's own registry over
215
+ `GET /v1/agents/:id` (`side_effect_tools`), or the local store when `INSIKA_DB`
216
+ points at it — and refuses unless `--eval-tools <a,b,c>` names every one of them.
217
+ The swap itself happens where the tools run (the deployment's eval profile; the
218
+ in-process overlay for a local graph, as in `examples/agent-tester/`); the
219
+ derivation is what keeps the client honest about what must be covered. An A2A target
220
+ has no reachable registry, so the explicit `--eval-tools` list is required there.
221
+ - `--persona-model` — the model playing the customer (default: the platform
222
+ `utility_model`). The judge that scores the transcript comes from the same
223
+ panel configuration as a replay.
224
+ - `--conv` — the conversation id. The default is unique per run
225
+ (`sim-<case id>-<random>`), because two runs on the same conv share the
226
+ deployment session: the second inherits the first's history — including
227
+ across the arms of an A/B comparison — and the grade stops measuring the
228
+ agent. Pass `--conv` only when continuing a session is the point.
229
+
230
+ Every simulated run is **`simulated: true`** — the transcript carries the flag end to
231
+ end, so a report never mixes a generated conversation with real traffic. A simulated
232
+ case (`persona:`) is one of the two shapes of a case; the replay Runner **skips** it
233
+ (the turns do not exist until a Simulator generates them) and only the simulate CLI
234
+ drives it.
235
+
160
236
  Cases live in two places, and it is the same YAML in both:
161
237
 
162
238
  - **`evals/golden/**`** in the repo — the curated corpus, reviewable in a pull request,
@@ -178,6 +254,55 @@ A case whose stored YAML no longer validates is **listed as broken** on the Eval
178
254
  rather than skipped in silence — a test suite that quietly shrinks is worse than a red
179
255
  one.
180
256
 
257
+ ### `run_persona_eval` — a QA agent that tests other agents
258
+
259
+ A tool (`lib/insika/tools/run_persona_eval.rb`), not a CLI: a QA agent picks one of its
260
+ authored simulated cases and runs it **in-process**, against the case's declared
261
+ target agent — the same Simulator + Judge machinery `evals:simulate` drives over HTTP,
262
+ minus the network hop.
263
+
264
+ ```ruby
265
+ tools %w[run_persona_eval save_artifact]
266
+ ```
267
+
268
+ The model only sees `case_id`, enumerated with the ids the tool can actually run
269
+ (never a free string it could invent) — every **simulated** case in the store, the same
270
+ `persona:` shape as above.
271
+
272
+ **Safety is derived here too, but there is no swap yet.** The tool computes the
273
+ target's reachable side-effect tools (`Evals::EvalProfile`, the same derivation the CLI
274
+ uses) and **refuses outright** — naming the tools — if that list is non-empty. Unlike
275
+ the CLI, nothing here actually swaps a side-effect tool for a dry-run: `Evals::
276
+ EvalProfile.registry` (the overlay) exists for exactly that, but nothing calls it yet.
277
+ So `run_persona_eval` only runs against **read-only** target agents today; wiring the
278
+ overlay into an in-process turn (so a target WITH a write tool can be tested safely) is
279
+ follow-up work, not something this tool claims to do.
280
+
281
+ **Budget**: the persona model + judge model calls are the cost of running the eval,
282
+ charged to the **calling** agent's own turn — never the target's (the target's own
283
+ turns bill normally, through the ordinary edge limiter, exactly as a real customer's
284
+ would). A hard cap on the calling agent skips the run — visibly, in the tool's own
285
+ result (`{skipped: true, reason: "budget", window: "daily"|"monthly"}`) — before a cent
286
+ is spent; a soft cap just runs (the ledger's own alert already warns).
287
+
288
+ Each run gets a **fresh session** (`eval-<case>-<random>`), never a reused one — a
289
+ reused session lets the target agent "remember" earlier runs, which is not what a
290
+ persona case is testing. See `examples/agent-tester/qa_scheduled.rb` for the full
291
+ loop: `schedule` fires the QA agent, it calls `run_persona_eval`, then
292
+ `save_artifact` publishes the quality report.
293
+
294
+ **Tenant isolation.** A golden case carries a `tenant` (`platform` — the
295
+ single-tenant default — unless the case declares one, the same convention
296
+ `save_artifact` uses for its own binding tenant). `run_persona_eval` only ever
297
+ lists or runs cases in the *calling* agent's own tenant: a case authored for
298
+ another tenant is invisible, not merely refused — the model cannot even learn
299
+ its id from the enum, and running it by a guessed id gets the same "unknown or
300
+ invalid" error a nonexistent id would (the two must not be distinguishable).
301
+ This is what makes "one QA agent per store" an actual boundary
302
+ in a deployment where several stores' persona cases live in the same
303
+ `GoldenStore` — without it, `qa-store-a` could enumerate and run
304
+ `qa-store-b`'s persona and read its `knows` in the transcript.
305
+
181
306
  ## Running
182
307
 
183
308
  ```bash
@@ -235,7 +360,7 @@ not a regression.
235
360
  The same gate machinery guards the two automated loops — [Refinement](REFINEMENT.md)
236
361
  (cloned-agent replay of a proposed edit) and [Harvest](HARVEST.md) (cloned-agent
237
362
  replay of a mined skill). **Judges are mandatory for those gates** in exactly three
238
- shapes, the P18 lesson: a gate without a recorded baseline refuses, an all-red
363
+ shapes: a gate without a recorded baseline refuses, an all-red
239
364
  baseline refuses, and a baseline recorded WITH judge scores, replayed with no judge
240
365
  configured, refuses — a rubric'd case with no verdict reads as a pass, so the
241
366
  candidate would beat a measurement it never took. A store with no golden cases cannot
data/docs/FACTS.md CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  title: Facts
3
- parent: Operate & prove it
4
- nav_order: 6
3
+ parent: Improve
4
+ nav_order: 5
5
5
  permalink: /facts/
6
6
  ---
7
7
 
@@ -92,7 +92,7 @@ visible — the operator's edit always wins, never a silent overwrite.
92
92
  ## Provenance
93
93
 
94
94
  An approved fact is written with `origin: "distilled:<session_ref>"` — the
95
- RFC-0031 provenance discipline: `"engine"` (the `remember` tool), `"operator"`
95
+ The provenance discipline: `"engine"` (the `remember` tool), `"operator"`
96
96
  (Studio edits), `"legacy"`, and `"distilled"` (+ the session that produced it).
97
97
  The Closed loop reads the same cell every later turn injects, and the
98
98
  distillation ledger suppresses re-proposing a fact that is already applied with
data/docs/HARVEST.md CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  title: Harvest
3
- parent: Operate & prove it
4
- nav_order: 7
3
+ parent: Improve
4
+ nav_order: 6
5
5
  permalink: /harvest/
6
6
  ---
7
7
 
@@ -44,7 +44,7 @@ filters, each drop counted and logged:
44
44
  regex; phrases match case/accent-folded at word boundaries. Every rejected
45
45
  candidate is logged with the rule id.
46
46
  2. **The grounding filter** — every product reference in a proposal must be
47
- in the union of the origin sessions' evidence ids (RFC-0029's ledger).
47
+ in the union of the origin sessions' evidence ids (the evidence ledger).
48
48
  A store without `grounding.matcher.sku` does not mine at all: product
49
49
  claims that cannot be verified are blocked by refusal, not by prompt.
50
50
  3. **Dedup** — an open `(agent, name)` tuple or a skill the store already has.
@@ -59,11 +59,10 @@ regression disqualifies** — the gate is a veto, never a score to argue with.
59
59
  Judges are mandatory in exactly the shapes the refinement gate already
60
60
  refuses: no recorded baseline, an all-red baseline, and a judged baseline
61
61
  replayed with no judge (a rubric'd case with no verdict would count as a
62
- pass — the P18 lesson, see [Evals](EVALS.md)).
62
+ pass, see [Evals](EVALS.md)).
63
63
 
64
64
  The conversion gate is the second ruler: the store's funnel metric over the
65
- criterion's window, compared to the **frozen baseline** (the RFC-0032
66
- `freeze_funnel_baseline`). Outcome is evidence — this gate can only say "the
65
+ criterion's window, compared to the **frozen baseline** (`freeze_funnel_baseline`). Outcome is evidence — this gate can only say "the
67
66
  store is measurably worse than the accepted state" or "there is nothing to
68
67
  compare against". It refuses on missing data, never passes: no frozen
69
68
  baseline, no criterion, no funnel store, a fold that has not converged — each