insika 0.7.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +191 -0
  3. data/README.md +9 -6
  4. data/bin/insika +44 -3
  5. data/docs/AGENTS.md +74 -14
  6. data/docs/API.md +73 -0
  7. data/docs/ARCHITECTURE.md +45 -44
  8. data/docs/ARTIFACTS.md +42 -0
  9. data/docs/CHANNELS.md +19 -2
  10. data/docs/CONTEXT.md +86 -38
  11. data/docs/DEPLOY.md +27 -7
  12. data/docs/EVALS.md +98 -8
  13. data/docs/FACTS.md +4 -0
  14. data/docs/KNOWLEDGE.md +7 -0
  15. data/docs/LOADTEST.md +15 -27
  16. data/docs/MEDIA.md +1 -1
  17. data/docs/OBSERVABILITY.md +48 -4
  18. data/docs/POLICY.md +14 -5
  19. data/docs/RELEASING.md +4 -0
  20. data/docs/RUNNING-LOCAL.md +2 -2
  21. data/docs/SECURITY.md +28 -2
  22. data/docs/SOAK.md +1 -1
  23. data/docs/TOOLS.md +150 -32
  24. data/docs/prompts/ADD-TOOL.md +12 -2
  25. data/docs/prompts/DIAGNOSE-TURN.md +3 -0
  26. data/docs/prompts/GO-LIVE.md +6 -4
  27. data/lib/insika/agent_profile.rb +47 -10
  28. data/lib/insika/channels/web/widget.js +33 -0
  29. data/lib/insika/channels/web.rb +5 -2
  30. data/lib/insika/chat_builder.rb +90 -37
  31. data/lib/insika/commands/agent_payload.rb +1 -1
  32. data/lib/insika/commands/run_distillation.rb +5 -8
  33. data/lib/insika/commands/seed_session.rb +118 -0
  34. data/lib/insika/compaction.rb +196 -0
  35. data/lib/insika/context/builder.rb +35 -11
  36. data/lib/insika/context/fragment.rb +4 -1
  37. data/lib/insika/context/priority.rb +8 -0
  38. data/lib/insika/context/provider.rb +5 -0
  39. data/lib/insika/context/providers/briefing.rb +61 -29
  40. data/lib/insika/context/providers/fence_notice.rb +27 -0
  41. data/lib/insika/context/providers/knowledge.rb +7 -4
  42. data/lib/insika/context/providers/memory.rb +8 -4
  43. data/lib/insika/context/providers/session.rb +50 -10
  44. data/lib/insika/context_trace_store.rb +11 -1
  45. data/lib/insika/doctor.rb +213 -10
  46. data/lib/insika/dsl/runtime.rb +5 -0
  47. data/lib/insika/dsl.rb +6 -0
  48. data/lib/insika/edge_limiter.rb +4 -1
  49. data/lib/insika/env_schema.rb +5 -6
  50. data/lib/insika/errors.rb +1 -0
  51. data/lib/insika/evals/assertions.rb +92 -6
  52. data/lib/insika/evals/golden.rb +91 -2
  53. data/lib/insika/evals/runner.rb +20 -0
  54. data/lib/insika/evals/simulator.rb +11 -2
  55. data/lib/insika/evals/transport.rb +118 -16
  56. data/lib/insika/evidence.rb +79 -12
  57. data/lib/insika/executor.rb +94 -23
  58. data/lib/insika/fence.rb +96 -0
  59. data/lib/insika/golden_store.rb +3 -0
  60. data/lib/insika/loop_detector.rb +5 -34
  61. data/lib/insika/mcp_store.rb +5 -2
  62. data/lib/insika/mcp_tool_registry.rb +8 -1
  63. data/lib/insika/memory_store.rb +12 -0
  64. data/lib/insika/overlay_tool_registry.rb +5 -0
  65. data/lib/insika/prefix_fingerprint.rb +32 -27
  66. data/lib/insika/profile_source.rb +8 -0
  67. data/lib/insika/server/app.rb +43 -1
  68. data/lib/insika/server/rack_app.rb +2 -0
  69. data/lib/insika/server/responses.rb +35 -8
  70. data/lib/insika/session_store.rb +38 -5
  71. data/lib/insika/settings_store.rb +18 -2
  72. data/lib/insika/soak/runner.rb +4 -4
  73. data/lib/insika/spoken_transcript.rb +31 -0
  74. data/lib/insika/studio/app.rb +34 -8
  75. data/lib/insika/studio/forms.rb +29 -3
  76. data/lib/insika/studio/views/_agent_tab_config.erb +5 -1
  77. data/lib/insika/studio/views/session.erb +1 -1
  78. data/lib/insika/studio/views/settings.erb +11 -0
  79. data/lib/insika/studio/views/tool_edit.erb +6 -2
  80. data/lib/insika/telemetry/recorder.rb +61 -1
  81. data/lib/insika/templates/daily-digest/README.md +9 -0
  82. data/lib/insika/templates/research-analyst/agent.rb +10 -0
  83. data/lib/insika/tool_assembly.rb +21 -13
  84. data/lib/insika/tool_batch.rb +67 -0
  85. data/lib/insika/tool_definition.rb +73 -10
  86. data/lib/insika/tool_envelope.rb +102 -2
  87. data/lib/insika/tool_store.rb +9 -4
  88. data/lib/insika/tool_trace_store.rb +1 -1
  89. data/lib/insika/tool_usage_report.rb +172 -0
  90. data/lib/insika/tools/data_defined_tool.rb +1 -0
  91. data/lib/insika/tools/present.rb +122 -0
  92. data/lib/insika/tools/run_persona_eval.rb +6 -1
  93. data/lib/insika/tools/tool_search.rb +4 -2
  94. data/lib/insika/turn_budget.rb +91 -0
  95. data/lib/insika/turn_state.rb +13 -1
  96. data/lib/insika/version.rb +1 -1
  97. data/lib/insika/wiring/graph.rb +7 -0
  98. data/lib/insika/wiring/graph_chat.rb +4 -0
  99. data/lib/insika.rb +11 -0
  100. metadata +10 -1
@@ -30,6 +30,7 @@ with one turn. Nothing more.
30
30
  | The need | The kind | Where it lives |
31
31
  |---|---|---|
32
32
  | Call an external HTTP API | **data tool** (`data_tool` in the DSL block) | a row in SQLite, editable at runtime |
33
+ | Show selected evidence cards | **presentation data tool** (`presentation`, no `request`) | the tool store; see [Tools](../TOOLS.md#presentation-tools-the-model-picks-ids-the-engine-shows-the-cards) |
33
34
  | Logic must run in-process | **code tool** (a Ruby class `< RubyLLM::Tool`) | the deployment image |
34
35
  | Adopt a whole external MCP server | **`mcp` instance** | durable config; its tools appear tagged `mcp:<name>` |
35
36
  | Teach a procedure (no data fetching) | **skill** (`skill "name", description:, instructions:`) | loads on demand via `load_skill` |
@@ -78,6 +79,12 @@ RULES:
78
79
  verbatim and arguments are checked against it at call time.
79
80
  - Author the FINAL url: the HTTP client does not follow redirects, and the egress guard
80
81
  cleared that host only.
82
+ - For a write that accepts catalog IDs, declare `requires_evidence` for those
83
+ parameters and ensure an allowed lookup declares `evidence`. Mark the write
84
+ `side_effect: true`; the engine serializes marked writes within a session.
85
+ - For presentation, follow the [card contract](../TOOLS.md#presentation-tools-the-model-picks-ids-the-engine-shows-the-cards):
86
+ pass known IDs to the presentation tool. Look up again if their cards are absent
87
+ from the session ledger or current information is needed.
81
88
  - Do not add a second capability "while we're here".
82
89
 
83
90
  ## Step 3 — Make sure it enters the tool-loop
@@ -96,8 +103,11 @@ instead of fighting it.
96
103
  ## Step 4 — Prove it with ONE turn
97
104
 
98
105
  Run one `reply()` whose message forces the call ("how many BRL is 1 USD right now?").
99
- The reply must use what the tool returned — if the model answers from imagination, the
100
- tool did not run: re-check Step 3 before touching the prompt.
106
+ Inspect the trace to confirm the call and result; the reply alone is not proof.
107
+ For a gated write, also verify that an unknown ID is blocked before any backend
108
+ request. For presentation, verify the selected `insika.ui` items. Use
109
+ [snapshot eval assertions](../EVALS.md#graders--what-a-turns-calls-and-reply-are-checked-against)
110
+ when the flow needs a repeatable precondition.
101
111
 
102
112
  ## Step 5 — Self-check
103
113
 
@@ -43,6 +43,9 @@ text), roughly **when**, and expected vs. actual. Reproduce once locally if chea
43
43
  | turn completed, customer got nothing | delivery is separate from the turn: check `channel_delivered` / `delivery_failed` | [Channels](../CHANNELS.md) |
44
44
  | freshly created agent returns empty turns | persona overflows the default `context_budget` (8000) | [Context](../CONTEXT.md) |
45
45
  | tool never called (or "missing") | not registered OR not allowed (`tools_allow`) | [Tools](../TOOLS.md) § troubleshooting |
46
+ | `tool_blocked` with `gate: provenance` | ID absent from session evidence; look up before retrying | [Tools](../TOOLS.md#provenance-checking-ids-before-a-write) |
47
+ | presentation shows no cards | inspect dropped IDs: `unknown`, `no_card`, `max`; a known ID may have no retained card | [API](../API.md#tool-and-presentation-sse-events) |
48
+ | cache reuse dropped | compare identity/tool-schema fingerprints with provider cache token accounting | [Context](../CONTEXT.md#the-provider-prefix-cache) |
46
49
  | identical `tool_call` repeated, then abort | the `max_tool_repeat` loop guard | [Agents](../AGENTS.md) § limits |
47
50
  | model gave up after one empty result | `tool_persistence` off (it is ON by default) | same |
48
51
  | `turn_stuck` event | the agent declared it cannot proceed — escalation signal, not a bug | [Agents](../AGENTS.md) § stuck |
@@ -40,7 +40,7 @@ convenience only:
40
40
  | Token | Gates | Rotating it |
41
41
  |---|---|---|
42
42
  | `ADMIN_TOKEN` | `/studio` login (the operator — just you) | safe, independent |
43
- | `OPENCLAW_GATEWAY_TOKEN` | Bearer for `/v1/responses` + `/v1/agents` (your API consumers) | both sides together, same step |
43
+ | `INSIKA_GATEWAY_TOKEN` | Bearer for `/v1/responses` + `/v1/agents` (your API consumers) | both sides together, same step |
44
44
 
45
45
  Generate each: `ruby -rsecurerandom -e 'puts SecureRandom.hex(24)'`. Set them as
46
46
  platform env vars. **Never** write either into a file, a commit, or your own output.
@@ -77,7 +77,7 @@ Any Docker host, same contract:
77
77
  ```bash
78
78
  docker build -t insika .
79
79
  docker run -p 9292:9292 -v insika-data:/data \
80
- -e DEEPSEEK_API_KEY=... -e ADMIN_TOKEN=... -e OPENCLAW_GATEWAY_TOKEN=... \
80
+ -e DEEPSEEK_API_KEY=... -e ADMIN_TOKEN=... -e INSIKA_GATEWAY_TOKEN=... \
81
81
  insika
82
82
  ```
83
83
 
@@ -87,12 +87,14 @@ In order, each with evidence:
87
87
 
88
88
  1. `curl https://<host>/up` → `{"status":"ok"}`.
89
89
  2. `bin/insika doctor` against the deployed volume (or via the platform's shell) —
90
- relay its findings verbatim; fix errors before continuing.
90
+ relay its findings verbatim; fix errors before continuing. Check evidence sources
91
+ for gated writes/presentation, review the fencing warning, and leave
92
+ `evals.seeding` off after snapshot evals. See [Deploy](../DEPLOY.md#strict-config-and-insika-doctor).
91
93
  3. One authenticated turn:
92
94
 
93
95
  ```bash
94
96
  curl -N https://<host>/v1/responses \
95
- -H "Authorization: Bearer $OPENCLAW_GATEWAY_TOKEN" \
97
+ -H "Authorization: Bearer $INSIKA_GATEWAY_TOKEN" \
96
98
  -H "Content-Type: application/json" \
97
99
  -d '{"model":"<agent-id>","user":"go-live-check","input":"hello"}'
98
100
  ```
@@ -57,13 +57,15 @@ module Insika
57
57
  :prompt_caching, # Anthropic prompt caching (R3): nil/false = OFF
58
58
  # (parity); true = ON. Same opt-in as `memory`. When ON
59
59
  # AND the resolved provider is Anthropic, ChatBuilder sets
60
- # ONE cache breakpoint at the end of the system block
61
- # (caches tools+system by the tools->system->messages
62
- # prefix order; immune to history eviction). PRE-AUDIT:
63
- # the system prompt MUST be byte-stable between turns —
64
- # a context provider injecting volatile content into
65
- # :system turns every turn into a paid cache WRITE with
66
- # no read hit. Enable only for stable-system agents.
60
+ # the cache breakpoint at the end of the IDENTITY layer
61
+ # of the system (prompt, skills, tool index); memory,
62
+ # knowledge and briefing render below it as a second
63
+ # block, so a per-turn change there never re-writes the
64
+ # cached prefix. Immune to history eviction too
65
+ # (tools->system->messages prefix order). What still
66
+ # costs a WRITE every turn: a provider declaring
67
+ # `layer :identity` while emitting per-turn bytes —
68
+ # the doctor's cache-layers check flags it.
67
69
  :tool_persistence, # the engine's "Tool discipline" block in the system
68
70
  # prompt (retry weak/empty tool results with a different
69
71
  # approach before giving up). THE ONE OPT-OUT FIELD:
@@ -81,6 +83,16 @@ module Insika
81
83
  # THE MODEL SEES: an older full result is only the first
82
84
  # occurrence; a model that wants an older detail re-calls
83
85
  # the tool. Cheap half of compaction for bloated histories.
86
+ :fencing, # third-party text is sanitized before the model reads
87
+ # it: nil/false = OFF (parity — bytes reach the model
88
+ # as-is); true = ON. Same opt-in as `memory`. When ON,
89
+ # every tool result (after the evidence reshape), every
90
+ # <memory> fact/note and every <knowledge> concept pass
91
+ # Insika::Fence (NFKC, invisible/control characters out,
92
+ # transcript- and tool-call-shaped tags removed, forged
93
+ # turn markers defused, per-leaf cap), and the FenceNotice
94
+ # sentence rides under the identity. Default OFF this
95
+ # release: the goldens were baselined on unfenced bytes.
84
96
  :params, # LLM generation params: a Hash with
85
97
  # temperature/max_tokens/thinking, applied to the chat at
86
98
  # stage 5. {} = provider defaults (parity).
@@ -322,7 +334,7 @@ module Insika
322
334
  policies: [], prompt_refs: [], limits: {}, approvals_required: nil,
323
335
  capabilities: nil, subagents: nil, tools_deferred: nil, memory: nil,
324
336
  prompt_caching: nil, tool_persistence: nil, tool_output_compression: nil,
325
- params: {}, model_policy: nil, guardrails: nil, sandbox: nil,
337
+ fencing: nil, params: {}, model_policy: nil, guardrails: nil, sandbox: nil,
326
338
  refinement: nil, capabilities_declared: nil, edge_stream: nil, metadata: {},
327
339
  budget: nil, reliability: nil, alerts: nil, routes: nil, stuck_signal: nil,
328
340
  outputs: nil, stt_prompt: nil, briefing_fields: nil, grounding: nil, funnel: nil,
@@ -333,7 +345,9 @@ outputs: nil, stt_prompt: nil, briefing_fields: nil, grounding: nil, funnel: nil
333
345
  tools_deny: Array(tools_deny), tools_allow_groups: tools_allow_groups, skills: skills,
334
346
  skills_eager: skills_eager,
335
347
  context_providers: context_providers, workflows_allow: workflows_allow,
336
- policies: Array(policies), prompt_refs: Array(prompt_refs),
348
+ policies: normalize_policies(policies, tools_allow: tools_allow, tools_deny: tools_deny,
349
+ tools_allow_groups: tools_allow_groups),
350
+ prompt_refs: Array(prompt_refs),
337
351
  limits: DEFAULT_LIMITS.merge(limits), approvals_required: approvals_required,
338
352
  capabilities: capabilities,
339
353
  # opt-in like capabilities: nil => NONE. Array-normalize a present value so
@@ -341,7 +355,7 @@ outputs: nil, stt_prompt: nil, briefing_fields: nil, grounding: nil, funnel: nil
341
355
  subagents: subagents.nil? ? nil : Array(subagents).map(&:to_s),
342
356
  tools_deferred: tools_deferred, memory: memory,
343
357
  prompt_caching: prompt_caching, tool_persistence: tool_persistence,
344
- tool_output_compression: tool_output_compression,
358
+ tool_output_compression: tool_output_compression, fencing: fencing,
345
359
  # The free-form hashes arrive with symbol keys (internal build) OR string
346
360
  # keys (StoredProfileSource JSON round-trip). Normalize to string keys ONCE
347
361
  # here — the single front door every profile passes through — so no reader
@@ -400,6 +414,29 @@ outputs: nil, stt_prompt: nil, briefing_fields: nil, grounding: nil, funnel: nil
400
414
  )
401
415
  end
402
416
 
417
+ # Declaring a tool allow/deny list IS opting into it. The list is only ever
418
+ # applied by the builtin `tool_allowlist` policy, and the Policy::Engine runs
419
+ # ONLY the policies a profile names — so a profile with `tools_allow: [a, b]`
420
+ # and no policies sent EVERY registered tool to the model, silently. A
421
+ # declared allowlist that does nothing is the failure mode; same rule as the
422
+ # "mcp:<name>" group auto-added to `tools_allow_groups` by the DSL.
423
+ #
424
+ # Presence, not emptiness, is the trigger for the two nil-able lists:
425
+ # `tools_allow: []` means "no tools" and must enforce just as hard.
426
+ # `tools_deny` has no nil state (it defaults to []), so only a non-empty
427
+ # deny list counts as a declaration.
428
+ #
429
+ # Appended, never prepended: profile-declared policies keep their order. The
430
+ # engine intersects allows and unions denies, so position changes only the
431
+ # audit order, not the outcome.
432
+ def self.normalize_policies(policies, tools_allow:, tools_deny:, tools_allow_groups:)
433
+ names = Array(policies)
434
+ declared = !tools_allow.nil? || !tools_allow_groups.nil? || !Array(tools_deny).empty?
435
+ return names unless declared && names.none? { |n| n.to_s == "tool_allowlist" }
436
+
437
+ names + ["tool_allowlist"]
438
+ end
439
+
403
440
  # nil/absent -> nil; a single Hash -> [Hash]; else an Array of Hashes —
404
441
  # deep-stringified so JSON round-trips stay stable.
405
442
  def self.normalize_schedules(list)
@@ -158,6 +158,11 @@
158
158
  log.scrollTop = log.scrollHeight;
159
159
  } else if (event === "working" && !bubbleEl && !note) {
160
160
  note = say("note", "working…");
161
+ } else if (event === "ui" && data.items && data.items.length) {
162
+ // A presentation tool picked cards: a plain list of caption + link.
163
+ // The host page restyles or replaces this; the protocol is the frame.
164
+ if (note) { note.remove(); note = null; }
165
+ show(data);
161
166
  } else if (event === "error") {
162
167
  if (note) { note.remove(); note = null; }
163
168
  throw new Error(data.message || "something went wrong");
@@ -257,6 +262,34 @@
257
262
 
258
263
  // --- helpers --------------------------------------------------------
259
264
 
265
+ function show(ui) {
266
+ var list = el("ul", "insika-m ui");
267
+ if (ui.title) {
268
+ var head = el("li", "title");
269
+ head.textContent = ui.title;
270
+ list.appendChild(head);
271
+ }
272
+ ui.items.forEach(function (item) {
273
+ var li = el("li", "");
274
+ var label = item.caption || item.id || item.url;
275
+ // Only http(s) becomes a link: the url is backend data, and a
276
+ // `javascript:` or `data:` scheme must never be one click away.
277
+ if (/^https?:\/\//i.test(item.url || "")) {
278
+ var a = el("a", "");
279
+ a.href = item.url;
280
+ a.target = "_blank";
281
+ a.rel = "noopener";
282
+ a.textContent = label;
283
+ li.appendChild(a);
284
+ } else {
285
+ li.textContent = label;
286
+ }
287
+ list.appendChild(li);
288
+ });
289
+ log.appendChild(list);
290
+ log.scrollTop = log.scrollHeight;
291
+ }
292
+
260
293
  function say(kind, content) {
261
294
  var node = el("div", "insika-m " + kind);
262
295
  node.textContent = content;
@@ -155,8 +155,9 @@ module Insika
155
155
  # path — `POST /messages` with an unknown id is a 404, not a new conversation.
156
156
  def mint_session_id = "#{@id}:#{SecureRandom.hex(16)}"
157
157
 
158
- # Turn Event -> SSE frame | nil. Four frames, which is the whole widget
159
- # protocol: what to type, what to say while a tool runs, and how it ended.
158
+ # Turn Event -> SSE frame | nil. Five frames, which is the whole widget
159
+ # protocol: what to type, what to say while a tool runs, what to show
160
+ # (a presentation tool's cards), and how it ended.
160
161
  #
161
162
  # `:intermediate` and `:thinking` are deliberately absent. `:content` is the
162
163
  # ANSWER — the model's narration on the way there is internal, and a
@@ -165,6 +166,8 @@ module Insika
165
166
  case event.type
166
167
  when :content then sse("delta", { delta: event.data[:delta].to_s })
167
168
  when :tool_call then sse("working", { name: event.data[:name].to_s })
169
+ when :ui then sse("ui", { component: event.data[:component].to_s, title: event.data[:title],
170
+ items: Array(event.data[:items]) })
168
171
  when :task_completed then sse("done", {})
169
172
  when :task_failed then sse("error", { message: event.data[:message].to_s })
170
173
  when :task_cancelled then sse("error", { message: "task cancelled" })
@@ -62,8 +62,7 @@ module Insika
62
62
 
63
63
  # Assembles the chat with the context (stage 2) and the Resolution's tools (stage 3).
64
64
  def configure_chat(chat, state)
65
- system = state.context.system.to_s
66
- apply_instructions(chat, system, state) unless system.empty?
65
+ apply_instructions(chat, state.context, state) unless state.context.system.to_s.empty?
67
66
 
68
67
  tools = Array(state.allowed_tools).dup
69
68
 
@@ -85,7 +84,8 @@ module Insika
85
84
  tools << Tools::ToolSearch.new(@tool_catalog, deferred_allowed, chat,
86
85
  tool_registry: @tool_registry,
87
86
  checkpoint_store: @checkpoint_store,
88
- event_stream: @event_stream, state: state)
87
+ event_stream: @event_stream, state: state,
88
+ trace_recorder: @tool_trace_store)
89
89
  end
90
90
 
91
91
  # load_skill is a system default (outside the allowlist), otherwise
@@ -263,32 +263,60 @@ module Insika
263
263
  ))
264
264
  end
265
265
 
266
- # R3: opt-in Anthropic prompt caching. When the agent enables
267
- # prompt_caching AND the resolved provider is Anthropic, wrap the system in
268
- # the provider's native Content helper with cache: true — ONE breakpoint at
269
- # the END of the system block. By Anthropic's prefix order
270
- # (tools -> system -> messages), a breakpoint on the last system block caches
271
- # tools + system together and is immune to history eviction (messages come
272
- # after it). RubyLLM::Content::Raw is Anthropic-specific: build_system_content
273
- # emits its blocks verbatim, so the cache_control rides along.
266
+ # Opt-in Anthropic prompt caching. When the agent enables prompt_caching
267
+ # AND the resolved provider is Anthropic, the system goes on the wire as TWO
268
+ # text blocks: the identity layer (prompt, skills, tool index) with the
269
+ # cache breakpoint at its end, then the volatile layer (memory, knowledge,
270
+ # briefing, request) plain. By Anthropic's prefix order
271
+ # (tools -> system -> messages) the breakpoint caches tools + identity
272
+ # together, immune to history eviction (messages come after it) AND to a
273
+ # memory fact or knowledge hit changing between turns (those bytes sit below
274
+ # the breakpoint, so they never enter the cached prefix). An empty volatile
275
+ # layer emits ONE block — byte-identical to the pre-split shape.
276
+ # RubyLLM::Content::Raw is Anthropic-specific: build_system_content emits
277
+ # its blocks verbatim, so the cache_control rides along.
274
278
  #
275
279
  # Any other case (caching off, or a non-Anthropic provider) uses the plain
276
280
  # string — OpenAI caches its prefix on its own; the Raw shape would confuse
277
281
  # non-Anthropic providers. The gem only supports MANUAL caching, and only for
278
282
  # Anthropic.
279
283
  #
280
- # PRE-AUDIT (why this is opt-in): the system must be BYTE-STABLE between turns
281
- # for a read hit. A context provider that injects volatile content into
282
- # :system (timestamps, per-turn data) makes every turn a paid cache WRITE
283
- # with no hit — worse than off. Enable only for stable-system agents.
284
- def apply_instructions(chat, system, state)
284
+ # What still breaks a read hit: a context provider that declares
285
+ # `layer :identity` and emits per-turn bytes (a timestamp, request data).
286
+ # The doctor's cache-layers check flags exactly that; the Studio cache tab
287
+ # shows it as `broke: <category>`.
288
+ def apply_instructions(chat, context, state)
285
289
  if state.profile.prompt_caching && anthropic_provider?(chat)
286
- chat.with_instructions(RubyLLM::Providers::Anthropic::Content.new(system, cache: true))
290
+ chat.with_instructions(RubyLLM::Providers::Anthropic::Content.new(parts: cache_blocks(context)))
287
291
  else
288
- chat.with_instructions(system)
292
+ chat.with_instructions(context.system.to_s)
289
293
  end
290
294
  end
291
295
 
296
+ # [{type:, text:, cache_control:?}] — identity block with the breakpoint,
297
+ # volatile block plain. Empty texts are skipped (Anthropic rejects an empty
298
+ # text block). `system` is authoritative: a package without the split (a
299
+ # custom builder's Struct) or one whose `system` was rewritten alone (an
300
+ # after_prompt hook doing `pkg.with(system: …)`) reads as all-identity —
301
+ # one block over the whole text, never bytes the hook did not put there.
302
+ def cache_blocks(context)
303
+ identity, volatile = system_layers(context)
304
+ blocks = []
305
+ blocks << { type: "text", text: identity, cache_control: { type: "ephemeral" } } unless identity.empty?
306
+ blocks << { type: "text", text: volatile } unless volatile.empty?
307
+ blocks
308
+ end
309
+
310
+ def system_layers(context)
311
+ system = context.system.to_s
312
+ return [system, ""] unless context.respond_to?(:system_volatile)
313
+
314
+ identity = context.system_identity.to_s
315
+ volatile = context.system_volatile.to_s
316
+ joined = [identity, volatile].reject(&:empty?).join("\n\n")
317
+ joined == system ? [identity, volatile] : [system, ""]
318
+ end
319
+
292
320
  # The RESOLVED provider (chat.model.provider is the slug string, e.g.
293
321
  # "anthropic"), authoritative even when the agent left provider nil and
294
322
  # RubyLLM inferred it from the model id. Any surface without a model (fakes,
@@ -332,20 +360,29 @@ module Insika
332
360
  end
333
361
 
334
362
  # RubyLLM's additive callbacks become events. load_skill becomes
335
- # :skill_activated. Adds the max_tool_calls counter: the loop is RubyLLM's;
336
- # here we only count and abort.
363
+ # :skill_activated. Hangs the turn's TurnBudget on them: the loop is
364
+ # RubyLLM's; here we only count, warn and abort.
337
365
  def wire_callbacks(chat, state, emit)
338
- # Per-TURN counter: safe as a closure local even under concurrent tool calls
339
- # (MRI fibers do not preempt between the read and the write). The per-CALL
340
- # correlation is NOT — it lives in fiber storage behind TurnState, because
341
- # each call gets its own fiber once tool concurrency is on.
342
- tool_calls = 0
343
- max_tool_calls = state.profile.limits[:max_tool_calls] || 50
366
+ # The turn's ONE tool-call counter. Safe as a closure-held object even
367
+ # under concurrent tool calls (MRI fibers do not preempt between the read
368
+ # and the write). The per-CALL correlation is NOT — it lives in fiber
369
+ # storage behind TurnState, because each call gets its own fiber once tool
370
+ # concurrency is on.
371
+ #
372
+ # It counts, warns at 10/5/2 calls remaining, and raises the
373
+ # `stage: :tool_limit` abort — the guard-rail that used to be inline here.
374
+ # Announcing the budget needs #add_message at the batch boundary; a chat
375
+ # without it (smoke shim, minimal double) still gets the count and the
376
+ # abort, just no notice.
377
+ budget = Insika::TurnBudget.new(
378
+ chat: chat, max: state.profile.limits[:max_tool_calls] || 50, emit: emit
379
+ )
344
380
 
345
381
  # the loop detector. Needs #after_message + #add_message for the
346
382
  # batch-boundary intervention; a chat without them (smoke shim, minimal
347
383
  # double) stays bounded by max_tool_calls alone — never half-wired.
348
- detector = if %i[after_message add_message].all? { |m| chat.respond_to?(m) }
384
+ appendable = %i[after_message add_message].all? { |m| chat.respond_to?(m) }
385
+ detector = if appendable
349
386
  repeat = state.profile.limits[:max_tool_repeat] || Insika::AgentProfile::DEFAULT_LIMITS[:max_tool_repeat]
350
387
  Insika::LoopDetector.new(chat: chat, limit: repeat, emit: emit) if repeat >= 2
351
388
  end
@@ -355,11 +392,7 @@ module Insika
355
392
  state.current_tool_call = tool_call
356
393
  # max_tool_calls guard-rail: stays inline (not as a registered hook)
357
394
  # because Hooks is shared across turns and has no unregister.
358
- tool_calls += 1
359
- if tool_calls > max_tool_calls
360
- raise Insika::TimeoutError.new("tool call limit exceeded (#{max_tool_calls})",
361
- stage: :tool_limit)
362
- end
395
+ budget.tool_call
363
396
 
364
397
  # AFTER the count, BEFORE the call runs — a post-warning
365
398
  # repeat raises here, so the stubborn loop pays for no extra call.
@@ -386,7 +419,7 @@ module Insika
386
419
  args = tool_call.arguments || {}
387
420
  emit.call(:knowledge_retrieved, { name: args["name"] || args[:name], agent: state.profile.id })
388
421
  else
389
- emit.call(:tool_call, { name: tool_call.name, arguments: tool_call.arguments })
422
+ emit.call(:tool_call, { name: tool_call.name, arguments: tool_call.arguments, call_id: tool_call.id }.compact)
390
423
  end
391
424
  end
392
425
 
@@ -394,13 +427,33 @@ module Insika
394
427
  # the RAW result — the only place a Tool::Halt (halt_when) is
395
428
  # still recognizable, and a halted batch must receive no intervention.
396
429
  detector&.tool_result(result)
430
+ budget.tool_result(result)
397
431
  result = @hooks.run_after(:tool, result)
398
- emit.call(:tool_result, { name: state.current_tool_name, result: result.to_s })
432
+ emit.call(:tool_result, { name: state.current_tool_name, call_id: state.current_tool_call&.id,
433
+ result: result.to_s }.compact.merge(tool_outcome(result)))
399
434
  end
400
435
 
401
- # the intervention appends at the batch boundary (the Nth tool
402
- # result closing) — never between two tool results of one batch.
403
- chat.after_message { |message| detector.message_ended(message) } if detector
436
+ # both appends land at the batch boundary (the Nth tool result closing) —
437
+ # never between two tool results of one batch.
438
+ return unless appendable
439
+
440
+ chat.after_message do |message|
441
+ detector&.message_ended(message)
442
+ budget.message_ended(message)
443
+ end
444
+ end
445
+
446
+ # How the call ENDED, for the :tool_result event (the edge publishes it and the
447
+ # evals grade it): the envelope's `{error:}` -> "error"; a gate that held the
448
+ # call (ToolEnvelope::Blocked) -> "blocked" + which gate; anything else ran ->
449
+ # "ok" — including a tool whose OWN answer happens to say "blocked". Read off
450
+ # the RAW result, before it is stringified for the event.
451
+ def tool_outcome(result)
452
+ return { status: "blocked", gate: result["gate"]&.to_s }.compact if result.is_a?(Insika::ToolEnvelope::Blocked)
453
+ return { status: "ok" } unless result.is_a?(Hash)
454
+ return { status: "error" } if result.key?(:error) || result.key?("error")
455
+
456
+ { status: "ok" }
404
457
  end
405
458
  end
406
459
  end
@@ -12,7 +12,7 @@ module Insika
12
12
  FIELDS = %i[id model provider base_prompt prompt_files tools_allow tools_deny
13
13
  tools_allow_groups skills skills_eager context_providers workflows_allow policies
14
14
  prompt_refs limits approvals_required capabilities subagents tools_deferred
15
- memory prompt_caching tool_persistence tool_output_compression budget reliability alerts
15
+ memory prompt_caching tool_persistence tool_output_compression fencing budget reliability alerts
16
16
  routes stuck_signal outputs stt_prompt briefing_fields grounding funnel followup
17
17
  params model_policy guardrails refinement capabilities_declared
18
18
  edge_stream metadata distill harvest knowledge schedules].freeze
@@ -148,8 +148,8 @@ module Insika
148
148
  !applied.nil? && applied.value == value && applied.origin.to_s.start_with?("distilled:")
149
149
  end
150
150
 
151
- # The prompt gets the transcript slice (masked through the
152
- # output filter first — the redaction rule), the
151
+ # The prompt gets the transcript slice (user/assistant only, masked
152
+ # through the output filter — the redaction rule), the
153
153
  # customer's CURRENT facts (so the model can avoid re-proposing applied
154
154
  # facts), and the answer rules (the pack prompt or DEFAULT_PROMPT).
155
155
  def build_prompt(config, session, baseline)
@@ -169,12 +169,9 @@ module Insika
169
169
  PROMPT
170
170
  end
171
171
 
172
- def render_transcript(messages)
173
- redacted, = Insika::Safety::Detectors.redact(
174
- messages.each_with_index.map { |m, i| "[#{i}] #{m['role']}: #{m['content']}" }.join("\n")
175
- )
176
- redacted
177
- end
172
+ # Only what people said — a `role: tool` message (a product description,
173
+ # a search result) is never a candidate customer fact.
174
+ def render_transcript(messages) = Insika::SpokenTranscript.render(messages)
178
175
 
179
176
  def utility_model
180
177
  return nil unless @settings_store
@@ -0,0 +1,118 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "time"
4
+
5
+ module Insika
6
+ module Commands
7
+ # Control command: loads a SNAPSHOT into a conversation before its first
8
+ # turn — the precondition an eval case starts from ("the customer already saw
9
+ # three products", "a stored preference"), without replaying the turns that
10
+ # would have produced it. Creates the session when it does not exist yet, then
11
+ # writes the four kinds of state a turn reads:
12
+ #
13
+ # history: [{ role:, content: }] -> the transcript, stamped `origin: engine`
14
+ # (nobody typed these; a report must not read them as the customer)
15
+ # evidence: { ids: [], cards: [] } -> the session evidence ledger (cards
16
+ # carry an `id`; a presentation tool shows them)
17
+ # memory: { facts: {}, notes: [] } -> the memory cell THIS turn will read
18
+ # briefing: { fields: {} } -> the session briefing
19
+ #
20
+ # Refuses a session that already has messages (ConflictError -> 409): seeding a
21
+ # used conversation is a test bug, not a merge — the case would assert against a
22
+ # state nobody wrote down. Synchronous; does not create a Task. -> Session.
23
+ class SeedSession
24
+ def initialize(session_store:, memory_store:, event_stream:)
25
+ @session_store = session_store
26
+ @memory_store = memory_store
27
+ @event_stream = event_stream
28
+ end
29
+
30
+ def call(command)
31
+ p = AgentPayload.symbolize(command.payload)
32
+ id = AgentPayload.presence(p[:id])
33
+ raise Insika::ValidationError, "id is required" if id.nil?
34
+
35
+ state = Coercion.deep_stringify(p[:state] || {})
36
+ raise Insika::ValidationError, "state must be a Hash" unless state.is_a?(Hash)
37
+
38
+ tenant = command.meta[:tenant] # the same source the turn reads — never the payload
39
+ customer = AgentPayload.presence(p[:customer])
40
+
41
+ session = @session_store.find(id) || @session_store.create(id: id, vars: {})
42
+ unless session.messages.empty?
43
+ raise Insika::ConflictError,
44
+ "session #{id} already has #{session.messages.size} message(s) — seed a fresh conversation"
45
+ end
46
+
47
+ seed_history(id, state["history"])
48
+ seed_evidence(id, state["evidence"])
49
+ seed_memory(memory_scope(tenant, customer, id), state["memory"])
50
+ seed_briefing(id, state["briefing"])
51
+
52
+ @event_stream.emit(Insika::Event.new(
53
+ type: :session_seeded,
54
+ data: { session_id: id, tenant: tenant, keys: state.keys }.compact,
55
+ meta: { session_id: id, at: Time.now.utc.iso8601 }
56
+ ))
57
+ @session_store.find(id)
58
+ end
59
+
60
+ private
61
+
62
+ def seed_history(id, history)
63
+ return if history.nil?
64
+ raise Insika::ValidationError, "state.history must be an array" unless history.is_a?(Array)
65
+
66
+ messages = history.map do |m|
67
+ raise Insika::ValidationError, "state.history entries must be { role:, content: }" unless m.is_a?(Hash)
68
+
69
+ role = m["role"].to_s
70
+ unless %w[user assistant].include?(role)
71
+ raise Insika::ValidationError, "state.history role must be user or assistant (got #{role.inspect})"
72
+ end
73
+
74
+ { "role" => role, "content" => m["content"].to_s, "origin" => Insika::MessageOrigin::ENGINE }
75
+ end
76
+ @session_store.append_messages(id, messages) unless messages.empty?
77
+ end
78
+
79
+ def seed_evidence(id, evidence)
80
+ return if evidence.nil?
81
+ raise Insika::ValidationError, "state.evidence must be { ids: [], cards: [] }" unless evidence.is_a?(Hash)
82
+
83
+ ids = Array(evidence["ids"]).map(&:to_s).reject(&:empty?)
84
+ cards = Insika::Evidence.valid_attachments(evidence["cards"]).select { |c| c["id"] }
85
+ # a card's id counts as seen: the card came from a search the case declares
86
+ ids = (ids + cards.map { |c| c["id"] }).uniq
87
+ @session_store.append_evidence(id, ids: ids, ungrounded: 0, cards: cards) unless ids.empty?
88
+ end
89
+
90
+ def seed_memory(scope, memory)
91
+ return if memory.nil?
92
+ raise Insika::ValidationError, "state.memory must be { facts: {}, notes: [] }" unless memory.is_a?(Hash)
93
+
94
+ facts = memory["facts"] || {}
95
+ raise Insika::ValidationError, "state.memory.facts must be a Hash" unless facts.is_a?(Hash)
96
+
97
+ facts.each { |key, value| @memory_store.put_fact(tenant: scope, key: key, value: value, origin: "operator") }
98
+ Array(memory["notes"]).each { |text| @memory_store.add_note(tenant: scope, text: text.to_s) }
99
+ end
100
+
101
+ def seed_briefing(id, briefing)
102
+ return if briefing.nil?
103
+ raise Insika::ValidationError, "state.briefing must be { fields: {} }" unless briefing.is_a?(Hash)
104
+
105
+ fields = briefing["fields"] || {}
106
+ raise Insika::ValidationError, "state.briefing.fields must be a Hash" unless fields.is_a?(Hash)
107
+
108
+ fields.each { |field, value| @session_store.update_briefing(id, field: field, value: value) }
109
+ end
110
+
111
+ # The SAME cell the turn will read — one rule, MemoryStore.scope_for. Seeding
112
+ # any other cell would pass the case against a memory the model never sees.
113
+ def memory_scope(tenant, customer, session_id)
114
+ MemoryStore.scope_for(tenant: tenant, customer: customer, session_id: session_id)
115
+ end
116
+ end
117
+ end
118
+ end