insika 0.8.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +75 -0
  3. data/README.md +5 -3
  4. data/bin/insika +1 -1
  5. data/docs/AGENTS.md +52 -11
  6. data/docs/API.md +73 -0
  7. data/docs/ARCHITECTURE.md +45 -44
  8. data/docs/CHANNELS.md +19 -2
  9. data/docs/CONTEXT.md +33 -27
  10. data/docs/DEPLOY.md +13 -2
  11. data/docs/EVALS.md +98 -8
  12. data/docs/FACTS.md +4 -0
  13. data/docs/KNOWLEDGE.md +7 -0
  14. data/docs/OBSERVABILITY.md +21 -6
  15. data/docs/POLICY.md +4 -1
  16. data/docs/RELEASING.md +4 -0
  17. data/docs/SECURITY.md +27 -1
  18. data/docs/TOOLS.md +127 -33
  19. data/docs/prompts/ADD-TOOL.md +12 -2
  20. data/docs/prompts/DIAGNOSE-TURN.md +3 -0
  21. data/docs/prompts/GO-LIVE.md +3 -1
  22. data/lib/insika/agent_profile.rb +21 -9
  23. data/lib/insika/channels/web/widget.js +33 -0
  24. data/lib/insika/channels/web.rb +5 -2
  25. data/lib/insika/chat_builder.rb +62 -20
  26. data/lib/insika/commands/agent_payload.rb +1 -1
  27. data/lib/insika/commands/run_distillation.rb +5 -8
  28. data/lib/insika/commands/seed_session.rb +118 -0
  29. data/lib/insika/context/builder.rb +29 -9
  30. data/lib/insika/context/priority.rb +2 -0
  31. data/lib/insika/context/provider.rb +5 -0
  32. data/lib/insika/context/providers/briefing.rb +11 -8
  33. data/lib/insika/context/providers/fence_notice.rb +27 -0
  34. data/lib/insika/context/providers/knowledge.rb +7 -4
  35. data/lib/insika/context/providers/memory.rb +8 -4
  36. data/lib/insika/context/providers/session.rb +7 -3
  37. data/lib/insika/doctor.rb +109 -1
  38. data/lib/insika/dsl/runtime.rb +1 -0
  39. data/lib/insika/dsl.rb +6 -0
  40. data/lib/insika/edge_limiter.rb +4 -1
  41. data/lib/insika/errors.rb +1 -0
  42. data/lib/insika/evals/assertions.rb +92 -6
  43. data/lib/insika/evals/golden.rb +91 -2
  44. data/lib/insika/evals/runner.rb +20 -0
  45. data/lib/insika/evals/simulator.rb +11 -2
  46. data/lib/insika/evals/transport.rb +117 -15
  47. data/lib/insika/evidence.rb +79 -12
  48. data/lib/insika/executor.rb +30 -23
  49. data/lib/insika/fence.rb +96 -0
  50. data/lib/insika/golden_store.rb +3 -0
  51. data/lib/insika/mcp_store.rb +5 -2
  52. data/lib/insika/mcp_tool_registry.rb +8 -1
  53. data/lib/insika/memory_store.rb +12 -0
  54. data/lib/insika/overlay_tool_registry.rb +5 -0
  55. data/lib/insika/prefix_fingerprint.rb +32 -27
  56. data/lib/insika/profile_source.rb +1 -0
  57. data/lib/insika/server/app.rb +43 -1
  58. data/lib/insika/server/rack_app.rb +2 -0
  59. data/lib/insika/server/responses.rb +31 -4
  60. data/lib/insika/session_store.rb +4 -1
  61. data/lib/insika/settings_store.rb +10 -1
  62. data/lib/insika/spoken_transcript.rb +31 -0
  63. data/lib/insika/studio/app.rb +6 -2
  64. data/lib/insika/studio/forms.rb +18 -3
  65. data/lib/insika/studio/views/_agent_tab_config.erb +5 -1
  66. data/lib/insika/studio/views/session.erb +1 -1
  67. data/lib/insika/studio/views/tool_edit.erb +6 -2
  68. data/lib/insika/telemetry/recorder.rb +13 -1
  69. data/lib/insika/tool_assembly.rb +21 -13
  70. data/lib/insika/tool_definition.rb +73 -10
  71. data/lib/insika/tool_envelope.rb +102 -2
  72. data/lib/insika/tool_store.rb +9 -4
  73. data/lib/insika/tool_trace_store.rb +1 -1
  74. data/lib/insika/tool_usage_report.rb +12 -2
  75. data/lib/insika/tools/data_defined_tool.rb +1 -0
  76. data/lib/insika/tools/present.rb +122 -0
  77. data/lib/insika/tools/run_persona_eval.rb +6 -1
  78. data/lib/insika/tools/tool_search.rb +4 -2
  79. data/lib/insika/turn_state.rb +13 -1
  80. data/lib/insika/version.rb +1 -1
  81. data/lib/insika/wiring/graph.rb +7 -0
  82. data/lib/insika/wiring/graph_chat.rb +4 -0
  83. data/lib/insika.rb +4 -0
  84. metadata +6 -1
@@ -39,7 +39,7 @@ module Insika
39
39
  source: id
40
40
  )
41
41
  end
42
- compaction ? [compaction_fragment(compaction)] + fragments : fragments
42
+ compaction ? [compaction_fragment(compaction, request)] + fragments : fragments
43
43
  end
44
44
 
45
45
  private
@@ -127,10 +127,14 @@ module Insika
127
127
  # provider-agnostic (a mid-history "system" message is not). Priority
128
128
  # COMPACTION (59): the "oldest unit" — under budget it drops before any
129
129
  # verbatim message. source "compaction" -> its own context-trace category.
130
- def compaction_fragment(state)
130
+ # Fenced when the agent has `fencing` on: the summary is model-written from
131
+ # customer text, and the notice promises the model this block is material.
132
+ def compaction_fragment(state, request)
133
+ summary = state["summary"].to_s
134
+ summary = Insika::Fence.sanitize_text(summary) if Insika::Fence.enabled?(request.profile)
131
135
  ContextFragment.build(
132
136
  content: { role: "user",
133
- content: "<conversation_summary>\n#{state['summary']}\n</conversation_summary>" },
137
+ content: "<conversation_summary>\n#{summary}\n</conversation_summary>" },
134
138
  placement: :history,
135
139
  priority: Context::Priority::COMPACTION,
136
140
  source: "compaction"
data/lib/insika/doctor.rb CHANGED
@@ -180,7 +180,8 @@ module Insika
180
180
  check_soak_envelope check_turn_timing check_grounding check_cache_layers
181
181
  check_memory_scopes check_funnel_declarations check_followup check_distill
182
182
  check_compaction check_harvest check_schedules check_guardrail_corpora
183
- check_tool_allowlist_policy]
183
+ check_tool_allowlist_policy check_fencing check_eval_seeding
184
+ check_presentation_tools check_provenance]
184
185
 
185
186
  def safe(check)
186
187
  Array(send(check))
@@ -836,6 +837,7 @@ def wrapped_content?(content) = /\A\s*\{\s*"[^"]+"\s*=>/.match?(content.to_s)
836
837
  # identity block bills a cache write every turn).
837
838
  IDENTITY_BUILTINS = %w[
838
839
  Insika::Context::Providers::Prompt
840
+ Insika::Context::Providers::FenceNotice
839
841
  Insika::Context::Providers::Skill
840
842
  Insika::Context::Providers::ToolSearch
841
843
  ].freeze
@@ -942,6 +944,71 @@ def wrapped_content?(content) = /\A\s*\{\s*"[^"]+"\s*=>/.match?(content.to_s)
942
944
 
943
945
  def names_tool_allowlist?(raw) = Array(raw["policies"]).any? { |p| p.to_s == "tool_allowlist" }
944
946
 
947
+ # A stored agent that allows a PRESENTATION tool but no tool declaring
948
+ # `evidence`. The presentation tool shows only cards an evidence tool returned
949
+ # this session, so with no evidence tool in reach every call drops every id —
950
+ # the model keeps asking to show products and nothing ever appears. Silent
951
+ # unless a presentation tool exists. Data tools only: a code tool's evidence
952
+ # lives in registry metadata the doctor does not read.
953
+ def check_presentation_tools
954
+ return [] unless @tool_store && @profile_source
955
+
956
+ raws = @tool_store.all_raw
957
+ presenting = raws.select { |r| r["presentation"] }.map { |r| r["name"].to_s }
958
+ return [] if presenting.empty?
959
+
960
+ evidence = raws.select { |r| r["evidence"] }.map { |r| r["name"].to_s }
961
+ bad = @profile_source.all.filter_map do |profile|
962
+ allowed = allowed_data_tools(profile, raws)
963
+ shows = presenting & allowed
964
+ next if shows.empty? || !(evidence & allowed).empty?
965
+
966
+ Finding.new(check: "presentation-tools", severity: :warn, fix: nil,
967
+ message: "agent '#{profile.id}' allows #{shows.join(', ')} but no tool declaring " \
968
+ "evidence — a presentation tool can only show cards an evidence tool " \
969
+ "returned this session, so every call would drop every id. Allow the " \
970
+ "search tool that declares `evidence`, or drop the presentation tool.")
971
+ end
972
+ return bad unless bad.empty?
973
+
974
+ [ok("presentation-tools", "#{presenting.length} presentation tool(s): every agent that shows cards " \
975
+ "also allows an evidence tool")]
976
+ end
977
+
978
+ # The data tools a profile may call, by the allowlist's own rules: nil lists =
979
+ # all; otherwise names ∪ groups; deny wins.
980
+ def allowed_data_tools(profile, raws)
981
+ names = profile.tools_allow.nil? ? nil : Array(profile.tools_allow).map(&:to_s)
982
+ groups = profile.tools_allow_groups.nil? ? nil : Array(profile.tools_allow_groups).map(&:to_s)
983
+ allowed = if names.nil? && groups.nil?
984
+ raws.map { |r| r["name"].to_s }
985
+ else
986
+ Array(names) | raws.select { |r| Array(groups).include?(r["group"].to_s) }.map { |r| r["name"].to_s }
987
+ end
988
+ allowed - Array(profile.tools_deny).map(&:to_s)
989
+ end
990
+
991
+ def check_provenance
992
+ return [] unless @tool_store && @profile_source.respond_to?(:all_raw)
993
+
994
+ tools = @tool_store.all_raw
995
+ @profile_source.all_raw.filter_map do |raw|
996
+ allowed = tools.select do |tool|
997
+ name = tool["name"]
998
+ unrestricted = raw["tools_allow"].nil? && raw["tools_allow_groups"].nil?
999
+ (unrestricted || Array(raw["tools_allow"]).include?(name) ||
1000
+ Array(raw["tools_allow_groups"]).include?(tool["group"])) &&
1001
+ !Array(raw["tools_deny"]).include?(name)
1002
+ end
1003
+ gated = allowed.select { |tool| tool["requires_evidence"] }
1004
+ next if gated.empty? || allowed.any? { |tool| tool["evidence"] }
1005
+
1006
+ Finding.new(check: "provenance", severity: :warn, fix: nil,
1007
+ message: "agent '#{raw["id"]}' allows #{gated.map { |t| t["name"] }.join(', ')} " \
1008
+ "with requires_evidence but no stored evidence tool — verify that an allowed code tool supplies evidence")
1009
+ end
1010
+ end
1011
+
945
1012
  def check_grounding
946
1013
  return [] unless @profile_source
947
1014
 
@@ -1325,6 +1392,47 @@ def wrapped_content?(content) = /\A\s*\{\s*"[^"]+"\s*=>/.match?(content.to_s)
1325
1392
  end
1326
1393
  end
1327
1394
 
1395
+ # `evals.seeding` opens POST /v1/conversations/:id/seed: a snapshot written
1396
+ # into a real conversation under the tenant token, before its first turn. Right
1397
+ # on the machine running snapshot evals; wrong left on in production, where any
1398
+ # consumer holding the token could write a customer's history. A warning, not
1399
+ # an error — the operator may be running the evals right now.
1400
+ def check_eval_seeding
1401
+ return [] unless @settings_store
1402
+
1403
+ on = ((@settings_store.get["evals"] || {})["seeding"]) == true
1404
+ return [ok("eval-seeding", "evals.seeding off — conversations cannot be seeded")] unless on
1405
+
1406
+ [Finding.new(check: "eval-seeding", severity: :warn, fix: nil,
1407
+ message: "evals.seeding is ON — POST /v1/conversations/:id/seed accepts a fabricated " \
1408
+ "conversation state under the tenant token. Fine while running snapshot " \
1409
+ "evals; turn it off (Studio > Settings, or `update_settings`) in production.")]
1410
+ end
1411
+
1412
+ # A customer-facing agent — one reachable through an inbound channel: any
1413
+ # agent when the relay is mounted (the event names the agent), the listed
1414
+ # ones for the widget — with `fencing` off reads third-party bytes as-is.
1415
+ # A warning, never an error: the default is off this release on purpose
1416
+ # (the goldens were baselined unfenced), so the message says exactly what
1417
+ # is unsanitized today. No inbound channel = nothing customer-facing = silent.
1418
+ def check_fencing
1419
+ return [] unless @profile_source
1420
+
1421
+ relay = Insika::Coercion.present?(@env["INSIKA_RELAY_TOKEN"])
1422
+ widget = @env["INSIKA_WIDGET_AGENTS"].to_s.split(",").map(&:strip).reject(&:empty?)
1423
+ return [] unless relay || widget.any?
1424
+
1425
+ exposed = @profile_source.all.select { |p| relay || widget.include?(p.id.to_s) }
1426
+ unfenced = exposed.reject { |p| Insika::Fence.enabled?(p) }.map(&:id)
1427
+ return [ok("fencing", "fencing on for every agent behind an inbound channel")] if unfenced.empty?
1428
+
1429
+ [Finding.new(check: "fencing", severity: :warn, fix: nil,
1430
+ message: "agent(s) #{unfenced.join(', ')}: reachable through an inbound channel with " \
1431
+ "`fencing` off — tool results, <memory> facts and <knowledge> concepts reach " \
1432
+ "the model byte-for-byte (invisible characters, forged turn markers and " \
1433
+ "transcript-shaped tags included). Set `fencing true` on the agent.")]
1434
+ end
1435
+
1328
1436
  # the harvest check — per profile WITH a harvest hash:
1329
1437
  # declared-without-model warn (D12), no grounding matcher warn (D3),
1330
1438
  # malformed negative list error (D4), else ok with the pending counts.
@@ -231,6 +231,7 @@ module Insika
231
231
  base: "", catalog: c[:prompt_catalog],
232
232
  agent_files: c[:agent_file_store], system_files: c[:system_file_store]
233
233
  ),
234
+ Insika::Context::Providers::FenceNotice.new,
234
235
  Insika::Context::Providers::Skill.new(catalog: c[:skill_catalog]),
235
236
  Insika::Context::Providers::SkillTrigger.new(catalog: c[:skill_catalog]),
236
237
  Insika::Context::Providers::ToolSearch.new(catalog: c[:tool_catalog]),
data/lib/insika/dsl.rb CHANGED
@@ -417,6 +417,12 @@ module Insika
417
417
  # SEES: repeated identical tool results collapse to a back-reference.
418
418
  def tool_output_compression(on = true) = @config[:tool_output_compression] = on
419
419
 
420
+ # Third-party text is sanitized before the model reads it (tool results,
421
+ # memory, knowledge) and a one-sentence notice names those blocks as data.
422
+ # Opt-in this release — the goldens were baselined on unfenced bytes:
423
+ # fencing true
424
+ def fencing(on = true) = @config[:fencing] = on
425
+
420
426
  # Content-safety guardrails — opt-in and configurable per agent.
421
427
  # Pure config-over-code: the hash is stored on the profile and consumed by
422
428
  # Safety::Config.from_profile. Merges, so repeated calls accumulate.
@@ -263,12 +263,15 @@ module Insika
263
263
 
264
264
  # Appends the note to the assembled system prompt: the real Data package is
265
265
  # immutable (with), the specs' minimal Struct is mutable — both duck-typed.
266
+ # The note is per-turn data, so it also lands in the volatile layer — below
267
+ # the cache breakpoint, never inside the cached identity prefix.
266
268
  def inject_budget_note(state, note)
267
269
  ctx = state.context
268
270
  return if ctx.nil?
269
271
 
270
272
  if ctx.respond_to?(:with)
271
- state.context = ctx.with(system: "#{ctx.system}\n\n#{note}")
273
+ volatile = [ctx.system_volatile, note].reject { |s| s.to_s.empty? }.join("\n\n")
274
+ state.context = ctx.with(system: "#{ctx.system}\n\n#{note}", system_volatile: volatile)
272
275
  elsif ctx.respond_to?(:system=)
273
276
  ctx.system = "#{ctx.system}\n\n#{note}"
274
277
  end
data/lib/insika/errors.rb CHANGED
@@ -8,6 +8,7 @@ module Insika
8
8
 
9
9
  class ValidationError < Error; end # Malformed Command -> HTTP 422, no Task created
10
10
  class NotFoundError < Error; end # nonexistent session/task/agent -> HTTP 404
11
+ class ConflictError < Error; end # the write contradicts state that already exists -> HTTP 409
11
12
 
12
13
  # Policy Engine denied -> :policy_denied event, task :failed
13
14
  class PolicyDenied < Error
@@ -13,14 +13,35 @@ module Insika
13
13
  # offline without a server.
14
14
  #
15
15
  # output_text: the final assistant text (for content checks)
16
- # tool_calls: [{ "name" =>, "status" => }] captured from the stream's tool events
16
+ # tool_calls: [{ "name" =>, "arguments" =>, "status" =>, "gate" => }] captured
17
+ # from the stream's tool events. `arguments` (a Hash) and
18
+ # `status` ("ok" | "error" | "blocked", with `gate` naming what
19
+ # held a blocked call) arrive from the item frames; an older
20
+ # deployment sends names only and they stay nil.
21
+ # ui: [{ "component" =>, "count" => }] — one per presentation call
22
+ # (the `insika.ui` frames); nil/[] when the turn showed nothing
17
23
  # error: transport/turn error string, or nil on a clean turn
18
- TurnResult = Struct.new(:output_text, :tool_calls, :error, keyword_init: true) do
24
+ TurnResult = Struct.new(:output_text, :tool_calls, :ui, :error, keyword_init: true) do
19
25
  def tool_names = Array(tool_calls).map { |t| (t["name"] || t[:name]).to_s }
20
26
 
21
- # A tool call whose status is anything but a success ("ok"/2xx/"success").
27
+ # Components that actually put a card in front of the customer (count > 0).
28
+ def shown_components
29
+ Array(ui).select { |u| (u["count"] || u[:count]).to_i.positive? }
30
+ .map { |u| (u["component"] || u[:component]).to_s }
31
+ end
32
+
33
+ # A tool call whose status is anything but a success ("ok"/2xx/"success"). A
34
+ # BLOCKED call is not an error: the tool never ran, a gate held it — that is
35
+ # the `blocked_gates` grader's business, not `must_not: tool_error`'s.
22
36
  def errored_tools
23
- Array(tool_calls).reject { |t| Assertions.ok_status?(t["status"] || t[:status]) }
37
+ Array(tool_calls).reject do |t|
38
+ status = t["status"] || t[:status]
39
+ Assertions.ok_status?(status) || status.to_s == "blocked"
40
+ end
41
+ end
42
+
43
+ def blocked_tools
44
+ Array(tool_calls).select { |t| (t["status"] || t[:status]).to_s == "blocked" }
24
45
  end
25
46
  end
26
47
 
@@ -144,8 +165,8 @@ module Insika
144
165
  # NOT `Array(turns)`: TurnResult is a Struct, so Array() would explode a single
145
166
  # one into its members and hand the policy checks three strings.
146
167
  conversation = turns.nil? || turns.empty? ? [result] : turns
147
- checks = tool_checks(golden, result) + must_not_checks(golden, result) +
148
- policy_checks(golden, conversation)
168
+ checks = tool_checks(golden, result) + call_checks(golden, result) + reply_checks(golden, result) +
169
+ ui_checks(golden, result) + must_not_checks(golden, result) + policy_checks(golden, conversation)
149
170
  CaseResult.new(id: golden.id, agent: golden.agent, error: nil, checks: checks,
150
171
  rubric: golden.rubric, judge: nil)
151
172
  end
@@ -163,6 +184,71 @@ module Insika
163
184
  end
164
185
  end
165
186
 
187
+ # The graders over the turn's CALLS. Each is its own Check, so the report names
188
+ # the one that failed instead of "tool checks failed". All read the LAST turn,
189
+ # like `tools_called`.
190
+ def call_checks(golden, result)
191
+ names = result.tool_names
192
+ checks = golden.never_calls.map do |name|
193
+ hit = names.include?(name)
194
+ Check.new(name: "never_calls:#{name}", pass: !hit, detail: hit ? "called #{name}" : "not called")
195
+ end
196
+
197
+ unless golden.calls_one_of.empty?
198
+ hit = golden.calls_one_of & names
199
+ checks << Check.new(name: "calls_one_of", pass: !hit.empty?,
200
+ detail: hit.empty? ? "none of #{golden.calls_one_of.join(', ')} called (saw: #{names.join(', ')})"
201
+ : "called #{hit.join(', ')}")
202
+ end
203
+
204
+ if (first = golden.first_tool)
205
+ checks << Check.new(name: "first_tool:#{first}", pass: names.first == first,
206
+ detail: "first call was #{names.first || 'none'}")
207
+ end
208
+
209
+ if (max = golden.max_tool_calls)
210
+ checks << Check.new(name: "max_tool_calls:#{max}", pass: names.size <= max,
211
+ detail: "#{names.size} call(s): #{names.join(', ')}")
212
+ end
213
+
214
+ blocked = result.blocked_tools.map { |t| "#{t['name'] || t[:name]}:#{t['gate'] || t[:gate]}" }
215
+ checks + golden.blocked_gates.map do |pair|
216
+ held = blocked.include?(pair)
217
+ Check.new(name: "blocked_gates:#{pair}", pass: held,
218
+ detail: held ? "held" : "not held (blocked: #{blocked.empty? ? 'none' : blocked.join(', ')})")
219
+ end
220
+ end
221
+
222
+ # Case-insensitive substrings over the PUBLISHED answer. `reply_omits` is where
223
+ # an internal id, a CPF or a raw tag leaking into the customer's text is pinned —
224
+ # the negative half of `reply_includes`.
225
+ def reply_checks(golden, result)
226
+ text = result.output_text.to_s.downcase
227
+ golden.reply_includes.map do |s|
228
+ hit = text.include?(s.downcase)
229
+ Check.new(name: "reply_includes:#{s}", pass: hit, detail: hit ? "present" : "missing from the reply")
230
+ end + golden.reply_omits.map do |s|
231
+ hit = text.include?(s.downcase)
232
+ Check.new(name: "reply_omits:#{s}", pass: !hit, detail: hit ? "present in the reply" : "absent")
233
+ end
234
+ end
235
+
236
+ # The graders over what the turn SHOWED. `ui_components` pins that a presentation
237
+ # tool put at least one card of that component in front of the customer;
238
+ # `no_ui` is its negative — a turn that must answer in text alone.
239
+ def ui_checks(golden, result)
240
+ shown = result.shown_components
241
+ checks = golden.ui_components.map do |component|
242
+ hit = shown.include?(component)
243
+ Check.new(name: "ui_components:#{component}", pass: hit,
244
+ detail: hit ? "shown" : "not shown (ui: #{shown.empty? ? 'none' : shown.join(', ')})")
245
+ end
246
+ return checks unless golden.no_ui?
247
+
248
+ checks << Check.new(name: "no_ui", pass: shown.empty?,
249
+ detail: shown.empty? ? "nothing shown" : "shown: #{shown.join(', ')}")
250
+ end
251
+
166
252
  # `must_not` detectors. "tool_error" is special (inspects statuses); the rest
167
253
  # are content detectors over the output text.
168
254
  def must_not_checks(golden, result)
@@ -20,8 +20,16 @@ module Insika
20
20
  # single-tenant default, like `save_artifact`'s own binding_tenant) unless the
21
21
  # case declares one. `run_persona_eval` uses it to keep a QA agent from ever
22
22
  # running (or even seeing) another tenant's persona case in the same store.
23
+ #
24
+ # `state`: the snapshot the conversation starts from — evidence ids, memory
25
+ # facts and notes, prior history, briefing fields — loaded into the session
26
+ # BEFORE turn 1. It is how a case tests "the customer already saw three products
27
+ # and says 'add the second one'" without replaying the search turn: no dependence
28
+ # on the model's first answer, one turn cheaper, and a messy state (a
29
+ # contradiction from six turns ago) becomes reproducible. {} = the case starts
30
+ # empty, as every case did before the key existed.
23
31
  Golden = Struct.new(:id, :agent, :turns, :expect, :requires, :reference, :source, :persona, :tenant,
24
- keyword_init: true) do
32
+ :state, keyword_init: true) do
25
33
  # The user messages to replay, in order. Empty for a persona case: a generated
26
34
  # conversation has no scripted turns.
27
35
  def user_turns = turns.map { |t| t["user"] }
@@ -44,6 +52,26 @@ module Insika
44
52
  # Names of the negative assertions to run (e.g. "pii_leak", "tool_error").
45
53
  def must_not = Array(expect["must_not"]).map(&:to_s)
46
54
 
55
+ # The graders over the turn's CALLS and its published REPLY — each optional,
56
+ # each deterministic. Every positive has a negative: `never_calls` pins what a
57
+ # correct turn does NOT do, which `tools_called` alone can never say.
58
+ def never_calls = Array(expect["never_calls"]).map(&:to_s)
59
+ def calls_one_of = Array(expect["calls_one_of"]).map(&:to_s)
60
+ def first_tool = GoldenLoader.presence(expect["first_tool"])
61
+ def max_tool_calls = expect["max_tool_calls"]
62
+ def reply_includes = Array(expect["reply_includes"]).map(&:to_s)
63
+ def reply_omits = Array(expect["reply_omits"]).map(&:to_s)
64
+ # "tool:gate" pairs that must appear among the turn's BLOCKED calls.
65
+ def blocked_gates = Array(expect["blocked_gates"]).map(&:to_s)
66
+ # UI components a presentation tool must have shown (`insika.ui` frames with
67
+ # at least one card), and its negative: `no_ui: true` pins a turn that shows
68
+ # nothing.
69
+ def ui_components = Array(expect["ui_components"]).map(&:to_s)
70
+ def no_ui? = expect["no_ui"] == true
71
+
72
+ def state = self[:state] || {}
73
+ def seeded? = !state.empty?
74
+
47
75
  # How much the agent should ask before acting. nil = the store
48
76
  # has no opinion and only the rubric decides.
49
77
  def policy = GoldenLoader.presence(expect["policy"])
@@ -111,6 +139,8 @@ module Insika
111
139
  raise InvalidGolden, "#{source}: 'expect' must be a mapping (case '#{id}')" unless expect.is_a?(Hash)
112
140
 
113
141
  validate_policy!(expect["policy"], id: id, source: source)
142
+ validate_graders!(expect, id: id, source: source)
143
+ state = normalize_state(raw["state"], id: id, source: source)
114
144
  requires = raw["requires"] || {}
115
145
  unless requires.is_a?(Hash)
116
146
  raise InvalidGolden, "#{source}: 'requires' must be a mapping (case '#{id}')"
@@ -121,7 +151,7 @@ module Insika
121
151
 
122
152
  Golden.new(id: id, agent: agent, turns: turns, expect: expect,
123
153
  requires: requires, reference: reference, source: source, persona: persona,
124
- tenant: tenant)
154
+ tenant: tenant, state: state)
125
155
  end
126
156
 
127
157
  # `persona:` is the alternative shape to `turns:`: the
@@ -174,6 +204,65 @@ module Insika
174
204
  { "role" => role, "text" => text }.merge(origin ? { "origin" => origin } : {})
175
205
  end
176
206
 
207
+ STATE_KEYS = %w[evidence memory history briefing].freeze
208
+
209
+ # state: the snapshot a case starts from. Absent -> {}. Only the four known
210
+ # keys, each in its own shape — a typo'd key (`evidences:`) would seed nothing
211
+ # and the case would go on passing against the wrong precondition. Refused at
212
+ # LOAD, not at seed time: a case that half-seeds is a hole in the net.
213
+ def normalize_state(raw, id:, source:)
214
+ return {} if raw.nil?
215
+
216
+ where = "#{source}: state (case '#{id}')"
217
+ raise InvalidGolden, "#{where} must be a mapping" unless raw.is_a?(Hash)
218
+
219
+ unknown = raw.keys.map(&:to_s) - STATE_KEYS
220
+ unless unknown.empty?
221
+ raise InvalidGolden, "#{where}: unknown key(s) #{unknown.join(', ')} — known: #{STATE_KEYS.join(', ')}"
222
+ end
223
+
224
+ state = raw.compact
225
+ mapping_with!(state["evidence"], "ids", Array, "#{where}.evidence")
226
+ mapping_with!(state["memory"], "facts", Hash, "#{where}.memory")
227
+ mapping_with!(state["memory"], "notes", Array, "#{where}.memory")
228
+ mapping_with!(state["briefing"], "fields", Hash, "#{where}.briefing")
229
+ history = state["history"]
230
+ unless history.nil? || (history.is_a?(Array) && history.all? { |m| history_message?(m) })
231
+ raise InvalidGolden, "#{where}.history must be [{ role: user|assistant, content: '…' }]"
232
+ end
233
+
234
+ state
235
+ end
236
+
237
+ def mapping_with!(value, key, type, where)
238
+ return if value.nil?
239
+ raise InvalidGolden, "#{where} must be a mapping" unless value.is_a?(Hash)
240
+ return if value[key].nil? || value[key].is_a?(type)
241
+
242
+ raise InvalidGolden, "#{where}.#{key} must be #{type == Array ? 'a list' : 'a mapping'}"
243
+ end
244
+
245
+ def history_message?(message)
246
+ message.is_a?(Hash) && %w[user assistant].include?(message["role"].to_s) &&
247
+ !presence(message["content"]).nil?
248
+ end
249
+
250
+ # The graders with a shape to get wrong: a non-integer `max_tool_calls` would
251
+ # compare against nil and pass; a `blocked_gates` entry without its gate would
252
+ # never match anything and pass. Refused at load, like `policy`.
253
+ def validate_graders!(expect, id:, source:)
254
+ max = expect["max_tool_calls"]
255
+ unless max.nil? || (max.is_a?(Integer) && max >= 0)
256
+ raise InvalidGolden, "#{source}: max_tool_calls must be a non-negative integer (case '#{id}')"
257
+ end
258
+
259
+ Array(expect["blocked_gates"]).each do |pair|
260
+ next if pair.to_s.match?(/\A[^:\s]+:[^:\s]+\z/)
261
+
262
+ raise InvalidGolden, "#{source}: blocked_gates entries are 'tool:gate' (got #{pair.inspect}, case '#{id}')"
263
+ end
264
+ end
265
+
177
266
  # A typo'd policy must not silently mean "no policy" — the case would go on
178
267
  # passing while the rule it was written for stopped being checked. The
179
268
  # `Assertions` constant is resolved at CALL time (this file loads first, and
@@ -66,6 +66,26 @@ module Insika
66
66
  # consumer needing a real Chat UUID as X-Chat-Id) supplies it via conv_map; otherwise the
67
67
  # synthetic "eval-<id>" keeps the adapter's own multi-turn continuation.
68
68
  conv = @conv_map[golden.id] || "eval-#{golden.id}"
69
+ if golden.seeded?
70
+ # The snapshot goes in BEFORE turn 1 — on the same conversation id the turns
71
+ # continue. A transport that cannot seed (A2A: a remote agent has no seed
72
+ # route) SKIPS the case with the reason, never runs it against an empty
73
+ # state and passes. A deployment that refuses seeding (the setting is off)
74
+ # is the same outcome: skipped and reported, which is what the `requires`
75
+ # discipline already promises for anything the deployment lacks.
76
+ unless @transport.respond_to?(:seed)
77
+ return RunCase.new(result: Assertions.skip(golden, "transport cannot seed state"), timings: [])
78
+ end
79
+
80
+ begin
81
+ @transport.seed(conv, golden.state)
82
+ rescue SeedRefused => e
83
+ return RunCase.new(result: Assertions.skip(golden, "deployment refuses seeding — #{e.message}"), timings: [])
84
+ rescue Insika::Error => e
85
+ failed = TurnResult.new(output_text: "", tool_calls: [], error: "seed failed: #{e.message}")
86
+ return RunCase.new(result: Assertions.evaluate(golden, failed), timings: [])
87
+ end
88
+ end
69
89
  turns = []
70
90
  timings = []
71
91
  spent = []
@@ -95,11 +95,20 @@ module Insika
95
95
  @safety = safety
96
96
  end
97
97
 
98
- # Runs one simulated conversation. -> SimulatedRun.
99
- def run(persona:, agent:, conv:)
98
+ # Runs one simulated conversation. -> SimulatedRun. `state` is the case's
99
+ # snapshot (Golden#state), loaded into the conversation before the persona's
100
+ # opening line — a simulated customer can start from "already saw three
101
+ # products" too. Empty/nil = the conversation starts empty, as before.
102
+ def run(persona:, agent:, conv:, state: nil)
100
103
  reason = @safety.refusal
101
104
  raise UnsafeTarget, reason if reason
102
105
 
106
+ if state && !state.empty?
107
+ raise Insika::Error, "transport cannot seed state" unless @transport.respond_to?(:seed)
108
+
109
+ @transport.seed(conv, state)
110
+ end
111
+
103
112
  transcript = []
104
113
  message = persona.opens_with
105
114
  stop = :max_turns