insika 0.7.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +191 -0
  3. data/README.md +9 -6
  4. data/bin/insika +44 -3
  5. data/docs/AGENTS.md +74 -14
  6. data/docs/API.md +73 -0
  7. data/docs/ARCHITECTURE.md +45 -44
  8. data/docs/ARTIFACTS.md +42 -0
  9. data/docs/CHANNELS.md +19 -2
  10. data/docs/CONTEXT.md +86 -38
  11. data/docs/DEPLOY.md +27 -7
  12. data/docs/EVALS.md +98 -8
  13. data/docs/FACTS.md +4 -0
  14. data/docs/KNOWLEDGE.md +7 -0
  15. data/docs/LOADTEST.md +15 -27
  16. data/docs/MEDIA.md +1 -1
  17. data/docs/OBSERVABILITY.md +48 -4
  18. data/docs/POLICY.md +14 -5
  19. data/docs/RELEASING.md +4 -0
  20. data/docs/RUNNING-LOCAL.md +2 -2
  21. data/docs/SECURITY.md +28 -2
  22. data/docs/SOAK.md +1 -1
  23. data/docs/TOOLS.md +150 -32
  24. data/docs/prompts/ADD-TOOL.md +12 -2
  25. data/docs/prompts/DIAGNOSE-TURN.md +3 -0
  26. data/docs/prompts/GO-LIVE.md +6 -4
  27. data/lib/insika/agent_profile.rb +47 -10
  28. data/lib/insika/channels/web/widget.js +33 -0
  29. data/lib/insika/channels/web.rb +5 -2
  30. data/lib/insika/chat_builder.rb +90 -37
  31. data/lib/insika/commands/agent_payload.rb +1 -1
  32. data/lib/insika/commands/run_distillation.rb +5 -8
  33. data/lib/insika/commands/seed_session.rb +118 -0
  34. data/lib/insika/compaction.rb +196 -0
  35. data/lib/insika/context/builder.rb +35 -11
  36. data/lib/insika/context/fragment.rb +4 -1
  37. data/lib/insika/context/priority.rb +8 -0
  38. data/lib/insika/context/provider.rb +5 -0
  39. data/lib/insika/context/providers/briefing.rb +61 -29
  40. data/lib/insika/context/providers/fence_notice.rb +27 -0
  41. data/lib/insika/context/providers/knowledge.rb +7 -4
  42. data/lib/insika/context/providers/memory.rb +8 -4
  43. data/lib/insika/context/providers/session.rb +50 -10
  44. data/lib/insika/context_trace_store.rb +11 -1
  45. data/lib/insika/doctor.rb +213 -10
  46. data/lib/insika/dsl/runtime.rb +5 -0
  47. data/lib/insika/dsl.rb +6 -0
  48. data/lib/insika/edge_limiter.rb +4 -1
  49. data/lib/insika/env_schema.rb +5 -6
  50. data/lib/insika/errors.rb +1 -0
  51. data/lib/insika/evals/assertions.rb +92 -6
  52. data/lib/insika/evals/golden.rb +91 -2
  53. data/lib/insika/evals/runner.rb +20 -0
  54. data/lib/insika/evals/simulator.rb +11 -2
  55. data/lib/insika/evals/transport.rb +118 -16
  56. data/lib/insika/evidence.rb +79 -12
  57. data/lib/insika/executor.rb +94 -23
  58. data/lib/insika/fence.rb +96 -0
  59. data/lib/insika/golden_store.rb +3 -0
  60. data/lib/insika/loop_detector.rb +5 -34
  61. data/lib/insika/mcp_store.rb +5 -2
  62. data/lib/insika/mcp_tool_registry.rb +8 -1
  63. data/lib/insika/memory_store.rb +12 -0
  64. data/lib/insika/overlay_tool_registry.rb +5 -0
  65. data/lib/insika/prefix_fingerprint.rb +32 -27
  66. data/lib/insika/profile_source.rb +8 -0
  67. data/lib/insika/server/app.rb +43 -1
  68. data/lib/insika/server/rack_app.rb +2 -0
  69. data/lib/insika/server/responses.rb +35 -8
  70. data/lib/insika/session_store.rb +38 -5
  71. data/lib/insika/settings_store.rb +18 -2
  72. data/lib/insika/soak/runner.rb +4 -4
  73. data/lib/insika/spoken_transcript.rb +31 -0
  74. data/lib/insika/studio/app.rb +34 -8
  75. data/lib/insika/studio/forms.rb +29 -3
  76. data/lib/insika/studio/views/_agent_tab_config.erb +5 -1
  77. data/lib/insika/studio/views/session.erb +1 -1
  78. data/lib/insika/studio/views/settings.erb +11 -0
  79. data/lib/insika/studio/views/tool_edit.erb +6 -2
  80. data/lib/insika/telemetry/recorder.rb +61 -1
  81. data/lib/insika/templates/daily-digest/README.md +9 -0
  82. data/lib/insika/templates/research-analyst/agent.rb +10 -0
  83. data/lib/insika/tool_assembly.rb +21 -13
  84. data/lib/insika/tool_batch.rb +67 -0
  85. data/lib/insika/tool_definition.rb +73 -10
  86. data/lib/insika/tool_envelope.rb +102 -2
  87. data/lib/insika/tool_store.rb +9 -4
  88. data/lib/insika/tool_trace_store.rb +1 -1
  89. data/lib/insika/tool_usage_report.rb +172 -0
  90. data/lib/insika/tools/data_defined_tool.rb +1 -0
  91. data/lib/insika/tools/present.rb +122 -0
  92. data/lib/insika/tools/run_persona_eval.rb +6 -1
  93. data/lib/insika/tools/tool_search.rb +4 -2
  94. data/lib/insika/turn_budget.rb +91 -0
  95. data/lib/insika/turn_state.rb +13 -1
  96. data/lib/insika/version.rb +1 -1
  97. data/lib/insika/wiring/graph.rb +7 -0
  98. data/lib/insika/wiring/graph_chat.rb +4 -0
  99. data/lib/insika.rb +11 -0
  100. metadata +10 -1
data/lib/insika/doctor.rb CHANGED
@@ -179,7 +179,9 @@ module Insika
179
179
  check_relay_channel check_web_widget check_skill_eager check_skill_drift check_shadow_parity
180
180
  check_soak_envelope check_turn_timing check_grounding check_cache_layers
181
181
  check_memory_scopes check_funnel_declarations check_followup check_distill
182
- check_harvest check_schedules check_guardrail_corpora]
182
+ check_compaction check_harvest check_schedules check_guardrail_corpora
183
+ check_tool_allowlist_policy check_fencing check_eval_seeding
184
+ check_presentation_tools check_provenance]
183
185
 
184
186
  def safe(check)
185
187
  Array(send(check))
@@ -722,26 +724,59 @@ module Insika
722
724
  # the mangled prompt on every turn, and nothing else would ever say so: the file is
723
725
  # present, non-empty, and the agent answers — worse than a crash. Found on the pilot
724
726
  # by an `insika refine` report, three weeks after the fact.
727
+ #
728
+ # The same sweep also WARNS (never errors) on a file that outgrew a prompt.
729
+ # Merchant packs are LLM-generated (generate-merchant-pack), and generated prose
730
+ # bloats: the pilot's 28 KB AGENTS.md is the local example, and the ETH Zurich
731
+ # instruction-file study puts the cost of that shape at 20%+ extra tokens per
732
+ # turn for no extra instruction-following. The thresholds are deliberately
733
+ # generous — a hand-written file never meets them; only the generated shape does.
725
734
  def check_prompt_files
726
735
  return [] unless @agent_file_store
727
736
 
728
737
  agents = @agent_file_store.agents
729
- wrapped = agents.flat_map do |agent|
730
- @agent_file_store.list(agent).filter_map do |name|
731
- next unless wrapped_content?(@agent_file_store.read(agent, name))
732
-
733
- Finding.new(check: "prompt-files", severity: :error, fix: nil,
734
- message: "agent '#{agent}' file '#{name}' holds a serialized object, not text — " \
735
- "the model receives `{\"content\" => …}` on one line, escapes and all. " \
736
- "Recover the markdown from inside the wrapper and write it back.")
738
+ wrapped = []
739
+ oversized = []
740
+ agents.each do |agent|
741
+ @agent_file_store.list(agent).each do |name|
742
+ content = @agent_file_store.read(agent, name)
743
+ if wrapped_content?(content)
744
+ wrapped << Finding.new(check: "prompt-files", severity: :error, fix: nil,
745
+ message: "agent '#{agent}' file '#{name}' holds a serialized object, not text — " \
746
+ "the model receives `{\"content\" => …}` on one line, escapes and all. " \
747
+ "Recover the markdown from inside the wrapper and write it back.")
748
+ elsif (finding = oversized_prompt(agent, name, content))
749
+ oversized << finding
750
+ end
737
751
  end
738
752
  end
739
- return wrapped if wrapped.any?
753
+ return wrapped + oversized if wrapped.any? || oversized.any?
740
754
 
741
755
  total = agents.sum { |a| @agent_file_store.list(a).length }
742
756
  [ok("prompt-files", "#{total} prompt file(s) across #{agents.length} agent(s): all text")]
743
757
  end
744
758
 
759
+ # WARN thresholds for one prompt file. ~6 000 estimated tokens (~24 KB) or 600
760
+ # lines: the pilot's generated 28 KB / 292-line AGENTS.md trips the token bar,
761
+ # every hand-written demo file stays far under both. Estimate = chars/4, the
762
+ # same yardstick as Insika::TokenEstimator — cheap and honest about being ±15%.
763
+ PROMPT_FILE_WARN_TOKENS = 6_000
764
+ PROMPT_FILE_WARN_LINES = 600
765
+
766
+ def oversized_prompt(agent, name, content)
767
+ text = content.to_s
768
+ tokens = Insika::TokenEstimator.estimate(text)
769
+ lines = text.lines.count
770
+ return nil if tokens <= PROMPT_FILE_WARN_TOKENS && lines <= PROMPT_FILE_WARN_LINES
771
+
772
+ Finding.new(check: "prompt-files", severity: :warn, fix: nil,
773
+ message: "agent '#{agent}' file '#{name}' is ~#{tokens} tokens over #{lines} line(s) " \
774
+ "(threshold: #{PROMPT_FILE_WARN_TOKENS} tokens / #{PROMPT_FILE_WARN_LINES} lines) — " \
775
+ "a prompt this large costs 20%+ more tokens on every turn for no better " \
776
+ "instruction-following. Trim it, or split the reference material into skills " \
777
+ "the agent loads on demand.")
778
+ end
779
+
745
780
  # Cheap and specific: Ruby's inspect of a Hash whose first key is a string. A real
746
781
  # prompt does not open with `{"…" =>`.
747
782
  def wrapped_content?(content) = /\A\s*\{\s*"[^"]+"\s*=>/.match?(content.to_s)
@@ -802,6 +837,7 @@ def wrapped_content?(content) = /\A\s*\{\s*"[^"]+"\s*=>/.match?(content.to_s)
802
837
  # identity block bills a cache write every turn).
803
838
  IDENTITY_BUILTINS = %w[
804
839
  Insika::Context::Providers::Prompt
840
+ Insika::Context::Providers::FenceNotice
805
841
  Insika::Context::Providers::Skill
806
842
  Insika::Context::Providers::ToolSearch
807
843
  ].freeze
@@ -868,6 +904,111 @@ def wrapped_content?(content) = /\A\s*\{\s*"[^"]+"\s*=>/.match?(content.to_s)
868
904
  # harmless but useless — every claim passes and the audit reads zero. A
869
905
  # warning, never an error: the pack owns matcher quality; the engine refuses
870
906
  # only uncompileable data.
907
+ # A stored agent that declares `tools_allow` / `tools_deny` /
908
+ # `tools_allow_groups` but does not name the `tool_allowlist` policy. Only
909
+ # that policy applies the lists, and the Policy::Engine runs ONLY the
910
+ # policies a profile names — so as written on disk the lists are inert and
911
+ # the model receives every registered tool (hundreds of schemas, tens of
912
+ # thousands of tokens, per request).
913
+ #
914
+ # AgentProfile.build now adds the policy whenever a list is declared, so a
915
+ # live turn off this record is already safe. The record itself stays wrong
916
+ # until something re-saves it, and anything that reads the stored shape
917
+ # directly — an older engine, a pack export, an operator auditing the
918
+ # Studio form — still sees an allowlist that does nothing. That is why this
919
+ # reads the RAW record and not `all`.
920
+ def check_tool_allowlist_policy
921
+ return [] unless @profile_source.respond_to?(:all_raw)
922
+
923
+ records = @profile_source.all_raw
924
+ bad = records.select { |r| tool_lists_declared?(r) && !names_tool_allowlist?(r) }
925
+ return [ok("tool-allowlist", "#{records.length} stored agent(s): every declared tool list names the policy")] if bad.empty?
926
+
927
+ bad.map do |r|
928
+ Finding.new(check: "tool-allowlist", severity: :error, fix: nil,
929
+ message: "agent '#{r["id"]}' declares a tool allow/deny list but its policies " \
930
+ "(#{Array(r["policies"]).join(", ").then { |p| p.empty? ? "none" : p }}) " \
931
+ "do not include tool_allowlist — as stored, the list is never applied and " \
932
+ "the model receives every registered tool. The engine repairs this when it " \
933
+ "loads the agent; re-save it (Studio, or the pack) so the stored policies " \
934
+ "match the intent.")
935
+ end
936
+ end
937
+
938
+ # Presence, not emptiness, for the two nil-able lists: `tools_allow: []`
939
+ # means "no tools" and is as much a declaration as a list of names.
940
+ # `tools_deny` has no nil state, so only a non-empty one counts.
941
+ def tool_lists_declared?(raw)
942
+ !raw["tools_allow"].nil? || !raw["tools_allow_groups"].nil? || !Array(raw["tools_deny"]).empty?
943
+ end
944
+
945
+ def names_tool_allowlist?(raw) = Array(raw["policies"]).any? { |p| p.to_s == "tool_allowlist" }
946
+
947
+ # A stored agent that allows a PRESENTATION tool but no tool declaring
948
+ # `evidence`. The presentation tool shows only cards an evidence tool returned
949
+ # this session, so with no evidence tool in reach every call drops every id —
950
+ # the model keeps asking to show products and nothing ever appears. Silent
951
+ # unless a presentation tool exists. Data tools only: a code tool's evidence
952
+ # lives in registry metadata the doctor does not read.
953
+ def check_presentation_tools
954
+ return [] unless @tool_store && @profile_source
955
+
956
+ raws = @tool_store.all_raw
957
+ presenting = raws.select { |r| r["presentation"] }.map { |r| r["name"].to_s }
958
+ return [] if presenting.empty?
959
+
960
+ evidence = raws.select { |r| r["evidence"] }.map { |r| r["name"].to_s }
961
+ bad = @profile_source.all.filter_map do |profile|
962
+ allowed = allowed_data_tools(profile, raws)
963
+ shows = presenting & allowed
964
+ next if shows.empty? || !(evidence & allowed).empty?
965
+
966
+ Finding.new(check: "presentation-tools", severity: :warn, fix: nil,
967
+ message: "agent '#{profile.id}' allows #{shows.join(', ')} but no tool declaring " \
968
+ "evidence — a presentation tool can only show cards an evidence tool " \
969
+ "returned this session, so every call would drop every id. Allow the " \
970
+ "search tool that declares `evidence`, or drop the presentation tool.")
971
+ end
972
+ return bad unless bad.empty?
973
+
974
+ [ok("presentation-tools", "#{presenting.length} presentation tool(s): every agent that shows cards " \
975
+ "also allows an evidence tool")]
976
+ end
977
+
978
+ # The data tools a profile may call, by the allowlist's own rules: nil lists =
979
+ # all; otherwise names ∪ groups; deny wins.
980
+ def allowed_data_tools(profile, raws)
981
+ names = profile.tools_allow.nil? ? nil : Array(profile.tools_allow).map(&:to_s)
982
+ groups = profile.tools_allow_groups.nil? ? nil : Array(profile.tools_allow_groups).map(&:to_s)
983
+ allowed = if names.nil? && groups.nil?
984
+ raws.map { |r| r["name"].to_s }
985
+ else
986
+ Array(names) | raws.select { |r| Array(groups).include?(r["group"].to_s) }.map { |r| r["name"].to_s }
987
+ end
988
+ allowed - Array(profile.tools_deny).map(&:to_s)
989
+ end
990
+
991
+ def check_provenance
992
+ return [] unless @tool_store && @profile_source.respond_to?(:all_raw)
993
+
994
+ tools = @tool_store.all_raw
995
+ @profile_source.all_raw.filter_map do |raw|
996
+ allowed = tools.select do |tool|
997
+ name = tool["name"]
998
+ unrestricted = raw["tools_allow"].nil? && raw["tools_allow_groups"].nil?
999
+ (unrestricted || Array(raw["tools_allow"]).include?(name) ||
1000
+ Array(raw["tools_allow_groups"]).include?(tool["group"])) &&
1001
+ !Array(raw["tools_deny"]).include?(name)
1002
+ end
1003
+ gated = allowed.select { |tool| tool["requires_evidence"] }
1004
+ next if gated.empty? || allowed.any? { |tool| tool["evidence"] }
1005
+
1006
+ Finding.new(check: "provenance", severity: :warn, fix: nil,
1007
+ message: "agent '#{raw["id"]}' allows #{gated.map { |t| t["name"] }.join(', ')} " \
1008
+ "with requires_evidence but no stored evidence tool — verify that an allowed code tool supplies evidence")
1009
+ end
1010
+ end
1011
+
871
1012
  def check_grounding
872
1013
  return [] unless @profile_source
873
1014
 
@@ -1230,6 +1371,68 @@ def wrapped_content?(content) = /\A\s*\{\s*"[^"]+"\s*=>/.match?(content.to_s)
1230
1371
  stale: @proposal_store.stale(limit: 10_000).size }
1231
1372
  end
1232
1373
 
1374
+ # the in-session compaction check (RFC-0044): platform-gated,
1375
+ # so ONE line — enabled with no resolvable model (no compaction.model, no
1376
+ # platform utility_model) can never run; the warn is the same "declared
1377
+ # but dead" signal as check_distill's (the engine never guesses a model).
1378
+ def check_compaction
1379
+ return [] unless @settings_store
1380
+
1381
+ settings = @settings_store.get
1382
+ config = settings["compaction"] || {}
1383
+ return [ok("compaction", "in-session compaction off")] unless Coercion.truthy?(config["enabled"])
1384
+
1385
+ if Coercion.presence(config["model"]).nil? && Coercion.presence(settings["utility_model"]).nil?
1386
+ [Finding.new(check: "compaction", severity: :warn, fix: nil,
1387
+ message: "in-session compaction is enabled but has no model slot — it will " \
1388
+ "never run (set compaction.model or the platform utility_model).")]
1389
+ else
1390
+ [ok("compaction", "in-session compaction on — keep_last #{config['keep_last']}, " \
1391
+ "compact_after #{config['compact_after']}")]
1392
+ end
1393
+ end
1394
+
1395
+ # `evals.seeding` opens POST /v1/conversations/:id/seed: a snapshot written
1396
+ # into a real conversation under the tenant token, before its first turn. Right
1397
+ # on the machine running snapshot evals; wrong left on in production, where any
1398
+ # consumer holding the token could write a customer's history. A warning, not
1399
+ # an error — the operator may be running the evals right now.
1400
+ def check_eval_seeding
1401
+ return [] unless @settings_store
1402
+
1403
+ on = ((@settings_store.get["evals"] || {})["seeding"]) == true
1404
+ return [ok("eval-seeding", "evals.seeding off — conversations cannot be seeded")] unless on
1405
+
1406
+ [Finding.new(check: "eval-seeding", severity: :warn, fix: nil,
1407
+ message: "evals.seeding is ON — POST /v1/conversations/:id/seed accepts a fabricated " \
1408
+ "conversation state under the tenant token. Fine while running snapshot " \
1409
+ "evals; turn it off (Studio > Settings, or `update_settings`) in production.")]
1410
+ end
1411
+
1412
+ # A customer-facing agent — one reachable through an inbound channel: any
1413
+ # agent when the relay is mounted (the event names the agent), the listed
1414
+ # ones for the widget — with `fencing` off reads third-party bytes as-is.
1415
+ # A warning, never an error: the default is off this release on purpose
1416
+ # (the goldens were baselined unfenced), so the message says exactly what
1417
+ # is unsanitized today. No inbound channel = nothing customer-facing = silent.
1418
+ def check_fencing
1419
+ return [] unless @profile_source
1420
+
1421
+ relay = Insika::Coercion.present?(@env["INSIKA_RELAY_TOKEN"])
1422
+ widget = @env["INSIKA_WIDGET_AGENTS"].to_s.split(",").map(&:strip).reject(&:empty?)
1423
+ return [] unless relay || widget.any?
1424
+
1425
+ exposed = @profile_source.all.select { |p| relay || widget.include?(p.id.to_s) }
1426
+ unfenced = exposed.reject { |p| Insika::Fence.enabled?(p) }.map(&:id)
1427
+ return [ok("fencing", "fencing on for every agent behind an inbound channel")] if unfenced.empty?
1428
+
1429
+ [Finding.new(check: "fencing", severity: :warn, fix: nil,
1430
+ message: "agent(s) #{unfenced.join(', ')}: reachable through an inbound channel with " \
1431
+ "`fencing` off — tool results, <memory> facts and <knowledge> concepts reach " \
1432
+ "the model byte-for-byte (invisible characters, forged turn markers and " \
1433
+ "transcript-shaped tags included). Set `fencing true` on the agent.")]
1434
+ end
1435
+
1233
1436
  # the harvest check — per profile WITH a harvest hash:
1234
1437
  # declared-without-model warn (D12), no grounding matcher warn (D3),
1235
1438
  # malformed negative list error (D4), else ok with the pending counts.
@@ -231,11 +231,16 @@ module Insika
231
231
  base: "", catalog: c[:prompt_catalog],
232
232
  agent_files: c[:agent_file_store], system_files: c[:system_file_store]
233
233
  ),
234
+ Insika::Context::Providers::FenceNotice.new,
234
235
  Insika::Context::Providers::Skill.new(catalog: c[:skill_catalog]),
235
236
  Insika::Context::Providers::SkillTrigger.new(catalog: c[:skill_catalog]),
236
237
  Insika::Context::Providers::ToolSearch.new(catalog: c[:tool_catalog]),
237
238
  Insika::Context::Providers::Memory.new(store: spine.memory_store),
238
239
  Insika::Context::Providers::Knowledge.new(store: spine.knowledge_store),
240
+ # Session briefing: read path. Inert for agents without briefing_fields.
241
+ # Before Session, so the durable head renders above the transcript and
242
+ # the tail recitation lands after it.
243
+ Insika::Context::Providers::Briefing.new(session_store: spine.session_store),
239
244
  Insika::Context::Providers::Session.new(session_store: spine.session_store)
240
245
  ] # NOT frozen: load_plugins appends plugin providers at boot
241
246
  end
data/lib/insika/dsl.rb CHANGED
@@ -417,6 +417,12 @@ module Insika
417
417
  # SEES: repeated identical tool results collapse to a back-reference.
418
418
  def tool_output_compression(on = true) = @config[:tool_output_compression] = on
419
419
 
420
+ # Third-party text is sanitized before the model reads it (tool results,
421
+ # memory, knowledge) and a one-sentence notice names those blocks as data.
422
+ # Opt-in this release — the goldens were baselined on unfenced bytes:
423
+ # fencing true
424
+ def fencing(on = true) = @config[:fencing] = on
425
+
420
426
  # Content-safety guardrails — opt-in and configurable per agent.
421
427
  # Pure config-over-code: the hash is stored on the profile and consumed by
422
428
  # Safety::Config.from_profile. Merges, so repeated calls accumulate.
@@ -263,12 +263,15 @@ module Insika
263
263
 
264
264
  # Appends the note to the assembled system prompt: the real Data package is
265
265
  # immutable (with), the specs' minimal Struct is mutable — both duck-typed.
266
+ # The note is per-turn data, so it also lands in the volatile layer — below
267
+ # the cache breakpoint, never inside the cached identity prefix.
266
268
  def inject_budget_note(state, note)
267
269
  ctx = state.context
268
270
  return if ctx.nil?
269
271
 
270
272
  if ctx.respond_to?(:with)
271
- state.context = ctx.with(system: "#{ctx.system}\n\n#{note}")
273
+ volatile = [ctx.system_volatile, note].reject { |s| s.to_s.empty? }.join("\n\n")
274
+ state.context = ctx.with(system: "#{ctx.system}\n\n#{note}", system_volatile: volatile)
272
275
  elsif ctx.respond_to?(:system=)
273
276
  ctx.system = "#{ctx.system}\n\n#{note}"
274
277
  end
@@ -72,10 +72,10 @@ module Insika
72
72
 
73
73
  # Prefixes the engine fully OWNS: an unknown key under one of these is a typo, not
74
74
  # a foreign var. INSIKA_ (current) and HARNESS_ (legacy, still honored during the
75
- # deprecation window). Deliberately NOT OPENCLAW_ (shared with the OpenClaw gateway
76
- # product, which sets its own OPENCLAW_HOME/_STATE_DIR/… — the engine merely borrows
77
- # 3 names for interop), nor LITESTREAM_ (the sidecar owns it), nor OTEL_ (the
78
- # OpenTelemetry SDK owns its env).
75
+ # deprecation window). Deliberately NOT OPENCLAW_ (the OpenClaw gateway product
76
+ # sets its own OPENCLAW_HOME/_STATE_DIR/… on the same host; the engine reads none
77
+ # of them), nor LITESTREAM_ (the sidecar owns it), nor OTEL_ (the OpenTelemetry
78
+ # SDK owns its env).
79
79
  OWNED_PREFIXES = [PREFIX, LEGACY_PREFIX].freeze
80
80
 
81
81
  BOOLEANS = %w[1 0 true false yes no on off].freeze
@@ -142,8 +142,7 @@ module Insika
142
142
  spec(name: "INSIKA_ROUTER_BACKEND_TIMEOUT", type: :integer, description: "`insika-router`'s connect/read timeout to a backend, in seconds (default 10)."),
143
143
  spec(name: "INSIKA_ROUTER_HOST", description: "bind address for `insika-router` itself (default 0.0.0.0)."),
144
144
  spec(name: "INSIKA_ROUTER_PORT", type: :integer, description: "listen port for `insika-router` itself (default 9090)."),
145
- spec(name: "OPENCLAW_GATEWAY_TOKEN", secret: true, description: "Bearer for /v1 + /a2a (falls back to ADMIN_TOKEN)."),
146
- spec(name: "OPENCLAW_AGENTS_DIR", type: :path, description: "Directory of OpenClaw-style agent packs."),
145
+ spec(name: "INSIKA_GATEWAY_TOKEN", secret: true, description: "Bearer for /v1 + /a2a (falls back to ADMIN_TOKEN when unset)."),
147
146
  spec(name: "INSIKA_PLUGIN_DIR", type: :path, description: "Workspace plugin root (directories with insika.plugin.yml). Loaded at boot; ids still need INSIKA_PLUGINS."),
148
147
  spec(name: "INSIKA_PLUGINS", type: :csv, description: "Plugin ids to enable from the workspace/bundled roots. Announced gems are enabled by installing them."),
149
148
  spec(name: "INSIKA_PLUGINS_DISABLED", type: :csv, description: "Plugin ids that never load — the absolute veto, wins over INSIKA_PLUGINS and over an announced gem."),
data/lib/insika/errors.rb CHANGED
@@ -8,6 +8,7 @@ module Insika
8
8
 
9
9
  class ValidationError < Error; end # Malformed Command -> HTTP 422, no Task created
10
10
  class NotFoundError < Error; end # nonexistent session/task/agent -> HTTP 404
11
+ class ConflictError < Error; end # the write contradicts state that already exists -> HTTP 409
11
12
 
12
13
  # Policy Engine denied -> :policy_denied event, task :failed
13
14
  class PolicyDenied < Error
@@ -13,14 +13,35 @@ module Insika
13
13
  # offline without a server.
14
14
  #
15
15
  # output_text: the final assistant text (for content checks)
16
- # tool_calls: [{ "name" =>, "status" => }] captured from the stream's tool events
16
+ # tool_calls: [{ "name" =>, "arguments" =>, "status" =>, "gate" => }] captured
17
+ # from the stream's tool events. `arguments` (a Hash) and
18
+ # `status` ("ok" | "error" | "blocked", with `gate` naming what
19
+ # held a blocked call) arrive from the item frames; an older
20
+ # deployment sends names only and they stay nil.
21
+ # ui: [{ "component" =>, "count" => }] — one per presentation call
22
+ # (the `insika.ui` frames); nil/[] when the turn showed nothing
17
23
  # error: transport/turn error string, or nil on a clean turn
18
- TurnResult = Struct.new(:output_text, :tool_calls, :error, keyword_init: true) do
24
+ TurnResult = Struct.new(:output_text, :tool_calls, :ui, :error, keyword_init: true) do
19
25
  def tool_names = Array(tool_calls).map { |t| (t["name"] || t[:name]).to_s }
20
26
 
21
- # A tool call whose status is anything but a success ("ok"/2xx/"success").
27
+ # Components that actually put a card in front of the customer (count > 0).
28
+ def shown_components
29
+ Array(ui).select { |u| (u["count"] || u[:count]).to_i.positive? }
30
+ .map { |u| (u["component"] || u[:component]).to_s }
31
+ end
32
+
33
+ # A tool call whose status is anything but a success ("ok"/2xx/"success"). A
34
+ # BLOCKED call is not an error: the tool never ran, a gate held it — that is
35
+ # the `blocked_gates` grader's business, not `must_not: tool_error`'s.
22
36
  def errored_tools
23
- Array(tool_calls).reject { |t| Assertions.ok_status?(t["status"] || t[:status]) }
37
+ Array(tool_calls).reject do |t|
38
+ status = t["status"] || t[:status]
39
+ Assertions.ok_status?(status) || status.to_s == "blocked"
40
+ end
41
+ end
42
+
43
+ def blocked_tools
44
+ Array(tool_calls).select { |t| (t["status"] || t[:status]).to_s == "blocked" }
24
45
  end
25
46
  end
26
47
 
@@ -144,8 +165,8 @@ module Insika
144
165
  # NOT `Array(turns)`: TurnResult is a Struct, so Array() would explode a single
145
166
  # one into its members and hand the policy checks three strings.
146
167
  conversation = turns.nil? || turns.empty? ? [result] : turns
147
- checks = tool_checks(golden, result) + must_not_checks(golden, result) +
148
- policy_checks(golden, conversation)
168
+ checks = tool_checks(golden, result) + call_checks(golden, result) + reply_checks(golden, result) +
169
+ ui_checks(golden, result) + must_not_checks(golden, result) + policy_checks(golden, conversation)
149
170
  CaseResult.new(id: golden.id, agent: golden.agent, error: nil, checks: checks,
150
171
  rubric: golden.rubric, judge: nil)
151
172
  end
@@ -163,6 +184,71 @@ module Insika
163
184
  end
164
185
  end
165
186
 
187
+ # The graders over the turn's CALLS. Each is its own Check, so the report names
188
+ # the one that failed instead of "tool checks failed". All read the LAST turn,
189
+ # like `tools_called`.
190
+ def call_checks(golden, result)
191
+ names = result.tool_names
192
+ checks = golden.never_calls.map do |name|
193
+ hit = names.include?(name)
194
+ Check.new(name: "never_calls:#{name}", pass: !hit, detail: hit ? "called #{name}" : "not called")
195
+ end
196
+
197
+ unless golden.calls_one_of.empty?
198
+ hit = golden.calls_one_of & names
199
+ checks << Check.new(name: "calls_one_of", pass: !hit.empty?,
200
+ detail: hit.empty? ? "none of #{golden.calls_one_of.join(', ')} called (saw: #{names.join(', ')})"
201
+ : "called #{hit.join(', ')}")
202
+ end
203
+
204
+ if (first = golden.first_tool)
205
+ checks << Check.new(name: "first_tool:#{first}", pass: names.first == first,
206
+ detail: "first call was #{names.first || 'none'}")
207
+ end
208
+
209
+ if (max = golden.max_tool_calls)
210
+ checks << Check.new(name: "max_tool_calls:#{max}", pass: names.size <= max,
211
+ detail: "#{names.size} call(s): #{names.join(', ')}")
212
+ end
213
+
214
+ blocked = result.blocked_tools.map { |t| "#{t['name'] || t[:name]}:#{t['gate'] || t[:gate]}" }
215
+ checks + golden.blocked_gates.map do |pair|
216
+ held = blocked.include?(pair)
217
+ Check.new(name: "blocked_gates:#{pair}", pass: held,
218
+ detail: held ? "held" : "not held (blocked: #{blocked.empty? ? 'none' : blocked.join(', ')})")
219
+ end
220
+ end
221
+
222
+ # Case-insensitive substrings over the PUBLISHED answer. `reply_omits` is where
223
+ # an internal id, a CPF or a raw tag leaking into the customer's text is pinned —
224
+ # the negative half of `reply_includes`.
225
+ def reply_checks(golden, result)
226
+ text = result.output_text.to_s.downcase
227
+ golden.reply_includes.map do |s|
228
+ hit = text.include?(s.downcase)
229
+ Check.new(name: "reply_includes:#{s}", pass: hit, detail: hit ? "present" : "missing from the reply")
230
+ end + golden.reply_omits.map do |s|
231
+ hit = text.include?(s.downcase)
232
+ Check.new(name: "reply_omits:#{s}", pass: !hit, detail: hit ? "present in the reply" : "absent")
233
+ end
234
+ end
235
+
236
+ # The graders over what the turn SHOWED. `ui_components` pins that a presentation
237
+ # tool put at least one card of that component in front of the customer;
238
+ # `no_ui` is its negative — a turn that must answer in text alone.
239
+ def ui_checks(golden, result)
240
+ shown = result.shown_components
241
+ checks = golden.ui_components.map do |component|
242
+ hit = shown.include?(component)
243
+ Check.new(name: "ui_components:#{component}", pass: hit,
244
+ detail: hit ? "shown" : "not shown (ui: #{shown.empty? ? 'none' : shown.join(', ')})")
245
+ end
246
+ return checks unless golden.no_ui?
247
+
248
+ checks << Check.new(name: "no_ui", pass: shown.empty?,
249
+ detail: shown.empty? ? "nothing shown" : "shown: #{shown.join(', ')}")
250
+ end
251
+
166
252
  # `must_not` detectors. "tool_error" is special (inspects statuses); the rest
167
253
  # are content detectors over the output text.
168
254
  def must_not_checks(golden, result)
@@ -20,8 +20,16 @@ module Insika
20
20
  # single-tenant default, like `save_artifact`'s own binding_tenant) unless the
21
21
  # case declares one. `run_persona_eval` uses it to keep a QA agent from ever
22
22
  # running (or even seeing) another tenant's persona case in the same store.
23
+ #
24
+ # `state`: the snapshot the conversation starts from — evidence ids, memory
25
+ # facts and notes, prior history, briefing fields — loaded into the session
26
+ # BEFORE turn 1. It is how a case tests "the customer already saw three products
27
+ # and says 'add the second one'" without replaying the search turn: no dependence
28
+ # on the model's first answer, one turn cheaper, and a messy state (a
29
+ # contradiction from six turns ago) becomes reproducible. {} = the case starts
30
+ # empty, as every case did before the key existed.
23
31
  Golden = Struct.new(:id, :agent, :turns, :expect, :requires, :reference, :source, :persona, :tenant,
24
- keyword_init: true) do
32
+ :state, keyword_init: true) do
25
33
  # The user messages to replay, in order. Empty for a persona case: a generated
26
34
  # conversation has no scripted turns.
27
35
  def user_turns = turns.map { |t| t["user"] }
@@ -44,6 +52,26 @@ module Insika
44
52
  # Names of the negative assertions to run (e.g. "pii_leak", "tool_error").
45
53
  def must_not = Array(expect["must_not"]).map(&:to_s)
46
54
 
55
+ # The graders over the turn's CALLS and its published REPLY — each optional,
56
+ # each deterministic. Every positive has a negative: `never_calls` pins what a
57
+ # correct turn does NOT do, which `tools_called` alone can never say.
58
+ def never_calls = Array(expect["never_calls"]).map(&:to_s)
59
+ def calls_one_of = Array(expect["calls_one_of"]).map(&:to_s)
60
+ def first_tool = GoldenLoader.presence(expect["first_tool"])
61
+ def max_tool_calls = expect["max_tool_calls"]
62
+ def reply_includes = Array(expect["reply_includes"]).map(&:to_s)
63
+ def reply_omits = Array(expect["reply_omits"]).map(&:to_s)
64
+ # "tool:gate" pairs that must appear among the turn's BLOCKED calls.
65
+ def blocked_gates = Array(expect["blocked_gates"]).map(&:to_s)
66
+ # UI components a presentation tool must have shown (`insika.ui` frames with
67
+ # at least one card), and its negative: `no_ui: true` pins a turn that shows
68
+ # nothing.
69
+ def ui_components = Array(expect["ui_components"]).map(&:to_s)
70
+ def no_ui? = expect["no_ui"] == true
71
+
72
+ def state = self[:state] || {}
73
+ def seeded? = !state.empty?
74
+
47
75
  # How much the agent should ask before acting. nil = the store
48
76
  # has no opinion and only the rubric decides.
49
77
  def policy = GoldenLoader.presence(expect["policy"])
@@ -111,6 +139,8 @@ module Insika
111
139
  raise InvalidGolden, "#{source}: 'expect' must be a mapping (case '#{id}')" unless expect.is_a?(Hash)
112
140
 
113
141
  validate_policy!(expect["policy"], id: id, source: source)
142
+ validate_graders!(expect, id: id, source: source)
143
+ state = normalize_state(raw["state"], id: id, source: source)
114
144
  requires = raw["requires"] || {}
115
145
  unless requires.is_a?(Hash)
116
146
  raise InvalidGolden, "#{source}: 'requires' must be a mapping (case '#{id}')"
@@ -121,7 +151,7 @@ module Insika
121
151
 
122
152
  Golden.new(id: id, agent: agent, turns: turns, expect: expect,
123
153
  requires: requires, reference: reference, source: source, persona: persona,
124
- tenant: tenant)
154
+ tenant: tenant, state: state)
125
155
  end
126
156
 
127
157
  # `persona:` is the alternative shape to `turns:`: the
@@ -174,6 +204,65 @@ module Insika
174
204
  { "role" => role, "text" => text }.merge(origin ? { "origin" => origin } : {})
175
205
  end
176
206
 
207
+ STATE_KEYS = %w[evidence memory history briefing].freeze
208
+
209
+ # state: the snapshot a case starts from. Absent -> {}. Only the four known
210
+ # keys, each in its own shape — a typo'd key (`evidences:`) would seed nothing
211
+ # and the case would go on passing against the wrong precondition. Refused at
212
+ # LOAD, not at seed time: a case that half-seeds is a hole in the net.
213
+ def normalize_state(raw, id:, source:)
214
+ return {} if raw.nil?
215
+
216
+ where = "#{source}: state (case '#{id}')"
217
+ raise InvalidGolden, "#{where} must be a mapping" unless raw.is_a?(Hash)
218
+
219
+ unknown = raw.keys.map(&:to_s) - STATE_KEYS
220
+ unless unknown.empty?
221
+ raise InvalidGolden, "#{where}: unknown key(s) #{unknown.join(', ')} — known: #{STATE_KEYS.join(', ')}"
222
+ end
223
+
224
+ state = raw.compact
225
+ mapping_with!(state["evidence"], "ids", Array, "#{where}.evidence")
226
+ mapping_with!(state["memory"], "facts", Hash, "#{where}.memory")
227
+ mapping_with!(state["memory"], "notes", Array, "#{where}.memory")
228
+ mapping_with!(state["briefing"], "fields", Hash, "#{where}.briefing")
229
+ history = state["history"]
230
+ unless history.nil? || (history.is_a?(Array) && history.all? { |m| history_message?(m) })
231
+ raise InvalidGolden, "#{where}.history must be [{ role: user|assistant, content: '…' }]"
232
+ end
233
+
234
+ state
235
+ end
236
+
237
+ def mapping_with!(value, key, type, where)
238
+ return if value.nil?
239
+ raise InvalidGolden, "#{where} must be a mapping" unless value.is_a?(Hash)
240
+ return if value[key].nil? || value[key].is_a?(type)
241
+
242
+ raise InvalidGolden, "#{where}.#{key} must be #{type == Array ? 'a list' : 'a mapping'}"
243
+ end
244
+
245
+ def history_message?(message)
246
+ message.is_a?(Hash) && %w[user assistant].include?(message["role"].to_s) &&
247
+ !presence(message["content"]).nil?
248
+ end
249
+
250
+ # The graders with a shape to get wrong: a non-integer `max_tool_calls` would
251
+ # compare against nil and pass; a `blocked_gates` entry without its gate would
252
+ # never match anything and pass. Refused at load, like `policy`.
253
+ def validate_graders!(expect, id:, source:)
254
+ max = expect["max_tool_calls"]
255
+ unless max.nil? || (max.is_a?(Integer) && max >= 0)
256
+ raise InvalidGolden, "#{source}: max_tool_calls must be a non-negative integer (case '#{id}')"
257
+ end
258
+
259
+ Array(expect["blocked_gates"]).each do |pair|
260
+ next if pair.to_s.match?(/\A[^:\s]+:[^:\s]+\z/)
261
+
262
+ raise InvalidGolden, "#{source}: blocked_gates entries are 'tool:gate' (got #{pair.inspect}, case '#{id}')"
263
+ end
264
+ end
265
+
177
266
  # A typo'd policy must not silently mean "no policy" — the case would go on
178
267
  # passing while the rule it was written for stopped being checked. The
179
268
  # `Assertions` constant is resolved at CALL time (this file loads first, and
@@ -66,6 +66,26 @@ module Insika
66
66
  # consumer needing a real Chat UUID as X-Chat-Id) supplies it via conv_map; otherwise the
67
67
  # synthetic "eval-<id>" keeps the adapter's own multi-turn continuation.
68
68
  conv = @conv_map[golden.id] || "eval-#{golden.id}"
69
+ if golden.seeded?
70
+ # The snapshot goes in BEFORE turn 1 — on the same conversation id the turns
71
+ # continue. A transport that cannot seed (A2A: a remote agent has no seed
72
+ # route) SKIPS the case with the reason, never runs it against an empty
73
+ # state and passes. A deployment that refuses seeding (the setting is off)
74
+ # is the same outcome: skipped and reported, which is what the `requires`
75
+ # discipline already promises for anything the deployment lacks.
76
+ unless @transport.respond_to?(:seed)
77
+ return RunCase.new(result: Assertions.skip(golden, "transport cannot seed state"), timings: [])
78
+ end
79
+
80
+ begin
81
+ @transport.seed(conv, golden.state)
82
+ rescue SeedRefused => e
83
+ return RunCase.new(result: Assertions.skip(golden, "deployment refuses seeding — #{e.message}"), timings: [])
84
+ rescue Insika::Error => e
85
+ failed = TurnResult.new(output_text: "", tool_calls: [], error: "seed failed: #{e.message}")
86
+ return RunCase.new(result: Assertions.evaluate(golden, failed), timings: [])
87
+ end
88
+ end
69
89
  turns = []
70
90
  timings = []
71
91
  spent = []