insika 0.7.0 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +191 -0
- data/README.md +9 -6
- data/bin/insika +44 -3
- data/docs/AGENTS.md +74 -14
- data/docs/API.md +73 -0
- data/docs/ARCHITECTURE.md +45 -44
- data/docs/ARTIFACTS.md +42 -0
- data/docs/CHANNELS.md +19 -2
- data/docs/CONTEXT.md +86 -38
- data/docs/DEPLOY.md +27 -7
- data/docs/EVALS.md +98 -8
- data/docs/FACTS.md +4 -0
- data/docs/KNOWLEDGE.md +7 -0
- data/docs/LOADTEST.md +15 -27
- data/docs/MEDIA.md +1 -1
- data/docs/OBSERVABILITY.md +48 -4
- data/docs/POLICY.md +14 -5
- data/docs/RELEASING.md +4 -0
- data/docs/RUNNING-LOCAL.md +2 -2
- data/docs/SECURITY.md +28 -2
- data/docs/SOAK.md +1 -1
- data/docs/TOOLS.md +150 -32
- data/docs/prompts/ADD-TOOL.md +12 -2
- data/docs/prompts/DIAGNOSE-TURN.md +3 -0
- data/docs/prompts/GO-LIVE.md +6 -4
- data/lib/insika/agent_profile.rb +47 -10
- data/lib/insika/channels/web/widget.js +33 -0
- data/lib/insika/channels/web.rb +5 -2
- data/lib/insika/chat_builder.rb +90 -37
- data/lib/insika/commands/agent_payload.rb +1 -1
- data/lib/insika/commands/run_distillation.rb +5 -8
- data/lib/insika/commands/seed_session.rb +118 -0
- data/lib/insika/compaction.rb +196 -0
- data/lib/insika/context/builder.rb +35 -11
- data/lib/insika/context/fragment.rb +4 -1
- data/lib/insika/context/priority.rb +8 -0
- data/lib/insika/context/provider.rb +5 -0
- data/lib/insika/context/providers/briefing.rb +61 -29
- data/lib/insika/context/providers/fence_notice.rb +27 -0
- data/lib/insika/context/providers/knowledge.rb +7 -4
- data/lib/insika/context/providers/memory.rb +8 -4
- data/lib/insika/context/providers/session.rb +50 -10
- data/lib/insika/context_trace_store.rb +11 -1
- data/lib/insika/doctor.rb +213 -10
- data/lib/insika/dsl/runtime.rb +5 -0
- data/lib/insika/dsl.rb +6 -0
- data/lib/insika/edge_limiter.rb +4 -1
- data/lib/insika/env_schema.rb +5 -6
- data/lib/insika/errors.rb +1 -0
- data/lib/insika/evals/assertions.rb +92 -6
- data/lib/insika/evals/golden.rb +91 -2
- data/lib/insika/evals/runner.rb +20 -0
- data/lib/insika/evals/simulator.rb +11 -2
- data/lib/insika/evals/transport.rb +118 -16
- data/lib/insika/evidence.rb +79 -12
- data/lib/insika/executor.rb +94 -23
- data/lib/insika/fence.rb +96 -0
- data/lib/insika/golden_store.rb +3 -0
- data/lib/insika/loop_detector.rb +5 -34
- data/lib/insika/mcp_store.rb +5 -2
- data/lib/insika/mcp_tool_registry.rb +8 -1
- data/lib/insika/memory_store.rb +12 -0
- data/lib/insika/overlay_tool_registry.rb +5 -0
- data/lib/insika/prefix_fingerprint.rb +32 -27
- data/lib/insika/profile_source.rb +8 -0
- data/lib/insika/server/app.rb +43 -1
- data/lib/insika/server/rack_app.rb +2 -0
- data/lib/insika/server/responses.rb +35 -8
- data/lib/insika/session_store.rb +38 -5
- data/lib/insika/settings_store.rb +18 -2
- data/lib/insika/soak/runner.rb +4 -4
- data/lib/insika/spoken_transcript.rb +31 -0
- data/lib/insika/studio/app.rb +34 -8
- data/lib/insika/studio/forms.rb +29 -3
- data/lib/insika/studio/views/_agent_tab_config.erb +5 -1
- data/lib/insika/studio/views/session.erb +1 -1
- data/lib/insika/studio/views/settings.erb +11 -0
- data/lib/insika/studio/views/tool_edit.erb +6 -2
- data/lib/insika/telemetry/recorder.rb +61 -1
- data/lib/insika/templates/daily-digest/README.md +9 -0
- data/lib/insika/templates/research-analyst/agent.rb +10 -0
- data/lib/insika/tool_assembly.rb +21 -13
- data/lib/insika/tool_batch.rb +67 -0
- data/lib/insika/tool_definition.rb +73 -10
- data/lib/insika/tool_envelope.rb +102 -2
- data/lib/insika/tool_store.rb +9 -4
- data/lib/insika/tool_trace_store.rb +1 -1
- data/lib/insika/tool_usage_report.rb +172 -0
- data/lib/insika/tools/data_defined_tool.rb +1 -0
- data/lib/insika/tools/present.rb +122 -0
- data/lib/insika/tools/run_persona_eval.rb +6 -1
- data/lib/insika/tools/tool_search.rb +4 -2
- data/lib/insika/turn_budget.rb +91 -0
- data/lib/insika/turn_state.rb +13 -1
- data/lib/insika/version.rb +1 -1
- data/lib/insika/wiring/graph.rb +7 -0
- data/lib/insika/wiring/graph_chat.rb +4 -0
- data/lib/insika.rb +11 -0
- metadata +10 -1
data/lib/insika/doctor.rb
CHANGED
|
@@ -179,7 +179,9 @@ module Insika
|
|
|
179
179
|
check_relay_channel check_web_widget check_skill_eager check_skill_drift check_shadow_parity
|
|
180
180
|
check_soak_envelope check_turn_timing check_grounding check_cache_layers
|
|
181
181
|
check_memory_scopes check_funnel_declarations check_followup check_distill
|
|
182
|
-
check_harvest check_schedules check_guardrail_corpora
|
|
182
|
+
check_compaction check_harvest check_schedules check_guardrail_corpora
|
|
183
|
+
check_tool_allowlist_policy check_fencing check_eval_seeding
|
|
184
|
+
check_presentation_tools check_provenance]
|
|
183
185
|
|
|
184
186
|
def safe(check)
|
|
185
187
|
Array(send(check))
|
|
@@ -722,26 +724,59 @@ module Insika
|
|
|
722
724
|
# the mangled prompt on every turn, and nothing else would ever say so: the file is
|
|
723
725
|
# present, non-empty, and the agent answers — worse than a crash. Found on the pilot
|
|
724
726
|
# by an `insika refine` report, three weeks after the fact.
|
|
727
|
+
#
|
|
728
|
+
# The same sweep also WARNS (never errors) on a file that outgrew a prompt.
|
|
729
|
+
# Merchant packs are LLM-generated (generate-merchant-pack), and generated prose
|
|
730
|
+
# bloats: the pilot's 28 KB AGENTS.md is the local example, and the ETH Zurich
|
|
731
|
+
# instruction-file study puts the cost of that shape at 20%+ extra tokens per
|
|
732
|
+
# turn for no extra instruction-following. The thresholds are deliberately
|
|
733
|
+
# generous — a hand-written file never meets them; only the generated shape does.
|
|
725
734
|
def check_prompt_files
|
|
726
735
|
return [] unless @agent_file_store
|
|
727
736
|
|
|
728
737
|
agents = @agent_file_store.agents
|
|
729
|
-
wrapped =
|
|
730
|
-
|
|
731
|
-
|
|
732
|
-
|
|
733
|
-
|
|
734
|
-
|
|
735
|
-
|
|
736
|
-
|
|
738
|
+
wrapped = []
|
|
739
|
+
oversized = []
|
|
740
|
+
agents.each do |agent|
|
|
741
|
+
@agent_file_store.list(agent).each do |name|
|
|
742
|
+
content = @agent_file_store.read(agent, name)
|
|
743
|
+
if wrapped_content?(content)
|
|
744
|
+
wrapped << Finding.new(check: "prompt-files", severity: :error, fix: nil,
|
|
745
|
+
message: "agent '#{agent}' file '#{name}' holds a serialized object, not text — " \
|
|
746
|
+
"the model receives `{\"content\" => …}` on one line, escapes and all. " \
|
|
747
|
+
"Recover the markdown from inside the wrapper and write it back.")
|
|
748
|
+
elsif (finding = oversized_prompt(agent, name, content))
|
|
749
|
+
oversized << finding
|
|
750
|
+
end
|
|
737
751
|
end
|
|
738
752
|
end
|
|
739
|
-
return wrapped if wrapped.any?
|
|
753
|
+
return wrapped + oversized if wrapped.any? || oversized.any?
|
|
740
754
|
|
|
741
755
|
total = agents.sum { |a| @agent_file_store.list(a).length }
|
|
742
756
|
[ok("prompt-files", "#{total} prompt file(s) across #{agents.length} agent(s): all text")]
|
|
743
757
|
end
|
|
744
758
|
|
|
759
|
+
# WARN thresholds for one prompt file. ~6 000 estimated tokens (~24 KB) or 600
|
|
760
|
+
# lines: the pilot's generated 28 KB / 292-line AGENTS.md trips the token bar,
|
|
761
|
+
# every hand-written demo file stays far under both. Estimate = chars/4, the
|
|
762
|
+
# same yardstick as Insika::TokenEstimator — cheap and honest about being ±15%.
|
|
763
|
+
PROMPT_FILE_WARN_TOKENS = 6_000
|
|
764
|
+
PROMPT_FILE_WARN_LINES = 600
|
|
765
|
+
|
|
766
|
+
def oversized_prompt(agent, name, content)
|
|
767
|
+
text = content.to_s
|
|
768
|
+
tokens = Insika::TokenEstimator.estimate(text)
|
|
769
|
+
lines = text.lines.count
|
|
770
|
+
return nil if tokens <= PROMPT_FILE_WARN_TOKENS && lines <= PROMPT_FILE_WARN_LINES
|
|
771
|
+
|
|
772
|
+
Finding.new(check: "prompt-files", severity: :warn, fix: nil,
|
|
773
|
+
message: "agent '#{agent}' file '#{name}' is ~#{tokens} tokens over #{lines} line(s) " \
|
|
774
|
+
"(threshold: #{PROMPT_FILE_WARN_TOKENS} tokens / #{PROMPT_FILE_WARN_LINES} lines) — " \
|
|
775
|
+
"a prompt this large costs 20%+ more tokens on every turn for no better " \
|
|
776
|
+
"instruction-following. Trim it, or split the reference material into skills " \
|
|
777
|
+
"the agent loads on demand.")
|
|
778
|
+
end
|
|
779
|
+
|
|
745
780
|
# Cheap and specific: Ruby's inspect of a Hash whose first key is a string. A real
|
|
746
781
|
# prompt does not open with `{"…" =>`.
|
|
747
782
|
def wrapped_content?(content) = /\A\s*\{\s*"[^"]+"\s*=>/.match?(content.to_s)
|
|
@@ -802,6 +837,7 @@ def wrapped_content?(content) = /\A\s*\{\s*"[^"]+"\s*=>/.match?(content.to_s)
|
|
|
802
837
|
# identity block bills a cache write every turn).
|
|
803
838
|
IDENTITY_BUILTINS = %w[
|
|
804
839
|
Insika::Context::Providers::Prompt
|
|
840
|
+
Insika::Context::Providers::FenceNotice
|
|
805
841
|
Insika::Context::Providers::Skill
|
|
806
842
|
Insika::Context::Providers::ToolSearch
|
|
807
843
|
].freeze
|
|
@@ -868,6 +904,111 @@ def wrapped_content?(content) = /\A\s*\{\s*"[^"]+"\s*=>/.match?(content.to_s)
|
|
|
868
904
|
# harmless but useless — every claim passes and the audit reads zero. A
|
|
869
905
|
# warning, never an error: the pack owns matcher quality; the engine refuses
|
|
870
906
|
# only uncompileable data.
|
|
907
|
+
# A stored agent that declares `tools_allow` / `tools_deny` /
|
|
908
|
+
# `tools_allow_groups` but does not name the `tool_allowlist` policy. Only
|
|
909
|
+
# that policy applies the lists, and the Policy::Engine runs ONLY the
|
|
910
|
+
# policies a profile names — so as written on disk the lists are inert and
|
|
911
|
+
# the model receives every registered tool (hundreds of schemas, tens of
|
|
912
|
+
# thousands of tokens, per request).
|
|
913
|
+
#
|
|
914
|
+
# AgentProfile.build now adds the policy whenever a list is declared, so a
|
|
915
|
+
# live turn off this record is already safe. The record itself stays wrong
|
|
916
|
+
# until something re-saves it, and anything that reads the stored shape
|
|
917
|
+
# directly — an older engine, a pack export, an operator auditing the
|
|
918
|
+
# Studio form — still sees an allowlist that does nothing. That is why this
|
|
919
|
+
# reads the RAW record and not `all`.
|
|
920
|
+
def check_tool_allowlist_policy
|
|
921
|
+
return [] unless @profile_source.respond_to?(:all_raw)
|
|
922
|
+
|
|
923
|
+
records = @profile_source.all_raw
|
|
924
|
+
bad = records.select { |r| tool_lists_declared?(r) && !names_tool_allowlist?(r) }
|
|
925
|
+
return [ok("tool-allowlist", "#{records.length} stored agent(s): every declared tool list names the policy")] if bad.empty?
|
|
926
|
+
|
|
927
|
+
bad.map do |r|
|
|
928
|
+
Finding.new(check: "tool-allowlist", severity: :error, fix: nil,
|
|
929
|
+
message: "agent '#{r["id"]}' declares a tool allow/deny list but its policies " \
|
|
930
|
+
"(#{Array(r["policies"]).join(", ").then { |p| p.empty? ? "none" : p }}) " \
|
|
931
|
+
"do not include tool_allowlist — as stored, the list is never applied and " \
|
|
932
|
+
"the model receives every registered tool. The engine repairs this when it " \
|
|
933
|
+
"loads the agent; re-save it (Studio, or the pack) so the stored policies " \
|
|
934
|
+
"match the intent.")
|
|
935
|
+
end
|
|
936
|
+
end
|
|
937
|
+
|
|
938
|
+
# Presence, not emptiness, for the two nil-able lists: `tools_allow: []`
|
|
939
|
+
# means "no tools" and is as much a declaration as a list of names.
|
|
940
|
+
# `tools_deny` has no nil state, so only a non-empty one counts.
|
|
941
|
+
def tool_lists_declared?(raw)
|
|
942
|
+
!raw["tools_allow"].nil? || !raw["tools_allow_groups"].nil? || !Array(raw["tools_deny"]).empty?
|
|
943
|
+
end
|
|
944
|
+
|
|
945
|
+
def names_tool_allowlist?(raw) = Array(raw["policies"]).any? { |p| p.to_s == "tool_allowlist" }
|
|
946
|
+
|
|
947
|
+
# A stored agent that allows a PRESENTATION tool but no tool declaring
|
|
948
|
+
# `evidence`. The presentation tool shows only cards an evidence tool returned
|
|
949
|
+
# this session, so with no evidence tool in reach every call drops every id —
|
|
950
|
+
# the model keeps asking to show products and nothing ever appears. Silent
|
|
951
|
+
# unless a presentation tool exists. Data tools only: a code tool's evidence
|
|
952
|
+
# lives in registry metadata the doctor does not read.
|
|
953
|
+
def check_presentation_tools
|
|
954
|
+
return [] unless @tool_store && @profile_source
|
|
955
|
+
|
|
956
|
+
raws = @tool_store.all_raw
|
|
957
|
+
presenting = raws.select { |r| r["presentation"] }.map { |r| r["name"].to_s }
|
|
958
|
+
return [] if presenting.empty?
|
|
959
|
+
|
|
960
|
+
evidence = raws.select { |r| r["evidence"] }.map { |r| r["name"].to_s }
|
|
961
|
+
bad = @profile_source.all.filter_map do |profile|
|
|
962
|
+
allowed = allowed_data_tools(profile, raws)
|
|
963
|
+
shows = presenting & allowed
|
|
964
|
+
next if shows.empty? || !(evidence & allowed).empty?
|
|
965
|
+
|
|
966
|
+
Finding.new(check: "presentation-tools", severity: :warn, fix: nil,
|
|
967
|
+
message: "agent '#{profile.id}' allows #{shows.join(', ')} but no tool declaring " \
|
|
968
|
+
"evidence — a presentation tool can only show cards an evidence tool " \
|
|
969
|
+
"returned this session, so every call would drop every id. Allow the " \
|
|
970
|
+
"search tool that declares `evidence`, or drop the presentation tool.")
|
|
971
|
+
end
|
|
972
|
+
return bad unless bad.empty?
|
|
973
|
+
|
|
974
|
+
[ok("presentation-tools", "#{presenting.length} presentation tool(s): every agent that shows cards " \
|
|
975
|
+
"also allows an evidence tool")]
|
|
976
|
+
end
|
|
977
|
+
|
|
978
|
+
# The data tools a profile may call, by the allowlist's own rules: nil lists =
|
|
979
|
+
# all; otherwise names ∪ groups; deny wins.
|
|
980
|
+
def allowed_data_tools(profile, raws)
|
|
981
|
+
names = profile.tools_allow.nil? ? nil : Array(profile.tools_allow).map(&:to_s)
|
|
982
|
+
groups = profile.tools_allow_groups.nil? ? nil : Array(profile.tools_allow_groups).map(&:to_s)
|
|
983
|
+
allowed = if names.nil? && groups.nil?
|
|
984
|
+
raws.map { |r| r["name"].to_s }
|
|
985
|
+
else
|
|
986
|
+
Array(names) | raws.select { |r| Array(groups).include?(r["group"].to_s) }.map { |r| r["name"].to_s }
|
|
987
|
+
end
|
|
988
|
+
allowed - Array(profile.tools_deny).map(&:to_s)
|
|
989
|
+
end
|
|
990
|
+
|
|
991
|
+
def check_provenance
|
|
992
|
+
return [] unless @tool_store && @profile_source.respond_to?(:all_raw)
|
|
993
|
+
|
|
994
|
+
tools = @tool_store.all_raw
|
|
995
|
+
@profile_source.all_raw.filter_map do |raw|
|
|
996
|
+
allowed = tools.select do |tool|
|
|
997
|
+
name = tool["name"]
|
|
998
|
+
unrestricted = raw["tools_allow"].nil? && raw["tools_allow_groups"].nil?
|
|
999
|
+
(unrestricted || Array(raw["tools_allow"]).include?(name) ||
|
|
1000
|
+
Array(raw["tools_allow_groups"]).include?(tool["group"])) &&
|
|
1001
|
+
!Array(raw["tools_deny"]).include?(name)
|
|
1002
|
+
end
|
|
1003
|
+
gated = allowed.select { |tool| tool["requires_evidence"] }
|
|
1004
|
+
next if gated.empty? || allowed.any? { |tool| tool["evidence"] }
|
|
1005
|
+
|
|
1006
|
+
Finding.new(check: "provenance", severity: :warn, fix: nil,
|
|
1007
|
+
message: "agent '#{raw["id"]}' allows #{gated.map { |t| t["name"] }.join(', ')} " \
|
|
1008
|
+
"with requires_evidence but no stored evidence tool — verify that an allowed code tool supplies evidence")
|
|
1009
|
+
end
|
|
1010
|
+
end
|
|
1011
|
+
|
|
871
1012
|
def check_grounding
|
|
872
1013
|
return [] unless @profile_source
|
|
873
1014
|
|
|
@@ -1230,6 +1371,68 @@ def wrapped_content?(content) = /\A\s*\{\s*"[^"]+"\s*=>/.match?(content.to_s)
|
|
|
1230
1371
|
stale: @proposal_store.stale(limit: 10_000).size }
|
|
1231
1372
|
end
|
|
1232
1373
|
|
|
1374
|
+
# the in-session compaction check (RFC-0044): platform-gated,
|
|
1375
|
+
# so ONE line — enabled with no resolvable model (no compaction.model, no
|
|
1376
|
+
# platform utility_model) can never run; the warn is the same "declared
|
|
1377
|
+
# but dead" signal as check_distill's (the engine never guesses a model).
|
|
1378
|
+
def check_compaction
|
|
1379
|
+
return [] unless @settings_store
|
|
1380
|
+
|
|
1381
|
+
settings = @settings_store.get
|
|
1382
|
+
config = settings["compaction"] || {}
|
|
1383
|
+
return [ok("compaction", "in-session compaction off")] unless Coercion.truthy?(config["enabled"])
|
|
1384
|
+
|
|
1385
|
+
if Coercion.presence(config["model"]).nil? && Coercion.presence(settings["utility_model"]).nil?
|
|
1386
|
+
[Finding.new(check: "compaction", severity: :warn, fix: nil,
|
|
1387
|
+
message: "in-session compaction is enabled but has no model slot — it will " \
|
|
1388
|
+
"never run (set compaction.model or the platform utility_model).")]
|
|
1389
|
+
else
|
|
1390
|
+
[ok("compaction", "in-session compaction on — keep_last #{config['keep_last']}, " \
|
|
1391
|
+
"compact_after #{config['compact_after']}")]
|
|
1392
|
+
end
|
|
1393
|
+
end
|
|
1394
|
+
|
|
1395
|
+
# `evals.seeding` opens POST /v1/conversations/:id/seed: a snapshot written
|
|
1396
|
+
# into a real conversation under the tenant token, before its first turn. Right
|
|
1397
|
+
# on the machine running snapshot evals; wrong left on in production, where any
|
|
1398
|
+
# consumer holding the token could write a customer's history. A warning, not
|
|
1399
|
+
# an error — the operator may be running the evals right now.
|
|
1400
|
+
def check_eval_seeding
|
|
1401
|
+
return [] unless @settings_store
|
|
1402
|
+
|
|
1403
|
+
on = ((@settings_store.get["evals"] || {})["seeding"]) == true
|
|
1404
|
+
return [ok("eval-seeding", "evals.seeding off — conversations cannot be seeded")] unless on
|
|
1405
|
+
|
|
1406
|
+
[Finding.new(check: "eval-seeding", severity: :warn, fix: nil,
|
|
1407
|
+
message: "evals.seeding is ON — POST /v1/conversations/:id/seed accepts a fabricated " \
|
|
1408
|
+
"conversation state under the tenant token. Fine while running snapshot " \
|
|
1409
|
+
"evals; turn it off (Studio > Settings, or `update_settings`) in production.")]
|
|
1410
|
+
end
|
|
1411
|
+
|
|
1412
|
+
# A customer-facing agent — one reachable through an inbound channel: any
|
|
1413
|
+
# agent when the relay is mounted (the event names the agent), the listed
|
|
1414
|
+
# ones for the widget — with `fencing` off reads third-party bytes as-is.
|
|
1415
|
+
# A warning, never an error: the default is off this release on purpose
|
|
1416
|
+
# (the goldens were baselined unfenced), so the message says exactly what
|
|
1417
|
+
# is unsanitized today. No inbound channel = nothing customer-facing = silent.
|
|
1418
|
+
def check_fencing
|
|
1419
|
+
return [] unless @profile_source
|
|
1420
|
+
|
|
1421
|
+
relay = Insika::Coercion.present?(@env["INSIKA_RELAY_TOKEN"])
|
|
1422
|
+
widget = @env["INSIKA_WIDGET_AGENTS"].to_s.split(",").map(&:strip).reject(&:empty?)
|
|
1423
|
+
return [] unless relay || widget.any?
|
|
1424
|
+
|
|
1425
|
+
exposed = @profile_source.all.select { |p| relay || widget.include?(p.id.to_s) }
|
|
1426
|
+
unfenced = exposed.reject { |p| Insika::Fence.enabled?(p) }.map(&:id)
|
|
1427
|
+
return [ok("fencing", "fencing on for every agent behind an inbound channel")] if unfenced.empty?
|
|
1428
|
+
|
|
1429
|
+
[Finding.new(check: "fencing", severity: :warn, fix: nil,
|
|
1430
|
+
message: "agent(s) #{unfenced.join(', ')}: reachable through an inbound channel with " \
|
|
1431
|
+
"`fencing` off — tool results, <memory> facts and <knowledge> concepts reach " \
|
|
1432
|
+
"the model byte-for-byte (invisible characters, forged turn markers and " \
|
|
1433
|
+
"transcript-shaped tags included). Set `fencing true` on the agent.")]
|
|
1434
|
+
end
|
|
1435
|
+
|
|
1233
1436
|
# the harvest check — per profile WITH a harvest hash:
|
|
1234
1437
|
# declared-without-model warn (D12), no grounding matcher warn (D3),
|
|
1235
1438
|
# malformed negative list error (D4), else ok with the pending counts.
|
data/lib/insika/dsl/runtime.rb
CHANGED
|
@@ -231,11 +231,16 @@ module Insika
|
|
|
231
231
|
base: "", catalog: c[:prompt_catalog],
|
|
232
232
|
agent_files: c[:agent_file_store], system_files: c[:system_file_store]
|
|
233
233
|
),
|
|
234
|
+
Insika::Context::Providers::FenceNotice.new,
|
|
234
235
|
Insika::Context::Providers::Skill.new(catalog: c[:skill_catalog]),
|
|
235
236
|
Insika::Context::Providers::SkillTrigger.new(catalog: c[:skill_catalog]),
|
|
236
237
|
Insika::Context::Providers::ToolSearch.new(catalog: c[:tool_catalog]),
|
|
237
238
|
Insika::Context::Providers::Memory.new(store: spine.memory_store),
|
|
238
239
|
Insika::Context::Providers::Knowledge.new(store: spine.knowledge_store),
|
|
240
|
+
# Session briefing: read path. Inert for agents without briefing_fields.
|
|
241
|
+
# Before Session, so the durable head renders above the transcript and
|
|
242
|
+
# the tail recitation lands after it.
|
|
243
|
+
Insika::Context::Providers::Briefing.new(session_store: spine.session_store),
|
|
239
244
|
Insika::Context::Providers::Session.new(session_store: spine.session_store)
|
|
240
245
|
] # NOT frozen: load_plugins appends plugin providers at boot
|
|
241
246
|
end
|
data/lib/insika/dsl.rb
CHANGED
|
@@ -417,6 +417,12 @@ module Insika
|
|
|
417
417
|
# SEES: repeated identical tool results collapse to a back-reference.
|
|
418
418
|
def tool_output_compression(on = true) = @config[:tool_output_compression] = on
|
|
419
419
|
|
|
420
|
+
# Third-party text is sanitized before the model reads it (tool results,
|
|
421
|
+
# memory, knowledge) and a one-sentence notice names those blocks as data.
|
|
422
|
+
# Opt-in this release — the goldens were baselined on unfenced bytes:
|
|
423
|
+
# fencing true
|
|
424
|
+
def fencing(on = true) = @config[:fencing] = on
|
|
425
|
+
|
|
420
426
|
# Content-safety guardrails — opt-in and configurable per agent.
|
|
421
427
|
# Pure config-over-code: the hash is stored on the profile and consumed by
|
|
422
428
|
# Safety::Config.from_profile. Merges, so repeated calls accumulate.
|
data/lib/insika/edge_limiter.rb
CHANGED
|
@@ -263,12 +263,15 @@ module Insika
|
|
|
263
263
|
|
|
264
264
|
# Appends the note to the assembled system prompt: the real Data package is
|
|
265
265
|
# immutable (with), the specs' minimal Struct is mutable — both duck-typed.
|
|
266
|
+
# The note is per-turn data, so it also lands in the volatile layer — below
|
|
267
|
+
# the cache breakpoint, never inside the cached identity prefix.
|
|
266
268
|
def inject_budget_note(state, note)
|
|
267
269
|
ctx = state.context
|
|
268
270
|
return if ctx.nil?
|
|
269
271
|
|
|
270
272
|
if ctx.respond_to?(:with)
|
|
271
|
-
|
|
273
|
+
volatile = [ctx.system_volatile, note].reject { |s| s.to_s.empty? }.join("\n\n")
|
|
274
|
+
state.context = ctx.with(system: "#{ctx.system}\n\n#{note}", system_volatile: volatile)
|
|
272
275
|
elsif ctx.respond_to?(:system=)
|
|
273
276
|
ctx.system = "#{ctx.system}\n\n#{note}"
|
|
274
277
|
end
|
data/lib/insika/env_schema.rb
CHANGED
|
@@ -72,10 +72,10 @@ module Insika
|
|
|
72
72
|
|
|
73
73
|
# Prefixes the engine fully OWNS: an unknown key under one of these is a typo, not
|
|
74
74
|
# a foreign var. INSIKA_ (current) and HARNESS_ (legacy, still honored during the
|
|
75
|
-
# deprecation window). Deliberately NOT OPENCLAW_ (
|
|
76
|
-
#
|
|
77
|
-
#
|
|
78
|
-
#
|
|
75
|
+
# deprecation window). Deliberately NOT OPENCLAW_ (the OpenClaw gateway product
|
|
76
|
+
# sets its own OPENCLAW_HOME/_STATE_DIR/… on the same host; the engine reads none
|
|
77
|
+
# of them), nor LITESTREAM_ (the sidecar owns it), nor OTEL_ (the OpenTelemetry
|
|
78
|
+
# SDK owns its env).
|
|
79
79
|
OWNED_PREFIXES = [PREFIX, LEGACY_PREFIX].freeze
|
|
80
80
|
|
|
81
81
|
BOOLEANS = %w[1 0 true false yes no on off].freeze
|
|
@@ -142,8 +142,7 @@ module Insika
|
|
|
142
142
|
spec(name: "INSIKA_ROUTER_BACKEND_TIMEOUT", type: :integer, description: "`insika-router`'s connect/read timeout to a backend, in seconds (default 10)."),
|
|
143
143
|
spec(name: "INSIKA_ROUTER_HOST", description: "bind address for `insika-router` itself (default 0.0.0.0)."),
|
|
144
144
|
spec(name: "INSIKA_ROUTER_PORT", type: :integer, description: "listen port for `insika-router` itself (default 9090)."),
|
|
145
|
-
spec(name: "
|
|
146
|
-
spec(name: "OPENCLAW_AGENTS_DIR", type: :path, description: "Directory of OpenClaw-style agent packs."),
|
|
145
|
+
spec(name: "INSIKA_GATEWAY_TOKEN", secret: true, description: "Bearer for /v1 + /a2a (falls back to ADMIN_TOKEN when unset)."),
|
|
147
146
|
spec(name: "INSIKA_PLUGIN_DIR", type: :path, description: "Workspace plugin root (directories with insika.plugin.yml). Loaded at boot; ids still need INSIKA_PLUGINS."),
|
|
148
147
|
spec(name: "INSIKA_PLUGINS", type: :csv, description: "Plugin ids to enable from the workspace/bundled roots. Announced gems are enabled by installing them."),
|
|
149
148
|
spec(name: "INSIKA_PLUGINS_DISABLED", type: :csv, description: "Plugin ids that never load — the absolute veto, wins over INSIKA_PLUGINS and over an announced gem."),
|
data/lib/insika/errors.rb
CHANGED
|
@@ -8,6 +8,7 @@ module Insika
|
|
|
8
8
|
|
|
9
9
|
class ValidationError < Error; end # Malformed Command -> HTTP 422, no Task created
|
|
10
10
|
class NotFoundError < Error; end # nonexistent session/task/agent -> HTTP 404
|
|
11
|
+
class ConflictError < Error; end # the write contradicts state that already exists -> HTTP 409
|
|
11
12
|
|
|
12
13
|
# Policy Engine denied -> :policy_denied event, task :failed
|
|
13
14
|
class PolicyDenied < Error
|
|
@@ -13,14 +13,35 @@ module Insika
|
|
|
13
13
|
# offline without a server.
|
|
14
14
|
#
|
|
15
15
|
# output_text: the final assistant text (for content checks)
|
|
16
|
-
# tool_calls: [{ "name" =>, "status" => }] captured
|
|
16
|
+
# tool_calls: [{ "name" =>, "arguments" =>, "status" =>, "gate" => }] captured
|
|
17
|
+
# from the stream's tool events. `arguments` (a Hash) and
|
|
18
|
+
# `status` ("ok" | "error" | "blocked", with `gate` naming what
|
|
19
|
+
# held a blocked call) arrive from the item frames; an older
|
|
20
|
+
# deployment sends names only and they stay nil.
|
|
21
|
+
# ui: [{ "component" =>, "count" => }] — one per presentation call
|
|
22
|
+
# (the `insika.ui` frames); nil/[] when the turn showed nothing
|
|
17
23
|
# error: transport/turn error string, or nil on a clean turn
|
|
18
|
-
TurnResult = Struct.new(:output_text, :tool_calls, :error, keyword_init: true) do
|
|
24
|
+
TurnResult = Struct.new(:output_text, :tool_calls, :ui, :error, keyword_init: true) do
|
|
19
25
|
def tool_names = Array(tool_calls).map { |t| (t["name"] || t[:name]).to_s }
|
|
20
26
|
|
|
21
|
-
#
|
|
27
|
+
# Components that actually put a card in front of the customer (count > 0).
|
|
28
|
+
def shown_components
|
|
29
|
+
Array(ui).select { |u| (u["count"] || u[:count]).to_i.positive? }
|
|
30
|
+
.map { |u| (u["component"] || u[:component]).to_s }
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
# A tool call whose status is anything but a success ("ok"/2xx/"success"). A
|
|
34
|
+
# BLOCKED call is not an error: the tool never ran, a gate held it — that is
|
|
35
|
+
# the `blocked_gates` grader's business, not `must_not: tool_error`'s.
|
|
22
36
|
def errored_tools
|
|
23
|
-
Array(tool_calls).reject
|
|
37
|
+
Array(tool_calls).reject do |t|
|
|
38
|
+
status = t["status"] || t[:status]
|
|
39
|
+
Assertions.ok_status?(status) || status.to_s == "blocked"
|
|
40
|
+
end
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
def blocked_tools
|
|
44
|
+
Array(tool_calls).select { |t| (t["status"] || t[:status]).to_s == "blocked" }
|
|
24
45
|
end
|
|
25
46
|
end
|
|
26
47
|
|
|
@@ -144,8 +165,8 @@ module Insika
|
|
|
144
165
|
# NOT `Array(turns)`: TurnResult is a Struct, so Array() would explode a single
|
|
145
166
|
# one into its members and hand the policy checks three strings.
|
|
146
167
|
conversation = turns.nil? || turns.empty? ? [result] : turns
|
|
147
|
-
checks = tool_checks(golden, result) +
|
|
148
|
-
policy_checks(golden, conversation)
|
|
168
|
+
checks = tool_checks(golden, result) + call_checks(golden, result) + reply_checks(golden, result) +
|
|
169
|
+
ui_checks(golden, result) + must_not_checks(golden, result) + policy_checks(golden, conversation)
|
|
149
170
|
CaseResult.new(id: golden.id, agent: golden.agent, error: nil, checks: checks,
|
|
150
171
|
rubric: golden.rubric, judge: nil)
|
|
151
172
|
end
|
|
@@ -163,6 +184,71 @@ module Insika
|
|
|
163
184
|
end
|
|
164
185
|
end
|
|
165
186
|
|
|
187
|
+
# The graders over the turn's CALLS. Each is its own Check, so the report names
|
|
188
|
+
# the one that failed instead of "tool checks failed". All read the LAST turn,
|
|
189
|
+
# like `tools_called`.
|
|
190
|
+
def call_checks(golden, result)
|
|
191
|
+
names = result.tool_names
|
|
192
|
+
checks = golden.never_calls.map do |name|
|
|
193
|
+
hit = names.include?(name)
|
|
194
|
+
Check.new(name: "never_calls:#{name}", pass: !hit, detail: hit ? "called #{name}" : "not called")
|
|
195
|
+
end
|
|
196
|
+
|
|
197
|
+
unless golden.calls_one_of.empty?
|
|
198
|
+
hit = golden.calls_one_of & names
|
|
199
|
+
checks << Check.new(name: "calls_one_of", pass: !hit.empty?,
|
|
200
|
+
detail: hit.empty? ? "none of #{golden.calls_one_of.join(', ')} called (saw: #{names.join(', ')})"
|
|
201
|
+
: "called #{hit.join(', ')}")
|
|
202
|
+
end
|
|
203
|
+
|
|
204
|
+
if (first = golden.first_tool)
|
|
205
|
+
checks << Check.new(name: "first_tool:#{first}", pass: names.first == first,
|
|
206
|
+
detail: "first call was #{names.first || 'none'}")
|
|
207
|
+
end
|
|
208
|
+
|
|
209
|
+
if (max = golden.max_tool_calls)
|
|
210
|
+
checks << Check.new(name: "max_tool_calls:#{max}", pass: names.size <= max,
|
|
211
|
+
detail: "#{names.size} call(s): #{names.join(', ')}")
|
|
212
|
+
end
|
|
213
|
+
|
|
214
|
+
blocked = result.blocked_tools.map { |t| "#{t['name'] || t[:name]}:#{t['gate'] || t[:gate]}" }
|
|
215
|
+
checks + golden.blocked_gates.map do |pair|
|
|
216
|
+
held = blocked.include?(pair)
|
|
217
|
+
Check.new(name: "blocked_gates:#{pair}", pass: held,
|
|
218
|
+
detail: held ? "held" : "not held (blocked: #{blocked.empty? ? 'none' : blocked.join(', ')})")
|
|
219
|
+
end
|
|
220
|
+
end
|
|
221
|
+
|
|
222
|
+
# Case-insensitive substrings over the PUBLISHED answer. `reply_omits` is where
|
|
223
|
+
# an internal id, a CPF or a raw tag leaking into the customer's text is pinned —
|
|
224
|
+
# the negative half of `reply_includes`.
|
|
225
|
+
def reply_checks(golden, result)
|
|
226
|
+
text = result.output_text.to_s.downcase
|
|
227
|
+
golden.reply_includes.map do |s|
|
|
228
|
+
hit = text.include?(s.downcase)
|
|
229
|
+
Check.new(name: "reply_includes:#{s}", pass: hit, detail: hit ? "present" : "missing from the reply")
|
|
230
|
+
end + golden.reply_omits.map do |s|
|
|
231
|
+
hit = text.include?(s.downcase)
|
|
232
|
+
Check.new(name: "reply_omits:#{s}", pass: !hit, detail: hit ? "present in the reply" : "absent")
|
|
233
|
+
end
|
|
234
|
+
end
|
|
235
|
+
|
|
236
|
+
# The graders over what the turn SHOWED. `ui_components` pins that a presentation
|
|
237
|
+
# tool put at least one card of that component in front of the customer;
|
|
238
|
+
# `no_ui` is its negative — a turn that must answer in text alone.
|
|
239
|
+
def ui_checks(golden, result)
|
|
240
|
+
shown = result.shown_components
|
|
241
|
+
checks = golden.ui_components.map do |component|
|
|
242
|
+
hit = shown.include?(component)
|
|
243
|
+
Check.new(name: "ui_components:#{component}", pass: hit,
|
|
244
|
+
detail: hit ? "shown" : "not shown (ui: #{shown.empty? ? 'none' : shown.join(', ')})")
|
|
245
|
+
end
|
|
246
|
+
return checks unless golden.no_ui?
|
|
247
|
+
|
|
248
|
+
checks << Check.new(name: "no_ui", pass: shown.empty?,
|
|
249
|
+
detail: shown.empty? ? "nothing shown" : "shown: #{shown.join(', ')}")
|
|
250
|
+
end
|
|
251
|
+
|
|
166
252
|
# `must_not` detectors. "tool_error" is special (inspects statuses); the rest
|
|
167
253
|
# are content detectors over the output text.
|
|
168
254
|
def must_not_checks(golden, result)
|
data/lib/insika/evals/golden.rb
CHANGED
|
@@ -20,8 +20,16 @@ module Insika
|
|
|
20
20
|
# single-tenant default, like `save_artifact`'s own binding_tenant) unless the
|
|
21
21
|
# case declares one. `run_persona_eval` uses it to keep a QA agent from ever
|
|
22
22
|
# running (or even seeing) another tenant's persona case in the same store.
|
|
23
|
+
#
|
|
24
|
+
# `state`: the snapshot the conversation starts from — evidence ids, memory
|
|
25
|
+
# facts and notes, prior history, briefing fields — loaded into the session
|
|
26
|
+
# BEFORE turn 1. It is how a case tests "the customer already saw three products
|
|
27
|
+
# and says 'add the second one'" without replaying the search turn: no dependence
|
|
28
|
+
# on the model's first answer, one turn cheaper, and a messy state (a
|
|
29
|
+
# contradiction from six turns ago) becomes reproducible. {} = the case starts
|
|
30
|
+
# empty, as every case did before the key existed.
|
|
23
31
|
Golden = Struct.new(:id, :agent, :turns, :expect, :requires, :reference, :source, :persona, :tenant,
|
|
24
|
-
keyword_init: true) do
|
|
32
|
+
:state, keyword_init: true) do
|
|
25
33
|
# The user messages to replay, in order. Empty for a persona case: a generated
|
|
26
34
|
# conversation has no scripted turns.
|
|
27
35
|
def user_turns = turns.map { |t| t["user"] }
|
|
@@ -44,6 +52,26 @@ module Insika
|
|
|
44
52
|
# Names of the negative assertions to run (e.g. "pii_leak", "tool_error").
|
|
45
53
|
def must_not = Array(expect["must_not"]).map(&:to_s)
|
|
46
54
|
|
|
55
|
+
# The graders over the turn's CALLS and its published REPLY — each optional,
|
|
56
|
+
# each deterministic. Every positive has a negative: `never_calls` pins what a
|
|
57
|
+
# correct turn does NOT do, which `tools_called` alone can never say.
|
|
58
|
+
def never_calls = Array(expect["never_calls"]).map(&:to_s)
|
|
59
|
+
def calls_one_of = Array(expect["calls_one_of"]).map(&:to_s)
|
|
60
|
+
def first_tool = GoldenLoader.presence(expect["first_tool"])
|
|
61
|
+
def max_tool_calls = expect["max_tool_calls"]
|
|
62
|
+
def reply_includes = Array(expect["reply_includes"]).map(&:to_s)
|
|
63
|
+
def reply_omits = Array(expect["reply_omits"]).map(&:to_s)
|
|
64
|
+
# "tool:gate" pairs that must appear among the turn's BLOCKED calls.
|
|
65
|
+
def blocked_gates = Array(expect["blocked_gates"]).map(&:to_s)
|
|
66
|
+
# UI components a presentation tool must have shown (`insika.ui` frames with
|
|
67
|
+
# at least one card), and its negative: `no_ui: true` pins a turn that shows
|
|
68
|
+
# nothing.
|
|
69
|
+
def ui_components = Array(expect["ui_components"]).map(&:to_s)
|
|
70
|
+
def no_ui? = expect["no_ui"] == true
|
|
71
|
+
|
|
72
|
+
def state = self[:state] || {}
|
|
73
|
+
def seeded? = !state.empty?
|
|
74
|
+
|
|
47
75
|
# How much the agent should ask before acting. nil = the store
|
|
48
76
|
# has no opinion and only the rubric decides.
|
|
49
77
|
def policy = GoldenLoader.presence(expect["policy"])
|
|
@@ -111,6 +139,8 @@ module Insika
|
|
|
111
139
|
raise InvalidGolden, "#{source}: 'expect' must be a mapping (case '#{id}')" unless expect.is_a?(Hash)
|
|
112
140
|
|
|
113
141
|
validate_policy!(expect["policy"], id: id, source: source)
|
|
142
|
+
validate_graders!(expect, id: id, source: source)
|
|
143
|
+
state = normalize_state(raw["state"], id: id, source: source)
|
|
114
144
|
requires = raw["requires"] || {}
|
|
115
145
|
unless requires.is_a?(Hash)
|
|
116
146
|
raise InvalidGolden, "#{source}: 'requires' must be a mapping (case '#{id}')"
|
|
@@ -121,7 +151,7 @@ module Insika
|
|
|
121
151
|
|
|
122
152
|
Golden.new(id: id, agent: agent, turns: turns, expect: expect,
|
|
123
153
|
requires: requires, reference: reference, source: source, persona: persona,
|
|
124
|
-
tenant: tenant)
|
|
154
|
+
tenant: tenant, state: state)
|
|
125
155
|
end
|
|
126
156
|
|
|
127
157
|
# `persona:` is the alternative shape to `turns:`: the
|
|
@@ -174,6 +204,65 @@ module Insika
|
|
|
174
204
|
{ "role" => role, "text" => text }.merge(origin ? { "origin" => origin } : {})
|
|
175
205
|
end
|
|
176
206
|
|
|
207
|
+
STATE_KEYS = %w[evidence memory history briefing].freeze
|
|
208
|
+
|
|
209
|
+
# state: the snapshot a case starts from. Absent -> {}. Only the four known
|
|
210
|
+
# keys, each in its own shape — a typo'd key (`evidences:`) would seed nothing
|
|
211
|
+
# and the case would go on passing against the wrong precondition. Refused at
|
|
212
|
+
# LOAD, not at seed time: a case that half-seeds is a hole in the net.
|
|
213
|
+
def normalize_state(raw, id:, source:)
|
|
214
|
+
return {} if raw.nil?
|
|
215
|
+
|
|
216
|
+
where = "#{source}: state (case '#{id}')"
|
|
217
|
+
raise InvalidGolden, "#{where} must be a mapping" unless raw.is_a?(Hash)
|
|
218
|
+
|
|
219
|
+
unknown = raw.keys.map(&:to_s) - STATE_KEYS
|
|
220
|
+
unless unknown.empty?
|
|
221
|
+
raise InvalidGolden, "#{where}: unknown key(s) #{unknown.join(', ')} — known: #{STATE_KEYS.join(', ')}"
|
|
222
|
+
end
|
|
223
|
+
|
|
224
|
+
state = raw.compact
|
|
225
|
+
mapping_with!(state["evidence"], "ids", Array, "#{where}.evidence")
|
|
226
|
+
mapping_with!(state["memory"], "facts", Hash, "#{where}.memory")
|
|
227
|
+
mapping_with!(state["memory"], "notes", Array, "#{where}.memory")
|
|
228
|
+
mapping_with!(state["briefing"], "fields", Hash, "#{where}.briefing")
|
|
229
|
+
history = state["history"]
|
|
230
|
+
unless history.nil? || (history.is_a?(Array) && history.all? { |m| history_message?(m) })
|
|
231
|
+
raise InvalidGolden, "#{where}.history must be [{ role: user|assistant, content: '…' }]"
|
|
232
|
+
end
|
|
233
|
+
|
|
234
|
+
state
|
|
235
|
+
end
|
|
236
|
+
|
|
237
|
+
def mapping_with!(value, key, type, where)
|
|
238
|
+
return if value.nil?
|
|
239
|
+
raise InvalidGolden, "#{where} must be a mapping" unless value.is_a?(Hash)
|
|
240
|
+
return if value[key].nil? || value[key].is_a?(type)
|
|
241
|
+
|
|
242
|
+
raise InvalidGolden, "#{where}.#{key} must be #{type == Array ? 'a list' : 'a mapping'}"
|
|
243
|
+
end
|
|
244
|
+
|
|
245
|
+
def history_message?(message)
|
|
246
|
+
message.is_a?(Hash) && %w[user assistant].include?(message["role"].to_s) &&
|
|
247
|
+
!presence(message["content"]).nil?
|
|
248
|
+
end
|
|
249
|
+
|
|
250
|
+
# The graders with a shape to get wrong: a non-integer `max_tool_calls` would
|
|
251
|
+
# compare against nil and pass; a `blocked_gates` entry without its gate would
|
|
252
|
+
# never match anything and pass. Refused at load, like `policy`.
|
|
253
|
+
def validate_graders!(expect, id:, source:)
|
|
254
|
+
max = expect["max_tool_calls"]
|
|
255
|
+
unless max.nil? || (max.is_a?(Integer) && max >= 0)
|
|
256
|
+
raise InvalidGolden, "#{source}: max_tool_calls must be a non-negative integer (case '#{id}')"
|
|
257
|
+
end
|
|
258
|
+
|
|
259
|
+
Array(expect["blocked_gates"]).each do |pair|
|
|
260
|
+
next if pair.to_s.match?(/\A[^:\s]+:[^:\s]+\z/)
|
|
261
|
+
|
|
262
|
+
raise InvalidGolden, "#{source}: blocked_gates entries are 'tool:gate' (got #{pair.inspect}, case '#{id}')"
|
|
263
|
+
end
|
|
264
|
+
end
|
|
265
|
+
|
|
177
266
|
# A typo'd policy must not silently mean "no policy" — the case would go on
|
|
178
267
|
# passing while the rule it was written for stopped being checked. The
|
|
179
268
|
# `Assertions` constant is resolved at CALL time (this file loads first, and
|
data/lib/insika/evals/runner.rb
CHANGED
|
@@ -66,6 +66,26 @@ module Insika
|
|
|
66
66
|
# consumer needing a real Chat UUID as X-Chat-Id) supplies it via conv_map; otherwise the
|
|
67
67
|
# synthetic "eval-<id>" keeps the adapter's own multi-turn continuation.
|
|
68
68
|
conv = @conv_map[golden.id] || "eval-#{golden.id}"
|
|
69
|
+
if golden.seeded?
|
|
70
|
+
# The snapshot goes in BEFORE turn 1 — on the same conversation id the turns
|
|
71
|
+
# continue. A transport that cannot seed (A2A: a remote agent has no seed
|
|
72
|
+
# route) SKIPS the case with the reason, never runs it against an empty
|
|
73
|
+
# state and passes. A deployment that refuses seeding (the setting is off)
|
|
74
|
+
# is the same outcome: skipped and reported, which is what the `requires`
|
|
75
|
+
# discipline already promises for anything the deployment lacks.
|
|
76
|
+
unless @transport.respond_to?(:seed)
|
|
77
|
+
return RunCase.new(result: Assertions.skip(golden, "transport cannot seed state"), timings: [])
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
begin
|
|
81
|
+
@transport.seed(conv, golden.state)
|
|
82
|
+
rescue SeedRefused => e
|
|
83
|
+
return RunCase.new(result: Assertions.skip(golden, "deployment refuses seeding — #{e.message}"), timings: [])
|
|
84
|
+
rescue Insika::Error => e
|
|
85
|
+
failed = TurnResult.new(output_text: "", tool_calls: [], error: "seed failed: #{e.message}")
|
|
86
|
+
return RunCase.new(result: Assertions.evaluate(golden, failed), timings: [])
|
|
87
|
+
end
|
|
88
|
+
end
|
|
69
89
|
turns = []
|
|
70
90
|
timings = []
|
|
71
91
|
spent = []
|