insika 0.8.0 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +75 -0
- data/README.md +5 -3
- data/bin/insika +1 -1
- data/docs/AGENTS.md +52 -11
- data/docs/API.md +73 -0
- data/docs/ARCHITECTURE.md +45 -44
- data/docs/CHANNELS.md +19 -2
- data/docs/CONTEXT.md +33 -27
- data/docs/DEPLOY.md +13 -2
- data/docs/EVALS.md +98 -8
- data/docs/FACTS.md +4 -0
- data/docs/KNOWLEDGE.md +7 -0
- data/docs/OBSERVABILITY.md +21 -6
- data/docs/POLICY.md +4 -1
- data/docs/RELEASING.md +4 -0
- data/docs/SECURITY.md +27 -1
- data/docs/TOOLS.md +127 -33
- data/docs/prompts/ADD-TOOL.md +12 -2
- data/docs/prompts/DIAGNOSE-TURN.md +3 -0
- data/docs/prompts/GO-LIVE.md +3 -1
- data/lib/insika/agent_profile.rb +21 -9
- data/lib/insika/channels/web/widget.js +33 -0
- data/lib/insika/channels/web.rb +5 -2
- data/lib/insika/chat_builder.rb +62 -20
- data/lib/insika/commands/agent_payload.rb +1 -1
- data/lib/insika/commands/run_distillation.rb +5 -8
- data/lib/insika/commands/seed_session.rb +118 -0
- data/lib/insika/context/builder.rb +29 -9
- data/lib/insika/context/priority.rb +2 -0
- data/lib/insika/context/provider.rb +5 -0
- data/lib/insika/context/providers/briefing.rb +11 -8
- data/lib/insika/context/providers/fence_notice.rb +27 -0
- data/lib/insika/context/providers/knowledge.rb +7 -4
- data/lib/insika/context/providers/memory.rb +8 -4
- data/lib/insika/context/providers/session.rb +7 -3
- data/lib/insika/doctor.rb +109 -1
- data/lib/insika/dsl/runtime.rb +1 -0
- data/lib/insika/dsl.rb +6 -0
- data/lib/insika/edge_limiter.rb +4 -1
- data/lib/insika/errors.rb +1 -0
- data/lib/insika/evals/assertions.rb +92 -6
- data/lib/insika/evals/golden.rb +91 -2
- data/lib/insika/evals/runner.rb +20 -0
- data/lib/insika/evals/simulator.rb +11 -2
- data/lib/insika/evals/transport.rb +117 -15
- data/lib/insika/evidence.rb +79 -12
- data/lib/insika/executor.rb +30 -23
- data/lib/insika/fence.rb +96 -0
- data/lib/insika/golden_store.rb +3 -0
- data/lib/insika/mcp_store.rb +5 -2
- data/lib/insika/mcp_tool_registry.rb +8 -1
- data/lib/insika/memory_store.rb +12 -0
- data/lib/insika/overlay_tool_registry.rb +5 -0
- data/lib/insika/prefix_fingerprint.rb +32 -27
- data/lib/insika/profile_source.rb +1 -0
- data/lib/insika/server/app.rb +43 -1
- data/lib/insika/server/rack_app.rb +2 -0
- data/lib/insika/server/responses.rb +31 -4
- data/lib/insika/session_store.rb +4 -1
- data/lib/insika/settings_store.rb +10 -1
- data/lib/insika/spoken_transcript.rb +31 -0
- data/lib/insika/studio/app.rb +6 -2
- data/lib/insika/studio/forms.rb +18 -3
- data/lib/insika/studio/views/_agent_tab_config.erb +5 -1
- data/lib/insika/studio/views/session.erb +1 -1
- data/lib/insika/studio/views/tool_edit.erb +6 -2
- data/lib/insika/telemetry/recorder.rb +13 -1
- data/lib/insika/tool_assembly.rb +21 -13
- data/lib/insika/tool_definition.rb +73 -10
- data/lib/insika/tool_envelope.rb +102 -2
- data/lib/insika/tool_store.rb +9 -4
- data/lib/insika/tool_trace_store.rb +1 -1
- data/lib/insika/tool_usage_report.rb +12 -2
- data/lib/insika/tools/data_defined_tool.rb +1 -0
- data/lib/insika/tools/present.rb +122 -0
- data/lib/insika/tools/run_persona_eval.rb +6 -1
- data/lib/insika/tools/tool_search.rb +4 -2
- data/lib/insika/turn_state.rb +13 -1
- data/lib/insika/version.rb +1 -1
- data/lib/insika/wiring/graph.rb +7 -0
- data/lib/insika/wiring/graph_chat.rb +4 -0
- data/lib/insika.rb +4 -0
- metadata +6 -1
|
@@ -39,7 +39,7 @@ module Insika
|
|
|
39
39
|
source: id
|
|
40
40
|
)
|
|
41
41
|
end
|
|
42
|
-
compaction ? [compaction_fragment(compaction)] + fragments : fragments
|
|
42
|
+
compaction ? [compaction_fragment(compaction, request)] + fragments : fragments
|
|
43
43
|
end
|
|
44
44
|
|
|
45
45
|
private
|
|
@@ -127,10 +127,14 @@ module Insika
|
|
|
127
127
|
# provider-agnostic (a mid-history "system" message is not). Priority
|
|
128
128
|
# COMPACTION (59): the "oldest unit" — under budget it drops before any
|
|
129
129
|
# verbatim message. source "compaction" -> its own context-trace category.
|
|
130
|
-
|
|
130
|
+
# Fenced when the agent has `fencing` on: the summary is model-written from
|
|
131
|
+
# customer text, and the notice promises the model this block is material.
|
|
132
|
+
def compaction_fragment(state, request)
|
|
133
|
+
summary = state["summary"].to_s
|
|
134
|
+
summary = Insika::Fence.sanitize_text(summary) if Insika::Fence.enabled?(request.profile)
|
|
131
135
|
ContextFragment.build(
|
|
132
136
|
content: { role: "user",
|
|
133
|
-
content: "<conversation_summary>\n#{
|
|
137
|
+
content: "<conversation_summary>\n#{summary}\n</conversation_summary>" },
|
|
134
138
|
placement: :history,
|
|
135
139
|
priority: Context::Priority::COMPACTION,
|
|
136
140
|
source: "compaction"
|
data/lib/insika/doctor.rb
CHANGED
|
@@ -180,7 +180,8 @@ module Insika
|
|
|
180
180
|
check_soak_envelope check_turn_timing check_grounding check_cache_layers
|
|
181
181
|
check_memory_scopes check_funnel_declarations check_followup check_distill
|
|
182
182
|
check_compaction check_harvest check_schedules check_guardrail_corpora
|
|
183
|
-
check_tool_allowlist_policy
|
|
183
|
+
check_tool_allowlist_policy check_fencing check_eval_seeding
|
|
184
|
+
check_presentation_tools check_provenance]
|
|
184
185
|
|
|
185
186
|
def safe(check)
|
|
186
187
|
Array(send(check))
|
|
@@ -836,6 +837,7 @@ def wrapped_content?(content) = /\A\s*\{\s*"[^"]+"\s*=>/.match?(content.to_s)
|
|
|
836
837
|
# identity block bills a cache write every turn).
|
|
837
838
|
IDENTITY_BUILTINS = %w[
|
|
838
839
|
Insika::Context::Providers::Prompt
|
|
840
|
+
Insika::Context::Providers::FenceNotice
|
|
839
841
|
Insika::Context::Providers::Skill
|
|
840
842
|
Insika::Context::Providers::ToolSearch
|
|
841
843
|
].freeze
|
|
@@ -942,6 +944,71 @@ def wrapped_content?(content) = /\A\s*\{\s*"[^"]+"\s*=>/.match?(content.to_s)
|
|
|
942
944
|
|
|
943
945
|
def names_tool_allowlist?(raw) = Array(raw["policies"]).any? { |p| p.to_s == "tool_allowlist" }
|
|
944
946
|
|
|
947
|
+
# A stored agent that allows a PRESENTATION tool but no tool declaring
|
|
948
|
+
# `evidence`. The presentation tool shows only cards an evidence tool returned
|
|
949
|
+
# this session, so with no evidence tool in reach every call drops every id —
|
|
950
|
+
# the model keeps asking to show products and nothing ever appears. Silent
|
|
951
|
+
# unless a presentation tool exists. Data tools only: a code tool's evidence
|
|
952
|
+
# lives in registry metadata the doctor does not read.
|
|
953
|
+
def check_presentation_tools
|
|
954
|
+
return [] unless @tool_store && @profile_source
|
|
955
|
+
|
|
956
|
+
raws = @tool_store.all_raw
|
|
957
|
+
presenting = raws.select { |r| r["presentation"] }.map { |r| r["name"].to_s }
|
|
958
|
+
return [] if presenting.empty?
|
|
959
|
+
|
|
960
|
+
evidence = raws.select { |r| r["evidence"] }.map { |r| r["name"].to_s }
|
|
961
|
+
bad = @profile_source.all.filter_map do |profile|
|
|
962
|
+
allowed = allowed_data_tools(profile, raws)
|
|
963
|
+
shows = presenting & allowed
|
|
964
|
+
next if shows.empty? || !(evidence & allowed).empty?
|
|
965
|
+
|
|
966
|
+
Finding.new(check: "presentation-tools", severity: :warn, fix: nil,
|
|
967
|
+
message: "agent '#{profile.id}' allows #{shows.join(', ')} but no tool declaring " \
|
|
968
|
+
"evidence — a presentation tool can only show cards an evidence tool " \
|
|
969
|
+
"returned this session, so every call would drop every id. Allow the " \
|
|
970
|
+
"search tool that declares `evidence`, or drop the presentation tool.")
|
|
971
|
+
end
|
|
972
|
+
return bad unless bad.empty?
|
|
973
|
+
|
|
974
|
+
[ok("presentation-tools", "#{presenting.length} presentation tool(s): every agent that shows cards " \
|
|
975
|
+
"also allows an evidence tool")]
|
|
976
|
+
end
|
|
977
|
+
|
|
978
|
+
# The data tools a profile may call, by the allowlist's own rules: nil lists =
|
|
979
|
+
# all; otherwise names ∪ groups; deny wins.
|
|
980
|
+
def allowed_data_tools(profile, raws)
|
|
981
|
+
names = profile.tools_allow.nil? ? nil : Array(profile.tools_allow).map(&:to_s)
|
|
982
|
+
groups = profile.tools_allow_groups.nil? ? nil : Array(profile.tools_allow_groups).map(&:to_s)
|
|
983
|
+
allowed = if names.nil? && groups.nil?
|
|
984
|
+
raws.map { |r| r["name"].to_s }
|
|
985
|
+
else
|
|
986
|
+
Array(names) | raws.select { |r| Array(groups).include?(r["group"].to_s) }.map { |r| r["name"].to_s }
|
|
987
|
+
end
|
|
988
|
+
allowed - Array(profile.tools_deny).map(&:to_s)
|
|
989
|
+
end
|
|
990
|
+
|
|
991
|
+
def check_provenance
|
|
992
|
+
return [] unless @tool_store && @profile_source.respond_to?(:all_raw)
|
|
993
|
+
|
|
994
|
+
tools = @tool_store.all_raw
|
|
995
|
+
@profile_source.all_raw.filter_map do |raw|
|
|
996
|
+
allowed = tools.select do |tool|
|
|
997
|
+
name = tool["name"]
|
|
998
|
+
unrestricted = raw["tools_allow"].nil? && raw["tools_allow_groups"].nil?
|
|
999
|
+
(unrestricted || Array(raw["tools_allow"]).include?(name) ||
|
|
1000
|
+
Array(raw["tools_allow_groups"]).include?(tool["group"])) &&
|
|
1001
|
+
!Array(raw["tools_deny"]).include?(name)
|
|
1002
|
+
end
|
|
1003
|
+
gated = allowed.select { |tool| tool["requires_evidence"] }
|
|
1004
|
+
next if gated.empty? || allowed.any? { |tool| tool["evidence"] }
|
|
1005
|
+
|
|
1006
|
+
Finding.new(check: "provenance", severity: :warn, fix: nil,
|
|
1007
|
+
message: "agent '#{raw["id"]}' allows #{gated.map { |t| t["name"] }.join(', ')} " \
|
|
1008
|
+
"with requires_evidence but no stored evidence tool — verify that an allowed code tool supplies evidence")
|
|
1009
|
+
end
|
|
1010
|
+
end
|
|
1011
|
+
|
|
945
1012
|
def check_grounding
|
|
946
1013
|
return [] unless @profile_source
|
|
947
1014
|
|
|
@@ -1325,6 +1392,47 @@ def wrapped_content?(content) = /\A\s*\{\s*"[^"]+"\s*=>/.match?(content.to_s)
|
|
|
1325
1392
|
end
|
|
1326
1393
|
end
|
|
1327
1394
|
|
|
1395
|
+
# `evals.seeding` opens POST /v1/conversations/:id/seed: a snapshot written
|
|
1396
|
+
# into a real conversation under the tenant token, before its first turn. Right
|
|
1397
|
+
# on the machine running snapshot evals; wrong left on in production, where any
|
|
1398
|
+
# consumer holding the token could write a customer's history. A warning, not
|
|
1399
|
+
# an error — the operator may be running the evals right now.
|
|
1400
|
+
def check_eval_seeding
|
|
1401
|
+
return [] unless @settings_store
|
|
1402
|
+
|
|
1403
|
+
on = ((@settings_store.get["evals"] || {})["seeding"]) == true
|
|
1404
|
+
return [ok("eval-seeding", "evals.seeding off — conversations cannot be seeded")] unless on
|
|
1405
|
+
|
|
1406
|
+
[Finding.new(check: "eval-seeding", severity: :warn, fix: nil,
|
|
1407
|
+
message: "evals.seeding is ON — POST /v1/conversations/:id/seed accepts a fabricated " \
|
|
1408
|
+
"conversation state under the tenant token. Fine while running snapshot " \
|
|
1409
|
+
"evals; turn it off (Studio > Settings, or `update_settings`) in production.")]
|
|
1410
|
+
end
|
|
1411
|
+
|
|
1412
|
+
# A customer-facing agent — one reachable through an inbound channel: any
|
|
1413
|
+
# agent when the relay is mounted (the event names the agent), the listed
|
|
1414
|
+
# ones for the widget — with `fencing` off reads third-party bytes as-is.
|
|
1415
|
+
# A warning, never an error: the default is off this release on purpose
|
|
1416
|
+
# (the goldens were baselined unfenced), so the message says exactly what
|
|
1417
|
+
# is unsanitized today. No inbound channel = nothing customer-facing = silent.
|
|
1418
|
+
def check_fencing
|
|
1419
|
+
return [] unless @profile_source
|
|
1420
|
+
|
|
1421
|
+
relay = Insika::Coercion.present?(@env["INSIKA_RELAY_TOKEN"])
|
|
1422
|
+
widget = @env["INSIKA_WIDGET_AGENTS"].to_s.split(",").map(&:strip).reject(&:empty?)
|
|
1423
|
+
return [] unless relay || widget.any?
|
|
1424
|
+
|
|
1425
|
+
exposed = @profile_source.all.select { |p| relay || widget.include?(p.id.to_s) }
|
|
1426
|
+
unfenced = exposed.reject { |p| Insika::Fence.enabled?(p) }.map(&:id)
|
|
1427
|
+
return [ok("fencing", "fencing on for every agent behind an inbound channel")] if unfenced.empty?
|
|
1428
|
+
|
|
1429
|
+
[Finding.new(check: "fencing", severity: :warn, fix: nil,
|
|
1430
|
+
message: "agent(s) #{unfenced.join(', ')}: reachable through an inbound channel with " \
|
|
1431
|
+
"`fencing` off — tool results, <memory> facts and <knowledge> concepts reach " \
|
|
1432
|
+
"the model byte-for-byte (invisible characters, forged turn markers and " \
|
|
1433
|
+
"transcript-shaped tags included). Set `fencing true` on the agent.")]
|
|
1434
|
+
end
|
|
1435
|
+
|
|
1328
1436
|
# the harvest check — per profile WITH a harvest hash:
|
|
1329
1437
|
# declared-without-model warn (D12), no grounding matcher warn (D3),
|
|
1330
1438
|
# malformed negative list error (D4), else ok with the pending counts.
|
data/lib/insika/dsl/runtime.rb
CHANGED
|
@@ -231,6 +231,7 @@ module Insika
|
|
|
231
231
|
base: "", catalog: c[:prompt_catalog],
|
|
232
232
|
agent_files: c[:agent_file_store], system_files: c[:system_file_store]
|
|
233
233
|
),
|
|
234
|
+
Insika::Context::Providers::FenceNotice.new,
|
|
234
235
|
Insika::Context::Providers::Skill.new(catalog: c[:skill_catalog]),
|
|
235
236
|
Insika::Context::Providers::SkillTrigger.new(catalog: c[:skill_catalog]),
|
|
236
237
|
Insika::Context::Providers::ToolSearch.new(catalog: c[:tool_catalog]),
|
data/lib/insika/dsl.rb
CHANGED
|
@@ -417,6 +417,12 @@ module Insika
|
|
|
417
417
|
# SEES: repeated identical tool results collapse to a back-reference.
|
|
418
418
|
def tool_output_compression(on = true) = @config[:tool_output_compression] = on
|
|
419
419
|
|
|
420
|
+
# Third-party text is sanitized before the model reads it (tool results,
|
|
421
|
+
# memory, knowledge) and a one-sentence notice names those blocks as data.
|
|
422
|
+
# Opt-in this release — the goldens were baselined on unfenced bytes:
|
|
423
|
+
# fencing true
|
|
424
|
+
def fencing(on = true) = @config[:fencing] = on
|
|
425
|
+
|
|
420
426
|
# Content-safety guardrails — opt-in and configurable per agent.
|
|
421
427
|
# Pure config-over-code: the hash is stored on the profile and consumed by
|
|
422
428
|
# Safety::Config.from_profile. Merges, so repeated calls accumulate.
|
data/lib/insika/edge_limiter.rb
CHANGED
|
@@ -263,12 +263,15 @@ module Insika
|
|
|
263
263
|
|
|
264
264
|
# Appends the note to the assembled system prompt: the real Data package is
|
|
265
265
|
# immutable (with), the specs' minimal Struct is mutable — both duck-typed.
|
|
266
|
+
# The note is per-turn data, so it also lands in the volatile layer — below
|
|
267
|
+
# the cache breakpoint, never inside the cached identity prefix.
|
|
266
268
|
def inject_budget_note(state, note)
|
|
267
269
|
ctx = state.context
|
|
268
270
|
return if ctx.nil?
|
|
269
271
|
|
|
270
272
|
if ctx.respond_to?(:with)
|
|
271
|
-
|
|
273
|
+
volatile = [ctx.system_volatile, note].reject { |s| s.to_s.empty? }.join("\n\n")
|
|
274
|
+
state.context = ctx.with(system: "#{ctx.system}\n\n#{note}", system_volatile: volatile)
|
|
272
275
|
elsif ctx.respond_to?(:system=)
|
|
273
276
|
ctx.system = "#{ctx.system}\n\n#{note}"
|
|
274
277
|
end
|
data/lib/insika/errors.rb
CHANGED
|
@@ -8,6 +8,7 @@ module Insika
|
|
|
8
8
|
|
|
9
9
|
class ValidationError < Error; end # Malformed Command -> HTTP 422, no Task created
|
|
10
10
|
class NotFoundError < Error; end # nonexistent session/task/agent -> HTTP 404
|
|
11
|
+
class ConflictError < Error; end # the write contradicts state that already exists -> HTTP 409
|
|
11
12
|
|
|
12
13
|
# Policy Engine denied -> :policy_denied event, task :failed
|
|
13
14
|
class PolicyDenied < Error
|
|
@@ -13,14 +13,35 @@ module Insika
|
|
|
13
13
|
# offline without a server.
|
|
14
14
|
#
|
|
15
15
|
# output_text: the final assistant text (for content checks)
|
|
16
|
-
# tool_calls: [{ "name" =>, "status" => }] captured
|
|
16
|
+
# tool_calls: [{ "name" =>, "arguments" =>, "status" =>, "gate" => }] captured
|
|
17
|
+
# from the stream's tool events. `arguments` (a Hash) and
|
|
18
|
+
# `status` ("ok" | "error" | "blocked", with `gate` naming what
|
|
19
|
+
# held a blocked call) arrive from the item frames; an older
|
|
20
|
+
# deployment sends names only and they stay nil.
|
|
21
|
+
# ui: [{ "component" =>, "count" => }] — one per presentation call
|
|
22
|
+
# (the `insika.ui` frames); nil/[] when the turn showed nothing
|
|
17
23
|
# error: transport/turn error string, or nil on a clean turn
|
|
18
|
-
TurnResult = Struct.new(:output_text, :tool_calls, :error, keyword_init: true) do
|
|
24
|
+
TurnResult = Struct.new(:output_text, :tool_calls, :ui, :error, keyword_init: true) do
|
|
19
25
|
def tool_names = Array(tool_calls).map { |t| (t["name"] || t[:name]).to_s }
|
|
20
26
|
|
|
21
|
-
#
|
|
27
|
+
# Components that actually put a card in front of the customer (count > 0).
|
|
28
|
+
def shown_components
|
|
29
|
+
Array(ui).select { |u| (u["count"] || u[:count]).to_i.positive? }
|
|
30
|
+
.map { |u| (u["component"] || u[:component]).to_s }
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
# A tool call whose status is anything but a success ("ok"/2xx/"success"). A
|
|
34
|
+
# BLOCKED call is not an error: the tool never ran, a gate held it — that is
|
|
35
|
+
# the `blocked_gates` grader's business, not `must_not: tool_error`'s.
|
|
22
36
|
def errored_tools
|
|
23
|
-
Array(tool_calls).reject
|
|
37
|
+
Array(tool_calls).reject do |t|
|
|
38
|
+
status = t["status"] || t[:status]
|
|
39
|
+
Assertions.ok_status?(status) || status.to_s == "blocked"
|
|
40
|
+
end
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
def blocked_tools
|
|
44
|
+
Array(tool_calls).select { |t| (t["status"] || t[:status]).to_s == "blocked" }
|
|
24
45
|
end
|
|
25
46
|
end
|
|
26
47
|
|
|
@@ -144,8 +165,8 @@ module Insika
|
|
|
144
165
|
# NOT `Array(turns)`: TurnResult is a Struct, so Array() would explode a single
|
|
145
166
|
# one into its members and hand the policy checks three strings.
|
|
146
167
|
conversation = turns.nil? || turns.empty? ? [result] : turns
|
|
147
|
-
checks = tool_checks(golden, result) +
|
|
148
|
-
policy_checks(golden, conversation)
|
|
168
|
+
checks = tool_checks(golden, result) + call_checks(golden, result) + reply_checks(golden, result) +
|
|
169
|
+
ui_checks(golden, result) + must_not_checks(golden, result) + policy_checks(golden, conversation)
|
|
149
170
|
CaseResult.new(id: golden.id, agent: golden.agent, error: nil, checks: checks,
|
|
150
171
|
rubric: golden.rubric, judge: nil)
|
|
151
172
|
end
|
|
@@ -163,6 +184,71 @@ module Insika
|
|
|
163
184
|
end
|
|
164
185
|
end
|
|
165
186
|
|
|
187
|
+
# The graders over the turn's CALLS. Each is its own Check, so the report names
|
|
188
|
+
# the one that failed instead of "tool checks failed". All read the LAST turn,
|
|
189
|
+
# like `tools_called`.
|
|
190
|
+
def call_checks(golden, result)
|
|
191
|
+
names = result.tool_names
|
|
192
|
+
checks = golden.never_calls.map do |name|
|
|
193
|
+
hit = names.include?(name)
|
|
194
|
+
Check.new(name: "never_calls:#{name}", pass: !hit, detail: hit ? "called #{name}" : "not called")
|
|
195
|
+
end
|
|
196
|
+
|
|
197
|
+
unless golden.calls_one_of.empty?
|
|
198
|
+
hit = golden.calls_one_of & names
|
|
199
|
+
checks << Check.new(name: "calls_one_of", pass: !hit.empty?,
|
|
200
|
+
detail: hit.empty? ? "none of #{golden.calls_one_of.join(', ')} called (saw: #{names.join(', ')})"
|
|
201
|
+
: "called #{hit.join(', ')}")
|
|
202
|
+
end
|
|
203
|
+
|
|
204
|
+
if (first = golden.first_tool)
|
|
205
|
+
checks << Check.new(name: "first_tool:#{first}", pass: names.first == first,
|
|
206
|
+
detail: "first call was #{names.first || 'none'}")
|
|
207
|
+
end
|
|
208
|
+
|
|
209
|
+
if (max = golden.max_tool_calls)
|
|
210
|
+
checks << Check.new(name: "max_tool_calls:#{max}", pass: names.size <= max,
|
|
211
|
+
detail: "#{names.size} call(s): #{names.join(', ')}")
|
|
212
|
+
end
|
|
213
|
+
|
|
214
|
+
blocked = result.blocked_tools.map { |t| "#{t['name'] || t[:name]}:#{t['gate'] || t[:gate]}" }
|
|
215
|
+
checks + golden.blocked_gates.map do |pair|
|
|
216
|
+
held = blocked.include?(pair)
|
|
217
|
+
Check.new(name: "blocked_gates:#{pair}", pass: held,
|
|
218
|
+
detail: held ? "held" : "not held (blocked: #{blocked.empty? ? 'none' : blocked.join(', ')})")
|
|
219
|
+
end
|
|
220
|
+
end
|
|
221
|
+
|
|
222
|
+
# Case-insensitive substrings over the PUBLISHED answer. `reply_omits` is where
|
|
223
|
+
# an internal id, a CPF or a raw tag leaking into the customer's text is pinned —
|
|
224
|
+
# the negative half of `reply_includes`.
|
|
225
|
+
def reply_checks(golden, result)
|
|
226
|
+
text = result.output_text.to_s.downcase
|
|
227
|
+
golden.reply_includes.map do |s|
|
|
228
|
+
hit = text.include?(s.downcase)
|
|
229
|
+
Check.new(name: "reply_includes:#{s}", pass: hit, detail: hit ? "present" : "missing from the reply")
|
|
230
|
+
end + golden.reply_omits.map do |s|
|
|
231
|
+
hit = text.include?(s.downcase)
|
|
232
|
+
Check.new(name: "reply_omits:#{s}", pass: !hit, detail: hit ? "present in the reply" : "absent")
|
|
233
|
+
end
|
|
234
|
+
end
|
|
235
|
+
|
|
236
|
+
# The graders over what the turn SHOWED. `ui_components` pins that a presentation
|
|
237
|
+
# tool put at least one card of that component in front of the customer;
|
|
238
|
+
# `no_ui` is its negative — a turn that must answer in text alone.
|
|
239
|
+
def ui_checks(golden, result)
|
|
240
|
+
shown = result.shown_components
|
|
241
|
+
checks = golden.ui_components.map do |component|
|
|
242
|
+
hit = shown.include?(component)
|
|
243
|
+
Check.new(name: "ui_components:#{component}", pass: hit,
|
|
244
|
+
detail: hit ? "shown" : "not shown (ui: #{shown.empty? ? 'none' : shown.join(', ')})")
|
|
245
|
+
end
|
|
246
|
+
return checks unless golden.no_ui?
|
|
247
|
+
|
|
248
|
+
checks << Check.new(name: "no_ui", pass: shown.empty?,
|
|
249
|
+
detail: shown.empty? ? "nothing shown" : "shown: #{shown.join(', ')}")
|
|
250
|
+
end
|
|
251
|
+
|
|
166
252
|
# `must_not` detectors. "tool_error" is special (inspects statuses); the rest
|
|
167
253
|
# are content detectors over the output text.
|
|
168
254
|
def must_not_checks(golden, result)
|
data/lib/insika/evals/golden.rb
CHANGED
|
@@ -20,8 +20,16 @@ module Insika
|
|
|
20
20
|
# single-tenant default, like `save_artifact`'s own binding_tenant) unless the
|
|
21
21
|
# case declares one. `run_persona_eval` uses it to keep a QA agent from ever
|
|
22
22
|
# running (or even seeing) another tenant's persona case in the same store.
|
|
23
|
+
#
|
|
24
|
+
# `state`: the snapshot the conversation starts from — evidence ids, memory
|
|
25
|
+
# facts and notes, prior history, briefing fields — loaded into the session
|
|
26
|
+
# BEFORE turn 1. It is how a case tests "the customer already saw three products
|
|
27
|
+
# and says 'add the second one'" without replaying the search turn: no dependence
|
|
28
|
+
# on the model's first answer, one turn cheaper, and a messy state (a
|
|
29
|
+
# contradiction from six turns ago) becomes reproducible. {} = the case starts
|
|
30
|
+
# empty, as every case did before the key existed.
|
|
23
31
|
Golden = Struct.new(:id, :agent, :turns, :expect, :requires, :reference, :source, :persona, :tenant,
|
|
24
|
-
keyword_init: true) do
|
|
32
|
+
:state, keyword_init: true) do
|
|
25
33
|
# The user messages to replay, in order. Empty for a persona case: a generated
|
|
26
34
|
# conversation has no scripted turns.
|
|
27
35
|
def user_turns = turns.map { |t| t["user"] }
|
|
@@ -44,6 +52,26 @@ module Insika
|
|
|
44
52
|
# Names of the negative assertions to run (e.g. "pii_leak", "tool_error").
|
|
45
53
|
def must_not = Array(expect["must_not"]).map(&:to_s)
|
|
46
54
|
|
|
55
|
+
# The graders over the turn's CALLS and its published REPLY — each optional,
|
|
56
|
+
# each deterministic. Every positive has a negative: `never_calls` pins what a
|
|
57
|
+
# correct turn does NOT do, which `tools_called` alone can never say.
|
|
58
|
+
def never_calls = Array(expect["never_calls"]).map(&:to_s)
|
|
59
|
+
def calls_one_of = Array(expect["calls_one_of"]).map(&:to_s)
|
|
60
|
+
def first_tool = GoldenLoader.presence(expect["first_tool"])
|
|
61
|
+
def max_tool_calls = expect["max_tool_calls"]
|
|
62
|
+
def reply_includes = Array(expect["reply_includes"]).map(&:to_s)
|
|
63
|
+
def reply_omits = Array(expect["reply_omits"]).map(&:to_s)
|
|
64
|
+
# "tool:gate" pairs that must appear among the turn's BLOCKED calls.
|
|
65
|
+
def blocked_gates = Array(expect["blocked_gates"]).map(&:to_s)
|
|
66
|
+
# UI components a presentation tool must have shown (`insika.ui` frames with
|
|
67
|
+
# at least one card), and its negative: `no_ui: true` pins a turn that shows
|
|
68
|
+
# nothing.
|
|
69
|
+
def ui_components = Array(expect["ui_components"]).map(&:to_s)
|
|
70
|
+
def no_ui? = expect["no_ui"] == true
|
|
71
|
+
|
|
72
|
+
def state = self[:state] || {}
|
|
73
|
+
def seeded? = !state.empty?
|
|
74
|
+
|
|
47
75
|
# How much the agent should ask before acting. nil = the store
|
|
48
76
|
# has no opinion and only the rubric decides.
|
|
49
77
|
def policy = GoldenLoader.presence(expect["policy"])
|
|
@@ -111,6 +139,8 @@ module Insika
|
|
|
111
139
|
raise InvalidGolden, "#{source}: 'expect' must be a mapping (case '#{id}')" unless expect.is_a?(Hash)
|
|
112
140
|
|
|
113
141
|
validate_policy!(expect["policy"], id: id, source: source)
|
|
142
|
+
validate_graders!(expect, id: id, source: source)
|
|
143
|
+
state = normalize_state(raw["state"], id: id, source: source)
|
|
114
144
|
requires = raw["requires"] || {}
|
|
115
145
|
unless requires.is_a?(Hash)
|
|
116
146
|
raise InvalidGolden, "#{source}: 'requires' must be a mapping (case '#{id}')"
|
|
@@ -121,7 +151,7 @@ module Insika
|
|
|
121
151
|
|
|
122
152
|
Golden.new(id: id, agent: agent, turns: turns, expect: expect,
|
|
123
153
|
requires: requires, reference: reference, source: source, persona: persona,
|
|
124
|
-
tenant: tenant)
|
|
154
|
+
tenant: tenant, state: state)
|
|
125
155
|
end
|
|
126
156
|
|
|
127
157
|
# `persona:` is the alternative shape to `turns:`: the
|
|
@@ -174,6 +204,65 @@ module Insika
|
|
|
174
204
|
{ "role" => role, "text" => text }.merge(origin ? { "origin" => origin } : {})
|
|
175
205
|
end
|
|
176
206
|
|
|
207
|
+
STATE_KEYS = %w[evidence memory history briefing].freeze
|
|
208
|
+
|
|
209
|
+
# state: the snapshot a case starts from. Absent -> {}. Only the four known
|
|
210
|
+
# keys, each in its own shape — a typo'd key (`evidences:`) would seed nothing
|
|
211
|
+
# and the case would go on passing against the wrong precondition. Refused at
|
|
212
|
+
# LOAD, not at seed time: a case that half-seeds is a hole in the net.
|
|
213
|
+
def normalize_state(raw, id:, source:)
|
|
214
|
+
return {} if raw.nil?
|
|
215
|
+
|
|
216
|
+
where = "#{source}: state (case '#{id}')"
|
|
217
|
+
raise InvalidGolden, "#{where} must be a mapping" unless raw.is_a?(Hash)
|
|
218
|
+
|
|
219
|
+
unknown = raw.keys.map(&:to_s) - STATE_KEYS
|
|
220
|
+
unless unknown.empty?
|
|
221
|
+
raise InvalidGolden, "#{where}: unknown key(s) #{unknown.join(', ')} — known: #{STATE_KEYS.join(', ')}"
|
|
222
|
+
end
|
|
223
|
+
|
|
224
|
+
state = raw.compact
|
|
225
|
+
mapping_with!(state["evidence"], "ids", Array, "#{where}.evidence")
|
|
226
|
+
mapping_with!(state["memory"], "facts", Hash, "#{where}.memory")
|
|
227
|
+
mapping_with!(state["memory"], "notes", Array, "#{where}.memory")
|
|
228
|
+
mapping_with!(state["briefing"], "fields", Hash, "#{where}.briefing")
|
|
229
|
+
history = state["history"]
|
|
230
|
+
unless history.nil? || (history.is_a?(Array) && history.all? { |m| history_message?(m) })
|
|
231
|
+
raise InvalidGolden, "#{where}.history must be [{ role: user|assistant, content: '…' }]"
|
|
232
|
+
end
|
|
233
|
+
|
|
234
|
+
state
|
|
235
|
+
end
|
|
236
|
+
|
|
237
|
+
def mapping_with!(value, key, type, where)
|
|
238
|
+
return if value.nil?
|
|
239
|
+
raise InvalidGolden, "#{where} must be a mapping" unless value.is_a?(Hash)
|
|
240
|
+
return if value[key].nil? || value[key].is_a?(type)
|
|
241
|
+
|
|
242
|
+
raise InvalidGolden, "#{where}.#{key} must be #{type == Array ? 'a list' : 'a mapping'}"
|
|
243
|
+
end
|
|
244
|
+
|
|
245
|
+
def history_message?(message)
|
|
246
|
+
message.is_a?(Hash) && %w[user assistant].include?(message["role"].to_s) &&
|
|
247
|
+
!presence(message["content"]).nil?
|
|
248
|
+
end
|
|
249
|
+
|
|
250
|
+
# The graders with a shape to get wrong: a non-integer `max_tool_calls` would
|
|
251
|
+
# compare against nil and pass; a `blocked_gates` entry without its gate would
|
|
252
|
+
# never match anything and pass. Refused at load, like `policy`.
|
|
253
|
+
def validate_graders!(expect, id:, source:)
|
|
254
|
+
max = expect["max_tool_calls"]
|
|
255
|
+
unless max.nil? || (max.is_a?(Integer) && max >= 0)
|
|
256
|
+
raise InvalidGolden, "#{source}: max_tool_calls must be a non-negative integer (case '#{id}')"
|
|
257
|
+
end
|
|
258
|
+
|
|
259
|
+
Array(expect["blocked_gates"]).each do |pair|
|
|
260
|
+
next if pair.to_s.match?(/\A[^:\s]+:[^:\s]+\z/)
|
|
261
|
+
|
|
262
|
+
raise InvalidGolden, "#{source}: blocked_gates entries are 'tool:gate' (got #{pair.inspect}, case '#{id}')"
|
|
263
|
+
end
|
|
264
|
+
end
|
|
265
|
+
|
|
177
266
|
# A typo'd policy must not silently mean "no policy" — the case would go on
|
|
178
267
|
# passing while the rule it was written for stopped being checked. The
|
|
179
268
|
# `Assertions` constant is resolved at CALL time (this file loads first, and
|
data/lib/insika/evals/runner.rb
CHANGED
|
@@ -66,6 +66,26 @@ module Insika
|
|
|
66
66
|
# consumer needing a real Chat UUID as X-Chat-Id) supplies it via conv_map; otherwise the
|
|
67
67
|
# synthetic "eval-<id>" keeps the adapter's own multi-turn continuation.
|
|
68
68
|
conv = @conv_map[golden.id] || "eval-#{golden.id}"
|
|
69
|
+
if golden.seeded?
|
|
70
|
+
# The snapshot goes in BEFORE turn 1 — on the same conversation id the turns
|
|
71
|
+
# continue. A transport that cannot seed (A2A: a remote agent has no seed
|
|
72
|
+
# route) SKIPS the case with the reason, never runs it against an empty
|
|
73
|
+
# state and passes. A deployment that refuses seeding (the setting is off)
|
|
74
|
+
# is the same outcome: skipped and reported, which is what the `requires`
|
|
75
|
+
# discipline already promises for anything the deployment lacks.
|
|
76
|
+
unless @transport.respond_to?(:seed)
|
|
77
|
+
return RunCase.new(result: Assertions.skip(golden, "transport cannot seed state"), timings: [])
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
begin
|
|
81
|
+
@transport.seed(conv, golden.state)
|
|
82
|
+
rescue SeedRefused => e
|
|
83
|
+
return RunCase.new(result: Assertions.skip(golden, "deployment refuses seeding — #{e.message}"), timings: [])
|
|
84
|
+
rescue Insika::Error => e
|
|
85
|
+
failed = TurnResult.new(output_text: "", tool_calls: [], error: "seed failed: #{e.message}")
|
|
86
|
+
return RunCase.new(result: Assertions.evaluate(golden, failed), timings: [])
|
|
87
|
+
end
|
|
88
|
+
end
|
|
69
89
|
turns = []
|
|
70
90
|
timings = []
|
|
71
91
|
spent = []
|
|
@@ -95,11 +95,20 @@ module Insika
|
|
|
95
95
|
@safety = safety
|
|
96
96
|
end
|
|
97
97
|
|
|
98
|
-
# Runs one simulated conversation. -> SimulatedRun.
|
|
99
|
-
|
|
98
|
+
# Runs one simulated conversation. -> SimulatedRun. `state` is the case's
|
|
99
|
+
# snapshot (Golden#state), loaded into the conversation before the persona's
|
|
100
|
+
# opening line — a simulated customer can start from "already saw three
|
|
101
|
+
# products" too. Empty/nil = the conversation starts empty, as before.
|
|
102
|
+
def run(persona:, agent:, conv:, state: nil)
|
|
100
103
|
reason = @safety.refusal
|
|
101
104
|
raise UnsafeTarget, reason if reason
|
|
102
105
|
|
|
106
|
+
if state && !state.empty?
|
|
107
|
+
raise Insika::Error, "transport cannot seed state" unless @transport.respond_to?(:seed)
|
|
108
|
+
|
|
109
|
+
@transport.seed(conv, state)
|
|
110
|
+
end
|
|
111
|
+
|
|
103
112
|
transcript = []
|
|
104
113
|
message = persona.opens_with
|
|
105
114
|
stop = :max_turns
|