insika 0.0.1 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (277) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +361 -0
  3. data/LICENSE +21 -0
  4. data/README.md +136 -2
  5. data/bin/insika +366 -0
  6. data/docs/AGENTS.md +618 -0
  7. data/docs/ARCHITECTURE.md +333 -0
  8. data/docs/BENCHMARK.md +114 -0
  9. data/docs/CHANNELS.md +453 -0
  10. data/docs/CONTEXT.md +117 -0
  11. data/docs/DEPLOY.md +354 -0
  12. data/docs/EMBEDDING.md +198 -0
  13. data/docs/EVALS.md +273 -0
  14. data/docs/LOADTEST.md +232 -0
  15. data/docs/OBSERVABILITY.md +374 -0
  16. data/docs/PLUGINS.md +211 -0
  17. data/docs/REFINEMENT.md +477 -0
  18. data/docs/RELEASING.md +70 -0
  19. data/docs/RUNNING-LOCAL.md +153 -0
  20. data/docs/SANDBOX.md +114 -0
  21. data/docs/SECURITY.md +375 -0
  22. data/docs/SKILLS.md +284 -0
  23. data/docs/TOOLS.md +302 -0
  24. data/docs/WHY.md +137 -0
  25. data/docs/WORKFLOWS.md +225 -0
  26. data/docs/build.md +14 -0
  27. data/docs/index.md +68 -0
  28. data/docs/onboarding/start.md +126 -0
  29. data/docs/operate.md +12 -0
  30. data/docs/ship.md +10 -0
  31. data/docs/understand.md +10 -0
  32. data/lib/insika/agent_file_store.rb +125 -0
  33. data/lib/insika/agent_profile.rb +255 -0
  34. data/lib/insika/alert_dispatcher.rb +139 -0
  35. data/lib/insika/allowlist.rb +28 -0
  36. data/lib/insika/baseline_store.rb +74 -0
  37. data/lib/insika/budget_ledger.rb +135 -0
  38. data/lib/insika/capability/resolved_tool.rb +34 -0
  39. data/lib/insika/capability_registry.rb +112 -0
  40. data/lib/insika/channel_delivery.rb +153 -0
  41. data/lib/insika/channel_registry.rb +30 -0
  42. data/lib/insika/channels/relay.rb +178 -0
  43. data/lib/insika/channels/web/widget.js +283 -0
  44. data/lib/insika/channels/web.rb +211 -0
  45. data/lib/insika/channels/webhook.rb +58 -0
  46. data/lib/insika/chat_builder.rb +303 -0
  47. data/lib/insika/checkpoint.rb +13 -0
  48. data/lib/insika/checkpoint_store.rb +153 -0
  49. data/lib/insika/circuit_state.rb +114 -0
  50. data/lib/insika/coercion.rb +58 -0
  51. data/lib/insika/command.rb +32 -0
  52. data/lib/insika/command_bus.rb +39 -0
  53. data/lib/insika/commands/agent_payload.rb +43 -0
  54. data/lib/insika/commands/approve_action.rb +46 -0
  55. data/lib/insika/commands/cancel_task.rb +33 -0
  56. data/lib/insika/commands/create_agent.rb +54 -0
  57. data/lib/insika/commands/create_session.rb +67 -0
  58. data/lib/insika/commands/delete_agent.rb +33 -0
  59. data/lib/insika/commands/delete_agent_file.rb +50 -0
  60. data/lib/insika/commands/delete_data_tool.rb +33 -0
  61. data/lib/insika/commands/delete_llm_provider.rb +36 -0
  62. data/lib/insika/commands/delete_mcp.rb +30 -0
  63. data/lib/insika/commands/delete_skill.rb +43 -0
  64. data/lib/insika/commands/delete_system_file.rb +29 -0
  65. data/lib/insika/commands/gate_refinement.rb +245 -0
  66. data/lib/insika/commands/import_mcp_tools.rb +48 -0
  67. data/lib/insika/commands/import_tools.rb +81 -0
  68. data/lib/insika/commands/issue_tenant_token.rb +41 -0
  69. data/lib/insika/commands/memory_add_note.rb +32 -0
  70. data/lib/insika/commands/memory_forget_fact.rb +32 -0
  71. data/lib/insika/commands/memory_put_fact.rb +35 -0
  72. data/lib/insika/commands/pause_task.rb +29 -0
  73. data/lib/insika/commands/resolve_refinement.rb +126 -0
  74. data/lib/insika/commands/restore_agent_file.rb +36 -0
  75. data/lib/insika/commands/restore_data_tool.rb +34 -0
  76. data/lib/insika/commands/restore_system_file.rb +31 -0
  77. data/lib/insika/commands/resume_task.rb +85 -0
  78. data/lib/insika/commands/revoke_token.rb +39 -0
  79. data/lib/insika/commands/rotate_tenant_token.rb +43 -0
  80. data/lib/insika/commands/run_refinement.rb +133 -0
  81. data/lib/insika/commands/send_message.rb +150 -0
  82. data/lib/insika/commands/set_agent_tools.rb +39 -0
  83. data/lib/insika/commands/set_skill_agents.rb +112 -0
  84. data/lib/insika/commands/trigger_workflow.rb +80 -0
  85. data/lib/insika/commands/update_agent.rb +49 -0
  86. data/lib/insika/commands/update_settings.rb +33 -0
  87. data/lib/insika/commands/upsert_llm_provider.rb +34 -0
  88. data/lib/insika/commands/upsert_mcp.rb +32 -0
  89. data/lib/insika/commands/write_agent_file.rb +57 -0
  90. data/lib/insika/commands/write_data_tool.rb +43 -0
  91. data/lib/insika/commands/write_golden.rb +58 -0
  92. data/lib/insika/commands/write_skill.rb +60 -0
  93. data/lib/insika/commands/write_system_file.rb +31 -0
  94. data/lib/insika/config_store.rb +89 -0
  95. data/lib/insika/context/builder.rb +166 -0
  96. data/lib/insika/context/catalog_provider.rb +23 -0
  97. data/lib/insika/context/fragment.rb +43 -0
  98. data/lib/insika/context/priority.rb +30 -0
  99. data/lib/insika/context/provider.rb +19 -0
  100. data/lib/insika/context/providers/memory.rb +60 -0
  101. data/lib/insika/context/providers/prompt.rb +105 -0
  102. data/lib/insika/context/providers/request.rb +32 -0
  103. data/lib/insika/context/providers/session.rb +123 -0
  104. data/lib/insika/context/providers/skill.rb +24 -0
  105. data/lib/insika/context/providers/skill_trigger.rb +128 -0
  106. data/lib/insika/context/providers/tool_search.rb +20 -0
  107. data/lib/insika/context_trace_store.rb +92 -0
  108. data/lib/insika/delegation_store.rb +153 -0
  109. data/lib/insika/doctor.rb +539 -0
  110. data/lib/insika/dsl/definition.rb +55 -0
  111. data/lib/insika/dsl/runtime.rb +382 -0
  112. data/lib/insika/dsl/server_boot.rb +98 -0
  113. data/lib/insika/dsl/system.rb +93 -0
  114. data/lib/insika/dsl/workflow_adapter.rb +59 -0
  115. data/lib/insika/dsl.rb +364 -0
  116. data/lib/insika/edge_limiter.rb +268 -0
  117. data/lib/insika/egress_guard.rb +75 -0
  118. data/lib/insika/env_schema.rb +249 -0
  119. data/lib/insika/errors.rb +201 -0
  120. data/lib/insika/evals/assertions.rb +247 -0
  121. data/lib/insika/evals/baseline.rb +69 -0
  122. data/lib/insika/evals/golden.rb +172 -0
  123. data/lib/insika/evals/judge.rb +225 -0
  124. data/lib/insika/evals/pairwise.rb +178 -0
  125. data/lib/insika/evals/report.rb +115 -0
  126. data/lib/insika/evals/runner.rb +141 -0
  127. data/lib/insika/evals/transport.rb +178 -0
  128. data/lib/insika/event.rb +18 -0
  129. data/lib/insika/event_stream.rb +132 -0
  130. data/lib/insika/executor.rb +1995 -0
  131. data/lib/insika/frontmatter.rb +42 -0
  132. data/lib/insika/golden_store.rb +145 -0
  133. data/lib/insika/hooks.rb +48 -0
  134. data/lib/insika/http_client.rb +63 -0
  135. data/lib/insika/inbound_log.rb +84 -0
  136. data/lib/insika/llm_configurator.rb +99 -0
  137. data/lib/insika/llm_provider_store.rb +83 -0
  138. data/lib/insika/loop_detector.rb +143 -0
  139. data/lib/insika/mcp_http_client.rb +67 -0
  140. data/lib/insika/mcp_store.rb +115 -0
  141. data/lib/insika/mcp_tool_ingestor.rb +143 -0
  142. data/lib/insika/memory_store.rb +93 -0
  143. data/lib/insika/message_origin.rb +76 -0
  144. data/lib/insika/middleware.rb +36 -0
  145. data/lib/insika/model_policy.rb +52 -0
  146. data/lib/insika/model_resolver.rb +176 -0
  147. data/lib/insika/model_selection.rb +115 -0
  148. data/lib/insika/onboarding.rb +208 -0
  149. data/lib/insika/outbox_store.rb +166 -0
  150. data/lib/insika/overlay_tool_registry.rb +102 -0
  151. data/lib/insika/pack.rb +102 -0
  152. data/lib/insika/pack_importer.rb +123 -0
  153. data/lib/insika/pending_action_store.rb +120 -0
  154. data/lib/insika/plugin/loader.rb +356 -0
  155. data/lib/insika/plugin.rb +35 -0
  156. data/lib/insika/policy/engine.rb +83 -0
  157. data/lib/insika/policy/policy.rb +120 -0
  158. data/lib/insika/policy_registry.rb +23 -0
  159. data/lib/insika/profile_source.rb +143 -0
  160. data/lib/insika/prompt_catalog.rb +61 -0
  161. data/lib/insika/provider_error_classifier.rb +160 -0
  162. data/lib/insika/queue_policy.rb +167 -0
  163. data/lib/insika/recovery.rb +168 -0
  164. data/lib/insika/refinement/candidate.rb +159 -0
  165. data/lib/insika/refinement/evidence_collector.rb +371 -0
  166. data/lib/insika/refinement/gate.rb +234 -0
  167. data/lib/insika/refinement/panel.rb +222 -0
  168. data/lib/insika/refinement/proposer.rb +262 -0
  169. data/lib/insika/refinement_store.rb +295 -0
  170. data/lib/insika/registry.rb +59 -0
  171. data/lib/insika/reliability.rb +185 -0
  172. data/lib/insika/safety/config.rb +109 -0
  173. data/lib/insika/safety/detectors.rb +176 -0
  174. data/lib/insika/safety/factory.rb +102 -0
  175. data/lib/insika/safety/input_guardrail.rb +102 -0
  176. data/lib/insika/safety/moderator.rb +94 -0
  177. data/lib/insika/safety/output_filter.rb +79 -0
  178. data/lib/insika/safety/output_validator.rb +101 -0
  179. data/lib/insika/safety/safe_responses.rb +47 -0
  180. data/lib/insika/sandbox/boundary.rb +93 -0
  181. data/lib/insika/sandbox/docker.rb +74 -0
  182. data/lib/insika/sandbox/local.rb +33 -0
  183. data/lib/insika/sandbox/runner.rb +80 -0
  184. data/lib/insika/sandbox.rb +85 -0
  185. data/lib/insika/schema_guard.rb +147 -0
  186. data/lib/insika/secret_masking.rb +34 -0
  187. data/lib/insika/server/a2a/agent_card.rb +27 -0
  188. data/lib/insika/server/a2a/app.rb +112 -0
  189. data/lib/insika/server/a2a/client.rb +101 -0
  190. data/lib/insika/server/a2a/errors.rb +32 -0
  191. data/lib/insika/server/a2a/http.rb +42 -0
  192. data/lib/insika/server/a2a/message.rb +27 -0
  193. data/lib/insika/server/a2a/protocol.rb +45 -0
  194. data/lib/insika/server/a2a/remotes.rb +25 -0
  195. data/lib/insika/server/a2a/task_projection.rb +40 -0
  196. data/lib/insika/server/app.rb +1022 -0
  197. data/lib/insika/server/boot.rb +119 -0
  198. data/lib/insika/server/rack_app.rb +118 -0
  199. data/lib/insika/server/responses.rb +165 -0
  200. data/lib/insika/server/sse_body.rb +96 -0
  201. data/lib/insika/server/tenant_auth.rb +61 -0
  202. data/lib/insika/session_actor.rb +162 -0
  203. data/lib/insika/session_store.rb +143 -0
  204. data/lib/insika/settings_store.rb +154 -0
  205. data/lib/insika/shutdown.rb +125 -0
  206. data/lib/insika/skill_catalog.rb +220 -0
  207. data/lib/insika/skill_store.rb +127 -0
  208. data/lib/insika/steer_injector.rb +110 -0
  209. data/lib/insika/store.rb +52 -0
  210. data/lib/insika/stores/memory.rb +123 -0
  211. data/lib/insika/stores/sqlite.rb +183 -0
  212. data/lib/insika/studio/app.rb +1693 -0
  213. data/lib/insika/studio/assets/dist/application.css +1 -0
  214. data/lib/insika/studio/assets/dist/application.js +70 -0
  215. data/lib/insika/studio/forms.rb +335 -0
  216. data/lib/insika/studio/nav_icons.rb +31 -0
  217. data/lib/insika/studio/views/_message.erb +44 -0
  218. data/lib/insika/studio/views/agent_detail.erb +285 -0
  219. data/lib/insika/studio/views/agents.erb +63 -0
  220. data/lib/insika/studio/views/approvals.erb +41 -0
  221. data/lib/insika/studio/views/chats.erb +34 -0
  222. data/lib/insika/studio/views/evals.erb +83 -0
  223. data/lib/insika/studio/views/home.erb +72 -0
  224. data/lib/insika/studio/views/layout.erb +94 -0
  225. data/lib/insika/studio/views/login.erb +17 -0
  226. data/lib/insika/studio/views/mcp.erb +91 -0
  227. data/lib/insika/studio/views/not_found.erb +5 -0
  228. data/lib/insika/studio/views/playground.erb +47 -0
  229. data/lib/insika/studio/views/refinement.erb +234 -0
  230. data/lib/insika/studio/views/session.erb +137 -0
  231. data/lib/insika/studio/views/settings.erb +168 -0
  232. data/lib/insika/studio/views/skills.erb +141 -0
  233. data/lib/insika/studio/views/system_files.erb +65 -0
  234. data/lib/insika/studio/views/task.erb +105 -0
  235. data/lib/insika/studio/views/tasks.erb +33 -0
  236. data/lib/insika/studio/views/tool_edit.erb +107 -0
  237. data/lib/insika/studio/views/tools.erb +89 -0
  238. data/lib/insika/subagent_graph.rb +96 -0
  239. data/lib/insika/system_file_store.rb +96 -0
  240. data/lib/insika/task_actor.rb +128 -0
  241. data/lib/insika/task_store.rb +250 -0
  242. data/lib/insika/telemetry/pricing.rb +104 -0
  243. data/lib/insika/telemetry/recorder.rb +228 -0
  244. data/lib/insika/telemetry.rb +127 -0
  245. data/lib/insika/testing/store_contract.rb +270 -0
  246. data/lib/insika/tick.rb +122 -0
  247. data/lib/insika/token_estimator.rb +16 -0
  248. data/lib/insika/token_store.rb +168 -0
  249. data/lib/insika/tool_assembly.rb +140 -0
  250. data/lib/insika/tool_catalog.rb +89 -0
  251. data/lib/insika/tool_definition.rb +518 -0
  252. data/lib/insika/tool_envelope.rb +140 -0
  253. data/lib/insika/tool_manifest.rb +218 -0
  254. data/lib/insika/tool_output_compressor.rb +100 -0
  255. data/lib/insika/tool_registry.rb +21 -0
  256. data/lib/insika/tool_store.rb +135 -0
  257. data/lib/insika/tool_trace_store.rb +92 -0
  258. data/lib/insika/tools/a2a_remote.rb +48 -0
  259. data/lib/insika/tools/agent_enum.rb +68 -0
  260. data/lib/insika/tools/concurrency.rb +54 -0
  261. data/lib/insika/tools/data_defined_tool.rb +219 -0
  262. data/lib/insika/tools/load_skill.rb +99 -0
  263. data/lib/insika/tools/remember.rb +53 -0
  264. data/lib/insika/tools/stuck_signal.rb +44 -0
  265. data/lib/insika/tools/subagent.rb +75 -0
  266. data/lib/insika/tools/subagents.rb +77 -0
  267. data/lib/insika/tools/tool_search.rb +94 -0
  268. data/lib/insika/turn_output.rb +139 -0
  269. data/lib/insika/turn_state.rb +162 -0
  270. data/lib/insika/turn_timing.rb +56 -0
  271. data/lib/insika/usage_ledger.rb +47 -0
  272. data/lib/insika/version.rb +3 -1
  273. data/lib/insika/wiring/graph.rb +249 -0
  274. data/lib/insika/workflow.rb +185 -0
  275. data/lib/insika/workflow_registry.rb +33 -0
  276. data/lib/insika.rb +220 -4
  277. metadata +412 -8
@@ -0,0 +1,247 @@
1
+ # frozen_string_literal: true
2
+
3
+ # the PII/secret patterns live in the RUNTIME (single source of
4
+ # truth) — the eval consumes them rather than keeping a divergent copy. Since the
5
+ # module moved under `lib/`, `Safety::Detectors` is loaded by `insika.rb` before this
6
+ # file; the explicit climb out of `evals/` that used to be here is gone.
7
+
8
+ module Insika
9
+ module Evals
10
+ # What the runner extracts from ONE replayed conversation, entirely from the
11
+ # public SSE stream of POST /v1/responses (no store reads — the eval stays a
12
+ # client). The assertion engine is pure over this value, so it's unit-testable
13
+ # offline without a server.
14
+ #
15
+ # output_text: the final assistant text (for content checks)
16
+ # tool_calls: [{ "name" =>, "status" => }] captured from the stream's tool events
17
+ # error: transport/turn error string, or nil on a clean turn
18
+ TurnResult = Struct.new(:output_text, :tool_calls, :error, keyword_init: true) do
19
+ def tool_names = Array(tool_calls).map { |t| (t["name"] || t[:name]).to_s }
20
+
21
+ # A tool call whose status is anything but a success ("ok"/2xx/"success").
22
+ def errored_tools
23
+ Array(tool_calls).reject { |t| Assertions.ok_status?(t["status"] || t[:status]) }
24
+ end
25
+ end
26
+
27
+ # A single check within a case (e.g. "tool:shipping_quote", "must_not:pii_leak").
28
+ Check = Struct.new(:name, :pass, :detail, keyword_init: true)
29
+
30
+ # The verdict for one golden case. `judge` (a Judge::Verdict) is attached AFTER
31
+ # the deterministic pass when the case has a rubric and a judge is configured;
32
+ # until then a rubric'd case is `judge_pending?` — it reads as "not fully
33
+ # evaluated", never a silent pass. A case passes only if the deterministic checks
34
+ # pass AND (there's no judge verdict OR it passed).
35
+ #
36
+ # `skipped` (a reason, nil = it ran) is the THIRD outcome: the
37
+ # deployment lacks something the case declared it needs, so there was nothing to
38
+ # assert. It is never a pass and never a failure — a suite of 40 cases where 12
39
+ # are skipped says something true, where 40 cases with 12 failures on capability
40
+ # grounds says nothing and gets ignored.
41
+ # `pairwise` (a Pairwise::Verdict) is attached when the case
42
+ # carries a `reference:` and a panel is configured. It DELIBERATELY does not enter
43
+ # `pass?`: "worse than the incumbent" is a judgement about a replacement decision,
44
+ # not a regression in the suite, and letting it fail a case would put an opinion
45
+ # about two conversations in the pre-merge gate.
46
+ CaseResult = Struct.new(:id, :agent, :checks, :error, :rubric, :judge, :skipped, :pairwise,
47
+ keyword_init: true) do
48
+ def skipped? = !skipped.nil?
49
+ def pass? = !skipped? && error.nil? && checks.all?(&:pass) && (judge.nil? || judge.pass)
50
+ def failures = checks.reject(&:pass)
51
+ # Has a rubric to score but no verdict yet (judge disabled / not run). A skipped
52
+ # case is not pending anything — nobody is going to judge a turn that never ran.
53
+ def judge_pending? = !skipped? && !rubric.to_s.strip.empty? && judge.nil?
54
+ end
55
+
56
+ # Deterministic evaluation — cheap, zero-token, zero-flakiness. It's the
57
+ # layer that catches the gross regressions (a tool stopped being called, a secret
58
+ # leaked, the turn errored). Subjective scoring is the LLM-judge in.
59
+ module Assertions
60
+ # Named negative detectors for `must_not` now live in the runtime. Kept as
61
+ # an alias so any external reference to Evals::Assertions::PII_DETECTORS still
62
+ # resolves; the values ARE the runtime's, never a fork.
63
+ PII_DETECTORS = Insika::Safety::Detectors::PII
64
+
65
+ # HOW MUCH THE AGENT SHOULD ASK BEFORE ACTING. Declared per
66
+ # case because it is a per-STORE decision, not a universal rule: sometimes the
67
+ # agent should establish the objective before searching ("energia, treino ou
68
+ # sono?" — a good store agent does this well), and sometimes asking again is the
69
+ # failure and it should just search. A global assertion would be wrong half
70
+ # the time; the judge is TOLD the policy (Judge#build_prompt) and this layer
71
+ # checks the half that needs no reader.
72
+ #
73
+ # Each rule is stated as the CUSTOMER-VISIBLE fact it checks. phrased
74
+ # this as "questions before the first tool call", written before: text a
75
+ # model emits before calling a tool never reaches the customer now (it rides
76
+ # `:intermediate`), and the eval is a client of `/v1/responses`, so what it can
77
+ # observe per turn is the published answer plus the tools that turn called.
78
+ # That is also the honest scope — a question nobody received is not a question.
79
+ POLICIES = {
80
+ # "UMA PERGUNTA POR VEZ" — the rule Insika broke twice under a 28 KB prompt.
81
+ "ask_once" => "at most one question per reply",
82
+ # Establish the objective before acting on a vague opener.
83
+ "investigate_first" => "asks before calling a tool, on the first turn",
84
+ # Act on the first plausible reading; refine after.
85
+ "act_fast" => "calls a tool on the first turn instead of asking"
86
+ }.freeze
87
+
88
+ module_function
89
+
90
+ # WHAT THIS DEPLOYMENT LACKS for the case to be worth running.
91
+ # -> [reason]; empty = run it.
92
+ #
93
+ # `available` is the deployment's answer for this agent:
94
+ # { "tools" => [names] | nil, "capabilities" => [names] }
95
+ # `tools` nil means an OPEN allowlist — the agent may call every registered tool,
96
+ # so no tool requirement can be judged missing and the case runs. That is the
97
+ # deliberate reading: "I could not rule it out" must not become a skip, or a
98
+ # permissive agent would quietly stop being tested.
99
+ def unmet_requirements(golden, available)
100
+ tools = available["tools"]
101
+ declared = Array(available["capabilities"]).map(&:to_s)
102
+
103
+ missing_tools = tools.nil? ? [] : golden.required_tools - Array(tools).map(&:to_s)
104
+ missing_caps = golden.required_capabilities - declared
105
+
106
+ reasons = []
107
+ reasons << "tool not available: #{missing_tools.join(', ')}" unless missing_tools.empty?
108
+ reasons << "capability not declared: #{missing_caps.join(', ')}" unless missing_caps.empty?
109
+ reasons
110
+ end
111
+
112
+ # The case did not run and MUST NOT read as either a pass or a failure.
113
+ def skip(golden, reason)
114
+ CaseResult.new(id: golden.id, agent: golden.agent, checks: [], error: nil,
115
+ rubric: nil, judge: nil, skipped: reason)
116
+ end
117
+
118
+ # A tool status counts as success when it's blank/"ok"/"success" or a 2xx code.
119
+ def ok_status?(status)
120
+ return true if status.nil?
121
+
122
+ s = status.to_s.strip.downcase
123
+ return true if s.empty? || %w[ok success succeeded done].include?(s)
124
+
125
+ code = Integer(s, exception: false)
126
+ code ? code.between?(200, 299) : false
127
+ end
128
+
129
+ # Golden + TurnResult -> CaseResult. A turn that failed to run yields a single
130
+ # failing check (there's nothing to assert on a turn that never produced output).
131
+ #
132
+ # `turns` is every turn of the conversation, in order; `result` is the last one
133
+ # (what the tool/content assertions have always run on). The policy checks need
134
+ # all of them — "one question per reply" is a rule about every reply, and the
135
+ # violation that motivated this was on the FIRST turn. Defaults to the single
136
+ # result so existing callers keep working.
137
+ def evaluate(golden, result, turns: nil)
138
+ if result.error
139
+ return CaseResult.new(id: golden.id, agent: golden.agent, error: result.error, rubric: nil, judge: nil,
140
+ checks: [Check.new(name: "turn", pass: false, detail: "turn error: #{result.error}")])
141
+ end
142
+
143
+ # NOT `Array(turns)`: TurnResult is a Struct, so Array() would explode a single
144
+ # one into its members and hand the policy checks three strings.
145
+ conversation = turns.nil? || turns.empty? ? [result] : turns
146
+ checks = tool_checks(golden, result) + must_not_checks(golden, result) +
147
+ policy_checks(golden, conversation)
148
+ CaseResult.new(id: golden.id, agent: golden.agent, error: nil, checks: checks,
149
+ rubric: golden.rubric, judge: nil)
150
+ end
151
+
152
+ # Each REQUIRED expected tool must appear in the turn's tool calls. Optional
153
+ # ("name?") tools are informational — present or not, they never fail.
154
+ def tool_checks(golden, result)
155
+ names = result.tool_names
156
+ golden.tools_called.filter_map do |t|
157
+ next if t[:optional]
158
+
159
+ present = names.include?(t[:name])
160
+ Check.new(name: "tool:#{t[:name]}", pass: present,
161
+ detail: present ? "called" : "expected but not called (saw: #{names.join(', ')})")
162
+ end
163
+ end
164
+
165
+ # `must_not` detectors. "tool_error" is special (inspects statuses); the rest
166
+ # are content detectors over the output text.
167
+ def must_not_checks(golden, result)
168
+ golden.must_not.map do |name|
169
+ if name == "tool_error"
170
+ bad = result.errored_tools
171
+ Check.new(name: "must_not:tool_error", pass: bad.empty?,
172
+ detail: bad.empty? ? "no tool errors" : "errored: #{bad.map { |t| t['name'] || t[:name] }.join(', ')}")
173
+ else
174
+ hit = detect(name, result.output_text.to_s)
175
+ Check.new(name: "must_not:#{name}", pass: hit.nil?,
176
+ detail: hit ? "matched #{hit.inspect}" : "clean")
177
+ end
178
+ end
179
+ end
180
+
181
+ # The declared `policy`, checked deterministically over the conversation. No
182
+ # policy -> no check (and nothing to explain in the report).
183
+ def policy_checks(golden, turns)
184
+ name = golden.policy
185
+ return [] if name.nil?
186
+
187
+ # Exhaustive on purpose: a policy added to POLICIES without a rule here would
188
+ # otherwise fall into whichever branch was last and check the wrong thing.
189
+ pass, detail = case name
190
+ when "ask_once" then ask_once(turns)
191
+ when "investigate_first" then investigate_first(turns.first)
192
+ when "act_fast" then act_fast(turns.first)
193
+ else raise ArgumentError, "policy #{name.inspect} has no rule"
194
+ end
195
+ [Check.new(name: "policy:#{name}", pass: pass, detail: detail)]
196
+ end
197
+
198
+ # Every reply asks at most one question. Reported with the offending turn and
199
+ # the reply itself — "2 questions" alone sends the reader digging.
200
+ def ask_once(turns)
201
+ offender = turns.each_with_index.find { |t, _| count_questions(t.output_text) > 1 }
202
+ return [true, "at most one question per reply"] unless offender
203
+
204
+ turn, i = offender
205
+ [false, "turn #{i + 1} asked #{count_questions(turn.output_text)} questions: " \
206
+ "#{turn.output_text.to_s.strip[0, 160].inspect}"]
207
+ end
208
+
209
+ def investigate_first(turn)
210
+ return [false, "no turn to check"] if turn.nil?
211
+
212
+ tools = turn.tool_names
213
+ return [false, "called #{tools.join(', ')} before asking anything"] unless tools.empty?
214
+ return [false, "answered without asking: #{turn.output_text.to_s.strip[0, 160].inspect}"] if
215
+ count_questions(turn.output_text).zero?
216
+
217
+ [true, "asked before acting"]
218
+ end
219
+
220
+ def act_fast(turn)
221
+ return [false, "no turn to check"] if turn.nil?
222
+
223
+ tools = turn.tool_names
224
+ return [true, "acted: called #{tools.join(', ')}"] unless tools.empty?
225
+
226
+ [false, "asked instead of acting: #{turn.output_text.to_s.strip[0, 160].inspect}"]
227
+ end
228
+
229
+ # Questions in ONE reply. Deliberately crude and deliberately documented: a run
230
+ # of "?" counts once ("já pensou??" is one question), and URLs are dropped first
231
+ # so a tracking link's query string is not read as the agent asking something.
232
+ # It is a policy signal, not grammar — and it already caught a real violation
233
+ # ("é pra você ou tá pensando em presentear alguém? E qual seu tamanho?").
234
+ def count_questions(text)
235
+ text.to_s.gsub(%r{https?://\S+}, " ").scan(/\?+/).size
236
+ end
237
+
238
+ # Runs a named detector over the text. "pii_leak" = union of all PII detectors;
239
+ # otherwise a single named pattern. Delegates to the runtime's single source
240
+ # which itself fails loud on an unknown name (a typo'd assertion must not
241
+ # silently pass).
242
+ def detect(name, text)
243
+ Insika::Safety::Detectors.detect(name, text)
244
+ end
245
+ end
246
+ end
247
+ end
@@ -0,0 +1,69 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "json"
4
+
5
+ module Insika
6
+ module Evals
7
+ # gating. A baseline is the accepted state of the golden
8
+ # set — `{ cases: { id => { pass, score } } }`. A gated run compares against it and
9
+ # blocks only on a REGRESSION, so known-failing cases don't wedge the gate while a
10
+ # real drop (a passing case that now fails, or a judge score that fell past the
11
+ # tolerance) does. That's the pre-merge gate for prompt/tool/model changes.
12
+ module Baseline
13
+ Regression = Struct.new(:id, :kind, :detail, keyword_init: true)
14
+
15
+ module_function
16
+
17
+ # [CaseResult] -> baseline hash. `at` is stamped by the caller (kept out of here
18
+ # so the module stays deterministic/testable).
19
+ # A SKIPPED case is left out entirely: writing it as `pass:
20
+ # false` would accept "this deployment cannot run it" as the accepted state,
21
+ # and the case would never block anywhere again.
22
+ def snapshot(results, at:)
23
+ {
24
+ "at" => at,
25
+ "cases" => results.reject(&:skipped?).each_with_object({}) do |r, h|
26
+ h[r.id] = { "pass" => r.pass?, "score" => (r.judge&.score) }
27
+ end
28
+ }
29
+ end
30
+
31
+ def load(path) = JSON.parse(File.read(path))
32
+
33
+ def write(path, results, at:)
34
+ File.write(path, JSON.pretty_generate(snapshot(results, at: at)))
35
+ end
36
+
37
+ # Compares a run against a loaded baseline. -> [Regression]. Only cases present
38
+ # in BOTH are compared: a new case (no baseline entry) never blocks the gate (it
39
+ # shows in the report as ❌ but is not a "regression"); document this in README.
40
+ # • pass→fail : baseline pass, now failing (hard regression).
41
+ # • pass→skipped: baseline pass, now unrunnable HERE. says the
42
+ # gate never blocks ON a skip, and it does not: a case that was
43
+ # already skipped or unknown stays silent. But a case that used
44
+ # to run on this deployment and no longer can means the agent
45
+ # lost a tool or a declaration — a suite shrinking in silence is
46
+ # exactly what this outcome was added to prevent. Re-baseline if
47
+ # the shrink is intended.
48
+ # • score-drop : baseline score - current judge score > tolerance (quality
49
+ # drift, even if the case still technically passes).
50
+ def compare(results, baseline, tolerance:)
51
+ base = baseline["cases"] || {}
52
+ results.filter_map do |r|
53
+ b = base[r.id]
54
+ next unless b
55
+
56
+ if b["pass"] && r.skipped?
57
+ Regression.new(id: r.id, kind: "pass→skipped",
58
+ detail: "was passing, now unrunnable here: #{r.skipped}")
59
+ elsif b["pass"] && !r.pass?
60
+ Regression.new(id: r.id, kind: "pass→fail", detail: "was passing, now failing")
61
+ elsif b["score"] && r.judge && (b["score"] - r.judge.score) > tolerance
62
+ Regression.new(id: r.id, kind: "score-drop",
63
+ detail: "judge #{b['score']} -> #{r.judge.score} (> #{tolerance})")
64
+ end
65
+ end
66
+ end
67
+ end
68
+ end
69
+ end
@@ -0,0 +1,172 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "yaml"
4
+
5
+ # Evals — the quality harness. It lives in `lib/` so the engine itself can
6
+ # call it (the refinement gate of needs to score a candidate agent, and a
7
+ # second copy of the judge would be the worst possible outcome), but it stays a
8
+ # CLIENT: it reaches a running deployment over HTTP through `HttpTransport` and never
9
+ # reads a store directly. `evals/run.rb` is a thin CLI over this module.
10
+ module Insika
11
+ module Evals
12
+ # A curated behavior case, loaded from a data file (evals/golden/<agent>/*.yml).
13
+ # Data, not code — same spirit as tools-as-data. See evals/README.md for the format.
14
+ Golden = Struct.new(:id, :agent, :turns, :expect, :requires, :reference, :source, keyword_init: true) do
15
+ # The user messages to replay, in order.
16
+ def user_turns = turns.map { |t| t["user"] }
17
+
18
+ # Tool refs the case expects; a trailing "?" marks OPTIONAL (never fails).
19
+ # -> [{ name:, optional: }]
20
+ def tools_called
21
+ Array(expect["tools_called"]).map do |ref|
22
+ s = ref.to_s
23
+ optional = s.end_with?("?")
24
+ { name: optional ? s[0..-2] : s, optional: optional }
25
+ end
26
+ end
27
+
28
+ # Names of the negative assertions to run (e.g. "pii_leak", "tool_error").
29
+ def must_not = Array(expect["must_not"]).map(&:to_s)
30
+
31
+ # How much the agent should ask before acting. nil = the store
32
+ # has no opinion and only the rubric decides.
33
+ def policy = GoldenLoader.presence(expect["policy"])
34
+
35
+ # What the DEPLOYMENT must have for this case to mean anything.
36
+ # Empty = runs everywhere.
37
+ def required_tools = Array(requires["tools"]).map(&:to_s)
38
+ def required_capabilities = Array(requires["capabilities"]).map(&:to_s)
39
+ def requirements? = !(required_tools + required_capabilities).empty?
40
+
41
+ # THE INCUMBENT'S CONVERSATION for the same opening — the other
42
+ # half of a pairwise comparison. Data in the case, not a store read: the eval is
43
+ # a client, and a pair that lives in one reviewable file cannot go stale against
44
+ # a database nobody looked at.
45
+ def reference_messages = Array(reference["messages"])
46
+ def reference_source = GoldenLoader.presence(reference["source"])
47
+ def reference? = !reference_messages.empty?
48
+
49
+ # Did a PERSON type part of the reference half? After a handoff the operator's
50
+ # words are stored as `role: assistant`, and comparing a model to a human
51
+ # and calling it a win is a lie in both directions — so the pair is LABELLED and
52
+ # the report never prints the outcome without it.
53
+ def human_assisted?
54
+ reference_messages.any? { |m| MessageOrigin.origin_of(m) == MessageOrigin::OPERATOR }
55
+ end
56
+
57
+ # LLM-judge rubric + threshold (consumed in — deferred here).
58
+ def rubric = expect["rubric"]
59
+ def min_score = expect["min_score"]
60
+ end
61
+
62
+ # Loads + validates golden files. Fails LOUD on a malformed case — a silently
63
+ # dropped golden is a hole in the safety net.
64
+ module GoldenLoader
65
+ class InvalidGolden < StandardError; end
66
+
67
+ module_function
68
+
69
+ # Loads every *.yml/*.yaml under `dir` (recursive), sorted by path for a stable
70
+ # run order. -> [Golden].
71
+ def load_dir(dir)
72
+ Dir.glob(File.join(dir, "**", "*.{yml,yaml}")).sort.map { |f| load_file(f) }
73
+ end
74
+
75
+ def load_file(path)
76
+ raw = YAML.safe_load(File.read(path), permitted_classes: [], aliases: false) || {}
77
+ build(raw, source: path)
78
+ rescue Psych::SyntaxError => e
79
+ raise InvalidGolden, "#{path}: invalid YAML — #{e.message}"
80
+ end
81
+
82
+ # hash (string keys) -> validated Golden. `source` is only for error messages.
83
+ def build(raw, source: "(inline)")
84
+ raise InvalidGolden, "#{source}: golden must be a mapping" unless raw.is_a?(Hash)
85
+
86
+ id = presence(raw["id"]) || (raise InvalidGolden, "#{source}: 'id' is required")
87
+ agent = presence(raw["agent"]) || (raise InvalidGolden, "#{source}: 'agent' is required (case '#{id}')")
88
+ turns = normalize_turns(raw["turns"], id: id, source: source)
89
+ expect = raw["expect"] || {}
90
+ raise InvalidGolden, "#{source}: 'expect' must be a mapping (case '#{id}')" unless expect.is_a?(Hash)
91
+
92
+ validate_policy!(expect["policy"], id: id, source: source)
93
+ requires = raw["requires"] || {}
94
+ unless requires.is_a?(Hash)
95
+ raise InvalidGolden, "#{source}: 'requires' must be a mapping (case '#{id}')"
96
+ end
97
+
98
+ reference = normalize_reference(raw["reference"], id: id, source: source)
99
+
100
+ Golden.new(id: id, agent: agent, turns: turns, expect: expect,
101
+ requires: requires, reference: reference, source: source)
102
+ end
103
+
104
+ # reference: { "source" => String?, "messages" => [{ "role" =>, "text" =>,
105
+ # "origin" => }] }. Absent -> {}, and the case simply has nothing to compare
106
+ # against. Malformed is REFUSED: a reference that half-loads would produce a
107
+ # pairwise verdict about a transcript nobody wrote.
108
+ def normalize_reference(raw, id:, source:)
109
+ return {} if raw.nil?
110
+ raise InvalidGolden, "#{source}: 'reference' must be a mapping (case '#{id}')" unless raw.is_a?(Hash)
111
+
112
+ messages = raw["messages"]
113
+ unless messages.is_a?(Array) && !messages.empty?
114
+ raise InvalidGolden, "#{source}: reference needs a non-empty 'messages' array (case '#{id}')"
115
+ end
116
+
117
+ { "source" => presence(raw["source"]),
118
+ "messages" => messages.each_with_index.map { |m, i| reference_message(m, i, id: id, source: source) } }.compact
119
+ end
120
+
121
+ def reference_message(raw, index, id:, source:)
122
+ where = "#{source}: reference.messages[#{index}] (case '#{id}')"
123
+ raise InvalidGolden, "#{where} must be a mapping" unless raw.is_a?(Hash)
124
+
125
+ role = presence(raw["role"])
126
+ raise InvalidGolden, "#{where} needs a 'role' of user or assistant" unless %w[user assistant].include?(role)
127
+
128
+ text = presence(raw["text"]) || (raise InvalidGolden, "#{where} needs a non-empty 'text'")
129
+ # The SAME closed vocabulary the engine stamps. A typo'd marker would
130
+ # read as "absent" downstream, which is how a human turn gets scored as the
131
+ # incumbent's model.
132
+ origin = begin
133
+ MessageOrigin.parse!(raw["origin"])
134
+ rescue Insika::ValidationError => e
135
+ raise InvalidGolden, "#{where}: #{e.message}"
136
+ end
137
+ { "role" => role, "text" => text }.merge(origin ? { "origin" => origin } : {})
138
+ end
139
+
140
+ # A typo'd policy must not silently mean "no policy" — the case would go on
141
+ # passing while the rule it was written for stopped being checked. The
142
+ # `Assertions` constant is resolved at CALL time (this file loads first, and
143
+ # assertions.rb touches `Safety::Detectors` at load time).
144
+ def validate_policy!(value, id:, source:)
145
+ name = presence(value)
146
+ return if name.nil? || Assertions::POLICIES.key?(name)
147
+
148
+ raise InvalidGolden, "#{source}: unknown policy #{name.inspect} (case '#{id}') — " \
149
+ "known: #{Assertions::POLICIES.keys.join(', ')}"
150
+ end
151
+
152
+ # turns: a non-empty array of { "user" => String }. Rejects anything else so a
153
+ # typo (e.g. `users:`) surfaces at load time, not as an empty replay.
154
+ def normalize_turns(turns, id:, source:)
155
+ unless turns.is_a?(Array) && !turns.empty?
156
+ raise InvalidGolden, "#{source}: 'turns' must be a non-empty array (case '#{id}')"
157
+ end
158
+
159
+ turns.each_with_index.map do |t, i|
160
+ user = t.is_a?(Hash) ? presence(t["user"]) : nil
161
+ user || (raise InvalidGolden, "#{source}: turns[#{i}] needs a non-empty 'user' (case '#{id}')")
162
+ { "user" => user }
163
+ end
164
+ end
165
+
166
+ def presence(v)
167
+ s = v.to_s.strip
168
+ s.empty? ? nil : s
169
+ end
170
+ end
171
+ end
172
+ end