insika 0.0.1 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (277) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +361 -0
  3. data/LICENSE +21 -0
  4. data/README.md +136 -2
  5. data/bin/insika +366 -0
  6. data/docs/AGENTS.md +618 -0
  7. data/docs/ARCHITECTURE.md +333 -0
  8. data/docs/BENCHMARK.md +114 -0
  9. data/docs/CHANNELS.md +453 -0
  10. data/docs/CONTEXT.md +117 -0
  11. data/docs/DEPLOY.md +354 -0
  12. data/docs/EMBEDDING.md +198 -0
  13. data/docs/EVALS.md +273 -0
  14. data/docs/LOADTEST.md +232 -0
  15. data/docs/OBSERVABILITY.md +374 -0
  16. data/docs/PLUGINS.md +211 -0
  17. data/docs/REFINEMENT.md +477 -0
  18. data/docs/RELEASING.md +70 -0
  19. data/docs/RUNNING-LOCAL.md +153 -0
  20. data/docs/SANDBOX.md +114 -0
  21. data/docs/SECURITY.md +375 -0
  22. data/docs/SKILLS.md +284 -0
  23. data/docs/TOOLS.md +302 -0
  24. data/docs/WHY.md +137 -0
  25. data/docs/WORKFLOWS.md +225 -0
  26. data/docs/build.md +14 -0
  27. data/docs/index.md +68 -0
  28. data/docs/onboarding/start.md +126 -0
  29. data/docs/operate.md +12 -0
  30. data/docs/ship.md +10 -0
  31. data/docs/understand.md +10 -0
  32. data/lib/insika/agent_file_store.rb +125 -0
  33. data/lib/insika/agent_profile.rb +255 -0
  34. data/lib/insika/alert_dispatcher.rb +139 -0
  35. data/lib/insika/allowlist.rb +28 -0
  36. data/lib/insika/baseline_store.rb +74 -0
  37. data/lib/insika/budget_ledger.rb +135 -0
  38. data/lib/insika/capability/resolved_tool.rb +34 -0
  39. data/lib/insika/capability_registry.rb +112 -0
  40. data/lib/insika/channel_delivery.rb +153 -0
  41. data/lib/insika/channel_registry.rb +30 -0
  42. data/lib/insika/channels/relay.rb +178 -0
  43. data/lib/insika/channels/web/widget.js +283 -0
  44. data/lib/insika/channels/web.rb +211 -0
  45. data/lib/insika/channels/webhook.rb +58 -0
  46. data/lib/insika/chat_builder.rb +303 -0
  47. data/lib/insika/checkpoint.rb +13 -0
  48. data/lib/insika/checkpoint_store.rb +153 -0
  49. data/lib/insika/circuit_state.rb +114 -0
  50. data/lib/insika/coercion.rb +58 -0
  51. data/lib/insika/command.rb +32 -0
  52. data/lib/insika/command_bus.rb +39 -0
  53. data/lib/insika/commands/agent_payload.rb +43 -0
  54. data/lib/insika/commands/approve_action.rb +46 -0
  55. data/lib/insika/commands/cancel_task.rb +33 -0
  56. data/lib/insika/commands/create_agent.rb +54 -0
  57. data/lib/insika/commands/create_session.rb +67 -0
  58. data/lib/insika/commands/delete_agent.rb +33 -0
  59. data/lib/insika/commands/delete_agent_file.rb +50 -0
  60. data/lib/insika/commands/delete_data_tool.rb +33 -0
  61. data/lib/insika/commands/delete_llm_provider.rb +36 -0
  62. data/lib/insika/commands/delete_mcp.rb +30 -0
  63. data/lib/insika/commands/delete_skill.rb +43 -0
  64. data/lib/insika/commands/delete_system_file.rb +29 -0
  65. data/lib/insika/commands/gate_refinement.rb +245 -0
  66. data/lib/insika/commands/import_mcp_tools.rb +48 -0
  67. data/lib/insika/commands/import_tools.rb +81 -0
  68. data/lib/insika/commands/issue_tenant_token.rb +41 -0
  69. data/lib/insika/commands/memory_add_note.rb +32 -0
  70. data/lib/insika/commands/memory_forget_fact.rb +32 -0
  71. data/lib/insika/commands/memory_put_fact.rb +35 -0
  72. data/lib/insika/commands/pause_task.rb +29 -0
  73. data/lib/insika/commands/resolve_refinement.rb +126 -0
  74. data/lib/insika/commands/restore_agent_file.rb +36 -0
  75. data/lib/insika/commands/restore_data_tool.rb +34 -0
  76. data/lib/insika/commands/restore_system_file.rb +31 -0
  77. data/lib/insika/commands/resume_task.rb +85 -0
  78. data/lib/insika/commands/revoke_token.rb +39 -0
  79. data/lib/insika/commands/rotate_tenant_token.rb +43 -0
  80. data/lib/insika/commands/run_refinement.rb +133 -0
  81. data/lib/insika/commands/send_message.rb +150 -0
  82. data/lib/insika/commands/set_agent_tools.rb +39 -0
  83. data/lib/insika/commands/set_skill_agents.rb +112 -0
  84. data/lib/insika/commands/trigger_workflow.rb +80 -0
  85. data/lib/insika/commands/update_agent.rb +49 -0
  86. data/lib/insika/commands/update_settings.rb +33 -0
  87. data/lib/insika/commands/upsert_llm_provider.rb +34 -0
  88. data/lib/insika/commands/upsert_mcp.rb +32 -0
  89. data/lib/insika/commands/write_agent_file.rb +57 -0
  90. data/lib/insika/commands/write_data_tool.rb +43 -0
  91. data/lib/insika/commands/write_golden.rb +58 -0
  92. data/lib/insika/commands/write_skill.rb +60 -0
  93. data/lib/insika/commands/write_system_file.rb +31 -0
  94. data/lib/insika/config_store.rb +89 -0
  95. data/lib/insika/context/builder.rb +166 -0
  96. data/lib/insika/context/catalog_provider.rb +23 -0
  97. data/lib/insika/context/fragment.rb +43 -0
  98. data/lib/insika/context/priority.rb +30 -0
  99. data/lib/insika/context/provider.rb +19 -0
  100. data/lib/insika/context/providers/memory.rb +60 -0
  101. data/lib/insika/context/providers/prompt.rb +105 -0
  102. data/lib/insika/context/providers/request.rb +32 -0
  103. data/lib/insika/context/providers/session.rb +123 -0
  104. data/lib/insika/context/providers/skill.rb +24 -0
  105. data/lib/insika/context/providers/skill_trigger.rb +128 -0
  106. data/lib/insika/context/providers/tool_search.rb +20 -0
  107. data/lib/insika/context_trace_store.rb +92 -0
  108. data/lib/insika/delegation_store.rb +153 -0
  109. data/lib/insika/doctor.rb +539 -0
  110. data/lib/insika/dsl/definition.rb +55 -0
  111. data/lib/insika/dsl/runtime.rb +382 -0
  112. data/lib/insika/dsl/server_boot.rb +98 -0
  113. data/lib/insika/dsl/system.rb +93 -0
  114. data/lib/insika/dsl/workflow_adapter.rb +59 -0
  115. data/lib/insika/dsl.rb +364 -0
  116. data/lib/insika/edge_limiter.rb +268 -0
  117. data/lib/insika/egress_guard.rb +75 -0
  118. data/lib/insika/env_schema.rb +249 -0
  119. data/lib/insika/errors.rb +201 -0
  120. data/lib/insika/evals/assertions.rb +247 -0
  121. data/lib/insika/evals/baseline.rb +69 -0
  122. data/lib/insika/evals/golden.rb +172 -0
  123. data/lib/insika/evals/judge.rb +225 -0
  124. data/lib/insika/evals/pairwise.rb +178 -0
  125. data/lib/insika/evals/report.rb +115 -0
  126. data/lib/insika/evals/runner.rb +141 -0
  127. data/lib/insika/evals/transport.rb +178 -0
  128. data/lib/insika/event.rb +18 -0
  129. data/lib/insika/event_stream.rb +132 -0
  130. data/lib/insika/executor.rb +1995 -0
  131. data/lib/insika/frontmatter.rb +42 -0
  132. data/lib/insika/golden_store.rb +145 -0
  133. data/lib/insika/hooks.rb +48 -0
  134. data/lib/insika/http_client.rb +63 -0
  135. data/lib/insika/inbound_log.rb +84 -0
  136. data/lib/insika/llm_configurator.rb +99 -0
  137. data/lib/insika/llm_provider_store.rb +83 -0
  138. data/lib/insika/loop_detector.rb +143 -0
  139. data/lib/insika/mcp_http_client.rb +67 -0
  140. data/lib/insika/mcp_store.rb +115 -0
  141. data/lib/insika/mcp_tool_ingestor.rb +143 -0
  142. data/lib/insika/memory_store.rb +93 -0
  143. data/lib/insika/message_origin.rb +76 -0
  144. data/lib/insika/middleware.rb +36 -0
  145. data/lib/insika/model_policy.rb +52 -0
  146. data/lib/insika/model_resolver.rb +176 -0
  147. data/lib/insika/model_selection.rb +115 -0
  148. data/lib/insika/onboarding.rb +208 -0
  149. data/lib/insika/outbox_store.rb +166 -0
  150. data/lib/insika/overlay_tool_registry.rb +102 -0
  151. data/lib/insika/pack.rb +102 -0
  152. data/lib/insika/pack_importer.rb +123 -0
  153. data/lib/insika/pending_action_store.rb +120 -0
  154. data/lib/insika/plugin/loader.rb +356 -0
  155. data/lib/insika/plugin.rb +35 -0
  156. data/lib/insika/policy/engine.rb +83 -0
  157. data/lib/insika/policy/policy.rb +120 -0
  158. data/lib/insika/policy_registry.rb +23 -0
  159. data/lib/insika/profile_source.rb +143 -0
  160. data/lib/insika/prompt_catalog.rb +61 -0
  161. data/lib/insika/provider_error_classifier.rb +160 -0
  162. data/lib/insika/queue_policy.rb +167 -0
  163. data/lib/insika/recovery.rb +168 -0
  164. data/lib/insika/refinement/candidate.rb +159 -0
  165. data/lib/insika/refinement/evidence_collector.rb +371 -0
  166. data/lib/insika/refinement/gate.rb +234 -0
  167. data/lib/insika/refinement/panel.rb +222 -0
  168. data/lib/insika/refinement/proposer.rb +262 -0
  169. data/lib/insika/refinement_store.rb +295 -0
  170. data/lib/insika/registry.rb +59 -0
  171. data/lib/insika/reliability.rb +185 -0
  172. data/lib/insika/safety/config.rb +109 -0
  173. data/lib/insika/safety/detectors.rb +176 -0
  174. data/lib/insika/safety/factory.rb +102 -0
  175. data/lib/insika/safety/input_guardrail.rb +102 -0
  176. data/lib/insika/safety/moderator.rb +94 -0
  177. data/lib/insika/safety/output_filter.rb +79 -0
  178. data/lib/insika/safety/output_validator.rb +101 -0
  179. data/lib/insika/safety/safe_responses.rb +47 -0
  180. data/lib/insika/sandbox/boundary.rb +93 -0
  181. data/lib/insika/sandbox/docker.rb +74 -0
  182. data/lib/insika/sandbox/local.rb +33 -0
  183. data/lib/insika/sandbox/runner.rb +80 -0
  184. data/lib/insika/sandbox.rb +85 -0
  185. data/lib/insika/schema_guard.rb +147 -0
  186. data/lib/insika/secret_masking.rb +34 -0
  187. data/lib/insika/server/a2a/agent_card.rb +27 -0
  188. data/lib/insika/server/a2a/app.rb +112 -0
  189. data/lib/insika/server/a2a/client.rb +101 -0
  190. data/lib/insika/server/a2a/errors.rb +32 -0
  191. data/lib/insika/server/a2a/http.rb +42 -0
  192. data/lib/insika/server/a2a/message.rb +27 -0
  193. data/lib/insika/server/a2a/protocol.rb +45 -0
  194. data/lib/insika/server/a2a/remotes.rb +25 -0
  195. data/lib/insika/server/a2a/task_projection.rb +40 -0
  196. data/lib/insika/server/app.rb +1022 -0
  197. data/lib/insika/server/boot.rb +119 -0
  198. data/lib/insika/server/rack_app.rb +118 -0
  199. data/lib/insika/server/responses.rb +165 -0
  200. data/lib/insika/server/sse_body.rb +96 -0
  201. data/lib/insika/server/tenant_auth.rb +61 -0
  202. data/lib/insika/session_actor.rb +162 -0
  203. data/lib/insika/session_store.rb +143 -0
  204. data/lib/insika/settings_store.rb +154 -0
  205. data/lib/insika/shutdown.rb +125 -0
  206. data/lib/insika/skill_catalog.rb +220 -0
  207. data/lib/insika/skill_store.rb +127 -0
  208. data/lib/insika/steer_injector.rb +110 -0
  209. data/lib/insika/store.rb +52 -0
  210. data/lib/insika/stores/memory.rb +123 -0
  211. data/lib/insika/stores/sqlite.rb +183 -0
  212. data/lib/insika/studio/app.rb +1693 -0
  213. data/lib/insika/studio/assets/dist/application.css +1 -0
  214. data/lib/insika/studio/assets/dist/application.js +70 -0
  215. data/lib/insika/studio/forms.rb +335 -0
  216. data/lib/insika/studio/nav_icons.rb +31 -0
  217. data/lib/insika/studio/views/_message.erb +44 -0
  218. data/lib/insika/studio/views/agent_detail.erb +285 -0
  219. data/lib/insika/studio/views/agents.erb +63 -0
  220. data/lib/insika/studio/views/approvals.erb +41 -0
  221. data/lib/insika/studio/views/chats.erb +34 -0
  222. data/lib/insika/studio/views/evals.erb +83 -0
  223. data/lib/insika/studio/views/home.erb +72 -0
  224. data/lib/insika/studio/views/layout.erb +94 -0
  225. data/lib/insika/studio/views/login.erb +17 -0
  226. data/lib/insika/studio/views/mcp.erb +91 -0
  227. data/lib/insika/studio/views/not_found.erb +5 -0
  228. data/lib/insika/studio/views/playground.erb +47 -0
  229. data/lib/insika/studio/views/refinement.erb +234 -0
  230. data/lib/insika/studio/views/session.erb +137 -0
  231. data/lib/insika/studio/views/settings.erb +168 -0
  232. data/lib/insika/studio/views/skills.erb +141 -0
  233. data/lib/insika/studio/views/system_files.erb +65 -0
  234. data/lib/insika/studio/views/task.erb +105 -0
  235. data/lib/insika/studio/views/tasks.erb +33 -0
  236. data/lib/insika/studio/views/tool_edit.erb +107 -0
  237. data/lib/insika/studio/views/tools.erb +89 -0
  238. data/lib/insika/subagent_graph.rb +96 -0
  239. data/lib/insika/system_file_store.rb +96 -0
  240. data/lib/insika/task_actor.rb +128 -0
  241. data/lib/insika/task_store.rb +250 -0
  242. data/lib/insika/telemetry/pricing.rb +104 -0
  243. data/lib/insika/telemetry/recorder.rb +228 -0
  244. data/lib/insika/telemetry.rb +127 -0
  245. data/lib/insika/testing/store_contract.rb +270 -0
  246. data/lib/insika/tick.rb +122 -0
  247. data/lib/insika/token_estimator.rb +16 -0
  248. data/lib/insika/token_store.rb +168 -0
  249. data/lib/insika/tool_assembly.rb +140 -0
  250. data/lib/insika/tool_catalog.rb +89 -0
  251. data/lib/insika/tool_definition.rb +518 -0
  252. data/lib/insika/tool_envelope.rb +140 -0
  253. data/lib/insika/tool_manifest.rb +218 -0
  254. data/lib/insika/tool_output_compressor.rb +100 -0
  255. data/lib/insika/tool_registry.rb +21 -0
  256. data/lib/insika/tool_store.rb +135 -0
  257. data/lib/insika/tool_trace_store.rb +92 -0
  258. data/lib/insika/tools/a2a_remote.rb +48 -0
  259. data/lib/insika/tools/agent_enum.rb +68 -0
  260. data/lib/insika/tools/concurrency.rb +54 -0
  261. data/lib/insika/tools/data_defined_tool.rb +219 -0
  262. data/lib/insika/tools/load_skill.rb +99 -0
  263. data/lib/insika/tools/remember.rb +53 -0
  264. data/lib/insika/tools/stuck_signal.rb +44 -0
  265. data/lib/insika/tools/subagent.rb +75 -0
  266. data/lib/insika/tools/subagents.rb +77 -0
  267. data/lib/insika/tools/tool_search.rb +94 -0
  268. data/lib/insika/turn_output.rb +139 -0
  269. data/lib/insika/turn_state.rb +162 -0
  270. data/lib/insika/turn_timing.rb +56 -0
  271. data/lib/insika/usage_ledger.rb +47 -0
  272. data/lib/insika/version.rb +3 -1
  273. data/lib/insika/wiring/graph.rb +249 -0
  274. data/lib/insika/workflow.rb +185 -0
  275. data/lib/insika/workflow_registry.rb +33 -0
  276. data/lib/insika.rb +220 -4
  277. metadata +412 -8
@@ -0,0 +1,225 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "json"
4
+
5
+ module Insika
6
+ module Evals
7
+ # The LLM-judge. Scores a golden's `rubric` against the
8
+ # actual assistant reply — the subjective layer on top of the deterministic
9
+ # asserts. Pure over an injected `ask` callable (prompt -> raw model text), so it's
10
+ # unit-testable without an LLM; the real ask (RubyLLM on the utility_model, temp 0)
11
+ # is built by the CLI.
12
+ #
13
+ # Conservative by construction: an unparseable judge reply scores 0 (fails) rather
14
+ # than silently passing.
15
+ #
16
+ # A PANEL, not a single voice. `quorum: N` samples ONE model N
17
+ # times, which measures that model's variance and little else — at temperature 0 it
18
+ # mostly returns the same answer, including the same blind spot. Two DIFFERENT
19
+ # models disagreeing about a rubric is the signal worth having, so `asks:` takes one
20
+ # callable per model: each judge is scored independently (its own samples, its own
21
+ # median, its own pass/fail against the case's `min_score`), then
22
+ # • `aggregate` combines the scores into the one the report and the baseline
23
+ # read (:median | :mean | :min — :min is the strict panel),
24
+ # • `min_agreement` decides the verdict: the FRACTION of judges that must pass on
25
+ # their own. 0.5 = a majority; 1.0 = unanimous.
26
+ # `quorum` still applies, per judge, so a panel can also be sampled.
27
+ class Judge
28
+ # `judges` carries the per-model scores, so a split panel is visible in the report
29
+ # instead of hiding inside an average.
30
+ Verdict = Struct.new(:score, :pass, :reason, :judges, keyword_init: true)
31
+
32
+ DEFAULT_MIN_SCORE = 0.7
33
+ AGGREGATES = %i[median mean min].freeze
34
+
35
+ # The store's operating policy, stated to the judge in the same words the
36
+ # deterministic layer checks. Empty when the store has no opinion — then the
37
+ # rubric alone decides, and inventing a default here would be inventing an
38
+ # opinion for someone else's store.
39
+ POLICY_INSTRUCTIONS = {
40
+ "ask_once" => "This store allows AT MOST ONE question per reply. Two questions in " \
41
+ "one message is a failure even if the content is otherwise good.",
42
+ "investigate_first" => "This store wants the objective established BEFORE acting: on a " \
43
+ "vague request the assistant should ask (once or twice, not a " \
44
+ "form), not search immediately.",
45
+ "act_fast" => "This store wants the assistant to ACT on the first plausible reading and " \
46
+ "refine after — asking something it could have answered by searching is a " \
47
+ "failure."
48
+ }.freeze
49
+
50
+ # ask: ->(prompt) { "<raw model text>" } — one judge (kept: the common case).
51
+ # asks: [callable, …] — a panel, one entry per model.
52
+ def initialize(ask: nil, asks: nil, quorum: 1, aggregate: :median, min_agreement: 0.5)
53
+ @asks = Array(asks || ask).compact
54
+ raise ArgumentError, "a judge needs at least one `ask`" if @asks.empty?
55
+
56
+ @quorum = [quorum.to_i, 1].max
57
+ @aggregate = aggregate.to_s.to_sym
58
+ raise ArgumentError, "unknown aggregate: #{aggregate}" unless AGGREGATES.include?(@aggregate)
59
+
60
+ @min_agreement = min_agreement.to_f.clamp(0.0, 1.0)
61
+ end
62
+
63
+ # Golden + the LAST TurnResult -> Verdict, or nil when there's nothing to judge
64
+ # (no rubric). `min_score` comes from the golden (default 0.7).
65
+ def score(golden:, result:)
66
+ rubric = golden.rubric.to_s.strip
67
+ return nil if rubric.empty?
68
+
69
+ prompt = build_prompt(rubric, golden.user_turns, result.output_text.to_s, golden.policy)
70
+ min = golden.min_score || DEFAULT_MIN_SCORE
71
+ panel = @asks.map { |ask| judge_once(ask, prompt, min) }
72
+
73
+ agreed = panel.count { |j| j[:pass] }
74
+ Verdict.new(score: combine(panel.map { |j| j[:score] }).round(3),
75
+ pass: (agreed.to_f / panel.length) >= @min_agreement,
76
+ reason: panel.map { |j| j[:reason] }.reject(&:empty?).first.to_s,
77
+ judges: panel.map { |j| j[:score] })
78
+ end
79
+
80
+ private
81
+
82
+ # One model's verdict: its own samples, its own median, its own pass/fail.
83
+ def judge_once(ask, prompt, min)
84
+ samples = Array.new(@quorum) { parse(ask.call(prompt).to_s) }
85
+ med = median(samples.map { |s| s[:score] })
86
+ { score: med, pass: med >= min,
87
+ reason: samples.map { |s| s[:reason] }.compact.reject(&:empty?).first.to_s }
88
+ end
89
+
90
+ def combine(scores)
91
+ case @aggregate
92
+ when :mean then scores.empty? ? 0.0 : scores.sum / scores.length.to_f
93
+ when :min then scores.min || 0.0
94
+ else median(scores)
95
+ end
96
+ end
97
+
98
+ # The `policy` is the one thing a rubric cannot carry alone: how
99
+ # much this store wants the agent to ask before acting is a per-store decision,
100
+ # and a judge that is not TOLD it will guess — half the time wrongly. The
101
+ # deterministic half is `Assertions.policy_checks`; this is the other half.
102
+ def build_prompt(rubric, user_turns, reply, policy = nil)
103
+ <<~PROMPT
104
+ You are a strict QA judge for a customer-service AI assistant. Judge the
105
+ ASSISTANT REPLY against the RUBRIC — nothing else.
106
+
107
+ RUBRIC:
108
+ #{rubric}
109
+ #{policy_clause(policy)}
110
+ CONVERSATION (user turns, in order):
111
+ #{user_turns.map { |t| "- #{t}" }.join("\n")}
112
+
113
+ ASSISTANT REPLY:
114
+ #{reply}
115
+
116
+ Score from 0.0 (fails the rubric) to 1.0 (fully meets it). Respond with ONLY a
117
+ JSON object, no prose:
118
+ {"score": <0..1>, "reason": "<one short sentence>"}
119
+ PROMPT
120
+ end
121
+
122
+ def policy_clause(policy)
123
+ instruction = POLICY_INSTRUCTIONS[policy.to_s]
124
+ return "" unless instruction
125
+
126
+ "\nSTORE POLICY (weigh this as part of the rubric):\n#{instruction}\n"
127
+ end
128
+
129
+ # Extracts the first {...} block and parses it. Any failure (no JSON, bad JSON,
130
+ # non-numeric score) -> score 0.0 with a diagnostic reason, so a broken judge
131
+ # never masquerades as a pass. Score is clamped to [0,1].
132
+ def parse(raw)
133
+ block = raw[/\{.*\}/m]
134
+ raise JSON::ParserError, "no JSON object" unless block
135
+
136
+ obj = JSON.parse(block)
137
+ score = Float(obj["score"])
138
+ { score: score.clamp(0.0, 1.0), reason: obj["reason"].to_s }
139
+ rescue JSON::ParserError, ArgumentError, TypeError
140
+ { score: 0.0, reason: "unparseable judge output: #{raw.to_s[0, 120].inspect}" }
141
+ end
142
+
143
+ def median(nums)
144
+ sorted = nums.compact.sort
145
+ return 0.0 if sorted.empty?
146
+
147
+ mid = sorted.length / 2
148
+ sorted.length.odd? ? sorted[mid] : (sorted[mid - 1] + sorted[mid]) / 2.0
149
+ end
150
+ end
151
+
152
+ # Builds the configured judge panel from `settings["evals"]`.
153
+ #
154
+ # It lives HERE and not in `evals/run.rb` because the CLI is no longer the only
155
+ # caller: the refinement gate scores a candidate with the SAME judges the operator
156
+ # configured, and is explicit that a second copy of the judge would be the
157
+ # worst possible outcome — the gate would then be grading against a rubric nobody
158
+ # tuned. One builder, two callers.
159
+ module JudgePanel
160
+ module_function
161
+
162
+ # settings: the `evals` hash (judges/quorum/aggregate/min_agreement).
163
+ # overrides: the CLI's flags, which win over the stored config.
164
+ # -> [Judge, [model names]] | nil when nobody is configured to ask. NIL AND NOT
165
+ # a no-op judge: a rubric'd case with no judge reads as `judge_pending`, which is
166
+ # visible, where a judge that always passes would be silent.
167
+ # `llm`: the graph's own RubyLLM context; nil = the global
168
+ # constant. Only the deployment root and the CLI build judges today, and both
169
+ # are one graph per process — the seam keeps the default honest for the day
170
+ # an embedded graph gates a candidate on its own credentials.
171
+ def build(settings, overrides: {}, chat_factory: nil, llm: nil)
172
+ settings = Coercion.deep_stringify(settings || {})
173
+ overrides = overrides.transform_keys(&:to_s)
174
+ models = resolve_models(settings, overrides)
175
+ return nil if models.empty?
176
+
177
+ factory = chat_factory || ->(model, provider) { ruby_llm_ask(model, provider, llm: llm) }
178
+ judge = Judge.new(
179
+ asks: models.map { |m| factory.call(m["model"], m["provider"]) },
180
+ quorum: overrides["quorum"] || settings["quorum"] || 1,
181
+ aggregate: overrides["aggregate"] || settings["aggregate"] || "median",
182
+ min_agreement: overrides["min_agreement"] || settings["min_agreement"] || 0.5
183
+ )
184
+ [judge, models.map { |m| m["model"] }]
185
+ end
186
+
187
+ # Sugar for the callers that only want the judge (the gate).
188
+ def judge(settings, **kw) = build(settings, **kw)&.first
189
+
190
+ # The SAME configured models, asked a different question. One
191
+ # builder because "who judges here" is one operator decision: a pairwise panel
192
+ # configured apart from the rubric panel would let a run be graded by judges
193
+ # nobody chose. -> [Pairwise, [model names]] | nil when nobody is configured.
194
+ def pairwise(settings, overrides: {}, chat_factory: nil, llm: nil)
195
+ settings = Coercion.deep_stringify(settings || {})
196
+ models = resolve_models(settings, overrides.transform_keys(&:to_s))
197
+ return nil if models.empty?
198
+
199
+ factory = chat_factory || ->(model, provider) { ruby_llm_ask(model, provider, llm: llm) }
200
+ [Pairwise.new(asks: models.map { |m| factory.call(m["model"], m["provider"]) }),
201
+ models.map { |m| m["model"] }]
202
+ end
203
+
204
+ def resolve_models(settings, overrides)
205
+ models = if Coercion.present?(overrides["judge_model"])
206
+ [{ "model" => overrides["judge_model"], "provider" => overrides["judge_provider"] }]
207
+ else
208
+ Array(settings["judges"])
209
+ end
210
+ models.map { |m| Coercion.deep_stringify(m) }.reject { |m| m["model"].to_s.strip.empty? }
211
+ end
212
+
213
+ # The default way to reach a model: RubyLLM, temperature 0, required lazily so
214
+ # nothing here loads a provider gem until a judge is actually configured.
215
+ def ruby_llm_ask(model, provider, llm: nil)
216
+ require "ruby_llm"
217
+ llm ||= RubyLLM
218
+ lambda do |prompt|
219
+ llm.chat(model: model, provider: provider, assume_model_exists: true)
220
+ .with_temperature(0).ask(prompt).content
221
+ end
222
+ end
223
+ end
224
+ end
225
+ end
@@ -0,0 +1,178 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "json"
4
+
5
+ module Insika
6
+ module Evals
7
+ # PAIRWISE AGAINST THE INCUMBENT — the number that answers "can we
8
+ # replace it". An absolute 0.72 says a reply cleared a bar we invented; it says
9
+ # nothing about whether the system already answering 403,231 chats would have done
10
+ # better with the same customer.
11
+ #
12
+ # Same opening, two transcripts, one question: which one served the customer
13
+ # better? Three outcomes (better / comparable / worse), plus two the panel can
14
+ # produce and must not hide — `split` when the judges disagree and `unknown` when
15
+ # none of them answered in a readable way.
16
+ #
17
+ # Three honesty rules, each of them the difference between a number people should
18
+ # trust and one they should not:
19
+ #
20
+ # 1. **Anonymous.** The judge sees "A" and "B" and is never told which one is
21
+ # Insika. Told, it would have an opinion about the new system rather than about
22
+ # the conversations.
23
+ # 2. **Both orders, every judge.** Position bias is THE known failure of pairwise
24
+ # LLM judging, and it is the same species as the kill criterion ("`better`
25
+ # tracks length, or politeness"). So each judge is asked twice with the
26
+ # transcripts swapped, and a verdict that FLIPS with presentation order is
27
+ # reported as `comparable` with `order_dependent: true` — a preference that
28
+ # depends on which one was printed first is not a preference.
29
+ # 3. **A split panel stays split.** Averaging "better" and "worse" into
30
+ # "comparable" invents agreement that nobody expressed.
31
+ #
32
+ # Cost: 2 provider calls per judge per case. This is why it never runs as
33
+ # part of the gate and is opt-in on the CLI.
34
+ class Pairwise
35
+ BETTER = "better"
36
+ COMPARABLE = "comparable"
37
+ WORSE = "worse"
38
+ SPLIT = "split"
39
+ UNKNOWN = "unknown"
40
+
41
+ # `vs` names WHO the reference half actually is: `agent` (model against model)
42
+ # or `human-assisted` (a person typed part of it,'s `origin: operator`).
43
+ # It rides on the verdict rather than beside it so a report cannot print the
44
+ # outcome without the label.
45
+ Verdict = Struct.new(:outcome, :reason, :vs, :judges, :order_dependent, keyword_init: true) do
46
+ def human_assisted? = vs == "human-assisted"
47
+ def decided? = [BETTER, COMPARABLE, WORSE].include?(outcome)
48
+ end
49
+
50
+ # asks: [->(prompt) { "<raw model text>" }] — the SAME panel the rubric judge
51
+ # uses (JudgePanel builds both), because "the judges the operator configured" is
52
+ # one decision, not two.
53
+ def initialize(asks:)
54
+ @asks = Array(asks).compact
55
+ raise ArgumentError, "a pairwise comparison needs at least one `ask`" if @asks.empty?
56
+ end
57
+
58
+ # golden + the run's [TurnResult] -> Verdict, or nil when the case carries no
59
+ # reference (most of them: a pair is curated by a human, like the case itself).
60
+ def compare(golden:, turns:)
61
+ return nil unless golden.reference?
62
+
63
+ ours = Pairwise.transcript(golden.user_turns, turns)
64
+ theirs = Pairwise.reference_transcript(golden.reference_messages)
65
+ return nil if ours.strip.empty?
66
+
67
+ panel = @asks.map { |ask| judge_once(ask, ours, theirs) }
68
+ combine(panel, vs: golden.human_assisted? ? "human-assisted" : "agent")
69
+ end
70
+
71
+ # The replayed conversation as the judge reads it: the user turns we sent,
72
+ # interleaved with the answers the deployment published. `turns` may be shorter
73
+ # than `user_turns` (an errored turn aborts the replay) — zip on what ran.
74
+ def self.transcript(user_turns, turns)
75
+ Array(turns).each_with_index.map do |t, i|
76
+ ["customer: #{user_turns[i]}", "assistant: #{t.output_text.to_s.strip}"]
77
+ end.flatten.join("\n")
78
+ end
79
+
80
+ # The incumbent's half. Human turns are NOT flagged to the judge: what it grades
81
+ # is the conversation as the customer received it, and telling it "a person wrote
82
+ # this one" is an invitation to grade the author instead. The fact is carried to
83
+ # the READER as `vs: human-assisted` instead, which is where it changes a decision.
84
+ def self.reference_transcript(messages)
85
+ Array(messages).map do |m|
86
+ speaker = m["role"].to_s == "user" ? "customer" : "assistant"
87
+ "#{speaker}: #{m['text'].to_s.strip}"
88
+ end.join("\n")
89
+ end
90
+
91
+ private
92
+
93
+ # One model, asked twice with the sides swapped. -> { outcome:, reason:,
94
+ # order_dependent: }
95
+ def judge_once(ask, ours, theirs)
96
+ first = outcome_of(parse(ask.call(prompt(ours, theirs))), ours_is: "A")
97
+ second = outcome_of(parse(ask.call(prompt(theirs, ours))), ours_is: "B")
98
+
99
+ return { outcome: UNKNOWN, reason: [first[:reason], second[:reason]].compact.first.to_s, order_dependent: false } \
100
+ if first[:outcome] == UNKNOWN && second[:outcome] == UNKNOWN
101
+
102
+ reason = [first[:reason], second[:reason]].reject { |r| r.to_s.empty? }.first.to_s
103
+ agreed = [first[:outcome], second[:outcome]].reject { |o| o == UNKNOWN }.uniq
104
+ return { outcome: agreed.first, reason: reason, order_dependent: false } if agreed.length == 1
105
+
106
+ { outcome: COMPARABLE, reason: reason, order_dependent: true }
107
+ end
108
+
109
+ # A strict majority decides; anything less is `split`. Unknown judges are left
110
+ # OUT of the tally (they expressed nothing) but stay visible in `judges`.
111
+ def combine(panel, vs:)
112
+ voted = panel.reject { |j| j[:outcome] == UNKNOWN }
113
+ winner, group = voted.group_by { |j| j[:outcome] }.max_by { |_, js| js.length }
114
+ outcome = if voted.empty? then UNKNOWN
115
+ elsif group.length * 2 > voted.length then winner
116
+ else SPLIT
117
+ end
118
+ # The reason comes from the judges that CARRIED the verdict, so a split or an
119
+ # unknown is explained by whoever produced it rather than by an outvoted one.
120
+ spoke = outcome == winner ? group : panel
121
+
122
+ Verdict.new(outcome: outcome, vs: vs,
123
+ reason: spoke.map { |j| j[:reason].to_s }.reject(&:empty?).first.to_s,
124
+ judges: panel.map { |j| j[:outcome] },
125
+ order_dependent: panel.any? { |j| j[:order_dependent] })
126
+ end
127
+
128
+ # "A won" means Insika won only when Insika WAS A. The mapping is the whole
129
+ # point of asking twice.
130
+ def outcome_of(parsed, ours_is:)
131
+ winner = parsed[:winner]
132
+ outcome = case winner
133
+ when "tie" then COMPARABLE
134
+ when nil then UNKNOWN
135
+ else winner == ours_is ? BETTER : WORSE
136
+ end
137
+ { outcome: outcome, reason: parsed[:reason] }
138
+ end
139
+
140
+ def prompt(a, b)
141
+ <<~PROMPT
142
+ You are comparing two customer-service conversations that began with the SAME
143
+ customer message. Decide which one SERVED THE CUSTOMER BETTER: did the customer
144
+ get what they came for, without detours, wrong information or invented facts?
145
+
146
+ Ignore length, tone, politeness, emoji and formatting UNLESS they changed what
147
+ the customer actually got. A short answer that solves the problem beats a long
148
+ one that does not.
149
+
150
+ CONVERSATION A:
151
+ #{a}
152
+
153
+ CONVERSATION B:
154
+ #{b}
155
+
156
+ Respond with ONLY a JSON object, no prose:
157
+ {"winner": "A" | "B" | "tie", "reason": "<one short sentence>"}
158
+ PROMPT
159
+ end
160
+
161
+ # An unreadable reply is UNKNOWN, never a preference: scoring it as a tie would
162
+ # quietly count a broken judge as evidence that the two systems are equivalent.
163
+ def parse(raw)
164
+ block = raw.to_s[/\{.*\}/m]
165
+ raise JSON::ParserError, "no JSON object" unless block
166
+
167
+ obj = JSON.parse(block)
168
+ winner = obj["winner"].to_s.strip.upcase
169
+ winner = "tie" if %w[TIE DRAW EQUAL COMPARABLE].include?(winner)
170
+ raise JSON::ParserError, "unknown winner" unless %w[A B tie].include?(winner)
171
+
172
+ { winner: winner, reason: obj["reason"].to_s }
173
+ rescue JSON::ParserError, TypeError
174
+ { winner: nil, reason: "unparseable pairwise output: #{raw.to_s[0, 120].inspect}" }
175
+ end
176
+ end
177
+ end
178
+ end
@@ -0,0 +1,115 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "json"
4
+
5
+ module Insika
6
+ module Evals
7
+ # Renders a run's [CaseResult] as a machine-readable JSON blob (for the baseline
8
+ # gating in) and a human-readable markdown summary.
9
+ # Pure over the results — takes a clock value in, never reads it (so callers stay
10
+ # deterministic/testable).
11
+ module Report
12
+ module_function
13
+
14
+ # -> Hash ready for JSON. `at` is an ISO-8601 string stamped by the caller.
15
+ def to_h(results, at:)
16
+ passed = results.count(&:pass?)
17
+ skipped = results.count(&:skipped?)
18
+ {
19
+ "at" => at,
20
+ "total" => results.size,
21
+ "passed" => passed,
22
+ # A skipped case is neither: counting it as failed is the lie this outcome
23
+ # exists to stop, and counting it as passed is worse.
24
+ "failed" => results.size - passed - skipped,
25
+ "skipped" => skipped,
26
+ "judge_pending" => results.count(&:judge_pending?),
27
+ # Absent (not an empty tally) when nothing was compared — a run with no
28
+ # pairwise is the normal one, and a zeroed block reads like every case tied.
29
+ "pairwise" => pairwise_summary(results),
30
+ "cases" => results.map do |r|
31
+ {
32
+ "id" => r.id, "agent" => r.agent, "pass" => r.pass?,
33
+ "skipped" => r.skipped,
34
+ "judge_pending" => r.judge_pending?, "error" => r.error,
35
+ "checks" => r.checks.map { |c| { "name" => c.name, "pass" => c.pass, "detail" => c.detail } },
36
+ "judge" => (r.judge && { "score" => r.judge.score, "pass" => r.judge.pass, "reason" => r.judge.reason }),
37
+ "pairwise" => (r.pairwise && { "outcome" => r.pairwise.outcome, "vs" => r.pairwise.vs,
38
+ "reason" => r.pairwise.reason, "judges" => r.pairwise.judges,
39
+ "order_dependent" => r.pairwise.order_dependent })
40
+ }
41
+ end
42
+ }.compact
43
+ end
44
+
45
+ # -> counts by outcome, or nil when no case carried a reference. `human_assisted`
46
+ # is counted separately and always printed: a "better" against a conversation a
47
+ # PERSON typed is a different claim from one against the incumbent's model, and
48
+ # the two must never be summed into one number somebody quotes.
49
+ def pairwise_summary(results)
50
+ compared = results.filter_map(&:pairwise)
51
+ return nil if compared.empty?
52
+
53
+ counts = compared.group_by(&:outcome).transform_values(&:size)
54
+ { "compared" => compared.size,
55
+ "human_assisted" => compared.count(&:human_assisted?),
56
+ "order_dependent" => compared.count(&:order_dependent),
57
+ "outcomes" => counts }
58
+ end
59
+
60
+ def to_json(results, at:)
61
+ JSON.pretty_generate(to_h(results, at: at))
62
+ end
63
+
64
+ PAIRWISE_MARK = { "better" => "🟢", "comparable" => "🟡", "worse" => "🔴",
65
+ "split" => "⚖️", "unknown" => "❔" }.freeze
66
+
67
+ # Never without `vs:` — see `pairwise_summary`.
68
+ def pairwise_line(v)
69
+ flip = v.order_dependent ? " (order-dependent)" : ""
70
+ "#{PAIRWISE_MARK.fetch(v.outcome, '·')} vs incumbent (#{v.vs}): #{v.outcome}#{flip} — #{v.reason}"
71
+ end
72
+
73
+ def pairwise_block(summary)
74
+ counts = summary["outcomes"].map { |k, n| "#{n} #{k}" }.join(" · ")
75
+ lines = ["", "**vs incumbent** (#{summary['compared']} compared): #{counts}"]
76
+ if summary["human_assisted"].positive?
77
+ lines << "- #{summary['human_assisted']} against a HUMAN-ASSISTED transcript (a person typed " \
78
+ "part of the reference — not a model-vs-model result)"
79
+ end
80
+ if summary["order_dependent"].positive?
81
+ lines << "- #{summary['order_dependent']} flipped when the transcripts were swapped, " \
82
+ "and were reported as comparable"
83
+ end
84
+ lines
85
+ end
86
+
87
+ # Human summary. One line per case; failing checks nested underneath.
88
+ def to_markdown(results, at:)
89
+ h = to_h(results, at: at)
90
+ lines = ["# Eval report — #{at}", "",
91
+ "**#{h['passed']}/#{h['total'] - h['skipped']} passed** · #{h['failed']} failed" \
92
+ "#{" · #{h['skipped']} skipped" if h['skipped'].positive?}" \
93
+ "#{" · #{h['judge_pending']} awaiting judge" if h['judge_pending'].positive?}", ""]
94
+ results.each do |r|
95
+ if r.skipped?
96
+ # WITH the reason, always: "12 skipped" alone is indistinguishable from a
97
+ # suite that quietly stopped testing anything.
98
+ lines << "- ⏭️ `#{r.id}` (#{r.agent}) — skipped: #{r.skipped}"
99
+ next
100
+ end
101
+
102
+ lines << "- #{r.pass? ? '✅' : '❌'} `#{r.id}` (#{r.agent})#{' ⏳ judge pending' if r.judge_pending?}"
103
+ r.failures.each { |c| lines << " - ❌ #{c.name}: #{c.detail}" }
104
+ if r.judge
105
+ v = r.judge
106
+ lines << " - #{v.pass ? '✅' : '❌'} judge: #{v.score} — #{v.reason}"
107
+ end
108
+ lines << " - #{pairwise_line(r.pairwise)}" if r.pairwise
109
+ end
110
+ lines.concat(pairwise_block(h["pairwise"])) if h["pairwise"]
111
+ "#{lines.join("\n")}\n"
112
+ end
113
+ end
114
+ end
115
+ end
@@ -0,0 +1,141 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative "golden"
4
+ require_relative "assertions"
5
+
6
+ module Insika
7
+ module Evals
8
+ # Replays golden cases through a Transport and evaluates each. Pure over the
9
+ # Transport (injected) — the fake makes the orchestration unit-testable offline;
10
+ # the HttpTransport makes a real run.
11
+ #
12
+ # Multi-turn: a case's turns replay IN ORDER under one conversation id
13
+ # (`eval-<id>`); earlier turns build context, the assertion runs on the LAST
14
+ # turn's result. A turn that errors aborts the rest of that conversation (there's
15
+ # nothing to continue from) and fails the case.
16
+ class Runner
17
+ # `tokens` is what the whole case cost, summed over its turns, or nil when no
18
+ # turn reported usage. `cached` is how much of that was served from the prompt
19
+ # cache, carried separately because it is the number that explains a total.
20
+ # Only the refinement gate reads them (records a run's cost); the
21
+ # report and the exit code are untouched.
22
+ RunCase = Struct.new(:result, :timings, :tokens, :cached, keyword_init: true)
23
+
24
+ # judge: an Evals::Judge (optional). When set, a case with a rubric whose turn
25
+ # ran cleanly gets a subjective verdict attached on top of the deterministic pass.
26
+ #
27
+ # capabilities: what the DEPLOYMENT has, per agent — anything answering
28
+ # `#for(agent_id)` with { "tools" =>, "capabilities" => } or nil. Used to skip a
29
+ # case the deployment cannot satisfy, BEFORE spending a turn on
30
+ # it. nil (or an unknown agent) = no resolution, and then a case with `requires`
31
+ # RUNS and says so in the report: "could not rule it out" is not a reason to
32
+ # stop testing something, and a suite that shrinks in silence is the failure
33
+ # this feature exists to avoid.
34
+ #
35
+ # pairwise: an Evals::Pairwise (optional). Only cases carrying a
36
+ # `reference:` are compared, and the verdict never touches pass/fail — it is the
37
+ # answer to "can we replace it", reported beside the suite's own verdict.
38
+ def initialize(transport:, judge: nil, conv_map: {}, capabilities: nil, pairwise: nil)
39
+ @transport = transport
40
+ @judge = judge
41
+ @conv_map = conv_map || {}
42
+ @capabilities = capabilities
43
+ @pairwise = pairwise
44
+ end
45
+
46
+ # [Golden] -> [RunCase]. Each RunCase carries the CaseResult (for the report) +
47
+ # per-turn timings (for `--mode perf`).
48
+ def run(goldens)
49
+ goldens.map { |g| run_case(g) }
50
+ end
51
+
52
+ def run_case(golden)
53
+ skip = skip_reason(golden)
54
+ return RunCase.new(result: Assertions.skip(golden, skip), timings: []) if skip
55
+
56
+ # A backend that resolves state from a pre-existing conversation (e.g. a
57
+ # consumer needing a real Chat UUID as X-Chat-Id) supplies it via conv_map; otherwise the
58
+ # synthetic "eval-<id>" keeps the adapter's own multi-turn continuation.
59
+ conv = @conv_map[golden.id] || "eval-#{golden.id}"
60
+ turns = []
61
+ timings = []
62
+ spent = []
63
+ cached = []
64
+ golden.user_turns.each do |message|
65
+ outcome = @transport.turn(agent: golden.agent, conv: conv, message: message)
66
+ timings << { ttfb: outcome.ttfb, total: outcome.total }
67
+ spent << billed_tokens(outcome.usage)
68
+ cached << cached_tokens(outcome.usage)
69
+ turns << outcome.result
70
+ break if outcome.result.error
71
+ end
72
+ last = turns.last
73
+ # Tool/content assertions read the last turn (unchanged); the policy checks
74
+ # read every turn — "one question per reply" is a rule about each of them.
75
+ result = Assertions.evaluate(golden, last, turns: turns)
76
+ # Subjective layer: only when a judge is configured, the case has a rubric, and
77
+ # the turn ran cleanly (nothing to judge on an errored turn).
78
+ result.judge = @judge.score(golden: golden, result: last) if @judge && result.rubric && result.error.nil?
79
+ # Against the incumbent. Same rule as the judge: nothing to
80
+ # compare on a turn that errored — half a conversation would lose the
81
+ # comparison for a reason that has nothing to do with the agent.
82
+ result.pairwise = @pairwise.compare(golden: golden, turns: turns) if @pairwise && result.error.nil?
83
+ RunCase.new(result: result, timings: timings, tokens: sum_tokens(spent),
84
+ cached: sum_tokens(cached))
85
+ end
86
+
87
+ private
88
+
89
+ # nil when NO turn reported usage; otherwise the sum of the ones that did. A
90
+ # partially-metered case is reported as what was actually measured, low rather
91
+ # than absent — a budget under-counting is a smaller lie than a budget that
92
+ # throws the number away because one leg was silent.
93
+ def sum_tokens(spent)
94
+ counted = spent.compact
95
+ counted.empty? ? nil : counted.sum
96
+ end
97
+
98
+ # What the turn actually SENT, cache included. The engine's `total_tokens` is
99
+ # input + output and DELIBERATELY excludes the cached prefix (`Executor#usage_of`
100
+ # reports `cached_tokens` alongside it), so reading it as the cost of a turn
101
+ # under-reads a cached identity by an order of magnitude.
102
+ #
103
+ # Measured on the pilot: one real turn reports `total_tokens: 88` with
104
+ # `cached_tokens: 26624`. A refinement budget built on the first number would
105
+ # have let a run send ~300× what its ceiling said. Cached tokens are cheaper
106
+ # than fresh ones; they are not free, and a ceiling has to see them.
107
+ def billed_tokens(usage)
108
+ return nil unless usage.is_a?(Hash)
109
+
110
+ total = read(usage, "total_tokens")
111
+ return nil if total.nil?
112
+
113
+ total + cached_tokens(usage).to_i
114
+ end
115
+
116
+ def cached_tokens(usage)
117
+ return nil unless usage.is_a?(Hash)
118
+
119
+ values = [read(usage, "cached_tokens"), read(usage, "cache_creation_tokens")].compact
120
+ values.empty? ? nil : values.sum
121
+ end
122
+
123
+ def read(usage, key)
124
+ value = usage[key] || usage[key.to_sym]
125
+ value&.to_i
126
+ end
127
+
128
+ # -> the reason to skip, or nil to run. A case with no `requires` always runs
129
+ # (and never even asks), which keeps the whole existing corpus untouched.
130
+ def skip_reason(golden)
131
+ return nil unless golden.requirements?
132
+
133
+ available = @capabilities&.for(golden.agent)
134
+ return nil if available.nil? # unresolved: run it, and the report says so
135
+
136
+ unmet = Assertions.unmet_requirements(golden, available)
137
+ unmet.empty? ? nil : unmet.join("; ")
138
+ end
139
+ end
140
+ end
141
+ end