insika 0.0.1 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +361 -0
- data/LICENSE +21 -0
- data/README.md +136 -2
- data/bin/insika +366 -0
- data/docs/AGENTS.md +618 -0
- data/docs/ARCHITECTURE.md +333 -0
- data/docs/BENCHMARK.md +114 -0
- data/docs/CHANNELS.md +453 -0
- data/docs/CONTEXT.md +117 -0
- data/docs/DEPLOY.md +354 -0
- data/docs/EMBEDDING.md +198 -0
- data/docs/EVALS.md +273 -0
- data/docs/LOADTEST.md +232 -0
- data/docs/OBSERVABILITY.md +374 -0
- data/docs/PLUGINS.md +211 -0
- data/docs/REFINEMENT.md +477 -0
- data/docs/RELEASING.md +70 -0
- data/docs/RUNNING-LOCAL.md +153 -0
- data/docs/SANDBOX.md +114 -0
- data/docs/SECURITY.md +375 -0
- data/docs/SKILLS.md +284 -0
- data/docs/TOOLS.md +302 -0
- data/docs/WHY.md +137 -0
- data/docs/WORKFLOWS.md +225 -0
- data/docs/build.md +14 -0
- data/docs/index.md +68 -0
- data/docs/onboarding/start.md +126 -0
- data/docs/operate.md +12 -0
- data/docs/ship.md +10 -0
- data/docs/understand.md +10 -0
- data/lib/insika/agent_file_store.rb +125 -0
- data/lib/insika/agent_profile.rb +255 -0
- data/lib/insika/alert_dispatcher.rb +139 -0
- data/lib/insika/allowlist.rb +28 -0
- data/lib/insika/baseline_store.rb +74 -0
- data/lib/insika/budget_ledger.rb +135 -0
- data/lib/insika/capability/resolved_tool.rb +34 -0
- data/lib/insika/capability_registry.rb +112 -0
- data/lib/insika/channel_delivery.rb +153 -0
- data/lib/insika/channel_registry.rb +30 -0
- data/lib/insika/channels/relay.rb +178 -0
- data/lib/insika/channels/web/widget.js +283 -0
- data/lib/insika/channels/web.rb +211 -0
- data/lib/insika/channels/webhook.rb +58 -0
- data/lib/insika/chat_builder.rb +303 -0
- data/lib/insika/checkpoint.rb +13 -0
- data/lib/insika/checkpoint_store.rb +153 -0
- data/lib/insika/circuit_state.rb +114 -0
- data/lib/insika/coercion.rb +58 -0
- data/lib/insika/command.rb +32 -0
- data/lib/insika/command_bus.rb +39 -0
- data/lib/insika/commands/agent_payload.rb +43 -0
- data/lib/insika/commands/approve_action.rb +46 -0
- data/lib/insika/commands/cancel_task.rb +33 -0
- data/lib/insika/commands/create_agent.rb +54 -0
- data/lib/insika/commands/create_session.rb +67 -0
- data/lib/insika/commands/delete_agent.rb +33 -0
- data/lib/insika/commands/delete_agent_file.rb +50 -0
- data/lib/insika/commands/delete_data_tool.rb +33 -0
- data/lib/insika/commands/delete_llm_provider.rb +36 -0
- data/lib/insika/commands/delete_mcp.rb +30 -0
- data/lib/insika/commands/delete_skill.rb +43 -0
- data/lib/insika/commands/delete_system_file.rb +29 -0
- data/lib/insika/commands/gate_refinement.rb +245 -0
- data/lib/insika/commands/import_mcp_tools.rb +48 -0
- data/lib/insika/commands/import_tools.rb +81 -0
- data/lib/insika/commands/issue_tenant_token.rb +41 -0
- data/lib/insika/commands/memory_add_note.rb +32 -0
- data/lib/insika/commands/memory_forget_fact.rb +32 -0
- data/lib/insika/commands/memory_put_fact.rb +35 -0
- data/lib/insika/commands/pause_task.rb +29 -0
- data/lib/insika/commands/resolve_refinement.rb +126 -0
- data/lib/insika/commands/restore_agent_file.rb +36 -0
- data/lib/insika/commands/restore_data_tool.rb +34 -0
- data/lib/insika/commands/restore_system_file.rb +31 -0
- data/lib/insika/commands/resume_task.rb +85 -0
- data/lib/insika/commands/revoke_token.rb +39 -0
- data/lib/insika/commands/rotate_tenant_token.rb +43 -0
- data/lib/insika/commands/run_refinement.rb +133 -0
- data/lib/insika/commands/send_message.rb +150 -0
- data/lib/insika/commands/set_agent_tools.rb +39 -0
- data/lib/insika/commands/set_skill_agents.rb +112 -0
- data/lib/insika/commands/trigger_workflow.rb +80 -0
- data/lib/insika/commands/update_agent.rb +49 -0
- data/lib/insika/commands/update_settings.rb +33 -0
- data/lib/insika/commands/upsert_llm_provider.rb +34 -0
- data/lib/insika/commands/upsert_mcp.rb +32 -0
- data/lib/insika/commands/write_agent_file.rb +57 -0
- data/lib/insika/commands/write_data_tool.rb +43 -0
- data/lib/insika/commands/write_golden.rb +58 -0
- data/lib/insika/commands/write_skill.rb +60 -0
- data/lib/insika/commands/write_system_file.rb +31 -0
- data/lib/insika/config_store.rb +89 -0
- data/lib/insika/context/builder.rb +166 -0
- data/lib/insika/context/catalog_provider.rb +23 -0
- data/lib/insika/context/fragment.rb +43 -0
- data/lib/insika/context/priority.rb +30 -0
- data/lib/insika/context/provider.rb +19 -0
- data/lib/insika/context/providers/memory.rb +60 -0
- data/lib/insika/context/providers/prompt.rb +105 -0
- data/lib/insika/context/providers/request.rb +32 -0
- data/lib/insika/context/providers/session.rb +123 -0
- data/lib/insika/context/providers/skill.rb +24 -0
- data/lib/insika/context/providers/skill_trigger.rb +128 -0
- data/lib/insika/context/providers/tool_search.rb +20 -0
- data/lib/insika/context_trace_store.rb +92 -0
- data/lib/insika/delegation_store.rb +153 -0
- data/lib/insika/doctor.rb +539 -0
- data/lib/insika/dsl/definition.rb +55 -0
- data/lib/insika/dsl/runtime.rb +382 -0
- data/lib/insika/dsl/server_boot.rb +98 -0
- data/lib/insika/dsl/system.rb +93 -0
- data/lib/insika/dsl/workflow_adapter.rb +59 -0
- data/lib/insika/dsl.rb +364 -0
- data/lib/insika/edge_limiter.rb +268 -0
- data/lib/insika/egress_guard.rb +75 -0
- data/lib/insika/env_schema.rb +249 -0
- data/lib/insika/errors.rb +201 -0
- data/lib/insika/evals/assertions.rb +247 -0
- data/lib/insika/evals/baseline.rb +69 -0
- data/lib/insika/evals/golden.rb +172 -0
- data/lib/insika/evals/judge.rb +225 -0
- data/lib/insika/evals/pairwise.rb +178 -0
- data/lib/insika/evals/report.rb +115 -0
- data/lib/insika/evals/runner.rb +141 -0
- data/lib/insika/evals/transport.rb +178 -0
- data/lib/insika/event.rb +18 -0
- data/lib/insika/event_stream.rb +132 -0
- data/lib/insika/executor.rb +1995 -0
- data/lib/insika/frontmatter.rb +42 -0
- data/lib/insika/golden_store.rb +145 -0
- data/lib/insika/hooks.rb +48 -0
- data/lib/insika/http_client.rb +63 -0
- data/lib/insika/inbound_log.rb +84 -0
- data/lib/insika/llm_configurator.rb +99 -0
- data/lib/insika/llm_provider_store.rb +83 -0
- data/lib/insika/loop_detector.rb +143 -0
- data/lib/insika/mcp_http_client.rb +67 -0
- data/lib/insika/mcp_store.rb +115 -0
- data/lib/insika/mcp_tool_ingestor.rb +143 -0
- data/lib/insika/memory_store.rb +93 -0
- data/lib/insika/message_origin.rb +76 -0
- data/lib/insika/middleware.rb +36 -0
- data/lib/insika/model_policy.rb +52 -0
- data/lib/insika/model_resolver.rb +176 -0
- data/lib/insika/model_selection.rb +115 -0
- data/lib/insika/onboarding.rb +208 -0
- data/lib/insika/outbox_store.rb +166 -0
- data/lib/insika/overlay_tool_registry.rb +102 -0
- data/lib/insika/pack.rb +102 -0
- data/lib/insika/pack_importer.rb +123 -0
- data/lib/insika/pending_action_store.rb +120 -0
- data/lib/insika/plugin/loader.rb +356 -0
- data/lib/insika/plugin.rb +35 -0
- data/lib/insika/policy/engine.rb +83 -0
- data/lib/insika/policy/policy.rb +120 -0
- data/lib/insika/policy_registry.rb +23 -0
- data/lib/insika/profile_source.rb +143 -0
- data/lib/insika/prompt_catalog.rb +61 -0
- data/lib/insika/provider_error_classifier.rb +160 -0
- data/lib/insika/queue_policy.rb +167 -0
- data/lib/insika/recovery.rb +168 -0
- data/lib/insika/refinement/candidate.rb +159 -0
- data/lib/insika/refinement/evidence_collector.rb +371 -0
- data/lib/insika/refinement/gate.rb +234 -0
- data/lib/insika/refinement/panel.rb +222 -0
- data/lib/insika/refinement/proposer.rb +262 -0
- data/lib/insika/refinement_store.rb +295 -0
- data/lib/insika/registry.rb +59 -0
- data/lib/insika/reliability.rb +185 -0
- data/lib/insika/safety/config.rb +109 -0
- data/lib/insika/safety/detectors.rb +176 -0
- data/lib/insika/safety/factory.rb +102 -0
- data/lib/insika/safety/input_guardrail.rb +102 -0
- data/lib/insika/safety/moderator.rb +94 -0
- data/lib/insika/safety/output_filter.rb +79 -0
- data/lib/insika/safety/output_validator.rb +101 -0
- data/lib/insika/safety/safe_responses.rb +47 -0
- data/lib/insika/sandbox/boundary.rb +93 -0
- data/lib/insika/sandbox/docker.rb +74 -0
- data/lib/insika/sandbox/local.rb +33 -0
- data/lib/insika/sandbox/runner.rb +80 -0
- data/lib/insika/sandbox.rb +85 -0
- data/lib/insika/schema_guard.rb +147 -0
- data/lib/insika/secret_masking.rb +34 -0
- data/lib/insika/server/a2a/agent_card.rb +27 -0
- data/lib/insika/server/a2a/app.rb +112 -0
- data/lib/insika/server/a2a/client.rb +101 -0
- data/lib/insika/server/a2a/errors.rb +32 -0
- data/lib/insika/server/a2a/http.rb +42 -0
- data/lib/insika/server/a2a/message.rb +27 -0
- data/lib/insika/server/a2a/protocol.rb +45 -0
- data/lib/insika/server/a2a/remotes.rb +25 -0
- data/lib/insika/server/a2a/task_projection.rb +40 -0
- data/lib/insika/server/app.rb +1022 -0
- data/lib/insika/server/boot.rb +119 -0
- data/lib/insika/server/rack_app.rb +118 -0
- data/lib/insika/server/responses.rb +165 -0
- data/lib/insika/server/sse_body.rb +96 -0
- data/lib/insika/server/tenant_auth.rb +61 -0
- data/lib/insika/session_actor.rb +162 -0
- data/lib/insika/session_store.rb +143 -0
- data/lib/insika/settings_store.rb +154 -0
- data/lib/insika/shutdown.rb +125 -0
- data/lib/insika/skill_catalog.rb +220 -0
- data/lib/insika/skill_store.rb +127 -0
- data/lib/insika/steer_injector.rb +110 -0
- data/lib/insika/store.rb +52 -0
- data/lib/insika/stores/memory.rb +123 -0
- data/lib/insika/stores/sqlite.rb +183 -0
- data/lib/insika/studio/app.rb +1693 -0
- data/lib/insika/studio/assets/dist/application.css +1 -0
- data/lib/insika/studio/assets/dist/application.js +70 -0
- data/lib/insika/studio/forms.rb +335 -0
- data/lib/insika/studio/nav_icons.rb +31 -0
- data/lib/insika/studio/views/_message.erb +44 -0
- data/lib/insika/studio/views/agent_detail.erb +285 -0
- data/lib/insika/studio/views/agents.erb +63 -0
- data/lib/insika/studio/views/approvals.erb +41 -0
- data/lib/insika/studio/views/chats.erb +34 -0
- data/lib/insika/studio/views/evals.erb +83 -0
- data/lib/insika/studio/views/home.erb +72 -0
- data/lib/insika/studio/views/layout.erb +94 -0
- data/lib/insika/studio/views/login.erb +17 -0
- data/lib/insika/studio/views/mcp.erb +91 -0
- data/lib/insika/studio/views/not_found.erb +5 -0
- data/lib/insika/studio/views/playground.erb +47 -0
- data/lib/insika/studio/views/refinement.erb +234 -0
- data/lib/insika/studio/views/session.erb +137 -0
- data/lib/insika/studio/views/settings.erb +168 -0
- data/lib/insika/studio/views/skills.erb +141 -0
- data/lib/insika/studio/views/system_files.erb +65 -0
- data/lib/insika/studio/views/task.erb +105 -0
- data/lib/insika/studio/views/tasks.erb +33 -0
- data/lib/insika/studio/views/tool_edit.erb +107 -0
- data/lib/insika/studio/views/tools.erb +89 -0
- data/lib/insika/subagent_graph.rb +96 -0
- data/lib/insika/system_file_store.rb +96 -0
- data/lib/insika/task_actor.rb +128 -0
- data/lib/insika/task_store.rb +250 -0
- data/lib/insika/telemetry/pricing.rb +104 -0
- data/lib/insika/telemetry/recorder.rb +228 -0
- data/lib/insika/telemetry.rb +127 -0
- data/lib/insika/testing/store_contract.rb +270 -0
- data/lib/insika/tick.rb +122 -0
- data/lib/insika/token_estimator.rb +16 -0
- data/lib/insika/token_store.rb +168 -0
- data/lib/insika/tool_assembly.rb +140 -0
- data/lib/insika/tool_catalog.rb +89 -0
- data/lib/insika/tool_definition.rb +518 -0
- data/lib/insika/tool_envelope.rb +140 -0
- data/lib/insika/tool_manifest.rb +218 -0
- data/lib/insika/tool_output_compressor.rb +100 -0
- data/lib/insika/tool_registry.rb +21 -0
- data/lib/insika/tool_store.rb +135 -0
- data/lib/insika/tool_trace_store.rb +92 -0
- data/lib/insika/tools/a2a_remote.rb +48 -0
- data/lib/insika/tools/agent_enum.rb +68 -0
- data/lib/insika/tools/concurrency.rb +54 -0
- data/lib/insika/tools/data_defined_tool.rb +219 -0
- data/lib/insika/tools/load_skill.rb +99 -0
- data/lib/insika/tools/remember.rb +53 -0
- data/lib/insika/tools/stuck_signal.rb +44 -0
- data/lib/insika/tools/subagent.rb +75 -0
- data/lib/insika/tools/subagents.rb +77 -0
- data/lib/insika/tools/tool_search.rb +94 -0
- data/lib/insika/turn_output.rb +139 -0
- data/lib/insika/turn_state.rb +162 -0
- data/lib/insika/turn_timing.rb +56 -0
- data/lib/insika/usage_ledger.rb +47 -0
- data/lib/insika/version.rb +3 -1
- data/lib/insika/wiring/graph.rb +249 -0
- data/lib/insika/workflow.rb +185 -0
- data/lib/insika/workflow_registry.rb +33 -0
- data/lib/insika.rb +220 -4
- metadata +412 -8
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "json"
|
|
4
|
+
|
|
5
|
+
module Insika
|
|
6
|
+
module Evals
|
|
7
|
+
# The LLM-judge. Scores a golden's `rubric` against the
|
|
8
|
+
# actual assistant reply — the subjective layer on top of the deterministic
|
|
9
|
+
# asserts. Pure over an injected `ask` callable (prompt -> raw model text), so it's
|
|
10
|
+
# unit-testable without an LLM; the real ask (RubyLLM on the utility_model, temp 0)
|
|
11
|
+
# is built by the CLI.
|
|
12
|
+
#
|
|
13
|
+
# Conservative by construction: an unparseable judge reply scores 0 (fails) rather
|
|
14
|
+
# than silently passing.
|
|
15
|
+
#
|
|
16
|
+
# A PANEL, not a single voice. `quorum: N` samples ONE model N
|
|
17
|
+
# times, which measures that model's variance and little else — at temperature 0 it
|
|
18
|
+
# mostly returns the same answer, including the same blind spot. Two DIFFERENT
|
|
19
|
+
# models disagreeing about a rubric is the signal worth having, so `asks:` takes one
|
|
20
|
+
# callable per model: each judge is scored independently (its own samples, its own
|
|
21
|
+
# median, its own pass/fail against the case's `min_score`), then
|
|
22
|
+
# • `aggregate` combines the scores into the one the report and the baseline
|
|
23
|
+
# read (:median | :mean | :min — :min is the strict panel),
|
|
24
|
+
# • `min_agreement` decides the verdict: the FRACTION of judges that must pass on
|
|
25
|
+
# their own. 0.5 = a majority; 1.0 = unanimous.
|
|
26
|
+
# `quorum` still applies, per judge, so a panel can also be sampled.
|
|
27
|
+
class Judge
|
|
28
|
+
# `judges` carries the per-model scores, so a split panel is visible in the report
|
|
29
|
+
# instead of hiding inside an average.
|
|
30
|
+
Verdict = Struct.new(:score, :pass, :reason, :judges, keyword_init: true)
|
|
31
|
+
|
|
32
|
+
DEFAULT_MIN_SCORE = 0.7
|
|
33
|
+
AGGREGATES = %i[median mean min].freeze
|
|
34
|
+
|
|
35
|
+
# The store's operating policy, stated to the judge in the same words the
|
|
36
|
+
# deterministic layer checks. Empty when the store has no opinion — then the
|
|
37
|
+
# rubric alone decides, and inventing a default here would be inventing an
|
|
38
|
+
# opinion for someone else's store.
|
|
39
|
+
POLICY_INSTRUCTIONS = {
|
|
40
|
+
"ask_once" => "This store allows AT MOST ONE question per reply. Two questions in " \
|
|
41
|
+
"one message is a failure even if the content is otherwise good.",
|
|
42
|
+
"investigate_first" => "This store wants the objective established BEFORE acting: on a " \
|
|
43
|
+
"vague request the assistant should ask (once or twice, not a " \
|
|
44
|
+
"form), not search immediately.",
|
|
45
|
+
"act_fast" => "This store wants the assistant to ACT on the first plausible reading and " \
|
|
46
|
+
"refine after — asking something it could have answered by searching is a " \
|
|
47
|
+
"failure."
|
|
48
|
+
}.freeze
|
|
49
|
+
|
|
50
|
+
# ask: ->(prompt) { "<raw model text>" } — one judge (kept: the common case).
|
|
51
|
+
# asks: [callable, …] — a panel, one entry per model.
|
|
52
|
+
def initialize(ask: nil, asks: nil, quorum: 1, aggregate: :median, min_agreement: 0.5)
|
|
53
|
+
@asks = Array(asks || ask).compact
|
|
54
|
+
raise ArgumentError, "a judge needs at least one `ask`" if @asks.empty?
|
|
55
|
+
|
|
56
|
+
@quorum = [quorum.to_i, 1].max
|
|
57
|
+
@aggregate = aggregate.to_s.to_sym
|
|
58
|
+
raise ArgumentError, "unknown aggregate: #{aggregate}" unless AGGREGATES.include?(@aggregate)
|
|
59
|
+
|
|
60
|
+
@min_agreement = min_agreement.to_f.clamp(0.0, 1.0)
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
# Golden + the LAST TurnResult -> Verdict, or nil when there's nothing to judge
|
|
64
|
+
# (no rubric). `min_score` comes from the golden (default 0.7).
|
|
65
|
+
def score(golden:, result:)
|
|
66
|
+
rubric = golden.rubric.to_s.strip
|
|
67
|
+
return nil if rubric.empty?
|
|
68
|
+
|
|
69
|
+
prompt = build_prompt(rubric, golden.user_turns, result.output_text.to_s, golden.policy)
|
|
70
|
+
min = golden.min_score || DEFAULT_MIN_SCORE
|
|
71
|
+
panel = @asks.map { |ask| judge_once(ask, prompt, min) }
|
|
72
|
+
|
|
73
|
+
agreed = panel.count { |j| j[:pass] }
|
|
74
|
+
Verdict.new(score: combine(panel.map { |j| j[:score] }).round(3),
|
|
75
|
+
pass: (agreed.to_f / panel.length) >= @min_agreement,
|
|
76
|
+
reason: panel.map { |j| j[:reason] }.reject(&:empty?).first.to_s,
|
|
77
|
+
judges: panel.map { |j| j[:score] })
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
private
|
|
81
|
+
|
|
82
|
+
# One model's verdict: its own samples, its own median, its own pass/fail.
|
|
83
|
+
def judge_once(ask, prompt, min)
|
|
84
|
+
samples = Array.new(@quorum) { parse(ask.call(prompt).to_s) }
|
|
85
|
+
med = median(samples.map { |s| s[:score] })
|
|
86
|
+
{ score: med, pass: med >= min,
|
|
87
|
+
reason: samples.map { |s| s[:reason] }.compact.reject(&:empty?).first.to_s }
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
def combine(scores)
|
|
91
|
+
case @aggregate
|
|
92
|
+
when :mean then scores.empty? ? 0.0 : scores.sum / scores.length.to_f
|
|
93
|
+
when :min then scores.min || 0.0
|
|
94
|
+
else median(scores)
|
|
95
|
+
end
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
# The `policy` is the one thing a rubric cannot carry alone: how
|
|
99
|
+
# much this store wants the agent to ask before acting is a per-store decision,
|
|
100
|
+
# and a judge that is not TOLD it will guess — half the time wrongly. The
|
|
101
|
+
# deterministic half is `Assertions.policy_checks`; this is the other half.
|
|
102
|
+
def build_prompt(rubric, user_turns, reply, policy = nil)
|
|
103
|
+
<<~PROMPT
|
|
104
|
+
You are a strict QA judge for a customer-service AI assistant. Judge the
|
|
105
|
+
ASSISTANT REPLY against the RUBRIC — nothing else.
|
|
106
|
+
|
|
107
|
+
RUBRIC:
|
|
108
|
+
#{rubric}
|
|
109
|
+
#{policy_clause(policy)}
|
|
110
|
+
CONVERSATION (user turns, in order):
|
|
111
|
+
#{user_turns.map { |t| "- #{t}" }.join("\n")}
|
|
112
|
+
|
|
113
|
+
ASSISTANT REPLY:
|
|
114
|
+
#{reply}
|
|
115
|
+
|
|
116
|
+
Score from 0.0 (fails the rubric) to 1.0 (fully meets it). Respond with ONLY a
|
|
117
|
+
JSON object, no prose:
|
|
118
|
+
{"score": <0..1>, "reason": "<one short sentence>"}
|
|
119
|
+
PROMPT
|
|
120
|
+
end
|
|
121
|
+
|
|
122
|
+
def policy_clause(policy)
|
|
123
|
+
instruction = POLICY_INSTRUCTIONS[policy.to_s]
|
|
124
|
+
return "" unless instruction
|
|
125
|
+
|
|
126
|
+
"\nSTORE POLICY (weigh this as part of the rubric):\n#{instruction}\n"
|
|
127
|
+
end
|
|
128
|
+
|
|
129
|
+
# Extracts the first {...} block and parses it. Any failure (no JSON, bad JSON,
|
|
130
|
+
# non-numeric score) -> score 0.0 with a diagnostic reason, so a broken judge
|
|
131
|
+
# never masquerades as a pass. Score is clamped to [0,1].
|
|
132
|
+
def parse(raw)
|
|
133
|
+
block = raw[/\{.*\}/m]
|
|
134
|
+
raise JSON::ParserError, "no JSON object" unless block
|
|
135
|
+
|
|
136
|
+
obj = JSON.parse(block)
|
|
137
|
+
score = Float(obj["score"])
|
|
138
|
+
{ score: score.clamp(0.0, 1.0), reason: obj["reason"].to_s }
|
|
139
|
+
rescue JSON::ParserError, ArgumentError, TypeError
|
|
140
|
+
{ score: 0.0, reason: "unparseable judge output: #{raw.to_s[0, 120].inspect}" }
|
|
141
|
+
end
|
|
142
|
+
|
|
143
|
+
def median(nums)
|
|
144
|
+
sorted = nums.compact.sort
|
|
145
|
+
return 0.0 if sorted.empty?
|
|
146
|
+
|
|
147
|
+
mid = sorted.length / 2
|
|
148
|
+
sorted.length.odd? ? sorted[mid] : (sorted[mid - 1] + sorted[mid]) / 2.0
|
|
149
|
+
end
|
|
150
|
+
end
|
|
151
|
+
|
|
152
|
+
# Builds the configured judge panel from `settings["evals"]`.
|
|
153
|
+
#
|
|
154
|
+
# It lives HERE and not in `evals/run.rb` because the CLI is no longer the only
|
|
155
|
+
# caller: the refinement gate scores a candidate with the SAME judges the operator
|
|
156
|
+
# configured, and is explicit that a second copy of the judge would be the
|
|
157
|
+
# worst possible outcome — the gate would then be grading against a rubric nobody
|
|
158
|
+
# tuned. One builder, two callers.
|
|
159
|
+
module JudgePanel
|
|
160
|
+
module_function
|
|
161
|
+
|
|
162
|
+
# settings: the `evals` hash (judges/quorum/aggregate/min_agreement).
|
|
163
|
+
# overrides: the CLI's flags, which win over the stored config.
|
|
164
|
+
# -> [Judge, [model names]] | nil when nobody is configured to ask. NIL AND NOT
|
|
165
|
+
# a no-op judge: a rubric'd case with no judge reads as `judge_pending`, which is
|
|
166
|
+
# visible, where a judge that always passes would be silent.
|
|
167
|
+
# `llm`: the graph's own RubyLLM context; nil = the global
|
|
168
|
+
# constant. Only the deployment root and the CLI build judges today, and both
|
|
169
|
+
# are one graph per process — the seam keeps the default honest for the day
|
|
170
|
+
# an embedded graph gates a candidate on its own credentials.
|
|
171
|
+
def build(settings, overrides: {}, chat_factory: nil, llm: nil)
|
|
172
|
+
settings = Coercion.deep_stringify(settings || {})
|
|
173
|
+
overrides = overrides.transform_keys(&:to_s)
|
|
174
|
+
models = resolve_models(settings, overrides)
|
|
175
|
+
return nil if models.empty?
|
|
176
|
+
|
|
177
|
+
factory = chat_factory || ->(model, provider) { ruby_llm_ask(model, provider, llm: llm) }
|
|
178
|
+
judge = Judge.new(
|
|
179
|
+
asks: models.map { |m| factory.call(m["model"], m["provider"]) },
|
|
180
|
+
quorum: overrides["quorum"] || settings["quorum"] || 1,
|
|
181
|
+
aggregate: overrides["aggregate"] || settings["aggregate"] || "median",
|
|
182
|
+
min_agreement: overrides["min_agreement"] || settings["min_agreement"] || 0.5
|
|
183
|
+
)
|
|
184
|
+
[judge, models.map { |m| m["model"] }]
|
|
185
|
+
end
|
|
186
|
+
|
|
187
|
+
# Sugar for the callers that only want the judge (the gate).
|
|
188
|
+
def judge(settings, **kw) = build(settings, **kw)&.first
|
|
189
|
+
|
|
190
|
+
# The SAME configured models, asked a different question. One
|
|
191
|
+
# builder because "who judges here" is one operator decision: a pairwise panel
|
|
192
|
+
# configured apart from the rubric panel would let a run be graded by judges
|
|
193
|
+
# nobody chose. -> [Pairwise, [model names]] | nil when nobody is configured.
|
|
194
|
+
def pairwise(settings, overrides: {}, chat_factory: nil, llm: nil)
|
|
195
|
+
settings = Coercion.deep_stringify(settings || {})
|
|
196
|
+
models = resolve_models(settings, overrides.transform_keys(&:to_s))
|
|
197
|
+
return nil if models.empty?
|
|
198
|
+
|
|
199
|
+
factory = chat_factory || ->(model, provider) { ruby_llm_ask(model, provider, llm: llm) }
|
|
200
|
+
[Pairwise.new(asks: models.map { |m| factory.call(m["model"], m["provider"]) }),
|
|
201
|
+
models.map { |m| m["model"] }]
|
|
202
|
+
end
|
|
203
|
+
|
|
204
|
+
def resolve_models(settings, overrides)
|
|
205
|
+
models = if Coercion.present?(overrides["judge_model"])
|
|
206
|
+
[{ "model" => overrides["judge_model"], "provider" => overrides["judge_provider"] }]
|
|
207
|
+
else
|
|
208
|
+
Array(settings["judges"])
|
|
209
|
+
end
|
|
210
|
+
models.map { |m| Coercion.deep_stringify(m) }.reject { |m| m["model"].to_s.strip.empty? }
|
|
211
|
+
end
|
|
212
|
+
|
|
213
|
+
# The default way to reach a model: RubyLLM, temperature 0, required lazily so
|
|
214
|
+
# nothing here loads a provider gem until a judge is actually configured.
|
|
215
|
+
def ruby_llm_ask(model, provider, llm: nil)
|
|
216
|
+
require "ruby_llm"
|
|
217
|
+
llm ||= RubyLLM
|
|
218
|
+
lambda do |prompt|
|
|
219
|
+
llm.chat(model: model, provider: provider, assume_model_exists: true)
|
|
220
|
+
.with_temperature(0).ask(prompt).content
|
|
221
|
+
end
|
|
222
|
+
end
|
|
223
|
+
end
|
|
224
|
+
end
|
|
225
|
+
end
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "json"
|
|
4
|
+
|
|
5
|
+
module Insika
|
|
6
|
+
module Evals
|
|
7
|
+
# PAIRWISE AGAINST THE INCUMBENT — the number that answers "can we
|
|
8
|
+
# replace it". An absolute 0.72 says a reply cleared a bar we invented; it says
|
|
9
|
+
# nothing about whether the system already answering 403,231 chats would have done
|
|
10
|
+
# better with the same customer.
|
|
11
|
+
#
|
|
12
|
+
# Same opening, two transcripts, one question: which one served the customer
|
|
13
|
+
# better? Three outcomes (better / comparable / worse), plus two the panel can
|
|
14
|
+
# produce and must not hide — `split` when the judges disagree and `unknown` when
|
|
15
|
+
# none of them answered in a readable way.
|
|
16
|
+
#
|
|
17
|
+
# Three honesty rules, each of them the difference between a number people should
|
|
18
|
+
# trust and one they should not:
|
|
19
|
+
#
|
|
20
|
+
# 1. **Anonymous.** The judge sees "A" and "B" and is never told which one is
|
|
21
|
+
# Insika. Told, it would have an opinion about the new system rather than about
|
|
22
|
+
# the conversations.
|
|
23
|
+
# 2. **Both orders, every judge.** Position bias is THE known failure of pairwise
|
|
24
|
+
# LLM judging, and it is the same species as the kill criterion ("`better`
|
|
25
|
+
# tracks length, or politeness"). So each judge is asked twice with the
|
|
26
|
+
# transcripts swapped, and a verdict that FLIPS with presentation order is
|
|
27
|
+
# reported as `comparable` with `order_dependent: true` — a preference that
|
|
28
|
+
# depends on which one was printed first is not a preference.
|
|
29
|
+
# 3. **A split panel stays split.** Averaging "better" and "worse" into
|
|
30
|
+
# "comparable" invents agreement that nobody expressed.
|
|
31
|
+
#
|
|
32
|
+
# Cost: 2 provider calls per judge per case. This is why it never runs as
|
|
33
|
+
# part of the gate and is opt-in on the CLI.
|
|
34
|
+
class Pairwise
|
|
35
|
+
BETTER = "better"
|
|
36
|
+
COMPARABLE = "comparable"
|
|
37
|
+
WORSE = "worse"
|
|
38
|
+
SPLIT = "split"
|
|
39
|
+
UNKNOWN = "unknown"
|
|
40
|
+
|
|
41
|
+
# `vs` names WHO the reference half actually is: `agent` (model against model)
|
|
42
|
+
# or `human-assisted` (a person typed part of it,'s `origin: operator`).
|
|
43
|
+
# It rides on the verdict rather than beside it so a report cannot print the
|
|
44
|
+
# outcome without the label.
|
|
45
|
+
Verdict = Struct.new(:outcome, :reason, :vs, :judges, :order_dependent, keyword_init: true) do
|
|
46
|
+
def human_assisted? = vs == "human-assisted"
|
|
47
|
+
def decided? = [BETTER, COMPARABLE, WORSE].include?(outcome)
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
# asks: [->(prompt) { "<raw model text>" }] — the SAME panel the rubric judge
|
|
51
|
+
# uses (JudgePanel builds both), because "the judges the operator configured" is
|
|
52
|
+
# one decision, not two.
|
|
53
|
+
def initialize(asks:)
|
|
54
|
+
@asks = Array(asks).compact
|
|
55
|
+
raise ArgumentError, "a pairwise comparison needs at least one `ask`" if @asks.empty?
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
# golden + the run's [TurnResult] -> Verdict, or nil when the case carries no
|
|
59
|
+
# reference (most of them: a pair is curated by a human, like the case itself).
|
|
60
|
+
def compare(golden:, turns:)
|
|
61
|
+
return nil unless golden.reference?
|
|
62
|
+
|
|
63
|
+
ours = Pairwise.transcript(golden.user_turns, turns)
|
|
64
|
+
theirs = Pairwise.reference_transcript(golden.reference_messages)
|
|
65
|
+
return nil if ours.strip.empty?
|
|
66
|
+
|
|
67
|
+
panel = @asks.map { |ask| judge_once(ask, ours, theirs) }
|
|
68
|
+
combine(panel, vs: golden.human_assisted? ? "human-assisted" : "agent")
|
|
69
|
+
end
|
|
70
|
+
|
|
71
|
+
# The replayed conversation as the judge reads it: the user turns we sent,
|
|
72
|
+
# interleaved with the answers the deployment published. `turns` may be shorter
|
|
73
|
+
# than `user_turns` (an errored turn aborts the replay) — zip on what ran.
|
|
74
|
+
def self.transcript(user_turns, turns)
|
|
75
|
+
Array(turns).each_with_index.map do |t, i|
|
|
76
|
+
["customer: #{user_turns[i]}", "assistant: #{t.output_text.to_s.strip}"]
|
|
77
|
+
end.flatten.join("\n")
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
# The incumbent's half. Human turns are NOT flagged to the judge: what it grades
|
|
81
|
+
# is the conversation as the customer received it, and telling it "a person wrote
|
|
82
|
+
# this one" is an invitation to grade the author instead. The fact is carried to
|
|
83
|
+
# the READER as `vs: human-assisted` instead, which is where it changes a decision.
|
|
84
|
+
def self.reference_transcript(messages)
|
|
85
|
+
Array(messages).map do |m|
|
|
86
|
+
speaker = m["role"].to_s == "user" ? "customer" : "assistant"
|
|
87
|
+
"#{speaker}: #{m['text'].to_s.strip}"
|
|
88
|
+
end.join("\n")
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
private
|
|
92
|
+
|
|
93
|
+
# One model, asked twice with the sides swapped. -> { outcome:, reason:,
|
|
94
|
+
# order_dependent: }
|
|
95
|
+
def judge_once(ask, ours, theirs)
|
|
96
|
+
first = outcome_of(parse(ask.call(prompt(ours, theirs))), ours_is: "A")
|
|
97
|
+
second = outcome_of(parse(ask.call(prompt(theirs, ours))), ours_is: "B")
|
|
98
|
+
|
|
99
|
+
return { outcome: UNKNOWN, reason: [first[:reason], second[:reason]].compact.first.to_s, order_dependent: false } \
|
|
100
|
+
if first[:outcome] == UNKNOWN && second[:outcome] == UNKNOWN
|
|
101
|
+
|
|
102
|
+
reason = [first[:reason], second[:reason]].reject { |r| r.to_s.empty? }.first.to_s
|
|
103
|
+
agreed = [first[:outcome], second[:outcome]].reject { |o| o == UNKNOWN }.uniq
|
|
104
|
+
return { outcome: agreed.first, reason: reason, order_dependent: false } if agreed.length == 1
|
|
105
|
+
|
|
106
|
+
{ outcome: COMPARABLE, reason: reason, order_dependent: true }
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
# A strict majority decides; anything less is `split`. Unknown judges are left
|
|
110
|
+
# OUT of the tally (they expressed nothing) but stay visible in `judges`.
|
|
111
|
+
def combine(panel, vs:)
|
|
112
|
+
voted = panel.reject { |j| j[:outcome] == UNKNOWN }
|
|
113
|
+
winner, group = voted.group_by { |j| j[:outcome] }.max_by { |_, js| js.length }
|
|
114
|
+
outcome = if voted.empty? then UNKNOWN
|
|
115
|
+
elsif group.length * 2 > voted.length then winner
|
|
116
|
+
else SPLIT
|
|
117
|
+
end
|
|
118
|
+
# The reason comes from the judges that CARRIED the verdict, so a split or an
|
|
119
|
+
# unknown is explained by whoever produced it rather than by an outvoted one.
|
|
120
|
+
spoke = outcome == winner ? group : panel
|
|
121
|
+
|
|
122
|
+
Verdict.new(outcome: outcome, vs: vs,
|
|
123
|
+
reason: spoke.map { |j| j[:reason].to_s }.reject(&:empty?).first.to_s,
|
|
124
|
+
judges: panel.map { |j| j[:outcome] },
|
|
125
|
+
order_dependent: panel.any? { |j| j[:order_dependent] })
|
|
126
|
+
end
|
|
127
|
+
|
|
128
|
+
# "A won" means Insika won only when Insika WAS A. The mapping is the whole
|
|
129
|
+
# point of asking twice.
|
|
130
|
+
def outcome_of(parsed, ours_is:)
|
|
131
|
+
winner = parsed[:winner]
|
|
132
|
+
outcome = case winner
|
|
133
|
+
when "tie" then COMPARABLE
|
|
134
|
+
when nil then UNKNOWN
|
|
135
|
+
else winner == ours_is ? BETTER : WORSE
|
|
136
|
+
end
|
|
137
|
+
{ outcome: outcome, reason: parsed[:reason] }
|
|
138
|
+
end
|
|
139
|
+
|
|
140
|
+
def prompt(a, b)
|
|
141
|
+
<<~PROMPT
|
|
142
|
+
You are comparing two customer-service conversations that began with the SAME
|
|
143
|
+
customer message. Decide which one SERVED THE CUSTOMER BETTER: did the customer
|
|
144
|
+
get what they came for, without detours, wrong information or invented facts?
|
|
145
|
+
|
|
146
|
+
Ignore length, tone, politeness, emoji and formatting UNLESS they changed what
|
|
147
|
+
the customer actually got. A short answer that solves the problem beats a long
|
|
148
|
+
one that does not.
|
|
149
|
+
|
|
150
|
+
CONVERSATION A:
|
|
151
|
+
#{a}
|
|
152
|
+
|
|
153
|
+
CONVERSATION B:
|
|
154
|
+
#{b}
|
|
155
|
+
|
|
156
|
+
Respond with ONLY a JSON object, no prose:
|
|
157
|
+
{"winner": "A" | "B" | "tie", "reason": "<one short sentence>"}
|
|
158
|
+
PROMPT
|
|
159
|
+
end
|
|
160
|
+
|
|
161
|
+
# An unreadable reply is UNKNOWN, never a preference: scoring it as a tie would
|
|
162
|
+
# quietly count a broken judge as evidence that the two systems are equivalent.
|
|
163
|
+
def parse(raw)
|
|
164
|
+
block = raw.to_s[/\{.*\}/m]
|
|
165
|
+
raise JSON::ParserError, "no JSON object" unless block
|
|
166
|
+
|
|
167
|
+
obj = JSON.parse(block)
|
|
168
|
+
winner = obj["winner"].to_s.strip.upcase
|
|
169
|
+
winner = "tie" if %w[TIE DRAW EQUAL COMPARABLE].include?(winner)
|
|
170
|
+
raise JSON::ParserError, "unknown winner" unless %w[A B tie].include?(winner)
|
|
171
|
+
|
|
172
|
+
{ winner: winner, reason: obj["reason"].to_s }
|
|
173
|
+
rescue JSON::ParserError, TypeError
|
|
174
|
+
{ winner: nil, reason: "unparseable pairwise output: #{raw.to_s[0, 120].inspect}" }
|
|
175
|
+
end
|
|
176
|
+
end
|
|
177
|
+
end
|
|
178
|
+
end
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "json"
|
|
4
|
+
|
|
5
|
+
module Insika
|
|
6
|
+
module Evals
|
|
7
|
+
# Renders a run's [CaseResult] as a machine-readable JSON blob (for the baseline
|
|
8
|
+
# gating in) and a human-readable markdown summary.
|
|
9
|
+
# Pure over the results — takes a clock value in, never reads it (so callers stay
|
|
10
|
+
# deterministic/testable).
|
|
11
|
+
module Report
|
|
12
|
+
module_function
|
|
13
|
+
|
|
14
|
+
# -> Hash ready for JSON. `at` is an ISO-8601 string stamped by the caller.
|
|
15
|
+
def to_h(results, at:)
|
|
16
|
+
passed = results.count(&:pass?)
|
|
17
|
+
skipped = results.count(&:skipped?)
|
|
18
|
+
{
|
|
19
|
+
"at" => at,
|
|
20
|
+
"total" => results.size,
|
|
21
|
+
"passed" => passed,
|
|
22
|
+
# A skipped case is neither: counting it as failed is the lie this outcome
|
|
23
|
+
# exists to stop, and counting it as passed is worse.
|
|
24
|
+
"failed" => results.size - passed - skipped,
|
|
25
|
+
"skipped" => skipped,
|
|
26
|
+
"judge_pending" => results.count(&:judge_pending?),
|
|
27
|
+
# Absent (not an empty tally) when nothing was compared — a run with no
|
|
28
|
+
# pairwise is the normal one, and a zeroed block reads like every case tied.
|
|
29
|
+
"pairwise" => pairwise_summary(results),
|
|
30
|
+
"cases" => results.map do |r|
|
|
31
|
+
{
|
|
32
|
+
"id" => r.id, "agent" => r.agent, "pass" => r.pass?,
|
|
33
|
+
"skipped" => r.skipped,
|
|
34
|
+
"judge_pending" => r.judge_pending?, "error" => r.error,
|
|
35
|
+
"checks" => r.checks.map { |c| { "name" => c.name, "pass" => c.pass, "detail" => c.detail } },
|
|
36
|
+
"judge" => (r.judge && { "score" => r.judge.score, "pass" => r.judge.pass, "reason" => r.judge.reason }),
|
|
37
|
+
"pairwise" => (r.pairwise && { "outcome" => r.pairwise.outcome, "vs" => r.pairwise.vs,
|
|
38
|
+
"reason" => r.pairwise.reason, "judges" => r.pairwise.judges,
|
|
39
|
+
"order_dependent" => r.pairwise.order_dependent })
|
|
40
|
+
}
|
|
41
|
+
end
|
|
42
|
+
}.compact
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
# -> counts by outcome, or nil when no case carried a reference. `human_assisted`
|
|
46
|
+
# is counted separately and always printed: a "better" against a conversation a
|
|
47
|
+
# PERSON typed is a different claim from one against the incumbent's model, and
|
|
48
|
+
# the two must never be summed into one number somebody quotes.
|
|
49
|
+
def pairwise_summary(results)
|
|
50
|
+
compared = results.filter_map(&:pairwise)
|
|
51
|
+
return nil if compared.empty?
|
|
52
|
+
|
|
53
|
+
counts = compared.group_by(&:outcome).transform_values(&:size)
|
|
54
|
+
{ "compared" => compared.size,
|
|
55
|
+
"human_assisted" => compared.count(&:human_assisted?),
|
|
56
|
+
"order_dependent" => compared.count(&:order_dependent),
|
|
57
|
+
"outcomes" => counts }
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
def to_json(results, at:)
|
|
61
|
+
JSON.pretty_generate(to_h(results, at: at))
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
PAIRWISE_MARK = { "better" => "🟢", "comparable" => "🟡", "worse" => "🔴",
|
|
65
|
+
"split" => "⚖️", "unknown" => "❔" }.freeze
|
|
66
|
+
|
|
67
|
+
# Never without `vs:` — see `pairwise_summary`.
|
|
68
|
+
def pairwise_line(v)
|
|
69
|
+
flip = v.order_dependent ? " (order-dependent)" : ""
|
|
70
|
+
"#{PAIRWISE_MARK.fetch(v.outcome, '·')} vs incumbent (#{v.vs}): #{v.outcome}#{flip} — #{v.reason}"
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
def pairwise_block(summary)
|
|
74
|
+
counts = summary["outcomes"].map { |k, n| "#{n} #{k}" }.join(" · ")
|
|
75
|
+
lines = ["", "**vs incumbent** (#{summary['compared']} compared): #{counts}"]
|
|
76
|
+
if summary["human_assisted"].positive?
|
|
77
|
+
lines << "- #{summary['human_assisted']} against a HUMAN-ASSISTED transcript (a person typed " \
|
|
78
|
+
"part of the reference — not a model-vs-model result)"
|
|
79
|
+
end
|
|
80
|
+
if summary["order_dependent"].positive?
|
|
81
|
+
lines << "- #{summary['order_dependent']} flipped when the transcripts were swapped, " \
|
|
82
|
+
"and were reported as comparable"
|
|
83
|
+
end
|
|
84
|
+
lines
|
|
85
|
+
end
|
|
86
|
+
|
|
87
|
+
# Human summary. One line per case; failing checks nested underneath.
|
|
88
|
+
def to_markdown(results, at:)
|
|
89
|
+
h = to_h(results, at: at)
|
|
90
|
+
lines = ["# Eval report — #{at}", "",
|
|
91
|
+
"**#{h['passed']}/#{h['total'] - h['skipped']} passed** · #{h['failed']} failed" \
|
|
92
|
+
"#{" · #{h['skipped']} skipped" if h['skipped'].positive?}" \
|
|
93
|
+
"#{" · #{h['judge_pending']} awaiting judge" if h['judge_pending'].positive?}", ""]
|
|
94
|
+
results.each do |r|
|
|
95
|
+
if r.skipped?
|
|
96
|
+
# WITH the reason, always: "12 skipped" alone is indistinguishable from a
|
|
97
|
+
# suite that quietly stopped testing anything.
|
|
98
|
+
lines << "- ⏭️ `#{r.id}` (#{r.agent}) — skipped: #{r.skipped}"
|
|
99
|
+
next
|
|
100
|
+
end
|
|
101
|
+
|
|
102
|
+
lines << "- #{r.pass? ? '✅' : '❌'} `#{r.id}` (#{r.agent})#{' ⏳ judge pending' if r.judge_pending?}"
|
|
103
|
+
r.failures.each { |c| lines << " - ❌ #{c.name}: #{c.detail}" }
|
|
104
|
+
if r.judge
|
|
105
|
+
v = r.judge
|
|
106
|
+
lines << " - #{v.pass ? '✅' : '❌'} judge: #{v.score} — #{v.reason}"
|
|
107
|
+
end
|
|
108
|
+
lines << " - #{pairwise_line(r.pairwise)}" if r.pairwise
|
|
109
|
+
end
|
|
110
|
+
lines.concat(pairwise_block(h["pairwise"])) if h["pairwise"]
|
|
111
|
+
"#{lines.join("\n")}\n"
|
|
112
|
+
end
|
|
113
|
+
end
|
|
114
|
+
end
|
|
115
|
+
end
|
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative "golden"
|
|
4
|
+
require_relative "assertions"
|
|
5
|
+
|
|
6
|
+
module Insika
|
|
7
|
+
module Evals
|
|
8
|
+
# Replays golden cases through a Transport and evaluates each. Pure over the
|
|
9
|
+
# Transport (injected) — the fake makes the orchestration unit-testable offline;
|
|
10
|
+
# the HttpTransport makes a real run.
|
|
11
|
+
#
|
|
12
|
+
# Multi-turn: a case's turns replay IN ORDER under one conversation id
|
|
13
|
+
# (`eval-<id>`); earlier turns build context, the assertion runs on the LAST
|
|
14
|
+
# turn's result. A turn that errors aborts the rest of that conversation (there's
|
|
15
|
+
# nothing to continue from) and fails the case.
|
|
16
|
+
class Runner
|
|
17
|
+
# `tokens` is what the whole case cost, summed over its turns, or nil when no
|
|
18
|
+
# turn reported usage. `cached` is how much of that was served from the prompt
|
|
19
|
+
# cache, carried separately because it is the number that explains a total.
|
|
20
|
+
# Only the refinement gate reads them (records a run's cost); the
|
|
21
|
+
# report and the exit code are untouched.
|
|
22
|
+
RunCase = Struct.new(:result, :timings, :tokens, :cached, keyword_init: true)
|
|
23
|
+
|
|
24
|
+
# judge: an Evals::Judge (optional). When set, a case with a rubric whose turn
|
|
25
|
+
# ran cleanly gets a subjective verdict attached on top of the deterministic pass.
|
|
26
|
+
#
|
|
27
|
+
# capabilities: what the DEPLOYMENT has, per agent — anything answering
|
|
28
|
+
# `#for(agent_id)` with { "tools" =>, "capabilities" => } or nil. Used to skip a
|
|
29
|
+
# case the deployment cannot satisfy, BEFORE spending a turn on
|
|
30
|
+
# it. nil (or an unknown agent) = no resolution, and then a case with `requires`
|
|
31
|
+
# RUNS and says so in the report: "could not rule it out" is not a reason to
|
|
32
|
+
# stop testing something, and a suite that shrinks in silence is the failure
|
|
33
|
+
# this feature exists to avoid.
|
|
34
|
+
#
|
|
35
|
+
# pairwise: an Evals::Pairwise (optional). Only cases carrying a
|
|
36
|
+
# `reference:` are compared, and the verdict never touches pass/fail — it is the
|
|
37
|
+
# answer to "can we replace it", reported beside the suite's own verdict.
|
|
38
|
+
def initialize(transport:, judge: nil, conv_map: {}, capabilities: nil, pairwise: nil)
|
|
39
|
+
@transport = transport
|
|
40
|
+
@judge = judge
|
|
41
|
+
@conv_map = conv_map || {}
|
|
42
|
+
@capabilities = capabilities
|
|
43
|
+
@pairwise = pairwise
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
# [Golden] -> [RunCase]. Each RunCase carries the CaseResult (for the report) +
|
|
47
|
+
# per-turn timings (for `--mode perf`).
|
|
48
|
+
def run(goldens)
|
|
49
|
+
goldens.map { |g| run_case(g) }
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
def run_case(golden)
|
|
53
|
+
skip = skip_reason(golden)
|
|
54
|
+
return RunCase.new(result: Assertions.skip(golden, skip), timings: []) if skip
|
|
55
|
+
|
|
56
|
+
# A backend that resolves state from a pre-existing conversation (e.g. a
|
|
57
|
+
# consumer needing a real Chat UUID as X-Chat-Id) supplies it via conv_map; otherwise the
|
|
58
|
+
# synthetic "eval-<id>" keeps the adapter's own multi-turn continuation.
|
|
59
|
+
conv = @conv_map[golden.id] || "eval-#{golden.id}"
|
|
60
|
+
turns = []
|
|
61
|
+
timings = []
|
|
62
|
+
spent = []
|
|
63
|
+
cached = []
|
|
64
|
+
golden.user_turns.each do |message|
|
|
65
|
+
outcome = @transport.turn(agent: golden.agent, conv: conv, message: message)
|
|
66
|
+
timings << { ttfb: outcome.ttfb, total: outcome.total }
|
|
67
|
+
spent << billed_tokens(outcome.usage)
|
|
68
|
+
cached << cached_tokens(outcome.usage)
|
|
69
|
+
turns << outcome.result
|
|
70
|
+
break if outcome.result.error
|
|
71
|
+
end
|
|
72
|
+
last = turns.last
|
|
73
|
+
# Tool/content assertions read the last turn (unchanged); the policy checks
|
|
74
|
+
# read every turn — "one question per reply" is a rule about each of them.
|
|
75
|
+
result = Assertions.evaluate(golden, last, turns: turns)
|
|
76
|
+
# Subjective layer: only when a judge is configured, the case has a rubric, and
|
|
77
|
+
# the turn ran cleanly (nothing to judge on an errored turn).
|
|
78
|
+
result.judge = @judge.score(golden: golden, result: last) if @judge && result.rubric && result.error.nil?
|
|
79
|
+
# Against the incumbent. Same rule as the judge: nothing to
|
|
80
|
+
# compare on a turn that errored — half a conversation would lose the
|
|
81
|
+
# comparison for a reason that has nothing to do with the agent.
|
|
82
|
+
result.pairwise = @pairwise.compare(golden: golden, turns: turns) if @pairwise && result.error.nil?
|
|
83
|
+
RunCase.new(result: result, timings: timings, tokens: sum_tokens(spent),
|
|
84
|
+
cached: sum_tokens(cached))
|
|
85
|
+
end
|
|
86
|
+
|
|
87
|
+
private
|
|
88
|
+
|
|
89
|
+
# nil when NO turn reported usage; otherwise the sum of the ones that did. A
|
|
90
|
+
# partially-metered case is reported as what was actually measured, low rather
|
|
91
|
+
# than absent — a budget under-counting is a smaller lie than a budget that
|
|
92
|
+
# throws the number away because one leg was silent.
|
|
93
|
+
def sum_tokens(spent)
|
|
94
|
+
counted = spent.compact
|
|
95
|
+
counted.empty? ? nil : counted.sum
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
# What the turn actually SENT, cache included. The engine's `total_tokens` is
|
|
99
|
+
# input + output and DELIBERATELY excludes the cached prefix (`Executor#usage_of`
|
|
100
|
+
# reports `cached_tokens` alongside it), so reading it as the cost of a turn
|
|
101
|
+
# under-reads a cached identity by an order of magnitude.
|
|
102
|
+
#
|
|
103
|
+
# Measured on the pilot: one real turn reports `total_tokens: 88` with
|
|
104
|
+
# `cached_tokens: 26624`. A refinement budget built on the first number would
|
|
105
|
+
# have let a run send ~300× what its ceiling said. Cached tokens are cheaper
|
|
106
|
+
# than fresh ones; they are not free, and a ceiling has to see them.
|
|
107
|
+
def billed_tokens(usage)
|
|
108
|
+
return nil unless usage.is_a?(Hash)
|
|
109
|
+
|
|
110
|
+
total = read(usage, "total_tokens")
|
|
111
|
+
return nil if total.nil?
|
|
112
|
+
|
|
113
|
+
total + cached_tokens(usage).to_i
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
def cached_tokens(usage)
|
|
117
|
+
return nil unless usage.is_a?(Hash)
|
|
118
|
+
|
|
119
|
+
values = [read(usage, "cached_tokens"), read(usage, "cache_creation_tokens")].compact
|
|
120
|
+
values.empty? ? nil : values.sum
|
|
121
|
+
end
|
|
122
|
+
|
|
123
|
+
def read(usage, key)
|
|
124
|
+
value = usage[key] || usage[key.to_sym]
|
|
125
|
+
value&.to_i
|
|
126
|
+
end
|
|
127
|
+
|
|
128
|
+
# -> the reason to skip, or nil to run. A case with no `requires` always runs
|
|
129
|
+
# (and never even asks), which keeps the whole existing corpus untouched.
|
|
130
|
+
def skip_reason(golden)
|
|
131
|
+
return nil unless golden.requirements?
|
|
132
|
+
|
|
133
|
+
available = @capabilities&.for(golden.agent)
|
|
134
|
+
return nil if available.nil? # unresolved: run it, and the report says so
|
|
135
|
+
|
|
136
|
+
unmet = Assertions.unmet_requirements(golden, available)
|
|
137
|
+
unmet.empty? ? nil : unmet.join("; ")
|
|
138
|
+
end
|
|
139
|
+
end
|
|
140
|
+
end
|
|
141
|
+
end
|