insika 0.0.1 → 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +295 -0
- data/LICENSE +21 -0
- data/README.md +136 -2
- data/bin/insika +351 -0
- data/docs/AGENTS.md +494 -0
- data/docs/ARCHITECTURE.md +333 -0
- data/docs/BENCHMARK.md +114 -0
- data/docs/CHANNELS.md +453 -0
- data/docs/CONTEXT.md +100 -0
- data/docs/DEPLOY.md +334 -0
- data/docs/EMBEDDING.md +194 -0
- data/docs/EVALS.md +273 -0
- data/docs/LOADTEST.md +231 -0
- data/docs/OBSERVABILITY.md +365 -0
- data/docs/PLUGINS.md +211 -0
- data/docs/REFINEMENT.md +477 -0
- data/docs/RELEASING.md +70 -0
- data/docs/RUNNING-LOCAL.md +153 -0
- data/docs/SANDBOX.md +114 -0
- data/docs/SECURITY.md +362 -0
- data/docs/SKILLS.md +98 -0
- data/docs/TOOLS.md +302 -0
- data/docs/WHY.md +137 -0
- data/docs/WORKFLOWS.md +225 -0
- data/docs/build.md +14 -0
- data/docs/index.md +68 -0
- data/docs/onboarding/start.md +126 -0
- data/docs/operate.md +12 -0
- data/docs/ship.md +10 -0
- data/docs/understand.md +10 -0
- data/lib/insika/agent_file_store.rb +125 -0
- data/lib/insika/agent_profile.rb +188 -0
- data/lib/insika/allowlist.rb +28 -0
- data/lib/insika/baseline_store.rb +74 -0
- data/lib/insika/capability/resolved_tool.rb +34 -0
- data/lib/insika/capability_registry.rb +112 -0
- data/lib/insika/channel_delivery.rb +150 -0
- data/lib/insika/channel_registry.rb +30 -0
- data/lib/insika/channels/relay.rb +178 -0
- data/lib/insika/channels/web/widget.js +283 -0
- data/lib/insika/channels/web.rb +211 -0
- data/lib/insika/chat_builder.rb +254 -0
- data/lib/insika/checkpoint.rb +13 -0
- data/lib/insika/checkpoint_store.rb +153 -0
- data/lib/insika/coercion.rb +50 -0
- data/lib/insika/command.rb +32 -0
- data/lib/insika/command_bus.rb +39 -0
- data/lib/insika/commands/agent_payload.rb +41 -0
- data/lib/insika/commands/approve_action.rb +46 -0
- data/lib/insika/commands/cancel_task.rb +33 -0
- data/lib/insika/commands/create_agent.rb +54 -0
- data/lib/insika/commands/create_session.rb +67 -0
- data/lib/insika/commands/delete_agent.rb +33 -0
- data/lib/insika/commands/delete_agent_file.rb +50 -0
- data/lib/insika/commands/delete_data_tool.rb +33 -0
- data/lib/insika/commands/delete_llm_provider.rb +36 -0
- data/lib/insika/commands/delete_mcp.rb +30 -0
- data/lib/insika/commands/delete_system_file.rb +29 -0
- data/lib/insika/commands/gate_refinement.rb +245 -0
- data/lib/insika/commands/import_mcp_tools.rb +48 -0
- data/lib/insika/commands/import_tools.rb +81 -0
- data/lib/insika/commands/memory_add_note.rb +32 -0
- data/lib/insika/commands/memory_forget_fact.rb +32 -0
- data/lib/insika/commands/memory_put_fact.rb +35 -0
- data/lib/insika/commands/pause_task.rb +29 -0
- data/lib/insika/commands/resolve_refinement.rb +126 -0
- data/lib/insika/commands/restore_agent_file.rb +36 -0
- data/lib/insika/commands/restore_data_tool.rb +34 -0
- data/lib/insika/commands/restore_system_file.rb +31 -0
- data/lib/insika/commands/resume_task.rb +85 -0
- data/lib/insika/commands/run_refinement.rb +133 -0
- data/lib/insika/commands/send_message.rb +150 -0
- data/lib/insika/commands/set_agent_tools.rb +39 -0
- data/lib/insika/commands/set_skill_agents.rb +71 -0
- data/lib/insika/commands/trigger_workflow.rb +80 -0
- data/lib/insika/commands/update_agent.rb +49 -0
- data/lib/insika/commands/update_settings.rb +33 -0
- data/lib/insika/commands/upsert_llm_provider.rb +34 -0
- data/lib/insika/commands/upsert_mcp.rb +32 -0
- data/lib/insika/commands/write_agent_file.rb +57 -0
- data/lib/insika/commands/write_data_tool.rb +43 -0
- data/lib/insika/commands/write_golden.rb +58 -0
- data/lib/insika/commands/write_skill.rb +50 -0
- data/lib/insika/commands/write_system_file.rb +31 -0
- data/lib/insika/config_store.rb +85 -0
- data/lib/insika/context/builder.rb +166 -0
- data/lib/insika/context/catalog_provider.rb +23 -0
- data/lib/insika/context/fragment.rb +19 -0
- data/lib/insika/context/priority.rb +29 -0
- data/lib/insika/context/provider.rb +19 -0
- data/lib/insika/context/providers/memory.rb +60 -0
- data/lib/insika/context/providers/prompt.rb +105 -0
- data/lib/insika/context/providers/request.rb +32 -0
- data/lib/insika/context/providers/session.rb +108 -0
- data/lib/insika/context/providers/skill.rb +20 -0
- data/lib/insika/context/providers/tool_search.rb +20 -0
- data/lib/insika/delegation_store.rb +153 -0
- data/lib/insika/doctor.rb +294 -0
- data/lib/insika/dsl/definition.rb +55 -0
- data/lib/insika/dsl/runtime.rb +379 -0
- data/lib/insika/dsl/server_boot.rb +97 -0
- data/lib/insika/dsl/system.rb +93 -0
- data/lib/insika/dsl/workflow_adapter.rb +59 -0
- data/lib/insika/dsl.rb +307 -0
- data/lib/insika/edge_limiter.rb +130 -0
- data/lib/insika/egress_guard.rb +75 -0
- data/lib/insika/env_schema.rb +246 -0
- data/lib/insika/errors.rb +145 -0
- data/lib/insika/evals/assertions.rb +247 -0
- data/lib/insika/evals/baseline.rb +69 -0
- data/lib/insika/evals/golden.rb +172 -0
- data/lib/insika/evals/judge.rb +225 -0
- data/lib/insika/evals/pairwise.rb +178 -0
- data/lib/insika/evals/report.rb +115 -0
- data/lib/insika/evals/runner.rb +141 -0
- data/lib/insika/evals/transport.rb +178 -0
- data/lib/insika/event.rb +18 -0
- data/lib/insika/event_stream.rb +114 -0
- data/lib/insika/executor.rb +1680 -0
- data/lib/insika/frontmatter.rb +42 -0
- data/lib/insika/golden_store.rb +145 -0
- data/lib/insika/hooks.rb +48 -0
- data/lib/insika/http_client.rb +63 -0
- data/lib/insika/inbound_log.rb +84 -0
- data/lib/insika/llm_configurator.rb +99 -0
- data/lib/insika/llm_provider_store.rb +83 -0
- data/lib/insika/mcp_http_client.rb +67 -0
- data/lib/insika/mcp_store.rb +115 -0
- data/lib/insika/mcp_tool_ingestor.rb +143 -0
- data/lib/insika/memory_store.rb +93 -0
- data/lib/insika/message_origin.rb +76 -0
- data/lib/insika/middleware.rb +36 -0
- data/lib/insika/model_policy.rb +52 -0
- data/lib/insika/model_resolver.rb +176 -0
- data/lib/insika/model_selection.rb +114 -0
- data/lib/insika/onboarding.rb +208 -0
- data/lib/insika/outbox_store.rb +166 -0
- data/lib/insika/overlay_tool_registry.rb +103 -0
- data/lib/insika/pack.rb +102 -0
- data/lib/insika/pack_importer.rb +121 -0
- data/lib/insika/pending_action_store.rb +120 -0
- data/lib/insika/plugin/loader.rb +356 -0
- data/lib/insika/plugin.rb +35 -0
- data/lib/insika/policy/engine.rb +83 -0
- data/lib/insika/policy/policy.rb +120 -0
- data/lib/insika/policy_registry.rb +23 -0
- data/lib/insika/profile_source.rb +137 -0
- data/lib/insika/prompt_catalog.rb +61 -0
- data/lib/insika/queue_policy.rb +167 -0
- data/lib/insika/recovery.rb +127 -0
- data/lib/insika/refinement/candidate.rb +159 -0
- data/lib/insika/refinement/evidence_collector.rb +371 -0
- data/lib/insika/refinement/gate.rb +234 -0
- data/lib/insika/refinement/panel.rb +222 -0
- data/lib/insika/refinement/proposer.rb +262 -0
- data/lib/insika/refinement_store.rb +295 -0
- data/lib/insika/registry.rb +59 -0
- data/lib/insika/safety/config.rb +109 -0
- data/lib/insika/safety/detectors.rb +176 -0
- data/lib/insika/safety/factory.rb +102 -0
- data/lib/insika/safety/input_guardrail.rb +87 -0
- data/lib/insika/safety/moderator.rb +86 -0
- data/lib/insika/safety/output_filter.rb +79 -0
- data/lib/insika/safety/output_validator.rb +101 -0
- data/lib/insika/safety/safe_responses.rb +47 -0
- data/lib/insika/sandbox/boundary.rb +93 -0
- data/lib/insika/sandbox/docker.rb +74 -0
- data/lib/insika/sandbox/local.rb +33 -0
- data/lib/insika/sandbox/runner.rb +80 -0
- data/lib/insika/sandbox.rb +85 -0
- data/lib/insika/schema_guard.rb +147 -0
- data/lib/insika/secret_masking.rb +34 -0
- data/lib/insika/server/a2a/agent_card.rb +27 -0
- data/lib/insika/server/a2a/app.rb +112 -0
- data/lib/insika/server/a2a/client.rb +101 -0
- data/lib/insika/server/a2a/errors.rb +32 -0
- data/lib/insika/server/a2a/http.rb +42 -0
- data/lib/insika/server/a2a/message.rb +27 -0
- data/lib/insika/server/a2a/protocol.rb +45 -0
- data/lib/insika/server/a2a/remotes.rb +25 -0
- data/lib/insika/server/a2a/task_projection.rb +40 -0
- data/lib/insika/server/admin_auth.rb +29 -0
- data/lib/insika/server/app.rb +850 -0
- data/lib/insika/server/boot.rb +119 -0
- data/lib/insika/server/rack_app.rb +110 -0
- data/lib/insika/server/responses.rb +155 -0
- data/lib/insika/server/sse_body.rb +96 -0
- data/lib/insika/session_actor.rb +162 -0
- data/lib/insika/session_store.rb +143 -0
- data/lib/insika/settings_store.rb +154 -0
- data/lib/insika/shutdown.rb +125 -0
- data/lib/insika/skill_catalog.rb +113 -0
- data/lib/insika/skill_store.rb +79 -0
- data/lib/insika/steer_injector.rb +110 -0
- data/lib/insika/store.rb +52 -0
- data/lib/insika/stores/memory.rb +123 -0
- data/lib/insika/stores/sqlite.rb +183 -0
- data/lib/insika/studio/app.rb +1571 -0
- data/lib/insika/studio/assets/dist/application.css +1 -0
- data/lib/insika/studio/assets/dist/application.js +69 -0
- data/lib/insika/studio/forms.rb +340 -0
- data/lib/insika/studio/nav_icons.rb +31 -0
- data/lib/insika/studio/views/_message.erb +44 -0
- data/lib/insika/studio/views/agent_detail.erb +285 -0
- data/lib/insika/studio/views/agents.erb +63 -0
- data/lib/insika/studio/views/approvals.erb +41 -0
- data/lib/insika/studio/views/chats.erb +34 -0
- data/lib/insika/studio/views/evals.erb +83 -0
- data/lib/insika/studio/views/home.erb +72 -0
- data/lib/insika/studio/views/layout.erb +94 -0
- data/lib/insika/studio/views/login.erb +17 -0
- data/lib/insika/studio/views/mcp.erb +91 -0
- data/lib/insika/studio/views/not_found.erb +5 -0
- data/lib/insika/studio/views/playground.erb +47 -0
- data/lib/insika/studio/views/refinement.erb +234 -0
- data/lib/insika/studio/views/session.erb +62 -0
- data/lib/insika/studio/views/settings.erb +173 -0
- data/lib/insika/studio/views/skills.erb +86 -0
- data/lib/insika/studio/views/system_files.erb +65 -0
- data/lib/insika/studio/views/task.erb +105 -0
- data/lib/insika/studio/views/tasks.erb +33 -0
- data/lib/insika/studio/views/tool_edit.erb +107 -0
- data/lib/insika/studio/views/tools.erb +89 -0
- data/lib/insika/subagent_graph.rb +96 -0
- data/lib/insika/system_file_store.rb +96 -0
- data/lib/insika/task_actor.rb +128 -0
- data/lib/insika/task_store.rb +250 -0
- data/lib/insika/telemetry/pricing.rb +104 -0
- data/lib/insika/telemetry/recorder.rb +228 -0
- data/lib/insika/telemetry.rb +127 -0
- data/lib/insika/testing/store_contract.rb +270 -0
- data/lib/insika/token_estimator.rb +16 -0
- data/lib/insika/tool_assembly.rb +140 -0
- data/lib/insika/tool_catalog.rb +89 -0
- data/lib/insika/tool_definition.rb +518 -0
- data/lib/insika/tool_envelope.rb +140 -0
- data/lib/insika/tool_manifest.rb +218 -0
- data/lib/insika/tool_registry.rb +21 -0
- data/lib/insika/tool_store.rb +135 -0
- data/lib/insika/tool_trace_store.rb +92 -0
- data/lib/insika/tools/a2a_remote.rb +48 -0
- data/lib/insika/tools/agent_enum.rb +68 -0
- data/lib/insika/tools/concurrency.rb +54 -0
- data/lib/insika/tools/data_defined_tool.rb +220 -0
- data/lib/insika/tools/load_skill.rb +41 -0
- data/lib/insika/tools/remember.rb +53 -0
- data/lib/insika/tools/subagent.rb +75 -0
- data/lib/insika/tools/subagents.rb +77 -0
- data/lib/insika/tools/tool_search.rb +94 -0
- data/lib/insika/turn_output.rb +139 -0
- data/lib/insika/turn_state.rb +158 -0
- data/lib/insika/turn_timing.rb +56 -0
- data/lib/insika/usage_ledger.rb +47 -0
- data/lib/insika/version.rb +3 -1
- data/lib/insika/wiring/graph.rb +198 -0
- data/lib/insika/workflow.rb +185 -0
- data/lib/insika/workflow_registry.rb +33 -0
- data/lib/insika.rb +203 -4
- metadata +395 -8
|
@@ -0,0 +1,371 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "json"
|
|
4
|
+
require "time"
|
|
5
|
+
|
|
6
|
+
module Insika
|
|
7
|
+
module Refinement
|
|
8
|
+
# RFC-0013 phase A. Reads a window of an agent's real traffic and emits RANKED
|
|
9
|
+
# FINDINGS — "here is what broke, how often, and in which conversations". No
|
|
10
|
+
# model runs here and nothing is written to the agent: this is the evidence half
|
|
11
|
+
# of the loop, and it is deliberately useful on its own.
|
|
12
|
+
#
|
|
13
|
+
# It reads ONLY durable data the engine already records:
|
|
14
|
+
# TaskStore — the turn, its Command (which carries the agent) and the
|
|
15
|
+
# executions (a failed turn keeps its error)
|
|
16
|
+
# SessionStore — the transcript (repetition, canned safe replies)
|
|
17
|
+
# ToolTraceStore — per-session tool calls with ok/args/result (already masked
|
|
18
|
+
# and clipped by the store itself)
|
|
19
|
+
#
|
|
20
|
+
# Two signals of RFC-0013 §3.3 are NOT computed here, and that is a finding about
|
|
21
|
+
# the engine rather than about an agent: guardrail decisions and edge-limit hits
|
|
22
|
+
# are emitted as EVENTS and never persisted, so the only durable footprint they
|
|
23
|
+
# leave is the canned safe reply in the transcript — which is exactly what the
|
|
24
|
+
# `safe_reply` finding matches. Attributing one to a specific rule needs a write
|
|
25
|
+
# path that does not exist yet.
|
|
26
|
+
#
|
|
27
|
+
# The agent of a turn comes from the Task's Command payload, not from the
|
|
28
|
+
# Session: a Session does not stamp which agent produced it.
|
|
29
|
+
class EvidenceCollector
|
|
30
|
+
include Coercion
|
|
31
|
+
|
|
32
|
+
DEFAULT_WINDOW = 200 # distinct sessions, most recent first
|
|
33
|
+
DEFAULT_MAX_FINDINGS = 20
|
|
34
|
+
MAX_PROVENANCE = 5 # session ids kept per finding
|
|
35
|
+
SNIPPET_CHARS = 160
|
|
36
|
+
REPETITION_JACCARD = 0.6
|
|
37
|
+
REPETITION_MIN_WORDS = 3 # "oi"/"sim" repeated is not a defect
|
|
38
|
+
|
|
39
|
+
# LEGACY FALLBACK, for transcripts written before messages carried an origin.
|
|
40
|
+
#
|
|
41
|
+
# A message the engine wrote itself — an injected context fragment, a delegation
|
|
42
|
+
# result delivered as a new turn — is persisted with `role: user` like any
|
|
43
|
+
# other, because it is what the model saw. Counting those as the customer
|
|
44
|
+
# repeating themselves turned the first production run into 219 false positives,
|
|
45
|
+
# every one of them the engine reading its own `<cacau_cep_obrigatorio>` back.
|
|
46
|
+
#
|
|
47
|
+
# `MessageOrigin` is the structural answer and is preferred whenever a message
|
|
48
|
+
# carries it. This regex stays for everything written before that field existed
|
|
49
|
+
# (the pilot's database is full of it) and for consumers that have not started
|
|
50
|
+
# declaring `origin` — it is a guess, and it only ever runs on messages that
|
|
51
|
+
# made no claim about themselves.
|
|
52
|
+
INJECTED_FRAGMENT_RE = /\A<[a-z][a-z0-9_:.-]*>/i
|
|
53
|
+
|
|
54
|
+
# Weight of a finding kind when ranking (count × severity).
|
|
55
|
+
SEVERITY = {
|
|
56
|
+
tool_error: 3, task_failed: 3, safe_reply: 2, repetition: 2, tool_unused: 1
|
|
57
|
+
}.freeze
|
|
58
|
+
|
|
59
|
+
# One defect, aggregated over the window. `key` is the stable identity used to
|
|
60
|
+
# dedupe/group (and, later, for a proposal to say which finding it addresses);
|
|
61
|
+
# `sessions` is provenance — ids only, capped, never content.
|
|
62
|
+
Finding = Data.define(:kind, :key, :title, :count, :severity, :sessions, :detail)
|
|
63
|
+
|
|
64
|
+
# What a run looked at, alongside what it found. `excluded` is reported rather
|
|
65
|
+
# than swallowed: a window that quietly dropped half the traffic reads like a
|
|
66
|
+
# clean deployment.
|
|
67
|
+
Report = Data.define(:agent_id, :window, :findings, :sessions_seen, :turns_seen,
|
|
68
|
+
:excluded)
|
|
69
|
+
|
|
70
|
+
def initialize(task_store:, session_store:, tool_trace_store:, profiles:,
|
|
71
|
+
settings_store: nil)
|
|
72
|
+
@task_store = task_store
|
|
73
|
+
@session_store = session_store
|
|
74
|
+
@tool_trace_store = tool_trace_store
|
|
75
|
+
@profiles = ProfileSource.coerce(profiles)
|
|
76
|
+
@settings_store = settings_store
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
# -> Report. `since` (ISO8601) wins over `last_sessions` when both are given:
|
|
80
|
+
# an incremental run ("what happened since the last one") is the common case.
|
|
81
|
+
#
|
|
82
|
+
# `exclude_sessions` drops sessions whose id starts with any of the given
|
|
83
|
+
# prefixes. It defaults to NOTHING — a report must not decide on its own what
|
|
84
|
+
# counts as real traffic — but a deployment that replays load tests or debug
|
|
85
|
+
# conversations into the same store needs it: on the pilot, `loadtest-` sessions
|
|
86
|
+
# outnumbered real ones and drowned every genuine finding.
|
|
87
|
+
def collect(agent_id:, last_sessions: DEFAULT_WINDOW, since: nil,
|
|
88
|
+
max_findings: DEFAULT_MAX_FINDINGS, exclude_sessions: [])
|
|
89
|
+
agent = agent_id.to_s
|
|
90
|
+
profile = @profiles[agent] ||
|
|
91
|
+
(raise Insika::NotFoundError, "agent '#{agent}' not configured")
|
|
92
|
+
|
|
93
|
+
tasks, excluded = window_tasks(agent, last_sessions: last_sessions, since: since,
|
|
94
|
+
exclude_sessions: Array(exclude_sessions))
|
|
95
|
+
session_ids = tasks.filter_map { |t| presence(t.session_id) }.uniq
|
|
96
|
+
traces = session_ids.to_h { |sid| [sid, @tool_trace_store.for_session(sid)] }
|
|
97
|
+
|
|
98
|
+
findings = [
|
|
99
|
+
*tool_error_findings(traces),
|
|
100
|
+
*task_failed_findings(tasks),
|
|
101
|
+
*repetition_findings(session_ids),
|
|
102
|
+
*safe_reply_findings(session_ids, profile),
|
|
103
|
+
# An EMPTY window says nothing about a tool being unused — it says the agent
|
|
104
|
+
# did not run. Without this guard every incremental run over quiet traffic
|
|
105
|
+
# would report the whole tool list as "never called".
|
|
106
|
+
*(tasks.empty? ? [] : tool_unused_findings(profile, traces))
|
|
107
|
+
]
|
|
108
|
+
|
|
109
|
+
Report.new(
|
|
110
|
+
agent_id: agent,
|
|
111
|
+
window: since ? { "since" => since.to_s } : { "last_sessions" => last_sessions },
|
|
112
|
+
findings: rank(findings).first(max_findings),
|
|
113
|
+
sessions_seen: session_ids.size, turns_seen: tasks.size, excluded: excluded
|
|
114
|
+
)
|
|
115
|
+
end
|
|
116
|
+
|
|
117
|
+
private
|
|
118
|
+
|
|
119
|
+
# The agent's turns, most recent first, and how many were excluded.
|
|
120
|
+
# -> [[Task], Integer]. O(n) over every task: the key is a UUID so `list` cannot
|
|
121
|
+
# order by time and there is no index by agent — same trade-off
|
|
122
|
+
# TaskStore#with_status already takes (one node, local SQLite).
|
|
123
|
+
def window_tasks(agent, last_sessions:, since:, exclude_sessions:)
|
|
124
|
+
# The id breaks the tie: `created_at` has second precision, so several turns
|
|
125
|
+
# can share it and `sort_by` alone would order them arbitrarily — two runs over
|
|
126
|
+
# the same data must produce the same window.
|
|
127
|
+
all = @task_store.each_id
|
|
128
|
+
.filter_map { |id| @task_store.find(id) }
|
|
129
|
+
.select { |t| agent_of(t) == agent }
|
|
130
|
+
.sort_by { |t| [t.created_at.to_s, t.id] }.reverse
|
|
131
|
+
|
|
132
|
+
kept = exclude_sessions.empty? ? all : all.reject { |t| excluded?(t, exclude_sessions) }
|
|
133
|
+
excluded = all.size - kept.size
|
|
134
|
+
|
|
135
|
+
return [kept.select { |t| t.created_at.to_s >= since.to_s }, excluded] if presence(since)
|
|
136
|
+
|
|
137
|
+
[take_until_sessions(kept, last_sessions), excluded]
|
|
138
|
+
end
|
|
139
|
+
|
|
140
|
+
def excluded?(task, prefixes)
|
|
141
|
+
sid = task.session_id.to_s
|
|
142
|
+
prefixes.any? { |p| sid.start_with?(p.to_s) }
|
|
143
|
+
end
|
|
144
|
+
|
|
145
|
+
# Walks newest-first and stops once `limit` DISTINCT sessions have been seen —
|
|
146
|
+
# so the window is "the last N conversations", not "the last N turns".
|
|
147
|
+
def take_until_sessions(tasks, limit)
|
|
148
|
+
seen = {}
|
|
149
|
+
tasks.take_while do |task|
|
|
150
|
+
sid = presence(task.session_id)
|
|
151
|
+
seen[sid] = true if sid
|
|
152
|
+
seen.size <= limit
|
|
153
|
+
end
|
|
154
|
+
end
|
|
155
|
+
|
|
156
|
+
def agent_of(task) = task.command.is_a?(Hash) ? task.command.dig("payload", "agent").to_s : ""
|
|
157
|
+
|
|
158
|
+
# --- findings ---------------------------------------------------------------
|
|
159
|
+
|
|
160
|
+
# A tool call that came back an error. Grouped by tool + a normalized error
|
|
161
|
+
# signature, so 40 instances of the same broken argument are ONE finding with
|
|
162
|
+
# count 40 instead of 40 rows nobody reads.
|
|
163
|
+
def tool_error_findings(traces)
|
|
164
|
+
group(traces_entries(traces).reject { |_sid, e| e["ok"] }) do |_sid, entry|
|
|
165
|
+
tool = entry["tool"].to_s
|
|
166
|
+
[:tool_error, "tool_error:#{tool}:#{error_signature(entry['result'])}",
|
|
167
|
+
"#{tool} failed: #{error_signature(entry['result'])}", nil]
|
|
168
|
+
end
|
|
169
|
+
end
|
|
170
|
+
|
|
171
|
+
# A turn that died. The error is whatever the Executor recorded on the
|
|
172
|
+
# execution, grouped by its normalized message.
|
|
173
|
+
def task_failed_findings(tasks)
|
|
174
|
+
failed = tasks.flat_map do |task|
|
|
175
|
+
task.executions.filter_map do |ex|
|
|
176
|
+
next if ex.error.nil?
|
|
177
|
+
|
|
178
|
+
[presence(task.session_id), ex.error]
|
|
179
|
+
end
|
|
180
|
+
end
|
|
181
|
+
group(failed) do |_sid, error|
|
|
182
|
+
sig = signature(error_text(error))
|
|
183
|
+
[:task_failed, "task_failed:#{sig}", "turn failed: #{sig}", nil]
|
|
184
|
+
end
|
|
185
|
+
end
|
|
186
|
+
|
|
187
|
+
# The customer said the same thing AGAIN AFTER THE AGENT ANSWERED — the outside view
|
|
188
|
+
# of an instruction the agent is not following. Heuristic on purpose (token overlap,
|
|
189
|
+
# no model call); the snippet is PII-redacted.
|
|
190
|
+
#
|
|
191
|
+
# "After the agent answered" is the load-bearing half, and RFC-0015 is what forced
|
|
192
|
+
# it to be said out loud. Two customer messages in a row is now ORDINARY: `collect`
|
|
193
|
+
# merges the fragments a person types into one turn and `steer` appends one into a
|
|
194
|
+
# run in flight, so a turn legitimately holds two of them. Someone still typing is
|
|
195
|
+
# not someone repeating themselves — and a steered message cannot be told apart in
|
|
196
|
+
# the transcript, because it correctly declares no origin (§7). The structure is the
|
|
197
|
+
# only honest signal: a reply has to sit between the two.
|
|
198
|
+
def repetition_findings(session_ids)
|
|
199
|
+
hits = session_ids.flat_map do |sid|
|
|
200
|
+
repeated_after_a_reply(messages(sid)).map { |text| [sid, text] }
|
|
201
|
+
end
|
|
202
|
+
group(hits) do |_sid, text|
|
|
203
|
+
[:repetition, "repetition", "customer repeated themselves", snippet(text)]
|
|
204
|
+
end
|
|
205
|
+
end
|
|
206
|
+
|
|
207
|
+
# A canned safe reply reached the customer: a guardrail block or an edge limit.
|
|
208
|
+
# Those decisions are events, not records (see the class comment), so the reply
|
|
209
|
+
# text IS the evidence.
|
|
210
|
+
#
|
|
211
|
+
# This is the one finding that reads the engine's OWN replies, so it looks at
|
|
212
|
+
# every assistant-role message rather than the agent's (`assistant_messages`
|
|
213
|
+
# deliberately excludes them). Two ways to recognise one, and the first is now
|
|
214
|
+
# exact: a reply the engine wrote SAYS so (`origin: engine`). The canned-text
|
|
215
|
+
# match stays for transcripts written before that — and for a text match the
|
|
216
|
+
# strings are emitted verbatim, so it is an exact compare, never a prefix.
|
|
217
|
+
def safe_reply_findings(session_ids, profile)
|
|
218
|
+
canned = canned_replies(profile)
|
|
219
|
+
|
|
220
|
+
hits = session_ids.flat_map do |sid|
|
|
221
|
+
messages(sid)
|
|
222
|
+
.select { |m| m["role"].to_s == "assistant" }
|
|
223
|
+
.select { |m| MessageOrigin.origin_of(m) == MessageOrigin::ENGINE || canned.include?(m["content"].to_s.strip) }
|
|
224
|
+
.map { |m| [sid, m["content"].to_s] }
|
|
225
|
+
end
|
|
226
|
+
group(hits) do |_sid, text|
|
|
227
|
+
[:safe_reply, "safe_reply", "a canned safe reply was served instead of an answer",
|
|
228
|
+
snippet(text)]
|
|
229
|
+
end
|
|
230
|
+
end
|
|
231
|
+
|
|
232
|
+
# A tool the agent is allowed to use and never used in the whole window —
|
|
233
|
+
# either the prompt does not mention it or it answers from memory instead.
|
|
234
|
+
# Skipped when `tools_allow` is nil (= every tool; nothing to compare against).
|
|
235
|
+
def tool_unused_findings(profile, traces)
|
|
236
|
+
allowed = profile.tools_allow
|
|
237
|
+
return [] if allowed.nil?
|
|
238
|
+
|
|
239
|
+
used = traces_entries(traces).map { |_sid, e| e["tool"].to_s }.uniq
|
|
240
|
+
(Array(allowed).map(&:to_s) - used).sort.map do |tool|
|
|
241
|
+
Finding.new(kind: :tool_unused, key: "tool_unused:#{tool}",
|
|
242
|
+
title: "#{tool} was never called in this window",
|
|
243
|
+
count: 1, severity: SEVERITY[:tool_unused], sessions: [], detail: nil)
|
|
244
|
+
end
|
|
245
|
+
end
|
|
246
|
+
|
|
247
|
+
# --- helpers ----------------------------------------------------------------
|
|
248
|
+
|
|
249
|
+
def traces_entries(traces)
|
|
250
|
+
traces.flat_map { |sid, entries| Array(entries).map { |e| [sid, e] } }
|
|
251
|
+
end
|
|
252
|
+
|
|
253
|
+
# [[session_id, item], …] -> [Finding] aggregated by the key the block returns.
|
|
254
|
+
# The block gets (session_id, item) and returns [kind, key, title, detail].
|
|
255
|
+
def group(pairs)
|
|
256
|
+
pairs.each_with_object({}) do |(sid, item), acc|
|
|
257
|
+
kind, key, title, detail = yield(sid, item)
|
|
258
|
+
f = acc[key] ||= { kind: kind, key: key, title: title, detail: detail, sessions: [], count: 0 }
|
|
259
|
+
f[:count] += 1
|
|
260
|
+
f[:sessions] << sid if sid && !f[:sessions].include?(sid)
|
|
261
|
+
end.values.map do |f|
|
|
262
|
+
Finding.new(kind: f[:kind], key: f[:key], title: f[:title], count: f[:count],
|
|
263
|
+
severity: SEVERITY.fetch(f[:kind], 1),
|
|
264
|
+
sessions: f[:sessions].first(MAX_PROVENANCE), detail: f[:detail])
|
|
265
|
+
end
|
|
266
|
+
end
|
|
267
|
+
|
|
268
|
+
# count × severity, then a stable tiebreak so two runs over the same window
|
|
269
|
+
# produce the same report.
|
|
270
|
+
def rank(findings)
|
|
271
|
+
findings.sort_by { |f| [-(f.count * f.severity), f.kind.to_s, f.key] }
|
|
272
|
+
end
|
|
273
|
+
|
|
274
|
+
def messages(session_id) = Array(@session_store.find(session_id)&.messages)
|
|
275
|
+
|
|
276
|
+
# What the AGENT actually replied — not a guardrail's canned safe reply, and not
|
|
277
|
+
# a human operator's typing in an imported transcript. Both are `role: assistant`
|
|
278
|
+
# and neither is the model, so scoring them as the agent's work is wrong in both
|
|
279
|
+
# directions: it flags text no model produced, and it credits the model with a
|
|
280
|
+
# human's rescue.
|
|
281
|
+
def assistant_messages(session_id)
|
|
282
|
+
messages(session_id).select { |m| MessageOrigin.agent?(m) }.map { |m| m["content"].to_s }
|
|
283
|
+
end
|
|
284
|
+
|
|
285
|
+
# Walks a session in order and returns the customer texts that repeat something the
|
|
286
|
+
# customer had ALREADY said and the agent had already answered. Two consecutive
|
|
287
|
+
# customer messages with nothing between them are one person typing (see
|
|
288
|
+
# #repetition_findings), so they never pair.
|
|
289
|
+
def repeated_after_a_reply(list)
|
|
290
|
+
previous = nil
|
|
291
|
+
answered = false
|
|
292
|
+
|
|
293
|
+
list.each_with_object([]) do |message, hits|
|
|
294
|
+
next answered = true if MessageOrigin.agent?(message)
|
|
295
|
+
next unless customer_text?(message)
|
|
296
|
+
|
|
297
|
+
text = message["content"].to_s
|
|
298
|
+
hits << text if previous && answered && similar?(previous, text)
|
|
299
|
+
previous = text
|
|
300
|
+
answered = false
|
|
301
|
+
end
|
|
302
|
+
end
|
|
303
|
+
|
|
304
|
+
# What the CUSTOMER actually said. A message that DECLARES its origin is taken at its
|
|
305
|
+
# word; one that declares nothing falls back to the tag heuristic, which is all a
|
|
306
|
+
# pre-origin transcript offers.
|
|
307
|
+
def customer_text?(message)
|
|
308
|
+
MessageOrigin.customer?(message) &&
|
|
309
|
+
!INJECTED_FRAGMENT_RE.match?(message["content"].to_s.lstrip)
|
|
310
|
+
end
|
|
311
|
+
|
|
312
|
+
def similar?(first, second)
|
|
313
|
+
a = words(first)
|
|
314
|
+
b = words(second)
|
|
315
|
+
return false if a.size < REPETITION_MIN_WORDS || b.size < REPETITION_MIN_WORDS
|
|
316
|
+
|
|
317
|
+
union = (a | b).size
|
|
318
|
+
union.positive? && ((a & b).size.to_f / union) >= REPETITION_JACCARD
|
|
319
|
+
end
|
|
320
|
+
|
|
321
|
+
def words(text) = text.to_s.downcase.scan(/[[:alnum:]]+/).uniq
|
|
322
|
+
|
|
323
|
+
# Every reply the deployment may emit INSTEAD of a real answer: the RFC-0009
|
|
324
|
+
# defaults, the agent's own overrides, and the edge limiter's reply (per-agent
|
|
325
|
+
# first, then the platform setting).
|
|
326
|
+
def canned_replies(profile)
|
|
327
|
+
agent_overrides = (profile.guardrails || {})["responses"]
|
|
328
|
+
[
|
|
329
|
+
*Insika::Safety::SafeResponses::DEFAULTS.values,
|
|
330
|
+
*Array(agent_overrides.is_a?(Hash) ? agent_overrides.values : nil),
|
|
331
|
+
(profile.limits || {})[:limit_response],
|
|
332
|
+
@settings_store&.get&.dig("edge", "limit_response")
|
|
333
|
+
].filter_map { |t| presence(t)&.strip }.uniq
|
|
334
|
+
end
|
|
335
|
+
|
|
336
|
+
# ToolTraceStore clips `result` to a String (JSON text when it was structured),
|
|
337
|
+
# so the error has to be dug back out of it.
|
|
338
|
+
def error_signature(result)
|
|
339
|
+
parsed = begin
|
|
340
|
+
JSON.parse(result.to_s)
|
|
341
|
+
rescue StandardError
|
|
342
|
+
nil
|
|
343
|
+
end
|
|
344
|
+
signature(parsed.is_a?(Hash) ? error_text(parsed) : result)
|
|
345
|
+
end
|
|
346
|
+
|
|
347
|
+
def error_text(error)
|
|
348
|
+
return error["error"] || error["message"] || error.to_s if error.is_a?(Hash)
|
|
349
|
+
|
|
350
|
+
error.to_s
|
|
351
|
+
end
|
|
352
|
+
|
|
353
|
+
# Groups variants of the same failure: collapse whitespace, blank out numbers
|
|
354
|
+
# and ids (a "product 4711 not found" is the same defect as "product 4712"),
|
|
355
|
+
# then cap the length.
|
|
356
|
+
def signature(text)
|
|
357
|
+
s = text.to_s.gsub(/\s+/, " ").strip
|
|
358
|
+
s = s.gsub(/\b[0-9a-f]{8}(?:-[0-9a-f]{4}){3}-[0-9a-f]{12}\b/i, "<id>").gsub(/\d+/, "<n>")
|
|
359
|
+
s = "(no message)" if s.empty?
|
|
360
|
+
s.length > 80 ? "#{s[0, 80]}…" : s
|
|
361
|
+
end
|
|
362
|
+
|
|
363
|
+
# Redacted, capped excerpt — a report is read by an operator, so it may carry
|
|
364
|
+
# customer words, but never a CPF, a phone number or a token.
|
|
365
|
+
def snippet(text)
|
|
366
|
+
redacted, = Insika::Safety::Detectors.redact(text.to_s.gsub(/\s+/, " ").strip)
|
|
367
|
+
redacted.length > SNIPPET_CHARS ? "#{redacted[0, SNIPPET_CHARS]}…" : redacted
|
|
368
|
+
end
|
|
369
|
+
end
|
|
370
|
+
end
|
|
371
|
+
end
|
|
@@ -0,0 +1,234 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Insika
|
|
4
|
+
module Refinement
|
|
5
|
+
# Scores a candidate by RUNNING it (RFC-0013 §3.5). Not by asking a model whether
|
|
6
|
+
# the edit looks good — that measures nothing, and D3 says so in one line.
|
|
7
|
+
#
|
|
8
|
+
# 1. clone the agent into a throwaway id (`<agent>-cand-<run8>`)
|
|
9
|
+
# 2. copy its instruction files, apply the candidate's edits to the COPY
|
|
10
|
+
# 3. replay the agent's golden set against the clone over the ordinary public
|
|
11
|
+
# surface (`POST /v1/responses`) — real turns, real tools, real guardrails
|
|
12
|
+
# 4. compare to the accepted baseline; ANY regression disqualifies
|
|
13
|
+
# 5. delete the clone, keep the report
|
|
14
|
+
#
|
|
15
|
+
# Step 3 is what makes this expensive and what makes it worth anything. The gate
|
|
16
|
+
# is the entire safety story of refinement: everything upstream can be wrong —
|
|
17
|
+
# a hallucinated rationale, a model that misread the evidence — and the worst
|
|
18
|
+
# outcome is still a candidate that fails to improve a score and never lands.
|
|
19
|
+
#
|
|
20
|
+
# The clone is deleted in an `ensure`, including when the replay raises. A
|
|
21
|
+
# leftover `-cand-` agent is servable at `/v1/responses` by anyone who knows the
|
|
22
|
+
# id, so leaking one is a real (if obscure) exposure, not just clutter.
|
|
23
|
+
class Gate
|
|
24
|
+
# The verdict for one candidate. `passed` is the only field the caller acts on;
|
|
25
|
+
# the rest is what an operator reads to decide whether the loop earns its keep.
|
|
26
|
+
# `tokens` is what the replay SENT, cache included, as the deployment reported
|
|
27
|
+
# it — nil when no turn carried usage. `cached` is how much of that came from
|
|
28
|
+
# the prompt cache, kept separate because it is what explains one candidate
|
|
29
|
+
# costing 8× another over the same cases. `tokens` is what the panel's budget
|
|
30
|
+
# (§3.9) spends and what the operator reads on the run: a gate is the expensive
|
|
31
|
+
# half of refinement and a loop whose cost is invisible is one nobody can decide
|
|
32
|
+
# to keep.
|
|
33
|
+
Report = Data.define(:candidate_id, :passed, :reason, :cases, :passed_cases,
|
|
34
|
+
:baseline_cases, :regressions, :report, :tokens, :cached) do
|
|
35
|
+
def to_h
|
|
36
|
+
{ "candidate_id" => candidate_id, "passed" => passed, "reason" => reason,
|
|
37
|
+
"cases" => cases, "passed_cases" => passed_cases, "baseline_cases" => baseline_cases,
|
|
38
|
+
"regressions" => regressions, "report" => report, "tokens" => tokens,
|
|
39
|
+
"cached" => cached }
|
|
40
|
+
end
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
DEFAULT_TOLERANCE = 0.05
|
|
44
|
+
|
|
45
|
+
# transport_factory: -> an Evals transport for the replay. A LAMBDA and not a
|
|
46
|
+
# transport, because the gate is constructed at boot and the deployment's own
|
|
47
|
+
# URL/token are what it has to talk to; passing the built object would freeze a
|
|
48
|
+
# credential the operator can rotate.
|
|
49
|
+
# capabilities_factory: -> an `Evals::HttpCapabilities` for the clone, or nil.
|
|
50
|
+
# Without it a case whose `requires` the agent cannot satisfy RUNS and fails
|
|
51
|
+
# (RFC-0014 §3.2 says it must skip) — and then the gate and `evals/run.rb`, the
|
|
52
|
+
# two callers of the one evaluator, disagree about what the corpus even
|
|
53
|
+
# measures. §3.7 exists to prevent exactly that.
|
|
54
|
+
def initialize(profiles:, agent_files:, goldens:, baselines:, transport_factory:,
|
|
55
|
+
capabilities_factory: nil, judge_factory: nil, tolerance: DEFAULT_TOLERANCE)
|
|
56
|
+
@profiles = profiles
|
|
57
|
+
@agent_files = agent_files
|
|
58
|
+
@goldens = goldens
|
|
59
|
+
@baselines = baselines
|
|
60
|
+
@transport_factory = transport_factory
|
|
61
|
+
@capabilities_factory = capabilities_factory
|
|
62
|
+
@judge_factory = judge_factory
|
|
63
|
+
@tolerance = tolerance
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
# -> Report. Never raises for an ordinary refusal (no cases, no baseline, a
|
|
67
|
+
# replay that blew up): those are verdicts, and a run that recorded WHY it could
|
|
68
|
+
# not gate is more useful than an exception in a log.
|
|
69
|
+
def score(agent_id:, candidate:, run_id:, tolerance: nil)
|
|
70
|
+
cases = @goldens.for_agent(agent_id)
|
|
71
|
+
return refusal(candidate, "the agent has no golden cases — nothing to gate against (RFC-0013 D4)") if cases.empty?
|
|
72
|
+
|
|
73
|
+
baseline = @baselines.get(agent_id)
|
|
74
|
+
# Without an accepted state, `Baseline.compare` compares nothing and reports
|
|
75
|
+
# zero regressions — a green light meaning "we did not look". Refusing is the
|
|
76
|
+
# only honest reading, and the fix is one command.
|
|
77
|
+
if baseline.nil?
|
|
78
|
+
return refusal(candidate, "no recorded baseline for '#{agent_id}' — " \
|
|
79
|
+
"run `insika evals:baseline import` or record one before gating")
|
|
80
|
+
end
|
|
81
|
+
|
|
82
|
+
# And an ALL-RED baseline is the same hole with a record in front of it.
|
|
83
|
+
# `compare` only reports a regression against a case the baseline had
|
|
84
|
+
# PASSING, so a baseline where nothing passes cannot produce one: every
|
|
85
|
+
# candidate sails through, including a harmful one.
|
|
86
|
+
#
|
|
87
|
+
# Found by running this against a real agent: a replay that 401'd recorded a
|
|
88
|
+
# baseline of two failures, and from then on the gate accepted everything —
|
|
89
|
+
# including an edit written to be harmful. "Known-failing cases do not wedge
|
|
90
|
+
# the gate" is the right rule for the pre-merge check (a red case is work in
|
|
91
|
+
# progress, not a blocker); here it degrades into "nothing can ever fail",
|
|
92
|
+
# and a gate that cannot fail is not a gate.
|
|
93
|
+
if passing_cases(baseline).zero?
|
|
94
|
+
return refusal(candidate, "the recorded baseline for '#{agent_id}' has no PASSING case " \
|
|
95
|
+
"(#{baseline_size(baseline)} recorded, all failing) — nothing could " \
|
|
96
|
+
"regress, so every candidate would pass. Fix the agent or the cases, " \
|
|
97
|
+
"then re-record the baseline from a green run")
|
|
98
|
+
end
|
|
99
|
+
|
|
100
|
+
# And a baseline JUDGED by a rubric, replayed with no judge, is the third
|
|
101
|
+
# shape of the same hole — the one this gate actually shipped with.
|
|
102
|
+
#
|
|
103
|
+
# `CaseResult#pass?` reads a missing judge verdict as a pass (a rubric'd case
|
|
104
|
+
# is `judge_pending?`, which nothing consults), so a replay with no judge
|
|
105
|
+
# scores every rubric case as passing. Compared against a baseline recorded
|
|
106
|
+
# WITH a judge, that is not a weaker measurement, it is an inverted one:
|
|
107
|
+
# every candidate reads as an improvement.
|
|
108
|
+
#
|
|
109
|
+
# Measured, not reasoned: gating the real pilot agent with `settings["evals"]`
|
|
110
|
+
# unset reported **6/6, no regression** against a baseline the same corpus had
|
|
111
|
+
# just scored **2/6** — `produto-sem-cep` was judged 0.0 and "passed". Both
|
|
112
|
+
# candidates on the panel cleared. That is §3.7's failure exactly: the CLI and
|
|
113
|
+
# the gate, the two callers of the one evaluator, disagreeing about what the
|
|
114
|
+
# corpus measures.
|
|
115
|
+
judge = @judge_factory&.call
|
|
116
|
+
if judge.nil? && judged?(baseline)
|
|
117
|
+
return refusal(candidate, "the recorded baseline for '#{agent_id}' carries judge scores but no " \
|
|
118
|
+
"judge is configured — a rubric'd case with no verdict counts as a " \
|
|
119
|
+
"PASS, so every candidate would beat it. Configure the judge panel " \
|
|
120
|
+
"(Studio → Settings → Evals, or `settings[\"evals\"][\"judges\"]`) or " \
|
|
121
|
+
"re-record the baseline without one")
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
clone_id = clone_id_for(agent_id, run_id)
|
|
125
|
+
begin
|
|
126
|
+
build_clone(agent_id, clone_id, candidate)
|
|
127
|
+
ran = replay(cases, clone_id, judge)
|
|
128
|
+
verdict(candidate, ran, baseline, tolerance || @tolerance)
|
|
129
|
+
rescue StandardError => e
|
|
130
|
+
refusal(candidate, "gate failed to run: #{e.class}: #{e.message}")
|
|
131
|
+
ensure
|
|
132
|
+
destroy_clone(clone_id)
|
|
133
|
+
end
|
|
134
|
+
end
|
|
135
|
+
|
|
136
|
+
# `<agent>-cand-<run8>`: recognizable at a glance in the Studio's agent list and
|
|
137
|
+
# in a provider bill, and scoped to the run so two gates cannot collide.
|
|
138
|
+
def clone_id_for(agent_id, run_id) = "#{agent_id}-cand-#{run_id.to_s.delete('-')[0, 8]}"
|
|
139
|
+
|
|
140
|
+
# How many accepted cases could actually regress. This is the gate's real
|
|
141
|
+
# strength, and it is worth being able to say out loud.
|
|
142
|
+
def passing_cases(baseline)
|
|
143
|
+
(baseline["cases"] || {}).count { |_id, entry| entry.is_a?(Hash) && entry["pass"] }
|
|
144
|
+
end
|
|
145
|
+
|
|
146
|
+
# Was this baseline recorded with a judge? A single scored case is enough: it
|
|
147
|
+
# proves the accepted state was measured by a rubric the replay has to match.
|
|
148
|
+
# A baseline with no scores at all was recorded blind too, so both sides are
|
|
149
|
+
# equally deterministic and the comparison, while weak, is not inverted.
|
|
150
|
+
def judged?(baseline)
|
|
151
|
+
(baseline["cases"] || {}).any? { |_id, entry| entry.is_a?(Hash) && !entry["score"].nil? }
|
|
152
|
+
end
|
|
153
|
+
|
|
154
|
+
private
|
|
155
|
+
|
|
156
|
+
def baseline_size(baseline) = (baseline["cases"] || {}).size
|
|
157
|
+
|
|
158
|
+
# Same profile, same tools, same guardrails — only the id and the instruction
|
|
159
|
+
# files differ. Copying the profile rather than editing the real one is what
|
|
160
|
+
# makes this safe to run against production: the live agent is never touched,
|
|
161
|
+
# not even for a moment.
|
|
162
|
+
def build_clone(agent_id, clone_id, candidate)
|
|
163
|
+
profile = @profiles.fetch(agent_id) ||
|
|
164
|
+
(raise Insika::NotFoundError, "agent '#{agent_id}' not configured")
|
|
165
|
+
@profiles.put(profile.with(id: clone_id))
|
|
166
|
+
|
|
167
|
+
contents = current_files(agent_id)
|
|
168
|
+
edited = candidate.apply(contents)
|
|
169
|
+
contents.merge(edited).each { |name, body| @agent_files.write(clone_id, name, body) }
|
|
170
|
+
end
|
|
171
|
+
|
|
172
|
+
# Every file the agent has, not only the ones the candidate touches: the clone
|
|
173
|
+
# has to be the same agent for the replay to mean anything.
|
|
174
|
+
def current_files(agent_id)
|
|
175
|
+
@agent_files.list(agent_id).each_with_object({}) do |name, acc|
|
|
176
|
+
acc[name] = @agent_files.read(agent_id, name).to_s
|
|
177
|
+
end
|
|
178
|
+
end
|
|
179
|
+
|
|
180
|
+
# The goldens name the REAL agent; the replay has to address the clone. The
|
|
181
|
+
# case is otherwise untouched — same turns, same rubric, same assertions — so
|
|
182
|
+
# what is measured is the edit and nothing else.
|
|
183
|
+
#
|
|
184
|
+
# The RunCases are kept whole (not `.map(&:result)`) because the token counts
|
|
185
|
+
# ride on them, and the budget is only honest if it sees what the replay spent.
|
|
186
|
+
def replay(cases, clone_id, judge)
|
|
187
|
+
retargeted = cases.map { |g| g.class.new(**g.to_h.merge(agent: clone_id)) }
|
|
188
|
+
runner = Insika::Evals::Runner.new(transport: @transport_factory.call, judge: judge,
|
|
189
|
+
capabilities: @capabilities_factory&.call)
|
|
190
|
+
runner.run(retargeted)
|
|
191
|
+
end
|
|
192
|
+
|
|
193
|
+
def verdict(candidate, ran, baseline, tolerance)
|
|
194
|
+
results = ran.map(&:result)
|
|
195
|
+
regressions = Insika::Evals::Baseline.compare(results, baseline, tolerance: tolerance)
|
|
196
|
+
passed = results.count(&:pass?)
|
|
197
|
+
graded = results.reject(&:skipped?).size
|
|
198
|
+
spent = ran.filter_map(&:tokens)
|
|
199
|
+
cached = ran.filter_map(&:cached)
|
|
200
|
+
|
|
201
|
+
Report.new(
|
|
202
|
+
candidate_id: candidate.id, passed: regressions.empty?,
|
|
203
|
+
reason: regressions.empty? ? nil : regression_reason(regressions),
|
|
204
|
+
cases: graded, passed_cases: passed,
|
|
205
|
+
baseline_cases: (baseline["cases"] || {}).size,
|
|
206
|
+
regressions: regressions.map { |r| { "id" => r.id, "kind" => r.kind, "detail" => r.detail } },
|
|
207
|
+
report: Insika::Evals::Report.to_h(results, at: Time.now.utc.iso8601),
|
|
208
|
+
tokens: spent.empty? ? nil : spent.sum,
|
|
209
|
+
cached: cached.empty? ? nil : cached.sum
|
|
210
|
+
)
|
|
211
|
+
end
|
|
212
|
+
|
|
213
|
+
def regression_reason(regressions)
|
|
214
|
+
"#{regressions.size} regression(s): " +
|
|
215
|
+
regressions.first(3).map { |r| "#{r.id} (#{r.kind})" }.join(", ")
|
|
216
|
+
end
|
|
217
|
+
|
|
218
|
+
def refusal(candidate, reason)
|
|
219
|
+
Report.new(candidate_id: candidate.id, passed: false, reason: reason,
|
|
220
|
+
cases: 0, passed_cases: 0, baseline_cases: 0, regressions: [], report: nil,
|
|
221
|
+
tokens: nil, cached: nil)
|
|
222
|
+
end
|
|
223
|
+
|
|
224
|
+
# Both halves, both tolerant of a missing one: this runs in an `ensure` after a
|
|
225
|
+
# failure that may have happened before either was created.
|
|
226
|
+
def destroy_clone(clone_id)
|
|
227
|
+
@agent_files.list(clone_id).each { |name| @agent_files.delete(clone_id, name) }
|
|
228
|
+
@profiles.delete(clone_id)
|
|
229
|
+
rescue StandardError => e
|
|
230
|
+
warn "[refinement] could not delete the gate clone '#{clone_id}': #{e.class}: #{e.message}"
|
|
231
|
+
end
|
|
232
|
+
end
|
|
233
|
+
end
|
|
234
|
+
end
|