insika 0.0.1 → 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (260) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +295 -0
  3. data/LICENSE +21 -0
  4. data/README.md +136 -2
  5. data/bin/insika +351 -0
  6. data/docs/AGENTS.md +494 -0
  7. data/docs/ARCHITECTURE.md +333 -0
  8. data/docs/BENCHMARK.md +114 -0
  9. data/docs/CHANNELS.md +453 -0
  10. data/docs/CONTEXT.md +100 -0
  11. data/docs/DEPLOY.md +334 -0
  12. data/docs/EMBEDDING.md +194 -0
  13. data/docs/EVALS.md +273 -0
  14. data/docs/LOADTEST.md +231 -0
  15. data/docs/OBSERVABILITY.md +365 -0
  16. data/docs/PLUGINS.md +211 -0
  17. data/docs/REFINEMENT.md +477 -0
  18. data/docs/RELEASING.md +70 -0
  19. data/docs/RUNNING-LOCAL.md +153 -0
  20. data/docs/SANDBOX.md +114 -0
  21. data/docs/SECURITY.md +362 -0
  22. data/docs/SKILLS.md +98 -0
  23. data/docs/TOOLS.md +302 -0
  24. data/docs/WHY.md +137 -0
  25. data/docs/WORKFLOWS.md +225 -0
  26. data/docs/build.md +14 -0
  27. data/docs/index.md +68 -0
  28. data/docs/onboarding/start.md +126 -0
  29. data/docs/operate.md +12 -0
  30. data/docs/ship.md +10 -0
  31. data/docs/understand.md +10 -0
  32. data/lib/insika/agent_file_store.rb +125 -0
  33. data/lib/insika/agent_profile.rb +188 -0
  34. data/lib/insika/allowlist.rb +28 -0
  35. data/lib/insika/baseline_store.rb +74 -0
  36. data/lib/insika/capability/resolved_tool.rb +34 -0
  37. data/lib/insika/capability_registry.rb +112 -0
  38. data/lib/insika/channel_delivery.rb +150 -0
  39. data/lib/insika/channel_registry.rb +30 -0
  40. data/lib/insika/channels/relay.rb +178 -0
  41. data/lib/insika/channels/web/widget.js +283 -0
  42. data/lib/insika/channels/web.rb +211 -0
  43. data/lib/insika/chat_builder.rb +254 -0
  44. data/lib/insika/checkpoint.rb +13 -0
  45. data/lib/insika/checkpoint_store.rb +153 -0
  46. data/lib/insika/coercion.rb +50 -0
  47. data/lib/insika/command.rb +32 -0
  48. data/lib/insika/command_bus.rb +39 -0
  49. data/lib/insika/commands/agent_payload.rb +41 -0
  50. data/lib/insika/commands/approve_action.rb +46 -0
  51. data/lib/insika/commands/cancel_task.rb +33 -0
  52. data/lib/insika/commands/create_agent.rb +54 -0
  53. data/lib/insika/commands/create_session.rb +67 -0
  54. data/lib/insika/commands/delete_agent.rb +33 -0
  55. data/lib/insika/commands/delete_agent_file.rb +50 -0
  56. data/lib/insika/commands/delete_data_tool.rb +33 -0
  57. data/lib/insika/commands/delete_llm_provider.rb +36 -0
  58. data/lib/insika/commands/delete_mcp.rb +30 -0
  59. data/lib/insika/commands/delete_system_file.rb +29 -0
  60. data/lib/insika/commands/gate_refinement.rb +245 -0
  61. data/lib/insika/commands/import_mcp_tools.rb +48 -0
  62. data/lib/insika/commands/import_tools.rb +81 -0
  63. data/lib/insika/commands/memory_add_note.rb +32 -0
  64. data/lib/insika/commands/memory_forget_fact.rb +32 -0
  65. data/lib/insika/commands/memory_put_fact.rb +35 -0
  66. data/lib/insika/commands/pause_task.rb +29 -0
  67. data/lib/insika/commands/resolve_refinement.rb +126 -0
  68. data/lib/insika/commands/restore_agent_file.rb +36 -0
  69. data/lib/insika/commands/restore_data_tool.rb +34 -0
  70. data/lib/insika/commands/restore_system_file.rb +31 -0
  71. data/lib/insika/commands/resume_task.rb +85 -0
  72. data/lib/insika/commands/run_refinement.rb +133 -0
  73. data/lib/insika/commands/send_message.rb +150 -0
  74. data/lib/insika/commands/set_agent_tools.rb +39 -0
  75. data/lib/insika/commands/set_skill_agents.rb +71 -0
  76. data/lib/insika/commands/trigger_workflow.rb +80 -0
  77. data/lib/insika/commands/update_agent.rb +49 -0
  78. data/lib/insika/commands/update_settings.rb +33 -0
  79. data/lib/insika/commands/upsert_llm_provider.rb +34 -0
  80. data/lib/insika/commands/upsert_mcp.rb +32 -0
  81. data/lib/insika/commands/write_agent_file.rb +57 -0
  82. data/lib/insika/commands/write_data_tool.rb +43 -0
  83. data/lib/insika/commands/write_golden.rb +58 -0
  84. data/lib/insika/commands/write_skill.rb +50 -0
  85. data/lib/insika/commands/write_system_file.rb +31 -0
  86. data/lib/insika/config_store.rb +85 -0
  87. data/lib/insika/context/builder.rb +166 -0
  88. data/lib/insika/context/catalog_provider.rb +23 -0
  89. data/lib/insika/context/fragment.rb +19 -0
  90. data/lib/insika/context/priority.rb +29 -0
  91. data/lib/insika/context/provider.rb +19 -0
  92. data/lib/insika/context/providers/memory.rb +60 -0
  93. data/lib/insika/context/providers/prompt.rb +105 -0
  94. data/lib/insika/context/providers/request.rb +32 -0
  95. data/lib/insika/context/providers/session.rb +108 -0
  96. data/lib/insika/context/providers/skill.rb +20 -0
  97. data/lib/insika/context/providers/tool_search.rb +20 -0
  98. data/lib/insika/delegation_store.rb +153 -0
  99. data/lib/insika/doctor.rb +294 -0
  100. data/lib/insika/dsl/definition.rb +55 -0
  101. data/lib/insika/dsl/runtime.rb +379 -0
  102. data/lib/insika/dsl/server_boot.rb +97 -0
  103. data/lib/insika/dsl/system.rb +93 -0
  104. data/lib/insika/dsl/workflow_adapter.rb +59 -0
  105. data/lib/insika/dsl.rb +307 -0
  106. data/lib/insika/edge_limiter.rb +130 -0
  107. data/lib/insika/egress_guard.rb +75 -0
  108. data/lib/insika/env_schema.rb +246 -0
  109. data/lib/insika/errors.rb +145 -0
  110. data/lib/insika/evals/assertions.rb +247 -0
  111. data/lib/insika/evals/baseline.rb +69 -0
  112. data/lib/insika/evals/golden.rb +172 -0
  113. data/lib/insika/evals/judge.rb +225 -0
  114. data/lib/insika/evals/pairwise.rb +178 -0
  115. data/lib/insika/evals/report.rb +115 -0
  116. data/lib/insika/evals/runner.rb +141 -0
  117. data/lib/insika/evals/transport.rb +178 -0
  118. data/lib/insika/event.rb +18 -0
  119. data/lib/insika/event_stream.rb +114 -0
  120. data/lib/insika/executor.rb +1680 -0
  121. data/lib/insika/frontmatter.rb +42 -0
  122. data/lib/insika/golden_store.rb +145 -0
  123. data/lib/insika/hooks.rb +48 -0
  124. data/lib/insika/http_client.rb +63 -0
  125. data/lib/insika/inbound_log.rb +84 -0
  126. data/lib/insika/llm_configurator.rb +99 -0
  127. data/lib/insika/llm_provider_store.rb +83 -0
  128. data/lib/insika/mcp_http_client.rb +67 -0
  129. data/lib/insika/mcp_store.rb +115 -0
  130. data/lib/insika/mcp_tool_ingestor.rb +143 -0
  131. data/lib/insika/memory_store.rb +93 -0
  132. data/lib/insika/message_origin.rb +76 -0
  133. data/lib/insika/middleware.rb +36 -0
  134. data/lib/insika/model_policy.rb +52 -0
  135. data/lib/insika/model_resolver.rb +176 -0
  136. data/lib/insika/model_selection.rb +114 -0
  137. data/lib/insika/onboarding.rb +208 -0
  138. data/lib/insika/outbox_store.rb +166 -0
  139. data/lib/insika/overlay_tool_registry.rb +103 -0
  140. data/lib/insika/pack.rb +102 -0
  141. data/lib/insika/pack_importer.rb +121 -0
  142. data/lib/insika/pending_action_store.rb +120 -0
  143. data/lib/insika/plugin/loader.rb +356 -0
  144. data/lib/insika/plugin.rb +35 -0
  145. data/lib/insika/policy/engine.rb +83 -0
  146. data/lib/insika/policy/policy.rb +120 -0
  147. data/lib/insika/policy_registry.rb +23 -0
  148. data/lib/insika/profile_source.rb +137 -0
  149. data/lib/insika/prompt_catalog.rb +61 -0
  150. data/lib/insika/queue_policy.rb +167 -0
  151. data/lib/insika/recovery.rb +127 -0
  152. data/lib/insika/refinement/candidate.rb +159 -0
  153. data/lib/insika/refinement/evidence_collector.rb +371 -0
  154. data/lib/insika/refinement/gate.rb +234 -0
  155. data/lib/insika/refinement/panel.rb +222 -0
  156. data/lib/insika/refinement/proposer.rb +262 -0
  157. data/lib/insika/refinement_store.rb +295 -0
  158. data/lib/insika/registry.rb +59 -0
  159. data/lib/insika/safety/config.rb +109 -0
  160. data/lib/insika/safety/detectors.rb +176 -0
  161. data/lib/insika/safety/factory.rb +102 -0
  162. data/lib/insika/safety/input_guardrail.rb +87 -0
  163. data/lib/insika/safety/moderator.rb +86 -0
  164. data/lib/insika/safety/output_filter.rb +79 -0
  165. data/lib/insika/safety/output_validator.rb +101 -0
  166. data/lib/insika/safety/safe_responses.rb +47 -0
  167. data/lib/insika/sandbox/boundary.rb +93 -0
  168. data/lib/insika/sandbox/docker.rb +74 -0
  169. data/lib/insika/sandbox/local.rb +33 -0
  170. data/lib/insika/sandbox/runner.rb +80 -0
  171. data/lib/insika/sandbox.rb +85 -0
  172. data/lib/insika/schema_guard.rb +147 -0
  173. data/lib/insika/secret_masking.rb +34 -0
  174. data/lib/insika/server/a2a/agent_card.rb +27 -0
  175. data/lib/insika/server/a2a/app.rb +112 -0
  176. data/lib/insika/server/a2a/client.rb +101 -0
  177. data/lib/insika/server/a2a/errors.rb +32 -0
  178. data/lib/insika/server/a2a/http.rb +42 -0
  179. data/lib/insika/server/a2a/message.rb +27 -0
  180. data/lib/insika/server/a2a/protocol.rb +45 -0
  181. data/lib/insika/server/a2a/remotes.rb +25 -0
  182. data/lib/insika/server/a2a/task_projection.rb +40 -0
  183. data/lib/insika/server/admin_auth.rb +29 -0
  184. data/lib/insika/server/app.rb +850 -0
  185. data/lib/insika/server/boot.rb +119 -0
  186. data/lib/insika/server/rack_app.rb +110 -0
  187. data/lib/insika/server/responses.rb +155 -0
  188. data/lib/insika/server/sse_body.rb +96 -0
  189. data/lib/insika/session_actor.rb +162 -0
  190. data/lib/insika/session_store.rb +143 -0
  191. data/lib/insika/settings_store.rb +154 -0
  192. data/lib/insika/shutdown.rb +125 -0
  193. data/lib/insika/skill_catalog.rb +113 -0
  194. data/lib/insika/skill_store.rb +79 -0
  195. data/lib/insika/steer_injector.rb +110 -0
  196. data/lib/insika/store.rb +52 -0
  197. data/lib/insika/stores/memory.rb +123 -0
  198. data/lib/insika/stores/sqlite.rb +183 -0
  199. data/lib/insika/studio/app.rb +1571 -0
  200. data/lib/insika/studio/assets/dist/application.css +1 -0
  201. data/lib/insika/studio/assets/dist/application.js +69 -0
  202. data/lib/insika/studio/forms.rb +340 -0
  203. data/lib/insika/studio/nav_icons.rb +31 -0
  204. data/lib/insika/studio/views/_message.erb +44 -0
  205. data/lib/insika/studio/views/agent_detail.erb +285 -0
  206. data/lib/insika/studio/views/agents.erb +63 -0
  207. data/lib/insika/studio/views/approvals.erb +41 -0
  208. data/lib/insika/studio/views/chats.erb +34 -0
  209. data/lib/insika/studio/views/evals.erb +83 -0
  210. data/lib/insika/studio/views/home.erb +72 -0
  211. data/lib/insika/studio/views/layout.erb +94 -0
  212. data/lib/insika/studio/views/login.erb +17 -0
  213. data/lib/insika/studio/views/mcp.erb +91 -0
  214. data/lib/insika/studio/views/not_found.erb +5 -0
  215. data/lib/insika/studio/views/playground.erb +47 -0
  216. data/lib/insika/studio/views/refinement.erb +234 -0
  217. data/lib/insika/studio/views/session.erb +62 -0
  218. data/lib/insika/studio/views/settings.erb +173 -0
  219. data/lib/insika/studio/views/skills.erb +86 -0
  220. data/lib/insika/studio/views/system_files.erb +65 -0
  221. data/lib/insika/studio/views/task.erb +105 -0
  222. data/lib/insika/studio/views/tasks.erb +33 -0
  223. data/lib/insika/studio/views/tool_edit.erb +107 -0
  224. data/lib/insika/studio/views/tools.erb +89 -0
  225. data/lib/insika/subagent_graph.rb +96 -0
  226. data/lib/insika/system_file_store.rb +96 -0
  227. data/lib/insika/task_actor.rb +128 -0
  228. data/lib/insika/task_store.rb +250 -0
  229. data/lib/insika/telemetry/pricing.rb +104 -0
  230. data/lib/insika/telemetry/recorder.rb +228 -0
  231. data/lib/insika/telemetry.rb +127 -0
  232. data/lib/insika/testing/store_contract.rb +270 -0
  233. data/lib/insika/token_estimator.rb +16 -0
  234. data/lib/insika/tool_assembly.rb +140 -0
  235. data/lib/insika/tool_catalog.rb +89 -0
  236. data/lib/insika/tool_definition.rb +518 -0
  237. data/lib/insika/tool_envelope.rb +140 -0
  238. data/lib/insika/tool_manifest.rb +218 -0
  239. data/lib/insika/tool_registry.rb +21 -0
  240. data/lib/insika/tool_store.rb +135 -0
  241. data/lib/insika/tool_trace_store.rb +92 -0
  242. data/lib/insika/tools/a2a_remote.rb +48 -0
  243. data/lib/insika/tools/agent_enum.rb +68 -0
  244. data/lib/insika/tools/concurrency.rb +54 -0
  245. data/lib/insika/tools/data_defined_tool.rb +220 -0
  246. data/lib/insika/tools/load_skill.rb +41 -0
  247. data/lib/insika/tools/remember.rb +53 -0
  248. data/lib/insika/tools/subagent.rb +75 -0
  249. data/lib/insika/tools/subagents.rb +77 -0
  250. data/lib/insika/tools/tool_search.rb +94 -0
  251. data/lib/insika/turn_output.rb +139 -0
  252. data/lib/insika/turn_state.rb +158 -0
  253. data/lib/insika/turn_timing.rb +56 -0
  254. data/lib/insika/usage_ledger.rb +47 -0
  255. data/lib/insika/version.rb +3 -1
  256. data/lib/insika/wiring/graph.rb +198 -0
  257. data/lib/insika/workflow.rb +185 -0
  258. data/lib/insika/workflow_registry.rb +33 -0
  259. data/lib/insika.rb +203 -4
  260. metadata +395 -8
@@ -0,0 +1,371 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "json"
4
+ require "time"
5
+
6
+ module Insika
7
+ module Refinement
8
+ # RFC-0013 phase A. Reads a window of an agent's real traffic and emits RANKED
9
+ # FINDINGS — "here is what broke, how often, and in which conversations". No
10
+ # model runs here and nothing is written to the agent: this is the evidence half
11
+ # of the loop, and it is deliberately useful on its own.
12
+ #
13
+ # It reads ONLY durable data the engine already records:
14
+ # TaskStore — the turn, its Command (which carries the agent) and the
15
+ # executions (a failed turn keeps its error)
16
+ # SessionStore — the transcript (repetition, canned safe replies)
17
+ # ToolTraceStore — per-session tool calls with ok/args/result (already masked
18
+ # and clipped by the store itself)
19
+ #
20
+ # Two signals of RFC-0013 §3.3 are NOT computed here, and that is a finding about
21
+ # the engine rather than about an agent: guardrail decisions and edge-limit hits
22
+ # are emitted as EVENTS and never persisted, so the only durable footprint they
23
+ # leave is the canned safe reply in the transcript — which is exactly what the
24
+ # `safe_reply` finding matches. Attributing one to a specific rule needs a write
25
+ # path that does not exist yet.
26
+ #
27
+ # The agent of a turn comes from the Task's Command payload, not from the
28
+ # Session: a Session does not stamp which agent produced it.
29
+ class EvidenceCollector
30
+ include Coercion
31
+
32
+ DEFAULT_WINDOW = 200 # distinct sessions, most recent first
33
+ DEFAULT_MAX_FINDINGS = 20
34
+ MAX_PROVENANCE = 5 # session ids kept per finding
35
+ SNIPPET_CHARS = 160
36
+ REPETITION_JACCARD = 0.6
37
+ REPETITION_MIN_WORDS = 3 # "oi"/"sim" repeated is not a defect
38
+
39
+ # LEGACY FALLBACK, for transcripts written before messages carried an origin.
40
+ #
41
+ # A message the engine wrote itself — an injected context fragment, a delegation
42
+ # result delivered as a new turn — is persisted with `role: user` like any
43
+ # other, because it is what the model saw. Counting those as the customer
44
+ # repeating themselves turned the first production run into 219 false positives,
45
+ # every one of them the engine reading its own `<cacau_cep_obrigatorio>` back.
46
+ #
47
+ # `MessageOrigin` is the structural answer and is preferred whenever a message
48
+ # carries it. This regex stays for everything written before that field existed
49
+ # (the pilot's database is full of it) and for consumers that have not started
50
+ # declaring `origin` — it is a guess, and it only ever runs on messages that
51
+ # made no claim about themselves.
52
+ INJECTED_FRAGMENT_RE = /\A<[a-z][a-z0-9_:.-]*>/i
53
+
54
+ # Weight of a finding kind when ranking (count × severity).
55
+ SEVERITY = {
56
+ tool_error: 3, task_failed: 3, safe_reply: 2, repetition: 2, tool_unused: 1
57
+ }.freeze
58
+
59
+ # One defect, aggregated over the window. `key` is the stable identity used to
60
+ # dedupe/group (and, later, for a proposal to say which finding it addresses);
61
+ # `sessions` is provenance — ids only, capped, never content.
62
+ Finding = Data.define(:kind, :key, :title, :count, :severity, :sessions, :detail)
63
+
64
+ # What a run looked at, alongside what it found. `excluded` is reported rather
65
+ # than swallowed: a window that quietly dropped half the traffic reads like a
66
+ # clean deployment.
67
+ Report = Data.define(:agent_id, :window, :findings, :sessions_seen, :turns_seen,
68
+ :excluded)
69
+
70
+ def initialize(task_store:, session_store:, tool_trace_store:, profiles:,
71
+ settings_store: nil)
72
+ @task_store = task_store
73
+ @session_store = session_store
74
+ @tool_trace_store = tool_trace_store
75
+ @profiles = ProfileSource.coerce(profiles)
76
+ @settings_store = settings_store
77
+ end
78
+
79
+ # -> Report. `since` (ISO8601) wins over `last_sessions` when both are given:
80
+ # an incremental run ("what happened since the last one") is the common case.
81
+ #
82
+ # `exclude_sessions` drops sessions whose id starts with any of the given
83
+ # prefixes. It defaults to NOTHING — a report must not decide on its own what
84
+ # counts as real traffic — but a deployment that replays load tests or debug
85
+ # conversations into the same store needs it: on the pilot, `loadtest-` sessions
86
+ # outnumbered real ones and drowned every genuine finding.
87
+ def collect(agent_id:, last_sessions: DEFAULT_WINDOW, since: nil,
88
+ max_findings: DEFAULT_MAX_FINDINGS, exclude_sessions: [])
89
+ agent = agent_id.to_s
90
+ profile = @profiles[agent] ||
91
+ (raise Insika::NotFoundError, "agent '#{agent}' not configured")
92
+
93
+ tasks, excluded = window_tasks(agent, last_sessions: last_sessions, since: since,
94
+ exclude_sessions: Array(exclude_sessions))
95
+ session_ids = tasks.filter_map { |t| presence(t.session_id) }.uniq
96
+ traces = session_ids.to_h { |sid| [sid, @tool_trace_store.for_session(sid)] }
97
+
98
+ findings = [
99
+ *tool_error_findings(traces),
100
+ *task_failed_findings(tasks),
101
+ *repetition_findings(session_ids),
102
+ *safe_reply_findings(session_ids, profile),
103
+ # An EMPTY window says nothing about a tool being unused — it says the agent
104
+ # did not run. Without this guard every incremental run over quiet traffic
105
+ # would report the whole tool list as "never called".
106
+ *(tasks.empty? ? [] : tool_unused_findings(profile, traces))
107
+ ]
108
+
109
+ Report.new(
110
+ agent_id: agent,
111
+ window: since ? { "since" => since.to_s } : { "last_sessions" => last_sessions },
112
+ findings: rank(findings).first(max_findings),
113
+ sessions_seen: session_ids.size, turns_seen: tasks.size, excluded: excluded
114
+ )
115
+ end
116
+
117
+ private
118
+
119
+ # The agent's turns, most recent first, and how many were excluded.
120
+ # -> [[Task], Integer]. O(n) over every task: the key is a UUID so `list` cannot
121
+ # order by time and there is no index by agent — same trade-off
122
+ # TaskStore#with_status already takes (one node, local SQLite).
123
+ def window_tasks(agent, last_sessions:, since:, exclude_sessions:)
124
+ # The id breaks the tie: `created_at` has second precision, so several turns
125
+ # can share it and `sort_by` alone would order them arbitrarily — two runs over
126
+ # the same data must produce the same window.
127
+ all = @task_store.each_id
128
+ .filter_map { |id| @task_store.find(id) }
129
+ .select { |t| agent_of(t) == agent }
130
+ .sort_by { |t| [t.created_at.to_s, t.id] }.reverse
131
+
132
+ kept = exclude_sessions.empty? ? all : all.reject { |t| excluded?(t, exclude_sessions) }
133
+ excluded = all.size - kept.size
134
+
135
+ return [kept.select { |t| t.created_at.to_s >= since.to_s }, excluded] if presence(since)
136
+
137
+ [take_until_sessions(kept, last_sessions), excluded]
138
+ end
139
+
140
+ def excluded?(task, prefixes)
141
+ sid = task.session_id.to_s
142
+ prefixes.any? { |p| sid.start_with?(p.to_s) }
143
+ end
144
+
145
+ # Walks newest-first and stops once `limit` DISTINCT sessions have been seen —
146
+ # so the window is "the last N conversations", not "the last N turns".
147
+ def take_until_sessions(tasks, limit)
148
+ seen = {}
149
+ tasks.take_while do |task|
150
+ sid = presence(task.session_id)
151
+ seen[sid] = true if sid
152
+ seen.size <= limit
153
+ end
154
+ end
155
+
156
+ def agent_of(task) = task.command.is_a?(Hash) ? task.command.dig("payload", "agent").to_s : ""
157
+
158
+ # --- findings ---------------------------------------------------------------
159
+
160
+ # A tool call that came back an error. Grouped by tool + a normalized error
161
+ # signature, so 40 instances of the same broken argument are ONE finding with
162
+ # count 40 instead of 40 rows nobody reads.
163
+ def tool_error_findings(traces)
164
+ group(traces_entries(traces).reject { |_sid, e| e["ok"] }) do |_sid, entry|
165
+ tool = entry["tool"].to_s
166
+ [:tool_error, "tool_error:#{tool}:#{error_signature(entry['result'])}",
167
+ "#{tool} failed: #{error_signature(entry['result'])}", nil]
168
+ end
169
+ end
170
+
171
+ # A turn that died. The error is whatever the Executor recorded on the
172
+ # execution, grouped by its normalized message.
173
+ def task_failed_findings(tasks)
174
+ failed = tasks.flat_map do |task|
175
+ task.executions.filter_map do |ex|
176
+ next if ex.error.nil?
177
+
178
+ [presence(task.session_id), ex.error]
179
+ end
180
+ end
181
+ group(failed) do |_sid, error|
182
+ sig = signature(error_text(error))
183
+ [:task_failed, "task_failed:#{sig}", "turn failed: #{sig}", nil]
184
+ end
185
+ end
186
+
187
+ # The customer said the same thing AGAIN AFTER THE AGENT ANSWERED — the outside view
188
+ # of an instruction the agent is not following. Heuristic on purpose (token overlap,
189
+ # no model call); the snippet is PII-redacted.
190
+ #
191
+ # "After the agent answered" is the load-bearing half, and RFC-0015 is what forced
192
+ # it to be said out loud. Two customer messages in a row is now ORDINARY: `collect`
193
+ # merges the fragments a person types into one turn and `steer` appends one into a
194
+ # run in flight, so a turn legitimately holds two of them. Someone still typing is
195
+ # not someone repeating themselves — and a steered message cannot be told apart in
196
+ # the transcript, because it correctly declares no origin (§7). The structure is the
197
+ # only honest signal: a reply has to sit between the two.
198
+ def repetition_findings(session_ids)
199
+ hits = session_ids.flat_map do |sid|
200
+ repeated_after_a_reply(messages(sid)).map { |text| [sid, text] }
201
+ end
202
+ group(hits) do |_sid, text|
203
+ [:repetition, "repetition", "customer repeated themselves", snippet(text)]
204
+ end
205
+ end
206
+
207
+ # A canned safe reply reached the customer: a guardrail block or an edge limit.
208
+ # Those decisions are events, not records (see the class comment), so the reply
209
+ # text IS the evidence.
210
+ #
211
+ # This is the one finding that reads the engine's OWN replies, so it looks at
212
+ # every assistant-role message rather than the agent's (`assistant_messages`
213
+ # deliberately excludes them). Two ways to recognise one, and the first is now
214
+ # exact: a reply the engine wrote SAYS so (`origin: engine`). The canned-text
215
+ # match stays for transcripts written before that — and for a text match the
216
+ # strings are emitted verbatim, so it is an exact compare, never a prefix.
217
+ def safe_reply_findings(session_ids, profile)
218
+ canned = canned_replies(profile)
219
+
220
+ hits = session_ids.flat_map do |sid|
221
+ messages(sid)
222
+ .select { |m| m["role"].to_s == "assistant" }
223
+ .select { |m| MessageOrigin.origin_of(m) == MessageOrigin::ENGINE || canned.include?(m["content"].to_s.strip) }
224
+ .map { |m| [sid, m["content"].to_s] }
225
+ end
226
+ group(hits) do |_sid, text|
227
+ [:safe_reply, "safe_reply", "a canned safe reply was served instead of an answer",
228
+ snippet(text)]
229
+ end
230
+ end
231
+
232
+ # A tool the agent is allowed to use and never used in the whole window —
233
+ # either the prompt does not mention it or it answers from memory instead.
234
+ # Skipped when `tools_allow` is nil (= every tool; nothing to compare against).
235
+ def tool_unused_findings(profile, traces)
236
+ allowed = profile.tools_allow
237
+ return [] if allowed.nil?
238
+
239
+ used = traces_entries(traces).map { |_sid, e| e["tool"].to_s }.uniq
240
+ (Array(allowed).map(&:to_s) - used).sort.map do |tool|
241
+ Finding.new(kind: :tool_unused, key: "tool_unused:#{tool}",
242
+ title: "#{tool} was never called in this window",
243
+ count: 1, severity: SEVERITY[:tool_unused], sessions: [], detail: nil)
244
+ end
245
+ end
246
+
247
+ # --- helpers ----------------------------------------------------------------
248
+
249
+ def traces_entries(traces)
250
+ traces.flat_map { |sid, entries| Array(entries).map { |e| [sid, e] } }
251
+ end
252
+
253
+ # [[session_id, item], …] -> [Finding] aggregated by the key the block returns.
254
+ # The block gets (session_id, item) and returns [kind, key, title, detail].
255
+ def group(pairs)
256
+ pairs.each_with_object({}) do |(sid, item), acc|
257
+ kind, key, title, detail = yield(sid, item)
258
+ f = acc[key] ||= { kind: kind, key: key, title: title, detail: detail, sessions: [], count: 0 }
259
+ f[:count] += 1
260
+ f[:sessions] << sid if sid && !f[:sessions].include?(sid)
261
+ end.values.map do |f|
262
+ Finding.new(kind: f[:kind], key: f[:key], title: f[:title], count: f[:count],
263
+ severity: SEVERITY.fetch(f[:kind], 1),
264
+ sessions: f[:sessions].first(MAX_PROVENANCE), detail: f[:detail])
265
+ end
266
+ end
267
+
268
+ # count × severity, then a stable tiebreak so two runs over the same window
269
+ # produce the same report.
270
+ def rank(findings)
271
+ findings.sort_by { |f| [-(f.count * f.severity), f.kind.to_s, f.key] }
272
+ end
273
+
274
+ def messages(session_id) = Array(@session_store.find(session_id)&.messages)
275
+
276
+ # What the AGENT actually replied — not a guardrail's canned safe reply, and not
277
+ # a human operator's typing in an imported transcript. Both are `role: assistant`
278
+ # and neither is the model, so scoring them as the agent's work is wrong in both
279
+ # directions: it flags text no model produced, and it credits the model with a
280
+ # human's rescue.
281
+ def assistant_messages(session_id)
282
+ messages(session_id).select { |m| MessageOrigin.agent?(m) }.map { |m| m["content"].to_s }
283
+ end
284
+
285
+ # Walks a session in order and returns the customer texts that repeat something the
286
+ # customer had ALREADY said and the agent had already answered. Two consecutive
287
+ # customer messages with nothing between them are one person typing (see
288
+ # #repetition_findings), so they never pair.
289
+ def repeated_after_a_reply(list)
290
+ previous = nil
291
+ answered = false
292
+
293
+ list.each_with_object([]) do |message, hits|
294
+ next answered = true if MessageOrigin.agent?(message)
295
+ next unless customer_text?(message)
296
+
297
+ text = message["content"].to_s
298
+ hits << text if previous && answered && similar?(previous, text)
299
+ previous = text
300
+ answered = false
301
+ end
302
+ end
303
+
304
+ # What the CUSTOMER actually said. A message that DECLARES its origin is taken at its
305
+ # word; one that declares nothing falls back to the tag heuristic, which is all a
306
+ # pre-origin transcript offers.
307
+ def customer_text?(message)
308
+ MessageOrigin.customer?(message) &&
309
+ !INJECTED_FRAGMENT_RE.match?(message["content"].to_s.lstrip)
310
+ end
311
+
312
+ def similar?(first, second)
313
+ a = words(first)
314
+ b = words(second)
315
+ return false if a.size < REPETITION_MIN_WORDS || b.size < REPETITION_MIN_WORDS
316
+
317
+ union = (a | b).size
318
+ union.positive? && ((a & b).size.to_f / union) >= REPETITION_JACCARD
319
+ end
320
+
321
+ def words(text) = text.to_s.downcase.scan(/[[:alnum:]]+/).uniq
322
+
323
+ # Every reply the deployment may emit INSTEAD of a real answer: the RFC-0009
324
+ # defaults, the agent's own overrides, and the edge limiter's reply (per-agent
325
+ # first, then the platform setting).
326
+ def canned_replies(profile)
327
+ agent_overrides = (profile.guardrails || {})["responses"]
328
+ [
329
+ *Insika::Safety::SafeResponses::DEFAULTS.values,
330
+ *Array(agent_overrides.is_a?(Hash) ? agent_overrides.values : nil),
331
+ (profile.limits || {})[:limit_response],
332
+ @settings_store&.get&.dig("edge", "limit_response")
333
+ ].filter_map { |t| presence(t)&.strip }.uniq
334
+ end
335
+
336
+ # ToolTraceStore clips `result` to a String (JSON text when it was structured),
337
+ # so the error has to be dug back out of it.
338
+ def error_signature(result)
339
+ parsed = begin
340
+ JSON.parse(result.to_s)
341
+ rescue StandardError
342
+ nil
343
+ end
344
+ signature(parsed.is_a?(Hash) ? error_text(parsed) : result)
345
+ end
346
+
347
+ def error_text(error)
348
+ return error["error"] || error["message"] || error.to_s if error.is_a?(Hash)
349
+
350
+ error.to_s
351
+ end
352
+
353
+ # Groups variants of the same failure: collapse whitespace, blank out numbers
354
+ # and ids (a "product 4711 not found" is the same defect as "product 4712"),
355
+ # then cap the length.
356
+ def signature(text)
357
+ s = text.to_s.gsub(/\s+/, " ").strip
358
+ s = s.gsub(/\b[0-9a-f]{8}(?:-[0-9a-f]{4}){3}-[0-9a-f]{12}\b/i, "<id>").gsub(/\d+/, "<n>")
359
+ s = "(no message)" if s.empty?
360
+ s.length > 80 ? "#{s[0, 80]}…" : s
361
+ end
362
+
363
+ # Redacted, capped excerpt — a report is read by an operator, so it may carry
364
+ # customer words, but never a CPF, a phone number or a token.
365
+ def snippet(text)
366
+ redacted, = Insika::Safety::Detectors.redact(text.to_s.gsub(/\s+/, " ").strip)
367
+ redacted.length > SNIPPET_CHARS ? "#{redacted[0, SNIPPET_CHARS]}…" : redacted
368
+ end
369
+ end
370
+ end
371
+ end
@@ -0,0 +1,234 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Insika
4
+ module Refinement
5
+ # Scores a candidate by RUNNING it (RFC-0013 §3.5). Not by asking a model whether
6
+ # the edit looks good — that measures nothing, and D3 says so in one line.
7
+ #
8
+ # 1. clone the agent into a throwaway id (`<agent>-cand-<run8>`)
9
+ # 2. copy its instruction files, apply the candidate's edits to the COPY
10
+ # 3. replay the agent's golden set against the clone over the ordinary public
11
+ # surface (`POST /v1/responses`) — real turns, real tools, real guardrails
12
+ # 4. compare to the accepted baseline; ANY regression disqualifies
13
+ # 5. delete the clone, keep the report
14
+ #
15
+ # Step 3 is what makes this expensive and what makes it worth anything. The gate
16
+ # is the entire safety story of refinement: everything upstream can be wrong —
17
+ # a hallucinated rationale, a model that misread the evidence — and the worst
18
+ # outcome is still a candidate that fails to improve a score and never lands.
19
+ #
20
+ # The clone is deleted in an `ensure`, including when the replay raises. A
21
+ # leftover `-cand-` agent is servable at `/v1/responses` by anyone who knows the
22
+ # id, so leaking one is a real (if obscure) exposure, not just clutter.
23
+ class Gate
24
+ # The verdict for one candidate. `passed` is the only field the caller acts on;
25
+ # the rest is what an operator reads to decide whether the loop earns its keep.
26
+ # `tokens` is what the replay SENT, cache included, as the deployment reported
27
+ # it — nil when no turn carried usage. `cached` is how much of that came from
28
+ # the prompt cache, kept separate because it is what explains one candidate
29
+ # costing 8× another over the same cases. `tokens` is what the panel's budget
30
+ # (§3.9) spends and what the operator reads on the run: a gate is the expensive
31
+ # half of refinement and a loop whose cost is invisible is one nobody can decide
32
+ # to keep.
33
+ Report = Data.define(:candidate_id, :passed, :reason, :cases, :passed_cases,
34
+ :baseline_cases, :regressions, :report, :tokens, :cached) do
35
+ def to_h
36
+ { "candidate_id" => candidate_id, "passed" => passed, "reason" => reason,
37
+ "cases" => cases, "passed_cases" => passed_cases, "baseline_cases" => baseline_cases,
38
+ "regressions" => regressions, "report" => report, "tokens" => tokens,
39
+ "cached" => cached }
40
+ end
41
+ end
42
+
43
+ DEFAULT_TOLERANCE = 0.05
44
+
45
+ # transport_factory: -> an Evals transport for the replay. A LAMBDA and not a
46
+ # transport, because the gate is constructed at boot and the deployment's own
47
+ # URL/token are what it has to talk to; passing the built object would freeze a
48
+ # credential the operator can rotate.
49
+ # capabilities_factory: -> an `Evals::HttpCapabilities` for the clone, or nil.
50
+ # Without it a case whose `requires` the agent cannot satisfy RUNS and fails
51
+ # (RFC-0014 §3.2 says it must skip) — and then the gate and `evals/run.rb`, the
52
+ # two callers of the one evaluator, disagree about what the corpus even
53
+ # measures. §3.7 exists to prevent exactly that.
54
+ def initialize(profiles:, agent_files:, goldens:, baselines:, transport_factory:,
55
+ capabilities_factory: nil, judge_factory: nil, tolerance: DEFAULT_TOLERANCE)
56
+ @profiles = profiles
57
+ @agent_files = agent_files
58
+ @goldens = goldens
59
+ @baselines = baselines
60
+ @transport_factory = transport_factory
61
+ @capabilities_factory = capabilities_factory
62
+ @judge_factory = judge_factory
63
+ @tolerance = tolerance
64
+ end
65
+
66
+ # -> Report. Never raises for an ordinary refusal (no cases, no baseline, a
67
+ # replay that blew up): those are verdicts, and a run that recorded WHY it could
68
+ # not gate is more useful than an exception in a log.
69
+ def score(agent_id:, candidate:, run_id:, tolerance: nil)
70
+ cases = @goldens.for_agent(agent_id)
71
+ return refusal(candidate, "the agent has no golden cases — nothing to gate against (RFC-0013 D4)") if cases.empty?
72
+
73
+ baseline = @baselines.get(agent_id)
74
+ # Without an accepted state, `Baseline.compare` compares nothing and reports
75
+ # zero regressions — a green light meaning "we did not look". Refusing is the
76
+ # only honest reading, and the fix is one command.
77
+ if baseline.nil?
78
+ return refusal(candidate, "no recorded baseline for '#{agent_id}' — " \
79
+ "run `insika evals:baseline import` or record one before gating")
80
+ end
81
+
82
+ # And an ALL-RED baseline is the same hole with a record in front of it.
83
+ # `compare` only reports a regression against a case the baseline had
84
+ # PASSING, so a baseline where nothing passes cannot produce one: every
85
+ # candidate sails through, including a harmful one.
86
+ #
87
+ # Found by running this against a real agent: a replay that 401'd recorded a
88
+ # baseline of two failures, and from then on the gate accepted everything —
89
+ # including an edit written to be harmful. "Known-failing cases do not wedge
90
+ # the gate" is the right rule for the pre-merge check (a red case is work in
91
+ # progress, not a blocker); here it degrades into "nothing can ever fail",
92
+ # and a gate that cannot fail is not a gate.
93
+ if passing_cases(baseline).zero?
94
+ return refusal(candidate, "the recorded baseline for '#{agent_id}' has no PASSING case " \
95
+ "(#{baseline_size(baseline)} recorded, all failing) — nothing could " \
96
+ "regress, so every candidate would pass. Fix the agent or the cases, " \
97
+ "then re-record the baseline from a green run")
98
+ end
99
+
100
+ # And a baseline JUDGED by a rubric, replayed with no judge, is the third
101
+ # shape of the same hole — the one this gate actually shipped with.
102
+ #
103
+ # `CaseResult#pass?` reads a missing judge verdict as a pass (a rubric'd case
104
+ # is `judge_pending?`, which nothing consults), so a replay with no judge
105
+ # scores every rubric case as passing. Compared against a baseline recorded
106
+ # WITH a judge, that is not a weaker measurement, it is an inverted one:
107
+ # every candidate reads as an improvement.
108
+ #
109
+ # Measured, not reasoned: gating the real pilot agent with `settings["evals"]`
110
+ # unset reported **6/6, no regression** against a baseline the same corpus had
111
+ # just scored **2/6** — `produto-sem-cep` was judged 0.0 and "passed". Both
112
+ # candidates on the panel cleared. That is §3.7's failure exactly: the CLI and
113
+ # the gate, the two callers of the one evaluator, disagreeing about what the
114
+ # corpus measures.
115
+ judge = @judge_factory&.call
116
+ if judge.nil? && judged?(baseline)
117
+ return refusal(candidate, "the recorded baseline for '#{agent_id}' carries judge scores but no " \
118
+ "judge is configured — a rubric'd case with no verdict counts as a " \
119
+ "PASS, so every candidate would beat it. Configure the judge panel " \
120
+ "(Studio → Settings → Evals, or `settings[\"evals\"][\"judges\"]`) or " \
121
+ "re-record the baseline without one")
122
+ end
123
+
124
+ clone_id = clone_id_for(agent_id, run_id)
125
+ begin
126
+ build_clone(agent_id, clone_id, candidate)
127
+ ran = replay(cases, clone_id, judge)
128
+ verdict(candidate, ran, baseline, tolerance || @tolerance)
129
+ rescue StandardError => e
130
+ refusal(candidate, "gate failed to run: #{e.class}: #{e.message}")
131
+ ensure
132
+ destroy_clone(clone_id)
133
+ end
134
+ end
135
+
136
+ # `<agent>-cand-<run8>`: recognizable at a glance in the Studio's agent list and
137
+ # in a provider bill, and scoped to the run so two gates cannot collide.
138
+ def clone_id_for(agent_id, run_id) = "#{agent_id}-cand-#{run_id.to_s.delete('-')[0, 8]}"
139
+
140
+ # How many accepted cases could actually regress. This is the gate's real
141
+ # strength, and it is worth being able to say out loud.
142
+ def passing_cases(baseline)
143
+ (baseline["cases"] || {}).count { |_id, entry| entry.is_a?(Hash) && entry["pass"] }
144
+ end
145
+
146
+ # Was this baseline recorded with a judge? A single scored case is enough: it
147
+ # proves the accepted state was measured by a rubric the replay has to match.
148
+ # A baseline with no scores at all was recorded blind too, so both sides are
149
+ # equally deterministic and the comparison, while weak, is not inverted.
150
+ def judged?(baseline)
151
+ (baseline["cases"] || {}).any? { |_id, entry| entry.is_a?(Hash) && !entry["score"].nil? }
152
+ end
153
+
154
+ private
155
+
156
+ def baseline_size(baseline) = (baseline["cases"] || {}).size
157
+
158
+ # Same profile, same tools, same guardrails — only the id and the instruction
159
+ # files differ. Copying the profile rather than editing the real one is what
160
+ # makes this safe to run against production: the live agent is never touched,
161
+ # not even for a moment.
162
+ def build_clone(agent_id, clone_id, candidate)
163
+ profile = @profiles.fetch(agent_id) ||
164
+ (raise Insika::NotFoundError, "agent '#{agent_id}' not configured")
165
+ @profiles.put(profile.with(id: clone_id))
166
+
167
+ contents = current_files(agent_id)
168
+ edited = candidate.apply(contents)
169
+ contents.merge(edited).each { |name, body| @agent_files.write(clone_id, name, body) }
170
+ end
171
+
172
+ # Every file the agent has, not only the ones the candidate touches: the clone
173
+ # has to be the same agent for the replay to mean anything.
174
+ def current_files(agent_id)
175
+ @agent_files.list(agent_id).each_with_object({}) do |name, acc|
176
+ acc[name] = @agent_files.read(agent_id, name).to_s
177
+ end
178
+ end
179
+
180
+ # The goldens name the REAL agent; the replay has to address the clone. The
181
+ # case is otherwise untouched — same turns, same rubric, same assertions — so
182
+ # what is measured is the edit and nothing else.
183
+ #
184
+ # The RunCases are kept whole (not `.map(&:result)`) because the token counts
185
+ # ride on them, and the budget is only honest if it sees what the replay spent.
186
+ def replay(cases, clone_id, judge)
187
+ retargeted = cases.map { |g| g.class.new(**g.to_h.merge(agent: clone_id)) }
188
+ runner = Insika::Evals::Runner.new(transport: @transport_factory.call, judge: judge,
189
+ capabilities: @capabilities_factory&.call)
190
+ runner.run(retargeted)
191
+ end
192
+
193
+ def verdict(candidate, ran, baseline, tolerance)
194
+ results = ran.map(&:result)
195
+ regressions = Insika::Evals::Baseline.compare(results, baseline, tolerance: tolerance)
196
+ passed = results.count(&:pass?)
197
+ graded = results.reject(&:skipped?).size
198
+ spent = ran.filter_map(&:tokens)
199
+ cached = ran.filter_map(&:cached)
200
+
201
+ Report.new(
202
+ candidate_id: candidate.id, passed: regressions.empty?,
203
+ reason: regressions.empty? ? nil : regression_reason(regressions),
204
+ cases: graded, passed_cases: passed,
205
+ baseline_cases: (baseline["cases"] || {}).size,
206
+ regressions: regressions.map { |r| { "id" => r.id, "kind" => r.kind, "detail" => r.detail } },
207
+ report: Insika::Evals::Report.to_h(results, at: Time.now.utc.iso8601),
208
+ tokens: spent.empty? ? nil : spent.sum,
209
+ cached: cached.empty? ? nil : cached.sum
210
+ )
211
+ end
212
+
213
+ def regression_reason(regressions)
214
+ "#{regressions.size} regression(s): " +
215
+ regressions.first(3).map { |r| "#{r.id} (#{r.kind})" }.join(", ")
216
+ end
217
+
218
+ def refusal(candidate, reason)
219
+ Report.new(candidate_id: candidate.id, passed: false, reason: reason,
220
+ cases: 0, passed_cases: 0, baseline_cases: 0, regressions: [], report: nil,
221
+ tokens: nil, cached: nil)
222
+ end
223
+
224
+ # Both halves, both tolerant of a missing one: this runs in an `ensure` after a
225
+ # failure that may have happened before either was created.
226
+ def destroy_clone(clone_id)
227
+ @agent_files.list(clone_id).each { |name| @agent_files.delete(clone_id, name) }
228
+ @profiles.delete(clone_id)
229
+ rescue StandardError => e
230
+ warn "[refinement] could not delete the gate clone '#{clone_id}': #{e.class}: #{e.message}"
231
+ end
232
+ end
233
+ end
234
+ end