insika 0.0.1 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +361 -0
- data/LICENSE +21 -0
- data/README.md +136 -2
- data/bin/insika +366 -0
- data/docs/AGENTS.md +618 -0
- data/docs/ARCHITECTURE.md +333 -0
- data/docs/BENCHMARK.md +114 -0
- data/docs/CHANNELS.md +453 -0
- data/docs/CONTEXT.md +117 -0
- data/docs/DEPLOY.md +354 -0
- data/docs/EMBEDDING.md +198 -0
- data/docs/EVALS.md +273 -0
- data/docs/LOADTEST.md +232 -0
- data/docs/OBSERVABILITY.md +374 -0
- data/docs/PLUGINS.md +211 -0
- data/docs/REFINEMENT.md +477 -0
- data/docs/RELEASING.md +70 -0
- data/docs/RUNNING-LOCAL.md +153 -0
- data/docs/SANDBOX.md +114 -0
- data/docs/SECURITY.md +375 -0
- data/docs/SKILLS.md +284 -0
- data/docs/TOOLS.md +302 -0
- data/docs/WHY.md +137 -0
- data/docs/WORKFLOWS.md +225 -0
- data/docs/build.md +14 -0
- data/docs/index.md +68 -0
- data/docs/onboarding/start.md +126 -0
- data/docs/operate.md +12 -0
- data/docs/ship.md +10 -0
- data/docs/understand.md +10 -0
- data/lib/insika/agent_file_store.rb +125 -0
- data/lib/insika/agent_profile.rb +255 -0
- data/lib/insika/alert_dispatcher.rb +139 -0
- data/lib/insika/allowlist.rb +28 -0
- data/lib/insika/baseline_store.rb +74 -0
- data/lib/insika/budget_ledger.rb +135 -0
- data/lib/insika/capability/resolved_tool.rb +34 -0
- data/lib/insika/capability_registry.rb +112 -0
- data/lib/insika/channel_delivery.rb +153 -0
- data/lib/insika/channel_registry.rb +30 -0
- data/lib/insika/channels/relay.rb +178 -0
- data/lib/insika/channels/web/widget.js +283 -0
- data/lib/insika/channels/web.rb +211 -0
- data/lib/insika/channels/webhook.rb +58 -0
- data/lib/insika/chat_builder.rb +303 -0
- data/lib/insika/checkpoint.rb +13 -0
- data/lib/insika/checkpoint_store.rb +153 -0
- data/lib/insika/circuit_state.rb +114 -0
- data/lib/insika/coercion.rb +58 -0
- data/lib/insika/command.rb +32 -0
- data/lib/insika/command_bus.rb +39 -0
- data/lib/insika/commands/agent_payload.rb +43 -0
- data/lib/insika/commands/approve_action.rb +46 -0
- data/lib/insika/commands/cancel_task.rb +33 -0
- data/lib/insika/commands/create_agent.rb +54 -0
- data/lib/insika/commands/create_session.rb +67 -0
- data/lib/insika/commands/delete_agent.rb +33 -0
- data/lib/insika/commands/delete_agent_file.rb +50 -0
- data/lib/insika/commands/delete_data_tool.rb +33 -0
- data/lib/insika/commands/delete_llm_provider.rb +36 -0
- data/lib/insika/commands/delete_mcp.rb +30 -0
- data/lib/insika/commands/delete_skill.rb +43 -0
- data/lib/insika/commands/delete_system_file.rb +29 -0
- data/lib/insika/commands/gate_refinement.rb +245 -0
- data/lib/insika/commands/import_mcp_tools.rb +48 -0
- data/lib/insika/commands/import_tools.rb +81 -0
- data/lib/insika/commands/issue_tenant_token.rb +41 -0
- data/lib/insika/commands/memory_add_note.rb +32 -0
- data/lib/insika/commands/memory_forget_fact.rb +32 -0
- data/lib/insika/commands/memory_put_fact.rb +35 -0
- data/lib/insika/commands/pause_task.rb +29 -0
- data/lib/insika/commands/resolve_refinement.rb +126 -0
- data/lib/insika/commands/restore_agent_file.rb +36 -0
- data/lib/insika/commands/restore_data_tool.rb +34 -0
- data/lib/insika/commands/restore_system_file.rb +31 -0
- data/lib/insika/commands/resume_task.rb +85 -0
- data/lib/insika/commands/revoke_token.rb +39 -0
- data/lib/insika/commands/rotate_tenant_token.rb +43 -0
- data/lib/insika/commands/run_refinement.rb +133 -0
- data/lib/insika/commands/send_message.rb +150 -0
- data/lib/insika/commands/set_agent_tools.rb +39 -0
- data/lib/insika/commands/set_skill_agents.rb +112 -0
- data/lib/insika/commands/trigger_workflow.rb +80 -0
- data/lib/insika/commands/update_agent.rb +49 -0
- data/lib/insika/commands/update_settings.rb +33 -0
- data/lib/insika/commands/upsert_llm_provider.rb +34 -0
- data/lib/insika/commands/upsert_mcp.rb +32 -0
- data/lib/insika/commands/write_agent_file.rb +57 -0
- data/lib/insika/commands/write_data_tool.rb +43 -0
- data/lib/insika/commands/write_golden.rb +58 -0
- data/lib/insika/commands/write_skill.rb +60 -0
- data/lib/insika/commands/write_system_file.rb +31 -0
- data/lib/insika/config_store.rb +89 -0
- data/lib/insika/context/builder.rb +166 -0
- data/lib/insika/context/catalog_provider.rb +23 -0
- data/lib/insika/context/fragment.rb +43 -0
- data/lib/insika/context/priority.rb +30 -0
- data/lib/insika/context/provider.rb +19 -0
- data/lib/insika/context/providers/memory.rb +60 -0
- data/lib/insika/context/providers/prompt.rb +105 -0
- data/lib/insika/context/providers/request.rb +32 -0
- data/lib/insika/context/providers/session.rb +123 -0
- data/lib/insika/context/providers/skill.rb +24 -0
- data/lib/insika/context/providers/skill_trigger.rb +128 -0
- data/lib/insika/context/providers/tool_search.rb +20 -0
- data/lib/insika/context_trace_store.rb +92 -0
- data/lib/insika/delegation_store.rb +153 -0
- data/lib/insika/doctor.rb +539 -0
- data/lib/insika/dsl/definition.rb +55 -0
- data/lib/insika/dsl/runtime.rb +382 -0
- data/lib/insika/dsl/server_boot.rb +98 -0
- data/lib/insika/dsl/system.rb +93 -0
- data/lib/insika/dsl/workflow_adapter.rb +59 -0
- data/lib/insika/dsl.rb +364 -0
- data/lib/insika/edge_limiter.rb +268 -0
- data/lib/insika/egress_guard.rb +75 -0
- data/lib/insika/env_schema.rb +249 -0
- data/lib/insika/errors.rb +201 -0
- data/lib/insika/evals/assertions.rb +247 -0
- data/lib/insika/evals/baseline.rb +69 -0
- data/lib/insika/evals/golden.rb +172 -0
- data/lib/insika/evals/judge.rb +225 -0
- data/lib/insika/evals/pairwise.rb +178 -0
- data/lib/insika/evals/report.rb +115 -0
- data/lib/insika/evals/runner.rb +141 -0
- data/lib/insika/evals/transport.rb +178 -0
- data/lib/insika/event.rb +18 -0
- data/lib/insika/event_stream.rb +132 -0
- data/lib/insika/executor.rb +1995 -0
- data/lib/insika/frontmatter.rb +42 -0
- data/lib/insika/golden_store.rb +145 -0
- data/lib/insika/hooks.rb +48 -0
- data/lib/insika/http_client.rb +63 -0
- data/lib/insika/inbound_log.rb +84 -0
- data/lib/insika/llm_configurator.rb +99 -0
- data/lib/insika/llm_provider_store.rb +83 -0
- data/lib/insika/loop_detector.rb +143 -0
- data/lib/insika/mcp_http_client.rb +67 -0
- data/lib/insika/mcp_store.rb +115 -0
- data/lib/insika/mcp_tool_ingestor.rb +143 -0
- data/lib/insika/memory_store.rb +93 -0
- data/lib/insika/message_origin.rb +76 -0
- data/lib/insika/middleware.rb +36 -0
- data/lib/insika/model_policy.rb +52 -0
- data/lib/insika/model_resolver.rb +176 -0
- data/lib/insika/model_selection.rb +115 -0
- data/lib/insika/onboarding.rb +208 -0
- data/lib/insika/outbox_store.rb +166 -0
- data/lib/insika/overlay_tool_registry.rb +102 -0
- data/lib/insika/pack.rb +102 -0
- data/lib/insika/pack_importer.rb +123 -0
- data/lib/insika/pending_action_store.rb +120 -0
- data/lib/insika/plugin/loader.rb +356 -0
- data/lib/insika/plugin.rb +35 -0
- data/lib/insika/policy/engine.rb +83 -0
- data/lib/insika/policy/policy.rb +120 -0
- data/lib/insika/policy_registry.rb +23 -0
- data/lib/insika/profile_source.rb +143 -0
- data/lib/insika/prompt_catalog.rb +61 -0
- data/lib/insika/provider_error_classifier.rb +160 -0
- data/lib/insika/queue_policy.rb +167 -0
- data/lib/insika/recovery.rb +168 -0
- data/lib/insika/refinement/candidate.rb +159 -0
- data/lib/insika/refinement/evidence_collector.rb +371 -0
- data/lib/insika/refinement/gate.rb +234 -0
- data/lib/insika/refinement/panel.rb +222 -0
- data/lib/insika/refinement/proposer.rb +262 -0
- data/lib/insika/refinement_store.rb +295 -0
- data/lib/insika/registry.rb +59 -0
- data/lib/insika/reliability.rb +185 -0
- data/lib/insika/safety/config.rb +109 -0
- data/lib/insika/safety/detectors.rb +176 -0
- data/lib/insika/safety/factory.rb +102 -0
- data/lib/insika/safety/input_guardrail.rb +102 -0
- data/lib/insika/safety/moderator.rb +94 -0
- data/lib/insika/safety/output_filter.rb +79 -0
- data/lib/insika/safety/output_validator.rb +101 -0
- data/lib/insika/safety/safe_responses.rb +47 -0
- data/lib/insika/sandbox/boundary.rb +93 -0
- data/lib/insika/sandbox/docker.rb +74 -0
- data/lib/insika/sandbox/local.rb +33 -0
- data/lib/insika/sandbox/runner.rb +80 -0
- data/lib/insika/sandbox.rb +85 -0
- data/lib/insika/schema_guard.rb +147 -0
- data/lib/insika/secret_masking.rb +34 -0
- data/lib/insika/server/a2a/agent_card.rb +27 -0
- data/lib/insika/server/a2a/app.rb +112 -0
- data/lib/insika/server/a2a/client.rb +101 -0
- data/lib/insika/server/a2a/errors.rb +32 -0
- data/lib/insika/server/a2a/http.rb +42 -0
- data/lib/insika/server/a2a/message.rb +27 -0
- data/lib/insika/server/a2a/protocol.rb +45 -0
- data/lib/insika/server/a2a/remotes.rb +25 -0
- data/lib/insika/server/a2a/task_projection.rb +40 -0
- data/lib/insika/server/app.rb +1022 -0
- data/lib/insika/server/boot.rb +119 -0
- data/lib/insika/server/rack_app.rb +118 -0
- data/lib/insika/server/responses.rb +165 -0
- data/lib/insika/server/sse_body.rb +96 -0
- data/lib/insika/server/tenant_auth.rb +61 -0
- data/lib/insika/session_actor.rb +162 -0
- data/lib/insika/session_store.rb +143 -0
- data/lib/insika/settings_store.rb +154 -0
- data/lib/insika/shutdown.rb +125 -0
- data/lib/insika/skill_catalog.rb +220 -0
- data/lib/insika/skill_store.rb +127 -0
- data/lib/insika/steer_injector.rb +110 -0
- data/lib/insika/store.rb +52 -0
- data/lib/insika/stores/memory.rb +123 -0
- data/lib/insika/stores/sqlite.rb +183 -0
- data/lib/insika/studio/app.rb +1693 -0
- data/lib/insika/studio/assets/dist/application.css +1 -0
- data/lib/insika/studio/assets/dist/application.js +70 -0
- data/lib/insika/studio/forms.rb +335 -0
- data/lib/insika/studio/nav_icons.rb +31 -0
- data/lib/insika/studio/views/_message.erb +44 -0
- data/lib/insika/studio/views/agent_detail.erb +285 -0
- data/lib/insika/studio/views/agents.erb +63 -0
- data/lib/insika/studio/views/approvals.erb +41 -0
- data/lib/insika/studio/views/chats.erb +34 -0
- data/lib/insika/studio/views/evals.erb +83 -0
- data/lib/insika/studio/views/home.erb +72 -0
- data/lib/insika/studio/views/layout.erb +94 -0
- data/lib/insika/studio/views/login.erb +17 -0
- data/lib/insika/studio/views/mcp.erb +91 -0
- data/lib/insika/studio/views/not_found.erb +5 -0
- data/lib/insika/studio/views/playground.erb +47 -0
- data/lib/insika/studio/views/refinement.erb +234 -0
- data/lib/insika/studio/views/session.erb +137 -0
- data/lib/insika/studio/views/settings.erb +168 -0
- data/lib/insika/studio/views/skills.erb +141 -0
- data/lib/insika/studio/views/system_files.erb +65 -0
- data/lib/insika/studio/views/task.erb +105 -0
- data/lib/insika/studio/views/tasks.erb +33 -0
- data/lib/insika/studio/views/tool_edit.erb +107 -0
- data/lib/insika/studio/views/tools.erb +89 -0
- data/lib/insika/subagent_graph.rb +96 -0
- data/lib/insika/system_file_store.rb +96 -0
- data/lib/insika/task_actor.rb +128 -0
- data/lib/insika/task_store.rb +250 -0
- data/lib/insika/telemetry/pricing.rb +104 -0
- data/lib/insika/telemetry/recorder.rb +228 -0
- data/lib/insika/telemetry.rb +127 -0
- data/lib/insika/testing/store_contract.rb +270 -0
- data/lib/insika/tick.rb +122 -0
- data/lib/insika/token_estimator.rb +16 -0
- data/lib/insika/token_store.rb +168 -0
- data/lib/insika/tool_assembly.rb +140 -0
- data/lib/insika/tool_catalog.rb +89 -0
- data/lib/insika/tool_definition.rb +518 -0
- data/lib/insika/tool_envelope.rb +140 -0
- data/lib/insika/tool_manifest.rb +218 -0
- data/lib/insika/tool_output_compressor.rb +100 -0
- data/lib/insika/tool_registry.rb +21 -0
- data/lib/insika/tool_store.rb +135 -0
- data/lib/insika/tool_trace_store.rb +92 -0
- data/lib/insika/tools/a2a_remote.rb +48 -0
- data/lib/insika/tools/agent_enum.rb +68 -0
- data/lib/insika/tools/concurrency.rb +54 -0
- data/lib/insika/tools/data_defined_tool.rb +219 -0
- data/lib/insika/tools/load_skill.rb +99 -0
- data/lib/insika/tools/remember.rb +53 -0
- data/lib/insika/tools/stuck_signal.rb +44 -0
- data/lib/insika/tools/subagent.rb +75 -0
- data/lib/insika/tools/subagents.rb +77 -0
- data/lib/insika/tools/tool_search.rb +94 -0
- data/lib/insika/turn_output.rb +139 -0
- data/lib/insika/turn_state.rb +162 -0
- data/lib/insika/turn_timing.rb +56 -0
- data/lib/insika/usage_ledger.rb +47 -0
- data/lib/insika/version.rb +3 -1
- data/lib/insika/wiring/graph.rb +249 -0
- data/lib/insika/workflow.rb +185 -0
- data/lib/insika/workflow_registry.rb +33 -0
- data/lib/insika.rb +220 -4
- metadata +412 -8
|
@@ -0,0 +1,247 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# the PII/secret patterns live in the RUNTIME (single source of
|
|
4
|
+
# truth) — the eval consumes them rather than keeping a divergent copy. Since the
|
|
5
|
+
# module moved under `lib/`, `Safety::Detectors` is loaded by `insika.rb` before this
|
|
6
|
+
# file; the explicit climb out of `evals/` that used to be here is gone.
|
|
7
|
+
|
|
8
|
+
module Insika
|
|
9
|
+
module Evals
|
|
10
|
+
# What the runner extracts from ONE replayed conversation, entirely from the
|
|
11
|
+
# public SSE stream of POST /v1/responses (no store reads — the eval stays a
|
|
12
|
+
# client). The assertion engine is pure over this value, so it's unit-testable
|
|
13
|
+
# offline without a server.
|
|
14
|
+
#
|
|
15
|
+
# output_text: the final assistant text (for content checks)
|
|
16
|
+
# tool_calls: [{ "name" =>, "status" => }] captured from the stream's tool events
|
|
17
|
+
# error: transport/turn error string, or nil on a clean turn
|
|
18
|
+
TurnResult = Struct.new(:output_text, :tool_calls, :error, keyword_init: true) do
|
|
19
|
+
def tool_names = Array(tool_calls).map { |t| (t["name"] || t[:name]).to_s }
|
|
20
|
+
|
|
21
|
+
# A tool call whose status is anything but a success ("ok"/2xx/"success").
|
|
22
|
+
def errored_tools
|
|
23
|
+
Array(tool_calls).reject { |t| Assertions.ok_status?(t["status"] || t[:status]) }
|
|
24
|
+
end
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
# A single check within a case (e.g. "tool:shipping_quote", "must_not:pii_leak").
|
|
28
|
+
Check = Struct.new(:name, :pass, :detail, keyword_init: true)
|
|
29
|
+
|
|
30
|
+
# The verdict for one golden case. `judge` (a Judge::Verdict) is attached AFTER
|
|
31
|
+
# the deterministic pass when the case has a rubric and a judge is configured;
|
|
32
|
+
# until then a rubric'd case is `judge_pending?` — it reads as "not fully
|
|
33
|
+
# evaluated", never a silent pass. A case passes only if the deterministic checks
|
|
34
|
+
# pass AND (there's no judge verdict OR it passed).
|
|
35
|
+
#
|
|
36
|
+
# `skipped` (a reason, nil = it ran) is the THIRD outcome: the
|
|
37
|
+
# deployment lacks something the case declared it needs, so there was nothing to
|
|
38
|
+
# assert. It is never a pass and never a failure — a suite of 40 cases where 12
|
|
39
|
+
# are skipped says something true, where 40 cases with 12 failures on capability
|
|
40
|
+
# grounds says nothing and gets ignored.
|
|
41
|
+
# `pairwise` (a Pairwise::Verdict) is attached when the case
|
|
42
|
+
# carries a `reference:` and a panel is configured. It DELIBERATELY does not enter
|
|
43
|
+
# `pass?`: "worse than the incumbent" is a judgement about a replacement decision,
|
|
44
|
+
# not a regression in the suite, and letting it fail a case would put an opinion
|
|
45
|
+
# about two conversations in the pre-merge gate.
|
|
46
|
+
CaseResult = Struct.new(:id, :agent, :checks, :error, :rubric, :judge, :skipped, :pairwise,
|
|
47
|
+
keyword_init: true) do
|
|
48
|
+
def skipped? = !skipped.nil?
|
|
49
|
+
def pass? = !skipped? && error.nil? && checks.all?(&:pass) && (judge.nil? || judge.pass)
|
|
50
|
+
def failures = checks.reject(&:pass)
|
|
51
|
+
# Has a rubric to score but no verdict yet (judge disabled / not run). A skipped
|
|
52
|
+
# case is not pending anything — nobody is going to judge a turn that never ran.
|
|
53
|
+
def judge_pending? = !skipped? && !rubric.to_s.strip.empty? && judge.nil?
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
# Deterministic evaluation — cheap, zero-token, zero-flakiness. It's the
|
|
57
|
+
# layer that catches the gross regressions (a tool stopped being called, a secret
|
|
58
|
+
# leaked, the turn errored). Subjective scoring is the LLM-judge in.
|
|
59
|
+
module Assertions
|
|
60
|
+
# Named negative detectors for `must_not` now live in the runtime. Kept as
|
|
61
|
+
# an alias so any external reference to Evals::Assertions::PII_DETECTORS still
|
|
62
|
+
# resolves; the values ARE the runtime's, never a fork.
|
|
63
|
+
PII_DETECTORS = Insika::Safety::Detectors::PII
|
|
64
|
+
|
|
65
|
+
# HOW MUCH THE AGENT SHOULD ASK BEFORE ACTING. Declared per
|
|
66
|
+
# case because it is a per-STORE decision, not a universal rule: sometimes the
|
|
67
|
+
# agent should establish the objective before searching ("energia, treino ou
|
|
68
|
+
# sono?" — a good store agent does this well), and sometimes asking again is the
|
|
69
|
+
# failure and it should just search. A global assertion would be wrong half
|
|
70
|
+
# the time; the judge is TOLD the policy (Judge#build_prompt) and this layer
|
|
71
|
+
# checks the half that needs no reader.
|
|
72
|
+
#
|
|
73
|
+
# Each rule is stated as the CUSTOMER-VISIBLE fact it checks. phrased
|
|
74
|
+
# this as "questions before the first tool call", written before: text a
|
|
75
|
+
# model emits before calling a tool never reaches the customer now (it rides
|
|
76
|
+
# `:intermediate`), and the eval is a client of `/v1/responses`, so what it can
|
|
77
|
+
# observe per turn is the published answer plus the tools that turn called.
|
|
78
|
+
# That is also the honest scope — a question nobody received is not a question.
|
|
79
|
+
POLICIES = {
|
|
80
|
+
# "UMA PERGUNTA POR VEZ" — the rule Insika broke twice under a 28 KB prompt.
|
|
81
|
+
"ask_once" => "at most one question per reply",
|
|
82
|
+
# Establish the objective before acting on a vague opener.
|
|
83
|
+
"investigate_first" => "asks before calling a tool, on the first turn",
|
|
84
|
+
# Act on the first plausible reading; refine after.
|
|
85
|
+
"act_fast" => "calls a tool on the first turn instead of asking"
|
|
86
|
+
}.freeze
|
|
87
|
+
|
|
88
|
+
module_function
|
|
89
|
+
|
|
90
|
+
# WHAT THIS DEPLOYMENT LACKS for the case to be worth running.
|
|
91
|
+
# -> [reason]; empty = run it.
|
|
92
|
+
#
|
|
93
|
+
# `available` is the deployment's answer for this agent:
|
|
94
|
+
# { "tools" => [names] | nil, "capabilities" => [names] }
|
|
95
|
+
# `tools` nil means an OPEN allowlist — the agent may call every registered tool,
|
|
96
|
+
# so no tool requirement can be judged missing and the case runs. That is the
|
|
97
|
+
# deliberate reading: "I could not rule it out" must not become a skip, or a
|
|
98
|
+
# permissive agent would quietly stop being tested.
|
|
99
|
+
def unmet_requirements(golden, available)
|
|
100
|
+
tools = available["tools"]
|
|
101
|
+
declared = Array(available["capabilities"]).map(&:to_s)
|
|
102
|
+
|
|
103
|
+
missing_tools = tools.nil? ? [] : golden.required_tools - Array(tools).map(&:to_s)
|
|
104
|
+
missing_caps = golden.required_capabilities - declared
|
|
105
|
+
|
|
106
|
+
reasons = []
|
|
107
|
+
reasons << "tool not available: #{missing_tools.join(', ')}" unless missing_tools.empty?
|
|
108
|
+
reasons << "capability not declared: #{missing_caps.join(', ')}" unless missing_caps.empty?
|
|
109
|
+
reasons
|
|
110
|
+
end
|
|
111
|
+
|
|
112
|
+
# The case did not run and MUST NOT read as either a pass or a failure.
|
|
113
|
+
def skip(golden, reason)
|
|
114
|
+
CaseResult.new(id: golden.id, agent: golden.agent, checks: [], error: nil,
|
|
115
|
+
rubric: nil, judge: nil, skipped: reason)
|
|
116
|
+
end
|
|
117
|
+
|
|
118
|
+
# A tool status counts as success when it's blank/"ok"/"success" or a 2xx code.
|
|
119
|
+
def ok_status?(status)
|
|
120
|
+
return true if status.nil?
|
|
121
|
+
|
|
122
|
+
s = status.to_s.strip.downcase
|
|
123
|
+
return true if s.empty? || %w[ok success succeeded done].include?(s)
|
|
124
|
+
|
|
125
|
+
code = Integer(s, exception: false)
|
|
126
|
+
code ? code.between?(200, 299) : false
|
|
127
|
+
end
|
|
128
|
+
|
|
129
|
+
# Golden + TurnResult -> CaseResult. A turn that failed to run yields a single
|
|
130
|
+
# failing check (there's nothing to assert on a turn that never produced output).
|
|
131
|
+
#
|
|
132
|
+
# `turns` is every turn of the conversation, in order; `result` is the last one
|
|
133
|
+
# (what the tool/content assertions have always run on). The policy checks need
|
|
134
|
+
# all of them — "one question per reply" is a rule about every reply, and the
|
|
135
|
+
# violation that motivated this was on the FIRST turn. Defaults to the single
|
|
136
|
+
# result so existing callers keep working.
|
|
137
|
+
def evaluate(golden, result, turns: nil)
|
|
138
|
+
if result.error
|
|
139
|
+
return CaseResult.new(id: golden.id, agent: golden.agent, error: result.error, rubric: nil, judge: nil,
|
|
140
|
+
checks: [Check.new(name: "turn", pass: false, detail: "turn error: #{result.error}")])
|
|
141
|
+
end
|
|
142
|
+
|
|
143
|
+
# NOT `Array(turns)`: TurnResult is a Struct, so Array() would explode a single
|
|
144
|
+
# one into its members and hand the policy checks three strings.
|
|
145
|
+
conversation = turns.nil? || turns.empty? ? [result] : turns
|
|
146
|
+
checks = tool_checks(golden, result) + must_not_checks(golden, result) +
|
|
147
|
+
policy_checks(golden, conversation)
|
|
148
|
+
CaseResult.new(id: golden.id, agent: golden.agent, error: nil, checks: checks,
|
|
149
|
+
rubric: golden.rubric, judge: nil)
|
|
150
|
+
end
|
|
151
|
+
|
|
152
|
+
# Each REQUIRED expected tool must appear in the turn's tool calls. Optional
|
|
153
|
+
# ("name?") tools are informational — present or not, they never fail.
|
|
154
|
+
def tool_checks(golden, result)
|
|
155
|
+
names = result.tool_names
|
|
156
|
+
golden.tools_called.filter_map do |t|
|
|
157
|
+
next if t[:optional]
|
|
158
|
+
|
|
159
|
+
present = names.include?(t[:name])
|
|
160
|
+
Check.new(name: "tool:#{t[:name]}", pass: present,
|
|
161
|
+
detail: present ? "called" : "expected but not called (saw: #{names.join(', ')})")
|
|
162
|
+
end
|
|
163
|
+
end
|
|
164
|
+
|
|
165
|
+
# `must_not` detectors. "tool_error" is special (inspects statuses); the rest
|
|
166
|
+
# are content detectors over the output text.
|
|
167
|
+
def must_not_checks(golden, result)
|
|
168
|
+
golden.must_not.map do |name|
|
|
169
|
+
if name == "tool_error"
|
|
170
|
+
bad = result.errored_tools
|
|
171
|
+
Check.new(name: "must_not:tool_error", pass: bad.empty?,
|
|
172
|
+
detail: bad.empty? ? "no tool errors" : "errored: #{bad.map { |t| t['name'] || t[:name] }.join(', ')}")
|
|
173
|
+
else
|
|
174
|
+
hit = detect(name, result.output_text.to_s)
|
|
175
|
+
Check.new(name: "must_not:#{name}", pass: hit.nil?,
|
|
176
|
+
detail: hit ? "matched #{hit.inspect}" : "clean")
|
|
177
|
+
end
|
|
178
|
+
end
|
|
179
|
+
end
|
|
180
|
+
|
|
181
|
+
# The declared `policy`, checked deterministically over the conversation. No
|
|
182
|
+
# policy -> no check (and nothing to explain in the report).
|
|
183
|
+
def policy_checks(golden, turns)
|
|
184
|
+
name = golden.policy
|
|
185
|
+
return [] if name.nil?
|
|
186
|
+
|
|
187
|
+
# Exhaustive on purpose: a policy added to POLICIES without a rule here would
|
|
188
|
+
# otherwise fall into whichever branch was last and check the wrong thing.
|
|
189
|
+
pass, detail = case name
|
|
190
|
+
when "ask_once" then ask_once(turns)
|
|
191
|
+
when "investigate_first" then investigate_first(turns.first)
|
|
192
|
+
when "act_fast" then act_fast(turns.first)
|
|
193
|
+
else raise ArgumentError, "policy #{name.inspect} has no rule"
|
|
194
|
+
end
|
|
195
|
+
[Check.new(name: "policy:#{name}", pass: pass, detail: detail)]
|
|
196
|
+
end
|
|
197
|
+
|
|
198
|
+
# Every reply asks at most one question. Reported with the offending turn and
|
|
199
|
+
# the reply itself — "2 questions" alone sends the reader digging.
|
|
200
|
+
def ask_once(turns)
|
|
201
|
+
offender = turns.each_with_index.find { |t, _| count_questions(t.output_text) > 1 }
|
|
202
|
+
return [true, "at most one question per reply"] unless offender
|
|
203
|
+
|
|
204
|
+
turn, i = offender
|
|
205
|
+
[false, "turn #{i + 1} asked #{count_questions(turn.output_text)} questions: " \
|
|
206
|
+
"#{turn.output_text.to_s.strip[0, 160].inspect}"]
|
|
207
|
+
end
|
|
208
|
+
|
|
209
|
+
def investigate_first(turn)
|
|
210
|
+
return [false, "no turn to check"] if turn.nil?
|
|
211
|
+
|
|
212
|
+
tools = turn.tool_names
|
|
213
|
+
return [false, "called #{tools.join(', ')} before asking anything"] unless tools.empty?
|
|
214
|
+
return [false, "answered without asking: #{turn.output_text.to_s.strip[0, 160].inspect}"] if
|
|
215
|
+
count_questions(turn.output_text).zero?
|
|
216
|
+
|
|
217
|
+
[true, "asked before acting"]
|
|
218
|
+
end
|
|
219
|
+
|
|
220
|
+
def act_fast(turn)
|
|
221
|
+
return [false, "no turn to check"] if turn.nil?
|
|
222
|
+
|
|
223
|
+
tools = turn.tool_names
|
|
224
|
+
return [true, "acted: called #{tools.join(', ')}"] unless tools.empty?
|
|
225
|
+
|
|
226
|
+
[false, "asked instead of acting: #{turn.output_text.to_s.strip[0, 160].inspect}"]
|
|
227
|
+
end
|
|
228
|
+
|
|
229
|
+
# Questions in ONE reply. Deliberately crude and deliberately documented: a run
|
|
230
|
+
# of "?" counts once ("já pensou??" is one question), and URLs are dropped first
|
|
231
|
+
# so a tracking link's query string is not read as the agent asking something.
|
|
232
|
+
# It is a policy signal, not grammar — and it already caught a real violation
|
|
233
|
+
# ("é pra você ou tá pensando em presentear alguém? E qual seu tamanho?").
|
|
234
|
+
def count_questions(text)
|
|
235
|
+
text.to_s.gsub(%r{https?://\S+}, " ").scan(/\?+/).size
|
|
236
|
+
end
|
|
237
|
+
|
|
238
|
+
# Runs a named detector over the text. "pii_leak" = union of all PII detectors;
|
|
239
|
+
# otherwise a single named pattern. Delegates to the runtime's single source
|
|
240
|
+
# which itself fails loud on an unknown name (a typo'd assertion must not
|
|
241
|
+
# silently pass).
|
|
242
|
+
def detect(name, text)
|
|
243
|
+
Insika::Safety::Detectors.detect(name, text)
|
|
244
|
+
end
|
|
245
|
+
end
|
|
246
|
+
end
|
|
247
|
+
end
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "json"
|
|
4
|
+
|
|
5
|
+
module Insika
|
|
6
|
+
module Evals
|
|
7
|
+
# gating. A baseline is the accepted state of the golden
|
|
8
|
+
# set — `{ cases: { id => { pass, score } } }`. A gated run compares against it and
|
|
9
|
+
# blocks only on a REGRESSION, so known-failing cases don't wedge the gate while a
|
|
10
|
+
# real drop (a passing case that now fails, or a judge score that fell past the
|
|
11
|
+
# tolerance) does. That's the pre-merge gate for prompt/tool/model changes.
|
|
12
|
+
module Baseline
|
|
13
|
+
Regression = Struct.new(:id, :kind, :detail, keyword_init: true)
|
|
14
|
+
|
|
15
|
+
module_function
|
|
16
|
+
|
|
17
|
+
# [CaseResult] -> baseline hash. `at` is stamped by the caller (kept out of here
|
|
18
|
+
# so the module stays deterministic/testable).
|
|
19
|
+
# A SKIPPED case is left out entirely: writing it as `pass:
|
|
20
|
+
# false` would accept "this deployment cannot run it" as the accepted state,
|
|
21
|
+
# and the case would never block anywhere again.
|
|
22
|
+
def snapshot(results, at:)
|
|
23
|
+
{
|
|
24
|
+
"at" => at,
|
|
25
|
+
"cases" => results.reject(&:skipped?).each_with_object({}) do |r, h|
|
|
26
|
+
h[r.id] = { "pass" => r.pass?, "score" => (r.judge&.score) }
|
|
27
|
+
end
|
|
28
|
+
}
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
def load(path) = JSON.parse(File.read(path))
|
|
32
|
+
|
|
33
|
+
def write(path, results, at:)
|
|
34
|
+
File.write(path, JSON.pretty_generate(snapshot(results, at: at)))
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
# Compares a run against a loaded baseline. -> [Regression]. Only cases present
|
|
38
|
+
# in BOTH are compared: a new case (no baseline entry) never blocks the gate (it
|
|
39
|
+
# shows in the report as ❌ but is not a "regression"); document this in README.
|
|
40
|
+
# • pass→fail : baseline pass, now failing (hard regression).
|
|
41
|
+
# • pass→skipped: baseline pass, now unrunnable HERE. says the
|
|
42
|
+
# gate never blocks ON a skip, and it does not: a case that was
|
|
43
|
+
# already skipped or unknown stays silent. But a case that used
|
|
44
|
+
# to run on this deployment and no longer can means the agent
|
|
45
|
+
# lost a tool or a declaration — a suite shrinking in silence is
|
|
46
|
+
# exactly what this outcome was added to prevent. Re-baseline if
|
|
47
|
+
# the shrink is intended.
|
|
48
|
+
# • score-drop : baseline score - current judge score > tolerance (quality
|
|
49
|
+
# drift, even if the case still technically passes).
|
|
50
|
+
def compare(results, baseline, tolerance:)
|
|
51
|
+
base = baseline["cases"] || {}
|
|
52
|
+
results.filter_map do |r|
|
|
53
|
+
b = base[r.id]
|
|
54
|
+
next unless b
|
|
55
|
+
|
|
56
|
+
if b["pass"] && r.skipped?
|
|
57
|
+
Regression.new(id: r.id, kind: "pass→skipped",
|
|
58
|
+
detail: "was passing, now unrunnable here: #{r.skipped}")
|
|
59
|
+
elsif b["pass"] && !r.pass?
|
|
60
|
+
Regression.new(id: r.id, kind: "pass→fail", detail: "was passing, now failing")
|
|
61
|
+
elsif b["score"] && r.judge && (b["score"] - r.judge.score) > tolerance
|
|
62
|
+
Regression.new(id: r.id, kind: "score-drop",
|
|
63
|
+
detail: "judge #{b['score']} -> #{r.judge.score} (> #{tolerance})")
|
|
64
|
+
end
|
|
65
|
+
end
|
|
66
|
+
end
|
|
67
|
+
end
|
|
68
|
+
end
|
|
69
|
+
end
|
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "yaml"
|
|
4
|
+
|
|
5
|
+
# Evals — the quality harness. It lives in `lib/` so the engine itself can
|
|
6
|
+
# call it (the refinement gate of needs to score a candidate agent, and a
|
|
7
|
+
# second copy of the judge would be the worst possible outcome), but it stays a
|
|
8
|
+
# CLIENT: it reaches a running deployment over HTTP through `HttpTransport` and never
|
|
9
|
+
# reads a store directly. `evals/run.rb` is a thin CLI over this module.
|
|
10
|
+
module Insika
|
|
11
|
+
module Evals
|
|
12
|
+
# A curated behavior case, loaded from a data file (evals/golden/<agent>/*.yml).
|
|
13
|
+
# Data, not code — same spirit as tools-as-data. See evals/README.md for the format.
|
|
14
|
+
Golden = Struct.new(:id, :agent, :turns, :expect, :requires, :reference, :source, keyword_init: true) do
|
|
15
|
+
# The user messages to replay, in order.
|
|
16
|
+
def user_turns = turns.map { |t| t["user"] }
|
|
17
|
+
|
|
18
|
+
# Tool refs the case expects; a trailing "?" marks OPTIONAL (never fails).
|
|
19
|
+
# -> [{ name:, optional: }]
|
|
20
|
+
def tools_called
|
|
21
|
+
Array(expect["tools_called"]).map do |ref|
|
|
22
|
+
s = ref.to_s
|
|
23
|
+
optional = s.end_with?("?")
|
|
24
|
+
{ name: optional ? s[0..-2] : s, optional: optional }
|
|
25
|
+
end
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
# Names of the negative assertions to run (e.g. "pii_leak", "tool_error").
|
|
29
|
+
def must_not = Array(expect["must_not"]).map(&:to_s)
|
|
30
|
+
|
|
31
|
+
# How much the agent should ask before acting. nil = the store
|
|
32
|
+
# has no opinion and only the rubric decides.
|
|
33
|
+
def policy = GoldenLoader.presence(expect["policy"])
|
|
34
|
+
|
|
35
|
+
# What the DEPLOYMENT must have for this case to mean anything.
|
|
36
|
+
# Empty = runs everywhere.
|
|
37
|
+
def required_tools = Array(requires["tools"]).map(&:to_s)
|
|
38
|
+
def required_capabilities = Array(requires["capabilities"]).map(&:to_s)
|
|
39
|
+
def requirements? = !(required_tools + required_capabilities).empty?
|
|
40
|
+
|
|
41
|
+
# THE INCUMBENT'S CONVERSATION for the same opening — the other
|
|
42
|
+
# half of a pairwise comparison. Data in the case, not a store read: the eval is
|
|
43
|
+
# a client, and a pair that lives in one reviewable file cannot go stale against
|
|
44
|
+
# a database nobody looked at.
|
|
45
|
+
def reference_messages = Array(reference["messages"])
|
|
46
|
+
def reference_source = GoldenLoader.presence(reference["source"])
|
|
47
|
+
def reference? = !reference_messages.empty?
|
|
48
|
+
|
|
49
|
+
# Did a PERSON type part of the reference half? After a handoff the operator's
|
|
50
|
+
# words are stored as `role: assistant`, and comparing a model to a human
|
|
51
|
+
# and calling it a win is a lie in both directions — so the pair is LABELLED and
|
|
52
|
+
# the report never prints the outcome without it.
|
|
53
|
+
def human_assisted?
|
|
54
|
+
reference_messages.any? { |m| MessageOrigin.origin_of(m) == MessageOrigin::OPERATOR }
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
# LLM-judge rubric + threshold (consumed in — deferred here).
|
|
58
|
+
def rubric = expect["rubric"]
|
|
59
|
+
def min_score = expect["min_score"]
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
# Loads + validates golden files. Fails LOUD on a malformed case — a silently
|
|
63
|
+
# dropped golden is a hole in the safety net.
|
|
64
|
+
module GoldenLoader
|
|
65
|
+
class InvalidGolden < StandardError; end
|
|
66
|
+
|
|
67
|
+
module_function
|
|
68
|
+
|
|
69
|
+
# Loads every *.yml/*.yaml under `dir` (recursive), sorted by path for a stable
|
|
70
|
+
# run order. -> [Golden].
|
|
71
|
+
def load_dir(dir)
|
|
72
|
+
Dir.glob(File.join(dir, "**", "*.{yml,yaml}")).sort.map { |f| load_file(f) }
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
def load_file(path)
|
|
76
|
+
raw = YAML.safe_load(File.read(path), permitted_classes: [], aliases: false) || {}
|
|
77
|
+
build(raw, source: path)
|
|
78
|
+
rescue Psych::SyntaxError => e
|
|
79
|
+
raise InvalidGolden, "#{path}: invalid YAML — #{e.message}"
|
|
80
|
+
end
|
|
81
|
+
|
|
82
|
+
# hash (string keys) -> validated Golden. `source` is only for error messages.
|
|
83
|
+
def build(raw, source: "(inline)")
|
|
84
|
+
raise InvalidGolden, "#{source}: golden must be a mapping" unless raw.is_a?(Hash)
|
|
85
|
+
|
|
86
|
+
id = presence(raw["id"]) || (raise InvalidGolden, "#{source}: 'id' is required")
|
|
87
|
+
agent = presence(raw["agent"]) || (raise InvalidGolden, "#{source}: 'agent' is required (case '#{id}')")
|
|
88
|
+
turns = normalize_turns(raw["turns"], id: id, source: source)
|
|
89
|
+
expect = raw["expect"] || {}
|
|
90
|
+
raise InvalidGolden, "#{source}: 'expect' must be a mapping (case '#{id}')" unless expect.is_a?(Hash)
|
|
91
|
+
|
|
92
|
+
validate_policy!(expect["policy"], id: id, source: source)
|
|
93
|
+
requires = raw["requires"] || {}
|
|
94
|
+
unless requires.is_a?(Hash)
|
|
95
|
+
raise InvalidGolden, "#{source}: 'requires' must be a mapping (case '#{id}')"
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
reference = normalize_reference(raw["reference"], id: id, source: source)
|
|
99
|
+
|
|
100
|
+
Golden.new(id: id, agent: agent, turns: turns, expect: expect,
|
|
101
|
+
requires: requires, reference: reference, source: source)
|
|
102
|
+
end
|
|
103
|
+
|
|
104
|
+
# reference: { "source" => String?, "messages" => [{ "role" =>, "text" =>,
|
|
105
|
+
# "origin" => }] }. Absent -> {}, and the case simply has nothing to compare
|
|
106
|
+
# against. Malformed is REFUSED: a reference that half-loads would produce a
|
|
107
|
+
# pairwise verdict about a transcript nobody wrote.
|
|
108
|
+
def normalize_reference(raw, id:, source:)
|
|
109
|
+
return {} if raw.nil?
|
|
110
|
+
raise InvalidGolden, "#{source}: 'reference' must be a mapping (case '#{id}')" unless raw.is_a?(Hash)
|
|
111
|
+
|
|
112
|
+
messages = raw["messages"]
|
|
113
|
+
unless messages.is_a?(Array) && !messages.empty?
|
|
114
|
+
raise InvalidGolden, "#{source}: reference needs a non-empty 'messages' array (case '#{id}')"
|
|
115
|
+
end
|
|
116
|
+
|
|
117
|
+
{ "source" => presence(raw["source"]),
|
|
118
|
+
"messages" => messages.each_with_index.map { |m, i| reference_message(m, i, id: id, source: source) } }.compact
|
|
119
|
+
end
|
|
120
|
+
|
|
121
|
+
def reference_message(raw, index, id:, source:)
|
|
122
|
+
where = "#{source}: reference.messages[#{index}] (case '#{id}')"
|
|
123
|
+
raise InvalidGolden, "#{where} must be a mapping" unless raw.is_a?(Hash)
|
|
124
|
+
|
|
125
|
+
role = presence(raw["role"])
|
|
126
|
+
raise InvalidGolden, "#{where} needs a 'role' of user or assistant" unless %w[user assistant].include?(role)
|
|
127
|
+
|
|
128
|
+
text = presence(raw["text"]) || (raise InvalidGolden, "#{where} needs a non-empty 'text'")
|
|
129
|
+
# The SAME closed vocabulary the engine stamps. A typo'd marker would
|
|
130
|
+
# read as "absent" downstream, which is how a human turn gets scored as the
|
|
131
|
+
# incumbent's model.
|
|
132
|
+
origin = begin
|
|
133
|
+
MessageOrigin.parse!(raw["origin"])
|
|
134
|
+
rescue Insika::ValidationError => e
|
|
135
|
+
raise InvalidGolden, "#{where}: #{e.message}"
|
|
136
|
+
end
|
|
137
|
+
{ "role" => role, "text" => text }.merge(origin ? { "origin" => origin } : {})
|
|
138
|
+
end
|
|
139
|
+
|
|
140
|
+
# A typo'd policy must not silently mean "no policy" — the case would go on
|
|
141
|
+
# passing while the rule it was written for stopped being checked. The
|
|
142
|
+
# `Assertions` constant is resolved at CALL time (this file loads first, and
|
|
143
|
+
# assertions.rb touches `Safety::Detectors` at load time).
|
|
144
|
+
def validate_policy!(value, id:, source:)
|
|
145
|
+
name = presence(value)
|
|
146
|
+
return if name.nil? || Assertions::POLICIES.key?(name)
|
|
147
|
+
|
|
148
|
+
raise InvalidGolden, "#{source}: unknown policy #{name.inspect} (case '#{id}') — " \
|
|
149
|
+
"known: #{Assertions::POLICIES.keys.join(', ')}"
|
|
150
|
+
end
|
|
151
|
+
|
|
152
|
+
# turns: a non-empty array of { "user" => String }. Rejects anything else so a
|
|
153
|
+
# typo (e.g. `users:`) surfaces at load time, not as an empty replay.
|
|
154
|
+
def normalize_turns(turns, id:, source:)
|
|
155
|
+
unless turns.is_a?(Array) && !turns.empty?
|
|
156
|
+
raise InvalidGolden, "#{source}: 'turns' must be a non-empty array (case '#{id}')"
|
|
157
|
+
end
|
|
158
|
+
|
|
159
|
+
turns.each_with_index.map do |t, i|
|
|
160
|
+
user = t.is_a?(Hash) ? presence(t["user"]) : nil
|
|
161
|
+
user || (raise InvalidGolden, "#{source}: turns[#{i}] needs a non-empty 'user' (case '#{id}')")
|
|
162
|
+
{ "user" => user }
|
|
163
|
+
end
|
|
164
|
+
end
|
|
165
|
+
|
|
166
|
+
def presence(v)
|
|
167
|
+
s = v.to_s.strip
|
|
168
|
+
s.empty? ? nil : s
|
|
169
|
+
end
|
|
170
|
+
end
|
|
171
|
+
end
|
|
172
|
+
end
|