insika 0.3.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +296 -0
- data/README.md +48 -12
- data/bin/insika +725 -0
- data/bin/insika-router +87 -0
- data/docs/AGENTS.md +116 -406
- data/docs/API.md +5 -5
- data/docs/ARCHITECTURE.md +3 -2
- data/docs/ARTIFACTS.md +137 -0
- data/docs/BENCHMARK.md +2 -2
- data/docs/CHANNELS.md +14 -14
- data/docs/CONTEXT.md +63 -19
- data/docs/DEMO.md +80 -0
- data/docs/DEPLOY.md +87 -10
- data/docs/EMBEDDING.md +1 -1
- data/docs/EVALS.md +128 -3
- data/docs/FACTS.md +3 -3
- data/docs/HARVEST.md +5 -6
- data/docs/KNOWLEDGE.md +290 -0
- data/docs/LOADTEST.md +17 -29
- data/docs/MEDIA.md +128 -0
- data/docs/OBSERVABILITY.md +46 -12
- data/docs/OUTCOMES.md +137 -0
- data/docs/PLUGINS.md +51 -6
- data/docs/POLICY.md +222 -0
- data/docs/REFINEMENT.md +14 -9
- data/docs/RELEASING.md +4 -4
- data/docs/ROUTER.md +213 -0
- data/docs/RUNNING-LOCAL.md +5 -5
- data/docs/SCHEDULING.md +121 -0
- data/docs/SECURITY.md +23 -7
- data/docs/SKILLS.md +11 -2
- data/docs/SOAK.md +3 -3
- data/docs/TEMPLATES.md +134 -0
- data/docs/TOOLS.md +176 -27
- data/docs/WHY.md +1 -1
- data/docs/WORKFLOWS.md +2 -2
- data/docs/_includes/head_custom.html +5 -0
- data/docs/_includes/title.html +13 -0
- data/docs/_sass/color_schemes/insika.scss +32 -0
- data/docs/_sass/custom/custom.scss +199 -0
- data/docs/_sass/custom/setup.scss +26 -0
- data/docs/assets/img/favicon.svg +7 -0
- data/docs/assets/img/insika-mark.svg +7 -0
- data/docs/core-concepts.md +21 -0
- data/docs/domain.md +4 -4
- data/docs/improve.md +20 -0
- data/docs/index.md +8 -5
- data/docs/integrate.md +20 -0
- data/docs/operate.md +13 -6
- data/docs/prompts/ADD-TOOL.md +118 -0
- data/docs/prompts/DIAGNOSE-TURN.md +65 -0
- data/docs/prompts/GO-LIVE.md +138 -0
- data/docs/prompts/RUN-EXAMPLES.md +70 -0
- data/docs/reference.md +19 -0
- data/docs/ship.md +10 -2
- data/docs/start-here.md +18 -0
- data/lib/insika/agent_profile.rb +99 -17
- data/lib/insika/artifact_signing.rb +82 -0
- data/lib/insika/artifact_store.rb +160 -0
- data/lib/insika/channel_delivery.rb +1 -1
- data/lib/insika/chat_builder.rb +50 -19
- data/lib/insika/commands/agent_payload.rb +2 -2
- data/lib/insika/commands/backfill_knowledge.rb +145 -0
- data/lib/insika/commands/delete_artifact.rb +35 -0
- data/lib/insika/commands/delete_concept.rb +34 -0
- data/lib/insika/commands/delete_mcp.rb +6 -2
- data/lib/insika/commands/delete_tenant_data.rb +15 -3
- data/lib/insika/commands/gate_refinement.rb +1 -1
- data/lib/insika/commands/refresh_mcp_tools.rb +47 -0
- data/lib/insika/commands/restore_concept.rb +34 -0
- data/lib/insika/commands/seed_demo_data.rb +31 -0
- data/lib/insika/commands/upsert_mcp.rb +6 -3
- data/lib/insika/commands/write_concept.rb +57 -0
- data/lib/insika/compaction.rb +196 -0
- data/lib/insika/context/builder.rb +6 -2
- data/lib/insika/context/fragment.rb +4 -1
- data/lib/insika/context/priority.rb +8 -0
- data/lib/insika/context/providers/briefing.rb +53 -24
- data/lib/insika/context/providers/knowledge.rb +108 -0
- data/lib/insika/context/providers/prompt.rb +30 -24
- data/lib/insika/context/providers/session.rb +46 -10
- data/lib/insika/context_trace_store.rb +11 -1
- data/lib/insika/cron.rb +189 -0
- data/lib/insika/demo/agent_attrs.rb +43 -0
- data/lib/insika/demo/golden_cases.rb +81 -0
- data/lib/insika/demo/seeder.rb +336 -0
- data/lib/insika/doctor.rb +280 -17
- data/lib/insika/dsl/definition.rb +3 -2
- data/lib/insika/dsl/runtime.rb +64 -79
- data/lib/insika/dsl/server_boot.rb +23 -1
- data/lib/insika/dsl/system.rb +10 -2
- data/lib/insika/dsl.rb +103 -2
- data/lib/insika/env_schema.rb +21 -7
- data/lib/insika/evals/golden.rb +41 -4
- data/lib/insika/evals/judge.rb +47 -2
- data/lib/insika/evals/pairwise.rb +11 -0
- data/lib/insika/evals/persona.rb +98 -0
- data/lib/insika/evals/runner.rb +9 -0
- data/lib/insika/evals/simulator.rb +225 -0
- data/lib/insika/evals/transport.rb +84 -2
- data/lib/insika/event_stream.rb +10 -0
- data/lib/insika/executor.rb +295 -55
- data/lib/insika/followup_policy.rb +2 -25
- data/lib/insika/golden_store.rb +16 -1
- data/lib/insika/grounding/matcher.rb +1 -1
- data/lib/insika/knowledge.rb +680 -0
- data/lib/insika/knowledge_store.rb +140 -0
- data/lib/insika/loop_detector.rb +5 -34
- data/lib/insika/mcp_client.rb +94 -0
- data/lib/insika/mcp_json.rb +74 -0
- data/lib/insika/mcp_live_tool.rb +43 -0
- data/lib/insika/mcp_store.rb +98 -26
- data/lib/insika/mcp_tool_ingestor.rb +30 -8
- data/lib/insika/mcp_tool_registry.rb +100 -0
- data/lib/insika/media.rb +115 -31
- data/lib/insika/message_origin.rb +1 -1
- data/lib/insika/middleware.rb +9 -0
- data/lib/insika/onboarding.rb +17 -1
- data/lib/insika/outcome_store.rb +1 -1
- data/lib/insika/overlay_tool_registry.rb +37 -17
- data/lib/insika/packaging.rb +2 -2
- data/lib/insika/profile_source.rb +15 -1
- data/lib/insika/prompt_catalog.rb +10 -0
- data/lib/insika/retention.rb +36 -1
- data/lib/insika/router/app.rb +157 -0
- data/lib/insika/router/backend_pool.rb +98 -0
- data/lib/insika/router/hash_ring.rb +55 -0
- data/lib/insika/router/proxy_body.rb +34 -0
- data/lib/insika/router/session_key.rb +54 -0
- data/lib/insika/router.rb +18 -0
- data/lib/insika/schedule.rb +177 -0
- data/lib/insika/schedule_engine.rb +314 -0
- data/lib/insika/schedule_store.rb +208 -0
- data/lib/insika/server/app.rb +105 -15
- data/lib/insika/server/rack_app.rb +5 -1
- data/lib/insika/server/responses.rb +5 -5
- data/lib/insika/session_store.rb +34 -4
- data/lib/insika/settings_store.rb +8 -1
- data/lib/insika/skill_catalog.rb +12 -0
- data/lib/insika/soak/runner.rb +4 -4
- data/lib/insika/steer_injector.rb +21 -10
- data/lib/insika/studio/app.rb +591 -47
- data/lib/insika/studio/assets/dist/application.css +1 -1
- data/lib/insika/studio/assets/dist/application.js +21 -21
- data/lib/insika/studio/forms.rb +57 -5
- data/lib/insika/studio/nav_icons.rb +14 -1
- data/lib/insika/studio/views/_agent_tab_cache.erb +25 -0
- data/lib/insika/studio/views/_agent_tab_config.erb +514 -0
- data/lib/insika/studio/views/_agent_tab_history.erb +24 -0
- data/lib/insika/studio/views/_agent_tab_loops.erb +54 -0
- data/lib/insika/studio/views/_agent_tab_memory.erb +51 -0
- data/lib/insika/studio/views/_agent_tab_outcomes.erb +31 -0
- data/lib/insika/studio/views/_agent_tab_prompts.erb +108 -0
- data/lib/insika/studio/views/_agent_tab_skills.erb +38 -0
- data/lib/insika/studio/views/_agents_master.erb +44 -0
- data/lib/insika/studio/views/_message.erb +49 -32
- data/lib/insika/studio/views/agent_detail.erb +61 -820
- data/lib/insika/studio/views/agents.erb +70 -57
- data/lib/insika/studio/views/artifact.erb +23 -0
- data/lib/insika/studio/views/artifacts.erb +59 -0
- data/lib/insika/studio/views/evals.erb +2 -2
- data/lib/insika/studio/views/facts.erb +1 -1
- data/lib/insika/studio/views/funnel.erb +1 -1
- data/lib/insika/studio/views/home.erb +106 -67
- data/lib/insika/studio/views/knowledge.erb +123 -0
- data/lib/insika/studio/views/layout.erb +14 -11
- data/lib/insika/studio/views/mcp.erb +174 -80
- data/lib/insika/studio/views/session.erb +231 -177
- data/lib/insika/studio/views/settings.erb +50 -1
- data/lib/insika/studio/views/skills.erb +1 -1
- data/lib/insika/studio/views/tools.erb +24 -9
- data/lib/insika/telemetry/recorder.rb +49 -1
- data/lib/insika/templates/browser-agent/README.md +36 -0
- data/lib/insika/templates/browser-agent/agent.rb +49 -0
- data/lib/insika/templates/daily-digest/README.md +47 -0
- data/lib/insika/templates/daily-digest/agent.rb +77 -0
- data/lib/insika/templates/repo-explorer/README.md +36 -0
- data/lib/insika/templates/repo-explorer/agent.rb +45 -0
- data/lib/insika/templates/research-analyst/README.md +26 -0
- data/lib/insika/templates/research-analyst/agent.rb +68 -0
- data/lib/insika/templates/review-panel/README.md +20 -0
- data/lib/insika/templates/review-panel/agent.rb +50 -0
- data/lib/insika/templates/travel-planner/README.md +35 -0
- data/lib/insika/templates/travel-planner/agent.rb +87 -0
- data/lib/insika/templates.rb +112 -0
- data/lib/insika/tick.rb +24 -12
- data/lib/insika/timezone.rb +45 -0
- data/lib/insika/tool_batch.rb +67 -0
- data/lib/insika/tool_usage_report.rb +162 -0
- data/lib/insika/tools/generate_image.rb +52 -7
- data/lib/insika/tools/load_knowledge.rb +74 -0
- data/lib/insika/tools/run_persona_eval.rb +328 -0
- data/lib/insika/tools/save_artifact.rb +95 -0
- data/lib/insika/turn_budget.rb +91 -0
- data/lib/insika/turn_output.rb +1 -1
- data/lib/insika/turn_state.rb +15 -4
- data/lib/insika/version.rb +1 -1
- data/lib/insika/wiring/graph.rb +184 -12
- data/lib/insika/wiring/graph_chat.rb +102 -0
- data/lib/insika.rb +64 -0
- metadata +109 -5
- data/docs/build.md +0 -14
- data/docs/understand.md +0 -10
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Insika
|
|
4
|
+
module Evals
|
|
5
|
+
# SIMULATED USERS. Two models talking: the target agent
|
|
6
|
+
# (reached through the same Transport seam as the Runner) and the simulated
|
|
7
|
+
# customer (a cheap model — the platform utility_model — playing a `Persona`).
|
|
8
|
+
#
|
|
9
|
+
# Pure over its seams, like the Runner: the Transport is injected (HttpTransport
|
|
10
|
+
# for a remote deployment, GraphTransport for the own graph, A2ATransport for an
|
|
11
|
+
# agent that only speaks A2A) and the persona model is an injected `ask`
|
|
12
|
+
# (prompt -> raw text), so the loop is unit-testable offline.
|
|
13
|
+
#
|
|
14
|
+
# Termination is recorded, never guessed: the persona's `max_turns`, the persona
|
|
15
|
+
# emitting `<<goal_met>>` (its goal is served) or `<<gave_up>>` (it abandons),
|
|
16
|
+
# or an errored agent turn. "Gave up at turn 3" is a finding, not a detail.
|
|
17
|
+
#
|
|
18
|
+
# SAFETY (rule fixed in the spec): a simulated conversation must not write for
|
|
19
|
+
# real. The `Safety` gate refuses the run unless the target is declared
|
|
20
|
+
# `staging` or the run uses an eval profile where the agent's side-effect tools
|
|
21
|
+
# are swapped for fakes — and the swap list is DERIVED from the tool registry
|
|
22
|
+
# (the engine marks `side_effect` on tools; see `EvalProfile`), never
|
|
23
|
+
# hand-maintained. Every run is `simulated: true`, so a report never mixes a
|
|
24
|
+
# generated conversation with real traffic.
|
|
25
|
+
class Simulator
|
|
26
|
+
STOP_GOAL_MET = "<<goal_met>>"
|
|
27
|
+
STOP_GAVE_UP = "<<gave_up>>"
|
|
28
|
+
STOPS = { STOP_GOAL_MET => :goal_met, STOP_GAVE_UP => :gave_up }.freeze
|
|
29
|
+
|
|
30
|
+
# A generated conversation. `transcript` is [{ role: "user"|"assistant",
|
|
31
|
+
# text:, tools: [names] }] in order; `stop` is one of :goal_met | :gave_up |
|
|
32
|
+
# :max_turns | :error; `simulated` is ALWAYS true — the flag that keeps a
|
|
33
|
+
# report from mixing generated traffic with real conversations (rule D).
|
|
34
|
+
SimulatedRun = Struct.new(:transcript, :stop, :turns, :error, keyword_init: true) do
|
|
35
|
+
def simulated = true
|
|
36
|
+
def simulated? = true
|
|
37
|
+
|
|
38
|
+
def to_h
|
|
39
|
+
{ "simulated" => true, "stop" => stop.to_s, "turns" => turns, "error" => error,
|
|
40
|
+
"transcript" => transcript }.compact
|
|
41
|
+
end
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
# The target-safety gate. A run is allowed when:
|
|
45
|
+
# · `staging` — the operator declares the target is a staging
|
|
46
|
+
# deployment (real tools, staging data), or
|
|
47
|
+
# · the derived `side_effect_tools` is EMPTY — the agent has nothing
|
|
48
|
+
# that can write, or
|
|
49
|
+
# · `eval_profile` AND `swapped_tools` covers EVERY derived
|
|
50
|
+
# side-effect tool — the eval profile swaps them all for dry-runs.
|
|
51
|
+
# Anything else is refused with the offending tool names. Refusals are loud:
|
|
52
|
+
# `UnsafeTarget` is raised BEFORE a single model call.
|
|
53
|
+
#
|
|
54
|
+
# `side_effect_tools` is the DERIVED list (see EvalProfile) — never a
|
|
55
|
+
# hand-typed claim; `swapped_tools` is what the run's eval profile declares
|
|
56
|
+
# it swaps. A bare `eval_profile: true` with a known side-effect list is
|
|
57
|
+
# REFUSED: an eval profile that leaves a write-capable tool unswapped is a
|
|
58
|
+
# trust-me flag wearing a safety's clothes.
|
|
59
|
+
Safety = Struct.new(:staging, :eval_profile, :side_effect_tools, :swapped_tools, keyword_init: true) do
|
|
60
|
+
def self.staging = new(staging: true)
|
|
61
|
+
|
|
62
|
+
# -> reason to refuse, or nil. `side_effect_tools` is the DERIVED list (the
|
|
63
|
+
# target's reachable tools the registry marks `side_effect`).
|
|
64
|
+
def refusal
|
|
65
|
+
return nil if staging
|
|
66
|
+
|
|
67
|
+
tools = Array(side_effect_tools).map(&:to_s).reject(&:empty?)
|
|
68
|
+
return nil if tools.empty?
|
|
69
|
+
|
|
70
|
+
if eval_profile
|
|
71
|
+
swapped = Array(swapped_tools).map(&:to_s)
|
|
72
|
+
uncovered = tools - swapped
|
|
73
|
+
return nil if uncovered.empty?
|
|
74
|
+
|
|
75
|
+
return "eval profile declares swapped tool(s) (#{swapped.join(', ')}) but the target " \
|
|
76
|
+
"also exposes side-effect tool(s) (#{uncovered.join(', ')}) — an eval profile " \
|
|
77
|
+
"must swap EVERY side-effect tool"
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
"the target agent exposes side-effect tool(s) (#{tools.join(', ')}) — a simulated " \
|
|
81
|
+
"conversation could write for real. Run against staging (--staging) or against an " \
|
|
82
|
+
"eval profile where these tools are swapped for fakes (--eval-profile --eval-tools ...)."
|
|
83
|
+
end
|
|
84
|
+
end
|
|
85
|
+
|
|
86
|
+
# Raised before any model call when the Safety gate refuses.
|
|
87
|
+
class UnsafeTarget < StandardError; end
|
|
88
|
+
|
|
89
|
+
# transport: the target agent seam (TurnOutcome per turn, like the Runner).
|
|
90
|
+
# ask: ->(prompt) { text } — the simulated customer (the cheap model).
|
|
91
|
+
# safety: a Safety — the gate evaluated before every run.
|
|
92
|
+
def initialize(transport:, ask:, safety:)
|
|
93
|
+
@transport = transport
|
|
94
|
+
@ask = ask
|
|
95
|
+
@safety = safety
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
# Runs one simulated conversation. -> SimulatedRun.
|
|
99
|
+
def run(persona:, agent:, conv:)
|
|
100
|
+
reason = @safety.refusal
|
|
101
|
+
raise UnsafeTarget, reason if reason
|
|
102
|
+
|
|
103
|
+
transcript = []
|
|
104
|
+
message = persona.opens_with
|
|
105
|
+
stop = :max_turns
|
|
106
|
+
|
|
107
|
+
1.upto(persona.max_turns) do |turn|
|
|
108
|
+
transcript << { role: "user", text: message }
|
|
109
|
+
outcome = @transport.turn(agent: agent, conv: conv, message: message)
|
|
110
|
+
if outcome.result.error
|
|
111
|
+
transcript << { role: "assistant", text: "", tools: [] }
|
|
112
|
+
return SimulatedRun.new(transcript: transcript, stop: :error, turns: turn,
|
|
113
|
+
error: outcome.result.error)
|
|
114
|
+
end
|
|
115
|
+
transcript << { role: "assistant", text: outcome.result.output_text,
|
|
116
|
+
tools: outcome.result.tool_names }
|
|
117
|
+
break if turn == persona.max_turns # the persona's budget is spent
|
|
118
|
+
|
|
119
|
+
reply = @ask.call(persona.prompt(transcript)).to_s
|
|
120
|
+
marker, text = strip_stop(reply)
|
|
121
|
+
if marker
|
|
122
|
+
transcript << { role: "user", text: text } unless text.empty?
|
|
123
|
+
return SimulatedRun.new(transcript: transcript, stop: marker, turns: turn)
|
|
124
|
+
end
|
|
125
|
+
if text.empty?
|
|
126
|
+
return SimulatedRun.new(transcript: transcript, stop: :error, turns: turn,
|
|
127
|
+
error: "the persona produced an empty message")
|
|
128
|
+
end
|
|
129
|
+
|
|
130
|
+
message = text
|
|
131
|
+
end
|
|
132
|
+
|
|
133
|
+
SimulatedRun.new(transcript: transcript, stop: stop, turns: persona.max_turns)
|
|
134
|
+
end
|
|
135
|
+
|
|
136
|
+
private
|
|
137
|
+
|
|
138
|
+
# -> [stop_reason | nil, text_without_marker]. A trailing stop marker ends
|
|
139
|
+
# the conversation; the text before it is the customer's final message.
|
|
140
|
+
def strip_stop(reply)
|
|
141
|
+
text = reply.strip
|
|
142
|
+
STOPS.each do |marker, reason|
|
|
143
|
+
return [reason, text.delete_suffix(marker).strip] if text.end_with?(marker)
|
|
144
|
+
end
|
|
145
|
+
[nil, text]
|
|
146
|
+
end
|
|
147
|
+
end
|
|
148
|
+
|
|
149
|
+
# The DERIVED eval profile: which of an agent's reachable tools can write for
|
|
150
|
+
# real, computed from the tool registry — the engine marks `side_effect` on
|
|
151
|
+
# tools (a data-tool's non-GET method, the MCP ingestor's `tools/call`), so the
|
|
152
|
+
# swap list is a fact of the deployment, never a hand-maintained list.
|
|
153
|
+
module EvalProfile
|
|
154
|
+
module_function
|
|
155
|
+
|
|
156
|
+
# profile + registry (answers #names and #side_effect?) ->
|
|
157
|
+
# [tool names] the agent can reach that are marked side-effect, sorted.
|
|
158
|
+
def side_effect_tools(profile, registry)
|
|
159
|
+
allowed = if profile.tools_allow.nil? then Array(registry.names)
|
|
160
|
+
else Array(profile.tools_allow).map(&:to_s)
|
|
161
|
+
end
|
|
162
|
+
denied = Array(profile.tools_deny).map(&:to_s)
|
|
163
|
+
(allowed - denied).select { |name| registry.side_effect?(name) }.sort
|
|
164
|
+
end
|
|
165
|
+
|
|
166
|
+
# -> bool: can a simulated run touch this agent without a staging
|
|
167
|
+
# declaration (no reachable side-effect tool)?
|
|
168
|
+
def safe?(profile, registry)
|
|
169
|
+
side_effect_tools(profile, registry).empty?
|
|
170
|
+
end
|
|
171
|
+
|
|
172
|
+
# A registry overlay that answers the swapped names with a DRY-RUN tool and
|
|
173
|
+
# delegates everything else to the base — the "side_effect -> fake" half of
|
|
174
|
+
# the eval profile, derived from the base registry (nothing hand-maintained).
|
|
175
|
+
# The overlay is a drop-in for the Executor's registry: same
|
|
176
|
+
# entries/resolve/side_effect? surface.
|
|
177
|
+
def registry(base, side_effect_tools:, dry_run: nil)
|
|
178
|
+
swapped = Array(side_effect_tools).map(&:to_s)
|
|
179
|
+
fake = dry_run || ->(name) { Simulator::DryRunTool.new(name) }
|
|
180
|
+
OverlayRegistry.new(base: base, swapped: swapped, fake: fake)
|
|
181
|
+
end
|
|
182
|
+
end
|
|
183
|
+
|
|
184
|
+
# The overlay behind `EvalProfile.registry`. Kept as a named class so the
|
|
185
|
+
# constant is assigned once at load time, not inside the method.
|
|
186
|
+
class EvalProfile::OverlayRegistry
|
|
187
|
+
def initialize(base:, swapped:, fake:)
|
|
188
|
+
@base = base
|
|
189
|
+
@swapped = swapped
|
|
190
|
+
@fake = fake
|
|
191
|
+
end
|
|
192
|
+
|
|
193
|
+
def names = @base.names
|
|
194
|
+
def entries = @base.entries
|
|
195
|
+
def side_effect?(name) = @swapped.include?(name.to_s) ? false : @base.side_effect?(name)
|
|
196
|
+
|
|
197
|
+
def resolve(name)
|
|
198
|
+
key = name.to_s
|
|
199
|
+
return @fake.call(key) if @swapped.include?(key)
|
|
200
|
+
|
|
201
|
+
@base.resolve(key)
|
|
202
|
+
end
|
|
203
|
+
end
|
|
204
|
+
end
|
|
205
|
+
end
|
|
206
|
+
|
|
207
|
+
# The dry-run fake lives with the Simulator (it is its "side_effect -> fake"
|
|
208
|
+
# convention). It answers the same surface as a RubyLLM tool (`call`) and never
|
|
209
|
+
# performs the real side effect. Its result carries `dry_run: true` so a reader
|
|
210
|
+
# can tell a swapped call from a real one in the transcript's tool trace.
|
|
211
|
+
class Insika::Evals::Simulator::DryRunTool
|
|
212
|
+
def initialize(name, description: nil)
|
|
213
|
+
@name = name.to_s
|
|
214
|
+
@description = description ||
|
|
215
|
+
"DRY-RUN of #{@name} — disabled for this simulated run, returns a canned envelope"
|
|
216
|
+
end
|
|
217
|
+
|
|
218
|
+
def name = @name
|
|
219
|
+
def description = @description
|
|
220
|
+
|
|
221
|
+
def call(args)
|
|
222
|
+
{ "dry_run" => true, "tool" => @name, "simulated" => true,
|
|
223
|
+
"note" => "side-effect tool disabled by the eval profile — the real call was NOT performed" }
|
|
224
|
+
end
|
|
225
|
+
end
|
|
@@ -108,7 +108,8 @@ module Insika
|
|
|
108
108
|
return nil unless res.code.to_i == 200
|
|
109
109
|
|
|
110
110
|
body = JSON.parse(res.body)
|
|
111
|
-
{ "tools" => body["tools"], "capabilities" => Array(body["capabilities"])
|
|
111
|
+
{ "tools" => body["tools"], "capabilities" => Array(body["capabilities"]),
|
|
112
|
+
"side_effect_tools" => body["side_effect_tools"] }
|
|
112
113
|
rescue StandardError, JSON::ParserError
|
|
113
114
|
nil
|
|
114
115
|
end
|
|
@@ -134,7 +135,7 @@ module Insika
|
|
|
134
135
|
req["Authorization"] = "Bearer #{@token}"
|
|
135
136
|
req["Content-Type"] = "application/json"
|
|
136
137
|
req["Accept"] = "text/event-stream"
|
|
137
|
-
req.body = JSON.generate(model: "
|
|
138
|
+
req.body = JSON.generate(model: "insika:#{agent}", user: conv, stream: true, input: message)
|
|
138
139
|
|
|
139
140
|
t0 = mono
|
|
140
141
|
ttfb = nil
|
|
@@ -174,5 +175,86 @@ module Insika
|
|
|
174
175
|
ttfb: nil, total: (mono - t0) * 1000.0)
|
|
175
176
|
end
|
|
176
177
|
end
|
|
178
|
+
|
|
179
|
+
# A transport over the deployment's OWN graph, in-process — no HTTP. This is
|
|
180
|
+
# how a simulated conversation (or a replay) exercises the local agent the
|
|
181
|
+
# way a customer would reach it, tools and guardrails included, without a
|
|
182
|
+
# server in between. `runtime` is anything answering the DSL Runtime contract
|
|
183
|
+
# (#chat(message, session_id:, agent:) -> text, raising on failure) — the DSL
|
|
184
|
+
# Definition/System runtime, or a test double.
|
|
185
|
+
#
|
|
186
|
+
# Tool activity is captured from the graph's event stream (the `:tool_call`
|
|
187
|
+
# events the ChatBuilder emits), so an in-process transcript records the same
|
|
188
|
+
# tool names an HTTP replay would — the Simulator's transcript is not blind to
|
|
189
|
+
# what the local agent called. `event_stream` is optional; when omitted it is
|
|
190
|
+
# read off the runtime's graph when one is reachable.
|
|
191
|
+
class GraphTransport
|
|
192
|
+
def initialize(runtime:, event_stream: nil)
|
|
193
|
+
@runtime = runtime
|
|
194
|
+
@event_stream = event_stream ||
|
|
195
|
+
(runtime.graph.event_stream if runtime.respond_to?(:graph) && runtime.graph)
|
|
196
|
+
end
|
|
197
|
+
|
|
198
|
+
def mono = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
199
|
+
|
|
200
|
+
def turn(agent:, conv:, message:)
|
|
201
|
+
t0 = mono
|
|
202
|
+
sub = @event_stream&.subscribe(types: [:tool_call])
|
|
203
|
+
begin
|
|
204
|
+
text = @runtime.chat(message, session_id: conv, agent: agent)
|
|
205
|
+
TurnOutcome.new(
|
|
206
|
+
result: TurnResult.new(output_text: text.to_s, tool_calls: drain_tools(sub), error: nil),
|
|
207
|
+
ttfb: nil, total: (mono - t0) * 1000.0, usage: nil
|
|
208
|
+
)
|
|
209
|
+
rescue Insika::Error => e
|
|
210
|
+
TurnOutcome.new(
|
|
211
|
+
result: TurnResult.new(output_text: "", tool_calls: drain_tools(sub), error: e.message),
|
|
212
|
+
ttfb: nil, total: (mono - t0) * 1000.0, usage: nil
|
|
213
|
+
)
|
|
214
|
+
ensure
|
|
215
|
+
sub&.close
|
|
216
|
+
end
|
|
217
|
+
end
|
|
218
|
+
|
|
219
|
+
private
|
|
220
|
+
|
|
221
|
+
# The same shape an HTTP replay produces: [{ "name" =>, "status" => nil }]
|
|
222
|
+
# (the stream carries no per-tool status — that lives in the trace store).
|
|
223
|
+
def drain_tools(sub)
|
|
224
|
+
return [] if sub.nil?
|
|
225
|
+
|
|
226
|
+
sub.drain_nonblocking.map { |ev| { "name" => ev.data[:name].to_s, "status" => nil } }
|
|
227
|
+
end
|
|
228
|
+
end
|
|
229
|
+
|
|
230
|
+
# Thin A2A transport: drives a REMOTE A2A agent (an agent that only speaks
|
|
231
|
+
# A2A) through the same `turn` seam the Simulator uses. The outbound A2A
|
|
232
|
+
# client already does the send+poll; this is the wrapper that makes it a
|
|
233
|
+
# Transport. `client` answers `#call(url, text, context_id:)` -> {text:} or
|
|
234
|
+
# {error:} — the `Server::A2A::Client` shape.
|
|
235
|
+
class A2ATransport
|
|
236
|
+
def initialize(client:, url:)
|
|
237
|
+
@client = client
|
|
238
|
+
@url = url
|
|
239
|
+
end
|
|
240
|
+
|
|
241
|
+
def mono = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
242
|
+
|
|
243
|
+
def turn(agent:, conv:, message:)
|
|
244
|
+
t0 = mono
|
|
245
|
+
result = @client.call(@url, message.to_s, context_id: conv)
|
|
246
|
+
if result[:error]
|
|
247
|
+
TurnOutcome.new(
|
|
248
|
+
result: TurnResult.new(output_text: "", tool_calls: [], error: result[:error].to_s),
|
|
249
|
+
ttfb: nil, total: (mono - t0) * 1000.0, usage: nil
|
|
250
|
+
)
|
|
251
|
+
else
|
|
252
|
+
TurnOutcome.new(
|
|
253
|
+
result: TurnResult.new(output_text: result[:text].to_s, tool_calls: [], error: nil),
|
|
254
|
+
ttfb: nil, total: (mono - t0) * 1000.0, usage: nil
|
|
255
|
+
)
|
|
256
|
+
end
|
|
257
|
+
end
|
|
258
|
+
end
|
|
177
259
|
end
|
|
178
260
|
end
|
data/lib/insika/event_stream.rb
CHANGED
|
@@ -85,6 +85,16 @@ module Insika
|
|
|
85
85
|
end
|
|
86
86
|
end
|
|
87
87
|
|
|
88
|
+
# Drains whatever is ALREADY queued without ever blocking. Safe in a
|
|
89
|
+
# cooperative reactor: between the `empty?` check and the `dequeue` no other
|
|
90
|
+
# fiber runs, so a non-empty dequeue never waits. The eval transports use it
|
|
91
|
+
# to collect the events a turn already emitted, AFTER the turn returned.
|
|
92
|
+
def drain_nonblocking
|
|
93
|
+
drained = []
|
|
94
|
+
drained << @queue.dequeue until @queue.empty?
|
|
95
|
+
drained
|
|
96
|
+
end
|
|
97
|
+
|
|
88
98
|
# Idempotent: a second CLOSED is harmless (the `each` stops at the first).
|
|
89
99
|
# `@on_close` fires only once (avoids removing the subscription twice).
|
|
90
100
|
def close
|