insika 0.3.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +180 -0
- data/README.md +45 -10
- data/bin/insika +684 -0
- data/bin/insika-router +87 -0
- data/docs/AGENTS.md +94 -403
- data/docs/API.md +5 -5
- data/docs/ARCHITECTURE.md +3 -2
- data/docs/ARTIFACTS.md +95 -0
- data/docs/BENCHMARK.md +2 -2
- data/docs/CHANNELS.md +14 -14
- data/docs/CONTEXT.md +9 -7
- data/docs/DEMO.md +80 -0
- data/docs/DEPLOY.md +71 -3
- data/docs/EMBEDDING.md +1 -1
- data/docs/EVALS.md +128 -3
- data/docs/FACTS.md +3 -3
- data/docs/HARVEST.md +5 -6
- data/docs/KNOWLEDGE.md +290 -0
- data/docs/LOADTEST.md +2 -2
- data/docs/MEDIA.md +128 -0
- data/docs/OBSERVABILITY.md +15 -10
- data/docs/OUTCOMES.md +137 -0
- data/docs/PLUGINS.md +51 -6
- data/docs/POLICY.md +216 -0
- data/docs/REFINEMENT.md +14 -9
- data/docs/RELEASING.md +4 -4
- data/docs/ROUTER.md +213 -0
- data/docs/RUNNING-LOCAL.md +3 -3
- data/docs/SCHEDULING.md +121 -0
- data/docs/SECURITY.md +22 -6
- data/docs/SKILLS.md +11 -2
- data/docs/SOAK.md +2 -2
- data/docs/TEMPLATES.md +134 -0
- data/docs/TOOLS.md +152 -27
- data/docs/WHY.md +1 -1
- data/docs/WORKFLOWS.md +2 -2
- data/docs/_includes/head_custom.html +5 -0
- data/docs/_includes/title.html +13 -0
- data/docs/_sass/color_schemes/insika.scss +32 -0
- data/docs/_sass/custom/custom.scss +199 -0
- data/docs/_sass/custom/setup.scss +26 -0
- data/docs/assets/img/favicon.svg +7 -0
- data/docs/assets/img/insika-mark.svg +7 -0
- data/docs/core-concepts.md +21 -0
- data/docs/domain.md +4 -4
- data/docs/improve.md +20 -0
- data/docs/index.md +8 -5
- data/docs/integrate.md +20 -0
- data/docs/operate.md +13 -6
- data/docs/prompts/ADD-TOOL.md +118 -0
- data/docs/prompts/DIAGNOSE-TURN.md +65 -0
- data/docs/prompts/GO-LIVE.md +138 -0
- data/docs/prompts/RUN-EXAMPLES.md +70 -0
- data/docs/reference.md +19 -0
- data/docs/ship.md +10 -2
- data/docs/start-here.md +18 -0
- data/lib/insika/agent_profile.rb +73 -16
- data/lib/insika/artifact_signing.rb +82 -0
- data/lib/insika/artifact_store.rb +160 -0
- data/lib/insika/channel_delivery.rb +1 -1
- data/lib/insika/chat_builder.rb +22 -2
- data/lib/insika/commands/agent_payload.rb +2 -2
- data/lib/insika/commands/backfill_knowledge.rb +145 -0
- data/lib/insika/commands/delete_artifact.rb +35 -0
- data/lib/insika/commands/delete_concept.rb +34 -0
- data/lib/insika/commands/delete_mcp.rb +6 -2
- data/lib/insika/commands/delete_tenant_data.rb +15 -3
- data/lib/insika/commands/gate_refinement.rb +1 -1
- data/lib/insika/commands/refresh_mcp_tools.rb +47 -0
- data/lib/insika/commands/restore_concept.rb +34 -0
- data/lib/insika/commands/seed_demo_data.rb +31 -0
- data/lib/insika/commands/upsert_mcp.rb +6 -3
- data/lib/insika/commands/write_concept.rb +57 -0
- data/lib/insika/context/priority.rb +2 -0
- data/lib/insika/context/providers/knowledge.rb +108 -0
- data/lib/insika/context/providers/prompt.rb +30 -24
- data/lib/insika/cron.rb +189 -0
- data/lib/insika/demo/agent_attrs.rb +43 -0
- data/lib/insika/demo/golden_cases.rb +81 -0
- data/lib/insika/demo/seeder.rb +336 -0
- data/lib/insika/doctor.rb +176 -8
- data/lib/insika/dsl/definition.rb +3 -2
- data/lib/insika/dsl/runtime.rb +60 -79
- data/lib/insika/dsl/server_boot.rb +23 -1
- data/lib/insika/dsl/system.rb +10 -2
- data/lib/insika/dsl.rb +103 -2
- data/lib/insika/env_schema.rb +16 -1
- data/lib/insika/evals/golden.rb +41 -4
- data/lib/insika/evals/judge.rb +47 -2
- data/lib/insika/evals/pairwise.rb +11 -0
- data/lib/insika/evals/persona.rb +98 -0
- data/lib/insika/evals/runner.rb +9 -0
- data/lib/insika/evals/simulator.rb +225 -0
- data/lib/insika/evals/transport.rb +83 -1
- data/lib/insika/event_stream.rb +10 -0
- data/lib/insika/executor.rb +231 -55
- data/lib/insika/followup_policy.rb +2 -25
- data/lib/insika/golden_store.rb +16 -1
- data/lib/insika/grounding/matcher.rb +1 -1
- data/lib/insika/knowledge.rb +680 -0
- data/lib/insika/knowledge_store.rb +140 -0
- data/lib/insika/mcp_client.rb +94 -0
- data/lib/insika/mcp_json.rb +74 -0
- data/lib/insika/mcp_live_tool.rb +43 -0
- data/lib/insika/mcp_store.rb +98 -26
- data/lib/insika/mcp_tool_ingestor.rb +30 -8
- data/lib/insika/mcp_tool_registry.rb +100 -0
- data/lib/insika/media.rb +115 -31
- data/lib/insika/message_origin.rb +1 -1
- data/lib/insika/middleware.rb +9 -0
- data/lib/insika/onboarding.rb +17 -1
- data/lib/insika/outcome_store.rb +1 -1
- data/lib/insika/overlay_tool_registry.rb +37 -17
- data/lib/insika/packaging.rb +2 -2
- data/lib/insika/profile_source.rb +8 -1
- data/lib/insika/prompt_catalog.rb +10 -0
- data/lib/insika/retention.rb +36 -1
- data/lib/insika/router/app.rb +157 -0
- data/lib/insika/router/backend_pool.rb +98 -0
- data/lib/insika/router/hash_ring.rb +55 -0
- data/lib/insika/router/proxy_body.rb +34 -0
- data/lib/insika/router/session_key.rb +54 -0
- data/lib/insika/router.rb +18 -0
- data/lib/insika/schedule.rb +177 -0
- data/lib/insika/schedule_engine.rb +314 -0
- data/lib/insika/schedule_store.rb +208 -0
- data/lib/insika/server/app.rb +105 -15
- data/lib/insika/server/rack_app.rb +5 -1
- data/lib/insika/server/responses.rb +1 -1
- data/lib/insika/skill_catalog.rb +12 -0
- data/lib/insika/steer_injector.rb +21 -10
- data/lib/insika/studio/app.rb +567 -45
- data/lib/insika/studio/assets/dist/application.css +1 -1
- data/lib/insika/studio/assets/dist/application.js +21 -21
- data/lib/insika/studio/forms.rb +46 -5
- data/lib/insika/studio/nav_icons.rb +14 -1
- data/lib/insika/studio/views/_agent_tab_cache.erb +25 -0
- data/lib/insika/studio/views/_agent_tab_config.erb +514 -0
- data/lib/insika/studio/views/_agent_tab_history.erb +24 -0
- data/lib/insika/studio/views/_agent_tab_loops.erb +54 -0
- data/lib/insika/studio/views/_agent_tab_memory.erb +51 -0
- data/lib/insika/studio/views/_agent_tab_outcomes.erb +31 -0
- data/lib/insika/studio/views/_agent_tab_prompts.erb +108 -0
- data/lib/insika/studio/views/_agent_tab_skills.erb +38 -0
- data/lib/insika/studio/views/_agents_master.erb +44 -0
- data/lib/insika/studio/views/_message.erb +49 -32
- data/lib/insika/studio/views/agent_detail.erb +61 -820
- data/lib/insika/studio/views/agents.erb +70 -57
- data/lib/insika/studio/views/artifact.erb +23 -0
- data/lib/insika/studio/views/artifacts.erb +59 -0
- data/lib/insika/studio/views/evals.erb +2 -2
- data/lib/insika/studio/views/facts.erb +1 -1
- data/lib/insika/studio/views/funnel.erb +1 -1
- data/lib/insika/studio/views/home.erb +106 -67
- data/lib/insika/studio/views/knowledge.erb +123 -0
- data/lib/insika/studio/views/layout.erb +14 -11
- data/lib/insika/studio/views/mcp.erb +174 -80
- data/lib/insika/studio/views/session.erb +231 -177
- data/lib/insika/studio/views/settings.erb +39 -1
- data/lib/insika/studio/views/skills.erb +1 -1
- data/lib/insika/studio/views/tools.erb +24 -9
- data/lib/insika/templates/browser-agent/README.md +36 -0
- data/lib/insika/templates/browser-agent/agent.rb +49 -0
- data/lib/insika/templates/daily-digest/README.md +38 -0
- data/lib/insika/templates/daily-digest/agent.rb +77 -0
- data/lib/insika/templates/repo-explorer/README.md +36 -0
- data/lib/insika/templates/repo-explorer/agent.rb +45 -0
- data/lib/insika/templates/research-analyst/README.md +26 -0
- data/lib/insika/templates/research-analyst/agent.rb +58 -0
- data/lib/insika/templates/review-panel/README.md +20 -0
- data/lib/insika/templates/review-panel/agent.rb +50 -0
- data/lib/insika/templates/travel-planner/README.md +35 -0
- data/lib/insika/templates/travel-planner/agent.rb +87 -0
- data/lib/insika/templates.rb +112 -0
- data/lib/insika/tick.rb +24 -12
- data/lib/insika/timezone.rb +45 -0
- data/lib/insika/tools/generate_image.rb +52 -7
- data/lib/insika/tools/load_knowledge.rb +74 -0
- data/lib/insika/tools/run_persona_eval.rb +328 -0
- data/lib/insika/tools/save_artifact.rb +95 -0
- data/lib/insika/turn_output.rb +1 -1
- data/lib/insika/turn_state.rb +15 -4
- data/lib/insika/version.rb +1 -1
- data/lib/insika/wiring/graph.rb +184 -12
- data/lib/insika/wiring/graph_chat.rb +102 -0
- data/lib/insika.rb +57 -0
- metadata +105 -5
- data/docs/build.md +0 -14
- data/docs/understand.md +0 -10
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Insika
|
|
4
|
+
module Evals
|
|
5
|
+
# A SIMULATED CUSTOMER — the data that turns a scripted case into a
|
|
6
|
+
# generated conversation. Pure data: goal, style, the opening
|
|
7
|
+
# message, the ONLY facts the persona may assert, and a hard turn cap. The
|
|
8
|
+
# persona is played by a model (the cheap utility_model) with this as its whole
|
|
9
|
+
# instruction; the anti-invention rule below is the soul of the feature — a
|
|
10
|
+
# simulator that invents an order number produces a conversation the agent
|
|
11
|
+
# could never have had, and a case that tests nothing.
|
|
12
|
+
Persona = Struct.new(:goal, :style, :opens_with, :knows, :max_turns, keyword_init: true) do
|
|
13
|
+
def to_h
|
|
14
|
+
{ "goal" => goal, "style" => style, "opens_with" => opens_with,
|
|
15
|
+
"knows" => knows, "max_turns" => max_turns }.compact
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
# The persona as a whole instruction. The `knows` facts are the ONLY
|
|
19
|
+
# assertions the persona may make; anything else is answered with
|
|
20
|
+
# ignorance — "não sei", "não tenho isso aqui" — exactly like a real
|
|
21
|
+
# customer who does not have the fact. `transcript` is the conversation
|
|
22
|
+
# so far, in order, as [{ role: "user"|"assistant", text: }].
|
|
23
|
+
def prompt(transcript)
|
|
24
|
+
facts = knows.map { |k, v| "- #{k}: #{v}" }.join("\n")
|
|
25
|
+
<<~PROMPT
|
|
26
|
+
You are simulating a customer in a chat with a store's virtual assistant.
|
|
27
|
+
Stay in character. You are not helping the assistant; you are the person
|
|
28
|
+
it serves.
|
|
29
|
+
|
|
30
|
+
GOAL: #{goal}
|
|
31
|
+
STYLE: #{style || "short, natural messages; answers what is asked"}
|
|
32
|
+
|
|
33
|
+
FACTS YOU KNOW — the ONLY facts you may assert:
|
|
34
|
+
#{facts}
|
|
35
|
+
|
|
36
|
+
RULES:
|
|
37
|
+
1. You may ONLY assert the facts above. Asked about anything else, you do
|
|
38
|
+
not know it — answer with ignorance, like a real customer without that
|
|
39
|
+
fact (you have no order number, no name, no date, no price beyond the
|
|
40
|
+
facts above). Never invent an order number, a name, a date, a price or
|
|
41
|
+
any other detail that is not in FACTS YOU KNOW.
|
|
42
|
+
2. Reply with ONLY the customer's next message.
|
|
43
|
+
3. When your goal has been met, end the message with the marker
|
|
44
|
+
<<goal_met>>. When you give up (the assistant cannot get you there),
|
|
45
|
+
end the message with the marker <<gave_up>>. Otherwise end with no
|
|
46
|
+
marker.
|
|
47
|
+
|
|
48
|
+
CONVERSATION SO FAR:
|
|
49
|
+
#{transcript.map { |m| "#{m[:role]}: #{m[:text]}" }.join("\n")}
|
|
50
|
+
|
|
51
|
+
Your next message:
|
|
52
|
+
PROMPT
|
|
53
|
+
end
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
# Loads + validates a persona mapping (the `persona:` key of a golden, or the
|
|
57
|
+
# `--persona` file of the simulate CLI). Fails LOUD on a malformed persona —
|
|
58
|
+
# a silently relaxed max_turns or a missing knows would produce a simulation
|
|
59
|
+
# that tests nothing.
|
|
60
|
+
module PersonaLoader
|
|
61
|
+
class InvalidPersona < StandardError; end
|
|
62
|
+
|
|
63
|
+
module_function
|
|
64
|
+
|
|
65
|
+
def build(raw, source: "(inline)")
|
|
66
|
+
raise InvalidPersona, "#{source}: persona must be a mapping" unless raw.is_a?(Hash)
|
|
67
|
+
|
|
68
|
+
goal = presence(raw["goal"])
|
|
69
|
+
raise InvalidPersona, "#{source}: persona needs a non-empty 'goal'" if goal.nil?
|
|
70
|
+
|
|
71
|
+
knows = raw["knows"]
|
|
72
|
+
unless knows.is_a?(Hash) && !knows.empty?
|
|
73
|
+
raise InvalidPersona, "#{source}: persona needs a non-empty 'knows' mapping (the only facts it may assert)"
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
opens = presence(raw["opens_with"])
|
|
77
|
+
raise InvalidPersona, "#{source}: persona needs a non-empty 'opens_with'" if opens.nil?
|
|
78
|
+
|
|
79
|
+
max = raw["max_turns"]
|
|
80
|
+
unless max.is_a?(Integer) && max.positive?
|
|
81
|
+
raise InvalidPersona, "#{source}: persona needs 'max_turns' as a positive integer"
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
Persona.new(
|
|
85
|
+
goal: goal, style: presence(raw["style"]),
|
|
86
|
+
opens_with: opens,
|
|
87
|
+
knows: knows.transform_keys(&:to_s).transform_values(&:to_s),
|
|
88
|
+
max_turns: max
|
|
89
|
+
)
|
|
90
|
+
end
|
|
91
|
+
|
|
92
|
+
def presence(v)
|
|
93
|
+
s = v.to_s.strip
|
|
94
|
+
s.empty? ? nil : s
|
|
95
|
+
end
|
|
96
|
+
end
|
|
97
|
+
end
|
|
98
|
+
end
|
data/lib/insika/evals/runner.rb
CHANGED
|
@@ -50,6 +50,15 @@ module Insika
|
|
|
50
50
|
end
|
|
51
51
|
|
|
52
52
|
def run_case(golden)
|
|
53
|
+
# A persona case is GENERATED, not replayed: the turns do not exist until a
|
|
54
|
+
# Simulator drives the conversation. The replay Runner cannot run it, and a
|
|
55
|
+
# silent no-op would read as a pass — so it is SKIPPED with the reason, and
|
|
56
|
+
# the Simulator (the simulate CLI) is the only driver.
|
|
57
|
+
if golden.simulated?
|
|
58
|
+
return RunCase.new(result: Assertions.skip(golden, "simulated case (persona) — drive it with `insika evals:simulate`"),
|
|
59
|
+
timings: [])
|
|
60
|
+
end
|
|
61
|
+
|
|
53
62
|
skip = skip_reason(golden)
|
|
54
63
|
return RunCase.new(result: Assertions.skip(golden, skip), timings: []) if skip
|
|
55
64
|
|
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Insika
|
|
4
|
+
module Evals
|
|
5
|
+
# SIMULATED USERS. Two models talking: the target agent
|
|
6
|
+
# (reached through the same Transport seam as the Runner) and the simulated
|
|
7
|
+
# customer (a cheap model — the platform utility_model — playing a `Persona`).
|
|
8
|
+
#
|
|
9
|
+
# Pure over its seams, like the Runner: the Transport is injected (HttpTransport
|
|
10
|
+
# for a remote deployment, GraphTransport for the own graph, A2ATransport for an
|
|
11
|
+
# agent that only speaks A2A) and the persona model is an injected `ask`
|
|
12
|
+
# (prompt -> raw text), so the loop is unit-testable offline.
|
|
13
|
+
#
|
|
14
|
+
# Termination is recorded, never guessed: the persona's `max_turns`, the persona
|
|
15
|
+
# emitting `<<goal_met>>` (its goal is served) or `<<gave_up>>` (it abandons),
|
|
16
|
+
# or an errored agent turn. "Gave up at turn 3" is a finding, not a detail.
|
|
17
|
+
#
|
|
18
|
+
# SAFETY (rule fixed in the spec): a simulated conversation must not write for
|
|
19
|
+
# real. The `Safety` gate refuses the run unless the target is declared
|
|
20
|
+
# `staging` or the run uses an eval profile where the agent's side-effect tools
|
|
21
|
+
# are swapped for fakes — and the swap list is DERIVED from the tool registry
|
|
22
|
+
# (the engine marks `side_effect` on tools; see `EvalProfile`), never
|
|
23
|
+
# hand-maintained. Every run is `simulated: true`, so a report never mixes a
|
|
24
|
+
# generated conversation with real traffic.
|
|
25
|
+
class Simulator
|
|
26
|
+
STOP_GOAL_MET = "<<goal_met>>"
|
|
27
|
+
STOP_GAVE_UP = "<<gave_up>>"
|
|
28
|
+
STOPS = { STOP_GOAL_MET => :goal_met, STOP_GAVE_UP => :gave_up }.freeze
|
|
29
|
+
|
|
30
|
+
# A generated conversation. `transcript` is [{ role: "user"|"assistant",
|
|
31
|
+
# text:, tools: [names] }] in order; `stop` is one of :goal_met | :gave_up |
|
|
32
|
+
# :max_turns | :error; `simulated` is ALWAYS true — the flag that keeps a
|
|
33
|
+
# report from mixing generated traffic with real conversations (rule D).
|
|
34
|
+
SimulatedRun = Struct.new(:transcript, :stop, :turns, :error, keyword_init: true) do
|
|
35
|
+
def simulated = true
|
|
36
|
+
def simulated? = true
|
|
37
|
+
|
|
38
|
+
def to_h
|
|
39
|
+
{ "simulated" => true, "stop" => stop.to_s, "turns" => turns, "error" => error,
|
|
40
|
+
"transcript" => transcript }.compact
|
|
41
|
+
end
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
# The target-safety gate. A run is allowed when:
|
|
45
|
+
# · `staging` — the operator declares the target is a staging
|
|
46
|
+
# deployment (real tools, staging data), or
|
|
47
|
+
# · the derived `side_effect_tools` is EMPTY — the agent has nothing
|
|
48
|
+
# that can write, or
|
|
49
|
+
# · `eval_profile` AND `swapped_tools` covers EVERY derived
|
|
50
|
+
# side-effect tool — the eval profile swaps them all for dry-runs.
|
|
51
|
+
# Anything else is refused with the offending tool names. Refusals are loud:
|
|
52
|
+
# `UnsafeTarget` is raised BEFORE a single model call.
|
|
53
|
+
#
|
|
54
|
+
# `side_effect_tools` is the DERIVED list (see EvalProfile) — never a
|
|
55
|
+
# hand-typed claim; `swapped_tools` is what the run's eval profile declares
|
|
56
|
+
# it swaps. A bare `eval_profile: true` with a known side-effect list is
|
|
57
|
+
# REFUSED: an eval profile that leaves a write-capable tool unswapped is a
|
|
58
|
+
# trust-me flag wearing a safety's clothes.
|
|
59
|
+
Safety = Struct.new(:staging, :eval_profile, :side_effect_tools, :swapped_tools, keyword_init: true) do
|
|
60
|
+
def self.staging = new(staging: true)
|
|
61
|
+
|
|
62
|
+
# -> reason to refuse, or nil. `side_effect_tools` is the DERIVED list (the
|
|
63
|
+
# target's reachable tools the registry marks `side_effect`).
|
|
64
|
+
def refusal
|
|
65
|
+
return nil if staging
|
|
66
|
+
|
|
67
|
+
tools = Array(side_effect_tools).map(&:to_s).reject(&:empty?)
|
|
68
|
+
return nil if tools.empty?
|
|
69
|
+
|
|
70
|
+
if eval_profile
|
|
71
|
+
swapped = Array(swapped_tools).map(&:to_s)
|
|
72
|
+
uncovered = tools - swapped
|
|
73
|
+
return nil if uncovered.empty?
|
|
74
|
+
|
|
75
|
+
return "eval profile declares swapped tool(s) (#{swapped.join(', ')}) but the target " \
|
|
76
|
+
"also exposes side-effect tool(s) (#{uncovered.join(', ')}) — an eval profile " \
|
|
77
|
+
"must swap EVERY side-effect tool"
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
"the target agent exposes side-effect tool(s) (#{tools.join(', ')}) — a simulated " \
|
|
81
|
+
"conversation could write for real. Run against staging (--staging) or against an " \
|
|
82
|
+
"eval profile where these tools are swapped for fakes (--eval-profile --eval-tools ...)."
|
|
83
|
+
end
|
|
84
|
+
end
|
|
85
|
+
|
|
86
|
+
# Raised before any model call when the Safety gate refuses.
|
|
87
|
+
class UnsafeTarget < StandardError; end
|
|
88
|
+
|
|
89
|
+
# transport: the target agent seam (TurnOutcome per turn, like the Runner).
|
|
90
|
+
# ask: ->(prompt) { text } — the simulated customer (the cheap model).
|
|
91
|
+
# safety: a Safety — the gate evaluated before every run.
|
|
92
|
+
def initialize(transport:, ask:, safety:)
|
|
93
|
+
@transport = transport
|
|
94
|
+
@ask = ask
|
|
95
|
+
@safety = safety
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
# Runs one simulated conversation. -> SimulatedRun.
|
|
99
|
+
def run(persona:, agent:, conv:)
|
|
100
|
+
reason = @safety.refusal
|
|
101
|
+
raise UnsafeTarget, reason if reason
|
|
102
|
+
|
|
103
|
+
transcript = []
|
|
104
|
+
message = persona.opens_with
|
|
105
|
+
stop = :max_turns
|
|
106
|
+
|
|
107
|
+
1.upto(persona.max_turns) do |turn|
|
|
108
|
+
transcript << { role: "user", text: message }
|
|
109
|
+
outcome = @transport.turn(agent: agent, conv: conv, message: message)
|
|
110
|
+
if outcome.result.error
|
|
111
|
+
transcript << { role: "assistant", text: "", tools: [] }
|
|
112
|
+
return SimulatedRun.new(transcript: transcript, stop: :error, turns: turn,
|
|
113
|
+
error: outcome.result.error)
|
|
114
|
+
end
|
|
115
|
+
transcript << { role: "assistant", text: outcome.result.output_text,
|
|
116
|
+
tools: outcome.result.tool_names }
|
|
117
|
+
break if turn == persona.max_turns # the persona's budget is spent
|
|
118
|
+
|
|
119
|
+
reply = @ask.call(persona.prompt(transcript)).to_s
|
|
120
|
+
marker, text = strip_stop(reply)
|
|
121
|
+
if marker
|
|
122
|
+
transcript << { role: "user", text: text } unless text.empty?
|
|
123
|
+
return SimulatedRun.new(transcript: transcript, stop: marker, turns: turn)
|
|
124
|
+
end
|
|
125
|
+
if text.empty?
|
|
126
|
+
return SimulatedRun.new(transcript: transcript, stop: :error, turns: turn,
|
|
127
|
+
error: "the persona produced an empty message")
|
|
128
|
+
end
|
|
129
|
+
|
|
130
|
+
message = text
|
|
131
|
+
end
|
|
132
|
+
|
|
133
|
+
SimulatedRun.new(transcript: transcript, stop: stop, turns: persona.max_turns)
|
|
134
|
+
end
|
|
135
|
+
|
|
136
|
+
private
|
|
137
|
+
|
|
138
|
+
# -> [stop_reason | nil, text_without_marker]. A trailing stop marker ends
|
|
139
|
+
# the conversation; the text before it is the customer's final message.
|
|
140
|
+
def strip_stop(reply)
|
|
141
|
+
text = reply.strip
|
|
142
|
+
STOPS.each do |marker, reason|
|
|
143
|
+
return [reason, text.delete_suffix(marker).strip] if text.end_with?(marker)
|
|
144
|
+
end
|
|
145
|
+
[nil, text]
|
|
146
|
+
end
|
|
147
|
+
end
|
|
148
|
+
|
|
149
|
+
# The DERIVED eval profile: which of an agent's reachable tools can write for
|
|
150
|
+
# real, computed from the tool registry — the engine marks `side_effect` on
|
|
151
|
+
# tools (a data-tool's non-GET method, the MCP ingestor's `tools/call`), so the
|
|
152
|
+
# swap list is a fact of the deployment, never a hand-maintained list.
|
|
153
|
+
module EvalProfile
|
|
154
|
+
module_function
|
|
155
|
+
|
|
156
|
+
# profile + registry (answers #names and #side_effect?) ->
|
|
157
|
+
# [tool names] the agent can reach that are marked side-effect, sorted.
|
|
158
|
+
def side_effect_tools(profile, registry)
|
|
159
|
+
allowed = if profile.tools_allow.nil? then Array(registry.names)
|
|
160
|
+
else Array(profile.tools_allow).map(&:to_s)
|
|
161
|
+
end
|
|
162
|
+
denied = Array(profile.tools_deny).map(&:to_s)
|
|
163
|
+
(allowed - denied).select { |name| registry.side_effect?(name) }.sort
|
|
164
|
+
end
|
|
165
|
+
|
|
166
|
+
# -> bool: can a simulated run touch this agent without a staging
|
|
167
|
+
# declaration (no reachable side-effect tool)?
|
|
168
|
+
def safe?(profile, registry)
|
|
169
|
+
side_effect_tools(profile, registry).empty?
|
|
170
|
+
end
|
|
171
|
+
|
|
172
|
+
# A registry overlay that answers the swapped names with a DRY-RUN tool and
|
|
173
|
+
# delegates everything else to the base — the "side_effect -> fake" half of
|
|
174
|
+
# the eval profile, derived from the base registry (nothing hand-maintained).
|
|
175
|
+
# The overlay is a drop-in for the Executor's registry: same
|
|
176
|
+
# entries/resolve/side_effect? surface.
|
|
177
|
+
def registry(base, side_effect_tools:, dry_run: nil)
|
|
178
|
+
swapped = Array(side_effect_tools).map(&:to_s)
|
|
179
|
+
fake = dry_run || ->(name) { Simulator::DryRunTool.new(name) }
|
|
180
|
+
OverlayRegistry.new(base: base, swapped: swapped, fake: fake)
|
|
181
|
+
end
|
|
182
|
+
end
|
|
183
|
+
|
|
184
|
+
# The overlay behind `EvalProfile.registry`. Kept as a named class so the
|
|
185
|
+
# constant is assigned once at load time, not inside the method.
|
|
186
|
+
class EvalProfile::OverlayRegistry
|
|
187
|
+
def initialize(base:, swapped:, fake:)
|
|
188
|
+
@base = base
|
|
189
|
+
@swapped = swapped
|
|
190
|
+
@fake = fake
|
|
191
|
+
end
|
|
192
|
+
|
|
193
|
+
def names = @base.names
|
|
194
|
+
def entries = @base.entries
|
|
195
|
+
def side_effect?(name) = @swapped.include?(name.to_s) ? false : @base.side_effect?(name)
|
|
196
|
+
|
|
197
|
+
def resolve(name)
|
|
198
|
+
key = name.to_s
|
|
199
|
+
return @fake.call(key) if @swapped.include?(key)
|
|
200
|
+
|
|
201
|
+
@base.resolve(key)
|
|
202
|
+
end
|
|
203
|
+
end
|
|
204
|
+
end
|
|
205
|
+
end
|
|
206
|
+
|
|
207
|
+
# The dry-run fake lives with the Simulator (it is its "side_effect -> fake"
|
|
208
|
+
# convention). It answers the same surface as a RubyLLM tool (`call`) and never
|
|
209
|
+
# performs the real side effect. Its result carries `dry_run: true` so a reader
|
|
210
|
+
# can tell a swapped call from a real one in the transcript's tool trace.
|
|
211
|
+
class Insika::Evals::Simulator::DryRunTool
|
|
212
|
+
def initialize(name, description: nil)
|
|
213
|
+
@name = name.to_s
|
|
214
|
+
@description = description ||
|
|
215
|
+
"DRY-RUN of #{@name} — disabled for this simulated run, returns a canned envelope"
|
|
216
|
+
end
|
|
217
|
+
|
|
218
|
+
def name = @name
|
|
219
|
+
def description = @description
|
|
220
|
+
|
|
221
|
+
def call(args)
|
|
222
|
+
{ "dry_run" => true, "tool" => @name, "simulated" => true,
|
|
223
|
+
"note" => "side-effect tool disabled by the eval profile — the real call was NOT performed" }
|
|
224
|
+
end
|
|
225
|
+
end
|
|
@@ -108,7 +108,8 @@ module Insika
|
|
|
108
108
|
return nil unless res.code.to_i == 200
|
|
109
109
|
|
|
110
110
|
body = JSON.parse(res.body)
|
|
111
|
-
{ "tools" => body["tools"], "capabilities" => Array(body["capabilities"])
|
|
111
|
+
{ "tools" => body["tools"], "capabilities" => Array(body["capabilities"]),
|
|
112
|
+
"side_effect_tools" => body["side_effect_tools"] }
|
|
112
113
|
rescue StandardError, JSON::ParserError
|
|
113
114
|
nil
|
|
114
115
|
end
|
|
@@ -174,5 +175,86 @@ module Insika
|
|
|
174
175
|
ttfb: nil, total: (mono - t0) * 1000.0)
|
|
175
176
|
end
|
|
176
177
|
end
|
|
178
|
+
|
|
179
|
+
# A transport over the deployment's OWN graph, in-process — no HTTP. This is
|
|
180
|
+
# how a simulated conversation (or a replay) exercises the local agent the
|
|
181
|
+
# way a customer would reach it, tools and guardrails included, without a
|
|
182
|
+
# server in between. `runtime` is anything answering the DSL Runtime contract
|
|
183
|
+
# (#chat(message, session_id:, agent:) -> text, raising on failure) — the DSL
|
|
184
|
+
# Definition/System runtime, or a test double.
|
|
185
|
+
#
|
|
186
|
+
# Tool activity is captured from the graph's event stream (the `:tool_call`
|
|
187
|
+
# events the ChatBuilder emits), so an in-process transcript records the same
|
|
188
|
+
# tool names an HTTP replay would — the Simulator's transcript is not blind to
|
|
189
|
+
# what the local agent called. `event_stream` is optional; when omitted it is
|
|
190
|
+
# read off the runtime's graph when one is reachable.
|
|
191
|
+
class GraphTransport
|
|
192
|
+
def initialize(runtime:, event_stream: nil)
|
|
193
|
+
@runtime = runtime
|
|
194
|
+
@event_stream = event_stream ||
|
|
195
|
+
(runtime.graph.event_stream if runtime.respond_to?(:graph) && runtime.graph)
|
|
196
|
+
end
|
|
197
|
+
|
|
198
|
+
def mono = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
199
|
+
|
|
200
|
+
def turn(agent:, conv:, message:)
|
|
201
|
+
t0 = mono
|
|
202
|
+
sub = @event_stream&.subscribe(types: [:tool_call])
|
|
203
|
+
begin
|
|
204
|
+
text = @runtime.chat(message, session_id: conv, agent: agent)
|
|
205
|
+
TurnOutcome.new(
|
|
206
|
+
result: TurnResult.new(output_text: text.to_s, tool_calls: drain_tools(sub), error: nil),
|
|
207
|
+
ttfb: nil, total: (mono - t0) * 1000.0, usage: nil
|
|
208
|
+
)
|
|
209
|
+
rescue Insika::Error => e
|
|
210
|
+
TurnOutcome.new(
|
|
211
|
+
result: TurnResult.new(output_text: "", tool_calls: drain_tools(sub), error: e.message),
|
|
212
|
+
ttfb: nil, total: (mono - t0) * 1000.0, usage: nil
|
|
213
|
+
)
|
|
214
|
+
ensure
|
|
215
|
+
sub&.close
|
|
216
|
+
end
|
|
217
|
+
end
|
|
218
|
+
|
|
219
|
+
private
|
|
220
|
+
|
|
221
|
+
# The same shape an HTTP replay produces: [{ "name" =>, "status" => nil }]
|
|
222
|
+
# (the stream carries no per-tool status — that lives in the trace store).
|
|
223
|
+
def drain_tools(sub)
|
|
224
|
+
return [] if sub.nil?
|
|
225
|
+
|
|
226
|
+
sub.drain_nonblocking.map { |ev| { "name" => ev.data[:name].to_s, "status" => nil } }
|
|
227
|
+
end
|
|
228
|
+
end
|
|
229
|
+
|
|
230
|
+
# Thin A2A transport: drives a REMOTE A2A agent (an agent that only speaks
|
|
231
|
+
# A2A) through the same `turn` seam the Simulator uses. The outbound A2A
|
|
232
|
+
# client already does the send+poll; this is the wrapper that makes it a
|
|
233
|
+
# Transport. `client` answers `#call(url, text, context_id:)` -> {text:} or
|
|
234
|
+
# {error:} — the `Server::A2A::Client` shape.
|
|
235
|
+
class A2ATransport
|
|
236
|
+
def initialize(client:, url:)
|
|
237
|
+
@client = client
|
|
238
|
+
@url = url
|
|
239
|
+
end
|
|
240
|
+
|
|
241
|
+
def mono = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
242
|
+
|
|
243
|
+
def turn(agent:, conv:, message:)
|
|
244
|
+
t0 = mono
|
|
245
|
+
result = @client.call(@url, message.to_s, context_id: conv)
|
|
246
|
+
if result[:error]
|
|
247
|
+
TurnOutcome.new(
|
|
248
|
+
result: TurnResult.new(output_text: "", tool_calls: [], error: result[:error].to_s),
|
|
249
|
+
ttfb: nil, total: (mono - t0) * 1000.0, usage: nil
|
|
250
|
+
)
|
|
251
|
+
else
|
|
252
|
+
TurnOutcome.new(
|
|
253
|
+
result: TurnResult.new(output_text: result[:text].to_s, tool_calls: [], error: nil),
|
|
254
|
+
ttfb: nil, total: (mono - t0) * 1000.0, usage: nil
|
|
255
|
+
)
|
|
256
|
+
end
|
|
257
|
+
end
|
|
258
|
+
end
|
|
177
259
|
end
|
|
178
260
|
end
|
data/lib/insika/event_stream.rb
CHANGED
|
@@ -85,6 +85,16 @@ module Insika
|
|
|
85
85
|
end
|
|
86
86
|
end
|
|
87
87
|
|
|
88
|
+
# Drains whatever is ALREADY queued without ever blocking. Safe in a
|
|
89
|
+
# cooperative reactor: between the `empty?` check and the `dequeue` no other
|
|
90
|
+
# fiber runs, so a non-empty dequeue never waits. The eval transports use it
|
|
91
|
+
# to collect the events a turn already emitted, AFTER the turn returned.
|
|
92
|
+
def drain_nonblocking
|
|
93
|
+
drained = []
|
|
94
|
+
drained << @queue.dequeue until @queue.empty?
|
|
95
|
+
drained
|
|
96
|
+
end
|
|
97
|
+
|
|
88
98
|
# Idempotent: a second CLOSED is harmless (the `each` stops at the first).
|
|
89
99
|
# `@on_close` fires only once (avoids removing the subscription twice).
|
|
90
100
|
def close
|