insika 0.3.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (190) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +180 -0
  3. data/README.md +45 -10
  4. data/bin/insika +684 -0
  5. data/bin/insika-router +87 -0
  6. data/docs/AGENTS.md +94 -403
  7. data/docs/API.md +5 -5
  8. data/docs/ARCHITECTURE.md +3 -2
  9. data/docs/ARTIFACTS.md +95 -0
  10. data/docs/BENCHMARK.md +2 -2
  11. data/docs/CHANNELS.md +14 -14
  12. data/docs/CONTEXT.md +9 -7
  13. data/docs/DEMO.md +80 -0
  14. data/docs/DEPLOY.md +71 -3
  15. data/docs/EMBEDDING.md +1 -1
  16. data/docs/EVALS.md +128 -3
  17. data/docs/FACTS.md +3 -3
  18. data/docs/HARVEST.md +5 -6
  19. data/docs/KNOWLEDGE.md +290 -0
  20. data/docs/LOADTEST.md +2 -2
  21. data/docs/MEDIA.md +128 -0
  22. data/docs/OBSERVABILITY.md +15 -10
  23. data/docs/OUTCOMES.md +137 -0
  24. data/docs/PLUGINS.md +51 -6
  25. data/docs/POLICY.md +216 -0
  26. data/docs/REFINEMENT.md +14 -9
  27. data/docs/RELEASING.md +4 -4
  28. data/docs/ROUTER.md +213 -0
  29. data/docs/RUNNING-LOCAL.md +3 -3
  30. data/docs/SCHEDULING.md +121 -0
  31. data/docs/SECURITY.md +22 -6
  32. data/docs/SKILLS.md +11 -2
  33. data/docs/SOAK.md +2 -2
  34. data/docs/TEMPLATES.md +134 -0
  35. data/docs/TOOLS.md +152 -27
  36. data/docs/WHY.md +1 -1
  37. data/docs/WORKFLOWS.md +2 -2
  38. data/docs/_includes/head_custom.html +5 -0
  39. data/docs/_includes/title.html +13 -0
  40. data/docs/_sass/color_schemes/insika.scss +32 -0
  41. data/docs/_sass/custom/custom.scss +199 -0
  42. data/docs/_sass/custom/setup.scss +26 -0
  43. data/docs/assets/img/favicon.svg +7 -0
  44. data/docs/assets/img/insika-mark.svg +7 -0
  45. data/docs/core-concepts.md +21 -0
  46. data/docs/domain.md +4 -4
  47. data/docs/improve.md +20 -0
  48. data/docs/index.md +8 -5
  49. data/docs/integrate.md +20 -0
  50. data/docs/operate.md +13 -6
  51. data/docs/prompts/ADD-TOOL.md +118 -0
  52. data/docs/prompts/DIAGNOSE-TURN.md +65 -0
  53. data/docs/prompts/GO-LIVE.md +138 -0
  54. data/docs/prompts/RUN-EXAMPLES.md +70 -0
  55. data/docs/reference.md +19 -0
  56. data/docs/ship.md +10 -2
  57. data/docs/start-here.md +18 -0
  58. data/lib/insika/agent_profile.rb +73 -16
  59. data/lib/insika/artifact_signing.rb +82 -0
  60. data/lib/insika/artifact_store.rb +160 -0
  61. data/lib/insika/channel_delivery.rb +1 -1
  62. data/lib/insika/chat_builder.rb +22 -2
  63. data/lib/insika/commands/agent_payload.rb +2 -2
  64. data/lib/insika/commands/backfill_knowledge.rb +145 -0
  65. data/lib/insika/commands/delete_artifact.rb +35 -0
  66. data/lib/insika/commands/delete_concept.rb +34 -0
  67. data/lib/insika/commands/delete_mcp.rb +6 -2
  68. data/lib/insika/commands/delete_tenant_data.rb +15 -3
  69. data/lib/insika/commands/gate_refinement.rb +1 -1
  70. data/lib/insika/commands/refresh_mcp_tools.rb +47 -0
  71. data/lib/insika/commands/restore_concept.rb +34 -0
  72. data/lib/insika/commands/seed_demo_data.rb +31 -0
  73. data/lib/insika/commands/upsert_mcp.rb +6 -3
  74. data/lib/insika/commands/write_concept.rb +57 -0
  75. data/lib/insika/context/priority.rb +2 -0
  76. data/lib/insika/context/providers/knowledge.rb +108 -0
  77. data/lib/insika/context/providers/prompt.rb +30 -24
  78. data/lib/insika/cron.rb +189 -0
  79. data/lib/insika/demo/agent_attrs.rb +43 -0
  80. data/lib/insika/demo/golden_cases.rb +81 -0
  81. data/lib/insika/demo/seeder.rb +336 -0
  82. data/lib/insika/doctor.rb +176 -8
  83. data/lib/insika/dsl/definition.rb +3 -2
  84. data/lib/insika/dsl/runtime.rb +60 -79
  85. data/lib/insika/dsl/server_boot.rb +23 -1
  86. data/lib/insika/dsl/system.rb +10 -2
  87. data/lib/insika/dsl.rb +103 -2
  88. data/lib/insika/env_schema.rb +16 -1
  89. data/lib/insika/evals/golden.rb +41 -4
  90. data/lib/insika/evals/judge.rb +47 -2
  91. data/lib/insika/evals/pairwise.rb +11 -0
  92. data/lib/insika/evals/persona.rb +98 -0
  93. data/lib/insika/evals/runner.rb +9 -0
  94. data/lib/insika/evals/simulator.rb +225 -0
  95. data/lib/insika/evals/transport.rb +83 -1
  96. data/lib/insika/event_stream.rb +10 -0
  97. data/lib/insika/executor.rb +231 -55
  98. data/lib/insika/followup_policy.rb +2 -25
  99. data/lib/insika/golden_store.rb +16 -1
  100. data/lib/insika/grounding/matcher.rb +1 -1
  101. data/lib/insika/knowledge.rb +680 -0
  102. data/lib/insika/knowledge_store.rb +140 -0
  103. data/lib/insika/mcp_client.rb +94 -0
  104. data/lib/insika/mcp_json.rb +74 -0
  105. data/lib/insika/mcp_live_tool.rb +43 -0
  106. data/lib/insika/mcp_store.rb +98 -26
  107. data/lib/insika/mcp_tool_ingestor.rb +30 -8
  108. data/lib/insika/mcp_tool_registry.rb +100 -0
  109. data/lib/insika/media.rb +115 -31
  110. data/lib/insika/message_origin.rb +1 -1
  111. data/lib/insika/middleware.rb +9 -0
  112. data/lib/insika/onboarding.rb +17 -1
  113. data/lib/insika/outcome_store.rb +1 -1
  114. data/lib/insika/overlay_tool_registry.rb +37 -17
  115. data/lib/insika/packaging.rb +2 -2
  116. data/lib/insika/profile_source.rb +8 -1
  117. data/lib/insika/prompt_catalog.rb +10 -0
  118. data/lib/insika/retention.rb +36 -1
  119. data/lib/insika/router/app.rb +157 -0
  120. data/lib/insika/router/backend_pool.rb +98 -0
  121. data/lib/insika/router/hash_ring.rb +55 -0
  122. data/lib/insika/router/proxy_body.rb +34 -0
  123. data/lib/insika/router/session_key.rb +54 -0
  124. data/lib/insika/router.rb +18 -0
  125. data/lib/insika/schedule.rb +177 -0
  126. data/lib/insika/schedule_engine.rb +314 -0
  127. data/lib/insika/schedule_store.rb +208 -0
  128. data/lib/insika/server/app.rb +105 -15
  129. data/lib/insika/server/rack_app.rb +5 -1
  130. data/lib/insika/server/responses.rb +1 -1
  131. data/lib/insika/skill_catalog.rb +12 -0
  132. data/lib/insika/steer_injector.rb +21 -10
  133. data/lib/insika/studio/app.rb +567 -45
  134. data/lib/insika/studio/assets/dist/application.css +1 -1
  135. data/lib/insika/studio/assets/dist/application.js +21 -21
  136. data/lib/insika/studio/forms.rb +46 -5
  137. data/lib/insika/studio/nav_icons.rb +14 -1
  138. data/lib/insika/studio/views/_agent_tab_cache.erb +25 -0
  139. data/lib/insika/studio/views/_agent_tab_config.erb +514 -0
  140. data/lib/insika/studio/views/_agent_tab_history.erb +24 -0
  141. data/lib/insika/studio/views/_agent_tab_loops.erb +54 -0
  142. data/lib/insika/studio/views/_agent_tab_memory.erb +51 -0
  143. data/lib/insika/studio/views/_agent_tab_outcomes.erb +31 -0
  144. data/lib/insika/studio/views/_agent_tab_prompts.erb +108 -0
  145. data/lib/insika/studio/views/_agent_tab_skills.erb +38 -0
  146. data/lib/insika/studio/views/_agents_master.erb +44 -0
  147. data/lib/insika/studio/views/_message.erb +49 -32
  148. data/lib/insika/studio/views/agent_detail.erb +61 -820
  149. data/lib/insika/studio/views/agents.erb +70 -57
  150. data/lib/insika/studio/views/artifact.erb +23 -0
  151. data/lib/insika/studio/views/artifacts.erb +59 -0
  152. data/lib/insika/studio/views/evals.erb +2 -2
  153. data/lib/insika/studio/views/facts.erb +1 -1
  154. data/lib/insika/studio/views/funnel.erb +1 -1
  155. data/lib/insika/studio/views/home.erb +106 -67
  156. data/lib/insika/studio/views/knowledge.erb +123 -0
  157. data/lib/insika/studio/views/layout.erb +14 -11
  158. data/lib/insika/studio/views/mcp.erb +174 -80
  159. data/lib/insika/studio/views/session.erb +231 -177
  160. data/lib/insika/studio/views/settings.erb +39 -1
  161. data/lib/insika/studio/views/skills.erb +1 -1
  162. data/lib/insika/studio/views/tools.erb +24 -9
  163. data/lib/insika/templates/browser-agent/README.md +36 -0
  164. data/lib/insika/templates/browser-agent/agent.rb +49 -0
  165. data/lib/insika/templates/daily-digest/README.md +38 -0
  166. data/lib/insika/templates/daily-digest/agent.rb +77 -0
  167. data/lib/insika/templates/repo-explorer/README.md +36 -0
  168. data/lib/insika/templates/repo-explorer/agent.rb +45 -0
  169. data/lib/insika/templates/research-analyst/README.md +26 -0
  170. data/lib/insika/templates/research-analyst/agent.rb +58 -0
  171. data/lib/insika/templates/review-panel/README.md +20 -0
  172. data/lib/insika/templates/review-panel/agent.rb +50 -0
  173. data/lib/insika/templates/travel-planner/README.md +35 -0
  174. data/lib/insika/templates/travel-planner/agent.rb +87 -0
  175. data/lib/insika/templates.rb +112 -0
  176. data/lib/insika/tick.rb +24 -12
  177. data/lib/insika/timezone.rb +45 -0
  178. data/lib/insika/tools/generate_image.rb +52 -7
  179. data/lib/insika/tools/load_knowledge.rb +74 -0
  180. data/lib/insika/tools/run_persona_eval.rb +328 -0
  181. data/lib/insika/tools/save_artifact.rb +95 -0
  182. data/lib/insika/turn_output.rb +1 -1
  183. data/lib/insika/turn_state.rb +15 -4
  184. data/lib/insika/version.rb +1 -1
  185. data/lib/insika/wiring/graph.rb +184 -12
  186. data/lib/insika/wiring/graph_chat.rb +102 -0
  187. data/lib/insika.rb +57 -0
  188. metadata +105 -5
  189. data/docs/build.md +0 -14
  190. data/docs/understand.md +0 -10
data/lib/insika/tick.rb CHANGED
@@ -3,21 +3,23 @@
3
3
  require "time"
4
4
 
5
5
  module Insika
6
- # The periodic tick: durability stops waiting for a reboot. One
7
- # pass does two things, in this order:
6
+ # The periodic tick: durability stops waiting for a reboot. One
7
+ # pass does three things, in this order:
8
8
  #
9
9
  # 1. DRAIN the outbox (`ChannelDelivery#sweep`) — replies a previous pass
10
10
  # (or process) recorded and never claimed. Ungated: every record carries
11
11
  # its own transactional claim, so N workers draining is safe.
12
- # 2. SWEEP stale orphaned tasks (`Recovery#run(stale_after:)`) — gated by a
13
- # bucketed claim (`Recovery.claim_sweep` on "tick:<epoch/interval>"), so
14
- # exactly one worker per window sweeps. The staleness threshold is the
15
- # liveness gate: a live :running turn is bounded by turn_timeout, so
16
- # anything untouched past it cannot be alive.
12
+ # 2. The engine's background duties, each gated by its OWN claim window so
13
+ # their O(n) scans never ride the 60 s loop: retention (daily), the
14
+ # outcome fold, the follow-up firer, and the recurring-schedule firer
15
+ # (the engine's own cron — it superseded the "point your own cron at
16
+ # the route" decision, see docs/SCHEDULING.md).
17
+ # 3. SWEEP stale orphaned tasks (`Recovery#run(stale_after:)`) — gated by a
18
+ # bucketed claim, so exactly one worker per window sweeps. The staleness
19
+ # threshold is the liveness gate: a live :running turn is bounded by
20
+ # turn_timeout, so anything untouched past it cannot be alive.
17
21
  #
18
- # It is NOT a job queue: no schedules, no priorities, no fan-out. The
19
- # refinement hook once pictured here is dropped by merit —
20
- # docs/REFINEMENT.md's "no scheduler in the engine" stands.
22
+ # It is NOT a job queue: no schedules queue, no priorities, no fan-out.
21
23
  class Tick
22
24
  # 60s: a customer waiting on WhatsApp is the deadline. 900s = 3x the
23
25
  # default turn_timeout (300s) — the rule, not the number: the threshold
@@ -31,7 +33,8 @@ module Insika
31
33
 
32
34
  def initialize(store:, recovery:, channel_delivery:, logger: nil,
33
35
  interval: DEFAULT_INTERVAL, stale_after: DEFAULT_STALE_AFTER,
34
- sleeper: nil, retention: nil, funnel: nil, followup: nil)
36
+ sleeper: nil, retention: nil, funnel: nil, followup: nil,
37
+ schedule: nil)
35
38
  @store = store
36
39
  @recovery = recovery
37
40
  @channel_delivery = channel_delivery
@@ -39,9 +42,10 @@ module Insika
39
42
  @interval = interval.to_i
40
43
  @stale_after = stale_after.to_i
41
44
  @sleeper = sleeper || method(:default_sleep)
42
- @retention = retention # WS8: the daily age-based sweep; nil = none
45
+ @retention = retention # the daily age-based sweep; nil = none
43
46
  @funnel = funnel # the tick-driven outcome fold; nil = none
44
47
  @followup = followup # the tick-driven follow-up firer; nil = none
48
+ @schedule = schedule # the recurring-schedule firer; nil = none
45
49
  end
46
50
 
47
51
  # the fold is wired after the Tick is built (the graph passes
@@ -53,6 +57,10 @@ module Insika
53
57
  # shape as `funnel` — the stores come from the spine).
54
58
  attr_accessor :followup
55
59
 
60
+ # the recurring-schedule firer, wired after the Tick is built
61
+ # (same shape — the stores come from the spine).
62
+ attr_accessor :schedule
63
+
56
64
  def enabled? = @interval.positive?
57
65
 
58
66
  # One pass, pure (no reactor needed): the serving loop calls it on a timer,
@@ -73,6 +81,10 @@ module Insika
73
81
  # OWN claim window so the O(n) scans never ride the 60 s loop.
74
82
  followup_summary = @followup&.run
75
83
  summary[:followup] = followup_summary if followup_summary
84
+ # the recurring-schedule firer — the tick's fourth duty,
85
+ # the same claim-window discipline as the follow-up firer.
86
+ schedule_summary = @schedule&.run
87
+ summary[:schedule] = schedule_summary if schedule_summary
76
88
  return summary unless claim_window
77
89
 
78
90
  result = @recovery.run(stale_after: @stale_after)
@@ -0,0 +1,45 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Insika
4
+ # IANA timezone handling through the OS tz database — the
5
+ # engine's only route to a zone NAME (Ruby stdlib's `Time#getlocal` takes an
6
+ # offset, not a zone name). Shared by FollowupPolicy (quiet hours), the cron
7
+ # parser (next-fire materialization) and the doctor (zone existence).
8
+ #
9
+ # IANA names are resolved by pointing Ruby's `TZ` at the zone for the
10
+ # computation. Save/restore keeps the global intact; under the engine's
11
+ # cooperative fiber model — no IO between the save and the restore — the
12
+ # mutation is atomic on the calling fiber.
13
+ module Timezone
14
+ # The candidate tz-data roots (TZDIR first — Ruby's own lookup env). The
15
+ # zone name maps to a FILE under the root ("America/Sao_Paulo" ->
16
+ # "America/Sao_Paulo").
17
+ TZ_ROOTS = ([ENV["TZDIR"]] +
18
+ %w[/usr/share/zoneinfo /usr/share/lib/zoneinfo /etc/zoneinfo])
19
+ .compact.freeze
20
+
21
+ module_function
22
+
23
+ # -> bool: is `zone` an IANA name the OS tz database knows? A bogus zone
24
+ # is a malformed declaration — refused where the doctor can name it (an
25
+ # unknown ENV["TZ"] silently behaves as UTC, so existence is checked
26
+ # against the database, not by asking Time).
27
+ def known?(zone)
28
+ zone = zone.to_s
29
+ return true if zone == "UTC" || zone == "Etc/UTC"
30
+
31
+ TZ_ROOTS.any? { |root| File.directory?(root) && File.exist?(File.join(root, zone)) }
32
+ end
33
+
34
+ # Yields `time` interpreted in the given IANA zone (via a save/restore of
35
+ # ENV["TZ"] — the stdlib-only route to the OS tz database). Returns the
36
+ # block's value.
37
+ def in_zone(zone, time)
38
+ previous = ENV["TZ"]
39
+ ENV["TZ"] = zone.to_s
40
+ yield time
41
+ ensure
42
+ ENV["TZ"] = previous
43
+ end
44
+ end
45
+ end
@@ -14,12 +14,42 @@ module Insika
14
14
  # (`channel.capabilities` includes "image_output") — nothing leaks by
15
15
  # default. The image is an envelope part, never part of the answer text;
16
16
  # the provider's tokens are merged into the turn's usage like any ask.
17
+ #
18
+ # Also EDITS — `source_image_urls` (or, absent that, the turn's
19
+ # own inbound photo) rides `paint(with:)`; `mask_url` rides `paint(mask:)`.
20
+ # What the edit MEANS (a try-on, a mockup) is the skill's business; this
21
+ # tool only transports the bytes.
17
22
  class GenerateImage < RubyLLM::Tool
18
- description "Generate an image and attach it to the reply as an output part. " \
19
- "Use when the customer asked for a picture or an image would help."
20
- param :prompt, desc: "What to draw, in detail"
21
- param :size, desc: "Optional canvas size, e.g. 1024x1024 (default from the agent config)",
22
- required: false
23
+ description "Generate an image, or EDIT one, and attach it to the reply as an " \
24
+ "output part. Use when the customer asked for a picture, or asked to " \
25
+ "transform/edit a photo (a virtual try-on, a mockup on their wall, a " \
26
+ "touch-up). Omitting source_image_urls generates a new image from the " \
27
+ "prompt alone — UNLESS this turn carries an inbound photo, in which case " \
28
+ "that photo is edited by default (pass source_image_urls explicitly to " \
29
+ "generate from scratch instead)."
30
+ # explicit JSON-schema form (the `param` DSL only reaches strings/scalars,
31
+ # and source_image_urls needs a typed array — the bare-array gotcha, #128).
32
+ params(
33
+ type: "object",
34
+ properties: {
35
+ prompt: { type: "string", description: "What to draw, or what edit to make, in detail" },
36
+ size: { type: "string",
37
+ description: "Optional canvas size, e.g. 1024x1024 (default from the agent config)" },
38
+ source_image_urls: {
39
+ type: "array",
40
+ description: "Image URLs to edit instead of generating from scratch — e.g. " \
41
+ "the photo the customer just sent in this conversation " \
42
+ "({{ctx.image_url}}), or any other URL from this chat. Omit to use " \
43
+ "the turn's inbound photo by default (if any), or to generate a " \
44
+ "fresh image when there is none.",
45
+ items: { type: "string" }
46
+ },
47
+ mask_url: { type: "string",
48
+ description: "Optional mask image URL marking which area of the " \
49
+ "source(s) to edit (transparent = editable)" }
50
+ },
51
+ required: %w[prompt]
52
+ )
23
53
 
24
54
  def name = "generate_image"
25
55
 
@@ -32,13 +62,28 @@ module Insika
32
62
  super()
33
63
  end
34
64
 
35
- def execute(prompt:, size: nil)
36
- cfg = @config.merge("size" => size.to_s).reject { |_, v| v.to_s.empty? }
65
+ def execute(prompt:, size: nil, source_image_urls: nil, mask_url: nil)
66
+ cfg = @config.merge("size" => size.to_s, "mask_url" => mask_url.to_s)
67
+ .reject { |_, v| v.to_s.empty? }
68
+ cfg = cfg.merge(source_config(source_image_urls))
37
69
  part, usage = @runner.generate_media_output(:image, prompt.to_s, cfg)
38
70
  @state.output_parts << part
39
71
  @runner.account_media_usage(@state, part, usage)
40
72
  "image generated and attached to the reply (#{part["mime_type"]})"
41
73
  end
74
+
75
+ private
76
+
77
+ # Explicit URLs win over the default; a turn with inbound images (no
78
+ # explicit URLs) hands `Output.generate_image` the ALREADY-FETCHED
79
+ # attachments (bypassing the URL fetch — they are bytes we hold).
80
+ def source_config(source_image_urls)
81
+ urls = Array(source_image_urls).map(&:to_s).reject(&:empty?)
82
+ return { "source_urls" => urls } if urls.any?
83
+ return {} unless Array(@state.image_attachments).any?
84
+
85
+ { "source_attachments" => @state.image_attachments }
86
+ end
42
87
  end
43
88
  end
44
89
  end
@@ -0,0 +1,74 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "ruby_llm"
4
+ require "time"
5
+
6
+ module Insika
7
+ module Tools
8
+ # Level 2 of knowledge's progressive disclosure: loads a learned
9
+ # concept's full body on demand, by the name the `<knowledge>` block
10
+ # listed. Same shape as `LoadSkill` — a system default (outside the
11
+ # allowlist), wired only when the profile opted in
12
+ # (`knowledge.retrieve`), never through `tools_allow`.
13
+ #
14
+ # `require "ruby_llm"` stays in THIS file (it inherits from
15
+ # RubyLLM::Tool), so it does NOT enter lib/insika.rb — the Executor
16
+ # loads it lazily inside create_chat, same as load_skill.
17
+ class LoadKnowledge < RubyLLM::Tool
18
+ description "Loads the complete content of a learned concept by name"
19
+ param :name, desc: "Exact concept name, as listed in <knowledge>"
20
+
21
+ # RubyLLM::Tool#name derives from self.class.name — for a nested class it
22
+ # produces "insika--tools--load_knowledge", not "load_knowledge" (which
23
+ # wire_callbacks/:knowledge_retrieved assume). Explicit override, same
24
+ # reason LoadSkill has one.
25
+ def name = "load_knowledge"
26
+
27
+ # trace_recorder/state are OPTIONAL (nil = no trace, parity): this tool
28
+ # is deliberately NOT enveloped (ToolAssembly#wrap_tools), so without
29
+ # recording HERE the call is missing from the Studio's trace — same
30
+ # shape as LoadSkill.
31
+ def initialize(store, agent_id, tenant: nil, trace_recorder: nil, state: nil)
32
+ @store = store
33
+ @agent_id = agent_id
34
+ @tenant = tenant
35
+ @trace_recorder = trace_recorder
36
+ @state = state
37
+ super()
38
+ end
39
+
40
+ def execute(name:)
41
+ started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
42
+ result = load(name)
43
+ trace(name, result, started)
44
+ result
45
+ end
46
+
47
+ private
48
+
49
+ def load(name)
50
+ content = @store.get(@agent_id, name.to_s, tenant: @tenant)
51
+ return { error: "concept '#{name}' not found" } unless content
52
+
53
+ content
54
+ end
55
+
56
+ # Mirrors LoadSkill#trace (same entry shape, so the Studio renders it
57
+ # like any other call). NEVER breaks the turn — the trace is
58
+ # observability, not the answer.
59
+ def trace(name, result, started)
60
+ return unless @trace_recorder && @state&.task&.session_id
61
+
62
+ @trace_recorder.record(
63
+ session_id: @state.task.session_id,
64
+ entry: { "turn" => @state.turn, "tool" => "load_knowledge", "call_id" => "",
65
+ "args" => { "name" => name.to_s }, "result" => result,
66
+ "ms" => ((Process.clock_gettime(Process::CLOCK_MONOTONIC) - started) * 1000).round,
67
+ "at" => Time.now.utc.iso8601 }
68
+ )
69
+ rescue StandardError
70
+ nil
71
+ end
72
+ end
73
+ end
74
+ end
@@ -0,0 +1,328 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "securerandom"
4
+ require "ruby_llm"
5
+ require_relative "agent_enum"
6
+
7
+ module Insika
8
+ module Tools
9
+ # `run_persona_eval` — a QA agent's own probe: pick an authored SIMULATED
10
+ # persona case (Evals::GoldenStore) and run it, in-process, against the
11
+ # case's declared target agent — the same Simulator + Judge machinery
12
+ # `insika evals:simulate` drives over HTTP, minus the CLI and the network
13
+ # hop.
14
+ #
15
+ # SAFETY is DERIVED, never a flag: the target's reachable side-effect tools
16
+ # are computed from the live registry (Evals::EvalProfile), the same way
17
+ # the CLI derives them. A read-only target needs no swap:
18
+ # Simulator::Safety's own "side_effect_tools.empty?" branch allows it
19
+ # directly. A target that DOES reach a side-effect tool gets the REAL
20
+ # swap (Evals::EvalProfile.registry — own overlay, wired here):
21
+ # every one of those tools resolves to a Simulator::DryRunTool for the
22
+ # duration of this ONE simulated conversation, run through a THROWAWAY
23
+ # Executor+Bus built fresh per call (`shadow_runtime`) — sharing every
24
+ # OTHER collaborator of the real graph (guardrails, policy, context
25
+ # assembly, skills/prompts, the real session/task/checkpoint stores), so
26
+ # the target is tested as faithfully as `--staging` ever was, minus the
27
+ # one write. Needs the real `graph:` (see `initialize`) — a caller that
28
+ # does not have one (an old-style double) falls back to refusing outright.
29
+ #
30
+ # BUDGET: the persona model + judge model calls are the cost of running the
31
+ # eval, charged to the CALLING agent's turn (never the target's — the
32
+ # target's own turns are billed normally, through the ordinary edge
33
+ # limiter, exactly as if a customer had sent those messages). A hard cap on
34
+ # the calling agent skips the run — visibly, in the tool result — before a
35
+ # cent is spent.
36
+ #
37
+ # TENANT ISOLATION: a persona case belongs to a tenant (Golden#tenant,
38
+ # "platform" by default); this tool only ever lists/runs cases in the
39
+ # CALLING agent's own tenant (calling_tenant, read off turn_context, never
40
+ # the model). One QA agent per store (the C3.2 plan) is what makes this
41
+ # meaningful -- without it, "qa-store-a" could enumerate and run
42
+ # "qa-store-b"'s persona and read its `knows` in the transcript.
43
+ class RunPersonaEval < RubyLLM::Tool
44
+ description "Run an authored simulated-customer persona case, in-process, against " \
45
+ "its target agent and score the whole conversation with the configured " \
46
+ "judge panel. Refuses if the target exposes a tool that could write for " \
47
+ "real."
48
+ param :case_id, desc: "Id of the persona case to run"
49
+
50
+ def name = "run_persona_eval"
51
+
52
+ # golden_store: Evals::GoldenStore — where authored persona cases live,
53
+ # scoped to the CALLING agent's own tenant (never the
54
+ # model's — see `calling_tenant`). A case authored for
55
+ # another tenant is invisible here, not merely undocumented.
56
+ # profiles: ProfileSource — resolves both the target agent (the
57
+ # case's `agent:`) and the CALLING agent (turn_context, for
58
+ # the budget check).
59
+ # tool_registry: the deployment's EFFECTIVE registry — what
60
+ # Evals::EvalProfile derives the target's side-effect tools
61
+ # from (the same registry a real turn resolves tools on).
62
+ # runtime: anything answering `#chat(message, session_id:, agent:)`
63
+ # (raises Insika::Error on failure) — Evals::GraphTransport's
64
+ # contract. Used AS-IS for a read-only target; a target
65
+ # with a reachable side-effect tool needs `graph:` instead
66
+ # (this alone cannot swap anything).
67
+ # graph: the real Wiring::Graph::Result — ONLY consulted to build
68
+ # the throwaway swapped-registry Executor+Bus
69
+ # (`shadow_runtime`) when the target has a reachable
70
+ # side-effect tool. nil (an old-style double) = that case
71
+ # refuses outright, same as before this existed.
72
+ # settings_store: where the platform `utility_model` (persona) and the
73
+ # judge panel (`evals.judges`) are configured.
74
+ # budget_ledger: WS2 counters — read before the run (skip on a hard cap),
75
+ # written after (persona + judge spend only).
76
+ # llm: this graph's own RubyLLM::Context, if it has one (a
77
+ # DSL-built graph's own credentials) — the persona/judge
78
+ # calls' preferred source, ahead of `runtime.llm`
79
+ # (kept for the existing double-based specs) and the
80
+ # process-wide RubyLLM constant.
81
+ def initialize(golden_store:, profiles:, tool_registry:, runtime:, settings_store:,
82
+ graph: nil, budget_ledger: nil, event_stream: nil, llm: nil)
83
+ @golden_store = golden_store
84
+ @profiles = profiles
85
+ @tool_registry = tool_registry
86
+ @runtime = runtime
87
+ @graph = graph
88
+ @settings_store = settings_store
89
+ @budget_ledger = budget_ledger
90
+ @event_stream = event_stream
91
+ @llm = llm
92
+ super()
93
+ end
94
+
95
+ # Per-turn bindings, deposited by the Executor (ToolAssembly's
96
+ # `turn_context=` seam — same as save_artifact/data-tools). Only the
97
+ # CALLING agent's id + declared tenant are used here (the budget check);
98
+ # never a tenant/agent the model types.
99
+ attr_reader :turn_context
100
+
101
+ def turn_context=(ctx)
102
+ @turn_context = (ctx || {}).each_with_object({}) { |(k, v), acc| acc[k.to_sym] = v }
103
+ end
104
+
105
+ # The runnable case ids, named so the model cannot guess one that does
106
+ # not exist (the Subagent tool's lesson — see AgentEnum).
107
+ def description
108
+ ids = case_ids
109
+ return super if ids.empty?
110
+
111
+ "#{super} Cases you may run: #{ids.join(', ')}."
112
+ end
113
+
114
+ def params_schema
115
+ Insika::Tools::AgentEnum.inject(super, case_ids, path: %i[case_id])
116
+ end
117
+
118
+ def execute(case_id:)
119
+ golden = @golden_store.find(case_id.to_s)
120
+ # The SAME error, whether the case does not exist or exists under another
121
+ # tenant — a QA agent must not be able to tell the two apart (that
122
+ # distinction is itself a leak: "case exists, just not yours").
123
+ unless golden&.simulated? && golden.tenant == calling_tenant
124
+ return { error: "unknown or invalid persona case '#{case_id}'" }
125
+ end
126
+
127
+ target = @profiles[golden.agent]
128
+ return { error: "persona case '#{case_id}' targets unknown agent '#{golden.agent}'" } unless target
129
+
130
+ derived = Insika::Evals::EvalProfile.side_effect_tools(target, @tool_registry)
131
+ return refuse_side_effects(golden.agent, derived) if !derived.empty? && @graph.nil?
132
+
133
+ skip = budget_skip
134
+ return skip if skip
135
+
136
+ meter = []
137
+ judge = build_judge(meter)
138
+ return { error: "no judge configured for this case — Studio -> Settings -> Evals" } unless judge
139
+
140
+ persona_ask = build_persona_ask(meter)
141
+ return persona_ask if persona_ask.is_a?(Hash) # {error:}
142
+
143
+ run_and_score(golden, derived, persona_ask, judge, meter)
144
+ rescue Insika::Evals::Simulator::UnsafeTarget => e
145
+ { error: e.message }
146
+ end
147
+
148
+ private
149
+
150
+ def refuse_side_effects(agent, derived)
151
+ { error: "target agent '#{agent}' exposes side-effect tool(s) (#{derived.join(', ')}) — " \
152
+ "run_persona_eval needs the real graph (no swap available for this caller)" }
153
+ end
154
+
155
+ def run_and_score(golden, derived, persona_ask, judge, meter)
156
+ runtime = derived.empty? ? @runtime : shadow_runtime(derived)
157
+ transport = Insika::Evals::GraphTransport.new(runtime: runtime, event_stream: @event_stream)
158
+ safety = Insika::Evals::Simulator::Safety.new(
159
+ side_effect_tools: derived, eval_profile: !derived.empty?, swapped_tools: derived
160
+ )
161
+ simulator = Insika::Evals::Simulator.new(transport: transport, ask: persona_ask, safety: safety)
162
+
163
+ conv = "eval-#{golden.id}-#{SecureRandom.hex(4)}" # a fresh session every run (never reused)
164
+ run = simulator.run(persona: golden.persona, agent: golden.agent, conv: conv)
165
+ verdict = judge.score_conversation(
166
+ rubric: golden.rubric, transcript: run.transcript, policy: golden.policy,
167
+ min_score: golden.min_score || Insika::Evals::Judge::DEFAULT_MIN_SCORE
168
+ )
169
+ account_spend(meter)
170
+
171
+ { case: golden.id, agent: golden.agent, stop: run.stop.to_s, turns: run.turns,
172
+ score: verdict&.score, pass: verdict&.pass, reason: verdict&.reason,
173
+ transcript: run.transcript.map { |m| { role: m[:role], text: m[:text] } } }
174
+ end
175
+
176
+ # A THROWAWAY Executor+Bus, built fresh for THIS call, over the SAME
177
+ # session/task/checkpoint stores, guardrails, policy engine, context
178
+ # assembly, skills/prompts and capabilities as the real graph — the
179
+ # ONLY thing different is the tool registry, which resolves every name
180
+ # in `derived` to a Simulator::DryRunTool (Evals::EvalProfile.registry)
181
+ # and everything else exactly as the real one does. Never persisted
182
+ # anywhere NEW: the simulated turn's session/task rows land in the
183
+ # SAME stores a `--staging` run's already do (RunPersonaEval's own
184
+ # never-reused `conv` id is what keeps it from colliding with a real
185
+ # customer session, same as before this existed).
186
+ def shadow_runtime(derived)
187
+ overlay = Insika::Evals::EvalProfile.registry(@tool_registry, side_effect_tools: derived)
188
+ executor = Insika::Executor.new(
189
+ context_builder: @graph.context_builder, policy_engine: @graph.policy_engine,
190
+ middleware: @graph.middleware, hooks: @graph.hooks,
191
+ tool_registry: overlay, skill_catalog: @graph.skill_catalog, profiles: @graph.profiles,
192
+ session_store: @graph.session_store, task_store: @graph.task_store,
193
+ checkpoint_store: @graph.checkpoint_store, event_stream: @graph.event_stream,
194
+ workflow_registry: @graph.workflow_registry, pending_action_store: @graph.pending_action_store,
195
+ capability_registry: @graph.capability_registry,
196
+ tool_catalog: Insika::ToolCatalog.new(tool_registry: overlay),
197
+ memory_store: @graph.memory_store, settings_store: @settings_store,
198
+ delegation_store: @graph.delegation_store, channel_delivery: @graph.channel_delivery,
199
+ llm: llm_context
200
+ )
201
+ bus = Insika::CommandBus.new
202
+ bus.register(:send_message, Insika::Commands::SendMessage.new(
203
+ profiles: @graph.profiles, session_store: @graph.session_store,
204
+ task_store: @graph.task_store, executor: executor,
205
+ inbound_log: @graph.inbound_log, contact_store: @graph.contact_store,
206
+ followup_store: @graph.followup_store, store: @graph.backend
207
+ ))
208
+ shadow = @graph.dup
209
+ shadow.bus = bus
210
+ shadow.executor = executor
211
+ Insika::Wiring::GraphChat.new(graph: shadow)
212
+ end
213
+
214
+ # -> [String] every case this tool can run: a valid, SIMULATED (persona:)
215
+ # golden belonging to the CALLING agent's own tenant. A scripted (turns:)
216
+ # case has nothing to simulate — the CLI skips it too — and a case outside
217
+ # `calling_tenant` is not merely un-runnable, it never appears at all (the
218
+ # model cannot even learn another tenant's case ids from the enum).
219
+ def case_ids
220
+ @golden_store.for_tenant(calling_tenant).select(&:simulated?).map(&:id)
221
+ end
222
+
223
+ # The CALLING agent's tenant — never the model's, never the target's. The
224
+ # same "declared command tenant, 'platform' is the single-tenant default"
225
+ # rule `save_artifact`'s `binding_tenant` uses. Every persona-case
226
+ # read/list/run in this tool is scoped to it.
227
+ def calling_tenant
228
+ Insika::Coercion.presence(turn_context&.dig(:command_tenant)) || "platform"
229
+ end
230
+
231
+ # settings["evals"] -> the configured judge panel, scoped to this
232
+ # graph's own RubyLLM credentials (never the process-wide default —
233
+ # a graph's judge spends the graph's own key) and METERED (every judge
234
+ # call's usage lands in `meter`). nil = nobody configured (the CLI's own
235
+ # rule: never guess a judge to spend money on).
236
+ def build_judge(meter)
237
+ judge, = Insika::Evals::JudgePanel.build(@settings_store.get["evals"] || {}, chat_factory: metered_factory(meter))
238
+ judge
239
+ end
240
+
241
+ # The persona model — the platform `utility_model` (never a caller-chosen
242
+ # model: the model does not get to pick what it costs to test itself),
243
+ # metered the same way. -> callable | {error:}.
244
+ def build_persona_ask(meter)
245
+ model = Insika::Coercion.presence(@settings_store.get["utility_model"])
246
+ return { error: "no persona model — set the platform utility_model (Studio -> Settings)" } if model.nil?
247
+
248
+ metered_factory(meter).call(model, nil)
249
+ end
250
+
251
+ # The explicit `llm:` wins (the real wiring passes it — see
252
+ # `Wiring::Graph.register_persona_eval_tool`); `@runtime.llm` is the
253
+ # fallback the existing double-based specs rely on (a fake `runtime`
254
+ # answering `#llm` with no `graph:` at all). nil = the process-wide
255
+ # RubyLLM constant.
256
+ def llm_context
257
+ @llm || (@runtime.respond_to?(:llm) ? @runtime.llm : nil)
258
+ end
259
+
260
+ # ->(model, provider) { ask } — a RubyLLM chat (this graph's own
261
+ # credentials, temperature 0) whose every `#ask` lands its Message in
262
+ # `sink` before handing back the text. Same shape as
263
+ # JudgePanel.ruby_llm_ask, except it does not throw the usage away.
264
+ # `assume_model_exists` is passed ONLY alongside an explicit `provider`
265
+ # (the judge panel's own shape — `settings["evals"]["judges"]` always
266
+ # carries one) — RubyLLM raises ArgumentError on `assume_model_exists:
267
+ # true` with no provider. The persona's own model (the platform
268
+ # `utility_model`, a bare ref with no companion provider setting
269
+ # anywhere in the schema) needs the OPPOSITE: no provider, no
270
+ # assume_model_exists, so RubyLLM resolves it from its own registry —
271
+ # exactly how a bare model ref already works everywhere else this
272
+ # codebase reaches for `utility_model`.
273
+ def metered_factory(sink)
274
+ lambda do |model, provider|
275
+ kwargs = { model: model }
276
+ if provider
277
+ kwargs[:provider] = provider
278
+ kwargs[:assume_model_exists] = true
279
+ end
280
+ chat = (llm_context || RubyLLM).chat(**kwargs).with_temperature(0)
281
+ lambda do |prompt|
282
+ msg = chat.ask(prompt)
283
+ sink << msg
284
+ msg.content
285
+ end
286
+ end
287
+ end
288
+
289
+ # -> truthy (a visible skip result) when the CALLING agent's own hard
290
+ # budget is already at/over a window cap; nil otherwise. Mirrors
291
+ # ScheduleEngine#budget_exhausted? — a HARD cap skips rather than spends
292
+ # (a soft one just runs; the ledger's own alert already warns).
293
+ def budget_skip
294
+ return nil unless @budget_ledger
295
+
296
+ agent_id = turn_context&.dig(:agent_id)
297
+ profile = agent_id && @profiles[agent_id]
298
+ budget = profile&.respond_to?(:budget) ? profile.budget : nil
299
+ return nil if budget.nil? || budget["soft"] == true
300
+
301
+ tenant = turn_context&.dig(:command_tenant)
302
+ window = %w[daily monthly].find do |w|
303
+ cap = budget[w].to_i
304
+ cap.positive? && @budget_ledger.current(tenant: tenant, agent: agent_id)[w.to_sym] >= cap
305
+ end
306
+ { skipped: true, reason: "budget", window: window } if window
307
+ end
308
+
309
+ # The turn's real billed spend for the persona + judge calls (the A4
310
+ # rule: total + cached + cache_creation), charged to the CALLING agent —
311
+ # never the target, whose own turns are billed normally by the edge
312
+ # limiter, exactly like a real customer's would be.
313
+ def account_spend(meter)
314
+ return if @budget_ledger.nil? || meter.empty?
315
+
316
+ tokens = meter.sum do |m|
317
+ m.input_tokens.to_i + m.output_tokens.to_i +
318
+ (m.respond_to?(:cached_tokens) ? m.cached_tokens.to_i : 0) +
319
+ (m.respond_to?(:cache_creation_tokens) ? m.cache_creation_tokens.to_i : 0)
320
+ end
321
+ return if tokens.zero?
322
+
323
+ @budget_ledger.add(tenant: turn_context&.dig(:command_tenant),
324
+ agent: turn_context&.dig(:agent_id), by: tokens)
325
+ end
326
+ end
327
+ end
328
+ end