insika 0.7.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +116 -0
- data/README.md +4 -3
- data/bin/insika +43 -2
- data/docs/AGENTS.md +22 -3
- data/docs/ARTIFACTS.md +42 -0
- data/docs/CONTEXT.md +54 -12
- data/docs/DEPLOY.md +16 -7
- data/docs/LOADTEST.md +15 -27
- data/docs/MEDIA.md +1 -1
- data/docs/OBSERVABILITY.md +31 -2
- data/docs/POLICY.md +10 -4
- data/docs/RUNNING-LOCAL.md +2 -2
- data/docs/SECURITY.md +1 -1
- data/docs/SOAK.md +1 -1
- data/docs/TOOLS.md +24 -0
- data/docs/prompts/GO-LIVE.md +3 -3
- data/lib/insika/agent_profile.rb +26 -1
- data/lib/insika/chat_builder.rb +28 -17
- data/lib/insika/compaction.rb +196 -0
- data/lib/insika/context/builder.rb +6 -2
- data/lib/insika/context/fragment.rb +4 -1
- data/lib/insika/context/priority.rb +6 -0
- data/lib/insika/context/providers/briefing.rb +53 -24
- data/lib/insika/context/providers/session.rb +46 -10
- data/lib/insika/context_trace_store.rb +11 -1
- data/lib/insika/doctor.rb +105 -10
- data/lib/insika/dsl/runtime.rb +4 -0
- data/lib/insika/env_schema.rb +5 -6
- data/lib/insika/evals/transport.rb +1 -1
- data/lib/insika/executor.rb +64 -0
- data/lib/insika/loop_detector.rb +5 -34
- data/lib/insika/profile_source.rb +7 -0
- data/lib/insika/server/responses.rb +4 -4
- data/lib/insika/session_store.rb +34 -4
- data/lib/insika/settings_store.rb +8 -1
- data/lib/insika/soak/runner.rb +4 -4
- data/lib/insika/studio/app.rb +28 -6
- data/lib/insika/studio/forms.rb +11 -0
- data/lib/insika/studio/views/settings.erb +11 -0
- data/lib/insika/telemetry/recorder.rb +49 -1
- data/lib/insika/templates/daily-digest/README.md +9 -0
- data/lib/insika/templates/research-analyst/agent.rb +10 -0
- data/lib/insika/tool_batch.rb +67 -0
- data/lib/insika/tool_usage_report.rb +162 -0
- data/lib/insika/turn_budget.rb +91 -0
- data/lib/insika/version.rb +1 -1
- data/lib/insika.rb +7 -0
- metadata +5 -1
data/lib/insika/session_store.rb
CHANGED
|
@@ -23,11 +23,11 @@ module Insika
|
|
|
23
23
|
KEY_PREFIX = "session:"
|
|
24
24
|
|
|
25
25
|
Session = Data.define(:id, :messages, :vars, :memory_refs,
|
|
26
|
-
:created_at, :updated_at, :briefing, :evidence) do
|
|
26
|
+
:created_at, :updated_at, :briefing, :evidence, :compaction) do
|
|
27
27
|
# Trailing members with defaults: an old record without the "briefing" /
|
|
28
|
-
# "evidence" keys reads as empty/nil without a migration.
|
|
28
|
+
# "evidence" / "compaction" keys reads as empty/nil without a migration.
|
|
29
29
|
def initialize(id:, messages:, vars:, memory_refs:, created_at:, updated_at:,
|
|
30
|
-
briefing: nil, evidence: nil)
|
|
30
|
+
briefing: nil, evidence: nil, compaction: nil)
|
|
31
31
|
super
|
|
32
32
|
end
|
|
33
33
|
end
|
|
@@ -150,6 +150,35 @@ module Insika
|
|
|
150
150
|
to_session(record)
|
|
151
151
|
end
|
|
152
152
|
|
|
153
|
+
# -> Session. Persists the in-session compaction state (RFC-0044): the
|
|
154
|
+
# summary of messages[0...upto] plus the boundary. `upto` is MONOTONIC —
|
|
155
|
+
# a write that does not move the boundary forward is a no-op (a stale
|
|
156
|
+
# double-write from a racing worker can never move it backwards). `runs`
|
|
157
|
+
# counts compactions over the session's lifetime (the trace reports it).
|
|
158
|
+
#
|
|
159
|
+
# CONCURRENCY NOTE: an unlocked RMW (read -> mutate -> set), like
|
|
160
|
+
# update_briefing — and safe for the same reason: nothing in the
|
|
161
|
+
# read/mutate/set path suspends, so no other writer can interleave
|
|
162
|
+
# mid-RMW. The LLM call that produced the summary happened BEFORE this
|
|
163
|
+
# method; only the plain write lives here.
|
|
164
|
+
def set_compaction(id, summary:, upto:, model: nil)
|
|
165
|
+
record = fetch!(id)
|
|
166
|
+
current = record["compaction"]
|
|
167
|
+
upto = Integer(upto)
|
|
168
|
+
return to_session(record) if current && upto <= current["upto"].to_i
|
|
169
|
+
|
|
170
|
+
record["compaction"] = {
|
|
171
|
+
"summary" => Coercion.utf8(summary.to_s),
|
|
172
|
+
"upto" => upto,
|
|
173
|
+
"runs" => (current ? current["runs"].to_i : 0) + 1,
|
|
174
|
+
"model" => Coercion.presence(model.to_s),
|
|
175
|
+
"at" => timestamp
|
|
176
|
+
}.compact
|
|
177
|
+
record["updated_at"] = timestamp
|
|
178
|
+
@store.set(SCOPE, key_for(id), record)
|
|
179
|
+
to_session(record)
|
|
180
|
+
end
|
|
181
|
+
|
|
153
182
|
# -> bool (delegates to the backend: false for a nonexistent id)
|
|
154
183
|
def delete(id)
|
|
155
184
|
@store.delete(SCOPE, key_for(id))
|
|
@@ -189,7 +218,8 @@ module Insika
|
|
|
189
218
|
created_at: record["created_at"],
|
|
190
219
|
updated_at: record["updated_at"],
|
|
191
220
|
briefing: record["briefing"] || { "fields" => {}, "next_step" => nil },
|
|
192
|
-
evidence: record["evidence"]
|
|
221
|
+
evidence: record["evidence"],
|
|
222
|
+
compaction: record["compaction"]
|
|
193
223
|
)
|
|
194
224
|
end
|
|
195
225
|
|
|
@@ -30,7 +30,14 @@ module Insika
|
|
|
30
30
|
"max_retries" => 2,
|
|
31
31
|
"turn_timeout" => 120,
|
|
32
32
|
"tool_timeout" => 30,
|
|
33
|
-
|
|
33
|
+
# In-session compaction (RFC-0044): when a session's UNCOMPACTED
|
|
34
|
+
# transcript grows past `compact_after` messages, everything but the
|
|
35
|
+
# last `keep_last` is summarized (model -> compaction.model, else the
|
|
36
|
+
# platform utility_model) into one history fragment. `prompt` replaces
|
|
37
|
+
# the engine default wholesale (the distill convention). enabled: false
|
|
38
|
+
# = parity (nothing runs). Additive keys — reads overlay DEFAULTS.
|
|
39
|
+
"compaction" => { "enabled" => false, "keep_last" => 20,
|
|
40
|
+
"compact_after" => 40, "model" => nil },
|
|
34
41
|
# Data lifecycle (WS8, phase 2): the RETENTION window in days. The
|
|
35
42
|
# tick's Retention sweep purges sessions (+traces), terminal tasks
|
|
36
43
|
# (+checkpoints), memory cells and outcomes older than this. nil/0 =
|
data/lib/insika/soak/runner.rb
CHANGED
|
@@ -50,7 +50,7 @@ module Insika
|
|
|
50
50
|
|
|
51
51
|
Environment:
|
|
52
52
|
INSIKA_URL base URL of the engine (default: http://localhost:9292)
|
|
53
|
-
|
|
53
|
+
INSIKA_GATEWAY_TOKEN Bearer; falls back to ADMIN_TOKEN, then "local-demo"
|
|
54
54
|
TXT
|
|
55
55
|
|
|
56
56
|
# Poisson inter-arrival seconds: `-mean * Math.log(1.0 - rand)`. Seeded,
|
|
@@ -245,7 +245,7 @@ module Insika
|
|
|
245
245
|
" agent: #{@agent}",
|
|
246
246
|
" shape: #{@envelope[:turns_per_hour]} turns/h poisson, #{@envelope[:session_turns]}-turn sessions, " \
|
|
247
247
|
"cap #{@envelope[:concurrency_cap]}, #{@envelope[:duration_hours]}h (#{@envelope[:warmup_hours]}h warmup)",
|
|
248
|
-
" sample: POST #{URI.join(target_url + '/', 'v1/responses')} model=
|
|
248
|
+
" sample: POST #{URI.join(target_url + '/', 'v1/responses')} model=insika:#{@agent} user=soak-1",
|
|
249
249
|
" out: #{@out}/"
|
|
250
250
|
]
|
|
251
251
|
unless @envelope.calibrated?
|
|
@@ -345,7 +345,7 @@ module Insika
|
|
|
345
345
|
|
|
346
346
|
private
|
|
347
347
|
|
|
348
|
-
def resolve_token(env) = env["
|
|
348
|
+
def resolve_token(env) = env["INSIKA_GATEWAY_TOKEN"] || env["ADMIN_TOKEN"] || "local-demo"
|
|
349
349
|
|
|
350
350
|
def install_traps
|
|
351
351
|
%w[INT TERM].each { |sig| Signal.trap(sig) { @stop_reason = "interrupted" } }
|
|
@@ -489,7 +489,7 @@ module Insika
|
|
|
489
489
|
req["Authorization"] = "Bearer #{@token}"
|
|
490
490
|
req["Content-Type"] = "application/json"
|
|
491
491
|
req["Accept"] = "text/event-stream"
|
|
492
|
-
req.body = JSON.generate(model: "
|
|
492
|
+
req.body = JSON.generate(model: "insika:#{agent}", user: user, stream: true, input: message)
|
|
493
493
|
|
|
494
494
|
t0 = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
495
495
|
ttfb = nil
|
data/lib/insika/studio/app.rb
CHANGED
|
@@ -2078,7 +2078,14 @@ end
|
|
|
2078
2078
|
"providers" => insika[:llm_provider_store] ? insika[:llm_provider_store].all.size : 0,
|
|
2079
2079
|
"MCP servers" => insika[:mcp_store] ? insika[:mcp_store].all.size : 0
|
|
2080
2080
|
}
|
|
2081
|
-
|
|
2081
|
+
# UTC, deliberately: sessions stamp `updated_at` with `Time.now.utc.iso8601`,
|
|
2082
|
+
# and every bucket below is keyed by a calendar part (date, hour) of that
|
|
2083
|
+
# stamp. A LOCAL `now` mixes two clocks — on a UTC-3 host, from 21:00 local
|
|
2084
|
+
# onward "today" is already tomorrow in UTC, so the day buckets stopped
|
|
2085
|
+
# matching and the 24h floor was built three hours in the future, silently
|
|
2086
|
+
# emptying both charts. Instant comparisons (`cutoff`) never had the bug;
|
|
2087
|
+
# calendar arithmetic did.
|
|
2088
|
+
now = Time.now.utc
|
|
2082
2089
|
cutoff = now - (5 * 60)
|
|
2083
2090
|
@active_now = sessions.count { |s| (t = parse_time(s.updated_at)) && t >= cutoff }
|
|
2084
2091
|
@recent = sessions.sort_by { |s| s.updated_at.to_s }.reverse.first(8)
|
|
@@ -2089,8 +2096,11 @@ end
|
|
|
2089
2096
|
# Trend affordances on the traffic stat cards: today vs yesterday from
|
|
2090
2097
|
# the 14-day series (config counts — agents/skills/tools/providers —
|
|
2091
2098
|
# have no daily shape and honestly show no trend).
|
|
2092
|
-
today,
|
|
2093
|
-
|
|
2099
|
+
# `@activity` is OLDEST FIRST and ends at today, so the last pair reads
|
|
2100
|
+
# [yesterday, today]. Destructured the other way round it reported a
|
|
2101
|
+
# first-conversation-of-the-day as "−1", every day.
|
|
2102
|
+
yesterday, today = @activity.last(2).map(&:last)
|
|
2103
|
+
@conv_trend = today.to_i - yesterday.to_i
|
|
2094
2104
|
@msg_trend = message_delta(sessions, now)
|
|
2095
2105
|
@persistence = insika.dig(:config, :persistence)
|
|
2096
2106
|
view("home")
|
|
@@ -2113,11 +2123,14 @@ end
|
|
|
2113
2123
|
end
|
|
2114
2124
|
|
|
2115
2125
|
# [[Date, count], …] — one bucket per day over the window, most-recent last.
|
|
2126
|
+
# `now` is UTC (render_home) and so is every `t` — see #utc_time. Both sides
|
|
2127
|
+
# of the bucket key must be read off the same clock or the join silently
|
|
2128
|
+
# misses.
|
|
2116
2129
|
def activity_by_day(sessions, days:, now:)
|
|
2117
2130
|
today = now.to_date
|
|
2118
2131
|
buckets = Hash.new(0)
|
|
2119
2132
|
sessions.each do |s|
|
|
2120
|
-
t =
|
|
2133
|
+
t = utc_time(s.updated_at) or next
|
|
2121
2134
|
buckets[t.to_date] += 1
|
|
2122
2135
|
end
|
|
2123
2136
|
(0...days).to_a.reverse.map { |i| d = today - i; [d, buckets[d]] }
|
|
@@ -2131,7 +2144,7 @@ end
|
|
|
2131
2144
|
floor = Time.utc(now.year, now.month, now.day, now.hour) - (hours - 1) * 3600
|
|
2132
2145
|
buckets = Hash.new(0)
|
|
2133
2146
|
sessions.each do |s|
|
|
2134
|
-
t =
|
|
2147
|
+
t = utc_time(s.updated_at) or next
|
|
2135
2148
|
h = Time.utc(t.year, t.month, t.day, t.hour)
|
|
2136
2149
|
buckets[h] += 1 if h >= floor
|
|
2137
2150
|
end
|
|
@@ -2144,11 +2157,20 @@ end
|
|
|
2144
2157
|
def message_delta(sessions, now)
|
|
2145
2158
|
today = now.to_date
|
|
2146
2159
|
sum = ->(date) do
|
|
2147
|
-
sessions.sum { |s| (t =
|
|
2160
|
+
sessions.sum { |s| (t = utc_time(s.updated_at)) && t.to_date == date ? Array(s.messages).size : 0 }
|
|
2148
2161
|
end
|
|
2149
2162
|
sum.call(today) - sum.call(today - 1)
|
|
2150
2163
|
end
|
|
2151
2164
|
|
|
2165
|
+
# A stamp read for its CALENDAR parts, always in UTC. `Time.parse` honours
|
|
2166
|
+
# whatever offset the string carries — ours are `Z`, but a record written by
|
|
2167
|
+
# anything else would otherwise bucket by its own zone. `getutc`, not `utc`:
|
|
2168
|
+
# the latter mutates the receiver.
|
|
2169
|
+
def utc_time(value)
|
|
2170
|
+
t = parse_time(value)
|
|
2171
|
+
t&.getutc
|
|
2172
|
+
end
|
|
2173
|
+
|
|
2152
2174
|
# --- History -------------------------------------------------------------
|
|
2153
2175
|
|
|
2154
2176
|
# Recent conversations (all agents — the Session doesn't stamp the agent that
|
data/lib/insika/studio/forms.rb
CHANGED
|
@@ -461,6 +461,17 @@ module Studio
|
|
|
461
461
|
# per-tenant map in the record — the view says so.
|
|
462
462
|
v = presence(r.params["memory_ttl_days"])
|
|
463
463
|
patch["memory_ttl_days"] = v.nil? ? nil : Integer(v)
|
|
464
|
+
# In-session compaction (RFC-0044). The checkbox is authoritative on
|
|
465
|
+
# this form (unchecked = disable); the numbers keep their stored value
|
|
466
|
+
# when cleared (deep_merge — the defaults backstop a fresh record);
|
|
467
|
+
# model blank = nil = the platform utility_model.
|
|
468
|
+
compaction = { "enabled" => r.params["compaction_enabled"] == "1",
|
|
469
|
+
"model" => presence(r.params["compaction_model"]) }
|
|
470
|
+
%w[keep_last compact_after].each do |f|
|
|
471
|
+
v = presence(r.params["compaction_#{f}"])
|
|
472
|
+
compaction[f] = Integer(v) if v
|
|
473
|
+
end
|
|
474
|
+
patch["compaction"] = compaction
|
|
464
475
|
patch
|
|
465
476
|
end
|
|
466
477
|
|
|
@@ -43,6 +43,17 @@
|
|
|
43
43
|
<label>Memory TTL (days)<input type="text" name="memory_ttl_days" value="<%= @settings["memory_ttl_days"] %>" inputmode="numeric" placeholder="blank = off"></label>
|
|
44
44
|
</div>
|
|
45
45
|
<p class="muted">Memory TTL: how many days a customer memory cell lives before the daily sweep prunes it (per-fact <code>expires_at</code> overrides always win). Editing here sets the platform default for <strong>every</strong> cell; a per-tenant map authored in the settings record is replaced by this save.</p>
|
|
46
|
+
<hr>
|
|
47
|
+
<label class="check">
|
|
48
|
+
<input type="checkbox" name="compaction_enabled" value="1"<%== " checked" if @settings.dig("compaction", "enabled") %>>
|
|
49
|
+
in-session compaction — summarize old turns past the threshold
|
|
50
|
+
</label>
|
|
51
|
+
<div class="grid-2">
|
|
52
|
+
<label>keep_last <span class="muted">(messages kept verbatim)</span><input type="text" name="compaction_keep_last" value="<%= @settings.dig("compaction", "keep_last") %>" inputmode="numeric"></label>
|
|
53
|
+
<label>compact_after <span class="muted">(uncompacted messages that trigger it)</span><input type="text" name="compaction_compact_after" value="<%= @settings.dig("compaction", "compact_after") %>" inputmode="numeric"></label>
|
|
54
|
+
</div>
|
|
55
|
+
<label>compaction model <span class="muted">(blank = the platform utility_model)</span><input type="text" name="compaction_model" value="<%= @settings.dig("compaction", "model") %>" placeholder="deepseek-v4-flash"></label>
|
|
56
|
+
<p class="muted">When a session's uncompacted transcript grows past <code>compact_after</code> messages, everything but the last <code>keep_last</code> is summarized by the cheap model into one <code><conversation_summary></code> fragment; the tail stays verbatim and the boundary is stable (the prompt cache holds after it). Runs post-turn, off the critical path.</p>
|
|
46
57
|
</form>
|
|
47
58
|
</div>
|
|
48
59
|
|
|
@@ -40,7 +40,8 @@ module Insika
|
|
|
40
40
|
# and units are part of the documented contract (docs/OBSERVABILITY.md) —
|
|
41
41
|
# renaming one breaks every dashboard built on it.
|
|
42
42
|
class Instruments
|
|
43
|
-
attr_reader :turns, :turn_duration, :tokens, :cost, :tool_calls, :tool_duration
|
|
43
|
+
attr_reader :turns, :turn_duration, :tokens, :cost, :tool_calls, :tool_duration,
|
|
44
|
+
:cache_hit_rate, :loop_intervened, :context_compacted
|
|
44
45
|
|
|
45
46
|
def initialize(meter)
|
|
46
47
|
@turns = meter.create_counter("insika.turns", unit: "{turn}",
|
|
@@ -55,6 +56,12 @@ module Insika
|
|
|
55
56
|
description: "Tool invocations")
|
|
56
57
|
@tool_duration = meter.create_histogram("insika.tool.duration", unit: "s",
|
|
57
58
|
description: "Wall time of a tool call")
|
|
59
|
+
@cache_hit_rate = meter.create_histogram("insika.cache.hit_rate", unit: "%",
|
|
60
|
+
description: "Prompt-cache hit rate of a turn (cached / billed prompt tokens)")
|
|
61
|
+
@loop_intervened = meter.create_counter("insika.tool.loop_intervened", unit: "{intervention}",
|
|
62
|
+
description: "Loop-detector warnings delivered to the model")
|
|
63
|
+
@context_compacted = meter.create_counter("insika.context.compacted", unit: "{compaction}",
|
|
64
|
+
description: "In-session compactions persisted (RFC-0044)")
|
|
58
65
|
end
|
|
59
66
|
end
|
|
60
67
|
|
|
@@ -73,6 +80,8 @@ module Insika
|
|
|
73
80
|
when :tool_call then start_tool(meta, data)
|
|
74
81
|
when :tool_result then finish_tool(meta)
|
|
75
82
|
when :data_tool_call then point_tool(meta, data)
|
|
83
|
+
when :tool_loop_intervened then count_loop(meta, data)
|
|
84
|
+
when :context_compacted then count_compaction(data)
|
|
76
85
|
when :task_completed then finish_turn(meta, data, :ok)
|
|
77
86
|
when :task_failed then finish_turn(meta, data, :error)
|
|
78
87
|
when :task_cancelled then finish_turn(meta, data, :cancelled)
|
|
@@ -166,6 +175,45 @@ module Insika
|
|
|
166
175
|
@instruments.turns.add(1, attributes: labels)
|
|
167
176
|
@instruments.turn_duration.record(seconds, attributes: labels) if seconds
|
|
168
177
|
count_usage(turn, usage)
|
|
178
|
+
count_cache_hit(turn, usage)
|
|
179
|
+
end
|
|
180
|
+
|
|
181
|
+
# Same arithmetic as the Executor's per-agent series (stamp_cache_hit):
|
|
182
|
+
# the billed prompt is fresh input + cache reads + cache writes, and the
|
|
183
|
+
# hit rate is reads over the whole billed prompt — always in [0,100]. A
|
|
184
|
+
# turn with no billed prompt tokens (no usage, usage without the fields)
|
|
185
|
+
# records nothing: absence is not a 0% hit.
|
|
186
|
+
def count_cache_hit(turn, usage)
|
|
187
|
+
return unless usage
|
|
188
|
+
|
|
189
|
+
billed = usage[:input_tokens].to_i + usage[:cached_tokens].to_i +
|
|
190
|
+
usage[:cache_creation_tokens].to_i
|
|
191
|
+
return unless billed.positive?
|
|
192
|
+
|
|
193
|
+
rate = (usage[:cached_tokens].to_i * 100.0) / billed
|
|
194
|
+
base = turn.labels.merge(attrs("insika.model" => usage[:model]&.to_s))
|
|
195
|
+
@instruments.cache_hit_rate.record(rate, attributes: base)
|
|
196
|
+
end
|
|
197
|
+
|
|
198
|
+
# The loop detector delivered its one-shot warning (`:tool_loop_intervened`,
|
|
199
|
+
# counts and the tool name, never arguments). An orphan event (no open turn)
|
|
200
|
+
# is ignored, like every other consumer of this stream.
|
|
201
|
+
def count_loop(meta, data)
|
|
202
|
+
return unless @instruments
|
|
203
|
+
|
|
204
|
+
turn = @turns[meta[:task_id]] or return
|
|
205
|
+
labels = turn.labels.merge(attrs("insika.tool" => data[:name]&.to_s))
|
|
206
|
+
@instruments.loop_intervened.add(1, attributes: labels)
|
|
207
|
+
end
|
|
208
|
+
|
|
209
|
+
# A compaction persisted (`:context_compacted`, RFC-0044) — counted by
|
|
210
|
+
# agent/model, INDEPENDENT of any open turn: it fires post-turn, usually
|
|
211
|
+
# after task_completed already closed the span.
|
|
212
|
+
def count_compaction(data)
|
|
213
|
+
return unless @instruments
|
|
214
|
+
|
|
215
|
+
labels = attrs("insika.agent" => data[:agent]&.to_s, "insika.model" => data[:model]&.to_s)
|
|
216
|
+
@instruments.context_compacted.add(1, attributes: labels)
|
|
169
217
|
end
|
|
170
218
|
|
|
171
219
|
# Tokens ride ONE counter split by `insika.token.type` (instead of four
|
|
@@ -31,6 +31,15 @@ in the Playground — the artifact lands on the Artifacts tab either way.
|
|
|
31
31
|
- The **numbers** are not. Swap the literal string in `agent.rb` for a
|
|
32
32
|
`data_tool` against your own sales API and nothing else changes.
|
|
33
33
|
|
|
34
|
+
## When the report needs real data
|
|
35
|
+
|
|
36
|
+
One agent doing 30–50 tool calls at one reasoning effort is how a report turn
|
|
37
|
+
hits the 300 s timeout. The recipe is to split the phases across agents —
|
|
38
|
+
`thinking: "low"` miners fanned out with `spawn_subagents`, a `thinking: "high"`
|
|
39
|
+
orchestrator that plans and writes. See
|
|
40
|
+
[Artifacts](https://github.com/guizaols/insika/blob/main/docs/ARTIFACTS.md), "Reasoning effort on a report turn",
|
|
41
|
+
and the `research-analyst` template for the fan-out shape.
|
|
42
|
+
|
|
34
43
|
## Edit it
|
|
35
44
|
|
|
36
45
|
The skill's instructions are the actual report spec — change the palette,
|
|
@@ -12,6 +12,12 @@
|
|
|
12
12
|
# of a business idea, IN PARALLEL (each in its own isolated context), then
|
|
13
13
|
# the lead synthesizes one recommendation.
|
|
14
14
|
#
|
|
15
|
+
# Reasoning effort is split by ROLE, not spread evenly: the specialists answer
|
|
16
|
+
# one narrow question each (`thinking: "low"`), the lead plans the delegation
|
|
17
|
+
# and weighs three answers against each other (`thinking: "high"`). A child
|
|
18
|
+
# inherits the environment as a DEFAULT only, so its own `params` wins. See
|
|
19
|
+
# docs/ARTIFACTS.md, "Reasoning effort on a report turn".
|
|
20
|
+
#
|
|
15
21
|
# DEEPSEEK_API_KEY=sk-... ruby research-analyst/agent.rb "a subscription box for specialty coffee"
|
|
16
22
|
# DEEPSEEK_API_KEY=sk-... ruby research-analyst/agent.rb --serve
|
|
17
23
|
require "insika"
|
|
@@ -21,19 +27,23 @@ team = Insika.system do
|
|
|
21
27
|
|
|
22
28
|
agent("market") do
|
|
23
29
|
model "deepseek-v4-flash"
|
|
30
|
+
params thinking: "low"
|
|
24
31
|
instructions "Research the MARKET angle of a business idea: audience, demand, competitors. Three sentences."
|
|
25
32
|
end
|
|
26
33
|
agent("technical") do
|
|
27
34
|
model "deepseek-v4-flash"
|
|
35
|
+
params thinking: "low"
|
|
28
36
|
instructions "Research the TECHNICAL/OPERATIONAL angle of a business idea: what it takes to build and run it. Three sentences."
|
|
29
37
|
end
|
|
30
38
|
agent("risk") do
|
|
31
39
|
model "deepseek-v4-flash"
|
|
40
|
+
params thinking: "low"
|
|
32
41
|
instructions "Research the RISK angle of a business idea: what could make it fail. Three sentences."
|
|
33
42
|
end
|
|
34
43
|
|
|
35
44
|
agent "analyst" do
|
|
36
45
|
model "deepseek-v4-flash"
|
|
46
|
+
params thinking: "high"
|
|
37
47
|
instructions <<~PROMPT
|
|
38
48
|
You are a research LEAD with no expertise of your own — never answer
|
|
39
49
|
from your own knowledge. Given a business idea, call spawn_subagents
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Insika
|
|
4
|
+
# The batch arithmetic behind every mid-turn `user` append.
|
|
5
|
+
#
|
|
6
|
+
# A model step that calls tools produces ONE assistant message announcing N
|
|
7
|
+
# tool calls, followed by N `role: tool` messages. Anthropic rejects a `user`
|
|
8
|
+
# message that lands between two of those results outright, so the ONLY valid
|
|
9
|
+
# append point inside a turn is the instant the Nth result closes the batch.
|
|
10
|
+
# SteerInjector discovered this rule; LoopDetector and TurnBudget both live by
|
|
11
|
+
# it, which is why the counting lives here instead of twice.
|
|
12
|
+
#
|
|
13
|
+
# Not a general-purpose helper: it answers one question ("did a batch just
|
|
14
|
+
# close?") and remembers one fact ("was this batch halted"), because a
|
|
15
|
+
# `halt_when` batch has no next model step and anything appended there would
|
|
16
|
+
# sit unread forever.
|
|
17
|
+
class ToolBatch
|
|
18
|
+
def initialize
|
|
19
|
+
@expected = nil # tool calls announced by the batch in flight (nil = none)
|
|
20
|
+
@seen = 0
|
|
21
|
+
@halted = false
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
# Feeds a RubyLLM message (duck-typed). True EXACTLY on the message that
|
|
25
|
+
# closes a batch of tool calls — the append boundary. Everything else,
|
|
26
|
+
# including the assistant message that opens the batch, is false.
|
|
27
|
+
def closed?(message)
|
|
28
|
+
role = field(message, :role).to_s
|
|
29
|
+
return open(message) if role == "assistant"
|
|
30
|
+
return false unless role == "tool" && @expected
|
|
31
|
+
|
|
32
|
+
@seen += 1
|
|
33
|
+
return false if @seen < @expected
|
|
34
|
+
|
|
35
|
+
@expected = nil
|
|
36
|
+
true
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
# From after_tool_result, with the RAW result: a Tool::Halt is only
|
|
40
|
+
# recognizable there.
|
|
41
|
+
def halt!(result)
|
|
42
|
+
@halted = true if defined?(RubyLLM::Tool::Halt) && result.is_a?(RubyLLM::Tool::Halt)
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
def halted? = @halted
|
|
46
|
+
|
|
47
|
+
private
|
|
48
|
+
|
|
49
|
+
# An assistant message with no tool calls is the model TALKING: the turn is
|
|
50
|
+
# ending, so nothing is in flight any more.
|
|
51
|
+
def open(message)
|
|
52
|
+
calls = field(message, :tool_calls)
|
|
53
|
+
size = calls.respond_to?(:size) ? calls.size : 0
|
|
54
|
+
@expected = size.zero? ? nil : size
|
|
55
|
+
@seen = 0
|
|
56
|
+
@halted = false
|
|
57
|
+
false
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
def field(message, name)
|
|
61
|
+
return message.public_send(name) if message.respond_to?(name)
|
|
62
|
+
return message[name] || message[name.to_s] if message.respond_to?(:[])
|
|
63
|
+
|
|
64
|
+
nil
|
|
65
|
+
end
|
|
66
|
+
end
|
|
67
|
+
end
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "time"
|
|
4
|
+
|
|
5
|
+
module Insika
|
|
6
|
+
# The tool AUDIT the per-session trace cannot answer: `tool_traces` records
|
|
7
|
+
# every call, but one session at a time, and nothing aggregates — so "which
|
|
8
|
+
# tools does this agent carry and never use?" had no surface at all. This
|
|
9
|
+
# report is that surface. Read-only by design: it names the candidates, the
|
|
10
|
+
# OPERATOR removes (a tool the report flags may still be the one a rare but
|
|
11
|
+
# critical flow needs).
|
|
12
|
+
#
|
|
13
|
+
# Attribution rides the task record: a session does not stamp its agent, but
|
|
14
|
+
# every task carries the agent in its command payload — tasks → sessions →
|
|
15
|
+
# tool_traces is the same read the Studio does. Three findings per agent:
|
|
16
|
+
#
|
|
17
|
+
# never_called — in `tools_allow`, zero calls in any stored trace. Dead
|
|
18
|
+
# weight: it costs schema tokens on every request and buys
|
|
19
|
+
# nothing (the pilot's `search_orders` is the known example).
|
|
20
|
+
# error_rate — called inside the window with > 30% conventional errors
|
|
21
|
+
# (the trace's own `ok` flag). Either the tool is broken or
|
|
22
|
+
# the model cannot hold its contract; both are operator work.
|
|
23
|
+
# stale — called at some point, but not once inside the window.
|
|
24
|
+
#
|
|
25
|
+
# Bounded and honest: it reads only what the stores kept (the trace caps at
|
|
26
|
+
# 200 entries/session), so a count here is "at least", never an exact total.
|
|
27
|
+
class ToolUsageReport
|
|
28
|
+
WINDOW_DAYS = 14
|
|
29
|
+
ERROR_RATE_THRESHOLD = 0.30
|
|
30
|
+
|
|
31
|
+
# One finding. kind: "never_called" | "error_rate" | "stale".
|
|
32
|
+
Row = Data.define(:agent, :tool, :kind, :detail) do
|
|
33
|
+
def to_h = { "agent" => agent, "tool" => tool, "kind" => kind, "detail" => detail }
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
Report = Data.define(:generated_at, :days, :agents, :rows) do
|
|
37
|
+
def to_h
|
|
38
|
+
{ "generated_at" => generated_at, "days" => days, "agents" => agents,
|
|
39
|
+
"rows" => rows.map(&:to_h) }
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
# Human report, grouped by agent. Silent agents still print their header —
|
|
43
|
+
# "nothing flagged" is a result, not an omission.
|
|
44
|
+
def to_s
|
|
45
|
+
lines = ["tool usage — last #{days} day(s), generated #{generated_at}"]
|
|
46
|
+
agents.each do |agent|
|
|
47
|
+
mine = rows.select { |r| r.agent == agent }
|
|
48
|
+
lines << "" << "#{agent}: #{mine.empty? ? 'nothing flagged' : "#{mine.length} finding(s)"}"
|
|
49
|
+
mine.each { |r| lines << format(" %-13s %s — %s", r.kind, r.tool, r.detail) }
|
|
50
|
+
end
|
|
51
|
+
lines.join("\n")
|
|
52
|
+
end
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
def initialize(task_store:, tool_trace_store:, profile_source:, now: nil)
|
|
56
|
+
@task_store = task_store
|
|
57
|
+
@tool_trace_store = tool_trace_store
|
|
58
|
+
@profile_source = profile_source
|
|
59
|
+
@now = now
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
# -> Report. `agent:` narrows to one agent (must still be a stored profile).
|
|
63
|
+
def generate(days: WINDOW_DAYS, agent: nil)
|
|
64
|
+
now = @now || Time.now.utc
|
|
65
|
+
cutoff = now - (days * 24 * 60 * 60)
|
|
66
|
+
profiles = @profile_source.all_raw
|
|
67
|
+
profiles = profiles.select { |r| r["id"].to_s == agent.to_s } if agent
|
|
68
|
+
sessions = sessions_by_agent
|
|
69
|
+
|
|
70
|
+
rows = profiles.flat_map do |record|
|
|
71
|
+
id = record["id"].to_s
|
|
72
|
+
stats = tool_stats(sessions[id] || [], cutoff)
|
|
73
|
+
never_called_rows(id, record, stats) +
|
|
74
|
+
error_rate_rows(id, stats, days) +
|
|
75
|
+
stale_rows(id, stats)
|
|
76
|
+
end
|
|
77
|
+
|
|
78
|
+
Report.new(generated_at: now.iso8601, days: days,
|
|
79
|
+
agents: profiles.map { |r| r["id"].to_s }.sort,
|
|
80
|
+
rows: rows.sort_by { |r| [r.agent, r.kind, r.tool] }.freeze)
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
private
|
|
84
|
+
|
|
85
|
+
# agent id -> [session ids], via the task records (the only place a session
|
|
86
|
+
# is tied to its agent). A task without agent or session (operator commands,
|
|
87
|
+
# workflows without a chat) contributes nothing.
|
|
88
|
+
def sessions_by_agent
|
|
89
|
+
acc = Hash.new { |h, k| h[k] = [] }
|
|
90
|
+
@task_store.each_id do |task_id|
|
|
91
|
+
task = @task_store.find(task_id) or next
|
|
92
|
+
agent = task.command.is_a?(Hash) ? task.command.dig("payload", "agent") : nil
|
|
93
|
+
next if agent.to_s.empty? || task.session_id.to_s.empty?
|
|
94
|
+
|
|
95
|
+
acc[agent.to_s] << task.session_id
|
|
96
|
+
end
|
|
97
|
+
acc.transform_values(&:uniq)
|
|
98
|
+
end
|
|
99
|
+
|
|
100
|
+
# tool name -> { calls:, errors:, window_calls:, window_errors:, last_at: }
|
|
101
|
+
# over every stored trace entry of the agent's sessions.
|
|
102
|
+
def tool_stats(session_ids, cutoff)
|
|
103
|
+
stats = Hash.new { |h, k| h[k] = { calls: 0, errors: 0, window_calls: 0, window_errors: 0, last_at: nil } }
|
|
104
|
+
session_ids.each do |sid|
|
|
105
|
+
@tool_trace_store.for_session(sid).each do |entry|
|
|
106
|
+
s = stats[entry["tool"].to_s]
|
|
107
|
+
at = parse_time(entry["at"])
|
|
108
|
+
error = entry["ok"] == false
|
|
109
|
+
s[:calls] += 1
|
|
110
|
+
s[:errors] += 1 if error
|
|
111
|
+
s[:last_at] = at if at && (s[:last_at].nil? || at > s[:last_at])
|
|
112
|
+
next unless at && at >= cutoff
|
|
113
|
+
|
|
114
|
+
s[:window_calls] += 1
|
|
115
|
+
s[:window_errors] += 1 if error
|
|
116
|
+
end
|
|
117
|
+
end
|
|
118
|
+
stats
|
|
119
|
+
end
|
|
120
|
+
|
|
121
|
+
def never_called_rows(agent, record, stats)
|
|
122
|
+
allow = record["tools_allow"]
|
|
123
|
+
return [] if allow.nil? # no allowlist declared -> nothing to audit against
|
|
124
|
+
|
|
125
|
+
Array(allow).map(&:to_s).reject { |t| stats.key?(t) }.map do |tool|
|
|
126
|
+
Row.new(agent: agent, tool: tool, kind: "never_called",
|
|
127
|
+
detail: "in tools_allow, never called in any stored trace — " \
|
|
128
|
+
"its schema still ships on every request")
|
|
129
|
+
end
|
|
130
|
+
end
|
|
131
|
+
|
|
132
|
+
def error_rate_rows(agent, stats, days)
|
|
133
|
+
stats.filter_map do |tool, s|
|
|
134
|
+
next if s[:window_calls].zero?
|
|
135
|
+
|
|
136
|
+
rate = s[:window_errors].to_f / s[:window_calls]
|
|
137
|
+
next if rate <= ERROR_RATE_THRESHOLD
|
|
138
|
+
|
|
139
|
+
Row.new(agent: agent, tool: tool, kind: "error_rate",
|
|
140
|
+
detail: "#{s[:window_errors]}/#{s[:window_calls]} call(s) errored in the last " \
|
|
141
|
+
"#{days} day(s) (#{(rate * 100).round}%)")
|
|
142
|
+
end
|
|
143
|
+
end
|
|
144
|
+
|
|
145
|
+
def stale_rows(agent, stats)
|
|
146
|
+
stats.filter_map do |tool, s|
|
|
147
|
+
next if s[:window_calls].positive? || s[:last_at].nil?
|
|
148
|
+
|
|
149
|
+
Row.new(agent: agent, tool: tool, kind: "stale",
|
|
150
|
+
detail: "last called #{s[:last_at].iso8601}, not once inside the window")
|
|
151
|
+
end
|
|
152
|
+
end
|
|
153
|
+
|
|
154
|
+
def parse_time(value)
|
|
155
|
+
return nil if value.to_s.empty?
|
|
156
|
+
|
|
157
|
+
Time.parse(value.to_s).utc
|
|
158
|
+
rescue ArgumentError
|
|
159
|
+
nil
|
|
160
|
+
end
|
|
161
|
+
end
|
|
162
|
+
end
|