samagotchi 0.3.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +162 -1
- data/README.md +29 -2
- data/bin/chi +60 -69
- data/docs/cli.md +211 -77
- data/docs/configuration.md +118 -21
- data/docs/desktop.md +39 -4
- data/docs/guardrails.md +11 -0
- data/docs/hooks.md +89 -7
- data/docs/memory.md +40 -0
- data/docs/plugins.md +50 -0
- data/docs/releasing.md +15 -12
- data/docs/sessions.md +20 -18
- data/lib/samagotchi/bootstrap/config_writer.rb +1 -2
- data/lib/samagotchi/bridge/sse_writer.rb +0 -3
- data/lib/samagotchi/bridge/turn_accumulator.rb +2 -0
- data/lib/samagotchi/bridge.rb +20 -12
- data/lib/samagotchi/bundles/skills/manifest.yml +10 -0
- data/lib/samagotchi/bundles/skills/plugin.rb +419 -0
- data/lib/samagotchi/bundles/source-links/hooks/source_links.rb +178 -5
- data/lib/samagotchi/bundles/source-links/manifest.yml +3 -3
- data/lib/samagotchi/bundles/source-links/source_links.md +1 -1
- data/lib/samagotchi/bundles/system/config_modification_protocol.md +9 -10
- data/lib/samagotchi/bundles/system/delegated.md +6 -7
- data/lib/samagotchi/bundles/system/identity.md +5 -0
- data/lib/samagotchi/bundles/system/manifest.yml +6 -6
- data/lib/samagotchi/bundles/system/memory_guide.md +26 -0
- data/lib/samagotchi/bundles/system/self_map.md +2 -1
- data/lib/samagotchi/client.rb +25 -26
- data/lib/samagotchi/commands/registry.rb +8 -0
- data/lib/samagotchi/config.rb +97 -113
- data/lib/samagotchi/desktop/macos/App.swift +12 -8
- data/lib/samagotchi/desktop/macos/ChiRunner.swift +4 -2
- data/lib/samagotchi/desktop/macos/Images.swift +113 -0
- data/lib/samagotchi/desktop/macos/Info.plist.erb +6 -0
- data/lib/samagotchi/desktop/macos/Panel.swift +112 -9
- data/lib/samagotchi/desktop/macos.rb +59 -8
- data/lib/samagotchi/desktop_command.rb +6 -3
- data/lib/samagotchi/edit_preview.rb +82 -0
- data/lib/samagotchi/engine.rb +236 -443
- data/lib/samagotchi/gem_update.rb +89 -0
- data/lib/samagotchi/guardrails/approval.rb +26 -4
- data/lib/samagotchi/guardrails/load_failures.rb +9 -3
- data/lib/samagotchi/host_registry.rb +8 -12
- data/lib/samagotchi/idle_client.rb +24 -15
- data/lib/samagotchi/idle_reminders.rb +2 -2
- data/lib/samagotchi/image_store.rb +10 -6
- data/lib/samagotchi/kernel_loop.rb +59 -123
- data/lib/samagotchi/live_versions.rb +65 -0
- data/lib/samagotchi/llm/api_key.rb +41 -0
- data/lib/samagotchi/llm/chat_loop.rb +77 -13
- data/lib/samagotchi/llm/errors.rb +38 -7
- data/lib/samagotchi/llm/http.rb +19 -22
- data/lib/samagotchi/llm/openai_chat.rb +22 -26
- data/lib/samagotchi/memory_bundle/installer.rb +65 -63
- data/lib/samagotchi/memory_bundle/provenance.rb +51 -12
- data/lib/samagotchi/memory_bundle/shipped_update.rb +157 -0
- data/lib/samagotchi/memory_bundle/status.rb +4 -1
- data/lib/samagotchi/memory_bundle/system_bundle.rb +81 -53
- data/lib/samagotchi/model_profile.rb +27 -10
- data/lib/samagotchi/note_command.rb +2 -1
- data/lib/samagotchi/prompt.rb +4 -2
- data/lib/samagotchi/reminder_store.rb +1 -9
- data/lib/samagotchi/reply_wait.rb +48 -4
- data/lib/samagotchi/self_report.rb +37 -5
- data/lib/samagotchi/send_command.rb +190 -17
- data/lib/samagotchi/session.rb +4 -2
- data/lib/samagotchi/session_commands.rb +38 -8
- data/lib/samagotchi/session_manager.rb +19 -53
- data/lib/samagotchi/system_prompt.rb +403 -0
- data/lib/samagotchi/terminal_ui/attach_launcher.rb +5 -3
- data/lib/samagotchi/terminal_ui/attached_loop.rb +141 -108
- data/lib/samagotchi/terminal_ui/attached_view.rb +27 -12
- data/lib/samagotchi/terminal_ui/event_renderer.rb +29 -8
- data/lib/samagotchi/terminal_ui/formatting.rb +41 -22
- data/lib/samagotchi/terminal_ui/input_support.rb +7 -23
- data/lib/samagotchi/terminal_ui/plain_surface.rb +13 -7
- data/lib/samagotchi/terminal_ui/question_prompt.rb +35 -0
- data/lib/samagotchi/terminal_ui/status_row.rb +81 -0
- data/lib/samagotchi/terminal_ui/surface.rb +1 -1
- data/lib/samagotchi/terminal_ui.rb +142 -923
- data/lib/samagotchi/text_diff.rb +181 -0
- data/lib/samagotchi/thinking.rb +126 -0
- data/lib/samagotchi/tool_activity.rb +52 -2
- data/lib/samagotchi/tool_runner.rb +37 -1
- data/lib/samagotchi/tools/ask_user_question.rb +41 -33
- data/lib/samagotchi/tools/edit.rb +23 -9
- data/lib/samagotchi/tools/execute.rb +3 -3
- data/lib/samagotchi/tools/output_guardrails.rb +8 -7
- data/lib/samagotchi/tools/read.rb +4 -4
- data/lib/samagotchi/tools/write.rb +4 -0
- data/lib/samagotchi/turn_flow.rb +12 -2
- data/lib/samagotchi/update_command.rb +309 -0
- data/lib/samagotchi/update_hint.rb +59 -0
- data/lib/samagotchi/version.rb +1 -1
- data/lib/samagotchi/vision_support.rb +6 -4
- data/lib/samagotchi/web/app.rb +173 -38
- data/lib/samagotchi/web/lan.rb +99 -0
- data/lib/samagotchi/web/message_parts.rb +19 -10
- data/lib/samagotchi/web/public/activity.js +10 -0
- data/lib/samagotchi/web/public/app.js +135 -78
- data/lib/samagotchi/web/public/chat_view.js +8 -1
- data/lib/samagotchi/web/public/data.js +2 -0
- data/lib/samagotchi/web/public/diff_view.js +58 -0
- data/lib/samagotchi/web/public/index.html +185 -18
- data/lib/samagotchi/web/public/model_pick.js +136 -0
- data/lib/samagotchi/web/public/model_picker.js +224 -0
- data/lib/samagotchi/web/public/notify.js +10 -0
- data/lib/samagotchi/web/public/question_card.js +3 -1
- data/lib/samagotchi/web/public/stage_model.js +110 -0
- data/lib/samagotchi/web/public/stage_view.js +580 -0
- data/lib/samagotchi/web/public/timing.js +6 -2
- data/lib/samagotchi/web/public/turn_events.js +38 -10
- data/lib/samagotchi/web/public/turn_model.js +11 -3
- data/lib/samagotchi/web/public/turn_view.js +76 -20
- data/lib/samagotchi/web/qr.rb +40 -0
- data/lib/samagotchi/web/server.rb +101 -11
- data/lib/samagotchi/web/token.rb +97 -0
- data/lib/samagotchi/worker.rb +5 -4
- metadata +38 -3
- data/lib/samagotchi/terminal_ui/legacy_surface.rb +0 -111
|
@@ -11,6 +11,7 @@ require_relative "client"
|
|
|
11
11
|
require_relative "llm/errors"
|
|
12
12
|
require_relative "log"
|
|
13
13
|
require_relative "empty_answer_retry"
|
|
14
|
+
require_relative "thinking"
|
|
14
15
|
require_relative "hooks"
|
|
15
16
|
require_relative "pending_input_queue"
|
|
16
17
|
require_relative "steer"
|
|
@@ -111,16 +112,10 @@ module Samagotchi
|
|
|
111
112
|
CONTEXT_LINE_PREFIX = "[CONTEXT: "
|
|
112
113
|
CONTEXT_LINE_KIND = "context"
|
|
113
114
|
CONTEXT_GUIDANCE_FROM_RANK = 2
|
|
114
|
-
CONTEXT_STATUS_ENABLED_ENV = "SAMAGOTCHI_CONTEXT_STATUS"
|
|
115
|
-
CONTEXT_CHARS_PER_TOKEN_ENV = "SAMAGOTCHI_CONTEXT_CHARS_PER_TOKEN"
|
|
116
|
-
CONTEXT_THRESHOLDS_ENV = "SAMAGOTCHI_CONTEXT_STATUS_THRESHOLDS"
|
|
117
|
-
CONTEXT_CADENCE_ENV = "SAMAGOTCHI_CONTEXT_STATUS_CADENCE"
|
|
118
115
|
|
|
119
116
|
DEFAULT_CONTEXT_CHARS_PER_TOKEN = 4.0
|
|
120
117
|
DEFAULT_CONTEXT_THRESHOLDS = [20, 40, 60, 80].freeze
|
|
121
|
-
DEFAULT_CONTEXT_CADENCE = 0
|
|
122
118
|
DEFAULT_MAX_TOOL_OUTPUT_CHARS = 10_000
|
|
123
|
-
TOOL_OUTPUT_CHARS_ENV = "SAMAGOTCHI_MAX_TOOL_OUTPUT_CHARS"
|
|
124
119
|
QWEN_INCOMPLETE_TOOL_CALL_RECOVERY_LIMIT = 2
|
|
125
120
|
QWEN_INCOMPLETE_TOOL_CALL_RECOVERY_PROMPT = "Continue the previous assistant message by finishing the open <tool_call> XML block. Output only the remaining XML needed to complete the tool call."
|
|
126
121
|
|
|
@@ -179,6 +174,9 @@ module Samagotchi
|
|
|
179
174
|
# The turn's request parameters (SamplingSettings.for), set by the Engine
|
|
180
175
|
# per turn; empty or nil sends none.
|
|
181
176
|
attr_accessor :sampling
|
|
177
|
+
# The turn's thinking level (Thinking.resolve), set by the Engine per
|
|
178
|
+
# turn; its own accessor, since the native path sends @sampling as is.
|
|
179
|
+
attr_accessor :thinking
|
|
182
180
|
# Tools::Peers (or the Engine's live view of it): the session
|
|
183
181
|
# list_sessions and send_note speak for; nil outside a session.
|
|
184
182
|
attr_accessor :peers
|
|
@@ -193,7 +191,7 @@ module Samagotchi
|
|
|
193
191
|
# @param cancel_controller [CancellationController, nil] optional cancellation source
|
|
194
192
|
# @param model_name [String, nil] optional per-run model override
|
|
195
193
|
# @param max_tool_output_chars [Integer, nil] per-output char cap for the
|
|
196
|
-
# :tool_call_completed event's `output:` (nil →
|
|
194
|
+
# :tool_call_completed event's `output:` (nil → max_tool_output_chars)
|
|
197
195
|
# @param pending_input [#call, nil] optional drain proc returning
|
|
198
196
|
# Array<String> of user steering messages queued while the turn runs.
|
|
199
197
|
# Drained at iteration boundaries (llama.cpp's /completion cannot accept
|
|
@@ -217,13 +215,17 @@ module Samagotchi
|
|
|
217
215
|
@retry_generation = false
|
|
218
216
|
context_status = nil
|
|
219
217
|
stream_splitter = ThoughtStreamSplitter.for_profile(@profile)
|
|
218
|
+
# Qwen with thinking off: an empty thought after the cue, so the model
|
|
219
|
+
# answers at once. Kept in the turn's model messages, so each tool-loop
|
|
220
|
+
# prompt starts with what the server already has cached.
|
|
221
|
+
prefill = Thinking.native(@thinking || Thinking::DEFAULT, @profile).prefill
|
|
220
222
|
partial_assistant_buffer = +""
|
|
221
223
|
|
|
222
224
|
effective_max_iterations = @no_interrupt ? 1000 : max_iterations
|
|
223
225
|
effective_max_tool_output_chars = resolve_output_char_cap(max_tool_output_chars)
|
|
224
226
|
effective_max_iterations.times do |iteration_index|
|
|
225
227
|
inject_pending_input!(conversation, pending_input, on_stream_event, iteration_index + 1, cancel_controller)
|
|
226
|
-
prompt, images = Prompt.format_with_images(conversation, profile: @profile, vision: @vision)
|
|
228
|
+
prompt, images = Prompt.format_with_images(conversation, profile: @profile, vision: @vision, prefill: prefill)
|
|
227
229
|
image_tokens = images.empty? ? 0 : ImagePlan.estimated_tokens(conversation)
|
|
228
230
|
context_window = ContextWindow.resolve(client: @client, model: resolved_model_name)
|
|
229
231
|
context_status = emit_context_status_event(on_stream_event, prompt, iteration_index: iteration_index, state: context_state, window: context_window,
|
|
@@ -232,7 +234,7 @@ module Samagotchi
|
|
|
232
234
|
# The model's own copy, on the tail (the prompt cache keeps its
|
|
233
235
|
# prefix), then the prompt again with it.
|
|
234
236
|
conversation << line
|
|
235
|
-
prompt, images = Prompt.format_with_images(conversation, profile: @profile, vision: @vision)
|
|
237
|
+
prompt, images = Prompt.format_with_images(conversation, profile: @profile, vision: @vision, prefill: prefill)
|
|
236
238
|
end
|
|
237
239
|
emit_stream_event(
|
|
238
240
|
on_stream_event,
|
|
@@ -294,6 +296,9 @@ module Samagotchi
|
|
|
294
296
|
images: images
|
|
295
297
|
)
|
|
296
298
|
)
|
|
299
|
+
# What the server's prompt count covers, so the next estimate adds
|
|
300
|
+
# only what the turn appended since (answer, tool results).
|
|
301
|
+
context_state[:counted] = { chars: prompt.length, image_tokens: image_tokens } if generation_usage
|
|
297
302
|
refresh_context_display(context_state, generation_usage, context_window)
|
|
298
303
|
emit_stream_event(
|
|
299
304
|
on_stream_event,
|
|
@@ -310,7 +315,7 @@ module Samagotchi
|
|
|
310
315
|
after_gen_event = { type: :after_generation, iteration: iteration_index + 1, response: response,
|
|
311
316
|
messages: AnswerDisplay.strip_all(conversation).map(&:dup).freeze }
|
|
312
317
|
fire_hook(:after_generation, after_gen_event) if @hooks
|
|
313
|
-
conversation << { role: "model", content: response }
|
|
318
|
+
conversation << { role: "model", content: prefill + response.to_s }
|
|
314
319
|
|
|
315
320
|
# Profile-specific parse (incl. Qwen unterminated-block recovery); the
|
|
316
321
|
# returned fragment (non-nil only for Qwen) is fed back on the next
|
|
@@ -365,6 +370,7 @@ module Samagotchi
|
|
|
365
370
|
image_counts = []
|
|
366
371
|
shown_params = []
|
|
367
372
|
shown_labels = []
|
|
373
|
+
diffs = []
|
|
368
374
|
results = calls.map.with_index do |call, call_index|
|
|
369
375
|
run = tool_runner.run(call, iteration: iteration_index + 1, call_index: call_index + 1,
|
|
370
376
|
call_count: calls.length, on_stream_event: on_stream_event,
|
|
@@ -374,6 +380,7 @@ module Samagotchi
|
|
|
374
380
|
image_counts << Array(run[:images]).size
|
|
375
381
|
shown_params << run[:shown_params]
|
|
376
382
|
shown_labels << run[:shown_label]
|
|
383
|
+
diffs << run[:diff]
|
|
377
384
|
run[:output]
|
|
378
385
|
end.join("\n\n---\n\n")
|
|
379
386
|
emit_stream_event(on_stream_event, type: :tool_dispatch_completed, iteration: iteration_index + 1, call_count: calls.length)
|
|
@@ -387,6 +394,9 @@ module Samagotchi
|
|
|
387
394
|
# a built-in), for the web's reload; the prompt never reads it.
|
|
388
395
|
tool_response[:tool_params] = shown_params if shown_params.any?
|
|
389
396
|
tool_response[:tool_labels] = shown_labels if shown_labels.any?
|
|
397
|
+
# What each edit/write changed (nil for other calls), in call order,
|
|
398
|
+
# for the web's reload; the prompt never reads it.
|
|
399
|
+
tool_response[:tool_diffs] = diffs if diffs.any?
|
|
390
400
|
conversation << tool_response
|
|
391
401
|
pending_tool_calls = true
|
|
392
402
|
rescue Client::RequestCancelled => e
|
|
@@ -514,18 +524,13 @@ module Samagotchi
|
|
|
514
524
|
|
|
515
525
|
# Resolve the per-output character cap for the emitted tool call events.
|
|
516
526
|
#
|
|
517
|
-
# Precedence: an explicit override wins, then
|
|
518
|
-
#
|
|
519
|
-
#
|
|
527
|
+
# Precedence: an explicit override wins, then max_tool_output_chars
|
|
528
|
+
# (Config). A non-positive value falls back to
|
|
529
|
+
# DEFAULT_MAX_TOOL_OUTPUT_CHARS (there is intentionally no "unlimited" — live UIs get a
|
|
520
530
|
# bounded `output:` plus a truthful `output_truncated:` flag).
|
|
521
531
|
# Class-level so the chat loop resolves it the same way.
|
|
522
532
|
def self.resolve_output_char_cap(override)
|
|
523
|
-
|
|
524
|
-
v = Samagotchi::Config.get("max_tool_output_chars") rescue nil
|
|
525
|
-
v.to_i if v
|
|
526
|
-
end
|
|
527
|
-
value = override || cfg_val || ENV[TOOL_OUTPUT_CHARS_ENV]
|
|
528
|
-
parsed = value.to_i
|
|
533
|
+
parsed = (override || Samagotchi::Config.get("max_tool_output_chars")).to_i
|
|
529
534
|
parsed.positive? ? parsed : DEFAULT_MAX_TOOL_OUTPUT_CHARS
|
|
530
535
|
end
|
|
531
536
|
|
|
@@ -552,8 +557,8 @@ module Samagotchi
|
|
|
552
557
|
end
|
|
553
558
|
|
|
554
559
|
def completion_n_predict
|
|
555
|
-
|
|
556
|
-
|
|
560
|
+
value = Samagotchi::Config.get("default.n_predict").to_i
|
|
561
|
+
value if value.positive?
|
|
557
562
|
end
|
|
558
563
|
|
|
559
564
|
def completion_model_name(override = nil)
|
|
@@ -612,7 +617,8 @@ module Samagotchi
|
|
|
612
617
|
def emit_context_status_event(on_stream_event, prompt, iteration_index:, state:, window: nil, image_tokens: 0)
|
|
613
618
|
return nil unless context_status_enabled?
|
|
614
619
|
|
|
615
|
-
usage = estimate_context_usage(prompt, server_usage: state[:server_usage], window: window, image_tokens: image_tokens
|
|
620
|
+
usage = estimate_context_usage(prompt, server_usage: state[:server_usage], window: window, image_tokens: image_tokens,
|
|
621
|
+
counted: state[:counted])
|
|
616
622
|
bucket = context_status_bucket(usage[:estimated_pct])
|
|
617
623
|
# The status line's value, every iteration; the gate below decides
|
|
618
624
|
# only the event and the model's guidance line.
|
|
@@ -674,20 +680,18 @@ module Samagotchi
|
|
|
674
680
|
end
|
|
675
681
|
|
|
676
682
|
def context_status_enabled?
|
|
677
|
-
|
|
678
|
-
unless cfg.nil?
|
|
679
|
-
return !!cfg
|
|
680
|
-
end
|
|
681
|
-
value = ENV[CONTEXT_STATUS_ENABLED_ENV]
|
|
682
|
-
return true if value.nil?
|
|
683
|
-
|
|
684
|
-
!(value == "0" || value.casecmp?("false"))
|
|
683
|
+
Samagotchi::Config.get("context.status") != false
|
|
685
684
|
end
|
|
686
685
|
|
|
687
686
|
# `window` is this iteration's ContextWindow::Resolved (resolved here when
|
|
688
687
|
# not given). A window the stream payload reports itself still wins.
|
|
689
688
|
# +image_tokens+: the images' estimate (their base64 is not in +prompt+).
|
|
690
|
-
|
|
689
|
+
# +counted+: the prompt the server's prompt_tokens counted ({chars:,
|
|
690
|
+
# image_tokens:}); what +prompt+ has on top of it (the answer, tool
|
|
691
|
+
# results since) is added as an estimate, so the value doesn't read low
|
|
692
|
+
# during a long tool loop. A prompt shorter than that one (trimmed) or
|
|
693
|
+
# none known: the server's count alone.
|
|
694
|
+
def estimate_context_usage(prompt, server_usage: nil, window: nil, image_tokens: 0, counted: nil)
|
|
691
695
|
window ||= ContextWindow.resolve(client: @client, model: @current_model_name)
|
|
692
696
|
window_source = window.source
|
|
693
697
|
if server_usage && server_usage[:context_window_tokens]
|
|
@@ -696,7 +700,7 @@ module Samagotchi
|
|
|
696
700
|
|
|
697
701
|
if server_usage && server_usage[:prompt_tokens]
|
|
698
702
|
window_tokens = server_usage[:context_window_tokens] || window.tokens
|
|
699
|
-
estimated_used_tokens = server_usage[:prompt_tokens]
|
|
703
|
+
estimated_used_tokens = server_usage[:prompt_tokens] + appended_tokens(prompt, image_tokens, counted)
|
|
700
704
|
estimated_remaining_tokens = [window_tokens - estimated_used_tokens, 0].max
|
|
701
705
|
estimated_pct = (estimated_used_tokens.to_f / window_tokens) * 100.0
|
|
702
706
|
|
|
@@ -725,31 +729,29 @@ module Samagotchi
|
|
|
725
729
|
}
|
|
726
730
|
end
|
|
727
731
|
|
|
732
|
+
# The estimate for what +prompt+ added since the +counted+ one.
|
|
733
|
+
def appended_tokens(prompt, image_tokens, counted)
|
|
734
|
+
return 0 unless counted
|
|
735
|
+
|
|
736
|
+
chars = prompt.length - counted[:chars]
|
|
737
|
+
return 0 unless chars.positive?
|
|
738
|
+
|
|
739
|
+
(chars / context_chars_per_token).ceil + [image_tokens - counted[:image_tokens].to_i, 0].max
|
|
740
|
+
end
|
|
741
|
+
|
|
728
742
|
def context_chars_per_token
|
|
729
|
-
|
|
730
|
-
if cfg && cfg.to_f.positive?
|
|
731
|
-
v = cfg.to_f
|
|
732
|
-
return v.positive? ? v : DEFAULT_CONTEXT_CHARS_PER_TOKEN
|
|
733
|
-
end
|
|
734
|
-
value = ENV.fetch(CONTEXT_CHARS_PER_TOKEN_ENV, DEFAULT_CONTEXT_CHARS_PER_TOKEN.to_s).to_f
|
|
743
|
+
value = Samagotchi::Config.get("context.chars_per_token").to_f
|
|
735
744
|
value.positive? ? value : DEFAULT_CONTEXT_CHARS_PER_TOKEN
|
|
736
745
|
end
|
|
737
746
|
|
|
738
747
|
def context_status_thresholds
|
|
739
|
-
|
|
740
|
-
raw = cfg && !cfg.to_s.strip.empty? ? cfg.to_s : ENV.fetch(CONTEXT_THRESHOLDS_ENV, DEFAULT_CONTEXT_THRESHOLDS.join(","))
|
|
748
|
+
raw = Samagotchi::Config.get("context.status_thresholds").to_s
|
|
741
749
|
parsed = raw.split(",").map { |value| value.strip.to_i }.select { |value| value.between?(1, 99) }.uniq.sort
|
|
742
750
|
parsed.empty? ? DEFAULT_CONTEXT_THRESHOLDS : parsed
|
|
743
751
|
end
|
|
744
752
|
|
|
745
753
|
def context_status_cadence
|
|
746
|
-
|
|
747
|
-
if !cfg.nil?
|
|
748
|
-
v = cfg.to_i
|
|
749
|
-
return [v, 0].max
|
|
750
|
-
end
|
|
751
|
-
value = ENV.fetch(CONTEXT_CADENCE_ENV, DEFAULT_CONTEXT_CADENCE.to_s).to_i
|
|
752
|
-
[value, 0].max
|
|
754
|
+
[Samagotchi::Config.get("context.status_cadence").to_i, 0].max
|
|
753
755
|
end
|
|
754
756
|
|
|
755
757
|
# 0 for the bucket under the first threshold, then one per threshold.
|
|
@@ -828,14 +830,6 @@ module Samagotchi
|
|
|
828
830
|
end
|
|
829
831
|
end
|
|
830
832
|
|
|
831
|
-
# Parse tool calls from raw model output using the active profile's
|
|
832
|
-
# ToolCallParser strategy (Gemma 4 or Qwen 3.6).
|
|
833
|
-
# Thought content is intentionally left intact while a tool-call turn is in
|
|
834
|
-
# progress to preserve same-turn reasoning context between tool calls.
|
|
835
|
-
def parse_tool_calls(text)
|
|
836
|
-
parser.parse(text.to_s)
|
|
837
|
-
end
|
|
838
|
-
|
|
839
833
|
public
|
|
840
834
|
|
|
841
835
|
# Public wrapper so other loops (e.g. the chat loop) can strip
|
|
@@ -974,76 +968,18 @@ module Samagotchi
|
|
|
974
968
|
}
|
|
975
969
|
end
|
|
976
970
|
|
|
971
|
+
# ask_user_question: validate the call, then hand the payload to the
|
|
972
|
+
# Engine's question flow (question_handler, which blocks until the user
|
|
973
|
+
# answers). Without one (headless), the payload as JSON so the model sees
|
|
974
|
+
# the options and can ask in plain text.
|
|
977
975
|
def handle_ask_user_question(call)
|
|
978
|
-
|
|
979
|
-
|
|
980
|
-
|
|
981
|
-
options = Samagotchi::Tools::AskUserQuestion.normalize_options_lenient(raw_opts)
|
|
982
|
-
# Fallback for case where raw was String like '["a","b"]' but lenient returned [] due to edge parse, try raw string of params
|
|
983
|
-
if options.empty? && raw_opts.is_a?(String)
|
|
984
|
-
options = Samagotchi::Tools::AskUserQuestion.normalize_options_lenient(raw_opts.to_s)
|
|
985
|
-
end
|
|
986
|
-
header = call[:header].to_s.strip
|
|
987
|
-
header = nil if header.empty?
|
|
988
|
-
multi = call[:multi_select]
|
|
989
|
-
free = call[:allow_freeform]
|
|
990
|
-
# Normalize booleans from string forms (Gemma passes "true"/"false" as strings)
|
|
991
|
-
multi = normalize_ask_bool(multi)
|
|
992
|
-
free = normalize_ask_bool(free)
|
|
993
|
-
|
|
994
|
-
if question.empty?
|
|
995
|
-
return "Error: ask_user_question requires 'question'"
|
|
996
|
-
end
|
|
997
|
-
# Dumb-model tolerant: salvage single-option parse glitches, but still require at least 1
|
|
998
|
-
if options.size < 1
|
|
999
|
-
alt = Samagotchi::Tools::AskUserQuestion.normalize_options_lenient(call[:content].to_s) if call[:content]
|
|
1000
|
-
options = alt unless alt.empty?
|
|
1001
|
-
end
|
|
1002
|
-
if options.empty?
|
|
1003
|
-
return "Error: ask_user_question requires 2-8 options (got 0). Provide e.g. options=[\"Cats\",\"Dogs\"]"
|
|
1004
|
-
end
|
|
1005
|
-
if options.size == 1
|
|
1006
|
-
# Allow single-option salvage for dumb models (will still render, user can answer or provide freeform)
|
|
1007
|
-
elsif options.size < 2 || options.size > 8
|
|
1008
|
-
return "Error: ask_user_question requires 2-8 options (got #{options.size}). Provide e.g. options=[\"Cats\",\"Dogs\"]"
|
|
1009
|
-
end
|
|
1010
|
-
|
|
1011
|
-
# If an Engine-level blocking handler is registered (TUI/Web), delegate
|
|
1012
|
-
# there (Engine sets question_handler). Otherwise fall back to a
|
|
1013
|
-
# non-blocking JSON preview so the model can still see a structured response.
|
|
1014
|
-
handler = @question_handler
|
|
1015
|
-
|
|
1016
|
-
payload = {
|
|
1017
|
-
question: question,
|
|
1018
|
-
options: options,
|
|
1019
|
-
header: header,
|
|
1020
|
-
multi_select: !!multi,
|
|
1021
|
-
allow_freeform: !!free
|
|
1022
|
-
}.compact
|
|
1023
|
-
|
|
1024
|
-
if handler
|
|
1025
|
-
begin
|
|
1026
|
-
result = handler.call(payload)
|
|
1027
|
-
return result.to_s
|
|
1028
|
-
rescue => e
|
|
1029
|
-
return "Error: ask_user_question handler failed: #{e.message}"
|
|
1030
|
-
end
|
|
1031
|
-
end
|
|
976
|
+
payload = Samagotchi::Tools::AskUserQuestion.validate(call)
|
|
977
|
+
return payload if payload.is_a?(String)
|
|
978
|
+
return JSON.pretty_generate(payload) unless @question_handler
|
|
1032
979
|
|
|
1033
|
-
|
|
1034
|
-
|
|
1035
|
-
|
|
1036
|
-
end
|
|
1037
|
-
|
|
1038
|
-
def normalize_ask_bool(v)
|
|
1039
|
-
return nil if v.nil?
|
|
1040
|
-
return v if v == true || v == false
|
|
1041
|
-
|
|
1042
|
-
s = v.to_s.strip.downcase
|
|
1043
|
-
return true if %w[1 true yes on].include?(s)
|
|
1044
|
-
return false if %w[0 false no off].include?(s)
|
|
1045
|
-
|
|
1046
|
-
nil
|
|
980
|
+
@question_handler.call(payload).to_s
|
|
981
|
+
rescue => e
|
|
982
|
+
"Error: ask_user_question handler failed: #{e.message}"
|
|
1047
983
|
end
|
|
1048
984
|
end
|
|
1049
985
|
end
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "json"
|
|
4
|
+
require "net/http"
|
|
5
|
+
require "socket"
|
|
6
|
+
require_relative "session"
|
|
7
|
+
require_relative "version"
|
|
8
|
+
|
|
9
|
+
module Samagotchi
|
|
10
|
+
# Which chi versions the running processes run, for `chi update`: session
|
|
11
|
+
# workers (their bridge.json sidecar names the version since chi update exists; an
|
|
12
|
+
# older one names none) and a `chi web` on its port (/api/info). Read-only:
|
|
13
|
+
# a dead worker's sidecar is left for the next client to clean up.
|
|
14
|
+
module LiveVersions
|
|
15
|
+
PROBE_TIMEOUT = 0.2
|
|
16
|
+
WEB_TIMEOUT = 0.5
|
|
17
|
+
|
|
18
|
+
# version is nil for a sidecar written before sidecars carried one.
|
|
19
|
+
Worker = Struct.new(:session_id, :version, keyword_init: true)
|
|
20
|
+
|
|
21
|
+
module_function
|
|
22
|
+
|
|
23
|
+
# @return [Array<Worker>] the workers whose Bridge answers, by session id
|
|
24
|
+
def workers(state_dir: Session.default_state_dir)
|
|
25
|
+
Dir[File.join(state_dir, "*", "bridge.json")].sort.filter_map do |sidecar|
|
|
26
|
+
data = JSON.parse(File.read(sidecar))
|
|
27
|
+
next unless data.is_a?(Hash) && listening?(data["port"].to_i)
|
|
28
|
+
|
|
29
|
+
Worker.new(session_id: File.basename(File.dirname(sidecar)), version: data["version"])
|
|
30
|
+
rescue JSON::ParserError, SystemCallError
|
|
31
|
+
nil
|
|
32
|
+
end
|
|
33
|
+
end
|
|
34
|
+
|
|
35
|
+
# The live workers on another version than +version+ (unknown counts).
|
|
36
|
+
def stale_workers(version = VERSION, state_dir: Session.default_state_dir)
|
|
37
|
+
workers(state_dir: state_dir).reject { |w| w.version == version }
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
# The version a chi web on host:port runs, or nil when nothing (or not
|
|
41
|
+
# chi web) answers.
|
|
42
|
+
def web_version(host, port, timeout: WEB_TIMEOUT)
|
|
43
|
+
web_info(host, port, timeout: timeout)&.fetch("version", nil).to_s.then { |v| v.empty? ? nil : v }
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
# The /api/info of a chi web on host:port, or nil when nothing (or not
|
|
47
|
+
# chi web) answers.
|
|
48
|
+
def web_info(host, port, timeout: WEB_TIMEOUT)
|
|
49
|
+
response = Net::HTTP.start(host, port, open_timeout: timeout, read_timeout: timeout) { |http| http.get("/api/info") }
|
|
50
|
+
info = response.code.to_i == 200 ? JSON.parse(response.body.to_s) : nil
|
|
51
|
+
info.is_a?(Hash) && info["app"] == "chi-web" ? info : nil
|
|
52
|
+
rescue StandardError
|
|
53
|
+
nil
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
def listening?(port, host: "127.0.0.1")
|
|
57
|
+
return false unless port.positive?
|
|
58
|
+
|
|
59
|
+
Socket.tcp(host, port, connect_timeout: PROBE_TIMEOUT).close
|
|
60
|
+
true
|
|
61
|
+
rescue StandardError
|
|
62
|
+
false
|
|
63
|
+
end
|
|
64
|
+
end
|
|
65
|
+
end
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative "errors"
|
|
4
|
+
|
|
5
|
+
module Samagotchi
|
|
6
|
+
module LLM
|
|
7
|
+
# A host's API key: the environment variable its api_key_env: names,
|
|
8
|
+
# sent as `Authorization: Bearer <key>` on every request LLM::HTTP makes
|
|
9
|
+
# for the host (llama.cpp started with --api-key, or a provider). The
|
|
10
|
+
# key itself never goes into a message.
|
|
11
|
+
ApiKey = Data.define(:env_name, :host, :env) do
|
|
12
|
+
# nil for a host without api_key_env: its requests carry no header.
|
|
13
|
+
def self.for(env_name, host:, env: ENV)
|
|
14
|
+
name = env_name.to_s.strip
|
|
15
|
+
name.empty? ? nil : new(env_name: name, host: host.to_s, env: env)
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
# What to try after a 401/403 from a host that has no api_key_env.
|
|
19
|
+
def self.missing_hint(host)
|
|
20
|
+
"the server may want an API key: put it in an environment variable and name it with api_key_env: on host #{host}"
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
# Sets the header, or raises AuthError when the variable is not set.
|
|
24
|
+
def authorize(request)
|
|
25
|
+
key = env[env_name].to_s
|
|
26
|
+
raise AuthError.new("#{host}: set #{env_name} (the API key for host #{host})", host: host) if key.strip.empty?
|
|
27
|
+
|
|
28
|
+
request["Authorization"] = "Bearer #{key}"
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
# What to try after a 401/403 with the key sent.
|
|
32
|
+
def hint = "check #{env_name} (the API key for host #{host})"
|
|
33
|
+
|
|
34
|
+
# Without the environment: it holds this key and every other secret.
|
|
35
|
+
def inspect = "#<#{self.class.name} #{env_name} host=#{host}>"
|
|
36
|
+
alias_method :to_s, :inspect
|
|
37
|
+
|
|
38
|
+
def pretty_print(printer) = printer.text(inspect)
|
|
39
|
+
end
|
|
40
|
+
end
|
|
41
|
+
end
|
|
@@ -17,6 +17,7 @@ require_relative "../vision_context"
|
|
|
17
17
|
require_relative "../log"
|
|
18
18
|
require_relative "../empty_answer_retry"
|
|
19
19
|
require_relative "../turn_note"
|
|
20
|
+
require_relative "../thinking"
|
|
20
21
|
|
|
21
22
|
module Samagotchi
|
|
22
23
|
module LLM
|
|
@@ -139,6 +140,34 @@ module Samagotchi
|
|
|
139
140
|
@kernel.respond_to?(:sampling) ? @kernel.sampling || {} : {}
|
|
140
141
|
end
|
|
141
142
|
|
|
143
|
+
# The turn's thinking level (the Engine sets it on the kernel).
|
|
144
|
+
def thinking
|
|
145
|
+
(@kernel.respond_to?(:thinking) && @kernel.thinking) || Thinking::DEFAULT
|
|
146
|
+
end
|
|
147
|
+
|
|
148
|
+
# The request fields the thinking level adds (Thinking.chat_fields);
|
|
149
|
+
# none for a model whose host refused them (#thinking_refused!).
|
|
150
|
+
def thinking_fields(model = nil)
|
|
151
|
+
return {} if model && (@thinking_refused ||= Set.new).include?(model)
|
|
152
|
+
|
|
153
|
+
Thinking.chat_fields(thinking)
|
|
154
|
+
end
|
|
155
|
+
|
|
156
|
+
# The host refused +model+'s thinking fields: leave them out from now on.
|
|
157
|
+
def thinking_refused!(model)
|
|
158
|
+
(@thinking_refused ||= Set.new) << model
|
|
159
|
+
end
|
|
160
|
+
|
|
161
|
+
# One generation's options: the thinking fields under the sampling
|
|
162
|
+
# (a sampling key wins, chat_template_kwargs merges per sub-key), the
|
|
163
|
+
# empty-answer retry's temperature on top, then every null dropped at
|
|
164
|
+
# any depth (a sampling null means "don't send it").
|
|
165
|
+
def request_options(retry_generation: false, model: nil)
|
|
166
|
+
options = deep_merge(thinking_fields(model), sampling)
|
|
167
|
+
options = EmptyAnswerRetry.sampling(options) if retry_generation
|
|
168
|
+
deep_compact(options)
|
|
169
|
+
end
|
|
170
|
+
|
|
142
171
|
def strip_model_thought(text)
|
|
143
172
|
@kernel.respond_to?(:strip_model_thought) ? @kernel.strip_model_thought(text) : text
|
|
144
173
|
end
|
|
@@ -180,7 +209,7 @@ module Samagotchi
|
|
|
180
209
|
# reasoning, never sent back), a result's tool_call_id, the image
|
|
181
210
|
# refs of a user message or a tool result, and a plugin tool result's
|
|
182
211
|
# tool_params and tool_labels (the live row's params line and label,
|
|
183
|
-
# never sent back).
|
|
212
|
+
# never sent back), and an edit/write result's tool_diffs (never sent).
|
|
184
213
|
def plain(conversation)
|
|
185
214
|
conversation.map do |entry|
|
|
186
215
|
content = entry[:content].is_a?(Array) ? entry[:content] : entry[:content].to_s
|
|
@@ -191,6 +220,7 @@ module Samagotchi
|
|
|
191
220
|
message[:thinking] = entry[:thinking] if entry[:thinking].is_a?(String) && !entry[:thinking].empty?
|
|
192
221
|
message[:tool_params] = entry[:tool_params] if entry[:tool_params]
|
|
193
222
|
message[:tool_labels] = entry[:tool_labels] if entry[:tool_labels]
|
|
223
|
+
message[:tool_diffs] = entry[:tool_diffs] if entry[:tool_diffs]
|
|
194
224
|
message[AnswerDisplay::KEY] = entry[AnswerDisplay::KEY] if entry[AnswerDisplay::KEY]
|
|
195
225
|
ContextNote::KEYS.each { |key| message[key] = entry[key] if entry.key?(key) }
|
|
196
226
|
message
|
|
@@ -199,6 +229,18 @@ module Samagotchi
|
|
|
199
229
|
|
|
200
230
|
private
|
|
201
231
|
|
|
232
|
+
def deep_merge(base, over)
|
|
233
|
+
base.merge(over) { |_key, a, b| a.is_a?(Hash) && b.is_a?(Hash) ? deep_merge(a, b) : b }
|
|
234
|
+
end
|
|
235
|
+
|
|
236
|
+
def deep_compact(hash)
|
|
237
|
+
hash.each_with_object({}) do |(key, value), out|
|
|
238
|
+
next if value.nil?
|
|
239
|
+
|
|
240
|
+
out[key] = value.is_a?(Hash) ? deep_compact(value) : value
|
|
241
|
+
end
|
|
242
|
+
end
|
|
243
|
+
|
|
202
244
|
# Ids of the calls whose assistant turn is followed by a tool message
|
|
203
245
|
# for every one of them (before the next non-tool message).
|
|
204
246
|
def paired_call_ids(conversation)
|
|
@@ -364,23 +406,22 @@ module Samagotchi
|
|
|
364
406
|
# text] when it was cancelled.
|
|
365
407
|
def generate(iteration)
|
|
366
408
|
window = @window = @loop.context_window(@model_name)
|
|
367
|
-
|
|
368
|
-
options = EmptyAnswerRetry.sampling(options) if @retry_generation
|
|
409
|
+
retry_generation = @retry_generation
|
|
369
410
|
@retry_generation = false
|
|
370
411
|
emit(type: :generation_started, iteration: iteration, context_window_tokens: window&.tokens,
|
|
371
412
|
context_window_source: window&.source)
|
|
372
413
|
@loop.fire_hook(:before_generation, { type: :before_generation, iteration: iteration })
|
|
373
414
|
streamed = +""
|
|
374
|
-
response =
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
415
|
+
response = begin
|
|
416
|
+
request(iteration, retry_generation, streamed)
|
|
417
|
+
rescue BadRequest => e
|
|
418
|
+
raise unless thinking_refused?(e)
|
|
419
|
+
|
|
420
|
+
# Once per model: asked again without the thinking fields.
|
|
421
|
+
@loop.thinking_refused!(@model_name)
|
|
422
|
+
emit(type: :thinking_refused, iteration: iteration, model: @model_name, level: @loop.thinking, detail: e.detail)
|
|
423
|
+
request(iteration, retry_generation, streamed)
|
|
424
|
+
end
|
|
384
425
|
record_context_status(response.usage, window)
|
|
385
426
|
emit(type: :generation_completed, iteration: iteration, content_length: response.text.length,
|
|
386
427
|
thinking_chars: response.reasoning.to_s.length, served_model: response.model,
|
|
@@ -393,6 +434,27 @@ module Samagotchi
|
|
|
393
434
|
[e.reason, streamed]
|
|
394
435
|
end
|
|
395
436
|
|
|
437
|
+
def request(iteration, retry_generation, streamed)
|
|
438
|
+
@loop.adapter.chat(
|
|
439
|
+
messages: @loop.wire_messages(@conversation), tools: @loop.tool_definitions, model: @model_name,
|
|
440
|
+
cancel_controller: @cancel_controller, session_id: @loop.session_id,
|
|
441
|
+
options: @loop.request_options(retry_generation: retry_generation, model: @model_name),
|
|
442
|
+
on_delta: lambda { |content:, reasoning:, payload:|
|
|
443
|
+
streamed << content
|
|
444
|
+
emit(type: :generation_chunk, iteration: iteration, content: reasoning + content, text: content,
|
|
445
|
+
thinking: reasoning, payload: payload)
|
|
446
|
+
},
|
|
447
|
+
on_retry: ->(**retry_event) { emit({ type: :generation_retrying, iteration: iteration }.merge(retry_event)) }
|
|
448
|
+
)
|
|
449
|
+
end
|
|
450
|
+
|
|
451
|
+
# A 400 about reasoning, for a request that carried thinking fields
|
|
452
|
+
# (not a missing-tools, image or context error).
|
|
453
|
+
def thinking_refused?(error)
|
|
454
|
+
error.reasoning_refused? && !error.is_a?(VisionUnsupported) && !error.tools_unsupported? &&
|
|
455
|
+
!error.context_overflow? && !@loop.thinking_fields(@model_name).empty?
|
|
456
|
+
end
|
|
457
|
+
|
|
396
458
|
# The status line's value from the server's counts for this request
|
|
397
459
|
# (prompt + answer); without them the last value stays.
|
|
398
460
|
def record_context_status(usage, window)
|
|
@@ -429,6 +491,8 @@ module Samagotchi
|
|
|
429
491
|
# A plugin tool's params line, for the web's reload; never sent.
|
|
430
492
|
entry[:tool_params] = run[:shown_params] if run[:shown_params]
|
|
431
493
|
entry[:tool_labels] = run[:shown_label] if run[:shown_label]
|
|
494
|
+
# What an edit/write changed, for the web's reload; never sent.
|
|
495
|
+
entry[:tool_diffs] = run[:diff] if run[:diff]
|
|
432
496
|
@conversation << entry
|
|
433
497
|
end
|
|
434
498
|
emit(type: :tool_dispatch_completed, iteration: iteration, call_count: tool_calls.length)
|