samagotchi 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/CHANGELOG.md +43 -0
- data/LICENSE +21 -0
- data/README.md +126 -0
- data/bin/chi +1140 -0
- data/docs/architecture.md +299 -0
- data/docs/cli.md +490 -0
- data/docs/configuration.md +494 -0
- data/docs/desktop.md +97 -0
- data/docs/guardrails.md +218 -0
- data/docs/hooks.md +309 -0
- data/docs/internals/background-tasks.md +26 -0
- data/docs/internals/context-telemetry.md +36 -0
- data/docs/internals/gemma4-contract.md +23 -0
- data/docs/internals/tool-guardrails.md +45 -0
- data/docs/memory.md +85 -0
- data/docs/plugins.md +819 -0
- data/docs/releasing.md +135 -0
- data/docs/sessions.md +155 -0
- data/lib/samagotchi/bridge/bounded_queue.rb +70 -0
- data/lib/samagotchi/bridge/card_store.rb +126 -0
- data/lib/samagotchi/bridge/event_id.rb +25 -0
- data/lib/samagotchi/bridge/ring_buffer.rb +63 -0
- data/lib/samagotchi/bridge/sse_writer.rb +248 -0
- data/lib/samagotchi/bridge/turn_accumulator.rb +189 -0
- data/lib/samagotchi/bridge.rb +993 -0
- data/lib/samagotchi/bridge_client/event_stream.rb +158 -0
- data/lib/samagotchi/bridge_client/sse_parser.rb +51 -0
- data/lib/samagotchi/bridge_client.rb +330 -0
- data/lib/samagotchi/bundle_needs.rb +97 -0
- data/lib/samagotchi/bundles/btw/manifest.yml +10 -0
- data/lib/samagotchi/bundles/btw/plugin.rb +100 -0
- data/lib/samagotchi/bundles/guardrails/guardrails/rules.yml +82 -0
- data/lib/samagotchi/bundles/guardrails/guardrails.md +14 -0
- data/lib/samagotchi/bundles/guardrails/manifest.yml +8 -0
- data/lib/samagotchi/bundles/known-names/hooks/known_names.rb +210 -0
- data/lib/samagotchi/bundles/known-names/known_names.md +3 -0
- data/lib/samagotchi/bundles/known-names/manifest.yml +14 -0
- data/lib/samagotchi/bundles/loop-guard/manifest.yml +10 -0
- data/lib/samagotchi/bundles/loop-guard/plugin.rb +158 -0
- data/lib/samagotchi/bundles/mcp/manifest.yml +11 -0
- data/lib/samagotchi/bundles/mcp/plugin.rb +631 -0
- data/lib/samagotchi/bundles/system/config_modification_protocol.md +149 -0
- data/lib/samagotchi/bundles/system/delegated.md +10 -0
- data/lib/samagotchi/bundles/system/identity.md +7 -0
- data/lib/samagotchi/bundles/system/manifest.yml +11 -0
- data/lib/samagotchi/bundles/system/memory_guide.md +107 -0
- data/lib/samagotchi/bundles/system/self_map.md +55 -0
- data/lib/samagotchi/cancellation_controller.rb +78 -0
- data/lib/samagotchi/client.rb +429 -0
- data/lib/samagotchi/commands/registry.rb +112 -0
- data/lib/samagotchi/config.rb +910 -0
- data/lib/samagotchi/context_note.rb +77 -0
- data/lib/samagotchi/context_quote.rb +21 -0
- data/lib/samagotchi/context_usage.rb +66 -0
- data/lib/samagotchi/context_window.rb +76 -0
- data/lib/samagotchi/debug_log.rb +110 -0
- data/lib/samagotchi/desktop/macos/App.swift +102 -0
- data/lib/samagotchi/desktop/macos/ChiRunner.swift +201 -0
- data/lib/samagotchi/desktop/macos/Hotkey.swift +42 -0
- data/lib/samagotchi/desktop/macos/Info.plist.erb +42 -0
- data/lib/samagotchi/desktop/macos/Panel.swift +383 -0
- data/lib/samagotchi/desktop/macos.rb +255 -0
- data/lib/samagotchi/desktop.rb +21 -0
- data/lib/samagotchi/desktop_command.rb +143 -0
- data/lib/samagotchi/engine.rb +2807 -0
- data/lib/samagotchi/guardrails/approval.rb +125 -0
- data/lib/samagotchi/guardrails/approvals.rb +177 -0
- data/lib/samagotchi/guardrails/context.rb +71 -0
- data/lib/samagotchi/guardrails/gate.rb +125 -0
- data/lib/samagotchi/guardrails/load_failures.rb +46 -0
- data/lib/samagotchi/guardrails/protected_paths.rb +77 -0
- data/lib/samagotchi/guardrails/rules.rb +199 -0
- data/lib/samagotchi/guardrails/targets.rb +119 -0
- data/lib/samagotchi/guardrails/verdict.rb +134 -0
- data/lib/samagotchi/guardrails.rb +18 -0
- data/lib/samagotchi/hooks/bundle_loader.rb +158 -0
- data/lib/samagotchi/hooks/loader.rb +162 -0
- data/lib/samagotchi/hooks/registry.rb +261 -0
- data/lib/samagotchi/hooks.rb +30 -0
- data/lib/samagotchi/host_registry.rb +315 -0
- data/lib/samagotchi/idle_client.rb +147 -0
- data/lib/samagotchi/idle_recap.rb +549 -0
- data/lib/samagotchi/idle_reminders.rb +101 -0
- data/lib/samagotchi/idle_scheduler.rb +76 -0
- data/lib/samagotchi/image_store.rb +393 -0
- data/lib/samagotchi/installed_gem.rb +38 -0
- data/lib/samagotchi/kernel_loop.rb +1017 -0
- data/lib/samagotchi/launch_mode.rb +34 -0
- data/lib/samagotchi/llm/backend.rb +28 -0
- data/lib/samagotchi/llm/chat_loop.rb +450 -0
- data/lib/samagotchi/llm/errors.rb +329 -0
- data/lib/samagotchi/llm/http.rb +412 -0
- data/lib/samagotchi/llm/model_result.rb +72 -0
- data/lib/samagotchi/llm/native_backend.rb +50 -0
- data/lib/samagotchi/llm/native_tool_normalizer.rb +277 -0
- data/lib/samagotchi/llm/openai_chat.rb +403 -0
- data/lib/samagotchi/llm/usage.rb +79 -0
- data/lib/samagotchi/log.rb +200 -0
- data/lib/samagotchi/log_line.rb +127 -0
- data/lib/samagotchi/log_path.rb +31 -0
- data/lib/samagotchi/log_subscriber.rb +163 -0
- data/lib/samagotchi/memory_bundle/builder.rb +364 -0
- data/lib/samagotchi/memory_bundle/index_updater.rb +123 -0
- data/lib/samagotchi/memory_bundle/installer.rb +528 -0
- data/lib/samagotchi/memory_bundle/listing.rb +72 -0
- data/lib/samagotchi/memory_bundle/manifest.rb +225 -0
- data/lib/samagotchi/memory_bundle/merger.rb +52 -0
- data/lib/samagotchi/memory_bundle/placeholder.rb +37 -0
- data/lib/samagotchi/memory_bundle/provenance.rb +257 -0
- data/lib/samagotchi/memory_bundle/source.rb +153 -0
- data/lib/samagotchi/memory_bundle/status.rb +107 -0
- data/lib/samagotchi/memory_bundle/system_bundle.rb +161 -0
- data/lib/samagotchi/memory_bundle/uninstaller.rb +128 -0
- data/lib/samagotchi/memory_bundle.rb +17 -0
- data/lib/samagotchi/memory_paths.rb +101 -0
- data/lib/samagotchi/model_overlay.rb +53 -0
- data/lib/samagotchi/model_profile.rb +309 -0
- data/lib/samagotchi/muted_memories.rb +66 -0
- data/lib/samagotchi/note_command.rb +163 -0
- data/lib/samagotchi/output_formatter.rb +100 -0
- data/lib/samagotchi/owner_lock.rb +110 -0
- data/lib/samagotchi/pending_input_queue.rb +48 -0
- data/lib/samagotchi/plugin/api.rb +362 -0
- data/lib/samagotchi/plugin/context.rb +193 -0
- data/lib/samagotchi/plugin/loader.rb +126 -0
- data/lib/samagotchi/plugin/service.rb +117 -0
- data/lib/samagotchi/plugin/sessions.rb +150 -0
- data/lib/samagotchi/plugin/side_question.rb +60 -0
- data/lib/samagotchi/plugin/tool_result.rb +24 -0
- data/lib/samagotchi/project_scope.rb +25 -0
- data/lib/samagotchi/prompt.rb +119 -0
- data/lib/samagotchi/prompt_literal_guard.rb +70 -0
- data/lib/samagotchi/recap_store.rb +92 -0
- data/lib/samagotchi/reminder_store.rb +165 -0
- data/lib/samagotchi/self_report.rb +195 -0
- data/lib/samagotchi/send_command.rb +170 -0
- data/lib/samagotchi/served_model.rb +32 -0
- data/lib/samagotchi/session.rb +508 -0
- data/lib/samagotchi/session_commands.rb +527 -0
- data/lib/samagotchi/session_delete_command.rb +105 -0
- data/lib/samagotchi/session_manager.rb +1049 -0
- data/lib/samagotchi/session_metrics.rb +466 -0
- data/lib/samagotchi/session_observer.rb +117 -0
- data/lib/samagotchi/terminal_ui/attach_launcher.rb +118 -0
- data/lib/samagotchi/terminal_ui/attached_loop.rb +1037 -0
- data/lib/samagotchi/terminal_ui/attached_view.rb +264 -0
- data/lib/samagotchi/terminal_ui/event_renderer.rb +192 -0
- data/lib/samagotchi/terminal_ui/formatting.rb +291 -0
- data/lib/samagotchi/terminal_ui/image_input.rb +36 -0
- data/lib/samagotchi/terminal_ui/input_support.rb +324 -0
- data/lib/samagotchi/terminal_ui/legacy_surface.rb +111 -0
- data/lib/samagotchi/terminal_ui/line_reader.rb +113 -0
- data/lib/samagotchi/terminal_ui/live_region.rb +36 -0
- data/lib/samagotchi/terminal_ui/plain_surface.rb +51 -0
- data/lib/samagotchi/terminal_ui/question_prompt.rb +153 -0
- data/lib/samagotchi/terminal_ui/question_slot.rb +131 -0
- data/lib/samagotchi/terminal_ui/reline_seam.rb +216 -0
- data/lib/samagotchi/terminal_ui/repl_input.rb +138 -0
- data/lib/samagotchi/terminal_ui/screen.rb +316 -0
- data/lib/samagotchi/terminal_ui/surface.rb +47 -0
- data/lib/samagotchi/terminal_ui/thinking_line.rb +101 -0
- data/lib/samagotchi/terminal_ui.rb +1992 -0
- data/lib/samagotchi/thinking_ticker.rb +110 -0
- data/lib/samagotchi/thought_stream_splitter.rb +149 -0
- data/lib/samagotchi/token_usage.rb +88 -0
- data/lib/samagotchi/tool_activity.rb +216 -0
- data/lib/samagotchi/tool_call_parser.rb +637 -0
- data/lib/samagotchi/tool_declarations.rb +561 -0
- data/lib/samagotchi/tool_runner.rb +211 -0
- data/lib/samagotchi/tools/args.rb +259 -0
- data/lib/samagotchi/tools/ask_user_question.rb +152 -0
- data/lib/samagotchi/tools/builtins.rb +122 -0
- data/lib/samagotchi/tools/cancel_reminder.rb +21 -0
- data/lib/samagotchi/tools/delegate.rb +167 -0
- data/lib/samagotchi/tools/delegate_result.rb +53 -0
- data/lib/samagotchi/tools/delegate_wait.rb +153 -0
- data/lib/samagotchi/tools/edit.rb +155 -0
- data/lib/samagotchi/tools/execute.rb +214 -0
- data/lib/samagotchi/tools/list_reminders.rb +20 -0
- data/lib/samagotchi/tools/list_sessions.rb +74 -0
- data/lib/samagotchi/tools/memory.rb +256 -0
- data/lib/samagotchi/tools/output_guardrails.rb +93 -0
- data/lib/samagotchi/tools/peers.rb +18 -0
- data/lib/samagotchi/tools/read.rb +182 -0
- data/lib/samagotchi/tools/register_reminder.rb +53 -0
- data/lib/samagotchi/tools/registry.rb +60 -0
- data/lib/samagotchi/tools/send_note.rb +49 -0
- data/lib/samagotchi/tools/task_create.rb +29 -0
- data/lib/samagotchi/tools/task_get.rb +39 -0
- data/lib/samagotchi/tools/task_list.rb +43 -0
- data/lib/samagotchi/tools/task_runtime.rb +311 -0
- data/lib/samagotchi/tools/task_stop.rb +29 -0
- data/lib/samagotchi/tools/task_wait.rb +104 -0
- data/lib/samagotchi/tools/tool_path.rb +18 -0
- data/lib/samagotchi/tools/web_fetch.rb +163 -0
- data/lib/samagotchi/tools/write.rb +26 -0
- data/lib/samagotchi/turn_flow.rb +242 -0
- data/lib/samagotchi/turn_note.rb +76 -0
- data/lib/samagotchi/turn_tally.rb +101 -0
- data/lib/samagotchi/version.rb +7 -0
- data/lib/samagotchi/vision_context.rb +132 -0
- data/lib/samagotchi/vision_support.rb +109 -0
- data/lib/samagotchi/web/app.rb +1349 -0
- data/lib/samagotchi/web/markdown_renderer.rb +107 -0
- data/lib/samagotchi/web/message_parts.rb +169 -0
- data/lib/samagotchi/web/public/activity.js +100 -0
- data/lib/samagotchi/web/public/annotations.js +67 -0
- data/lib/samagotchi/web/public/app.js +2382 -0
- data/lib/samagotchi/web/public/card.js +74 -0
- data/lib/samagotchi/web/public/chat_view.js +360 -0
- data/lib/samagotchi/web/public/chunk_router.js +25 -0
- data/lib/samagotchi/web/public/command_complete.js +39 -0
- data/lib/samagotchi/web/public/composer_size.js +19 -0
- data/lib/samagotchi/web/public/copy.js +142 -0
- data/lib/samagotchi/web/public/ctx.js +35 -0
- data/lib/samagotchi/web/public/data.js +256 -0
- data/lib/samagotchi/web/public/format.js +232 -0
- data/lib/samagotchi/web/public/hold.js +78 -0
- data/lib/samagotchi/web/public/images.js +77 -0
- data/lib/samagotchi/web/public/index.html +568 -0
- data/lib/samagotchi/web/public/init_row.js +60 -0
- data/lib/samagotchi/web/public/model_pick.js +23 -0
- data/lib/samagotchi/web/public/question_card.js +100 -0
- data/lib/samagotchi/web/public/route.js +17 -0
- data/lib/samagotchi/web/public/scope.js +36 -0
- data/lib/samagotchi/web/public/scroll.js +24 -0
- data/lib/samagotchi/web/public/sentences.js +88 -0
- data/lib/samagotchi/web/public/sessions_list.js +60 -0
- data/lib/samagotchi/web/public/strip.js +25 -0
- data/lib/samagotchi/web/public/tally.js +37 -0
- data/lib/samagotchi/web/public/thinking_ticker.js +79 -0
- data/lib/samagotchi/web/public/timing.js +185 -0
- data/lib/samagotchi/web/public/turn_events.js +209 -0
- data/lib/samagotchi/web/public/turn_model.js +204 -0
- data/lib/samagotchi/web/public/turn_view.js +587 -0
- data/lib/samagotchi/web/server.rb +183 -0
- data/lib/samagotchi/web/session_hub.rb +329 -0
- data/lib/samagotchi/web/session_summary.rb +85 -0
- data/lib/samagotchi/worker.rb +635 -0
- data/lib/samagotchi/worker_idle_exit.rb +87 -0
- data/lib/samagotchi.rb +12 -0
- metadata +374 -0
|
@@ -0,0 +1,315 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative "config"
|
|
4
|
+
require_relative "client"
|
|
5
|
+
require_relative "llm/openai_chat"
|
|
6
|
+
|
|
7
|
+
module Samagotchi
|
|
8
|
+
# HostRegistry manages multiple model hosts (llama.cpp / mlx / oMLX) and
|
|
9
|
+
# provides lazy model discovery aggregation.
|
|
10
|
+
#
|
|
11
|
+
# - Hosts are defined in config.yml `hosts:` section or synthesized from
|
|
12
|
+
# SAMAGOTCHI_SERVER_HOST/PORT env (via Config).
|
|
13
|
+
# - Discovery is lazy: list_all_models is the explicit trigger (called by
|
|
14
|
+
# /models), not on startup. Each host's list is cached (60s, 10 minutes
|
|
15
|
+
# for a remote host) with skip-on-error; lists are LLM::ModelInfo.
|
|
16
|
+
# - Routing: client_for_model resolves a (possibly qualified) model string
|
|
17
|
+
# to the appropriate Client instance. A remote host is chosen only by
|
|
18
|
+
# exact model id, host:model or an alias, never by a substring.
|
|
19
|
+
class HostRegistry
|
|
20
|
+
CACHE_TTL_SECONDS = 60
|
|
21
|
+
REMOTE_CACHE_TTL_SECONDS = 600
|
|
22
|
+
LIST_TIMEOUT_SECONDS = 3
|
|
23
|
+
# Seconds a remote host's stream may take to show something (a queued
|
|
24
|
+
# free model on OpenRouter can send only keep-alives for minutes).
|
|
25
|
+
REMOTE_FIRST_TOKEN_TIMEOUT = 120
|
|
26
|
+
|
|
27
|
+
# url: the configured url, when the entry has one (host, port and scheme
|
|
28
|
+
# come from it); api_key_env: the variable holding the host's API key;
|
|
29
|
+
# profile: the configured prompt profile name, if any;
|
|
30
|
+
# first_token_timeout: the configured first-token limit (see #first_token_limit);
|
|
31
|
+
# vision: the configured true/false (VisionSupport), nil when unset.
|
|
32
|
+
HostEntry = Struct.new(:name, :host, :port, :transport, :client, :api, :scheme, :url, :api_key_env, :profile,
|
|
33
|
+
:first_token_timeout, :vision, keyword_init: true) do
|
|
34
|
+
# Talks the OpenAI chat API (the chat loop); nil and raw apis use the
|
|
35
|
+
# raw-prompt loop.
|
|
36
|
+
def chat? = api == :openai
|
|
37
|
+
|
|
38
|
+
# The server root, e.g. for llama.cpp's own endpoints and recap.
|
|
39
|
+
def root_url = "#{scheme || "http"}://#{host}:#{port}"
|
|
40
|
+
|
|
41
|
+
# The OpenAI-compatible API base the chat loop talks to: the url as
|
|
42
|
+
# configured, else the root's /v1.
|
|
43
|
+
def openai_base_url = url || "#{root_url}/v1"
|
|
44
|
+
|
|
45
|
+
# A provider on the network rather than a local server: it needs a key
|
|
46
|
+
# or speaks https. Its model list is cached longer and it is never
|
|
47
|
+
# picked by a substring of a model name.
|
|
48
|
+
def remote? = !api_key_env.to_s.empty? || scheme == "https"
|
|
49
|
+
|
|
50
|
+
def models_ttl = remote? ? REMOTE_CACHE_TTL_SECONDS : CACHE_TTL_SECONDS
|
|
51
|
+
|
|
52
|
+
# Seconds a streamed answer may take to show something, or nil: the
|
|
53
|
+
# host's first_token_timeout, else server.first_token_timeout, else
|
|
54
|
+
# 120 for a remote host (a local server's long prompt eval is normal,
|
|
55
|
+
# and read_timeout catches a dead one). 0 turns it off.
|
|
56
|
+
def first_token_limit
|
|
57
|
+
seconds = first_token_timeout
|
|
58
|
+
seconds = HostRegistry.configured_first_token_timeout if seconds.nil?
|
|
59
|
+
seconds = remote? ? REMOTE_FIRST_TOKEN_TIMEOUT : nil if seconds.nil?
|
|
60
|
+
seconds&.positive? ? seconds : nil
|
|
61
|
+
end
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
# Where a model's requests go: the host entry, the client to use and the
|
|
65
|
+
# model name to send (the host prefix stripped).
|
|
66
|
+
ModelTarget = Data.define(:model, :entry, :bare_model, :client) do
|
|
67
|
+
def root_url = entry.root_url
|
|
68
|
+
def openai_base_url = entry.openai_base_url
|
|
69
|
+
end
|
|
70
|
+
|
|
71
|
+
# A client that every target uses instead of its host's own (specs inject
|
|
72
|
+
# a stub this way; the Engine/TUI `client:` keyword sets it).
|
|
73
|
+
attr_accessor :client_override
|
|
74
|
+
|
|
75
|
+
# @param clock [#call, nil] monotonic seconds (specs)
|
|
76
|
+
def initialize(hosts_config: nil, env: ENV, client_override: nil, clock: nil)
|
|
77
|
+
@client_override = client_override
|
|
78
|
+
@clock = clock || -> { Process.clock_gettime(Process::CLOCK_MONOTONIC) }
|
|
79
|
+
@adapters = {}
|
|
80
|
+
@host_lists = {}
|
|
81
|
+
raw = hosts_config || ConfigFile.hosts_config(env: env)
|
|
82
|
+
@entries = {}
|
|
83
|
+
raw.each do |key, cfg|
|
|
84
|
+
# cfg: {name:, host:, port:, transport:, original_name:}
|
|
85
|
+
transport = cfg[:transport]
|
|
86
|
+
entry = HostEntry.new(name: key.to_s.downcase, host: cfg[:host], port: cfg[:port].to_i, transport: transport,
|
|
87
|
+
api: cfg[:api]&.to_sym, scheme: cfg[:scheme], url: cfg[:url], api_key_env: cfg[:api_key_env],
|
|
88
|
+
profile: cfg[:profile], first_token_timeout: cfg[:first_token_timeout],
|
|
89
|
+
vision: cfg[:vision])
|
|
90
|
+
entry.client = Client.new(host: cfg[:host], port: cfg[:port], transport: transport, scheme: cfg[:scheme],
|
|
91
|
+
first_token_timeout: entry.first_token_limit, name: entry.name)
|
|
92
|
+
@entries[entry.name] = entry
|
|
93
|
+
end
|
|
94
|
+
# Fallback single entry (should already be synthesized by hosts_config, but guard)
|
|
95
|
+
if @entries.empty?
|
|
96
|
+
host = Config.get("server.host")
|
|
97
|
+
host = "localhost" if host.empty?
|
|
98
|
+
port = Config.get("server.port").to_i
|
|
99
|
+
port = 8080 if port <= 0
|
|
100
|
+
@entries["default"] = HostEntry.new(name: "default", host: host, port: port, transport: nil, client: Client.new(host: host, port: port, name: "default"))
|
|
101
|
+
end
|
|
102
|
+
@mutex = Mutex.new
|
|
103
|
+
@cache = nil
|
|
104
|
+
@cache_at = nil
|
|
105
|
+
@model_index = nil # downcased model_id => host_name
|
|
106
|
+
end
|
|
107
|
+
|
|
108
|
+
def entries
|
|
109
|
+
@entries
|
|
110
|
+
end
|
|
111
|
+
|
|
112
|
+
# server.first_token_timeout, or nil when unset or unreadable.
|
|
113
|
+
def self.configured_first_token_timeout
|
|
114
|
+
Config.get("server.first_token_timeout")
|
|
115
|
+
rescue StandardError
|
|
116
|
+
nil
|
|
117
|
+
end
|
|
118
|
+
|
|
119
|
+
def entry_names
|
|
120
|
+
@entries.keys
|
|
121
|
+
end
|
|
122
|
+
|
|
123
|
+
def default_entry
|
|
124
|
+
@entries["default"] || @entries.values.first
|
|
125
|
+
end
|
|
126
|
+
|
|
127
|
+
def find_entry(name)
|
|
128
|
+
@entries[name.to_s.strip.downcase]
|
|
129
|
+
end
|
|
130
|
+
|
|
131
|
+
# Parse host-qualified model string using known host names.
|
|
132
|
+
# Returns [host_name_or_nil, bare_model]
|
|
133
|
+
def parse_qualified_model(raw)
|
|
134
|
+
ConfigFile.parse_host_qualified_model(raw, hosts: @entries)
|
|
135
|
+
end
|
|
136
|
+
|
|
137
|
+
# Resolve model string (already alias-resolved, may be qualified) to a HostEntry.
|
|
138
|
+
# If qualified explicitly, return that host. If unqualified, try cached model index,
|
|
139
|
+
# else fallback to default host.
|
|
140
|
+
def host_for_model(raw_model)
|
|
141
|
+
host_ref, bare = parse_qualified_model(raw_model)
|
|
142
|
+
# If qualified, try to resolve alias on the bare part (small-box:small -> small-box:gemma-small)
|
|
143
|
+
if host_ref && bare
|
|
144
|
+
begin
|
|
145
|
+
aliases = ConfigFile.model_aliases
|
|
146
|
+
resolved = aliases.fetch(bare.downcase, bare)
|
|
147
|
+
bare = resolved if resolved != bare
|
|
148
|
+
rescue StandardError
|
|
149
|
+
nil
|
|
150
|
+
end
|
|
151
|
+
entry = find_entry(host_ref)
|
|
152
|
+
return [entry, bare] if entry
|
|
153
|
+
# Unknown prefix — treat as bare model on default host
|
|
154
|
+
return [default_entry, raw_model.to_s.strip]
|
|
155
|
+
end
|
|
156
|
+
# Unqualified: also try alias resolution for discovery (small -> gemma-small or small -> small-box:gemma-small)
|
|
157
|
+
begin
|
|
158
|
+
aliases = ConfigFile.model_aliases
|
|
159
|
+
resolved = aliases.fetch(bare.to_s.strip.downcase, bare)
|
|
160
|
+
if resolved != bare
|
|
161
|
+
# If alias points to a qualified ref, re-parse it
|
|
162
|
+
q_host, q_bare = parse_qualified_model(resolved)
|
|
163
|
+
if q_host
|
|
164
|
+
entry = find_entry(q_host)
|
|
165
|
+
return [entry, q_bare] if entry
|
|
166
|
+
end
|
|
167
|
+
bare = resolved
|
|
168
|
+
end
|
|
169
|
+
rescue StandardError
|
|
170
|
+
nil
|
|
171
|
+
end
|
|
172
|
+
bare_down = bare.to_s.strip.downcase
|
|
173
|
+
# Try cached index (populated after list_all_models)
|
|
174
|
+
idx = @mutex.synchronize { @model_index }
|
|
175
|
+
if idx && idx.key?(bare_down)
|
|
176
|
+
host_name = idx[bare_down]
|
|
177
|
+
entry = find_entry(host_name)
|
|
178
|
+
return [entry, bare] if entry
|
|
179
|
+
end
|
|
180
|
+
# Fallback: try substring match in cached aggregated results if available
|
|
181
|
+
# (lightweight: scan cached model lists). Remote hosts match exactly only.
|
|
182
|
+
cached = @mutex.synchronize { @cache }
|
|
183
|
+
if cached
|
|
184
|
+
cached.each do |hname, data|
|
|
185
|
+
next unless data[:models]
|
|
186
|
+
data[:models].each do |m|
|
|
187
|
+
if m.id.downcase == bare_down
|
|
188
|
+
entry = find_entry(hname)
|
|
189
|
+
return [entry, bare] if entry
|
|
190
|
+
end
|
|
191
|
+
end
|
|
192
|
+
end
|
|
193
|
+
cached.each do |hname, data|
|
|
194
|
+
next unless data[:models]
|
|
195
|
+
next if find_entry(hname)&.remote?
|
|
196
|
+
|
|
197
|
+
data[:models].each do |m|
|
|
198
|
+
if m.id.downcase.include?(bare_down)
|
|
199
|
+
entry = find_entry(hname)
|
|
200
|
+
return [entry, bare] if entry
|
|
201
|
+
end
|
|
202
|
+
end
|
|
203
|
+
end
|
|
204
|
+
end
|
|
205
|
+
[default_entry, bare]
|
|
206
|
+
end
|
|
207
|
+
|
|
208
|
+
# The chat adapter for a host (one per host, so its cached model list
|
|
209
|
+
# serves the context window). Only chat hosts use it for turns.
|
|
210
|
+
# @return [LLM::OpenAIChat]
|
|
211
|
+
def adapter_for(entry)
|
|
212
|
+
@mutex.synchronize do
|
|
213
|
+
@adapters[entry.name] ||= LLM::OpenAIChat.for(entry, models_ttl: entry.models_ttl,
|
|
214
|
+
first_token_timeout: entry.first_token_limit)
|
|
215
|
+
end
|
|
216
|
+
end
|
|
217
|
+
|
|
218
|
+
# A host's models as ModelInfo: a chat host's from its adapter, a raw
|
|
219
|
+
# host's from its Client (ids from the server's own list shape).
|
|
220
|
+
def list_models_for(entry)
|
|
221
|
+
return adapter_for(entry).list_models if entry.chat? && !@client_override
|
|
222
|
+
|
|
223
|
+
Array(client_for(entry).list_models).map do |raw|
|
|
224
|
+
if raw.is_a?(Hash)
|
|
225
|
+
id = raw["id"] || raw[:id] || raw["model"] || raw["name"]
|
|
226
|
+
LLM::ModelInfo.new(id: id.to_s, context_window: nil, supports_tools: nil, raw: raw)
|
|
227
|
+
else
|
|
228
|
+
LLM::ModelInfo.new(id: raw.to_s, context_window: nil, supports_tools: nil, raw: {})
|
|
229
|
+
end
|
|
230
|
+
end
|
|
231
|
+
end
|
|
232
|
+
|
|
233
|
+
def client_for_model(raw_model)
|
|
234
|
+
host_entry, bare = host_for_model(raw_model)
|
|
235
|
+
[client_for(host_entry), bare, host_entry]
|
|
236
|
+
end
|
|
237
|
+
|
|
238
|
+
# The single host/model resolution: alias and host routing (host_for_model)
|
|
239
|
+
# plus the name sent to the server (bare_name).
|
|
240
|
+
# @param raw_model [String] a model name, alias or host:model ref
|
|
241
|
+
# @return [ModelTarget]
|
|
242
|
+
def resolve(raw_model)
|
|
243
|
+
entry, = host_for_model(raw_model)
|
|
244
|
+
ModelTarget.new(model: raw_model, entry: entry, bare_model: bare_name(raw_model), client: client_for(entry))
|
|
245
|
+
end
|
|
246
|
+
|
|
247
|
+
# The model name without a known host prefix ("box:gemma" → "gemma").
|
|
248
|
+
# Aliases are not applied here.
|
|
249
|
+
def bare_name(full_ref)
|
|
250
|
+
_, bare = parse_qualified_model(full_ref)
|
|
251
|
+
bare.to_s.strip.empty? ? full_ref.to_s.strip : bare
|
|
252
|
+
end
|
|
253
|
+
|
|
254
|
+
def client_for(entry)
|
|
255
|
+
@client_override || entry.client
|
|
256
|
+
end
|
|
257
|
+
|
|
258
|
+
# List models on all hosts in parallel. On error per-host, skip with error entry (no failover).
|
|
259
|
+
# Returns { host_name => { host:, port:, transport:, models: [LLM::ModelInfo], error: nil|String } }
|
|
260
|
+
# Also populates the model index. Unless forced, a host's list is reused
|
|
261
|
+
# for its TTL (60s; 10 minutes for a remote host).
|
|
262
|
+
def list_all_models(force: true)
|
|
263
|
+
results = {}
|
|
264
|
+
results_mutex = Mutex.new
|
|
265
|
+
threads = @entries.map do |name, entry|
|
|
266
|
+
fresh = !force && fresh_list(name, entry)
|
|
267
|
+
next results_mutex.synchronize { results[name] = fresh } if fresh
|
|
268
|
+
|
|
269
|
+
Thread.new do
|
|
270
|
+
begin
|
|
271
|
+
models = list_models_for(entry)
|
|
272
|
+
data = { host: entry.host, port: entry.port, transport: entry.transport, models: models, error: nil }
|
|
273
|
+
@mutex.synchronize { @host_lists[name] = { data: data, at: @clock.call } }
|
|
274
|
+
rescue StandardError => e
|
|
275
|
+
data = { host: entry.host, port: entry.port, transport: entry.transport, models: [], error: e.message }
|
|
276
|
+
end
|
|
277
|
+
results_mutex.synchronize { results[name] = data }
|
|
278
|
+
end
|
|
279
|
+
end
|
|
280
|
+
threads.each { |thread| thread.join if thread.is_a?(Thread) }
|
|
281
|
+
|
|
282
|
+
# Build model index: model_id downcased -> host_name (first host wins)
|
|
283
|
+
index = {}
|
|
284
|
+
results.each do |hname, data|
|
|
285
|
+
next if data[:error]
|
|
286
|
+
Array(data[:models]).each do |m|
|
|
287
|
+
mid = m.id.to_s
|
|
288
|
+
next if mid.strip.empty?
|
|
289
|
+
down = mid.downcase
|
|
290
|
+
index[down] = hname unless index.key?(down)
|
|
291
|
+
end
|
|
292
|
+
end
|
|
293
|
+
|
|
294
|
+
@mutex.synchronize do
|
|
295
|
+
@cache = results
|
|
296
|
+
@cache_at = @clock.call
|
|
297
|
+
@model_index = index
|
|
298
|
+
end
|
|
299
|
+
results
|
|
300
|
+
end
|
|
301
|
+
|
|
302
|
+
def cached_results
|
|
303
|
+
@mutex.synchronize { @cache }
|
|
304
|
+
end
|
|
305
|
+
|
|
306
|
+
private
|
|
307
|
+
|
|
308
|
+
def fresh_list(name, entry)
|
|
309
|
+
@mutex.synchronize do
|
|
310
|
+
cached = @host_lists[name]
|
|
311
|
+
cached[:data] if cached && (@clock.call - cached[:at]) < entry.models_ttl
|
|
312
|
+
end
|
|
313
|
+
end
|
|
314
|
+
end
|
|
315
|
+
end
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "json"
|
|
4
|
+
require_relative "llm/openai_chat"
|
|
5
|
+
|
|
6
|
+
module Samagotchi
|
|
7
|
+
# Standalone summarizer for the idle session-recap feature.
|
|
8
|
+
#
|
|
9
|
+
# Asks an OpenAI-compatible /chat/completions endpoint (the local llama.cpp,
|
|
10
|
+
# or a recap host's) through its own OpenAIChat: one plain request, no
|
|
11
|
+
# tools, no retries, its own timeout. It is NEVER wired into the Engine's
|
|
12
|
+
# kernel/backend, so it shares no engine concurrency. The Engine's idle
|
|
13
|
+
# detector only snapshots session.messages and calls #summarize; this object
|
|
14
|
+
# owns the HTTP boundary and is fully decoupled (configurable base_url + model).
|
|
15
|
+
#
|
|
16
|
+
# The request targets /chat/completions, NOT /completions (a raw
|
|
17
|
+
# completions endpoint loops on gemma think tokens).
|
|
18
|
+
#
|
|
19
|
+
# #summarize raises SummarizeError on any failure; the idle detector isolates
|
|
20
|
+
# that so a failed recap never breaks the active session.
|
|
21
|
+
class IdleClient
|
|
22
|
+
# Raised when summarization fails (server down, timeout, malformed body…).
|
|
23
|
+
class SummarizeError < StandardError; end
|
|
24
|
+
|
|
25
|
+
# What #summarize returns: the recap text and the model that answered
|
|
26
|
+
# (the server's name for it; nil when the reply names none). #to_s is
|
|
27
|
+
# the text.
|
|
28
|
+
Summary = Data.define(:text, :model) do
|
|
29
|
+
def to_s = text
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
DEFAULT_TIMEOUT_SECONDS = 30.0
|
|
33
|
+
# Up to 10 sentences (recap.sentences) need ~350 tokens with thinking off.
|
|
34
|
+
MAX_TOKENS = 512
|
|
35
|
+
|
|
36
|
+
# Request fields that turn thinking off. A reasoning model otherwise
|
|
37
|
+
# spends the budget thinking and the recap stops mid-sentence.
|
|
38
|
+
# - chat_template_kwargs.enable_thinking: the chat template's switch
|
|
39
|
+
# (llama.cpp with Qwen/Gemma templates); templates without it ignore it.
|
|
40
|
+
# - reasoning_effort "none": the OpenAI-style knob, for servers that
|
|
41
|
+
# ignore the template switch (Splash thought until max_tokens).
|
|
42
|
+
THINKING_OFF = {
|
|
43
|
+
chat_template_kwargs: { enable_thinking: false },
|
|
44
|
+
reasoning_effort: "none"
|
|
45
|
+
}.freeze
|
|
46
|
+
|
|
47
|
+
# @param base_url [String] the OpenAI API base, e.g. http://host:8081/v1
|
|
48
|
+
# @param api_key_env [String, nil] the variable holding the host's key
|
|
49
|
+
# @param timeout [Numeric] HTTP request timeout. Kept to the recap's own
|
|
50
|
+
# wait budget (not the chat's global request_timeout) so an abandoned
|
|
51
|
+
# summarize thread can't outlive the recap attempt by minutes.
|
|
52
|
+
def initialize(model:, base_url: nil, api_key_env: nil, timeout: DEFAULT_TIMEOUT_SECONDS, env: ENV)
|
|
53
|
+
@model = model
|
|
54
|
+
# A recap is best-effort: one short attempt, no retries. The idle job
|
|
55
|
+
# tries again after the next activity, never on its own.
|
|
56
|
+
@chat = LLM::OpenAIChat.new(base_url: base_url.to_s, host_name: "recap", api_key_env: api_key_env,
|
|
57
|
+
stream: false, retries: false, timeout: timeout, env: env, purpose: "recap")
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
THINK_RE = /<\|think\|.*?\|think\|>/m
|
|
61
|
+
LITERAL_THINK_RE = /\[\[SAMAGOTCHI_LITERAL_THINK_OPEN\]\].*?\[\[SAMAGOTCHI_LITERAL_THINK_CLOSE\]\]/m
|
|
62
|
+
|
|
63
|
+
# Strip thinking blocks (gemma <|think|>…, qwen prompt literals) and
|
|
64
|
+
# collapse the blank lines they leave. Shared with IdleRecap's
|
|
65
|
+
# transcript filter.
|
|
66
|
+
def self.strip_thinking(text)
|
|
67
|
+
text.to_s
|
|
68
|
+
.gsub(THINK_RE, "")
|
|
69
|
+
.gsub(LITERAL_THINK_RE, "")
|
|
70
|
+
.gsub(/\n\n+/, "\n")
|
|
71
|
+
.strip
|
|
72
|
+
end
|
|
73
|
+
|
|
74
|
+
# +text+ up to its last sentence end (. ! ? plus closing quotes or
|
|
75
|
+
# brackets, then whitespace or the end), or "" when none finished.
|
|
76
|
+
def self.full_sentences(text)
|
|
77
|
+
text.to_s[/\A.*[.!?]["')\]`*]*(?=\s|\z)/m].to_s
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
# Summarize an already-built recap prompt: a string (one user message) or
|
|
81
|
+
# a list of chat messages. Returns a Summary of the cleaned prose, or nil
|
|
82
|
+
# when there is nothing to summarize. Any failure raises SummarizeError
|
|
83
|
+
# (the caller isolates it).
|
|
84
|
+
# @return [Summary, nil]
|
|
85
|
+
def summarize(prompt)
|
|
86
|
+
messages = prompt.is_a?(Array) ? prompt : [{ role: "user", content: prompt.to_s.strip }]
|
|
87
|
+
return nil if messages.all? { |m| m[:content].to_s.strip.empty? }
|
|
88
|
+
|
|
89
|
+
content, served = generate(messages)
|
|
90
|
+
cleaned = content.to_s.strip
|
|
91
|
+
cleaned.empty? ? nil : Summary.new(text: cleaned, model: served)
|
|
92
|
+
rescue SummarizeError
|
|
93
|
+
raise
|
|
94
|
+
rescue StandardError => e
|
|
95
|
+
raise SummarizeError, "recap summarization failed: #{e.class}: #{e.message}"
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
# One side answer (a plugin's ctx.ask_model): +messages+ as they are, no
|
|
99
|
+
# tools, thinking off. An answer cut off by +max_tokens+ is kept as it
|
|
100
|
+
# is, marked with "…". Cancelling +cancel_controller+ aborts the request.
|
|
101
|
+
# @return [Summary] the answer ("" when the model said nothing)
|
|
102
|
+
# @raise [SummarizeError] any failure but a cancel
|
|
103
|
+
# @raise [LLM::RequestCancelled] +cancel_controller+ was cancelled
|
|
104
|
+
def ask(messages, max_tokens: MAX_TOKENS, cancel_controller: nil)
|
|
105
|
+
content, served = generate(messages, max_tokens: max_tokens, cancel_controller: cancel_controller, whole_sentences: false)
|
|
106
|
+
Summary.new(text: content.to_s.strip, model: served)
|
|
107
|
+
rescue SummarizeError, LLM::RequestCancelled
|
|
108
|
+
raise
|
|
109
|
+
rescue StandardError => e
|
|
110
|
+
raise SummarizeError, "the model request failed: #{e.class}: #{e.message}"
|
|
111
|
+
end
|
|
112
|
+
|
|
113
|
+
private
|
|
114
|
+
|
|
115
|
+
# One plain /chat/completions request. Returns the cleaned assistant text
|
|
116
|
+
# ("" when there is nothing after stripping) and the served model's name
|
|
117
|
+
# (nil when the reply names none). Raises SummarizeError when
|
|
118
|
+
# the reply has neither content nor reasoning_content. Cut off by
|
|
119
|
+
# +max_tokens+, the text keeps its finished sentences (+whole_sentences+)
|
|
120
|
+
# or all of it, with "…".
|
|
121
|
+
def generate(messages, max_tokens: MAX_TOKENS, cancel_controller: nil, whole_sentences: true)
|
|
122
|
+
response = @chat.chat(
|
|
123
|
+
messages: messages, model: @model, tools: [], cancel_controller: cancel_controller,
|
|
124
|
+
options: { max_tokens: max_tokens, **THINKING_OFF }
|
|
125
|
+
)
|
|
126
|
+
content = response.text
|
|
127
|
+
reasoning = response.reasoning
|
|
128
|
+
raise SummarizeError, "server returned no parseable assistant content" if content.empty? && reasoning.empty?
|
|
129
|
+
|
|
130
|
+
cut_off = response.finish_reason == "length"
|
|
131
|
+
# Reasoning cut off by max_tokens is the model's thinking, not a recap
|
|
132
|
+
# (a server that ignores the thinking switch thinks until the limit).
|
|
133
|
+
return ["", response.model] if content.empty? && cut_off
|
|
134
|
+
|
|
135
|
+
# Prefer content, fall back to a finished reasoning_content (e.g.
|
|
136
|
+
# Qwen3.6), and strip thinking tokens some models (Qwen, Gemma) leave
|
|
137
|
+
# in the text.
|
|
138
|
+
text = self.class.strip_thinking(content.empty? ? reasoning : content)
|
|
139
|
+
# Cut off by max_tokens: keep the sentences that finished ("" if none).
|
|
140
|
+
text = whole_sentences ? self.class.full_sentences(text) : "#{text}…" if cut_off && !text.empty?
|
|
141
|
+
[text, response.model]
|
|
142
|
+
rescue LLM::ProtocolError => e
|
|
143
|
+
raise SummarizeError, "server returned no parseable assistant content (#{e.message})"
|
|
144
|
+
end
|
|
145
|
+
|
|
146
|
+
end
|
|
147
|
+
end
|