samagotchi 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (243) hide show
  1. checksums.yaml +7 -0
  2. data/CHANGELOG.md +43 -0
  3. data/LICENSE +21 -0
  4. data/README.md +126 -0
  5. data/bin/chi +1140 -0
  6. data/docs/architecture.md +299 -0
  7. data/docs/cli.md +490 -0
  8. data/docs/configuration.md +494 -0
  9. data/docs/desktop.md +97 -0
  10. data/docs/guardrails.md +218 -0
  11. data/docs/hooks.md +309 -0
  12. data/docs/internals/background-tasks.md +26 -0
  13. data/docs/internals/context-telemetry.md +36 -0
  14. data/docs/internals/gemma4-contract.md +23 -0
  15. data/docs/internals/tool-guardrails.md +45 -0
  16. data/docs/memory.md +85 -0
  17. data/docs/plugins.md +819 -0
  18. data/docs/releasing.md +135 -0
  19. data/docs/sessions.md +155 -0
  20. data/lib/samagotchi/bridge/bounded_queue.rb +70 -0
  21. data/lib/samagotchi/bridge/card_store.rb +126 -0
  22. data/lib/samagotchi/bridge/event_id.rb +25 -0
  23. data/lib/samagotchi/bridge/ring_buffer.rb +63 -0
  24. data/lib/samagotchi/bridge/sse_writer.rb +248 -0
  25. data/lib/samagotchi/bridge/turn_accumulator.rb +189 -0
  26. data/lib/samagotchi/bridge.rb +993 -0
  27. data/lib/samagotchi/bridge_client/event_stream.rb +158 -0
  28. data/lib/samagotchi/bridge_client/sse_parser.rb +51 -0
  29. data/lib/samagotchi/bridge_client.rb +330 -0
  30. data/lib/samagotchi/bundle_needs.rb +97 -0
  31. data/lib/samagotchi/bundles/btw/manifest.yml +10 -0
  32. data/lib/samagotchi/bundles/btw/plugin.rb +100 -0
  33. data/lib/samagotchi/bundles/guardrails/guardrails/rules.yml +82 -0
  34. data/lib/samagotchi/bundles/guardrails/guardrails.md +14 -0
  35. data/lib/samagotchi/bundles/guardrails/manifest.yml +8 -0
  36. data/lib/samagotchi/bundles/known-names/hooks/known_names.rb +210 -0
  37. data/lib/samagotchi/bundles/known-names/known_names.md +3 -0
  38. data/lib/samagotchi/bundles/known-names/manifest.yml +14 -0
  39. data/lib/samagotchi/bundles/loop-guard/manifest.yml +10 -0
  40. data/lib/samagotchi/bundles/loop-guard/plugin.rb +158 -0
  41. data/lib/samagotchi/bundles/mcp/manifest.yml +11 -0
  42. data/lib/samagotchi/bundles/mcp/plugin.rb +631 -0
  43. data/lib/samagotchi/bundles/system/config_modification_protocol.md +149 -0
  44. data/lib/samagotchi/bundles/system/delegated.md +10 -0
  45. data/lib/samagotchi/bundles/system/identity.md +7 -0
  46. data/lib/samagotchi/bundles/system/manifest.yml +11 -0
  47. data/lib/samagotchi/bundles/system/memory_guide.md +107 -0
  48. data/lib/samagotchi/bundles/system/self_map.md +55 -0
  49. data/lib/samagotchi/cancellation_controller.rb +78 -0
  50. data/lib/samagotchi/client.rb +429 -0
  51. data/lib/samagotchi/commands/registry.rb +112 -0
  52. data/lib/samagotchi/config.rb +910 -0
  53. data/lib/samagotchi/context_note.rb +77 -0
  54. data/lib/samagotchi/context_quote.rb +21 -0
  55. data/lib/samagotchi/context_usage.rb +66 -0
  56. data/lib/samagotchi/context_window.rb +76 -0
  57. data/lib/samagotchi/debug_log.rb +110 -0
  58. data/lib/samagotchi/desktop/macos/App.swift +102 -0
  59. data/lib/samagotchi/desktop/macos/ChiRunner.swift +201 -0
  60. data/lib/samagotchi/desktop/macos/Hotkey.swift +42 -0
  61. data/lib/samagotchi/desktop/macos/Info.plist.erb +42 -0
  62. data/lib/samagotchi/desktop/macos/Panel.swift +383 -0
  63. data/lib/samagotchi/desktop/macos.rb +255 -0
  64. data/lib/samagotchi/desktop.rb +21 -0
  65. data/lib/samagotchi/desktop_command.rb +143 -0
  66. data/lib/samagotchi/engine.rb +2807 -0
  67. data/lib/samagotchi/guardrails/approval.rb +125 -0
  68. data/lib/samagotchi/guardrails/approvals.rb +177 -0
  69. data/lib/samagotchi/guardrails/context.rb +71 -0
  70. data/lib/samagotchi/guardrails/gate.rb +125 -0
  71. data/lib/samagotchi/guardrails/load_failures.rb +46 -0
  72. data/lib/samagotchi/guardrails/protected_paths.rb +77 -0
  73. data/lib/samagotchi/guardrails/rules.rb +199 -0
  74. data/lib/samagotchi/guardrails/targets.rb +119 -0
  75. data/lib/samagotchi/guardrails/verdict.rb +134 -0
  76. data/lib/samagotchi/guardrails.rb +18 -0
  77. data/lib/samagotchi/hooks/bundle_loader.rb +158 -0
  78. data/lib/samagotchi/hooks/loader.rb +162 -0
  79. data/lib/samagotchi/hooks/registry.rb +261 -0
  80. data/lib/samagotchi/hooks.rb +30 -0
  81. data/lib/samagotchi/host_registry.rb +315 -0
  82. data/lib/samagotchi/idle_client.rb +147 -0
  83. data/lib/samagotchi/idle_recap.rb +549 -0
  84. data/lib/samagotchi/idle_reminders.rb +101 -0
  85. data/lib/samagotchi/idle_scheduler.rb +76 -0
  86. data/lib/samagotchi/image_store.rb +393 -0
  87. data/lib/samagotchi/installed_gem.rb +38 -0
  88. data/lib/samagotchi/kernel_loop.rb +1017 -0
  89. data/lib/samagotchi/launch_mode.rb +34 -0
  90. data/lib/samagotchi/llm/backend.rb +28 -0
  91. data/lib/samagotchi/llm/chat_loop.rb +450 -0
  92. data/lib/samagotchi/llm/errors.rb +329 -0
  93. data/lib/samagotchi/llm/http.rb +412 -0
  94. data/lib/samagotchi/llm/model_result.rb +72 -0
  95. data/lib/samagotchi/llm/native_backend.rb +50 -0
  96. data/lib/samagotchi/llm/native_tool_normalizer.rb +277 -0
  97. data/lib/samagotchi/llm/openai_chat.rb +403 -0
  98. data/lib/samagotchi/llm/usage.rb +79 -0
  99. data/lib/samagotchi/log.rb +200 -0
  100. data/lib/samagotchi/log_line.rb +127 -0
  101. data/lib/samagotchi/log_path.rb +31 -0
  102. data/lib/samagotchi/log_subscriber.rb +163 -0
  103. data/lib/samagotchi/memory_bundle/builder.rb +364 -0
  104. data/lib/samagotchi/memory_bundle/index_updater.rb +123 -0
  105. data/lib/samagotchi/memory_bundle/installer.rb +528 -0
  106. data/lib/samagotchi/memory_bundle/listing.rb +72 -0
  107. data/lib/samagotchi/memory_bundle/manifest.rb +225 -0
  108. data/lib/samagotchi/memory_bundle/merger.rb +52 -0
  109. data/lib/samagotchi/memory_bundle/placeholder.rb +37 -0
  110. data/lib/samagotchi/memory_bundle/provenance.rb +257 -0
  111. data/lib/samagotchi/memory_bundle/source.rb +153 -0
  112. data/lib/samagotchi/memory_bundle/status.rb +107 -0
  113. data/lib/samagotchi/memory_bundle/system_bundle.rb +161 -0
  114. data/lib/samagotchi/memory_bundle/uninstaller.rb +128 -0
  115. data/lib/samagotchi/memory_bundle.rb +17 -0
  116. data/lib/samagotchi/memory_paths.rb +101 -0
  117. data/lib/samagotchi/model_overlay.rb +53 -0
  118. data/lib/samagotchi/model_profile.rb +309 -0
  119. data/lib/samagotchi/muted_memories.rb +66 -0
  120. data/lib/samagotchi/note_command.rb +163 -0
  121. data/lib/samagotchi/output_formatter.rb +100 -0
  122. data/lib/samagotchi/owner_lock.rb +110 -0
  123. data/lib/samagotchi/pending_input_queue.rb +48 -0
  124. data/lib/samagotchi/plugin/api.rb +362 -0
  125. data/lib/samagotchi/plugin/context.rb +193 -0
  126. data/lib/samagotchi/plugin/loader.rb +126 -0
  127. data/lib/samagotchi/plugin/service.rb +117 -0
  128. data/lib/samagotchi/plugin/sessions.rb +150 -0
  129. data/lib/samagotchi/plugin/side_question.rb +60 -0
  130. data/lib/samagotchi/plugin/tool_result.rb +24 -0
  131. data/lib/samagotchi/project_scope.rb +25 -0
  132. data/lib/samagotchi/prompt.rb +119 -0
  133. data/lib/samagotchi/prompt_literal_guard.rb +70 -0
  134. data/lib/samagotchi/recap_store.rb +92 -0
  135. data/lib/samagotchi/reminder_store.rb +165 -0
  136. data/lib/samagotchi/self_report.rb +195 -0
  137. data/lib/samagotchi/send_command.rb +170 -0
  138. data/lib/samagotchi/served_model.rb +32 -0
  139. data/lib/samagotchi/session.rb +508 -0
  140. data/lib/samagotchi/session_commands.rb +527 -0
  141. data/lib/samagotchi/session_delete_command.rb +105 -0
  142. data/lib/samagotchi/session_manager.rb +1049 -0
  143. data/lib/samagotchi/session_metrics.rb +466 -0
  144. data/lib/samagotchi/session_observer.rb +117 -0
  145. data/lib/samagotchi/terminal_ui/attach_launcher.rb +118 -0
  146. data/lib/samagotchi/terminal_ui/attached_loop.rb +1037 -0
  147. data/lib/samagotchi/terminal_ui/attached_view.rb +264 -0
  148. data/lib/samagotchi/terminal_ui/event_renderer.rb +192 -0
  149. data/lib/samagotchi/terminal_ui/formatting.rb +291 -0
  150. data/lib/samagotchi/terminal_ui/image_input.rb +36 -0
  151. data/lib/samagotchi/terminal_ui/input_support.rb +324 -0
  152. data/lib/samagotchi/terminal_ui/legacy_surface.rb +111 -0
  153. data/lib/samagotchi/terminal_ui/line_reader.rb +113 -0
  154. data/lib/samagotchi/terminal_ui/live_region.rb +36 -0
  155. data/lib/samagotchi/terminal_ui/plain_surface.rb +51 -0
  156. data/lib/samagotchi/terminal_ui/question_prompt.rb +153 -0
  157. data/lib/samagotchi/terminal_ui/question_slot.rb +131 -0
  158. data/lib/samagotchi/terminal_ui/reline_seam.rb +216 -0
  159. data/lib/samagotchi/terminal_ui/repl_input.rb +138 -0
  160. data/lib/samagotchi/terminal_ui/screen.rb +316 -0
  161. data/lib/samagotchi/terminal_ui/surface.rb +47 -0
  162. data/lib/samagotchi/terminal_ui/thinking_line.rb +101 -0
  163. data/lib/samagotchi/terminal_ui.rb +1992 -0
  164. data/lib/samagotchi/thinking_ticker.rb +110 -0
  165. data/lib/samagotchi/thought_stream_splitter.rb +149 -0
  166. data/lib/samagotchi/token_usage.rb +88 -0
  167. data/lib/samagotchi/tool_activity.rb +216 -0
  168. data/lib/samagotchi/tool_call_parser.rb +637 -0
  169. data/lib/samagotchi/tool_declarations.rb +561 -0
  170. data/lib/samagotchi/tool_runner.rb +211 -0
  171. data/lib/samagotchi/tools/args.rb +259 -0
  172. data/lib/samagotchi/tools/ask_user_question.rb +152 -0
  173. data/lib/samagotchi/tools/builtins.rb +122 -0
  174. data/lib/samagotchi/tools/cancel_reminder.rb +21 -0
  175. data/lib/samagotchi/tools/delegate.rb +167 -0
  176. data/lib/samagotchi/tools/delegate_result.rb +53 -0
  177. data/lib/samagotchi/tools/delegate_wait.rb +153 -0
  178. data/lib/samagotchi/tools/edit.rb +155 -0
  179. data/lib/samagotchi/tools/execute.rb +214 -0
  180. data/lib/samagotchi/tools/list_reminders.rb +20 -0
  181. data/lib/samagotchi/tools/list_sessions.rb +74 -0
  182. data/lib/samagotchi/tools/memory.rb +256 -0
  183. data/lib/samagotchi/tools/output_guardrails.rb +93 -0
  184. data/lib/samagotchi/tools/peers.rb +18 -0
  185. data/lib/samagotchi/tools/read.rb +182 -0
  186. data/lib/samagotchi/tools/register_reminder.rb +53 -0
  187. data/lib/samagotchi/tools/registry.rb +60 -0
  188. data/lib/samagotchi/tools/send_note.rb +49 -0
  189. data/lib/samagotchi/tools/task_create.rb +29 -0
  190. data/lib/samagotchi/tools/task_get.rb +39 -0
  191. data/lib/samagotchi/tools/task_list.rb +43 -0
  192. data/lib/samagotchi/tools/task_runtime.rb +311 -0
  193. data/lib/samagotchi/tools/task_stop.rb +29 -0
  194. data/lib/samagotchi/tools/task_wait.rb +104 -0
  195. data/lib/samagotchi/tools/tool_path.rb +18 -0
  196. data/lib/samagotchi/tools/web_fetch.rb +163 -0
  197. data/lib/samagotchi/tools/write.rb +26 -0
  198. data/lib/samagotchi/turn_flow.rb +242 -0
  199. data/lib/samagotchi/turn_note.rb +76 -0
  200. data/lib/samagotchi/turn_tally.rb +101 -0
  201. data/lib/samagotchi/version.rb +7 -0
  202. data/lib/samagotchi/vision_context.rb +132 -0
  203. data/lib/samagotchi/vision_support.rb +109 -0
  204. data/lib/samagotchi/web/app.rb +1349 -0
  205. data/lib/samagotchi/web/markdown_renderer.rb +107 -0
  206. data/lib/samagotchi/web/message_parts.rb +169 -0
  207. data/lib/samagotchi/web/public/activity.js +100 -0
  208. data/lib/samagotchi/web/public/annotations.js +67 -0
  209. data/lib/samagotchi/web/public/app.js +2382 -0
  210. data/lib/samagotchi/web/public/card.js +74 -0
  211. data/lib/samagotchi/web/public/chat_view.js +360 -0
  212. data/lib/samagotchi/web/public/chunk_router.js +25 -0
  213. data/lib/samagotchi/web/public/command_complete.js +39 -0
  214. data/lib/samagotchi/web/public/composer_size.js +19 -0
  215. data/lib/samagotchi/web/public/copy.js +142 -0
  216. data/lib/samagotchi/web/public/ctx.js +35 -0
  217. data/lib/samagotchi/web/public/data.js +256 -0
  218. data/lib/samagotchi/web/public/format.js +232 -0
  219. data/lib/samagotchi/web/public/hold.js +78 -0
  220. data/lib/samagotchi/web/public/images.js +77 -0
  221. data/lib/samagotchi/web/public/index.html +568 -0
  222. data/lib/samagotchi/web/public/init_row.js +60 -0
  223. data/lib/samagotchi/web/public/model_pick.js +23 -0
  224. data/lib/samagotchi/web/public/question_card.js +100 -0
  225. data/lib/samagotchi/web/public/route.js +17 -0
  226. data/lib/samagotchi/web/public/scope.js +36 -0
  227. data/lib/samagotchi/web/public/scroll.js +24 -0
  228. data/lib/samagotchi/web/public/sentences.js +88 -0
  229. data/lib/samagotchi/web/public/sessions_list.js +60 -0
  230. data/lib/samagotchi/web/public/strip.js +25 -0
  231. data/lib/samagotchi/web/public/tally.js +37 -0
  232. data/lib/samagotchi/web/public/thinking_ticker.js +79 -0
  233. data/lib/samagotchi/web/public/timing.js +185 -0
  234. data/lib/samagotchi/web/public/turn_events.js +209 -0
  235. data/lib/samagotchi/web/public/turn_model.js +204 -0
  236. data/lib/samagotchi/web/public/turn_view.js +587 -0
  237. data/lib/samagotchi/web/server.rb +183 -0
  238. data/lib/samagotchi/web/session_hub.rb +329 -0
  239. data/lib/samagotchi/web/session_summary.rb +85 -0
  240. data/lib/samagotchi/worker.rb +635 -0
  241. data/lib/samagotchi/worker_idle_exit.rb +87 -0
  242. data/lib/samagotchi.rb +12 -0
  243. metadata +374 -0
@@ -0,0 +1,315 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative "config"
4
+ require_relative "client"
5
+ require_relative "llm/openai_chat"
6
+
7
+ module Samagotchi
8
+ # HostRegistry manages multiple model hosts (llama.cpp / mlx / oMLX) and
9
+ # provides lazy model discovery aggregation.
10
+ #
11
+ # - Hosts are defined in config.yml `hosts:` section or synthesized from
12
+ # SAMAGOTCHI_SERVER_HOST/PORT env (via Config).
13
+ # - Discovery is lazy: list_all_models is the explicit trigger (called by
14
+ # /models), not on startup. Each host's list is cached (60s, 10 minutes
15
+ # for a remote host) with skip-on-error; lists are LLM::ModelInfo.
16
+ # - Routing: client_for_model resolves a (possibly qualified) model string
17
+ # to the appropriate Client instance. A remote host is chosen only by
18
+ # exact model id, host:model or an alias, never by a substring.
19
+ class HostRegistry
20
+ CACHE_TTL_SECONDS = 60
21
+ REMOTE_CACHE_TTL_SECONDS = 600
22
+ LIST_TIMEOUT_SECONDS = 3
23
+ # Seconds a remote host's stream may take to show something (a queued
24
+ # free model on OpenRouter can send only keep-alives for minutes).
25
+ REMOTE_FIRST_TOKEN_TIMEOUT = 120
26
+
27
+ # url: the configured url, when the entry has one (host, port and scheme
28
+ # come from it); api_key_env: the variable holding the host's API key;
29
+ # profile: the configured prompt profile name, if any;
30
+ # first_token_timeout: the configured first-token limit (see #first_token_limit);
31
+ # vision: the configured true/false (VisionSupport), nil when unset.
32
+ HostEntry = Struct.new(:name, :host, :port, :transport, :client, :api, :scheme, :url, :api_key_env, :profile,
33
+ :first_token_timeout, :vision, keyword_init: true) do
34
+ # Talks the OpenAI chat API (the chat loop); nil and raw apis use the
35
+ # raw-prompt loop.
36
+ def chat? = api == :openai
37
+
38
+ # The server root, e.g. for llama.cpp's own endpoints and recap.
39
+ def root_url = "#{scheme || "http"}://#{host}:#{port}"
40
+
41
+ # The OpenAI-compatible API base the chat loop talks to: the url as
42
+ # configured, else the root's /v1.
43
+ def openai_base_url = url || "#{root_url}/v1"
44
+
45
+ # A provider on the network rather than a local server: it needs a key
46
+ # or speaks https. Its model list is cached longer and it is never
47
+ # picked by a substring of a model name.
48
+ def remote? = !api_key_env.to_s.empty? || scheme == "https"
49
+
50
+ def models_ttl = remote? ? REMOTE_CACHE_TTL_SECONDS : CACHE_TTL_SECONDS
51
+
52
+ # Seconds a streamed answer may take to show something, or nil: the
53
+ # host's first_token_timeout, else server.first_token_timeout, else
54
+ # 120 for a remote host (a local server's long prompt eval is normal,
55
+ # and read_timeout catches a dead one). 0 turns it off.
56
+ def first_token_limit
57
+ seconds = first_token_timeout
58
+ seconds = HostRegistry.configured_first_token_timeout if seconds.nil?
59
+ seconds = remote? ? REMOTE_FIRST_TOKEN_TIMEOUT : nil if seconds.nil?
60
+ seconds&.positive? ? seconds : nil
61
+ end
62
+ end
63
+
64
+ # Where a model's requests go: the host entry, the client to use and the
65
+ # model name to send (the host prefix stripped).
66
+ ModelTarget = Data.define(:model, :entry, :bare_model, :client) do
67
+ def root_url = entry.root_url
68
+ def openai_base_url = entry.openai_base_url
69
+ end
70
+
71
+ # A client that every target uses instead of its host's own (specs inject
72
+ # a stub this way; the Engine/TUI `client:` keyword sets it).
73
+ attr_accessor :client_override
74
+
75
+ # @param clock [#call, nil] monotonic seconds (specs)
76
+ def initialize(hosts_config: nil, env: ENV, client_override: nil, clock: nil)
77
+ @client_override = client_override
78
+ @clock = clock || -> { Process.clock_gettime(Process::CLOCK_MONOTONIC) }
79
+ @adapters = {}
80
+ @host_lists = {}
81
+ raw = hosts_config || ConfigFile.hosts_config(env: env)
82
+ @entries = {}
83
+ raw.each do |key, cfg|
84
+ # cfg: {name:, host:, port:, transport:, original_name:}
85
+ transport = cfg[:transport]
86
+ entry = HostEntry.new(name: key.to_s.downcase, host: cfg[:host], port: cfg[:port].to_i, transport: transport,
87
+ api: cfg[:api]&.to_sym, scheme: cfg[:scheme], url: cfg[:url], api_key_env: cfg[:api_key_env],
88
+ profile: cfg[:profile], first_token_timeout: cfg[:first_token_timeout],
89
+ vision: cfg[:vision])
90
+ entry.client = Client.new(host: cfg[:host], port: cfg[:port], transport: transport, scheme: cfg[:scheme],
91
+ first_token_timeout: entry.first_token_limit, name: entry.name)
92
+ @entries[entry.name] = entry
93
+ end
94
+ # Fallback single entry (should already be synthesized by hosts_config, but guard)
95
+ if @entries.empty?
96
+ host = Config.get("server.host")
97
+ host = "localhost" if host.empty?
98
+ port = Config.get("server.port").to_i
99
+ port = 8080 if port <= 0
100
+ @entries["default"] = HostEntry.new(name: "default", host: host, port: port, transport: nil, client: Client.new(host: host, port: port, name: "default"))
101
+ end
102
+ @mutex = Mutex.new
103
+ @cache = nil
104
+ @cache_at = nil
105
+ @model_index = nil # downcased model_id => host_name
106
+ end
107
+
108
+ def entries
109
+ @entries
110
+ end
111
+
112
+ # server.first_token_timeout, or nil when unset or unreadable.
113
+ def self.configured_first_token_timeout
114
+ Config.get("server.first_token_timeout")
115
+ rescue StandardError
116
+ nil
117
+ end
118
+
119
+ def entry_names
120
+ @entries.keys
121
+ end
122
+
123
+ def default_entry
124
+ @entries["default"] || @entries.values.first
125
+ end
126
+
127
+ def find_entry(name)
128
+ @entries[name.to_s.strip.downcase]
129
+ end
130
+
131
+ # Parse host-qualified model string using known host names.
132
+ # Returns [host_name_or_nil, bare_model]
133
+ def parse_qualified_model(raw)
134
+ ConfigFile.parse_host_qualified_model(raw, hosts: @entries)
135
+ end
136
+
137
+ # Resolve model string (already alias-resolved, may be qualified) to a HostEntry.
138
+ # If qualified explicitly, return that host. If unqualified, try cached model index,
139
+ # else fallback to default host.
140
+ def host_for_model(raw_model)
141
+ host_ref, bare = parse_qualified_model(raw_model)
142
+ # If qualified, try to resolve alias on the bare part (small-box:small -> small-box:gemma-small)
143
+ if host_ref && bare
144
+ begin
145
+ aliases = ConfigFile.model_aliases
146
+ resolved = aliases.fetch(bare.downcase, bare)
147
+ bare = resolved if resolved != bare
148
+ rescue StandardError
149
+ nil
150
+ end
151
+ entry = find_entry(host_ref)
152
+ return [entry, bare] if entry
153
+ # Unknown prefix — treat as bare model on default host
154
+ return [default_entry, raw_model.to_s.strip]
155
+ end
156
+ # Unqualified: also try alias resolution for discovery (small -> gemma-small or small -> small-box:gemma-small)
157
+ begin
158
+ aliases = ConfigFile.model_aliases
159
+ resolved = aliases.fetch(bare.to_s.strip.downcase, bare)
160
+ if resolved != bare
161
+ # If alias points to a qualified ref, re-parse it
162
+ q_host, q_bare = parse_qualified_model(resolved)
163
+ if q_host
164
+ entry = find_entry(q_host)
165
+ return [entry, q_bare] if entry
166
+ end
167
+ bare = resolved
168
+ end
169
+ rescue StandardError
170
+ nil
171
+ end
172
+ bare_down = bare.to_s.strip.downcase
173
+ # Try cached index (populated after list_all_models)
174
+ idx = @mutex.synchronize { @model_index }
175
+ if idx && idx.key?(bare_down)
176
+ host_name = idx[bare_down]
177
+ entry = find_entry(host_name)
178
+ return [entry, bare] if entry
179
+ end
180
+ # Fallback: try substring match in cached aggregated results if available
181
+ # (lightweight: scan cached model lists). Remote hosts match exactly only.
182
+ cached = @mutex.synchronize { @cache }
183
+ if cached
184
+ cached.each do |hname, data|
185
+ next unless data[:models]
186
+ data[:models].each do |m|
187
+ if m.id.downcase == bare_down
188
+ entry = find_entry(hname)
189
+ return [entry, bare] if entry
190
+ end
191
+ end
192
+ end
193
+ cached.each do |hname, data|
194
+ next unless data[:models]
195
+ next if find_entry(hname)&.remote?
196
+
197
+ data[:models].each do |m|
198
+ if m.id.downcase.include?(bare_down)
199
+ entry = find_entry(hname)
200
+ return [entry, bare] if entry
201
+ end
202
+ end
203
+ end
204
+ end
205
+ [default_entry, bare]
206
+ end
207
+
208
+ # The chat adapter for a host (one per host, so its cached model list
209
+ # serves the context window). Only chat hosts use it for turns.
210
+ # @return [LLM::OpenAIChat]
211
+ def adapter_for(entry)
212
+ @mutex.synchronize do
213
+ @adapters[entry.name] ||= LLM::OpenAIChat.for(entry, models_ttl: entry.models_ttl,
214
+ first_token_timeout: entry.first_token_limit)
215
+ end
216
+ end
217
+
218
+ # A host's models as ModelInfo: a chat host's from its adapter, a raw
219
+ # host's from its Client (ids from the server's own list shape).
220
+ def list_models_for(entry)
221
+ return adapter_for(entry).list_models if entry.chat? && !@client_override
222
+
223
+ Array(client_for(entry).list_models).map do |raw|
224
+ if raw.is_a?(Hash)
225
+ id = raw["id"] || raw[:id] || raw["model"] || raw["name"]
226
+ LLM::ModelInfo.new(id: id.to_s, context_window: nil, supports_tools: nil, raw: raw)
227
+ else
228
+ LLM::ModelInfo.new(id: raw.to_s, context_window: nil, supports_tools: nil, raw: {})
229
+ end
230
+ end
231
+ end
232
+
233
+ def client_for_model(raw_model)
234
+ host_entry, bare = host_for_model(raw_model)
235
+ [client_for(host_entry), bare, host_entry]
236
+ end
237
+
238
+ # The single host/model resolution: alias and host routing (host_for_model)
239
+ # plus the name sent to the server (bare_name).
240
+ # @param raw_model [String] a model name, alias or host:model ref
241
+ # @return [ModelTarget]
242
+ def resolve(raw_model)
243
+ entry, = host_for_model(raw_model)
244
+ ModelTarget.new(model: raw_model, entry: entry, bare_model: bare_name(raw_model), client: client_for(entry))
245
+ end
246
+
247
+ # The model name without a known host prefix ("box:gemma" → "gemma").
248
+ # Aliases are not applied here.
249
+ def bare_name(full_ref)
250
+ _, bare = parse_qualified_model(full_ref)
251
+ bare.to_s.strip.empty? ? full_ref.to_s.strip : bare
252
+ end
253
+
254
+ def client_for(entry)
255
+ @client_override || entry.client
256
+ end
257
+
258
+ # List models on all hosts in parallel. On error per-host, skip with error entry (no failover).
259
+ # Returns { host_name => { host:, port:, transport:, models: [LLM::ModelInfo], error: nil|String } }
260
+ # Also populates the model index. Unless forced, a host's list is reused
261
+ # for its TTL (60s; 10 minutes for a remote host).
262
+ def list_all_models(force: true)
263
+ results = {}
264
+ results_mutex = Mutex.new
265
+ threads = @entries.map do |name, entry|
266
+ fresh = !force && fresh_list(name, entry)
267
+ next results_mutex.synchronize { results[name] = fresh } if fresh
268
+
269
+ Thread.new do
270
+ begin
271
+ models = list_models_for(entry)
272
+ data = { host: entry.host, port: entry.port, transport: entry.transport, models: models, error: nil }
273
+ @mutex.synchronize { @host_lists[name] = { data: data, at: @clock.call } }
274
+ rescue StandardError => e
275
+ data = { host: entry.host, port: entry.port, transport: entry.transport, models: [], error: e.message }
276
+ end
277
+ results_mutex.synchronize { results[name] = data }
278
+ end
279
+ end
280
+ threads.each { |thread| thread.join if thread.is_a?(Thread) }
281
+
282
+ # Build model index: model_id downcased -> host_name (first host wins)
283
+ index = {}
284
+ results.each do |hname, data|
285
+ next if data[:error]
286
+ Array(data[:models]).each do |m|
287
+ mid = m.id.to_s
288
+ next if mid.strip.empty?
289
+ down = mid.downcase
290
+ index[down] = hname unless index.key?(down)
291
+ end
292
+ end
293
+
294
+ @mutex.synchronize do
295
+ @cache = results
296
+ @cache_at = @clock.call
297
+ @model_index = index
298
+ end
299
+ results
300
+ end
301
+
302
+ def cached_results
303
+ @mutex.synchronize { @cache }
304
+ end
305
+
306
+ private
307
+
308
+ def fresh_list(name, entry)
309
+ @mutex.synchronize do
310
+ cached = @host_lists[name]
311
+ cached[:data] if cached && (@clock.call - cached[:at]) < entry.models_ttl
312
+ end
313
+ end
314
+ end
315
+ end
@@ -0,0 +1,147 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "json"
4
+ require_relative "llm/openai_chat"
5
+
6
+ module Samagotchi
7
+ # Standalone summarizer for the idle session-recap feature.
8
+ #
9
+ # Asks an OpenAI-compatible /chat/completions endpoint (the local llama.cpp,
10
+ # or a recap host's) through its own OpenAIChat: one plain request, no
11
+ # tools, no retries, its own timeout. It is NEVER wired into the Engine's
12
+ # kernel/backend, so it shares no engine concurrency. The Engine's idle
13
+ # detector only snapshots session.messages and calls #summarize; this object
14
+ # owns the HTTP boundary and is fully decoupled (configurable base_url + model).
15
+ #
16
+ # The request targets /chat/completions, NOT /completions (a raw
17
+ # completions endpoint loops on gemma think tokens).
18
+ #
19
+ # #summarize raises SummarizeError on any failure; the idle detector isolates
20
+ # that so a failed recap never breaks the active session.
21
+ class IdleClient
22
+ # Raised when summarization fails (server down, timeout, malformed body…).
23
+ class SummarizeError < StandardError; end
24
+
25
+ # What #summarize returns: the recap text and the model that answered
26
+ # (the server's name for it; nil when the reply names none). #to_s is
27
+ # the text.
28
+ Summary = Data.define(:text, :model) do
29
+ def to_s = text
30
+ end
31
+
32
+ DEFAULT_TIMEOUT_SECONDS = 30.0
33
+ # Up to 10 sentences (recap.sentences) need ~350 tokens with thinking off.
34
+ MAX_TOKENS = 512
35
+
36
+ # Request fields that turn thinking off. A reasoning model otherwise
37
+ # spends the budget thinking and the recap stops mid-sentence.
38
+ # - chat_template_kwargs.enable_thinking: the chat template's switch
39
+ # (llama.cpp with Qwen/Gemma templates); templates without it ignore it.
40
+ # - reasoning_effort "none": the OpenAI-style knob, for servers that
41
+ # ignore the template switch (Splash thought until max_tokens).
42
+ THINKING_OFF = {
43
+ chat_template_kwargs: { enable_thinking: false },
44
+ reasoning_effort: "none"
45
+ }.freeze
46
+
47
+ # @param base_url [String] the OpenAI API base, e.g. http://host:8081/v1
48
+ # @param api_key_env [String, nil] the variable holding the host's key
49
+ # @param timeout [Numeric] HTTP request timeout. Kept to the recap's own
50
+ # wait budget (not the chat's global request_timeout) so an abandoned
51
+ # summarize thread can't outlive the recap attempt by minutes.
52
+ def initialize(model:, base_url: nil, api_key_env: nil, timeout: DEFAULT_TIMEOUT_SECONDS, env: ENV)
53
+ @model = model
54
+ # A recap is best-effort: one short attempt, no retries. The idle job
55
+ # tries again after the next activity, never on its own.
56
+ @chat = LLM::OpenAIChat.new(base_url: base_url.to_s, host_name: "recap", api_key_env: api_key_env,
57
+ stream: false, retries: false, timeout: timeout, env: env, purpose: "recap")
58
+ end
59
+
60
+ THINK_RE = /<\|think\|.*?\|think\|>/m
61
+ LITERAL_THINK_RE = /\[\[SAMAGOTCHI_LITERAL_THINK_OPEN\]\].*?\[\[SAMAGOTCHI_LITERAL_THINK_CLOSE\]\]/m
62
+
63
+ # Strip thinking blocks (gemma <|think|>…, qwen prompt literals) and
64
+ # collapse the blank lines they leave. Shared with IdleRecap's
65
+ # transcript filter.
66
+ def self.strip_thinking(text)
67
+ text.to_s
68
+ .gsub(THINK_RE, "")
69
+ .gsub(LITERAL_THINK_RE, "")
70
+ .gsub(/\n\n+/, "\n")
71
+ .strip
72
+ end
73
+
74
+ # +text+ up to its last sentence end (. ! ? plus closing quotes or
75
+ # brackets, then whitespace or the end), or "" when none finished.
76
+ def self.full_sentences(text)
77
+ text.to_s[/\A.*[.!?]["')\]`*]*(?=\s|\z)/m].to_s
78
+ end
79
+
80
+ # Summarize an already-built recap prompt: a string (one user message) or
81
+ # a list of chat messages. Returns a Summary of the cleaned prose, or nil
82
+ # when there is nothing to summarize. Any failure raises SummarizeError
83
+ # (the caller isolates it).
84
+ # @return [Summary, nil]
85
+ def summarize(prompt)
86
+ messages = prompt.is_a?(Array) ? prompt : [{ role: "user", content: prompt.to_s.strip }]
87
+ return nil if messages.all? { |m| m[:content].to_s.strip.empty? }
88
+
89
+ content, served = generate(messages)
90
+ cleaned = content.to_s.strip
91
+ cleaned.empty? ? nil : Summary.new(text: cleaned, model: served)
92
+ rescue SummarizeError
93
+ raise
94
+ rescue StandardError => e
95
+ raise SummarizeError, "recap summarization failed: #{e.class}: #{e.message}"
96
+ end
97
+
98
+ # One side answer (a plugin's ctx.ask_model): +messages+ as they are, no
99
+ # tools, thinking off. An answer cut off by +max_tokens+ is kept as it
100
+ # is, marked with "…". Cancelling +cancel_controller+ aborts the request.
101
+ # @return [Summary] the answer ("" when the model said nothing)
102
+ # @raise [SummarizeError] any failure but a cancel
103
+ # @raise [LLM::RequestCancelled] +cancel_controller+ was cancelled
104
+ def ask(messages, max_tokens: MAX_TOKENS, cancel_controller: nil)
105
+ content, served = generate(messages, max_tokens: max_tokens, cancel_controller: cancel_controller, whole_sentences: false)
106
+ Summary.new(text: content.to_s.strip, model: served)
107
+ rescue SummarizeError, LLM::RequestCancelled
108
+ raise
109
+ rescue StandardError => e
110
+ raise SummarizeError, "the model request failed: #{e.class}: #{e.message}"
111
+ end
112
+
113
+ private
114
+
115
+ # One plain /chat/completions request. Returns the cleaned assistant text
116
+ # ("" when there is nothing after stripping) and the served model's name
117
+ # (nil when the reply names none). Raises SummarizeError when
118
+ # the reply has neither content nor reasoning_content. Cut off by
119
+ # +max_tokens+, the text keeps its finished sentences (+whole_sentences+)
120
+ # or all of it, with "…".
121
+ def generate(messages, max_tokens: MAX_TOKENS, cancel_controller: nil, whole_sentences: true)
122
+ response = @chat.chat(
123
+ messages: messages, model: @model, tools: [], cancel_controller: cancel_controller,
124
+ options: { max_tokens: max_tokens, **THINKING_OFF }
125
+ )
126
+ content = response.text
127
+ reasoning = response.reasoning
128
+ raise SummarizeError, "server returned no parseable assistant content" if content.empty? && reasoning.empty?
129
+
130
+ cut_off = response.finish_reason == "length"
131
+ # Reasoning cut off by max_tokens is the model's thinking, not a recap
132
+ # (a server that ignores the thinking switch thinks until the limit).
133
+ return ["", response.model] if content.empty? && cut_off
134
+
135
+ # Prefer content, fall back to a finished reasoning_content (e.g.
136
+ # Qwen3.6), and strip thinking tokens some models (Qwen, Gemma) leave
137
+ # in the text.
138
+ text = self.class.strip_thinking(content.empty? ? reasoning : content)
139
+ # Cut off by max_tokens: keep the sentences that finished ("" if none).
140
+ text = whole_sentences ? self.class.full_sentences(text) : "#{text}…" if cut_off && !text.empty?
141
+ [text, response.model]
142
+ rescue LLM::ProtocolError => e
143
+ raise SummarizeError, "server returned no parseable assistant content (#{e.message})"
144
+ end
145
+
146
+ end
147
+ end