samagotchi 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (243) hide show
  1. checksums.yaml +7 -0
  2. data/CHANGELOG.md +43 -0
  3. data/LICENSE +21 -0
  4. data/README.md +126 -0
  5. data/bin/chi +1140 -0
  6. data/docs/architecture.md +299 -0
  7. data/docs/cli.md +490 -0
  8. data/docs/configuration.md +494 -0
  9. data/docs/desktop.md +97 -0
  10. data/docs/guardrails.md +218 -0
  11. data/docs/hooks.md +309 -0
  12. data/docs/internals/background-tasks.md +26 -0
  13. data/docs/internals/context-telemetry.md +36 -0
  14. data/docs/internals/gemma4-contract.md +23 -0
  15. data/docs/internals/tool-guardrails.md +45 -0
  16. data/docs/memory.md +85 -0
  17. data/docs/plugins.md +819 -0
  18. data/docs/releasing.md +135 -0
  19. data/docs/sessions.md +155 -0
  20. data/lib/samagotchi/bridge/bounded_queue.rb +70 -0
  21. data/lib/samagotchi/bridge/card_store.rb +126 -0
  22. data/lib/samagotchi/bridge/event_id.rb +25 -0
  23. data/lib/samagotchi/bridge/ring_buffer.rb +63 -0
  24. data/lib/samagotchi/bridge/sse_writer.rb +248 -0
  25. data/lib/samagotchi/bridge/turn_accumulator.rb +189 -0
  26. data/lib/samagotchi/bridge.rb +993 -0
  27. data/lib/samagotchi/bridge_client/event_stream.rb +158 -0
  28. data/lib/samagotchi/bridge_client/sse_parser.rb +51 -0
  29. data/lib/samagotchi/bridge_client.rb +330 -0
  30. data/lib/samagotchi/bundle_needs.rb +97 -0
  31. data/lib/samagotchi/bundles/btw/manifest.yml +10 -0
  32. data/lib/samagotchi/bundles/btw/plugin.rb +100 -0
  33. data/lib/samagotchi/bundles/guardrails/guardrails/rules.yml +82 -0
  34. data/lib/samagotchi/bundles/guardrails/guardrails.md +14 -0
  35. data/lib/samagotchi/bundles/guardrails/manifest.yml +8 -0
  36. data/lib/samagotchi/bundles/known-names/hooks/known_names.rb +210 -0
  37. data/lib/samagotchi/bundles/known-names/known_names.md +3 -0
  38. data/lib/samagotchi/bundles/known-names/manifest.yml +14 -0
  39. data/lib/samagotchi/bundles/loop-guard/manifest.yml +10 -0
  40. data/lib/samagotchi/bundles/loop-guard/plugin.rb +158 -0
  41. data/lib/samagotchi/bundles/mcp/manifest.yml +11 -0
  42. data/lib/samagotchi/bundles/mcp/plugin.rb +631 -0
  43. data/lib/samagotchi/bundles/system/config_modification_protocol.md +149 -0
  44. data/lib/samagotchi/bundles/system/delegated.md +10 -0
  45. data/lib/samagotchi/bundles/system/identity.md +7 -0
  46. data/lib/samagotchi/bundles/system/manifest.yml +11 -0
  47. data/lib/samagotchi/bundles/system/memory_guide.md +107 -0
  48. data/lib/samagotchi/bundles/system/self_map.md +55 -0
  49. data/lib/samagotchi/cancellation_controller.rb +78 -0
  50. data/lib/samagotchi/client.rb +429 -0
  51. data/lib/samagotchi/commands/registry.rb +112 -0
  52. data/lib/samagotchi/config.rb +910 -0
  53. data/lib/samagotchi/context_note.rb +77 -0
  54. data/lib/samagotchi/context_quote.rb +21 -0
  55. data/lib/samagotchi/context_usage.rb +66 -0
  56. data/lib/samagotchi/context_window.rb +76 -0
  57. data/lib/samagotchi/debug_log.rb +110 -0
  58. data/lib/samagotchi/desktop/macos/App.swift +102 -0
  59. data/lib/samagotchi/desktop/macos/ChiRunner.swift +201 -0
  60. data/lib/samagotchi/desktop/macos/Hotkey.swift +42 -0
  61. data/lib/samagotchi/desktop/macos/Info.plist.erb +42 -0
  62. data/lib/samagotchi/desktop/macos/Panel.swift +383 -0
  63. data/lib/samagotchi/desktop/macos.rb +255 -0
  64. data/lib/samagotchi/desktop.rb +21 -0
  65. data/lib/samagotchi/desktop_command.rb +143 -0
  66. data/lib/samagotchi/engine.rb +2807 -0
  67. data/lib/samagotchi/guardrails/approval.rb +125 -0
  68. data/lib/samagotchi/guardrails/approvals.rb +177 -0
  69. data/lib/samagotchi/guardrails/context.rb +71 -0
  70. data/lib/samagotchi/guardrails/gate.rb +125 -0
  71. data/lib/samagotchi/guardrails/load_failures.rb +46 -0
  72. data/lib/samagotchi/guardrails/protected_paths.rb +77 -0
  73. data/lib/samagotchi/guardrails/rules.rb +199 -0
  74. data/lib/samagotchi/guardrails/targets.rb +119 -0
  75. data/lib/samagotchi/guardrails/verdict.rb +134 -0
  76. data/lib/samagotchi/guardrails.rb +18 -0
  77. data/lib/samagotchi/hooks/bundle_loader.rb +158 -0
  78. data/lib/samagotchi/hooks/loader.rb +162 -0
  79. data/lib/samagotchi/hooks/registry.rb +261 -0
  80. data/lib/samagotchi/hooks.rb +30 -0
  81. data/lib/samagotchi/host_registry.rb +315 -0
  82. data/lib/samagotchi/idle_client.rb +147 -0
  83. data/lib/samagotchi/idle_recap.rb +549 -0
  84. data/lib/samagotchi/idle_reminders.rb +101 -0
  85. data/lib/samagotchi/idle_scheduler.rb +76 -0
  86. data/lib/samagotchi/image_store.rb +393 -0
  87. data/lib/samagotchi/installed_gem.rb +38 -0
  88. data/lib/samagotchi/kernel_loop.rb +1017 -0
  89. data/lib/samagotchi/launch_mode.rb +34 -0
  90. data/lib/samagotchi/llm/backend.rb +28 -0
  91. data/lib/samagotchi/llm/chat_loop.rb +450 -0
  92. data/lib/samagotchi/llm/errors.rb +329 -0
  93. data/lib/samagotchi/llm/http.rb +412 -0
  94. data/lib/samagotchi/llm/model_result.rb +72 -0
  95. data/lib/samagotchi/llm/native_backend.rb +50 -0
  96. data/lib/samagotchi/llm/native_tool_normalizer.rb +277 -0
  97. data/lib/samagotchi/llm/openai_chat.rb +403 -0
  98. data/lib/samagotchi/llm/usage.rb +79 -0
  99. data/lib/samagotchi/log.rb +200 -0
  100. data/lib/samagotchi/log_line.rb +127 -0
  101. data/lib/samagotchi/log_path.rb +31 -0
  102. data/lib/samagotchi/log_subscriber.rb +163 -0
  103. data/lib/samagotchi/memory_bundle/builder.rb +364 -0
  104. data/lib/samagotchi/memory_bundle/index_updater.rb +123 -0
  105. data/lib/samagotchi/memory_bundle/installer.rb +528 -0
  106. data/lib/samagotchi/memory_bundle/listing.rb +72 -0
  107. data/lib/samagotchi/memory_bundle/manifest.rb +225 -0
  108. data/lib/samagotchi/memory_bundle/merger.rb +52 -0
  109. data/lib/samagotchi/memory_bundle/placeholder.rb +37 -0
  110. data/lib/samagotchi/memory_bundle/provenance.rb +257 -0
  111. data/lib/samagotchi/memory_bundle/source.rb +153 -0
  112. data/lib/samagotchi/memory_bundle/status.rb +107 -0
  113. data/lib/samagotchi/memory_bundle/system_bundle.rb +161 -0
  114. data/lib/samagotchi/memory_bundle/uninstaller.rb +128 -0
  115. data/lib/samagotchi/memory_bundle.rb +17 -0
  116. data/lib/samagotchi/memory_paths.rb +101 -0
  117. data/lib/samagotchi/model_overlay.rb +53 -0
  118. data/lib/samagotchi/model_profile.rb +309 -0
  119. data/lib/samagotchi/muted_memories.rb +66 -0
  120. data/lib/samagotchi/note_command.rb +163 -0
  121. data/lib/samagotchi/output_formatter.rb +100 -0
  122. data/lib/samagotchi/owner_lock.rb +110 -0
  123. data/lib/samagotchi/pending_input_queue.rb +48 -0
  124. data/lib/samagotchi/plugin/api.rb +362 -0
  125. data/lib/samagotchi/plugin/context.rb +193 -0
  126. data/lib/samagotchi/plugin/loader.rb +126 -0
  127. data/lib/samagotchi/plugin/service.rb +117 -0
  128. data/lib/samagotchi/plugin/sessions.rb +150 -0
  129. data/lib/samagotchi/plugin/side_question.rb +60 -0
  130. data/lib/samagotchi/plugin/tool_result.rb +24 -0
  131. data/lib/samagotchi/project_scope.rb +25 -0
  132. data/lib/samagotchi/prompt.rb +119 -0
  133. data/lib/samagotchi/prompt_literal_guard.rb +70 -0
  134. data/lib/samagotchi/recap_store.rb +92 -0
  135. data/lib/samagotchi/reminder_store.rb +165 -0
  136. data/lib/samagotchi/self_report.rb +195 -0
  137. data/lib/samagotchi/send_command.rb +170 -0
  138. data/lib/samagotchi/served_model.rb +32 -0
  139. data/lib/samagotchi/session.rb +508 -0
  140. data/lib/samagotchi/session_commands.rb +527 -0
  141. data/lib/samagotchi/session_delete_command.rb +105 -0
  142. data/lib/samagotchi/session_manager.rb +1049 -0
  143. data/lib/samagotchi/session_metrics.rb +466 -0
  144. data/lib/samagotchi/session_observer.rb +117 -0
  145. data/lib/samagotchi/terminal_ui/attach_launcher.rb +118 -0
  146. data/lib/samagotchi/terminal_ui/attached_loop.rb +1037 -0
  147. data/lib/samagotchi/terminal_ui/attached_view.rb +264 -0
  148. data/lib/samagotchi/terminal_ui/event_renderer.rb +192 -0
  149. data/lib/samagotchi/terminal_ui/formatting.rb +291 -0
  150. data/lib/samagotchi/terminal_ui/image_input.rb +36 -0
  151. data/lib/samagotchi/terminal_ui/input_support.rb +324 -0
  152. data/lib/samagotchi/terminal_ui/legacy_surface.rb +111 -0
  153. data/lib/samagotchi/terminal_ui/line_reader.rb +113 -0
  154. data/lib/samagotchi/terminal_ui/live_region.rb +36 -0
  155. data/lib/samagotchi/terminal_ui/plain_surface.rb +51 -0
  156. data/lib/samagotchi/terminal_ui/question_prompt.rb +153 -0
  157. data/lib/samagotchi/terminal_ui/question_slot.rb +131 -0
  158. data/lib/samagotchi/terminal_ui/reline_seam.rb +216 -0
  159. data/lib/samagotchi/terminal_ui/repl_input.rb +138 -0
  160. data/lib/samagotchi/terminal_ui/screen.rb +316 -0
  161. data/lib/samagotchi/terminal_ui/surface.rb +47 -0
  162. data/lib/samagotchi/terminal_ui/thinking_line.rb +101 -0
  163. data/lib/samagotchi/terminal_ui.rb +1992 -0
  164. data/lib/samagotchi/thinking_ticker.rb +110 -0
  165. data/lib/samagotchi/thought_stream_splitter.rb +149 -0
  166. data/lib/samagotchi/token_usage.rb +88 -0
  167. data/lib/samagotchi/tool_activity.rb +216 -0
  168. data/lib/samagotchi/tool_call_parser.rb +637 -0
  169. data/lib/samagotchi/tool_declarations.rb +561 -0
  170. data/lib/samagotchi/tool_runner.rb +211 -0
  171. data/lib/samagotchi/tools/args.rb +259 -0
  172. data/lib/samagotchi/tools/ask_user_question.rb +152 -0
  173. data/lib/samagotchi/tools/builtins.rb +122 -0
  174. data/lib/samagotchi/tools/cancel_reminder.rb +21 -0
  175. data/lib/samagotchi/tools/delegate.rb +167 -0
  176. data/lib/samagotchi/tools/delegate_result.rb +53 -0
  177. data/lib/samagotchi/tools/delegate_wait.rb +153 -0
  178. data/lib/samagotchi/tools/edit.rb +155 -0
  179. data/lib/samagotchi/tools/execute.rb +214 -0
  180. data/lib/samagotchi/tools/list_reminders.rb +20 -0
  181. data/lib/samagotchi/tools/list_sessions.rb +74 -0
  182. data/lib/samagotchi/tools/memory.rb +256 -0
  183. data/lib/samagotchi/tools/output_guardrails.rb +93 -0
  184. data/lib/samagotchi/tools/peers.rb +18 -0
  185. data/lib/samagotchi/tools/read.rb +182 -0
  186. data/lib/samagotchi/tools/register_reminder.rb +53 -0
  187. data/lib/samagotchi/tools/registry.rb +60 -0
  188. data/lib/samagotchi/tools/send_note.rb +49 -0
  189. data/lib/samagotchi/tools/task_create.rb +29 -0
  190. data/lib/samagotchi/tools/task_get.rb +39 -0
  191. data/lib/samagotchi/tools/task_list.rb +43 -0
  192. data/lib/samagotchi/tools/task_runtime.rb +311 -0
  193. data/lib/samagotchi/tools/task_stop.rb +29 -0
  194. data/lib/samagotchi/tools/task_wait.rb +104 -0
  195. data/lib/samagotchi/tools/tool_path.rb +18 -0
  196. data/lib/samagotchi/tools/web_fetch.rb +163 -0
  197. data/lib/samagotchi/tools/write.rb +26 -0
  198. data/lib/samagotchi/turn_flow.rb +242 -0
  199. data/lib/samagotchi/turn_note.rb +76 -0
  200. data/lib/samagotchi/turn_tally.rb +101 -0
  201. data/lib/samagotchi/version.rb +7 -0
  202. data/lib/samagotchi/vision_context.rb +132 -0
  203. data/lib/samagotchi/vision_support.rb +109 -0
  204. data/lib/samagotchi/web/app.rb +1349 -0
  205. data/lib/samagotchi/web/markdown_renderer.rb +107 -0
  206. data/lib/samagotchi/web/message_parts.rb +169 -0
  207. data/lib/samagotchi/web/public/activity.js +100 -0
  208. data/lib/samagotchi/web/public/annotations.js +67 -0
  209. data/lib/samagotchi/web/public/app.js +2382 -0
  210. data/lib/samagotchi/web/public/card.js +74 -0
  211. data/lib/samagotchi/web/public/chat_view.js +360 -0
  212. data/lib/samagotchi/web/public/chunk_router.js +25 -0
  213. data/lib/samagotchi/web/public/command_complete.js +39 -0
  214. data/lib/samagotchi/web/public/composer_size.js +19 -0
  215. data/lib/samagotchi/web/public/copy.js +142 -0
  216. data/lib/samagotchi/web/public/ctx.js +35 -0
  217. data/lib/samagotchi/web/public/data.js +256 -0
  218. data/lib/samagotchi/web/public/format.js +232 -0
  219. data/lib/samagotchi/web/public/hold.js +78 -0
  220. data/lib/samagotchi/web/public/images.js +77 -0
  221. data/lib/samagotchi/web/public/index.html +568 -0
  222. data/lib/samagotchi/web/public/init_row.js +60 -0
  223. data/lib/samagotchi/web/public/model_pick.js +23 -0
  224. data/lib/samagotchi/web/public/question_card.js +100 -0
  225. data/lib/samagotchi/web/public/route.js +17 -0
  226. data/lib/samagotchi/web/public/scope.js +36 -0
  227. data/lib/samagotchi/web/public/scroll.js +24 -0
  228. data/lib/samagotchi/web/public/sentences.js +88 -0
  229. data/lib/samagotchi/web/public/sessions_list.js +60 -0
  230. data/lib/samagotchi/web/public/strip.js +25 -0
  231. data/lib/samagotchi/web/public/tally.js +37 -0
  232. data/lib/samagotchi/web/public/thinking_ticker.js +79 -0
  233. data/lib/samagotchi/web/public/timing.js +185 -0
  234. data/lib/samagotchi/web/public/turn_events.js +209 -0
  235. data/lib/samagotchi/web/public/turn_model.js +204 -0
  236. data/lib/samagotchi/web/public/turn_view.js +587 -0
  237. data/lib/samagotchi/web/server.rb +183 -0
  238. data/lib/samagotchi/web/session_hub.rb +329 -0
  239. data/lib/samagotchi/web/session_summary.rb +85 -0
  240. data/lib/samagotchi/worker.rb +635 -0
  241. data/lib/samagotchi/worker_idle_exit.rb +87 -0
  242. data/lib/samagotchi.rb +12 -0
  243. metadata +374 -0
@@ -0,0 +1,1017 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative "model_profile"
4
+ require_relative "tool_call_parser"
5
+ require_relative "config"
6
+ require_relative "context_usage"
7
+ require_relative "context_window"
8
+ require_relative "prompt"
9
+ require_relative "prompt_literal_guard"
10
+ require_relative "client"
11
+ require_relative "llm/errors"
12
+ require_relative "log"
13
+ require_relative "hooks"
14
+ require_relative "pending_input_queue"
15
+ require_relative "thought_stream_splitter"
16
+ require_relative "tools/builtins"
17
+ require_relative "muted_memories"
18
+ require_relative "tool_activity"
19
+ require_relative "tool_runner"
20
+
21
+ module Samagotchi
22
+ # The KernelLoop drives the model ↔ tool interaction cycle.
23
+ #
24
+ # Flow:
25
+ # 1. Format the conversation using the active profile and call llama.cpp.
26
+ # 2. Parse the response for tool-call blocks in the profile's format.
27
+ # 3. Dispatch each tool call, collect results.
28
+ # 4. Inject results as a tool_response message and repeat from step 1.
29
+ # 5. Stop when the model emits no tool calls or max_iterations is reached.
30
+ #
31
+ # Supports multiple model profiles:
32
+ # - Gemma 4: <|tool_call>call:NAME{params}<tool_call|>
33
+ # - Qwen 3.6: <tool_call><function=NAME><parameter=KEY>VALUE</parameter></function></tool_call>
34
+ class KernelLoop
35
+ Result = Struct.new(:output, :conversation, :exhausted, :pending_tool_calls, :tool_activity, :canceled, :cancellation_reason, :context_status, keyword_init: true) do
36
+ def to_s
37
+ output.to_s
38
+ end
39
+
40
+ alias to_str to_s
41
+
42
+ def ==(other)
43
+ if other.is_a?(self.class)
44
+ super
45
+ else
46
+ to_s == other
47
+ end
48
+ end
49
+
50
+ def exhausted?
51
+ exhausted
52
+ end
53
+
54
+ def pending_tool_calls?
55
+ pending_tool_calls
56
+ end
57
+
58
+ def resumable?
59
+ exhausted? && pending_tool_calls?
60
+ end
61
+
62
+ def canceled?
63
+ canceled
64
+ end
65
+
66
+ # Struct/Enumerable defines include? with collection semantics, but the
67
+ # historical KernelLoop#run contract returned a String. Keep include?
68
+ # aligned with String#include? for backward compatibility.
69
+ def include?(needle)
70
+ to_s.include?(needle)
71
+ end
72
+
73
+ # Keep compatibility with existing callers/specs that treat run() as a
74
+ # plain string (e.g., include?, match, start_with?).
75
+ def method_missing(name, *args, &block)
76
+ return to_s.public_send(name, *args, &block) if to_s.respond_to?(name)
77
+
78
+ super
79
+ end
80
+
81
+ def respond_to_missing?(name, include_private = false)
82
+ to_s.respond_to?(name, include_private) || super
83
+ end
84
+ end
85
+
86
+ # The built-in tool classes (Tools::Builtins registers them).
87
+ TOOLS = Tools::Builtins::CLASSES
88
+
89
+ # What a tool handler (call, kctx) gets from the kernel: the reminder
90
+ # store, the peers, the model key, and the memory read and ask-user
91
+ # flows that need its state.
92
+ class ToolContext
93
+ def initialize(kernel)
94
+ @kernel = kernel
95
+ end
96
+
97
+ def reminder_store = @kernel.reminder_store
98
+ def peers = @kernel.peers
99
+ def model_key = @kernel.model_key
100
+ def muted_memory_read(call) = @kernel.__send__(:muted_memory_read, Tools::MemoryRead, call)
101
+ def ask_user_question(call) = @kernel.__send__(:handle_ask_user_question, call)
102
+ end
103
+
104
+ CONTEXT_STATUS_PREFIX = "CONTEXT_STATUS"
105
+ # The model's own line about its context (a tail system message, kind
106
+ # CONTEXT_LINE_KIND), left once per rise into a bucket whose guidance
107
+ # asks for a change: from the second threshold (40% by default) up.
108
+ CONTEXT_LINE_PREFIX = "[CONTEXT: "
109
+ CONTEXT_LINE_KIND = "context"
110
+ CONTEXT_GUIDANCE_FROM_RANK = 2
111
+ CONTEXT_STATUS_ENABLED_ENV = "SAMAGOTCHI_CONTEXT_STATUS"
112
+ CONTEXT_CHARS_PER_TOKEN_ENV = "SAMAGOTCHI_CONTEXT_CHARS_PER_TOKEN"
113
+ CONTEXT_THRESHOLDS_ENV = "SAMAGOTCHI_CONTEXT_STATUS_THRESHOLDS"
114
+ CONTEXT_CADENCE_ENV = "SAMAGOTCHI_CONTEXT_STATUS_CADENCE"
115
+
116
+ DEFAULT_CONTEXT_CHARS_PER_TOKEN = 4.0
117
+ DEFAULT_CONTEXT_THRESHOLDS = [20, 40, 60, 80].freeze
118
+ DEFAULT_CONTEXT_CADENCE = 0
119
+ DEFAULT_MAX_TOOL_OUTPUT_CHARS = 10_000
120
+ TOOL_OUTPUT_CHARS_ENV = "SAMAGOTCHI_MAX_TOOL_OUTPUT_CHARS"
121
+ QWEN_INCOMPLETE_TOOL_CALL_RECOVERY_LIMIT = 2
122
+ QWEN_INCOMPLETE_TOOL_CALL_RECOVERY_PROMPT = "Continue the previous assistant message by finishing the open <tool_call> XML block. Output only the remaining XML needed to complete the tool call."
123
+
124
+ # @param tools [Tools::Registry, nil] the tools calls dispatch to (the
125
+ # Engine's; nil: the built-ins alone)
126
+ def initialize(client: nil, profile: nil, model_name: nil, no_interrupt: false, hooks: nil, reminder_store: nil, model_key: nil,
127
+ tools: nil)
128
+ @client = client || Client.new
129
+ @tools = tools || Tools::Builtins.default
130
+ @no_interrupt = no_interrupt
131
+ resolved_model_name = ModelProfile.required_model_name(model_name)
132
+ # The resolved model id actually used for this run (per-run override wins
133
+ # over the config alias); on every debug dump so we can see exactly
134
+ # which model each request went to. The Engine sets it per turn too:
135
+ # the chat loop dispatches tools here without going through #run.
136
+ @current_model_name = resolved_model_name
137
+ @profile = profile ? ModelProfile.normalize(profile) : ModelProfile.from_model_name(resolved_model_name)
138
+ # Where @profile came from, as /stats shows it (a Resolution's label
139
+ # once the Engine resolves one, see #use_profile!).
140
+ @profile_source = profile ? "given" : "name"
141
+ @hooks = hooks
142
+ @reminder_store = reminder_store
143
+ @model_key = model_key
144
+ end
145
+
146
+ # @return [ReminderStore, nil] the reminder store for inspection (used by
147
+ # Engine to share the same store with the KernelLoop when TerminalUI
148
+ # creates both).
149
+ attr_reader :reminder_store
150
+
151
+ # @return [ModelProfile] the active prompt profile
152
+ attr_reader :profile
153
+
154
+ # @return [Samagotchi::Hooks::Registry, nil] hooks registry shared with Engine.
155
+ # Engine owns the registry; KernelLoop only fires events. Accessor allows
156
+ # Engine to propagate its registry to an externally-created kernel (TUI path).
157
+ attr_accessor :hooks
158
+ # @return [Tools::Registry] the tools #dispatch runs. The Engine sets
159
+ # its own on a kernel built before it (the REPL's), as with hooks.
160
+ attr_accessor :tools
161
+ attr_accessor :client
162
+ attr_accessor :model_key
163
+ attr_accessor :current_model_name
164
+ # @return [Array<String>, nil] the session's muted memories (normalized
165
+ # names, see MutedMemories); memory_read refuses them. The Engine sets it.
166
+ attr_accessor :muted_memory_names
167
+ # @return [Proc, nil] answers ask_user_question (payload → answer string);
168
+ # Engine sets it to its blocking request_question.
169
+ attr_accessor :question_handler
170
+ # The Guardrails::Gate ToolRunner asks before each call; the Engine sets
171
+ # it (nil: ToolRunner's own, hooks only).
172
+ attr_accessor :guardrail_gate
173
+ # The turn's VisionContext (images: capability, files, limits), set by
174
+ # the Engine per turn; nil sends no images (placeholders instead).
175
+ attr_accessor :vision
176
+ # Tools::Peers (or the Engine's live view of it): the session
177
+ # list_sessions and send_note speak for; nil outside a session.
178
+ attr_accessor :peers
179
+
180
+ # Run the conversation loop and return the final model response plus
181
+ # resumable conversation state when execution stops at max_iterations.
182
+ #
183
+ # @param messages [Array<Hash>, Result] conversation so far ({role:, content:})
184
+ # or a previous Result to resume
185
+ # @param max_iterations [Integer] safety cap on tool-call rounds
186
+ # @param on_stream_event [Proc, nil] optional callback for generation events
187
+ # @param cancel_controller [CancellationController, nil] optional cancellation source
188
+ # @param model_name [String, nil] optional per-run model override
189
+ # @param max_tool_output_chars [Integer, nil] per-output char cap for the
190
+ # :tool_call_completed event's `output:` (nil → env/DEFAULT_MAX_TOOL_OUTPUT_CHARS)
191
+ # @param pending_input [#call, nil] optional drain proc returning
192
+ # Array<String> of user steering messages queued while the turn runs.
193
+ # Drained at iteration boundaries (llama.cpp's /completion cannot accept
194
+ # steering mid-stream); drained lines merge into ONE user message appended
195
+ # at the conversation tail (prefix KV cache preserved) and a
196
+ # :pending_input_merged stream event is emitted.
197
+ # @return [Result] final visible response with continuation metadata
198
+ def run(messages, max_iterations: 100, on_stream_event: nil, cancel_controller: nil, model_name: nil, max_tool_output_chars: nil, pending_input: nil)
199
+ resolved_model_name = completion_model_name(model_name)
200
+ @current_model_name = resolved_model_name
201
+
202
+ conversation = prepare_conversation(messages)
203
+ context_state = initial_context_status_state(conversation)
204
+ exhausted = false
205
+ pending_tool_calls = false
206
+ tool_activity = []
207
+ qwen_recovery_attempts = 0
208
+ qwen_partial_tool_call = nil
209
+ context_status = nil
210
+ stream_splitter = ThoughtStreamSplitter.for_profile(@profile)
211
+ partial_assistant_buffer = +""
212
+
213
+ effective_max_iterations = @no_interrupt ? 1000 : max_iterations
214
+ effective_max_tool_output_chars = resolve_output_char_cap(max_tool_output_chars)
215
+ effective_max_iterations.times do |iteration_index|
216
+ inject_pending_input!(conversation, pending_input, on_stream_event, iteration_index + 1, cancel_controller)
217
+ prompt, images = Prompt.format_with_images(conversation, profile: @profile, vision: @vision)
218
+ image_tokens = images.empty? ? 0 : ImagePlan.estimated_tokens(conversation)
219
+ context_window = ContextWindow.resolve(client: @client, model: resolved_model_name)
220
+ context_status = emit_context_status_event(on_stream_event, prompt, iteration_index: iteration_index, state: context_state, window: context_window,
221
+ image_tokens: image_tokens) || context_status
222
+ if (line = context_state.delete(:guidance))
223
+ # The model's own copy, on the tail (the prompt cache keeps its
224
+ # prefix), then the prompt again with it.
225
+ conversation << line
226
+ prompt, images = Prompt.format_with_images(conversation, profile: @profile, vision: @vision)
227
+ end
228
+ emit_stream_event(
229
+ on_stream_event,
230
+ type: :generation_started,
231
+ iteration: iteration_index + 1,
232
+ context_window_tokens: context_window.tokens,
233
+ context_window_source: context_window.source,
234
+ profile: @profile.name,
235
+ profile_source: @profile_source
236
+ )
237
+ served_model = nil
238
+ # This generation's own server counts (the run-long
239
+ # context_state[:server_usage] can hold an earlier one's).
240
+ generation_usage = nil
241
+ # The thinking this generation streamed, for the log (a stuck
242
+ # thinking generation shows as thinking_chars=N content_length=…).
243
+ streamed_thinking = 0
244
+ # Fire :before_generation hook
245
+ gen_event = { type: :before_generation, iteration: iteration_index + 1 }
246
+ fire_hook(:before_generation, gen_event) if @hooks
247
+ response = @client.complete(
248
+ prompt,
249
+ **complete_kwargs(
250
+ cancel_controller: cancel_controller,
251
+ model_name: resolved_model_name,
252
+ on_chunk: lambda { |chunk|
253
+ generation_usage = capture_server_usage(chunk[:payload], context_state) || generation_usage
254
+ # llama.cpp names the loaded model in the stream's last payload.
255
+ named = chunk[:payload]["model"] if chunk[:payload].is_a?(Hash)
256
+ served_model = named if named.is_a?(String) && !named.strip.empty?
257
+ split = stream_splitter.feed(chunk[:content])
258
+ partial_assistant_buffer << split[:text]
259
+ streamed_thinking += split[:thinking].to_s.length
260
+ if on_stream_event
261
+ emit_stream_event(
262
+ on_stream_event,
263
+ type: :generation_chunk,
264
+ iteration: iteration_index + 1,
265
+ content: chunk[:content],
266
+ thinking: split[:thinking],
267
+ payload: chunk[:payload]
268
+ )
269
+ end
270
+ },
271
+ on_retry: lambda { |retry_event|
272
+ # The retry streams from the start: its counts replace these.
273
+ generation_usage = nil
274
+ streamed_thinking = 0
275
+ next unless on_stream_event
276
+
277
+ emit_stream_event(
278
+ on_stream_event,
279
+ {
280
+ type: :generation_retrying,
281
+ iteration: iteration_index + 1
282
+ }.merge(retry_event)
283
+ )
284
+ },
285
+ images: images
286
+ )
287
+ )
288
+ refresh_context_display(context_state, generation_usage, context_window)
289
+ emit_stream_event(
290
+ on_stream_event,
291
+ type: :generation_completed,
292
+ iteration: iteration_index + 1,
293
+ content_length: response.to_s.length,
294
+ thinking_chars: thinking_chars(response.to_s, streamed_thinking),
295
+ served_model: served_model,
296
+ requested_model: resolved_model_name
297
+ )
298
+ dump_log("response", response, iteration: iteration_index + 1)
299
+ # Fire :after_generation hook (after LLM returns, before tool parse),
300
+ # with a read-only copy of the conversation as sent.
301
+ after_gen_event = { type: :after_generation, iteration: iteration_index + 1, response: response,
302
+ messages: conversation.map(&:dup).freeze }
303
+ fire_hook(:after_generation, after_gen_event) if @hooks
304
+ conversation << { role: "model", content: response }
305
+
306
+ # Profile-specific parse (incl. Qwen unterminated-block recovery); the
307
+ # returned fragment (non-nil only for Qwen) is fed back on the next
308
+ # iteration if the model opened a tool-call block it did not close.
309
+ calls, qwen_partial_tool_call = parser.parse_with_recovery(response, qwen_partial_tool_call)
310
+ calls = calls.map do |call|
311
+ PromptLiteralGuard.restore_call(call, profile: @profile)
312
+ end
313
+ qwen_incomplete_tool_call = !qwen_partial_tool_call.nil?
314
+
315
+ if calls.empty?
316
+ if qwen_incomplete_tool_call && qwen_recovery_attempts < QWEN_INCOMPLETE_TOOL_CALL_RECOVERY_LIMIT
317
+ qwen_recovery_attempts += 1
318
+ conversation << { role: "user", content: QWEN_INCOMPLETE_TOOL_CALL_RECOVERY_PROMPT, preserve_literals: true }
319
+ pending_tool_calls = false
320
+ next
321
+ end
322
+
323
+ pending_tool_calls = false
324
+ answer = -> { PromptLiteralGuard.restore(strip_thought_blocks(response), profile: @profile) }
325
+ unless inject_pending_input!(conversation, pending_input, on_stream_event, iteration_index + 1, cancel_controller, answer: answer)
326
+ break
327
+ end
328
+ # Queued steering keeps the turn going: loop again so the model
329
+ # answers the injected message instead of stopping here.
330
+ next
331
+ end
332
+
333
+ qwen_recovery_attempts = 0
334
+ qwen_partial_tool_call = nil
335
+
336
+ emit_stream_event(on_stream_event, type: :tool_dispatch_started, iteration: iteration_index + 1, call_count: calls.length)
337
+ tool_images = []
338
+ image_counts = []
339
+ shown_params = []
340
+ shown_labels = []
341
+ results = calls.map.with_index do |call, call_index|
342
+ run = tool_runner.run(call, iteration: iteration_index + 1, call_index: call_index + 1,
343
+ call_count: calls.length, on_stream_event: on_stream_event,
344
+ max_tool_output_chars: effective_max_tool_output_chars)
345
+ tool_activity << run[:activity]
346
+ tool_images.concat(Array(run[:images]))
347
+ image_counts << Array(run[:images]).size
348
+ shown_params << run[:shown_params]
349
+ shown_labels << run[:shown_label]
350
+ run[:output]
351
+ end.join("\n\n---\n\n")
352
+ emit_stream_event(on_stream_event, type: :tool_dispatch_completed, iteration: iteration_index + 1, call_count: calls.length)
353
+ # The joined results carry every call's images, in call order.
354
+ tool_response = { role: "tool_response", content: results }
355
+ tool_response[:images] = tool_images unless tool_images.empty?
356
+ # How many of them each call returned, in call order, so the web's
357
+ # reload puts each on its own tool row; the prompt never reads it.
358
+ tool_response[:image_counts] = image_counts unless tool_images.empty?
359
+ # A plugin tool's params line, one per call in call order (nil for
360
+ # a built-in), for the web's reload; the prompt never reads it.
361
+ tool_response[:tool_params] = shown_params if shown_params.any?
362
+ tool_response[:tool_labels] = shown_labels if shown_labels.any?
363
+ conversation << tool_response
364
+ pending_tool_calls = true
365
+ rescue Client::RequestCancelled => e
366
+ emit_stream_event(
367
+ on_stream_event,
368
+ type: :generation_cancelled,
369
+ iteration: iteration_index + 1,
370
+ reason: e.reason
371
+ )
372
+ return cancelled_result(conversation, tool_activity: tool_activity, reason: e.reason, partial_assistant_text: partial_assistant_buffer)
373
+ end
374
+
375
+ if pending_tool_calls && tool_response_turn?(conversation.last)
376
+ exhausted = true
377
+ end
378
+
379
+ output = strip_thought_blocks(last_model_content(conversation))
380
+ # A turn stopped at the limit ends on a call it never ran: show only its text.
381
+ output = parser.strip_tool_calls(output) if exhausted
382
+ Result.new(
383
+ output: PromptLiteralGuard.restore(output, profile: @profile),
384
+ conversation: duplicate_conversation(conversation),
385
+ exhausted: exhausted,
386
+ pending_tool_calls: pending_tool_calls,
387
+ tool_activity: tool_activity,
388
+ canceled: false,
389
+ cancellation_reason: nil,
390
+ context_status: context_state[:display] || context_status
391
+ )
392
+ rescue StandardError => e
393
+ LLM::FailedTurn.attach(e, conversation && duplicate_conversation(conversation))
394
+ raise
395
+ end
396
+
397
+ # Use a resolved profile (ModelProfile::Resolution) from now on,
398
+ # whatever model name later runs carry.
399
+ def use_profile!(resolution)
400
+ @profile = resolution.profile
401
+ @profile_source = resolution.label
402
+ end
403
+
404
+ def sync_model_key!(key)
405
+ @model_key = key
406
+ end
407
+
408
+ private
409
+
410
+ # memory_read with the session's mutes applied: a blank name (the index)
411
+ # loses the muted memories' lines; a muted name in a comma list is
412
+ # refused with its own error line and the rest is read as usual.
413
+ def muted_memory_read(tool, call)
414
+ muted = Array(@muted_memory_names)
415
+ content = call[:content].to_s
416
+ read = ->(names) { tool.call(names, scope: call[:scope], model_key: @model_key) }
417
+ return read.call(content) if muted.empty?
418
+ return MutedMemories.filter_index(read.call(content), muted) if content.strip.empty?
419
+
420
+ names = Tools::MemoryRead.parse_names(content)
421
+ refused, allowed = names.partition { |name| MutedMemories.muted?(name, muted) }
422
+ return read.call(content) if refused.empty?
423
+
424
+ errors = refused.map { |name| "Error: memory '#{name}' is muted for this session" }
425
+ return errors.join("\n") if allowed.empty?
426
+
427
+ [read.call(allowed.join(",")), *errors].join(Tools::MemoryRead::SEPARATOR)
428
+ end
429
+
430
+ def emit_stream_event(callback, event)
431
+ callback&.call(event)
432
+ rescue StandardError
433
+ nil
434
+ end
435
+
436
+ # Drain the pending input queue (if any) and, when messages are waiting,
437
+ # append them as ONE merged user message at the conversation tail and emit
438
+ # :pending_input_merged. Tail-append only: head mutation would invalidate
439
+ # the server-side prefix KV cache. Returns true when a message was injected.
440
+ # After a cancel the input stays queued: it runs as the next turn instead
441
+ # of dying with this one. +answer+ (a proc, called only on a merge) is the
442
+ # answer the merge follows: the UIs show it, the turn summary has only the
443
+ # last one.
444
+ def inject_pending_input!(conversation, pending_input, on_stream_event, iteration, cancel_controller = nil, answer: nil)
445
+ return false unless pending_input
446
+ return false if cancel_controller&.cancelled?
447
+
448
+ lines = begin
449
+ pending_input.call
450
+ rescue StandardError
451
+ nil
452
+ end
453
+ return false if lines.nil? || lines.empty?
454
+
455
+ content = lines.map { |line| line.to_s.strip }.reject(&:empty?).join("\n\n")
456
+ return false if content.empty?
457
+
458
+ answer = answer.call.to_s if answer
459
+ conversation << { role: "user", content: content }
460
+ emit_stream_event(
461
+ on_stream_event,
462
+ type: :pending_input_merged,
463
+ iteration: iteration,
464
+ count: lines.length,
465
+ content: content,
466
+ answer: answer.to_s.strip.empty? ? nil : answer
467
+ )
468
+ true
469
+ end
470
+
471
+ # ── Hook dispatch helper ───────────────────────────────────────────────────
472
+
473
+ # Fire a named hook on the registry (if present).
474
+ # Hooks are dispatched synchronously; the event hash is passed by reference
475
+ # so hooks can mutate fields (e.g. :before_tool_call can modify :call).
476
+ def tool_runner
477
+ @tool_runner ||= ToolRunner.new(self)
478
+ end
479
+
480
+ def fire_hook(name, event)
481
+ return unless @hooks
482
+ @hooks.fire(name, event)
483
+ rescue StandardError
484
+ # A failing hook must not break the turn.
485
+ end
486
+
487
+ # Coerce a value to boolean — handles true/false, nil, and string "true"/"false".
488
+ def truthy?(val) = Tools::Builtins.truthy?(val)
489
+
490
+ def tool_context = @tool_context ||= ToolContext.new(self)
491
+
492
+ # ── Output char cap resolution ─────────────────────────────────────────────
493
+
494
+ # Resolve the per-output character cap for the emitted tool call events.
495
+ #
496
+ # Precedence: an explicit override wins, then the SAMAGOTCHI_MAX_TOOL_OUTPUT_CHARS
497
+ # env var, then DEFAULT_MAX_TOOL_OUTPUT_CHARS. A non-positive value falls back
498
+ # to the default (there is intentionally no "unlimited" — live UIs get a
499
+ # bounded `output:` plus a truthful `output_truncated:` flag).
500
+ # Class-level so the chat loop resolves it the same way.
501
+ def self.resolve_output_char_cap(override)
502
+ cfg_val = begin
503
+ v = Samagotchi::Config.get("max_tool_output_chars") rescue nil
504
+ v.to_i if v
505
+ end
506
+ value = override || cfg_val || ENV[TOOL_OUTPUT_CHARS_ENV]
507
+ parsed = value.to_i
508
+ parsed.positive? ? parsed : DEFAULT_MAX_TOOL_OUTPUT_CHARS
509
+ end
510
+
511
+ def resolve_output_char_cap(override)
512
+ self.class.resolve_output_char_cap(override)
513
+ end
514
+
515
+ def complete_kwargs(cancel_controller:, model_name: nil, on_chunk: nil, on_retry: nil, images: [])
516
+ kwargs = {}
517
+ # Only a request with images names them: a text-only call is unchanged.
518
+ kwargs[:images] = images unless images.empty?
519
+ kwargs[:on_chunk] = on_chunk if on_chunk
520
+ kwargs[:on_retry] = on_retry if on_retry && client_supports_keyword?(:on_retry)
521
+ kwargs[:cancel_controller] = cancel_controller if cancel_controller && client_supports_keyword?(:cancel_controller)
522
+ kwargs[:stop] = @profile.stop_sequences if client_supports_keyword?(:stop)
523
+ n_predict = completion_n_predict
524
+ kwargs[:n_predict] = n_predict if n_predict && client_supports_keyword?(:n_predict)
525
+ resolved_model_name = completion_model_name(model_name)
526
+ kwargs[:model] = resolved_model_name if resolved_model_name && client_supports_keyword?(:model)
527
+ kwargs
528
+ end
529
+
530
+ def completion_n_predict
531
+ v = Samagotchi::Config.get("default.n_predict") rescue nil
532
+ v.to_i if v && v.to_i.positive?
533
+ end
534
+
535
+ def completion_model_name(override = nil)
536
+ ModelProfile.required_model_name(override)
537
+ end
538
+
539
+ def client_supports_keyword?(keyword)
540
+ @client_complete_keyword_support ||= {}
541
+ return @client_complete_keyword_support[keyword] if @client_complete_keyword_support.key?(keyword)
542
+
543
+ @client_complete_keyword_support[keyword] = begin
544
+ parameters = @client.method(:complete).parameters
545
+ parameters.any? { |kind, name| (kind == :key || kind == :keyreq) && name == keyword } ||
546
+ parameters.any? { |kind, _name| kind == :keyrest }
547
+ rescue StandardError
548
+ false
549
+ end
550
+ end
551
+
552
+ def cancelled_result(conversation, tool_activity:, reason:, partial_assistant_text: "")
553
+ partial = partial_assistant_text.to_s.strip
554
+ conversation = duplicate_conversation(conversation)
555
+ # Salvage the already-streamed visible reply (thought/tool_call lanes
556
+ # were never routed into the buffer, so unterminated tool_call fragments
557
+ # cannot leak) so a follow-up steering message continues with the model's
558
+ # half-finished work in context instead of losing it.
559
+ unless partial.empty?
560
+ conversation << { role: "model", content: "#{partial}\n[interrupted]", interrupted: true }
561
+ end
562
+ Result.new(
563
+ output: "",
564
+ conversation: conversation,
565
+ exhausted: false,
566
+ pending_tool_calls: false,
567
+ tool_activity: tool_activity,
568
+ canceled: true,
569
+ cancellation_reason: reason
570
+ )
571
+ end
572
+
573
+ # A payload dump (model response, tool call/result, context status): the
574
+ # debug level only, tagged with the model it came from.
575
+ def dump_log(event, payload, **fields)
576
+ return unless Log.level?(:debug)
577
+
578
+ Log.debug(:model, event, payload: payload, model: @current_model_name, **fields)
579
+ end
580
+
581
+ # Estimate context usage for this iteration's prompt and, when the emit
582
+ # gate fires, surface it to stream consumers as a :context_status event.
583
+ # A rise into a bucket that asks the model for a change also leaves a
584
+ # short line for it (state[:guidance], see context_guidance_message):
585
+ # not the telemetry, which the model no longer receives (it used to be injected as a
586
+ # synthetic system message); the returned {est_pct:, bucket:} hash feeds
587
+ # the Result's context_status for UI status lines (nil when not emitted).
588
+ def emit_context_status_event(on_stream_event, prompt, iteration_index:, state:, window: nil, image_tokens: 0)
589
+ return nil unless context_status_enabled?
590
+
591
+ usage = estimate_context_usage(prompt, server_usage: state[:server_usage], window: window, image_tokens: image_tokens)
592
+ bucket = context_status_bucket(usage[:estimated_pct])
593
+ # The status line's value, every iteration; the gate below decides
594
+ # only the event and the model's guidance line.
595
+ state[:display] = { est_pct: usage[:estimated_pct], bucket: bucket }
596
+ emit_status = should_emit_context_status?(state: state, bucket: bucket, iteration_index: iteration_index)
597
+ previous_bucket = state[:last_bucket]
598
+ state[:last_bucket] = bucket
599
+ return nil unless emit_status
600
+
601
+ state[:guidance] = context_guidance_message(usage: usage, bucket: bucket) if guidance_due?(previous_bucket, bucket)
602
+
603
+ status_message = context_status_message(usage: usage, bucket: bucket, source: usage[:source])
604
+ emit_stream_event(
605
+ on_stream_event,
606
+ type: :context_status,
607
+ iteration: iteration_index + 1,
608
+ status: status_message,
609
+ usage: usage,
610
+ bucket: bucket,
611
+ source: usage[:source]
612
+ )
613
+ dump_log("context_status", status_message, iteration: iteration_index + 1, bucket: bucket)
614
+ { est_pct: usage[:estimated_pct], bucket: bucket }
615
+ end
616
+
617
+ # @return [Hash, nil] the payload's normalized counts, when it has any
618
+ def capture_server_usage(payload, state)
619
+ normalized = ContextUsage.normalize(payload)
620
+ state[:server_usage] = normalized if normalized
621
+ normalized
622
+ end
623
+
624
+ # After a generation: the status line's value from what the server
625
+ # reported for it (prompt + answer), so a turn's value counts its last
626
+ # answer. Without counts the pre-generation estimate stays.
627
+ def refresh_context_display(state, usage, window)
628
+ return unless usage
629
+
630
+ display = context_display(used_tokens: usage[:total_tokens],
631
+ window_tokens: usage[:context_window_tokens] || window&.tokens)
632
+ state[:display] = display if display
633
+ end
634
+
635
+ def initial_context_status_state(conversation)
636
+ { last_bucket: extract_last_context_status_bucket(conversation) }
637
+ end
638
+
639
+ # The bucket of the last status line the conversation holds: the model's
640
+ # own line, or a legacy session's injected telemetry.
641
+ def extract_last_context_status_bucket(conversation)
642
+ message = conversation.reverse.find do |entry|
643
+ content = entry[:content].to_s
644
+ entry[:role] == "system" && (content.start_with?(CONTEXT_STATUS_PREFIX) || content.start_with?(CONTEXT_LINE_PREFIX))
645
+ end
646
+ return nil unless message
647
+
648
+ match = message[:content].match(/\bbucket=([a-z0-9_]+)/)
649
+ match && match[1]
650
+ end
651
+
652
+ def context_status_enabled?
653
+ cfg = begin Samagotchi::Config.get("context.status") rescue nil end
654
+ unless cfg.nil?
655
+ return !!cfg
656
+ end
657
+ value = ENV[CONTEXT_STATUS_ENABLED_ENV]
658
+ return true if value.nil?
659
+
660
+ !(value == "0" || value.casecmp?("false"))
661
+ end
662
+
663
+ # `window` is this iteration's ContextWindow::Resolved (resolved here when
664
+ # not given). A window the stream payload reports itself still wins.
665
+ # +image_tokens+: the images' estimate (their base64 is not in +prompt+).
666
+ def estimate_context_usage(prompt, server_usage: nil, window: nil, image_tokens: 0)
667
+ window ||= ContextWindow.resolve(client: @client, model: @current_model_name)
668
+ window_source = window.source
669
+ if server_usage && server_usage[:context_window_tokens]
670
+ window_source = :server
671
+ end
672
+
673
+ if server_usage && server_usage[:prompt_tokens]
674
+ window_tokens = server_usage[:context_window_tokens] || window.tokens
675
+ estimated_used_tokens = server_usage[:prompt_tokens]
676
+ estimated_remaining_tokens = [window_tokens - estimated_used_tokens, 0].max
677
+ estimated_pct = (estimated_used_tokens.to_f / window_tokens) * 100.0
678
+
679
+ return {
680
+ window_tokens: window_tokens,
681
+ window_source: window_source,
682
+ estimated_used_tokens: estimated_used_tokens,
683
+ estimated_remaining_tokens: estimated_remaining_tokens,
684
+ estimated_pct: estimated_pct,
685
+ source: "server"
686
+ }
687
+ end
688
+
689
+ window_tokens = window.tokens
690
+ estimated_used_tokens = (prompt.length / context_chars_per_token).ceil + image_tokens
691
+ estimated_remaining_tokens = [window_tokens - estimated_used_tokens, 0].max
692
+ estimated_pct = (estimated_used_tokens.to_f / window_tokens) * 100.0
693
+
694
+ {
695
+ window_tokens: window_tokens,
696
+ window_source: window_source,
697
+ estimated_used_tokens: estimated_used_tokens,
698
+ estimated_remaining_tokens: estimated_remaining_tokens,
699
+ estimated_pct: estimated_pct,
700
+ source: "estimate"
701
+ }
702
+ end
703
+
704
+ def context_chars_per_token
705
+ cfg = begin Samagotchi::Config.get("context.chars_per_token") rescue nil end
706
+ if cfg && cfg.to_f.positive?
707
+ v = cfg.to_f
708
+ return v.positive? ? v : DEFAULT_CONTEXT_CHARS_PER_TOKEN
709
+ end
710
+ value = ENV.fetch(CONTEXT_CHARS_PER_TOKEN_ENV, DEFAULT_CONTEXT_CHARS_PER_TOKEN.to_s).to_f
711
+ value.positive? ? value : DEFAULT_CONTEXT_CHARS_PER_TOKEN
712
+ end
713
+
714
+ def context_status_thresholds
715
+ cfg = begin Samagotchi::Config.get("context.status_thresholds") rescue nil end
716
+ raw = cfg && !cfg.to_s.strip.empty? ? cfg.to_s : ENV.fetch(CONTEXT_THRESHOLDS_ENV, DEFAULT_CONTEXT_THRESHOLDS.join(","))
717
+ parsed = raw.split(",").map { |value| value.strip.to_i }.select { |value| value.between?(1, 99) }.uniq.sort
718
+ parsed.empty? ? DEFAULT_CONTEXT_THRESHOLDS : parsed
719
+ end
720
+
721
+ def context_status_cadence
722
+ cfg = begin Samagotchi::Config.get("context.status_cadence") rescue nil end
723
+ if !cfg.nil?
724
+ v = cfg.to_i
725
+ return [v, 0].max
726
+ end
727
+ value = ENV.fetch(CONTEXT_CADENCE_ENV, DEFAULT_CONTEXT_CADENCE.to_s).to_i
728
+ [value, 0].max
729
+ end
730
+
731
+ # 0 for the bucket under the first threshold, then one per threshold.
732
+ def bucket_rank(bucket)
733
+ return 0 if bucket.nil? || bucket.to_s.start_with?("under")
734
+
735
+ (context_status_thresholds.index(bucket.to_s.delete_suffix("plus").to_i) || -1) + 1
736
+ end
737
+
738
+ # A rise (never a fall or a cadence tick) into a bucket whose guidance
739
+ # asks for a change. With no previous bucket (a first turn, a resumed
740
+ # session with no line yet), the first bucket counts as a rise from 0.
741
+ def guidance_due?(previous, bucket)
742
+ rank = bucket_rank(bucket)
743
+ rank >= CONTEXT_GUIDANCE_FROM_RANK && rank > bucket_rank(previous)
744
+ end
745
+
746
+ def context_guidance_message(usage:, bucket:)
747
+ how = usage[:source].to_s == "server" ? "as the server reports" : "estimated"
748
+ { role: "system", kind: CONTEXT_LINE_KIND,
749
+ content: "#{CONTEXT_LINE_PREFIX}about #{usage[:estimated_pct].to_f.round}% of the context window is in use " \
750
+ "(#{how}; bucket=#{bucket}). #{context_status_guidance(bucket)}]" }
751
+ end
752
+
753
+ def context_status_bucket(estimated_pct)
754
+ thresholds = context_status_thresholds
755
+ bucket = "under#{thresholds.first}"
756
+ thresholds.each do |threshold|
757
+ bucket = "#{threshold}plus" if estimated_pct >= threshold
758
+ end
759
+ bucket
760
+ end
761
+
762
+ def should_emit_context_status?(state:, bucket:, iteration_index:)
763
+ last_bucket = state[:last_bucket]
764
+ below_threshold_bucket = "under#{context_status_thresholds.first}"
765
+ bucket_changed = if last_bucket.nil?
766
+ bucket != below_threshold_bucket
767
+ else
768
+ bucket != last_bucket
769
+ end
770
+
771
+ cadence = context_status_cadence
772
+ cadence_due = cadence.positive? && ((iteration_index + 1) % cadence).zero?
773
+ bucket_changed || cadence_due
774
+ end
775
+
776
+ def context_status_message(usage:, bucket:, source:)
777
+ format(
778
+ "%<prefix>s window_tokens=%<window>d window_src=%<window_src>s est_used_tokens=%<used>d est_remaining_tokens=%<remaining>d est_pct=%<pct>.1f bucket=%<bucket>s thresholds=%<thresholds>s src=%<src>s guidance=%<guidance>s",
779
+ prefix: CONTEXT_STATUS_PREFIX,
780
+ window: usage[:window_tokens],
781
+ window_src: usage[:window_source] || "default",
782
+ used: usage[:estimated_used_tokens],
783
+ remaining: usage[:estimated_remaining_tokens],
784
+ pct: usage[:estimated_pct],
785
+ bucket: bucket,
786
+ thresholds: context_status_thresholds.join(","),
787
+ src: source,
788
+ guidance: context_status_guidance(bucket)
789
+ )
790
+ end
791
+
792
+ def context_status_guidance(bucket)
793
+ case bucket
794
+ when "under20", "20plus"
795
+ "context healthy — proceed normally"
796
+ when "40plus"
797
+ "context moderate — prefer targeted and range reads over full-file dumps"
798
+ when "60plus"
799
+ "context elevated — be concise, prefer range reads, avoid re-reading large files"
800
+ when "80plus"
801
+ "context critical — summarize aggressively, avoid large outputs, delegate broad work to subagents"
802
+ else
803
+ "context healthy — proceed normally"
804
+ end
805
+ end
806
+
807
+ # Parse tool calls from raw model output using the active profile's
808
+ # ToolCallParser strategy (Gemma 4 or Qwen 3.6).
809
+ # Thought content is intentionally left intact while a tool-call turn is in
810
+ # progress to preserve same-turn reasoning context between tool calls.
811
+ def parse_tool_calls(text)
812
+ parser.parse(text.to_s)
813
+ end
814
+
815
+ public
816
+
817
+ # Public wrapper so other loops (e.g. the chat loop) can strip
818
+ # per-profile thought blocks from finished model text without duplicating the
819
+ # Gemma 4 / Qwen 3.6 logic. Mirrors the native loop's "strip before deciding
820
+ # whether the model called a tool / returning the final answer".
821
+ def strip_model_thought(text)
822
+ strip_thought_blocks(text)
823
+ end
824
+
825
+ # The status line's context value ({est_pct:, bucket:}) for +used_tokens+
826
+ # of +window_tokens+; nil without both, or with context.status off. The
827
+ # chat loop builds its value with it too.
828
+ def context_display(used_tokens:, window_tokens:)
829
+ return nil unless context_status_enabled?
830
+ return nil unless ContextWindow.positive_integer?(used_tokens) && ContextWindow.positive_integer?(window_tokens)
831
+
832
+ pct = (used_tokens.to_f / window_tokens) * 100.0
833
+ { est_pct: pct, bucket: context_status_bucket(pct) }
834
+ end
835
+
836
+ # Per-profile parse strategy. Rebuilt when the active profile changes
837
+ # (the profile may be re-inferred per run when not explicitly pinned).
838
+ # Public so other loops (and specs) can reach the active profile's parser.
839
+ def parser
840
+ @parser = ToolCallParser.for_profile(@profile) if @parser_profile != @profile
841
+ @parser_profile = @profile
842
+ @parser
843
+ end
844
+ private
845
+
846
+ # The generation's thinking: what the stream split into the thinking
847
+ # lane (Qwen), else what the profile's thought blocks hold (Gemma's
848
+ # stream isn't split).
849
+ def thinking_chars(response, streamed)
850
+ return streamed if streamed.positive?
851
+
852
+ response.length - strip_thought_blocks(response).length
853
+ end
854
+
855
+ # Remove thought blocks from model output. The format depends on profile
856
+ # (Gemma 4 <|think|>/channel blocks vs Qwen 3.6 literal think tokens);
857
+ # the per-profile logic lives in ToolCallParser.
858
+ def strip_thought_blocks(text)
859
+ parser.strip_thought(text)
860
+ end
861
+
862
+ def sanitize_history(messages)
863
+ messages.map do |m|
864
+ if m[:role] == "model"
865
+ { role: m[:role], content: strip_thought_blocks(m[:content].to_s) }
866
+ else
867
+ m.dup
868
+ end
869
+ end
870
+ end
871
+
872
+ def prepare_conversation(messages)
873
+ if messages.is_a?(Result)
874
+ duplicate_conversation(messages.conversation)
875
+ else
876
+ # Standard multi-turn compliance: never pass prior raw thought blocks.
877
+ sanitize_history(messages)
878
+ end
879
+ end
880
+
881
+ def duplicate_conversation(messages)
882
+ messages.map(&:dup)
883
+ end
884
+
885
+ def last_model_content(conversation)
886
+ message = conversation.reverse.find { |entry| entry[:role] == "model" }
887
+ message ? message[:content].to_s : ""
888
+ end
889
+
890
+ def tool_response_turn?(message)
891
+ message && message[:role] == "tool_response"
892
+ end
893
+
894
+ public
895
+ # Public entry point for executing an ALREADY-NORMALIZED internal tool call
896
+ # (the {name:, content:, path:, scope:, …} shape).
897
+ #
898
+ # Other agentic loops — notably the chat loop's native tool calls
899
+ # — need to execute tool calls through this single path so tool execution,
900
+ # unknown-tool handling, and activity events are shared, not duplicated. Callers
901
+ # are responsible for normalizing the provider's native call into this shape
902
+ # first (see Samagotchi::LLM::NativeToolNormalizer); dispatch itself never
903
+ # parses provider text.
904
+ def dispatch_tool_call(call)
905
+ dispatch(call)
906
+ end
907
+
908
+ private
909
+ def dispatch(call)
910
+ entry = @tools[call[:name]]
911
+ unless entry
912
+ available = @tools.names.join(", ")
913
+ result = "Error: unknown tool '#{call[:name]}'. Available: #{available}"
914
+ return {
915
+ output: result,
916
+ activity: ToolActivity.tool_activity_event(call[:name], call, result, registry: @tools)
917
+ }
918
+ end
919
+
920
+ dump_log("tool_call", call[:content], tool: call[:name], path: call[:path], scope: call[:scope])
921
+
922
+ result = entry.handler.call(call, tool_context)
923
+
924
+ dump_log("tool_result", result, tool: call[:name])
925
+ dispatched = {
926
+ output: "[#{call[:name]}]\n#{result}",
927
+ activity: ToolActivity.tool_activity_event(call[:name], call, result, registry: @tools)
928
+ }
929
+ # Images the tool returned (read's ImageResult, a plugin's ToolResult):
930
+ # ToolRunner attaches them (or says why not).
931
+ if result.respond_to?(:images) && !Array(result.images).empty?
932
+ dispatched[:images] = Array(result.images)
933
+ dispatched[:image_only] = true if result.respond_to?(:image_only?) && result.image_only?
934
+ end
935
+ dispatched
936
+ rescue => e
937
+ dump_log("tool_error", e.message, tool: call[:name], error: e.class.name)
938
+ result = "Error: #{e.message}"
939
+ {
940
+ output: "[#{call[:name]}] #{result}",
941
+ activity: ToolActivity.tool_activity_event(call[:name], call, result, registry: @tools)
942
+ }
943
+ end
944
+
945
+ def handle_ask_user_question(call)
946
+ question = (call[:question] || call[:content]).to_s.strip
947
+ raw_opts = call[:options]
948
+ # Dumb-model tolerant: raw may be String JSON, Array, or malformed with brackets/quotes
949
+ options = Samagotchi::Tools::AskUserQuestion.normalize_options_lenient(raw_opts)
950
+ # Fallback for case where raw was String like '["a","b"]' but lenient returned [] due to edge parse, try raw string of params
951
+ if options.empty? && raw_opts.is_a?(String)
952
+ options = Samagotchi::Tools::AskUserQuestion.normalize_options_lenient(raw_opts.to_s)
953
+ end
954
+ header = call[:header].to_s.strip
955
+ header = nil if header.empty?
956
+ multi = call[:multi_select]
957
+ free = call[:allow_freeform]
958
+ # Normalize booleans from string forms (Gemma passes "true"/"false" as strings)
959
+ multi = normalize_ask_bool(multi)
960
+ free = normalize_ask_bool(free)
961
+
962
+ if question.empty?
963
+ return "Error: ask_user_question requires 'question'"
964
+ end
965
+ # Dumb-model tolerant: salvage single-option parse glitches, but still require at least 1
966
+ if options.size < 1
967
+ alt = Samagotchi::Tools::AskUserQuestion.normalize_options_lenient(call[:content].to_s) if call[:content]
968
+ options = alt unless alt.empty?
969
+ end
970
+ if options.empty?
971
+ return "Error: ask_user_question requires 2-8 options (got 0). Provide e.g. options=[\"Cats\",\"Dogs\"]"
972
+ end
973
+ if options.size == 1
974
+ # Allow single-option salvage for dumb models (will still render, user can answer or provide freeform)
975
+ elsif options.size < 2 || options.size > 8
976
+ return "Error: ask_user_question requires 2-8 options (got #{options.size}). Provide e.g. options=[\"Cats\",\"Dogs\"]"
977
+ end
978
+
979
+ # If an Engine-level blocking handler is registered (TUI/Web), delegate
980
+ # there (Engine sets question_handler). Otherwise fall back to a
981
+ # non-blocking JSON preview so the model can still see a structured response.
982
+ handler = @question_handler
983
+
984
+ payload = {
985
+ question: question,
986
+ options: options,
987
+ header: header,
988
+ multi_select: !!multi,
989
+ allow_freeform: !!free
990
+ }.compact
991
+
992
+ if handler
993
+ begin
994
+ result = handler.call(payload)
995
+ return result.to_s
996
+ rescue => e
997
+ return "Error: ask_user_question handler failed: #{e.message}"
998
+ end
999
+ end
1000
+
1001
+ # Headless fallback: return JSON so model sees structured options and can
1002
+ # fallback to plain text qualification.
1003
+ JSON.pretty_generate(payload)
1004
+ end
1005
+
1006
+ def normalize_ask_bool(v)
1007
+ return nil if v.nil?
1008
+ return v if v == true || v == false
1009
+
1010
+ s = v.to_s.strip.downcase
1011
+ return true if %w[1 true yes on].include?(s)
1012
+ return false if %w[0 false no off].include?(s)
1013
+
1014
+ nil
1015
+ end
1016
+ end
1017
+ end