samagotchi 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/CHANGELOG.md +43 -0
- data/LICENSE +21 -0
- data/README.md +126 -0
- data/bin/chi +1140 -0
- data/docs/architecture.md +299 -0
- data/docs/cli.md +490 -0
- data/docs/configuration.md +494 -0
- data/docs/desktop.md +97 -0
- data/docs/guardrails.md +218 -0
- data/docs/hooks.md +309 -0
- data/docs/internals/background-tasks.md +26 -0
- data/docs/internals/context-telemetry.md +36 -0
- data/docs/internals/gemma4-contract.md +23 -0
- data/docs/internals/tool-guardrails.md +45 -0
- data/docs/memory.md +85 -0
- data/docs/plugins.md +819 -0
- data/docs/releasing.md +135 -0
- data/docs/sessions.md +155 -0
- data/lib/samagotchi/bridge/bounded_queue.rb +70 -0
- data/lib/samagotchi/bridge/card_store.rb +126 -0
- data/lib/samagotchi/bridge/event_id.rb +25 -0
- data/lib/samagotchi/bridge/ring_buffer.rb +63 -0
- data/lib/samagotchi/bridge/sse_writer.rb +248 -0
- data/lib/samagotchi/bridge/turn_accumulator.rb +189 -0
- data/lib/samagotchi/bridge.rb +993 -0
- data/lib/samagotchi/bridge_client/event_stream.rb +158 -0
- data/lib/samagotchi/bridge_client/sse_parser.rb +51 -0
- data/lib/samagotchi/bridge_client.rb +330 -0
- data/lib/samagotchi/bundle_needs.rb +97 -0
- data/lib/samagotchi/bundles/btw/manifest.yml +10 -0
- data/lib/samagotchi/bundles/btw/plugin.rb +100 -0
- data/lib/samagotchi/bundles/guardrails/guardrails/rules.yml +82 -0
- data/lib/samagotchi/bundles/guardrails/guardrails.md +14 -0
- data/lib/samagotchi/bundles/guardrails/manifest.yml +8 -0
- data/lib/samagotchi/bundles/known-names/hooks/known_names.rb +210 -0
- data/lib/samagotchi/bundles/known-names/known_names.md +3 -0
- data/lib/samagotchi/bundles/known-names/manifest.yml +14 -0
- data/lib/samagotchi/bundles/loop-guard/manifest.yml +10 -0
- data/lib/samagotchi/bundles/loop-guard/plugin.rb +158 -0
- data/lib/samagotchi/bundles/mcp/manifest.yml +11 -0
- data/lib/samagotchi/bundles/mcp/plugin.rb +631 -0
- data/lib/samagotchi/bundles/system/config_modification_protocol.md +149 -0
- data/lib/samagotchi/bundles/system/delegated.md +10 -0
- data/lib/samagotchi/bundles/system/identity.md +7 -0
- data/lib/samagotchi/bundles/system/manifest.yml +11 -0
- data/lib/samagotchi/bundles/system/memory_guide.md +107 -0
- data/lib/samagotchi/bundles/system/self_map.md +55 -0
- data/lib/samagotchi/cancellation_controller.rb +78 -0
- data/lib/samagotchi/client.rb +429 -0
- data/lib/samagotchi/commands/registry.rb +112 -0
- data/lib/samagotchi/config.rb +910 -0
- data/lib/samagotchi/context_note.rb +77 -0
- data/lib/samagotchi/context_quote.rb +21 -0
- data/lib/samagotchi/context_usage.rb +66 -0
- data/lib/samagotchi/context_window.rb +76 -0
- data/lib/samagotchi/debug_log.rb +110 -0
- data/lib/samagotchi/desktop/macos/App.swift +102 -0
- data/lib/samagotchi/desktop/macos/ChiRunner.swift +201 -0
- data/lib/samagotchi/desktop/macos/Hotkey.swift +42 -0
- data/lib/samagotchi/desktop/macos/Info.plist.erb +42 -0
- data/lib/samagotchi/desktop/macos/Panel.swift +383 -0
- data/lib/samagotchi/desktop/macos.rb +255 -0
- data/lib/samagotchi/desktop.rb +21 -0
- data/lib/samagotchi/desktop_command.rb +143 -0
- data/lib/samagotchi/engine.rb +2807 -0
- data/lib/samagotchi/guardrails/approval.rb +125 -0
- data/lib/samagotchi/guardrails/approvals.rb +177 -0
- data/lib/samagotchi/guardrails/context.rb +71 -0
- data/lib/samagotchi/guardrails/gate.rb +125 -0
- data/lib/samagotchi/guardrails/load_failures.rb +46 -0
- data/lib/samagotchi/guardrails/protected_paths.rb +77 -0
- data/lib/samagotchi/guardrails/rules.rb +199 -0
- data/lib/samagotchi/guardrails/targets.rb +119 -0
- data/lib/samagotchi/guardrails/verdict.rb +134 -0
- data/lib/samagotchi/guardrails.rb +18 -0
- data/lib/samagotchi/hooks/bundle_loader.rb +158 -0
- data/lib/samagotchi/hooks/loader.rb +162 -0
- data/lib/samagotchi/hooks/registry.rb +261 -0
- data/lib/samagotchi/hooks.rb +30 -0
- data/lib/samagotchi/host_registry.rb +315 -0
- data/lib/samagotchi/idle_client.rb +147 -0
- data/lib/samagotchi/idle_recap.rb +549 -0
- data/lib/samagotchi/idle_reminders.rb +101 -0
- data/lib/samagotchi/idle_scheduler.rb +76 -0
- data/lib/samagotchi/image_store.rb +393 -0
- data/lib/samagotchi/installed_gem.rb +38 -0
- data/lib/samagotchi/kernel_loop.rb +1017 -0
- data/lib/samagotchi/launch_mode.rb +34 -0
- data/lib/samagotchi/llm/backend.rb +28 -0
- data/lib/samagotchi/llm/chat_loop.rb +450 -0
- data/lib/samagotchi/llm/errors.rb +329 -0
- data/lib/samagotchi/llm/http.rb +412 -0
- data/lib/samagotchi/llm/model_result.rb +72 -0
- data/lib/samagotchi/llm/native_backend.rb +50 -0
- data/lib/samagotchi/llm/native_tool_normalizer.rb +277 -0
- data/lib/samagotchi/llm/openai_chat.rb +403 -0
- data/lib/samagotchi/llm/usage.rb +79 -0
- data/lib/samagotchi/log.rb +200 -0
- data/lib/samagotchi/log_line.rb +127 -0
- data/lib/samagotchi/log_path.rb +31 -0
- data/lib/samagotchi/log_subscriber.rb +163 -0
- data/lib/samagotchi/memory_bundle/builder.rb +364 -0
- data/lib/samagotchi/memory_bundle/index_updater.rb +123 -0
- data/lib/samagotchi/memory_bundle/installer.rb +528 -0
- data/lib/samagotchi/memory_bundle/listing.rb +72 -0
- data/lib/samagotchi/memory_bundle/manifest.rb +225 -0
- data/lib/samagotchi/memory_bundle/merger.rb +52 -0
- data/lib/samagotchi/memory_bundle/placeholder.rb +37 -0
- data/lib/samagotchi/memory_bundle/provenance.rb +257 -0
- data/lib/samagotchi/memory_bundle/source.rb +153 -0
- data/lib/samagotchi/memory_bundle/status.rb +107 -0
- data/lib/samagotchi/memory_bundle/system_bundle.rb +161 -0
- data/lib/samagotchi/memory_bundle/uninstaller.rb +128 -0
- data/lib/samagotchi/memory_bundle.rb +17 -0
- data/lib/samagotchi/memory_paths.rb +101 -0
- data/lib/samagotchi/model_overlay.rb +53 -0
- data/lib/samagotchi/model_profile.rb +309 -0
- data/lib/samagotchi/muted_memories.rb +66 -0
- data/lib/samagotchi/note_command.rb +163 -0
- data/lib/samagotchi/output_formatter.rb +100 -0
- data/lib/samagotchi/owner_lock.rb +110 -0
- data/lib/samagotchi/pending_input_queue.rb +48 -0
- data/lib/samagotchi/plugin/api.rb +362 -0
- data/lib/samagotchi/plugin/context.rb +193 -0
- data/lib/samagotchi/plugin/loader.rb +126 -0
- data/lib/samagotchi/plugin/service.rb +117 -0
- data/lib/samagotchi/plugin/sessions.rb +150 -0
- data/lib/samagotchi/plugin/side_question.rb +60 -0
- data/lib/samagotchi/plugin/tool_result.rb +24 -0
- data/lib/samagotchi/project_scope.rb +25 -0
- data/lib/samagotchi/prompt.rb +119 -0
- data/lib/samagotchi/prompt_literal_guard.rb +70 -0
- data/lib/samagotchi/recap_store.rb +92 -0
- data/lib/samagotchi/reminder_store.rb +165 -0
- data/lib/samagotchi/self_report.rb +195 -0
- data/lib/samagotchi/send_command.rb +170 -0
- data/lib/samagotchi/served_model.rb +32 -0
- data/lib/samagotchi/session.rb +508 -0
- data/lib/samagotchi/session_commands.rb +527 -0
- data/lib/samagotchi/session_delete_command.rb +105 -0
- data/lib/samagotchi/session_manager.rb +1049 -0
- data/lib/samagotchi/session_metrics.rb +466 -0
- data/lib/samagotchi/session_observer.rb +117 -0
- data/lib/samagotchi/terminal_ui/attach_launcher.rb +118 -0
- data/lib/samagotchi/terminal_ui/attached_loop.rb +1037 -0
- data/lib/samagotchi/terminal_ui/attached_view.rb +264 -0
- data/lib/samagotchi/terminal_ui/event_renderer.rb +192 -0
- data/lib/samagotchi/terminal_ui/formatting.rb +291 -0
- data/lib/samagotchi/terminal_ui/image_input.rb +36 -0
- data/lib/samagotchi/terminal_ui/input_support.rb +324 -0
- data/lib/samagotchi/terminal_ui/legacy_surface.rb +111 -0
- data/lib/samagotchi/terminal_ui/line_reader.rb +113 -0
- data/lib/samagotchi/terminal_ui/live_region.rb +36 -0
- data/lib/samagotchi/terminal_ui/plain_surface.rb +51 -0
- data/lib/samagotchi/terminal_ui/question_prompt.rb +153 -0
- data/lib/samagotchi/terminal_ui/question_slot.rb +131 -0
- data/lib/samagotchi/terminal_ui/reline_seam.rb +216 -0
- data/lib/samagotchi/terminal_ui/repl_input.rb +138 -0
- data/lib/samagotchi/terminal_ui/screen.rb +316 -0
- data/lib/samagotchi/terminal_ui/surface.rb +47 -0
- data/lib/samagotchi/terminal_ui/thinking_line.rb +101 -0
- data/lib/samagotchi/terminal_ui.rb +1992 -0
- data/lib/samagotchi/thinking_ticker.rb +110 -0
- data/lib/samagotchi/thought_stream_splitter.rb +149 -0
- data/lib/samagotchi/token_usage.rb +88 -0
- data/lib/samagotchi/tool_activity.rb +216 -0
- data/lib/samagotchi/tool_call_parser.rb +637 -0
- data/lib/samagotchi/tool_declarations.rb +561 -0
- data/lib/samagotchi/tool_runner.rb +211 -0
- data/lib/samagotchi/tools/args.rb +259 -0
- data/lib/samagotchi/tools/ask_user_question.rb +152 -0
- data/lib/samagotchi/tools/builtins.rb +122 -0
- data/lib/samagotchi/tools/cancel_reminder.rb +21 -0
- data/lib/samagotchi/tools/delegate.rb +167 -0
- data/lib/samagotchi/tools/delegate_result.rb +53 -0
- data/lib/samagotchi/tools/delegate_wait.rb +153 -0
- data/lib/samagotchi/tools/edit.rb +155 -0
- data/lib/samagotchi/tools/execute.rb +214 -0
- data/lib/samagotchi/tools/list_reminders.rb +20 -0
- data/lib/samagotchi/tools/list_sessions.rb +74 -0
- data/lib/samagotchi/tools/memory.rb +256 -0
- data/lib/samagotchi/tools/output_guardrails.rb +93 -0
- data/lib/samagotchi/tools/peers.rb +18 -0
- data/lib/samagotchi/tools/read.rb +182 -0
- data/lib/samagotchi/tools/register_reminder.rb +53 -0
- data/lib/samagotchi/tools/registry.rb +60 -0
- data/lib/samagotchi/tools/send_note.rb +49 -0
- data/lib/samagotchi/tools/task_create.rb +29 -0
- data/lib/samagotchi/tools/task_get.rb +39 -0
- data/lib/samagotchi/tools/task_list.rb +43 -0
- data/lib/samagotchi/tools/task_runtime.rb +311 -0
- data/lib/samagotchi/tools/task_stop.rb +29 -0
- data/lib/samagotchi/tools/task_wait.rb +104 -0
- data/lib/samagotchi/tools/tool_path.rb +18 -0
- data/lib/samagotchi/tools/web_fetch.rb +163 -0
- data/lib/samagotchi/tools/write.rb +26 -0
- data/lib/samagotchi/turn_flow.rb +242 -0
- data/lib/samagotchi/turn_note.rb +76 -0
- data/lib/samagotchi/turn_tally.rb +101 -0
- data/lib/samagotchi/version.rb +7 -0
- data/lib/samagotchi/vision_context.rb +132 -0
- data/lib/samagotchi/vision_support.rb +109 -0
- data/lib/samagotchi/web/app.rb +1349 -0
- data/lib/samagotchi/web/markdown_renderer.rb +107 -0
- data/lib/samagotchi/web/message_parts.rb +169 -0
- data/lib/samagotchi/web/public/activity.js +100 -0
- data/lib/samagotchi/web/public/annotations.js +67 -0
- data/lib/samagotchi/web/public/app.js +2382 -0
- data/lib/samagotchi/web/public/card.js +74 -0
- data/lib/samagotchi/web/public/chat_view.js +360 -0
- data/lib/samagotchi/web/public/chunk_router.js +25 -0
- data/lib/samagotchi/web/public/command_complete.js +39 -0
- data/lib/samagotchi/web/public/composer_size.js +19 -0
- data/lib/samagotchi/web/public/copy.js +142 -0
- data/lib/samagotchi/web/public/ctx.js +35 -0
- data/lib/samagotchi/web/public/data.js +256 -0
- data/lib/samagotchi/web/public/format.js +232 -0
- data/lib/samagotchi/web/public/hold.js +78 -0
- data/lib/samagotchi/web/public/images.js +77 -0
- data/lib/samagotchi/web/public/index.html +568 -0
- data/lib/samagotchi/web/public/init_row.js +60 -0
- data/lib/samagotchi/web/public/model_pick.js +23 -0
- data/lib/samagotchi/web/public/question_card.js +100 -0
- data/lib/samagotchi/web/public/route.js +17 -0
- data/lib/samagotchi/web/public/scope.js +36 -0
- data/lib/samagotchi/web/public/scroll.js +24 -0
- data/lib/samagotchi/web/public/sentences.js +88 -0
- data/lib/samagotchi/web/public/sessions_list.js +60 -0
- data/lib/samagotchi/web/public/strip.js +25 -0
- data/lib/samagotchi/web/public/tally.js +37 -0
- data/lib/samagotchi/web/public/thinking_ticker.js +79 -0
- data/lib/samagotchi/web/public/timing.js +185 -0
- data/lib/samagotchi/web/public/turn_events.js +209 -0
- data/lib/samagotchi/web/public/turn_model.js +204 -0
- data/lib/samagotchi/web/public/turn_view.js +587 -0
- data/lib/samagotchi/web/server.rb +183 -0
- data/lib/samagotchi/web/session_hub.rb +329 -0
- data/lib/samagotchi/web/session_summary.rb +85 -0
- data/lib/samagotchi/worker.rb +635 -0
- data/lib/samagotchi/worker_idle_exit.rb +87 -0
- data/lib/samagotchi.rb +12 -0
- metadata +374 -0
|
@@ -0,0 +1,429 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "net/http"
|
|
4
|
+
require "json"
|
|
5
|
+
require "uri"
|
|
6
|
+
require_relative "config"
|
|
7
|
+
require_relative "cancellation_controller"
|
|
8
|
+
require_relative "llm/http"
|
|
9
|
+
require_relative "vision_context"
|
|
10
|
+
require_relative "vision_support"
|
|
11
|
+
|
|
12
|
+
module Samagotchi
|
|
13
|
+
# Thin HTTP client for llama.cpp's native /completion endpoint, or an
|
|
14
|
+
# OpenAI-compatible /v1/completions endpoint (e.g. mlx_lm.server or oMLX).
|
|
15
|
+
# Configure via environment variables (see Samagotchi::Config):
|
|
16
|
+
# SAMAGOTCHI_SERVER_HOST (default: localhost)
|
|
17
|
+
# SAMAGOTCHI_SERVER_PORT (default: 8080; oMLX's default is 8000, set it to match)
|
|
18
|
+
# SAMAGOTCHI_SERVER_OPEN_TIMEOUT (default: 10 seconds)
|
|
19
|
+
# SAMAGOTCHI_SERVER_READ_TIMEOUT (default: 600 seconds)
|
|
20
|
+
# SAMAGOTCHI_SERVER_TRANSPORT (llama_cpp|mlx|omlx, default: llama_cpp)
|
|
21
|
+
class Client
|
|
22
|
+
# The shared HTTP layer's errors, under their old names.
|
|
23
|
+
RequestCancelled = LLM::RequestCancelled
|
|
24
|
+
RetryExhausted = LLM::RetryExhausted
|
|
25
|
+
|
|
26
|
+
# The /props probe runs before a turn's generation, so it gets a short
|
|
27
|
+
# budget and no retry (see #server_props).
|
|
28
|
+
CONTEXT_WINDOW_PROBE_OPEN_TIMEOUT = 1
|
|
29
|
+
CONTEXT_WINDOW_PROBE_READ_TIMEOUT = 2
|
|
30
|
+
|
|
31
|
+
SERVER_TRANSPORT_ENV = "SAMAGOTCHI_SERVER_TRANSPORT"
|
|
32
|
+
DEFAULT_TRANSPORT = :llama_cpp
|
|
33
|
+
VALID_TRANSPORTS = %i[llama_cpp mlx omlx].freeze
|
|
34
|
+
|
|
35
|
+
# Wire-format strategy for one server transport. `Client` keeps the
|
|
36
|
+
# transport-agnostic request/retry/stream loop; everything that differs
|
|
37
|
+
# between llama.cpp's native API and the OpenAI-compatible servers
|
|
38
|
+
# (mlx_lm.server, oMLX) lives here: endpoint paths, payload keys, streamed
|
|
39
|
+
# content parsing, and the request's `model` field semantics.
|
|
40
|
+
class Transport
|
|
41
|
+
attr_reader :name
|
|
42
|
+
|
|
43
|
+
# @param name [Symbol] one of Client::VALID_TRANSPORTS
|
|
44
|
+
# @param model_resolver [Proc, nil] client-installed resolver for the
|
|
45
|
+
# request's `model` field (oMLX resolves against its /v1/models list;
|
|
46
|
+
# mlx installs one that always returns nil to omit the field); nil
|
|
47
|
+
# means forward the selector verbatim (llama.cpp default)
|
|
48
|
+
def initialize(name, model_resolver: nil)
|
|
49
|
+
@name = name
|
|
50
|
+
@model_resolver = model_resolver
|
|
51
|
+
end
|
|
52
|
+
|
|
53
|
+
def label
|
|
54
|
+
# oMLX gets its own label so its error paths read "omlx ...", not "mlx ...".
|
|
55
|
+
@name == :llama_cpp ? "llama.cpp" : @name.to_s
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
def completion_path
|
|
59
|
+
openai_compatible? ? "/v1/completions" : "/completion"
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
def models_path
|
|
63
|
+
openai_compatible? ? "/v1/models" : "/models"
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
def token_limit_key
|
|
67
|
+
openai_compatible? ? :max_tokens : :n_predict
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
# Where the server describes itself (context window, chat template), or
|
|
71
|
+
# nil when it has no such route. llama.cpp's /props carries the per-slot
|
|
72
|
+
# n_ctx (-c split across --parallel slots) and the chat template.
|
|
73
|
+
# mlx_lm.server and oMLX expose neither.
|
|
74
|
+
def props_path
|
|
75
|
+
openai_compatible? ? nil : "/props"
|
|
76
|
+
end
|
|
77
|
+
|
|
78
|
+
def context_window_from(body)
|
|
79
|
+
n_ctx = body.is_a?(Hash) ? body.dig("default_generation_settings", "n_ctx") : nil
|
|
80
|
+
n_ctx.is_a?(Integer) && n_ctx.positive? ? n_ctx : nil
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
# Text content carried by one streamed `data:` payload.
|
|
84
|
+
def content_from_payload(payload)
|
|
85
|
+
openai_compatible? ? payload.dig("choices", 0, "text").to_s : payload.fetch("content", "")
|
|
86
|
+
end
|
|
87
|
+
|
|
88
|
+
# The request's `model` field for this transport (nil = omit the field):
|
|
89
|
+
# - llama.cpp: forward the selector (SAMAGOTCHI_DEFAULT_MODEL) verbatim.
|
|
90
|
+
# - mlx_lm.server: omit `model` entirely (use whatever was loaded via the
|
|
91
|
+
# server's own `--model` CLI flag).
|
|
92
|
+
# - oMLX: MUST send a model id that exists in the server's `/v1/models`
|
|
93
|
+
# list, or oMLX 400s with "model: Field required". The client-installed
|
|
94
|
+
# resolver maps the short selector (e.g. `gemma-4-26b-a4b-it-4bit`) to
|
|
95
|
+
# the exact registered id (which may be prefixed, e.g.
|
|
96
|
+
# `mlx-community--...`) by matching against the loaded /v1/models list.
|
|
97
|
+
def model_for_payload(model)
|
|
98
|
+
return @model_resolver.call(model) if @model_resolver
|
|
99
|
+
|
|
100
|
+
value = model.to_s.strip
|
|
101
|
+
value.empty? ? nil : value
|
|
102
|
+
end
|
|
103
|
+
|
|
104
|
+
private
|
|
105
|
+
|
|
106
|
+
def openai_compatible?
|
|
107
|
+
@name == :mlx || @name == :omlx
|
|
108
|
+
end
|
|
109
|
+
end
|
|
110
|
+
|
|
111
|
+
# One /props probe's outcome. `answered?` is false when the probe failed:
|
|
112
|
+
# a network error, a timeout or any non-200 (llama.cpp answers 503 while
|
|
113
|
+
# it loads a model). `body` is the parsed JSON of a 200, or nil when it
|
|
114
|
+
# isn't JSON.
|
|
115
|
+
ServerProps = Data.define(:body, :status) do
|
|
116
|
+
def answered?
|
|
117
|
+
status == :ok
|
|
118
|
+
end
|
|
119
|
+
end
|
|
120
|
+
|
|
121
|
+
# Moved to its own file; the old name keeps working.
|
|
122
|
+
CancellationController = Samagotchi::CancellationController
|
|
123
|
+
|
|
124
|
+
# @param sleeper [#call, nil] waits between retries (specs pass a no-op)
|
|
125
|
+
# @param scheme [String, nil] "https" for a TLS server (default http)
|
|
126
|
+
# @param first_token_timeout [Numeric, nil] seconds a completion may take
|
|
127
|
+
# to stream its first text (LLM::HTTP); nil: no limit
|
|
128
|
+
# @param name [String, nil] the host's config name, for error lines
|
|
129
|
+
# (default: the transport's label)
|
|
130
|
+
def initialize(host: nil, port: nil, open_timeout: nil, read_timeout: nil, transport: nil, sleeper: nil, scheme: nil,
|
|
131
|
+
first_token_timeout: nil, name: nil)
|
|
132
|
+
# Unified config precedence: CLI > ENV > file > default (via Samagotchi::Config)
|
|
133
|
+
cfg_host = nil; cfg_port = nil; cfg_transport_raw = nil
|
|
134
|
+
begin
|
|
135
|
+
cfg_host = Samagotchi::Config.get("server.host")
|
|
136
|
+
cfg_port = Samagotchi::Config.get("server.port")
|
|
137
|
+
cfg_open_timeout = Samagotchi::Config.get("server.open_timeout")
|
|
138
|
+
cfg_read_timeout = Samagotchi::Config.get("server.read_timeout")
|
|
139
|
+
cfg_transport_raw = Samagotchi::Config.get("server.transport")
|
|
140
|
+
rescue StandardError
|
|
141
|
+
nil
|
|
142
|
+
end
|
|
143
|
+
@host = host || cfg_host
|
|
144
|
+
@port = (port || cfg_port).to_i
|
|
145
|
+
@scheme = scheme || "http"
|
|
146
|
+
@open_timeout = (open_timeout || cfg_open_timeout).to_i
|
|
147
|
+
@read_timeout = (read_timeout || cfg_read_timeout).to_i
|
|
148
|
+
transport_fallback = cfg_transport_raw || ENV.fetch(SERVER_TRANSPORT_ENV, DEFAULT_TRANSPORT.to_s)
|
|
149
|
+
@transport = build_transport(resolve_transport(transport || transport_fallback))
|
|
150
|
+
@props_cache = {}
|
|
151
|
+
@props_mutex = Mutex.new
|
|
152
|
+
@first_token_timeout = first_token_timeout
|
|
153
|
+
@host_name = name
|
|
154
|
+
@label = name.to_s.empty? ? @transport.label : name.to_s
|
|
155
|
+
@http = LLM::HTTP.new(label: @label, open_timeout: @open_timeout, read_timeout: @read_timeout,
|
|
156
|
+
sleeper: sleeper, first_token_timeout: first_token_timeout)
|
|
157
|
+
end
|
|
158
|
+
|
|
159
|
+
# Seconds a completion may take to stream its first text, or nil.
|
|
160
|
+
attr_reader :first_token_timeout
|
|
161
|
+
|
|
162
|
+
# The host's config name, or nil when built without one.
|
|
163
|
+
attr_reader :host_name
|
|
164
|
+
|
|
165
|
+
# The wire-format strategy for this client's transport.
|
|
166
|
+
def transport
|
|
167
|
+
@transport
|
|
168
|
+
end
|
|
169
|
+
|
|
170
|
+
# Send a raw prompt and return the model's completion text.
|
|
171
|
+
#
|
|
172
|
+
# llama.cpp can stream completion chunks as newline-delimited `data: {...}`
|
|
173
|
+
# records. We consume that stream and still return a single joined string so
|
|
174
|
+
# the rest of the harness API stays unchanged.
|
|
175
|
+
#
|
|
176
|
+
# @param prompt [String] full formatted prompt string
|
|
177
|
+
# @param stop [Array<String>] stop sequences
|
|
178
|
+
# @param n_predict [Integer, nil] optional max tokens to generate
|
|
179
|
+
# @param model [String, nil] optional llama.cpp model identifier
|
|
180
|
+
# @param on_chunk [Proc, nil] optional callback per streamed chunk
|
|
181
|
+
# @param cancel_controller [CancellationController, nil] cancellation source for in-flight requests
|
|
182
|
+
# @param on_retry [Proc, nil] optional callback before retry sleep
|
|
183
|
+
# @param images [Array<String>] base64 images, one per
|
|
184
|
+
# ImagePlan::NATIVE_PLACEHOLDER in the prompt (llama.cpp only)
|
|
185
|
+
# @return [String] the generated text
|
|
186
|
+
def complete(prompt, stop: ["<end_of_turn>", "<|tool_response>"], n_predict: nil, model: nil, on_chunk: nil, cancel_controller: nil, on_retry: nil,
|
|
187
|
+
images: [])
|
|
188
|
+
images = Array(images)
|
|
189
|
+
return stream_completion(scrub_utf8(prompt), stop, n_predict, model, on_chunk, cancel_controller, on_retry) if images.empty?
|
|
190
|
+
|
|
191
|
+
# The media marker is random per server process: a restart between the
|
|
192
|
+
# /props read and the request makes the prompt fail to tokenize, so
|
|
193
|
+
# the marker is read again once.
|
|
194
|
+
attempts = 0
|
|
195
|
+
begin
|
|
196
|
+
attempts += 1
|
|
197
|
+
payload_prompt = { prompt_string: scrub_utf8(prompt.gsub(ImagePlan::NATIVE_PLACEHOLDER, media_marker!(model))),
|
|
198
|
+
multimodal_data: images }
|
|
199
|
+
stream_completion(payload_prompt, stop, n_predict, model, on_chunk, cancel_controller, on_retry)
|
|
200
|
+
rescue LLM::BadRequest => e
|
|
201
|
+
raise unless attempts == 1 && e.message.include?("Failed to tokenize prompt")
|
|
202
|
+
|
|
203
|
+
invalidate_context_window!
|
|
204
|
+
retry
|
|
205
|
+
end
|
|
206
|
+
end
|
|
207
|
+
|
|
208
|
+
private def stream_completion(prompt, stop, n_predict, model, on_chunk, cancel_controller, on_retry)
|
|
209
|
+
uri = completion_uri
|
|
210
|
+
request = Net::HTTP::Post.new(uri)
|
|
211
|
+
request["Content-Type"] = "application/json"
|
|
212
|
+
request.body = completion_payload(prompt, stop: stop, n_predict: n_predict, model: model).to_json
|
|
213
|
+
|
|
214
|
+
result = +""
|
|
215
|
+
reset_on_retry = lambda do |event|
|
|
216
|
+
# The retry streams the answer from the start again.
|
|
217
|
+
result = +""
|
|
218
|
+
on_retry&.call(**event)
|
|
219
|
+
end
|
|
220
|
+
@http.stream_lines(uri, request, cancel_controller: cancel_controller, on_retry: reset_on_retry,
|
|
221
|
+
on_network_error: ->(_error) { invalidate_context_window! },
|
|
222
|
+
log_fields: { model: model, purpose: "chat" }) do |line, shown|
|
|
223
|
+
parsed_chunk = parse_stream_line(line)
|
|
224
|
+
next unless parsed_chunk
|
|
225
|
+
|
|
226
|
+
content, payload = parsed_chunk
|
|
227
|
+
shown.call unless content.to_s.empty?
|
|
228
|
+
result << content
|
|
229
|
+
on_chunk&.call(content: content, payload: payload)
|
|
230
|
+
end
|
|
231
|
+
result
|
|
232
|
+
rescue RequestCancelled, LLM::ProviderError
|
|
233
|
+
raise
|
|
234
|
+
rescue StandardError => e
|
|
235
|
+
raise "#{@transport.label} request failed (#{@host}:#{@port}): #{e.message}"
|
|
236
|
+
end
|
|
237
|
+
|
|
238
|
+
# The running llama.cpp's media marker (/props), or a VisionUnsupported.
|
|
239
|
+
def media_marker!(model)
|
|
240
|
+
marker = @transport.props_path && VisionSupport.media_marker(server_props(model: model))
|
|
241
|
+
return marker if marker
|
|
242
|
+
|
|
243
|
+
raise LLM::VisionUnsupported.new("#{@label}: can't reach /props for the media marker", host: @label)
|
|
244
|
+
end
|
|
245
|
+
|
|
246
|
+
def list_models
|
|
247
|
+
uri = URI("#{@scheme}://#{@host}:#{@port}#{@transport.models_path}")
|
|
248
|
+
response = @http.fetch(uri, Net::HTTP::Get.new(uri), log_fields: { purpose: "models" })
|
|
249
|
+
parsed = JSON.parse(response.body.to_s)
|
|
250
|
+
parsed.fetch("data", parsed)
|
|
251
|
+
rescue LLM::ProviderError
|
|
252
|
+
raise
|
|
253
|
+
rescue StandardError => e
|
|
254
|
+
raise "#{@transport.label} model listing failed (#{@host}:#{@port}): #{e.message}"
|
|
255
|
+
end
|
|
256
|
+
|
|
257
|
+
# What the running server says about itself (/props), as a ServerProps,
|
|
258
|
+
# or nil when the transport has no such route. One GET with short
|
|
259
|
+
# timeouts and no retry: it runs before generation and must never hold up
|
|
260
|
+
# a turn. The probe names the model (`?model=`): a llama.cpp router
|
|
261
|
+
# answers a stub without it, and a single-model server ignores it.
|
|
262
|
+
# Whatever the server answers (a non-200 too) is cached per model; a
|
|
263
|
+
# network failure is not, so the next call asks again.
|
|
264
|
+
def server_props(model: nil)
|
|
265
|
+
path = @transport.props_path
|
|
266
|
+
return nil unless path
|
|
267
|
+
|
|
268
|
+
key = model.to_s
|
|
269
|
+
@props_mutex.synchronize do
|
|
270
|
+
return @props_cache[key] if @props_cache.key?(key)
|
|
271
|
+
end
|
|
272
|
+
|
|
273
|
+
props = probe_props(path, key)
|
|
274
|
+
@props_mutex.synchronize { @props_cache[key] = props } unless props.status == :network_error
|
|
275
|
+
props
|
|
276
|
+
end
|
|
277
|
+
|
|
278
|
+
# The context window (tokens) the running server was started with, or nil
|
|
279
|
+
# when the transport reports none or the probe fails (see #server_props).
|
|
280
|
+
def context_window(model: nil)
|
|
281
|
+
props = server_props(model: model)
|
|
282
|
+
props&.answered? ? @transport.context_window_from(props.body) : nil
|
|
283
|
+
rescue StandardError
|
|
284
|
+
nil
|
|
285
|
+
end
|
|
286
|
+
|
|
287
|
+
# Forget cached /props answers: the server may have restarted with
|
|
288
|
+
# another -c, or a model switch may have loaded one with a different
|
|
289
|
+
# window.
|
|
290
|
+
def invalidate_context_window!
|
|
291
|
+
@props_mutex.synchronize { @props_cache.clear }
|
|
292
|
+
end
|
|
293
|
+
|
|
294
|
+
private
|
|
295
|
+
|
|
296
|
+
def probe_props(path, model)
|
|
297
|
+
query = model.empty? ? "" : "?#{URI.encode_www_form(model: model)}"
|
|
298
|
+
uri = URI("#{@scheme}://#{@host}:#{@port}#{path}#{query}")
|
|
299
|
+
response = @http.fetch(uri, Net::HTTP::Get.new(uri), retries: false, check_status: false,
|
|
300
|
+
log_fields: { model: model.empty? ? nil : model, purpose: "probe" },
|
|
301
|
+
open_timeout: CONTEXT_WINDOW_PROBE_OPEN_TIMEOUT,
|
|
302
|
+
read_timeout: CONTEXT_WINDOW_PROBE_READ_TIMEOUT)
|
|
303
|
+
return ServerProps.new(body: nil, status: :http_error) unless response.code.to_s == "200"
|
|
304
|
+
|
|
305
|
+
ServerProps.new(body: parse_props(response.body), status: :ok)
|
|
306
|
+
rescue StandardError
|
|
307
|
+
ServerProps.new(body: nil, status: :network_error)
|
|
308
|
+
end
|
|
309
|
+
|
|
310
|
+
def parse_props(body)
|
|
311
|
+
JSON.parse(body.to_s)
|
|
312
|
+
rescue JSON::ParserError
|
|
313
|
+
nil
|
|
314
|
+
end
|
|
315
|
+
|
|
316
|
+
def resolve_transport(transport)
|
|
317
|
+
value = (transport || ENV.fetch(SERVER_TRANSPORT_ENV, DEFAULT_TRANSPORT.to_s)).to_s.strip.downcase.to_sym
|
|
318
|
+
VALID_TRANSPORTS.include?(value) ? value : DEFAULT_TRANSPORT
|
|
319
|
+
end
|
|
320
|
+
|
|
321
|
+
# Build the wire-format strategy for a resolved transport name. oMLX gets
|
|
322
|
+
# a resolver that maps the short selector to the exact /v1/models id (see
|
|
323
|
+
# #resolve_omlx_model); mlx gets a resolver that always returns nil so the
|
|
324
|
+
# `model` field is omitted; llama.cpp uses the strategy's default (forward
|
|
325
|
+
# the selector verbatim).
|
|
326
|
+
def build_transport(name)
|
|
327
|
+
resolver = case name
|
|
328
|
+
when :omlx
|
|
329
|
+
->(model) { resolve_omlx_model(model) }
|
|
330
|
+
when :mlx
|
|
331
|
+
->(_model) { nil }
|
|
332
|
+
end
|
|
333
|
+
Transport.new(name, model_resolver: resolver)
|
|
334
|
+
end
|
|
335
|
+
|
|
336
|
+
def completion_uri
|
|
337
|
+
URI("#{@scheme}://#{@host}:#{@port}#{@transport.completion_path}")
|
|
338
|
+
end
|
|
339
|
+
|
|
340
|
+
def completion_payload(prompt, stop:, n_predict:, model:)
|
|
341
|
+
payload = { prompt: prompt, stop: stop, stream: true }
|
|
342
|
+
payload[@transport.token_limit_key] = n_predict if n_predict && n_predict.to_i.positive?
|
|
343
|
+
model_name = @transport.model_for_payload(model)
|
|
344
|
+
payload[:model] = model_name if model_name
|
|
345
|
+
payload
|
|
346
|
+
end
|
|
347
|
+
|
|
348
|
+
# Conversation content (system prompt + tool responses + model output) can
|
|
349
|
+
# contain invalid UTF-8 — e.g. a shell/file op writes a garbled multibyte
|
|
350
|
+
# sequence (a truncated em-dash, a stray replacement byte). `JSON#to_json`
|
|
351
|
+
# raises `JSON::GeneratorError` on such input, which would abort the whole
|
|
352
|
+
# turn. Scrub the payload first: only offending bytes are replaced with "?",
|
|
353
|
+
# every valid UTF-8 string passes through untouched. Non-UTF-8 encodings are
|
|
354
|
+
# left alone because `#to_json` already handles them without raising.
|
|
355
|
+
def scrub_utf8(obj)
|
|
356
|
+
case obj
|
|
357
|
+
when String
|
|
358
|
+
str = obj.to_s
|
|
359
|
+
str.encoding == Encoding::UTF_8 && !str.valid_encoding? ? str.scrub("?") : str
|
|
360
|
+
when Array
|
|
361
|
+
obj.map { |element| scrub_utf8(element) }
|
|
362
|
+
when Hash
|
|
363
|
+
obj.each_with_object({}) { |(key, value), memo| memo[scrub_utf8(key)] = scrub_utf8(value) }
|
|
364
|
+
else
|
|
365
|
+
obj
|
|
366
|
+
end
|
|
367
|
+
end
|
|
368
|
+
|
|
369
|
+
# Resolve a user-facing SAMAGOTCHI_DEFAULT_MODEL selector to the exact id oMLX
|
|
370
|
+
# expects in the request body (an id from its `/v1/models` list).
|
|
371
|
+
#
|
|
372
|
+
# Resolution order: exact (case-insensitive) match first, then the first
|
|
373
|
+
# substring match, else the selector passes through unchanged so oMLX returns
|
|
374
|
+
# its own 404 listing the available models. An empty selector yields nil (no
|
|
375
|
+
# model field). The id list is loaded once per client and memoized, but
|
|
376
|
+
# resolution runs on every completion so a runtime model switch re-resolves.
|
|
377
|
+
def resolve_omlx_model(raw)
|
|
378
|
+
value = raw.to_s.strip
|
|
379
|
+
return nil if value.empty?
|
|
380
|
+
|
|
381
|
+
ids = fetch_omlx_model_ids
|
|
382
|
+
return value if ids.empty?
|
|
383
|
+
|
|
384
|
+
ids.find { |id| id.casecmp?(value) } ||
|
|
385
|
+
ids.select { |id| id.downcase.include?(value.downcase) }.first ||
|
|
386
|
+
value
|
|
387
|
+
end
|
|
388
|
+
|
|
389
|
+
# Load and memoize the list of model ids from oMLX's `/v1/models`.
|
|
390
|
+
#
|
|
391
|
+
# Only memoize on success: a failed `list_models` must leave the cache unset
|
|
392
|
+
# so the next completion retries (a transient blip shouldn't disable
|
|
393
|
+
# resolution for the whole session). On failure we return [] so an unknown
|
|
394
|
+
# selector still passes through raw, letting oMLX return its own 400/404
|
|
395
|
+
# (server decides), as documented in docs/configuration.md.
|
|
396
|
+
def fetch_omlx_model_ids
|
|
397
|
+
return @omlx_model_ids if defined?(@omlx_model_ids)
|
|
398
|
+
|
|
399
|
+
begin
|
|
400
|
+
ids = list_models.map { |m| m.is_a?(Hash) ? m["id"] : m }
|
|
401
|
+
rescue StandardError
|
|
402
|
+
return []
|
|
403
|
+
end
|
|
404
|
+
|
|
405
|
+
@omlx_model_ids = ids
|
|
406
|
+
ids
|
|
407
|
+
end
|
|
408
|
+
|
|
409
|
+
# Returns [content, payload] for a streamed SSE line, or nil to skip
|
|
410
|
+
# (blank lines, non-data lines, and the mlx/oMLX `[DONE]` sentinel).
|
|
411
|
+
# Raises the ProviderError of a server's error event.
|
|
412
|
+
def parse_stream_line(line)
|
|
413
|
+
error = LLM::HTTP.sse_error(line, host: @label)
|
|
414
|
+
raise error if error
|
|
415
|
+
return nil if line.empty? || !line.start_with?("data: ")
|
|
416
|
+
|
|
417
|
+
data = line.delete_prefix("data: ")
|
|
418
|
+
return nil if data == "[DONE]"
|
|
419
|
+
|
|
420
|
+
payload = begin
|
|
421
|
+
JSON.parse(data)
|
|
422
|
+
rescue JSON::ParserError => e
|
|
423
|
+
raise LLM::ProtocolError.new("#{@label}: malformed stream chunk: #{e.message[0, 200]}", host: @label)
|
|
424
|
+
end
|
|
425
|
+
content = @transport.content_from_payload(payload)
|
|
426
|
+
[content, payload]
|
|
427
|
+
end
|
|
428
|
+
end
|
|
429
|
+
end
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Samagotchi
|
|
4
|
+
module Commands
|
|
5
|
+
# The slash (and bang) commands a session knows, in lookup order: the
|
|
6
|
+
# built-ins SessionCommands registers, and later the ones bundles add.
|
|
7
|
+
#
|
|
8
|
+
# An entry the UI runs itself (/stats, /exit, …) is +local+: it is listed
|
|
9
|
+
# for Tab completion and help, and #lookup never returns it.
|
|
10
|
+
class Registry
|
|
11
|
+
# @!attribute id [Symbol] the entry's key (:model, :rollback, …)
|
|
12
|
+
# @!attribute name [String] "/model", "!rollback", …
|
|
13
|
+
# @!attribute anytime [Boolean] may run while a turn runs
|
|
14
|
+
# @!attribute local [Boolean] the UI runs it; completion and help only
|
|
15
|
+
# @!attribute uis [Array<Symbol>, nil] a local entry's UIs (:repl,
|
|
16
|
+
# :attached) for completion; nil means every UI
|
|
17
|
+
# @!attribute match [#call] line (stripped) → whether it is this command
|
|
18
|
+
# @!attribute handler [Proc, nil] runs it; SessionCommands instance_execs
|
|
19
|
+
# a built-in's with the line, and calls a bundle's with the text
|
|
20
|
+
# after the name
|
|
21
|
+
# @!attribute source [String] "core", or the bundle that added it
|
|
22
|
+
Entry = Struct.new(:id, :name, :description, :anytime, :local, :uis, :match, :handler, :source,
|
|
23
|
+
keyword_init: true) do
|
|
24
|
+
def match?(text) = match.call(text)
|
|
25
|
+
def slash? = name.start_with?("/")
|
|
26
|
+
def in_ui?(ui) = uis.nil? || uis.include?(ui)
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
def initialize
|
|
30
|
+
@entries = []
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
# A UI without an Engine (attached) learns the session's commands from
|
|
34
|
+
# its snapshot (#listing): +base+'s entries as they are (their match
|
|
35
|
+
# and ids), then each listed one +base+ lacks, matched by its name
|
|
36
|
+
# (a bundle's command; the worker runs it).
|
|
37
|
+
# @param listing [Array<Hash>] #listing, with String or Symbol keys
|
|
38
|
+
# @return [Registry]
|
|
39
|
+
def self.from_listing(listing, base:)
|
|
40
|
+
registry = new
|
|
41
|
+
base.entries.each { |entry| registry.send(:add, entry) }
|
|
42
|
+
known = base.entries.map(&:name)
|
|
43
|
+
Array(listing).each do |item|
|
|
44
|
+
item = item.transform_keys(&:to_sym)
|
|
45
|
+
name = item[:name].to_s
|
|
46
|
+
next if name.empty? || known.include?(name)
|
|
47
|
+
|
|
48
|
+
known << name
|
|
49
|
+
registry.register(name, item[:description].to_s, anytime: item[:anytime] == true, local: item[:local] == true,
|
|
50
|
+
uis: item[:uis]&.map(&:to_sym), source: item[:source] || "core")
|
|
51
|
+
end
|
|
52
|
+
registry
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
# @param match [#call, nil] the default matches the name alone or the
|
|
56
|
+
# name, a space and arguments
|
|
57
|
+
# @return [Entry]
|
|
58
|
+
def register(name, description, id: nil, anytime: false, local: false, uis: nil, match: nil,
|
|
59
|
+
source: "core", &handler)
|
|
60
|
+
raise ArgumentError, "command #{name} is already registered" if @entries.any? { |entry| entry.name == name }
|
|
61
|
+
|
|
62
|
+
entry = Entry.new(id: id || name.delete_prefix("/").to_sym, name: name, description: description,
|
|
63
|
+
anytime: anytime, local: local, uis: uis, match: match || default_match(name),
|
|
64
|
+
handler: handler, source: source)
|
|
65
|
+
@entries << entry
|
|
66
|
+
entry
|
|
67
|
+
end
|
|
68
|
+
|
|
69
|
+
# @return [Entry, nil] the first entry the session runs that matches +line+
|
|
70
|
+
def lookup(line)
|
|
71
|
+
text = line.to_s.strip
|
|
72
|
+
@entries.find { |entry| !entry.local && entry.match?(text) }
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
def command?(line) = !lookup(line).nil?
|
|
76
|
+
|
|
77
|
+
# @return [Array<Entry>] every entry, in registration order
|
|
78
|
+
def entries = @entries.dup
|
|
79
|
+
|
|
80
|
+
# What a snapshot tells a UI without an Engine (see .from_listing).
|
|
81
|
+
# @return [Array<Hash>] {name:, description:, anytime:, local:, uis:,
|
|
82
|
+
# source:} per entry, in registration order
|
|
83
|
+
def listing
|
|
84
|
+
@entries.map do |entry|
|
|
85
|
+
{ name: entry.name, description: entry.description, anytime: entry.anytime ? true : false,
|
|
86
|
+
local: entry.local ? true : false, uis: entry.uis&.map(&:to_s), source: entry.source }
|
|
87
|
+
end
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
# @param ui [Symbol] :repl or :attached
|
|
91
|
+
# @return [Array<String>] the /names Tab offers in +ui+, sorted
|
|
92
|
+
def completions(ui)
|
|
93
|
+
@entries.select { |entry| entry.slash? && entry.in_ui?(ui) }.map(&:name).uniq.sort
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
def freeze
|
|
97
|
+
@entries.freeze
|
|
98
|
+
super
|
|
99
|
+
end
|
|
100
|
+
|
|
101
|
+
private
|
|
102
|
+
|
|
103
|
+
def add(entry)
|
|
104
|
+
@entries << entry
|
|
105
|
+
end
|
|
106
|
+
|
|
107
|
+
def default_match(name)
|
|
108
|
+
->(text) { text == name || text.start_with?("#{name} ") }
|
|
109
|
+
end
|
|
110
|
+
end
|
|
111
|
+
end
|
|
112
|
+
end
|