samagotchi 0.2.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (96) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +116 -1
  3. data/README.md +40 -4
  4. data/bin/chi +86 -19
  5. data/docs/cli.md +108 -5
  6. data/docs/configuration.md +229 -45
  7. data/docs/desktop.md +6 -0
  8. data/docs/hooks.md +126 -5
  9. data/docs/plugins.md +68 -2
  10. data/docs/releasing.md +18 -8
  11. data/docs/sessions.md +30 -4
  12. data/lib/samagotchi/answer_display.rb +95 -0
  13. data/lib/samagotchi/archive_store.rb +90 -0
  14. data/lib/samagotchi/bootstrap/config_writer.rb +342 -0
  15. data/lib/samagotchi/bootstrap/probe.rb +262 -0
  16. data/lib/samagotchi/bootstrap_command.rb +347 -0
  17. data/lib/samagotchi/bridge/pending_card.rb +89 -0
  18. data/lib/samagotchi/bridge/turn_accumulator.rb +14 -3
  19. data/lib/samagotchi/bridge.rb +9 -0
  20. data/lib/samagotchi/bridge_client.rb +6 -2
  21. data/lib/samagotchi/bundles/check-in/manifest.yml +10 -0
  22. data/lib/samagotchi/bundles/check-in/plugin.rb +244 -0
  23. data/lib/samagotchi/bundles/source-links/hooks/source_links.rb +358 -0
  24. data/lib/samagotchi/bundles/source-links/manifest.yml +14 -0
  25. data/lib/samagotchi/bundles/source-links/source_links.md +5 -0
  26. data/lib/samagotchi/bundles/system/config_modification_protocol.md +10 -6
  27. data/lib/samagotchi/bundles/system/manifest.yml +3 -3
  28. data/lib/samagotchi/bundles/system/self_map.md +8 -2
  29. data/lib/samagotchi/client.rb +72 -13
  30. data/lib/samagotchi/config.rb +196 -36
  31. data/lib/samagotchi/desktop/macos/ChiRunner.swift +13 -7
  32. data/lib/samagotchi/desktop/macos/Panel.swift +71 -19
  33. data/lib/samagotchi/empty_answer_retry.rb +43 -0
  34. data/lib/samagotchi/engine.rb +233 -36
  35. data/lib/samagotchi/guardrails/approval.rb +9 -0
  36. data/lib/samagotchi/guardrails/scratch_writes.rb +40 -0
  37. data/lib/samagotchi/guardrails.rb +1 -0
  38. data/lib/samagotchi/hooks/registry.rb +24 -5
  39. data/lib/samagotchi/host_registry.rb +4 -3
  40. data/lib/samagotchi/idle_recap.rb +5 -1
  41. data/lib/samagotchi/kernel_loop.rb +47 -15
  42. data/lib/samagotchi/llm/chat_loop.rb +59 -20
  43. data/lib/samagotchi/llm/errors.rb +21 -3
  44. data/lib/samagotchi/llm/http.rb +42 -13
  45. data/lib/samagotchi/llm/openai_chat.rb +12 -4
  46. data/lib/samagotchi/log_subscriber.rb +18 -3
  47. data/lib/samagotchi/model_profile.rb +1 -1
  48. data/lib/samagotchi/plugin/context.rb +22 -1
  49. data/lib/samagotchi/plugin/sessions.rb +3 -1
  50. data/lib/samagotchi/reply_wait.rb +126 -0
  51. data/lib/samagotchi/sampling_settings.rb +58 -0
  52. data/lib/samagotchi/self_report.rb +1 -0
  53. data/lib/samagotchi/send_command.rb +153 -7
  54. data/lib/samagotchi/session.rb +52 -11
  55. data/lib/samagotchi/session_archive_command.rb +107 -0
  56. data/lib/samagotchi/session_commands.rb +11 -2
  57. data/lib/samagotchi/session_manager.rb +114 -9
  58. data/lib/samagotchi/session_metrics.rb +222 -106
  59. data/lib/samagotchi/steer.rb +72 -0
  60. data/lib/samagotchi/terminal_ui/attached_loop.rb +39 -6
  61. data/lib/samagotchi/terminal_ui/event_renderer.rb +13 -8
  62. data/lib/samagotchi/terminal_ui/formatting.rb +31 -8
  63. data/lib/samagotchi/terminal_ui/input_support.rb +3 -0
  64. data/lib/samagotchi/terminal_ui.rb +77 -4
  65. data/lib/samagotchi/tool_activity.rb +3 -1
  66. data/lib/samagotchi/tools/builtins.rb +15 -4
  67. data/lib/samagotchi/tools/delegate_wait.rb +26 -69
  68. data/lib/samagotchi/tools/execute.rb +52 -14
  69. data/lib/samagotchi/tools/task_runtime.rb +19 -0
  70. data/lib/samagotchi/tools/task_wait.rb +27 -3
  71. data/lib/samagotchi/turn_note.rb +60 -6
  72. data/lib/samagotchi/version.rb +1 -1
  73. data/lib/samagotchi/vision_support.rb +2 -6
  74. data/lib/samagotchi/web/app.rb +88 -4
  75. data/lib/samagotchi/web/public/activity.js +10 -1
  76. data/lib/samagotchi/web/public/annotate_presets.js +26 -0
  77. data/lib/samagotchi/web/public/annotations.js +13 -0
  78. data/lib/samagotchi/web/public/app.js +437 -88
  79. data/lib/samagotchi/web/public/card.js +5 -3
  80. data/lib/samagotchi/web/public/chat_view.js +10 -1
  81. data/lib/samagotchi/web/public/copy.js +20 -4
  82. data/lib/samagotchi/web/public/ctx.js +15 -0
  83. data/lib/samagotchi/web/public/data.js +21 -6
  84. data/lib/samagotchi/web/public/format.js +9 -0
  85. data/lib/samagotchi/web/public/index.html +38 -2
  86. data/lib/samagotchi/web/public/notify.js +175 -0
  87. data/lib/samagotchi/web/public/question_card.js +2 -1
  88. data/lib/samagotchi/web/public/sessions_list.js +7 -0
  89. data/lib/samagotchi/web/public/timing.js +39 -14
  90. data/lib/samagotchi/web/public/turn_events.js +46 -0
  91. data/lib/samagotchi/web/public/turn_view.js +47 -7
  92. data/lib/samagotchi/web/server.rb +8 -4
  93. data/lib/samagotchi/web/session_hub.rb +2 -1
  94. data/lib/samagotchi/web/session_summary.rb +24 -1
  95. data/lib/samagotchi/worker.rb +11 -0
  96. metadata +20 -1
@@ -18,7 +18,7 @@ Single registry `Samagotchi::Config` (`Config::ENTRIES` in `lib/samagotchi/confi
18
18
 
19
19
  Sections forbid `_`/`-` (`SECTION_RE` `/\A[a-z0-9]+\z/`); leaves keep `snake_case` in YAML (`base_url`) and become kebab in CLI (`base-url`) via registry derivation — no generic string split, registry lookup avoids flat vs nested collision.
20
20
 
21
- **Universal entries** (`expose: [:env,:config,:cli]`): `default.model`, `server.host/port/transport/open_timeout/read_timeout`, `server.first_token_timeout` (env and config only), `recap.model/base_url/host_ref/inactivity/timeout/min_user_turns/sentences`, `session.retention_days/max_count/keep_status/sweep_interval_hours/idle_exit_minutes`, `session.shared/keep_empty/max_children` (env and config only), `log.file/disable`, `status.line/width_mode/max_width/fixed_width`, `context.status/window_tokens/chars_per_token/status_thresholds/status_cadence`, `thinking.ui/preview_lines/render_interval`, `n_predict`, `max_tool_output_chars`, `retry.max/base_delay/max_delay`, `read.*`, `execute.*`, `web.port/host`, `no_interrupt`, `no_default_input` etc. (`Config::ENTRIES`; `expose` says which of env/config/cli each takes). Precedence is `CLI > ENV > file > default`.
21
+ **Universal entries** (`expose: [:env,:config,:cli]`): `default.model`, `server.host/port/transport/open_timeout/read_timeout`, `server.first_token_timeout` (env and config only), `recap.model/base_url/host_ref/inactivity/timeout/min_user_turns/sentences`, `session.retention_days/max_count/keep_status/sweep_interval_hours/idle_exit_minutes`, `session.shared/keep_empty/max_children` (env and config only), `log.file/disable`, `status.line/width_mode/max_width/fixed_width`, `context.status/window_tokens/chars_per_token/status_thresholds/status_cadence`, `thinking.ui/render_interval/turn_preamble`, `default.n_predict`, `max_tool_output_chars`, `retry.max/base_delay/max_delay`, `read.*`, `execute.*`, `web.port/host`, `no_interrupt`, `no_default_input` etc. (`Config::ENTRIES`; `expose` says which of env/config/cli each takes). Precedence is `CLI > ENV > file > default`.
22
22
 
23
23
  Example `config.yml` (new nested form, preferred):
24
24
 
@@ -52,13 +52,13 @@ log:
52
52
  disable: false
53
53
  ```
54
54
 
55
- Legacy flat keys (`SAMAGOTCHI_DEFAULT_MODEL`, `SAMAGOTCHI_N_PREDICT` etc. at top-level) are still read via fallback in `Config.lookup_yaml` but warn `Warning: config key 'SAMAGOTCHI_DEFAULT_MODEL' is legacy UPPER — use 'default.model'` (`ConfigFile.load_global_env!`). Migrate them to nested form and remove the flat entry. The old `LLAMA_HOST`/`LLAMA_PORT` aliases were removed; use `server.host`/`server.port` (nested) or `SAMAGOTCHI_SERVER_HOST`/`SAMAGOTCHI_SERVER_PORT`.
55
+ Legacy flat keys (`SAMAGOTCHI_DEFAULT_MODEL`, `SAMAGOTCHI_N_PREDICT` etc. at top-level) are still read via fallback in `Config.lookup_yaml` but warn `Warning: config key 'SAMAGOTCHI_DEFAULT_MODEL' is legacy UPPER — use 'default.model'` (`ConfigFile.load_global_env!`). When a file has both, the nested key wins and the warning names both. Migrate them to nested form and remove the flat entry. The old `LLAMA_HOST`/`LLAMA_PORT` aliases were removed; use `server.host`/`server.port` (nested) or `SAMAGOTCHI_SERVER_HOST`/`SAMAGOTCHI_SERVER_PORT`.
56
56
 
57
57
  **Excluded maps** (YAML-only, not part of the flat registry; skipped by scalar loader):
58
58
 
59
59
  - `model_aliases:` map of alias → model id (`ConfigFile.resolve_model_alias`). Keys lowercased on write (`ConfigFile.write_model_alias!`). Values may be bare `model` or qualified `host:model` (hybrid).
60
- - `hosts:` map of `name → {host, port | url, transport, api, api_key_env, profile, first_token_timeout, enabled}` (`ConfigFile.hosts_config`, `host_registry.rb` `HostEntry`). Names lowercased; `url:` (http/https, optional path) replaces host/port, never both; `api_key_env:` names the env var holding the API key (never write a key into config.yml); `transport` overrides `server.transport`; `first_token_timeout` (seconds, `0` = off; a negative or non-number warns and is ignored) overrides `server.first_token_timeout` for that host; workers inherit via `SAMAGOTCHI_HOSTS_JSON` (`hosts_json_for_env`, `session_manager.rb`).
61
- - `models:` map of model id or alias → `{profile}` (`ConfigFile.model_settings`). Keys match case-insensitively. `profile` (here or on a host) is `qwen36|gemma4`: the raw prompt format for native hosts. Precedence: `--profile`/`SAMAGOTCHI_MODEL_PROFILE` > `models:` > `hosts.<name>.profile` > the llama.cpp server's chat template > the name (`qwen`/`gemma`) > `qwen36` (`ModelProfile.resolve`). Set one when a model's name hides its family (e.g. a Qwen fine-tune under another name on mlx, which has no template to read).
60
+ - `hosts:` map of `name → {host, port | url, transport, api, api_key_env, profile, first_token_timeout, vision, sampling, enabled}` (`ConfigFile.hosts_config`, `host_registry.rb` `HostEntry`). Names lowercased; `url:` (http/https, optional path) replaces host/port, never both; `api_key_env:` names the env var holding the API key (never write a key into config.yml); `transport` overrides `server.transport`; `first_token_timeout` (seconds, `0` = off; a negative or non-number warns and is ignored) overrides `server.first_token_timeout` for that host; workers inherit via `SAMAGOTCHI_HOSTS_JSON` (`hosts_json_for_env`, `session_manager.rb`).
61
+ - `models:` map of model id or alias → `{profile, vision, sampling}` (`ConfigFile.model_settings`). `sampling:` (here or on a host) is a map of request fields passed to the provider as written (`temperature`, `top_p`, `presence_penalty`, `repeat_penalty`, …; a model's fields win over its host's per field; `null` = don't send; chi's own fields like `max_tokens`/`stream` are refused with a warning; `docs/configuration.md` "Sampling"). Keys match case-insensitively. `profile` (here or on a host) is `qwen36|gemma4`: the raw prompt format for native hosts. Precedence: `--profile`/`SAMAGOTCHI_MODEL_PROFILE` > `models:` > `hosts.<name>.profile` > the llama.cpp server's chat template > the name (`qwen`/`gemma`) > `qwen36` (`ModelProfile.resolve`). Set one when a model's name hides its family (e.g. a Qwen fine-tune under another name on mlx, which has no template to read).
62
62
  - `hooks:` map of `hooks_dir` + per-event lists `{path, on_error}` (`Hooks::Loader.load`). `hooks_dir` may start with `~`.
63
63
  - `guardrails:` tool-call rules (`docs/guardrails.md`; read in `Engine#guardrail_rules`, parsed by `Guardrails::Rules.parse`): `enabled` (bool, default true; `false` drops rules and hooks' asks, a deny still applies; env `SAMAGOTCHI_GUARDRAILS_ENABLED`), `rules:` (list), `disable:` (list of rule ids, `id` or `bundle:id`, switching off a bundle's or config rule without editing it). A rule takes only `id`, `tool`, `command`, `path`, `verdict`, `reason`, `scopes`: `id` required; at least one of `tool` (a name, `shell` = execute + task_create, a `File.fnmatch` glob like `"mcp_*"`, or a list), `command` (a Ruby regex on the shell command), `path` (a glob, or `outside_repo`); `verdict` `ask|deny`; `scopes` (for `ask`) a subset of `once, session, repo, rule`. **Any parse error (an unknown key, a bad regex, no verdict) makes chi deny every tool call** until fixed, so validate with `YAML.safe_load` and keep the list shape. Rules load when a session starts: restart the worker/REPL after an edit. Installed bundles' rule files (`chi bundle install guardrails`) add to them; `/guardrails` lists what loaded.
64
64
 
@@ -92,6 +92,10 @@ bundles:
92
92
  deny_after: 2 # same call + same result N times in a turn -> deny the next
93
93
  stop_after: 4 # stop the turn at this many denies
94
94
  mode: deny # deny | notify (warn only); ignore_tools: [task_wait, ...]
95
+ check-in:
96
+ after: 50 # tool calls in one turn with no answer before the first check
97
+ every: 50 # then again every N more
98
+ mode: ask # ask (a card) | nudge (nudge the model by itself) | notify; message:, ignore_tools: [...]
95
99
  mcp: # tools become mcp_<server>_<tool>; /mcp lists them
96
100
  timeout: 60 # per call, seconds; startup_timeout: 10
97
101
  servers:
@@ -113,7 +117,7 @@ A guardrail rule's `tool:` may be a glob (`tool: "mcp_*"`, verdict `ask`) to cov
113
117
  1. **Read** the current file via `read` tool (or `ConfigFile.global_path`). If `File.file?` false, start from `{}`.
114
118
  2. `YAML.safe_load` (permitted_classes: [], aliases: false). If data nil or not Hash, treat as `{}` or raise with path.
115
119
  3. Mutate the intended **nested** key in the raw hash. Preserve all other keys byte-for-byte where possible. Example for default model: `raw_data["default"] ||= {}; raw_data["default"]["model"] = "new-model"; raw_data.delete("SAMAGOTCHI_DEFAULT_MODEL")` to migrate legacy.
116
- 4. **Validate** (see below) before writing. Also run `Samagotchi::Config.validate_yaml_sections` — it rejects top-level `_` (suggest `default.model`) and warns on legacy flat keys.
120
+ 4. **Validate** (see below) before writing. Also run `Samagotchi::Config.validate_yaml_sections` — it returns one `config: unknown key '…' (did you mean '…'?)` per key chi doesn't read (every config-exposed `Config::ENTRIES` key is known as written, including the section-less `max_tool_output_chars` and `skip_agent_md`; names under `hosts:`/`models:`/`model_aliases:`/`hooks:`/`bundles:`/`memories:` are free-form, host and model entries are checked against `Config::MAP_ENTRY_KEYS`). An empty list means no warning at start.
117
121
  5. **Write atomically**: `FileUtils.mkdir_p(File.dirname(path))`, `File.write("#{path}.tmp", YAML.dump(raw_data))`, `File.rename("#{path}.tmp", path)`.
118
122
  6. Update in-process state: `write_default_model!` sets `ENV["SAMAGOTCHI_DEFAULT_MODEL"]` and `Samagotchi::Config.reload!`; otherwise the harness picks it up on next `Config.get` (live resolve) or restart. CLI overrides (`--default-model`) win over file until process exit.
119
123
 
@@ -132,7 +136,7 @@ A guardrail rule's `tool:` may be a glob (`tool: "mcp_*"`, verdict `ask`) to cov
132
136
  - **Hosts**: each entry needs `host`, `port` 1-65535, `transport` and `api` optional (a raw `api` must match `transport`), name must match `/\A[a-z0-9][a-z0-9._-]*\z/i`.
133
137
  - **Hooks**: each entry must have `path` (relative to `hooks_dir`), `on_error` is `skip` (default) or `log`. Class name must match file basename snake→Pascal.
134
138
  - **Scalars via registry**: `Config.coerce` validates `String/Numeric/true/false` per `type: :string/:integer/:float/:bool/:enum`; invalid values warn and fall back to entry `default`.
135
- - **Sections**: `validate_yaml_sections` rejects top-level keys containing `_` (suggest dotted) and section names containing `_`/`-`.
139
+ - **Keys**: `validate_yaml_sections` flags unknown keys (with a suggestion), a registry section that isn't a mapping, and env/CLI-only keys (`model.profile`).
136
140
 
137
141
  ## Tools to use
138
142
 
@@ -1,11 +1,11 @@
1
1
  ---
2
2
  name: samagotchi-system
3
- version: 0.2.0
3
+ version: 0.3.0
4
4
  scope: system
5
5
  description: Default system memories — identity, self map, config modification protocol, memory guide and the delegated-session rules
6
6
  files:
7
7
  identity.md: sha256:8b594100bee4797b563bd6425dc6f3ff93e1bdfbe9fc5fa1d66b42093c4795e2
8
- self_map.md: sha256:d42481b930ef6e59deb1e5c989a9a171e834569b42be9a8d3dda6b2379b0a5ff
9
- config_modification_protocol.md: sha256:c67e5fc1611a09203e3ebb1af2c14cacc2878b42a9fb0eb3d190c3b2daddafc3
8
+ self_map.md: sha256:67eb6223728a72788fa0d75aa56affea2ca48962de2e445293bc12b15f5f70c0
9
+ config_modification_protocol.md: sha256:1d1047f3f99fc9ceb89ef191cbcf3d68bc996addd00d7b6eb567a63b8ff677ea
10
10
  memory_guide.md: sha256:6dd1ecfd6a377c4f2cfc2dc299c1ef3835f10d9f313616fd3eb23e5fd3202858
11
11
  delegated.md: sha256:8f7965b6165c3526abb0cd39e44a8b73a8d6a26755049bdcd007baf0b0771087
@@ -13,11 +13,15 @@
13
13
  - Sessions: each is one file, `<sessions dir>/<id>.json`. A `<id>/` dir beside it exists only
14
14
  for background/web workers (input/, notes/, output/, pid). List them with `chi sessions list`
15
15
  (`--live` for the ones a worker runs now); delete one with `chi sessions delete ID` (never by hand).
16
+ An archived session (`chi sessions archive ID`, a marker file `<id>/archived`) is hidden from
17
+ every list and kept for good; `chi sessions list --archived` shows it.
16
18
  - `[CONTEXT NOTE from …]` messages are context notes (`chi note`, or another session's
17
19
  `send_note`): background, not requests. `list_sessions` + `send_note` tell another session
18
20
  something without starting a turn there; see `docs/sessions.md` "Context notes".
19
21
  - `chi send -m TEXT <id>` is the other half: the text goes in as the user's message and a
20
22
  turn runs (stdin piped too = quoted context above it); see `docs/sessions.md` "Sending a message".
23
+ `chi send --new --wait -m TEXT` starts a session the user can watch in the web and prints its
24
+ answer (exit 3: it waits for the user's answer); see "Starting a session".
21
25
  - `delegate` hands a task to a child session that runs in parallel and returns only its final
22
26
  reply (`delegate_result` waits for it; `session:` sends a follow-up to a child). A child is a
23
27
  normal session: it shows in `chi sessions list` with `↳ <parent>`, and the user can steer it
@@ -26,8 +30,10 @@
26
30
  commands (`anytime:` ones run mid-turn), model tools (they can return images I see), hooks,
27
31
  cards, side answers (`ask_model`),
28
32
  child sessions and services (e.g. MCP server processes). Shipped: `btw` (`/btw` side question),
29
- `mcp` (MCP server tools; a screenshot comes as a picture), `guardrails` (rules), `known-names` (typo guard), `loop-guard`
30
- (denies a repeated tool call with the same result, stops the turn after a few). `chi bundle list`
33
+ `mcp` (MCP server tools; a screenshot comes as a picture), `guardrails` (rules), `known-names` (typo guard), `source-links`
34
+ (turns source refs like JIRA-123 in an answer into links in the web, and a one-line note), `loop-guard`
35
+ (denies a repeated tool call with the same result, stops the turn after a few), `check-in` (after N tool
36
+ calls with no answer, a card asks the user to nudge me, let me go on or stop; `/checkin`). `chi bundle list`
31
37
  shows installed + available; `chi bundle install <name>`. Settings: config.yml `bundles: <name>:`
32
38
  (`config_modification_protocol`), read at session start: after an install or a settings change,
33
39
  tell the user to restart the session. API and bundle docs: `docs/plugins.md`.
@@ -8,6 +8,7 @@ require_relative "cancellation_controller"
8
8
  require_relative "llm/http"
9
9
  require_relative "vision_context"
10
10
  require_relative "vision_support"
11
+ require_relative "sampling_settings"
11
12
 
12
13
  module Samagotchi
13
14
  # Thin HTTP client for llama.cpp's native /completion endpoint, or an
@@ -27,6 +28,10 @@ module Samagotchi
27
28
  # budget and no retry (see #server_props).
28
29
  CONTEXT_WINDOW_PROBE_OPEN_TIMEOUT = 1
29
30
  CONTEXT_WINDOW_PROBE_READ_TIMEOUT = 2
31
+ # Seconds a probe that got no answer (refused, timed out) is remembered:
32
+ # a down or hung server then costs one probe per window, not one per
33
+ # generation, and a new turn doesn't ask again at once.
34
+ PROPS_FAILURE_TTL = 30
30
35
 
31
36
  SERVER_TRANSPORT_ENV = "SAMAGOTCHI_SERVER_TRANSPORT"
32
37
  DEFAULT_TRANSPORT = :llama_cpp
@@ -109,7 +114,7 @@ module Samagotchi
109
114
  end
110
115
 
111
116
  # One /props probe's outcome. `answered?` is false when the probe failed:
112
- # a network error, a timeout or any non-200 (llama.cpp answers 503 while
117
+ # a network error, a timeout, a turn's cancel or any non-200 (llama.cpp answers 503 while
113
118
  # it loads a model). `body` is the parsed JSON of a 200, or nil when it
114
119
  # isn't JSON.
115
120
  ServerProps = Data.define(:body, :status) do
@@ -121,6 +126,21 @@ module Samagotchi
121
126
  # Moved to its own file; the old name keeps working.
122
127
  CancellationController = Samagotchi::CancellationController
123
128
 
129
+ PROBE_CANCEL_KEY = :samagotchi_probe_cancel
130
+
131
+ # The cancel a /props probe made on this thread listens to: a turn sets
132
+ # its own (Engine#run_turn), so a Stop cuts the probes it makes before
133
+ # its first request. Probes on other threads (chi self, /model, a
134
+ # status snapshot) have none and run to their timeout.
135
+ def self.probe_cancel = Thread.current[PROBE_CANCEL_KEY]
136
+
137
+ # Sets this thread's probe cancel; returns the one it replaces.
138
+ def self.swap_probe_cancel(controller)
139
+ previous = Thread.current[PROBE_CANCEL_KEY]
140
+ Thread.current[PROBE_CANCEL_KEY] = controller
141
+ previous
142
+ end
143
+
124
144
  # @param sleeper [#call, nil] waits between retries (specs pass a no-op)
125
145
  # @param scheme [String, nil] "https" for a TLS server (default http)
126
146
  # @param first_token_timeout [Numeric, nil] seconds a completion may take
@@ -148,6 +168,7 @@ module Samagotchi
148
168
  transport_fallback = cfg_transport_raw || ENV.fetch(SERVER_TRANSPORT_ENV, DEFAULT_TRANSPORT.to_s)
149
169
  @transport = build_transport(resolve_transport(transport || transport_fallback))
150
170
  @props_cache = {}
171
+ @props_failures = {}
151
172
  @props_mutex = Mutex.new
152
173
  @first_token_timeout = first_token_timeout
153
174
  @host_name = name
@@ -182,11 +203,14 @@ module Samagotchi
182
203
  # @param on_retry [Proc, nil] optional callback before retry sleep
183
204
  # @param images [Array<String>] base64 images, one per
184
205
  # ImagePlan::NATIVE_PLACEHOLDER in the prompt (llama.cpp only)
206
+ # @param sampling [Hash] request fields to add (SamplingSettings):
207
+ # temperature, penalties, …; they can't replace the fields above
185
208
  # @return [String] the generated text
186
209
  def complete(prompt, stop: ["<end_of_turn>", "<|tool_response>"], n_predict: nil, model: nil, on_chunk: nil, cancel_controller: nil, on_retry: nil,
187
- images: [])
210
+ images: [], sampling: {})
188
211
  images = Array(images)
189
- return stream_completion(scrub_utf8(prompt), stop, n_predict, model, on_chunk, cancel_controller, on_retry) if images.empty?
212
+ request = { stop: stop, n_predict: n_predict, model: model, sampling: sampling }
213
+ return stream_completion(scrub_utf8(prompt), request, on_chunk, cancel_controller, on_retry) if images.empty?
190
214
 
191
215
  # The media marker is random per server process: a restart between the
192
216
  # /props read and the request makes the prompt fail to tokenize, so
@@ -196,7 +220,7 @@ module Samagotchi
196
220
  attempts += 1
197
221
  payload_prompt = { prompt_string: scrub_utf8(prompt.gsub(ImagePlan::NATIVE_PLACEHOLDER, media_marker!(model))),
198
222
  multimodal_data: images }
199
- stream_completion(payload_prompt, stop, n_predict, model, on_chunk, cancel_controller, on_retry)
223
+ stream_completion(payload_prompt, request, on_chunk, cancel_controller, on_retry)
200
224
  rescue LLM::BadRequest => e
201
225
  raise unless attempts == 1 && e.message.include?("Failed to tokenize prompt")
202
226
 
@@ -205,11 +229,14 @@ module Samagotchi
205
229
  end
206
230
  end
207
231
 
208
- private def stream_completion(prompt, stop, n_predict, model, on_chunk, cancel_controller, on_retry)
232
+ OPENAI_API_HINT = "does this host speak the OpenAI API? set `api: openai` on it"
233
+
234
+ private def stream_completion(prompt, fields, on_chunk, cancel_controller, on_retry)
209
235
  uri = completion_uri
210
236
  request = Net::HTTP::Post.new(uri)
211
237
  request["Content-Type"] = "application/json"
212
- request.body = completion_payload(prompt, stop: stop, n_predict: n_predict, model: model).to_json
238
+ request.body = completion_payload(prompt, **fields).to_json
239
+ model = fields[:model]
213
240
 
214
241
  result = +""
215
242
  reset_on_retry = lambda do |event|
@@ -219,7 +246,8 @@ module Samagotchi
219
246
  end
220
247
  @http.stream_lines(uri, request, cancel_controller: cancel_controller, on_retry: reset_on_retry,
221
248
  on_network_error: ->(_error) { invalidate_context_window! },
222
- log_fields: { model: model, purpose: "chat" }) do |line, shown|
249
+ log_fields: { model: model, purpose: "chat",
250
+ sampling: SamplingSettings.log_text(sendable_sampling(fields[:sampling])) }) do |line, shown|
223
251
  parsed_chunk = parse_stream_line(line)
224
252
  next unless parsed_chunk
225
253
 
@@ -229,6 +257,11 @@ module Samagotchi
229
257
  on_chunk&.call(content: content, payload: payload)
230
258
  end
231
259
  result
260
+ rescue LLM::BadRequest => e
261
+ raise unless e.status == 404 && @transport.name == :llama_cpp
262
+
263
+ # No native /completion here: likely an OpenAI-compatible server.
264
+ raise LLM::BadRequest.new(e.message, host: e.host, status: e.status, attempts: e.attempts, hint: OPENAI_API_HINT)
232
265
  rescue RequestCancelled, LLM::ProviderError
233
266
  raise
234
267
  rescue StandardError => e
@@ -260,7 +293,9 @@ module Samagotchi
260
293
  # a turn. The probe names the model (`?model=`): a llama.cpp router
261
294
  # answers a stub without it, and a single-model server ignores it.
262
295
  # Whatever the server answers (a non-200 too) is cached per model; a
263
- # network failure is not, so the next call asks again.
296
+ # network failure for PROPS_FAILURE_TTL seconds, then the next call asks
297
+ # again. A probe cut by this thread's Client.probe_cancel answers
298
+ # :cancelled, uncached, and the turn's next request ends it.
264
299
  def server_props(model: nil)
265
300
  path = @transport.props_path
266
301
  return nil unless path
@@ -268,10 +303,21 @@ module Samagotchi
268
303
  key = model.to_s
269
304
  @props_mutex.synchronize do
270
305
  return @props_cache[key] if @props_cache.key?(key)
306
+
307
+ failed, failed_at = @props_failures[key]
308
+ return failed if failed && monotonic_now - failed_at < PROPS_FAILURE_TTL
271
309
  end
272
310
 
273
311
  props = probe_props(path, key)
274
- @props_mutex.synchronize { @props_cache[key] = props } unless props.status == :network_error
312
+ @props_mutex.synchronize do
313
+ case props.status
314
+ when :cancelled then nil
315
+ when :network_error then @props_failures[key] = [props, monotonic_now]
316
+ else
317
+ @props_cache[key] = props
318
+ @props_failures.delete(key)
319
+ end
320
+ end
275
321
  props
276
322
  end
277
323
 
@@ -286,23 +332,30 @@ module Samagotchi
286
332
 
287
333
  # Forget cached /props answers: the server may have restarted with
288
334
  # another -c, or a model switch may have loaded one with a different
289
- # window.
335
+ # window. A recent failure stays until its PROPS_FAILURE_TTL ends: it
336
+ # holds no stale window, and asking a hung server again at every turn's
337
+ # start is what it saves.
290
338
  def invalidate_context_window!
291
339
  @props_mutex.synchronize { @props_cache.clear }
292
340
  end
293
341
 
294
342
  private
295
343
 
344
+ def monotonic_now = Process.clock_gettime(Process::CLOCK_MONOTONIC)
345
+
296
346
  def probe_props(path, model)
297
347
  query = model.empty? ? "" : "?#{URI.encode_www_form(model: model)}"
298
348
  uri = URI("#{@scheme}://#{@host}:#{@port}#{path}#{query}")
299
349
  response = @http.fetch(uri, Net::HTTP::Get.new(uri), retries: false, check_status: false,
300
350
  log_fields: { model: model.empty? ? nil : model, purpose: "probe" },
301
351
  open_timeout: CONTEXT_WINDOW_PROBE_OPEN_TIMEOUT,
302
- read_timeout: CONTEXT_WINDOW_PROBE_READ_TIMEOUT)
352
+ read_timeout: CONTEXT_WINDOW_PROBE_READ_TIMEOUT,
353
+ cancel_controller: Client.probe_cancel)
303
354
  return ServerProps.new(body: nil, status: :http_error) unless response.code.to_s == "200"
304
355
 
305
356
  ServerProps.new(body: parse_props(response.body), status: :ok)
357
+ rescue RequestCancelled
358
+ ServerProps.new(body: nil, status: :cancelled)
306
359
  rescue StandardError
307
360
  ServerProps.new(body: nil, status: :network_error)
308
361
  end
@@ -337,12 +390,18 @@ module Samagotchi
337
390
  URI("#{@scheme}://#{@host}:#{@port}#{@transport.completion_path}")
338
391
  end
339
392
 
340
- def completion_payload(prompt, stop:, n_predict:, model:)
393
+ def completion_payload(prompt, stop:, n_predict:, model:, sampling: {})
341
394
  payload = { prompt: prompt, stop: stop, stream: true }
342
395
  payload[@transport.token_limit_key] = n_predict if n_predict && n_predict.to_i.positive?
343
396
  model_name = @transport.model_for_payload(model)
344
397
  payload[:model] = model_name if model_name
345
- payload
398
+ sendable_sampling(sampling).merge(payload)
399
+ end
400
+
401
+ # The sampling fields that go out: no nil (nothing to drop here: the
402
+ # server's defaults apply) and none of the reserved request fields.
403
+ def sendable_sampling(sampling)
404
+ (sampling || {}).reject { |key, value| value.nil? || ConfigFile::SAMPLING_RESERVED_KEYS.include?(key.to_s) }
346
405
  end
347
406
 
348
407
  # Conversation content (system prompt + tool responses + model output) can