samagotchi 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +116 -1
- data/README.md +40 -4
- data/bin/chi +86 -19
- data/docs/cli.md +108 -5
- data/docs/configuration.md +229 -45
- data/docs/desktop.md +6 -0
- data/docs/hooks.md +126 -5
- data/docs/plugins.md +68 -2
- data/docs/releasing.md +18 -8
- data/docs/sessions.md +30 -4
- data/lib/samagotchi/answer_display.rb +95 -0
- data/lib/samagotchi/archive_store.rb +90 -0
- data/lib/samagotchi/bootstrap/config_writer.rb +342 -0
- data/lib/samagotchi/bootstrap/probe.rb +262 -0
- data/lib/samagotchi/bootstrap_command.rb +347 -0
- data/lib/samagotchi/bridge/pending_card.rb +89 -0
- data/lib/samagotchi/bridge/turn_accumulator.rb +14 -3
- data/lib/samagotchi/bridge.rb +9 -0
- data/lib/samagotchi/bridge_client.rb +6 -2
- data/lib/samagotchi/bundles/check-in/manifest.yml +10 -0
- data/lib/samagotchi/bundles/check-in/plugin.rb +244 -0
- data/lib/samagotchi/bundles/source-links/hooks/source_links.rb +358 -0
- data/lib/samagotchi/bundles/source-links/manifest.yml +14 -0
- data/lib/samagotchi/bundles/source-links/source_links.md +5 -0
- data/lib/samagotchi/bundles/system/config_modification_protocol.md +10 -6
- data/lib/samagotchi/bundles/system/manifest.yml +3 -3
- data/lib/samagotchi/bundles/system/self_map.md +8 -2
- data/lib/samagotchi/client.rb +72 -13
- data/lib/samagotchi/config.rb +196 -36
- data/lib/samagotchi/desktop/macos/ChiRunner.swift +13 -7
- data/lib/samagotchi/desktop/macos/Panel.swift +71 -19
- data/lib/samagotchi/empty_answer_retry.rb +43 -0
- data/lib/samagotchi/engine.rb +233 -36
- data/lib/samagotchi/guardrails/approval.rb +9 -0
- data/lib/samagotchi/guardrails/scratch_writes.rb +40 -0
- data/lib/samagotchi/guardrails.rb +1 -0
- data/lib/samagotchi/hooks/registry.rb +24 -5
- data/lib/samagotchi/host_registry.rb +4 -3
- data/lib/samagotchi/idle_recap.rb +5 -1
- data/lib/samagotchi/kernel_loop.rb +47 -15
- data/lib/samagotchi/llm/chat_loop.rb +59 -20
- data/lib/samagotchi/llm/errors.rb +21 -3
- data/lib/samagotchi/llm/http.rb +42 -13
- data/lib/samagotchi/llm/openai_chat.rb +12 -4
- data/lib/samagotchi/log_subscriber.rb +18 -3
- data/lib/samagotchi/model_profile.rb +1 -1
- data/lib/samagotchi/plugin/context.rb +22 -1
- data/lib/samagotchi/plugin/sessions.rb +3 -1
- data/lib/samagotchi/reply_wait.rb +126 -0
- data/lib/samagotchi/sampling_settings.rb +58 -0
- data/lib/samagotchi/self_report.rb +1 -0
- data/lib/samagotchi/send_command.rb +153 -7
- data/lib/samagotchi/session.rb +52 -11
- data/lib/samagotchi/session_archive_command.rb +107 -0
- data/lib/samagotchi/session_commands.rb +11 -2
- data/lib/samagotchi/session_manager.rb +114 -9
- data/lib/samagotchi/session_metrics.rb +222 -106
- data/lib/samagotchi/steer.rb +72 -0
- data/lib/samagotchi/terminal_ui/attached_loop.rb +39 -6
- data/lib/samagotchi/terminal_ui/event_renderer.rb +13 -8
- data/lib/samagotchi/terminal_ui/formatting.rb +31 -8
- data/lib/samagotchi/terminal_ui/input_support.rb +3 -0
- data/lib/samagotchi/terminal_ui.rb +77 -4
- data/lib/samagotchi/tool_activity.rb +3 -1
- data/lib/samagotchi/tools/builtins.rb +15 -4
- data/lib/samagotchi/tools/delegate_wait.rb +26 -69
- data/lib/samagotchi/tools/execute.rb +52 -14
- data/lib/samagotchi/tools/task_runtime.rb +19 -0
- data/lib/samagotchi/tools/task_wait.rb +27 -3
- data/lib/samagotchi/turn_note.rb +60 -6
- data/lib/samagotchi/version.rb +1 -1
- data/lib/samagotchi/vision_support.rb +2 -6
- data/lib/samagotchi/web/app.rb +88 -4
- data/lib/samagotchi/web/public/activity.js +10 -1
- data/lib/samagotchi/web/public/annotate_presets.js +26 -0
- data/lib/samagotchi/web/public/annotations.js +13 -0
- data/lib/samagotchi/web/public/app.js +437 -88
- data/lib/samagotchi/web/public/card.js +5 -3
- data/lib/samagotchi/web/public/chat_view.js +10 -1
- data/lib/samagotchi/web/public/copy.js +20 -4
- data/lib/samagotchi/web/public/ctx.js +15 -0
- data/lib/samagotchi/web/public/data.js +21 -6
- data/lib/samagotchi/web/public/format.js +9 -0
- data/lib/samagotchi/web/public/index.html +38 -2
- data/lib/samagotchi/web/public/notify.js +175 -0
- data/lib/samagotchi/web/public/question_card.js +2 -1
- data/lib/samagotchi/web/public/sessions_list.js +7 -0
- data/lib/samagotchi/web/public/timing.js +39 -14
- data/lib/samagotchi/web/public/turn_events.js +46 -0
- data/lib/samagotchi/web/public/turn_view.js +47 -7
- data/lib/samagotchi/web/server.rb +8 -4
- data/lib/samagotchi/web/session_hub.rb +2 -1
- data/lib/samagotchi/web/session_summary.rb +24 -1
- data/lib/samagotchi/worker.rb +11 -0
- metadata +20 -1
|
@@ -18,7 +18,7 @@ Single registry `Samagotchi::Config` (`Config::ENTRIES` in `lib/samagotchi/confi
|
|
|
18
18
|
|
|
19
19
|
Sections forbid `_`/`-` (`SECTION_RE` `/\A[a-z0-9]+\z/`); leaves keep `snake_case` in YAML (`base_url`) and become kebab in CLI (`base-url`) via registry derivation — no generic string split, registry lookup avoids flat vs nested collision.
|
|
20
20
|
|
|
21
|
-
**Universal entries** (`expose: [:env,:config,:cli]`): `default.model`, `server.host/port/transport/open_timeout/read_timeout`, `server.first_token_timeout` (env and config only), `recap.model/base_url/host_ref/inactivity/timeout/min_user_turns/sentences`, `session.retention_days/max_count/keep_status/sweep_interval_hours/idle_exit_minutes`, `session.shared/keep_empty/max_children` (env and config only), `log.file/disable`, `status.line/width_mode/max_width/fixed_width`, `context.status/window_tokens/chars_per_token/status_thresholds/status_cadence`, `thinking.ui/
|
|
21
|
+
**Universal entries** (`expose: [:env,:config,:cli]`): `default.model`, `server.host/port/transport/open_timeout/read_timeout`, `server.first_token_timeout` (env and config only), `recap.model/base_url/host_ref/inactivity/timeout/min_user_turns/sentences`, `session.retention_days/max_count/keep_status/sweep_interval_hours/idle_exit_minutes`, `session.shared/keep_empty/max_children` (env and config only), `log.file/disable`, `status.line/width_mode/max_width/fixed_width`, `context.status/window_tokens/chars_per_token/status_thresholds/status_cadence`, `thinking.ui/render_interval/turn_preamble`, `default.n_predict`, `max_tool_output_chars`, `retry.max/base_delay/max_delay`, `read.*`, `execute.*`, `web.port/host`, `no_interrupt`, `no_default_input` etc. (`Config::ENTRIES`; `expose` says which of env/config/cli each takes). Precedence is `CLI > ENV > file > default`.
|
|
22
22
|
|
|
23
23
|
Example `config.yml` (new nested form, preferred):
|
|
24
24
|
|
|
@@ -52,13 +52,13 @@ log:
|
|
|
52
52
|
disable: false
|
|
53
53
|
```
|
|
54
54
|
|
|
55
|
-
Legacy flat keys (`SAMAGOTCHI_DEFAULT_MODEL`, `SAMAGOTCHI_N_PREDICT` etc. at top-level) are still read via fallback in `Config.lookup_yaml` but warn `Warning: config key 'SAMAGOTCHI_DEFAULT_MODEL' is legacy UPPER — use 'default.model'` (`ConfigFile.load_global_env!`). Migrate them to nested form and remove the flat entry. The old `LLAMA_HOST`/`LLAMA_PORT` aliases were removed; use `server.host`/`server.port` (nested) or `SAMAGOTCHI_SERVER_HOST`/`SAMAGOTCHI_SERVER_PORT`.
|
|
55
|
+
Legacy flat keys (`SAMAGOTCHI_DEFAULT_MODEL`, `SAMAGOTCHI_N_PREDICT` etc. at top-level) are still read via fallback in `Config.lookup_yaml` but warn `Warning: config key 'SAMAGOTCHI_DEFAULT_MODEL' is legacy UPPER — use 'default.model'` (`ConfigFile.load_global_env!`). When a file has both, the nested key wins and the warning names both. Migrate them to nested form and remove the flat entry. The old `LLAMA_HOST`/`LLAMA_PORT` aliases were removed; use `server.host`/`server.port` (nested) or `SAMAGOTCHI_SERVER_HOST`/`SAMAGOTCHI_SERVER_PORT`.
|
|
56
56
|
|
|
57
57
|
**Excluded maps** (YAML-only, not part of the flat registry; skipped by scalar loader):
|
|
58
58
|
|
|
59
59
|
- `model_aliases:` map of alias → model id (`ConfigFile.resolve_model_alias`). Keys lowercased on write (`ConfigFile.write_model_alias!`). Values may be bare `model` or qualified `host:model` (hybrid).
|
|
60
|
-
- `hosts:` map of `name → {host, port | url, transport, api, api_key_env, profile, first_token_timeout, enabled}` (`ConfigFile.hosts_config`, `host_registry.rb` `HostEntry`). Names lowercased; `url:` (http/https, optional path) replaces host/port, never both; `api_key_env:` names the env var holding the API key (never write a key into config.yml); `transport` overrides `server.transport`; `first_token_timeout` (seconds, `0` = off; a negative or non-number warns and is ignored) overrides `server.first_token_timeout` for that host; workers inherit via `SAMAGOTCHI_HOSTS_JSON` (`hosts_json_for_env`, `session_manager.rb`).
|
|
61
|
-
- `models:` map of model id or alias → `{profile}` (`ConfigFile.model_settings`). Keys match case-insensitively. `profile` (here or on a host) is `qwen36|gemma4`: the raw prompt format for native hosts. Precedence: `--profile`/`SAMAGOTCHI_MODEL_PROFILE` > `models:` > `hosts.<name>.profile` > the llama.cpp server's chat template > the name (`qwen`/`gemma`) > `qwen36` (`ModelProfile.resolve`). Set one when a model's name hides its family (e.g. a Qwen fine-tune under another name on mlx, which has no template to read).
|
|
60
|
+
- `hosts:` map of `name → {host, port | url, transport, api, api_key_env, profile, first_token_timeout, vision, sampling, enabled}` (`ConfigFile.hosts_config`, `host_registry.rb` `HostEntry`). Names lowercased; `url:` (http/https, optional path) replaces host/port, never both; `api_key_env:` names the env var holding the API key (never write a key into config.yml); `transport` overrides `server.transport`; `first_token_timeout` (seconds, `0` = off; a negative or non-number warns and is ignored) overrides `server.first_token_timeout` for that host; workers inherit via `SAMAGOTCHI_HOSTS_JSON` (`hosts_json_for_env`, `session_manager.rb`).
|
|
61
|
+
- `models:` map of model id or alias → `{profile, vision, sampling}` (`ConfigFile.model_settings`). `sampling:` (here or on a host) is a map of request fields passed to the provider as written (`temperature`, `top_p`, `presence_penalty`, `repeat_penalty`, …; a model's fields win over its host's per field; `null` = don't send; chi's own fields like `max_tokens`/`stream` are refused with a warning; `docs/configuration.md` "Sampling"). Keys match case-insensitively. `profile` (here or on a host) is `qwen36|gemma4`: the raw prompt format for native hosts. Precedence: `--profile`/`SAMAGOTCHI_MODEL_PROFILE` > `models:` > `hosts.<name>.profile` > the llama.cpp server's chat template > the name (`qwen`/`gemma`) > `qwen36` (`ModelProfile.resolve`). Set one when a model's name hides its family (e.g. a Qwen fine-tune under another name on mlx, which has no template to read).
|
|
62
62
|
- `hooks:` map of `hooks_dir` + per-event lists `{path, on_error}` (`Hooks::Loader.load`). `hooks_dir` may start with `~`.
|
|
63
63
|
- `guardrails:` tool-call rules (`docs/guardrails.md`; read in `Engine#guardrail_rules`, parsed by `Guardrails::Rules.parse`): `enabled` (bool, default true; `false` drops rules and hooks' asks, a deny still applies; env `SAMAGOTCHI_GUARDRAILS_ENABLED`), `rules:` (list), `disable:` (list of rule ids, `id` or `bundle:id`, switching off a bundle's or config rule without editing it). A rule takes only `id`, `tool`, `command`, `path`, `verdict`, `reason`, `scopes`: `id` required; at least one of `tool` (a name, `shell` = execute + task_create, a `File.fnmatch` glob like `"mcp_*"`, or a list), `command` (a Ruby regex on the shell command), `path` (a glob, or `outside_repo`); `verdict` `ask|deny`; `scopes` (for `ask`) a subset of `once, session, repo, rule`. **Any parse error (an unknown key, a bad regex, no verdict) makes chi deny every tool call** until fixed, so validate with `YAML.safe_load` and keep the list shape. Rules load when a session starts: restart the worker/REPL after an edit. Installed bundles' rule files (`chi bundle install guardrails`) add to them; `/guardrails` lists what loaded.
|
|
64
64
|
|
|
@@ -92,6 +92,10 @@ bundles:
|
|
|
92
92
|
deny_after: 2 # same call + same result N times in a turn -> deny the next
|
|
93
93
|
stop_after: 4 # stop the turn at this many denies
|
|
94
94
|
mode: deny # deny | notify (warn only); ignore_tools: [task_wait, ...]
|
|
95
|
+
check-in:
|
|
96
|
+
after: 50 # tool calls in one turn with no answer before the first check
|
|
97
|
+
every: 50 # then again every N more
|
|
98
|
+
mode: ask # ask (a card) | nudge (nudge the model by itself) | notify; message:, ignore_tools: [...]
|
|
95
99
|
mcp: # tools become mcp_<server>_<tool>; /mcp lists them
|
|
96
100
|
timeout: 60 # per call, seconds; startup_timeout: 10
|
|
97
101
|
servers:
|
|
@@ -113,7 +117,7 @@ A guardrail rule's `tool:` may be a glob (`tool: "mcp_*"`, verdict `ask`) to cov
|
|
|
113
117
|
1. **Read** the current file via `read` tool (or `ConfigFile.global_path`). If `File.file?` false, start from `{}`.
|
|
114
118
|
2. `YAML.safe_load` (permitted_classes: [], aliases: false). If data nil or not Hash, treat as `{}` or raise with path.
|
|
115
119
|
3. Mutate the intended **nested** key in the raw hash. Preserve all other keys byte-for-byte where possible. Example for default model: `raw_data["default"] ||= {}; raw_data["default"]["model"] = "new-model"; raw_data.delete("SAMAGOTCHI_DEFAULT_MODEL")` to migrate legacy.
|
|
116
|
-
4. **Validate** (see below) before writing. Also run `Samagotchi::Config.validate_yaml_sections` — it
|
|
120
|
+
4. **Validate** (see below) before writing. Also run `Samagotchi::Config.validate_yaml_sections` — it returns one `config: unknown key '…' (did you mean '…'?)` per key chi doesn't read (every config-exposed `Config::ENTRIES` key is known as written, including the section-less `max_tool_output_chars` and `skip_agent_md`; names under `hosts:`/`models:`/`model_aliases:`/`hooks:`/`bundles:`/`memories:` are free-form, host and model entries are checked against `Config::MAP_ENTRY_KEYS`). An empty list means no warning at start.
|
|
117
121
|
5. **Write atomically**: `FileUtils.mkdir_p(File.dirname(path))`, `File.write("#{path}.tmp", YAML.dump(raw_data))`, `File.rename("#{path}.tmp", path)`.
|
|
118
122
|
6. Update in-process state: `write_default_model!` sets `ENV["SAMAGOTCHI_DEFAULT_MODEL"]` and `Samagotchi::Config.reload!`; otherwise the harness picks it up on next `Config.get` (live resolve) or restart. CLI overrides (`--default-model`) win over file until process exit.
|
|
119
123
|
|
|
@@ -132,7 +136,7 @@ A guardrail rule's `tool:` may be a glob (`tool: "mcp_*"`, verdict `ask`) to cov
|
|
|
132
136
|
- **Hosts**: each entry needs `host`, `port` 1-65535, `transport` and `api` optional (a raw `api` must match `transport`), name must match `/\A[a-z0-9][a-z0-9._-]*\z/i`.
|
|
133
137
|
- **Hooks**: each entry must have `path` (relative to `hooks_dir`), `on_error` is `skip` (default) or `log`. Class name must match file basename snake→Pascal.
|
|
134
138
|
- **Scalars via registry**: `Config.coerce` validates `String/Numeric/true/false` per `type: :string/:integer/:float/:bool/:enum`; invalid values warn and fall back to entry `default`.
|
|
135
|
-
- **
|
|
139
|
+
- **Keys**: `validate_yaml_sections` flags unknown keys (with a suggestion), a registry section that isn't a mapping, and env/CLI-only keys (`model.profile`).
|
|
136
140
|
|
|
137
141
|
## Tools to use
|
|
138
142
|
|
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: samagotchi-system
|
|
3
|
-
version: 0.
|
|
3
|
+
version: 0.3.0
|
|
4
4
|
scope: system
|
|
5
5
|
description: Default system memories — identity, self map, config modification protocol, memory guide and the delegated-session rules
|
|
6
6
|
files:
|
|
7
7
|
identity.md: sha256:8b594100bee4797b563bd6425dc6f3ff93e1bdfbe9fc5fa1d66b42093c4795e2
|
|
8
|
-
self_map.md: sha256:
|
|
9
|
-
config_modification_protocol.md: sha256:
|
|
8
|
+
self_map.md: sha256:67eb6223728a72788fa0d75aa56affea2ca48962de2e445293bc12b15f5f70c0
|
|
9
|
+
config_modification_protocol.md: sha256:1d1047f3f99fc9ceb89ef191cbcf3d68bc996addd00d7b6eb567a63b8ff677ea
|
|
10
10
|
memory_guide.md: sha256:6dd1ecfd6a377c4f2cfc2dc299c1ef3835f10d9f313616fd3eb23e5fd3202858
|
|
11
11
|
delegated.md: sha256:8f7965b6165c3526abb0cd39e44a8b73a8d6a26755049bdcd007baf0b0771087
|
|
@@ -13,11 +13,15 @@
|
|
|
13
13
|
- Sessions: each is one file, `<sessions dir>/<id>.json`. A `<id>/` dir beside it exists only
|
|
14
14
|
for background/web workers (input/, notes/, output/, pid). List them with `chi sessions list`
|
|
15
15
|
(`--live` for the ones a worker runs now); delete one with `chi sessions delete ID` (never by hand).
|
|
16
|
+
An archived session (`chi sessions archive ID`, a marker file `<id>/archived`) is hidden from
|
|
17
|
+
every list and kept for good; `chi sessions list --archived` shows it.
|
|
16
18
|
- `[CONTEXT NOTE from …]` messages are context notes (`chi note`, or another session's
|
|
17
19
|
`send_note`): background, not requests. `list_sessions` + `send_note` tell another session
|
|
18
20
|
something without starting a turn there; see `docs/sessions.md` "Context notes".
|
|
19
21
|
- `chi send -m TEXT <id>` is the other half: the text goes in as the user's message and a
|
|
20
22
|
turn runs (stdin piped too = quoted context above it); see `docs/sessions.md` "Sending a message".
|
|
23
|
+
`chi send --new --wait -m TEXT` starts a session the user can watch in the web and prints its
|
|
24
|
+
answer (exit 3: it waits for the user's answer); see "Starting a session".
|
|
21
25
|
- `delegate` hands a task to a child session that runs in parallel and returns only its final
|
|
22
26
|
reply (`delegate_result` waits for it; `session:` sends a follow-up to a child). A child is a
|
|
23
27
|
normal session: it shows in `chi sessions list` with `↳ <parent>`, and the user can steer it
|
|
@@ -26,8 +30,10 @@
|
|
|
26
30
|
commands (`anytime:` ones run mid-turn), model tools (they can return images I see), hooks,
|
|
27
31
|
cards, side answers (`ask_model`),
|
|
28
32
|
child sessions and services (e.g. MCP server processes). Shipped: `btw` (`/btw` side question),
|
|
29
|
-
`mcp` (MCP server tools; a screenshot comes as a picture), `guardrails` (rules), `known-names` (typo guard), `
|
|
30
|
-
(
|
|
33
|
+
`mcp` (MCP server tools; a screenshot comes as a picture), `guardrails` (rules), `known-names` (typo guard), `source-links`
|
|
34
|
+
(turns source refs like JIRA-123 in an answer into links in the web, and a one-line note), `loop-guard`
|
|
35
|
+
(denies a repeated tool call with the same result, stops the turn after a few), `check-in` (after N tool
|
|
36
|
+
calls with no answer, a card asks the user to nudge me, let me go on or stop; `/checkin`). `chi bundle list`
|
|
31
37
|
shows installed + available; `chi bundle install <name>`. Settings: config.yml `bundles: <name>:`
|
|
32
38
|
(`config_modification_protocol`), read at session start: after an install or a settings change,
|
|
33
39
|
tell the user to restart the session. API and bundle docs: `docs/plugins.md`.
|
data/lib/samagotchi/client.rb
CHANGED
|
@@ -8,6 +8,7 @@ require_relative "cancellation_controller"
|
|
|
8
8
|
require_relative "llm/http"
|
|
9
9
|
require_relative "vision_context"
|
|
10
10
|
require_relative "vision_support"
|
|
11
|
+
require_relative "sampling_settings"
|
|
11
12
|
|
|
12
13
|
module Samagotchi
|
|
13
14
|
# Thin HTTP client for llama.cpp's native /completion endpoint, or an
|
|
@@ -27,6 +28,10 @@ module Samagotchi
|
|
|
27
28
|
# budget and no retry (see #server_props).
|
|
28
29
|
CONTEXT_WINDOW_PROBE_OPEN_TIMEOUT = 1
|
|
29
30
|
CONTEXT_WINDOW_PROBE_READ_TIMEOUT = 2
|
|
31
|
+
# Seconds a probe that got no answer (refused, timed out) is remembered:
|
|
32
|
+
# a down or hung server then costs one probe per window, not one per
|
|
33
|
+
# generation, and a new turn doesn't ask again at once.
|
|
34
|
+
PROPS_FAILURE_TTL = 30
|
|
30
35
|
|
|
31
36
|
SERVER_TRANSPORT_ENV = "SAMAGOTCHI_SERVER_TRANSPORT"
|
|
32
37
|
DEFAULT_TRANSPORT = :llama_cpp
|
|
@@ -109,7 +114,7 @@ module Samagotchi
|
|
|
109
114
|
end
|
|
110
115
|
|
|
111
116
|
# One /props probe's outcome. `answered?` is false when the probe failed:
|
|
112
|
-
# a network error, a timeout or any non-200 (llama.cpp answers 503 while
|
|
117
|
+
# a network error, a timeout, a turn's cancel or any non-200 (llama.cpp answers 503 while
|
|
113
118
|
# it loads a model). `body` is the parsed JSON of a 200, or nil when it
|
|
114
119
|
# isn't JSON.
|
|
115
120
|
ServerProps = Data.define(:body, :status) do
|
|
@@ -121,6 +126,21 @@ module Samagotchi
|
|
|
121
126
|
# Moved to its own file; the old name keeps working.
|
|
122
127
|
CancellationController = Samagotchi::CancellationController
|
|
123
128
|
|
|
129
|
+
PROBE_CANCEL_KEY = :samagotchi_probe_cancel
|
|
130
|
+
|
|
131
|
+
# The cancel a /props probe made on this thread listens to: a turn sets
|
|
132
|
+
# its own (Engine#run_turn), so a Stop cuts the probes it makes before
|
|
133
|
+
# its first request. Probes on other threads (chi self, /model, a
|
|
134
|
+
# status snapshot) have none and run to their timeout.
|
|
135
|
+
def self.probe_cancel = Thread.current[PROBE_CANCEL_KEY]
|
|
136
|
+
|
|
137
|
+
# Sets this thread's probe cancel; returns the one it replaces.
|
|
138
|
+
def self.swap_probe_cancel(controller)
|
|
139
|
+
previous = Thread.current[PROBE_CANCEL_KEY]
|
|
140
|
+
Thread.current[PROBE_CANCEL_KEY] = controller
|
|
141
|
+
previous
|
|
142
|
+
end
|
|
143
|
+
|
|
124
144
|
# @param sleeper [#call, nil] waits between retries (specs pass a no-op)
|
|
125
145
|
# @param scheme [String, nil] "https" for a TLS server (default http)
|
|
126
146
|
# @param first_token_timeout [Numeric, nil] seconds a completion may take
|
|
@@ -148,6 +168,7 @@ module Samagotchi
|
|
|
148
168
|
transport_fallback = cfg_transport_raw || ENV.fetch(SERVER_TRANSPORT_ENV, DEFAULT_TRANSPORT.to_s)
|
|
149
169
|
@transport = build_transport(resolve_transport(transport || transport_fallback))
|
|
150
170
|
@props_cache = {}
|
|
171
|
+
@props_failures = {}
|
|
151
172
|
@props_mutex = Mutex.new
|
|
152
173
|
@first_token_timeout = first_token_timeout
|
|
153
174
|
@host_name = name
|
|
@@ -182,11 +203,14 @@ module Samagotchi
|
|
|
182
203
|
# @param on_retry [Proc, nil] optional callback before retry sleep
|
|
183
204
|
# @param images [Array<String>] base64 images, one per
|
|
184
205
|
# ImagePlan::NATIVE_PLACEHOLDER in the prompt (llama.cpp only)
|
|
206
|
+
# @param sampling [Hash] request fields to add (SamplingSettings):
|
|
207
|
+
# temperature, penalties, …; they can't replace the fields above
|
|
185
208
|
# @return [String] the generated text
|
|
186
209
|
def complete(prompt, stop: ["<end_of_turn>", "<|tool_response>"], n_predict: nil, model: nil, on_chunk: nil, cancel_controller: nil, on_retry: nil,
|
|
187
|
-
images: [])
|
|
210
|
+
images: [], sampling: {})
|
|
188
211
|
images = Array(images)
|
|
189
|
-
|
|
212
|
+
request = { stop: stop, n_predict: n_predict, model: model, sampling: sampling }
|
|
213
|
+
return stream_completion(scrub_utf8(prompt), request, on_chunk, cancel_controller, on_retry) if images.empty?
|
|
190
214
|
|
|
191
215
|
# The media marker is random per server process: a restart between the
|
|
192
216
|
# /props read and the request makes the prompt fail to tokenize, so
|
|
@@ -196,7 +220,7 @@ module Samagotchi
|
|
|
196
220
|
attempts += 1
|
|
197
221
|
payload_prompt = { prompt_string: scrub_utf8(prompt.gsub(ImagePlan::NATIVE_PLACEHOLDER, media_marker!(model))),
|
|
198
222
|
multimodal_data: images }
|
|
199
|
-
stream_completion(payload_prompt,
|
|
223
|
+
stream_completion(payload_prompt, request, on_chunk, cancel_controller, on_retry)
|
|
200
224
|
rescue LLM::BadRequest => e
|
|
201
225
|
raise unless attempts == 1 && e.message.include?("Failed to tokenize prompt")
|
|
202
226
|
|
|
@@ -205,11 +229,14 @@ module Samagotchi
|
|
|
205
229
|
end
|
|
206
230
|
end
|
|
207
231
|
|
|
208
|
-
|
|
232
|
+
OPENAI_API_HINT = "does this host speak the OpenAI API? set `api: openai` on it"
|
|
233
|
+
|
|
234
|
+
private def stream_completion(prompt, fields, on_chunk, cancel_controller, on_retry)
|
|
209
235
|
uri = completion_uri
|
|
210
236
|
request = Net::HTTP::Post.new(uri)
|
|
211
237
|
request["Content-Type"] = "application/json"
|
|
212
|
-
request.body = completion_payload(prompt,
|
|
238
|
+
request.body = completion_payload(prompt, **fields).to_json
|
|
239
|
+
model = fields[:model]
|
|
213
240
|
|
|
214
241
|
result = +""
|
|
215
242
|
reset_on_retry = lambda do |event|
|
|
@@ -219,7 +246,8 @@ module Samagotchi
|
|
|
219
246
|
end
|
|
220
247
|
@http.stream_lines(uri, request, cancel_controller: cancel_controller, on_retry: reset_on_retry,
|
|
221
248
|
on_network_error: ->(_error) { invalidate_context_window! },
|
|
222
|
-
log_fields: { model: model, purpose: "chat"
|
|
249
|
+
log_fields: { model: model, purpose: "chat",
|
|
250
|
+
sampling: SamplingSettings.log_text(sendable_sampling(fields[:sampling])) }) do |line, shown|
|
|
223
251
|
parsed_chunk = parse_stream_line(line)
|
|
224
252
|
next unless parsed_chunk
|
|
225
253
|
|
|
@@ -229,6 +257,11 @@ module Samagotchi
|
|
|
229
257
|
on_chunk&.call(content: content, payload: payload)
|
|
230
258
|
end
|
|
231
259
|
result
|
|
260
|
+
rescue LLM::BadRequest => e
|
|
261
|
+
raise unless e.status == 404 && @transport.name == :llama_cpp
|
|
262
|
+
|
|
263
|
+
# No native /completion here: likely an OpenAI-compatible server.
|
|
264
|
+
raise LLM::BadRequest.new(e.message, host: e.host, status: e.status, attempts: e.attempts, hint: OPENAI_API_HINT)
|
|
232
265
|
rescue RequestCancelled, LLM::ProviderError
|
|
233
266
|
raise
|
|
234
267
|
rescue StandardError => e
|
|
@@ -260,7 +293,9 @@ module Samagotchi
|
|
|
260
293
|
# a turn. The probe names the model (`?model=`): a llama.cpp router
|
|
261
294
|
# answers a stub without it, and a single-model server ignores it.
|
|
262
295
|
# Whatever the server answers (a non-200 too) is cached per model; a
|
|
263
|
-
# network failure
|
|
296
|
+
# network failure for PROPS_FAILURE_TTL seconds, then the next call asks
|
|
297
|
+
# again. A probe cut by this thread's Client.probe_cancel answers
|
|
298
|
+
# :cancelled, uncached, and the turn's next request ends it.
|
|
264
299
|
def server_props(model: nil)
|
|
265
300
|
path = @transport.props_path
|
|
266
301
|
return nil unless path
|
|
@@ -268,10 +303,21 @@ module Samagotchi
|
|
|
268
303
|
key = model.to_s
|
|
269
304
|
@props_mutex.synchronize do
|
|
270
305
|
return @props_cache[key] if @props_cache.key?(key)
|
|
306
|
+
|
|
307
|
+
failed, failed_at = @props_failures[key]
|
|
308
|
+
return failed if failed && monotonic_now - failed_at < PROPS_FAILURE_TTL
|
|
271
309
|
end
|
|
272
310
|
|
|
273
311
|
props = probe_props(path, key)
|
|
274
|
-
@props_mutex.synchronize
|
|
312
|
+
@props_mutex.synchronize do
|
|
313
|
+
case props.status
|
|
314
|
+
when :cancelled then nil
|
|
315
|
+
when :network_error then @props_failures[key] = [props, monotonic_now]
|
|
316
|
+
else
|
|
317
|
+
@props_cache[key] = props
|
|
318
|
+
@props_failures.delete(key)
|
|
319
|
+
end
|
|
320
|
+
end
|
|
275
321
|
props
|
|
276
322
|
end
|
|
277
323
|
|
|
@@ -286,23 +332,30 @@ module Samagotchi
|
|
|
286
332
|
|
|
287
333
|
# Forget cached /props answers: the server may have restarted with
|
|
288
334
|
# another -c, or a model switch may have loaded one with a different
|
|
289
|
-
# window.
|
|
335
|
+
# window. A recent failure stays until its PROPS_FAILURE_TTL ends: it
|
|
336
|
+
# holds no stale window, and asking a hung server again at every turn's
|
|
337
|
+
# start is what it saves.
|
|
290
338
|
def invalidate_context_window!
|
|
291
339
|
@props_mutex.synchronize { @props_cache.clear }
|
|
292
340
|
end
|
|
293
341
|
|
|
294
342
|
private
|
|
295
343
|
|
|
344
|
+
def monotonic_now = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
345
|
+
|
|
296
346
|
def probe_props(path, model)
|
|
297
347
|
query = model.empty? ? "" : "?#{URI.encode_www_form(model: model)}"
|
|
298
348
|
uri = URI("#{@scheme}://#{@host}:#{@port}#{path}#{query}")
|
|
299
349
|
response = @http.fetch(uri, Net::HTTP::Get.new(uri), retries: false, check_status: false,
|
|
300
350
|
log_fields: { model: model.empty? ? nil : model, purpose: "probe" },
|
|
301
351
|
open_timeout: CONTEXT_WINDOW_PROBE_OPEN_TIMEOUT,
|
|
302
|
-
read_timeout: CONTEXT_WINDOW_PROBE_READ_TIMEOUT
|
|
352
|
+
read_timeout: CONTEXT_WINDOW_PROBE_READ_TIMEOUT,
|
|
353
|
+
cancel_controller: Client.probe_cancel)
|
|
303
354
|
return ServerProps.new(body: nil, status: :http_error) unless response.code.to_s == "200"
|
|
304
355
|
|
|
305
356
|
ServerProps.new(body: parse_props(response.body), status: :ok)
|
|
357
|
+
rescue RequestCancelled
|
|
358
|
+
ServerProps.new(body: nil, status: :cancelled)
|
|
306
359
|
rescue StandardError
|
|
307
360
|
ServerProps.new(body: nil, status: :network_error)
|
|
308
361
|
end
|
|
@@ -337,12 +390,18 @@ module Samagotchi
|
|
|
337
390
|
URI("#{@scheme}://#{@host}:#{@port}#{@transport.completion_path}")
|
|
338
391
|
end
|
|
339
392
|
|
|
340
|
-
def completion_payload(prompt, stop:, n_predict:, model:)
|
|
393
|
+
def completion_payload(prompt, stop:, n_predict:, model:, sampling: {})
|
|
341
394
|
payload = { prompt: prompt, stop: stop, stream: true }
|
|
342
395
|
payload[@transport.token_limit_key] = n_predict if n_predict && n_predict.to_i.positive?
|
|
343
396
|
model_name = @transport.model_for_payload(model)
|
|
344
397
|
payload[:model] = model_name if model_name
|
|
345
|
-
payload
|
|
398
|
+
sendable_sampling(sampling).merge(payload)
|
|
399
|
+
end
|
|
400
|
+
|
|
401
|
+
# The sampling fields that go out: no nil (nothing to drop here: the
|
|
402
|
+
# server's defaults apply) and none of the reserved request fields.
|
|
403
|
+
def sendable_sampling(sampling)
|
|
404
|
+
(sampling || {}).reject { |key, value| value.nil? || ConfigFile::SAMPLING_RESERVED_KEYS.include?(key.to_s) }
|
|
346
405
|
end
|
|
347
406
|
|
|
348
407
|
# Conversation content (system prompt + tool responses + model output) can
|