lithe 0.9.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. lithe-0.9.4/LICENSE +21 -0
  2. lithe-0.9.4/PKG-INFO +334 -0
  3. lithe-0.9.4/README.md +287 -0
  4. lithe-0.9.4/lithe/__init__.py +64 -0
  5. lithe-0.9.4/lithe/actions.py +105 -0
  6. lithe-0.9.4/lithe/bundles/__init__.py +62 -0
  7. lithe-0.9.4/lithe/bundles/_textmatch.py +115 -0
  8. lithe-0.9.4/lithe/bundles/admin.py +133 -0
  9. lithe-0.9.4/lithe/bundles/download.py +272 -0
  10. lithe-0.9.4/lithe/bundles/host.py +602 -0
  11. lithe-0.9.4/lithe/bundles/images.py +362 -0
  12. lithe-0.9.4/lithe/bundles/mcp.py +742 -0
  13. lithe-0.9.4/lithe/bundles/patch.py +487 -0
  14. lithe-0.9.4/lithe/bundles/sandbox.py +249 -0
  15. lithe-0.9.4/lithe/bundles/skills.py +485 -0
  16. lithe-0.9.4/lithe/bundles/store/__init__.py +18 -0
  17. lithe-0.9.4/lithe/bundles/store/jsonl.py +392 -0
  18. lithe-0.9.4/lithe/bundles/store/protocol.py +214 -0
  19. lithe-0.9.4/lithe/bundles/subagents.py +543 -0
  20. lithe-0.9.4/lithe/bundles/todos.py +275 -0
  21. lithe-0.9.4/lithe/bundles/workspace.py +654 -0
  22. lithe-0.9.4/lithe/context.py +82 -0
  23. lithe-0.9.4/lithe/events.py +85 -0
  24. lithe-0.9.4/lithe/llm.py +304 -0
  25. lithe-0.9.4/lithe/memory.py +353 -0
  26. lithe-0.9.4/lithe/modes.py +70 -0
  27. lithe-0.9.4/lithe/py.typed +0 -0
  28. lithe-0.9.4/lithe/runtime.py +899 -0
  29. lithe-0.9.4/lithe/skills.py +28 -0
  30. lithe-0.9.4/lithe/tools.py +281 -0
  31. lithe-0.9.4/lithe/transports.py +726 -0
  32. lithe-0.9.4/lithe.egg-info/PKG-INFO +334 -0
  33. lithe-0.9.4/lithe.egg-info/SOURCES.txt +57 -0
  34. lithe-0.9.4/lithe.egg-info/dependency_links.txt +1 -0
  35. lithe-0.9.4/lithe.egg-info/requires.txt +6 -0
  36. lithe-0.9.4/lithe.egg-info/top_level.txt +1 -0
  37. lithe-0.9.4/pyproject.toml +52 -0
  38. lithe-0.9.4/setup.cfg +4 -0
  39. lithe-0.9.4/tests/test_lithe.py +375 -0
  40. lithe-0.9.4/tests/test_lithe_action_bridge.py +229 -0
  41. lithe-0.9.4/tests/test_lithe_admin.py +76 -0
  42. lithe-0.9.4/tests/test_lithe_download_bundle.py +301 -0
  43. lithe-0.9.4/tests/test_lithe_e2e.py +113 -0
  44. lithe-0.9.4/tests/test_lithe_host_store.py +618 -0
  45. lithe-0.9.4/tests/test_lithe_images_bundle.py +202 -0
  46. lithe-0.9.4/tests/test_lithe_llm.py +101 -0
  47. lithe-0.9.4/tests/test_lithe_mcp_bundle.py +310 -0
  48. lithe-0.9.4/tests/test_lithe_mcp_http.py +155 -0
  49. lithe-0.9.4/tests/test_lithe_memory.py +145 -0
  50. lithe-0.9.4/tests/test_lithe_p2_round.py +842 -0
  51. lithe-0.9.4/tests/test_lithe_patch.py +571 -0
  52. lithe-0.9.4/tests/test_lithe_runtime.py +1251 -0
  53. lithe-0.9.4/tests/test_lithe_sandbox_bundle.py +213 -0
  54. lithe-0.9.4/tests/test_lithe_skills_bundle.py +87 -0
  55. lithe-0.9.4/tests/test_lithe_skills_packages.py +152 -0
  56. lithe-0.9.4/tests/test_lithe_subagents.py +506 -0
  57. lithe-0.9.4/tests/test_lithe_todos_bundle.py +205 -0
  58. lithe-0.9.4/tests/test_lithe_transports.py +618 -0
  59. lithe-0.9.4/tests/test_lithe_workspace_bundle.py +449 -0
lithe-0.9.4/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 betaloop contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
lithe-0.9.4/PKG-INFO ADDED
@@ -0,0 +1,334 @@
1
+ Metadata-Version: 2.4
2
+ Name: lithe
3
+ Version: 0.9.4
4
+ Summary: A reusable, storage-free ReAct agent kernel (engine + framework + protocols) with optional capability bundles.
5
+ License: MIT License
6
+
7
+ Copyright (c) 2026 betaloop contributors
8
+
9
+ Permission is hereby granted, free of charge, to any person obtaining a copy
10
+ of this software and associated documentation files (the "Software"), to deal
11
+ in the Software without restriction, including without limitation the rights
12
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
13
+ copies of the Software, and to permit persons to whom the Software is
14
+ furnished to do so, subject to the following conditions:
15
+
16
+ The above copyright notice and this permission notice shall be included in all
17
+ copies or substantial portions of the Software.
18
+
19
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
20
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
21
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
22
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
23
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
24
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
25
+ SOFTWARE.
26
+
27
+ Keywords: agent,llm,react,kernel,tool-calling,mcp
28
+ Classifier: Development Status :: 4 - Beta
29
+ Classifier: Intended Audience :: Developers
30
+ Classifier: License :: OSI Approved :: MIT License
31
+ Classifier: Programming Language :: Python :: 3
32
+ Classifier: Programming Language :: Python :: 3.10
33
+ Classifier: Programming Language :: Python :: 3.11
34
+ Classifier: Programming Language :: Python :: 3.12
35
+ Classifier: Programming Language :: Python :: 3.13
36
+ Classifier: Topic :: Software Development :: Libraries :: Application Frameworks
37
+ Classifier: Typing :: Typed
38
+ Requires-Python: >=3.10
39
+ Description-Content-Type: text/markdown
40
+ License-File: LICENSE
41
+ Requires-Dist: httpx>=0.24
42
+ Provides-Extra: dev
43
+ Requires-Dist: pytest>=8; extra == "dev"
44
+ Requires-Dist: pytest-asyncio>=0.23; extra == "dev"
45
+ Requires-Dist: ruff>=0.4; extra == "dev"
46
+ Dynamic: license-file
47
+
48
+ # lithe
49
+
50
+ A reusable, **storage-free** ReAct agent kernel: the engine, framework and
51
+ protocols that drive a tool-calling agent, plus optional capability bundles. It
52
+ knows nothing about how runs are stored (or even whether they are) — persistence
53
+ is an optional `EventSink` a host plugs in. Any application (a thesis-writing
54
+ platform, a coding agent, ...) implements its own tools + prompt + storage and
55
+ reuses this kernel.
56
+
57
+ ## Core (zero I/O, zero business)
58
+ - `runtime` — `AgentRuntime` ReAct loop + event stream. Supports cancellation
59
+ (`run(..., stop=Event|callable)` → `cancelled` event, `status="cancelled"`;
60
+ checked between streaming deltas too, closing the in-flight model stream
61
+ instead of paying for a response nobody wants) and streaming
62
+ (`LLMConfig(stream=True)` → `assistant_delta` events while the
63
+ model generates; a final full `assistant` event always follows). Every
64
+ `tool_call` event is announced before any of the step's tools execute, so a
65
+ slow tool never hides what is pending on the frontend; result events still
66
+ follow in model order. Every model call emits a
67
+ `usage` event — prompt/completion/total tokens, that call's
68
+ cost, and context fullness (`context_tokens`, `context_chars`,
69
+ `context_window`, `context_percent` when `LLMConfig(context_window=...)` is
70
+ set) — so a frontend can show live token/context gauges; `RunStats` and the
71
+ host `done` event carry the cumulative breakdown. Run budgets
72
+ (`max_cost` / `max_total_tokens`) cut a runaway run short with
73
+ `status="budget_exceeded"` — no further tool execution, no further model
74
+ calls — and a host may seed `stats` with prior-conversation totals to
75
+ budget across runs. A stuck model reissuing the *identical* (tool, args)
76
+ call more than `repeat_call_limit` (default 3) times gets an inline nudge
77
+ inside that tool's result, feeding self-correction (`None` disables).
78
+ Malformed tool-call arguments come back to the model as failed tool
79
+ results instead of executing with empty/wrong args. A run that exhausts
80
+ its step budget gets a forced toolless wrap-up call (`tool_choice="none"`)
81
+ so it ends with the model's summary, reporting `status="max_steps"` (and
82
+ never executing the stubborn model's further tool calls).
83
+ `LLMConfig(temperature=..., max_tokens=...)` are forwarded on every call,
84
+ and a generation cut off by the token cap (`finish_reason="length"`) is
85
+ marked — the notice rides in the final text, the per-call `usage` event
86
+ carries `finish_reason`, and a tool_call whose arguments JSON was truncated
87
+ gets a "cut off by max_tokens, re-issue the call" error instead of a bare
88
+ "invalid JSON" that invites a byte-identical retry. An empty response (no
89
+ text, no tool calls) is retried once and otherwise ends the run with
90
+ `status="empty_response"` instead of an empty "success". A missing
91
+ `tool_call` id is synthesized at one point (streamed and non-streamed
92
+ alike), so a gateway that omits ids never poisons the next request with
93
+ `tool_call_id: null` — and streamed calls no longer mint colliding
94
+ `call_0`-style ids across steps. Synthetic run endings (max-step /
95
+ budget notices) are recorded like any assistant turn, so sinks and the
96
+ next run's replay see why the run stopped. Tool calls execute in model
97
+ order with consecutive READ tools parallel and every WRITE/META tool alone
98
+ (no write races); `ToolSpec(timeout=...)` cancels a hung call. A mid-run
99
+ context budget (`context_budget`, default 400k chars) shrinks old tool
100
+ results head+tail so long runs don't blow the context window. Sinks are
101
+ error-isolated (a broken display/record sink logs instead of killing the
102
+ run; `strict_records=True` opts record failures back into fatal).
103
+ `AgentRuntime(envelope=True)` emits the `run_start`/`done` envelope itself
104
+ for hosts driving the runtime directly, and `AgentRuntime(http_client=...)`
105
+ (forwarded by `AgentHost`) reuses one host-owned `httpx.AsyncClient` across
106
+ runs — connection pooling, limits, proxy/verify config for service hosts.
107
+ - `llm` — OpenAI-compatible chat client (retry / jittered backoff / response-shape
108
+ validation / `Retry-After`-aware 429 handling / fatal-4xx fail-fast) + SSE
109
+ streaming helpers; transports normalize usage to
110
+ `prompt_tokens`/`completion_tokens`/`total_tokens` across chat-completions
111
+ and Responses shapes
112
+ - `tools` — `ToolRegistry` (register / unregister / dispatch / mode filtering /
113
+ argument validation / per-tool timeout) + pre-dispatch **middleware** via
114
+ ``add_middleware`` (audit / quota / human-in-the-loop confirmation of write
115
+ tools)
116
+ - `actions` — `Action` + `UndoEngine` (pure, storage-free undo; reverters may
117
+ be sync or async)
118
+ - `memory` — `replay_messages` / `recap_text` / `window_with_recap` /
119
+ `run_timeline` (reconstruct a stored turn's ordered event timeline) +
120
+ `MemoryProvider`; replay reconciliation is two-sided *and* window-safe
121
+ (a tool row whose calling assistant fell outside the window is dropped,
122
+ not sent as an orphan first message)
123
+ - `events` — `EventSink` (display + record channels) + SSE serialization
124
+ (`to_sse` degrades non-JSON values via `str()` — a `datetime` inside a
125
+ tool's `ui` payload can't crash the host's SSE layer)
126
+ - `context` / `modes` — `AgentContext` + `AgentMode` / `ToolCategory`;
127
+ host-defined modes via ``register_mode(name, categories)`` (unknown modes
128
+ raise instead of silently degrading to read-only). `AgentContext.shared`
129
+ is per-run state shared *by reference* with subagent contexts — the
130
+ vehicle for cross-context coordination (the workspace stale-file guard's
131
+ revision map, the run's cancellation handle)
132
+
133
+ ## Optional bundles (`lithe.bundles`)
134
+ - `host` — `AgentHost` host-adapter framework: message assembly, run envelope
135
+ (`run_start`/`done`), error funneling, `StoreSink` (persist via a store),
136
+ `undo_run` (zero-config: bundled tools register reverters keyed by their
137
+ action kinds, and reverters see the host's `extra` context; a reversion
138
+ whose *status mark* fails surfaces in the report instead of being
139
+ swallowed), `DictToolAdapter` (wrap a dict-based tool system — specs
140
+ without a handler log a warning instead of silently vanishing, and a
141
+ `reverters=` map wires custom action kinds into undo).
142
+ Eliminates the per-host boilerplate round 1
143
+ left behind. `host.run(..., stop=...)` forwards cancellation to the runtime;
144
+ `AgentHost(max_cost=..., max_total_tokens=..., repeat_call_limit=...)`
145
+ forwards the runtime's budget / repeat guards to every run it builds;
146
+ `AgentHost(http_client=...)` shares one HTTP client across runs, and
147
+ `AgentHost(capture_actions=False)` turns off `StoreSink`'s automatic
148
+ action capture for hosts that persist actions themselves.
149
+ - `store` — `RunStore`/`ConversationStore`/`BlobStore` Protocols +
150
+ `JsonlRunStore` (default, **zero-database** JSONL + content-addressed blob
151
+ spillover). Hosts wanting a DB implement the Protocols; the default needs
152
+ none. Action values larger than `spill_threshold` (default 8KB) externalize
153
+ to blobs and rehydrate transparently on read; id counters are in-memory so
154
+ appends don't rescan the stream. `list_actions(subagent=...)` filters by
155
+ subagent during the fold (unselected rows skip blob rehydration — the
156
+ subagent engine's snapshots stay O(one worker) on long runs); torn lines
157
+ log a warning and count in `dropped_lines` instead of vanishing silently;
158
+ `fsync=True` flushes each append for crash-durability.
159
+ - `subagents` — `SubagentEngine` + `SubagentRoster`/`SubagentSpec` + `delegate`
160
+ tool: isolated worker agents the orchestrator hands subtasks to, tagged so undo
161
+ still reverts them while the orchestrator's context stays lean.
162
+ `delegate_parallel` fans independent tasks out concurrently (bounded by
163
+ `max_parallel`, failures isolated per agent, duplicate agents in one batch
164
+ rejected — they would claim each other's actions); `SubagentEngine(on_subagent_event=...)`
165
+ streams live `subagent_progress` heartbeats to a host push channel (text,
166
+ args and summaries capped — a 50KB `write_file` payload never rides the
167
+ callback); `SubagentSpec(transport=...)` routes a subagent to a different
168
+ endpoint. `make_delegate_tool(engine, timeout=...)` caps one delegation's
169
+ wall time so a hung worker cannot hold the orchestrator's step forever.
170
+ Budget caps apply per subagent run (each delegation gets its own
171
+ `max_cost` / `max_total_tokens`) while each delegation's spend folds into
172
+ the parent run's `done` event, stats and store row (with a
173
+ `subagent_*` breakdown); cancellation of the orchestrating run
174
+ propagates into an in-flight subagent (its stop handle rides in
175
+ `ctx.shared`); a delegation's own mutations are identified by an
176
+ id-membership snapshot, so stores with opaque (non-integer) action ids
177
+ count them correctly.
178
+ - `admin` — `tool_categories` / `list_tools_admin` / `list_tool_packages_admin` /
179
+ `check_packages` over a registry + display packages (admin-panel source;
180
+ packages are grouping only, tools stay per-name togglable).
181
+ - `patch` — `apply_patch`: line-oriented multi-file edits via the Codex
182
+ `*** Begin Patch` envelope (add/update/move/delete files, `@@` chunks of
183
+ context/`-`/`+` lines, `*** End of File` anchoring). Chunk location runs a
184
+ four-pass fuzzy ladder (exact → trailing-ws → strip → Unicode-punctuation
185
+ fold); `*** End of File` chunks run the tail-anchored ladder at full
186
+ strength before any forward match, so a whitespace-mismatched tail beats
187
+ an exact look-alike earlier in the file instead of silently editing the
188
+ wrong site; two same-position insertions keep document order. Application
189
+ is all-or-nothing against an in-memory overlay (later
190
+ sections of the same file chain; create-then-edit works), so a bad hunk
191
+ leaves the workspace untouched. Same `file_change` events + one new undo
192
+ kind (`file_delete`).
193
+ - `workspace` — sandboxed file I/O + read/write/edit/list/search/glob tools +
194
+ file undo reverters. Directory walks (`list_files` / `search_files` /
195
+ `glob_files`) never follow symlinks — code executed by `run_code` could
196
+ otherwise plant a link to a host file and read it back through a walk,
197
+ bypassing the path guard (symlinks list as an opaque `symlink` type;
198
+ `list_files(dirs=...)` routes the requested folders through the same path
199
+ guard). `read_file` prefixes every line with its 1-based
200
+ number and supports `offset`/`limit` line-window reads. `edit_file` refuses
201
+ ambiguous `old_text` (multi-match) unless `replace_all` is set, returns a
202
+ diff, and falls back to a whole-line fuzzy match (shared ladder,
203
+ trailing-blank aligned like `apply_patch`) when the
204
+ exact substring misses. All write tools guard against stale content: a
205
+ file read this run and changed out-of-band is refused with "re-read it"
206
+ instead of being clobbered; the revision map lives in `ctx.shared`, so the
207
+ guard spans the orchestrator and every (including parallel) subagent of
208
+ the run. `search_files` greps content
209
+ by regex (dir / glob filters) with a 30s tool timeout, a 10s scan budget
210
+ and a 10k-char line cap (catastrophic-backtracking patterns and huge
211
+ minified lines can't hang the loop); `glob_files` matches paths by pattern.
212
+ - `todos` — a per-scope task list the agent plans against: `TodoStore`
213
+ (pure container) + `JsonTodoStore` (atomic tmp+rename JSON persistence;
214
+ malformed items are dropped on load instead of crashing the system
215
+ prompt) + `update_todos`/`list_todos` tools (`todo_replace` reverter) +
216
+ `todos_block` for splicing the list into a system prompt.
217
+ - `images` — tool-tier image perception, no kernel changes. `image_info`:
218
+ stdlib-only header probe (PNG/JPEG/GIF/BMP/WEBP — dimensions, dpi, color
219
+ mode) answering deterministic questions with zero model calls; it stats
220
+ first and reads only a bounded header window off the event loop.
221
+ `analyze_image`: ONE vision-model call (OpenAI `image_url` data-URL block +
222
+ the question), registered only when a `LLMConfig` is passed — the host's
223
+ main config reuses the main model, a dedicated one routes vision elsewhere.
224
+ Size is checked by stat before reading (a 2GB upload is refused, not
225
+ loaded), file reads run off the event loop, and `detail="auto"` shares a
226
+ cache key with an omitted detail (the API treats them identically — no
227
+ double billing). Images never enter the main conversation (answers are
228
+ memoized per file hash + question), so context budget / trimming / replay
229
+ stay untouched.
230
+ - `sandbox` — Python code execution (bubblewrap or passthrough backend);
231
+ output truncation keeps head+tail so tracebacks at the end stay visible.
232
+ `run_code`/`run_file` are WRITE-classified (executing model-written code
233
+ can mutate the workspace): invisible in read-only modes and never run in
234
+ parallel with other tool calls. Their side effects produce no undo
235
+ records (the tool descriptions say so) — durable edits belong in
236
+ `write_file`/`edit_file`/`apply_patch`. Note the contract difference:
237
+ `register_code_tools` takes `workspace_for(ctx) -> root path` (a `str`),
238
+ while the workspace/images bundles take `workspace_for(ctx) -> Workspace`.
239
+ - `download` — `download_file(url, path?)`: stream one HTTP(S) resource into
240
+ the workspace (default `downloads/`), the controlled ingress the
241
+ network-isolated sandbox deliberately lacks — fetching is a bounded,
242
+ audited tool while executing model code stays offline. SSRF guard: http/
243
+ https only, redirects followed manually with every hop re-validated, and
244
+ all resolved addresses of every hop must be globally routable
245
+ (`ipaddress.is_global` rejects loopback/private/link-local/CGN/reserved/
246
+ multicast — cloud metadata included); checking all A/AAAA records up front
247
+ narrows DNS rebinding to the TTL window. Size cap by `Content-Length`
248
+ pre-check plus streaming cutoff (a lying header gets cut mid-stream and
249
+ the partial `.part` file removed; the file lands atomically via rename).
250
+ Existing targets are refused (the model picks a new name), so nothing
251
+ undoable is mutated. Defaults: 64MB cap, 120s total budget (kernel
252
+ `ToolSpec` timeout as backstop), 5 redirects; WRITE-classified like the
253
+ other workspace-mutating tools.
254
+ - `skills` — markdown skill libraries (`SkillLibrary` flat dir; package-aware
255
+ `SkillPackages` with `RemoteSkillSource` registry mirrors — refresh
256
+ failures clean their staging dir, log, and keep the previous cache) +
257
+ `load_skill` tool. Skills do file/network I/O, hence a bundle — `import
258
+ lithe` stays zero-I/O (deprecated `lithe.skills` alias kept; it
259
+ warns on import).
260
+ - `mcp` — `MCPManager` + `MCPServerConfig` + `parse_servers`: bridge external
261
+ MCP servers into the
262
+ registry — stdio transport (`command=[...]`, e.g. `npx -y @z_ai/mcp-server`)
263
+ or streamable-http (`url=` + `headers=` for auth, `Mcp-Session-Id` handled
264
+ automatically). `tools/list` pagination (`nextCursor`) is followed, so
265
+ paginated servers don't silently lose half their tools. Spawned servers
266
+ get only a safe env allow-list plus the configured `env` (PATH/locale/
267
+ HOME/TMPDIR) — never all of `os.environ` with its secrets — unless
268
+ `inherit_env=True` opts back into the legacy behavior. URLs log as
269
+ scheme://host only (credentials in the query or userinfo never reach a
270
+ log line). Sessions outlive registry rebuilds and lazily self-heal: a dead
271
+ session is restarted on the next tool call (`revive`), and
272
+ `attach`/`ensure` re-register tools on any fresh registry (`ensure` is the
273
+ idempotent variant for hosts that cache registries across runs).
274
+ `tool_allowlist`/`tool_blocklist` filter by the server-side tool name.
275
+ `readOnlyHint` annotations map to the READ category; MCP tools carry no undo
276
+ reverters. Stdlib-only
277
+ (newline-delimited JSON-RPC / plain POST + SSE), secrets stay host-side:
278
+
279
+ ```python
280
+ from lithe.bundles import MCPManager, MCPServerConfig
281
+
282
+ manager = MCPManager([MCPServerConfig(
283
+ name="zai", command=["npx", "-y", "@z_ai/mcp-server"],
284
+ env={"Z_AI_API_KEY": "...", "Z_AI_MODE": "ZHIPU"},
285
+ default_category="read")])
286
+ await manager.attach(registry) # registry gains zai__* tools
287
+ ```
288
+
289
+ ## Install
290
+ ```bash
291
+ pip install lithe
292
+ # local dev (editable + test/lint deps):
293
+ pip install -e ".[dev]"
294
+ ```
295
+
296
+ ## Storage model
297
+ The kernel stores nothing. A host provides:
298
+ - an **`EventSink`** (write side) — persists message records however it likes
299
+ (DB / file / nowhere);
300
+ - a **`MemoryProvider`** (read side, optional) — replays prior turns;
301
+ - an **`UndoEngine`** fed from wherever the host kept actions.
302
+
303
+ ## Minimal host sketch
304
+ ```python
305
+ from lithe import AgentContext, LLMConfig, ToolRegistry
306
+ from lithe.bundles import AgentHost, JsonlRunStore
307
+ from lithe.bundles.workspace import Workspace, register_file_tools
308
+
309
+ registry = ToolRegistry()
310
+ # workspace_for(ctx) -> Workspace (a sandboxed root per user):
311
+ register_file_tools(registry, lambda ctx: Workspace(f"/data/{ctx.user_id}"))
312
+ store = JsonlRunStore("/var/lib/myapp/agent") # zero-DB default
313
+ host = AgentHost(registry,
314
+ LLMConfig(model=..., base_url=..., api_key=...),
315
+ store, build_system_prompt=my_prompt_builder)
316
+
317
+ ctx = AgentContext(run_id=rid, user_id=uid)
318
+ async for event in host.run(ctx, task, history=prior_turns):
319
+ ... # forward run_start / step / tool_call / tool_result / done to your frontend
320
+ ```
321
+ A host supplies only its **tools**, **system prompt**, and (optionally) a store
322
+ backend — the engine, persistence, run envelope, undo, and (via `subagents`)
323
+ delegation are all reused.
324
+
325
+ ## More
326
+ - `examples/minimal_host.py` — a runnable, offline minimal host (tools → run →
327
+ events → undo); runs in CI.
328
+ - `examples/live_host.py` — the same flow against a real OpenAI-compatible
329
+ endpoint (set `BETA_API_KEY` / `BETA_BASE_URL` / `BETA_MODEL`; no default
330
+ endpoint, it never spends tokens by accident).
331
+ - `CHANGELOG.md` — what changed and when.
332
+
333
+ ## License
334
+ MIT
lithe-0.9.4/README.md ADDED
@@ -0,0 +1,287 @@
1
+ # lithe
2
+
3
+ A reusable, **storage-free** ReAct agent kernel: the engine, framework and
4
+ protocols that drive a tool-calling agent, plus optional capability bundles. It
5
+ knows nothing about how runs are stored (or even whether they are) — persistence
6
+ is an optional `EventSink` a host plugs in. Any application (a thesis-writing
7
+ platform, a coding agent, ...) implements its own tools + prompt + storage and
8
+ reuses this kernel.
9
+
10
+ ## Core (zero I/O, zero business)
11
+ - `runtime` — `AgentRuntime` ReAct loop + event stream. Supports cancellation
12
+ (`run(..., stop=Event|callable)` → `cancelled` event, `status="cancelled"`;
13
+ checked between streaming deltas too, closing the in-flight model stream
14
+ instead of paying for a response nobody wants) and streaming
15
+ (`LLMConfig(stream=True)` → `assistant_delta` events while the
16
+ model generates; a final full `assistant` event always follows). Every
17
+ `tool_call` event is announced before any of the step's tools execute, so a
18
+ slow tool never hides what is pending on the frontend; result events still
19
+ follow in model order. Every model call emits a
20
+ `usage` event — prompt/completion/total tokens, that call's
21
+ cost, and context fullness (`context_tokens`, `context_chars`,
22
+ `context_window`, `context_percent` when `LLMConfig(context_window=...)` is
23
+ set) — so a frontend can show live token/context gauges; `RunStats` and the
24
+ host `done` event carry the cumulative breakdown. Run budgets
25
+ (`max_cost` / `max_total_tokens`) cut a runaway run short with
26
+ `status="budget_exceeded"` — no further tool execution, no further model
27
+ calls — and a host may seed `stats` with prior-conversation totals to
28
+ budget across runs. A stuck model reissuing the *identical* (tool, args)
29
+ call more than `repeat_call_limit` (default 3) times gets an inline nudge
30
+ inside that tool's result, feeding self-correction (`None` disables).
31
+ Malformed tool-call arguments come back to the model as failed tool
32
+ results instead of executing with empty/wrong args. A run that exhausts
33
+ its step budget gets a forced toolless wrap-up call (`tool_choice="none"`)
34
+ so it ends with the model's summary, reporting `status="max_steps"` (and
35
+ never executing the stubborn model's further tool calls).
36
+ `LLMConfig(temperature=..., max_tokens=...)` are forwarded on every call,
37
+ and a generation cut off by the token cap (`finish_reason="length"`) is
38
+ marked — the notice rides in the final text, the per-call `usage` event
39
+ carries `finish_reason`, and a tool_call whose arguments JSON was truncated
40
+ gets a "cut off by max_tokens, re-issue the call" error instead of a bare
41
+ "invalid JSON" that invites a byte-identical retry. An empty response (no
42
+ text, no tool calls) is retried once and otherwise ends the run with
43
+ `status="empty_response"` instead of an empty "success". A missing
44
+ `tool_call` id is synthesized at one point (streamed and non-streamed
45
+ alike), so a gateway that omits ids never poisons the next request with
46
+ `tool_call_id: null` — and streamed calls no longer mint colliding
47
+ `call_0`-style ids across steps. Synthetic run endings (max-step /
48
+ budget notices) are recorded like any assistant turn, so sinks and the
49
+ next run's replay see why the run stopped. Tool calls execute in model
50
+ order with consecutive READ tools parallel and every WRITE/META tool alone
51
+ (no write races); `ToolSpec(timeout=...)` cancels a hung call. A mid-run
52
+ context budget (`context_budget`, default 400k chars) shrinks old tool
53
+ results head+tail so long runs don't blow the context window. Sinks are
54
+ error-isolated (a broken display/record sink logs instead of killing the
55
+ run; `strict_records=True` opts record failures back into fatal).
56
+ `AgentRuntime(envelope=True)` emits the `run_start`/`done` envelope itself
57
+ for hosts driving the runtime directly, and `AgentRuntime(http_client=...)`
58
+ (forwarded by `AgentHost`) reuses one host-owned `httpx.AsyncClient` across
59
+ runs — connection pooling, limits, proxy/verify config for service hosts.
60
+ - `llm` — OpenAI-compatible chat client (retry / jittered backoff / response-shape
61
+ validation / `Retry-After`-aware 429 handling / fatal-4xx fail-fast) + SSE
62
+ streaming helpers; transports normalize usage to
63
+ `prompt_tokens`/`completion_tokens`/`total_tokens` across chat-completions
64
+ and Responses shapes
65
+ - `tools` — `ToolRegistry` (register / unregister / dispatch / mode filtering /
66
+ argument validation / per-tool timeout) + pre-dispatch **middleware** via
67
+ ``add_middleware`` (audit / quota / human-in-the-loop confirmation of write
68
+ tools)
69
+ - `actions` — `Action` + `UndoEngine` (pure, storage-free undo; reverters may
70
+ be sync or async)
71
+ - `memory` — `replay_messages` / `recap_text` / `window_with_recap` /
72
+ `run_timeline` (reconstruct a stored turn's ordered event timeline) +
73
+ `MemoryProvider`; replay reconciliation is two-sided *and* window-safe
74
+ (a tool row whose calling assistant fell outside the window is dropped,
75
+ not sent as an orphan first message)
76
+ - `events` — `EventSink` (display + record channels) + SSE serialization
77
+ (`to_sse` degrades non-JSON values via `str()` — a `datetime` inside a
78
+ tool's `ui` payload can't crash the host's SSE layer)
79
+ - `context` / `modes` — `AgentContext` + `AgentMode` / `ToolCategory`;
80
+ host-defined modes via ``register_mode(name, categories)`` (unknown modes
81
+ raise instead of silently degrading to read-only). `AgentContext.shared`
82
+ is per-run state shared *by reference* with subagent contexts — the
83
+ vehicle for cross-context coordination (the workspace stale-file guard's
84
+ revision map, the run's cancellation handle)
85
+
86
+ ## Optional bundles (`lithe.bundles`)
87
+ - `host` — `AgentHost` host-adapter framework: message assembly, run envelope
88
+ (`run_start`/`done`), error funneling, `StoreSink` (persist via a store),
89
+ `undo_run` (zero-config: bundled tools register reverters keyed by their
90
+ action kinds, and reverters see the host's `extra` context; a reversion
91
+ whose *status mark* fails surfaces in the report instead of being
92
+ swallowed), `DictToolAdapter` (wrap a dict-based tool system — specs
93
+ without a handler log a warning instead of silently vanishing, and a
94
+ `reverters=` map wires custom action kinds into undo).
95
+ Eliminates the per-host boilerplate round 1
96
+ left behind. `host.run(..., stop=...)` forwards cancellation to the runtime;
97
+ `AgentHost(max_cost=..., max_total_tokens=..., repeat_call_limit=...)`
98
+ forwards the runtime's budget / repeat guards to every run it builds;
99
+ `AgentHost(http_client=...)` shares one HTTP client across runs, and
100
+ `AgentHost(capture_actions=False)` turns off `StoreSink`'s automatic
101
+ action capture for hosts that persist actions themselves.
102
+ - `store` — `RunStore`/`ConversationStore`/`BlobStore` Protocols +
103
+ `JsonlRunStore` (default, **zero-database** JSONL + content-addressed blob
104
+ spillover). Hosts wanting a DB implement the Protocols; the default needs
105
+ none. Action values larger than `spill_threshold` (default 8KB) externalize
106
+ to blobs and rehydrate transparently on read; id counters are in-memory so
107
+ appends don't rescan the stream. `list_actions(subagent=...)` filters by
108
+ subagent during the fold (unselected rows skip blob rehydration — the
109
+ subagent engine's snapshots stay O(one worker) on long runs); torn lines
110
+ log a warning and count in `dropped_lines` instead of vanishing silently;
111
+ `fsync=True` flushes each append for crash-durability.
112
+ - `subagents` — `SubagentEngine` + `SubagentRoster`/`SubagentSpec` + `delegate`
113
+ tool: isolated worker agents the orchestrator hands subtasks to, tagged so undo
114
+ still reverts them while the orchestrator's context stays lean.
115
+ `delegate_parallel` fans independent tasks out concurrently (bounded by
116
+ `max_parallel`, failures isolated per agent, duplicate agents in one batch
117
+ rejected — they would claim each other's actions); `SubagentEngine(on_subagent_event=...)`
118
+ streams live `subagent_progress` heartbeats to a host push channel (text,
119
+ args and summaries capped — a 50KB `write_file` payload never rides the
120
+ callback); `SubagentSpec(transport=...)` routes a subagent to a different
121
+ endpoint. `make_delegate_tool(engine, timeout=...)` caps one delegation's
122
+ wall time so a hung worker cannot hold the orchestrator's step forever.
123
+ Budget caps apply per subagent run (each delegation gets its own
124
+ `max_cost` / `max_total_tokens`) while each delegation's spend folds into
125
+ the parent run's `done` event, stats and store row (with a
126
+ `subagent_*` breakdown); cancellation of the orchestrating run
127
+ propagates into an in-flight subagent (its stop handle rides in
128
+ `ctx.shared`); a delegation's own mutations are identified by an
129
+ id-membership snapshot, so stores with opaque (non-integer) action ids
130
+ count them correctly.
131
+ - `admin` — `tool_categories` / `list_tools_admin` / `list_tool_packages_admin` /
132
+ `check_packages` over a registry + display packages (admin-panel source;
133
+ packages are grouping only, tools stay per-name togglable).
134
+ - `patch` — `apply_patch`: line-oriented multi-file edits via the Codex
135
+ `*** Begin Patch` envelope (add/update/move/delete files, `@@` chunks of
136
+ context/`-`/`+` lines, `*** End of File` anchoring). Chunk location runs a
137
+ four-pass fuzzy ladder (exact → trailing-ws → strip → Unicode-punctuation
138
+ fold); `*** End of File` chunks run the tail-anchored ladder at full
139
+ strength before any forward match, so a whitespace-mismatched tail beats
140
+ an exact look-alike earlier in the file instead of silently editing the
141
+ wrong site; two same-position insertions keep document order. Application
142
+ is all-or-nothing against an in-memory overlay (later
143
+ sections of the same file chain; create-then-edit works), so a bad hunk
144
+ leaves the workspace untouched. Same `file_change` events + one new undo
145
+ kind (`file_delete`).
146
+ - `workspace` — sandboxed file I/O + read/write/edit/list/search/glob tools +
147
+ file undo reverters. Directory walks (`list_files` / `search_files` /
148
+ `glob_files`) never follow symlinks — code executed by `run_code` could
149
+ otherwise plant a link to a host file and read it back through a walk,
150
+ bypassing the path guard (symlinks list as an opaque `symlink` type;
151
+ `list_files(dirs=...)` routes the requested folders through the same path
152
+ guard). `read_file` prefixes every line with its 1-based
153
+ number and supports `offset`/`limit` line-window reads. `edit_file` refuses
154
+ ambiguous `old_text` (multi-match) unless `replace_all` is set, returns a
155
+ diff, and falls back to a whole-line fuzzy match (shared ladder,
156
+ trailing-blank aligned like `apply_patch`) when the
157
+ exact substring misses. All write tools guard against stale content: a
158
+ file read this run and changed out-of-band is refused with "re-read it"
159
+ instead of being clobbered; the revision map lives in `ctx.shared`, so the
160
+ guard spans the orchestrator and every (including parallel) subagent of
161
+ the run. `search_files` greps content
162
+ by regex (dir / glob filters) with a 30s tool timeout, a 10s scan budget
163
+ and a 10k-char line cap (catastrophic-backtracking patterns and huge
164
+ minified lines can't hang the loop); `glob_files` matches paths by pattern.
165
+ - `todos` — a per-scope task list the agent plans against: `TodoStore`
166
+ (pure container) + `JsonTodoStore` (atomic tmp+rename JSON persistence;
167
+ malformed items are dropped on load instead of crashing the system
168
+ prompt) + `update_todos`/`list_todos` tools (`todo_replace` reverter) +
169
+ `todos_block` for splicing the list into a system prompt.
170
+ - `images` — tool-tier image perception, no kernel changes. `image_info`:
171
+ stdlib-only header probe (PNG/JPEG/GIF/BMP/WEBP — dimensions, dpi, color
172
+ mode) answering deterministic questions with zero model calls; it stats
173
+ first and reads only a bounded header window off the event loop.
174
+ `analyze_image`: ONE vision-model call (OpenAI `image_url` data-URL block +
175
+ the question), registered only when a `LLMConfig` is passed — the host's
176
+ main config reuses the main model, a dedicated one routes vision elsewhere.
177
+ Size is checked by stat before reading (a 2GB upload is refused, not
178
+ loaded), file reads run off the event loop, and `detail="auto"` shares a
179
+ cache key with an omitted detail (the API treats them identically — no
180
+ double billing). Images never enter the main conversation (answers are
181
+ memoized per file hash + question), so context budget / trimming / replay
182
+ stay untouched.
183
+ - `sandbox` — Python code execution (bubblewrap or passthrough backend);
184
+ output truncation keeps head+tail so tracebacks at the end stay visible.
185
+ `run_code`/`run_file` are WRITE-classified (executing model-written code
186
+ can mutate the workspace): invisible in read-only modes and never run in
187
+ parallel with other tool calls. Their side effects produce no undo
188
+ records (the tool descriptions say so) — durable edits belong in
189
+ `write_file`/`edit_file`/`apply_patch`. Note the contract difference:
190
+ `register_code_tools` takes `workspace_for(ctx) -> root path` (a `str`),
191
+ while the workspace/images bundles take `workspace_for(ctx) -> Workspace`.
192
+ - `download` — `download_file(url, path?)`: stream one HTTP(S) resource into
193
+ the workspace (default `downloads/`), the controlled ingress the
194
+ network-isolated sandbox deliberately lacks — fetching is a bounded,
195
+ audited tool while executing model code stays offline. SSRF guard: http/
196
+ https only, redirects followed manually with every hop re-validated, and
197
+ all resolved addresses of every hop must be globally routable
198
+ (`ipaddress.is_global` rejects loopback/private/link-local/CGN/reserved/
199
+ multicast — cloud metadata included); checking all A/AAAA records up front
200
+ narrows DNS rebinding to the TTL window. Size cap by `Content-Length`
201
+ pre-check plus streaming cutoff (a lying header gets cut mid-stream and
202
+ the partial `.part` file removed; the file lands atomically via rename).
203
+ Existing targets are refused (the model picks a new name), so nothing
204
+ undoable is mutated. Defaults: 64MB cap, 120s total budget (kernel
205
+ `ToolSpec` timeout as backstop), 5 redirects; WRITE-classified like the
206
+ other workspace-mutating tools.
207
+ - `skills` — markdown skill libraries (`SkillLibrary` flat dir; package-aware
208
+ `SkillPackages` with `RemoteSkillSource` registry mirrors — refresh
209
+ failures clean their staging dir, log, and keep the previous cache) +
210
+ `load_skill` tool. Skills do file/network I/O, hence a bundle — `import
211
+ lithe` stays zero-I/O (deprecated `lithe.skills` alias kept; it
212
+ warns on import).
213
+ - `mcp` — `MCPManager` + `MCPServerConfig` + `parse_servers`: bridge external
214
+ MCP servers into the
215
+ registry — stdio transport (`command=[...]`, e.g. `npx -y @z_ai/mcp-server`)
216
+ or streamable-http (`url=` + `headers=` for auth, `Mcp-Session-Id` handled
217
+ automatically). `tools/list` pagination (`nextCursor`) is followed, so
218
+ paginated servers don't silently lose half their tools. Spawned servers
219
+ get only a safe env allow-list plus the configured `env` (PATH/locale/
220
+ HOME/TMPDIR) — never all of `os.environ` with its secrets — unless
221
+ `inherit_env=True` opts back into the legacy behavior. URLs log as
222
+ scheme://host only (credentials in the query or userinfo never reach a
223
+ log line). Sessions outlive registry rebuilds and lazily self-heal: a dead
224
+ session is restarted on the next tool call (`revive`), and
225
+ `attach`/`ensure` re-register tools on any fresh registry (`ensure` is the
226
+ idempotent variant for hosts that cache registries across runs).
227
+ `tool_allowlist`/`tool_blocklist` filter by the server-side tool name.
228
+ `readOnlyHint` annotations map to the READ category; MCP tools carry no undo
229
+ reverters. Stdlib-only
230
+ (newline-delimited JSON-RPC / plain POST + SSE), secrets stay host-side:
231
+
232
+ ```python
233
+ from lithe.bundles import MCPManager, MCPServerConfig
234
+
235
+ manager = MCPManager([MCPServerConfig(
236
+ name="zai", command=["npx", "-y", "@z_ai/mcp-server"],
237
+ env={"Z_AI_API_KEY": "...", "Z_AI_MODE": "ZHIPU"},
238
+ default_category="read")])
239
+ await manager.attach(registry) # registry gains zai__* tools
240
+ ```
241
+
242
+ ## Install
243
+ ```bash
244
+ pip install lithe
245
+ # local dev (editable + test/lint deps):
246
+ pip install -e ".[dev]"
247
+ ```
248
+
249
+ ## Storage model
250
+ The kernel stores nothing. A host provides:
251
+ - an **`EventSink`** (write side) — persists message records however it likes
252
+ (DB / file / nowhere);
253
+ - a **`MemoryProvider`** (read side, optional) — replays prior turns;
254
+ - an **`UndoEngine`** fed from wherever the host kept actions.
255
+
256
+ ## Minimal host sketch
257
+ ```python
258
+ from lithe import AgentContext, LLMConfig, ToolRegistry
259
+ from lithe.bundles import AgentHost, JsonlRunStore
260
+ from lithe.bundles.workspace import Workspace, register_file_tools
261
+
262
+ registry = ToolRegistry()
263
+ # workspace_for(ctx) -> Workspace (a sandboxed root per user):
264
+ register_file_tools(registry, lambda ctx: Workspace(f"/data/{ctx.user_id}"))
265
+ store = JsonlRunStore("/var/lib/myapp/agent") # zero-DB default
266
+ host = AgentHost(registry,
267
+ LLMConfig(model=..., base_url=..., api_key=...),
268
+ store, build_system_prompt=my_prompt_builder)
269
+
270
+ ctx = AgentContext(run_id=rid, user_id=uid)
271
+ async for event in host.run(ctx, task, history=prior_turns):
272
+ ... # forward run_start / step / tool_call / tool_result / done to your frontend
273
+ ```
274
+ A host supplies only its **tools**, **system prompt**, and (optionally) a store
275
+ backend — the engine, persistence, run envelope, undo, and (via `subagents`)
276
+ delegation are all reused.
277
+
278
+ ## More
279
+ - `examples/minimal_host.py` — a runnable, offline minimal host (tools → run →
280
+ events → undo); runs in CI.
281
+ - `examples/live_host.py` — the same flow against a real OpenAI-compatible
282
+ endpoint (set `BETA_API_KEY` / `BETA_BASE_URL` / `BETA_MODEL`; no default
283
+ endpoint, it never spends tokens by accident).
284
+ - `CHANGELOG.md` — what changed and when.
285
+
286
+ ## License
287
+ MIT