lithe 0.9.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lithe-0.9.4/LICENSE +21 -0
- lithe-0.9.4/PKG-INFO +334 -0
- lithe-0.9.4/README.md +287 -0
- lithe-0.9.4/lithe/__init__.py +64 -0
- lithe-0.9.4/lithe/actions.py +105 -0
- lithe-0.9.4/lithe/bundles/__init__.py +62 -0
- lithe-0.9.4/lithe/bundles/_textmatch.py +115 -0
- lithe-0.9.4/lithe/bundles/admin.py +133 -0
- lithe-0.9.4/lithe/bundles/download.py +272 -0
- lithe-0.9.4/lithe/bundles/host.py +602 -0
- lithe-0.9.4/lithe/bundles/images.py +362 -0
- lithe-0.9.4/lithe/bundles/mcp.py +742 -0
- lithe-0.9.4/lithe/bundles/patch.py +487 -0
- lithe-0.9.4/lithe/bundles/sandbox.py +249 -0
- lithe-0.9.4/lithe/bundles/skills.py +485 -0
- lithe-0.9.4/lithe/bundles/store/__init__.py +18 -0
- lithe-0.9.4/lithe/bundles/store/jsonl.py +392 -0
- lithe-0.9.4/lithe/bundles/store/protocol.py +214 -0
- lithe-0.9.4/lithe/bundles/subagents.py +543 -0
- lithe-0.9.4/lithe/bundles/todos.py +275 -0
- lithe-0.9.4/lithe/bundles/workspace.py +654 -0
- lithe-0.9.4/lithe/context.py +82 -0
- lithe-0.9.4/lithe/events.py +85 -0
- lithe-0.9.4/lithe/llm.py +304 -0
- lithe-0.9.4/lithe/memory.py +353 -0
- lithe-0.9.4/lithe/modes.py +70 -0
- lithe-0.9.4/lithe/py.typed +0 -0
- lithe-0.9.4/lithe/runtime.py +899 -0
- lithe-0.9.4/lithe/skills.py +28 -0
- lithe-0.9.4/lithe/tools.py +281 -0
- lithe-0.9.4/lithe/transports.py +726 -0
- lithe-0.9.4/lithe.egg-info/PKG-INFO +334 -0
- lithe-0.9.4/lithe.egg-info/SOURCES.txt +57 -0
- lithe-0.9.4/lithe.egg-info/dependency_links.txt +1 -0
- lithe-0.9.4/lithe.egg-info/requires.txt +6 -0
- lithe-0.9.4/lithe.egg-info/top_level.txt +1 -0
- lithe-0.9.4/pyproject.toml +52 -0
- lithe-0.9.4/setup.cfg +4 -0
- lithe-0.9.4/tests/test_lithe.py +375 -0
- lithe-0.9.4/tests/test_lithe_action_bridge.py +229 -0
- lithe-0.9.4/tests/test_lithe_admin.py +76 -0
- lithe-0.9.4/tests/test_lithe_download_bundle.py +301 -0
- lithe-0.9.4/tests/test_lithe_e2e.py +113 -0
- lithe-0.9.4/tests/test_lithe_host_store.py +618 -0
- lithe-0.9.4/tests/test_lithe_images_bundle.py +202 -0
- lithe-0.9.4/tests/test_lithe_llm.py +101 -0
- lithe-0.9.4/tests/test_lithe_mcp_bundle.py +310 -0
- lithe-0.9.4/tests/test_lithe_mcp_http.py +155 -0
- lithe-0.9.4/tests/test_lithe_memory.py +145 -0
- lithe-0.9.4/tests/test_lithe_p2_round.py +842 -0
- lithe-0.9.4/tests/test_lithe_patch.py +571 -0
- lithe-0.9.4/tests/test_lithe_runtime.py +1251 -0
- lithe-0.9.4/tests/test_lithe_sandbox_bundle.py +213 -0
- lithe-0.9.4/tests/test_lithe_skills_bundle.py +87 -0
- lithe-0.9.4/tests/test_lithe_skills_packages.py +152 -0
- lithe-0.9.4/tests/test_lithe_subagents.py +506 -0
- lithe-0.9.4/tests/test_lithe_todos_bundle.py +205 -0
- lithe-0.9.4/tests/test_lithe_transports.py +618 -0
- lithe-0.9.4/tests/test_lithe_workspace_bundle.py +449 -0
lithe-0.9.4/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 betaloop contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
lithe-0.9.4/PKG-INFO
ADDED
|
@@ -0,0 +1,334 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: lithe
|
|
3
|
+
Version: 0.9.4
|
|
4
|
+
Summary: A reusable, storage-free ReAct agent kernel (engine + framework + protocols) with optional capability bundles.
|
|
5
|
+
License: MIT License
|
|
6
|
+
|
|
7
|
+
Copyright (c) 2026 betaloop contributors
|
|
8
|
+
|
|
9
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
10
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
11
|
+
in the Software without restriction, including without limitation the rights
|
|
12
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
13
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
14
|
+
furnished to do so, subject to the following conditions:
|
|
15
|
+
|
|
16
|
+
The above copyright notice and this permission notice shall be included in all
|
|
17
|
+
copies or substantial portions of the Software.
|
|
18
|
+
|
|
19
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
20
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
21
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
22
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
23
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
24
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
25
|
+
SOFTWARE.
|
|
26
|
+
|
|
27
|
+
Keywords: agent,llm,react,kernel,tool-calling,mcp
|
|
28
|
+
Classifier: Development Status :: 4 - Beta
|
|
29
|
+
Classifier: Intended Audience :: Developers
|
|
30
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
31
|
+
Classifier: Programming Language :: Python :: 3
|
|
32
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
33
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
34
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
35
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
36
|
+
Classifier: Topic :: Software Development :: Libraries :: Application Frameworks
|
|
37
|
+
Classifier: Typing :: Typed
|
|
38
|
+
Requires-Python: >=3.10
|
|
39
|
+
Description-Content-Type: text/markdown
|
|
40
|
+
License-File: LICENSE
|
|
41
|
+
Requires-Dist: httpx>=0.24
|
|
42
|
+
Provides-Extra: dev
|
|
43
|
+
Requires-Dist: pytest>=8; extra == "dev"
|
|
44
|
+
Requires-Dist: pytest-asyncio>=0.23; extra == "dev"
|
|
45
|
+
Requires-Dist: ruff>=0.4; extra == "dev"
|
|
46
|
+
Dynamic: license-file
|
|
47
|
+
|
|
48
|
+
# lithe
|
|
49
|
+
|
|
50
|
+
A reusable, **storage-free** ReAct agent kernel: the engine, framework and
|
|
51
|
+
protocols that drive a tool-calling agent, plus optional capability bundles. It
|
|
52
|
+
knows nothing about how runs are stored (or even whether they are) — persistence
|
|
53
|
+
is an optional `EventSink` a host plugs in. Any application (a thesis-writing
|
|
54
|
+
platform, a coding agent, ...) implements its own tools + prompt + storage and
|
|
55
|
+
reuses this kernel.
|
|
56
|
+
|
|
57
|
+
## Core (zero I/O, zero business)
|
|
58
|
+
- `runtime` — `AgentRuntime` ReAct loop + event stream. Supports cancellation
|
|
59
|
+
(`run(..., stop=Event|callable)` → `cancelled` event, `status="cancelled"`;
|
|
60
|
+
checked between streaming deltas too, closing the in-flight model stream
|
|
61
|
+
instead of paying for a response nobody wants) and streaming
|
|
62
|
+
(`LLMConfig(stream=True)` → `assistant_delta` events while the
|
|
63
|
+
model generates; a final full `assistant` event always follows). Every
|
|
64
|
+
`tool_call` event is announced before any of the step's tools execute, so a
|
|
65
|
+
slow tool never hides what is pending on the frontend; result events still
|
|
66
|
+
follow in model order. Every model call emits a
|
|
67
|
+
`usage` event — prompt/completion/total tokens, that call's
|
|
68
|
+
cost, and context fullness (`context_tokens`, `context_chars`,
|
|
69
|
+
`context_window`, `context_percent` when `LLMConfig(context_window=...)` is
|
|
70
|
+
set) — so a frontend can show live token/context gauges; `RunStats` and the
|
|
71
|
+
host `done` event carry the cumulative breakdown. Run budgets
|
|
72
|
+
(`max_cost` / `max_total_tokens`) cut a runaway run short with
|
|
73
|
+
`status="budget_exceeded"` — no further tool execution, no further model
|
|
74
|
+
calls — and a host may seed `stats` with prior-conversation totals to
|
|
75
|
+
budget across runs. A stuck model reissuing the *identical* (tool, args)
|
|
76
|
+
call more than `repeat_call_limit` (default 3) times gets an inline nudge
|
|
77
|
+
inside that tool's result, feeding self-correction (`None` disables).
|
|
78
|
+
Malformed tool-call arguments come back to the model as failed tool
|
|
79
|
+
results instead of executing with empty/wrong args. A run that exhausts
|
|
80
|
+
its step budget gets a forced toolless wrap-up call (`tool_choice="none"`)
|
|
81
|
+
so it ends with the model's summary, reporting `status="max_steps"` (and
|
|
82
|
+
never executing the stubborn model's further tool calls).
|
|
83
|
+
`LLMConfig(temperature=..., max_tokens=...)` are forwarded on every call,
|
|
84
|
+
and a generation cut off by the token cap (`finish_reason="length"`) is
|
|
85
|
+
marked — the notice rides in the final text, the per-call `usage` event
|
|
86
|
+
carries `finish_reason`, and a tool_call whose arguments JSON was truncated
|
|
87
|
+
gets a "cut off by max_tokens, re-issue the call" error instead of a bare
|
|
88
|
+
"invalid JSON" that invites a byte-identical retry. An empty response (no
|
|
89
|
+
text, no tool calls) is retried once and otherwise ends the run with
|
|
90
|
+
`status="empty_response"` instead of an empty "success". A missing
|
|
91
|
+
`tool_call` id is synthesized at one point (streamed and non-streamed
|
|
92
|
+
alike), so a gateway that omits ids never poisons the next request with
|
|
93
|
+
`tool_call_id: null` — and streamed calls no longer mint colliding
|
|
94
|
+
`call_0`-style ids across steps. Synthetic run endings (max-step /
|
|
95
|
+
budget notices) are recorded like any assistant turn, so sinks and the
|
|
96
|
+
next run's replay see why the run stopped. Tool calls execute in model
|
|
97
|
+
order with consecutive READ tools parallel and every WRITE/META tool alone
|
|
98
|
+
(no write races); `ToolSpec(timeout=...)` cancels a hung call. A mid-run
|
|
99
|
+
context budget (`context_budget`, default 400k chars) shrinks old tool
|
|
100
|
+
results head+tail so long runs don't blow the context window. Sinks are
|
|
101
|
+
error-isolated (a broken display/record sink logs instead of killing the
|
|
102
|
+
run; `strict_records=True` opts record failures back into fatal).
|
|
103
|
+
`AgentRuntime(envelope=True)` emits the `run_start`/`done` envelope itself
|
|
104
|
+
for hosts driving the runtime directly, and `AgentRuntime(http_client=...)`
|
|
105
|
+
(forwarded by `AgentHost`) reuses one host-owned `httpx.AsyncClient` across
|
|
106
|
+
runs — connection pooling, limits, proxy/verify config for service hosts.
|
|
107
|
+
- `llm` — OpenAI-compatible chat client (retry / jittered backoff / response-shape
|
|
108
|
+
validation / `Retry-After`-aware 429 handling / fatal-4xx fail-fast) + SSE
|
|
109
|
+
streaming helpers; transports normalize usage to
|
|
110
|
+
`prompt_tokens`/`completion_tokens`/`total_tokens` across chat-completions
|
|
111
|
+
and Responses shapes
|
|
112
|
+
- `tools` — `ToolRegistry` (register / unregister / dispatch / mode filtering /
|
|
113
|
+
argument validation / per-tool timeout) + pre-dispatch **middleware** via
|
|
114
|
+
``add_middleware`` (audit / quota / human-in-the-loop confirmation of write
|
|
115
|
+
tools)
|
|
116
|
+
- `actions` — `Action` + `UndoEngine` (pure, storage-free undo; reverters may
|
|
117
|
+
be sync or async)
|
|
118
|
+
- `memory` — `replay_messages` / `recap_text` / `window_with_recap` /
|
|
119
|
+
`run_timeline` (reconstruct a stored turn's ordered event timeline) +
|
|
120
|
+
`MemoryProvider`; replay reconciliation is two-sided *and* window-safe
|
|
121
|
+
(a tool row whose calling assistant fell outside the window is dropped,
|
|
122
|
+
not sent as an orphan first message)
|
|
123
|
+
- `events` — `EventSink` (display + record channels) + SSE serialization
|
|
124
|
+
(`to_sse` degrades non-JSON values via `str()` — a `datetime` inside a
|
|
125
|
+
tool's `ui` payload can't crash the host's SSE layer)
|
|
126
|
+
- `context` / `modes` — `AgentContext` + `AgentMode` / `ToolCategory`;
|
|
127
|
+
host-defined modes via ``register_mode(name, categories)`` (unknown modes
|
|
128
|
+
raise instead of silently degrading to read-only). `AgentContext.shared`
|
|
129
|
+
is per-run state shared *by reference* with subagent contexts — the
|
|
130
|
+
vehicle for cross-context coordination (the workspace stale-file guard's
|
|
131
|
+
revision map, the run's cancellation handle)
|
|
132
|
+
|
|
133
|
+
## Optional bundles (`lithe.bundles`)
|
|
134
|
+
- `host` — `AgentHost` host-adapter framework: message assembly, run envelope
|
|
135
|
+
(`run_start`/`done`), error funneling, `StoreSink` (persist via a store),
|
|
136
|
+
`undo_run` (zero-config: bundled tools register reverters keyed by their
|
|
137
|
+
action kinds, and reverters see the host's `extra` context; a reversion
|
|
138
|
+
whose *status mark* fails surfaces in the report instead of being
|
|
139
|
+
swallowed), `DictToolAdapter` (wrap a dict-based tool system — specs
|
|
140
|
+
without a handler log a warning instead of silently vanishing, and a
|
|
141
|
+
`reverters=` map wires custom action kinds into undo).
|
|
142
|
+
Eliminates the per-host boilerplate round 1
|
|
143
|
+
left behind. `host.run(..., stop=...)` forwards cancellation to the runtime;
|
|
144
|
+
`AgentHost(max_cost=..., max_total_tokens=..., repeat_call_limit=...)`
|
|
145
|
+
forwards the runtime's budget / repeat guards to every run it builds;
|
|
146
|
+
`AgentHost(http_client=...)` shares one HTTP client across runs, and
|
|
147
|
+
`AgentHost(capture_actions=False)` turns off `StoreSink`'s automatic
|
|
148
|
+
action capture for hosts that persist actions themselves.
|
|
149
|
+
- `store` — `RunStore`/`ConversationStore`/`BlobStore` Protocols +
|
|
150
|
+
`JsonlRunStore` (default, **zero-database** JSONL + content-addressed blob
|
|
151
|
+
spillover). Hosts wanting a DB implement the Protocols; the default needs
|
|
152
|
+
none. Action values larger than `spill_threshold` (default 8KB) externalize
|
|
153
|
+
to blobs and rehydrate transparently on read; id counters are in-memory so
|
|
154
|
+
appends don't rescan the stream. `list_actions(subagent=...)` filters by
|
|
155
|
+
subagent during the fold (unselected rows skip blob rehydration — the
|
|
156
|
+
subagent engine's snapshots stay O(one worker) on long runs); torn lines
|
|
157
|
+
log a warning and count in `dropped_lines` instead of vanishing silently;
|
|
158
|
+
`fsync=True` flushes each append for crash-durability.
|
|
159
|
+
- `subagents` — `SubagentEngine` + `SubagentRoster`/`SubagentSpec` + `delegate`
|
|
160
|
+
tool: isolated worker agents the orchestrator hands subtasks to, tagged so undo
|
|
161
|
+
still reverts them while the orchestrator's context stays lean.
|
|
162
|
+
`delegate_parallel` fans independent tasks out concurrently (bounded by
|
|
163
|
+
`max_parallel`, failures isolated per agent, duplicate agents in one batch
|
|
164
|
+
rejected — they would claim each other's actions); `SubagentEngine(on_subagent_event=...)`
|
|
165
|
+
streams live `subagent_progress` heartbeats to a host push channel (text,
|
|
166
|
+
args and summaries capped — a 50KB `write_file` payload never rides the
|
|
167
|
+
callback); `SubagentSpec(transport=...)` routes a subagent to a different
|
|
168
|
+
endpoint. `make_delegate_tool(engine, timeout=...)` caps one delegation's
|
|
169
|
+
wall time so a hung worker cannot hold the orchestrator's step forever.
|
|
170
|
+
Budget caps apply per subagent run (each delegation gets its own
|
|
171
|
+
`max_cost` / `max_total_tokens`) while each delegation's spend folds into
|
|
172
|
+
the parent run's `done` event, stats and store row (with a
|
|
173
|
+
`subagent_*` breakdown); cancellation of the orchestrating run
|
|
174
|
+
propagates into an in-flight subagent (its stop handle rides in
|
|
175
|
+
`ctx.shared`); a delegation's own mutations are identified by an
|
|
176
|
+
id-membership snapshot, so stores with opaque (non-integer) action ids
|
|
177
|
+
count them correctly.
|
|
178
|
+
- `admin` — `tool_categories` / `list_tools_admin` / `list_tool_packages_admin` /
|
|
179
|
+
`check_packages` over a registry + display packages (admin-panel source;
|
|
180
|
+
packages are grouping only, tools stay per-name togglable).
|
|
181
|
+
- `patch` — `apply_patch`: line-oriented multi-file edits via the Codex
|
|
182
|
+
`*** Begin Patch` envelope (add/update/move/delete files, `@@` chunks of
|
|
183
|
+
context/`-`/`+` lines, `*** End of File` anchoring). Chunk location runs a
|
|
184
|
+
four-pass fuzzy ladder (exact → trailing-ws → strip → Unicode-punctuation
|
|
185
|
+
fold); `*** End of File` chunks run the tail-anchored ladder at full
|
|
186
|
+
strength before any forward match, so a whitespace-mismatched tail beats
|
|
187
|
+
an exact look-alike earlier in the file instead of silently editing the
|
|
188
|
+
wrong site; two same-position insertions keep document order. Application
|
|
189
|
+
is all-or-nothing against an in-memory overlay (later
|
|
190
|
+
sections of the same file chain; create-then-edit works), so a bad hunk
|
|
191
|
+
leaves the workspace untouched. Same `file_change` events + one new undo
|
|
192
|
+
kind (`file_delete`).
|
|
193
|
+
- `workspace` — sandboxed file I/O + read/write/edit/list/search/glob tools +
|
|
194
|
+
file undo reverters. Directory walks (`list_files` / `search_files` /
|
|
195
|
+
`glob_files`) never follow symlinks — code executed by `run_code` could
|
|
196
|
+
otherwise plant a link to a host file and read it back through a walk,
|
|
197
|
+
bypassing the path guard (symlinks list as an opaque `symlink` type;
|
|
198
|
+
`list_files(dirs=...)` routes the requested folders through the same path
|
|
199
|
+
guard). `read_file` prefixes every line with its 1-based
|
|
200
|
+
number and supports `offset`/`limit` line-window reads. `edit_file` refuses
|
|
201
|
+
ambiguous `old_text` (multi-match) unless `replace_all` is set, returns a
|
|
202
|
+
diff, and falls back to a whole-line fuzzy match (shared ladder,
|
|
203
|
+
trailing-blank aligned like `apply_patch`) when the
|
|
204
|
+
exact substring misses. All write tools guard against stale content: a
|
|
205
|
+
file read this run and changed out-of-band is refused with "re-read it"
|
|
206
|
+
instead of being clobbered; the revision map lives in `ctx.shared`, so the
|
|
207
|
+
guard spans the orchestrator and every (including parallel) subagent of
|
|
208
|
+
the run. `search_files` greps content
|
|
209
|
+
by regex (dir / glob filters) with a 30s tool timeout, a 10s scan budget
|
|
210
|
+
and a 10k-char line cap (catastrophic-backtracking patterns and huge
|
|
211
|
+
minified lines can't hang the loop); `glob_files` matches paths by pattern.
|
|
212
|
+
- `todos` — a per-scope task list the agent plans against: `TodoStore`
|
|
213
|
+
(pure container) + `JsonTodoStore` (atomic tmp+rename JSON persistence;
|
|
214
|
+
malformed items are dropped on load instead of crashing the system
|
|
215
|
+
prompt) + `update_todos`/`list_todos` tools (`todo_replace` reverter) +
|
|
216
|
+
`todos_block` for splicing the list into a system prompt.
|
|
217
|
+
- `images` — tool-tier image perception, no kernel changes. `image_info`:
|
|
218
|
+
stdlib-only header probe (PNG/JPEG/GIF/BMP/WEBP — dimensions, dpi, color
|
|
219
|
+
mode) answering deterministic questions with zero model calls; it stats
|
|
220
|
+
first and reads only a bounded header window off the event loop.
|
|
221
|
+
`analyze_image`: ONE vision-model call (OpenAI `image_url` data-URL block +
|
|
222
|
+
the question), registered only when a `LLMConfig` is passed — the host's
|
|
223
|
+
main config reuses the main model, a dedicated one routes vision elsewhere.
|
|
224
|
+
Size is checked by stat before reading (a 2GB upload is refused, not
|
|
225
|
+
loaded), file reads run off the event loop, and `detail="auto"` shares a
|
|
226
|
+
cache key with an omitted detail (the API treats them identically — no
|
|
227
|
+
double billing). Images never enter the main conversation (answers are
|
|
228
|
+
memoized per file hash + question), so context budget / trimming / replay
|
|
229
|
+
stay untouched.
|
|
230
|
+
- `sandbox` — Python code execution (bubblewrap or passthrough backend);
|
|
231
|
+
output truncation keeps head+tail so tracebacks at the end stay visible.
|
|
232
|
+
`run_code`/`run_file` are WRITE-classified (executing model-written code
|
|
233
|
+
can mutate the workspace): invisible in read-only modes and never run in
|
|
234
|
+
parallel with other tool calls. Their side effects produce no undo
|
|
235
|
+
records (the tool descriptions say so) — durable edits belong in
|
|
236
|
+
`write_file`/`edit_file`/`apply_patch`. Note the contract difference:
|
|
237
|
+
`register_code_tools` takes `workspace_for(ctx) -> root path` (a `str`),
|
|
238
|
+
while the workspace/images bundles take `workspace_for(ctx) -> Workspace`.
|
|
239
|
+
- `download` — `download_file(url, path?)`: stream one HTTP(S) resource into
|
|
240
|
+
the workspace (default `downloads/`), the controlled ingress the
|
|
241
|
+
network-isolated sandbox deliberately lacks — fetching is a bounded,
|
|
242
|
+
audited tool while executing model code stays offline. SSRF guard: http/
|
|
243
|
+
https only, redirects followed manually with every hop re-validated, and
|
|
244
|
+
all resolved addresses of every hop must be globally routable
|
|
245
|
+
(`ipaddress.is_global` rejects loopback/private/link-local/CGN/reserved/
|
|
246
|
+
multicast — cloud metadata included); checking all A/AAAA records up front
|
|
247
|
+
narrows DNS rebinding to the TTL window. Size cap by `Content-Length`
|
|
248
|
+
pre-check plus streaming cutoff (a lying header gets cut mid-stream and
|
|
249
|
+
the partial `.part` file removed; the file lands atomically via rename).
|
|
250
|
+
Existing targets are refused (the model picks a new name), so nothing
|
|
251
|
+
undoable is mutated. Defaults: 64MB cap, 120s total budget (kernel
|
|
252
|
+
`ToolSpec` timeout as backstop), 5 redirects; WRITE-classified like the
|
|
253
|
+
other workspace-mutating tools.
|
|
254
|
+
- `skills` — markdown skill libraries (`SkillLibrary` flat dir; package-aware
|
|
255
|
+
`SkillPackages` with `RemoteSkillSource` registry mirrors — refresh
|
|
256
|
+
failures clean their staging dir, log, and keep the previous cache) +
|
|
257
|
+
`load_skill` tool. Skills do file/network I/O, hence a bundle — `import
|
|
258
|
+
lithe` stays zero-I/O (deprecated `lithe.skills` alias kept; it
|
|
259
|
+
warns on import).
|
|
260
|
+
- `mcp` — `MCPManager` + `MCPServerConfig` + `parse_servers`: bridge external
|
|
261
|
+
MCP servers into the
|
|
262
|
+
registry — stdio transport (`command=[...]`, e.g. `npx -y @z_ai/mcp-server`)
|
|
263
|
+
or streamable-http (`url=` + `headers=` for auth, `Mcp-Session-Id` handled
|
|
264
|
+
automatically). `tools/list` pagination (`nextCursor`) is followed, so
|
|
265
|
+
paginated servers don't silently lose half their tools. Spawned servers
|
|
266
|
+
get only a safe env allow-list plus the configured `env` (PATH/locale/
|
|
267
|
+
HOME/TMPDIR) — never all of `os.environ` with its secrets — unless
|
|
268
|
+
`inherit_env=True` opts back into the legacy behavior. URLs log as
|
|
269
|
+
scheme://host only (credentials in the query or userinfo never reach a
|
|
270
|
+
log line). Sessions outlive registry rebuilds and lazily self-heal: a dead
|
|
271
|
+
session is restarted on the next tool call (`revive`), and
|
|
272
|
+
`attach`/`ensure` re-register tools on any fresh registry (`ensure` is the
|
|
273
|
+
idempotent variant for hosts that cache registries across runs).
|
|
274
|
+
`tool_allowlist`/`tool_blocklist` filter by the server-side tool name.
|
|
275
|
+
`readOnlyHint` annotations map to the READ category; MCP tools carry no undo
|
|
276
|
+
reverters. Stdlib-only
|
|
277
|
+
(newline-delimited JSON-RPC / plain POST + SSE), secrets stay host-side:
|
|
278
|
+
|
|
279
|
+
```python
|
|
280
|
+
from lithe.bundles import MCPManager, MCPServerConfig
|
|
281
|
+
|
|
282
|
+
manager = MCPManager([MCPServerConfig(
|
|
283
|
+
name="zai", command=["npx", "-y", "@z_ai/mcp-server"],
|
|
284
|
+
env={"Z_AI_API_KEY": "...", "Z_AI_MODE": "ZHIPU"},
|
|
285
|
+
default_category="read")])
|
|
286
|
+
await manager.attach(registry) # registry gains zai__* tools
|
|
287
|
+
```
|
|
288
|
+
|
|
289
|
+
## Install
|
|
290
|
+
```bash
|
|
291
|
+
pip install lithe
|
|
292
|
+
# local dev (editable + test/lint deps):
|
|
293
|
+
pip install -e ".[dev]"
|
|
294
|
+
```
|
|
295
|
+
|
|
296
|
+
## Storage model
|
|
297
|
+
The kernel stores nothing. A host provides:
|
|
298
|
+
- an **`EventSink`** (write side) — persists message records however it likes
|
|
299
|
+
(DB / file / nowhere);
|
|
300
|
+
- a **`MemoryProvider`** (read side, optional) — replays prior turns;
|
|
301
|
+
- an **`UndoEngine`** fed from wherever the host kept actions.
|
|
302
|
+
|
|
303
|
+
## Minimal host sketch
|
|
304
|
+
```python
|
|
305
|
+
from lithe import AgentContext, LLMConfig, ToolRegistry
|
|
306
|
+
from lithe.bundles import AgentHost, JsonlRunStore
|
|
307
|
+
from lithe.bundles.workspace import Workspace, register_file_tools
|
|
308
|
+
|
|
309
|
+
registry = ToolRegistry()
|
|
310
|
+
# workspace_for(ctx) -> Workspace (a sandboxed root per user):
|
|
311
|
+
register_file_tools(registry, lambda ctx: Workspace(f"/data/{ctx.user_id}"))
|
|
312
|
+
store = JsonlRunStore("/var/lib/myapp/agent") # zero-DB default
|
|
313
|
+
host = AgentHost(registry,
|
|
314
|
+
LLMConfig(model=..., base_url=..., api_key=...),
|
|
315
|
+
store, build_system_prompt=my_prompt_builder)
|
|
316
|
+
|
|
317
|
+
ctx = AgentContext(run_id=rid, user_id=uid)
|
|
318
|
+
async for event in host.run(ctx, task, history=prior_turns):
|
|
319
|
+
... # forward run_start / step / tool_call / tool_result / done to your frontend
|
|
320
|
+
```
|
|
321
|
+
A host supplies only its **tools**, **system prompt**, and (optionally) a store
|
|
322
|
+
backend — the engine, persistence, run envelope, undo, and (via `subagents`)
|
|
323
|
+
delegation are all reused.
|
|
324
|
+
|
|
325
|
+
## More
|
|
326
|
+
- `examples/minimal_host.py` — a runnable, offline minimal host (tools → run →
|
|
327
|
+
events → undo); runs in CI.
|
|
328
|
+
- `examples/live_host.py` — the same flow against a real OpenAI-compatible
|
|
329
|
+
endpoint (set `BETA_API_KEY` / `BETA_BASE_URL` / `BETA_MODEL`; no default
|
|
330
|
+
endpoint, it never spends tokens by accident).
|
|
331
|
+
- `CHANGELOG.md` — what changed and when.
|
|
332
|
+
|
|
333
|
+
## License
|
|
334
|
+
MIT
|
lithe-0.9.4/README.md
ADDED
|
@@ -0,0 +1,287 @@
|
|
|
1
|
+
# lithe
|
|
2
|
+
|
|
3
|
+
A reusable, **storage-free** ReAct agent kernel: the engine, framework and
|
|
4
|
+
protocols that drive a tool-calling agent, plus optional capability bundles. It
|
|
5
|
+
knows nothing about how runs are stored (or even whether they are) — persistence
|
|
6
|
+
is an optional `EventSink` a host plugs in. Any application (a thesis-writing
|
|
7
|
+
platform, a coding agent, ...) implements its own tools + prompt + storage and
|
|
8
|
+
reuses this kernel.
|
|
9
|
+
|
|
10
|
+
## Core (zero I/O, zero business)
|
|
11
|
+
- `runtime` — `AgentRuntime` ReAct loop + event stream. Supports cancellation
|
|
12
|
+
(`run(..., stop=Event|callable)` → `cancelled` event, `status="cancelled"`;
|
|
13
|
+
checked between streaming deltas too, closing the in-flight model stream
|
|
14
|
+
instead of paying for a response nobody wants) and streaming
|
|
15
|
+
(`LLMConfig(stream=True)` → `assistant_delta` events while the
|
|
16
|
+
model generates; a final full `assistant` event always follows). Every
|
|
17
|
+
`tool_call` event is announced before any of the step's tools execute, so a
|
|
18
|
+
slow tool never hides what is pending on the frontend; result events still
|
|
19
|
+
follow in model order. Every model call emits a
|
|
20
|
+
`usage` event — prompt/completion/total tokens, that call's
|
|
21
|
+
cost, and context fullness (`context_tokens`, `context_chars`,
|
|
22
|
+
`context_window`, `context_percent` when `LLMConfig(context_window=...)` is
|
|
23
|
+
set) — so a frontend can show live token/context gauges; `RunStats` and the
|
|
24
|
+
host `done` event carry the cumulative breakdown. Run budgets
|
|
25
|
+
(`max_cost` / `max_total_tokens`) cut a runaway run short with
|
|
26
|
+
`status="budget_exceeded"` — no further tool execution, no further model
|
|
27
|
+
calls — and a host may seed `stats` with prior-conversation totals to
|
|
28
|
+
budget across runs. A stuck model reissuing the *identical* (tool, args)
|
|
29
|
+
call more than `repeat_call_limit` (default 3) times gets an inline nudge
|
|
30
|
+
inside that tool's result, feeding self-correction (`None` disables).
|
|
31
|
+
Malformed tool-call arguments come back to the model as failed tool
|
|
32
|
+
results instead of executing with empty/wrong args. A run that exhausts
|
|
33
|
+
its step budget gets a forced toolless wrap-up call (`tool_choice="none"`)
|
|
34
|
+
so it ends with the model's summary, reporting `status="max_steps"` (and
|
|
35
|
+
never executing the stubborn model's further tool calls).
|
|
36
|
+
`LLMConfig(temperature=..., max_tokens=...)` are forwarded on every call,
|
|
37
|
+
and a generation cut off by the token cap (`finish_reason="length"`) is
|
|
38
|
+
marked — the notice rides in the final text, the per-call `usage` event
|
|
39
|
+
carries `finish_reason`, and a tool_call whose arguments JSON was truncated
|
|
40
|
+
gets a "cut off by max_tokens, re-issue the call" error instead of a bare
|
|
41
|
+
"invalid JSON" that invites a byte-identical retry. An empty response (no
|
|
42
|
+
text, no tool calls) is retried once and otherwise ends the run with
|
|
43
|
+
`status="empty_response"` instead of an empty "success". A missing
|
|
44
|
+
`tool_call` id is synthesized at one point (streamed and non-streamed
|
|
45
|
+
alike), so a gateway that omits ids never poisons the next request with
|
|
46
|
+
`tool_call_id: null` — and streamed calls no longer mint colliding
|
|
47
|
+
`call_0`-style ids across steps. Synthetic run endings (max-step /
|
|
48
|
+
budget notices) are recorded like any assistant turn, so sinks and the
|
|
49
|
+
next run's replay see why the run stopped. Tool calls execute in model
|
|
50
|
+
order with consecutive READ tools parallel and every WRITE/META tool alone
|
|
51
|
+
(no write races); `ToolSpec(timeout=...)` cancels a hung call. A mid-run
|
|
52
|
+
context budget (`context_budget`, default 400k chars) shrinks old tool
|
|
53
|
+
results head+tail so long runs don't blow the context window. Sinks are
|
|
54
|
+
error-isolated (a broken display/record sink logs instead of killing the
|
|
55
|
+
run; `strict_records=True` opts record failures back into fatal).
|
|
56
|
+
`AgentRuntime(envelope=True)` emits the `run_start`/`done` envelope itself
|
|
57
|
+
for hosts driving the runtime directly, and `AgentRuntime(http_client=...)`
|
|
58
|
+
(forwarded by `AgentHost`) reuses one host-owned `httpx.AsyncClient` across
|
|
59
|
+
runs — connection pooling, limits, proxy/verify config for service hosts.
|
|
60
|
+
- `llm` — OpenAI-compatible chat client (retry / jittered backoff / response-shape
|
|
61
|
+
validation / `Retry-After`-aware 429 handling / fatal-4xx fail-fast) + SSE
|
|
62
|
+
streaming helpers; transports normalize usage to
|
|
63
|
+
`prompt_tokens`/`completion_tokens`/`total_tokens` across chat-completions
|
|
64
|
+
and Responses shapes
|
|
65
|
+
- `tools` — `ToolRegistry` (register / unregister / dispatch / mode filtering /
|
|
66
|
+
argument validation / per-tool timeout) + pre-dispatch **middleware** via
|
|
67
|
+
``add_middleware`` (audit / quota / human-in-the-loop confirmation of write
|
|
68
|
+
tools)
|
|
69
|
+
- `actions` — `Action` + `UndoEngine` (pure, storage-free undo; reverters may
|
|
70
|
+
be sync or async)
|
|
71
|
+
- `memory` — `replay_messages` / `recap_text` / `window_with_recap` /
|
|
72
|
+
`run_timeline` (reconstruct a stored turn's ordered event timeline) +
|
|
73
|
+
`MemoryProvider`; replay reconciliation is two-sided *and* window-safe
|
|
74
|
+
(a tool row whose calling assistant fell outside the window is dropped,
|
|
75
|
+
not sent as an orphan first message)
|
|
76
|
+
- `events` — `EventSink` (display + record channels) + SSE serialization
|
|
77
|
+
(`to_sse` degrades non-JSON values via `str()` — a `datetime` inside a
|
|
78
|
+
tool's `ui` payload can't crash the host's SSE layer)
|
|
79
|
+
- `context` / `modes` — `AgentContext` + `AgentMode` / `ToolCategory`;
|
|
80
|
+
host-defined modes via ``register_mode(name, categories)`` (unknown modes
|
|
81
|
+
raise instead of silently degrading to read-only). `AgentContext.shared`
|
|
82
|
+
is per-run state shared *by reference* with subagent contexts — the
|
|
83
|
+
vehicle for cross-context coordination (the workspace stale-file guard's
|
|
84
|
+
revision map, the run's cancellation handle)
|
|
85
|
+
|
|
86
|
+
## Optional bundles (`lithe.bundles`)
|
|
87
|
+
- `host` — `AgentHost` host-adapter framework: message assembly, run envelope
|
|
88
|
+
(`run_start`/`done`), error funneling, `StoreSink` (persist via a store),
|
|
89
|
+
`undo_run` (zero-config: bundled tools register reverters keyed by their
|
|
90
|
+
action kinds, and reverters see the host's `extra` context; a reversion
|
|
91
|
+
whose *status mark* fails surfaces in the report instead of being
|
|
92
|
+
swallowed), `DictToolAdapter` (wrap a dict-based tool system — specs
|
|
93
|
+
without a handler log a warning instead of silently vanishing, and a
|
|
94
|
+
`reverters=` map wires custom action kinds into undo).
|
|
95
|
+
Eliminates the per-host boilerplate round 1
|
|
96
|
+
left behind. `host.run(..., stop=...)` forwards cancellation to the runtime;
|
|
97
|
+
`AgentHost(max_cost=..., max_total_tokens=..., repeat_call_limit=...)`
|
|
98
|
+
forwards the runtime's budget / repeat guards to every run it builds;
|
|
99
|
+
`AgentHost(http_client=...)` shares one HTTP client across runs, and
|
|
100
|
+
`AgentHost(capture_actions=False)` turns off `StoreSink`'s automatic
|
|
101
|
+
action capture for hosts that persist actions themselves.
|
|
102
|
+
- `store` — `RunStore`/`ConversationStore`/`BlobStore` Protocols +
|
|
103
|
+
`JsonlRunStore` (default, **zero-database** JSONL + content-addressed blob
|
|
104
|
+
spillover). Hosts wanting a DB implement the Protocols; the default needs
|
|
105
|
+
none. Action values larger than `spill_threshold` (default 8KB) externalize
|
|
106
|
+
to blobs and rehydrate transparently on read; id counters are in-memory so
|
|
107
|
+
appends don't rescan the stream. `list_actions(subagent=...)` filters by
|
|
108
|
+
subagent during the fold (unselected rows skip blob rehydration — the
|
|
109
|
+
subagent engine's snapshots stay O(one worker) on long runs); torn lines
|
|
110
|
+
log a warning and count in `dropped_lines` instead of vanishing silently;
|
|
111
|
+
`fsync=True` flushes each append for crash-durability.
|
|
112
|
+
- `subagents` — `SubagentEngine` + `SubagentRoster`/`SubagentSpec` + `delegate`
|
|
113
|
+
tool: isolated worker agents the orchestrator hands subtasks to, tagged so undo
|
|
114
|
+
still reverts them while the orchestrator's context stays lean.
|
|
115
|
+
`delegate_parallel` fans independent tasks out concurrently (bounded by
|
|
116
|
+
`max_parallel`, failures isolated per agent, duplicate agents in one batch
|
|
117
|
+
rejected — they would claim each other's actions); `SubagentEngine(on_subagent_event=...)`
|
|
118
|
+
streams live `subagent_progress` heartbeats to a host push channel (text,
|
|
119
|
+
args and summaries capped — a 50KB `write_file` payload never rides the
|
|
120
|
+
callback); `SubagentSpec(transport=...)` routes a subagent to a different
|
|
121
|
+
endpoint. `make_delegate_tool(engine, timeout=...)` caps one delegation's
|
|
122
|
+
wall time so a hung worker cannot hold the orchestrator's step forever.
|
|
123
|
+
Budget caps apply per subagent run (each delegation gets its own
|
|
124
|
+
`max_cost` / `max_total_tokens`) while each delegation's spend folds into
|
|
125
|
+
the parent run's `done` event, stats and store row (with a
|
|
126
|
+
`subagent_*` breakdown); cancellation of the orchestrating run
|
|
127
|
+
propagates into an in-flight subagent (its stop handle rides in
|
|
128
|
+
`ctx.shared`); a delegation's own mutations are identified by an
|
|
129
|
+
id-membership snapshot, so stores with opaque (non-integer) action ids
|
|
130
|
+
count them correctly.
|
|
131
|
+
- `admin` — `tool_categories` / `list_tools_admin` / `list_tool_packages_admin` /
|
|
132
|
+
`check_packages` over a registry + display packages (admin-panel source;
|
|
133
|
+
packages are grouping only, tools stay per-name togglable).
|
|
134
|
+
- `patch` — `apply_patch`: line-oriented multi-file edits via the Codex
|
|
135
|
+
`*** Begin Patch` envelope (add/update/move/delete files, `@@` chunks of
|
|
136
|
+
context/`-`/`+` lines, `*** End of File` anchoring). Chunk location runs a
|
|
137
|
+
four-pass fuzzy ladder (exact → trailing-ws → strip → Unicode-punctuation
|
|
138
|
+
fold); `*** End of File` chunks run the tail-anchored ladder at full
|
|
139
|
+
strength before any forward match, so a whitespace-mismatched tail beats
|
|
140
|
+
an exact look-alike earlier in the file instead of silently editing the
|
|
141
|
+
wrong site; two same-position insertions keep document order. Application
|
|
142
|
+
is all-or-nothing against an in-memory overlay (later
|
|
143
|
+
sections of the same file chain; create-then-edit works), so a bad hunk
|
|
144
|
+
leaves the workspace untouched. Same `file_change` events + one new undo
|
|
145
|
+
kind (`file_delete`).
|
|
146
|
+
- `workspace` — sandboxed file I/O + read/write/edit/list/search/glob tools +
|
|
147
|
+
file undo reverters. Directory walks (`list_files` / `search_files` /
|
|
148
|
+
`glob_files`) never follow symlinks — code executed by `run_code` could
|
|
149
|
+
otherwise plant a link to a host file and read it back through a walk,
|
|
150
|
+
bypassing the path guard (symlinks list as an opaque `symlink` type;
|
|
151
|
+
`list_files(dirs=...)` routes the requested folders through the same path
|
|
152
|
+
guard). `read_file` prefixes every line with its 1-based
|
|
153
|
+
number and supports `offset`/`limit` line-window reads. `edit_file` refuses
|
|
154
|
+
ambiguous `old_text` (multi-match) unless `replace_all` is set, returns a
|
|
155
|
+
diff, and falls back to a whole-line fuzzy match (shared ladder,
|
|
156
|
+
trailing-blank aligned like `apply_patch`) when the
|
|
157
|
+
exact substring misses. All write tools guard against stale content: a
|
|
158
|
+
file read this run and changed out-of-band is refused with "re-read it"
|
|
159
|
+
instead of being clobbered; the revision map lives in `ctx.shared`, so the
|
|
160
|
+
guard spans the orchestrator and every (including parallel) subagent of
|
|
161
|
+
the run. `search_files` greps content
|
|
162
|
+
by regex (dir / glob filters) with a 30s tool timeout, a 10s scan budget
|
|
163
|
+
and a 10k-char line cap (catastrophic-backtracking patterns and huge
|
|
164
|
+
minified lines can't hang the loop); `glob_files` matches paths by pattern.
|
|
165
|
+
- `todos` — a per-scope task list the agent plans against: `TodoStore`
|
|
166
|
+
(pure container) + `JsonTodoStore` (atomic tmp+rename JSON persistence;
|
|
167
|
+
malformed items are dropped on load instead of crashing the system
|
|
168
|
+
prompt) + `update_todos`/`list_todos` tools (`todo_replace` reverter) +
|
|
169
|
+
`todos_block` for splicing the list into a system prompt.
|
|
170
|
+
- `images` — tool-tier image perception, no kernel changes. `image_info`:
|
|
171
|
+
stdlib-only header probe (PNG/JPEG/GIF/BMP/WEBP — dimensions, dpi, color
|
|
172
|
+
mode) answering deterministic questions with zero model calls; it stats
|
|
173
|
+
first and reads only a bounded header window off the event loop.
|
|
174
|
+
`analyze_image`: ONE vision-model call (OpenAI `image_url` data-URL block +
|
|
175
|
+
the question), registered only when a `LLMConfig` is passed — the host's
|
|
176
|
+
main config reuses the main model, a dedicated one routes vision elsewhere.
|
|
177
|
+
Size is checked by stat before reading (a 2GB upload is refused, not
|
|
178
|
+
loaded), file reads run off the event loop, and `detail="auto"` shares a
|
|
179
|
+
cache key with an omitted detail (the API treats them identically — no
|
|
180
|
+
double billing). Images never enter the main conversation (answers are
|
|
181
|
+
memoized per file hash + question), so context budget / trimming / replay
|
|
182
|
+
stay untouched.
|
|
183
|
+
- `sandbox` — Python code execution (bubblewrap or passthrough backend);
|
|
184
|
+
output truncation keeps head+tail so tracebacks at the end stay visible.
|
|
185
|
+
`run_code`/`run_file` are WRITE-classified (executing model-written code
|
|
186
|
+
can mutate the workspace): invisible in read-only modes and never run in
|
|
187
|
+
parallel with other tool calls. Their side effects produce no undo
|
|
188
|
+
records (the tool descriptions say so) — durable edits belong in
|
|
189
|
+
`write_file`/`edit_file`/`apply_patch`. Note the contract difference:
|
|
190
|
+
`register_code_tools` takes `workspace_for(ctx) -> root path` (a `str`),
|
|
191
|
+
while the workspace/images bundles take `workspace_for(ctx) -> Workspace`.
|
|
192
|
+
- `download` — `download_file(url, path?)`: stream one HTTP(S) resource into
|
|
193
|
+
the workspace (default `downloads/`), the controlled ingress the
|
|
194
|
+
network-isolated sandbox deliberately lacks — fetching is a bounded,
|
|
195
|
+
audited tool while executing model code stays offline. SSRF guard: http/
|
|
196
|
+
https only, redirects followed manually with every hop re-validated, and
|
|
197
|
+
all resolved addresses of every hop must be globally routable
|
|
198
|
+
(`ipaddress.is_global` rejects loopback/private/link-local/CGN/reserved/
|
|
199
|
+
multicast — cloud metadata included); checking all A/AAAA records up front
|
|
200
|
+
narrows DNS rebinding to the TTL window. Size cap by `Content-Length`
|
|
201
|
+
pre-check plus streaming cutoff (a lying header gets cut mid-stream and
|
|
202
|
+
the partial `.part` file removed; the file lands atomically via rename).
|
|
203
|
+
Existing targets are refused (the model picks a new name), so nothing
|
|
204
|
+
undoable is mutated. Defaults: 64MB cap, 120s total budget (kernel
|
|
205
|
+
`ToolSpec` timeout as backstop), 5 redirects; WRITE-classified like the
|
|
206
|
+
other workspace-mutating tools.
|
|
207
|
+
- `skills` — markdown skill libraries (`SkillLibrary` flat dir; package-aware
|
|
208
|
+
`SkillPackages` with `RemoteSkillSource` registry mirrors — refresh
|
|
209
|
+
failures clean their staging dir, log, and keep the previous cache) +
|
|
210
|
+
`load_skill` tool. Skills do file/network I/O, hence a bundle — `import
|
|
211
|
+
lithe` stays zero-I/O (deprecated `lithe.skills` alias kept; it
|
|
212
|
+
warns on import).
|
|
213
|
+
- `mcp` — `MCPManager` + `MCPServerConfig` + `parse_servers`: bridge external
|
|
214
|
+
MCP servers into the
|
|
215
|
+
registry — stdio transport (`command=[...]`, e.g. `npx -y @z_ai/mcp-server`)
|
|
216
|
+
or streamable-http (`url=` + `headers=` for auth, `Mcp-Session-Id` handled
|
|
217
|
+
automatically). `tools/list` pagination (`nextCursor`) is followed, so
|
|
218
|
+
paginated servers don't silently lose half their tools. Spawned servers
|
|
219
|
+
get only a safe env allow-list plus the configured `env` (PATH/locale/
|
|
220
|
+
HOME/TMPDIR) — never all of `os.environ` with its secrets — unless
|
|
221
|
+
`inherit_env=True` opts back into the legacy behavior. URLs log as
|
|
222
|
+
scheme://host only (credentials in the query or userinfo never reach a
|
|
223
|
+
log line). Sessions outlive registry rebuilds and lazily self-heal: a dead
|
|
224
|
+
session is restarted on the next tool call (`revive`), and
|
|
225
|
+
`attach`/`ensure` re-register tools on any fresh registry (`ensure` is the
|
|
226
|
+
idempotent variant for hosts that cache registries across runs).
|
|
227
|
+
`tool_allowlist`/`tool_blocklist` filter by the server-side tool name.
|
|
228
|
+
`readOnlyHint` annotations map to the READ category; MCP tools carry no undo
|
|
229
|
+
reverters. Stdlib-only
|
|
230
|
+
(newline-delimited JSON-RPC / plain POST + SSE), secrets stay host-side:
|
|
231
|
+
|
|
232
|
+
```python
|
|
233
|
+
from lithe.bundles import MCPManager, MCPServerConfig
|
|
234
|
+
|
|
235
|
+
manager = MCPManager([MCPServerConfig(
|
|
236
|
+
name="zai", command=["npx", "-y", "@z_ai/mcp-server"],
|
|
237
|
+
env={"Z_AI_API_KEY": "...", "Z_AI_MODE": "ZHIPU"},
|
|
238
|
+
default_category="read")])
|
|
239
|
+
await manager.attach(registry) # registry gains zai__* tools
|
|
240
|
+
```
|
|
241
|
+
|
|
242
|
+
## Install
|
|
243
|
+
```bash
|
|
244
|
+
pip install lithe
|
|
245
|
+
# local dev (editable + test/lint deps):
|
|
246
|
+
pip install -e ".[dev]"
|
|
247
|
+
```
|
|
248
|
+
|
|
249
|
+
## Storage model
|
|
250
|
+
The kernel stores nothing. A host provides:
|
|
251
|
+
- an **`EventSink`** (write side) — persists message records however it likes
|
|
252
|
+
(DB / file / nowhere);
|
|
253
|
+
- a **`MemoryProvider`** (read side, optional) — replays prior turns;
|
|
254
|
+
- an **`UndoEngine`** fed from wherever the host kept actions.
|
|
255
|
+
|
|
256
|
+
## Minimal host sketch
|
|
257
|
+
```python
|
|
258
|
+
from lithe import AgentContext, LLMConfig, ToolRegistry
|
|
259
|
+
from lithe.bundles import AgentHost, JsonlRunStore
|
|
260
|
+
from lithe.bundles.workspace import Workspace, register_file_tools
|
|
261
|
+
|
|
262
|
+
registry = ToolRegistry()
|
|
263
|
+
# workspace_for(ctx) -> Workspace (a sandboxed root per user):
|
|
264
|
+
register_file_tools(registry, lambda ctx: Workspace(f"/data/{ctx.user_id}"))
|
|
265
|
+
store = JsonlRunStore("/var/lib/myapp/agent") # zero-DB default
|
|
266
|
+
host = AgentHost(registry,
|
|
267
|
+
LLMConfig(model=..., base_url=..., api_key=...),
|
|
268
|
+
store, build_system_prompt=my_prompt_builder)
|
|
269
|
+
|
|
270
|
+
ctx = AgentContext(run_id=rid, user_id=uid)
|
|
271
|
+
async for event in host.run(ctx, task, history=prior_turns):
|
|
272
|
+
... # forward run_start / step / tool_call / tool_result / done to your frontend
|
|
273
|
+
```
|
|
274
|
+
A host supplies only its **tools**, **system prompt**, and (optionally) a store
|
|
275
|
+
backend — the engine, persistence, run envelope, undo, and (via `subagents`)
|
|
276
|
+
delegation are all reused.
|
|
277
|
+
|
|
278
|
+
## More
|
|
279
|
+
- `examples/minimal_host.py` — a runnable, offline minimal host (tools → run →
|
|
280
|
+
events → undo); runs in CI.
|
|
281
|
+
- `examples/live_host.py` — the same flow against a real OpenAI-compatible
|
|
282
|
+
endpoint (set `BETA_API_KEY` / `BETA_BASE_URL` / `BETA_MODEL`; no default
|
|
283
|
+
endpoint, it never spends tokens by accident).
|
|
284
|
+
- `CHANGELOG.md` — what changed and when.
|
|
285
|
+
|
|
286
|
+
## License
|
|
287
|
+
MIT
|