otto-cli-agent 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent/README.md +22 -0
- agent/__init__.py +0 -0
- agent/cli/README.md +77 -0
- agent/cli/__init__.py +0 -0
- agent/cli/art.py +371 -0
- agent/cli/chat.py +309 -0
- agent/cli/clipboard.py +106 -0
- agent/cli/context.py +56 -0
- agent/cli/doctor.py +60 -0
- agent/cli/errors.py +58 -0
- agent/cli/eval.py +130 -0
- agent/cli/eval_claw.py +414 -0
- agent/cli/eval_compaction.py +106 -0
- agent/cli/eval_hle.py +92 -0
- agent/cli/eval_memory.py +253 -0
- agent/cli/eval_swe.py +172 -0
- agent/cli/lessons.py +97 -0
- agent/cli/main.py +67 -0
- agent/cli/modals.py +570 -0
- agent/cli/models.py +64 -0
- agent/cli/output.py +54 -0
- agent/cli/route.py +78 -0
- agent/cli/sessions.py +125 -0
- agent/cli/setup_screen.py +562 -0
- agent/cli/shell.py +548 -0
- agent/cli/tui.py +1807 -0
- agent/cli/ui.py +14 -0
- agent/cli/usage_panel.py +159 -0
- agent/config/README.md +7 -0
- agent/config/__init__.py +0 -0
- agent/config/envfile.py +76 -0
- agent/eval/README.md +76 -0
- agent/eval/__init__.py +0 -0
- agent/eval/claw_bench.py +1031 -0
- agent/eval/compaction_bench.py +229 -0
- agent/eval/data/README.md +10 -0
- agent/eval/data/claw/README.md +108 -0
- agent/eval/data/claw/llm_judge-gemini.patch +57 -0
- agent/eval/data/claw/otto.yaml +35 -0
- agent/eval/failures.py +276 -0
- agent/eval/golden/README.md +33 -0
- agent/eval/golden/code_01.json +6 -0
- agent/eval/golden/code_02.json +6 -0
- agent/eval/golden/code_03.json +6 -0
- agent/eval/golden/code_04.json +6 -0
- agent/eval/golden/code_05.json +6 -0
- agent/eval/golden/code_06.json +6 -0
- agent/eval/golden/math_01.json +6 -0
- agent/eval/golden/math_02.json +6 -0
- agent/eval/golden/math_03.json +6 -0
- agent/eval/golden/math_04.json +6 -0
- agent/eval/golden/math_05.json +6 -0
- agent/eval/golden/math_06.json +6 -0
- agent/eval/golden/nphard_gcp_01.json +6 -0
- agent/eval/golden/nphard_ksp_01.json +6 -0
- agent/eval/golden/nphard_math_binpacking_01.json +6 -0
- agent/eval/golden/nphard_math_clique_01.json +6 -0
- agent/eval/golden/nphard_math_setcover_01.json +6 -0
- agent/eval/golden/nphard_math_subsetsum_01.json +6 -0
- agent/eval/golden/nphard_tsp_01.json +6 -0
- agent/eval/golden/nphard_tsp_02.json +6 -0
- agent/eval/hle_bench.py +273 -0
- agent/eval/langfuse_sync.py +172 -0
- agent/eval/memory_bench.py +538 -0
- agent/eval/runner.py +174 -0
- agent/eval/single_agent.py +120 -0
- agent/eval/swe_bench.py +604 -0
- agent/eval/terminal_bench.py +345 -0
- agent/memory/README.md +102 -0
- agent/memory/__init__.py +42 -0
- agent/memory/embeddings.py +302 -0
- agent/memory/hashing.py +15 -0
- agent/memory/lessons.py +483 -0
- agent/memory/queue.py +531 -0
- agent/memory/retrieval.py +493 -0
- agent/memory/session.py +60 -0
- agent/memory/sessions.py +436 -0
- agent/memory/store.py +429 -0
- agent/memory/tokens.py +60 -0
- agent/memory/wiring.py +146 -0
- agent/pipeline/README.md +135 -0
- agent/pipeline/__init__.py +0 -0
- agent/pipeline/browsing.py +609 -0
- agent/pipeline/budget.py +403 -0
- agent/pipeline/codemap.py +254 -0
- agent/pipeline/evidence.py +325 -0
- agent/pipeline/execution.py +67 -0
- agent/pipeline/modes.py +137 -0
- agent/pipeline/native.py +1137 -0
- agent/pipeline/nodes.py +3644 -0
- agent/pipeline/pricing.py +209 -0
- agent/pipeline/progress.py +139 -0
- agent/pipeline/rag.py +139 -0
- agent/pipeline/research.py +1325 -0
- agent/pipeline/run.py +528 -0
- agent/pipeline/screen.py +77 -0
- agent/pipeline/state.py +220 -0
- agent/pipeline/toolkit.py +328 -0
- agent/pipeline/tools.py +1990 -0
- agent/pipeline/tracing.py +147 -0
- agent/pipeline/usage.py +251 -0
- agent/pipeline/vision.py +84 -0
- agent/pipeline/walkthrough.py +735 -0
- agent/pipeline/workspace.py +229 -0
- agent/router/README.md +60 -0
- agent/router/__init__.py +0 -0
- agent/router/automap.py +114 -0
- agent/router/health.py +229 -0
- agent/router/llm_provider/README.md +38 -0
- agent/router/llm_provider/__init__.py +202 -0
- agent/router/llm_provider/anthropic_provider.py +128 -0
- agent/router/llm_provider/base.py +507 -0
- agent/router/llm_provider/custom.py +152 -0
- agent/router/llm_provider/gemini_provider.py +122 -0
- agent/router/llm_provider/inception_provider.py +687 -0
- agent/router/llm_provider/openai_provider.py +151 -0
- agent/router/llm_provider/retired.py +145 -0
- agent/router/llm_provider/temperature.py +371 -0
- agent/router/mapping.py +579 -0
- agent/router/outcomes.py +363 -0
- agent/router/overrides.py +389 -0
- agent/router/reload.py +28 -0
- agent/router/router.py +413 -0
- agent/router/setup.py +123 -0
- otto_cli_agent-0.1.0.dist-info/METADATA +115 -0
- otto_cli_agent-0.1.0.dist-info/RECORD +129 -0
- otto_cli_agent-0.1.0.dist-info/WHEEL +4 -0
- otto_cli_agent-0.1.0.dist-info/entry_points.txt +2 -0
- otto_cli_agent-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,202 @@
|
|
|
1
|
+
"""Provider registry: Inception, OpenAI, Anthropic, Gemini, and any named
|
|
2
|
+
OpenAI-compatible endpoint (agent/router/llm_provider/custom.py).
|
|
3
|
+
Still a registry, not a hardcoded import: a second vendor showing up later is
|
|
4
|
+
a dict entry here, not a rewrite of every call site that reaches through
|
|
5
|
+
get_provider()/provider_names(). Importing this package stays cheap for the
|
|
6
|
+
same reason it always did -- lazy import, so a missing optional dependency
|
|
7
|
+
breaks one provider, not the whole app.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import importlib
|
|
13
|
+
import os
|
|
14
|
+
from functools import lru_cache
|
|
15
|
+
|
|
16
|
+
from langchain_core.language_models import BaseChatModel
|
|
17
|
+
|
|
18
|
+
from .base import (
|
|
19
|
+
AuthError,
|
|
20
|
+
BaseProvider,
|
|
21
|
+
Capability,
|
|
22
|
+
CapabilityNotSupported,
|
|
23
|
+
HealthReport,
|
|
24
|
+
ModelInfo,
|
|
25
|
+
ModelNotFound,
|
|
26
|
+
ProviderError,
|
|
27
|
+
ProviderStatus,
|
|
28
|
+
ProviderUnavailable,
|
|
29
|
+
SupportsEdit,
|
|
30
|
+
SupportsEmbeddings,
|
|
31
|
+
SupportsFIM,
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
__all__ = [
|
|
35
|
+
"AuthError",
|
|
36
|
+
"BaseProvider",
|
|
37
|
+
"Capability",
|
|
38
|
+
"CapabilityNotSupported",
|
|
39
|
+
"HealthReport",
|
|
40
|
+
"ModelInfo",
|
|
41
|
+
"ModelNotFound",
|
|
42
|
+
"ProviderError",
|
|
43
|
+
"ProviderStatus",
|
|
44
|
+
"ProviderUnavailable",
|
|
45
|
+
"SupportsEdit",
|
|
46
|
+
"SupportsEmbeddings",
|
|
47
|
+
"SupportsFIM",
|
|
48
|
+
"UnknownProvider",
|
|
49
|
+
"FALLBACK_MODEL_SPEC",
|
|
50
|
+
"default_model_spec",
|
|
51
|
+
"provider_names",
|
|
52
|
+
"builtin_provider_names",
|
|
53
|
+
"provider_class",
|
|
54
|
+
"register_custom",
|
|
55
|
+
"unregister_custom",
|
|
56
|
+
"is_custom",
|
|
57
|
+
"get_provider",
|
|
58
|
+
"parse_spec",
|
|
59
|
+
"get_chat_model",
|
|
60
|
+
"health_report",
|
|
61
|
+
"all_models",
|
|
62
|
+
"reset",
|
|
63
|
+
]
|
|
64
|
+
|
|
65
|
+
#: Used when OTTO_DEFAULT_MODEL is unset. Mercury 2.5 is Inception's current
|
|
66
|
+
#: chat model (2026-09-09; mercury-2 is still documented but no longer what
|
|
67
|
+
#: a fresh install should reach for).
|
|
68
|
+
FALLBACK_MODEL_SPEC = "inception:mercury-2.5"
|
|
69
|
+
|
|
70
|
+
_PROVIDER_MODULES: dict[str, tuple[str, str]] = {
|
|
71
|
+
"inception": (".inception_provider", "InceptionProvider"),
|
|
72
|
+
"anthropic": (".anthropic_provider", "AnthropicProvider"),
|
|
73
|
+
"openai": (".openai_provider", "OpenAIProvider"),
|
|
74
|
+
"gemini": (".gemini_provider", "GeminiProvider"),
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
class UnknownProvider(ProviderError):
|
|
79
|
+
"""Spec named a provider that is not registered."""
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def default_model_spec() -> str:
|
|
83
|
+
"""Resolve the default model, honouring OTTO_DEFAULT_MODEL.
|
|
84
|
+
|
|
85
|
+
A function, not a module constant: this package may well be imported before
|
|
86
|
+
the Typer callback calls `load_dotenv`, and a constant evaluated at import
|
|
87
|
+
time would freeze whatever the environment happened to hold then.
|
|
88
|
+
"""
|
|
89
|
+
return os.environ.get("OTTO_DEFAULT_MODEL") or FALLBACK_MODEL_SPEC
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
#: Providers registered at runtime -- the custom OpenAI-compatible endpoints
|
|
93
|
+
#: a person adds in the TUI's setup screen (agent/router/llm_provider/custom.py
|
|
94
|
+
#: builds the class, agent/router/overrides.py registers it from
|
|
95
|
+
#: ~/.otto/routes.json at startup). Kept apart from the static table so the
|
|
96
|
+
#: built-ins stay importable by name and a removed endpoint leaves no trace.
|
|
97
|
+
_CUSTOM: dict[str, type[BaseProvider]] = {}
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def builtin_provider_names() -> tuple[str, ...]:
|
|
101
|
+
return tuple(_PROVIDER_MODULES)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def provider_names() -> tuple[str, ...]:
|
|
105
|
+
return tuple(_PROVIDER_MODULES) + tuple(_CUSTOM)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def is_custom(name: str) -> bool:
|
|
109
|
+
return name in _CUSTOM
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def register_custom(name: str, cls: type[BaseProvider]) -> None:
|
|
113
|
+
"""Make `name` resolvable through provider_class()/get_provider().
|
|
114
|
+
Clears the caches, so a re-registered name is never served stale."""
|
|
115
|
+
if name in _PROVIDER_MODULES:
|
|
116
|
+
raise UnknownProvider(f"{name!r} is a built-in provider and cannot be replaced")
|
|
117
|
+
_CUSTOM[name] = cls
|
|
118
|
+
reset()
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def unregister_custom(name: str) -> None:
|
|
122
|
+
if _CUSTOM.pop(name, None) is not None:
|
|
123
|
+
reset()
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
@lru_cache(maxsize=None)
|
|
127
|
+
def provider_class(name: str) -> type[BaseProvider]:
|
|
128
|
+
"""Import and return a provider class by name."""
|
|
129
|
+
if name in _CUSTOM:
|
|
130
|
+
return _CUSTOM[name]
|
|
131
|
+
try:
|
|
132
|
+
module_path, attr = _PROVIDER_MODULES[name]
|
|
133
|
+
except KeyError:
|
|
134
|
+
raise UnknownProvider(
|
|
135
|
+
f"unknown provider {name!r}. Known: {', '.join(provider_names())}"
|
|
136
|
+
) from None
|
|
137
|
+
module = importlib.import_module(module_path, __package__)
|
|
138
|
+
return getattr(module, attr)
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
@lru_cache(maxsize=None)
|
|
142
|
+
def get_provider(name: str) -> BaseProvider:
|
|
143
|
+
"""Return a constructed, cached provider. Raises `AuthError` without a key."""
|
|
144
|
+
return provider_class(name)()
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def parse_spec(spec: str) -> tuple[str, str]:
|
|
148
|
+
"""Split ``"provider:model"``. A bare model name uses the default provider."""
|
|
149
|
+
provider, sep, model_id = spec.partition(":")
|
|
150
|
+
if not sep:
|
|
151
|
+
default_provider, _ = parse_spec(default_model_spec())
|
|
152
|
+
return default_provider, spec
|
|
153
|
+
if not model_id:
|
|
154
|
+
raise UnknownProvider(f"spec {spec!r} names a provider but no model")
|
|
155
|
+
return provider, model_id
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def get_chat_model(spec: str | None = None, **kwargs) -> BaseChatModel:
|
|
159
|
+
"""Resolve ``"provider:model"`` to a LangChain chat model.
|
|
160
|
+
|
|
161
|
+
Validates the id against the provider's catalogue first, so a typo fails
|
|
162
|
+
here with the list of alternatives rather than as a vendor 404 midway
|
|
163
|
+
through a graph run.
|
|
164
|
+
"""
|
|
165
|
+
provider_name, model_id = parse_spec(spec or default_model_spec())
|
|
166
|
+
provider = get_provider(provider_name)
|
|
167
|
+
provider.resolve_model(model_id)
|
|
168
|
+
return provider.chat_model(model_id, **kwargs)
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def health_report() -> list[HealthReport]:
|
|
172
|
+
"""Check every registered provider. Never raises -- this is the doctor command."""
|
|
173
|
+
reports: list[HealthReport] = []
|
|
174
|
+
for name in provider_names():
|
|
175
|
+
try:
|
|
176
|
+
reports.append(provider_class(name).check())
|
|
177
|
+
except Exception as exc: # import failure, e.g. SDK not installed
|
|
178
|
+
reports.append(
|
|
179
|
+
HealthReport(name, ProviderStatus.ERROR, detail=f"import failed: {exc}")
|
|
180
|
+
)
|
|
181
|
+
return reports
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def all_models(capability: Capability | None = None) -> list[ModelInfo]:
|
|
185
|
+
"""Every model across every configured provider, skipping unusable ones."""
|
|
186
|
+
models: list[ModelInfo] = []
|
|
187
|
+
for name in provider_names():
|
|
188
|
+
cls = provider_class(name)
|
|
189
|
+
if not cls.is_configured():
|
|
190
|
+
continue
|
|
191
|
+
try:
|
|
192
|
+
models.extend(get_provider(name).list_models(capability))
|
|
193
|
+
except ProviderError:
|
|
194
|
+
# One dead key must not blank out the rest.
|
|
195
|
+
continue
|
|
196
|
+
return models
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def reset() -> None:
|
|
200
|
+
"""Drop cached provider instances -- call after changing keys in-process."""
|
|
201
|
+
get_provider.cache_clear()
|
|
202
|
+
provider_class.cache_clear()
|
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
"""Anthropic provider.
|
|
2
|
+
|
|
3
|
+
Discovery goes through the `anthropic` SDK; chat goes through `ChatAnthropic`.
|
|
4
|
+
|
|
5
|
+
That split is deliberate and differs from `inception_provider.py`. Inception has
|
|
6
|
+
no LangChain integration, so a custom `BaseChatModel` was the only way to reach
|
|
7
|
+
it. Anthropic has a maintained one that already handles content blocks, tool
|
|
8
|
+
binding, structured output, prompt caching and extended thinking -- reproducing
|
|
9
|
+
that by hand over the raw SDK would be a few hundred lines of duplication with
|
|
10
|
+
no capability gained.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
from anthropic import (
|
|
18
|
+
Anthropic,
|
|
19
|
+
APIConnectionError,
|
|
20
|
+
APIStatusError,
|
|
21
|
+
AnthropicError,
|
|
22
|
+
AuthenticationError,
|
|
23
|
+
PermissionDeniedError,
|
|
24
|
+
)
|
|
25
|
+
from langchain_anthropic import ChatAnthropic
|
|
26
|
+
from langchain_core.language_models import BaseChatModel
|
|
27
|
+
|
|
28
|
+
from agent.router.llm_provider.base import (
|
|
29
|
+
AuthError,
|
|
30
|
+
BaseProvider,
|
|
31
|
+
Capability,
|
|
32
|
+
ModelInfo,
|
|
33
|
+
ProviderError,
|
|
34
|
+
ProviderUnavailable,
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
__all__ = ["AnthropicProvider"]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _translate(exc: Exception) -> ProviderError:
|
|
41
|
+
# AuthenticationError and PermissionDeniedError both subclass APIStatusError,
|
|
42
|
+
# so they must be tested before it.
|
|
43
|
+
if isinstance(exc, (AuthenticationError, PermissionDeniedError)):
|
|
44
|
+
return AuthError(f"anthropic: key rejected ({exc})")
|
|
45
|
+
if isinstance(exc, APIConnectionError): # APITimeoutError subclasses this
|
|
46
|
+
return ProviderUnavailable(f"anthropic: {exc}")
|
|
47
|
+
if isinstance(exc, APIStatusError):
|
|
48
|
+
if exc.status_code in (401, 403):
|
|
49
|
+
return AuthError(f"anthropic: HTTP {exc.status_code}")
|
|
50
|
+
return ProviderError(f"anthropic: HTTP {exc.status_code}")
|
|
51
|
+
return ProviderError(f"anthropic: {exc}")
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _capabilities_of(model: Any) -> set[Capability]:
|
|
55
|
+
"""Read capabilities off `ModelInfo.capabilities`.
|
|
56
|
+
|
|
57
|
+
Anthropic publishes structured capability flags, so nothing here is guessed
|
|
58
|
+
from the model name -- only TOOLS is assumed, because the Models API exposes
|
|
59
|
+
no tool-use flag and every current Claude model supports tool use.
|
|
60
|
+
"""
|
|
61
|
+
caps = {Capability.CHAT, Capability.TOOLS}
|
|
62
|
+
|
|
63
|
+
published = getattr(model, "capabilities", None)
|
|
64
|
+
if published is None:
|
|
65
|
+
# Older API versions omit the block entirely. Report the floor rather
|
|
66
|
+
# than inventing flags.
|
|
67
|
+
return caps
|
|
68
|
+
|
|
69
|
+
if getattr(published.image_input, "supported", False):
|
|
70
|
+
caps.add(Capability.VISION)
|
|
71
|
+
if getattr(published.structured_outputs, "supported", False):
|
|
72
|
+
caps.add(Capability.STRUCTURED_OUTPUT)
|
|
73
|
+
# Extended thinking and effort levels are two expressions of the same idea.
|
|
74
|
+
if getattr(published.thinking, "supported", False) or getattr(
|
|
75
|
+
published.effort, "supported", False
|
|
76
|
+
):
|
|
77
|
+
caps.add(Capability.REASONING)
|
|
78
|
+
|
|
79
|
+
return caps
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
class AnthropicProvider(BaseProvider):
|
|
83
|
+
name = "anthropic"
|
|
84
|
+
env_var = "ANTHROPIC_API_KEY"
|
|
85
|
+
default_base_url = None
|
|
86
|
+
capabilities = frozenset(
|
|
87
|
+
{
|
|
88
|
+
Capability.CHAT,
|
|
89
|
+
Capability.TOOLS,
|
|
90
|
+
Capability.VISION,
|
|
91
|
+
Capability.REASONING,
|
|
92
|
+
Capability.STRUCTURED_OUTPUT,
|
|
93
|
+
}
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
def _build_client(self) -> Anthropic:
|
|
97
|
+
kwargs: dict[str, Any] = {"api_key": self._api_key}
|
|
98
|
+
if self._base_url:
|
|
99
|
+
kwargs["base_url"] = self._base_url
|
|
100
|
+
return Anthropic(**kwargs)
|
|
101
|
+
|
|
102
|
+
def _fetch_models(self) -> list[ModelInfo]:
|
|
103
|
+
try:
|
|
104
|
+
# The Models API paginates and defaults to 20 per page. In this SDK
|
|
105
|
+
# SyncPage.__iter__ walks iter_pages() itself, so plain iteration
|
|
106
|
+
# already spans every page -- but the `limit` still matters: it sets
|
|
107
|
+
# the page size, and without it you pay a round trip per 20 models.
|
|
108
|
+
models = list(self._client.models.list(limit=1000))
|
|
109
|
+
except AnthropicError as exc:
|
|
110
|
+
raise _translate(exc) from exc
|
|
111
|
+
|
|
112
|
+
return [
|
|
113
|
+
ModelInfo(
|
|
114
|
+
id=model.id,
|
|
115
|
+
provider=self.name,
|
|
116
|
+
display_name=model.display_name,
|
|
117
|
+
capabilities=frozenset(_capabilities_of(model)),
|
|
118
|
+
context_window=model.max_input_tokens,
|
|
119
|
+
max_output_tokens=model.max_tokens,
|
|
120
|
+
raw=model.model_dump(mode="json"),
|
|
121
|
+
)
|
|
122
|
+
for model in models
|
|
123
|
+
]
|
|
124
|
+
|
|
125
|
+
def chat_model(self, model_id: str, **kwargs: Any) -> BaseChatModel:
|
|
126
|
+
if self._base_url:
|
|
127
|
+
kwargs.setdefault("base_url", self._base_url)
|
|
128
|
+
return ChatAnthropic(model=model_id, api_key=self._api_key, **kwargs)
|