otto-cli-agent 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (129) hide show
  1. agent/README.md +22 -0
  2. agent/__init__.py +0 -0
  3. agent/cli/README.md +77 -0
  4. agent/cli/__init__.py +0 -0
  5. agent/cli/art.py +371 -0
  6. agent/cli/chat.py +309 -0
  7. agent/cli/clipboard.py +106 -0
  8. agent/cli/context.py +56 -0
  9. agent/cli/doctor.py +60 -0
  10. agent/cli/errors.py +58 -0
  11. agent/cli/eval.py +130 -0
  12. agent/cli/eval_claw.py +414 -0
  13. agent/cli/eval_compaction.py +106 -0
  14. agent/cli/eval_hle.py +92 -0
  15. agent/cli/eval_memory.py +253 -0
  16. agent/cli/eval_swe.py +172 -0
  17. agent/cli/lessons.py +97 -0
  18. agent/cli/main.py +67 -0
  19. agent/cli/modals.py +570 -0
  20. agent/cli/models.py +64 -0
  21. agent/cli/output.py +54 -0
  22. agent/cli/route.py +78 -0
  23. agent/cli/sessions.py +125 -0
  24. agent/cli/setup_screen.py +562 -0
  25. agent/cli/shell.py +548 -0
  26. agent/cli/tui.py +1807 -0
  27. agent/cli/ui.py +14 -0
  28. agent/cli/usage_panel.py +159 -0
  29. agent/config/README.md +7 -0
  30. agent/config/__init__.py +0 -0
  31. agent/config/envfile.py +76 -0
  32. agent/eval/README.md +76 -0
  33. agent/eval/__init__.py +0 -0
  34. agent/eval/claw_bench.py +1031 -0
  35. agent/eval/compaction_bench.py +229 -0
  36. agent/eval/data/README.md +10 -0
  37. agent/eval/data/claw/README.md +108 -0
  38. agent/eval/data/claw/llm_judge-gemini.patch +57 -0
  39. agent/eval/data/claw/otto.yaml +35 -0
  40. agent/eval/failures.py +276 -0
  41. agent/eval/golden/README.md +33 -0
  42. agent/eval/golden/code_01.json +6 -0
  43. agent/eval/golden/code_02.json +6 -0
  44. agent/eval/golden/code_03.json +6 -0
  45. agent/eval/golden/code_04.json +6 -0
  46. agent/eval/golden/code_05.json +6 -0
  47. agent/eval/golden/code_06.json +6 -0
  48. agent/eval/golden/math_01.json +6 -0
  49. agent/eval/golden/math_02.json +6 -0
  50. agent/eval/golden/math_03.json +6 -0
  51. agent/eval/golden/math_04.json +6 -0
  52. agent/eval/golden/math_05.json +6 -0
  53. agent/eval/golden/math_06.json +6 -0
  54. agent/eval/golden/nphard_gcp_01.json +6 -0
  55. agent/eval/golden/nphard_ksp_01.json +6 -0
  56. agent/eval/golden/nphard_math_binpacking_01.json +6 -0
  57. agent/eval/golden/nphard_math_clique_01.json +6 -0
  58. agent/eval/golden/nphard_math_setcover_01.json +6 -0
  59. agent/eval/golden/nphard_math_subsetsum_01.json +6 -0
  60. agent/eval/golden/nphard_tsp_01.json +6 -0
  61. agent/eval/golden/nphard_tsp_02.json +6 -0
  62. agent/eval/hle_bench.py +273 -0
  63. agent/eval/langfuse_sync.py +172 -0
  64. agent/eval/memory_bench.py +538 -0
  65. agent/eval/runner.py +174 -0
  66. agent/eval/single_agent.py +120 -0
  67. agent/eval/swe_bench.py +604 -0
  68. agent/eval/terminal_bench.py +345 -0
  69. agent/memory/README.md +102 -0
  70. agent/memory/__init__.py +42 -0
  71. agent/memory/embeddings.py +302 -0
  72. agent/memory/hashing.py +15 -0
  73. agent/memory/lessons.py +483 -0
  74. agent/memory/queue.py +531 -0
  75. agent/memory/retrieval.py +493 -0
  76. agent/memory/session.py +60 -0
  77. agent/memory/sessions.py +436 -0
  78. agent/memory/store.py +429 -0
  79. agent/memory/tokens.py +60 -0
  80. agent/memory/wiring.py +146 -0
  81. agent/pipeline/README.md +135 -0
  82. agent/pipeline/__init__.py +0 -0
  83. agent/pipeline/browsing.py +609 -0
  84. agent/pipeline/budget.py +403 -0
  85. agent/pipeline/codemap.py +254 -0
  86. agent/pipeline/evidence.py +325 -0
  87. agent/pipeline/execution.py +67 -0
  88. agent/pipeline/modes.py +137 -0
  89. agent/pipeline/native.py +1137 -0
  90. agent/pipeline/nodes.py +3644 -0
  91. agent/pipeline/pricing.py +209 -0
  92. agent/pipeline/progress.py +139 -0
  93. agent/pipeline/rag.py +139 -0
  94. agent/pipeline/research.py +1325 -0
  95. agent/pipeline/run.py +528 -0
  96. agent/pipeline/screen.py +77 -0
  97. agent/pipeline/state.py +220 -0
  98. agent/pipeline/toolkit.py +328 -0
  99. agent/pipeline/tools.py +1990 -0
  100. agent/pipeline/tracing.py +147 -0
  101. agent/pipeline/usage.py +251 -0
  102. agent/pipeline/vision.py +84 -0
  103. agent/pipeline/walkthrough.py +735 -0
  104. agent/pipeline/workspace.py +229 -0
  105. agent/router/README.md +60 -0
  106. agent/router/__init__.py +0 -0
  107. agent/router/automap.py +114 -0
  108. agent/router/health.py +229 -0
  109. agent/router/llm_provider/README.md +38 -0
  110. agent/router/llm_provider/__init__.py +202 -0
  111. agent/router/llm_provider/anthropic_provider.py +128 -0
  112. agent/router/llm_provider/base.py +507 -0
  113. agent/router/llm_provider/custom.py +152 -0
  114. agent/router/llm_provider/gemini_provider.py +122 -0
  115. agent/router/llm_provider/inception_provider.py +687 -0
  116. agent/router/llm_provider/openai_provider.py +151 -0
  117. agent/router/llm_provider/retired.py +145 -0
  118. agent/router/llm_provider/temperature.py +371 -0
  119. agent/router/mapping.py +579 -0
  120. agent/router/outcomes.py +363 -0
  121. agent/router/overrides.py +389 -0
  122. agent/router/reload.py +28 -0
  123. agent/router/router.py +413 -0
  124. agent/router/setup.py +123 -0
  125. otto_cli_agent-0.1.0.dist-info/METADATA +115 -0
  126. otto_cli_agent-0.1.0.dist-info/RECORD +129 -0
  127. otto_cli_agent-0.1.0.dist-info/WHEEL +4 -0
  128. otto_cli_agent-0.1.0.dist-info/entry_points.txt +2 -0
  129. otto_cli_agent-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,202 @@
1
+ """Provider registry: Inception, OpenAI, Anthropic, Gemini, and any named
2
+ OpenAI-compatible endpoint (agent/router/llm_provider/custom.py).
3
+ Still a registry, not a hardcoded import: a second vendor showing up later is
4
+ a dict entry here, not a rewrite of every call site that reaches through
5
+ get_provider()/provider_names(). Importing this package stays cheap for the
6
+ same reason it always did -- lazy import, so a missing optional dependency
7
+ breaks one provider, not the whole app.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import importlib
13
+ import os
14
+ from functools import lru_cache
15
+
16
+ from langchain_core.language_models import BaseChatModel
17
+
18
+ from .base import (
19
+ AuthError,
20
+ BaseProvider,
21
+ Capability,
22
+ CapabilityNotSupported,
23
+ HealthReport,
24
+ ModelInfo,
25
+ ModelNotFound,
26
+ ProviderError,
27
+ ProviderStatus,
28
+ ProviderUnavailable,
29
+ SupportsEdit,
30
+ SupportsEmbeddings,
31
+ SupportsFIM,
32
+ )
33
+
34
+ __all__ = [
35
+ "AuthError",
36
+ "BaseProvider",
37
+ "Capability",
38
+ "CapabilityNotSupported",
39
+ "HealthReport",
40
+ "ModelInfo",
41
+ "ModelNotFound",
42
+ "ProviderError",
43
+ "ProviderStatus",
44
+ "ProviderUnavailable",
45
+ "SupportsEdit",
46
+ "SupportsEmbeddings",
47
+ "SupportsFIM",
48
+ "UnknownProvider",
49
+ "FALLBACK_MODEL_SPEC",
50
+ "default_model_spec",
51
+ "provider_names",
52
+ "builtin_provider_names",
53
+ "provider_class",
54
+ "register_custom",
55
+ "unregister_custom",
56
+ "is_custom",
57
+ "get_provider",
58
+ "parse_spec",
59
+ "get_chat_model",
60
+ "health_report",
61
+ "all_models",
62
+ "reset",
63
+ ]
64
+
65
+ #: Used when OTTO_DEFAULT_MODEL is unset. Mercury 2.5 is Inception's current
66
+ #: chat model (2026-09-09; mercury-2 is still documented but no longer what
67
+ #: a fresh install should reach for).
68
+ FALLBACK_MODEL_SPEC = "inception:mercury-2.5"
69
+
70
+ _PROVIDER_MODULES: dict[str, tuple[str, str]] = {
71
+ "inception": (".inception_provider", "InceptionProvider"),
72
+ "anthropic": (".anthropic_provider", "AnthropicProvider"),
73
+ "openai": (".openai_provider", "OpenAIProvider"),
74
+ "gemini": (".gemini_provider", "GeminiProvider"),
75
+ }
76
+
77
+
78
+ class UnknownProvider(ProviderError):
79
+ """Spec named a provider that is not registered."""
80
+
81
+
82
+ def default_model_spec() -> str:
83
+ """Resolve the default model, honouring OTTO_DEFAULT_MODEL.
84
+
85
+ A function, not a module constant: this package may well be imported before
86
+ the Typer callback calls `load_dotenv`, and a constant evaluated at import
87
+ time would freeze whatever the environment happened to hold then.
88
+ """
89
+ return os.environ.get("OTTO_DEFAULT_MODEL") or FALLBACK_MODEL_SPEC
90
+
91
+
92
+ #: Providers registered at runtime -- the custom OpenAI-compatible endpoints
93
+ #: a person adds in the TUI's setup screen (agent/router/llm_provider/custom.py
94
+ #: builds the class, agent/router/overrides.py registers it from
95
+ #: ~/.otto/routes.json at startup). Kept apart from the static table so the
96
+ #: built-ins stay importable by name and a removed endpoint leaves no trace.
97
+ _CUSTOM: dict[str, type[BaseProvider]] = {}
98
+
99
+
100
+ def builtin_provider_names() -> tuple[str, ...]:
101
+ return tuple(_PROVIDER_MODULES)
102
+
103
+
104
+ def provider_names() -> tuple[str, ...]:
105
+ return tuple(_PROVIDER_MODULES) + tuple(_CUSTOM)
106
+
107
+
108
+ def is_custom(name: str) -> bool:
109
+ return name in _CUSTOM
110
+
111
+
112
+ def register_custom(name: str, cls: type[BaseProvider]) -> None:
113
+ """Make `name` resolvable through provider_class()/get_provider().
114
+ Clears the caches, so a re-registered name is never served stale."""
115
+ if name in _PROVIDER_MODULES:
116
+ raise UnknownProvider(f"{name!r} is a built-in provider and cannot be replaced")
117
+ _CUSTOM[name] = cls
118
+ reset()
119
+
120
+
121
+ def unregister_custom(name: str) -> None:
122
+ if _CUSTOM.pop(name, None) is not None:
123
+ reset()
124
+
125
+
126
+ @lru_cache(maxsize=None)
127
+ def provider_class(name: str) -> type[BaseProvider]:
128
+ """Import and return a provider class by name."""
129
+ if name in _CUSTOM:
130
+ return _CUSTOM[name]
131
+ try:
132
+ module_path, attr = _PROVIDER_MODULES[name]
133
+ except KeyError:
134
+ raise UnknownProvider(
135
+ f"unknown provider {name!r}. Known: {', '.join(provider_names())}"
136
+ ) from None
137
+ module = importlib.import_module(module_path, __package__)
138
+ return getattr(module, attr)
139
+
140
+
141
+ @lru_cache(maxsize=None)
142
+ def get_provider(name: str) -> BaseProvider:
143
+ """Return a constructed, cached provider. Raises `AuthError` without a key."""
144
+ return provider_class(name)()
145
+
146
+
147
+ def parse_spec(spec: str) -> tuple[str, str]:
148
+ """Split ``"provider:model"``. A bare model name uses the default provider."""
149
+ provider, sep, model_id = spec.partition(":")
150
+ if not sep:
151
+ default_provider, _ = parse_spec(default_model_spec())
152
+ return default_provider, spec
153
+ if not model_id:
154
+ raise UnknownProvider(f"spec {spec!r} names a provider but no model")
155
+ return provider, model_id
156
+
157
+
158
+ def get_chat_model(spec: str | None = None, **kwargs) -> BaseChatModel:
159
+ """Resolve ``"provider:model"`` to a LangChain chat model.
160
+
161
+ Validates the id against the provider's catalogue first, so a typo fails
162
+ here with the list of alternatives rather than as a vendor 404 midway
163
+ through a graph run.
164
+ """
165
+ provider_name, model_id = parse_spec(spec or default_model_spec())
166
+ provider = get_provider(provider_name)
167
+ provider.resolve_model(model_id)
168
+ return provider.chat_model(model_id, **kwargs)
169
+
170
+
171
+ def health_report() -> list[HealthReport]:
172
+ """Check every registered provider. Never raises -- this is the doctor command."""
173
+ reports: list[HealthReport] = []
174
+ for name in provider_names():
175
+ try:
176
+ reports.append(provider_class(name).check())
177
+ except Exception as exc: # import failure, e.g. SDK not installed
178
+ reports.append(
179
+ HealthReport(name, ProviderStatus.ERROR, detail=f"import failed: {exc}")
180
+ )
181
+ return reports
182
+
183
+
184
+ def all_models(capability: Capability | None = None) -> list[ModelInfo]:
185
+ """Every model across every configured provider, skipping unusable ones."""
186
+ models: list[ModelInfo] = []
187
+ for name in provider_names():
188
+ cls = provider_class(name)
189
+ if not cls.is_configured():
190
+ continue
191
+ try:
192
+ models.extend(get_provider(name).list_models(capability))
193
+ except ProviderError:
194
+ # One dead key must not blank out the rest.
195
+ continue
196
+ return models
197
+
198
+
199
+ def reset() -> None:
200
+ """Drop cached provider instances -- call after changing keys in-process."""
201
+ get_provider.cache_clear()
202
+ provider_class.cache_clear()
@@ -0,0 +1,128 @@
1
+ """Anthropic provider.
2
+
3
+ Discovery goes through the `anthropic` SDK; chat goes through `ChatAnthropic`.
4
+
5
+ That split is deliberate and differs from `inception_provider.py`. Inception has
6
+ no LangChain integration, so a custom `BaseChatModel` was the only way to reach
7
+ it. Anthropic has a maintained one that already handles content blocks, tool
8
+ binding, structured output, prompt caching and extended thinking -- reproducing
9
+ that by hand over the raw SDK would be a few hundred lines of duplication with
10
+ no capability gained.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ from typing import Any
16
+
17
+ from anthropic import (
18
+ Anthropic,
19
+ APIConnectionError,
20
+ APIStatusError,
21
+ AnthropicError,
22
+ AuthenticationError,
23
+ PermissionDeniedError,
24
+ )
25
+ from langchain_anthropic import ChatAnthropic
26
+ from langchain_core.language_models import BaseChatModel
27
+
28
+ from agent.router.llm_provider.base import (
29
+ AuthError,
30
+ BaseProvider,
31
+ Capability,
32
+ ModelInfo,
33
+ ProviderError,
34
+ ProviderUnavailable,
35
+ )
36
+
37
+ __all__ = ["AnthropicProvider"]
38
+
39
+
40
+ def _translate(exc: Exception) -> ProviderError:
41
+ # AuthenticationError and PermissionDeniedError both subclass APIStatusError,
42
+ # so they must be tested before it.
43
+ if isinstance(exc, (AuthenticationError, PermissionDeniedError)):
44
+ return AuthError(f"anthropic: key rejected ({exc})")
45
+ if isinstance(exc, APIConnectionError): # APITimeoutError subclasses this
46
+ return ProviderUnavailable(f"anthropic: {exc}")
47
+ if isinstance(exc, APIStatusError):
48
+ if exc.status_code in (401, 403):
49
+ return AuthError(f"anthropic: HTTP {exc.status_code}")
50
+ return ProviderError(f"anthropic: HTTP {exc.status_code}")
51
+ return ProviderError(f"anthropic: {exc}")
52
+
53
+
54
+ def _capabilities_of(model: Any) -> set[Capability]:
55
+ """Read capabilities off `ModelInfo.capabilities`.
56
+
57
+ Anthropic publishes structured capability flags, so nothing here is guessed
58
+ from the model name -- only TOOLS is assumed, because the Models API exposes
59
+ no tool-use flag and every current Claude model supports tool use.
60
+ """
61
+ caps = {Capability.CHAT, Capability.TOOLS}
62
+
63
+ published = getattr(model, "capabilities", None)
64
+ if published is None:
65
+ # Older API versions omit the block entirely. Report the floor rather
66
+ # than inventing flags.
67
+ return caps
68
+
69
+ if getattr(published.image_input, "supported", False):
70
+ caps.add(Capability.VISION)
71
+ if getattr(published.structured_outputs, "supported", False):
72
+ caps.add(Capability.STRUCTURED_OUTPUT)
73
+ # Extended thinking and effort levels are two expressions of the same idea.
74
+ if getattr(published.thinking, "supported", False) or getattr(
75
+ published.effort, "supported", False
76
+ ):
77
+ caps.add(Capability.REASONING)
78
+
79
+ return caps
80
+
81
+
82
+ class AnthropicProvider(BaseProvider):
83
+ name = "anthropic"
84
+ env_var = "ANTHROPIC_API_KEY"
85
+ default_base_url = None
86
+ capabilities = frozenset(
87
+ {
88
+ Capability.CHAT,
89
+ Capability.TOOLS,
90
+ Capability.VISION,
91
+ Capability.REASONING,
92
+ Capability.STRUCTURED_OUTPUT,
93
+ }
94
+ )
95
+
96
+ def _build_client(self) -> Anthropic:
97
+ kwargs: dict[str, Any] = {"api_key": self._api_key}
98
+ if self._base_url:
99
+ kwargs["base_url"] = self._base_url
100
+ return Anthropic(**kwargs)
101
+
102
+ def _fetch_models(self) -> list[ModelInfo]:
103
+ try:
104
+ # The Models API paginates and defaults to 20 per page. In this SDK
105
+ # SyncPage.__iter__ walks iter_pages() itself, so plain iteration
106
+ # already spans every page -- but the `limit` still matters: it sets
107
+ # the page size, and without it you pay a round trip per 20 models.
108
+ models = list(self._client.models.list(limit=1000))
109
+ except AnthropicError as exc:
110
+ raise _translate(exc) from exc
111
+
112
+ return [
113
+ ModelInfo(
114
+ id=model.id,
115
+ provider=self.name,
116
+ display_name=model.display_name,
117
+ capabilities=frozenset(_capabilities_of(model)),
118
+ context_window=model.max_input_tokens,
119
+ max_output_tokens=model.max_tokens,
120
+ raw=model.model_dump(mode="json"),
121
+ )
122
+ for model in models
123
+ ]
124
+
125
+ def chat_model(self, model_id: str, **kwargs: Any) -> BaseChatModel:
126
+ if self._base_url:
127
+ kwargs.setdefault("base_url", self._base_url)
128
+ return ChatAnthropic(model=model_id, api_key=self._api_key, **kwargs)