cat-stack 2.5.1__tar.gz → 2.5.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cat_stack-2.5.1 → cat_stack-2.5.3}/PKG-INFO +7 -5
- {cat_stack-2.5.1 → cat_stack-2.5.3}/README.md +3 -1
- {cat_stack-2.5.1 → cat_stack-2.5.3}/pyproject.toml +2 -2
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/__about__.py +1 -1
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/_providers.py +74 -4
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/image_functions.py +14 -13
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/pdf_functions.py +17 -15
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/text_functions_ensemble.py +28 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/.gitignore +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/LICENSE +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/cat_stack/__init__.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/__init__.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/_batch.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/_category_analysis.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/_chunked.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/_embeddings.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/_formatter.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/_pilot_test.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/_prompts.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/_review_ui.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/_tiebreaker.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/_utils.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/_web_fetch.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/_wrapper_helpers.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/calls/CoVe.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/calls/__init__.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/calls/image_CoVe.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/calls/image_stepback.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/calls/pdf_CoVe.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/calls/pdf_stepback.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/calls/stepback.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/calls/top_n.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/classify.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/collapse_themes.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/explore.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/extract.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/images/circle.png +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/images/cube.png +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/images/diamond.png +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/images/overlapping_pentagons.png +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/images/rectangles.png +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/model_reference_list.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/prompt_tune.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/summarize.py +0 -0
- {cat_stack-2.5.1 → cat_stack-2.5.3}/src/catstack/text_functions.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: cat-stack
|
|
3
|
-
Version: 2.5.
|
|
3
|
+
Version: 2.5.3
|
|
4
4
|
Summary: Domain-agnostic text, image, PDF, and DOCX classification engine powered by LLMs
|
|
5
5
|
Project-URL: Documentation, https://github.com/chrissoria/cat-stack#readme
|
|
6
6
|
Project-URL: Issues, https://github.com/chrissoria/cat-stack/issues
|
|
@@ -24,9 +24,9 @@ Requires-Dist: pandas
|
|
|
24
24
|
Requires-Dist: requests
|
|
25
25
|
Requires-Dist: tqdm
|
|
26
26
|
Provides-Extra: agent
|
|
27
|
-
Requires-Dist: cat-claws[claude]>=0.3.
|
|
27
|
+
Requires-Dist: cat-claws[claude]>=0.3.3; extra == 'agent'
|
|
28
28
|
Provides-Extra: codex-agent
|
|
29
|
-
Requires-Dist: cat-claws[codex]>=0.3.
|
|
29
|
+
Requires-Dist: cat-claws[codex]>=0.3.3; extra == 'codex-agent'
|
|
30
30
|
Provides-Extra: docx
|
|
31
31
|
Requires-Dist: python-docx>=1.0.0; extra == 'docx'
|
|
32
32
|
Provides-Extra: embeddings
|
|
@@ -239,7 +239,9 @@ OpenAI, Anthropic, Google (Gemini), Mistral, Perplexity, xAI (Grok), HuggingFace
|
|
|
239
239
|
|
|
240
240
|
All providers use the same `(model_name, provider, api_key)` tuple format. Provider is auto-detected from model name if omitted.
|
|
241
241
|
|
|
242
|
-
**Subscription backends (no API key).** Three `model_source` values authenticate through a chat subscription instead of a metered key — leave `api_key` unset: `"claude-agent"` (Claude subscription via the Agent SDK; `pip install "cat-stack[agent]"`), `"claude-code"` (the Claude Code CLI, if installed — no extra needed), and `"codex-agent"` (ChatGPT subscription; `pip install "cat-stack[codex-agent]"`). Classification, extraction, exploration, and summarization all route through them.
|
|
242
|
+
**Subscription backends (no API key).** Three `model_source` values authenticate through a chat subscription instead of a metered key — leave `api_key` unset: `"claude-agent"` (Claude subscription via the Agent SDK; `pip install "cat-stack[agent]"`), `"claude-code"` (the Claude Code CLI, if installed — no extra needed), and `"codex-agent"` (ChatGPT subscription; `pip install "cat-stack[codex-agent]"`). Classification, extraction, exploration, and summarization all route through them, including images and PDF pages on the two agent backends. They use the agent CLI's own sign-in (separate from the Claude desktop app's): runs check it up front and, if you're signed out, open a single browser window to sign in once, then continue. See the [cat-claws README](https://github.com/chrissoria/cat-agent#signing-in-once).
|
|
243
|
+
|
|
244
|
+
**Reproducibility note:** the same nominal model can behave differently across access routes. In a seed-matched benchmark (Claude Sonnet 5, temperature 0), extraction through the Agent SDK produced more varied label phrasings than the direct API (89% vs. 74% unique raw labels), while final consolidated taxonomies were equivalent. If raw label counts matter to your analysis, record the access route alongside the model version.
|
|
243
245
|
|
|
244
246
|
## Features
|
|
245
247
|
|
|
@@ -198,7 +198,9 @@ OpenAI, Anthropic, Google (Gemini), Mistral, Perplexity, xAI (Grok), HuggingFace
|
|
|
198
198
|
|
|
199
199
|
All providers use the same `(model_name, provider, api_key)` tuple format. Provider is auto-detected from model name if omitted.
|
|
200
200
|
|
|
201
|
-
**Subscription backends (no API key).** Three `model_source` values authenticate through a chat subscription instead of a metered key — leave `api_key` unset: `"claude-agent"` (Claude subscription via the Agent SDK; `pip install "cat-stack[agent]"`), `"claude-code"` (the Claude Code CLI, if installed — no extra needed), and `"codex-agent"` (ChatGPT subscription; `pip install "cat-stack[codex-agent]"`). Classification, extraction, exploration, and summarization all route through them.
|
|
201
|
+
**Subscription backends (no API key).** Three `model_source` values authenticate through a chat subscription instead of a metered key — leave `api_key` unset: `"claude-agent"` (Claude subscription via the Agent SDK; `pip install "cat-stack[agent]"`), `"claude-code"` (the Claude Code CLI, if installed — no extra needed), and `"codex-agent"` (ChatGPT subscription; `pip install "cat-stack[codex-agent]"`). Classification, extraction, exploration, and summarization all route through them, including images and PDF pages on the two agent backends. They use the agent CLI's own sign-in (separate from the Claude desktop app's): runs check it up front and, if you're signed out, open a single browser window to sign in once, then continue. See the [cat-claws README](https://github.com/chrissoria/cat-agent#signing-in-once).
|
|
202
|
+
|
|
203
|
+
**Reproducibility note:** the same nominal model can behave differently across access routes. In a seed-matched benchmark (Claude Sonnet 5, temperature 0), extraction through the Agent SDK produced more varied label phrasings than the direct API (89% vs. 74% unique raw labels), while final consolidated taxonomies were equivalent. If raw label counts matter to your analysis, record the access route alongside the model version.
|
|
202
204
|
|
|
203
205
|
## Features
|
|
204
206
|
|
|
@@ -42,8 +42,8 @@ embeddings = ["sentence-transformers>=2.2.0"]
|
|
|
42
42
|
# `agent` keeps its historical meaning (the Claude backend) so every shipped
|
|
43
43
|
# install hint stays true; `codex-agent` matches the provider string so the
|
|
44
44
|
# error-message hint is copy-pasteable.
|
|
45
|
-
agent = ["cat-claws[claude]>=0.3.
|
|
46
|
-
codex-agent = ["cat-claws[codex]>=0.3.
|
|
45
|
+
agent = ["cat-claws[claude]>=0.3.3"]
|
|
46
|
+
codex-agent = ["cat-claws[codex]>=0.3.3"]
|
|
47
47
|
|
|
48
48
|
[project.urls]
|
|
49
49
|
Documentation = "https://github.com/chrissoria/cat-stack#readme"
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# SPDX-FileCopyrightText: 2025-present Christopher Soria <chrissoria@berkeley.edu>
|
|
2
2
|
#
|
|
3
3
|
# SPDX-License-Identifier: GPL-3.0-or-later
|
|
4
|
-
__version__ = "2.5.
|
|
4
|
+
__version__ = "2.5.3"
|
|
5
5
|
__author__ = "Chris Soria"
|
|
6
6
|
__email__ = "chrissoria@berkeley.edu"
|
|
7
7
|
__title__ = "cat-stack"
|
|
@@ -735,8 +735,10 @@ PROVIDER_CONFIG = {
|
|
|
735
735
|
|
|
736
736
|
|
|
737
737
|
# Providers that route through complete() with no HTTP endpoint of their own
|
|
738
|
-
# (subscription logins / CLI). Features that build a direct HTTP request
|
|
739
|
-
#
|
|
738
|
+
# (subscription logins / CLI). Features that build a direct HTTP request can't
|
|
739
|
+
# use them — guard with a clear error, not a deep crash. (Images and PDF pages
|
|
740
|
+
# DO work on the cat-claws agent backends: they go through the adapters'
|
|
741
|
+
# images= argument, see _split_agent_content / _call_agent_image.)
|
|
740
742
|
_SUBSCRIPTION_PROVIDERS = ("claude-code", "claude-agent", "codex-agent")
|
|
741
743
|
|
|
742
744
|
# Agent-SDK backends routed through cat-claws: provider -> (adapter name,
|
|
@@ -748,6 +750,64 @@ _AGENT_BACKENDS = {
|
|
|
748
750
|
}
|
|
749
751
|
|
|
750
752
|
|
|
753
|
+
def _split_agent_content(content):
|
|
754
|
+
"""Split one message's content into (text, images) for the cat-claws
|
|
755
|
+
adapters.
|
|
756
|
+
|
|
757
|
+
Plain-string content is text only. Multimodal content is a list of blocks
|
|
758
|
+
in whichever provider shape the prompt builder produced (Anthropic
|
|
759
|
+
`image` + base64 `source`, OpenAI `image_url` data URL, Google
|
|
760
|
+
`inline_data`); text blocks are joined and image blocks become the
|
|
761
|
+
adapters' ``{"media_type", "data"}`` dicts. Unknown block types are
|
|
762
|
+
dropped rather than stringified into the prompt.
|
|
763
|
+
"""
|
|
764
|
+
if not isinstance(content, list):
|
|
765
|
+
return content, []
|
|
766
|
+
texts, images = [], []
|
|
767
|
+
for block in content:
|
|
768
|
+
if not isinstance(block, dict):
|
|
769
|
+
continue
|
|
770
|
+
kind = block.get("type")
|
|
771
|
+
if kind == "text":
|
|
772
|
+
texts.append(block.get("text", ""))
|
|
773
|
+
elif kind == "image":
|
|
774
|
+
src = block.get("source") or {}
|
|
775
|
+
if src.get("type") == "base64" and src.get("data"):
|
|
776
|
+
images.append({"media_type": src.get("media_type") or "image/png",
|
|
777
|
+
"data": src["data"]})
|
|
778
|
+
elif kind == "image_url":
|
|
779
|
+
url = (block.get("image_url") or {}).get("url", "")
|
|
780
|
+
if url.startswith("data:") and ";base64," in url:
|
|
781
|
+
header, data = url.split(",", 1)
|
|
782
|
+
images.append({"media_type": header[5:].split(";", 1)[0] or "image/png",
|
|
783
|
+
"data": data})
|
|
784
|
+
elif kind == "inline_data" and block.get("data"):
|
|
785
|
+
images.append({"media_type": block.get("mime_type") or "image/png",
|
|
786
|
+
"data": block["data"]})
|
|
787
|
+
for im in images: # the API wants image/jpeg, not image/jpg
|
|
788
|
+
if im["media_type"] == "image/jpg":
|
|
789
|
+
im["media_type"] = "image/jpeg"
|
|
790
|
+
return "\n\n".join(t for t in texts if t), images
|
|
791
|
+
|
|
792
|
+
|
|
793
|
+
def _require_agent_sign_in(provider):
|
|
794
|
+
"""Preflight for the cat-claws subscription backends: stop before any row
|
|
795
|
+
runs when the agent CLI is signed out, instead of every row failing on
|
|
796
|
+
"not logged in". cat-claws >= 0.3.3 checks with a cheap status call (no
|
|
797
|
+
model call); if signed out it opens ONE browser sign-in automatically and
|
|
798
|
+
continues, or, where no browser can open, raises NotSignedInError with
|
|
799
|
+
instructions worded for the Claude app or a terminal. No-op for other
|
|
800
|
+
providers, and when cat-claws is missing or older.
|
|
801
|
+
"""
|
|
802
|
+
if provider not in _AGENT_BACKENDS:
|
|
803
|
+
return
|
|
804
|
+
try:
|
|
805
|
+
from catclaws import ensure_signed_in
|
|
806
|
+
except ImportError:
|
|
807
|
+
return
|
|
808
|
+
ensure_signed_in(_AGENT_BACKENDS[provider][0])
|
|
809
|
+
|
|
810
|
+
|
|
751
811
|
def _require_http_provider(model_source, feature):
|
|
752
812
|
"""Raise a clear error when an HTTP-only feature is used with a
|
|
753
813
|
subscription/CLI provider (claude-code / claude-agent / codex-agent)."""
|
|
@@ -1272,17 +1332,26 @@ class UnifiedLLMClient:
|
|
|
1272
1332
|
)
|
|
1273
1333
|
import asyncio
|
|
1274
1334
|
|
|
1335
|
+
# Multimodal content (image summaries, rendered PDF pages) arrives as
|
|
1336
|
+
# a list of blocks: text goes to the prompt, images to the adapter's
|
|
1337
|
+
# `images=` argument (base.AgentAdapter.one_shot contract).
|
|
1275
1338
|
system_parts = []
|
|
1276
1339
|
user_parts = []
|
|
1340
|
+
images = []
|
|
1277
1341
|
for msg in messages:
|
|
1342
|
+
text, msg_images = _split_agent_content(msg["content"])
|
|
1278
1343
|
if msg["role"] == "system":
|
|
1279
|
-
system_parts.append(
|
|
1344
|
+
system_parts.append(text)
|
|
1280
1345
|
elif msg["role"] in ("user", "assistant"):
|
|
1281
|
-
user_parts.append(
|
|
1346
|
+
user_parts.append(text)
|
|
1347
|
+
images.extend(msg_images)
|
|
1282
1348
|
system_prompt = "\n\n".join(system_parts) if system_parts else None
|
|
1283
1349
|
user_prompt = "\n\n".join(user_parts)
|
|
1284
1350
|
|
|
1285
1351
|
adapter = get_adapter(adapter_name)
|
|
1352
|
+
# Only pass images when there are some, so text-only calls are
|
|
1353
|
+
# byte-identical to before.
|
|
1354
|
+
image_kwargs = {"images": images} if images else {}
|
|
1286
1355
|
try:
|
|
1287
1356
|
return asyncio.run(
|
|
1288
1357
|
adapter.one_shot(
|
|
@@ -1290,6 +1359,7 @@ class UnifiedLLMClient:
|
|
|
1290
1359
|
system_prompt=system_prompt,
|
|
1291
1360
|
model=self.model,
|
|
1292
1361
|
thinking_budget=thinking_budget or 0,
|
|
1362
|
+
**image_kwargs,
|
|
1293
1363
|
)
|
|
1294
1364
|
)
|
|
1295
1365
|
except Exception as e:
|
|
@@ -154,12 +154,8 @@ def image_multi_class(
|
|
|
154
154
|
"(the text-only CLI shim). Use model_source='claude-agent' (the cat-claws "
|
|
155
155
|
"subscription backend) or an API-key provider."
|
|
156
156
|
)
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
"Image classification is not yet supported with "
|
|
160
|
-
"model_source='codex-agent'. Use model_source='claude-agent' (the "
|
|
161
|
-
"multimodal subscription backend) or an API-key provider."
|
|
162
|
-
)
|
|
157
|
+
from ._providers import _require_agent_sign_in
|
|
158
|
+
_require_agent_sign_in(model_source)
|
|
163
159
|
|
|
164
160
|
image_files = _load_image_files(image_input)
|
|
165
161
|
|
|
@@ -677,15 +673,18 @@ Provide the final categorization in the same JSON format:"""
|
|
|
677
673
|
|
|
678
674
|
return """{"1":"e"}""", "Max retries exceeded"
|
|
679
675
|
|
|
680
|
-
def
|
|
681
|
-
"""Image classification via
|
|
682
|
-
subscription, no API key).
|
|
676
|
+
def _call_agent_image(base_text, encoded, media_type):
|
|
677
|
+
"""Image classification via a cat-claws multimodal adapter
|
|
678
|
+
(claude-agent or codex-agent: subscription login, no API key).
|
|
679
|
+
Returns (reply, error) like _call_anthropic."""
|
|
680
|
+
from ._providers import _AGENT_BACKENDS
|
|
681
|
+
adapter_name, install_hint = _AGENT_BACKENDS[model_source]
|
|
683
682
|
try:
|
|
684
683
|
from catclaws._adapters import get_adapter
|
|
685
684
|
except ImportError:
|
|
686
|
-
return None,
|
|
685
|
+
return None, f"cat-claws is not installed. Run: {install_hint}"
|
|
687
686
|
import asyncio
|
|
688
|
-
adapter = get_adapter(
|
|
687
|
+
adapter = get_adapter(adapter_name)
|
|
689
688
|
_system = ("You are an image classification engine. Follow the user's "
|
|
690
689
|
"instructions exactly and reply with only what they ask for.")
|
|
691
690
|
try:
|
|
@@ -738,9 +737,11 @@ Provide the final categorization in the same JSON format:"""
|
|
|
738
737
|
image_content = {"type": "image_url", "image_url": {"url": encoded_image, "detail": "high"}}
|
|
739
738
|
return _call_mistral(prompt, step2_prompt, step3_prompt, step4_prompt, image_content)
|
|
740
739
|
|
|
741
|
-
elif model_source
|
|
740
|
+
elif model_source in ("claude-agent", "codex-agent"):
|
|
742
741
|
media_type = f"image/{ext}" if ext else "image/jpeg"
|
|
743
|
-
|
|
742
|
+
if media_type == "image/jpg":
|
|
743
|
+
media_type = "image/jpeg"
|
|
744
|
+
return _call_agent_image(base_prompt_text, encoded, media_type)
|
|
744
745
|
|
|
745
746
|
else:
|
|
746
747
|
raise ValueError("Unknown source! Choose from OpenAI, Anthropic, Perplexity, Google, xAI, Huggingface, or Mistral")
|
|
@@ -399,12 +399,8 @@ def pdf_multi_class(
|
|
|
399
399
|
"(the text-only CLI shim). Use model_source='claude-agent' (the cat-claws "
|
|
400
400
|
"subscription backend) or an API-key provider."
|
|
401
401
|
)
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
"PDF classification is not yet supported with "
|
|
405
|
-
"model_source='codex-agent'. Use model_source='claude-agent' (the "
|
|
406
|
-
"multimodal subscription backend) or an API-key provider."
|
|
407
|
-
)
|
|
402
|
+
from ._providers import _require_agent_sign_in
|
|
403
|
+
_require_agent_sign_in(model_source)
|
|
408
404
|
|
|
409
405
|
# Providers with native PDF support (only used in image/both modes)
|
|
410
406
|
native_pdf_providers = {"anthropic", "google"}
|
|
@@ -1192,23 +1188,27 @@ Provide the final categorization in the same JSON format:"""
|
|
|
1192
1188
|
|
|
1193
1189
|
return """{"1":"e"}""", "Max retries exceeded"
|
|
1194
1190
|
|
|
1195
|
-
def
|
|
1196
|
-
"""PDF-page classification via
|
|
1197
|
-
|
|
1198
|
-
|
|
1191
|
+
def _call_agent_pdf(base_text, encoded_image):
|
|
1192
|
+
"""PDF-page classification via a cat-claws adapter (claude-agent or
|
|
1193
|
+
codex-agent: subscription login, no API key). The page is rendered to
|
|
1194
|
+
an image (PDF-as-images); encoded_image=None sends text only (mode
|
|
1195
|
+
"text"). Returns (reply, error)."""
|
|
1196
|
+
from ._providers import _AGENT_BACKENDS
|
|
1197
|
+
adapter_name, install_hint = _AGENT_BACKENDS[model_source]
|
|
1199
1198
|
try:
|
|
1200
1199
|
from catclaws._adapters import get_adapter
|
|
1201
1200
|
except ImportError:
|
|
1202
|
-
return None,
|
|
1201
|
+
return None, f"cat-claws is not installed. Run: {install_hint}"
|
|
1203
1202
|
import asyncio
|
|
1204
|
-
adapter = get_adapter(
|
|
1203
|
+
adapter = get_adapter(adapter_name)
|
|
1205
1204
|
_system = ("You are a document-page classification engine. Follow the "
|
|
1206
1205
|
"user's instructions exactly and reply with only what they ask for.")
|
|
1207
1206
|
try:
|
|
1208
1207
|
reply, error = asyncio.run(adapter.one_shot(
|
|
1209
1208
|
base_text, system_prompt=_system, model=user_model,
|
|
1210
1209
|
thinking_budget=thinking_budget or 0,
|
|
1211
|
-
images
|
|
1210
|
+
**({"images": [{"media_type": "image/png", "data": encoded_image}]}
|
|
1211
|
+
if encoded_image else {}),
|
|
1212
1212
|
))
|
|
1213
1213
|
return (None, error) if error else (reply, None)
|
|
1214
1214
|
except Exception as e:
|
|
@@ -1244,6 +1244,8 @@ Provide the final categorization in the same JSON format:"""
|
|
|
1244
1244
|
return _call_openai_text_only(base_prompt_text, step2_prompt, step3_prompt, step4_prompt)
|
|
1245
1245
|
elif model_source == "mistral":
|
|
1246
1246
|
return _call_mistral_text_only(base_prompt_text, step2_prompt, step3_prompt, step4_prompt)
|
|
1247
|
+
elif model_source in ("claude-agent", "codex-agent"):
|
|
1248
|
+
return _call_agent_pdf(base_prompt_text, None)
|
|
1247
1249
|
else:
|
|
1248
1250
|
raise ValueError(f"Unknown source! Choose from OpenAI, Anthropic, Perplexity, Google, xAI, Huggingface, or Mistral")
|
|
1249
1251
|
|
|
@@ -1293,12 +1295,12 @@ Provide the final categorization in the same JSON format:"""
|
|
|
1293
1295
|
prompt_data = _build_prompt_google_pdf(encoded_pdf, base_prompt_text)
|
|
1294
1296
|
return _call_google(prompt_data, step2_prompt, step3_prompt, step4_prompt, base_prompt_text)
|
|
1295
1297
|
|
|
1296
|
-
elif model_source
|
|
1298
|
+
elif model_source in ("claude-agent", "codex-agent"):
|
|
1297
1299
|
image_bytes, is_valid = _extract_page_as_image_bytes(pdf_path, page_index)
|
|
1298
1300
|
if not is_valid:
|
|
1299
1301
|
return None, "Failed to render PDF page to image"
|
|
1300
1302
|
encoded_image = _encode_bytes_to_base64(image_bytes)
|
|
1301
|
-
return
|
|
1303
|
+
return _call_agent_pdf(base_prompt_text, encoded_image)
|
|
1302
1304
|
|
|
1303
1305
|
# Handle providers requiring image conversion
|
|
1304
1306
|
else:
|
|
@@ -53,6 +53,7 @@ from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
|
53
53
|
from typing import Optional, Callable, Union
|
|
54
54
|
|
|
55
55
|
from ._utils import _extract_balanced_json
|
|
56
|
+
from ._providers import _require_agent_sign_in
|
|
56
57
|
from .text_functions import (
|
|
57
58
|
UnifiedLLMClient,
|
|
58
59
|
detect_provider,
|
|
@@ -684,6 +685,7 @@ def prepare_model_configs(
|
|
|
684
685
|
"Install: pip install cat-stack[agent]\n"
|
|
685
686
|
+ "="*60
|
|
686
687
|
)
|
|
688
|
+
_require_agent_sign_in(detected_provider)
|
|
687
689
|
elif detected_provider == "codex-agent":
|
|
688
690
|
try:
|
|
689
691
|
import catclaws # noqa: F401
|
|
@@ -696,6 +698,7 @@ def prepare_model_configs(
|
|
|
696
698
|
'Install: pip install "cat-stack[codex-agent]"\n'
|
|
697
699
|
+ "="*60
|
|
698
700
|
)
|
|
701
|
+
_require_agent_sign_in(detected_provider)
|
|
699
702
|
else:
|
|
700
703
|
# Validate API key exists for cloud providers
|
|
701
704
|
if not api_key:
|
|
@@ -3863,6 +3866,17 @@ multi_class_ensemble = classify_ensemble
|
|
|
3863
3866
|
# Summarization helpers
|
|
3864
3867
|
# =============================================================================
|
|
3865
3868
|
|
|
3869
|
+
def _format_item_errors(errors: dict, multi_model: bool) -> str:
|
|
3870
|
+
"""One row's failure reasons for the `error_message` column: the bare
|
|
3871
|
+
message for a single model, "model: message" pairs for an ensemble.
|
|
3872
|
+
Empty string when the row had no errors."""
|
|
3873
|
+
if not errors:
|
|
3874
|
+
return ""
|
|
3875
|
+
if not multi_model:
|
|
3876
|
+
return "; ".join(str(e) for e in errors.values())
|
|
3877
|
+
return "; ".join(f"{m}: {e}" for m, e in errors.items())
|
|
3878
|
+
|
|
3879
|
+
|
|
3866
3880
|
def _save_partial_summarize_results(all_results, model_configs, model_names, is_pdf_mode, filename, save_directory):
|
|
3867
3881
|
"""Save partial summarization results to CSV for safety/incremental saves."""
|
|
3868
3882
|
rows = []
|
|
@@ -3895,6 +3909,7 @@ def _save_partial_summarize_results(all_results, model_configs, model_names, is_
|
|
|
3895
3909
|
) else "partial"
|
|
3896
3910
|
else:
|
|
3897
3911
|
row["processing_status"] = "success"
|
|
3912
|
+
row["error_message"] = _format_item_errors(entry["errors"], len(model_configs) > 1)
|
|
3898
3913
|
|
|
3899
3914
|
rows.append(row)
|
|
3900
3915
|
|
|
@@ -4511,6 +4526,9 @@ def summarize_ensemble(
|
|
|
4511
4526
|
|
|
4512
4527
|
if error and error != "skipped":
|
|
4513
4528
|
still_failed.append((idx, model_name))
|
|
4529
|
+
# Keep the latest reason so the output reports what the
|
|
4530
|
+
# final attempt actually hit.
|
|
4531
|
+
all_results[idx]["errors"][model_name] = error
|
|
4514
4532
|
else:
|
|
4515
4533
|
# Update the stored result
|
|
4516
4534
|
all_results[idx]["model_results"][model_name] = json_result
|
|
@@ -4631,10 +4649,20 @@ def summarize_ensemble(
|
|
|
4631
4649
|
else:
|
|
4632
4650
|
row["processing_status"] = "success"
|
|
4633
4651
|
|
|
4652
|
+
# Why a row failed: previously collected per item but never written
|
|
4653
|
+
# out, so failures were silent (status "error", no reason).
|
|
4654
|
+
row["error_message"] = _format_item_errors(entry["errors"], len(model_configs) > 1)
|
|
4655
|
+
|
|
4634
4656
|
rows.append(row)
|
|
4635
4657
|
|
|
4636
4658
|
df = pd.DataFrame(rows)
|
|
4637
4659
|
|
|
4660
|
+
n_failed = int((df["error_message"] != "").sum()) if "error_message" in df else 0
|
|
4661
|
+
if n_failed:
|
|
4662
|
+
first = df.loc[df["error_message"] != "", "error_message"].iloc[0]
|
|
4663
|
+
print(f"\n[CatLLM] WARNING: {n_failed} of {len(df)} item(s) had errors "
|
|
4664
|
+
f"(see the error_message column). First error: {first}")
|
|
4665
|
+
|
|
4638
4666
|
# Save to file if requested
|
|
4639
4667
|
if filename:
|
|
4640
4668
|
save_path = os.path.join(save_directory, filename) if save_directory else filename
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|