cat-stack 2.5.0__tar.gz → 2.5.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cat_stack-2.5.0 → cat_stack-2.5.2}/PKG-INFO +12 -6
- {cat_stack-2.5.0 → cat_stack-2.5.2}/README.md +8 -2
- {cat_stack-2.5.0 → cat_stack-2.5.2}/pyproject.toml +2 -2
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/__about__.py +1 -1
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/_providers.py +56 -4
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/collapse_themes.py +33 -5
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/explore.py +4 -2
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/extract.py +4 -2
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/image_functions.py +12 -13
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/pdf_functions.py +15 -15
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/text_functions_ensemble.py +25 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/.gitignore +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/LICENSE +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/cat_stack/__init__.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/__init__.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/_batch.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/_category_analysis.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/_chunked.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/_embeddings.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/_formatter.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/_pilot_test.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/_prompts.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/_review_ui.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/_tiebreaker.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/_utils.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/_web_fetch.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/_wrapper_helpers.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/calls/CoVe.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/calls/__init__.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/calls/image_CoVe.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/calls/image_stepback.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/calls/pdf_CoVe.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/calls/pdf_stepback.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/calls/stepback.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/calls/top_n.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/classify.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/images/circle.png +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/images/cube.png +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/images/diamond.png +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/images/overlapping_pentagons.png +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/images/rectangles.png +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/model_reference_list.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/prompt_tune.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/summarize.py +0 -0
- {cat_stack-2.5.0 → cat_stack-2.5.2}/src/catstack/text_functions.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: cat-stack
|
|
3
|
-
Version: 2.5.
|
|
3
|
+
Version: 2.5.2
|
|
4
4
|
Summary: Domain-agnostic text, image, PDF, and DOCX classification engine powered by LLMs
|
|
5
5
|
Project-URL: Documentation, https://github.com/chrissoria/cat-stack#readme
|
|
6
6
|
Project-URL: Issues, https://github.com/chrissoria/cat-stack/issues
|
|
@@ -24,9 +24,9 @@ Requires-Dist: pandas
|
|
|
24
24
|
Requires-Dist: requests
|
|
25
25
|
Requires-Dist: tqdm
|
|
26
26
|
Provides-Extra: agent
|
|
27
|
-
Requires-Dist: cat-claws[claude]>=0.3.
|
|
27
|
+
Requires-Dist: cat-claws[claude]>=0.3.2; extra == 'agent'
|
|
28
28
|
Provides-Extra: codex-agent
|
|
29
|
-
Requires-Dist: cat-claws[codex]>=0.3.
|
|
29
|
+
Requires-Dist: cat-claws[codex]>=0.3.2; extra == 'codex-agent'
|
|
30
30
|
Provides-Extra: docx
|
|
31
31
|
Requires-Dist: python-docx>=1.0.0; extra == 'docx'
|
|
32
32
|
Provides-Extra: embeddings
|
|
@@ -138,7 +138,7 @@ cat.classify(
|
|
|
138
138
|
```
|
|
139
139
|
|
|
140
140
|
### `extract()`
|
|
141
|
-
Discover categories from a corpus using LLM-driven exploration.
|
|
141
|
+
Discover categories from a corpus using LLM-driven exploration. Since v2.5.0, text consolidation runs the full explore → `collapse_themes()` pipeline (`engine="collapse"`, the default): the entire raw label inventory reaches the semantic merge — Jaro-Winkler dedup, embedding pre-merge, quality-controlled LLM passes, then a count-guided reduction to at most `max_categories`. Pass `engine="legacy"` to reproduce pre-2.5 runs (single merge call over a truncated inventory). `collapse_kwargs` forwards options to `collapse_themes()`; `max_workers` parallelizes both extraction and consolidation.
|
|
142
142
|
|
|
143
143
|
```python
|
|
144
144
|
cat.extract(
|
|
@@ -197,7 +197,7 @@ cat.collapse_themes(
|
|
|
197
197
|
| Parameter | Default | Description |
|
|
198
198
|
| --- | --- | --- |
|
|
199
199
|
| `input_data` | — | List of category labels, or a frequency `Series`/`dict` (`label -> count`). |
|
|
200
|
-
| `api_key` | `None` | API key for the LLM provider (
|
|
200
|
+
| `api_key` | `None` | API key for the LLM provider. Not required for subscription/CLI backends (`claude-code`, `claude-agent`, `codex-agent`) or `ollama`. |
|
|
201
201
|
| `description` | `""` | The survey question or context, used in the merge prompt. |
|
|
202
202
|
| `passes` | `1` | Number of merge iterations, or `"auto"` to iterate until the embedding-quality benchmark peaks. |
|
|
203
203
|
| `max_passes` | `10` | Cap on iterations when `passes="auto"`. |
|
|
@@ -207,6 +207,8 @@ cat.collapse_themes(
|
|
|
207
207
|
| `embedding_merge_threshold` | `0.92` | Cosine similarity at/above which labels are merged in the pre-LLM embedding step. `None`/`>=1.0` disables it. |
|
|
208
208
|
| `shuffle` | `True` | Randomize order each pass so batch composition varies (improves convergence stability). |
|
|
209
209
|
| `final_consolidation` | `0.82` | Cosine threshold for one greedy global embedding re-merge after all passes, collapsing cross-batch duplicates. Conservative by design (errs toward keeping categories). `False`/`None` skips it. |
|
|
210
|
+
| `top_n` | `None` | If set, a final global LLM call consolidates the surviving list into at most N categories, guided by each label's generation count (frequent themes favored; overlapping labels merged rather than dropped). Guaranteed `<= top_n` via truncation plus a deterministic top-N-by-count fallback. |
|
|
211
|
+
| `prune` | `False` | `True` = drop conceptual duplicates keeping one representative verbatim; never renames or merges merely-related labels. |
|
|
210
212
|
| `user_model` | `"gpt-4o"` | Model for the merge phase. Use a capable model — small models can degenerate. |
|
|
211
213
|
| `model_source` | `"auto"` | Provider for `user_model` (`"auto"`, `"openai"`, `"huggingface"`, …). |
|
|
212
214
|
| `unique_model` | `None` | If set, run an initial extract-unique thinning phase on this (typically cheaper) model before the merge phase. `None` skips the phase (backward compatible). |
|
|
@@ -237,6 +239,10 @@ OpenAI, Anthropic, Google (Gemini), Mistral, Perplexity, xAI (Grok), HuggingFace
|
|
|
237
239
|
|
|
238
240
|
All providers use the same `(model_name, provider, api_key)` tuple format. Provider is auto-detected from model name if omitted.
|
|
239
241
|
|
|
242
|
+
**Subscription backends (no API key).** Three `model_source` values authenticate through a chat subscription instead of a metered key — leave `api_key` unset: `"claude-agent"` (Claude subscription via the Agent SDK; `pip install "cat-stack[agent]"`), `"claude-code"` (the Claude Code CLI, if installed — no extra needed), and `"codex-agent"` (ChatGPT subscription; `pip install "cat-stack[codex-agent]"`). Classification, extraction, exploration, and summarization all route through them.
|
|
243
|
+
|
|
244
|
+
**Reproducibility note:** the same nominal model can behave differently across access routes. In a seed-matched benchmark (Claude Sonnet 5, temperature 0), extraction through the Agent SDK produced more varied label phrasings than the direct API (89% vs. 74% unique raw labels), while final consolidated taxonomies were equivalent. If raw label counts matter to your analysis, record the access route alongside the model version.
|
|
245
|
+
|
|
240
246
|
## Features
|
|
241
247
|
|
|
242
248
|
- **Automatic prompt optimization** (`prompt_tune`) — correct a small sample in a browser UI, and the system generates per-category instructions that improve accuracy
|
|
@@ -97,7 +97,7 @@ cat.classify(
|
|
|
97
97
|
```
|
|
98
98
|
|
|
99
99
|
### `extract()`
|
|
100
|
-
Discover categories from a corpus using LLM-driven exploration.
|
|
100
|
+
Discover categories from a corpus using LLM-driven exploration. Since v2.5.0, text consolidation runs the full explore → `collapse_themes()` pipeline (`engine="collapse"`, the default): the entire raw label inventory reaches the semantic merge — Jaro-Winkler dedup, embedding pre-merge, quality-controlled LLM passes, then a count-guided reduction to at most `max_categories`. Pass `engine="legacy"` to reproduce pre-2.5 runs (single merge call over a truncated inventory). `collapse_kwargs` forwards options to `collapse_themes()`; `max_workers` parallelizes both extraction and consolidation.
|
|
101
101
|
|
|
102
102
|
```python
|
|
103
103
|
cat.extract(
|
|
@@ -156,7 +156,7 @@ cat.collapse_themes(
|
|
|
156
156
|
| Parameter | Default | Description |
|
|
157
157
|
| --- | --- | --- |
|
|
158
158
|
| `input_data` | — | List of category labels, or a frequency `Series`/`dict` (`label -> count`). |
|
|
159
|
-
| `api_key` | `None` | API key for the LLM provider (
|
|
159
|
+
| `api_key` | `None` | API key for the LLM provider. Not required for subscription/CLI backends (`claude-code`, `claude-agent`, `codex-agent`) or `ollama`. |
|
|
160
160
|
| `description` | `""` | The survey question or context, used in the merge prompt. |
|
|
161
161
|
| `passes` | `1` | Number of merge iterations, or `"auto"` to iterate until the embedding-quality benchmark peaks. |
|
|
162
162
|
| `max_passes` | `10` | Cap on iterations when `passes="auto"`. |
|
|
@@ -166,6 +166,8 @@ cat.collapse_themes(
|
|
|
166
166
|
| `embedding_merge_threshold` | `0.92` | Cosine similarity at/above which labels are merged in the pre-LLM embedding step. `None`/`>=1.0` disables it. |
|
|
167
167
|
| `shuffle` | `True` | Randomize order each pass so batch composition varies (improves convergence stability). |
|
|
168
168
|
| `final_consolidation` | `0.82` | Cosine threshold for one greedy global embedding re-merge after all passes, collapsing cross-batch duplicates. Conservative by design (errs toward keeping categories). `False`/`None` skips it. |
|
|
169
|
+
| `top_n` | `None` | If set, a final global LLM call consolidates the surviving list into at most N categories, guided by each label's generation count (frequent themes favored; overlapping labels merged rather than dropped). Guaranteed `<= top_n` via truncation plus a deterministic top-N-by-count fallback. |
|
|
170
|
+
| `prune` | `False` | `True` = drop conceptual duplicates keeping one representative verbatim; never renames or merges merely-related labels. |
|
|
169
171
|
| `user_model` | `"gpt-4o"` | Model for the merge phase. Use a capable model — small models can degenerate. |
|
|
170
172
|
| `model_source` | `"auto"` | Provider for `user_model` (`"auto"`, `"openai"`, `"huggingface"`, …). |
|
|
171
173
|
| `unique_model` | `None` | If set, run an initial extract-unique thinning phase on this (typically cheaper) model before the merge phase. `None` skips the phase (backward compatible). |
|
|
@@ -196,6 +198,10 @@ OpenAI, Anthropic, Google (Gemini), Mistral, Perplexity, xAI (Grok), HuggingFace
|
|
|
196
198
|
|
|
197
199
|
All providers use the same `(model_name, provider, api_key)` tuple format. Provider is auto-detected from model name if omitted.
|
|
198
200
|
|
|
201
|
+
**Subscription backends (no API key).** Three `model_source` values authenticate through a chat subscription instead of a metered key — leave `api_key` unset: `"claude-agent"` (Claude subscription via the Agent SDK; `pip install "cat-stack[agent]"`), `"claude-code"` (the Claude Code CLI, if installed — no extra needed), and `"codex-agent"` (ChatGPT subscription; `pip install "cat-stack[codex-agent]"`). Classification, extraction, exploration, and summarization all route through them.
|
|
202
|
+
|
|
203
|
+
**Reproducibility note:** the same nominal model can behave differently across access routes. In a seed-matched benchmark (Claude Sonnet 5, temperature 0), extraction through the Agent SDK produced more varied label phrasings than the direct API (89% vs. 74% unique raw labels), while final consolidated taxonomies were equivalent. If raw label counts matter to your analysis, record the access route alongside the model version.
|
|
204
|
+
|
|
199
205
|
## Features
|
|
200
206
|
|
|
201
207
|
- **Automatic prompt optimization** (`prompt_tune`) — correct a small sample in a browser UI, and the system generates per-category instructions that improve accuracy
|
|
@@ -42,8 +42,8 @@ embeddings = ["sentence-transformers>=2.2.0"]
|
|
|
42
42
|
# `agent` keeps its historical meaning (the Claude backend) so every shipped
|
|
43
43
|
# install hint stays true; `codex-agent` matches the provider string so the
|
|
44
44
|
# error-message hint is copy-pasteable.
|
|
45
|
-
agent = ["cat-claws[claude]>=0.3.
|
|
46
|
-
codex-agent = ["cat-claws[codex]>=0.3.
|
|
45
|
+
agent = ["cat-claws[claude]>=0.3.2"]
|
|
46
|
+
codex-agent = ["cat-claws[codex]>=0.3.2"]
|
|
47
47
|
|
|
48
48
|
[project.urls]
|
|
49
49
|
Documentation = "https://github.com/chrissoria/cat-stack#readme"
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# SPDX-FileCopyrightText: 2025-present Christopher Soria <chrissoria@berkeley.edu>
|
|
2
2
|
#
|
|
3
3
|
# SPDX-License-Identifier: GPL-3.0-or-later
|
|
4
|
-
__version__ = "2.5.
|
|
4
|
+
__version__ = "2.5.2"
|
|
5
5
|
__author__ = "Chris Soria"
|
|
6
6
|
__email__ = "chrissoria@berkeley.edu"
|
|
7
7
|
__title__ = "cat-stack"
|
|
@@ -735,8 +735,10 @@ PROVIDER_CONFIG = {
|
|
|
735
735
|
|
|
736
736
|
|
|
737
737
|
# Providers that route through complete() with no HTTP endpoint of their own
|
|
738
|
-
# (subscription logins / CLI). Features that build a direct HTTP request
|
|
739
|
-
#
|
|
738
|
+
# (subscription logins / CLI). Features that build a direct HTTP request can't
|
|
739
|
+
# use them — guard with a clear error, not a deep crash. (Images and PDF pages
|
|
740
|
+
# DO work on the cat-claws agent backends: they go through the adapters'
|
|
741
|
+
# images= argument, see _split_agent_content / _call_agent_image.)
|
|
740
742
|
_SUBSCRIPTION_PROVIDERS = ("claude-code", "claude-agent", "codex-agent")
|
|
741
743
|
|
|
742
744
|
# Agent-SDK backends routed through cat-claws: provider -> (adapter name,
|
|
@@ -748,6 +750,46 @@ _AGENT_BACKENDS = {
|
|
|
748
750
|
}
|
|
749
751
|
|
|
750
752
|
|
|
753
|
+
def _split_agent_content(content):
|
|
754
|
+
"""Split one message's content into (text, images) for the cat-claws
|
|
755
|
+
adapters.
|
|
756
|
+
|
|
757
|
+
Plain-string content is text only. Multimodal content is a list of blocks
|
|
758
|
+
in whichever provider shape the prompt builder produced (Anthropic
|
|
759
|
+
`image` + base64 `source`, OpenAI `image_url` data URL, Google
|
|
760
|
+
`inline_data`); text blocks are joined and image blocks become the
|
|
761
|
+
adapters' ``{"media_type", "data"}`` dicts. Unknown block types are
|
|
762
|
+
dropped rather than stringified into the prompt.
|
|
763
|
+
"""
|
|
764
|
+
if not isinstance(content, list):
|
|
765
|
+
return content, []
|
|
766
|
+
texts, images = [], []
|
|
767
|
+
for block in content:
|
|
768
|
+
if not isinstance(block, dict):
|
|
769
|
+
continue
|
|
770
|
+
kind = block.get("type")
|
|
771
|
+
if kind == "text":
|
|
772
|
+
texts.append(block.get("text", ""))
|
|
773
|
+
elif kind == "image":
|
|
774
|
+
src = block.get("source") or {}
|
|
775
|
+
if src.get("type") == "base64" and src.get("data"):
|
|
776
|
+
images.append({"media_type": src.get("media_type") or "image/png",
|
|
777
|
+
"data": src["data"]})
|
|
778
|
+
elif kind == "image_url":
|
|
779
|
+
url = (block.get("image_url") or {}).get("url", "")
|
|
780
|
+
if url.startswith("data:") and ";base64," in url:
|
|
781
|
+
header, data = url.split(",", 1)
|
|
782
|
+
images.append({"media_type": header[5:].split(";", 1)[0] or "image/png",
|
|
783
|
+
"data": data})
|
|
784
|
+
elif kind == "inline_data" and block.get("data"):
|
|
785
|
+
images.append({"media_type": block.get("mime_type") or "image/png",
|
|
786
|
+
"data": block["data"]})
|
|
787
|
+
for im in images: # the API wants image/jpeg, not image/jpg
|
|
788
|
+
if im["media_type"] == "image/jpg":
|
|
789
|
+
im["media_type"] = "image/jpeg"
|
|
790
|
+
return "\n\n".join(t for t in texts if t), images
|
|
791
|
+
|
|
792
|
+
|
|
751
793
|
def _require_http_provider(model_source, feature):
|
|
752
794
|
"""Raise a clear error when an HTTP-only feature is used with a
|
|
753
795
|
subscription/CLI provider (claude-code / claude-agent / codex-agent)."""
|
|
@@ -1272,17 +1314,26 @@ class UnifiedLLMClient:
|
|
|
1272
1314
|
)
|
|
1273
1315
|
import asyncio
|
|
1274
1316
|
|
|
1317
|
+
# Multimodal content (image summaries, rendered PDF pages) arrives as
|
|
1318
|
+
# a list of blocks: text goes to the prompt, images to the adapter's
|
|
1319
|
+
# `images=` argument (base.AgentAdapter.one_shot contract).
|
|
1275
1320
|
system_parts = []
|
|
1276
1321
|
user_parts = []
|
|
1322
|
+
images = []
|
|
1277
1323
|
for msg in messages:
|
|
1324
|
+
text, msg_images = _split_agent_content(msg["content"])
|
|
1278
1325
|
if msg["role"] == "system":
|
|
1279
|
-
system_parts.append(
|
|
1326
|
+
system_parts.append(text)
|
|
1280
1327
|
elif msg["role"] in ("user", "assistant"):
|
|
1281
|
-
user_parts.append(
|
|
1328
|
+
user_parts.append(text)
|
|
1329
|
+
images.extend(msg_images)
|
|
1282
1330
|
system_prompt = "\n\n".join(system_parts) if system_parts else None
|
|
1283
1331
|
user_prompt = "\n\n".join(user_parts)
|
|
1284
1332
|
|
|
1285
1333
|
adapter = get_adapter(adapter_name)
|
|
1334
|
+
# Only pass images when there are some, so text-only calls are
|
|
1335
|
+
# byte-identical to before.
|
|
1336
|
+
image_kwargs = {"images": images} if images else {}
|
|
1286
1337
|
try:
|
|
1287
1338
|
return asyncio.run(
|
|
1288
1339
|
adapter.one_shot(
|
|
@@ -1290,6 +1341,7 @@ class UnifiedLLMClient:
|
|
|
1290
1341
|
system_prompt=system_prompt,
|
|
1291
1342
|
model=self.model,
|
|
1292
1343
|
thinking_budget=thinking_budget or 0,
|
|
1344
|
+
**image_kwargs,
|
|
1293
1345
|
)
|
|
1294
1346
|
)
|
|
1295
1347
|
except Exception as e:
|
|
@@ -25,7 +25,7 @@ import numpy as np
|
|
|
25
25
|
import pandas as pd
|
|
26
26
|
from jellyfish import jaro_winkler_similarity
|
|
27
27
|
|
|
28
|
-
from ._providers import UnifiedLLMClient, detect_provider
|
|
28
|
+
from ._providers import UnifiedLLMClient, detect_provider, _SUBSCRIPTION_PROVIDERS
|
|
29
29
|
from ._utils import _clean_label
|
|
30
30
|
|
|
31
31
|
__all__ = [
|
|
@@ -301,6 +301,19 @@ def _count_guidance(current, input_data):
|
|
|
301
301
|
return {lbl: max(int(c), orig.get(_norm_key(lbl), 0)) for lbl, c in cur.items()}
|
|
302
302
|
|
|
303
303
|
|
|
304
|
+
def _as_label_list(x):
|
|
305
|
+
"""Enforce the documented list[str] return contract at exit points.
|
|
306
|
+
|
|
307
|
+
Dict / Series / DataFrame inputs can reach a return untouched (e.g.
|
|
308
|
+
passes=0 with a top_n no-op), which would leak the input container —
|
|
309
|
+
and, via the filename write, save counts instead of labels."""
|
|
310
|
+
if isinstance(x, list):
|
|
311
|
+
return [str(v) for v in x]
|
|
312
|
+
if isinstance(x, dict):
|
|
313
|
+
return [str(k) for k in x]
|
|
314
|
+
return [str(k) for k in _to_counts(x)]
|
|
315
|
+
|
|
316
|
+
|
|
304
317
|
def _to_counts(input_data):
|
|
305
318
|
"""Coerce the accepted input forms into a {category: count} dict."""
|
|
306
319
|
if isinstance(input_data, pd.DataFrame):
|
|
@@ -425,7 +438,9 @@ def collapse_themes(
|
|
|
425
438
|
input_data: Themes to collapse. list[str] (duplicates allowed), pandas
|
|
426
439
|
Series, dict {category: count}, or DataFrame with "category"
|
|
427
440
|
[and optional "count"] columns.
|
|
428
|
-
api_key (str): API key for the model provider.
|
|
441
|
+
api_key (str): API key for the model provider. Not required for
|
|
442
|
+
subscription/CLI backends (`claude-code`, `claude-agent`,
|
|
443
|
+
`codex-agent`) or `ollama`.
|
|
429
444
|
description (str): Data/question context, injected into the prompt — e.g.
|
|
430
445
|
the survey question the categories came from. Helps the model judge
|
|
431
446
|
which distinctions matter.
|
|
@@ -505,9 +520,6 @@ def collapse_themes(
|
|
|
505
520
|
... aggressive=True, passes="auto", max_workers=8,
|
|
506
521
|
... )
|
|
507
522
|
"""
|
|
508
|
-
if not api_key:
|
|
509
|
-
raise ValueError("collapse_themes() needs an api_key for the LLM call.")
|
|
510
|
-
|
|
511
523
|
mode = "merge" if aggressive else "unique"
|
|
512
524
|
|
|
513
525
|
# The main (merge) phase runs on merge_model if given, else user_model. A separate
|
|
@@ -518,6 +530,20 @@ def collapse_themes(
|
|
|
518
530
|
merge_name = merge_model or user_model
|
|
519
531
|
merge_src = merge_model_source if merge_model else model_source
|
|
520
532
|
merge_provider = detect_provider(merge_name, merge_src)
|
|
533
|
+
|
|
534
|
+
# A key is only needed for providers that bill one; the subscription/CLI
|
|
535
|
+
# backends and ollama run keyless.
|
|
536
|
+
_keyless = set(_SUBSCRIPTION_PROVIDERS) | {"ollama"}
|
|
537
|
+
_providers_used = {merge_provider}
|
|
538
|
+
if unique_model:
|
|
539
|
+
_providers_used.add(detect_provider(unique_model, unique_model_source))
|
|
540
|
+
if not api_key and not _providers_used <= _keyless:
|
|
541
|
+
raise ValueError(
|
|
542
|
+
"collapse_themes() needs an api_key for the LLM call. "
|
|
543
|
+
"(Not required for the claude-code/claude-agent/codex-agent "
|
|
544
|
+
"backends or ollama.)"
|
|
545
|
+
)
|
|
546
|
+
|
|
521
547
|
client = UnifiedLLMClient(provider=merge_provider, api_key=api_key, model=merge_name)
|
|
522
548
|
|
|
523
549
|
def _run(cl, items, md, p):
|
|
@@ -589,6 +615,7 @@ def collapse_themes(
|
|
|
589
615
|
if top_n and len(items) > int(top_n):
|
|
590
616
|
items = _select_top_n(client, _count_guidance(items, input_data),
|
|
591
617
|
int(top_n), description, creativity, dedupe_threshold)
|
|
618
|
+
items = _as_label_list(items)
|
|
592
619
|
if filename:
|
|
593
620
|
pd.DataFrame({"category": items}).to_csv(filename, index=False)
|
|
594
621
|
print(f"Collapsed categories saved to {filename}")
|
|
@@ -674,6 +701,7 @@ def collapse_themes(
|
|
|
674
701
|
current = _select_top_n(client, _count_guidance(current, input_data),
|
|
675
702
|
int(top_n), description, creativity, dedupe_threshold)
|
|
676
703
|
|
|
704
|
+
current = _as_label_list(current)
|
|
677
705
|
if filename:
|
|
678
706
|
pd.DataFrame({"category": current}).to_csv(filename, index=False)
|
|
679
707
|
print(f"Collapsed categories saved to {filename}")
|
|
@@ -16,7 +16,7 @@ from .text_functions import explore_common_categories
|
|
|
16
16
|
|
|
17
17
|
def explore(
|
|
18
18
|
input_data,
|
|
19
|
-
api_key,
|
|
19
|
+
api_key=None,
|
|
20
20
|
description="",
|
|
21
21
|
max_categories=12,
|
|
22
22
|
categories_per_chunk=10,
|
|
@@ -46,7 +46,9 @@ def explore(
|
|
|
46
46
|
|
|
47
47
|
Args:
|
|
48
48
|
input_data: List of text responses or pandas Series.
|
|
49
|
-
api_key (str): API key for the model provider.
|
|
49
|
+
api_key (str, optional): API key for the model provider. Not required
|
|
50
|
+
for subscription/CLI backends (claude-code, claude-agent,
|
|
51
|
+
codex-agent) or ollama.
|
|
50
52
|
description (str): Description of the data context. Content-neutral —
|
|
51
53
|
for survey responses this is the question that was asked; for
|
|
52
54
|
documents or posts this describes what the content is about.
|
|
@@ -41,7 +41,7 @@ from .collapse_themes import collapse_themes
|
|
|
41
41
|
|
|
42
42
|
def extract(
|
|
43
43
|
input_data,
|
|
44
|
-
api_key,
|
|
44
|
+
api_key=None,
|
|
45
45
|
input_type="auto",
|
|
46
46
|
description="",
|
|
47
47
|
survey_question=None,
|
|
@@ -79,7 +79,9 @@ def extract(
|
|
|
79
79
|
- For text: list of text responses or pandas Series
|
|
80
80
|
- For image: directory path, single file, or list of image paths
|
|
81
81
|
- For pdf: directory path, single file, or list of PDF paths
|
|
82
|
-
api_key (str): API key for the model provider.
|
|
82
|
+
api_key (str, optional): API key for the model provider. Not required
|
|
83
|
+
for subscription/CLI backends (claude-code, claude-agent,
|
|
84
|
+
codex-agent) or ollama.
|
|
83
85
|
input_type (str): Type of input data. Options:
|
|
84
86
|
- "auto" (default): Auto-detect from file extensions
|
|
85
87
|
- "text": Text responses
|
|
@@ -154,12 +154,6 @@ def image_multi_class(
|
|
|
154
154
|
"(the text-only CLI shim). Use model_source='claude-agent' (the cat-claws "
|
|
155
155
|
"subscription backend) or an API-key provider."
|
|
156
156
|
)
|
|
157
|
-
if model_source == "codex-agent":
|
|
158
|
-
raise ValueError(
|
|
159
|
-
"Image classification is not yet supported with "
|
|
160
|
-
"model_source='codex-agent'. Use model_source='claude-agent' (the "
|
|
161
|
-
"multimodal subscription backend) or an API-key provider."
|
|
162
|
-
)
|
|
163
157
|
|
|
164
158
|
image_files = _load_image_files(image_input)
|
|
165
159
|
|
|
@@ -677,15 +671,18 @@ Provide the final categorization in the same JSON format:"""
|
|
|
677
671
|
|
|
678
672
|
return """{"1":"e"}""", "Max retries exceeded"
|
|
679
673
|
|
|
680
|
-
def
|
|
681
|
-
"""Image classification via
|
|
682
|
-
subscription, no API key).
|
|
674
|
+
def _call_agent_image(base_text, encoded, media_type):
|
|
675
|
+
"""Image classification via a cat-claws multimodal adapter
|
|
676
|
+
(claude-agent or codex-agent: subscription login, no API key).
|
|
677
|
+
Returns (reply, error) like _call_anthropic."""
|
|
678
|
+
from ._providers import _AGENT_BACKENDS
|
|
679
|
+
adapter_name, install_hint = _AGENT_BACKENDS[model_source]
|
|
683
680
|
try:
|
|
684
681
|
from catclaws._adapters import get_adapter
|
|
685
682
|
except ImportError:
|
|
686
|
-
return None,
|
|
683
|
+
return None, f"cat-claws is not installed. Run: {install_hint}"
|
|
687
684
|
import asyncio
|
|
688
|
-
adapter = get_adapter(
|
|
685
|
+
adapter = get_adapter(adapter_name)
|
|
689
686
|
_system = ("You are an image classification engine. Follow the user's "
|
|
690
687
|
"instructions exactly and reply with only what they ask for.")
|
|
691
688
|
try:
|
|
@@ -738,9 +735,11 @@ Provide the final categorization in the same JSON format:"""
|
|
|
738
735
|
image_content = {"type": "image_url", "image_url": {"url": encoded_image, "detail": "high"}}
|
|
739
736
|
return _call_mistral(prompt, step2_prompt, step3_prompt, step4_prompt, image_content)
|
|
740
737
|
|
|
741
|
-
elif model_source
|
|
738
|
+
elif model_source in ("claude-agent", "codex-agent"):
|
|
742
739
|
media_type = f"image/{ext}" if ext else "image/jpeg"
|
|
743
|
-
|
|
740
|
+
if media_type == "image/jpg":
|
|
741
|
+
media_type = "image/jpeg"
|
|
742
|
+
return _call_agent_image(base_prompt_text, encoded, media_type)
|
|
744
743
|
|
|
745
744
|
else:
|
|
746
745
|
raise ValueError("Unknown source! Choose from OpenAI, Anthropic, Perplexity, Google, xAI, Huggingface, or Mistral")
|
|
@@ -399,12 +399,6 @@ def pdf_multi_class(
|
|
|
399
399
|
"(the text-only CLI shim). Use model_source='claude-agent' (the cat-claws "
|
|
400
400
|
"subscription backend) or an API-key provider."
|
|
401
401
|
)
|
|
402
|
-
if model_source == "codex-agent":
|
|
403
|
-
raise ValueError(
|
|
404
|
-
"PDF classification is not yet supported with "
|
|
405
|
-
"model_source='codex-agent'. Use model_source='claude-agent' (the "
|
|
406
|
-
"multimodal subscription backend) or an API-key provider."
|
|
407
|
-
)
|
|
408
402
|
|
|
409
403
|
# Providers with native PDF support (only used in image/both modes)
|
|
410
404
|
native_pdf_providers = {"anthropic", "google"}
|
|
@@ -1192,23 +1186,27 @@ Provide the final categorization in the same JSON format:"""
|
|
|
1192
1186
|
|
|
1193
1187
|
return """{"1":"e"}""", "Max retries exceeded"
|
|
1194
1188
|
|
|
1195
|
-
def
|
|
1196
|
-
"""PDF-page classification via
|
|
1197
|
-
|
|
1198
|
-
|
|
1189
|
+
def _call_agent_pdf(base_text, encoded_image):
|
|
1190
|
+
"""PDF-page classification via a cat-claws adapter (claude-agent or
|
|
1191
|
+
codex-agent: subscription login, no API key). The page is rendered to
|
|
1192
|
+
an image (PDF-as-images); encoded_image=None sends text only (mode
|
|
1193
|
+
"text"). Returns (reply, error)."""
|
|
1194
|
+
from ._providers import _AGENT_BACKENDS
|
|
1195
|
+
adapter_name, install_hint = _AGENT_BACKENDS[model_source]
|
|
1199
1196
|
try:
|
|
1200
1197
|
from catclaws._adapters import get_adapter
|
|
1201
1198
|
except ImportError:
|
|
1202
|
-
return None,
|
|
1199
|
+
return None, f"cat-claws is not installed. Run: {install_hint}"
|
|
1203
1200
|
import asyncio
|
|
1204
|
-
adapter = get_adapter(
|
|
1201
|
+
adapter = get_adapter(adapter_name)
|
|
1205
1202
|
_system = ("You are a document-page classification engine. Follow the "
|
|
1206
1203
|
"user's instructions exactly and reply with only what they ask for.")
|
|
1207
1204
|
try:
|
|
1208
1205
|
reply, error = asyncio.run(adapter.one_shot(
|
|
1209
1206
|
base_text, system_prompt=_system, model=user_model,
|
|
1210
1207
|
thinking_budget=thinking_budget or 0,
|
|
1211
|
-
images
|
|
1208
|
+
**({"images": [{"media_type": "image/png", "data": encoded_image}]}
|
|
1209
|
+
if encoded_image else {}),
|
|
1212
1210
|
))
|
|
1213
1211
|
return (None, error) if error else (reply, None)
|
|
1214
1212
|
except Exception as e:
|
|
@@ -1244,6 +1242,8 @@ Provide the final categorization in the same JSON format:"""
|
|
|
1244
1242
|
return _call_openai_text_only(base_prompt_text, step2_prompt, step3_prompt, step4_prompt)
|
|
1245
1243
|
elif model_source == "mistral":
|
|
1246
1244
|
return _call_mistral_text_only(base_prompt_text, step2_prompt, step3_prompt, step4_prompt)
|
|
1245
|
+
elif model_source in ("claude-agent", "codex-agent"):
|
|
1246
|
+
return _call_agent_pdf(base_prompt_text, None)
|
|
1247
1247
|
else:
|
|
1248
1248
|
raise ValueError(f"Unknown source! Choose from OpenAI, Anthropic, Perplexity, Google, xAI, Huggingface, or Mistral")
|
|
1249
1249
|
|
|
@@ -1293,12 +1293,12 @@ Provide the final categorization in the same JSON format:"""
|
|
|
1293
1293
|
prompt_data = _build_prompt_google_pdf(encoded_pdf, base_prompt_text)
|
|
1294
1294
|
return _call_google(prompt_data, step2_prompt, step3_prompt, step4_prompt, base_prompt_text)
|
|
1295
1295
|
|
|
1296
|
-
elif model_source
|
|
1296
|
+
elif model_source in ("claude-agent", "codex-agent"):
|
|
1297
1297
|
image_bytes, is_valid = _extract_page_as_image_bytes(pdf_path, page_index)
|
|
1298
1298
|
if not is_valid:
|
|
1299
1299
|
return None, "Failed to render PDF page to image"
|
|
1300
1300
|
encoded_image = _encode_bytes_to_base64(image_bytes)
|
|
1301
|
-
return
|
|
1301
|
+
return _call_agent_pdf(base_prompt_text, encoded_image)
|
|
1302
1302
|
|
|
1303
1303
|
# Handle providers requiring image conversion
|
|
1304
1304
|
else:
|
|
@@ -3863,6 +3863,17 @@ multi_class_ensemble = classify_ensemble
|
|
|
3863
3863
|
# Summarization helpers
|
|
3864
3864
|
# =============================================================================
|
|
3865
3865
|
|
|
3866
|
+
def _format_item_errors(errors: dict, multi_model: bool) -> str:
|
|
3867
|
+
"""One row's failure reasons for the `error_message` column: the bare
|
|
3868
|
+
message for a single model, "model: message" pairs for an ensemble.
|
|
3869
|
+
Empty string when the row had no errors."""
|
|
3870
|
+
if not errors:
|
|
3871
|
+
return ""
|
|
3872
|
+
if not multi_model:
|
|
3873
|
+
return "; ".join(str(e) for e in errors.values())
|
|
3874
|
+
return "; ".join(f"{m}: {e}" for m, e in errors.items())
|
|
3875
|
+
|
|
3876
|
+
|
|
3866
3877
|
def _save_partial_summarize_results(all_results, model_configs, model_names, is_pdf_mode, filename, save_directory):
|
|
3867
3878
|
"""Save partial summarization results to CSV for safety/incremental saves."""
|
|
3868
3879
|
rows = []
|
|
@@ -3895,6 +3906,7 @@ def _save_partial_summarize_results(all_results, model_configs, model_names, is_
|
|
|
3895
3906
|
) else "partial"
|
|
3896
3907
|
else:
|
|
3897
3908
|
row["processing_status"] = "success"
|
|
3909
|
+
row["error_message"] = _format_item_errors(entry["errors"], len(model_configs) > 1)
|
|
3898
3910
|
|
|
3899
3911
|
rows.append(row)
|
|
3900
3912
|
|
|
@@ -4511,6 +4523,9 @@ def summarize_ensemble(
|
|
|
4511
4523
|
|
|
4512
4524
|
if error and error != "skipped":
|
|
4513
4525
|
still_failed.append((idx, model_name))
|
|
4526
|
+
# Keep the latest reason so the output reports what the
|
|
4527
|
+
# final attempt actually hit.
|
|
4528
|
+
all_results[idx]["errors"][model_name] = error
|
|
4514
4529
|
else:
|
|
4515
4530
|
# Update the stored result
|
|
4516
4531
|
all_results[idx]["model_results"][model_name] = json_result
|
|
@@ -4631,10 +4646,20 @@ def summarize_ensemble(
|
|
|
4631
4646
|
else:
|
|
4632
4647
|
row["processing_status"] = "success"
|
|
4633
4648
|
|
|
4649
|
+
# Why a row failed: previously collected per item but never written
|
|
4650
|
+
# out, so failures were silent (status "error", no reason).
|
|
4651
|
+
row["error_message"] = _format_item_errors(entry["errors"], len(model_configs) > 1)
|
|
4652
|
+
|
|
4634
4653
|
rows.append(row)
|
|
4635
4654
|
|
|
4636
4655
|
df = pd.DataFrame(rows)
|
|
4637
4656
|
|
|
4657
|
+
n_failed = int((df["error_message"] != "").sum()) if "error_message" in df else 0
|
|
4658
|
+
if n_failed:
|
|
4659
|
+
first = df.loc[df["error_message"] != "", "error_message"].iloc[0]
|
|
4660
|
+
print(f"\n[CatLLM] WARNING: {n_failed} of {len(df)} item(s) had errors "
|
|
4661
|
+
f"(see the error_message column). First error: {first}")
|
|
4662
|
+
|
|
4638
4663
|
# Save to file if requested
|
|
4639
4664
|
if filename:
|
|
4640
4665
|
save_path = os.path.join(save_directory, filename) if save_directory else filename
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|