cat-stack 2.5.0__tar.gz → 2.5.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. {cat_stack-2.5.0 → cat_stack-2.5.1}/PKG-INFO +9 -5
  2. {cat_stack-2.5.0 → cat_stack-2.5.1}/README.md +6 -2
  3. {cat_stack-2.5.0 → cat_stack-2.5.1}/pyproject.toml +2 -2
  4. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/__about__.py +1 -1
  5. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/collapse_themes.py +33 -5
  6. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/explore.py +4 -2
  7. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/extract.py +4 -2
  8. {cat_stack-2.5.0 → cat_stack-2.5.1}/.gitignore +0 -0
  9. {cat_stack-2.5.0 → cat_stack-2.5.1}/LICENSE +0 -0
  10. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/cat_stack/__init__.py +0 -0
  11. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/__init__.py +0 -0
  12. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/_batch.py +0 -0
  13. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/_category_analysis.py +0 -0
  14. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/_chunked.py +0 -0
  15. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/_embeddings.py +0 -0
  16. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/_formatter.py +0 -0
  17. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/_pilot_test.py +0 -0
  18. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/_prompts.py +0 -0
  19. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/_providers.py +0 -0
  20. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/_review_ui.py +0 -0
  21. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/_tiebreaker.py +0 -0
  22. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/_utils.py +0 -0
  23. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/_web_fetch.py +0 -0
  24. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/_wrapper_helpers.py +0 -0
  25. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/calls/CoVe.py +0 -0
  26. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/calls/__init__.py +0 -0
  27. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/calls/image_CoVe.py +0 -0
  28. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/calls/image_stepback.py +0 -0
  29. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/calls/pdf_CoVe.py +0 -0
  30. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/calls/pdf_stepback.py +0 -0
  31. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/calls/stepback.py +0 -0
  32. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/calls/top_n.py +0 -0
  33. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/classify.py +0 -0
  34. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/image_functions.py +0 -0
  35. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/images/circle.png +0 -0
  36. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/images/cube.png +0 -0
  37. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/images/diamond.png +0 -0
  38. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/images/overlapping_pentagons.png +0 -0
  39. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/images/rectangles.png +0 -0
  40. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/model_reference_list.py +0 -0
  41. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/pdf_functions.py +0 -0
  42. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/prompt_tune.py +0 -0
  43. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/summarize.py +0 -0
  44. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/text_functions.py +0 -0
  45. {cat_stack-2.5.0 → cat_stack-2.5.1}/src/catstack/text_functions_ensemble.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: cat-stack
3
- Version: 2.5.0
3
+ Version: 2.5.1
4
4
  Summary: Domain-agnostic text, image, PDF, and DOCX classification engine powered by LLMs
5
5
  Project-URL: Documentation, https://github.com/chrissoria/cat-stack#readme
6
6
  Project-URL: Issues, https://github.com/chrissoria/cat-stack/issues
@@ -24,9 +24,9 @@ Requires-Dist: pandas
24
24
  Requires-Dist: requests
25
25
  Requires-Dist: tqdm
26
26
  Provides-Extra: agent
27
- Requires-Dist: cat-claws[claude]>=0.3.0; extra == 'agent'
27
+ Requires-Dist: cat-claws[claude]>=0.3.1; extra == 'agent'
28
28
  Provides-Extra: codex-agent
29
- Requires-Dist: cat-claws[codex]>=0.3.0; extra == 'codex-agent'
29
+ Requires-Dist: cat-claws[codex]>=0.3.1; extra == 'codex-agent'
30
30
  Provides-Extra: docx
31
31
  Requires-Dist: python-docx>=1.0.0; extra == 'docx'
32
32
  Provides-Extra: embeddings
@@ -138,7 +138,7 @@ cat.classify(
138
138
  ```
139
139
 
140
140
  ### `extract()`
141
- Discover categories from a corpus using LLM-driven exploration.
141
+ Discover categories from a corpus using LLM-driven exploration. Since v2.5.0, text consolidation runs the full explore → `collapse_themes()` pipeline (`engine="collapse"`, the default): the entire raw label inventory reaches the semantic merge — Jaro-Winkler dedup, embedding pre-merge, quality-controlled LLM passes, then a count-guided reduction to at most `max_categories`. Pass `engine="legacy"` to reproduce pre-2.5 runs (single merge call over a truncated inventory). `collapse_kwargs` forwards options to `collapse_themes()`; `max_workers` parallelizes both extraction and consolidation.
142
142
 
143
143
  ```python
144
144
  cat.extract(
@@ -197,7 +197,7 @@ cat.collapse_themes(
197
197
  | Parameter | Default | Description |
198
198
  | --- | --- | --- |
199
199
  | `input_data` | — | List of category labels, or a frequency `Series`/`dict` (`label -> count`). |
200
- | `api_key` | `None` | API key for the LLM provider (required). |
200
+ | `api_key` | `None` | API key for the LLM provider. Not required for subscription/CLI backends (`claude-code`, `claude-agent`, `codex-agent`) or `ollama`. |
201
201
  | `description` | `""` | The survey question or context, used in the merge prompt. |
202
202
  | `passes` | `1` | Number of merge iterations, or `"auto"` to iterate until the embedding-quality benchmark peaks. |
203
203
  | `max_passes` | `10` | Cap on iterations when `passes="auto"`. |
@@ -207,6 +207,8 @@ cat.collapse_themes(
207
207
  | `embedding_merge_threshold` | `0.92` | Cosine similarity at/above which labels are merged in the pre-LLM embedding step. `None`/`>=1.0` disables it. |
208
208
  | `shuffle` | `True` | Randomize order each pass so batch composition varies (improves convergence stability). |
209
209
  | `final_consolidation` | `0.82` | Cosine threshold for one greedy global embedding re-merge after all passes, collapsing cross-batch duplicates. Conservative by design (errs toward keeping categories). `False`/`None` skips it. |
210
+ | `top_n` | `None` | If set, a final global LLM call consolidates the surviving list into at most N categories, guided by each label's generation count (frequent themes favored; overlapping labels merged rather than dropped). Guaranteed `<= top_n` via truncation plus a deterministic top-N-by-count fallback. |
211
+ | `prune` | `False` | `True` = drop conceptual duplicates keeping one representative verbatim; never renames or merges merely-related labels. |
210
212
  | `user_model` | `"gpt-4o"` | Model for the merge phase. Use a capable model — small models can degenerate. |
211
213
  | `model_source` | `"auto"` | Provider for `user_model` (`"auto"`, `"openai"`, `"huggingface"`, …). |
212
214
  | `unique_model` | `None` | If set, run an initial extract-unique thinning phase on this (typically cheaper) model before the merge phase. `None` skips the phase (backward compatible). |
@@ -237,6 +239,8 @@ OpenAI, Anthropic, Google (Gemini), Mistral, Perplexity, xAI (Grok), HuggingFace
237
239
 
238
240
  All providers use the same `(model_name, provider, api_key)` tuple format. Provider is auto-detected from model name if omitted.
239
241
 
242
+ **Subscription backends (no API key).** Three `model_source` values authenticate through a chat subscription instead of a metered key — leave `api_key` unset: `"claude-agent"` (Claude subscription via the Agent SDK; `pip install "cat-stack[agent]"`), `"claude-code"` (the Claude Code CLI, if installed — no extra needed), and `"codex-agent"` (ChatGPT subscription; `pip install "cat-stack[codex-agent]"`). Classification, extraction, exploration, and summarization all route through them.
243
+
240
244
  ## Features
241
245
 
242
246
  - **Automatic prompt optimization** (`prompt_tune`) — correct a small sample in a browser UI, and the system generates per-category instructions that improve accuracy
@@ -97,7 +97,7 @@ cat.classify(
97
97
  ```
98
98
 
99
99
  ### `extract()`
100
- Discover categories from a corpus using LLM-driven exploration.
100
+ Discover categories from a corpus using LLM-driven exploration. Since v2.5.0, text consolidation runs the full explore → `collapse_themes()` pipeline (`engine="collapse"`, the default): the entire raw label inventory reaches the semantic merge — Jaro-Winkler dedup, embedding pre-merge, quality-controlled LLM passes, then a count-guided reduction to at most `max_categories`. Pass `engine="legacy"` to reproduce pre-2.5 runs (single merge call over a truncated inventory). `collapse_kwargs` forwards options to `collapse_themes()`; `max_workers` parallelizes both extraction and consolidation.
101
101
 
102
102
  ```python
103
103
  cat.extract(
@@ -156,7 +156,7 @@ cat.collapse_themes(
156
156
  | Parameter | Default | Description |
157
157
  | --- | --- | --- |
158
158
  | `input_data` | — | List of category labels, or a frequency `Series`/`dict` (`label -> count`). |
159
- | `api_key` | `None` | API key for the LLM provider (required). |
159
+ | `api_key` | `None` | API key for the LLM provider. Not required for subscription/CLI backends (`claude-code`, `claude-agent`, `codex-agent`) or `ollama`. |
160
160
  | `description` | `""` | The survey question or context, used in the merge prompt. |
161
161
  | `passes` | `1` | Number of merge iterations, or `"auto"` to iterate until the embedding-quality benchmark peaks. |
162
162
  | `max_passes` | `10` | Cap on iterations when `passes="auto"`. |
@@ -166,6 +166,8 @@ cat.collapse_themes(
166
166
  | `embedding_merge_threshold` | `0.92` | Cosine similarity at/above which labels are merged in the pre-LLM embedding step. `None`/`>=1.0` disables it. |
167
167
  | `shuffle` | `True` | Randomize order each pass so batch composition varies (improves convergence stability). |
168
168
  | `final_consolidation` | `0.82` | Cosine threshold for one greedy global embedding re-merge after all passes, collapsing cross-batch duplicates. Conservative by design (errs toward keeping categories). `False`/`None` skips it. |
169
+ | `top_n` | `None` | If set, a final global LLM call consolidates the surviving list into at most N categories, guided by each label's generation count (frequent themes favored; overlapping labels merged rather than dropped). Guaranteed `<= top_n` via truncation plus a deterministic top-N-by-count fallback. |
170
+ | `prune` | `False` | `True` = drop conceptual duplicates keeping one representative verbatim; never renames or merges merely-related labels. |
169
171
  | `user_model` | `"gpt-4o"` | Model for the merge phase. Use a capable model — small models can degenerate. |
170
172
  | `model_source` | `"auto"` | Provider for `user_model` (`"auto"`, `"openai"`, `"huggingface"`, …). |
171
173
  | `unique_model` | `None` | If set, run an initial extract-unique thinning phase on this (typically cheaper) model before the merge phase. `None` skips the phase (backward compatible). |
@@ -196,6 +198,8 @@ OpenAI, Anthropic, Google (Gemini), Mistral, Perplexity, xAI (Grok), HuggingFace
196
198
 
197
199
  All providers use the same `(model_name, provider, api_key)` tuple format. Provider is auto-detected from model name if omitted.
198
200
 
201
+ **Subscription backends (no API key).** Three `model_source` values authenticate through a chat subscription instead of a metered key — leave `api_key` unset: `"claude-agent"` (Claude subscription via the Agent SDK; `pip install "cat-stack[agent]"`), `"claude-code"` (the Claude Code CLI, if installed — no extra needed), and `"codex-agent"` (ChatGPT subscription; `pip install "cat-stack[codex-agent]"`). Classification, extraction, exploration, and summarization all route through them.
202
+
199
203
  ## Features
200
204
 
201
205
  - **Automatic prompt optimization** (`prompt_tune`) — correct a small sample in a browser UI, and the system generates per-category instructions that improve accuracy
@@ -42,8 +42,8 @@ embeddings = ["sentence-transformers>=2.2.0"]
42
42
  # `agent` keeps its historical meaning (the Claude backend) so every shipped
43
43
  # install hint stays true; `codex-agent` matches the provider string so the
44
44
  # error-message hint is copy-pasteable.
45
- agent = ["cat-claws[claude]>=0.3.0"]
46
- codex-agent = ["cat-claws[codex]>=0.3.0"]
45
+ agent = ["cat-claws[claude]>=0.3.1"]
46
+ codex-agent = ["cat-claws[codex]>=0.3.1"]
47
47
 
48
48
  [project.urls]
49
49
  Documentation = "https://github.com/chrissoria/cat-stack#readme"
@@ -1,7 +1,7 @@
1
1
  # SPDX-FileCopyrightText: 2025-present Christopher Soria <chrissoria@berkeley.edu>
2
2
  #
3
3
  # SPDX-License-Identifier: GPL-3.0-or-later
4
- __version__ = "2.5.0"
4
+ __version__ = "2.5.1"
5
5
  __author__ = "Chris Soria"
6
6
  __email__ = "chrissoria@berkeley.edu"
7
7
  __title__ = "cat-stack"
@@ -25,7 +25,7 @@ import numpy as np
25
25
  import pandas as pd
26
26
  from jellyfish import jaro_winkler_similarity
27
27
 
28
- from ._providers import UnifiedLLMClient, detect_provider
28
+ from ._providers import UnifiedLLMClient, detect_provider, _SUBSCRIPTION_PROVIDERS
29
29
  from ._utils import _clean_label
30
30
 
31
31
  __all__ = [
@@ -301,6 +301,19 @@ def _count_guidance(current, input_data):
301
301
  return {lbl: max(int(c), orig.get(_norm_key(lbl), 0)) for lbl, c in cur.items()}
302
302
 
303
303
 
304
+ def _as_label_list(x):
305
+ """Enforce the documented list[str] return contract at exit points.
306
+
307
+ Dict / Series / DataFrame inputs can reach a return untouched (e.g.
308
+ passes=0 with a top_n no-op), which would leak the input container —
309
+ and, via the filename write, save counts instead of labels."""
310
+ if isinstance(x, list):
311
+ return [str(v) for v in x]
312
+ if isinstance(x, dict):
313
+ return [str(k) for k in x]
314
+ return [str(k) for k in _to_counts(x)]
315
+
316
+
304
317
  def _to_counts(input_data):
305
318
  """Coerce the accepted input forms into a {category: count} dict."""
306
319
  if isinstance(input_data, pd.DataFrame):
@@ -425,7 +438,9 @@ def collapse_themes(
425
438
  input_data: Themes to collapse. list[str] (duplicates allowed), pandas
426
439
  Series, dict {category: count}, or DataFrame with "category"
427
440
  [and optional "count"] columns.
428
- api_key (str): API key for the model provider.
441
+ api_key (str): API key for the model provider. Not required for
442
+ subscription/CLI backends (`claude-code`, `claude-agent`,
443
+ `codex-agent`) or `ollama`.
429
444
  description (str): Data/question context, injected into the prompt — e.g.
430
445
  the survey question the categories came from. Helps the model judge
431
446
  which distinctions matter.
@@ -505,9 +520,6 @@ def collapse_themes(
505
520
  ... aggressive=True, passes="auto", max_workers=8,
506
521
  ... )
507
522
  """
508
- if not api_key:
509
- raise ValueError("collapse_themes() needs an api_key for the LLM call.")
510
-
511
523
  mode = "merge" if aggressive else "unique"
512
524
 
513
525
  # The main (merge) phase runs on merge_model if given, else user_model. A separate
@@ -518,6 +530,20 @@ def collapse_themes(
518
530
  merge_name = merge_model or user_model
519
531
  merge_src = merge_model_source if merge_model else model_source
520
532
  merge_provider = detect_provider(merge_name, merge_src)
533
+
534
+ # A key is only needed for providers that bill one; the subscription/CLI
535
+ # backends and ollama run keyless.
536
+ _keyless = set(_SUBSCRIPTION_PROVIDERS) | {"ollama"}
537
+ _providers_used = {merge_provider}
538
+ if unique_model:
539
+ _providers_used.add(detect_provider(unique_model, unique_model_source))
540
+ if not api_key and not _providers_used <= _keyless:
541
+ raise ValueError(
542
+ "collapse_themes() needs an api_key for the LLM call. "
543
+ "(Not required for the claude-code/claude-agent/codex-agent "
544
+ "backends or ollama.)"
545
+ )
546
+
521
547
  client = UnifiedLLMClient(provider=merge_provider, api_key=api_key, model=merge_name)
522
548
 
523
549
  def _run(cl, items, md, p):
@@ -589,6 +615,7 @@ def collapse_themes(
589
615
  if top_n and len(items) > int(top_n):
590
616
  items = _select_top_n(client, _count_guidance(items, input_data),
591
617
  int(top_n), description, creativity, dedupe_threshold)
618
+ items = _as_label_list(items)
592
619
  if filename:
593
620
  pd.DataFrame({"category": items}).to_csv(filename, index=False)
594
621
  print(f"Collapsed categories saved to {filename}")
@@ -674,6 +701,7 @@ def collapse_themes(
674
701
  current = _select_top_n(client, _count_guidance(current, input_data),
675
702
  int(top_n), description, creativity, dedupe_threshold)
676
703
 
704
+ current = _as_label_list(current)
677
705
  if filename:
678
706
  pd.DataFrame({"category": current}).to_csv(filename, index=False)
679
707
  print(f"Collapsed categories saved to {filename}")
@@ -16,7 +16,7 @@ from .text_functions import explore_common_categories
16
16
 
17
17
  def explore(
18
18
  input_data,
19
- api_key,
19
+ api_key=None,
20
20
  description="",
21
21
  max_categories=12,
22
22
  categories_per_chunk=10,
@@ -46,7 +46,9 @@ def explore(
46
46
 
47
47
  Args:
48
48
  input_data: List of text responses or pandas Series.
49
- api_key (str): API key for the model provider.
49
+ api_key (str, optional): API key for the model provider. Not required
50
+ for subscription/CLI backends (claude-code, claude-agent,
51
+ codex-agent) or ollama.
50
52
  description (str): Description of the data context. Content-neutral —
51
53
  for survey responses this is the question that was asked; for
52
54
  documents or posts this describes what the content is about.
@@ -41,7 +41,7 @@ from .collapse_themes import collapse_themes
41
41
 
42
42
  def extract(
43
43
  input_data,
44
- api_key,
44
+ api_key=None,
45
45
  input_type="auto",
46
46
  description="",
47
47
  survey_question=None,
@@ -79,7 +79,9 @@ def extract(
79
79
  - For text: list of text responses or pandas Series
80
80
  - For image: directory path, single file, or list of image paths
81
81
  - For pdf: directory path, single file, or list of PDF paths
82
- api_key (str): API key for the model provider.
82
+ api_key (str, optional): API key for the model provider. Not required
83
+ for subscription/CLI backends (claude-code, claude-agent,
84
+ codex-agent) or ollama.
83
85
  input_type (str): Type of input data. Options:
84
86
  - "auto" (default): Auto-detect from file extensions
85
87
  - "text": Text responses
File without changes
File without changes