cat-stack 2.5.2__tar.gz → 2.5.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. {cat_stack-2.5.2 → cat_stack-2.5.4}/PKG-INFO +4 -4
  2. {cat_stack-2.5.2 → cat_stack-2.5.4}/README.md +1 -1
  3. {cat_stack-2.5.2 → cat_stack-2.5.4}/pyproject.toml +2 -2
  4. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/__about__.py +1 -1
  5. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/_providers.py +18 -0
  6. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/image_functions.py +2 -0
  7. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/pdf_functions.py +2 -0
  8. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/text_functions_ensemble.py +61 -11
  9. {cat_stack-2.5.2 → cat_stack-2.5.4}/.gitignore +0 -0
  10. {cat_stack-2.5.2 → cat_stack-2.5.4}/LICENSE +0 -0
  11. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/cat_stack/__init__.py +0 -0
  12. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/__init__.py +0 -0
  13. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/_batch.py +0 -0
  14. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/_category_analysis.py +0 -0
  15. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/_chunked.py +0 -0
  16. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/_embeddings.py +0 -0
  17. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/_formatter.py +0 -0
  18. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/_pilot_test.py +0 -0
  19. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/_prompts.py +0 -0
  20. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/_review_ui.py +0 -0
  21. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/_tiebreaker.py +0 -0
  22. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/_utils.py +0 -0
  23. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/_web_fetch.py +0 -0
  24. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/_wrapper_helpers.py +0 -0
  25. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/calls/CoVe.py +0 -0
  26. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/calls/__init__.py +0 -0
  27. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/calls/image_CoVe.py +0 -0
  28. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/calls/image_stepback.py +0 -0
  29. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/calls/pdf_CoVe.py +0 -0
  30. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/calls/pdf_stepback.py +0 -0
  31. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/calls/stepback.py +0 -0
  32. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/calls/top_n.py +0 -0
  33. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/classify.py +0 -0
  34. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/collapse_themes.py +0 -0
  35. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/explore.py +0 -0
  36. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/extract.py +0 -0
  37. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/images/circle.png +0 -0
  38. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/images/cube.png +0 -0
  39. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/images/diamond.png +0 -0
  40. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/images/overlapping_pentagons.png +0 -0
  41. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/images/rectangles.png +0 -0
  42. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/model_reference_list.py +0 -0
  43. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/prompt_tune.py +0 -0
  44. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/summarize.py +0 -0
  45. {cat_stack-2.5.2 → cat_stack-2.5.4}/src/catstack/text_functions.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: cat-stack
3
- Version: 2.5.2
3
+ Version: 2.5.4
4
4
  Summary: Domain-agnostic text, image, PDF, and DOCX classification engine powered by LLMs
5
5
  Project-URL: Documentation, https://github.com/chrissoria/cat-stack#readme
6
6
  Project-URL: Issues, https://github.com/chrissoria/cat-stack/issues
@@ -24,9 +24,9 @@ Requires-Dist: pandas
24
24
  Requires-Dist: requests
25
25
  Requires-Dist: tqdm
26
26
  Provides-Extra: agent
27
- Requires-Dist: cat-claws[claude]>=0.3.2; extra == 'agent'
27
+ Requires-Dist: cat-claws[claude]>=0.3.3; extra == 'agent'
28
28
  Provides-Extra: codex-agent
29
- Requires-Dist: cat-claws[codex]>=0.3.2; extra == 'codex-agent'
29
+ Requires-Dist: cat-claws[codex]>=0.3.3; extra == 'codex-agent'
30
30
  Provides-Extra: docx
31
31
  Requires-Dist: python-docx>=1.0.0; extra == 'docx'
32
32
  Provides-Extra: embeddings
@@ -239,7 +239,7 @@ OpenAI, Anthropic, Google (Gemini), Mistral, Perplexity, xAI (Grok), HuggingFace
239
239
 
240
240
  All providers use the same `(model_name, provider, api_key)` tuple format. Provider is auto-detected from model name if omitted.
241
241
 
242
- **Subscription backends (no API key).** Three `model_source` values authenticate through a chat subscription instead of a metered key — leave `api_key` unset: `"claude-agent"` (Claude subscription via the Agent SDK; `pip install "cat-stack[agent]"`), `"claude-code"` (the Claude Code CLI, if installed — no extra needed), and `"codex-agent"` (ChatGPT subscription; `pip install "cat-stack[codex-agent]"`). Classification, extraction, exploration, and summarization all route through them.
242
+ **Subscription backends (no API key).** Three `model_source` values authenticate through a chat subscription instead of a metered key — leave `api_key` unset: `"claude-agent"` (Claude subscription via the Agent SDK; `pip install "cat-stack[agent]"`), `"claude-code"` (the Claude Code CLI, if installed — no extra needed), and `"codex-agent"` (ChatGPT subscription; `pip install "cat-stack[codex-agent]"`). Classification, extraction, exploration, and summarization all route through them, including images and PDF pages on the two agent backends. They use the agent CLI's own sign-in (separate from the Claude desktop app's): runs check it up front and, if you're signed out, open a single browser window to sign in once, then continue. See the [cat-claws README](https://github.com/chrissoria/cat-agent#signing-in-once).
243
243
 
244
244
  **Reproducibility note:** the same nominal model can behave differently across access routes. In a seed-matched benchmark (Claude Sonnet 5, temperature 0), extraction through the Agent SDK produced more varied label phrasings than the direct API (89% vs. 74% unique raw labels), while final consolidated taxonomies were equivalent. If raw label counts matter to your analysis, record the access route alongside the model version.
245
245
 
@@ -198,7 +198,7 @@ OpenAI, Anthropic, Google (Gemini), Mistral, Perplexity, xAI (Grok), HuggingFace
198
198
 
199
199
  All providers use the same `(model_name, provider, api_key)` tuple format. Provider is auto-detected from model name if omitted.
200
200
 
201
- **Subscription backends (no API key).** Three `model_source` values authenticate through a chat subscription instead of a metered key — leave `api_key` unset: `"claude-agent"` (Claude subscription via the Agent SDK; `pip install "cat-stack[agent]"`), `"claude-code"` (the Claude Code CLI, if installed — no extra needed), and `"codex-agent"` (ChatGPT subscription; `pip install "cat-stack[codex-agent]"`). Classification, extraction, exploration, and summarization all route through them.
201
+ **Subscription backends (no API key).** Three `model_source` values authenticate through a chat subscription instead of a metered key — leave `api_key` unset: `"claude-agent"` (Claude subscription via the Agent SDK; `pip install "cat-stack[agent]"`), `"claude-code"` (the Claude Code CLI, if installed — no extra needed), and `"codex-agent"` (ChatGPT subscription; `pip install "cat-stack[codex-agent]"`). Classification, extraction, exploration, and summarization all route through them, including images and PDF pages on the two agent backends. They use the agent CLI's own sign-in (separate from the Claude desktop app's): runs check it up front and, if you're signed out, open a single browser window to sign in once, then continue. See the [cat-claws README](https://github.com/chrissoria/cat-agent#signing-in-once).
202
202
 
203
203
  **Reproducibility note:** the same nominal model can behave differently across access routes. In a seed-matched benchmark (Claude Sonnet 5, temperature 0), extraction through the Agent SDK produced more varied label phrasings than the direct API (89% vs. 74% unique raw labels), while final consolidated taxonomies were equivalent. If raw label counts matter to your analysis, record the access route alongside the model version.
204
204
 
@@ -42,8 +42,8 @@ embeddings = ["sentence-transformers>=2.2.0"]
42
42
  # `agent` keeps its historical meaning (the Claude backend) so every shipped
43
43
  # install hint stays true; `codex-agent` matches the provider string so the
44
44
  # error-message hint is copy-pasteable.
45
- agent = ["cat-claws[claude]>=0.3.2"]
46
- codex-agent = ["cat-claws[codex]>=0.3.2"]
45
+ agent = ["cat-claws[claude]>=0.3.3"]
46
+ codex-agent = ["cat-claws[codex]>=0.3.3"]
47
47
 
48
48
  [project.urls]
49
49
  Documentation = "https://github.com/chrissoria/cat-stack#readme"
@@ -1,7 +1,7 @@
1
1
  # SPDX-FileCopyrightText: 2025-present Christopher Soria <chrissoria@berkeley.edu>
2
2
  #
3
3
  # SPDX-License-Identifier: GPL-3.0-or-later
4
- __version__ = "2.5.2"
4
+ __version__ = "2.5.4"
5
5
  __author__ = "Chris Soria"
6
6
  __email__ = "chrissoria@berkeley.edu"
7
7
  __title__ = "cat-stack"
@@ -790,6 +790,24 @@ def _split_agent_content(content):
790
790
  return "\n\n".join(t for t in texts if t), images
791
791
 
792
792
 
793
+ def _require_agent_sign_in(provider):
794
+ """Preflight for the cat-claws subscription backends: stop before any row
795
+ runs when the agent CLI is signed out, instead of every row failing on
796
+ "not logged in". cat-claws >= 0.3.3 checks with a cheap status call (no
797
+ model call); if signed out it opens ONE browser sign-in automatically and
798
+ continues, or, where no browser can open, raises NotSignedInError with
799
+ instructions worded for the Claude app or a terminal. No-op for other
800
+ providers, and when cat-claws is missing or older.
801
+ """
802
+ if provider not in _AGENT_BACKENDS:
803
+ return
804
+ try:
805
+ from catclaws import ensure_signed_in
806
+ except ImportError:
807
+ return
808
+ ensure_signed_in(_AGENT_BACKENDS[provider][0])
809
+
810
+
793
811
  def _require_http_provider(model_source, feature):
794
812
  """Raise a clear error when an HTTP-only feature is used with a
795
813
  subscription/CLI provider (claude-code / claude-agent / codex-agent)."""
@@ -154,6 +154,8 @@ def image_multi_class(
154
154
  "(the text-only CLI shim). Use model_source='claude-agent' (the cat-claws "
155
155
  "subscription backend) or an API-key provider."
156
156
  )
157
+ from ._providers import _require_agent_sign_in
158
+ _require_agent_sign_in(model_source)
157
159
 
158
160
  image_files = _load_image_files(image_input)
159
161
 
@@ -399,6 +399,8 @@ def pdf_multi_class(
399
399
  "(the text-only CLI shim). Use model_source='claude-agent' (the cat-claws "
400
400
  "subscription backend) or an API-key provider."
401
401
  )
402
+ from ._providers import _require_agent_sign_in
403
+ _require_agent_sign_in(model_source)
402
404
 
403
405
  # Providers with native PDF support (only used in image/both modes)
404
406
  native_pdf_providers = {"anthropic", "google"}
@@ -53,6 +53,7 @@ from concurrent.futures import ThreadPoolExecutor, as_completed
53
53
  from typing import Optional, Callable, Union
54
54
 
55
55
  from ._utils import _extract_balanced_json
56
+ from ._providers import _require_agent_sign_in
56
57
  from .text_functions import (
57
58
  UnifiedLLMClient,
58
59
  detect_provider,
@@ -684,6 +685,7 @@ def prepare_model_configs(
684
685
  "Install: pip install cat-stack[agent]\n"
685
686
  + "="*60
686
687
  )
688
+ _require_agent_sign_in(detected_provider)
687
689
  elif detected_provider == "codex-agent":
688
690
  try:
689
691
  import catclaws # noqa: F401
@@ -696,6 +698,7 @@ def prepare_model_configs(
696
698
  'Install: pip install "cat-stack[codex-agent]"\n'
697
699
  + "="*60
698
700
  )
701
+ _require_agent_sign_in(detected_provider)
699
702
  else:
700
703
  # Validate API key exists for cloud providers
701
704
  if not api_key:
@@ -1487,6 +1490,53 @@ def _extract_json_for_summary(reply: str) -> str:
1487
1490
  return '{"summary": ""}'
1488
1491
 
1489
1492
 
1493
+ _FENCE_RE = re.compile(r"^\s*```(?:json)?\s*|\s*```\s*$", re.IGNORECASE)
1494
+ _THINK_RE = re.compile(r"<think>.*?</think>", re.DOTALL)
1495
+ # `"summary": "` ... last `"` ... then only closing braces / a fence
1496
+ _LENIENT_SUMMARY_RE = re.compile(r'"summary"\s*:\s*"(.*)"\s*\}*\s*(?:```)?\s*$', re.DOTALL)
1497
+
1498
+
1499
+ def _parse_summary_reply(reply) -> tuple:
1500
+ """Read a summary out of a model reply as tolerantly as is safe.
1501
+
1502
+ Returns ``(json_str, error)``: ``json_str`` is a ``{"summary": ...}``
1503
+ object and ``error`` is None, or ``error`` explains why no summary could
1504
+ be read (so the row is retried and the reason reaches `error_message`,
1505
+ instead of the row failing silently as it used to).
1506
+
1507
+ Tried in order:
1508
+ 1. strict: the first JSON object in the reply (code fences and
1509
+ <think> blocks are fine);
1510
+ 2. lenient: a `"summary": "..."` object that is not valid JSON, most
1511
+ often unescaped double quotes or raw newlines inside the text
1512
+ (common with quoted category labels or non-English text);
1513
+ 3. prose: a reply with no JSON object at all is taken as the summary
1514
+ itself (the model answered in plain text).
1515
+ """
1516
+ if reply is None or not str(reply).strip():
1517
+ return '{"summary": ""}', "the model returned an empty reply"
1518
+ text = _THINK_RE.sub("", str(reply)).strip()
1519
+
1520
+ json_str = _extract_json_for_summary(text)
1521
+ ok, _ = extract_summary_from_json(json_str)
1522
+ if ok:
1523
+ return json_str, None
1524
+
1525
+ m = _LENIENT_SUMMARY_RE.search(text)
1526
+ if m:
1527
+ value = m.group(1).replace('\\"', '"').replace("\\n", "\n").strip()
1528
+ if value:
1529
+ return json.dumps({"summary": value}, ensure_ascii=False), None
1530
+
1531
+ if "{" not in text:
1532
+ prose = _FENCE_RE.sub("", text).strip()
1533
+ if prose:
1534
+ return json.dumps({"summary": prose}, ensure_ascii=False), None
1535
+
1536
+ snippet = " ".join(text.split())[:200]
1537
+ return '{"summary": ""}', f"could not read a summary from the model's reply: {snippet!r}"
1538
+
1539
+
1490
1540
  def extract_summary_from_json(json_str: str) -> tuple:
1491
1541
  """
1492
1542
  Extract summary from JSON response.
@@ -4227,10 +4277,10 @@ def summarize_ensemble(
4227
4277
  if error:
4228
4278
  return (model_name, '{"summary": ""}', error)
4229
4279
 
4230
- # Extract JSON from response
4231
- json_str = _extract_json_for_summary(response)
4232
-
4233
- return (model_name, json_str, None)
4280
+ # Read the summary; an unreadable reply is an error (retried,
4281
+ # reported), never a silent empty summary
4282
+ json_str, parse_error = _parse_summary_reply(response)
4283
+ return (model_name, json_str, parse_error)
4234
4284
 
4235
4285
  except Exception as e:
4236
4286
  error_msg = str(e)
@@ -4291,8 +4341,8 @@ def summarize_ensemble(
4291
4341
  if error:
4292
4342
  return (model_name, '{"summary": ""}', error)
4293
4343
 
4294
- json_str = _extract_json_for_summary(response)
4295
- return (model_name, json_str, None)
4344
+ json_str, parse_error = _parse_summary_reply(response)
4345
+ return (model_name, json_str, parse_error)
4296
4346
 
4297
4347
  except Exception as e:
4298
4348
  return (model_name, '{"summary": ""}', str(e))
@@ -4343,10 +4393,10 @@ def summarize_ensemble(
4343
4393
  if error:
4344
4394
  return (model_name, '{"summary": ""}', error)
4345
4395
 
4346
- # Extract JSON from response
4347
- json_str = _extract_json_for_summary(response)
4348
-
4349
- return (model_name, json_str, None)
4396
+ # Read the summary; an unreadable reply is an error (retried,
4397
+ # reported), never a silent empty summary
4398
+ json_str, parse_error = _parse_summary_reply(response)
4399
+ return (model_name, json_str, parse_error)
4350
4400
 
4351
4401
  except Exception as e:
4352
4402
  error_msg = str(e)
@@ -4732,7 +4782,7 @@ Provide your answer in JSON format: {{"summary": "your synthesized summary"}}"""
4732
4782
  max_retries=max_retries,
4733
4783
  )
4734
4784
 
4735
- json_str = _extract_json_for_summary(response)
4785
+ json_str, _ = _parse_summary_reply(response)
4736
4786
  is_valid, summary = extract_summary_from_json(json_str)
4737
4787
 
4738
4788
  if is_valid:
File without changes
File without changes