content-core 2.0.2__tar.gz → 2.0.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (113) hide show
  1. {content_core-2.0.2 → content_core-2.0.4}/CHANGELOG.md +13 -0
  2. {content_core-2.0.2 → content_core-2.0.4}/CLAUDE.md +32 -0
  3. {content_core-2.0.2 → content_core-2.0.4}/PKG-INFO +1 -1
  4. {content_core-2.0.2 → content_core-2.0.4}/SKILL.md +3 -2
  5. {content_core-2.0.2 → content_core-2.0.4}/pyproject.toml +1 -1
  6. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/__init__.py +4 -2
  7. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/cli.py +6 -0
  8. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/common/state.py +17 -0
  9. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/extraction.py +101 -19
  10. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/url/youtube.py +2 -0
  11. {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_routing.py +85 -2
  12. {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_youtube_parsing.py +12 -0
  13. {content_core-2.0.2 → content_core-2.0.4}/uv.lock +1 -1
  14. content_core-2.0.2/test_coverage_branch_report.md +0 -480
  15. {content_core-2.0.2 → content_core-2.0.4}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
  16. {content_core-2.0.2 → content_core-2.0.4}/.github/workflows/claude-code-review.yml +0 -0
  17. {content_core-2.0.2 → content_core-2.0.4}/.github/workflows/claude.yml +0 -0
  18. {content_core-2.0.2 → content_core-2.0.4}/.github/workflows/create-tag.yml +0 -0
  19. {content_core-2.0.2 → content_core-2.0.4}/.github/workflows/publish.yml +0 -0
  20. {content_core-2.0.2 → content_core-2.0.4}/.github/workflows/test.yml +0 -0
  21. {content_core-2.0.2 → content_core-2.0.4}/.gitignore +0 -0
  22. {content_core-2.0.2 → content_core-2.0.4}/.python-version +0 -0
  23. {content_core-2.0.2 → content_core-2.0.4}/CONTRIBUTING.md +0 -0
  24. {content_core-2.0.2 → content_core-2.0.4}/LICENSE +0 -0
  25. {content_core-2.0.2 → content_core-2.0.4}/Makefile +0 -0
  26. {content_core-2.0.2 → content_core-2.0.4}/README.md +0 -0
  27. {content_core-2.0.2 → content_core-2.0.4}/REFACTOR.md +0 -0
  28. {content_core-2.0.2 → content_core-2.0.4}/TEST_PLAN.md +0 -0
  29. {content_core-2.0.2 → content_core-2.0.4}/docs/cli.md +0 -0
  30. {content_core-2.0.2 → content_core-2.0.4}/docs/mcp.md +0 -0
  31. {content_core-2.0.2 → content_core-2.0.4}/docs/processors.md +0 -0
  32. {content_core-2.0.2 → content_core-2.0.4}/docs/usage.md +0 -0
  33. {content_core-2.0.2 → content_core-2.0.4}/examples/main.py +0 -0
  34. {content_core-2.0.2 → content_core-2.0.4}/prompts/content/cleanup.jinja +0 -0
  35. {content_core-2.0.2 → content_core-2.0.4}/prompts/content/summarize.jinja +0 -0
  36. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/common/__init__.py +0 -0
  37. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/common/exceptions.py +0 -0
  38. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/common/retry.py +0 -0
  39. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/common/types.py +0 -0
  40. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/config.py +0 -0
  41. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/content/__init__.py +0 -0
  42. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/content/extraction/__init__.py +0 -0
  43. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/content/identification/__init__.py +0 -0
  44. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/content/identification/file_detector.py +0 -0
  45. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/content/summary/__init__.py +0 -0
  46. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/content/summary/core.py +0 -0
  47. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/logging.py +0 -0
  48. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/mcp/__init__.py +0 -0
  49. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/mcp/server.py +0 -0
  50. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/models.py +0 -0
  51. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/document/__init__.py +0 -0
  52. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/document/docling.py +0 -0
  53. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/document/docx.py +0 -0
  54. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/document/epub.py +0 -0
  55. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/document/pdf.py +0 -0
  56. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/document/pptx.py +0 -0
  57. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/document/xlsx.py +0 -0
  58. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/media/__init__.py +0 -0
  59. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/media/audio.py +0 -0
  60. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/media/video.py +0 -0
  61. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/protocol.py +0 -0
  62. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/text.py +0 -0
  63. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/url/__init__.py +0 -0
  64. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/url/bs4.py +0 -0
  65. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/url/crawl4ai.py +0 -0
  66. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/url/firecrawl.py +0 -0
  67. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/url/jina.py +0 -0
  68. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/url/reddit.py +0 -0
  69. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/py.typed +0 -0
  70. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/templated_message.py +0 -0
  71. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/tools/__init__.py +0 -0
  72. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/tools/extract.py +0 -0
  73. {content_core-2.0.2 → content_core-2.0.4}/src/content_core/tools/summarize.py +0 -0
  74. {content_core-2.0.2 → content_core-2.0.4}/tests/conftest.py +0 -0
  75. {content_core-2.0.2 → content_core-2.0.4}/tests/e2e/__init__.py +0 -0
  76. {content_core-2.0.2 → content_core-2.0.4}/tests/e2e/test_docling.py +0 -0
  77. {content_core-2.0.2 → content_core-2.0.4}/tests/e2e/test_media.py +0 -0
  78. {content_core-2.0.2 → content_core-2.0.4}/tests/e2e/test_remote.py +0 -0
  79. {content_core-2.0.2 → content_core-2.0.4}/tests/e2e/test_url_engines.py +0 -0
  80. {content_core-2.0.2 → content_core-2.0.4}/tests/e2e/test_youtube.py +0 -0
  81. {content_core-2.0.2 → content_core-2.0.4}/tests/input_content/file.docx +0 -0
  82. {content_core-2.0.2 → content_core-2.0.4}/tests/input_content/file.epub +0 -0
  83. {content_core-2.0.2 → content_core-2.0.4}/tests/input_content/file.md +0 -0
  84. {content_core-2.0.2 → content_core-2.0.4}/tests/input_content/file.mp3 +0 -0
  85. {content_core-2.0.2 → content_core-2.0.4}/tests/input_content/file.mp4 +0 -0
  86. {content_core-2.0.2 → content_core-2.0.4}/tests/input_content/file.pdf +0 -0
  87. {content_core-2.0.2 → content_core-2.0.4}/tests/input_content/file.pptx +0 -0
  88. {content_core-2.0.2 → content_core-2.0.4}/tests/input_content/file.txt +0 -0
  89. {content_core-2.0.2 → content_core-2.0.4}/tests/input_content/file.xlsx +0 -0
  90. {content_core-2.0.2 → content_core-2.0.4}/tests/input_content/file_audio.mp3 +0 -0
  91. {content_core-2.0.2 → content_core-2.0.4}/tests/input_content/new_pdf.pdf +0 -0
  92. {content_core-2.0.2 → content_core-2.0.4}/tests/integration/conftest.py +0 -0
  93. {content_core-2.0.2 → content_core-2.0.4}/tests/integration/test_cli_v2.py +0 -0
  94. {content_core-2.0.2 → content_core-2.0.4}/tests/integration/test_extraction.py +0 -0
  95. {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_audio_concurrency.py +0 -0
  96. {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_cli.py +0 -0
  97. {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_config_file.py +0 -0
  98. {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_config_v2.py +0 -0
  99. {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_docling_extraction.py +0 -0
  100. {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_epub_extraction.py +0 -0
  101. {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_file_detector.py +0 -0
  102. {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_file_detector_critical.py +0 -0
  103. {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_file_detector_performance.py +0 -0
  104. {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_mcp_v2.py +0 -0
  105. {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_media_pipeline.py +0 -0
  106. {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_models_v2.py +0 -0
  107. {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_office_extraction.py +0 -0
  108. {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_pdf_extraction.py +0 -0
  109. {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_pdf_helpers.py +0 -0
  110. {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_reddit_extraction.py +0 -0
  111. {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_retry.py +0 -0
  112. {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_text_processing.py +0 -0
  113. {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_url_engine_select.py +0 -0
@@ -7,6 +7,19 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+ ## [2.0.4] - 2026-07-12
11
+
12
+ ### Added
13
+ - `check_file_support(file_path, config)` public API for cheap pre-flight validation of whether a file can be extracted, without running extraction; returns a `FileSupport` verdict
14
+
15
+ ### Fixed
16
+ - YouTube `youtube.com/live/<id>` and `youtube.com/shorts/<id>` URLs were not recognized, causing title and transcript extraction to fail
17
+
18
+ ## [2.0.3] - 2026-04-13
19
+
20
+ ### Added
21
+ - `--version` flag to CLI (`content-core --version`)
22
+
10
23
  ## [2.0.2] - 2026-04-13
11
24
 
12
25
  ### Fixed
@@ -21,6 +21,38 @@ content-core mcp
21
21
  content-core config list|set|delete
22
22
  ```
23
23
 
24
+ ## For Automated Agents
25
+
26
+ If you are an automated coding agent (harny, Claude Code in headless mode, etc.) working on this repo:
27
+
28
+ ### Install command
29
+
30
+ ```bash
31
+ uv sync --group dev
32
+ ```
33
+
34
+ ### Validator command
35
+
36
+ ```bash
37
+ uv run pytest tests/unit tests/integration -v
38
+ ```
39
+
40
+ This is the gate. It returns clean pass/fail in ~15s, requires no credentials or network, and matches what CI enforces on `main` (CI runs only `tests/unit`; integration tests are also safe locally — no network/API keys). For targeted feedback during iteration, use `uv run pytest -k "<keyword>"` per the table in the Testing section below.
41
+
42
+ ### Do NOT use as gates
43
+
44
+ - **`ruff check`** — available via `make ruff` but NOT enforced in CI. Including it as a validator gate would surface unaudited pre-existing findings unrelated to your task.
45
+ - **`mypy`** — not enforced in CI. No baseline configured. Same risk.
46
+ - **`make test-e2e` / `make test-e2e-heavy`** — require network access, API keys for LLM/STT providers, and large model downloads. Will fail in a sandboxed worktree.
47
+
48
+ If you believe a task genuinely requires lint or type cleanup, do it as a *separate task* with explicit scope, not as a side effect of an unrelated change.
49
+
50
+ ### Scope guidance
51
+
52
+ - Single processor changes (`processors/url/*.py`, `processors/document/*.py`, `processors/media/*.py`) are the natural unit of work — keep changes confined to one file plus its matching test in `tests/unit/`.
53
+ - Avoid touching `extraction.py` (orchestrator) or `config.py` unless the task explicitly targets routing or configuration.
54
+ - New processors should follow the `Processor` protocol in `src/content_core/processors/protocol.py`.
55
+
24
56
  ## Codebase Structure
25
57
 
26
58
  ```
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: content-core
3
- Version: 2.0.2
3
+ Version: 2.0.4
4
4
  Summary: Extract what matters from any media source. Available as Python Library, macOS Service, CLI and MCP Server
5
5
  Author-email: LUIS NOVO <lfnovo@gmail.com>
6
6
  License-File: LICENSE
@@ -152,8 +152,9 @@ If summarization fails with an API key error, fall back to `extract_content` and
152
152
 
153
153
  ## Guidelines
154
154
 
155
- - Prefer the MCP tools if available — they are async and more efficient than CLI calls
156
- - If MCP is not available, use the CLI via Bash with `uvx content-core`
155
+ - For small/medium content (articles, short pages): prefer MCP tools if available — they are async and more efficient
156
+ - For large content (long documents, full books, lengthy transcripts): prefer the CLI via Bash, redirecting output to a file (`uvx content-core extract "URL" > output.md`). This avoids flooding the agent's context window with large payloads. Read only the relevant sections from the file as needed.
157
+ - If MCP is not available, always use the CLI via Bash with `uvx content-core`
157
158
  - For URLs: extraction works without any API key
158
159
  - For audio/video: requires `OPENAI_API_KEY` (or another STT provider key)
159
160
  - For summarization: requires an LLM API key
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "content-core"
3
- version = "2.0.2"
3
+ version = "2.0.4"
4
4
  description = "Extract what matters from any media source. Available as Python Library, macOS Service, CLI and MCP Server"
5
5
  readme = "README.md"
6
6
  homepage = "https://github.com/lfnovo/content-core"
@@ -5,9 +5,9 @@ load_dotenv()
5
5
 
6
6
  from content_core.config import ContentCoreConfig
7
7
  from content_core.content.summary import summarize
8
- from content_core.extraction import extract_content
8
+ from content_core.extraction import check_file_support, extract_content
9
9
  from content_core.logging import configure_logging
10
- from content_core.common.state import ExtractionInput, ExtractionOutput
10
+ from content_core.common.state import ExtractionInput, ExtractionOutput, FileSupport
11
11
 
12
12
  # Convenience alias
13
13
  extract = extract_content
@@ -18,8 +18,10 @@ configure_logging(debug=False)
18
18
  __all__ = [
19
19
  "extract_content",
20
20
  "extract",
21
+ "check_file_support",
21
22
  "summarize",
22
23
  "ContentCoreConfig",
23
24
  "ExtractionInput",
24
25
  "ExtractionOutput",
26
+ "FileSupport",
25
27
  ]
@@ -7,7 +7,13 @@ import click
7
7
  from content_core.logging import configure_logging
8
8
 
9
9
 
10
+ def _get_version():
11
+ from importlib.metadata import version
12
+ return version("content-core")
13
+
14
+
10
15
  @click.group()
16
+ @click.version_option(package_name="content-core")
11
17
  @click.option("--debug", is_flag=True, help="Enable debug logging")
12
18
  def cli(debug):
13
19
  """Content Core — Extract and summarize content from any source."""
@@ -29,7 +29,24 @@ class ExtractionOutput(BaseModel):
29
29
  metadata: dict = Field(default_factory=dict)
30
30
 
31
31
 
32
+ class FileSupport(BaseModel):
33
+ """Verdict from a pre-flight file-support check.
34
+
35
+ Returned by ``check_file_support`` so callers can validate an upload
36
+ cheaply (identification + routing only, no extraction) before committing
37
+ to a full extraction job.
38
+ """
39
+
40
+ supported: bool
41
+ file_path: str
42
+ identified_type: str = "" # MIME type detected for the file
43
+ document_engine: str = "" # engine the verdict was computed for
44
+ processor: Optional[str] = None # processor that would handle it, if supported
45
+ reason: Optional[str] = None # human-readable explanation when unsupported
46
+
47
+
32
48
  __all__ = [
33
49
  "ExtractionInput",
34
50
  "ExtractionOutput",
51
+ "FileSupport",
35
52
  ]
@@ -11,7 +11,7 @@ from content_core.common.exceptions import InvalidInputError, UnsupportedTypeExc
11
11
  from content_core.common.retry import retry_download
12
12
  from content_core.config import ContentCoreConfig, get_default_config
13
13
  from content_core.logging import logger
14
- from content_core.common.state import ExtractionOutput
14
+ from content_core.common.state import ExtractionOutput, FileSupport
15
15
 
16
16
  # Import processor v2 functions
17
17
  from content_core.processors.media.audio import transcribe_audio
@@ -105,6 +105,90 @@ async def _extract_url(url: str, cfg: ContentCoreConfig) -> ExtractionOutput:
105
105
  return await extract_from_url(url, cfg)
106
106
 
107
107
 
108
+ def _route_for_mime(mime: str, cfg: ContentCoreConfig) -> str | None:
109
+ """Return the processor that would handle a file of this MIME type.
110
+
111
+ Single source of truth for file-support routing, shared by ``_extract_file``
112
+ (actual extraction) and ``check_file_support`` (pre-flight validation), so
113
+ the pre-flight answer can never disagree with what extraction really does.
114
+
115
+ Returns the processor name ("docling", "pdf", "epub", "office", "video",
116
+ "audio", "text") or ``None`` if the type is unsupported.
117
+ """
118
+ engine = cfg.document_engine
119
+ if engine == "docling" or (
120
+ engine == "auto" and DOCLING_AVAILABLE and mime in DOCLING_SUPPORTED
121
+ ):
122
+ if DOCLING_AVAILABLE and extract_docling is not None:
123
+ return "docling"
124
+
125
+ if mime in SUPPORTED_PDF_TYPES:
126
+ return "pdf"
127
+ if mime in SUPPORTED_EPUB_TYPES:
128
+ return "epub"
129
+ if mime in SUPPORTED_OFFICE_TYPES:
130
+ return "office"
131
+ if mime.startswith("video/"):
132
+ return "video"
133
+ if mime.startswith("audio/"):
134
+ return "audio"
135
+ if mime == "text/plain":
136
+ return "text"
137
+ return None
138
+
139
+
140
+ async def check_file_support(
141
+ file_path: str, config: ContentCoreConfig | None = None
142
+ ) -> FileSupport:
143
+ """Check whether content-core can extract a given file, without extracting.
144
+
145
+ Runs the same identification + routing that ``extract_content`` uses, but
146
+ stops before extraction, so the answer carries the same authority as a real
147
+ run at a fraction of the cost (it only reads the file header). "Unsupported"
148
+ is returned as a verdict rather than raised, since it is an expected outcome
149
+ at ingestion time.
150
+
151
+ Args:
152
+ file_path: Local file path to validate.
153
+ config: Optional config override. The ``document_engine`` setting is
154
+ honored, since supported types vary by engine.
155
+
156
+ Returns:
157
+ FileSupport verdict with ``supported``, the ``identified_type`` (MIME),
158
+ the ``processor`` that would handle it (when supported), and a ``reason``
159
+ when it would not.
160
+ """
161
+ from content_core.content.identification import get_file_type
162
+
163
+ cfg = config or get_default_config()
164
+
165
+ # Identification can fail outright for unrecognizable files -- that is itself
166
+ # an "unsupported" verdict (extract_content would raise here too), not an
167
+ # error the caller should have to catch.
168
+ try:
169
+ mime = await get_file_type(file_path)
170
+ except UnsupportedTypeException as exc:
171
+ return FileSupport(
172
+ supported=False,
173
+ file_path=file_path,
174
+ identified_type="",
175
+ document_engine=cfg.document_engine,
176
+ processor=None,
177
+ reason=str(exc),
178
+ )
179
+
180
+ processor = _route_for_mime(mime, cfg)
181
+ supported = processor is not None
182
+ return FileSupport(
183
+ supported=supported,
184
+ file_path=file_path,
185
+ identified_type=mime,
186
+ document_engine=cfg.document_engine,
187
+ processor=processor,
188
+ reason=None if supported else f"Unsupported file type: {mime}",
189
+ )
190
+
191
+
108
192
  async def _extract_file(
109
193
  path: str, cfg: ContentCoreConfig, delete_after: bool = False
110
194
  ) -> ExtractionOutput:
@@ -116,32 +200,30 @@ async def _extract_file(
116
200
 
117
201
  used_docling = False
118
202
  try:
203
+ route = _route_for_mime(mime, cfg)
204
+
119
205
  # Docling routing (if enabled and supported)
120
- engine = cfg.document_engine
121
- if engine == "docling" or (
122
- engine == "auto" and DOCLING_AVAILABLE and mime in DOCLING_SUPPORTED
123
- ):
124
- if DOCLING_AVAILABLE and extract_docling is not None:
125
- used_docling = True
126
- result = await extract_docling(path, cfg)
127
- if not result.title:
128
- result.title = os.path.basename(path)
129
- result.identified_type = mime
130
- result.source_type = "file"
131
- return result
206
+ if route == "docling":
207
+ used_docling = True
208
+ result = await extract_docling(path, cfg)
209
+ if not result.title:
210
+ result.title = os.path.basename(path)
211
+ result.identified_type = mime
212
+ result.source_type = "file"
213
+ return result
132
214
 
133
215
  # Standard processors
134
- if mime in SUPPORTED_PDF_TYPES:
216
+ if route == "pdf":
135
217
  result = await extract_pdf_file(path, cfg)
136
- elif mime in SUPPORTED_EPUB_TYPES:
218
+ elif route == "epub":
137
219
  result = await extract_epub_file(path, cfg)
138
- elif mime in SUPPORTED_OFFICE_TYPES:
220
+ elif route == "office":
139
221
  result = await extract_office(path, mime, cfg)
140
- elif mime.startswith("video/"):
222
+ elif route == "video":
141
223
  result = await extract_video(path, cfg)
142
- elif mime.startswith("audio/"):
224
+ elif route == "audio":
143
225
  result = await transcribe_audio(path, cfg)
144
- elif mime == "text/plain":
226
+ elif route == "text":
145
227
  result = await extract_text_file(path, cfg)
146
228
  else:
147
229
  raise UnsupportedTypeException(f"Unsupported file type: {mime}")
@@ -60,6 +60,8 @@ async def _extract_youtube_id(url):
60
60
  r"(?:" # Group start
61
61
  r"/embed/" # Embed URL
62
62
  r"|/v/" # Older video URL
63
+ r"|/live/" # Livestream URL (active or ended)
64
+ r"|/shorts/" # Shorts URL
63
65
  r"|/watch\?v=" # Standard watch URL
64
66
  r"|/watch\?.+&v=" # Other watch URL
65
67
  r")" # Group end
@@ -7,8 +7,8 @@ import pytest
7
7
 
8
8
  from content_core.common.exceptions import InvalidInputError, UnsupportedTypeException
9
9
  from content_core.config import ContentCoreConfig
10
- from content_core.extraction import extract_content
11
- from content_core.common.state import ExtractionOutput
10
+ from content_core.extraction import check_file_support, extract_content
11
+ from content_core.common.state import ExtractionOutput, FileSupport
12
12
 
13
13
 
14
14
  def _make_output(**kwargs) -> ExtractionOutput:
@@ -324,3 +324,86 @@ async def test_docling_flags_warning_without_engine():
324
324
 
325
325
  mock_logger.warning.assert_called_once()
326
326
  assert "docling" in mock_logger.warning.call_args[0][0].lower()
327
+
328
+
329
+ # ---------------------------------------------------------------------------
330
+ # 14. check_file_support pre-flight verdict
331
+ # ---------------------------------------------------------------------------
332
+ @pytest.mark.asyncio
333
+ async def test_check_file_support_supported():
334
+ cfg = ContentCoreConfig(document_engine="simple")
335
+ with patch(
336
+ "content_core.content.identification.get_file_type",
337
+ new_callable=AsyncMock,
338
+ return_value="application/pdf",
339
+ ):
340
+ result = await check_file_support("/tmp/test.pdf", config=cfg)
341
+ assert isinstance(result, FileSupport)
342
+ assert result.supported is True
343
+ assert result.identified_type == "application/pdf"
344
+ assert result.processor == "pdf"
345
+ assert result.reason is None
346
+ assert result.document_engine == "simple"
347
+ assert result.file_path == "/tmp/test.pdf"
348
+
349
+
350
+ @pytest.mark.asyncio
351
+ async def test_check_file_support_unsupported():
352
+ cfg = ContentCoreConfig(document_engine="simple")
353
+ with patch(
354
+ "content_core.content.identification.get_file_type",
355
+ new_callable=AsyncMock,
356
+ return_value="application/x-unknown-binary",
357
+ ):
358
+ result = await check_file_support("/tmp/test.bin", config=cfg)
359
+ assert result.supported is False
360
+ assert result.processor is None
361
+ assert result.reason is not None
362
+ assert "application/x-unknown-binary" in result.reason
363
+
364
+
365
+ @pytest.mark.asyncio
366
+ async def test_check_file_support_unidentifiable_returns_verdict():
367
+ """A file whose type can't be determined is a verdict, not a raised error."""
368
+ cfg = ContentCoreConfig(document_engine="simple")
369
+ with patch(
370
+ "content_core.content.identification.get_file_type",
371
+ new_callable=AsyncMock,
372
+ side_effect=UnsupportedTypeException("Unable to determine file type for: x"),
373
+ ):
374
+ result = await check_file_support("/tmp/mystery.xyz", config=cfg)
375
+ assert result.supported is False
376
+ assert result.processor is None
377
+ assert result.identified_type == ""
378
+ assert "determine file type" in result.reason
379
+
380
+
381
+ @pytest.mark.asyncio
382
+ async def test_check_file_support_does_not_extract():
383
+ """The pre-flight check must never invoke a real extractor."""
384
+ cfg = ContentCoreConfig(document_engine="simple")
385
+ with patch(
386
+ "content_core.content.identification.get_file_type",
387
+ new_callable=AsyncMock,
388
+ return_value="application/pdf",
389
+ ), patch(
390
+ "content_core.extraction.extract_pdf_file", new_callable=AsyncMock
391
+ ) as mock_pdf:
392
+ await check_file_support("/tmp/test.pdf", config=cfg)
393
+ mock_pdf.assert_not_awaited()
394
+
395
+
396
+ @pytest.mark.asyncio
397
+ async def test_check_file_support_agrees_with_extraction():
398
+ """The verdict must never disagree with what extract_content actually does."""
399
+ cfg = ContentCoreConfig(document_engine="simple")
400
+ with patch(
401
+ "content_core.content.identification.get_file_type",
402
+ new_callable=AsyncMock,
403
+ return_value="application/x-unknown-binary",
404
+ ):
405
+ verdict = await check_file_support("/tmp/test.bin", config=cfg)
406
+ assert verdict.supported is False
407
+ # extraction of the same type raises, confirming the verdict
408
+ with pytest.raises(UnsupportedTypeException):
409
+ await extract_content(file_path="/tmp/test.bin", config=cfg)
@@ -25,6 +25,18 @@ class TestExtractYoutubeId:
25
25
  )
26
26
  assert result == "dQw4w9WgXcQ"
27
27
 
28
+ async def test_live_url(self):
29
+ result = await _extract_youtube_id(
30
+ "https://www.youtube.com/live/dQw4w9WgXcQ"
31
+ )
32
+ assert result == "dQw4w9WgXcQ"
33
+
34
+ async def test_shorts_url(self):
35
+ result = await _extract_youtube_id(
36
+ "https://www.youtube.com/shorts/dQw4w9WgXcQ"
37
+ )
38
+ assert result == "dQw4w9WgXcQ"
39
+
28
40
  async def test_url_with_extra_params(self):
29
41
  result = await _extract_youtube_id(
30
42
  "https://www.youtube.com/watch?v=dQw4w9WgXcQ&t=120"
@@ -775,7 +775,7 @@ wheels = [
775
775
 
776
776
  [[package]]
777
777
  name = "content-core"
778
- version = "2.0.2"
778
+ version = "2.0.4"
779
779
  source = { editable = "." }
780
780
  dependencies = [
781
781
  { name = "ai-prompter" },