content-core 2.0.2__tar.gz → 2.0.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {content_core-2.0.2 → content_core-2.0.4}/CHANGELOG.md +13 -0
- {content_core-2.0.2 → content_core-2.0.4}/CLAUDE.md +32 -0
- {content_core-2.0.2 → content_core-2.0.4}/PKG-INFO +1 -1
- {content_core-2.0.2 → content_core-2.0.4}/SKILL.md +3 -2
- {content_core-2.0.2 → content_core-2.0.4}/pyproject.toml +1 -1
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/__init__.py +4 -2
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/cli.py +6 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/common/state.py +17 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/extraction.py +101 -19
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/url/youtube.py +2 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_routing.py +85 -2
- {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_youtube_parsing.py +12 -0
- {content_core-2.0.2 → content_core-2.0.4}/uv.lock +1 -1
- content_core-2.0.2/test_coverage_branch_report.md +0 -480
- {content_core-2.0.2 → content_core-2.0.4}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/.github/workflows/claude-code-review.yml +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/.github/workflows/claude.yml +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/.github/workflows/create-tag.yml +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/.github/workflows/publish.yml +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/.github/workflows/test.yml +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/.gitignore +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/.python-version +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/CONTRIBUTING.md +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/LICENSE +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/Makefile +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/README.md +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/REFACTOR.md +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/TEST_PLAN.md +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/docs/cli.md +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/docs/mcp.md +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/docs/processors.md +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/docs/usage.md +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/examples/main.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/prompts/content/cleanup.jinja +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/prompts/content/summarize.jinja +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/common/__init__.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/common/exceptions.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/common/retry.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/common/types.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/config.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/content/__init__.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/content/extraction/__init__.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/content/identification/__init__.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/content/identification/file_detector.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/content/summary/__init__.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/content/summary/core.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/logging.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/mcp/__init__.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/mcp/server.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/models.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/document/__init__.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/document/docling.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/document/docx.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/document/epub.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/document/pdf.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/document/pptx.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/document/xlsx.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/media/__init__.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/media/audio.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/media/video.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/protocol.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/text.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/url/__init__.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/url/bs4.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/url/crawl4ai.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/url/firecrawl.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/url/jina.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/processors/url/reddit.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/py.typed +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/templated_message.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/tools/__init__.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/tools/extract.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/src/content_core/tools/summarize.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/conftest.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/e2e/__init__.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/e2e/test_docling.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/e2e/test_media.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/e2e/test_remote.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/e2e/test_url_engines.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/e2e/test_youtube.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/input_content/file.docx +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/input_content/file.epub +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/input_content/file.md +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/input_content/file.mp3 +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/input_content/file.mp4 +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/input_content/file.pdf +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/input_content/file.pptx +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/input_content/file.txt +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/input_content/file.xlsx +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/input_content/file_audio.mp3 +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/input_content/new_pdf.pdf +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/integration/conftest.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/integration/test_cli_v2.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/integration/test_extraction.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_audio_concurrency.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_cli.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_config_file.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_config_v2.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_docling_extraction.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_epub_extraction.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_file_detector.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_file_detector_critical.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_file_detector_performance.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_mcp_v2.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_media_pipeline.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_models_v2.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_office_extraction.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_pdf_extraction.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_pdf_helpers.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_reddit_extraction.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_retry.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_text_processing.py +0 -0
- {content_core-2.0.2 → content_core-2.0.4}/tests/unit/test_url_engine_select.py +0 -0
|
@@ -7,6 +7,19 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
## [2.0.4] - 2026-07-12
|
|
11
|
+
|
|
12
|
+
### Added
|
|
13
|
+
- `check_file_support(file_path, config)` public API for cheap pre-flight validation of whether a file can be extracted, without running extraction; returns a `FileSupport` verdict
|
|
14
|
+
|
|
15
|
+
### Fixed
|
|
16
|
+
- YouTube `youtube.com/live/<id>` and `youtube.com/shorts/<id>` URLs were not recognized, causing title and transcript extraction to fail
|
|
17
|
+
|
|
18
|
+
## [2.0.3] - 2026-04-13
|
|
19
|
+
|
|
20
|
+
### Added
|
|
21
|
+
- `--version` flag to CLI (`content-core --version`)
|
|
22
|
+
|
|
10
23
|
## [2.0.2] - 2026-04-13
|
|
11
24
|
|
|
12
25
|
### Fixed
|
|
@@ -21,6 +21,38 @@ content-core mcp
|
|
|
21
21
|
content-core config list|set|delete
|
|
22
22
|
```
|
|
23
23
|
|
|
24
|
+
## For Automated Agents
|
|
25
|
+
|
|
26
|
+
If you are an automated coding agent (harny, Claude Code in headless mode, etc.) working on this repo:
|
|
27
|
+
|
|
28
|
+
### Install command
|
|
29
|
+
|
|
30
|
+
```bash
|
|
31
|
+
uv sync --group dev
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
### Validator command
|
|
35
|
+
|
|
36
|
+
```bash
|
|
37
|
+
uv run pytest tests/unit tests/integration -v
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
This is the gate. It returns clean pass/fail in ~15s, requires no credentials or network, and matches what CI enforces on `main` (CI runs only `tests/unit`; integration tests are also safe locally — no network/API keys). For targeted feedback during iteration, use `uv run pytest -k "<keyword>"` per the table in the Testing section below.
|
|
41
|
+
|
|
42
|
+
### Do NOT use as gates
|
|
43
|
+
|
|
44
|
+
- **`ruff check`** — available via `make ruff` but NOT enforced in CI. Including it as a validator gate would surface unaudited pre-existing findings unrelated to your task.
|
|
45
|
+
- **`mypy`** — not enforced in CI. No baseline configured. Same risk.
|
|
46
|
+
- **`make test-e2e` / `make test-e2e-heavy`** — require network access, API keys for LLM/STT providers, and large model downloads. Will fail in a sandboxed worktree.
|
|
47
|
+
|
|
48
|
+
If you believe a task genuinely requires lint or type cleanup, do it as a *separate task* with explicit scope, not as a side effect of an unrelated change.
|
|
49
|
+
|
|
50
|
+
### Scope guidance
|
|
51
|
+
|
|
52
|
+
- Single processor changes (`processors/url/*.py`, `processors/document/*.py`, `processors/media/*.py`) are the natural unit of work — keep changes confined to one file plus its matching test in `tests/unit/`.
|
|
53
|
+
- Avoid touching `extraction.py` (orchestrator) or `config.py` unless the task explicitly targets routing or configuration.
|
|
54
|
+
- New processors should follow the `Processor` protocol in `src/content_core/processors/protocol.py`.
|
|
55
|
+
|
|
24
56
|
## Codebase Structure
|
|
25
57
|
|
|
26
58
|
```
|
|
@@ -152,8 +152,9 @@ If summarization fails with an API key error, fall back to `extract_content` and
|
|
|
152
152
|
|
|
153
153
|
## Guidelines
|
|
154
154
|
|
|
155
|
-
-
|
|
156
|
-
-
|
|
155
|
+
- For small/medium content (articles, short pages): prefer MCP tools if available — they are async and more efficient
|
|
156
|
+
- For large content (long documents, full books, lengthy transcripts): prefer the CLI via Bash, redirecting output to a file (`uvx content-core extract "URL" > output.md`). This avoids flooding the agent's context window with large payloads. Read only the relevant sections from the file as needed.
|
|
157
|
+
- If MCP is not available, always use the CLI via Bash with `uvx content-core`
|
|
157
158
|
- For URLs: extraction works without any API key
|
|
158
159
|
- For audio/video: requires `OPENAI_API_KEY` (or another STT provider key)
|
|
159
160
|
- For summarization: requires an LLM API key
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "content-core"
|
|
3
|
-
version = "2.0.
|
|
3
|
+
version = "2.0.4"
|
|
4
4
|
description = "Extract what matters from any media source. Available as Python Library, macOS Service, CLI and MCP Server"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
homepage = "https://github.com/lfnovo/content-core"
|
|
@@ -5,9 +5,9 @@ load_dotenv()
|
|
|
5
5
|
|
|
6
6
|
from content_core.config import ContentCoreConfig
|
|
7
7
|
from content_core.content.summary import summarize
|
|
8
|
-
from content_core.extraction import extract_content
|
|
8
|
+
from content_core.extraction import check_file_support, extract_content
|
|
9
9
|
from content_core.logging import configure_logging
|
|
10
|
-
from content_core.common.state import ExtractionInput, ExtractionOutput
|
|
10
|
+
from content_core.common.state import ExtractionInput, ExtractionOutput, FileSupport
|
|
11
11
|
|
|
12
12
|
# Convenience alias
|
|
13
13
|
extract = extract_content
|
|
@@ -18,8 +18,10 @@ configure_logging(debug=False)
|
|
|
18
18
|
__all__ = [
|
|
19
19
|
"extract_content",
|
|
20
20
|
"extract",
|
|
21
|
+
"check_file_support",
|
|
21
22
|
"summarize",
|
|
22
23
|
"ContentCoreConfig",
|
|
23
24
|
"ExtractionInput",
|
|
24
25
|
"ExtractionOutput",
|
|
26
|
+
"FileSupport",
|
|
25
27
|
]
|
|
@@ -7,7 +7,13 @@ import click
|
|
|
7
7
|
from content_core.logging import configure_logging
|
|
8
8
|
|
|
9
9
|
|
|
10
|
+
def _get_version():
|
|
11
|
+
from importlib.metadata import version
|
|
12
|
+
return version("content-core")
|
|
13
|
+
|
|
14
|
+
|
|
10
15
|
@click.group()
|
|
16
|
+
@click.version_option(package_name="content-core")
|
|
11
17
|
@click.option("--debug", is_flag=True, help="Enable debug logging")
|
|
12
18
|
def cli(debug):
|
|
13
19
|
"""Content Core — Extract and summarize content from any source."""
|
|
@@ -29,7 +29,24 @@ class ExtractionOutput(BaseModel):
|
|
|
29
29
|
metadata: dict = Field(default_factory=dict)
|
|
30
30
|
|
|
31
31
|
|
|
32
|
+
class FileSupport(BaseModel):
|
|
33
|
+
"""Verdict from a pre-flight file-support check.
|
|
34
|
+
|
|
35
|
+
Returned by ``check_file_support`` so callers can validate an upload
|
|
36
|
+
cheaply (identification + routing only, no extraction) before committing
|
|
37
|
+
to a full extraction job.
|
|
38
|
+
"""
|
|
39
|
+
|
|
40
|
+
supported: bool
|
|
41
|
+
file_path: str
|
|
42
|
+
identified_type: str = "" # MIME type detected for the file
|
|
43
|
+
document_engine: str = "" # engine the verdict was computed for
|
|
44
|
+
processor: Optional[str] = None # processor that would handle it, if supported
|
|
45
|
+
reason: Optional[str] = None # human-readable explanation when unsupported
|
|
46
|
+
|
|
47
|
+
|
|
32
48
|
__all__ = [
|
|
33
49
|
"ExtractionInput",
|
|
34
50
|
"ExtractionOutput",
|
|
51
|
+
"FileSupport",
|
|
35
52
|
]
|
|
@@ -11,7 +11,7 @@ from content_core.common.exceptions import InvalidInputError, UnsupportedTypeExc
|
|
|
11
11
|
from content_core.common.retry import retry_download
|
|
12
12
|
from content_core.config import ContentCoreConfig, get_default_config
|
|
13
13
|
from content_core.logging import logger
|
|
14
|
-
from content_core.common.state import ExtractionOutput
|
|
14
|
+
from content_core.common.state import ExtractionOutput, FileSupport
|
|
15
15
|
|
|
16
16
|
# Import processor v2 functions
|
|
17
17
|
from content_core.processors.media.audio import transcribe_audio
|
|
@@ -105,6 +105,90 @@ async def _extract_url(url: str, cfg: ContentCoreConfig) -> ExtractionOutput:
|
|
|
105
105
|
return await extract_from_url(url, cfg)
|
|
106
106
|
|
|
107
107
|
|
|
108
|
+
def _route_for_mime(mime: str, cfg: ContentCoreConfig) -> str | None:
|
|
109
|
+
"""Return the processor that would handle a file of this MIME type.
|
|
110
|
+
|
|
111
|
+
Single source of truth for file-support routing, shared by ``_extract_file``
|
|
112
|
+
(actual extraction) and ``check_file_support`` (pre-flight validation), so
|
|
113
|
+
the pre-flight answer can never disagree with what extraction really does.
|
|
114
|
+
|
|
115
|
+
Returns the processor name ("docling", "pdf", "epub", "office", "video",
|
|
116
|
+
"audio", "text") or ``None`` if the type is unsupported.
|
|
117
|
+
"""
|
|
118
|
+
engine = cfg.document_engine
|
|
119
|
+
if engine == "docling" or (
|
|
120
|
+
engine == "auto" and DOCLING_AVAILABLE and mime in DOCLING_SUPPORTED
|
|
121
|
+
):
|
|
122
|
+
if DOCLING_AVAILABLE and extract_docling is not None:
|
|
123
|
+
return "docling"
|
|
124
|
+
|
|
125
|
+
if mime in SUPPORTED_PDF_TYPES:
|
|
126
|
+
return "pdf"
|
|
127
|
+
if mime in SUPPORTED_EPUB_TYPES:
|
|
128
|
+
return "epub"
|
|
129
|
+
if mime in SUPPORTED_OFFICE_TYPES:
|
|
130
|
+
return "office"
|
|
131
|
+
if mime.startswith("video/"):
|
|
132
|
+
return "video"
|
|
133
|
+
if mime.startswith("audio/"):
|
|
134
|
+
return "audio"
|
|
135
|
+
if mime == "text/plain":
|
|
136
|
+
return "text"
|
|
137
|
+
return None
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
async def check_file_support(
|
|
141
|
+
file_path: str, config: ContentCoreConfig | None = None
|
|
142
|
+
) -> FileSupport:
|
|
143
|
+
"""Check whether content-core can extract a given file, without extracting.
|
|
144
|
+
|
|
145
|
+
Runs the same identification + routing that ``extract_content`` uses, but
|
|
146
|
+
stops before extraction, so the answer carries the same authority as a real
|
|
147
|
+
run at a fraction of the cost (it only reads the file header). "Unsupported"
|
|
148
|
+
is returned as a verdict rather than raised, since it is an expected outcome
|
|
149
|
+
at ingestion time.
|
|
150
|
+
|
|
151
|
+
Args:
|
|
152
|
+
file_path: Local file path to validate.
|
|
153
|
+
config: Optional config override. The ``document_engine`` setting is
|
|
154
|
+
honored, since supported types vary by engine.
|
|
155
|
+
|
|
156
|
+
Returns:
|
|
157
|
+
FileSupport verdict with ``supported``, the ``identified_type`` (MIME),
|
|
158
|
+
the ``processor`` that would handle it (when supported), and a ``reason``
|
|
159
|
+
when it would not.
|
|
160
|
+
"""
|
|
161
|
+
from content_core.content.identification import get_file_type
|
|
162
|
+
|
|
163
|
+
cfg = config or get_default_config()
|
|
164
|
+
|
|
165
|
+
# Identification can fail outright for unrecognizable files -- that is itself
|
|
166
|
+
# an "unsupported" verdict (extract_content would raise here too), not an
|
|
167
|
+
# error the caller should have to catch.
|
|
168
|
+
try:
|
|
169
|
+
mime = await get_file_type(file_path)
|
|
170
|
+
except UnsupportedTypeException as exc:
|
|
171
|
+
return FileSupport(
|
|
172
|
+
supported=False,
|
|
173
|
+
file_path=file_path,
|
|
174
|
+
identified_type="",
|
|
175
|
+
document_engine=cfg.document_engine,
|
|
176
|
+
processor=None,
|
|
177
|
+
reason=str(exc),
|
|
178
|
+
)
|
|
179
|
+
|
|
180
|
+
processor = _route_for_mime(mime, cfg)
|
|
181
|
+
supported = processor is not None
|
|
182
|
+
return FileSupport(
|
|
183
|
+
supported=supported,
|
|
184
|
+
file_path=file_path,
|
|
185
|
+
identified_type=mime,
|
|
186
|
+
document_engine=cfg.document_engine,
|
|
187
|
+
processor=processor,
|
|
188
|
+
reason=None if supported else f"Unsupported file type: {mime}",
|
|
189
|
+
)
|
|
190
|
+
|
|
191
|
+
|
|
108
192
|
async def _extract_file(
|
|
109
193
|
path: str, cfg: ContentCoreConfig, delete_after: bool = False
|
|
110
194
|
) -> ExtractionOutput:
|
|
@@ -116,32 +200,30 @@ async def _extract_file(
|
|
|
116
200
|
|
|
117
201
|
used_docling = False
|
|
118
202
|
try:
|
|
203
|
+
route = _route_for_mime(mime, cfg)
|
|
204
|
+
|
|
119
205
|
# Docling routing (if enabled and supported)
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
result.title = os.path.basename(path)
|
|
129
|
-
result.identified_type = mime
|
|
130
|
-
result.source_type = "file"
|
|
131
|
-
return result
|
|
206
|
+
if route == "docling":
|
|
207
|
+
used_docling = True
|
|
208
|
+
result = await extract_docling(path, cfg)
|
|
209
|
+
if not result.title:
|
|
210
|
+
result.title = os.path.basename(path)
|
|
211
|
+
result.identified_type = mime
|
|
212
|
+
result.source_type = "file"
|
|
213
|
+
return result
|
|
132
214
|
|
|
133
215
|
# Standard processors
|
|
134
|
-
if
|
|
216
|
+
if route == "pdf":
|
|
135
217
|
result = await extract_pdf_file(path, cfg)
|
|
136
|
-
elif
|
|
218
|
+
elif route == "epub":
|
|
137
219
|
result = await extract_epub_file(path, cfg)
|
|
138
|
-
elif
|
|
220
|
+
elif route == "office":
|
|
139
221
|
result = await extract_office(path, mime, cfg)
|
|
140
|
-
elif
|
|
222
|
+
elif route == "video":
|
|
141
223
|
result = await extract_video(path, cfg)
|
|
142
|
-
elif
|
|
224
|
+
elif route == "audio":
|
|
143
225
|
result = await transcribe_audio(path, cfg)
|
|
144
|
-
elif
|
|
226
|
+
elif route == "text":
|
|
145
227
|
result = await extract_text_file(path, cfg)
|
|
146
228
|
else:
|
|
147
229
|
raise UnsupportedTypeException(f"Unsupported file type: {mime}")
|
|
@@ -60,6 +60,8 @@ async def _extract_youtube_id(url):
|
|
|
60
60
|
r"(?:" # Group start
|
|
61
61
|
r"/embed/" # Embed URL
|
|
62
62
|
r"|/v/" # Older video URL
|
|
63
|
+
r"|/live/" # Livestream URL (active or ended)
|
|
64
|
+
r"|/shorts/" # Shorts URL
|
|
63
65
|
r"|/watch\?v=" # Standard watch URL
|
|
64
66
|
r"|/watch\?.+&v=" # Other watch URL
|
|
65
67
|
r")" # Group end
|
|
@@ -7,8 +7,8 @@ import pytest
|
|
|
7
7
|
|
|
8
8
|
from content_core.common.exceptions import InvalidInputError, UnsupportedTypeException
|
|
9
9
|
from content_core.config import ContentCoreConfig
|
|
10
|
-
from content_core.extraction import extract_content
|
|
11
|
-
from content_core.common.state import ExtractionOutput
|
|
10
|
+
from content_core.extraction import check_file_support, extract_content
|
|
11
|
+
from content_core.common.state import ExtractionOutput, FileSupport
|
|
12
12
|
|
|
13
13
|
|
|
14
14
|
def _make_output(**kwargs) -> ExtractionOutput:
|
|
@@ -324,3 +324,86 @@ async def test_docling_flags_warning_without_engine():
|
|
|
324
324
|
|
|
325
325
|
mock_logger.warning.assert_called_once()
|
|
326
326
|
assert "docling" in mock_logger.warning.call_args[0][0].lower()
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
# ---------------------------------------------------------------------------
|
|
330
|
+
# 14. check_file_support pre-flight verdict
|
|
331
|
+
# ---------------------------------------------------------------------------
|
|
332
|
+
@pytest.mark.asyncio
|
|
333
|
+
async def test_check_file_support_supported():
|
|
334
|
+
cfg = ContentCoreConfig(document_engine="simple")
|
|
335
|
+
with patch(
|
|
336
|
+
"content_core.content.identification.get_file_type",
|
|
337
|
+
new_callable=AsyncMock,
|
|
338
|
+
return_value="application/pdf",
|
|
339
|
+
):
|
|
340
|
+
result = await check_file_support("/tmp/test.pdf", config=cfg)
|
|
341
|
+
assert isinstance(result, FileSupport)
|
|
342
|
+
assert result.supported is True
|
|
343
|
+
assert result.identified_type == "application/pdf"
|
|
344
|
+
assert result.processor == "pdf"
|
|
345
|
+
assert result.reason is None
|
|
346
|
+
assert result.document_engine == "simple"
|
|
347
|
+
assert result.file_path == "/tmp/test.pdf"
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
@pytest.mark.asyncio
|
|
351
|
+
async def test_check_file_support_unsupported():
|
|
352
|
+
cfg = ContentCoreConfig(document_engine="simple")
|
|
353
|
+
with patch(
|
|
354
|
+
"content_core.content.identification.get_file_type",
|
|
355
|
+
new_callable=AsyncMock,
|
|
356
|
+
return_value="application/x-unknown-binary",
|
|
357
|
+
):
|
|
358
|
+
result = await check_file_support("/tmp/test.bin", config=cfg)
|
|
359
|
+
assert result.supported is False
|
|
360
|
+
assert result.processor is None
|
|
361
|
+
assert result.reason is not None
|
|
362
|
+
assert "application/x-unknown-binary" in result.reason
|
|
363
|
+
|
|
364
|
+
|
|
365
|
+
@pytest.mark.asyncio
|
|
366
|
+
async def test_check_file_support_unidentifiable_returns_verdict():
|
|
367
|
+
"""A file whose type can't be determined is a verdict, not a raised error."""
|
|
368
|
+
cfg = ContentCoreConfig(document_engine="simple")
|
|
369
|
+
with patch(
|
|
370
|
+
"content_core.content.identification.get_file_type",
|
|
371
|
+
new_callable=AsyncMock,
|
|
372
|
+
side_effect=UnsupportedTypeException("Unable to determine file type for: x"),
|
|
373
|
+
):
|
|
374
|
+
result = await check_file_support("/tmp/mystery.xyz", config=cfg)
|
|
375
|
+
assert result.supported is False
|
|
376
|
+
assert result.processor is None
|
|
377
|
+
assert result.identified_type == ""
|
|
378
|
+
assert "determine file type" in result.reason
|
|
379
|
+
|
|
380
|
+
|
|
381
|
+
@pytest.mark.asyncio
|
|
382
|
+
async def test_check_file_support_does_not_extract():
|
|
383
|
+
"""The pre-flight check must never invoke a real extractor."""
|
|
384
|
+
cfg = ContentCoreConfig(document_engine="simple")
|
|
385
|
+
with patch(
|
|
386
|
+
"content_core.content.identification.get_file_type",
|
|
387
|
+
new_callable=AsyncMock,
|
|
388
|
+
return_value="application/pdf",
|
|
389
|
+
), patch(
|
|
390
|
+
"content_core.extraction.extract_pdf_file", new_callable=AsyncMock
|
|
391
|
+
) as mock_pdf:
|
|
392
|
+
await check_file_support("/tmp/test.pdf", config=cfg)
|
|
393
|
+
mock_pdf.assert_not_awaited()
|
|
394
|
+
|
|
395
|
+
|
|
396
|
+
@pytest.mark.asyncio
|
|
397
|
+
async def test_check_file_support_agrees_with_extraction():
|
|
398
|
+
"""The verdict must never disagree with what extract_content actually does."""
|
|
399
|
+
cfg = ContentCoreConfig(document_engine="simple")
|
|
400
|
+
with patch(
|
|
401
|
+
"content_core.content.identification.get_file_type",
|
|
402
|
+
new_callable=AsyncMock,
|
|
403
|
+
return_value="application/x-unknown-binary",
|
|
404
|
+
):
|
|
405
|
+
verdict = await check_file_support("/tmp/test.bin", config=cfg)
|
|
406
|
+
assert verdict.supported is False
|
|
407
|
+
# extraction of the same type raises, confirming the verdict
|
|
408
|
+
with pytest.raises(UnsupportedTypeException):
|
|
409
|
+
await extract_content(file_path="/tmp/test.bin", config=cfg)
|
|
@@ -25,6 +25,18 @@ class TestExtractYoutubeId:
|
|
|
25
25
|
)
|
|
26
26
|
assert result == "dQw4w9WgXcQ"
|
|
27
27
|
|
|
28
|
+
async def test_live_url(self):
|
|
29
|
+
result = await _extract_youtube_id(
|
|
30
|
+
"https://www.youtube.com/live/dQw4w9WgXcQ"
|
|
31
|
+
)
|
|
32
|
+
assert result == "dQw4w9WgXcQ"
|
|
33
|
+
|
|
34
|
+
async def test_shorts_url(self):
|
|
35
|
+
result = await _extract_youtube_id(
|
|
36
|
+
"https://www.youtube.com/shorts/dQw4w9WgXcQ"
|
|
37
|
+
)
|
|
38
|
+
assert result == "dQw4w9WgXcQ"
|
|
39
|
+
|
|
28
40
|
async def test_url_with_extra_params(self):
|
|
29
41
|
result = await _extract_youtube_id(
|
|
30
42
|
"https://www.youtube.com/watch?v=dQw4w9WgXcQ&t=120"
|