docspectra 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
docspectra/__init__.py ADDED
@@ -0,0 +1,8 @@
1
+ """DocSpectra: schema-driven document parsing — any file in, spectral JSON out."""
2
+
3
+ from importlib.metadata import PackageNotFoundError, version
4
+
5
+ try:
6
+ __version__ = version('docspectra')
7
+ except PackageNotFoundError:
8
+ __version__ = '0.1.0'
@@ -0,0 +1,209 @@
1
+ """The only PydanticAI boundary: xtremeparse's AgentRunner adapted here.
2
+ Swapping agent frameworks means rewriting this module and nothing else."""
3
+
4
+ from __future__ import annotations
5
+
6
+ import json
7
+ import re
8
+ from typing import Sequence
9
+
10
+ import httpx
11
+ from pydantic_ai import Agent, NativeOutput
12
+ from pydantic_ai.models import Model
13
+ from pydantic_ai.output import StructuredDict
14
+
15
+ from xtremeparse.contracts import AgentResult, Issue, JSONSchema
16
+ from xtremeparse.scheduling import report_tokens
17
+
18
+
19
+ # qwen commercial/numeric models think by default and dashscope rejects
20
+ # tool_choice='required' (structured output) while thinking — same quirk
21
+ # argus works around; generalized to any qwen3.x numeric series
22
+ _THINKING_OFF = re.compile(
23
+ r'^qwen-(plus|flash|turbo)(-.*)?$|'
24
+ r'^qwen3(\.\d+)?-(omni-flash|vl-plus|flash|plus|\d+b(-a\d+b)?)(-.*)?$')
25
+
26
+
27
+ class PydanticAIRunner:
28
+ """Adapts the AgentRunner protocol to PydanticAI.
29
+
30
+ The shared ``content`` prefix rides its own system message (Agent
31
+ ``instructions``) so provider prefix caching sees identical bytes
32
+ across every call of one extraction; the unit card, scope and
33
+ feedback trail in the user message — a history round already
34
+ carries the card and scope, so the caller passes ``''`` for them
35
+ and only the per-round feedback is fresh. Structured output goes
36
+ through
37
+ the provider's native JSON-schema mode — tool-mode definitions
38
+ serialize ahead of the system message and displace the shared
39
+ prefix, killing cache reuse across specialists. Array-shaped result
40
+ schemas are wrapped in an ``items`` object (native output requires
41
+ an object schema) and unwrapped on the way out. StructuredDict
42
+ guarantees only a str-keyed dict — deep conformance is the injected
43
+ validator's job. Sampling defaults to temperature 0 (extraction is
44
+ deterministic-intent); ``model_settings`` overrides any field.
45
+ Actual token usage is reported to the dispatching scheduler via
46
+ xtremeparse's ``report_tokens``. ``tools`` is accepted (frozen
47
+ protocol) but not implemented yet."""
48
+
49
+ def __init__(self, model: Model, *, model_settings: dict | None = None,
50
+ strict_output: bool = True):
51
+ self.model = model
52
+ # deterministic-intent extraction: sampling heat off by default
53
+ self.model_settings = {'temperature': 0, **(model_settings or {})}
54
+ # strict keeps the schema in a server-side grammar; without it some
55
+ # providers inject the schema into the prompt behind the payload and
56
+ # break the cacheable prefix — hosts whose strict grammar rejects
57
+ # the schema shape turn it off here
58
+ self.strict_output = strict_output
59
+
60
+ async def run(self, *, instructions: str, result_schema: JSONSchema, content: str,
61
+ scope: str = None, tools: list = None, history: list = None,
62
+ feedback: Sequence[Issue] = None) -> AgentResult:
63
+ if tools:
64
+ raise NotImplementedError('tool-calling is deferred; see design.md §5.1')
65
+ free = result_schema.get('type') == 'string' # free-text mode: lib parses
66
+ wrapped = not free and result_schema.get('type') == 'array'
67
+ schema = _wrap(result_schema) if wrapped else result_schema
68
+ output_type = str if free else NativeOutput(StructuredDict(schema),
69
+ strict=self.strict_output)
70
+ agent = Agent(self.model, output_type=output_type, instructions=content)
71
+ result = await agent.run(_tail(instructions, scope, feedback),
72
+ message_history=history,
73
+ model_settings=self._model_settings())
74
+ await report_tokens(result.usage.input_tokens + result.usage.output_tokens)
75
+ data = result.output.get('items') if wrapped else result.output
76
+ return AgentResult(data=data, history=result.all_messages())
77
+
78
+ def _model_settings(self):
79
+ settings = dict(self.model_settings)
80
+ if _THINKING_OFF.match(getattr(self.model, 'model_name', '')):
81
+ settings['extra_body'] = {'enable_thinking': False}
82
+ return settings
83
+
84
+
85
+ def enable_tracing() -> None:
86
+ """Turn on the framework's native OTel span emission for every agent
87
+ (process-wide). Every adapter in this boundary provides this hook so
88
+ hosts enable tracing without naming the framework."""
89
+ Agent.instrument_all(True)
90
+
91
+
92
+ def _json_body(body: dict) -> bytes:
93
+ """SDK serialization style — ensure_ascii would double CJK bytes."""
94
+ return json.dumps(body, ensure_ascii=False, separators=(',', ':')).encode()
95
+
96
+
97
+ def _stripped_headers(headers: httpx.Headers) -> httpx.Headers:
98
+ """Headers for a rewritten body: the length/encoding headers are stale."""
99
+ headers = headers.copy()
100
+ headers.pop('content-length', None)
101
+ headers.pop('content-encoding', None)
102
+ return headers
103
+
104
+
105
+ def marked_request(request: httpx.Request) -> httpx.Request:
106
+ """Rewrite a chat-completions request so its leading string-content
107
+ system message becomes a cache-marked content block (dashscope's
108
+ ``cache_control`` explicit-cache extension). The marker must close
109
+ the shared payload at a message boundary — marking mid-message does
110
+ not match. Non-matching shapes pass through untouched."""
111
+ body = json.loads(request.read())
112
+ messages = body.get('messages')
113
+ if (messages and messages[0].get('role') == 'system'
114
+ and isinstance(messages[0].get('content'), str)):
115
+ messages[0]['content'] = [{'type': 'text', 'text': messages[0]['content'],
116
+ 'cache_control': {'type': 'ephemeral'}}]
117
+ return httpx.Request(request.method, request.url,
118
+ headers=_stripped_headers(request.headers),
119
+ content=_json_body(body))
120
+ return request
121
+
122
+
123
+ def _unthinking(request: httpx.Request) -> httpx.Request:
124
+ """Drops ``enable_thinking: false`` into a chat-completions body."""
125
+ body = json.loads(request.read())
126
+ return httpx.Request(request.method, request.url,
127
+ headers=_stripped_headers(request.headers),
128
+ content=_json_body(body | {'enable_thinking': False}))
129
+
130
+
131
+ def _rename_cache_write(response: httpx.Response) -> httpx.Response:
132
+ """Relabel dashscope's cache-write usage field to the OpenAI name
133
+ pydantic-ai reads (``prompt_tokens_details.cache_write_tokens``);
134
+ without the relabel the count never reaches the usage object or
135
+ the span."""
136
+ body = response.content
137
+ if b'"cache_creation_input_tokens"' not in body:
138
+ return response
139
+ return httpx.Response(response.status_code,
140
+ headers=_stripped_headers(response.headers),
141
+ content=body.replace(b'"cache_creation_input_tokens"',
142
+ b'"cache_write_tokens"'))
143
+
144
+
145
+ class _WireTransport(httpx.AsyncHTTPTransport):
146
+ """Applies provider wire transforms — requests before they leave,
147
+ chat-completion responses as they return; the openai client beneath
148
+ is untouched."""
149
+
150
+ def __init__(self, transforms: list):
151
+ super().__init__() # the base owns the connection pool
152
+ self.transforms = transforms
153
+
154
+ async def handle_async_request(self, request: httpx.Request) -> httpx.Response:
155
+ if request.url.path.endswith('/chat/completions'):
156
+ for transform in self.transforms:
157
+ request = transform(request)
158
+ response = await super().handle_async_request(request)
159
+ if 'application/json' in response.headers.get('content-type', ''):
160
+ await response.aread() # a rewritable body must be read first
161
+ return _rename_cache_write(response)
162
+ return response
163
+ return await super().handle_async_request(request)
164
+
165
+
166
+ def openai_model(base_url: str, api_key: str, model: str, *, explicit_cache: bool = False,
167
+ thinking_off: bool = False,
168
+ http_client: httpx.AsyncClient | None = None):
169
+ """A pydantic-ai Model for any OpenAI-compatible endpoint — the
170
+ provider mechanism lives inside the agents boundary; hosts pass
171
+ only deployment config. Requires the ``docspectra[pydantic-ai]`` extra.
172
+ ``explicit_cache`` marks the leading system message as a provider
173
+ cache block (dashscope explicit cache): paired with a runner that
174
+ puts the shared payload in Agent instructions, the call that runs
175
+ first pays cache creation once and every call behind it hits; the
176
+ wire transport relabels dashscope's non-standard cache-write usage
177
+ field so the count also lands on the span. Providers that reject
178
+ the extension must leave it off.
179
+ ``thinking_off`` drops ``enable_thinking: false`` into every request
180
+ — for consumers that build their own pydantic-ai agents outside the
181
+ runner's model_settings (pydantic-evals LLMJudge: tools +
182
+ tool_choice='required', which dashscope rejects while thinking).
183
+ A host ``http_client`` (proxies, connection tuning) is used
184
+ verbatim."""
185
+ try:
186
+ from pydantic_ai.models.openai import OpenAIChatModel
187
+ from pydantic_ai.providers.openai import OpenAIProvider
188
+ except ImportError as e:
189
+ raise ImportError('install the pydantic-ai extra: docspectra[pydantic-ai]') from e
190
+ if transforms := ([marked_request] if explicit_cache else []) \
191
+ + ([_unthinking] if thinking_off else []):
192
+ http_client = httpx.AsyncClient(transport=_WireTransport(transforms))
193
+ return OpenAIChatModel(model, provider=OpenAIProvider(
194
+ base_url=base_url, api_key=api_key, http_client=http_client))
195
+
196
+
197
+ def _wrap(schema: JSONSchema) -> JSONSchema:
198
+ return {'type': 'object', 'properties': {'items': schema},
199
+ 'required': ['items'], 'additionalProperties': False}
200
+
201
+
202
+ def _tail(instructions: str, scope, feedback) -> str:
203
+ parts = [instructions] if instructions else []
204
+ if scope is not None:
205
+ parts.append(f'\n\n---\nAssigned material:\n{scope}')
206
+ if feedback:
207
+ lines = '\n'.join(f'- {i.path}: {i.message}' for i in feedback)
208
+ parts.append(f'\n\n---\nValidation errors to fix:\n{lines}')
209
+ return ''.join(parts)
@@ -0,0 +1,38 @@
1
+ """File → text. Provider engines as optional extras (``docspectra[pdf-inspect]``,
2
+ ``docspectra[anydoc]``, ``docspectra[mineru]``), magic-byte type routing,
3
+ and file-level decline chains: ``DOCSPECTRA_PARSER_CONFIG`` orders each type's
4
+ chain and a provider whose gate fires on a specific file declines, letting the
5
+ next engine take it."""
6
+
7
+ from docspectra.parser.contracts import (Attempt, Declined, FileInput,
8
+ ImageAsset, MAX_ASSETS, MAX_ASSET_BYTES,
9
+ ParseResult)
10
+ from docspectra.parser.core import (
11
+ ParseDeclinedError,
12
+ Provider,
13
+ configure,
14
+ parse,
15
+ parse_path,
16
+ provider,
17
+ register,
18
+ )
19
+ from docspectra.parser.sniff import sniff, validate_declared
20
+
21
+ __all__ = [
22
+ 'Attempt',
23
+ 'Declined',
24
+ 'FileInput',
25
+ 'ImageAsset',
26
+ 'MAX_ASSETS',
27
+ 'MAX_ASSET_BYTES',
28
+ 'ParseDeclinedError',
29
+ 'ParseResult',
30
+ 'Provider',
31
+ 'configure',
32
+ 'parse',
33
+ 'parse_path',
34
+ 'provider',
35
+ 'register',
36
+ 'sniff',
37
+ 'validate_declared',
38
+ ]
@@ -0,0 +1,64 @@
1
+ """Parser contracts: result shapes and control-flow values shared by core
2
+ and providers."""
3
+
4
+ from __future__ import annotations
5
+
6
+ from dataclasses import dataclass, field
7
+ from pathlib import Path
8
+ from typing import Literal
9
+
10
+ FileInput = bytes | str | Path
11
+
12
+
13
+ @dataclass
14
+ class ImageAsset:
15
+ """One embedded image carried faithfully, uninterpreted.
16
+
17
+ ``name`` matches the ``![](...)`` reference in ParseResult.text;
18
+ ``source`` is engine provenance ('pdf p2', 'word/media/image1.png').
19
+ """
20
+
21
+ name: str
22
+ mime: str
23
+ data: bytes
24
+ source: str
25
+
26
+
27
+ # payload-budget policy for ParseResult.images — every provider skips
28
+ # assets over these limits and records the skip count in meta
29
+ MAX_ASSETS = 50
30
+ MAX_ASSET_BYTES = 5 * 1024 * 1024
31
+
32
+
33
+ @dataclass
34
+ class Attempt:
35
+ """One provider's outcome on a file, recorded by core on success and
36
+ failure alike — the eval suite reads routing through it."""
37
+
38
+ provider: str
39
+ outcome: Literal['success', 'declined', 'error']
40
+ reason: str | None = None
41
+
42
+
43
+ @dataclass
44
+ class Declined:
45
+ """Expected control flow: this engine cannot read *this* file, or
46
+ read it into a known false-positive shape (a gate fired). Never an
47
+ exception — the chain moves to the next provider."""
48
+
49
+ reason: str
50
+
51
+
52
+ @dataclass
53
+ class ParseResult:
54
+ """The outcome of a successful parse. ``format`` is constrained —
55
+ extend the Literal, never free-form. ``meta`` is provider diagnostics
56
+ (gate hits, page counts, task ids, timings). ``provider`` and
57
+ ``attempts`` are stamped by core, not the provider."""
58
+
59
+ text: str
60
+ provider: str = '' # engine that accepted the file (core-stamped)
61
+ format: Literal['markdown'] = 'markdown'
62
+ images: list[ImageAsset] = field(default_factory=list)
63
+ attempts: list[Attempt] = field(default_factory=list)
64
+ meta: dict = field(default_factory=dict)
@@ -0,0 +1,192 @@
1
+ """Provider registry, decline chains, and the async parse entry.
2
+
3
+ Core carries zero parsing dependencies: built-in providers are in-tree
4
+ modules under ``docspectra.providers``, try-imported at init — an absent
5
+ extra is skipped silently (an installed-but-broken provider is recorded
6
+ and surfaced in config errors instead). Engines declare themselves with
7
+ :func:`provider`; chains are ordered by ``DOCSPECTRA_PARSER_CONFIG`` (a
8
+ JSON env var mapping a document type to an ordered provider list),
9
+ falling back to the derived default chain — the canonical order of
10
+ installed providers that declared the type. Declines are return values,
11
+ hard errors fall through; chain exhaustion raises
12
+ :class:`ParseDeclinedError` with every attempt aggregated."""
13
+
14
+ from __future__ import annotations
15
+
16
+ import importlib
17
+ import json
18
+ import os
19
+ from dataclasses import dataclass
20
+ from pathlib import Path
21
+ from typing import Awaitable, Callable, Sequence
22
+
23
+ from docspectra.parser.contracts import Attempt, Declined, ParseResult
24
+ from docspectra.parser.sniff import sniff, validate_declared
25
+
26
+ _BUILTIN_MODULES = ('txt', 'pdf_inspect', 'anydoc', 'mineru')
27
+ _BUILTIN_NAMES = {m.replace('_', '-') for m in _BUILTIN_MODULES}
28
+ _CONFIG_ENV = 'DOCSPECTRA_PARSER_CONFIG'
29
+
30
+
31
+ @dataclass
32
+ class Provider:
33
+ """One engine: the async parse entry plus the types it declares."""
34
+
35
+ name: str
36
+ types: tuple[str, ...]
37
+ parse: Callable[..., Awaitable[ParseResult | Declined]]
38
+
39
+
40
+ class ParseDeclinedError(Exception):
41
+ """Every provider in the chain declined or errored on this file.
42
+
43
+ ``attempts`` aggregates every outcome. A hard error raised through
44
+ the chain is chained as the cause.
45
+ """
46
+
47
+ def __init__(self, attempts: list[Attempt]):
48
+ self.attempts = attempts
49
+ detail = '; '.join(
50
+ f"{a.provider}: {a.outcome}" + (f' ({a.reason})' if a.reason else '')
51
+ for a in attempts
52
+ )
53
+ super().__init__(f'every provider declined: {detail}')
54
+
55
+
56
+ _providers: dict[str, Provider] = {} # keyed by normalized (hyphen) engine name
57
+ _builtins_loaded = False
58
+ _config: dict | None = None # effective chain config; resolved from env on first use
59
+ _import_errors: dict[str, str] = {} # provider module → failure, when installed-but-broken
60
+
61
+
62
+ def _norm(name: str) -> str:
63
+ return name.replace('_', '-')
64
+
65
+
66
+ def register(name: str, types: Sequence[str], fn) -> None:
67
+ """Programmatic registration, same surface as the decorator. Names are
68
+ normalized (hyphen↔underscore) and registration is last-write-wins."""
69
+ _providers[_norm(name)] = Provider(_norm(name), tuple(types), fn)
70
+
71
+
72
+ def provider(name: str, types: Sequence[str]):
73
+ """Declare an engine around an async ``parse(data, *, name=None)
74
+ -> ParseResult | Declined``."""
75
+
76
+ def deco(fn):
77
+ register(name, types, fn)
78
+ return fn
79
+
80
+ return deco
81
+
82
+
83
+ def ensure_builtins() -> None:
84
+ global _builtins_loaded
85
+ if _builtins_loaded:
86
+ return
87
+ _builtins_loaded = True
88
+ for module in _BUILTIN_MODULES:
89
+ try:
90
+ importlib.import_module(f'docspectra.providers.{module}')
91
+ except ImportError as e:
92
+ if not isinstance(e, ModuleNotFoundError) \
93
+ or (e.name or '').startswith('docspectra'):
94
+ _import_errors[_norm(module)] = f'{type(e).__name__}: {e}'
95
+
96
+
97
+ def configure(config: dict | None = None) -> None:
98
+ """Pin chains to an explicit ``{type: [engine, ...]}`` mapping, or
99
+ reload them from DOCSPECTRA_PARSER_CONFIG (None, validated here —
100
+ the fail-fast hook a host calls at startup)."""
101
+ global _config
102
+ ensure_builtins()
103
+ _config = _validate(config) if config is not None else _env_config()
104
+
105
+
106
+ async def parse(data: bytes, *, name: str | None = None) -> ParseResult:
107
+ """Parse one file (bytes are the contract) into a ParseResult,
108
+ walking the type's decline chain. Raises ValueError on unroutable
109
+ input, on a declared binary extension whose content fails magic
110
+ verification, or when no provider is installed; ParseDeclinedError
111
+ when every provider declines or errors."""
112
+ ensure_builtins()
113
+ validate_declared(data, name) # declared-binary gate, before any provider
114
+ doc_type = sniff(data, name)
115
+ if doc_type is None:
116
+ raise ValueError(f'cannot determine document type of {name or "input"}')
117
+ chain = _chain_for(doc_type)
118
+ if not chain:
119
+ raise ValueError(f'no provider installed for document type {doc_type!r}')
120
+ attempts: list[Attempt] = []
121
+ cause = None
122
+ for provider_name in chain:
123
+ p = _providers[provider_name]
124
+ try:
125
+ result = await p.parse(data, name=name)
126
+ except Exception as e:
127
+ attempts.append(Attempt(p.name, 'error', f'{type(e).__name__}: {e}'))
128
+ cause = cause or e
129
+ continue
130
+ if isinstance(result, Declined):
131
+ attempts.append(Attempt(p.name, 'declined', result.reason))
132
+ continue
133
+ result.provider = p.name
134
+ result.attempts = [*attempts, Attempt(p.name, 'success')]
135
+ return result
136
+ raise ParseDeclinedError(attempts) from cause
137
+
138
+
139
+ async def parse_path(path: str | Path) -> ParseResult:
140
+ """Convenience wrapper: read the file and parse it, name from the path."""
141
+ path = Path(path)
142
+ return await parse(path.read_bytes(), name=path.name)
143
+
144
+
145
+ def _chain_for(doc_type: str) -> list[str]:
146
+ config = _effective_config()
147
+ if doc_type in config:
148
+ return list(config[doc_type])
149
+ # derived default chain: canonical order of installed providers that
150
+ # declared the type — extends automatically as built-ins are added
151
+ return [p.name for p in _providers.values() if doc_type in p.types]
152
+
153
+
154
+ def _effective_config() -> dict:
155
+ global _config
156
+ if _config is None:
157
+ _config = _env_config()
158
+ return _config
159
+
160
+
161
+ def _env_config() -> dict:
162
+ raw = os.environ.get(_CONFIG_ENV)
163
+ if not raw:
164
+ return {}
165
+ try:
166
+ config = json.loads(raw)
167
+ except json.JSONDecodeError as e:
168
+ raise ValueError(f'{_CONFIG_ENV} is not valid JSON: {e}') from e
169
+ if not isinstance(config, dict):
170
+ raise ValueError(f'{_CONFIG_ENV} must be a JSON object of type → engine list')
171
+ return _validate(config)
172
+
173
+
174
+ def _validate(config: dict) -> dict:
175
+ validated = {}
176
+ for doc_type, names in config.items():
177
+ if not isinstance(names, list) or not all(isinstance(n, str) for n in names):
178
+ raise ValueError(f'{_CONFIG_ENV}[{doc_type!r}] must be a list of engine names')
179
+ normed = []
180
+ for name in names:
181
+ key = _norm(name)
182
+ if key not in _providers:
183
+ if key in _BUILTIN_NAMES:
184
+ detail = f' ({_import_errors[key]})' if key in _import_errors else ''
185
+ raise ValueError(
186
+ f'provider {name!r} is configured but not installed{detail} — '
187
+ f'install the matching extra (docspectra[{key}])')
188
+ raise ValueError(
189
+ f'unknown parser provider {name!r}; known: {sorted(_providers)}')
190
+ normed.append(key)
191
+ validated[doc_type] = normed
192
+ return validated
@@ -0,0 +1,97 @@
1
+ """Document-type sniffing: magic bytes first, extension fallback, no
2
+ dependencies. Bytes win on conflict for routing; a filename declaring a
3
+ binary extension is gated by :func:`validate_declared` before any
4
+ provider runs.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import io
10
+ import zipfile
11
+ from pathlib import Path
12
+
13
+ _PDF = b'%PDF-'
14
+ _ZIP = b'PK\x03\x04'
15
+ _OLE2 = b'\xd0\xcf\x11\xe0\xa1\xb1\x1a\xe1'
16
+
17
+ _IMAGES = (
18
+ (b'\xff\xd8\xff', 'jpg'),
19
+ (b'\x89PNG\r\n\x1a\n', 'png'),
20
+ (b'GIF87a', 'gif'),
21
+ (b'GIF89a', 'gif'),
22
+ (b'BM', 'bmp'),
23
+ (b'II*\x00', 'tiff'),
24
+ (b'MM\x00*', 'tiff'),
25
+ (b'\x00\x00\x00\x0cjP \r\n\x87\n', 'jp2'),
26
+ )
27
+
28
+ # extension aliases, normalized to the canonical type
29
+ _ALIASES = {'jpeg': 'jpg', 'tif': 'tiff', 'htm': 'html'}
30
+
31
+ # OLE2 magic is coarse — the legacy office extensions form one family
32
+ _FAMILY = {'doc': frozenset({'doc', 'ppt', 'xls'})}
33
+
34
+ # types whose declared extension must verify against content magic: the
35
+ # magic tables' codomain (zip parts + images + webp), family expanded
36
+ _BINARY_TYPES = ({'pdf', 'docx', 'xlsx', 'pptx', 'webp'}
37
+ | {t for _, t in _IMAGES} | set().union(*_FAMILY.values()))
38
+
39
+
40
+ def sniff(data: bytes, name: str | None = None) -> str | None:
41
+ """Route bytes to a canonical document type. ``name`` feeds the
42
+ extension fallback when magic is inconclusive; None when neither
43
+ yields a type."""
44
+ if data.startswith(_PDF):
45
+ return 'pdf'
46
+ if data.startswith(_ZIP):
47
+ return _zip_type(data) or _extension_type(name)
48
+ if data.startswith(_OLE2):
49
+ return 'doc' # coarse: doc/ppt/xls share the legacy chain
50
+ if t := _image_type(data):
51
+ return t
52
+ return _extension_type(name)
53
+
54
+
55
+ def validate_declared(data: bytes, name: str | None = None) -> None:
56
+ """Gate a declared binary extension against content magic: raise
57
+ ValueError('file content is not a XXX') — XXX the extension's
58
+ canonical type — when the bytes do not verify. Text and unknown
59
+ extensions are never validated; the extension fallback owns them."""
60
+ declared = _extension_type(name)
61
+ if declared not in _BINARY_TYPES or _matches(data, declared):
62
+ return
63
+ raise ValueError(f'file content is not a {declared}')
64
+
65
+
66
+ def _matches(data: bytes, declared: str) -> bool:
67
+ magic = sniff(data) # no name: pure magic, extension fallback off
68
+ return magic is not None and declared in _FAMILY.get(magic, {magic})
69
+
70
+
71
+ def _zip_type(data: bytes) -> str | None:
72
+ try:
73
+ with zipfile.ZipFile(io.BytesIO(data)) as z:
74
+ for info in z.infolist():
75
+ if info.filename.startswith('word/'):
76
+ return 'docx'
77
+ if info.filename.startswith('xl/'):
78
+ return 'xlsx'
79
+ if info.filename.startswith('ppt/'):
80
+ return 'pptx'
81
+ except (zipfile.BadZipFile, OSError):
82
+ return None
83
+ return None
84
+
85
+
86
+ def _image_type(data: bytes) -> str | None:
87
+ if data.startswith(b'RIFF') and data[8:12] == b'WEBP':
88
+ return 'webp'
89
+ for magic, t in _IMAGES:
90
+ if data.startswith(magic):
91
+ return t
92
+ return None
93
+
94
+
95
+ def _extension_type(name: str | None) -> str | None:
96
+ ext = Path(name or '').suffix.lower().lstrip('.')
97
+ return _ALIASES.get(ext, ext) or None
@@ -0,0 +1,42 @@
1
+ """Built-in provider engines. Core try-imports each module at init; a
2
+ module whose extra is absent fails to import and is skipped silently."""
3
+
4
+ from functools import wraps
5
+
6
+ from opentelemetry.trace import get_tracer
7
+
8
+ from docspectra.parser import Declined, ParseResult
9
+
10
+ # shared by the provider engines; spans are no-ops until a host
11
+ # registers a tracer provider
12
+ tracer = get_tracer('docspectra')
13
+
14
+
15
+ def traced(fn):
16
+ """One span per parse call, named after the engine module — engine
17
+ registrations and their span names can then never drift apart. The
18
+ span carries the outcome bookkeeping (outcome, decline reason,
19
+ input bytes, the meta diagnostics — never text: spans are an
20
+ account ledger like the sidecar, PII stays out; batch_id stays off
21
+ too, it points at remote state holding the file)."""
22
+ span_name = f'{fn.__module__.rsplit(".", 1)[-1]}.parse'
23
+
24
+ @wraps(fn)
25
+ async def wrapper(*args, **kwargs):
26
+ with tracer.start_as_current_span(span_name) as span:
27
+ result = await fn(*args, **kwargs)
28
+ if args and isinstance(args[0], (bytes, bytearray)):
29
+ span.set_attribute('parse.bytes', len(args[0]))
30
+ if isinstance(result, Declined):
31
+ span.set_attribute('parse.outcome', 'declined')
32
+ span.set_attribute('parse.reason', result.reason)
33
+ else:
34
+ span.set_attribute('parse.outcome', 'success')
35
+ span.set_attribute('parse.chars', len(result.text))
36
+ span.set_attribute('parse.images', len(result.images))
37
+ for key, value in result.meta.items():
38
+ if key != 'batch_id':
39
+ span.set_attribute(f'parse.{key}', value)
40
+ return result
41
+
42
+ return wrapper
@@ -0,0 +1,122 @@
1
+ """Shared plumbing behind the local pdf engines (pdf-inspect and anydoc):
2
+ the post-classify tail — two byte-side decline gates over pypdf (the
3
+ word-strip fingerprint and the content-stream ratio) plus one text-side
4
+ gate over the engine markdown (the table wall), the XObject image
5
+ collection, and the markdown clean-up — pure pypdf plus regex, so either
6
+ extra can host it without the other's engine."""
7
+
8
+ from __future__ import annotations
9
+
10
+ import io
11
+ import re
12
+
13
+ from pypdf import PdfReader
14
+
15
+ from docspectra.parser import (Declined, ImageAsset, MAX_ASSETS,
16
+ MAX_ASSET_BYTES, ParseResult)
17
+
18
+ RATIO_THRESHOLD = 200 # B/char; measured 4,968 vs clean max 82 (parser.md §1)
19
+ STRIP_W, STRIP_H = 80, 100 # pixel dims of word-shaped raster strips
20
+ TABLE_SHARE_MIN = 0.7 # of all text chars; judgment call — measured wall
21
+ # 0.99, corpus real tables a minority share (parser.md §1)
22
+ WALL_ROW_CHARS = 500 # chars in one row; the engine's own paragraph-wall
23
+ # constant, measured wall row ~1,300 (parser.md §1)
24
+ _MIME = {'/DCTDecode': 'image/jpeg', '/JPXDecode': 'image/jp2',
25
+ '/CCITTFaxDecode': 'image/tiff'}
26
+
27
+ _U_TAG = re.compile(r'</?u>')
28
+ _BULLET = re.compile(r'^l ', re.MULTILINE) # the engine renders bullets as 'l'
29
+
30
+
31
+ def scan(data: bytes, text: str, **meta) -> ParseResult | Declined:
32
+ """The shared post-classify tail: the two pypdf gates and the
33
+ text-side table wall, image collection, and the cleaned ParseResult.
34
+ ``meta`` adds engine-specific diagnostics between ``pages`` and
35
+ ``skipped_assets``."""
36
+ reader = PdfReader(io.BytesIO(data))
37
+ for page in reader.pages:
38
+ if strip_hit(page):
39
+ return Declined('fingerprint_gate')
40
+ if ratio_hit(reader, len(text)) >= RATIO_THRESHOLD:
41
+ return Declined('ratio_gate')
42
+ if table_wall_hit(text):
43
+ return Declined('table_wall_gate')
44
+ images, skipped = collect_images(reader)
45
+ return ParseResult(text=clean(text), images=images,
46
+ meta={'pages': len(reader.pages), **meta,
47
+ 'skipped_assets': skipped})
48
+
49
+
50
+ def clean(markdown: str) -> str:
51
+ return _BULLET.sub('• ', _U_TAG.sub('', markdown))
52
+
53
+
54
+ def table_wall_hit(text: str) -> bool:
55
+ """One markdown table swallowing the page's flowing prose — the
56
+ layout-band false positive of geometry-based table detection (a
57
+ borderless template frame reconstructed as a grid, reading order
58
+ scrambled). Both conditions matter: a document that merely carries
59
+ a real table stays, and a page of genuine data cells (high table
60
+ share, short rows) stays."""
61
+ lines = [line for line in text.splitlines() if line.strip()]
62
+ rows = [line for line in lines
63
+ if line.startswith('|') and line.strip('|- ')]
64
+ total = sum(map(len, lines))
65
+ return (bool(rows)
66
+ and sum(map(len, rows)) / total >= TABLE_SHARE_MIN
67
+ and max(map(len, rows)) > WALL_ROW_CHARS)
68
+
69
+
70
+ def strip_hit(page) -> bool:
71
+ """A wide flat image XObject — a rasterized text strip among a real
72
+ text layer (the raster-strip class). Missing height counts as the threshold."""
73
+ return any(obj.get('/Width', 0) >= STRIP_W and obj.get('/Height', STRIP_H) <= STRIP_H
74
+ for _, obj in _iter_images(page))
75
+
76
+
77
+ def ratio_hit(reader: PdfReader, nchars: int) -> float:
78
+ """Content-stream bytes per delivered text char; outlined glyphs
79
+ (the outlined-glyph class) inflate the stream without any text to show for it."""
80
+ nbytes = 0
81
+ for page in reader.pages:
82
+ if (contents := page.get_contents()) is not None:
83
+ nbytes += len(contents.get_data())
84
+ return nbytes / nchars if nchars else float('inf')
85
+
86
+
87
+ def collect_images(reader: PdfReader) -> tuple[list[ImageAsset], int]:
88
+ images, skipped = [], 0
89
+ for page_no, page in enumerate(reader.pages, 1):
90
+ for key, obj in _iter_images(page):
91
+ if len(images) >= MAX_ASSETS:
92
+ skipped += 1
93
+ continue
94
+ try:
95
+ data = obj.get_data()
96
+ except Exception:
97
+ skipped += 1
98
+ continue
99
+ if len(data) > MAX_ASSET_BYTES:
100
+ skipped += 1
101
+ continue
102
+ # names are opaque per-engine keys: the engine emits no
103
+ # image references, so nothing in the text reconciles to them
104
+ images.append(ImageAsset(
105
+ name=f'p{page_no}-{str(key).lstrip("/")}',
106
+ mime=_MIME.get(str(obj.get('/Filter', '')), 'application/octet-stream'),
107
+ data=data, source=f'pdf p{page_no} {key}'))
108
+ return images, skipped
109
+
110
+
111
+ def _iter_images(page):
112
+ """Yield (key, ImageObject) pairs from the page's own XObject dict —
113
+ the measured walk behind the fingerprint gate (parser.md §1)."""
114
+ resources = _deref(page.get('/Resources') or {})
115
+ for key, ref in _deref(resources.get('/XObject') or {}).items():
116
+ obj = _deref(ref)
117
+ if obj.get('/Subtype') == '/Image':
118
+ yield key, obj
119
+
120
+
121
+ def _deref(obj):
122
+ return obj.get_object() if hasattr(obj, 'get_object') else obj
@@ -0,0 +1,65 @@
1
+ """Local superset engine (extra: anydoc): firecrawl-anydoc (Rust) over
2
+ pdf plus every office family. The pdf branch runs the shared decline
3
+ gates — classify from the engine's own ImageBased/Scanned rejection,
4
+ the rest from ``_pdfscan``; office formats (doc/docx, xls/xlsx, ppt/pptx,
5
+ odt/ods/odp, rtf, epub, csv) convert to markdown with their embedded
6
+ images carried as ImageAssets. Docx headers and footers are page
7
+ furniture the engine drops by design."""
8
+
9
+ from __future__ import annotations
10
+
11
+ import anydoc
12
+
13
+ from docspectra.parser import (Declined, ImageAsset, MAX_ASSETS,
14
+ MAX_ASSET_BYTES, ParseResult, provider,
15
+ sniff)
16
+ from docspectra.providers import traced
17
+ from docspectra.providers._pdfscan import scan
18
+
19
+ # sniff type → engine Format for the office branch; None lets the engine
20
+ # detect from the bytes (the OLE2 family — 'doc' covers legacy doc/xls/ppt
21
+ # alike, and xls has no Format name of its own). An explicit map, not a
22
+ # passthrough: a sniff/engine vocabulary drift must fail here, not inside
23
+ # the engine as a runtime error.
24
+ _FORMAT: dict[str, str | None] = {
25
+ 'docx': 'docx', 'doc': None,
26
+ 'ppt': 'ppt', 'pptx': 'pptx', 'xls': None, 'xlsx': 'xlsx',
27
+ 'odt': 'odt', 'ods': 'ods', 'odp': 'odp',
28
+ 'rtf': 'rtf', 'epub': 'epub', 'csv': 'csv',
29
+ }
30
+
31
+
32
+ @provider('anydoc', ('pdf', *_FORMAT))
33
+ @traced
34
+ async def parse_anydoc(data: bytes, *, name: str | None = None) -> ParseResult | Declined:
35
+ # core already sniffed for routing; re-sniffed here because the
36
+ # provider contract carries (data, name) and the engine Format and
37
+ # the pdf/office branch both hang off the type
38
+ if (doc_type := sniff(data, name)) == 'pdf':
39
+ return _parse_pdf(data)
40
+ return _parse_office(data, doc_type)
41
+
42
+
43
+ def _parse_pdf(data: bytes) -> ParseResult | Declined:
44
+ try:
45
+ text = anydoc.to_markdown_bytes(data)
46
+ except anydoc.UnsupportedError:
47
+ # the engine's own rejection: ImageBased/Scanned pages need OCR
48
+ return Declined('classify_gate')
49
+ return scan(data, text)
50
+
51
+
52
+ def _parse_office(data: bytes, doc_type: str) -> ParseResult:
53
+ # signature-less formats (csv) must name their format explicitly
54
+ fmt = _FORMAT[doc_type]
55
+ text = anydoc.to_markdown_bytes(data, fmt)
56
+ images, skipped = [], 0
57
+ for asset in anydoc.to_document(data, fmt).assets:
58
+ if len(images) >= MAX_ASSETS or len(asset.data) > MAX_ASSET_BYTES:
59
+ skipped += 1
60
+ continue
61
+ images.append(ImageAsset(
62
+ name=asset.origin_part, mime=asset.media_type, data=asset.data,
63
+ source=f'anydoc {asset.origin_part}'))
64
+ return ParseResult(text=text, images=images,
65
+ meta={'skipped_assets': skipped})
@@ -0,0 +1,213 @@
1
+ """Cloud engine (extra: mineru): MinerU extraction service via signed
2
+ upload + batch poll, rate-limited by dual xtremeflow lanes (submit
3
+ 50/min, status 1000/min — the official ceilings, per-minute numbers
4
+ divided by 60 into float rps). 429/5xx raise RetryException and retry
5
+ through auto_backoff; other 4xx (quota, auth) surface as hard errors.
6
+ Results are cached in-process keyed by (sha256, model_version) — memory
7
+ only, never disk, evicted by count and byte budget."""
8
+
9
+ from __future__ import annotations
10
+
11
+ import asyncio
12
+ import hashlib
13
+ import io
14
+ import os
15
+ import time
16
+ import zipfile
17
+ from collections import OrderedDict
18
+
19
+ import httpx
20
+
21
+ from docspectra.parser import (ImageAsset, MAX_ASSETS, MAX_ASSET_BYTES,
22
+ ParseResult, provider)
23
+ from docspectra.providers import traced
24
+
25
+ from xtremeflow.scheduler.rate_limit import RetryException, auto_backoff
26
+ from xtremeflow.scheduler.request import RequestRateScheduler # deep import: not re-exported
27
+
28
+ BASE = 'https://mineru.net/api/v4'
29
+ POLL_INTERVAL = 2.0
30
+ _CACHE_MAX = 8 # ~1 MB per file — roughly 8 MB resident
31
+ _CACHE_BYTES = 8 * 1024 * 1024
32
+ _UPLOAD_RETRIES = 3
33
+ _MIME = {'jpg': 'image/jpeg', 'jpeg': 'image/jpeg', 'png': 'image/png',
34
+ 'gif': 'image/gif', 'webp': 'image/webp', 'svg': 'image/svg+xml',
35
+ 'bmp': 'image/bmp', 'tiff': 'image/tiff'}
36
+
37
+
38
+ class MineruError(Exception):
39
+ """Hard error: quota, auth, timeout, or a failed extraction — falls
40
+ through the chain like any provider exception."""
41
+
42
+
43
+ @provider('mineru', ('pdf', 'doc', 'docx', 'ppt', 'pptx', 'xls', 'xlsx',
44
+ 'png', 'jpg', 'jpeg', 'jp2', 'webp', 'gif', 'bmp', 'tiff'))
45
+ @traced
46
+ async def parse_mineru(data: bytes, *, name: str | None = None) -> ParseResult:
47
+ token = os.environ.get('MINERU_API_TOKEN')
48
+ if not token:
49
+ raise MineruError('MINERU_API_TOKEN is not set')
50
+ model = os.environ.get('MINERU_MODEL_VERSION', 'pipeline')
51
+ key = (hashlib.sha256(data).hexdigest(), model)
52
+ if cached := _cache_get(key):
53
+ # per-call facade: share the immutable payload, never the cached
54
+ # meta (a batch id for a batch this call never submitted)
55
+ return ParseResult(text=cached.text, images=cached.images,
56
+ meta={'model_version': model, 'cache_hit': True})
57
+ timeout = float(os.environ.get('MINERU_TIMEOUT', '300'))
58
+ submit_lane, status_lane = _lanes()
59
+ async with httpx.AsyncClient(timeout=30.0) as client:
60
+ task = await submit_lane.start_task(_apply_url(client, token, model, name))
61
+ batch_id, file_url = await task
62
+ await _upload(client, file_url, data)
63
+ zip_url = await _poll(client, token, batch_id, timeout, status_lane)
64
+ if not zip_url:
65
+ raise MineruError(f'mineru batch {batch_id} finished without a result zip')
66
+ result_zip = await _download(client, zip_url)
67
+ markdown, images, skipped = _unzip(result_zip)
68
+ result = ParseResult(
69
+ text=markdown, images=images,
70
+ meta={'model_version': model, 'batch_id': batch_id, 'skipped_assets': skipped})
71
+ _cache_put(key, result)
72
+ return result
73
+
74
+
75
+ @auto_backoff()
76
+ async def _apply_url(client: httpx.AsyncClient, token: str, model: str,
77
+ name: str | None) -> tuple[str, str]:
78
+ res = await client.post(f'{BASE}/file-urls/batch', headers=_headers(token),
79
+ json={'files': [{'name': name or 'file', 'data_id': 'f'}],
80
+ 'model_version': model})
81
+ _check(res)
82
+ payload = res.json()
83
+ if payload.get('code') != 0:
84
+ raise MineruError(f'mineru file-urls failed: {payload.get("msg")}')
85
+ return payload['data']['batch_id'], payload['data']['file_urls'][0]
86
+
87
+
88
+ async def _upload(client: httpx.AsyncClient, url: str, data: bytes) -> None:
89
+ """S3 single PUT is atomic — an interrupted PUT leaves no object, so
90
+ retrying the same signed URL is safe. Outside the lanes: S3 does not
91
+ count against mineru's submit rate."""
92
+ for attempt in range(_UPLOAD_RETRIES):
93
+ res = await client.put(url, content=data, timeout=120.0)
94
+ if res.status_code == 429 or res.status_code >= 500:
95
+ await asyncio.sleep(2 ** attempt)
96
+ continue
97
+ if res.status_code >= 400:
98
+ raise MineruError(f'mineru upload failed: {res.status_code}: {res.text[:200]}')
99
+ return
100
+ raise MineruError(f'mineru upload failed after {_UPLOAD_RETRIES} attempts: {res.status_code}')
101
+
102
+
103
+ async def _poll(client: httpx.AsyncClient, token: str, batch_id: str,
104
+ timeout: float, status_lane: RequestRateScheduler) -> str:
105
+ deadline = time.monotonic() + timeout
106
+ while True:
107
+ task = await status_lane.start_task(_poll_once(client, token, batch_id))
108
+ item = await task
109
+ state = item['state']
110
+ if state == 'done':
111
+ return item.get('full_zip_url')
112
+ if state == 'failed':
113
+ raise MineruError(f'mineru extract failed: {item.get("err_msg")}')
114
+ if time.monotonic() >= deadline:
115
+ raise MineruError(f'mineru timed out after {timeout:.0f}s (state {state})')
116
+ await asyncio.sleep(POLL_INTERVAL)
117
+
118
+
119
+ @auto_backoff()
120
+ async def _poll_once(client: httpx.AsyncClient, token: str, batch_id: str) -> dict:
121
+ res = await client.get(f'{BASE}/extract-results/batch/{batch_id}',
122
+ headers=_headers(token))
123
+ _check(res)
124
+ payload = res.json()
125
+ if payload.get('code') != 0:
126
+ raise MineruError(f'mineru extract-results failed: {payload.get("msg")}')
127
+ return payload['data']['extract_result'][0]
128
+
129
+
130
+ async def _download(client: httpx.AsyncClient, zip_url: str) -> bytes:
131
+ res = await client.get(zip_url, timeout=120.0)
132
+ if res.status_code >= 400:
133
+ raise MineruError(f'mineru result download failed: {res.status_code}')
134
+ return res.content
135
+
136
+
137
+ def _unzip(result_zip: bytes) -> tuple[str, list[ImageAsset], int]:
138
+ """Result zip → markdown (full.md) + ImageAssets; assets reconcile
139
+ with the markdown's ``![](images/...)`` references by name."""
140
+ markdown = ''
141
+ found = False
142
+ images: list[ImageAsset] = []
143
+ skipped = 0
144
+ with zipfile.ZipFile(io.BytesIO(result_zip)) as z:
145
+ for name in z.namelist():
146
+ if name.endswith('full.md'):
147
+ found = True
148
+ markdown = z.read(name).decode('utf-8', errors='replace')
149
+ elif name.startswith('images/') and not name.endswith('/'):
150
+ if len(images) >= MAX_ASSETS or z.getinfo(name).file_size > MAX_ASSET_BYTES:
151
+ skipped += 1
152
+ continue
153
+ ext = name.rsplit('.', 1)[-1].lower()
154
+ images.append(ImageAsset(name=name, mime=_MIME.get(ext, 'application/octet-stream'),
155
+ data=z.read(name), source=f'mineru {name}'))
156
+ if not found:
157
+ raise MineruError('mineru result zip has no full.md')
158
+ return markdown, images, skipped
159
+
160
+
161
+ def _check(res: httpx.Response) -> None:
162
+ if res.status_code >= 400:
163
+ msg = f'mineru {res.status_code}: {res.text[:200]}'
164
+ if res.status_code == 429 or res.status_code >= 500:
165
+ retry_after = res.headers.get('Retry-After')
166
+ raise RetryException(msg, retry_after=float(retry_after) if retry_after else None)
167
+ raise MineruError(msg)
168
+
169
+
170
+ def _headers(token: str) -> dict:
171
+ return {'Content-Type': 'application/json', 'Authorization': f'Bearer {token}'}
172
+
173
+
174
+ _lanes_cache: tuple[RequestRateScheduler, RequestRateScheduler] | None = None
175
+
176
+
177
+ def _lanes() -> tuple[RequestRateScheduler, RequestRateScheduler]:
178
+ global _lanes_cache
179
+ if _lanes_cache is None:
180
+ spec = dict(kv.split(':', 1) for kv in
181
+ os.environ.get('MINERU_CONCURRENCY', 'submit:50|status:1000').split('|'))
182
+ missing = [k for k in ('submit', 'status') if k not in spec]
183
+ if missing:
184
+ raise ValueError(f'MINERU_CONCURRENCY is missing the {missing} lane')
185
+ _lanes_cache = (
186
+ RequestRateScheduler(max_rps=float(spec['submit']) / 60, max_concurrency=8),
187
+ RequestRateScheduler(max_rps=float(spec['status']) / 60, max_concurrency=32),
188
+ )
189
+ return _lanes_cache
190
+
191
+
192
+ _cache: OrderedDict[tuple[str, str], ParseResult] = OrderedDict()
193
+ _cache_bytes = 0
194
+
195
+
196
+ def _result_size(result: ParseResult) -> int:
197
+ return len(result.text) + sum(len(img.data) for img in result.images)
198
+
199
+
200
+ def _cache_get(key: tuple[str, str]) -> ParseResult | None:
201
+ if cached := _cache.get(key):
202
+ _cache.move_to_end(key)
203
+ return cached
204
+
205
+
206
+ def _cache_put(key: tuple[str, str], result: ParseResult) -> None:
207
+ global _cache_bytes
208
+ _cache[key] = result
209
+ _cache.move_to_end(key)
210
+ _cache_bytes += _result_size(result)
211
+ while len(_cache) > _CACHE_MAX or _cache_bytes > _CACHE_BYTES:
212
+ _, old = _cache.popitem(last=False)
213
+ _cache_bytes -= _result_size(old)
@@ -0,0 +1,21 @@
1
+ """Local pdf engine (extra: pdf-inspect): pdf-inspector extraction with
2
+ a sampling-classify decline gate (raster pages); the fingerprint/ratio
3
+ gates, image collection, and post-processing live in ``_pdfscan``."""
4
+
5
+ from __future__ import annotations
6
+
7
+ import pdf_inspector
8
+
9
+ from docspectra.parser import Declined, ParseResult, provider
10
+ from docspectra.providers import traced
11
+ from docspectra.providers._pdfscan import scan
12
+
13
+
14
+ @provider('pdf-inspect', ('pdf',))
15
+ @traced
16
+ async def parse_pdf(data: bytes, *, name: str | None = None) -> ParseResult | Declined:
17
+ det = pdf_inspector.process_pdf_bytes(data)
18
+ if det.pages_needing_ocr:
19
+ return Declined('classify_gate')
20
+ return scan(data, det.markdown or '', pdf_type=str(det.pdf_type),
21
+ encoding_issues=det.has_encoding_issues)
@@ -0,0 +1,15 @@
1
+ """txt/md passthrough — the one built-in provider that needs no extra."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from docspectra.parser import Declined, ParseResult, provider
6
+ from docspectra.providers import traced
7
+
8
+
9
+ @provider('txt', ('txt', 'md', 'markdown'))
10
+ @traced
11
+ async def parse_txt(data: bytes, *, name: str | None = None) -> ParseResult | Declined:
12
+ try:
13
+ return ParseResult(text=data.decode('utf-8-sig'))
14
+ except UnicodeDecodeError:
15
+ return Declined('utf-8 decode failed')
@@ -0,0 +1,5 @@
1
+ """The parse service: file + JSON schema → JSON (parser → xtremeparse extractor)."""
2
+
3
+ from docspectra.service.parse import GenericIssue, ParseService, schema_validator
4
+
5
+ __all__ = ['ParseService', 'schema_validator', 'GenericIssue']
@@ -0,0 +1,60 @@
1
+ """The parse service: file + JSON schema → JSON (parser → extractor)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+
7
+ from jsonschema import Draft202012Validator
8
+
9
+ from xtremeparse import Extractor, ExtractionResult
10
+
11
+ from docspectra.parser import FileInput, parse as parse_file, parse_path
12
+
13
+
14
+ class ParseService:
15
+ """One file and one JSON Schema in, best-effort schema-shaped JSON
16
+ out — validated and self-corrected against a generic JSON Schema
17
+ validator (hosts with their own validator, e.g. docxcast, inject it
18
+ through the extractor directly)."""
19
+
20
+ def __init__(self, runner, *, scheduler=None):
21
+ self._extractor = Extractor(runner, scheduler=scheduler)
22
+
23
+ async def parse(self, data: FileInput, schema: dict, *,
24
+ name: str | None = None, validator=None) -> ExtractionResult:
25
+ document = await parse_file(data, name=name) if isinstance(data, bytes) \
26
+ else await parse_path(data)
27
+ return await self._extractor.extract(document.text, schema,
28
+ validator=validator or schema_validator(schema))
29
+
30
+
31
+ @dataclass
32
+ class GenericIssue:
33
+ """JSON-Schema error shaped to the Issue protocol; array indices
34
+ render bracketed (`jobs[0].company`) so correction routing matches."""
35
+
36
+ path: str
37
+ message: str
38
+ code: str
39
+ expected: object = None
40
+ got: object = None
41
+
42
+
43
+ def schema_validator(schema: dict):
44
+ """Validate data against a JSON Schema, returning error-level issues."""
45
+ validator = Draft202012Validator(schema)
46
+
47
+ def validate(data: dict) -> list:
48
+ return [GenericIssue(_path(e), e.message, e.validator) for e in validator.iter_errors(data)]
49
+
50
+ return validate
51
+
52
+
53
+ def _path(error) -> str:
54
+ path = ''
55
+ for part in error.absolute_path:
56
+ if isinstance(part, int):
57
+ path = f'{path}[{part}]'
58
+ else:
59
+ path = f'{path}.{part}' if path else str(part)
60
+ return path or '$'
@@ -0,0 +1,72 @@
1
+ Metadata-Version: 2.5
2
+ Name: docspectra
3
+ Version: 0.1.0
4
+ Summary: DocSpectra: schema-driven document parsing — any file in, spectral JSON out
5
+ Project-URL: Homepage, https://github.com/flowjzh/docspectra
6
+ Project-URL: Repository, https://github.com/flowjzh/docspectra.git
7
+ Project-URL: Issues, https://github.com/flowjzh/docspectra/issues
8
+ Author-email: Flow Jiang <flowjzh@gmail.com>
9
+ License-Expression: MIT
10
+ License-File: LICENSE
11
+ Keywords: documents,extraction,json-schema,llm,parsing
12
+ Classifier: Development Status :: 3 - Alpha
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3 :: Only
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
22
+ Classifier: Typing :: Typed
23
+ Requires-Python: >=3.10
24
+ Requires-Dist: jsonschema>=4.23
25
+ Requires-Dist: opentelemetry-api>=1.40.0
26
+ Requires-Dist: xtremeparse>=0.1.0
27
+ Provides-Extra: anydoc
28
+ Requires-Dist: firecrawl-anydoc; extra == 'anydoc'
29
+ Requires-Dist: pypdf; extra == 'anydoc'
30
+ Provides-Extra: mineru
31
+ Requires-Dist: httpx>=0.27; extra == 'mineru'
32
+ Requires-Dist: xtremeflow>=0.4.4; extra == 'mineru'
33
+ Provides-Extra: pdf-inspect
34
+ Requires-Dist: pdf-inspector; extra == 'pdf-inspect'
35
+ Requires-Dist: pypdf; extra == 'pdf-inspect'
36
+ Provides-Extra: pydantic-ai
37
+ Requires-Dist: pydantic-ai-slim[openai]>=2.30.0; extra == 'pydantic-ai'
38
+ Description-Content-Type: text/markdown
39
+
40
+ # DocSpectra
41
+
42
+ <img width="600" alt="DocSpectra" src="https://github.com/user-attachments/assets/1d858fb3-67c5-41f6-b0db-3fb6ab39fe4a" />
43
+
44
+ > **Any file in. Spectral JSON out. Fast.**
45
+
46
+ DocSpectra is a schema-driven document-parsing library: hand it a file in
47
+ any supported format and a JSON Schema, get back schema-conforming JSON —
48
+ extracted by extreme-concurrency LLM specialists that chunk, route, fan
49
+ out and self-correct (powered by
50
+ [xtremeparse](https://github.com/flowjzh/xtremeparse)).
51
+
52
+ ```
53
+ file + JSON schema ──► docspectra.parse ──► JSON
54
+ ```
55
+
56
+ ## Layout
57
+
58
+ - `docspectra/parser` — file → text (provider engines as optional extras:
59
+ `pdf-inspect`, `anydoc`, `mineru`; magic-byte type routing;
60
+ file-level decline chains — a provider whose gate fires on a specific
61
+ file declines and the next engine takes it)
62
+ - `docspectra/agents` — the only PydanticAI boundary (AgentRunner adapter)
63
+ - `docspectra/service` — `parse`: file + JSON schema → JSON
64
+
65
+ DocSpectra is deliberately generic: JSON Schema in and out, no document
66
+ vendors, no HTTP server, no domain tools. Applications assemble it with
67
+ their own schema sources and renderers (e.g. bridging
68
+ [DocXCast](https://github.com/flowjzh/docxcast) templates).
69
+
70
+ ## Status
71
+
72
+ Parser layer implemented and tested.
@@ -0,0 +1,18 @@
1
+ docspectra/__init__.py,sha256=BFvaJmQhD1jgrnqnoUXnTbd4Q34pjXJDvIH-Yp-vJqI,248
2
+ docspectra/agents/__init__.py,sha256=0_Cilm03Cxyr-J5joWKtv-a0dVT0ouPOOxRJTtEheLA,10239
3
+ docspectra/parser/__init__.py,sha256=OHaFVRg6oSJRh3WwM9Rs9tTa7ubK5pZp-ynl4JpIbZQ,1057
4
+ docspectra/parser/contracts.py,sha256=bLr7q5ypU6xaXJ3aWLqiaOYS4a2aYOp3QhFH5aT3MdA,1855
5
+ docspectra/parser/core.py,sha256=YB-uXbKTgpUcSrKcgJHDtc-qK0HRlzy_ypZW15gJxds,7156
6
+ docspectra/parser/sniff.py,sha256=YGJW-9przYVpqLhr5d3VUX2zGNyEwGf4K7YF3e84m_0,3282
7
+ docspectra/providers/__init__.py,sha256=HxwE6bzjhpLFVwa-VH07-TeyFMs0iJ5Xog5HjLcX-jc,1780
8
+ docspectra/providers/_pdfscan.py,sha256=8tOyZ-GUIk0M34-quEkyMXP2C3ziRuGiFKPq0Tx0bM4,5104
9
+ docspectra/providers/anydoc.py,sha256=UT_o6ReffKls3RztTCvDh7C3uzWaYYJMWLPkOrw0ebE,2752
10
+ docspectra/providers/mineru.py,sha256=s02CMeUEhmRNuvPXnEYd1xYAD_ftSjPbBxIJN9OfMWk,8904
11
+ docspectra/providers/pdf_inspect.py,sha256=GyDauwjQBrowa1a14hSSaxLQDfXMW64ZJ34a7_OPdHM,800
12
+ docspectra/providers/txt.py,sha256=Fg8vCB7tpotEZIxPKh_eTgej6A6Nb1O0EwtLxnaBYEA,502
13
+ docspectra/service/__init__.py,sha256=yG6mGKeariyx4EeQcMM2kDrvx99vETxaKCAIqwI2hXw,236
14
+ docspectra/service/parse.py,sha256=zXmRkHHs6fk7qm7rEh4SgyuNPzA3cEO0VzS3SrNbB_Q,1980
15
+ docspectra-0.1.0.dist-info/METADATA,sha256=adDfdK8pxcTpeQIYo-IjSlJdgJO8R3YaNhA91DdacZo,2943
16
+ docspectra-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
17
+ docspectra-0.1.0.dist-info/licenses/LICENSE,sha256=h8-551rXql0Z6aYn_7eita2vDlYtGqP12OGxrmlaQ74,1067
18
+ docspectra-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Flow Jiang
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.