docspectra 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docspectra/__init__.py +8 -0
- docspectra/agents/__init__.py +209 -0
- docspectra/parser/__init__.py +38 -0
- docspectra/parser/contracts.py +64 -0
- docspectra/parser/core.py +192 -0
- docspectra/parser/sniff.py +97 -0
- docspectra/providers/__init__.py +42 -0
- docspectra/providers/_pdfscan.py +122 -0
- docspectra/providers/anydoc.py +65 -0
- docspectra/providers/mineru.py +213 -0
- docspectra/providers/pdf_inspect.py +21 -0
- docspectra/providers/txt.py +15 -0
- docspectra/service/__init__.py +5 -0
- docspectra/service/parse.py +60 -0
- docspectra-0.1.0.dist-info/METADATA +72 -0
- docspectra-0.1.0.dist-info/RECORD +18 -0
- docspectra-0.1.0.dist-info/WHEEL +4 -0
- docspectra-0.1.0.dist-info/licenses/LICENSE +21 -0
docspectra/__init__.py
ADDED
|
@@ -0,0 +1,209 @@
|
|
|
1
|
+
"""The only PydanticAI boundary: xtremeparse's AgentRunner adapted here.
|
|
2
|
+
Swapping agent frameworks means rewriting this module and nothing else."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import json
|
|
7
|
+
import re
|
|
8
|
+
from typing import Sequence
|
|
9
|
+
|
|
10
|
+
import httpx
|
|
11
|
+
from pydantic_ai import Agent, NativeOutput
|
|
12
|
+
from pydantic_ai.models import Model
|
|
13
|
+
from pydantic_ai.output import StructuredDict
|
|
14
|
+
|
|
15
|
+
from xtremeparse.contracts import AgentResult, Issue, JSONSchema
|
|
16
|
+
from xtremeparse.scheduling import report_tokens
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
# qwen commercial/numeric models think by default and dashscope rejects
|
|
20
|
+
# tool_choice='required' (structured output) while thinking — same quirk
|
|
21
|
+
# argus works around; generalized to any qwen3.x numeric series
|
|
22
|
+
_THINKING_OFF = re.compile(
|
|
23
|
+
r'^qwen-(plus|flash|turbo)(-.*)?$|'
|
|
24
|
+
r'^qwen3(\.\d+)?-(omni-flash|vl-plus|flash|plus|\d+b(-a\d+b)?)(-.*)?$')
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class PydanticAIRunner:
|
|
28
|
+
"""Adapts the AgentRunner protocol to PydanticAI.
|
|
29
|
+
|
|
30
|
+
The shared ``content`` prefix rides its own system message (Agent
|
|
31
|
+
``instructions``) so provider prefix caching sees identical bytes
|
|
32
|
+
across every call of one extraction; the unit card, scope and
|
|
33
|
+
feedback trail in the user message — a history round already
|
|
34
|
+
carries the card and scope, so the caller passes ``''`` for them
|
|
35
|
+
and only the per-round feedback is fresh. Structured output goes
|
|
36
|
+
through
|
|
37
|
+
the provider's native JSON-schema mode — tool-mode definitions
|
|
38
|
+
serialize ahead of the system message and displace the shared
|
|
39
|
+
prefix, killing cache reuse across specialists. Array-shaped result
|
|
40
|
+
schemas are wrapped in an ``items`` object (native output requires
|
|
41
|
+
an object schema) and unwrapped on the way out. StructuredDict
|
|
42
|
+
guarantees only a str-keyed dict — deep conformance is the injected
|
|
43
|
+
validator's job. Sampling defaults to temperature 0 (extraction is
|
|
44
|
+
deterministic-intent); ``model_settings`` overrides any field.
|
|
45
|
+
Actual token usage is reported to the dispatching scheduler via
|
|
46
|
+
xtremeparse's ``report_tokens``. ``tools`` is accepted (frozen
|
|
47
|
+
protocol) but not implemented yet."""
|
|
48
|
+
|
|
49
|
+
def __init__(self, model: Model, *, model_settings: dict | None = None,
|
|
50
|
+
strict_output: bool = True):
|
|
51
|
+
self.model = model
|
|
52
|
+
# deterministic-intent extraction: sampling heat off by default
|
|
53
|
+
self.model_settings = {'temperature': 0, **(model_settings or {})}
|
|
54
|
+
# strict keeps the schema in a server-side grammar; without it some
|
|
55
|
+
# providers inject the schema into the prompt behind the payload and
|
|
56
|
+
# break the cacheable prefix — hosts whose strict grammar rejects
|
|
57
|
+
# the schema shape turn it off here
|
|
58
|
+
self.strict_output = strict_output
|
|
59
|
+
|
|
60
|
+
async def run(self, *, instructions: str, result_schema: JSONSchema, content: str,
|
|
61
|
+
scope: str = None, tools: list = None, history: list = None,
|
|
62
|
+
feedback: Sequence[Issue] = None) -> AgentResult:
|
|
63
|
+
if tools:
|
|
64
|
+
raise NotImplementedError('tool-calling is deferred; see design.md §5.1')
|
|
65
|
+
free = result_schema.get('type') == 'string' # free-text mode: lib parses
|
|
66
|
+
wrapped = not free and result_schema.get('type') == 'array'
|
|
67
|
+
schema = _wrap(result_schema) if wrapped else result_schema
|
|
68
|
+
output_type = str if free else NativeOutput(StructuredDict(schema),
|
|
69
|
+
strict=self.strict_output)
|
|
70
|
+
agent = Agent(self.model, output_type=output_type, instructions=content)
|
|
71
|
+
result = await agent.run(_tail(instructions, scope, feedback),
|
|
72
|
+
message_history=history,
|
|
73
|
+
model_settings=self._model_settings())
|
|
74
|
+
await report_tokens(result.usage.input_tokens + result.usage.output_tokens)
|
|
75
|
+
data = result.output.get('items') if wrapped else result.output
|
|
76
|
+
return AgentResult(data=data, history=result.all_messages())
|
|
77
|
+
|
|
78
|
+
def _model_settings(self):
|
|
79
|
+
settings = dict(self.model_settings)
|
|
80
|
+
if _THINKING_OFF.match(getattr(self.model, 'model_name', '')):
|
|
81
|
+
settings['extra_body'] = {'enable_thinking': False}
|
|
82
|
+
return settings
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def enable_tracing() -> None:
|
|
86
|
+
"""Turn on the framework's native OTel span emission for every agent
|
|
87
|
+
(process-wide). Every adapter in this boundary provides this hook so
|
|
88
|
+
hosts enable tracing without naming the framework."""
|
|
89
|
+
Agent.instrument_all(True)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _json_body(body: dict) -> bytes:
|
|
93
|
+
"""SDK serialization style — ensure_ascii would double CJK bytes."""
|
|
94
|
+
return json.dumps(body, ensure_ascii=False, separators=(',', ':')).encode()
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _stripped_headers(headers: httpx.Headers) -> httpx.Headers:
|
|
98
|
+
"""Headers for a rewritten body: the length/encoding headers are stale."""
|
|
99
|
+
headers = headers.copy()
|
|
100
|
+
headers.pop('content-length', None)
|
|
101
|
+
headers.pop('content-encoding', None)
|
|
102
|
+
return headers
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def marked_request(request: httpx.Request) -> httpx.Request:
|
|
106
|
+
"""Rewrite a chat-completions request so its leading string-content
|
|
107
|
+
system message becomes a cache-marked content block (dashscope's
|
|
108
|
+
``cache_control`` explicit-cache extension). The marker must close
|
|
109
|
+
the shared payload at a message boundary — marking mid-message does
|
|
110
|
+
not match. Non-matching shapes pass through untouched."""
|
|
111
|
+
body = json.loads(request.read())
|
|
112
|
+
messages = body.get('messages')
|
|
113
|
+
if (messages and messages[0].get('role') == 'system'
|
|
114
|
+
and isinstance(messages[0].get('content'), str)):
|
|
115
|
+
messages[0]['content'] = [{'type': 'text', 'text': messages[0]['content'],
|
|
116
|
+
'cache_control': {'type': 'ephemeral'}}]
|
|
117
|
+
return httpx.Request(request.method, request.url,
|
|
118
|
+
headers=_stripped_headers(request.headers),
|
|
119
|
+
content=_json_body(body))
|
|
120
|
+
return request
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _unthinking(request: httpx.Request) -> httpx.Request:
|
|
124
|
+
"""Drops ``enable_thinking: false`` into a chat-completions body."""
|
|
125
|
+
body = json.loads(request.read())
|
|
126
|
+
return httpx.Request(request.method, request.url,
|
|
127
|
+
headers=_stripped_headers(request.headers),
|
|
128
|
+
content=_json_body(body | {'enable_thinking': False}))
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _rename_cache_write(response: httpx.Response) -> httpx.Response:
|
|
132
|
+
"""Relabel dashscope's cache-write usage field to the OpenAI name
|
|
133
|
+
pydantic-ai reads (``prompt_tokens_details.cache_write_tokens``);
|
|
134
|
+
without the relabel the count never reaches the usage object or
|
|
135
|
+
the span."""
|
|
136
|
+
body = response.content
|
|
137
|
+
if b'"cache_creation_input_tokens"' not in body:
|
|
138
|
+
return response
|
|
139
|
+
return httpx.Response(response.status_code,
|
|
140
|
+
headers=_stripped_headers(response.headers),
|
|
141
|
+
content=body.replace(b'"cache_creation_input_tokens"',
|
|
142
|
+
b'"cache_write_tokens"'))
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
class _WireTransport(httpx.AsyncHTTPTransport):
|
|
146
|
+
"""Applies provider wire transforms — requests before they leave,
|
|
147
|
+
chat-completion responses as they return; the openai client beneath
|
|
148
|
+
is untouched."""
|
|
149
|
+
|
|
150
|
+
def __init__(self, transforms: list):
|
|
151
|
+
super().__init__() # the base owns the connection pool
|
|
152
|
+
self.transforms = transforms
|
|
153
|
+
|
|
154
|
+
async def handle_async_request(self, request: httpx.Request) -> httpx.Response:
|
|
155
|
+
if request.url.path.endswith('/chat/completions'):
|
|
156
|
+
for transform in self.transforms:
|
|
157
|
+
request = transform(request)
|
|
158
|
+
response = await super().handle_async_request(request)
|
|
159
|
+
if 'application/json' in response.headers.get('content-type', ''):
|
|
160
|
+
await response.aread() # a rewritable body must be read first
|
|
161
|
+
return _rename_cache_write(response)
|
|
162
|
+
return response
|
|
163
|
+
return await super().handle_async_request(request)
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def openai_model(base_url: str, api_key: str, model: str, *, explicit_cache: bool = False,
|
|
167
|
+
thinking_off: bool = False,
|
|
168
|
+
http_client: httpx.AsyncClient | None = None):
|
|
169
|
+
"""A pydantic-ai Model for any OpenAI-compatible endpoint — the
|
|
170
|
+
provider mechanism lives inside the agents boundary; hosts pass
|
|
171
|
+
only deployment config. Requires the ``docspectra[pydantic-ai]`` extra.
|
|
172
|
+
``explicit_cache`` marks the leading system message as a provider
|
|
173
|
+
cache block (dashscope explicit cache): paired with a runner that
|
|
174
|
+
puts the shared payload in Agent instructions, the call that runs
|
|
175
|
+
first pays cache creation once and every call behind it hits; the
|
|
176
|
+
wire transport relabels dashscope's non-standard cache-write usage
|
|
177
|
+
field so the count also lands on the span. Providers that reject
|
|
178
|
+
the extension must leave it off.
|
|
179
|
+
``thinking_off`` drops ``enable_thinking: false`` into every request
|
|
180
|
+
— for consumers that build their own pydantic-ai agents outside the
|
|
181
|
+
runner's model_settings (pydantic-evals LLMJudge: tools +
|
|
182
|
+
tool_choice='required', which dashscope rejects while thinking).
|
|
183
|
+
A host ``http_client`` (proxies, connection tuning) is used
|
|
184
|
+
verbatim."""
|
|
185
|
+
try:
|
|
186
|
+
from pydantic_ai.models.openai import OpenAIChatModel
|
|
187
|
+
from pydantic_ai.providers.openai import OpenAIProvider
|
|
188
|
+
except ImportError as e:
|
|
189
|
+
raise ImportError('install the pydantic-ai extra: docspectra[pydantic-ai]') from e
|
|
190
|
+
if transforms := ([marked_request] if explicit_cache else []) \
|
|
191
|
+
+ ([_unthinking] if thinking_off else []):
|
|
192
|
+
http_client = httpx.AsyncClient(transport=_WireTransport(transforms))
|
|
193
|
+
return OpenAIChatModel(model, provider=OpenAIProvider(
|
|
194
|
+
base_url=base_url, api_key=api_key, http_client=http_client))
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def _wrap(schema: JSONSchema) -> JSONSchema:
|
|
198
|
+
return {'type': 'object', 'properties': {'items': schema},
|
|
199
|
+
'required': ['items'], 'additionalProperties': False}
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def _tail(instructions: str, scope, feedback) -> str:
|
|
203
|
+
parts = [instructions] if instructions else []
|
|
204
|
+
if scope is not None:
|
|
205
|
+
parts.append(f'\n\n---\nAssigned material:\n{scope}')
|
|
206
|
+
if feedback:
|
|
207
|
+
lines = '\n'.join(f'- {i.path}: {i.message}' for i in feedback)
|
|
208
|
+
parts.append(f'\n\n---\nValidation errors to fix:\n{lines}')
|
|
209
|
+
return ''.join(parts)
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""File → text. Provider engines as optional extras (``docspectra[pdf-inspect]``,
|
|
2
|
+
``docspectra[anydoc]``, ``docspectra[mineru]``), magic-byte type routing,
|
|
3
|
+
and file-level decline chains: ``DOCSPECTRA_PARSER_CONFIG`` orders each type's
|
|
4
|
+
chain and a provider whose gate fires on a specific file declines, letting the
|
|
5
|
+
next engine take it."""
|
|
6
|
+
|
|
7
|
+
from docspectra.parser.contracts import (Attempt, Declined, FileInput,
|
|
8
|
+
ImageAsset, MAX_ASSETS, MAX_ASSET_BYTES,
|
|
9
|
+
ParseResult)
|
|
10
|
+
from docspectra.parser.core import (
|
|
11
|
+
ParseDeclinedError,
|
|
12
|
+
Provider,
|
|
13
|
+
configure,
|
|
14
|
+
parse,
|
|
15
|
+
parse_path,
|
|
16
|
+
provider,
|
|
17
|
+
register,
|
|
18
|
+
)
|
|
19
|
+
from docspectra.parser.sniff import sniff, validate_declared
|
|
20
|
+
|
|
21
|
+
__all__ = [
|
|
22
|
+
'Attempt',
|
|
23
|
+
'Declined',
|
|
24
|
+
'FileInput',
|
|
25
|
+
'ImageAsset',
|
|
26
|
+
'MAX_ASSETS',
|
|
27
|
+
'MAX_ASSET_BYTES',
|
|
28
|
+
'ParseDeclinedError',
|
|
29
|
+
'ParseResult',
|
|
30
|
+
'Provider',
|
|
31
|
+
'configure',
|
|
32
|
+
'parse',
|
|
33
|
+
'parse_path',
|
|
34
|
+
'provider',
|
|
35
|
+
'register',
|
|
36
|
+
'sniff',
|
|
37
|
+
'validate_declared',
|
|
38
|
+
]
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
"""Parser contracts: result shapes and control-flow values shared by core
|
|
2
|
+
and providers."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
from dataclasses import dataclass, field
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Literal
|
|
9
|
+
|
|
10
|
+
FileInput = bytes | str | Path
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass
|
|
14
|
+
class ImageAsset:
|
|
15
|
+
"""One embedded image carried faithfully, uninterpreted.
|
|
16
|
+
|
|
17
|
+
``name`` matches the ```` reference in ParseResult.text;
|
|
18
|
+
``source`` is engine provenance ('pdf p2', 'word/media/image1.png').
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
name: str
|
|
22
|
+
mime: str
|
|
23
|
+
data: bytes
|
|
24
|
+
source: str
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
# payload-budget policy for ParseResult.images — every provider skips
|
|
28
|
+
# assets over these limits and records the skip count in meta
|
|
29
|
+
MAX_ASSETS = 50
|
|
30
|
+
MAX_ASSET_BYTES = 5 * 1024 * 1024
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass
|
|
34
|
+
class Attempt:
|
|
35
|
+
"""One provider's outcome on a file, recorded by core on success and
|
|
36
|
+
failure alike — the eval suite reads routing through it."""
|
|
37
|
+
|
|
38
|
+
provider: str
|
|
39
|
+
outcome: Literal['success', 'declined', 'error']
|
|
40
|
+
reason: str | None = None
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
@dataclass
|
|
44
|
+
class Declined:
|
|
45
|
+
"""Expected control flow: this engine cannot read *this* file, or
|
|
46
|
+
read it into a known false-positive shape (a gate fired). Never an
|
|
47
|
+
exception — the chain moves to the next provider."""
|
|
48
|
+
|
|
49
|
+
reason: str
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
@dataclass
|
|
53
|
+
class ParseResult:
|
|
54
|
+
"""The outcome of a successful parse. ``format`` is constrained —
|
|
55
|
+
extend the Literal, never free-form. ``meta`` is provider diagnostics
|
|
56
|
+
(gate hits, page counts, task ids, timings). ``provider`` and
|
|
57
|
+
``attempts`` are stamped by core, not the provider."""
|
|
58
|
+
|
|
59
|
+
text: str
|
|
60
|
+
provider: str = '' # engine that accepted the file (core-stamped)
|
|
61
|
+
format: Literal['markdown'] = 'markdown'
|
|
62
|
+
images: list[ImageAsset] = field(default_factory=list)
|
|
63
|
+
attempts: list[Attempt] = field(default_factory=list)
|
|
64
|
+
meta: dict = field(default_factory=dict)
|
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
"""Provider registry, decline chains, and the async parse entry.
|
|
2
|
+
|
|
3
|
+
Core carries zero parsing dependencies: built-in providers are in-tree
|
|
4
|
+
modules under ``docspectra.providers``, try-imported at init — an absent
|
|
5
|
+
extra is skipped silently (an installed-but-broken provider is recorded
|
|
6
|
+
and surfaced in config errors instead). Engines declare themselves with
|
|
7
|
+
:func:`provider`; chains are ordered by ``DOCSPECTRA_PARSER_CONFIG`` (a
|
|
8
|
+
JSON env var mapping a document type to an ordered provider list),
|
|
9
|
+
falling back to the derived default chain — the canonical order of
|
|
10
|
+
installed providers that declared the type. Declines are return values,
|
|
11
|
+
hard errors fall through; chain exhaustion raises
|
|
12
|
+
:class:`ParseDeclinedError` with every attempt aggregated."""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import importlib
|
|
17
|
+
import json
|
|
18
|
+
import os
|
|
19
|
+
from dataclasses import dataclass
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import Awaitable, Callable, Sequence
|
|
22
|
+
|
|
23
|
+
from docspectra.parser.contracts import Attempt, Declined, ParseResult
|
|
24
|
+
from docspectra.parser.sniff import sniff, validate_declared
|
|
25
|
+
|
|
26
|
+
_BUILTIN_MODULES = ('txt', 'pdf_inspect', 'anydoc', 'mineru')
|
|
27
|
+
_BUILTIN_NAMES = {m.replace('_', '-') for m in _BUILTIN_MODULES}
|
|
28
|
+
_CONFIG_ENV = 'DOCSPECTRA_PARSER_CONFIG'
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass
|
|
32
|
+
class Provider:
|
|
33
|
+
"""One engine: the async parse entry plus the types it declares."""
|
|
34
|
+
|
|
35
|
+
name: str
|
|
36
|
+
types: tuple[str, ...]
|
|
37
|
+
parse: Callable[..., Awaitable[ParseResult | Declined]]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class ParseDeclinedError(Exception):
|
|
41
|
+
"""Every provider in the chain declined or errored on this file.
|
|
42
|
+
|
|
43
|
+
``attempts`` aggregates every outcome. A hard error raised through
|
|
44
|
+
the chain is chained as the cause.
|
|
45
|
+
"""
|
|
46
|
+
|
|
47
|
+
def __init__(self, attempts: list[Attempt]):
|
|
48
|
+
self.attempts = attempts
|
|
49
|
+
detail = '; '.join(
|
|
50
|
+
f"{a.provider}: {a.outcome}" + (f' ({a.reason})' if a.reason else '')
|
|
51
|
+
for a in attempts
|
|
52
|
+
)
|
|
53
|
+
super().__init__(f'every provider declined: {detail}')
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
_providers: dict[str, Provider] = {} # keyed by normalized (hyphen) engine name
|
|
57
|
+
_builtins_loaded = False
|
|
58
|
+
_config: dict | None = None # effective chain config; resolved from env on first use
|
|
59
|
+
_import_errors: dict[str, str] = {} # provider module → failure, when installed-but-broken
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _norm(name: str) -> str:
|
|
63
|
+
return name.replace('_', '-')
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def register(name: str, types: Sequence[str], fn) -> None:
|
|
67
|
+
"""Programmatic registration, same surface as the decorator. Names are
|
|
68
|
+
normalized (hyphen↔underscore) and registration is last-write-wins."""
|
|
69
|
+
_providers[_norm(name)] = Provider(_norm(name), tuple(types), fn)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def provider(name: str, types: Sequence[str]):
|
|
73
|
+
"""Declare an engine around an async ``parse(data, *, name=None)
|
|
74
|
+
-> ParseResult | Declined``."""
|
|
75
|
+
|
|
76
|
+
def deco(fn):
|
|
77
|
+
register(name, types, fn)
|
|
78
|
+
return fn
|
|
79
|
+
|
|
80
|
+
return deco
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def ensure_builtins() -> None:
|
|
84
|
+
global _builtins_loaded
|
|
85
|
+
if _builtins_loaded:
|
|
86
|
+
return
|
|
87
|
+
_builtins_loaded = True
|
|
88
|
+
for module in _BUILTIN_MODULES:
|
|
89
|
+
try:
|
|
90
|
+
importlib.import_module(f'docspectra.providers.{module}')
|
|
91
|
+
except ImportError as e:
|
|
92
|
+
if not isinstance(e, ModuleNotFoundError) \
|
|
93
|
+
or (e.name or '').startswith('docspectra'):
|
|
94
|
+
_import_errors[_norm(module)] = f'{type(e).__name__}: {e}'
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def configure(config: dict | None = None) -> None:
|
|
98
|
+
"""Pin chains to an explicit ``{type: [engine, ...]}`` mapping, or
|
|
99
|
+
reload them from DOCSPECTRA_PARSER_CONFIG (None, validated here —
|
|
100
|
+
the fail-fast hook a host calls at startup)."""
|
|
101
|
+
global _config
|
|
102
|
+
ensure_builtins()
|
|
103
|
+
_config = _validate(config) if config is not None else _env_config()
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
async def parse(data: bytes, *, name: str | None = None) -> ParseResult:
|
|
107
|
+
"""Parse one file (bytes are the contract) into a ParseResult,
|
|
108
|
+
walking the type's decline chain. Raises ValueError on unroutable
|
|
109
|
+
input, on a declared binary extension whose content fails magic
|
|
110
|
+
verification, or when no provider is installed; ParseDeclinedError
|
|
111
|
+
when every provider declines or errors."""
|
|
112
|
+
ensure_builtins()
|
|
113
|
+
validate_declared(data, name) # declared-binary gate, before any provider
|
|
114
|
+
doc_type = sniff(data, name)
|
|
115
|
+
if doc_type is None:
|
|
116
|
+
raise ValueError(f'cannot determine document type of {name or "input"}')
|
|
117
|
+
chain = _chain_for(doc_type)
|
|
118
|
+
if not chain:
|
|
119
|
+
raise ValueError(f'no provider installed for document type {doc_type!r}')
|
|
120
|
+
attempts: list[Attempt] = []
|
|
121
|
+
cause = None
|
|
122
|
+
for provider_name in chain:
|
|
123
|
+
p = _providers[provider_name]
|
|
124
|
+
try:
|
|
125
|
+
result = await p.parse(data, name=name)
|
|
126
|
+
except Exception as e:
|
|
127
|
+
attempts.append(Attempt(p.name, 'error', f'{type(e).__name__}: {e}'))
|
|
128
|
+
cause = cause or e
|
|
129
|
+
continue
|
|
130
|
+
if isinstance(result, Declined):
|
|
131
|
+
attempts.append(Attempt(p.name, 'declined', result.reason))
|
|
132
|
+
continue
|
|
133
|
+
result.provider = p.name
|
|
134
|
+
result.attempts = [*attempts, Attempt(p.name, 'success')]
|
|
135
|
+
return result
|
|
136
|
+
raise ParseDeclinedError(attempts) from cause
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
async def parse_path(path: str | Path) -> ParseResult:
|
|
140
|
+
"""Convenience wrapper: read the file and parse it, name from the path."""
|
|
141
|
+
path = Path(path)
|
|
142
|
+
return await parse(path.read_bytes(), name=path.name)
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def _chain_for(doc_type: str) -> list[str]:
|
|
146
|
+
config = _effective_config()
|
|
147
|
+
if doc_type in config:
|
|
148
|
+
return list(config[doc_type])
|
|
149
|
+
# derived default chain: canonical order of installed providers that
|
|
150
|
+
# declared the type — extends automatically as built-ins are added
|
|
151
|
+
return [p.name for p in _providers.values() if doc_type in p.types]
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _effective_config() -> dict:
|
|
155
|
+
global _config
|
|
156
|
+
if _config is None:
|
|
157
|
+
_config = _env_config()
|
|
158
|
+
return _config
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _env_config() -> dict:
|
|
162
|
+
raw = os.environ.get(_CONFIG_ENV)
|
|
163
|
+
if not raw:
|
|
164
|
+
return {}
|
|
165
|
+
try:
|
|
166
|
+
config = json.loads(raw)
|
|
167
|
+
except json.JSONDecodeError as e:
|
|
168
|
+
raise ValueError(f'{_CONFIG_ENV} is not valid JSON: {e}') from e
|
|
169
|
+
if not isinstance(config, dict):
|
|
170
|
+
raise ValueError(f'{_CONFIG_ENV} must be a JSON object of type → engine list')
|
|
171
|
+
return _validate(config)
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def _validate(config: dict) -> dict:
|
|
175
|
+
validated = {}
|
|
176
|
+
for doc_type, names in config.items():
|
|
177
|
+
if not isinstance(names, list) or not all(isinstance(n, str) for n in names):
|
|
178
|
+
raise ValueError(f'{_CONFIG_ENV}[{doc_type!r}] must be a list of engine names')
|
|
179
|
+
normed = []
|
|
180
|
+
for name in names:
|
|
181
|
+
key = _norm(name)
|
|
182
|
+
if key not in _providers:
|
|
183
|
+
if key in _BUILTIN_NAMES:
|
|
184
|
+
detail = f' ({_import_errors[key]})' if key in _import_errors else ''
|
|
185
|
+
raise ValueError(
|
|
186
|
+
f'provider {name!r} is configured but not installed{detail} — '
|
|
187
|
+
f'install the matching extra (docspectra[{key}])')
|
|
188
|
+
raise ValueError(
|
|
189
|
+
f'unknown parser provider {name!r}; known: {sorted(_providers)}')
|
|
190
|
+
normed.append(key)
|
|
191
|
+
validated[doc_type] = normed
|
|
192
|
+
return validated
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
"""Document-type sniffing: magic bytes first, extension fallback, no
|
|
2
|
+
dependencies. Bytes win on conflict for routing; a filename declaring a
|
|
3
|
+
binary extension is gated by :func:`validate_declared` before any
|
|
4
|
+
provider runs.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import io
|
|
10
|
+
import zipfile
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
_PDF = b'%PDF-'
|
|
14
|
+
_ZIP = b'PK\x03\x04'
|
|
15
|
+
_OLE2 = b'\xd0\xcf\x11\xe0\xa1\xb1\x1a\xe1'
|
|
16
|
+
|
|
17
|
+
_IMAGES = (
|
|
18
|
+
(b'\xff\xd8\xff', 'jpg'),
|
|
19
|
+
(b'\x89PNG\r\n\x1a\n', 'png'),
|
|
20
|
+
(b'GIF87a', 'gif'),
|
|
21
|
+
(b'GIF89a', 'gif'),
|
|
22
|
+
(b'BM', 'bmp'),
|
|
23
|
+
(b'II*\x00', 'tiff'),
|
|
24
|
+
(b'MM\x00*', 'tiff'),
|
|
25
|
+
(b'\x00\x00\x00\x0cjP \r\n\x87\n', 'jp2'),
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
# extension aliases, normalized to the canonical type
|
|
29
|
+
_ALIASES = {'jpeg': 'jpg', 'tif': 'tiff', 'htm': 'html'}
|
|
30
|
+
|
|
31
|
+
# OLE2 magic is coarse — the legacy office extensions form one family
|
|
32
|
+
_FAMILY = {'doc': frozenset({'doc', 'ppt', 'xls'})}
|
|
33
|
+
|
|
34
|
+
# types whose declared extension must verify against content magic: the
|
|
35
|
+
# magic tables' codomain (zip parts + images + webp), family expanded
|
|
36
|
+
_BINARY_TYPES = ({'pdf', 'docx', 'xlsx', 'pptx', 'webp'}
|
|
37
|
+
| {t for _, t in _IMAGES} | set().union(*_FAMILY.values()))
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def sniff(data: bytes, name: str | None = None) -> str | None:
|
|
41
|
+
"""Route bytes to a canonical document type. ``name`` feeds the
|
|
42
|
+
extension fallback when magic is inconclusive; None when neither
|
|
43
|
+
yields a type."""
|
|
44
|
+
if data.startswith(_PDF):
|
|
45
|
+
return 'pdf'
|
|
46
|
+
if data.startswith(_ZIP):
|
|
47
|
+
return _zip_type(data) or _extension_type(name)
|
|
48
|
+
if data.startswith(_OLE2):
|
|
49
|
+
return 'doc' # coarse: doc/ppt/xls share the legacy chain
|
|
50
|
+
if t := _image_type(data):
|
|
51
|
+
return t
|
|
52
|
+
return _extension_type(name)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def validate_declared(data: bytes, name: str | None = None) -> None:
|
|
56
|
+
"""Gate a declared binary extension against content magic: raise
|
|
57
|
+
ValueError('file content is not a XXX') — XXX the extension's
|
|
58
|
+
canonical type — when the bytes do not verify. Text and unknown
|
|
59
|
+
extensions are never validated; the extension fallback owns them."""
|
|
60
|
+
declared = _extension_type(name)
|
|
61
|
+
if declared not in _BINARY_TYPES or _matches(data, declared):
|
|
62
|
+
return
|
|
63
|
+
raise ValueError(f'file content is not a {declared}')
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _matches(data: bytes, declared: str) -> bool:
|
|
67
|
+
magic = sniff(data) # no name: pure magic, extension fallback off
|
|
68
|
+
return magic is not None and declared in _FAMILY.get(magic, {magic})
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _zip_type(data: bytes) -> str | None:
|
|
72
|
+
try:
|
|
73
|
+
with zipfile.ZipFile(io.BytesIO(data)) as z:
|
|
74
|
+
for info in z.infolist():
|
|
75
|
+
if info.filename.startswith('word/'):
|
|
76
|
+
return 'docx'
|
|
77
|
+
if info.filename.startswith('xl/'):
|
|
78
|
+
return 'xlsx'
|
|
79
|
+
if info.filename.startswith('ppt/'):
|
|
80
|
+
return 'pptx'
|
|
81
|
+
except (zipfile.BadZipFile, OSError):
|
|
82
|
+
return None
|
|
83
|
+
return None
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _image_type(data: bytes) -> str | None:
|
|
87
|
+
if data.startswith(b'RIFF') and data[8:12] == b'WEBP':
|
|
88
|
+
return 'webp'
|
|
89
|
+
for magic, t in _IMAGES:
|
|
90
|
+
if data.startswith(magic):
|
|
91
|
+
return t
|
|
92
|
+
return None
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _extension_type(name: str | None) -> str | None:
|
|
96
|
+
ext = Path(name or '').suffix.lower().lstrip('.')
|
|
97
|
+
return _ALIASES.get(ext, ext) or None
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
"""Built-in provider engines. Core try-imports each module at init; a
|
|
2
|
+
module whose extra is absent fails to import and is skipped silently."""
|
|
3
|
+
|
|
4
|
+
from functools import wraps
|
|
5
|
+
|
|
6
|
+
from opentelemetry.trace import get_tracer
|
|
7
|
+
|
|
8
|
+
from docspectra.parser import Declined, ParseResult
|
|
9
|
+
|
|
10
|
+
# shared by the provider engines; spans are no-ops until a host
|
|
11
|
+
# registers a tracer provider
|
|
12
|
+
tracer = get_tracer('docspectra')
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def traced(fn):
|
|
16
|
+
"""One span per parse call, named after the engine module — engine
|
|
17
|
+
registrations and their span names can then never drift apart. The
|
|
18
|
+
span carries the outcome bookkeeping (outcome, decline reason,
|
|
19
|
+
input bytes, the meta diagnostics — never text: spans are an
|
|
20
|
+
account ledger like the sidecar, PII stays out; batch_id stays off
|
|
21
|
+
too, it points at remote state holding the file)."""
|
|
22
|
+
span_name = f'{fn.__module__.rsplit(".", 1)[-1]}.parse'
|
|
23
|
+
|
|
24
|
+
@wraps(fn)
|
|
25
|
+
async def wrapper(*args, **kwargs):
|
|
26
|
+
with tracer.start_as_current_span(span_name) as span:
|
|
27
|
+
result = await fn(*args, **kwargs)
|
|
28
|
+
if args and isinstance(args[0], (bytes, bytearray)):
|
|
29
|
+
span.set_attribute('parse.bytes', len(args[0]))
|
|
30
|
+
if isinstance(result, Declined):
|
|
31
|
+
span.set_attribute('parse.outcome', 'declined')
|
|
32
|
+
span.set_attribute('parse.reason', result.reason)
|
|
33
|
+
else:
|
|
34
|
+
span.set_attribute('parse.outcome', 'success')
|
|
35
|
+
span.set_attribute('parse.chars', len(result.text))
|
|
36
|
+
span.set_attribute('parse.images', len(result.images))
|
|
37
|
+
for key, value in result.meta.items():
|
|
38
|
+
if key != 'batch_id':
|
|
39
|
+
span.set_attribute(f'parse.{key}', value)
|
|
40
|
+
return result
|
|
41
|
+
|
|
42
|
+
return wrapper
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
"""Shared plumbing behind the local pdf engines (pdf-inspect and anydoc):
|
|
2
|
+
the post-classify tail — two byte-side decline gates over pypdf (the
|
|
3
|
+
word-strip fingerprint and the content-stream ratio) plus one text-side
|
|
4
|
+
gate over the engine markdown (the table wall), the XObject image
|
|
5
|
+
collection, and the markdown clean-up — pure pypdf plus regex, so either
|
|
6
|
+
extra can host it without the other's engine."""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import io
|
|
11
|
+
import re
|
|
12
|
+
|
|
13
|
+
from pypdf import PdfReader
|
|
14
|
+
|
|
15
|
+
from docspectra.parser import (Declined, ImageAsset, MAX_ASSETS,
|
|
16
|
+
MAX_ASSET_BYTES, ParseResult)
|
|
17
|
+
|
|
18
|
+
RATIO_THRESHOLD = 200 # B/char; measured 4,968 vs clean max 82 (parser.md §1)
|
|
19
|
+
STRIP_W, STRIP_H = 80, 100 # pixel dims of word-shaped raster strips
|
|
20
|
+
TABLE_SHARE_MIN = 0.7 # of all text chars; judgment call — measured wall
|
|
21
|
+
# 0.99, corpus real tables a minority share (parser.md §1)
|
|
22
|
+
WALL_ROW_CHARS = 500 # chars in one row; the engine's own paragraph-wall
|
|
23
|
+
# constant, measured wall row ~1,300 (parser.md §1)
|
|
24
|
+
_MIME = {'/DCTDecode': 'image/jpeg', '/JPXDecode': 'image/jp2',
|
|
25
|
+
'/CCITTFaxDecode': 'image/tiff'}
|
|
26
|
+
|
|
27
|
+
_U_TAG = re.compile(r'</?u>')
|
|
28
|
+
_BULLET = re.compile(r'^l ', re.MULTILINE) # the engine renders bullets as 'l'
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def scan(data: bytes, text: str, **meta) -> ParseResult | Declined:
|
|
32
|
+
"""The shared post-classify tail: the two pypdf gates and the
|
|
33
|
+
text-side table wall, image collection, and the cleaned ParseResult.
|
|
34
|
+
``meta`` adds engine-specific diagnostics between ``pages`` and
|
|
35
|
+
``skipped_assets``."""
|
|
36
|
+
reader = PdfReader(io.BytesIO(data))
|
|
37
|
+
for page in reader.pages:
|
|
38
|
+
if strip_hit(page):
|
|
39
|
+
return Declined('fingerprint_gate')
|
|
40
|
+
if ratio_hit(reader, len(text)) >= RATIO_THRESHOLD:
|
|
41
|
+
return Declined('ratio_gate')
|
|
42
|
+
if table_wall_hit(text):
|
|
43
|
+
return Declined('table_wall_gate')
|
|
44
|
+
images, skipped = collect_images(reader)
|
|
45
|
+
return ParseResult(text=clean(text), images=images,
|
|
46
|
+
meta={'pages': len(reader.pages), **meta,
|
|
47
|
+
'skipped_assets': skipped})
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def clean(markdown: str) -> str:
|
|
51
|
+
return _BULLET.sub('• ', _U_TAG.sub('', markdown))
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def table_wall_hit(text: str) -> bool:
|
|
55
|
+
"""One markdown table swallowing the page's flowing prose — the
|
|
56
|
+
layout-band false positive of geometry-based table detection (a
|
|
57
|
+
borderless template frame reconstructed as a grid, reading order
|
|
58
|
+
scrambled). Both conditions matter: a document that merely carries
|
|
59
|
+
a real table stays, and a page of genuine data cells (high table
|
|
60
|
+
share, short rows) stays."""
|
|
61
|
+
lines = [line for line in text.splitlines() if line.strip()]
|
|
62
|
+
rows = [line for line in lines
|
|
63
|
+
if line.startswith('|') and line.strip('|- ')]
|
|
64
|
+
total = sum(map(len, lines))
|
|
65
|
+
return (bool(rows)
|
|
66
|
+
and sum(map(len, rows)) / total >= TABLE_SHARE_MIN
|
|
67
|
+
and max(map(len, rows)) > WALL_ROW_CHARS)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def strip_hit(page) -> bool:
|
|
71
|
+
"""A wide flat image XObject — a rasterized text strip among a real
|
|
72
|
+
text layer (the raster-strip class). Missing height counts as the threshold."""
|
|
73
|
+
return any(obj.get('/Width', 0) >= STRIP_W and obj.get('/Height', STRIP_H) <= STRIP_H
|
|
74
|
+
for _, obj in _iter_images(page))
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def ratio_hit(reader: PdfReader, nchars: int) -> float:
|
|
78
|
+
"""Content-stream bytes per delivered text char; outlined glyphs
|
|
79
|
+
(the outlined-glyph class) inflate the stream without any text to show for it."""
|
|
80
|
+
nbytes = 0
|
|
81
|
+
for page in reader.pages:
|
|
82
|
+
if (contents := page.get_contents()) is not None:
|
|
83
|
+
nbytes += len(contents.get_data())
|
|
84
|
+
return nbytes / nchars if nchars else float('inf')
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def collect_images(reader: PdfReader) -> tuple[list[ImageAsset], int]:
|
|
88
|
+
images, skipped = [], 0
|
|
89
|
+
for page_no, page in enumerate(reader.pages, 1):
|
|
90
|
+
for key, obj in _iter_images(page):
|
|
91
|
+
if len(images) >= MAX_ASSETS:
|
|
92
|
+
skipped += 1
|
|
93
|
+
continue
|
|
94
|
+
try:
|
|
95
|
+
data = obj.get_data()
|
|
96
|
+
except Exception:
|
|
97
|
+
skipped += 1
|
|
98
|
+
continue
|
|
99
|
+
if len(data) > MAX_ASSET_BYTES:
|
|
100
|
+
skipped += 1
|
|
101
|
+
continue
|
|
102
|
+
# names are opaque per-engine keys: the engine emits no
|
|
103
|
+
# image references, so nothing in the text reconciles to them
|
|
104
|
+
images.append(ImageAsset(
|
|
105
|
+
name=f'p{page_no}-{str(key).lstrip("/")}',
|
|
106
|
+
mime=_MIME.get(str(obj.get('/Filter', '')), 'application/octet-stream'),
|
|
107
|
+
data=data, source=f'pdf p{page_no} {key}'))
|
|
108
|
+
return images, skipped
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _iter_images(page):
|
|
112
|
+
"""Yield (key, ImageObject) pairs from the page's own XObject dict —
|
|
113
|
+
the measured walk behind the fingerprint gate (parser.md §1)."""
|
|
114
|
+
resources = _deref(page.get('/Resources') or {})
|
|
115
|
+
for key, ref in _deref(resources.get('/XObject') or {}).items():
|
|
116
|
+
obj = _deref(ref)
|
|
117
|
+
if obj.get('/Subtype') == '/Image':
|
|
118
|
+
yield key, obj
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _deref(obj):
|
|
122
|
+
return obj.get_object() if hasattr(obj, 'get_object') else obj
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
"""Local superset engine (extra: anydoc): firecrawl-anydoc (Rust) over
|
|
2
|
+
pdf plus every office family. The pdf branch runs the shared decline
|
|
3
|
+
gates — classify from the engine's own ImageBased/Scanned rejection,
|
|
4
|
+
the rest from ``_pdfscan``; office formats (doc/docx, xls/xlsx, ppt/pptx,
|
|
5
|
+
odt/ods/odp, rtf, epub, csv) convert to markdown with their embedded
|
|
6
|
+
images carried as ImageAssets. Docx headers and footers are page
|
|
7
|
+
furniture the engine drops by design."""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import anydoc
|
|
12
|
+
|
|
13
|
+
from docspectra.parser import (Declined, ImageAsset, MAX_ASSETS,
|
|
14
|
+
MAX_ASSET_BYTES, ParseResult, provider,
|
|
15
|
+
sniff)
|
|
16
|
+
from docspectra.providers import traced
|
|
17
|
+
from docspectra.providers._pdfscan import scan
|
|
18
|
+
|
|
19
|
+
# sniff type → engine Format for the office branch; None lets the engine
|
|
20
|
+
# detect from the bytes (the OLE2 family — 'doc' covers legacy doc/xls/ppt
|
|
21
|
+
# alike, and xls has no Format name of its own). An explicit map, not a
|
|
22
|
+
# passthrough: a sniff/engine vocabulary drift must fail here, not inside
|
|
23
|
+
# the engine as a runtime error.
|
|
24
|
+
_FORMAT: dict[str, str | None] = {
|
|
25
|
+
'docx': 'docx', 'doc': None,
|
|
26
|
+
'ppt': 'ppt', 'pptx': 'pptx', 'xls': None, 'xlsx': 'xlsx',
|
|
27
|
+
'odt': 'odt', 'ods': 'ods', 'odp': 'odp',
|
|
28
|
+
'rtf': 'rtf', 'epub': 'epub', 'csv': 'csv',
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@provider('anydoc', ('pdf', *_FORMAT))
|
|
33
|
+
@traced
|
|
34
|
+
async def parse_anydoc(data: bytes, *, name: str | None = None) -> ParseResult | Declined:
|
|
35
|
+
# core already sniffed for routing; re-sniffed here because the
|
|
36
|
+
# provider contract carries (data, name) and the engine Format and
|
|
37
|
+
# the pdf/office branch both hang off the type
|
|
38
|
+
if (doc_type := sniff(data, name)) == 'pdf':
|
|
39
|
+
return _parse_pdf(data)
|
|
40
|
+
return _parse_office(data, doc_type)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _parse_pdf(data: bytes) -> ParseResult | Declined:
|
|
44
|
+
try:
|
|
45
|
+
text = anydoc.to_markdown_bytes(data)
|
|
46
|
+
except anydoc.UnsupportedError:
|
|
47
|
+
# the engine's own rejection: ImageBased/Scanned pages need OCR
|
|
48
|
+
return Declined('classify_gate')
|
|
49
|
+
return scan(data, text)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _parse_office(data: bytes, doc_type: str) -> ParseResult:
|
|
53
|
+
# signature-less formats (csv) must name their format explicitly
|
|
54
|
+
fmt = _FORMAT[doc_type]
|
|
55
|
+
text = anydoc.to_markdown_bytes(data, fmt)
|
|
56
|
+
images, skipped = [], 0
|
|
57
|
+
for asset in anydoc.to_document(data, fmt).assets:
|
|
58
|
+
if len(images) >= MAX_ASSETS or len(asset.data) > MAX_ASSET_BYTES:
|
|
59
|
+
skipped += 1
|
|
60
|
+
continue
|
|
61
|
+
images.append(ImageAsset(
|
|
62
|
+
name=asset.origin_part, mime=asset.media_type, data=asset.data,
|
|
63
|
+
source=f'anydoc {asset.origin_part}'))
|
|
64
|
+
return ParseResult(text=text, images=images,
|
|
65
|
+
meta={'skipped_assets': skipped})
|
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
"""Cloud engine (extra: mineru): MinerU extraction service via signed
|
|
2
|
+
upload + batch poll, rate-limited by dual xtremeflow lanes (submit
|
|
3
|
+
50/min, status 1000/min — the official ceilings, per-minute numbers
|
|
4
|
+
divided by 60 into float rps). 429/5xx raise RetryException and retry
|
|
5
|
+
through auto_backoff; other 4xx (quota, auth) surface as hard errors.
|
|
6
|
+
Results are cached in-process keyed by (sha256, model_version) — memory
|
|
7
|
+
only, never disk, evicted by count and byte budget."""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import asyncio
|
|
12
|
+
import hashlib
|
|
13
|
+
import io
|
|
14
|
+
import os
|
|
15
|
+
import time
|
|
16
|
+
import zipfile
|
|
17
|
+
from collections import OrderedDict
|
|
18
|
+
|
|
19
|
+
import httpx
|
|
20
|
+
|
|
21
|
+
from docspectra.parser import (ImageAsset, MAX_ASSETS, MAX_ASSET_BYTES,
|
|
22
|
+
ParseResult, provider)
|
|
23
|
+
from docspectra.providers import traced
|
|
24
|
+
|
|
25
|
+
from xtremeflow.scheduler.rate_limit import RetryException, auto_backoff
|
|
26
|
+
from xtremeflow.scheduler.request import RequestRateScheduler # deep import: not re-exported
|
|
27
|
+
|
|
28
|
+
BASE = 'https://mineru.net/api/v4'
|
|
29
|
+
POLL_INTERVAL = 2.0
|
|
30
|
+
_CACHE_MAX = 8 # ~1 MB per file — roughly 8 MB resident
|
|
31
|
+
_CACHE_BYTES = 8 * 1024 * 1024
|
|
32
|
+
_UPLOAD_RETRIES = 3
|
|
33
|
+
_MIME = {'jpg': 'image/jpeg', 'jpeg': 'image/jpeg', 'png': 'image/png',
|
|
34
|
+
'gif': 'image/gif', 'webp': 'image/webp', 'svg': 'image/svg+xml',
|
|
35
|
+
'bmp': 'image/bmp', 'tiff': 'image/tiff'}
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class MineruError(Exception):
|
|
39
|
+
"""Hard error: quota, auth, timeout, or a failed extraction — falls
|
|
40
|
+
through the chain like any provider exception."""
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
@provider('mineru', ('pdf', 'doc', 'docx', 'ppt', 'pptx', 'xls', 'xlsx',
|
|
44
|
+
'png', 'jpg', 'jpeg', 'jp2', 'webp', 'gif', 'bmp', 'tiff'))
|
|
45
|
+
@traced
|
|
46
|
+
async def parse_mineru(data: bytes, *, name: str | None = None) -> ParseResult:
|
|
47
|
+
token = os.environ.get('MINERU_API_TOKEN')
|
|
48
|
+
if not token:
|
|
49
|
+
raise MineruError('MINERU_API_TOKEN is not set')
|
|
50
|
+
model = os.environ.get('MINERU_MODEL_VERSION', 'pipeline')
|
|
51
|
+
key = (hashlib.sha256(data).hexdigest(), model)
|
|
52
|
+
if cached := _cache_get(key):
|
|
53
|
+
# per-call facade: share the immutable payload, never the cached
|
|
54
|
+
# meta (a batch id for a batch this call never submitted)
|
|
55
|
+
return ParseResult(text=cached.text, images=cached.images,
|
|
56
|
+
meta={'model_version': model, 'cache_hit': True})
|
|
57
|
+
timeout = float(os.environ.get('MINERU_TIMEOUT', '300'))
|
|
58
|
+
submit_lane, status_lane = _lanes()
|
|
59
|
+
async with httpx.AsyncClient(timeout=30.0) as client:
|
|
60
|
+
task = await submit_lane.start_task(_apply_url(client, token, model, name))
|
|
61
|
+
batch_id, file_url = await task
|
|
62
|
+
await _upload(client, file_url, data)
|
|
63
|
+
zip_url = await _poll(client, token, batch_id, timeout, status_lane)
|
|
64
|
+
if not zip_url:
|
|
65
|
+
raise MineruError(f'mineru batch {batch_id} finished without a result zip')
|
|
66
|
+
result_zip = await _download(client, zip_url)
|
|
67
|
+
markdown, images, skipped = _unzip(result_zip)
|
|
68
|
+
result = ParseResult(
|
|
69
|
+
text=markdown, images=images,
|
|
70
|
+
meta={'model_version': model, 'batch_id': batch_id, 'skipped_assets': skipped})
|
|
71
|
+
_cache_put(key, result)
|
|
72
|
+
return result
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
@auto_backoff()
|
|
76
|
+
async def _apply_url(client: httpx.AsyncClient, token: str, model: str,
|
|
77
|
+
name: str | None) -> tuple[str, str]:
|
|
78
|
+
res = await client.post(f'{BASE}/file-urls/batch', headers=_headers(token),
|
|
79
|
+
json={'files': [{'name': name or 'file', 'data_id': 'f'}],
|
|
80
|
+
'model_version': model})
|
|
81
|
+
_check(res)
|
|
82
|
+
payload = res.json()
|
|
83
|
+
if payload.get('code') != 0:
|
|
84
|
+
raise MineruError(f'mineru file-urls failed: {payload.get("msg")}')
|
|
85
|
+
return payload['data']['batch_id'], payload['data']['file_urls'][0]
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
async def _upload(client: httpx.AsyncClient, url: str, data: bytes) -> None:
|
|
89
|
+
"""S3 single PUT is atomic — an interrupted PUT leaves no object, so
|
|
90
|
+
retrying the same signed URL is safe. Outside the lanes: S3 does not
|
|
91
|
+
count against mineru's submit rate."""
|
|
92
|
+
for attempt in range(_UPLOAD_RETRIES):
|
|
93
|
+
res = await client.put(url, content=data, timeout=120.0)
|
|
94
|
+
if res.status_code == 429 or res.status_code >= 500:
|
|
95
|
+
await asyncio.sleep(2 ** attempt)
|
|
96
|
+
continue
|
|
97
|
+
if res.status_code >= 400:
|
|
98
|
+
raise MineruError(f'mineru upload failed: {res.status_code}: {res.text[:200]}')
|
|
99
|
+
return
|
|
100
|
+
raise MineruError(f'mineru upload failed after {_UPLOAD_RETRIES} attempts: {res.status_code}')
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
async def _poll(client: httpx.AsyncClient, token: str, batch_id: str,
|
|
104
|
+
timeout: float, status_lane: RequestRateScheduler) -> str:
|
|
105
|
+
deadline = time.monotonic() + timeout
|
|
106
|
+
while True:
|
|
107
|
+
task = await status_lane.start_task(_poll_once(client, token, batch_id))
|
|
108
|
+
item = await task
|
|
109
|
+
state = item['state']
|
|
110
|
+
if state == 'done':
|
|
111
|
+
return item.get('full_zip_url')
|
|
112
|
+
if state == 'failed':
|
|
113
|
+
raise MineruError(f'mineru extract failed: {item.get("err_msg")}')
|
|
114
|
+
if time.monotonic() >= deadline:
|
|
115
|
+
raise MineruError(f'mineru timed out after {timeout:.0f}s (state {state})')
|
|
116
|
+
await asyncio.sleep(POLL_INTERVAL)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
@auto_backoff()
|
|
120
|
+
async def _poll_once(client: httpx.AsyncClient, token: str, batch_id: str) -> dict:
|
|
121
|
+
res = await client.get(f'{BASE}/extract-results/batch/{batch_id}',
|
|
122
|
+
headers=_headers(token))
|
|
123
|
+
_check(res)
|
|
124
|
+
payload = res.json()
|
|
125
|
+
if payload.get('code') != 0:
|
|
126
|
+
raise MineruError(f'mineru extract-results failed: {payload.get("msg")}')
|
|
127
|
+
return payload['data']['extract_result'][0]
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
async def _download(client: httpx.AsyncClient, zip_url: str) -> bytes:
|
|
131
|
+
res = await client.get(zip_url, timeout=120.0)
|
|
132
|
+
if res.status_code >= 400:
|
|
133
|
+
raise MineruError(f'mineru result download failed: {res.status_code}')
|
|
134
|
+
return res.content
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def _unzip(result_zip: bytes) -> tuple[str, list[ImageAsset], int]:
|
|
138
|
+
"""Result zip → markdown (full.md) + ImageAssets; assets reconcile
|
|
139
|
+
with the markdown's ```` references by name."""
|
|
140
|
+
markdown = ''
|
|
141
|
+
found = False
|
|
142
|
+
images: list[ImageAsset] = []
|
|
143
|
+
skipped = 0
|
|
144
|
+
with zipfile.ZipFile(io.BytesIO(result_zip)) as z:
|
|
145
|
+
for name in z.namelist():
|
|
146
|
+
if name.endswith('full.md'):
|
|
147
|
+
found = True
|
|
148
|
+
markdown = z.read(name).decode('utf-8', errors='replace')
|
|
149
|
+
elif name.startswith('images/') and not name.endswith('/'):
|
|
150
|
+
if len(images) >= MAX_ASSETS or z.getinfo(name).file_size > MAX_ASSET_BYTES:
|
|
151
|
+
skipped += 1
|
|
152
|
+
continue
|
|
153
|
+
ext = name.rsplit('.', 1)[-1].lower()
|
|
154
|
+
images.append(ImageAsset(name=name, mime=_MIME.get(ext, 'application/octet-stream'),
|
|
155
|
+
data=z.read(name), source=f'mineru {name}'))
|
|
156
|
+
if not found:
|
|
157
|
+
raise MineruError('mineru result zip has no full.md')
|
|
158
|
+
return markdown, images, skipped
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _check(res: httpx.Response) -> None:
|
|
162
|
+
if res.status_code >= 400:
|
|
163
|
+
msg = f'mineru {res.status_code}: {res.text[:200]}'
|
|
164
|
+
if res.status_code == 429 or res.status_code >= 500:
|
|
165
|
+
retry_after = res.headers.get('Retry-After')
|
|
166
|
+
raise RetryException(msg, retry_after=float(retry_after) if retry_after else None)
|
|
167
|
+
raise MineruError(msg)
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _headers(token: str) -> dict:
|
|
171
|
+
return {'Content-Type': 'application/json', 'Authorization': f'Bearer {token}'}
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
_lanes_cache: tuple[RequestRateScheduler, RequestRateScheduler] | None = None
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _lanes() -> tuple[RequestRateScheduler, RequestRateScheduler]:
|
|
178
|
+
global _lanes_cache
|
|
179
|
+
if _lanes_cache is None:
|
|
180
|
+
spec = dict(kv.split(':', 1) for kv in
|
|
181
|
+
os.environ.get('MINERU_CONCURRENCY', 'submit:50|status:1000').split('|'))
|
|
182
|
+
missing = [k for k in ('submit', 'status') if k not in spec]
|
|
183
|
+
if missing:
|
|
184
|
+
raise ValueError(f'MINERU_CONCURRENCY is missing the {missing} lane')
|
|
185
|
+
_lanes_cache = (
|
|
186
|
+
RequestRateScheduler(max_rps=float(spec['submit']) / 60, max_concurrency=8),
|
|
187
|
+
RequestRateScheduler(max_rps=float(spec['status']) / 60, max_concurrency=32),
|
|
188
|
+
)
|
|
189
|
+
return _lanes_cache
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
_cache: OrderedDict[tuple[str, str], ParseResult] = OrderedDict()
|
|
193
|
+
_cache_bytes = 0
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _result_size(result: ParseResult) -> int:
|
|
197
|
+
return len(result.text) + sum(len(img.data) for img in result.images)
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def _cache_get(key: tuple[str, str]) -> ParseResult | None:
|
|
201
|
+
if cached := _cache.get(key):
|
|
202
|
+
_cache.move_to_end(key)
|
|
203
|
+
return cached
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def _cache_put(key: tuple[str, str], result: ParseResult) -> None:
|
|
207
|
+
global _cache_bytes
|
|
208
|
+
_cache[key] = result
|
|
209
|
+
_cache.move_to_end(key)
|
|
210
|
+
_cache_bytes += _result_size(result)
|
|
211
|
+
while len(_cache) > _CACHE_MAX or _cache_bytes > _CACHE_BYTES:
|
|
212
|
+
_, old = _cache.popitem(last=False)
|
|
213
|
+
_cache_bytes -= _result_size(old)
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""Local pdf engine (extra: pdf-inspect): pdf-inspector extraction with
|
|
2
|
+
a sampling-classify decline gate (raster pages); the fingerprint/ratio
|
|
3
|
+
gates, image collection, and post-processing live in ``_pdfscan``."""
|
|
4
|
+
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import pdf_inspector
|
|
8
|
+
|
|
9
|
+
from docspectra.parser import Declined, ParseResult, provider
|
|
10
|
+
from docspectra.providers import traced
|
|
11
|
+
from docspectra.providers._pdfscan import scan
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@provider('pdf-inspect', ('pdf',))
|
|
15
|
+
@traced
|
|
16
|
+
async def parse_pdf(data: bytes, *, name: str | None = None) -> ParseResult | Declined:
|
|
17
|
+
det = pdf_inspector.process_pdf_bytes(data)
|
|
18
|
+
if det.pages_needing_ocr:
|
|
19
|
+
return Declined('classify_gate')
|
|
20
|
+
return scan(data, det.markdown or '', pdf_type=str(det.pdf_type),
|
|
21
|
+
encoding_issues=det.has_encoding_issues)
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
"""txt/md passthrough — the one built-in provider that needs no extra."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from docspectra.parser import Declined, ParseResult, provider
|
|
6
|
+
from docspectra.providers import traced
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@provider('txt', ('txt', 'md', 'markdown'))
|
|
10
|
+
@traced
|
|
11
|
+
async def parse_txt(data: bytes, *, name: str | None = None) -> ParseResult | Declined:
|
|
12
|
+
try:
|
|
13
|
+
return ParseResult(text=data.decode('utf-8-sig'))
|
|
14
|
+
except UnicodeDecodeError:
|
|
15
|
+
return Declined('utf-8 decode failed')
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""The parse service: file + JSON schema → JSON (parser → extractor)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
|
|
7
|
+
from jsonschema import Draft202012Validator
|
|
8
|
+
|
|
9
|
+
from xtremeparse import Extractor, ExtractionResult
|
|
10
|
+
|
|
11
|
+
from docspectra.parser import FileInput, parse as parse_file, parse_path
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class ParseService:
|
|
15
|
+
"""One file and one JSON Schema in, best-effort schema-shaped JSON
|
|
16
|
+
out — validated and self-corrected against a generic JSON Schema
|
|
17
|
+
validator (hosts with their own validator, e.g. docxcast, inject it
|
|
18
|
+
through the extractor directly)."""
|
|
19
|
+
|
|
20
|
+
def __init__(self, runner, *, scheduler=None):
|
|
21
|
+
self._extractor = Extractor(runner, scheduler=scheduler)
|
|
22
|
+
|
|
23
|
+
async def parse(self, data: FileInput, schema: dict, *,
|
|
24
|
+
name: str | None = None, validator=None) -> ExtractionResult:
|
|
25
|
+
document = await parse_file(data, name=name) if isinstance(data, bytes) \
|
|
26
|
+
else await parse_path(data)
|
|
27
|
+
return await self._extractor.extract(document.text, schema,
|
|
28
|
+
validator=validator or schema_validator(schema))
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass
|
|
32
|
+
class GenericIssue:
|
|
33
|
+
"""JSON-Schema error shaped to the Issue protocol; array indices
|
|
34
|
+
render bracketed (`jobs[0].company`) so correction routing matches."""
|
|
35
|
+
|
|
36
|
+
path: str
|
|
37
|
+
message: str
|
|
38
|
+
code: str
|
|
39
|
+
expected: object = None
|
|
40
|
+
got: object = None
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def schema_validator(schema: dict):
|
|
44
|
+
"""Validate data against a JSON Schema, returning error-level issues."""
|
|
45
|
+
validator = Draft202012Validator(schema)
|
|
46
|
+
|
|
47
|
+
def validate(data: dict) -> list:
|
|
48
|
+
return [GenericIssue(_path(e), e.message, e.validator) for e in validator.iter_errors(data)]
|
|
49
|
+
|
|
50
|
+
return validate
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _path(error) -> str:
|
|
54
|
+
path = ''
|
|
55
|
+
for part in error.absolute_path:
|
|
56
|
+
if isinstance(part, int):
|
|
57
|
+
path = f'{path}[{part}]'
|
|
58
|
+
else:
|
|
59
|
+
path = f'{path}.{part}' if path else str(part)
|
|
60
|
+
return path or '$'
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: docspectra
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: DocSpectra: schema-driven document parsing — any file in, spectral JSON out
|
|
5
|
+
Project-URL: Homepage, https://github.com/flowjzh/docspectra
|
|
6
|
+
Project-URL: Repository, https://github.com/flowjzh/docspectra.git
|
|
7
|
+
Project-URL: Issues, https://github.com/flowjzh/docspectra/issues
|
|
8
|
+
Author-email: Flow Jiang <flowjzh@gmail.com>
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: documents,extraction,json-schema,llm,parsing
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
22
|
+
Classifier: Typing :: Typed
|
|
23
|
+
Requires-Python: >=3.10
|
|
24
|
+
Requires-Dist: jsonschema>=4.23
|
|
25
|
+
Requires-Dist: opentelemetry-api>=1.40.0
|
|
26
|
+
Requires-Dist: xtremeparse>=0.1.0
|
|
27
|
+
Provides-Extra: anydoc
|
|
28
|
+
Requires-Dist: firecrawl-anydoc; extra == 'anydoc'
|
|
29
|
+
Requires-Dist: pypdf; extra == 'anydoc'
|
|
30
|
+
Provides-Extra: mineru
|
|
31
|
+
Requires-Dist: httpx>=0.27; extra == 'mineru'
|
|
32
|
+
Requires-Dist: xtremeflow>=0.4.4; extra == 'mineru'
|
|
33
|
+
Provides-Extra: pdf-inspect
|
|
34
|
+
Requires-Dist: pdf-inspector; extra == 'pdf-inspect'
|
|
35
|
+
Requires-Dist: pypdf; extra == 'pdf-inspect'
|
|
36
|
+
Provides-Extra: pydantic-ai
|
|
37
|
+
Requires-Dist: pydantic-ai-slim[openai]>=2.30.0; extra == 'pydantic-ai'
|
|
38
|
+
Description-Content-Type: text/markdown
|
|
39
|
+
|
|
40
|
+
# DocSpectra
|
|
41
|
+
|
|
42
|
+
<img width="600" alt="DocSpectra" src="https://github.com/user-attachments/assets/1d858fb3-67c5-41f6-b0db-3fb6ab39fe4a" />
|
|
43
|
+
|
|
44
|
+
> **Any file in. Spectral JSON out. Fast.**
|
|
45
|
+
|
|
46
|
+
DocSpectra is a schema-driven document-parsing library: hand it a file in
|
|
47
|
+
any supported format and a JSON Schema, get back schema-conforming JSON —
|
|
48
|
+
extracted by extreme-concurrency LLM specialists that chunk, route, fan
|
|
49
|
+
out and self-correct (powered by
|
|
50
|
+
[xtremeparse](https://github.com/flowjzh/xtremeparse)).
|
|
51
|
+
|
|
52
|
+
```
|
|
53
|
+
file + JSON schema ──► docspectra.parse ──► JSON
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
## Layout
|
|
57
|
+
|
|
58
|
+
- `docspectra/parser` — file → text (provider engines as optional extras:
|
|
59
|
+
`pdf-inspect`, `anydoc`, `mineru`; magic-byte type routing;
|
|
60
|
+
file-level decline chains — a provider whose gate fires on a specific
|
|
61
|
+
file declines and the next engine takes it)
|
|
62
|
+
- `docspectra/agents` — the only PydanticAI boundary (AgentRunner adapter)
|
|
63
|
+
- `docspectra/service` — `parse`: file + JSON schema → JSON
|
|
64
|
+
|
|
65
|
+
DocSpectra is deliberately generic: JSON Schema in and out, no document
|
|
66
|
+
vendors, no HTTP server, no domain tools. Applications assemble it with
|
|
67
|
+
their own schema sources and renderers (e.g. bridging
|
|
68
|
+
[DocXCast](https://github.com/flowjzh/docxcast) templates).
|
|
69
|
+
|
|
70
|
+
## Status
|
|
71
|
+
|
|
72
|
+
Parser layer implemented and tested.
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
docspectra/__init__.py,sha256=BFvaJmQhD1jgrnqnoUXnTbd4Q34pjXJDvIH-Yp-vJqI,248
|
|
2
|
+
docspectra/agents/__init__.py,sha256=0_Cilm03Cxyr-J5joWKtv-a0dVT0ouPOOxRJTtEheLA,10239
|
|
3
|
+
docspectra/parser/__init__.py,sha256=OHaFVRg6oSJRh3WwM9Rs9tTa7ubK5pZp-ynl4JpIbZQ,1057
|
|
4
|
+
docspectra/parser/contracts.py,sha256=bLr7q5ypU6xaXJ3aWLqiaOYS4a2aYOp3QhFH5aT3MdA,1855
|
|
5
|
+
docspectra/parser/core.py,sha256=YB-uXbKTgpUcSrKcgJHDtc-qK0HRlzy_ypZW15gJxds,7156
|
|
6
|
+
docspectra/parser/sniff.py,sha256=YGJW-9przYVpqLhr5d3VUX2zGNyEwGf4K7YF3e84m_0,3282
|
|
7
|
+
docspectra/providers/__init__.py,sha256=HxwE6bzjhpLFVwa-VH07-TeyFMs0iJ5Xog5HjLcX-jc,1780
|
|
8
|
+
docspectra/providers/_pdfscan.py,sha256=8tOyZ-GUIk0M34-quEkyMXP2C3ziRuGiFKPq0Tx0bM4,5104
|
|
9
|
+
docspectra/providers/anydoc.py,sha256=UT_o6ReffKls3RztTCvDh7C3uzWaYYJMWLPkOrw0ebE,2752
|
|
10
|
+
docspectra/providers/mineru.py,sha256=s02CMeUEhmRNuvPXnEYd1xYAD_ftSjPbBxIJN9OfMWk,8904
|
|
11
|
+
docspectra/providers/pdf_inspect.py,sha256=GyDauwjQBrowa1a14hSSaxLQDfXMW64ZJ34a7_OPdHM,800
|
|
12
|
+
docspectra/providers/txt.py,sha256=Fg8vCB7tpotEZIxPKh_eTgej6A6Nb1O0EwtLxnaBYEA,502
|
|
13
|
+
docspectra/service/__init__.py,sha256=yG6mGKeariyx4EeQcMM2kDrvx99vETxaKCAIqwI2hXw,236
|
|
14
|
+
docspectra/service/parse.py,sha256=zXmRkHHs6fk7qm7rEh4SgyuNPzA3cEO0VzS3SrNbB_Q,1980
|
|
15
|
+
docspectra-0.1.0.dist-info/METADATA,sha256=adDfdK8pxcTpeQIYo-IjSlJdgJO8R3YaNhA91DdacZo,2943
|
|
16
|
+
docspectra-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
17
|
+
docspectra-0.1.0.dist-info/licenses/LICENSE,sha256=h8-551rXql0Z6aYn_7eita2vDlYtGqP12OGxrmlaQ74,1067
|
|
18
|
+
docspectra-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Flow Jiang
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|