langparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +55 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +12 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/registry.py +38 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/text.py +96 -0
- langparse/chunkers/workbook.py +942 -0
- langparse/cli.py +329 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +210 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +164 -0
- langparse/engines/pdf/mineru.py +259 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +134 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +70 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +220 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/progress.py +77 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +11 -0
- langparse/services/batch_service.py +339 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +523 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/services/workbook_quality_benchmark.py +230 -0
- langparse/types.py +97 -0
- langparse/workbooks/__init__.py +103 -0
- langparse/workbooks/adapters.py +474 -0
- langparse/workbooks/assembly.py +993 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/bundle-v1.schema.json +71 -0
- langparse/workbooks/bundle.py +341 -0
- langparse/workbooks/classification.py +393 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/labels.py +14 -0
- langparse/workbooks/lineage.py +117 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/objects.py +229 -0
- langparse/workbooks/quality/__init__.py +23 -0
- langparse/workbooks/quality/bundle.py +53 -0
- langparse/workbooks/quality/evaluator.py +266 -0
- langparse/workbooks/quality/facts.py +142 -0
- langparse/workbooks/quality/schema.py +462 -0
- langparse/workbooks/reference_types.py +73 -0
- langparse/workbooks/references.py +178 -0
- langparse/workbooks/regions.py +932 -0
- langparse/workbooks/rendering.py +222 -0
- langparse/workbooks/tables.py +477 -0
- langparse/workbooks/types.py +257 -0
- langparse-0.1.0.dist-info/METADATA +790 -0
- langparse-0.1.0.dist-info/RECORD +101 -0
- langparse-0.1.0.dist-info/WHEEL +5 -0
- langparse-0.1.0.dist-info/entry_points.txt +2 -0
- langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,220 @@
|
|
|
1
|
+
import csv
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
|
|
4
|
+
from langparse.core.parser import BaseParser
|
|
5
|
+
from langparse.parsers.sniff import looks_like_ole_binary, looks_like_zip_ooxml
|
|
6
|
+
from langparse.progress import ProgressCallback, ProgressReporter
|
|
7
|
+
from langparse.types import ParsedDocumentResult, ParsedElement, ParseDiagnostics, ParsedPageResult
|
|
8
|
+
from langparse.workbooks.modeling import (
|
|
9
|
+
RequiredWorkbookDisambiguationError,
|
|
10
|
+
WorkbookDisambiguation,
|
|
11
|
+
)
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class ExcelParser(BaseParser):
|
|
15
|
+
"""
|
|
16
|
+
Parses workbook facts and exposes one compatibility part per sheet.
|
|
17
|
+
|
|
18
|
+
Sheets keep stable ordinals but are not pages, so Excel results are always
|
|
19
|
+
marked ``paginated=False``.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
def __init__(
|
|
23
|
+
self,
|
|
24
|
+
disambiguation: str | WorkbookDisambiguation | None = None,
|
|
25
|
+
*,
|
|
26
|
+
model: str | None = None,
|
|
27
|
+
api_key: str | None = None,
|
|
28
|
+
base_url: str | None = None,
|
|
29
|
+
) -> None:
|
|
30
|
+
if isinstance(disambiguation, WorkbookDisambiguation):
|
|
31
|
+
self.disambiguation = disambiguation
|
|
32
|
+
elif isinstance(disambiguation, str):
|
|
33
|
+
mode = disambiguation.lower()
|
|
34
|
+
if mode == "off":
|
|
35
|
+
self.disambiguation = WorkbookDisambiguation.off()
|
|
36
|
+
elif mode in ("auto", "required"):
|
|
37
|
+
from langparse.workbooks.modeling.openai_adapter import (
|
|
38
|
+
OpenAIWorkbookStructureAdapter,
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
adapter = OpenAIWorkbookStructureAdapter.from_env(
|
|
42
|
+
cli_model=model,
|
|
43
|
+
cli_api_key=api_key,
|
|
44
|
+
cli_base_url=base_url,
|
|
45
|
+
)
|
|
46
|
+
if mode == "auto":
|
|
47
|
+
self.disambiguation = WorkbookDisambiguation.auto(adapter)
|
|
48
|
+
else:
|
|
49
|
+
self.disambiguation = WorkbookDisambiguation.required(adapter)
|
|
50
|
+
else:
|
|
51
|
+
raise ValueError(f"Invalid disambiguation mode: {disambiguation}")
|
|
52
|
+
elif model is not None:
|
|
53
|
+
from langparse.workbooks.modeling.openai_adapter import (
|
|
54
|
+
OpenAIWorkbookStructureAdapter,
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
adapter = OpenAIWorkbookStructureAdapter.from_env(
|
|
58
|
+
cli_model=model,
|
|
59
|
+
cli_api_key=api_key,
|
|
60
|
+
cli_base_url=base_url,
|
|
61
|
+
)
|
|
62
|
+
self.disambiguation = WorkbookDisambiguation.auto(adapter)
|
|
63
|
+
else:
|
|
64
|
+
self.disambiguation = (
|
|
65
|
+
WorkbookDisambiguation.off() if disambiguation is None else disambiguation
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
def parse_result(
|
|
69
|
+
self, file_path: str | Path, *, progress_callback: ProgressCallback | None = None, **kwargs
|
|
70
|
+
) -> ParsedDocumentResult:
|
|
71
|
+
path = self._resolve_existing_path(file_path)
|
|
72
|
+
reporter = ProgressReporter(path, progress_callback)
|
|
73
|
+
|
|
74
|
+
if looks_like_zip_ooxml(path):
|
|
75
|
+
return self._parse_ooxml(path, reporter)
|
|
76
|
+
|
|
77
|
+
try:
|
|
78
|
+
import pandas as pd
|
|
79
|
+
except ImportError:
|
|
80
|
+
raise ImportError(
|
|
81
|
+
"pandas and openpyxl are required. Install with `pip install pandas openpyxl`."
|
|
82
|
+
) from None
|
|
83
|
+
|
|
84
|
+
# The extension names one of .csv/.xls/.xlsx, but it can lie -- a
|
|
85
|
+
# workbook re-exported or renamed to .csv (or vice versa) would
|
|
86
|
+
# otherwise be handed to the wrong pandas reader. Content decides:
|
|
87
|
+
# a real workbook is either a ZIP-OOXML or legacy-OLE container;
|
|
88
|
+
# anything else is read as delimited text regardless of its label.
|
|
89
|
+
reporter.emit("extracting")
|
|
90
|
+
is_legacy_workbook = looks_like_ole_binary(path)
|
|
91
|
+
if is_legacy_workbook:
|
|
92
|
+
sheets = pd.read_excel(path, sheet_name=None)
|
|
93
|
+
else:
|
|
94
|
+
with path.open(encoding="utf-8-sig", newline="") as stream:
|
|
95
|
+
sample = stream.read(65536)
|
|
96
|
+
try:
|
|
97
|
+
delimiter = csv.Sniffer().sniff(sample, delimiters=",\t;|").delimiter
|
|
98
|
+
except csv.Error:
|
|
99
|
+
# A sample can end inside a quoted multiline field. The header
|
|
100
|
+
# still provides a delimiter without counting commas in prose.
|
|
101
|
+
try:
|
|
102
|
+
delimiter = (
|
|
103
|
+
csv.Sniffer().sniff(sample.splitlines()[0], delimiters=",\t;|").delimiter
|
|
104
|
+
)
|
|
105
|
+
except (csv.Error, IndexError):
|
|
106
|
+
delimiter = ","
|
|
107
|
+
sheets = {
|
|
108
|
+
None: pd.read_csv(
|
|
109
|
+
path, sep=delimiter, encoding="utf-8-sig", dtype=str, keep_default_na=False
|
|
110
|
+
)
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
reporter.emit("rendering")
|
|
114
|
+
pages = [
|
|
115
|
+
self._page_for_sheet(index + 1, sheet_name, frame)
|
|
116
|
+
for index, (sheet_name, frame) in enumerate(sheets.items())
|
|
117
|
+
]
|
|
118
|
+
|
|
119
|
+
return ParsedDocumentResult(
|
|
120
|
+
source=str(path),
|
|
121
|
+
filename=path.name,
|
|
122
|
+
engine="excel",
|
|
123
|
+
pages=pages,
|
|
124
|
+
markdown_content="\n".join(page.markdown_content for page in pages),
|
|
125
|
+
metadata={"extension": path.suffix, "sheet_count": len(pages)},
|
|
126
|
+
paginated=False,
|
|
127
|
+
diagnostics=(
|
|
128
|
+
ParseDiagnostics(
|
|
129
|
+
status="partial",
|
|
130
|
+
reconstruction_passed=False,
|
|
131
|
+
unsupported_features=[
|
|
132
|
+
"Legacy OLE workbooks use the pandas compatibility adapter; "
|
|
133
|
+
"rich workbook facts are not available in Phase 1."
|
|
134
|
+
],
|
|
135
|
+
)
|
|
136
|
+
if is_legacy_workbook
|
|
137
|
+
else None
|
|
138
|
+
),
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
def _parse_ooxml(self, path: Path, reporter: ProgressReporter) -> ParsedDocumentResult:
|
|
142
|
+
try:
|
|
143
|
+
from langparse.workbooks.adapters import OOXMLWorkbookAdapter
|
|
144
|
+
from langparse.workbooks.assembly import assemble_baseline, assemble_workbook
|
|
145
|
+
from langparse.workbooks.rendering import (
|
|
146
|
+
compatibility_pages,
|
|
147
|
+
render_workbook_markdown,
|
|
148
|
+
)
|
|
149
|
+
except ImportError:
|
|
150
|
+
raise ImportError(
|
|
151
|
+
"openpyxl is required for OOXML workbooks. "
|
|
152
|
+
"Install with `pip install langparse[excel]`."
|
|
153
|
+
) from None
|
|
154
|
+
|
|
155
|
+
reporter.emit("extracting")
|
|
156
|
+
snapshot = OOXMLWorkbookAdapter().snapshot(path)
|
|
157
|
+
reporter.emit("assembling")
|
|
158
|
+
try:
|
|
159
|
+
structure, diagnostics = assemble_workbook(
|
|
160
|
+
snapshot,
|
|
161
|
+
disambiguation=self.disambiguation,
|
|
162
|
+
)
|
|
163
|
+
except RequiredWorkbookDisambiguationError:
|
|
164
|
+
raise
|
|
165
|
+
except Exception as exc:
|
|
166
|
+
structure, diagnostics = assemble_baseline(snapshot)
|
|
167
|
+
diagnostics.status = "partial"
|
|
168
|
+
diagnostics.warnings.append(
|
|
169
|
+
f"Semantic workbook assembly failed; retained raw-grid fallback: {type(exc).__name__}"
|
|
170
|
+
)
|
|
171
|
+
from langparse.workbooks.objects import attach_objects
|
|
172
|
+
|
|
173
|
+
attach_objects(snapshot, structure, diagnostics)
|
|
174
|
+
reporter.emit("rendering")
|
|
175
|
+
pages = compatibility_pages(snapshot, structure)
|
|
176
|
+
markdown = render_workbook_markdown(snapshot, structure)
|
|
177
|
+
return ParsedDocumentResult(
|
|
178
|
+
source=str(path),
|
|
179
|
+
filename=path.name,
|
|
180
|
+
engine="excel",
|
|
181
|
+
pages=pages,
|
|
182
|
+
markdown_content=markdown,
|
|
183
|
+
metadata={"extension": path.suffix, "sheet_count": len(snapshot.sheets)},
|
|
184
|
+
paginated=False,
|
|
185
|
+
structure=structure,
|
|
186
|
+
diagnostics=diagnostics,
|
|
187
|
+
)
|
|
188
|
+
|
|
189
|
+
def _page_for_sheet(self, page_number: int, sheet_name, frame) -> ParsedPageResult:
|
|
190
|
+
table_markdown = frame.to_markdown(index=False, disable_numparse=True)
|
|
191
|
+
heading = f"## Sheet: {sheet_name}\n" if sheet_name is not None else ""
|
|
192
|
+
markdown_content = f"{heading}\n{table_markdown}\n" if heading else table_markdown
|
|
193
|
+
|
|
194
|
+
rows = [[str(column) for column in frame.columns]]
|
|
195
|
+
rows.extend([[self._cell_text(value) for value in row] for row in frame.values])
|
|
196
|
+
|
|
197
|
+
return ParsedPageResult(
|
|
198
|
+
page_number=page_number,
|
|
199
|
+
markdown_content=markdown_content,
|
|
200
|
+
plain_text=table_markdown,
|
|
201
|
+
elements=[ParsedElement(kind="table", text=table_markdown)],
|
|
202
|
+
tables=[{"rows": rows, "sheet_name": sheet_name}],
|
|
203
|
+
metadata={"part_kind": "sheet", "sheet_name": sheet_name, "source_range": None},
|
|
204
|
+
)
|
|
205
|
+
|
|
206
|
+
def _cell_text(self, value) -> str:
|
|
207
|
+
"""
|
|
208
|
+
Render one cell for the structured table.
|
|
209
|
+
|
|
210
|
+
Blank cells arrive as NaN rather than None, and a single blank promotes
|
|
211
|
+
an integer column to float -- so a naive str() yields "nan" and "1.0"
|
|
212
|
+
where the source held an empty cell and 1.
|
|
213
|
+
"""
|
|
214
|
+
import pandas as pd
|
|
215
|
+
|
|
216
|
+
if value is None or pd.isna(value):
|
|
217
|
+
return ""
|
|
218
|
+
if isinstance(value, float) and value.is_integer():
|
|
219
|
+
return str(int(value))
|
|
220
|
+
return str(value)
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
from pathlib import Path
|
|
2
|
+
|
|
3
|
+
from langparse.core.parser import BaseParser
|
|
4
|
+
from langparse.types import ParsedDocumentResult, ParsedPageResult
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class MarkdownParser(BaseParser):
|
|
8
|
+
"""
|
|
9
|
+
A simple parser for Markdown files.
|
|
10
|
+
|
|
11
|
+
Markdown is a flow format with no page boundaries, so the whole file is one
|
|
12
|
+
unpaginated page and the content passes through verbatim -- including any
|
|
13
|
+
page markers the source already carries.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
def parse_result(self, file_path: str | Path, **kwargs) -> ParsedDocumentResult:
|
|
17
|
+
path = self._resolve_existing_path(file_path)
|
|
18
|
+
content = path.read_text(encoding="utf-8")
|
|
19
|
+
|
|
20
|
+
return ParsedDocumentResult(
|
|
21
|
+
source=str(path),
|
|
22
|
+
filename=path.name,
|
|
23
|
+
engine="markdown",
|
|
24
|
+
pages=[
|
|
25
|
+
ParsedPageResult(
|
|
26
|
+
page_number=1,
|
|
27
|
+
markdown_content=content,
|
|
28
|
+
plain_text=content,
|
|
29
|
+
)
|
|
30
|
+
],
|
|
31
|
+
markdown_content=content,
|
|
32
|
+
metadata={"extension": path.suffix},
|
|
33
|
+
paginated=False,
|
|
34
|
+
)
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
from pathlib import Path
|
|
2
|
+
from typing import Literal
|
|
3
|
+
|
|
4
|
+
from langparse.config import settings
|
|
5
|
+
from langparse.core.parser import BaseParser
|
|
6
|
+
from langparse.services.parse_service import ParseService
|
|
7
|
+
from langparse.types import ParsedDocumentResult
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class PDFParser(BaseParser):
|
|
11
|
+
"""
|
|
12
|
+
A universal PDF parser that delegates to specific engines.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
def __init__(
|
|
16
|
+
self,
|
|
17
|
+
engine: Literal["simple", "mineru", "deepdoc"] = None,
|
|
18
|
+
**engine_kwargs,
|
|
19
|
+
):
|
|
20
|
+
# Engine name resolves as: argument > config > default. Construction is
|
|
21
|
+
# delegated so selection errors read the same here as from the CLI.
|
|
22
|
+
self.engine_name = engine or settings.get("default_pdf_engine", "simple")
|
|
23
|
+
self.engine = ParseService().create_engine(self.engine_name, **engine_kwargs)
|
|
24
|
+
|
|
25
|
+
def parse_result(self, file_path: str | Path, **kwargs) -> ParsedDocumentResult:
|
|
26
|
+
return ParseService().parse_result(
|
|
27
|
+
file_path,
|
|
28
|
+
engine_name=self.engine_name,
|
|
29
|
+
engine=self.engine,
|
|
30
|
+
**kwargs,
|
|
31
|
+
)
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
from langparse.parsers.sniff import sniff_kind
|
|
6
|
+
|
|
7
|
+
#: Single source of truth for which extensions LangParse handles and which
|
|
8
|
+
#: parser family owns each one. AutoParser, ParseService routing, directory
|
|
9
|
+
#: expansion and CLI error messages all read from here so they cannot drift.
|
|
10
|
+
PARSER_KIND_BY_EXTENSION: dict[str, str] = {
|
|
11
|
+
".pdf": "pdf",
|
|
12
|
+
".docx": "docx",
|
|
13
|
+
".doc": "docx",
|
|
14
|
+
".xlsx": "excel",
|
|
15
|
+
".xlsm": "excel",
|
|
16
|
+
".xls": "excel",
|
|
17
|
+
".csv": "excel",
|
|
18
|
+
".md": "markdown",
|
|
19
|
+
".txt": "markdown",
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
SUPPORTED_EXTENSIONS = frozenset(PARSER_KIND_BY_EXTENSION)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def parser_kind_for(file_path) -> str | None:
|
|
26
|
+
"""Return the parser family for a path, or None when unsupported.
|
|
27
|
+
|
|
28
|
+
Extensions can lie -- a renamed or re-exported file can carry the wrong
|
|
29
|
+
suffix. When the path exists, its content is sniffed first (magic bytes /
|
|
30
|
+
OOXML zip contents); a conclusive sniff overrides the extension. An
|
|
31
|
+
inconclusive sniff (plain text, legacy OLE binary, missing file) falls
|
|
32
|
+
back to the extension, exactly as before.
|
|
33
|
+
"""
|
|
34
|
+
path = Path(file_path)
|
|
35
|
+
sniffed = sniff_kind(path)
|
|
36
|
+
if sniffed is not None:
|
|
37
|
+
return sniffed
|
|
38
|
+
return PARSER_KIND_BY_EXTENSION.get(path.suffix.lower())
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def is_supported(file_path) -> bool:
|
|
42
|
+
return parser_kind_for(file_path) is not None
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def unsupported_extension_error(file_path) -> ValueError:
|
|
46
|
+
suffix = Path(file_path).suffix.lower() or "(no extension)"
|
|
47
|
+
supported = ", ".join(sorted(SUPPORTED_EXTENSIONS))
|
|
48
|
+
return ValueError(f"Unsupported file extension: {suffix}. Supported: {supported}")
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import zipfile
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
#: Magic-byte signatures for formats we can identify without a dependency.
|
|
7
|
+
_PDF_SIGNATURE = b"%PDF-"
|
|
8
|
+
_ZIP_SIGNATURE = b"PK\x03\x04"
|
|
9
|
+
_OLE_CFB_SIGNATURE = b"\xd0\xcf\x11\xe0\xa1\xb1\x1a\xe1"
|
|
10
|
+
|
|
11
|
+
#: OOXML zip packages are only distinguishable by which top-level folder
|
|
12
|
+
#: their parts live under -- the file extension plays no part in this.
|
|
13
|
+
_OOXML_KIND_BY_NAMELIST_PREFIX = {
|
|
14
|
+
"word/": "docx",
|
|
15
|
+
"xl/": "excel",
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def sniff_kind(file_path) -> str | None:
|
|
20
|
+
"""Best-effort parser-kind detection from file content, independent of extension.
|
|
21
|
+
|
|
22
|
+
Extensions can lie -- a renamed or re-exported file can carry the wrong
|
|
23
|
+
suffix. Returns a `PARSER_KIND_BY_EXTENSION` value when the signature is
|
|
24
|
+
conclusive (PDF, OOXML docx/xlsx). Returns None when it isn't -- legacy
|
|
25
|
+
OLE binaries, plain text (csv/md/txt), a missing file, or an unrecognized
|
|
26
|
+
layout -- so the caller falls back to the extension.
|
|
27
|
+
"""
|
|
28
|
+
header = _read_header(file_path)
|
|
29
|
+
if header is None:
|
|
30
|
+
return None
|
|
31
|
+
if header.startswith(_PDF_SIGNATURE):
|
|
32
|
+
return "pdf"
|
|
33
|
+
if header.startswith(_ZIP_SIGNATURE):
|
|
34
|
+
return _sniff_ooxml_kind(file_path)
|
|
35
|
+
return None
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def looks_like_zip_ooxml(file_path) -> bool:
|
|
39
|
+
"""True when the file is any ZIP-based Office document (docx/xlsx/...),
|
|
40
|
+
regardless of extension. Used to catch a workbook mislabeled `.csv`."""
|
|
41
|
+
header = _read_header(file_path)
|
|
42
|
+
return header is not None and header.startswith(_ZIP_SIGNATURE)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def looks_like_ole_binary(file_path) -> bool:
|
|
46
|
+
"""True when the file is a legacy OLE Compound File Binary container
|
|
47
|
+
(pre-2007 .doc/.xls), regardless of extension. The signature alone can't
|
|
48
|
+
tell a Word doc from a workbook -- callers only check this after routing
|
|
49
|
+
has already settled the family via the extension."""
|
|
50
|
+
header = _read_header(file_path)
|
|
51
|
+
return header is not None and header.startswith(_OLE_CFB_SIGNATURE)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _read_header(file_path, size: int = 8) -> bytes | None:
|
|
55
|
+
path = Path(file_path)
|
|
56
|
+
try:
|
|
57
|
+
with path.open("rb") as handle:
|
|
58
|
+
return handle.read(size)
|
|
59
|
+
except OSError:
|
|
60
|
+
return None
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _sniff_ooxml_kind(file_path) -> str | None:
|
|
64
|
+
try:
|
|
65
|
+
with zipfile.ZipFile(file_path) as archive:
|
|
66
|
+
names = archive.namelist()
|
|
67
|
+
except zipfile.BadZipFile:
|
|
68
|
+
return None
|
|
69
|
+
for prefix, kind in _OOXML_KIND_BY_NAMELIST_PREFIX.items():
|
|
70
|
+
if any(name.startswith(prefix) for name in names):
|
|
71
|
+
return kind
|
|
72
|
+
return None
|
langparse/progress.py
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""Optional, synchronous progress observations; no task storage or dependencies."""
|
|
2
|
+
|
|
3
|
+
from collections.abc import Callable, Iterator
|
|
4
|
+
from contextlib import contextmanager
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Literal
|
|
8
|
+
|
|
9
|
+
from langparse.logging import get_logger
|
|
10
|
+
|
|
11
|
+
logger = get_logger(__name__)
|
|
12
|
+
ProgressState = Literal["running", "completed", "failed", "skipped"]
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass(frozen=True)
|
|
16
|
+
class ProgressEvent:
|
|
17
|
+
"""Percent is 0–100 within phase, never an estimate of remaining time.
|
|
18
|
+
|
|
19
|
+
A file's lifecycle ends only at phase='file', state='completed'/'failed'.
|
|
20
|
+
Completion does not imply that the result's quality diagnostics are successful.
|
|
21
|
+
Unknown amounts remain None. An empty source identifies the batch itself.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
source: str
|
|
25
|
+
phase: str
|
|
26
|
+
state: ProgressState = "running"
|
|
27
|
+
completed_units: int | None = None
|
|
28
|
+
total_units: int | None = None
|
|
29
|
+
unit: str | None = None
|
|
30
|
+
percent: float | None = None
|
|
31
|
+
message: str = ""
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
ProgressCallback = Callable[[ProgressEvent], None]
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class ProgressReporter:
|
|
38
|
+
"""Internal adapter keeping observer failures out of parsing and fallback paths."""
|
|
39
|
+
|
|
40
|
+
def __init__(self, source: str | Path, callback: ProgressCallback | None = None):
|
|
41
|
+
self.source = str(source)
|
|
42
|
+
self.callback = callback
|
|
43
|
+
|
|
44
|
+
def emit(
|
|
45
|
+
self,
|
|
46
|
+
phase: str,
|
|
47
|
+
*,
|
|
48
|
+
state: ProgressState = "running",
|
|
49
|
+
completed_units: int | None = None,
|
|
50
|
+
total_units: int | None = None,
|
|
51
|
+
unit: str | None = None,
|
|
52
|
+
percent: float | None = None,
|
|
53
|
+
message: str = "",
|
|
54
|
+
) -> None:
|
|
55
|
+
if self.callback is None:
|
|
56
|
+
return
|
|
57
|
+
if percent is None and completed_units is not None and total_units:
|
|
58
|
+
percent = 100.0 * completed_units / total_units
|
|
59
|
+
event = ProgressEvent(
|
|
60
|
+
self.source, phase, state, completed_units, total_units, unit, percent, message
|
|
61
|
+
)
|
|
62
|
+
try:
|
|
63
|
+
self.callback(event)
|
|
64
|
+
except Exception as exc:
|
|
65
|
+
# Do not include observer exceptions' text: it may contain credentials.
|
|
66
|
+
logger.warning("Progress callback failed (%s)", type(exc).__name__)
|
|
67
|
+
|
|
68
|
+
@contextmanager
|
|
69
|
+
def operation(self, phase: str) -> Iterator[None]:
|
|
70
|
+
self.emit(phase)
|
|
71
|
+
try:
|
|
72
|
+
yield
|
|
73
|
+
except Exception as exc:
|
|
74
|
+
self.emit(phase, state="failed", message=type(exc).__name__)
|
|
75
|
+
raise
|
|
76
|
+
else:
|
|
77
|
+
self.emit(phase, state="completed")
|
langparse/py.typed
ADDED
|
File without changes
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
from langparse.services.batch_service import BatchParseService
|
|
2
|
+
from langparse.services.benchmark_service import BenchmarkService
|
|
3
|
+
from langparse.services.parse_service import ParseService
|
|
4
|
+
from langparse.services.workbook_quality_benchmark import WorkbookQualityBenchmarkService
|
|
5
|
+
|
|
6
|
+
__all__ = [
|
|
7
|
+
"BatchParseService",
|
|
8
|
+
"BenchmarkService",
|
|
9
|
+
"ParseService",
|
|
10
|
+
"WorkbookQualityBenchmarkService",
|
|
11
|
+
]
|