langparse 0.1.0rc1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (85) hide show
  1. langparse/__init__.py +35 -0
  2. langparse/autoparser.py +25 -0
  3. langparse/chunkers/__init__.py +0 -0
  4. langparse/chunkers/blocks.py +151 -0
  5. langparse/chunkers/profiles.py +53 -0
  6. langparse/chunkers/semantic.py +242 -0
  7. langparse/chunkers/workbook.py +900 -0
  8. langparse/cli.py +257 -0
  9. langparse/config.py +169 -0
  10. langparse/core/__init__.py +0 -0
  11. langparse/core/chunker.py +16 -0
  12. langparse/core/engine.py +37 -0
  13. langparse/core/parser.py +35 -0
  14. langparse/core/rendering.py +49 -0
  15. langparse/engines/__init__.py +1 -0
  16. langparse/engines/pdf/__init__.py +1 -0
  17. langparse/engines/pdf/deepdoc/__init__.py +55 -0
  18. langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
  19. langparse/engines/pdf/deepdoc/model_loader.py +101 -0
  20. langparse/engines/pdf/deepdoc/ocr.py +641 -0
  21. langparse/engines/pdf/deepdoc/operators.py +684 -0
  22. langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
  23. langparse/engines/pdf/deepdoc/postprocess.py +339 -0
  24. langparse/engines/pdf/deepdoc/recognizer.py +418 -0
  25. langparse/engines/pdf/deepdoc/rendering.py +202 -0
  26. langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
  27. langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
  28. langparse/engines/pdf/deepdoc/utils.py +36 -0
  29. langparse/engines/pdf/deepdoc_engine.py +140 -0
  30. langparse/engines/pdf/mineru.py +235 -0
  31. langparse/engines/pdf/mineru_client.py +318 -0
  32. langparse/engines/pdf/mineru_service.py +225 -0
  33. langparse/engines/pdf/ocr.py +101 -0
  34. langparse/engines/pdf/other.py +20 -0
  35. langparse/engines/pdf/simple.py +127 -0
  36. langparse/engines/pdf/vision_llm.py +27 -0
  37. langparse/errors.py +52 -0
  38. langparse/logging.py +27 -0
  39. langparse/metrics.py +129 -0
  40. langparse/parsers/__init__.py +0 -0
  41. langparse/parsers/docx_parser.py +114 -0
  42. langparse/parsers/excel_parser.py +190 -0
  43. langparse/parsers/markdown_parser.py +34 -0
  44. langparse/parsers/pdf_parser.py +31 -0
  45. langparse/parsers/registry.py +48 -0
  46. langparse/parsers/sniff.py +72 -0
  47. langparse/py.typed +0 -0
  48. langparse/services/__init__.py +5 -0
  49. langparse/services/batch_service.py +253 -0
  50. langparse/services/benchmark_service.py +202 -0
  51. langparse/services/fidelity.py +154 -0
  52. langparse/services/output_paths.py +86 -0
  53. langparse/services/parse_service.py +468 -0
  54. langparse/services/quality.py +65 -0
  55. langparse/services/workbook_ambiguity_benchmark.py +563 -0
  56. langparse/types.py +94 -0
  57. langparse/workbooks/__init__.py +97 -0
  58. langparse/workbooks/adapters.py +309 -0
  59. langparse/workbooks/assembly.py +928 -0
  60. langparse/workbooks/blocks.py +209 -0
  61. langparse/workbooks/classification.py +381 -0
  62. langparse/workbooks/continuation.py +577 -0
  63. langparse/workbooks/evaluation/__init__.py +45 -0
  64. langparse/workbooks/evaluation/evaluator.py +381 -0
  65. langparse/workbooks/evaluation/schema.py +419 -0
  66. langparse/workbooks/modeling/__init__.py +52 -0
  67. langparse/workbooks/modeling/cache.py +20 -0
  68. langparse/workbooks/modeling/config.py +87 -0
  69. langparse/workbooks/modeling/contract.py +628 -0
  70. langparse/workbooks/modeling/disambiguation.py +800 -0
  71. langparse/workbooks/modeling/openai_adapter.py +192 -0
  72. langparse/workbooks/modeling/policy.py +79 -0
  73. langparse/workbooks/modeling/ports.py +44 -0
  74. langparse/workbooks/modeling/pricing.py +17 -0
  75. langparse/workbooks/modeling/types.py +251 -0
  76. langparse/workbooks/regions.py +77 -0
  77. langparse/workbooks/rendering.py +216 -0
  78. langparse/workbooks/tables.py +375 -0
  79. langparse/workbooks/types.py +239 -0
  80. langparse-0.1.0rc1.dist-info/METADATA +720 -0
  81. langparse-0.1.0rc1.dist-info/RECORD +85 -0
  82. langparse-0.1.0rc1.dist-info/WHEEL +5 -0
  83. langparse-0.1.0rc1.dist-info/entry_points.txt +2 -0
  84. langparse-0.1.0rc1.dist-info/licenses/LICENSE +192 -0
  85. langparse-0.1.0rc1.dist-info/top_level.txt +1 -0
langparse/__init__.py ADDED
@@ -0,0 +1,35 @@
1
+ from langparse.autoparser import AutoParser
2
+ from langparse.chunkers.semantic import SemanticChunker
3
+ from langparse.core.chunker import BaseChunker
4
+ from langparse.core.parser import BaseParser
5
+ from langparse.metrics import BatchItemResult, BatchRunResult, ParseMetrics
6
+ from langparse.parsers.docx_parser import DocxParser
7
+ from langparse.parsers.excel_parser import ExcelParser
8
+ from langparse.parsers.markdown_parser import MarkdownParser
9
+ from langparse.parsers.pdf_parser import PDFParser
10
+ from langparse.types import (
11
+ Chunk,
12
+ Document,
13
+ ParsedDocumentResult,
14
+ ParsedElement,
15
+ ParsedPageResult,
16
+ )
17
+
18
+ __all__ = [
19
+ "Document",
20
+ "Chunk",
21
+ "ParsedDocumentResult",
22
+ "ParsedPageResult",
23
+ "ParsedElement",
24
+ "BaseParser",
25
+ "BaseChunker",
26
+ "AutoParser",
27
+ "PDFParser",
28
+ "MarkdownParser",
29
+ "DocxParser",
30
+ "ExcelParser",
31
+ "SemanticChunker",
32
+ "ParseMetrics",
33
+ "BatchItemResult",
34
+ "BatchRunResult",
35
+ ]
@@ -0,0 +1,25 @@
1
+ from pathlib import Path
2
+
3
+ from langparse.core.rendering import document_from_result
4
+ from langparse.types import Document, ParsedDocumentResult
5
+
6
+
7
+ class AutoParser:
8
+ """
9
+ Facade that parses any supported file without the caller picking a parser.
10
+
11
+ Extension routing lives in `ParseService`, driven by
12
+ `langparse.parsers.registry`, so this stays a convenience wrapper rather
13
+ than a second place formats can be registered and drift.
14
+ """
15
+
16
+ @staticmethod
17
+ def parse_result(file_path: str | Path, **kwargs) -> ParsedDocumentResult:
18
+ from langparse.services.parse_service import ParseService
19
+
20
+ engine_name = kwargs.pop("engine", None) or "simple"
21
+ return ParseService().parse_result(file_path, engine_name=engine_name, **kwargs)
22
+
23
+ @staticmethod
24
+ def parse(file_path: str | Path, **kwargs) -> Document:
25
+ return document_from_result(AutoParser.parse_result(file_path, **kwargs))
File without changes
@@ -0,0 +1,151 @@
1
+ """
2
+ Scan Markdown into typed blocks.
3
+
4
+ Chunking used to run a heading regex over the whole document, which cannot tell
5
+ a real heading from a `#` comment inside a fenced code block -- recognising a
6
+ fence requires tracking state across lines. Scanning into typed blocks fixes
7
+ that structurally, and the same block types are what let the packer treat
8
+ tables and code differently when they overflow a chunk.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import re
14
+ from dataclasses import dataclass, field
15
+
16
+ HEADING_RE = re.compile(r"^(#{1,6})\s+(.+?)\s*$")
17
+ FENCE_RE = re.compile(r"^(\s*)(`{3,}|~{3,})(.*)$")
18
+ PAGE_MARKER_RE = re.compile(r"^\s*<!--\s*page_number:\s*(\d+)\s*-->\s*$")
19
+ TABLE_SEPARATOR_RE = re.compile(r"^\s*\|?\s*:?-{3,}:?\s*(\|\s*:?-{3,}:?\s*)*\|?\s*$")
20
+
21
+ HEADING = "heading"
22
+ TABLE = "table"
23
+ CODE = "code"
24
+ PARAGRAPH = "paragraph"
25
+ PAGE_MARKER = "page_marker"
26
+
27
+
28
+ @dataclass
29
+ class Block:
30
+ kind: str
31
+ text: str
32
+ #: heading only
33
+ level: int = 0
34
+ title: str = ""
35
+ #: table only
36
+ rows: list[list[str]] = field(default_factory=list)
37
+ has_header: bool = False
38
+ #: page_marker only
39
+ page_number: int = 0
40
+
41
+
42
+ def scan_blocks(markdown: str) -> list[Block]:
43
+ """Split Markdown into an ordered list of typed blocks."""
44
+ if not markdown:
45
+ return []
46
+
47
+ lines = markdown.splitlines()
48
+ blocks: list[Block] = []
49
+ index = 0
50
+
51
+ while index < len(lines):
52
+ line = lines[index]
53
+
54
+ if not line.strip():
55
+ index += 1
56
+ continue
57
+
58
+ fence = FENCE_RE.match(line)
59
+ if fence:
60
+ index = _consume_fence(lines, index, fence.group(2), blocks)
61
+ continue
62
+
63
+ page_marker = PAGE_MARKER_RE.match(line)
64
+ if page_marker:
65
+ blocks.append(Block(kind=PAGE_MARKER, text=line, page_number=int(page_marker.group(1))))
66
+ index += 1
67
+ continue
68
+
69
+ heading = HEADING_RE.match(line)
70
+ if heading:
71
+ blocks.append(
72
+ Block(
73
+ kind=HEADING,
74
+ text=line,
75
+ level=len(heading.group(1)),
76
+ title=heading.group(2),
77
+ )
78
+ )
79
+ index += 1
80
+ continue
81
+
82
+ if _starts_table(lines, index):
83
+ index = _consume_table(lines, index, blocks)
84
+ continue
85
+
86
+ index = _consume_paragraph(lines, index, blocks)
87
+
88
+ return blocks
89
+
90
+
91
+ def _consume_fence(lines: list[str], index: int, marker: str, blocks: list[Block]) -> int:
92
+ """Consume a fenced block. An unclosed fence runs to end of input."""
93
+ fence_char = marker[0]
94
+ collected = [lines[index]]
95
+ index += 1
96
+
97
+ while index < len(lines):
98
+ collected.append(lines[index])
99
+ closing = FENCE_RE.match(lines[index])
100
+ index += 1
101
+ if closing and closing.group(2)[0] == fence_char and len(closing.group(2)) >= len(marker):
102
+ break
103
+
104
+ blocks.append(Block(kind=CODE, text="\n".join(collected)))
105
+ return index
106
+
107
+
108
+ def _starts_table(lines: list[str], index: int) -> bool:
109
+ """A table needs a pipe row followed by a separator row; pipes alone are prose."""
110
+ if not lines[index].lstrip().startswith("|"):
111
+ return False
112
+ return index + 1 < len(lines) and bool(TABLE_SEPARATOR_RE.match(lines[index + 1]))
113
+
114
+
115
+ def _consume_table(lines: list[str], index: int, blocks: list[Block]) -> int:
116
+ collected: list[str] = []
117
+ while index < len(lines) and lines[index].lstrip().startswith("|"):
118
+ collected.append(lines[index])
119
+ index += 1
120
+
121
+ rows = [_split_table_row(line) for line in collected if not TABLE_SEPARATOR_RE.match(line)]
122
+ blocks.append(Block(kind=TABLE, text="\n".join(collected), rows=rows, has_header=bool(rows)))
123
+ return index
124
+
125
+
126
+ def _split_table_row(line: str) -> list[str]:
127
+ stripped = line.strip()
128
+ if stripped.startswith("|"):
129
+ stripped = stripped[1:]
130
+ if stripped.endswith("|"):
131
+ stripped = stripped[:-1]
132
+ return [cell.strip() for cell in stripped.split("|")]
133
+
134
+
135
+ def _consume_paragraph(lines: list[str], index: int, blocks: list[Block]) -> int:
136
+ collected: list[str] = []
137
+ while index < len(lines) and lines[index].strip():
138
+ line = lines[index]
139
+ if (
140
+ FENCE_RE.match(line)
141
+ or HEADING_RE.match(line)
142
+ or PAGE_MARKER_RE.match(line)
143
+ or _starts_table(lines, index)
144
+ ):
145
+ break
146
+ collected.append(line)
147
+ index += 1
148
+
149
+ if collected:
150
+ blocks.append(Block(kind=PARAGRAPH, text="\n".join(collected)))
151
+ return index
@@ -0,0 +1,53 @@
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import dataclass
4
+ from enum import Enum
5
+
6
+
7
+ class WorkbookChunkProfile(str, Enum):
8
+ RETRIEVAL = "retrieval"
9
+ ANALYSIS = "analysis"
10
+
11
+
12
+ @dataclass(frozen=True)
13
+ class WorkbookChunkPolicy:
14
+ name: WorkbookChunkProfile
15
+ version: int
16
+ default_max_chunk_size: int
17
+ analysis_records: bool
18
+
19
+
20
+ class ChunkProfileNotSupportedError(ValueError):
21
+ """Raised when a valid chunk profile cannot represent the parsed input."""
22
+
23
+
24
+ _POLICIES = {
25
+ WorkbookChunkProfile.RETRIEVAL: WorkbookChunkPolicy(
26
+ name=WorkbookChunkProfile.RETRIEVAL,
27
+ version=1,
28
+ default_max_chunk_size=1000,
29
+ analysis_records=False,
30
+ ),
31
+ WorkbookChunkProfile.ANALYSIS: WorkbookChunkPolicy(
32
+ name=WorkbookChunkProfile.ANALYSIS,
33
+ version=1,
34
+ default_max_chunk_size=4000,
35
+ analysis_records=True,
36
+ ),
37
+ }
38
+
39
+
40
+ def resolve_workbook_chunk_policy(
41
+ profile: str | WorkbookChunkProfile | None,
42
+ ) -> WorkbookChunkPolicy:
43
+ if profile is None:
44
+ selected = WorkbookChunkProfile.RETRIEVAL
45
+ else:
46
+ try:
47
+ selected = WorkbookChunkProfile(profile)
48
+ except ValueError:
49
+ available = ", ".join(sorted(item.value for item in WorkbookChunkProfile))
50
+ raise ValueError(
51
+ f"Unknown workbook chunk profile {profile!r}. Available: {available}"
52
+ ) from None
53
+ return _POLICIES[selected]
@@ -0,0 +1,242 @@
1
+ from __future__ import annotations
2
+
3
+ import re
4
+ from collections.abc import Callable
5
+ from dataclasses import dataclass, field
6
+
7
+ from langparse.chunkers.blocks import CODE, HEADING, PAGE_MARKER, TABLE, Block, scan_blocks
8
+ from langparse.core.chunker import BaseChunker
9
+ from langparse.types import Chunk, Document
10
+
11
+ SENTENCE_END_RE = re.compile(r"(?<=[.!?。!?])\s+")
12
+ JOIN = "\n\n"
13
+
14
+
15
+ @dataclass
16
+ class _Section:
17
+ header: str | None = None
18
+ header_level: int = 0
19
+ header_path: str = ""
20
+ blocks: list[Block] = field(default_factory=list)
21
+ page_numbers: set = field(default_factory=set)
22
+
23
+
24
+ @dataclass
25
+ class _Unit:
26
+ """One packable piece of content. Oversized units get a chunk to themselves."""
27
+
28
+ text: str
29
+ oversized: bool = False
30
+
31
+
32
+ class SemanticChunker(BaseChunker):
33
+ """
34
+ Chunks text on Markdown structure, keeping chunks within a size budget.
35
+
36
+ Sections come from heading structure; within a section, blocks are packed
37
+ greedily up to `max_chunk_size`. Size is measured by `length_function`, so
38
+ callers embedding against a token budget can pass a tokenizer's encoder
39
+ instead of the default character count.
40
+ """
41
+
42
+ def __init__(
43
+ self,
44
+ max_chunk_size: int = 1000,
45
+ overlap: int = 0,
46
+ length_function: Callable[[str], int] = len,
47
+ ):
48
+ if overlap >= max_chunk_size:
49
+ raise ValueError("overlap must be smaller than max_chunk_size")
50
+ self.max_chunk_size = max_chunk_size
51
+ self.overlap = overlap
52
+ self.length_function = length_function
53
+
54
+ def chunk(self, document: Document, **kwargs) -> list[Chunk]:
55
+ sections = self._sections(scan_blocks(document.content))
56
+
57
+ chunks: list[Chunk] = []
58
+ for section in sections:
59
+ units = self._units_for(section.blocks)
60
+ for text, oversized in self._pack(units):
61
+ metadata = document.metadata.copy()
62
+ metadata.update(
63
+ {
64
+ "header": section.header,
65
+ "header_level": section.header_level,
66
+ "header_path": section.header_path,
67
+ "page_numbers": sorted(section.page_numbers) or [1],
68
+ "chunk_index": len(chunks),
69
+ }
70
+ )
71
+ if oversized:
72
+ metadata["oversized"] = True
73
+ chunks.append(Chunk(content=text, metadata=metadata))
74
+
75
+ return chunks
76
+
77
+ # -- sectioning ---------------------------------------------------------
78
+
79
+ def _sections(self, blocks: list[Block]) -> list[_Section]:
80
+ sections: list[_Section] = []
81
+ header_stack: list[tuple[int, str]] = []
82
+ current_page = 1
83
+ current = _Section(page_numbers={current_page})
84
+
85
+ for block in blocks:
86
+ if block.kind == PAGE_MARKER:
87
+ current_page = block.page_number
88
+ current.page_numbers.add(current_page)
89
+ continue
90
+
91
+ if block.kind == HEADING:
92
+ if current.blocks:
93
+ sections.append(current)
94
+ while header_stack and header_stack[-1][0] >= block.level:
95
+ header_stack.pop()
96
+ header_stack.append((block.level, block.title))
97
+ current = _Section(
98
+ header=block.title,
99
+ header_level=block.level,
100
+ header_path=" > ".join(title for _, title in header_stack),
101
+ blocks=[block],
102
+ page_numbers={current_page},
103
+ )
104
+ continue
105
+
106
+ current.blocks.append(block)
107
+
108
+ if current.blocks:
109
+ sections.append(current)
110
+ return sections
111
+
112
+ # -- unit expansion -----------------------------------------------------
113
+
114
+ def _units_for(self, blocks: list[Block]) -> list[_Unit]:
115
+ units: list[_Unit] = []
116
+ for block in blocks:
117
+ if self._fits(block.text):
118
+ units.append(_Unit(block.text))
119
+ elif block.kind == TABLE:
120
+ units.extend(_Unit(part) for part in self._split_table(block))
121
+ elif block.kind == CODE:
122
+ # Splitting a fenced block would leave unterminated fences, so it
123
+ # travels whole and is flagged for the caller to notice.
124
+ units.append(_Unit(block.text, oversized=True))
125
+ else:
126
+ units.extend(_Unit(part) for part in self._split_prose(block.text))
127
+ return units
128
+
129
+ def _split_table(self, block: Block) -> list[str]:
130
+ """Split by row, repeating the header so each part reads on its own."""
131
+ if not block.rows:
132
+ return [block.text]
133
+
134
+ header, *data_rows = block.rows
135
+ header_markdown = [_row_markdown(header), _separator_markdown(len(header))]
136
+
137
+ parts: list[str] = []
138
+ pending: list[str] = []
139
+
140
+ for row in data_rows:
141
+ candidate = pending + [_row_markdown(row)]
142
+ # Measure the rendered candidate rather than summing row sizes: a
143
+ # token counter does not charge exactly one unit per newline.
144
+ if pending and not self._fits("\n".join(header_markdown + candidate)):
145
+ parts.append("\n".join(header_markdown + pending))
146
+ pending = [_row_markdown(row)]
147
+ else:
148
+ pending = candidate
149
+
150
+ if pending:
151
+ parts.append("\n".join(header_markdown + pending))
152
+ return parts or [block.text]
153
+
154
+ def _split_prose(self, text: str) -> list[str]:
155
+ parts: list[str] = []
156
+ pending = ""
157
+ for sentence in SENTENCE_END_RE.split(text):
158
+ if not sentence:
159
+ continue
160
+ candidate = f"{pending} {sentence}".strip() if pending else sentence
161
+ if pending and not self._fits(candidate):
162
+ parts.append(pending)
163
+ pending = sentence
164
+ else:
165
+ pending = candidate
166
+
167
+ while not self._fits(pending):
168
+ head, pending = self._cut_to_fit(pending)
169
+ parts.append(head)
170
+
171
+ if pending:
172
+ parts.append(pending)
173
+ return parts
174
+
175
+ def _cut_to_fit(self, text: str) -> tuple[str, str]:
176
+ """Hard-split a run that no sentence boundary can bring under budget."""
177
+ low, high, best = 1, len(text), 1
178
+ while low <= high:
179
+ middle = (low + high) // 2
180
+ if self.length_function(text[:middle]) <= self.max_chunk_size:
181
+ best = middle
182
+ low = middle + 1
183
+ else:
184
+ high = middle - 1
185
+ return text[:best], text[best:]
186
+
187
+ # -- packing ------------------------------------------------------------
188
+
189
+ def _pack(self, units: list[_Unit]) -> list[tuple[str, bool]]:
190
+ packed: list[tuple[str, bool]] = []
191
+ pending: list[str] = []
192
+
193
+ def flush():
194
+ if pending:
195
+ packed.append((JOIN.join(pending), False))
196
+ pending.clear()
197
+
198
+ for unit in units:
199
+ if unit.oversized:
200
+ flush()
201
+ packed.append((unit.text, True))
202
+ continue
203
+
204
+ if pending and not self._fits(JOIN.join(pending + [unit.text])):
205
+ flush()
206
+ pending.append(unit.text)
207
+
208
+ flush()
209
+ return self._apply_overlap(packed)
210
+
211
+ def _apply_overlap(self, packed: list[tuple[str, bool]]) -> list[tuple[str, bool]]:
212
+ if self.overlap <= 0 or len(packed) < 2:
213
+ return packed
214
+
215
+ result = [packed[0]]
216
+ for index in range(1, len(packed)):
217
+ text, oversized = packed[index]
218
+ tail = self._tail(packed[index - 1][0])
219
+ result.append((f"{tail}{JOIN}{text}" if tail else text, oversized))
220
+ return result
221
+
222
+ def _tail(self, text: str) -> str:
223
+ """Longest whitespace-aligned suffix that fits the overlap budget."""
224
+ words = text.split()
225
+ tail = ""
226
+ for count in range(1, len(words) + 1):
227
+ candidate = " ".join(words[-count:])
228
+ if self.length_function(candidate) > self.overlap:
229
+ break
230
+ tail = candidate
231
+ return tail
232
+
233
+ def _fits(self, text: str) -> bool:
234
+ return self.length_function(text) <= self.max_chunk_size
235
+
236
+
237
+ def _row_markdown(row: list[str]) -> str:
238
+ return f"| {' | '.join(row)} |"
239
+
240
+
241
+ def _separator_markdown(width: int) -> str:
242
+ return f"| {' | '.join(['---'] * width)} |"