langparse 0.1.0rc1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +35 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +0 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/workbook.py +900 -0
- langparse/cli.py +257 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +202 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +140 -0
- langparse/engines/pdf/mineru.py +235 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +127 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +52 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +190 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +5 -0
- langparse/services/batch_service.py +253 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +468 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/types.py +94 -0
- langparse/workbooks/__init__.py +97 -0
- langparse/workbooks/adapters.py +309 -0
- langparse/workbooks/assembly.py +928 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/classification.py +381 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/regions.py +77 -0
- langparse/workbooks/rendering.py +216 -0
- langparse/workbooks/tables.py +375 -0
- langparse/workbooks/types.py +239 -0
- langparse-0.1.0rc1.dist-info/METADATA +720 -0
- langparse-0.1.0rc1.dist-info/RECORD +85 -0
- langparse-0.1.0rc1.dist-info/WHEEL +5 -0
- langparse-0.1.0rc1.dist-info/entry_points.txt +2 -0
- langparse-0.1.0rc1.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0rc1.dist-info/top_level.txt +1 -0
langparse/__init__.py
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
from langparse.autoparser import AutoParser
|
|
2
|
+
from langparse.chunkers.semantic import SemanticChunker
|
|
3
|
+
from langparse.core.chunker import BaseChunker
|
|
4
|
+
from langparse.core.parser import BaseParser
|
|
5
|
+
from langparse.metrics import BatchItemResult, BatchRunResult, ParseMetrics
|
|
6
|
+
from langparse.parsers.docx_parser import DocxParser
|
|
7
|
+
from langparse.parsers.excel_parser import ExcelParser
|
|
8
|
+
from langparse.parsers.markdown_parser import MarkdownParser
|
|
9
|
+
from langparse.parsers.pdf_parser import PDFParser
|
|
10
|
+
from langparse.types import (
|
|
11
|
+
Chunk,
|
|
12
|
+
Document,
|
|
13
|
+
ParsedDocumentResult,
|
|
14
|
+
ParsedElement,
|
|
15
|
+
ParsedPageResult,
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
__all__ = [
|
|
19
|
+
"Document",
|
|
20
|
+
"Chunk",
|
|
21
|
+
"ParsedDocumentResult",
|
|
22
|
+
"ParsedPageResult",
|
|
23
|
+
"ParsedElement",
|
|
24
|
+
"BaseParser",
|
|
25
|
+
"BaseChunker",
|
|
26
|
+
"AutoParser",
|
|
27
|
+
"PDFParser",
|
|
28
|
+
"MarkdownParser",
|
|
29
|
+
"DocxParser",
|
|
30
|
+
"ExcelParser",
|
|
31
|
+
"SemanticChunker",
|
|
32
|
+
"ParseMetrics",
|
|
33
|
+
"BatchItemResult",
|
|
34
|
+
"BatchRunResult",
|
|
35
|
+
]
|
langparse/autoparser.py
ADDED
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
from pathlib import Path
|
|
2
|
+
|
|
3
|
+
from langparse.core.rendering import document_from_result
|
|
4
|
+
from langparse.types import Document, ParsedDocumentResult
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class AutoParser:
|
|
8
|
+
"""
|
|
9
|
+
Facade that parses any supported file without the caller picking a parser.
|
|
10
|
+
|
|
11
|
+
Extension routing lives in `ParseService`, driven by
|
|
12
|
+
`langparse.parsers.registry`, so this stays a convenience wrapper rather
|
|
13
|
+
than a second place formats can be registered and drift.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
@staticmethod
|
|
17
|
+
def parse_result(file_path: str | Path, **kwargs) -> ParsedDocumentResult:
|
|
18
|
+
from langparse.services.parse_service import ParseService
|
|
19
|
+
|
|
20
|
+
engine_name = kwargs.pop("engine", None) or "simple"
|
|
21
|
+
return ParseService().parse_result(file_path, engine_name=engine_name, **kwargs)
|
|
22
|
+
|
|
23
|
+
@staticmethod
|
|
24
|
+
def parse(file_path: str | Path, **kwargs) -> Document:
|
|
25
|
+
return document_from_result(AutoParser.parse_result(file_path, **kwargs))
|
|
File without changes
|
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Scan Markdown into typed blocks.
|
|
3
|
+
|
|
4
|
+
Chunking used to run a heading regex over the whole document, which cannot tell
|
|
5
|
+
a real heading from a `#` comment inside a fenced code block -- recognising a
|
|
6
|
+
fence requires tracking state across lines. Scanning into typed blocks fixes
|
|
7
|
+
that structurally, and the same block types are what let the packer treat
|
|
8
|
+
tables and code differently when they overflow a chunk.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import re
|
|
14
|
+
from dataclasses import dataclass, field
|
|
15
|
+
|
|
16
|
+
HEADING_RE = re.compile(r"^(#{1,6})\s+(.+?)\s*$")
|
|
17
|
+
FENCE_RE = re.compile(r"^(\s*)(`{3,}|~{3,})(.*)$")
|
|
18
|
+
PAGE_MARKER_RE = re.compile(r"^\s*<!--\s*page_number:\s*(\d+)\s*-->\s*$")
|
|
19
|
+
TABLE_SEPARATOR_RE = re.compile(r"^\s*\|?\s*:?-{3,}:?\s*(\|\s*:?-{3,}:?\s*)*\|?\s*$")
|
|
20
|
+
|
|
21
|
+
HEADING = "heading"
|
|
22
|
+
TABLE = "table"
|
|
23
|
+
CODE = "code"
|
|
24
|
+
PARAGRAPH = "paragraph"
|
|
25
|
+
PAGE_MARKER = "page_marker"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass
|
|
29
|
+
class Block:
|
|
30
|
+
kind: str
|
|
31
|
+
text: str
|
|
32
|
+
#: heading only
|
|
33
|
+
level: int = 0
|
|
34
|
+
title: str = ""
|
|
35
|
+
#: table only
|
|
36
|
+
rows: list[list[str]] = field(default_factory=list)
|
|
37
|
+
has_header: bool = False
|
|
38
|
+
#: page_marker only
|
|
39
|
+
page_number: int = 0
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def scan_blocks(markdown: str) -> list[Block]:
|
|
43
|
+
"""Split Markdown into an ordered list of typed blocks."""
|
|
44
|
+
if not markdown:
|
|
45
|
+
return []
|
|
46
|
+
|
|
47
|
+
lines = markdown.splitlines()
|
|
48
|
+
blocks: list[Block] = []
|
|
49
|
+
index = 0
|
|
50
|
+
|
|
51
|
+
while index < len(lines):
|
|
52
|
+
line = lines[index]
|
|
53
|
+
|
|
54
|
+
if not line.strip():
|
|
55
|
+
index += 1
|
|
56
|
+
continue
|
|
57
|
+
|
|
58
|
+
fence = FENCE_RE.match(line)
|
|
59
|
+
if fence:
|
|
60
|
+
index = _consume_fence(lines, index, fence.group(2), blocks)
|
|
61
|
+
continue
|
|
62
|
+
|
|
63
|
+
page_marker = PAGE_MARKER_RE.match(line)
|
|
64
|
+
if page_marker:
|
|
65
|
+
blocks.append(Block(kind=PAGE_MARKER, text=line, page_number=int(page_marker.group(1))))
|
|
66
|
+
index += 1
|
|
67
|
+
continue
|
|
68
|
+
|
|
69
|
+
heading = HEADING_RE.match(line)
|
|
70
|
+
if heading:
|
|
71
|
+
blocks.append(
|
|
72
|
+
Block(
|
|
73
|
+
kind=HEADING,
|
|
74
|
+
text=line,
|
|
75
|
+
level=len(heading.group(1)),
|
|
76
|
+
title=heading.group(2),
|
|
77
|
+
)
|
|
78
|
+
)
|
|
79
|
+
index += 1
|
|
80
|
+
continue
|
|
81
|
+
|
|
82
|
+
if _starts_table(lines, index):
|
|
83
|
+
index = _consume_table(lines, index, blocks)
|
|
84
|
+
continue
|
|
85
|
+
|
|
86
|
+
index = _consume_paragraph(lines, index, blocks)
|
|
87
|
+
|
|
88
|
+
return blocks
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _consume_fence(lines: list[str], index: int, marker: str, blocks: list[Block]) -> int:
|
|
92
|
+
"""Consume a fenced block. An unclosed fence runs to end of input."""
|
|
93
|
+
fence_char = marker[0]
|
|
94
|
+
collected = [lines[index]]
|
|
95
|
+
index += 1
|
|
96
|
+
|
|
97
|
+
while index < len(lines):
|
|
98
|
+
collected.append(lines[index])
|
|
99
|
+
closing = FENCE_RE.match(lines[index])
|
|
100
|
+
index += 1
|
|
101
|
+
if closing and closing.group(2)[0] == fence_char and len(closing.group(2)) >= len(marker):
|
|
102
|
+
break
|
|
103
|
+
|
|
104
|
+
blocks.append(Block(kind=CODE, text="\n".join(collected)))
|
|
105
|
+
return index
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _starts_table(lines: list[str], index: int) -> bool:
|
|
109
|
+
"""A table needs a pipe row followed by a separator row; pipes alone are prose."""
|
|
110
|
+
if not lines[index].lstrip().startswith("|"):
|
|
111
|
+
return False
|
|
112
|
+
return index + 1 < len(lines) and bool(TABLE_SEPARATOR_RE.match(lines[index + 1]))
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _consume_table(lines: list[str], index: int, blocks: list[Block]) -> int:
|
|
116
|
+
collected: list[str] = []
|
|
117
|
+
while index < len(lines) and lines[index].lstrip().startswith("|"):
|
|
118
|
+
collected.append(lines[index])
|
|
119
|
+
index += 1
|
|
120
|
+
|
|
121
|
+
rows = [_split_table_row(line) for line in collected if not TABLE_SEPARATOR_RE.match(line)]
|
|
122
|
+
blocks.append(Block(kind=TABLE, text="\n".join(collected), rows=rows, has_header=bool(rows)))
|
|
123
|
+
return index
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def _split_table_row(line: str) -> list[str]:
|
|
127
|
+
stripped = line.strip()
|
|
128
|
+
if stripped.startswith("|"):
|
|
129
|
+
stripped = stripped[1:]
|
|
130
|
+
if stripped.endswith("|"):
|
|
131
|
+
stripped = stripped[:-1]
|
|
132
|
+
return [cell.strip() for cell in stripped.split("|")]
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _consume_paragraph(lines: list[str], index: int, blocks: list[Block]) -> int:
|
|
136
|
+
collected: list[str] = []
|
|
137
|
+
while index < len(lines) and lines[index].strip():
|
|
138
|
+
line = lines[index]
|
|
139
|
+
if (
|
|
140
|
+
FENCE_RE.match(line)
|
|
141
|
+
or HEADING_RE.match(line)
|
|
142
|
+
or PAGE_MARKER_RE.match(line)
|
|
143
|
+
or _starts_table(lines, index)
|
|
144
|
+
):
|
|
145
|
+
break
|
|
146
|
+
collected.append(line)
|
|
147
|
+
index += 1
|
|
148
|
+
|
|
149
|
+
if collected:
|
|
150
|
+
blocks.append(Block(kind=PARAGRAPH, text="\n".join(collected)))
|
|
151
|
+
return index
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
from enum import Enum
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class WorkbookChunkProfile(str, Enum):
|
|
8
|
+
RETRIEVAL = "retrieval"
|
|
9
|
+
ANALYSIS = "analysis"
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@dataclass(frozen=True)
|
|
13
|
+
class WorkbookChunkPolicy:
|
|
14
|
+
name: WorkbookChunkProfile
|
|
15
|
+
version: int
|
|
16
|
+
default_max_chunk_size: int
|
|
17
|
+
analysis_records: bool
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class ChunkProfileNotSupportedError(ValueError):
|
|
21
|
+
"""Raised when a valid chunk profile cannot represent the parsed input."""
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
_POLICIES = {
|
|
25
|
+
WorkbookChunkProfile.RETRIEVAL: WorkbookChunkPolicy(
|
|
26
|
+
name=WorkbookChunkProfile.RETRIEVAL,
|
|
27
|
+
version=1,
|
|
28
|
+
default_max_chunk_size=1000,
|
|
29
|
+
analysis_records=False,
|
|
30
|
+
),
|
|
31
|
+
WorkbookChunkProfile.ANALYSIS: WorkbookChunkPolicy(
|
|
32
|
+
name=WorkbookChunkProfile.ANALYSIS,
|
|
33
|
+
version=1,
|
|
34
|
+
default_max_chunk_size=4000,
|
|
35
|
+
analysis_records=True,
|
|
36
|
+
),
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def resolve_workbook_chunk_policy(
|
|
41
|
+
profile: str | WorkbookChunkProfile | None,
|
|
42
|
+
) -> WorkbookChunkPolicy:
|
|
43
|
+
if profile is None:
|
|
44
|
+
selected = WorkbookChunkProfile.RETRIEVAL
|
|
45
|
+
else:
|
|
46
|
+
try:
|
|
47
|
+
selected = WorkbookChunkProfile(profile)
|
|
48
|
+
except ValueError:
|
|
49
|
+
available = ", ".join(sorted(item.value for item in WorkbookChunkProfile))
|
|
50
|
+
raise ValueError(
|
|
51
|
+
f"Unknown workbook chunk profile {profile!r}. Available: {available}"
|
|
52
|
+
) from None
|
|
53
|
+
return _POLICIES[selected]
|
|
@@ -0,0 +1,242 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from collections.abc import Callable
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
|
|
7
|
+
from langparse.chunkers.blocks import CODE, HEADING, PAGE_MARKER, TABLE, Block, scan_blocks
|
|
8
|
+
from langparse.core.chunker import BaseChunker
|
|
9
|
+
from langparse.types import Chunk, Document
|
|
10
|
+
|
|
11
|
+
SENTENCE_END_RE = re.compile(r"(?<=[.!?。!?])\s+")
|
|
12
|
+
JOIN = "\n\n"
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass
|
|
16
|
+
class _Section:
|
|
17
|
+
header: str | None = None
|
|
18
|
+
header_level: int = 0
|
|
19
|
+
header_path: str = ""
|
|
20
|
+
blocks: list[Block] = field(default_factory=list)
|
|
21
|
+
page_numbers: set = field(default_factory=set)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass
|
|
25
|
+
class _Unit:
|
|
26
|
+
"""One packable piece of content. Oversized units get a chunk to themselves."""
|
|
27
|
+
|
|
28
|
+
text: str
|
|
29
|
+
oversized: bool = False
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class SemanticChunker(BaseChunker):
|
|
33
|
+
"""
|
|
34
|
+
Chunks text on Markdown structure, keeping chunks within a size budget.
|
|
35
|
+
|
|
36
|
+
Sections come from heading structure; within a section, blocks are packed
|
|
37
|
+
greedily up to `max_chunk_size`. Size is measured by `length_function`, so
|
|
38
|
+
callers embedding against a token budget can pass a tokenizer's encoder
|
|
39
|
+
instead of the default character count.
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
def __init__(
|
|
43
|
+
self,
|
|
44
|
+
max_chunk_size: int = 1000,
|
|
45
|
+
overlap: int = 0,
|
|
46
|
+
length_function: Callable[[str], int] = len,
|
|
47
|
+
):
|
|
48
|
+
if overlap >= max_chunk_size:
|
|
49
|
+
raise ValueError("overlap must be smaller than max_chunk_size")
|
|
50
|
+
self.max_chunk_size = max_chunk_size
|
|
51
|
+
self.overlap = overlap
|
|
52
|
+
self.length_function = length_function
|
|
53
|
+
|
|
54
|
+
def chunk(self, document: Document, **kwargs) -> list[Chunk]:
|
|
55
|
+
sections = self._sections(scan_blocks(document.content))
|
|
56
|
+
|
|
57
|
+
chunks: list[Chunk] = []
|
|
58
|
+
for section in sections:
|
|
59
|
+
units = self._units_for(section.blocks)
|
|
60
|
+
for text, oversized in self._pack(units):
|
|
61
|
+
metadata = document.metadata.copy()
|
|
62
|
+
metadata.update(
|
|
63
|
+
{
|
|
64
|
+
"header": section.header,
|
|
65
|
+
"header_level": section.header_level,
|
|
66
|
+
"header_path": section.header_path,
|
|
67
|
+
"page_numbers": sorted(section.page_numbers) or [1],
|
|
68
|
+
"chunk_index": len(chunks),
|
|
69
|
+
}
|
|
70
|
+
)
|
|
71
|
+
if oversized:
|
|
72
|
+
metadata["oversized"] = True
|
|
73
|
+
chunks.append(Chunk(content=text, metadata=metadata))
|
|
74
|
+
|
|
75
|
+
return chunks
|
|
76
|
+
|
|
77
|
+
# -- sectioning ---------------------------------------------------------
|
|
78
|
+
|
|
79
|
+
def _sections(self, blocks: list[Block]) -> list[_Section]:
|
|
80
|
+
sections: list[_Section] = []
|
|
81
|
+
header_stack: list[tuple[int, str]] = []
|
|
82
|
+
current_page = 1
|
|
83
|
+
current = _Section(page_numbers={current_page})
|
|
84
|
+
|
|
85
|
+
for block in blocks:
|
|
86
|
+
if block.kind == PAGE_MARKER:
|
|
87
|
+
current_page = block.page_number
|
|
88
|
+
current.page_numbers.add(current_page)
|
|
89
|
+
continue
|
|
90
|
+
|
|
91
|
+
if block.kind == HEADING:
|
|
92
|
+
if current.blocks:
|
|
93
|
+
sections.append(current)
|
|
94
|
+
while header_stack and header_stack[-1][0] >= block.level:
|
|
95
|
+
header_stack.pop()
|
|
96
|
+
header_stack.append((block.level, block.title))
|
|
97
|
+
current = _Section(
|
|
98
|
+
header=block.title,
|
|
99
|
+
header_level=block.level,
|
|
100
|
+
header_path=" > ".join(title for _, title in header_stack),
|
|
101
|
+
blocks=[block],
|
|
102
|
+
page_numbers={current_page},
|
|
103
|
+
)
|
|
104
|
+
continue
|
|
105
|
+
|
|
106
|
+
current.blocks.append(block)
|
|
107
|
+
|
|
108
|
+
if current.blocks:
|
|
109
|
+
sections.append(current)
|
|
110
|
+
return sections
|
|
111
|
+
|
|
112
|
+
# -- unit expansion -----------------------------------------------------
|
|
113
|
+
|
|
114
|
+
def _units_for(self, blocks: list[Block]) -> list[_Unit]:
|
|
115
|
+
units: list[_Unit] = []
|
|
116
|
+
for block in blocks:
|
|
117
|
+
if self._fits(block.text):
|
|
118
|
+
units.append(_Unit(block.text))
|
|
119
|
+
elif block.kind == TABLE:
|
|
120
|
+
units.extend(_Unit(part) for part in self._split_table(block))
|
|
121
|
+
elif block.kind == CODE:
|
|
122
|
+
# Splitting a fenced block would leave unterminated fences, so it
|
|
123
|
+
# travels whole and is flagged for the caller to notice.
|
|
124
|
+
units.append(_Unit(block.text, oversized=True))
|
|
125
|
+
else:
|
|
126
|
+
units.extend(_Unit(part) for part in self._split_prose(block.text))
|
|
127
|
+
return units
|
|
128
|
+
|
|
129
|
+
def _split_table(self, block: Block) -> list[str]:
|
|
130
|
+
"""Split by row, repeating the header so each part reads on its own."""
|
|
131
|
+
if not block.rows:
|
|
132
|
+
return [block.text]
|
|
133
|
+
|
|
134
|
+
header, *data_rows = block.rows
|
|
135
|
+
header_markdown = [_row_markdown(header), _separator_markdown(len(header))]
|
|
136
|
+
|
|
137
|
+
parts: list[str] = []
|
|
138
|
+
pending: list[str] = []
|
|
139
|
+
|
|
140
|
+
for row in data_rows:
|
|
141
|
+
candidate = pending + [_row_markdown(row)]
|
|
142
|
+
# Measure the rendered candidate rather than summing row sizes: a
|
|
143
|
+
# token counter does not charge exactly one unit per newline.
|
|
144
|
+
if pending and not self._fits("\n".join(header_markdown + candidate)):
|
|
145
|
+
parts.append("\n".join(header_markdown + pending))
|
|
146
|
+
pending = [_row_markdown(row)]
|
|
147
|
+
else:
|
|
148
|
+
pending = candidate
|
|
149
|
+
|
|
150
|
+
if pending:
|
|
151
|
+
parts.append("\n".join(header_markdown + pending))
|
|
152
|
+
return parts or [block.text]
|
|
153
|
+
|
|
154
|
+
def _split_prose(self, text: str) -> list[str]:
|
|
155
|
+
parts: list[str] = []
|
|
156
|
+
pending = ""
|
|
157
|
+
for sentence in SENTENCE_END_RE.split(text):
|
|
158
|
+
if not sentence:
|
|
159
|
+
continue
|
|
160
|
+
candidate = f"{pending} {sentence}".strip() if pending else sentence
|
|
161
|
+
if pending and not self._fits(candidate):
|
|
162
|
+
parts.append(pending)
|
|
163
|
+
pending = sentence
|
|
164
|
+
else:
|
|
165
|
+
pending = candidate
|
|
166
|
+
|
|
167
|
+
while not self._fits(pending):
|
|
168
|
+
head, pending = self._cut_to_fit(pending)
|
|
169
|
+
parts.append(head)
|
|
170
|
+
|
|
171
|
+
if pending:
|
|
172
|
+
parts.append(pending)
|
|
173
|
+
return parts
|
|
174
|
+
|
|
175
|
+
def _cut_to_fit(self, text: str) -> tuple[str, str]:
|
|
176
|
+
"""Hard-split a run that no sentence boundary can bring under budget."""
|
|
177
|
+
low, high, best = 1, len(text), 1
|
|
178
|
+
while low <= high:
|
|
179
|
+
middle = (low + high) // 2
|
|
180
|
+
if self.length_function(text[:middle]) <= self.max_chunk_size:
|
|
181
|
+
best = middle
|
|
182
|
+
low = middle + 1
|
|
183
|
+
else:
|
|
184
|
+
high = middle - 1
|
|
185
|
+
return text[:best], text[best:]
|
|
186
|
+
|
|
187
|
+
# -- packing ------------------------------------------------------------
|
|
188
|
+
|
|
189
|
+
def _pack(self, units: list[_Unit]) -> list[tuple[str, bool]]:
|
|
190
|
+
packed: list[tuple[str, bool]] = []
|
|
191
|
+
pending: list[str] = []
|
|
192
|
+
|
|
193
|
+
def flush():
|
|
194
|
+
if pending:
|
|
195
|
+
packed.append((JOIN.join(pending), False))
|
|
196
|
+
pending.clear()
|
|
197
|
+
|
|
198
|
+
for unit in units:
|
|
199
|
+
if unit.oversized:
|
|
200
|
+
flush()
|
|
201
|
+
packed.append((unit.text, True))
|
|
202
|
+
continue
|
|
203
|
+
|
|
204
|
+
if pending and not self._fits(JOIN.join(pending + [unit.text])):
|
|
205
|
+
flush()
|
|
206
|
+
pending.append(unit.text)
|
|
207
|
+
|
|
208
|
+
flush()
|
|
209
|
+
return self._apply_overlap(packed)
|
|
210
|
+
|
|
211
|
+
def _apply_overlap(self, packed: list[tuple[str, bool]]) -> list[tuple[str, bool]]:
|
|
212
|
+
if self.overlap <= 0 or len(packed) < 2:
|
|
213
|
+
return packed
|
|
214
|
+
|
|
215
|
+
result = [packed[0]]
|
|
216
|
+
for index in range(1, len(packed)):
|
|
217
|
+
text, oversized = packed[index]
|
|
218
|
+
tail = self._tail(packed[index - 1][0])
|
|
219
|
+
result.append((f"{tail}{JOIN}{text}" if tail else text, oversized))
|
|
220
|
+
return result
|
|
221
|
+
|
|
222
|
+
def _tail(self, text: str) -> str:
|
|
223
|
+
"""Longest whitespace-aligned suffix that fits the overlap budget."""
|
|
224
|
+
words = text.split()
|
|
225
|
+
tail = ""
|
|
226
|
+
for count in range(1, len(words) + 1):
|
|
227
|
+
candidate = " ".join(words[-count:])
|
|
228
|
+
if self.length_function(candidate) > self.overlap:
|
|
229
|
+
break
|
|
230
|
+
tail = candidate
|
|
231
|
+
return tail
|
|
232
|
+
|
|
233
|
+
def _fits(self, text: str) -> bool:
|
|
234
|
+
return self.length_function(text) <= self.max_chunk_size
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
def _row_markdown(row: list[str]) -> str:
|
|
238
|
+
return f"| {' | '.join(row)} |"
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def _separator_markdown(width: int) -> str:
|
|
242
|
+
return f"| {' | '.join(['---'] * width)} |"
|