langparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +55 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +12 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/registry.py +38 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/text.py +96 -0
- langparse/chunkers/workbook.py +942 -0
- langparse/cli.py +329 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +210 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +164 -0
- langparse/engines/pdf/mineru.py +259 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +134 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +70 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +220 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/progress.py +77 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +11 -0
- langparse/services/batch_service.py +339 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +523 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/services/workbook_quality_benchmark.py +230 -0
- langparse/types.py +97 -0
- langparse/workbooks/__init__.py +103 -0
- langparse/workbooks/adapters.py +474 -0
- langparse/workbooks/assembly.py +993 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/bundle-v1.schema.json +71 -0
- langparse/workbooks/bundle.py +341 -0
- langparse/workbooks/classification.py +393 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/labels.py +14 -0
- langparse/workbooks/lineage.py +117 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/objects.py +229 -0
- langparse/workbooks/quality/__init__.py +23 -0
- langparse/workbooks/quality/bundle.py +53 -0
- langparse/workbooks/quality/evaluator.py +266 -0
- langparse/workbooks/quality/facts.py +142 -0
- langparse/workbooks/quality/schema.py +462 -0
- langparse/workbooks/reference_types.py +73 -0
- langparse/workbooks/references.py +178 -0
- langparse/workbooks/regions.py +932 -0
- langparse/workbooks/rendering.py +222 -0
- langparse/workbooks/tables.py +477 -0
- langparse/workbooks/types.py +257 -0
- langparse-0.1.0.dist-info/METADATA +790 -0
- langparse-0.1.0.dist-info/RECORD +101 -0
- langparse-0.1.0.dist-info/WHEEL +5 -0
- langparse-0.1.0.dist-info/entry_points.txt +2 -0
- langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0.dist-info/top_level.txt +1 -0
langparse/__init__.py
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
from importlib.metadata import version as _distribution_version
|
|
2
|
+
|
|
3
|
+
__version__ = _distribution_version("langparse")
|
|
4
|
+
|
|
5
|
+
from langparse.autoparser import AutoParser
|
|
6
|
+
from langparse.chunkers import (
|
|
7
|
+
FixedTokenChunker,
|
|
8
|
+
SemanticChunker,
|
|
9
|
+
SlidingWindowChunker,
|
|
10
|
+
available_chunkers,
|
|
11
|
+
create_chunker,
|
|
12
|
+
register_chunker,
|
|
13
|
+
)
|
|
14
|
+
from langparse.core.chunker import BaseChunker
|
|
15
|
+
from langparse.core.parser import BaseParser
|
|
16
|
+
from langparse.metrics import BatchItemResult, BatchRunResult, ParseMetrics
|
|
17
|
+
from langparse.parsers.docx_parser import DocxParser
|
|
18
|
+
from langparse.parsers.excel_parser import ExcelParser
|
|
19
|
+
from langparse.parsers.markdown_parser import MarkdownParser
|
|
20
|
+
from langparse.parsers.pdf_parser import PDFParser
|
|
21
|
+
from langparse.progress import ProgressCallback, ProgressEvent
|
|
22
|
+
from langparse.types import (
|
|
23
|
+
Chunk,
|
|
24
|
+
Document,
|
|
25
|
+
ParsedDocumentResult,
|
|
26
|
+
ParsedElement,
|
|
27
|
+
ParsedPageResult,
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
__all__ = [
|
|
31
|
+
"__version__",
|
|
32
|
+
"Document",
|
|
33
|
+
"Chunk",
|
|
34
|
+
"ParsedDocumentResult",
|
|
35
|
+
"ParsedPageResult",
|
|
36
|
+
"ParsedElement",
|
|
37
|
+
"BaseParser",
|
|
38
|
+
"BaseChunker",
|
|
39
|
+
"AutoParser",
|
|
40
|
+
"PDFParser",
|
|
41
|
+
"MarkdownParser",
|
|
42
|
+
"DocxParser",
|
|
43
|
+
"ExcelParser",
|
|
44
|
+
"SemanticChunker",
|
|
45
|
+
"FixedTokenChunker",
|
|
46
|
+
"SlidingWindowChunker",
|
|
47
|
+
"available_chunkers",
|
|
48
|
+
"create_chunker",
|
|
49
|
+
"register_chunker",
|
|
50
|
+
"ParseMetrics",
|
|
51
|
+
"ProgressCallback",
|
|
52
|
+
"ProgressEvent",
|
|
53
|
+
"BatchItemResult",
|
|
54
|
+
"BatchRunResult",
|
|
55
|
+
]
|
langparse/autoparser.py
ADDED
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
from pathlib import Path
|
|
2
|
+
|
|
3
|
+
from langparse.core.rendering import document_from_result
|
|
4
|
+
from langparse.types import Document, ParsedDocumentResult
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class AutoParser:
|
|
8
|
+
"""
|
|
9
|
+
Facade that parses any supported file without the caller picking a parser.
|
|
10
|
+
|
|
11
|
+
Extension routing lives in `ParseService`, driven by
|
|
12
|
+
`langparse.parsers.registry`, so this stays a convenience wrapper rather
|
|
13
|
+
than a second place formats can be registered and drift.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
@staticmethod
|
|
17
|
+
def parse_result(file_path: str | Path, **kwargs) -> ParsedDocumentResult:
|
|
18
|
+
from langparse.services.parse_service import ParseService
|
|
19
|
+
|
|
20
|
+
engine_name = kwargs.pop("engine", None) or "simple"
|
|
21
|
+
return ParseService().parse_result(file_path, engine_name=engine_name, **kwargs)
|
|
22
|
+
|
|
23
|
+
@staticmethod
|
|
24
|
+
def parse(file_path: str | Path, **kwargs) -> Document:
|
|
25
|
+
return document_from_result(AutoParser.parse_result(file_path, **kwargs))
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
from langparse.chunkers.registry import available_chunkers, create_chunker, register_chunker
|
|
2
|
+
from langparse.chunkers.semantic import SemanticChunker
|
|
3
|
+
from langparse.chunkers.text import FixedTokenChunker, SlidingWindowChunker
|
|
4
|
+
|
|
5
|
+
__all__ = [
|
|
6
|
+
"FixedTokenChunker",
|
|
7
|
+
"SlidingWindowChunker",
|
|
8
|
+
"SemanticChunker",
|
|
9
|
+
"available_chunkers",
|
|
10
|
+
"create_chunker",
|
|
11
|
+
"register_chunker",
|
|
12
|
+
]
|
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Scan Markdown into typed blocks.
|
|
3
|
+
|
|
4
|
+
Chunking used to run a heading regex over the whole document, which cannot tell
|
|
5
|
+
a real heading from a `#` comment inside a fenced code block -- recognising a
|
|
6
|
+
fence requires tracking state across lines. Scanning into typed blocks fixes
|
|
7
|
+
that structurally, and the same block types are what let the packer treat
|
|
8
|
+
tables and code differently when they overflow a chunk.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import re
|
|
14
|
+
from dataclasses import dataclass, field
|
|
15
|
+
|
|
16
|
+
HEADING_RE = re.compile(r"^(#{1,6})\s+(.+?)\s*$")
|
|
17
|
+
FENCE_RE = re.compile(r"^(\s*)(`{3,}|~{3,})(.*)$")
|
|
18
|
+
PAGE_MARKER_RE = re.compile(r"^\s*<!--\s*page_number:\s*(\d+)\s*-->\s*$")
|
|
19
|
+
TABLE_SEPARATOR_RE = re.compile(r"^\s*\|?\s*:?-{3,}:?\s*(\|\s*:?-{3,}:?\s*)*\|?\s*$")
|
|
20
|
+
|
|
21
|
+
HEADING = "heading"
|
|
22
|
+
TABLE = "table"
|
|
23
|
+
CODE = "code"
|
|
24
|
+
PARAGRAPH = "paragraph"
|
|
25
|
+
PAGE_MARKER = "page_marker"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass
|
|
29
|
+
class Block:
|
|
30
|
+
kind: str
|
|
31
|
+
text: str
|
|
32
|
+
#: heading only
|
|
33
|
+
level: int = 0
|
|
34
|
+
title: str = ""
|
|
35
|
+
#: table only
|
|
36
|
+
rows: list[list[str]] = field(default_factory=list)
|
|
37
|
+
has_header: bool = False
|
|
38
|
+
#: page_marker only
|
|
39
|
+
page_number: int = 0
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def scan_blocks(markdown: str) -> list[Block]:
|
|
43
|
+
"""Split Markdown into an ordered list of typed blocks."""
|
|
44
|
+
if not markdown:
|
|
45
|
+
return []
|
|
46
|
+
|
|
47
|
+
lines = markdown.splitlines()
|
|
48
|
+
blocks: list[Block] = []
|
|
49
|
+
index = 0
|
|
50
|
+
|
|
51
|
+
while index < len(lines):
|
|
52
|
+
line = lines[index]
|
|
53
|
+
|
|
54
|
+
if not line.strip():
|
|
55
|
+
index += 1
|
|
56
|
+
continue
|
|
57
|
+
|
|
58
|
+
fence = FENCE_RE.match(line)
|
|
59
|
+
if fence:
|
|
60
|
+
index = _consume_fence(lines, index, fence.group(2), blocks)
|
|
61
|
+
continue
|
|
62
|
+
|
|
63
|
+
page_marker = PAGE_MARKER_RE.match(line)
|
|
64
|
+
if page_marker:
|
|
65
|
+
blocks.append(Block(kind=PAGE_MARKER, text=line, page_number=int(page_marker.group(1))))
|
|
66
|
+
index += 1
|
|
67
|
+
continue
|
|
68
|
+
|
|
69
|
+
heading = HEADING_RE.match(line)
|
|
70
|
+
if heading:
|
|
71
|
+
blocks.append(
|
|
72
|
+
Block(
|
|
73
|
+
kind=HEADING,
|
|
74
|
+
text=line,
|
|
75
|
+
level=len(heading.group(1)),
|
|
76
|
+
title=heading.group(2),
|
|
77
|
+
)
|
|
78
|
+
)
|
|
79
|
+
index += 1
|
|
80
|
+
continue
|
|
81
|
+
|
|
82
|
+
if _starts_table(lines, index):
|
|
83
|
+
index = _consume_table(lines, index, blocks)
|
|
84
|
+
continue
|
|
85
|
+
|
|
86
|
+
index = _consume_paragraph(lines, index, blocks)
|
|
87
|
+
|
|
88
|
+
return blocks
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _consume_fence(lines: list[str], index: int, marker: str, blocks: list[Block]) -> int:
|
|
92
|
+
"""Consume a fenced block. An unclosed fence runs to end of input."""
|
|
93
|
+
fence_char = marker[0]
|
|
94
|
+
collected = [lines[index]]
|
|
95
|
+
index += 1
|
|
96
|
+
|
|
97
|
+
while index < len(lines):
|
|
98
|
+
collected.append(lines[index])
|
|
99
|
+
closing = FENCE_RE.match(lines[index])
|
|
100
|
+
index += 1
|
|
101
|
+
if closing and closing.group(2)[0] == fence_char and len(closing.group(2)) >= len(marker):
|
|
102
|
+
break
|
|
103
|
+
|
|
104
|
+
blocks.append(Block(kind=CODE, text="\n".join(collected)))
|
|
105
|
+
return index
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _starts_table(lines: list[str], index: int) -> bool:
|
|
109
|
+
"""A table needs a pipe row followed by a separator row; pipes alone are prose."""
|
|
110
|
+
if not lines[index].lstrip().startswith("|"):
|
|
111
|
+
return False
|
|
112
|
+
return index + 1 < len(lines) and bool(TABLE_SEPARATOR_RE.match(lines[index + 1]))
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _consume_table(lines: list[str], index: int, blocks: list[Block]) -> int:
|
|
116
|
+
collected: list[str] = []
|
|
117
|
+
while index < len(lines) and lines[index].lstrip().startswith("|"):
|
|
118
|
+
collected.append(lines[index])
|
|
119
|
+
index += 1
|
|
120
|
+
|
|
121
|
+
rows = [_split_table_row(line) for line in collected if not TABLE_SEPARATOR_RE.match(line)]
|
|
122
|
+
blocks.append(Block(kind=TABLE, text="\n".join(collected), rows=rows, has_header=bool(rows)))
|
|
123
|
+
return index
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def _split_table_row(line: str) -> list[str]:
|
|
127
|
+
stripped = line.strip()
|
|
128
|
+
if stripped.startswith("|"):
|
|
129
|
+
stripped = stripped[1:]
|
|
130
|
+
if stripped.endswith("|"):
|
|
131
|
+
stripped = stripped[:-1]
|
|
132
|
+
return [cell.strip() for cell in stripped.split("|")]
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _consume_paragraph(lines: list[str], index: int, blocks: list[Block]) -> int:
|
|
136
|
+
collected: list[str] = []
|
|
137
|
+
while index < len(lines) and lines[index].strip():
|
|
138
|
+
line = lines[index]
|
|
139
|
+
if (
|
|
140
|
+
FENCE_RE.match(line)
|
|
141
|
+
or HEADING_RE.match(line)
|
|
142
|
+
or PAGE_MARKER_RE.match(line)
|
|
143
|
+
or _starts_table(lines, index)
|
|
144
|
+
):
|
|
145
|
+
break
|
|
146
|
+
collected.append(line)
|
|
147
|
+
index += 1
|
|
148
|
+
|
|
149
|
+
if collected:
|
|
150
|
+
blocks.append(Block(kind=PARAGRAPH, text="\n".join(collected)))
|
|
151
|
+
return index
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
from enum import Enum
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class WorkbookChunkProfile(str, Enum):
|
|
8
|
+
RETRIEVAL = "retrieval"
|
|
9
|
+
ANALYSIS = "analysis"
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@dataclass(frozen=True)
|
|
13
|
+
class WorkbookChunkPolicy:
|
|
14
|
+
name: WorkbookChunkProfile
|
|
15
|
+
version: int
|
|
16
|
+
default_max_chunk_size: int
|
|
17
|
+
analysis_records: bool
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class ChunkProfileNotSupportedError(ValueError):
|
|
21
|
+
"""Raised when a valid chunk profile cannot represent the parsed input."""
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
_POLICIES = {
|
|
25
|
+
WorkbookChunkProfile.RETRIEVAL: WorkbookChunkPolicy(
|
|
26
|
+
name=WorkbookChunkProfile.RETRIEVAL,
|
|
27
|
+
version=1,
|
|
28
|
+
default_max_chunk_size=1000,
|
|
29
|
+
analysis_records=False,
|
|
30
|
+
),
|
|
31
|
+
WorkbookChunkProfile.ANALYSIS: WorkbookChunkPolicy(
|
|
32
|
+
name=WorkbookChunkProfile.ANALYSIS,
|
|
33
|
+
version=1,
|
|
34
|
+
default_max_chunk_size=4000,
|
|
35
|
+
analysis_records=True,
|
|
36
|
+
),
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def resolve_workbook_chunk_policy(
|
|
41
|
+
profile: str | WorkbookChunkProfile | None,
|
|
42
|
+
) -> WorkbookChunkPolicy:
|
|
43
|
+
if profile is None:
|
|
44
|
+
selected = WorkbookChunkProfile.RETRIEVAL
|
|
45
|
+
else:
|
|
46
|
+
try:
|
|
47
|
+
selected = WorkbookChunkProfile(profile)
|
|
48
|
+
except ValueError:
|
|
49
|
+
available = ", ".join(sorted(item.value for item in WorkbookChunkProfile))
|
|
50
|
+
raise ValueError(
|
|
51
|
+
f"Unknown workbook chunk profile {profile!r}. Available: {available}"
|
|
52
|
+
) from None
|
|
53
|
+
return _POLICIES[selected]
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""Explicit registry for document-text chunkers; workbook routing stays structural."""
|
|
2
|
+
|
|
3
|
+
from collections.abc import Callable
|
|
4
|
+
|
|
5
|
+
from langparse.chunkers.semantic import SemanticChunker
|
|
6
|
+
from langparse.chunkers.text import FixedTokenChunker, SlidingWindowChunker, _validate_budget
|
|
7
|
+
from langparse.core.chunker import BaseChunker
|
|
8
|
+
|
|
9
|
+
_FACTORIES: dict[str, Callable[..., BaseChunker]] = {
|
|
10
|
+
"semantic": SemanticChunker,
|
|
11
|
+
"fixed-token": FixedTokenChunker,
|
|
12
|
+
"sliding-window": SlidingWindowChunker,
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def register_chunker(name: str, factory: Callable[..., BaseChunker]) -> None:
|
|
17
|
+
if not isinstance(name, str) or not name.strip() or not callable(factory):
|
|
18
|
+
raise ValueError("A non-empty name and callable chunker factory are required")
|
|
19
|
+
if name in _FACTORIES:
|
|
20
|
+
raise ValueError(f"Chunk strategy {name!r} is already registered")
|
|
21
|
+
_FACTORIES[name] = factory
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def available_chunkers() -> tuple[str, ...]:
|
|
25
|
+
return tuple(sorted(_FACTORIES))
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def create_chunker(name: str = "semantic", **options) -> BaseChunker:
|
|
29
|
+
if name not in _FACTORIES:
|
|
30
|
+
raise ValueError(
|
|
31
|
+
f"Unknown chunk strategy {name!r}. Available: {', '.join(available_chunkers())}"
|
|
32
|
+
)
|
|
33
|
+
if name == "semantic":
|
|
34
|
+
_validate_budget(options.get("max_chunk_size", 1000), options.get("overlap", 0))
|
|
35
|
+
result = _FACTORIES[name](**options)
|
|
36
|
+
if not isinstance(result, BaseChunker):
|
|
37
|
+
raise TypeError("Chunker factory must return a BaseChunker")
|
|
38
|
+
return result
|
|
@@ -0,0 +1,242 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from collections.abc import Callable
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
|
|
7
|
+
from langparse.chunkers.blocks import CODE, HEADING, PAGE_MARKER, TABLE, Block, scan_blocks
|
|
8
|
+
from langparse.core.chunker import BaseChunker
|
|
9
|
+
from langparse.types import Chunk, Document
|
|
10
|
+
|
|
11
|
+
SENTENCE_END_RE = re.compile(r"(?<=[.!?。!?])\s+")
|
|
12
|
+
JOIN = "\n\n"
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass
|
|
16
|
+
class _Section:
|
|
17
|
+
header: str | None = None
|
|
18
|
+
header_level: int = 0
|
|
19
|
+
header_path: str = ""
|
|
20
|
+
blocks: list[Block] = field(default_factory=list)
|
|
21
|
+
page_numbers: set = field(default_factory=set)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass
|
|
25
|
+
class _Unit:
|
|
26
|
+
"""One packable piece of content. Oversized units get a chunk to themselves."""
|
|
27
|
+
|
|
28
|
+
text: str
|
|
29
|
+
oversized: bool = False
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class SemanticChunker(BaseChunker):
|
|
33
|
+
"""
|
|
34
|
+
Chunks text on Markdown structure, keeping chunks within a size budget.
|
|
35
|
+
|
|
36
|
+
Sections come from heading structure; within a section, blocks are packed
|
|
37
|
+
greedily up to `max_chunk_size`. Size is measured by `length_function`, so
|
|
38
|
+
callers embedding against a token budget can pass a tokenizer's encoder
|
|
39
|
+
instead of the default character count.
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
def __init__(
|
|
43
|
+
self,
|
|
44
|
+
max_chunk_size: int = 1000,
|
|
45
|
+
overlap: int = 0,
|
|
46
|
+
length_function: Callable[[str], int] = len,
|
|
47
|
+
):
|
|
48
|
+
if overlap >= max_chunk_size:
|
|
49
|
+
raise ValueError("overlap must be smaller than max_chunk_size")
|
|
50
|
+
self.max_chunk_size = max_chunk_size
|
|
51
|
+
self.overlap = overlap
|
|
52
|
+
self.length_function = length_function
|
|
53
|
+
|
|
54
|
+
def chunk(self, document: Document, **kwargs) -> list[Chunk]:
|
|
55
|
+
sections = self._sections(scan_blocks(document.content))
|
|
56
|
+
|
|
57
|
+
chunks: list[Chunk] = []
|
|
58
|
+
for section in sections:
|
|
59
|
+
units = self._units_for(section.blocks)
|
|
60
|
+
for text, oversized in self._pack(units):
|
|
61
|
+
metadata = document.metadata.copy()
|
|
62
|
+
metadata.update(
|
|
63
|
+
{
|
|
64
|
+
"header": section.header,
|
|
65
|
+
"header_level": section.header_level,
|
|
66
|
+
"header_path": section.header_path,
|
|
67
|
+
"page_numbers": sorted(section.page_numbers) or [1],
|
|
68
|
+
"chunk_index": len(chunks),
|
|
69
|
+
}
|
|
70
|
+
)
|
|
71
|
+
if oversized:
|
|
72
|
+
metadata["oversized"] = True
|
|
73
|
+
chunks.append(Chunk(content=text, metadata=metadata))
|
|
74
|
+
|
|
75
|
+
return chunks
|
|
76
|
+
|
|
77
|
+
# -- sectioning ---------------------------------------------------------
|
|
78
|
+
|
|
79
|
+
def _sections(self, blocks: list[Block]) -> list[_Section]:
|
|
80
|
+
sections: list[_Section] = []
|
|
81
|
+
header_stack: list[tuple[int, str]] = []
|
|
82
|
+
current_page = 1
|
|
83
|
+
current = _Section(page_numbers={current_page})
|
|
84
|
+
|
|
85
|
+
for block in blocks:
|
|
86
|
+
if block.kind == PAGE_MARKER:
|
|
87
|
+
current_page = block.page_number
|
|
88
|
+
current.page_numbers.add(current_page)
|
|
89
|
+
continue
|
|
90
|
+
|
|
91
|
+
if block.kind == HEADING:
|
|
92
|
+
if current.blocks:
|
|
93
|
+
sections.append(current)
|
|
94
|
+
while header_stack and header_stack[-1][0] >= block.level:
|
|
95
|
+
header_stack.pop()
|
|
96
|
+
header_stack.append((block.level, block.title))
|
|
97
|
+
current = _Section(
|
|
98
|
+
header=block.title,
|
|
99
|
+
header_level=block.level,
|
|
100
|
+
header_path=" > ".join(title for _, title in header_stack),
|
|
101
|
+
blocks=[block],
|
|
102
|
+
page_numbers={current_page},
|
|
103
|
+
)
|
|
104
|
+
continue
|
|
105
|
+
|
|
106
|
+
current.blocks.append(block)
|
|
107
|
+
|
|
108
|
+
if current.blocks:
|
|
109
|
+
sections.append(current)
|
|
110
|
+
return sections
|
|
111
|
+
|
|
112
|
+
# -- unit expansion -----------------------------------------------------
|
|
113
|
+
|
|
114
|
+
def _units_for(self, blocks: list[Block]) -> list[_Unit]:
|
|
115
|
+
units: list[_Unit] = []
|
|
116
|
+
for block in blocks:
|
|
117
|
+
if self._fits(block.text):
|
|
118
|
+
units.append(_Unit(block.text))
|
|
119
|
+
elif block.kind == TABLE:
|
|
120
|
+
units.extend(_Unit(part) for part in self._split_table(block))
|
|
121
|
+
elif block.kind == CODE:
|
|
122
|
+
# Splitting a fenced block would leave unterminated fences, so it
|
|
123
|
+
# travels whole and is flagged for the caller to notice.
|
|
124
|
+
units.append(_Unit(block.text, oversized=True))
|
|
125
|
+
else:
|
|
126
|
+
units.extend(_Unit(part) for part in self._split_prose(block.text))
|
|
127
|
+
return units
|
|
128
|
+
|
|
129
|
+
def _split_table(self, block: Block) -> list[str]:
|
|
130
|
+
"""Split by row, repeating the header so each part reads on its own."""
|
|
131
|
+
if not block.rows:
|
|
132
|
+
return [block.text]
|
|
133
|
+
|
|
134
|
+
header, *data_rows = block.rows
|
|
135
|
+
header_markdown = [_row_markdown(header), _separator_markdown(len(header))]
|
|
136
|
+
|
|
137
|
+
parts: list[str] = []
|
|
138
|
+
pending: list[str] = []
|
|
139
|
+
|
|
140
|
+
for row in data_rows:
|
|
141
|
+
candidate = pending + [_row_markdown(row)]
|
|
142
|
+
# Measure the rendered candidate rather than summing row sizes: a
|
|
143
|
+
# token counter does not charge exactly one unit per newline.
|
|
144
|
+
if pending and not self._fits("\n".join(header_markdown + candidate)):
|
|
145
|
+
parts.append("\n".join(header_markdown + pending))
|
|
146
|
+
pending = [_row_markdown(row)]
|
|
147
|
+
else:
|
|
148
|
+
pending = candidate
|
|
149
|
+
|
|
150
|
+
if pending:
|
|
151
|
+
parts.append("\n".join(header_markdown + pending))
|
|
152
|
+
return parts or [block.text]
|
|
153
|
+
|
|
154
|
+
def _split_prose(self, text: str) -> list[str]:
|
|
155
|
+
parts: list[str] = []
|
|
156
|
+
pending = ""
|
|
157
|
+
for sentence in SENTENCE_END_RE.split(text):
|
|
158
|
+
if not sentence:
|
|
159
|
+
continue
|
|
160
|
+
candidate = f"{pending} {sentence}".strip() if pending else sentence
|
|
161
|
+
if pending and not self._fits(candidate):
|
|
162
|
+
parts.append(pending)
|
|
163
|
+
pending = sentence
|
|
164
|
+
else:
|
|
165
|
+
pending = candidate
|
|
166
|
+
|
|
167
|
+
while not self._fits(pending):
|
|
168
|
+
head, pending = self._cut_to_fit(pending)
|
|
169
|
+
parts.append(head)
|
|
170
|
+
|
|
171
|
+
if pending:
|
|
172
|
+
parts.append(pending)
|
|
173
|
+
return parts
|
|
174
|
+
|
|
175
|
+
def _cut_to_fit(self, text: str) -> tuple[str, str]:
|
|
176
|
+
"""Hard-split a run that no sentence boundary can bring under budget."""
|
|
177
|
+
low, high, best = 1, len(text), 1
|
|
178
|
+
while low <= high:
|
|
179
|
+
middle = (low + high) // 2
|
|
180
|
+
if self.length_function(text[:middle]) <= self.max_chunk_size:
|
|
181
|
+
best = middle
|
|
182
|
+
low = middle + 1
|
|
183
|
+
else:
|
|
184
|
+
high = middle - 1
|
|
185
|
+
return text[:best], text[best:]
|
|
186
|
+
|
|
187
|
+
# -- packing ------------------------------------------------------------
|
|
188
|
+
|
|
189
|
+
def _pack(self, units: list[_Unit]) -> list[tuple[str, bool]]:
|
|
190
|
+
packed: list[tuple[str, bool]] = []
|
|
191
|
+
pending: list[str] = []
|
|
192
|
+
|
|
193
|
+
def flush():
|
|
194
|
+
if pending:
|
|
195
|
+
packed.append((JOIN.join(pending), False))
|
|
196
|
+
pending.clear()
|
|
197
|
+
|
|
198
|
+
for unit in units:
|
|
199
|
+
if unit.oversized:
|
|
200
|
+
flush()
|
|
201
|
+
packed.append((unit.text, True))
|
|
202
|
+
continue
|
|
203
|
+
|
|
204
|
+
if pending and not self._fits(JOIN.join(pending + [unit.text])):
|
|
205
|
+
flush()
|
|
206
|
+
pending.append(unit.text)
|
|
207
|
+
|
|
208
|
+
flush()
|
|
209
|
+
return self._apply_overlap(packed)
|
|
210
|
+
|
|
211
|
+
def _apply_overlap(self, packed: list[tuple[str, bool]]) -> list[tuple[str, bool]]:
|
|
212
|
+
if self.overlap <= 0 or len(packed) < 2:
|
|
213
|
+
return packed
|
|
214
|
+
|
|
215
|
+
result = [packed[0]]
|
|
216
|
+
for index in range(1, len(packed)):
|
|
217
|
+
text, oversized = packed[index]
|
|
218
|
+
tail = self._tail(packed[index - 1][0])
|
|
219
|
+
result.append((f"{tail}{JOIN}{text}" if tail else text, oversized))
|
|
220
|
+
return result
|
|
221
|
+
|
|
222
|
+
def _tail(self, text: str) -> str:
|
|
223
|
+
"""Longest whitespace-aligned suffix that fits the overlap budget."""
|
|
224
|
+
words = text.split()
|
|
225
|
+
tail = ""
|
|
226
|
+
for count in range(1, len(words) + 1):
|
|
227
|
+
candidate = " ".join(words[-count:])
|
|
228
|
+
if self.length_function(candidate) > self.overlap:
|
|
229
|
+
break
|
|
230
|
+
tail = candidate
|
|
231
|
+
return tail
|
|
232
|
+
|
|
233
|
+
def _fits(self, text: str) -> bool:
|
|
234
|
+
return self.length_function(text) <= self.max_chunk_size
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
def _row_markdown(row: list[str]) -> str:
|
|
238
|
+
return f"| {' | '.join(row)} |"
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def _separator_markdown(width: int) -> str:
|
|
242
|
+
return f"| {' | '.join(['---'] * width)} |"
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
"""Deterministic text windows with explicit measurement units."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from collections.abc import Callable, Sequence
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from langparse.core.chunker import BaseChunker
|
|
10
|
+
from langparse.types import Chunk, Document
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def _validate_budget(size: int, overlap: int) -> None:
|
|
14
|
+
if not isinstance(size, int) or isinstance(size, bool) or size <= 0:
|
|
15
|
+
raise ValueError("max_chunk_size must be a positive integer")
|
|
16
|
+
if not isinstance(overlap, int) or isinstance(overlap, bool) or not 0 <= overlap < size:
|
|
17
|
+
raise ValueError("overlap must be an integer between 0 and max_chunk_size - 1")
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class FixedTokenChunker(BaseChunker):
|
|
21
|
+
"""Split encoded tokens. Default lexical units are NOT model tokens.
|
|
22
|
+
|
|
23
|
+
The default codec preserves whitespace and groups Unicode words, whitespace,
|
|
24
|
+
and individual punctuation. Supply both encoder and decoder to use a model's
|
|
25
|
+
tokenizer; the caller owns that codec's lossless decoding contract.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
def __init__(
|
|
29
|
+
self,
|
|
30
|
+
max_chunk_size: int = 1000,
|
|
31
|
+
overlap: int = 0,
|
|
32
|
+
*,
|
|
33
|
+
encoder: Callable[[str], Sequence[Any]] | None = None,
|
|
34
|
+
decoder: Callable[[Sequence[Any]], str] | None = None,
|
|
35
|
+
):
|
|
36
|
+
_validate_budget(max_chunk_size, overlap)
|
|
37
|
+
if (encoder is None) != (decoder is None):
|
|
38
|
+
raise ValueError("encoder and decoder must be supplied together")
|
|
39
|
+
self.max_chunk_size = max_chunk_size
|
|
40
|
+
self.overlap = overlap
|
|
41
|
+
self.encoder = encoder or (lambda text: re.findall(r"\s+|\w+|[^\w\s]", text))
|
|
42
|
+
self.decoder = decoder or "".join
|
|
43
|
+
self.tokenizer = "custom" if encoder is not None else "lexical"
|
|
44
|
+
|
|
45
|
+
def chunk(self, document: Document, **kwargs) -> list[Chunk]:
|
|
46
|
+
tokens = list(self.encoder(document.content))
|
|
47
|
+
chunks = []
|
|
48
|
+
for start in range(0, len(tokens), self.max_chunk_size - self.overlap):
|
|
49
|
+
end = min(start + self.max_chunk_size, len(tokens))
|
|
50
|
+
chunks.append(
|
|
51
|
+
Chunk(
|
|
52
|
+
content=self.decoder(tokens[start:end]),
|
|
53
|
+
metadata={
|
|
54
|
+
**document.metadata,
|
|
55
|
+
"chunk_strategy": "fixed-token",
|
|
56
|
+
"chunk_index": len(chunks),
|
|
57
|
+
"tokenizer": self.tokenizer,
|
|
58
|
+
"token_start": start,
|
|
59
|
+
"token_end": end,
|
|
60
|
+
"token_count": end - start,
|
|
61
|
+
},
|
|
62
|
+
)
|
|
63
|
+
)
|
|
64
|
+
if end == len(tokens):
|
|
65
|
+
break
|
|
66
|
+
return chunks
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
class SlidingWindowChunker(BaseChunker):
|
|
70
|
+
"""Character windows over the complete text, including block separators."""
|
|
71
|
+
|
|
72
|
+
def __init__(self, max_chunk_size: int = 1000, overlap: int = 0):
|
|
73
|
+
_validate_budget(max_chunk_size, overlap)
|
|
74
|
+
self.max_chunk_size = max_chunk_size
|
|
75
|
+
self.overlap = overlap
|
|
76
|
+
|
|
77
|
+
def chunk(self, document: Document, **kwargs) -> list[Chunk]:
|
|
78
|
+
chunks = []
|
|
79
|
+
for start in range(0, len(document.content), self.max_chunk_size - self.overlap):
|
|
80
|
+
end = min(start + self.max_chunk_size, len(document.content))
|
|
81
|
+
chunks.append(
|
|
82
|
+
Chunk(
|
|
83
|
+
content=document.content[start:end],
|
|
84
|
+
metadata={
|
|
85
|
+
**document.metadata,
|
|
86
|
+
"chunk_strategy": "sliding-window",
|
|
87
|
+
"chunk_index": len(chunks),
|
|
88
|
+
"size_unit": "character",
|
|
89
|
+
"char_start": start,
|
|
90
|
+
"char_end": end,
|
|
91
|
+
},
|
|
92
|
+
)
|
|
93
|
+
)
|
|
94
|
+
if end == len(document.content):
|
|
95
|
+
break
|
|
96
|
+
return chunks
|