langparse 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. langparse/__init__.py +55 -0
  2. langparse/autoparser.py +25 -0
  3. langparse/chunkers/__init__.py +12 -0
  4. langparse/chunkers/blocks.py +151 -0
  5. langparse/chunkers/profiles.py +53 -0
  6. langparse/chunkers/registry.py +38 -0
  7. langparse/chunkers/semantic.py +242 -0
  8. langparse/chunkers/text.py +96 -0
  9. langparse/chunkers/workbook.py +942 -0
  10. langparse/cli.py +329 -0
  11. langparse/config.py +169 -0
  12. langparse/core/__init__.py +0 -0
  13. langparse/core/chunker.py +16 -0
  14. langparse/core/engine.py +37 -0
  15. langparse/core/parser.py +35 -0
  16. langparse/core/rendering.py +49 -0
  17. langparse/engines/__init__.py +1 -0
  18. langparse/engines/pdf/__init__.py +1 -0
  19. langparse/engines/pdf/deepdoc/__init__.py +55 -0
  20. langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
  21. langparse/engines/pdf/deepdoc/model_loader.py +101 -0
  22. langparse/engines/pdf/deepdoc/ocr.py +641 -0
  23. langparse/engines/pdf/deepdoc/operators.py +684 -0
  24. langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
  25. langparse/engines/pdf/deepdoc/postprocess.py +339 -0
  26. langparse/engines/pdf/deepdoc/recognizer.py +418 -0
  27. langparse/engines/pdf/deepdoc/rendering.py +210 -0
  28. langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
  29. langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
  30. langparse/engines/pdf/deepdoc/utils.py +36 -0
  31. langparse/engines/pdf/deepdoc_engine.py +164 -0
  32. langparse/engines/pdf/mineru.py +259 -0
  33. langparse/engines/pdf/mineru_client.py +318 -0
  34. langparse/engines/pdf/mineru_service.py +225 -0
  35. langparse/engines/pdf/ocr.py +101 -0
  36. langparse/engines/pdf/other.py +20 -0
  37. langparse/engines/pdf/simple.py +134 -0
  38. langparse/engines/pdf/vision_llm.py +27 -0
  39. langparse/errors.py +70 -0
  40. langparse/logging.py +27 -0
  41. langparse/metrics.py +129 -0
  42. langparse/parsers/__init__.py +0 -0
  43. langparse/parsers/docx_parser.py +114 -0
  44. langparse/parsers/excel_parser.py +220 -0
  45. langparse/parsers/markdown_parser.py +34 -0
  46. langparse/parsers/pdf_parser.py +31 -0
  47. langparse/parsers/registry.py +48 -0
  48. langparse/parsers/sniff.py +72 -0
  49. langparse/progress.py +77 -0
  50. langparse/py.typed +0 -0
  51. langparse/services/__init__.py +11 -0
  52. langparse/services/batch_service.py +339 -0
  53. langparse/services/benchmark_service.py +202 -0
  54. langparse/services/fidelity.py +154 -0
  55. langparse/services/output_paths.py +86 -0
  56. langparse/services/parse_service.py +523 -0
  57. langparse/services/quality.py +65 -0
  58. langparse/services/workbook_ambiguity_benchmark.py +563 -0
  59. langparse/services/workbook_quality_benchmark.py +230 -0
  60. langparse/types.py +97 -0
  61. langparse/workbooks/__init__.py +103 -0
  62. langparse/workbooks/adapters.py +474 -0
  63. langparse/workbooks/assembly.py +993 -0
  64. langparse/workbooks/blocks.py +209 -0
  65. langparse/workbooks/bundle-v1.schema.json +71 -0
  66. langparse/workbooks/bundle.py +341 -0
  67. langparse/workbooks/classification.py +393 -0
  68. langparse/workbooks/continuation.py +577 -0
  69. langparse/workbooks/evaluation/__init__.py +45 -0
  70. langparse/workbooks/evaluation/evaluator.py +381 -0
  71. langparse/workbooks/evaluation/schema.py +419 -0
  72. langparse/workbooks/labels.py +14 -0
  73. langparse/workbooks/lineage.py +117 -0
  74. langparse/workbooks/modeling/__init__.py +52 -0
  75. langparse/workbooks/modeling/cache.py +20 -0
  76. langparse/workbooks/modeling/config.py +87 -0
  77. langparse/workbooks/modeling/contract.py +628 -0
  78. langparse/workbooks/modeling/disambiguation.py +800 -0
  79. langparse/workbooks/modeling/openai_adapter.py +192 -0
  80. langparse/workbooks/modeling/policy.py +79 -0
  81. langparse/workbooks/modeling/ports.py +44 -0
  82. langparse/workbooks/modeling/pricing.py +17 -0
  83. langparse/workbooks/modeling/types.py +251 -0
  84. langparse/workbooks/objects.py +229 -0
  85. langparse/workbooks/quality/__init__.py +23 -0
  86. langparse/workbooks/quality/bundle.py +53 -0
  87. langparse/workbooks/quality/evaluator.py +266 -0
  88. langparse/workbooks/quality/facts.py +142 -0
  89. langparse/workbooks/quality/schema.py +462 -0
  90. langparse/workbooks/reference_types.py +73 -0
  91. langparse/workbooks/references.py +178 -0
  92. langparse/workbooks/regions.py +932 -0
  93. langparse/workbooks/rendering.py +222 -0
  94. langparse/workbooks/tables.py +477 -0
  95. langparse/workbooks/types.py +257 -0
  96. langparse-0.1.0.dist-info/METADATA +790 -0
  97. langparse-0.1.0.dist-info/RECORD +101 -0
  98. langparse-0.1.0.dist-info/WHEEL +5 -0
  99. langparse-0.1.0.dist-info/entry_points.txt +2 -0
  100. langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
  101. langparse-0.1.0.dist-info/top_level.txt +1 -0
langparse/__init__.py ADDED
@@ -0,0 +1,55 @@
1
+ from importlib.metadata import version as _distribution_version
2
+
3
+ __version__ = _distribution_version("langparse")
4
+
5
+ from langparse.autoparser import AutoParser
6
+ from langparse.chunkers import (
7
+ FixedTokenChunker,
8
+ SemanticChunker,
9
+ SlidingWindowChunker,
10
+ available_chunkers,
11
+ create_chunker,
12
+ register_chunker,
13
+ )
14
+ from langparse.core.chunker import BaseChunker
15
+ from langparse.core.parser import BaseParser
16
+ from langparse.metrics import BatchItemResult, BatchRunResult, ParseMetrics
17
+ from langparse.parsers.docx_parser import DocxParser
18
+ from langparse.parsers.excel_parser import ExcelParser
19
+ from langparse.parsers.markdown_parser import MarkdownParser
20
+ from langparse.parsers.pdf_parser import PDFParser
21
+ from langparse.progress import ProgressCallback, ProgressEvent
22
+ from langparse.types import (
23
+ Chunk,
24
+ Document,
25
+ ParsedDocumentResult,
26
+ ParsedElement,
27
+ ParsedPageResult,
28
+ )
29
+
30
+ __all__ = [
31
+ "__version__",
32
+ "Document",
33
+ "Chunk",
34
+ "ParsedDocumentResult",
35
+ "ParsedPageResult",
36
+ "ParsedElement",
37
+ "BaseParser",
38
+ "BaseChunker",
39
+ "AutoParser",
40
+ "PDFParser",
41
+ "MarkdownParser",
42
+ "DocxParser",
43
+ "ExcelParser",
44
+ "SemanticChunker",
45
+ "FixedTokenChunker",
46
+ "SlidingWindowChunker",
47
+ "available_chunkers",
48
+ "create_chunker",
49
+ "register_chunker",
50
+ "ParseMetrics",
51
+ "ProgressCallback",
52
+ "ProgressEvent",
53
+ "BatchItemResult",
54
+ "BatchRunResult",
55
+ ]
@@ -0,0 +1,25 @@
1
+ from pathlib import Path
2
+
3
+ from langparse.core.rendering import document_from_result
4
+ from langparse.types import Document, ParsedDocumentResult
5
+
6
+
7
+ class AutoParser:
8
+ """
9
+ Facade that parses any supported file without the caller picking a parser.
10
+
11
+ Extension routing lives in `ParseService`, driven by
12
+ `langparse.parsers.registry`, so this stays a convenience wrapper rather
13
+ than a second place formats can be registered and drift.
14
+ """
15
+
16
+ @staticmethod
17
+ def parse_result(file_path: str | Path, **kwargs) -> ParsedDocumentResult:
18
+ from langparse.services.parse_service import ParseService
19
+
20
+ engine_name = kwargs.pop("engine", None) or "simple"
21
+ return ParseService().parse_result(file_path, engine_name=engine_name, **kwargs)
22
+
23
+ @staticmethod
24
+ def parse(file_path: str | Path, **kwargs) -> Document:
25
+ return document_from_result(AutoParser.parse_result(file_path, **kwargs))
@@ -0,0 +1,12 @@
1
+ from langparse.chunkers.registry import available_chunkers, create_chunker, register_chunker
2
+ from langparse.chunkers.semantic import SemanticChunker
3
+ from langparse.chunkers.text import FixedTokenChunker, SlidingWindowChunker
4
+
5
+ __all__ = [
6
+ "FixedTokenChunker",
7
+ "SlidingWindowChunker",
8
+ "SemanticChunker",
9
+ "available_chunkers",
10
+ "create_chunker",
11
+ "register_chunker",
12
+ ]
@@ -0,0 +1,151 @@
1
+ """
2
+ Scan Markdown into typed blocks.
3
+
4
+ Chunking used to run a heading regex over the whole document, which cannot tell
5
+ a real heading from a `#` comment inside a fenced code block -- recognising a
6
+ fence requires tracking state across lines. Scanning into typed blocks fixes
7
+ that structurally, and the same block types are what let the packer treat
8
+ tables and code differently when they overflow a chunk.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import re
14
+ from dataclasses import dataclass, field
15
+
16
+ HEADING_RE = re.compile(r"^(#{1,6})\s+(.+?)\s*$")
17
+ FENCE_RE = re.compile(r"^(\s*)(`{3,}|~{3,})(.*)$")
18
+ PAGE_MARKER_RE = re.compile(r"^\s*<!--\s*page_number:\s*(\d+)\s*-->\s*$")
19
+ TABLE_SEPARATOR_RE = re.compile(r"^\s*\|?\s*:?-{3,}:?\s*(\|\s*:?-{3,}:?\s*)*\|?\s*$")
20
+
21
+ HEADING = "heading"
22
+ TABLE = "table"
23
+ CODE = "code"
24
+ PARAGRAPH = "paragraph"
25
+ PAGE_MARKER = "page_marker"
26
+
27
+
28
+ @dataclass
29
+ class Block:
30
+ kind: str
31
+ text: str
32
+ #: heading only
33
+ level: int = 0
34
+ title: str = ""
35
+ #: table only
36
+ rows: list[list[str]] = field(default_factory=list)
37
+ has_header: bool = False
38
+ #: page_marker only
39
+ page_number: int = 0
40
+
41
+
42
+ def scan_blocks(markdown: str) -> list[Block]:
43
+ """Split Markdown into an ordered list of typed blocks."""
44
+ if not markdown:
45
+ return []
46
+
47
+ lines = markdown.splitlines()
48
+ blocks: list[Block] = []
49
+ index = 0
50
+
51
+ while index < len(lines):
52
+ line = lines[index]
53
+
54
+ if not line.strip():
55
+ index += 1
56
+ continue
57
+
58
+ fence = FENCE_RE.match(line)
59
+ if fence:
60
+ index = _consume_fence(lines, index, fence.group(2), blocks)
61
+ continue
62
+
63
+ page_marker = PAGE_MARKER_RE.match(line)
64
+ if page_marker:
65
+ blocks.append(Block(kind=PAGE_MARKER, text=line, page_number=int(page_marker.group(1))))
66
+ index += 1
67
+ continue
68
+
69
+ heading = HEADING_RE.match(line)
70
+ if heading:
71
+ blocks.append(
72
+ Block(
73
+ kind=HEADING,
74
+ text=line,
75
+ level=len(heading.group(1)),
76
+ title=heading.group(2),
77
+ )
78
+ )
79
+ index += 1
80
+ continue
81
+
82
+ if _starts_table(lines, index):
83
+ index = _consume_table(lines, index, blocks)
84
+ continue
85
+
86
+ index = _consume_paragraph(lines, index, blocks)
87
+
88
+ return blocks
89
+
90
+
91
+ def _consume_fence(lines: list[str], index: int, marker: str, blocks: list[Block]) -> int:
92
+ """Consume a fenced block. An unclosed fence runs to end of input."""
93
+ fence_char = marker[0]
94
+ collected = [lines[index]]
95
+ index += 1
96
+
97
+ while index < len(lines):
98
+ collected.append(lines[index])
99
+ closing = FENCE_RE.match(lines[index])
100
+ index += 1
101
+ if closing and closing.group(2)[0] == fence_char and len(closing.group(2)) >= len(marker):
102
+ break
103
+
104
+ blocks.append(Block(kind=CODE, text="\n".join(collected)))
105
+ return index
106
+
107
+
108
+ def _starts_table(lines: list[str], index: int) -> bool:
109
+ """A table needs a pipe row followed by a separator row; pipes alone are prose."""
110
+ if not lines[index].lstrip().startswith("|"):
111
+ return False
112
+ return index + 1 < len(lines) and bool(TABLE_SEPARATOR_RE.match(lines[index + 1]))
113
+
114
+
115
+ def _consume_table(lines: list[str], index: int, blocks: list[Block]) -> int:
116
+ collected: list[str] = []
117
+ while index < len(lines) and lines[index].lstrip().startswith("|"):
118
+ collected.append(lines[index])
119
+ index += 1
120
+
121
+ rows = [_split_table_row(line) for line in collected if not TABLE_SEPARATOR_RE.match(line)]
122
+ blocks.append(Block(kind=TABLE, text="\n".join(collected), rows=rows, has_header=bool(rows)))
123
+ return index
124
+
125
+
126
+ def _split_table_row(line: str) -> list[str]:
127
+ stripped = line.strip()
128
+ if stripped.startswith("|"):
129
+ stripped = stripped[1:]
130
+ if stripped.endswith("|"):
131
+ stripped = stripped[:-1]
132
+ return [cell.strip() for cell in stripped.split("|")]
133
+
134
+
135
+ def _consume_paragraph(lines: list[str], index: int, blocks: list[Block]) -> int:
136
+ collected: list[str] = []
137
+ while index < len(lines) and lines[index].strip():
138
+ line = lines[index]
139
+ if (
140
+ FENCE_RE.match(line)
141
+ or HEADING_RE.match(line)
142
+ or PAGE_MARKER_RE.match(line)
143
+ or _starts_table(lines, index)
144
+ ):
145
+ break
146
+ collected.append(line)
147
+ index += 1
148
+
149
+ if collected:
150
+ blocks.append(Block(kind=PARAGRAPH, text="\n".join(collected)))
151
+ return index
@@ -0,0 +1,53 @@
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import dataclass
4
+ from enum import Enum
5
+
6
+
7
+ class WorkbookChunkProfile(str, Enum):
8
+ RETRIEVAL = "retrieval"
9
+ ANALYSIS = "analysis"
10
+
11
+
12
+ @dataclass(frozen=True)
13
+ class WorkbookChunkPolicy:
14
+ name: WorkbookChunkProfile
15
+ version: int
16
+ default_max_chunk_size: int
17
+ analysis_records: bool
18
+
19
+
20
+ class ChunkProfileNotSupportedError(ValueError):
21
+ """Raised when a valid chunk profile cannot represent the parsed input."""
22
+
23
+
24
+ _POLICIES = {
25
+ WorkbookChunkProfile.RETRIEVAL: WorkbookChunkPolicy(
26
+ name=WorkbookChunkProfile.RETRIEVAL,
27
+ version=1,
28
+ default_max_chunk_size=1000,
29
+ analysis_records=False,
30
+ ),
31
+ WorkbookChunkProfile.ANALYSIS: WorkbookChunkPolicy(
32
+ name=WorkbookChunkProfile.ANALYSIS,
33
+ version=1,
34
+ default_max_chunk_size=4000,
35
+ analysis_records=True,
36
+ ),
37
+ }
38
+
39
+
40
+ def resolve_workbook_chunk_policy(
41
+ profile: str | WorkbookChunkProfile | None,
42
+ ) -> WorkbookChunkPolicy:
43
+ if profile is None:
44
+ selected = WorkbookChunkProfile.RETRIEVAL
45
+ else:
46
+ try:
47
+ selected = WorkbookChunkProfile(profile)
48
+ except ValueError:
49
+ available = ", ".join(sorted(item.value for item in WorkbookChunkProfile))
50
+ raise ValueError(
51
+ f"Unknown workbook chunk profile {profile!r}. Available: {available}"
52
+ ) from None
53
+ return _POLICIES[selected]
@@ -0,0 +1,38 @@
1
+ """Explicit registry for document-text chunkers; workbook routing stays structural."""
2
+
3
+ from collections.abc import Callable
4
+
5
+ from langparse.chunkers.semantic import SemanticChunker
6
+ from langparse.chunkers.text import FixedTokenChunker, SlidingWindowChunker, _validate_budget
7
+ from langparse.core.chunker import BaseChunker
8
+
9
+ _FACTORIES: dict[str, Callable[..., BaseChunker]] = {
10
+ "semantic": SemanticChunker,
11
+ "fixed-token": FixedTokenChunker,
12
+ "sliding-window": SlidingWindowChunker,
13
+ }
14
+
15
+
16
+ def register_chunker(name: str, factory: Callable[..., BaseChunker]) -> None:
17
+ if not isinstance(name, str) or not name.strip() or not callable(factory):
18
+ raise ValueError("A non-empty name and callable chunker factory are required")
19
+ if name in _FACTORIES:
20
+ raise ValueError(f"Chunk strategy {name!r} is already registered")
21
+ _FACTORIES[name] = factory
22
+
23
+
24
+ def available_chunkers() -> tuple[str, ...]:
25
+ return tuple(sorted(_FACTORIES))
26
+
27
+
28
+ def create_chunker(name: str = "semantic", **options) -> BaseChunker:
29
+ if name not in _FACTORIES:
30
+ raise ValueError(
31
+ f"Unknown chunk strategy {name!r}. Available: {', '.join(available_chunkers())}"
32
+ )
33
+ if name == "semantic":
34
+ _validate_budget(options.get("max_chunk_size", 1000), options.get("overlap", 0))
35
+ result = _FACTORIES[name](**options)
36
+ if not isinstance(result, BaseChunker):
37
+ raise TypeError("Chunker factory must return a BaseChunker")
38
+ return result
@@ -0,0 +1,242 @@
1
+ from __future__ import annotations
2
+
3
+ import re
4
+ from collections.abc import Callable
5
+ from dataclasses import dataclass, field
6
+
7
+ from langparse.chunkers.blocks import CODE, HEADING, PAGE_MARKER, TABLE, Block, scan_blocks
8
+ from langparse.core.chunker import BaseChunker
9
+ from langparse.types import Chunk, Document
10
+
11
+ SENTENCE_END_RE = re.compile(r"(?<=[.!?。!?])\s+")
12
+ JOIN = "\n\n"
13
+
14
+
15
+ @dataclass
16
+ class _Section:
17
+ header: str | None = None
18
+ header_level: int = 0
19
+ header_path: str = ""
20
+ blocks: list[Block] = field(default_factory=list)
21
+ page_numbers: set = field(default_factory=set)
22
+
23
+
24
+ @dataclass
25
+ class _Unit:
26
+ """One packable piece of content. Oversized units get a chunk to themselves."""
27
+
28
+ text: str
29
+ oversized: bool = False
30
+
31
+
32
+ class SemanticChunker(BaseChunker):
33
+ """
34
+ Chunks text on Markdown structure, keeping chunks within a size budget.
35
+
36
+ Sections come from heading structure; within a section, blocks are packed
37
+ greedily up to `max_chunk_size`. Size is measured by `length_function`, so
38
+ callers embedding against a token budget can pass a tokenizer's encoder
39
+ instead of the default character count.
40
+ """
41
+
42
+ def __init__(
43
+ self,
44
+ max_chunk_size: int = 1000,
45
+ overlap: int = 0,
46
+ length_function: Callable[[str], int] = len,
47
+ ):
48
+ if overlap >= max_chunk_size:
49
+ raise ValueError("overlap must be smaller than max_chunk_size")
50
+ self.max_chunk_size = max_chunk_size
51
+ self.overlap = overlap
52
+ self.length_function = length_function
53
+
54
+ def chunk(self, document: Document, **kwargs) -> list[Chunk]:
55
+ sections = self._sections(scan_blocks(document.content))
56
+
57
+ chunks: list[Chunk] = []
58
+ for section in sections:
59
+ units = self._units_for(section.blocks)
60
+ for text, oversized in self._pack(units):
61
+ metadata = document.metadata.copy()
62
+ metadata.update(
63
+ {
64
+ "header": section.header,
65
+ "header_level": section.header_level,
66
+ "header_path": section.header_path,
67
+ "page_numbers": sorted(section.page_numbers) or [1],
68
+ "chunk_index": len(chunks),
69
+ }
70
+ )
71
+ if oversized:
72
+ metadata["oversized"] = True
73
+ chunks.append(Chunk(content=text, metadata=metadata))
74
+
75
+ return chunks
76
+
77
+ # -- sectioning ---------------------------------------------------------
78
+
79
+ def _sections(self, blocks: list[Block]) -> list[_Section]:
80
+ sections: list[_Section] = []
81
+ header_stack: list[tuple[int, str]] = []
82
+ current_page = 1
83
+ current = _Section(page_numbers={current_page})
84
+
85
+ for block in blocks:
86
+ if block.kind == PAGE_MARKER:
87
+ current_page = block.page_number
88
+ current.page_numbers.add(current_page)
89
+ continue
90
+
91
+ if block.kind == HEADING:
92
+ if current.blocks:
93
+ sections.append(current)
94
+ while header_stack and header_stack[-1][0] >= block.level:
95
+ header_stack.pop()
96
+ header_stack.append((block.level, block.title))
97
+ current = _Section(
98
+ header=block.title,
99
+ header_level=block.level,
100
+ header_path=" > ".join(title for _, title in header_stack),
101
+ blocks=[block],
102
+ page_numbers={current_page},
103
+ )
104
+ continue
105
+
106
+ current.blocks.append(block)
107
+
108
+ if current.blocks:
109
+ sections.append(current)
110
+ return sections
111
+
112
+ # -- unit expansion -----------------------------------------------------
113
+
114
+ def _units_for(self, blocks: list[Block]) -> list[_Unit]:
115
+ units: list[_Unit] = []
116
+ for block in blocks:
117
+ if self._fits(block.text):
118
+ units.append(_Unit(block.text))
119
+ elif block.kind == TABLE:
120
+ units.extend(_Unit(part) for part in self._split_table(block))
121
+ elif block.kind == CODE:
122
+ # Splitting a fenced block would leave unterminated fences, so it
123
+ # travels whole and is flagged for the caller to notice.
124
+ units.append(_Unit(block.text, oversized=True))
125
+ else:
126
+ units.extend(_Unit(part) for part in self._split_prose(block.text))
127
+ return units
128
+
129
+ def _split_table(self, block: Block) -> list[str]:
130
+ """Split by row, repeating the header so each part reads on its own."""
131
+ if not block.rows:
132
+ return [block.text]
133
+
134
+ header, *data_rows = block.rows
135
+ header_markdown = [_row_markdown(header), _separator_markdown(len(header))]
136
+
137
+ parts: list[str] = []
138
+ pending: list[str] = []
139
+
140
+ for row in data_rows:
141
+ candidate = pending + [_row_markdown(row)]
142
+ # Measure the rendered candidate rather than summing row sizes: a
143
+ # token counter does not charge exactly one unit per newline.
144
+ if pending and not self._fits("\n".join(header_markdown + candidate)):
145
+ parts.append("\n".join(header_markdown + pending))
146
+ pending = [_row_markdown(row)]
147
+ else:
148
+ pending = candidate
149
+
150
+ if pending:
151
+ parts.append("\n".join(header_markdown + pending))
152
+ return parts or [block.text]
153
+
154
+ def _split_prose(self, text: str) -> list[str]:
155
+ parts: list[str] = []
156
+ pending = ""
157
+ for sentence in SENTENCE_END_RE.split(text):
158
+ if not sentence:
159
+ continue
160
+ candidate = f"{pending} {sentence}".strip() if pending else sentence
161
+ if pending and not self._fits(candidate):
162
+ parts.append(pending)
163
+ pending = sentence
164
+ else:
165
+ pending = candidate
166
+
167
+ while not self._fits(pending):
168
+ head, pending = self._cut_to_fit(pending)
169
+ parts.append(head)
170
+
171
+ if pending:
172
+ parts.append(pending)
173
+ return parts
174
+
175
+ def _cut_to_fit(self, text: str) -> tuple[str, str]:
176
+ """Hard-split a run that no sentence boundary can bring under budget."""
177
+ low, high, best = 1, len(text), 1
178
+ while low <= high:
179
+ middle = (low + high) // 2
180
+ if self.length_function(text[:middle]) <= self.max_chunk_size:
181
+ best = middle
182
+ low = middle + 1
183
+ else:
184
+ high = middle - 1
185
+ return text[:best], text[best:]
186
+
187
+ # -- packing ------------------------------------------------------------
188
+
189
+ def _pack(self, units: list[_Unit]) -> list[tuple[str, bool]]:
190
+ packed: list[tuple[str, bool]] = []
191
+ pending: list[str] = []
192
+
193
+ def flush():
194
+ if pending:
195
+ packed.append((JOIN.join(pending), False))
196
+ pending.clear()
197
+
198
+ for unit in units:
199
+ if unit.oversized:
200
+ flush()
201
+ packed.append((unit.text, True))
202
+ continue
203
+
204
+ if pending and not self._fits(JOIN.join(pending + [unit.text])):
205
+ flush()
206
+ pending.append(unit.text)
207
+
208
+ flush()
209
+ return self._apply_overlap(packed)
210
+
211
+ def _apply_overlap(self, packed: list[tuple[str, bool]]) -> list[tuple[str, bool]]:
212
+ if self.overlap <= 0 or len(packed) < 2:
213
+ return packed
214
+
215
+ result = [packed[0]]
216
+ for index in range(1, len(packed)):
217
+ text, oversized = packed[index]
218
+ tail = self._tail(packed[index - 1][0])
219
+ result.append((f"{tail}{JOIN}{text}" if tail else text, oversized))
220
+ return result
221
+
222
+ def _tail(self, text: str) -> str:
223
+ """Longest whitespace-aligned suffix that fits the overlap budget."""
224
+ words = text.split()
225
+ tail = ""
226
+ for count in range(1, len(words) + 1):
227
+ candidate = " ".join(words[-count:])
228
+ if self.length_function(candidate) > self.overlap:
229
+ break
230
+ tail = candidate
231
+ return tail
232
+
233
+ def _fits(self, text: str) -> bool:
234
+ return self.length_function(text) <= self.max_chunk_size
235
+
236
+
237
+ def _row_markdown(row: list[str]) -> str:
238
+ return f"| {' | '.join(row)} |"
239
+
240
+
241
+ def _separator_markdown(width: int) -> str:
242
+ return f"| {' | '.join(['---'] * width)} |"
@@ -0,0 +1,96 @@
1
+ """Deterministic text windows with explicit measurement units."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from collections.abc import Callable, Sequence
7
+ from typing import Any
8
+
9
+ from langparse.core.chunker import BaseChunker
10
+ from langparse.types import Chunk, Document
11
+
12
+
13
+ def _validate_budget(size: int, overlap: int) -> None:
14
+ if not isinstance(size, int) or isinstance(size, bool) or size <= 0:
15
+ raise ValueError("max_chunk_size must be a positive integer")
16
+ if not isinstance(overlap, int) or isinstance(overlap, bool) or not 0 <= overlap < size:
17
+ raise ValueError("overlap must be an integer between 0 and max_chunk_size - 1")
18
+
19
+
20
+ class FixedTokenChunker(BaseChunker):
21
+ """Split encoded tokens. Default lexical units are NOT model tokens.
22
+
23
+ The default codec preserves whitespace and groups Unicode words, whitespace,
24
+ and individual punctuation. Supply both encoder and decoder to use a model's
25
+ tokenizer; the caller owns that codec's lossless decoding contract.
26
+ """
27
+
28
+ def __init__(
29
+ self,
30
+ max_chunk_size: int = 1000,
31
+ overlap: int = 0,
32
+ *,
33
+ encoder: Callable[[str], Sequence[Any]] | None = None,
34
+ decoder: Callable[[Sequence[Any]], str] | None = None,
35
+ ):
36
+ _validate_budget(max_chunk_size, overlap)
37
+ if (encoder is None) != (decoder is None):
38
+ raise ValueError("encoder and decoder must be supplied together")
39
+ self.max_chunk_size = max_chunk_size
40
+ self.overlap = overlap
41
+ self.encoder = encoder or (lambda text: re.findall(r"\s+|\w+|[^\w\s]", text))
42
+ self.decoder = decoder or "".join
43
+ self.tokenizer = "custom" if encoder is not None else "lexical"
44
+
45
+ def chunk(self, document: Document, **kwargs) -> list[Chunk]:
46
+ tokens = list(self.encoder(document.content))
47
+ chunks = []
48
+ for start in range(0, len(tokens), self.max_chunk_size - self.overlap):
49
+ end = min(start + self.max_chunk_size, len(tokens))
50
+ chunks.append(
51
+ Chunk(
52
+ content=self.decoder(tokens[start:end]),
53
+ metadata={
54
+ **document.metadata,
55
+ "chunk_strategy": "fixed-token",
56
+ "chunk_index": len(chunks),
57
+ "tokenizer": self.tokenizer,
58
+ "token_start": start,
59
+ "token_end": end,
60
+ "token_count": end - start,
61
+ },
62
+ )
63
+ )
64
+ if end == len(tokens):
65
+ break
66
+ return chunks
67
+
68
+
69
+ class SlidingWindowChunker(BaseChunker):
70
+ """Character windows over the complete text, including block separators."""
71
+
72
+ def __init__(self, max_chunk_size: int = 1000, overlap: int = 0):
73
+ _validate_budget(max_chunk_size, overlap)
74
+ self.max_chunk_size = max_chunk_size
75
+ self.overlap = overlap
76
+
77
+ def chunk(self, document: Document, **kwargs) -> list[Chunk]:
78
+ chunks = []
79
+ for start in range(0, len(document.content), self.max_chunk_size - self.overlap):
80
+ end = min(start + self.max_chunk_size, len(document.content))
81
+ chunks.append(
82
+ Chunk(
83
+ content=document.content[start:end],
84
+ metadata={
85
+ **document.metadata,
86
+ "chunk_strategy": "sliding-window",
87
+ "chunk_index": len(chunks),
88
+ "size_unit": "character",
89
+ "char_start": start,
90
+ "char_end": end,
91
+ },
92
+ )
93
+ )
94
+ if end == len(document.content):
95
+ break
96
+ return chunks