lumberjack-py 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lumberjack/__init__.py +4 -0
- lumberjack/_internal/__init__.py +1 -0
- lumberjack/_internal/block_splitter.py +504 -0
- lumberjack/_internal/formats.py +55 -0
- lumberjack/_internal/options.py +112 -0
- lumberjack/_internal/pipeline.py +218 -0
- lumberjack/_internal/rendering.py +12 -0
- lumberjack/block.py +192 -0
- lumberjack/cli.py +142 -0
- lumberjack/finalizer.py +103 -0
- lumberjack/lumberjack.py +66 -0
- lumberjack/models.py +316 -0
- lumberjack/normalizer.py +16 -0
- lumberjack/parser/__init__.py +24 -0
- lumberjack/parser/auto.py +139 -0
- lumberjack/parser/docx/__init__.py +3 -0
- lumberjack/parser/docx/parser.py +422 -0
- lumberjack/parser/html/__init__.py +3 -0
- lumberjack/parser/html/parser.py +479 -0
- lumberjack/parser/html/table_parser.py +361 -0
- lumberjack/parser/markdown/__init__.py +15 -0
- lumberjack/parser/markdown/parser.py +1037 -0
- lumberjack/parser/markdown/plugins/__init__.py +3 -0
- lumberjack/parser/markdown/plugins/brackets_plugin.py +93 -0
- lumberjack/protocols.py +43 -0
- lumberjack/py.typed +0 -0
- lumberjack/splitter/__init__.py +32 -0
- lumberjack/splitter/base.py +336 -0
- lumberjack/splitter/context.py +118 -0
- lumberjack/splitter/exact.py +365 -0
- lumberjack/splitter/incremental.py +518 -0
- lumberjack/splitter/section.py +54 -0
- lumberjack/splitter/sibling.py +21 -0
- lumberjack/splitter/subtree.py +22 -0
- lumberjack/splitter/topology/__init__.py +1 -0
- lumberjack/splitter/topology/section.py +52 -0
- lumberjack/splitter/topology/sibling.py +114 -0
- lumberjack/splitter/topology/subtree.py +64 -0
- lumberjack/tokenizer.py +194 -0
- lumberjack/transformer.py +103 -0
- lumberjack/web/__init__.py +5 -0
- lumberjack/web/__main__.py +22 -0
- lumberjack/web/app.py +43 -0
- lumberjack/web/routes.py +205 -0
- lumberjack_py-0.4.0.dist-info/METADATA +197 -0
- lumberjack_py-0.4.0.dist-info/RECORD +49 -0
- lumberjack_py-0.4.0.dist-info/WHEEL +4 -0
- lumberjack_py-0.4.0.dist-info/entry_points.txt +3 -0
- lumberjack_py-0.4.0.dist-info/licenses/LICENSE +21 -0
lumberjack/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Private implementation helpers for Lumberjack's public components."""
|
|
@@ -0,0 +1,504 @@
|
|
|
1
|
+
"""Internal oversized-block splitting helpers."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from typing import TYPE_CHECKING
|
|
7
|
+
|
|
8
|
+
from lumberjack.block import (
|
|
9
|
+
BlockOption,
|
|
10
|
+
HTMLTableConfig,
|
|
11
|
+
MarkdownTableConfig,
|
|
12
|
+
default_block_config,
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
from ..parser.html.table_parser import HTMLTableParser, HTMLTableRow
|
|
16
|
+
|
|
17
|
+
if TYPE_CHECKING:
|
|
18
|
+
from ..models import DocumentBlock
|
|
19
|
+
from ..protocols import TokenizerProtocol
|
|
20
|
+
|
|
21
|
+
SENTENCE_BREAK_RE = re.compile(r"(?<=[.!?\u3002\uff01\uff1f])\s+")
|
|
22
|
+
PROTECTED_SPAN_RE = re.compile(r"<https?://[^\s>]+>|https?://[^\s)>\]]+")
|
|
23
|
+
TABLE_DELIMITER_CELL_RE = re.compile(r":?-+(:?-+)*:?")
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class BlockSplitter:
|
|
27
|
+
"""Splits oversized text blocks into token-bounded pieces."""
|
|
28
|
+
|
|
29
|
+
def __init__(
|
|
30
|
+
self,
|
|
31
|
+
tokenizer: TokenizerProtocol,
|
|
32
|
+
*,
|
|
33
|
+
max_tokens: int,
|
|
34
|
+
block_options: dict[str, BlockOption],
|
|
35
|
+
) -> None:
|
|
36
|
+
self.tokenizer = tokenizer
|
|
37
|
+
self.max_tokens = max_tokens
|
|
38
|
+
self.block_options = block_options
|
|
39
|
+
self._html_table_parser = HTMLTableParser()
|
|
40
|
+
|
|
41
|
+
def split_oversized_block(
|
|
42
|
+
self,
|
|
43
|
+
block: DocumentBlock,
|
|
44
|
+
*,
|
|
45
|
+
default_budget: int,
|
|
46
|
+
) -> list[tuple[str, int]] | None:
|
|
47
|
+
config = self._block_config(block.kind)
|
|
48
|
+
if config is None or not config.split:
|
|
49
|
+
return None
|
|
50
|
+
|
|
51
|
+
budget = self._block_budget(block.kind, default_budget)
|
|
52
|
+
|
|
53
|
+
if block.kind in {"code_block", "code_fence"}:
|
|
54
|
+
return self.split_code_block(block, max_tokens=budget)
|
|
55
|
+
|
|
56
|
+
if block.kind == "list" and block.children:
|
|
57
|
+
return self.split_list_block(block, max_tokens=budget)
|
|
58
|
+
|
|
59
|
+
if block.kind == "table":
|
|
60
|
+
return self.split_table_block(block, default_budget=default_budget)
|
|
61
|
+
|
|
62
|
+
if block.kind == "html_table":
|
|
63
|
+
return self.split_html_table_block(
|
|
64
|
+
block,
|
|
65
|
+
default_budget=default_budget,
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
return self.split_text(block.text, max_tokens=budget)
|
|
69
|
+
|
|
70
|
+
def _block_config(self, kind: str) -> BlockOption:
|
|
71
|
+
return self.block_options.get(kind.lower(), default_block_config(kind))
|
|
72
|
+
|
|
73
|
+
def _block_budget(self, kind: str, default_budget: int | None = None) -> int:
|
|
74
|
+
config = self._block_config(kind)
|
|
75
|
+
if config and config.max_tokens:
|
|
76
|
+
return config.max_tokens
|
|
77
|
+
if default_budget is not None:
|
|
78
|
+
return default_budget
|
|
79
|
+
return self.max_tokens
|
|
80
|
+
|
|
81
|
+
def _repeat_header(self, kind: str) -> bool:
|
|
82
|
+
config = self._block_config(kind)
|
|
83
|
+
return (
|
|
84
|
+
not isinstance(config, MarkdownTableConfig | HTMLTableConfig)
|
|
85
|
+
or config.repeat_header
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
def split_code_block(
|
|
89
|
+
self,
|
|
90
|
+
block: DocumentBlock,
|
|
91
|
+
*,
|
|
92
|
+
max_tokens: int,
|
|
93
|
+
) -> list[tuple[str, int]]:
|
|
94
|
+
info = str(block.attrs.get("info") or block.attrs.get("language") or "").strip()
|
|
95
|
+
literal = str(block.attrs.get("literal") or "")
|
|
96
|
+
open_fence = f"```{info}".rstrip()
|
|
97
|
+
close_fence = "```"
|
|
98
|
+
empty_render = f"{open_fence}\n\n{close_fence}"
|
|
99
|
+
wrapper_tokens = self.tokenizer.count(empty_render, cache=True)
|
|
100
|
+
if wrapper_tokens >= max_tokens:
|
|
101
|
+
return [(block.text, self.tokenizer.count(block.text, cache=True))]
|
|
102
|
+
|
|
103
|
+
code_budget = max_tokens - wrapper_tokens
|
|
104
|
+
pieces = self.split_text(literal, max_tokens=code_budget)
|
|
105
|
+
result: list[tuple[str, int]] = []
|
|
106
|
+
for piece, _piece_tokens in pieces:
|
|
107
|
+
wrapped = f"{open_fence}\n{piece}\n{close_fence}"
|
|
108
|
+
# Fences add tokens beyond the literal, so recount the wrapped text.
|
|
109
|
+
result.append((wrapped, self.tokenizer.count(wrapped, cache=True)))
|
|
110
|
+
return result
|
|
111
|
+
|
|
112
|
+
def split_table_block(
|
|
113
|
+
self,
|
|
114
|
+
block: DocumentBlock,
|
|
115
|
+
*,
|
|
116
|
+
default_budget: int | None = None,
|
|
117
|
+
) -> list[tuple[str, int]]:
|
|
118
|
+
# Handle markdown table
|
|
119
|
+
lines = [line.rstrip() for line in block.text.splitlines() if line.strip()]
|
|
120
|
+
max_tokens = self._block_budget(block.kind, default_budget)
|
|
121
|
+
repeat_header = self._repeat_header(block.kind)
|
|
122
|
+
if len(lines) < 3 or not self.is_table_delimiter_row(lines[1]):
|
|
123
|
+
return self.split_text(block.text, max_tokens=max_tokens)
|
|
124
|
+
|
|
125
|
+
def emit_piece(piece_header: list[str], rows: list[str]) -> tuple[str, int]:
|
|
126
|
+
piece = self.render_table_piece(piece_header, rows)
|
|
127
|
+
return (piece, self.tokenizer.count(piece, cache=True))
|
|
128
|
+
|
|
129
|
+
header = lines[:2]
|
|
130
|
+
rows = lines[2:]
|
|
131
|
+
pieces: list[tuple[str, int]] = []
|
|
132
|
+
current_rows: list[str] = []
|
|
133
|
+
|
|
134
|
+
for row in rows:
|
|
135
|
+
candidate_rows = [*current_rows, row]
|
|
136
|
+
candidate_header = header if repeat_header or not pieces else []
|
|
137
|
+
candidate = self.render_table_piece(candidate_header, candidate_rows)
|
|
138
|
+
candidate_tokens = self.tokenizer.count(candidate, cache=True)
|
|
139
|
+
if current_rows and candidate_tokens > max_tokens:
|
|
140
|
+
piece_header = header if repeat_header or not pieces else []
|
|
141
|
+
pieces.append(emit_piece(piece_header, current_rows))
|
|
142
|
+
current_rows = [row]
|
|
143
|
+
single_header = header if repeat_header or not pieces else []
|
|
144
|
+
single_row = self.render_table_piece(single_header, current_rows)
|
|
145
|
+
single_tokens = self.tokenizer.count(single_row, cache=True)
|
|
146
|
+
if single_tokens > max_tokens:
|
|
147
|
+
pieces.append((single_row, single_tokens))
|
|
148
|
+
current_rows = []
|
|
149
|
+
continue
|
|
150
|
+
|
|
151
|
+
if not current_rows and candidate_tokens > max_tokens:
|
|
152
|
+
pieces.append((candidate, candidate_tokens))
|
|
153
|
+
continue
|
|
154
|
+
|
|
155
|
+
current_rows = candidate_rows
|
|
156
|
+
|
|
157
|
+
if current_rows:
|
|
158
|
+
piece_header = header if repeat_header or not pieces else []
|
|
159
|
+
pieces.append(emit_piece(piece_header, current_rows))
|
|
160
|
+
|
|
161
|
+
return pieces or [(block.text, self.tokenizer.count(block.text, cache=True))]
|
|
162
|
+
|
|
163
|
+
def is_table_delimiter_row(self, line: str) -> bool:
|
|
164
|
+
cells = [cell.strip() for cell in line.strip().strip("|").split("|")]
|
|
165
|
+
return bool(cells) and all(
|
|
166
|
+
cell and TABLE_DELIMITER_CELL_RE.fullmatch(cell) for cell in cells
|
|
167
|
+
)
|
|
168
|
+
|
|
169
|
+
def render_table_piece(self, header: list[str], rows: list[str]) -> str:
|
|
170
|
+
return "\n".join([*header, *rows])
|
|
171
|
+
|
|
172
|
+
def split_html_table_block(
|
|
173
|
+
self,
|
|
174
|
+
block: DocumentBlock,
|
|
175
|
+
*,
|
|
176
|
+
default_budget: int | None = None,
|
|
177
|
+
) -> list[tuple[str, int]]:
|
|
178
|
+
"""Split an HTML table while preserving its original HTML format.
|
|
179
|
+
|
|
180
|
+
This method extracts HTML tables and splits them by rows while keeping
|
|
181
|
+
the HTML structure intact, without converting to markdown format.
|
|
182
|
+
"""
|
|
183
|
+
max_tokens = self._block_budget(block.kind, default_budget)
|
|
184
|
+
repeat_header = self._repeat_header(block.kind)
|
|
185
|
+
tables = self._html_table_parser.extract_tables(block.text)
|
|
186
|
+
if not tables:
|
|
187
|
+
return [(block.text, self.tokenizer.count(block.text, cache=True))]
|
|
188
|
+
|
|
189
|
+
def emit(html: str) -> tuple[str, int]:
|
|
190
|
+
return (html, self.tokenizer.count(html, cache=True))
|
|
191
|
+
|
|
192
|
+
pieces: list[tuple[str, int]] = []
|
|
193
|
+
for html_table in tables:
|
|
194
|
+
# Get the raw HTML content
|
|
195
|
+
table_html = html_table.raw_html
|
|
196
|
+
|
|
197
|
+
# Extract the opening <table> tag with all attributes
|
|
198
|
+
table_open_tag = ""
|
|
199
|
+
table_match = re.search(r"<table\b[^>]*>", table_html, re.IGNORECASE)
|
|
200
|
+
if table_match:
|
|
201
|
+
table_open_tag = table_match.group(0)
|
|
202
|
+
|
|
203
|
+
# Extract caption if present
|
|
204
|
+
caption_html = ""
|
|
205
|
+
if html_table.caption:
|
|
206
|
+
caption_match = self._html_table_parser.CAPTION_RE.search(table_html)
|
|
207
|
+
if caption_match:
|
|
208
|
+
caption_html = caption_match.group(0)
|
|
209
|
+
|
|
210
|
+
# Split by rows while preserving HTML structure
|
|
211
|
+
header_rows = list(html_table.headers)
|
|
212
|
+
data_rows = list(html_table.rows)
|
|
213
|
+
|
|
214
|
+
if not data_rows:
|
|
215
|
+
pieces.append(emit(table_html))
|
|
216
|
+
continue
|
|
217
|
+
|
|
218
|
+
# Group rows by token budget
|
|
219
|
+
current_rows: list[HTMLTableRow] = []
|
|
220
|
+
pieces_count = 0
|
|
221
|
+
|
|
222
|
+
for row in data_rows:
|
|
223
|
+
test_rows = [*current_rows, row]
|
|
224
|
+
# Build test HTML to check token count
|
|
225
|
+
candidate_headers = (
|
|
226
|
+
header_rows if repeat_header or pieces_count == 0 else []
|
|
227
|
+
)
|
|
228
|
+
test_html = self._build_html_table_piece(
|
|
229
|
+
table_open_tag, caption_html, candidate_headers, test_rows
|
|
230
|
+
)
|
|
231
|
+
test_tokens = self.tokenizer.count(test_html, cache=True)
|
|
232
|
+
|
|
233
|
+
if current_rows and test_tokens > max_tokens:
|
|
234
|
+
# Emit current group. ``piece_html`` differs from
|
|
235
|
+
# ``test_html`` (it drops the candidate row), so it needs
|
|
236
|
+
# its own count — but its headers share the same
|
|
237
|
+
# ``pieces_count == 0`` decision as ``candidate_headers``.
|
|
238
|
+
piece_headers = (
|
|
239
|
+
header_rows if repeat_header or pieces_count == 0 else []
|
|
240
|
+
)
|
|
241
|
+
piece_html = self._build_html_table_piece(
|
|
242
|
+
table_open_tag, caption_html, piece_headers, current_rows
|
|
243
|
+
)
|
|
244
|
+
pieces.append(emit(piece_html))
|
|
245
|
+
current_rows = [row]
|
|
246
|
+
pieces_count += 1
|
|
247
|
+
elif not current_rows and test_tokens > max_tokens:
|
|
248
|
+
# Single row exceeds budget, emit ``test_html`` as is —
|
|
249
|
+
# reuse the count we just computed for the budget check.
|
|
250
|
+
pieces.append((test_html, test_tokens))
|
|
251
|
+
current_rows = []
|
|
252
|
+
pieces_count += 1
|
|
253
|
+
else:
|
|
254
|
+
current_rows.append(row)
|
|
255
|
+
|
|
256
|
+
# Don't forget remaining rows
|
|
257
|
+
if current_rows:
|
|
258
|
+
piece_headers = (
|
|
259
|
+
header_rows if repeat_header or pieces_count == 0 else []
|
|
260
|
+
)
|
|
261
|
+
piece_html = self._build_html_table_piece(
|
|
262
|
+
table_open_tag, caption_html, piece_headers, current_rows
|
|
263
|
+
)
|
|
264
|
+
pieces.append(emit(piece_html))
|
|
265
|
+
pieces_count += 1
|
|
266
|
+
|
|
267
|
+
return (
|
|
268
|
+
pieces
|
|
269
|
+
if pieces
|
|
270
|
+
else [(block.text, self.tokenizer.count(block.text, cache=True))]
|
|
271
|
+
)
|
|
272
|
+
|
|
273
|
+
def _build_html_table_piece(
|
|
274
|
+
self,
|
|
275
|
+
table_open_tag: str,
|
|
276
|
+
caption_html: str,
|
|
277
|
+
header_rows: list[HTMLTableRow],
|
|
278
|
+
data_rows: list[HTMLTableRow],
|
|
279
|
+
) -> str:
|
|
280
|
+
"""Build a complete HTML table piece from components.
|
|
281
|
+
|
|
282
|
+
Args:
|
|
283
|
+
table_open_tag: Complete opening <table> tag with attributes.
|
|
284
|
+
caption_html: Raw HTML caption string.
|
|
285
|
+
header_rows: List of header row objects.
|
|
286
|
+
data_rows: List of data row objects to include.
|
|
287
|
+
|
|
288
|
+
Returns:
|
|
289
|
+
Complete HTML table string.
|
|
290
|
+
"""
|
|
291
|
+
lines: list[str] = [table_open_tag if table_open_tag else "<table>"]
|
|
292
|
+
|
|
293
|
+
# Add caption
|
|
294
|
+
if caption_html:
|
|
295
|
+
lines.append(caption_html)
|
|
296
|
+
|
|
297
|
+
# Add header rows
|
|
298
|
+
for header_row in header_rows:
|
|
299
|
+
lines.append(header_row.raw_html)
|
|
300
|
+
|
|
301
|
+
# Add data rows
|
|
302
|
+
for data_row in data_rows:
|
|
303
|
+
lines.append(data_row.raw_html)
|
|
304
|
+
|
|
305
|
+
lines.append("</table>")
|
|
306
|
+
return "\n".join(lines)
|
|
307
|
+
|
|
308
|
+
def split_list_block(
|
|
309
|
+
self,
|
|
310
|
+
block: DocumentBlock,
|
|
311
|
+
*,
|
|
312
|
+
max_tokens: int,
|
|
313
|
+
) -> list[tuple[str, int]]:
|
|
314
|
+
items = [child.text for child in block.children if child.text]
|
|
315
|
+
if len(items) <= 1:
|
|
316
|
+
return self.split_text(
|
|
317
|
+
block.text,
|
|
318
|
+
max_tokens=max_tokens,
|
|
319
|
+
)
|
|
320
|
+
|
|
321
|
+
packed = self.pack_parts(
|
|
322
|
+
items,
|
|
323
|
+
max_tokens,
|
|
324
|
+
separator="\n",
|
|
325
|
+
)
|
|
326
|
+
if all(tokens <= max_tokens for _, tokens in packed):
|
|
327
|
+
return packed
|
|
328
|
+
|
|
329
|
+
pieces: list[tuple[str, int]] = []
|
|
330
|
+
for item in items:
|
|
331
|
+
item_tokens = self.tokenizer.count(item, cache=True)
|
|
332
|
+
if item_tokens <= max_tokens:
|
|
333
|
+
pieces.append((item, item_tokens))
|
|
334
|
+
continue
|
|
335
|
+
pieces.extend(
|
|
336
|
+
self.split_text(
|
|
337
|
+
item,
|
|
338
|
+
max_tokens=max_tokens,
|
|
339
|
+
)
|
|
340
|
+
)
|
|
341
|
+
return pieces
|
|
342
|
+
|
|
343
|
+
def split_text(
|
|
344
|
+
self,
|
|
345
|
+
text: str,
|
|
346
|
+
*,
|
|
347
|
+
max_tokens: int,
|
|
348
|
+
) -> list[tuple[str, int]]:
|
|
349
|
+
text_tokens = self.tokenizer.count(text, cache=True)
|
|
350
|
+
if text_tokens <= max_tokens:
|
|
351
|
+
return [(text, text_tokens)]
|
|
352
|
+
|
|
353
|
+
if any(
|
|
354
|
+
self.tokenizer.count(m.group(0), cache=True) > max_tokens
|
|
355
|
+
for m in PROTECTED_SPAN_RE.finditer(text)
|
|
356
|
+
):
|
|
357
|
+
return [(text, text_tokens)]
|
|
358
|
+
|
|
359
|
+
for separator in ("\n\n", "\n"):
|
|
360
|
+
parts = [part.strip() for part in text.split(separator) if part.strip()]
|
|
361
|
+
if len(parts) > 1:
|
|
362
|
+
packed = self._pack_fitting_parts(
|
|
363
|
+
parts,
|
|
364
|
+
max_tokens,
|
|
365
|
+
separator=separator,
|
|
366
|
+
)
|
|
367
|
+
if packed is not None:
|
|
368
|
+
return packed
|
|
369
|
+
|
|
370
|
+
sentence_parts = [
|
|
371
|
+
part.strip() for part in SENTENCE_BREAK_RE.split(text) if part.strip()
|
|
372
|
+
]
|
|
373
|
+
if len(sentence_parts) > 1:
|
|
374
|
+
packed = self._pack_fitting_parts(
|
|
375
|
+
sentence_parts,
|
|
376
|
+
max_tokens,
|
|
377
|
+
separator=" ",
|
|
378
|
+
)
|
|
379
|
+
if packed is not None:
|
|
380
|
+
return packed
|
|
381
|
+
|
|
382
|
+
word_parts = [part for part in text.split(" ") if part]
|
|
383
|
+
if len(word_parts) > 1:
|
|
384
|
+
packed = self._pack_fitting_parts(
|
|
385
|
+
word_parts,
|
|
386
|
+
max_tokens,
|
|
387
|
+
separator=" ",
|
|
388
|
+
)
|
|
389
|
+
if packed is not None:
|
|
390
|
+
return packed
|
|
391
|
+
|
|
392
|
+
return self.hard_split(text, max_tokens)
|
|
393
|
+
|
|
394
|
+
def _pack_fitting_parts(
|
|
395
|
+
self,
|
|
396
|
+
parts: list[str],
|
|
397
|
+
max_tokens: int,
|
|
398
|
+
*,
|
|
399
|
+
separator: str,
|
|
400
|
+
) -> list[tuple[str, int]] | None:
|
|
401
|
+
"""Pack one fallback level only when every atomic part fits."""
|
|
402
|
+
part_token_counts = [self.tokenizer.count(part, cache=True) for part in parts]
|
|
403
|
+
if any(tokens > max_tokens for tokens in part_token_counts):
|
|
404
|
+
return None
|
|
405
|
+
return self.pack_parts(
|
|
406
|
+
parts,
|
|
407
|
+
max_tokens,
|
|
408
|
+
separator=separator,
|
|
409
|
+
part_token_counts=part_token_counts,
|
|
410
|
+
)
|
|
411
|
+
|
|
412
|
+
def pack_parts(
|
|
413
|
+
self,
|
|
414
|
+
parts: list[str],
|
|
415
|
+
max_tokens: int,
|
|
416
|
+
*,
|
|
417
|
+
separator: str,
|
|
418
|
+
part_token_counts: list[int] | None = None,
|
|
419
|
+
) -> list[tuple[str, int]]:
|
|
420
|
+
if part_token_counts is not None and len(part_token_counts) != len(parts):
|
|
421
|
+
raise ValueError("part_token_counts must match parts")
|
|
422
|
+
|
|
423
|
+
packed: list[tuple[str, int]] = []
|
|
424
|
+
current_parts: list[str] = []
|
|
425
|
+
current_joined = ""
|
|
426
|
+
current_tokens = 0
|
|
427
|
+
for index, part in enumerate(parts):
|
|
428
|
+
candidate_text = (
|
|
429
|
+
current_joined + separator + part if current_joined else part
|
|
430
|
+
)
|
|
431
|
+
candidate_tokens = self.tokenizer.count(candidate_text, cache=True)
|
|
432
|
+
if current_parts and candidate_tokens > max_tokens:
|
|
433
|
+
packed.append((current_joined, current_tokens))
|
|
434
|
+
current_parts = [part]
|
|
435
|
+
current_joined = part
|
|
436
|
+
current_tokens = (
|
|
437
|
+
part_token_counts[index]
|
|
438
|
+
if part_token_counts is not None
|
|
439
|
+
else self.tokenizer.count(part, cache=True)
|
|
440
|
+
)
|
|
441
|
+
else:
|
|
442
|
+
current_parts.append(part)
|
|
443
|
+
current_joined = candidate_text
|
|
444
|
+
current_tokens = candidate_tokens
|
|
445
|
+
if current_parts:
|
|
446
|
+
packed.append((current_joined, current_tokens))
|
|
447
|
+
return packed
|
|
448
|
+
|
|
449
|
+
def hard_split(
|
|
450
|
+
self,
|
|
451
|
+
text: str,
|
|
452
|
+
max_tokens: int,
|
|
453
|
+
) -> list[tuple[str, int]]:
|
|
454
|
+
"""Split with bounded prefix searches instead of growing recounts."""
|
|
455
|
+
result: list[tuple[str, int]] = []
|
|
456
|
+
start = 0
|
|
457
|
+
|
|
458
|
+
while start < len(text):
|
|
459
|
+
lower = start
|
|
460
|
+
step = 1
|
|
461
|
+
upper = min(len(text), start + step)
|
|
462
|
+
|
|
463
|
+
while upper < len(text):
|
|
464
|
+
if self.tokenizer.count(text[start:upper], cache=True) > max_tokens:
|
|
465
|
+
break
|
|
466
|
+
lower = upper
|
|
467
|
+
step *= 2
|
|
468
|
+
upper = min(len(text), start + step)
|
|
469
|
+
|
|
470
|
+
if upper == len(text) and (
|
|
471
|
+
self.tokenizer.count(text[start:upper], cache=True) <= max_tokens
|
|
472
|
+
):
|
|
473
|
+
lower = upper
|
|
474
|
+
|
|
475
|
+
if lower == start and (
|
|
476
|
+
self.tokenizer.count(text[start : start + 1], cache=True) > max_tokens
|
|
477
|
+
):
|
|
478
|
+
lower = start + 1
|
|
479
|
+
|
|
480
|
+
if lower < upper and lower < len(text):
|
|
481
|
+
left = max(start + 1, lower + 1)
|
|
482
|
+
right = upper
|
|
483
|
+
best = lower
|
|
484
|
+
while left <= right:
|
|
485
|
+
middle = (left + right) // 2
|
|
486
|
+
if (
|
|
487
|
+
self.tokenizer.count(text[start:middle], cache=True)
|
|
488
|
+
<= max_tokens
|
|
489
|
+
):
|
|
490
|
+
best = middle
|
|
491
|
+
left = middle + 1
|
|
492
|
+
else:
|
|
493
|
+
right = middle - 1
|
|
494
|
+
lower = best
|
|
495
|
+
|
|
496
|
+
if lower == start:
|
|
497
|
+
lower = start + 1
|
|
498
|
+
|
|
499
|
+
piece = text[start:lower].strip()
|
|
500
|
+
if piece:
|
|
501
|
+
result.append((piece, self.tokenizer.count(piece, cache=True)))
|
|
502
|
+
start = lower
|
|
503
|
+
|
|
504
|
+
return result
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
SUPPORTED_FORMATS = frozenset({"auto", "markdown", "html", "docx"})
|
|
6
|
+
TEXT_FORMATS = frozenset({"markdown", "html"})
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def detect_format(source: str | bytes | Path, format: str) -> str:
|
|
10
|
+
"""Resolve the input format from an explicit hint and source shape."""
|
|
11
|
+
if format not in SUPPORTED_FORMATS:
|
|
12
|
+
msg = f"Unsupported input format: {format}"
|
|
13
|
+
raise ValueError(msg)
|
|
14
|
+
|
|
15
|
+
if format != "auto":
|
|
16
|
+
return format
|
|
17
|
+
|
|
18
|
+
if isinstance(source, bytes):
|
|
19
|
+
return "docx"
|
|
20
|
+
|
|
21
|
+
if isinstance(source, Path):
|
|
22
|
+
return detect_format_from_filename(source.name)
|
|
23
|
+
|
|
24
|
+
return "markdown"
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def detect_format_from_filename(filename: str) -> str:
|
|
28
|
+
"""Detect an input format from a filename extension."""
|
|
29
|
+
suffix = Path(filename).suffix.lower()
|
|
30
|
+
if suffix == ".docx":
|
|
31
|
+
return "docx"
|
|
32
|
+
if suffix in {".html", ".htm"}:
|
|
33
|
+
return "html"
|
|
34
|
+
return "markdown"
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def read_text_input(source: str | bytes | Path) -> str:
|
|
38
|
+
"""Read Markdown or HTML text from any supported source shape."""
|
|
39
|
+
if isinstance(source, Path):
|
|
40
|
+
return source.read_text(encoding="utf-8")
|
|
41
|
+
if isinstance(source, bytes):
|
|
42
|
+
return source.decode("utf-8")
|
|
43
|
+
return source
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def read_docx_input(source: str | bytes | Path) -> bytes:
|
|
47
|
+
"""Read DOCX binary content from any supported source shape."""
|
|
48
|
+
if isinstance(source, Path):
|
|
49
|
+
return source.read_bytes()
|
|
50
|
+
if isinstance(source, str):
|
|
51
|
+
raise TypeError(
|
|
52
|
+
"Expected bytes or a .docx file path for DOCX format, got a text string. "
|
|
53
|
+
"Pass a Path or bytes instead."
|
|
54
|
+
)
|
|
55
|
+
return source
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
"""CLI and Web adapters for public block configuration objects."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
from collections.abc import Mapping
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from lumberjack.block import (
|
|
10
|
+
BlockConfig,
|
|
11
|
+
BlockKind,
|
|
12
|
+
BlockOption,
|
|
13
|
+
CustomBlockConfig,
|
|
14
|
+
HTMLTableConfig,
|
|
15
|
+
MarkdownTableConfig,
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
BASE_FIELDS = frozenset({"isolated", "split", "max_tokens"})
|
|
19
|
+
TABLE_FIELDS = frozenset({"repeat_header"})
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def block_config_from_mapping(kind: str, config: Mapping[str, Any]) -> BlockOption:
|
|
23
|
+
"""Convert one external mapping into a typed public block config."""
|
|
24
|
+
normalized = kind.strip().lower()
|
|
25
|
+
is_table = normalized in {BlockKind.TABLE, BlockKind.HTML_TABLE}
|
|
26
|
+
valid_fields = BASE_FIELDS | (TABLE_FIELDS if is_table else frozenset())
|
|
27
|
+
unknown = set(config) - valid_fields
|
|
28
|
+
if unknown:
|
|
29
|
+
names = ", ".join(sorted(unknown))
|
|
30
|
+
valid = ", ".join(sorted(valid_fields))
|
|
31
|
+
raise ValueError(
|
|
32
|
+
f"Unknown block config field(s) for {kind!r}: {names}. Valid fields: {valid}"
|
|
33
|
+
)
|
|
34
|
+
base = {name: config[name] for name in BASE_FIELDS if name in config}
|
|
35
|
+
if normalized == BlockKind.TABLE:
|
|
36
|
+
return MarkdownTableConfig(
|
|
37
|
+
**base, repeat_header=config.get("repeat_header", True)
|
|
38
|
+
)
|
|
39
|
+
if normalized == BlockKind.HTML_TABLE:
|
|
40
|
+
return HTMLTableConfig(**base, repeat_header=config.get("repeat_header", True))
|
|
41
|
+
try:
|
|
42
|
+
return BlockConfig(BlockKind(normalized), **base)
|
|
43
|
+
except ValueError:
|
|
44
|
+
return CustomBlockConfig(normalized, **base)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def parse_block_config_mapping(
|
|
48
|
+
raw: Mapping[str, Any] | None,
|
|
49
|
+
) -> list[BlockOption] | None:
|
|
50
|
+
if raw is None:
|
|
51
|
+
return None
|
|
52
|
+
result: list[BlockOption] = []
|
|
53
|
+
for kind, config in raw.items():
|
|
54
|
+
if not isinstance(config, Mapping):
|
|
55
|
+
raise TypeError(f"block_configs[{kind!r}] must be an object")
|
|
56
|
+
result.append(block_config_from_mapping(kind, config))
|
|
57
|
+
return result
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def parse_block_config_json(raw: str) -> list[BlockOption] | None:
|
|
61
|
+
if not raw or not raw.strip():
|
|
62
|
+
return None
|
|
63
|
+
try:
|
|
64
|
+
parsed = json.loads(raw)
|
|
65
|
+
except json.JSONDecodeError as exc:
|
|
66
|
+
raise ValueError("Invalid block_configs JSON") from exc
|
|
67
|
+
if not isinstance(parsed, Mapping):
|
|
68
|
+
raise TypeError("block_configs must be a JSON object")
|
|
69
|
+
return parse_block_config_mapping(parsed)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _parse_cli_entry(entry: str, block_kinds: frozenset[str]) -> BlockOption:
|
|
73
|
+
parts = [part.strip() for part in entry.split(":")]
|
|
74
|
+
kind = parts[0].lower() if parts else ""
|
|
75
|
+
if not kind:
|
|
76
|
+
raise ValueError("block config kind cannot be empty")
|
|
77
|
+
if kind not in block_kinds:
|
|
78
|
+
valid = ", ".join(sorted(block_kinds))
|
|
79
|
+
raise ValueError(f"Unknown block kind: {kind!r} (valid: {valid})")
|
|
80
|
+
|
|
81
|
+
config: dict[str, object] = {}
|
|
82
|
+
for token in parts[1:]:
|
|
83
|
+
lowered = token.lower()
|
|
84
|
+
if not token:
|
|
85
|
+
continue
|
|
86
|
+
if lowered == "isolated":
|
|
87
|
+
config["isolated"] = True
|
|
88
|
+
elif lowered == "nosplit":
|
|
89
|
+
config["split"] = False
|
|
90
|
+
else:
|
|
91
|
+
try:
|
|
92
|
+
config["max_tokens"] = int(token)
|
|
93
|
+
except ValueError as exc:
|
|
94
|
+
raise ValueError(f"Unknown block config token: {token!r}") from exc
|
|
95
|
+
return block_config_from_mapping(kind, config)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def parse_cli_block_configs(
|
|
99
|
+
entries: list[str],
|
|
100
|
+
*,
|
|
101
|
+
block_kinds: frozenset[str],
|
|
102
|
+
json_config: str = "",
|
|
103
|
+
) -> list[BlockOption]:
|
|
104
|
+
"""Parse CLI config, with JSON entries overriding short-form entries."""
|
|
105
|
+
indexed = {
|
|
106
|
+
str(config.kind): config
|
|
107
|
+
for config in (_parse_cli_entry(entry, block_kinds) for entry in entries)
|
|
108
|
+
}
|
|
109
|
+
json_options = parse_block_config_json(json_config)
|
|
110
|
+
if json_options:
|
|
111
|
+
indexed.update((str(config.kind), config) for config in json_options)
|
|
112
|
+
return list(indexed.values())
|