lumberjack-py 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. lumberjack/__init__.py +4 -0
  2. lumberjack/_internal/__init__.py +1 -0
  3. lumberjack/_internal/block_splitter.py +504 -0
  4. lumberjack/_internal/formats.py +55 -0
  5. lumberjack/_internal/options.py +112 -0
  6. lumberjack/_internal/pipeline.py +218 -0
  7. lumberjack/_internal/rendering.py +12 -0
  8. lumberjack/block.py +192 -0
  9. lumberjack/cli.py +142 -0
  10. lumberjack/finalizer.py +103 -0
  11. lumberjack/lumberjack.py +66 -0
  12. lumberjack/models.py +316 -0
  13. lumberjack/normalizer.py +16 -0
  14. lumberjack/parser/__init__.py +24 -0
  15. lumberjack/parser/auto.py +139 -0
  16. lumberjack/parser/docx/__init__.py +3 -0
  17. lumberjack/parser/docx/parser.py +422 -0
  18. lumberjack/parser/html/__init__.py +3 -0
  19. lumberjack/parser/html/parser.py +479 -0
  20. lumberjack/parser/html/table_parser.py +361 -0
  21. lumberjack/parser/markdown/__init__.py +15 -0
  22. lumberjack/parser/markdown/parser.py +1037 -0
  23. lumberjack/parser/markdown/plugins/__init__.py +3 -0
  24. lumberjack/parser/markdown/plugins/brackets_plugin.py +93 -0
  25. lumberjack/protocols.py +43 -0
  26. lumberjack/py.typed +0 -0
  27. lumberjack/splitter/__init__.py +32 -0
  28. lumberjack/splitter/base.py +336 -0
  29. lumberjack/splitter/context.py +118 -0
  30. lumberjack/splitter/exact.py +365 -0
  31. lumberjack/splitter/incremental.py +518 -0
  32. lumberjack/splitter/section.py +54 -0
  33. lumberjack/splitter/sibling.py +21 -0
  34. lumberjack/splitter/subtree.py +22 -0
  35. lumberjack/splitter/topology/__init__.py +1 -0
  36. lumberjack/splitter/topology/section.py +52 -0
  37. lumberjack/splitter/topology/sibling.py +114 -0
  38. lumberjack/splitter/topology/subtree.py +64 -0
  39. lumberjack/tokenizer.py +194 -0
  40. lumberjack/transformer.py +103 -0
  41. lumberjack/web/__init__.py +5 -0
  42. lumberjack/web/__main__.py +22 -0
  43. lumberjack/web/app.py +43 -0
  44. lumberjack/web/routes.py +205 -0
  45. lumberjack_py-0.4.0.dist-info/METADATA +197 -0
  46. lumberjack_py-0.4.0.dist-info/RECORD +49 -0
  47. lumberjack_py-0.4.0.dist-info/WHEEL +4 -0
  48. lumberjack_py-0.4.0.dist-info/entry_points.txt +3 -0
  49. lumberjack_py-0.4.0.dist-info/licenses/LICENSE +21 -0
lumberjack/__init__.py ADDED
@@ -0,0 +1,4 @@
1
+ from .lumberjack import Lumberjack
2
+ from .models import Document
3
+
4
+ __all__ = ["Document", "Lumberjack"]
@@ -0,0 +1 @@
1
+ """Private implementation helpers for Lumberjack's public components."""
@@ -0,0 +1,504 @@
1
+ """Internal oversized-block splitting helpers."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from typing import TYPE_CHECKING
7
+
8
+ from lumberjack.block import (
9
+ BlockOption,
10
+ HTMLTableConfig,
11
+ MarkdownTableConfig,
12
+ default_block_config,
13
+ )
14
+
15
+ from ..parser.html.table_parser import HTMLTableParser, HTMLTableRow
16
+
17
+ if TYPE_CHECKING:
18
+ from ..models import DocumentBlock
19
+ from ..protocols import TokenizerProtocol
20
+
21
+ SENTENCE_BREAK_RE = re.compile(r"(?<=[.!?\u3002\uff01\uff1f])\s+")
22
+ PROTECTED_SPAN_RE = re.compile(r"<https?://[^\s>]+>|https?://[^\s)>\]]+")
23
+ TABLE_DELIMITER_CELL_RE = re.compile(r":?-+(:?-+)*:?")
24
+
25
+
26
+ class BlockSplitter:
27
+ """Splits oversized text blocks into token-bounded pieces."""
28
+
29
+ def __init__(
30
+ self,
31
+ tokenizer: TokenizerProtocol,
32
+ *,
33
+ max_tokens: int,
34
+ block_options: dict[str, BlockOption],
35
+ ) -> None:
36
+ self.tokenizer = tokenizer
37
+ self.max_tokens = max_tokens
38
+ self.block_options = block_options
39
+ self._html_table_parser = HTMLTableParser()
40
+
41
+ def split_oversized_block(
42
+ self,
43
+ block: DocumentBlock,
44
+ *,
45
+ default_budget: int,
46
+ ) -> list[tuple[str, int]] | None:
47
+ config = self._block_config(block.kind)
48
+ if config is None or not config.split:
49
+ return None
50
+
51
+ budget = self._block_budget(block.kind, default_budget)
52
+
53
+ if block.kind in {"code_block", "code_fence"}:
54
+ return self.split_code_block(block, max_tokens=budget)
55
+
56
+ if block.kind == "list" and block.children:
57
+ return self.split_list_block(block, max_tokens=budget)
58
+
59
+ if block.kind == "table":
60
+ return self.split_table_block(block, default_budget=default_budget)
61
+
62
+ if block.kind == "html_table":
63
+ return self.split_html_table_block(
64
+ block,
65
+ default_budget=default_budget,
66
+ )
67
+
68
+ return self.split_text(block.text, max_tokens=budget)
69
+
70
+ def _block_config(self, kind: str) -> BlockOption:
71
+ return self.block_options.get(kind.lower(), default_block_config(kind))
72
+
73
+ def _block_budget(self, kind: str, default_budget: int | None = None) -> int:
74
+ config = self._block_config(kind)
75
+ if config and config.max_tokens:
76
+ return config.max_tokens
77
+ if default_budget is not None:
78
+ return default_budget
79
+ return self.max_tokens
80
+
81
+ def _repeat_header(self, kind: str) -> bool:
82
+ config = self._block_config(kind)
83
+ return (
84
+ not isinstance(config, MarkdownTableConfig | HTMLTableConfig)
85
+ or config.repeat_header
86
+ )
87
+
88
+ def split_code_block(
89
+ self,
90
+ block: DocumentBlock,
91
+ *,
92
+ max_tokens: int,
93
+ ) -> list[tuple[str, int]]:
94
+ info = str(block.attrs.get("info") or block.attrs.get("language") or "").strip()
95
+ literal = str(block.attrs.get("literal") or "")
96
+ open_fence = f"```{info}".rstrip()
97
+ close_fence = "```"
98
+ empty_render = f"{open_fence}\n\n{close_fence}"
99
+ wrapper_tokens = self.tokenizer.count(empty_render, cache=True)
100
+ if wrapper_tokens >= max_tokens:
101
+ return [(block.text, self.tokenizer.count(block.text, cache=True))]
102
+
103
+ code_budget = max_tokens - wrapper_tokens
104
+ pieces = self.split_text(literal, max_tokens=code_budget)
105
+ result: list[tuple[str, int]] = []
106
+ for piece, _piece_tokens in pieces:
107
+ wrapped = f"{open_fence}\n{piece}\n{close_fence}"
108
+ # Fences add tokens beyond the literal, so recount the wrapped text.
109
+ result.append((wrapped, self.tokenizer.count(wrapped, cache=True)))
110
+ return result
111
+
112
+ def split_table_block(
113
+ self,
114
+ block: DocumentBlock,
115
+ *,
116
+ default_budget: int | None = None,
117
+ ) -> list[tuple[str, int]]:
118
+ # Handle markdown table
119
+ lines = [line.rstrip() for line in block.text.splitlines() if line.strip()]
120
+ max_tokens = self._block_budget(block.kind, default_budget)
121
+ repeat_header = self._repeat_header(block.kind)
122
+ if len(lines) < 3 or not self.is_table_delimiter_row(lines[1]):
123
+ return self.split_text(block.text, max_tokens=max_tokens)
124
+
125
+ def emit_piece(piece_header: list[str], rows: list[str]) -> tuple[str, int]:
126
+ piece = self.render_table_piece(piece_header, rows)
127
+ return (piece, self.tokenizer.count(piece, cache=True))
128
+
129
+ header = lines[:2]
130
+ rows = lines[2:]
131
+ pieces: list[tuple[str, int]] = []
132
+ current_rows: list[str] = []
133
+
134
+ for row in rows:
135
+ candidate_rows = [*current_rows, row]
136
+ candidate_header = header if repeat_header or not pieces else []
137
+ candidate = self.render_table_piece(candidate_header, candidate_rows)
138
+ candidate_tokens = self.tokenizer.count(candidate, cache=True)
139
+ if current_rows and candidate_tokens > max_tokens:
140
+ piece_header = header if repeat_header or not pieces else []
141
+ pieces.append(emit_piece(piece_header, current_rows))
142
+ current_rows = [row]
143
+ single_header = header if repeat_header or not pieces else []
144
+ single_row = self.render_table_piece(single_header, current_rows)
145
+ single_tokens = self.tokenizer.count(single_row, cache=True)
146
+ if single_tokens > max_tokens:
147
+ pieces.append((single_row, single_tokens))
148
+ current_rows = []
149
+ continue
150
+
151
+ if not current_rows and candidate_tokens > max_tokens:
152
+ pieces.append((candidate, candidate_tokens))
153
+ continue
154
+
155
+ current_rows = candidate_rows
156
+
157
+ if current_rows:
158
+ piece_header = header if repeat_header or not pieces else []
159
+ pieces.append(emit_piece(piece_header, current_rows))
160
+
161
+ return pieces or [(block.text, self.tokenizer.count(block.text, cache=True))]
162
+
163
+ def is_table_delimiter_row(self, line: str) -> bool:
164
+ cells = [cell.strip() for cell in line.strip().strip("|").split("|")]
165
+ return bool(cells) and all(
166
+ cell and TABLE_DELIMITER_CELL_RE.fullmatch(cell) for cell in cells
167
+ )
168
+
169
+ def render_table_piece(self, header: list[str], rows: list[str]) -> str:
170
+ return "\n".join([*header, *rows])
171
+
172
+ def split_html_table_block(
173
+ self,
174
+ block: DocumentBlock,
175
+ *,
176
+ default_budget: int | None = None,
177
+ ) -> list[tuple[str, int]]:
178
+ """Split an HTML table while preserving its original HTML format.
179
+
180
+ This method extracts HTML tables and splits them by rows while keeping
181
+ the HTML structure intact, without converting to markdown format.
182
+ """
183
+ max_tokens = self._block_budget(block.kind, default_budget)
184
+ repeat_header = self._repeat_header(block.kind)
185
+ tables = self._html_table_parser.extract_tables(block.text)
186
+ if not tables:
187
+ return [(block.text, self.tokenizer.count(block.text, cache=True))]
188
+
189
+ def emit(html: str) -> tuple[str, int]:
190
+ return (html, self.tokenizer.count(html, cache=True))
191
+
192
+ pieces: list[tuple[str, int]] = []
193
+ for html_table in tables:
194
+ # Get the raw HTML content
195
+ table_html = html_table.raw_html
196
+
197
+ # Extract the opening <table> tag with all attributes
198
+ table_open_tag = ""
199
+ table_match = re.search(r"<table\b[^>]*>", table_html, re.IGNORECASE)
200
+ if table_match:
201
+ table_open_tag = table_match.group(0)
202
+
203
+ # Extract caption if present
204
+ caption_html = ""
205
+ if html_table.caption:
206
+ caption_match = self._html_table_parser.CAPTION_RE.search(table_html)
207
+ if caption_match:
208
+ caption_html = caption_match.group(0)
209
+
210
+ # Split by rows while preserving HTML structure
211
+ header_rows = list(html_table.headers)
212
+ data_rows = list(html_table.rows)
213
+
214
+ if not data_rows:
215
+ pieces.append(emit(table_html))
216
+ continue
217
+
218
+ # Group rows by token budget
219
+ current_rows: list[HTMLTableRow] = []
220
+ pieces_count = 0
221
+
222
+ for row in data_rows:
223
+ test_rows = [*current_rows, row]
224
+ # Build test HTML to check token count
225
+ candidate_headers = (
226
+ header_rows if repeat_header or pieces_count == 0 else []
227
+ )
228
+ test_html = self._build_html_table_piece(
229
+ table_open_tag, caption_html, candidate_headers, test_rows
230
+ )
231
+ test_tokens = self.tokenizer.count(test_html, cache=True)
232
+
233
+ if current_rows and test_tokens > max_tokens:
234
+ # Emit current group. ``piece_html`` differs from
235
+ # ``test_html`` (it drops the candidate row), so it needs
236
+ # its own count — but its headers share the same
237
+ # ``pieces_count == 0`` decision as ``candidate_headers``.
238
+ piece_headers = (
239
+ header_rows if repeat_header or pieces_count == 0 else []
240
+ )
241
+ piece_html = self._build_html_table_piece(
242
+ table_open_tag, caption_html, piece_headers, current_rows
243
+ )
244
+ pieces.append(emit(piece_html))
245
+ current_rows = [row]
246
+ pieces_count += 1
247
+ elif not current_rows and test_tokens > max_tokens:
248
+ # Single row exceeds budget, emit ``test_html`` as is —
249
+ # reuse the count we just computed for the budget check.
250
+ pieces.append((test_html, test_tokens))
251
+ current_rows = []
252
+ pieces_count += 1
253
+ else:
254
+ current_rows.append(row)
255
+
256
+ # Don't forget remaining rows
257
+ if current_rows:
258
+ piece_headers = (
259
+ header_rows if repeat_header or pieces_count == 0 else []
260
+ )
261
+ piece_html = self._build_html_table_piece(
262
+ table_open_tag, caption_html, piece_headers, current_rows
263
+ )
264
+ pieces.append(emit(piece_html))
265
+ pieces_count += 1
266
+
267
+ return (
268
+ pieces
269
+ if pieces
270
+ else [(block.text, self.tokenizer.count(block.text, cache=True))]
271
+ )
272
+
273
+ def _build_html_table_piece(
274
+ self,
275
+ table_open_tag: str,
276
+ caption_html: str,
277
+ header_rows: list[HTMLTableRow],
278
+ data_rows: list[HTMLTableRow],
279
+ ) -> str:
280
+ """Build a complete HTML table piece from components.
281
+
282
+ Args:
283
+ table_open_tag: Complete opening <table> tag with attributes.
284
+ caption_html: Raw HTML caption string.
285
+ header_rows: List of header row objects.
286
+ data_rows: List of data row objects to include.
287
+
288
+ Returns:
289
+ Complete HTML table string.
290
+ """
291
+ lines: list[str] = [table_open_tag if table_open_tag else "<table>"]
292
+
293
+ # Add caption
294
+ if caption_html:
295
+ lines.append(caption_html)
296
+
297
+ # Add header rows
298
+ for header_row in header_rows:
299
+ lines.append(header_row.raw_html)
300
+
301
+ # Add data rows
302
+ for data_row in data_rows:
303
+ lines.append(data_row.raw_html)
304
+
305
+ lines.append("</table>")
306
+ return "\n".join(lines)
307
+
308
+ def split_list_block(
309
+ self,
310
+ block: DocumentBlock,
311
+ *,
312
+ max_tokens: int,
313
+ ) -> list[tuple[str, int]]:
314
+ items = [child.text for child in block.children if child.text]
315
+ if len(items) <= 1:
316
+ return self.split_text(
317
+ block.text,
318
+ max_tokens=max_tokens,
319
+ )
320
+
321
+ packed = self.pack_parts(
322
+ items,
323
+ max_tokens,
324
+ separator="\n",
325
+ )
326
+ if all(tokens <= max_tokens for _, tokens in packed):
327
+ return packed
328
+
329
+ pieces: list[tuple[str, int]] = []
330
+ for item in items:
331
+ item_tokens = self.tokenizer.count(item, cache=True)
332
+ if item_tokens <= max_tokens:
333
+ pieces.append((item, item_tokens))
334
+ continue
335
+ pieces.extend(
336
+ self.split_text(
337
+ item,
338
+ max_tokens=max_tokens,
339
+ )
340
+ )
341
+ return pieces
342
+
343
+ def split_text(
344
+ self,
345
+ text: str,
346
+ *,
347
+ max_tokens: int,
348
+ ) -> list[tuple[str, int]]:
349
+ text_tokens = self.tokenizer.count(text, cache=True)
350
+ if text_tokens <= max_tokens:
351
+ return [(text, text_tokens)]
352
+
353
+ if any(
354
+ self.tokenizer.count(m.group(0), cache=True) > max_tokens
355
+ for m in PROTECTED_SPAN_RE.finditer(text)
356
+ ):
357
+ return [(text, text_tokens)]
358
+
359
+ for separator in ("\n\n", "\n"):
360
+ parts = [part.strip() for part in text.split(separator) if part.strip()]
361
+ if len(parts) > 1:
362
+ packed = self._pack_fitting_parts(
363
+ parts,
364
+ max_tokens,
365
+ separator=separator,
366
+ )
367
+ if packed is not None:
368
+ return packed
369
+
370
+ sentence_parts = [
371
+ part.strip() for part in SENTENCE_BREAK_RE.split(text) if part.strip()
372
+ ]
373
+ if len(sentence_parts) > 1:
374
+ packed = self._pack_fitting_parts(
375
+ sentence_parts,
376
+ max_tokens,
377
+ separator=" ",
378
+ )
379
+ if packed is not None:
380
+ return packed
381
+
382
+ word_parts = [part for part in text.split(" ") if part]
383
+ if len(word_parts) > 1:
384
+ packed = self._pack_fitting_parts(
385
+ word_parts,
386
+ max_tokens,
387
+ separator=" ",
388
+ )
389
+ if packed is not None:
390
+ return packed
391
+
392
+ return self.hard_split(text, max_tokens)
393
+
394
+ def _pack_fitting_parts(
395
+ self,
396
+ parts: list[str],
397
+ max_tokens: int,
398
+ *,
399
+ separator: str,
400
+ ) -> list[tuple[str, int]] | None:
401
+ """Pack one fallback level only when every atomic part fits."""
402
+ part_token_counts = [self.tokenizer.count(part, cache=True) for part in parts]
403
+ if any(tokens > max_tokens for tokens in part_token_counts):
404
+ return None
405
+ return self.pack_parts(
406
+ parts,
407
+ max_tokens,
408
+ separator=separator,
409
+ part_token_counts=part_token_counts,
410
+ )
411
+
412
+ def pack_parts(
413
+ self,
414
+ parts: list[str],
415
+ max_tokens: int,
416
+ *,
417
+ separator: str,
418
+ part_token_counts: list[int] | None = None,
419
+ ) -> list[tuple[str, int]]:
420
+ if part_token_counts is not None and len(part_token_counts) != len(parts):
421
+ raise ValueError("part_token_counts must match parts")
422
+
423
+ packed: list[tuple[str, int]] = []
424
+ current_parts: list[str] = []
425
+ current_joined = ""
426
+ current_tokens = 0
427
+ for index, part in enumerate(parts):
428
+ candidate_text = (
429
+ current_joined + separator + part if current_joined else part
430
+ )
431
+ candidate_tokens = self.tokenizer.count(candidate_text, cache=True)
432
+ if current_parts and candidate_tokens > max_tokens:
433
+ packed.append((current_joined, current_tokens))
434
+ current_parts = [part]
435
+ current_joined = part
436
+ current_tokens = (
437
+ part_token_counts[index]
438
+ if part_token_counts is not None
439
+ else self.tokenizer.count(part, cache=True)
440
+ )
441
+ else:
442
+ current_parts.append(part)
443
+ current_joined = candidate_text
444
+ current_tokens = candidate_tokens
445
+ if current_parts:
446
+ packed.append((current_joined, current_tokens))
447
+ return packed
448
+
449
+ def hard_split(
450
+ self,
451
+ text: str,
452
+ max_tokens: int,
453
+ ) -> list[tuple[str, int]]:
454
+ """Split with bounded prefix searches instead of growing recounts."""
455
+ result: list[tuple[str, int]] = []
456
+ start = 0
457
+
458
+ while start < len(text):
459
+ lower = start
460
+ step = 1
461
+ upper = min(len(text), start + step)
462
+
463
+ while upper < len(text):
464
+ if self.tokenizer.count(text[start:upper], cache=True) > max_tokens:
465
+ break
466
+ lower = upper
467
+ step *= 2
468
+ upper = min(len(text), start + step)
469
+
470
+ if upper == len(text) and (
471
+ self.tokenizer.count(text[start:upper], cache=True) <= max_tokens
472
+ ):
473
+ lower = upper
474
+
475
+ if lower == start and (
476
+ self.tokenizer.count(text[start : start + 1], cache=True) > max_tokens
477
+ ):
478
+ lower = start + 1
479
+
480
+ if lower < upper and lower < len(text):
481
+ left = max(start + 1, lower + 1)
482
+ right = upper
483
+ best = lower
484
+ while left <= right:
485
+ middle = (left + right) // 2
486
+ if (
487
+ self.tokenizer.count(text[start:middle], cache=True)
488
+ <= max_tokens
489
+ ):
490
+ best = middle
491
+ left = middle + 1
492
+ else:
493
+ right = middle - 1
494
+ lower = best
495
+
496
+ if lower == start:
497
+ lower = start + 1
498
+
499
+ piece = text[start:lower].strip()
500
+ if piece:
501
+ result.append((piece, self.tokenizer.count(piece, cache=True)))
502
+ start = lower
503
+
504
+ return result
@@ -0,0 +1,55 @@
1
+ from __future__ import annotations
2
+
3
+ from pathlib import Path
4
+
5
+ SUPPORTED_FORMATS = frozenset({"auto", "markdown", "html", "docx"})
6
+ TEXT_FORMATS = frozenset({"markdown", "html"})
7
+
8
+
9
+ def detect_format(source: str | bytes | Path, format: str) -> str:
10
+ """Resolve the input format from an explicit hint and source shape."""
11
+ if format not in SUPPORTED_FORMATS:
12
+ msg = f"Unsupported input format: {format}"
13
+ raise ValueError(msg)
14
+
15
+ if format != "auto":
16
+ return format
17
+
18
+ if isinstance(source, bytes):
19
+ return "docx"
20
+
21
+ if isinstance(source, Path):
22
+ return detect_format_from_filename(source.name)
23
+
24
+ return "markdown"
25
+
26
+
27
+ def detect_format_from_filename(filename: str) -> str:
28
+ """Detect an input format from a filename extension."""
29
+ suffix = Path(filename).suffix.lower()
30
+ if suffix == ".docx":
31
+ return "docx"
32
+ if suffix in {".html", ".htm"}:
33
+ return "html"
34
+ return "markdown"
35
+
36
+
37
+ def read_text_input(source: str | bytes | Path) -> str:
38
+ """Read Markdown or HTML text from any supported source shape."""
39
+ if isinstance(source, Path):
40
+ return source.read_text(encoding="utf-8")
41
+ if isinstance(source, bytes):
42
+ return source.decode("utf-8")
43
+ return source
44
+
45
+
46
+ def read_docx_input(source: str | bytes | Path) -> bytes:
47
+ """Read DOCX binary content from any supported source shape."""
48
+ if isinstance(source, Path):
49
+ return source.read_bytes()
50
+ if isinstance(source, str):
51
+ raise TypeError(
52
+ "Expected bytes or a .docx file path for DOCX format, got a text string. "
53
+ "Pass a Path or bytes instead."
54
+ )
55
+ return source
@@ -0,0 +1,112 @@
1
+ """CLI and Web adapters for public block configuration objects."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ from collections.abc import Mapping
7
+ from typing import Any
8
+
9
+ from lumberjack.block import (
10
+ BlockConfig,
11
+ BlockKind,
12
+ BlockOption,
13
+ CustomBlockConfig,
14
+ HTMLTableConfig,
15
+ MarkdownTableConfig,
16
+ )
17
+
18
+ BASE_FIELDS = frozenset({"isolated", "split", "max_tokens"})
19
+ TABLE_FIELDS = frozenset({"repeat_header"})
20
+
21
+
22
+ def block_config_from_mapping(kind: str, config: Mapping[str, Any]) -> BlockOption:
23
+ """Convert one external mapping into a typed public block config."""
24
+ normalized = kind.strip().lower()
25
+ is_table = normalized in {BlockKind.TABLE, BlockKind.HTML_TABLE}
26
+ valid_fields = BASE_FIELDS | (TABLE_FIELDS if is_table else frozenset())
27
+ unknown = set(config) - valid_fields
28
+ if unknown:
29
+ names = ", ".join(sorted(unknown))
30
+ valid = ", ".join(sorted(valid_fields))
31
+ raise ValueError(
32
+ f"Unknown block config field(s) for {kind!r}: {names}. Valid fields: {valid}"
33
+ )
34
+ base = {name: config[name] for name in BASE_FIELDS if name in config}
35
+ if normalized == BlockKind.TABLE:
36
+ return MarkdownTableConfig(
37
+ **base, repeat_header=config.get("repeat_header", True)
38
+ )
39
+ if normalized == BlockKind.HTML_TABLE:
40
+ return HTMLTableConfig(**base, repeat_header=config.get("repeat_header", True))
41
+ try:
42
+ return BlockConfig(BlockKind(normalized), **base)
43
+ except ValueError:
44
+ return CustomBlockConfig(normalized, **base)
45
+
46
+
47
+ def parse_block_config_mapping(
48
+ raw: Mapping[str, Any] | None,
49
+ ) -> list[BlockOption] | None:
50
+ if raw is None:
51
+ return None
52
+ result: list[BlockOption] = []
53
+ for kind, config in raw.items():
54
+ if not isinstance(config, Mapping):
55
+ raise TypeError(f"block_configs[{kind!r}] must be an object")
56
+ result.append(block_config_from_mapping(kind, config))
57
+ return result
58
+
59
+
60
+ def parse_block_config_json(raw: str) -> list[BlockOption] | None:
61
+ if not raw or not raw.strip():
62
+ return None
63
+ try:
64
+ parsed = json.loads(raw)
65
+ except json.JSONDecodeError as exc:
66
+ raise ValueError("Invalid block_configs JSON") from exc
67
+ if not isinstance(parsed, Mapping):
68
+ raise TypeError("block_configs must be a JSON object")
69
+ return parse_block_config_mapping(parsed)
70
+
71
+
72
+ def _parse_cli_entry(entry: str, block_kinds: frozenset[str]) -> BlockOption:
73
+ parts = [part.strip() for part in entry.split(":")]
74
+ kind = parts[0].lower() if parts else ""
75
+ if not kind:
76
+ raise ValueError("block config kind cannot be empty")
77
+ if kind not in block_kinds:
78
+ valid = ", ".join(sorted(block_kinds))
79
+ raise ValueError(f"Unknown block kind: {kind!r} (valid: {valid})")
80
+
81
+ config: dict[str, object] = {}
82
+ for token in parts[1:]:
83
+ lowered = token.lower()
84
+ if not token:
85
+ continue
86
+ if lowered == "isolated":
87
+ config["isolated"] = True
88
+ elif lowered == "nosplit":
89
+ config["split"] = False
90
+ else:
91
+ try:
92
+ config["max_tokens"] = int(token)
93
+ except ValueError as exc:
94
+ raise ValueError(f"Unknown block config token: {token!r}") from exc
95
+ return block_config_from_mapping(kind, config)
96
+
97
+
98
+ def parse_cli_block_configs(
99
+ entries: list[str],
100
+ *,
101
+ block_kinds: frozenset[str],
102
+ json_config: str = "",
103
+ ) -> list[BlockOption]:
104
+ """Parse CLI config, with JSON entries overriding short-form entries."""
105
+ indexed = {
106
+ str(config.kind): config
107
+ for config in (_parse_cli_entry(entry, block_kinds) for entry in entries)
108
+ }
109
+ json_options = parse_block_config_json(json_config)
110
+ if json_options:
111
+ indexed.update((str(config.kind), config) for config in json_options)
112
+ return list(indexed.values())