graph-knowledge-doc-parser 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. graph_knowledge_doc_parser-0.1.0.dist-info/METADATA +326 -0
  2. graph_knowledge_doc_parser-0.1.0.dist-info/RECORD +38 -0
  3. graph_knowledge_doc_parser-0.1.0.dist-info/WHEEL +4 -0
  4. graph_knowledge_doc_parser-0.1.0.dist-info/entry_points.txt +3 -0
  5. kg_doc_parser/__init__.py +9 -0
  6. kg_doc_parser/cast_hinting.py +19 -0
  7. kg_doc_parser/document_ingester_logger.py +766 -0
  8. kg_doc_parser/models.py +277 -0
  9. kg_doc_parser/ocr.py +752 -0
  10. kg_doc_parser/pdf2png.py +286 -0
  11. kg_doc_parser/semantic_document_splitting_layerwise_edits.py +3302 -0
  12. kg_doc_parser/text_processing_utils.py +30 -0
  13. kg_doc_parser/utils/__init__.py +0 -0
  14. kg_doc_parser/utils/bounded_threadpool_executor.py +37 -0
  15. kg_doc_parser/utils/file_loaders.py +405 -0
  16. kg_doc_parser/utils/langchain.py +220 -0
  17. kg_doc_parser/utils/log.py +135 -0
  18. kg_doc_parser/utils/version_chaining.py +1278 -0
  19. kg_doc_parser/workflow_ingest/__init__.py +187 -0
  20. kg_doc_parser/workflow_ingest/_kogwistar.py +13 -0
  21. kg_doc_parser/workflow_ingest/adapters.py +212 -0
  22. kg_doc_parser/workflow_ingest/cache.py +63 -0
  23. kg_doc_parser/workflow_ingest/cli.py +324 -0
  24. kg_doc_parser/workflow_ingest/clients.py +444 -0
  25. kg_doc_parser/workflow_ingest/demo_harness.py +427 -0
  26. kg_doc_parser/workflow_ingest/design.py +208 -0
  27. kg_doc_parser/workflow_ingest/handlers.py +617 -0
  28. kg_doc_parser/workflow_ingest/models.py +575 -0
  29. kg_doc_parser/workflow_ingest/ocr_pipeline.py +1581 -0
  30. kg_doc_parser/workflow_ingest/page_index.py +473 -0
  31. kg_doc_parser/workflow_ingest/parser_core.py +862 -0
  32. kg_doc_parser/workflow_ingest/parsing.py +249 -0
  33. kg_doc_parser/workflow_ingest/probe.py +164 -0
  34. kg_doc_parser/workflow_ingest/providers.py +412 -0
  35. kg_doc_parser/workflow_ingest/runners.py +546 -0
  36. kg_doc_parser/workflow_ingest/semantics.py +231 -0
  37. kg_doc_parser/workflow_ingest/service.py +112 -0
  38. kg_doc_parser/workflow_ingest/smoke_assets.py +62 -0
@@ -0,0 +1,473 @@
1
+ from __future__ import annotations
2
+
3
+ """Reusable page-index document parsing for text and Markdown inputs.
4
+
5
+ The pipeline keeps a fast heuristic mode for deterministic structure extraction
6
+ and an Ollama-backed mode that reuses the existing parser provider boundary.
7
+ Both modes normalize raw content into page-aware source units and return a
8
+ semantic tree with hydrated spans.
9
+
10
+ Example CLI
11
+ -----------
12
+ Heuristic mode:
13
+
14
+ .venv\\Scripts\\python.exe -m pytest \
15
+ tests/test_workflow_ingest_page_index_pipeline.py::test_page_index_heuristic_parses_text_and_markdown[text] -q
16
+
17
+ .venv\\Scripts\\python.exe -m pytest \
18
+ tests/test_workflow_ingest_page_index_pipeline.py::test_page_index_heuristic_parses_text_and_markdown[markdown] -q
19
+
20
+ Ollama mode with a local Gemma parser model:
21
+
22
+ set KG_DOC_PARSER_PROVIDER=ollama
23
+ set KG_DOC_PARSER_MODEL=gemma4
24
+ set KG_DOC_PARSER_BASE_URL=http://127.0.0.1:11434
25
+ .venv\\Scripts\\python.exe -m pytest \
26
+ tests/test_workflow_ingest_page_index_pipeline.py::test_page_index_ollama_smoke_parses_text_and_markdown[text] -q
27
+
28
+ set KG_DOC_PARSER_PROVIDER=ollama
29
+ set KG_DOC_PARSER_MODEL=gemma4
30
+ set KG_DOC_PARSER_BASE_URL=http://127.0.0.1:11434
31
+ .venv\\Scripts\\python.exe -m pytest \
32
+ tests/test_workflow_ingest_page_index_pipeline.py::test_page_index_ollama_smoke_parses_text_and_markdown[markdown] -q
33
+ """
34
+
35
+ import re
36
+ from dataclasses import dataclass
37
+ from typing import Any, Literal
38
+
39
+ from pydantic import BaseModel, Field
40
+
41
+ from .adapters import build_authoritative_source_map, build_parser_input_dict, build_parser_source_map
42
+ from .models import GroundedSourceRecord, NormalizedPage, NormalizedSourceCollection, SourceUnit, WorkflowIngestInput
43
+ from .providers import WorkflowProviderSettings, build_chat_model_for_role
44
+ from .semantics import HydratedTextPointer, SemanticNode, compute_pointer_coverage, correct_and_validate_pointer
45
+
46
+ PageIndexMode = Literal["heuristic", "ollama"]
47
+ PageIndexSourceFormat = Literal["text", "markdown"]
48
+ PageIndexNodeType = Literal["SECTION", "SUBSECTION", "PARAGRAPH", "TERM"]
49
+
50
+
51
+ class PageIndexBlockSpec(BaseModel):
52
+ """Recursive structural block emitted by the page-index parser."""
53
+
54
+ title: str
55
+ node_type: PageIndexNodeType
56
+ excerpt: str
57
+ child_nodes: list["PageIndexBlockSpec"] = Field(default_factory=list)
58
+
59
+
60
+ PageIndexBlockSpec.model_rebuild()
61
+
62
+
63
+ @dataclass(slots=True)
64
+ class PageIndexParseResult:
65
+ mode: PageIndexMode
66
+ source_format: PageIndexSourceFormat
67
+ workflow_input: WorkflowIngestInput
68
+ authoritative_source_map: dict[str, GroundedSourceRecord]
69
+ parser_input_dict: dict[str, Any]
70
+ parser_source_map: dict[str, dict[str, Any]]
71
+ semantic_tree: SemanticNode
72
+ coverage: dict[str, Any]
73
+
74
+
75
+ @dataclass(slots=True)
76
+ class _PageUnit:
77
+ page_number: int
78
+ unit_id: str
79
+ text: str
80
+
81
+
82
+ @dataclass(slots=True)
83
+ class _BlockSpan:
84
+ start_char: int
85
+ end_char: int
86
+ text: str
87
+ node_type: PageIndexNodeType
88
+ title: str
89
+ heading_level: int | None = None
90
+
91
+
92
+ def _split_pages(raw_text: str) -> list[str]:
93
+ """Split a logical document into page-sized chunks."""
94
+
95
+ pages = re.split(r"\f|^\s*--- PAGE BREAK ---\s*$", raw_text, flags=re.MULTILINE)
96
+ return [page.strip("\n") for page in pages if page.strip()]
97
+
98
+
99
+ def build_page_index_workflow_input(
100
+ *,
101
+ document_id: str,
102
+ title: str,
103
+ raw_text: str,
104
+ source_format: PageIndexSourceFormat,
105
+ ) -> WorkflowIngestInput:
106
+ pages = _split_pages(raw_text)
107
+ normalized_pages: list[NormalizedPage] = []
108
+ for page_number, page_text in enumerate(pages, start=1):
109
+ normalized_pages.append(
110
+ NormalizedPage(
111
+ page_number=page_number,
112
+ units=[
113
+ SourceUnit(
114
+ modality="text",
115
+ page_number=page_number,
116
+ cluster_number=0,
117
+ text=page_text,
118
+ embedding_space="default_text",
119
+ metadata={"source_format": source_format},
120
+ )
121
+ ],
122
+ metadata={"source_format": source_format},
123
+ )
124
+ )
125
+ return WorkflowIngestInput(
126
+ request_id=document_id,
127
+ collections=[
128
+ NormalizedSourceCollection(
129
+ collection_id=document_id,
130
+ title=title,
131
+ modality="text",
132
+ pages=normalized_pages,
133
+ embedding_spaces=["default_text"],
134
+ metadata={"source_format": source_format},
135
+ )
136
+ ],
137
+ )
138
+
139
+
140
+ def _split_page_blocks(page_text: str, *, source_format: PageIndexSourceFormat = "text") -> list[_BlockSpan]:
141
+ blocks: list[_BlockSpan] = []
142
+ cursor = 0
143
+ paragraph_start: int | None = None
144
+ paragraph_end = 0
145
+ saw_nonblank = False
146
+
147
+ def _flush_paragraph() -> None:
148
+ nonlocal paragraph_start, paragraph_end
149
+ if paragraph_start is None:
150
+ return
151
+ raw = page_text[paragraph_start:paragraph_end]
152
+ trimmed = raw.strip()
153
+ if trimmed:
154
+ relative_start = raw.find(trimmed)
155
+ start_char = paragraph_start + relative_start
156
+ end_char = start_char + len(trimmed) - 1
157
+ blocks.append(_classify_block(trimmed, start_char, end_char, source_format=source_format, is_first=False))
158
+ paragraph_start = None
159
+
160
+ for line in page_text.splitlines(keepends=True):
161
+ line_start = cursor
162
+ line_end = cursor + len(line)
163
+ stripped = line.strip()
164
+ if not stripped:
165
+ _flush_paragraph()
166
+ cursor = line_end
167
+ continue
168
+ line_block = _classify_block(
169
+ stripped,
170
+ line_start,
171
+ line_end - 1,
172
+ source_format=source_format,
173
+ is_first=not saw_nonblank,
174
+ )
175
+ saw_nonblank = True
176
+ if line_block.node_type != "PARAGRAPH":
177
+ _flush_paragraph()
178
+ blocks.append(line_block)
179
+ else:
180
+ if paragraph_start is None:
181
+ paragraph_start = line_start
182
+ paragraph_end = line_end
183
+ cursor = line_end
184
+ _flush_paragraph()
185
+ return blocks
186
+
187
+
188
+ def _classify_block(text: str, start_char: int, end_char: int, *, source_format: PageIndexSourceFormat = "text", is_first: bool = False) -> _BlockSpan:
189
+ stripped = text.strip()
190
+ first_line = stripped.splitlines()[0].strip()
191
+ md_heading = re.match(r"^(#{1,6})\s+(.*)$", first_line)
192
+ if source_format == "markdown" and md_heading:
193
+ level = len(md_heading.group(1))
194
+ title = md_heading.group(2).strip() or first_line.lstrip("#").strip()
195
+ node_type: PageIndexNodeType = "SECTION" if level <= 2 else "SUBSECTION"
196
+ return _BlockSpan(start_char=start_char, end_char=end_char, text=stripped, node_type=node_type, title=title, heading_level=level)
197
+
198
+ plain_heading = re.match(r"^(Section|Clause|Article|Definitions?)\b[:\s].*", first_line, flags=re.IGNORECASE)
199
+ numbered_section = re.match(r"^\d+(?:\.\d+)+\s+\S+", first_line)
200
+ term_like = re.match(r"^\s*(?:\d+[.)]|[-*+])\s+\S+", first_line)
201
+ all_caps_heading = (
202
+ len(first_line.split()) <= 8
203
+ and any(ch.isalpha() for ch in first_line)
204
+ and first_line.upper() == first_line
205
+ )
206
+ title_like_first_line = is_first and len(first_line.split()) <= 6 and not first_line.endswith((".", "!", "?"))
207
+
208
+ if plain_heading or numbered_section or all_caps_heading or title_like_first_line:
209
+ if title_like_first_line or (all_caps_heading and is_first):
210
+ level = 1
211
+ elif numbered_section:
212
+ level = max(2, first_line.count(".") + 2)
213
+ else:
214
+ level = 2
215
+ title = first_line.rstrip(":").strip()
216
+ node_type = "SECTION" if level <= 2 else "SUBSECTION"
217
+ return _BlockSpan(start_char=start_char, end_char=end_char, text=stripped, node_type=node_type, title=title, heading_level=level)
218
+ if term_like:
219
+ title = re.sub(r"^\s*(?:\d+[.)]|[-*+])\s+", "", first_line).strip()
220
+ return _BlockSpan(start_char=start_char, end_char=end_char, text=stripped, node_type="TERM", title=title or first_line, heading_level=None)
221
+ title = first_line[:80].rstrip()
222
+ return _BlockSpan(start_char=start_char, end_char=end_char, text=stripped, node_type="PARAGRAPH", title=title, heading_level=None)
223
+
224
+
225
+ def _heuristic_page_outline(page_text: str, *, page_number: int, source_format: PageIndexSourceFormat) -> list[PageIndexBlockSpec]:
226
+ blocks = _split_page_blocks(page_text, source_format=source_format)
227
+ stack: list[tuple[int, PageIndexBlockSpec]] = []
228
+ roots: list[PageIndexBlockSpec] = []
229
+ for index, block in enumerate(blocks):
230
+ classified = _classify_block(block.text, block.start_char, block.end_char, source_format=source_format, is_first=index == 0)
231
+ spec = PageIndexBlockSpec(
232
+ title=classified.title,
233
+ node_type=classified.node_type,
234
+ excerpt=classified.text,
235
+ )
236
+ if classified.node_type in {"SECTION", "SUBSECTION"}:
237
+ while stack and stack[-1][0] >= int(classified.heading_level or 1):
238
+ stack.pop()
239
+ if stack:
240
+ stack[-1][1].child_nodes.append(spec)
241
+ else:
242
+ roots.append(spec)
243
+ stack.append((int(classified.heading_level or 1), spec))
244
+ continue
245
+ parent = stack[-1][1] if stack else None
246
+ if parent is None:
247
+ roots.append(spec)
248
+ else:
249
+ parent.child_nodes.append(spec)
250
+ return roots
251
+
252
+
253
+ def _llm_page_outline(
254
+ *,
255
+ page_text: str,
256
+ page_number: int,
257
+ source_format: PageIndexSourceFormat,
258
+ provider_settings: WorkflowProviderSettings,
259
+ ) -> list[PageIndexBlockSpec]:
260
+ chat = build_chat_model_for_role("parser", provider_settings)
261
+ structured = chat.with_structured_output(PageIndexBlockSpec, include_raw=True)
262
+ from langchain_core.messages import HumanMessage, SystemMessage
263
+
264
+ prompt = (
265
+ "You are a document parser for a page-index pipeline.\n"
266
+ "Return a hierarchy of section, subsection, paragraph, and term blocks.\n"
267
+ "Prefer a deeper tree when the document contains nested numbering, subclauses, or subheadings.\n"
268
+ "Do not flatten nested structure into one section with many children if the text supports a parent/child relationship.\n"
269
+ "Treat headings and numbered clauses as hierarchy cues: page title > section > subsection > paragraph > term.\n"
270
+ "Keep paragraphs grouped under the nearest heading, and keep terms nested under the clause or subsection they belong to.\n"
271
+ "Use verbatim excerpts from the supplied page text.\n"
272
+ f"Source format: {source_format}\n"
273
+ f"Page number: {page_number}\n"
274
+ "Do not invent content. Keep excerpts short but exact."
275
+ )
276
+ payload = structured.invoke(
277
+ [
278
+ SystemMessage(content=prompt),
279
+ HumanMessage(content=page_text),
280
+ ]
281
+ )
282
+ parsed = payload.get("parsed") if isinstance(payload, dict) else payload
283
+ if parsed is None:
284
+ error = payload.get("parsing_error") if isinstance(payload, dict) else None
285
+ raise ValueError(f"ollama page index parse failed: {error!r}")
286
+ if isinstance(parsed, PageIndexBlockSpec):
287
+ return [parsed]
288
+ if isinstance(parsed, list):
289
+ return [PageIndexBlockSpec.model_validate(item) for item in parsed]
290
+ if isinstance(parsed, dict) and "child_nodes" in parsed:
291
+ return [PageIndexBlockSpec.model_validate(parsed)]
292
+ raise TypeError(f"unexpected ollama page index payload: {type(parsed)!r}")
293
+
294
+
295
+ def _find_page_unit(authoritative_source_map: dict[str, GroundedSourceRecord], page_number: int) -> tuple[str, str]:
296
+ for unit_id, record in authoritative_source_map.items():
297
+ if record.page_number == page_number and record.participates_in_semantic_text:
298
+ return unit_id, record.text
299
+ raise ValueError(f"missing text page for page number {page_number}")
300
+
301
+
302
+ def _resolve_pointer(
303
+ *,
304
+ unit_id: str,
305
+ page_text: str,
306
+ excerpt: str,
307
+ start_at: int = 0,
308
+ ) -> HydratedTextPointer:
309
+ needle = excerpt.strip() or excerpt or page_text.strip()
310
+ candidate = HydratedTextPointer(
311
+ source_cluster_id=unit_id,
312
+ start_char=max(0, start_at),
313
+ end_char=max(0, start_at + max(len(needle), 1) - 1),
314
+ verbatim_text=needle,
315
+ )
316
+ resolved = correct_and_validate_pointer(candidate, {unit_id: {"text": page_text}})
317
+ if resolved is None:
318
+ raise ValueError(f"unable to resolve excerpt against page text for {unit_id!r}: {needle!r}")
319
+ return resolved
320
+
321
+
322
+ def _make_semantic_node(
323
+ *,
324
+ title: str,
325
+ node_type: str,
326
+ parent_id: str | None,
327
+ level_from_root: int,
328
+ pointers: list[HydratedTextPointer],
329
+ ) -> SemanticNode:
330
+ return SemanticNode(
331
+ title=title,
332
+ node_type=node_type,
333
+ parent_id=parent_id,
334
+ level_from_root=level_from_root,
335
+ total_content_pointers=pointers,
336
+ child_nodes=[],
337
+ )
338
+
339
+
340
+ def _materialize_block_tree(
341
+ *,
342
+ block_specs: list[PageIndexBlockSpec],
343
+ page_text: str,
344
+ unit_id: str,
345
+ parent_id: str,
346
+ level_from_root: int,
347
+ start_at: int = 0,
348
+ ) -> tuple[list[SemanticNode], int]:
349
+ nodes: list[SemanticNode] = []
350
+ cursor = start_at
351
+ for spec in block_specs:
352
+ pointer = _resolve_pointer(unit_id=unit_id, page_text=page_text, excerpt=spec.excerpt, start_at=cursor)
353
+ cursor = pointer.end_char + 1
354
+ node = _make_semantic_node(
355
+ title=spec.title,
356
+ node_type=spec.node_type,
357
+ parent_id=parent_id,
358
+ level_from_root=level_from_root,
359
+ pointers=[pointer],
360
+ )
361
+ child_nodes, cursor = _materialize_block_tree(
362
+ block_specs=spec.child_nodes,
363
+ page_text=page_text,
364
+ unit_id=unit_id,
365
+ parent_id=node.node_id or parent_id,
366
+ level_from_root=level_from_root + 1,
367
+ start_at=cursor,
368
+ )
369
+ node.child_nodes.extend(child_nodes)
370
+ nodes.append(node)
371
+ return nodes, cursor
372
+
373
+
374
+ def parse_page_index_document(
375
+ *,
376
+ document_id: str,
377
+ title: str,
378
+ raw_text: str,
379
+ source_format: PageIndexSourceFormat = "text",
380
+ mode: PageIndexMode = "heuristic",
381
+ provider_settings: WorkflowProviderSettings | None = None,
382
+ ) -> PageIndexParseResult:
383
+ """Parse a plain text or Markdown document into a page-index semantic tree."""
384
+
385
+ workflow_input = build_page_index_workflow_input(
386
+ document_id=document_id,
387
+ title=title,
388
+ raw_text=raw_text,
389
+ source_format=source_format,
390
+ )
391
+ authoritative_source_map = build_authoritative_source_map(workflow_input)
392
+ parser_input_dict = build_parser_input_dict(workflow_input.collections[0])
393
+ parser_source_map = build_parser_source_map(authoritative_source_map)
394
+
395
+ page_units = sorted(
396
+ (
397
+ (unit_id, record)
398
+ for unit_id, record in authoritative_source_map.items()
399
+ if record.participates_in_semantic_text
400
+ ),
401
+ key=lambda item: (item[1].page_number, item[1].cluster_number or 0, item[0]),
402
+ )
403
+
404
+ root_pointers: list[HydratedTextPointer] = []
405
+ page_nodes: list[SemanticNode] = []
406
+ for page_number, (unit_id, record) in enumerate(page_units, start=1):
407
+ root_pointers.append(
408
+ HydratedTextPointer(
409
+ source_cluster_id=unit_id,
410
+ start_char=0,
411
+ end_char=max(0, len(record.text) - 1),
412
+ verbatim_text=record.text,
413
+ )
414
+ )
415
+ page_text = record.text
416
+ if mode == "heuristic":
417
+ block_specs = _heuristic_page_outline(page_text, page_number=page_number, source_format=source_format)
418
+ elif mode == "ollama":
419
+ settings = provider_settings or WorkflowProviderSettings.from_env()
420
+ if settings.parser.provider != "ollama":
421
+ raise ValueError("ollama mode requires KG_DOC_PARSER_PROVIDER=ollama")
422
+ block_specs = _llm_page_outline(
423
+ page_text=page_text,
424
+ page_number=page_number,
425
+ source_format=source_format,
426
+ provider_settings=settings,
427
+ )
428
+ else: # pragma: no cover - Literal guards this in type-checked code.
429
+ raise ValueError(f"unsupported page index mode: {mode}")
430
+
431
+ page_node = _make_semantic_node(
432
+ title=f"Page {page_number}",
433
+ node_type="PAGE",
434
+ parent_id=None,
435
+ level_from_root=1,
436
+ pointers=[
437
+ HydratedTextPointer(
438
+ source_cluster_id=unit_id,
439
+ start_char=0,
440
+ end_char=max(0, len(page_text) - 1),
441
+ verbatim_text=page_text,
442
+ )
443
+ ],
444
+ )
445
+ child_nodes, _ = _materialize_block_tree(
446
+ block_specs=block_specs,
447
+ page_text=page_text,
448
+ unit_id=unit_id,
449
+ parent_id=page_node.node_id or document_id,
450
+ level_from_root=2,
451
+ )
452
+ page_node.child_nodes.extend(child_nodes)
453
+ page_nodes.append(page_node)
454
+
455
+ semantic_tree = SemanticNode(
456
+ title=title,
457
+ node_type="DOCUMENT_ROOT",
458
+ parent_id=None,
459
+ level_from_root=0,
460
+ total_content_pointers=root_pointers,
461
+ child_nodes=page_nodes,
462
+ )
463
+ coverage = compute_pointer_coverage(semantic_tree, parser_source_map)
464
+ return PageIndexParseResult(
465
+ mode=mode,
466
+ source_format=source_format,
467
+ workflow_input=workflow_input,
468
+ authoritative_source_map=authoritative_source_map,
469
+ parser_input_dict=parser_input_dict,
470
+ parser_source_map=parser_source_map,
471
+ semantic_tree=semantic_tree,
472
+ coverage=coverage,
473
+ )