graph-knowledge-doc-parser 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graph_knowledge_doc_parser-0.1.0.dist-info/METADATA +326 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/RECORD +38 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/WHEEL +4 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/entry_points.txt +3 -0
- kg_doc_parser/__init__.py +9 -0
- kg_doc_parser/cast_hinting.py +19 -0
- kg_doc_parser/document_ingester_logger.py +766 -0
- kg_doc_parser/models.py +277 -0
- kg_doc_parser/ocr.py +752 -0
- kg_doc_parser/pdf2png.py +286 -0
- kg_doc_parser/semantic_document_splitting_layerwise_edits.py +3302 -0
- kg_doc_parser/text_processing_utils.py +30 -0
- kg_doc_parser/utils/__init__.py +0 -0
- kg_doc_parser/utils/bounded_threadpool_executor.py +37 -0
- kg_doc_parser/utils/file_loaders.py +405 -0
- kg_doc_parser/utils/langchain.py +220 -0
- kg_doc_parser/utils/log.py +135 -0
- kg_doc_parser/utils/version_chaining.py +1278 -0
- kg_doc_parser/workflow_ingest/__init__.py +187 -0
- kg_doc_parser/workflow_ingest/_kogwistar.py +13 -0
- kg_doc_parser/workflow_ingest/adapters.py +212 -0
- kg_doc_parser/workflow_ingest/cache.py +63 -0
- kg_doc_parser/workflow_ingest/cli.py +324 -0
- kg_doc_parser/workflow_ingest/clients.py +444 -0
- kg_doc_parser/workflow_ingest/demo_harness.py +427 -0
- kg_doc_parser/workflow_ingest/design.py +208 -0
- kg_doc_parser/workflow_ingest/handlers.py +617 -0
- kg_doc_parser/workflow_ingest/models.py +575 -0
- kg_doc_parser/workflow_ingest/ocr_pipeline.py +1581 -0
- kg_doc_parser/workflow_ingest/page_index.py +473 -0
- kg_doc_parser/workflow_ingest/parser_core.py +862 -0
- kg_doc_parser/workflow_ingest/parsing.py +249 -0
- kg_doc_parser/workflow_ingest/probe.py +164 -0
- kg_doc_parser/workflow_ingest/providers.py +412 -0
- kg_doc_parser/workflow_ingest/runners.py +546 -0
- kg_doc_parser/workflow_ingest/semantics.py +231 -0
- kg_doc_parser/workflow_ingest/service.py +112 -0
- kg_doc_parser/workflow_ingest/smoke_assets.py +62 -0
|
@@ -0,0 +1,473 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
"""Reusable page-index document parsing for text and Markdown inputs.
|
|
4
|
+
|
|
5
|
+
The pipeline keeps a fast heuristic mode for deterministic structure extraction
|
|
6
|
+
and an Ollama-backed mode that reuses the existing parser provider boundary.
|
|
7
|
+
Both modes normalize raw content into page-aware source units and return a
|
|
8
|
+
semantic tree with hydrated spans.
|
|
9
|
+
|
|
10
|
+
Example CLI
|
|
11
|
+
-----------
|
|
12
|
+
Heuristic mode:
|
|
13
|
+
|
|
14
|
+
.venv\\Scripts\\python.exe -m pytest \
|
|
15
|
+
tests/test_workflow_ingest_page_index_pipeline.py::test_page_index_heuristic_parses_text_and_markdown[text] -q
|
|
16
|
+
|
|
17
|
+
.venv\\Scripts\\python.exe -m pytest \
|
|
18
|
+
tests/test_workflow_ingest_page_index_pipeline.py::test_page_index_heuristic_parses_text_and_markdown[markdown] -q
|
|
19
|
+
|
|
20
|
+
Ollama mode with a local Gemma parser model:
|
|
21
|
+
|
|
22
|
+
set KG_DOC_PARSER_PROVIDER=ollama
|
|
23
|
+
set KG_DOC_PARSER_MODEL=gemma4
|
|
24
|
+
set KG_DOC_PARSER_BASE_URL=http://127.0.0.1:11434
|
|
25
|
+
.venv\\Scripts\\python.exe -m pytest \
|
|
26
|
+
tests/test_workflow_ingest_page_index_pipeline.py::test_page_index_ollama_smoke_parses_text_and_markdown[text] -q
|
|
27
|
+
|
|
28
|
+
set KG_DOC_PARSER_PROVIDER=ollama
|
|
29
|
+
set KG_DOC_PARSER_MODEL=gemma4
|
|
30
|
+
set KG_DOC_PARSER_BASE_URL=http://127.0.0.1:11434
|
|
31
|
+
.venv\\Scripts\\python.exe -m pytest \
|
|
32
|
+
tests/test_workflow_ingest_page_index_pipeline.py::test_page_index_ollama_smoke_parses_text_and_markdown[markdown] -q
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
import re
|
|
36
|
+
from dataclasses import dataclass
|
|
37
|
+
from typing import Any, Literal
|
|
38
|
+
|
|
39
|
+
from pydantic import BaseModel, Field
|
|
40
|
+
|
|
41
|
+
from .adapters import build_authoritative_source_map, build_parser_input_dict, build_parser_source_map
|
|
42
|
+
from .models import GroundedSourceRecord, NormalizedPage, NormalizedSourceCollection, SourceUnit, WorkflowIngestInput
|
|
43
|
+
from .providers import WorkflowProviderSettings, build_chat_model_for_role
|
|
44
|
+
from .semantics import HydratedTextPointer, SemanticNode, compute_pointer_coverage, correct_and_validate_pointer
|
|
45
|
+
|
|
46
|
+
PageIndexMode = Literal["heuristic", "ollama"]
|
|
47
|
+
PageIndexSourceFormat = Literal["text", "markdown"]
|
|
48
|
+
PageIndexNodeType = Literal["SECTION", "SUBSECTION", "PARAGRAPH", "TERM"]
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class PageIndexBlockSpec(BaseModel):
|
|
52
|
+
"""Recursive structural block emitted by the page-index parser."""
|
|
53
|
+
|
|
54
|
+
title: str
|
|
55
|
+
node_type: PageIndexNodeType
|
|
56
|
+
excerpt: str
|
|
57
|
+
child_nodes: list["PageIndexBlockSpec"] = Field(default_factory=list)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
PageIndexBlockSpec.model_rebuild()
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
@dataclass(slots=True)
|
|
64
|
+
class PageIndexParseResult:
|
|
65
|
+
mode: PageIndexMode
|
|
66
|
+
source_format: PageIndexSourceFormat
|
|
67
|
+
workflow_input: WorkflowIngestInput
|
|
68
|
+
authoritative_source_map: dict[str, GroundedSourceRecord]
|
|
69
|
+
parser_input_dict: dict[str, Any]
|
|
70
|
+
parser_source_map: dict[str, dict[str, Any]]
|
|
71
|
+
semantic_tree: SemanticNode
|
|
72
|
+
coverage: dict[str, Any]
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
@dataclass(slots=True)
|
|
76
|
+
class _PageUnit:
|
|
77
|
+
page_number: int
|
|
78
|
+
unit_id: str
|
|
79
|
+
text: str
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
@dataclass(slots=True)
|
|
83
|
+
class _BlockSpan:
|
|
84
|
+
start_char: int
|
|
85
|
+
end_char: int
|
|
86
|
+
text: str
|
|
87
|
+
node_type: PageIndexNodeType
|
|
88
|
+
title: str
|
|
89
|
+
heading_level: int | None = None
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _split_pages(raw_text: str) -> list[str]:
|
|
93
|
+
"""Split a logical document into page-sized chunks."""
|
|
94
|
+
|
|
95
|
+
pages = re.split(r"\f|^\s*--- PAGE BREAK ---\s*$", raw_text, flags=re.MULTILINE)
|
|
96
|
+
return [page.strip("\n") for page in pages if page.strip()]
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def build_page_index_workflow_input(
|
|
100
|
+
*,
|
|
101
|
+
document_id: str,
|
|
102
|
+
title: str,
|
|
103
|
+
raw_text: str,
|
|
104
|
+
source_format: PageIndexSourceFormat,
|
|
105
|
+
) -> WorkflowIngestInput:
|
|
106
|
+
pages = _split_pages(raw_text)
|
|
107
|
+
normalized_pages: list[NormalizedPage] = []
|
|
108
|
+
for page_number, page_text in enumerate(pages, start=1):
|
|
109
|
+
normalized_pages.append(
|
|
110
|
+
NormalizedPage(
|
|
111
|
+
page_number=page_number,
|
|
112
|
+
units=[
|
|
113
|
+
SourceUnit(
|
|
114
|
+
modality="text",
|
|
115
|
+
page_number=page_number,
|
|
116
|
+
cluster_number=0,
|
|
117
|
+
text=page_text,
|
|
118
|
+
embedding_space="default_text",
|
|
119
|
+
metadata={"source_format": source_format},
|
|
120
|
+
)
|
|
121
|
+
],
|
|
122
|
+
metadata={"source_format": source_format},
|
|
123
|
+
)
|
|
124
|
+
)
|
|
125
|
+
return WorkflowIngestInput(
|
|
126
|
+
request_id=document_id,
|
|
127
|
+
collections=[
|
|
128
|
+
NormalizedSourceCollection(
|
|
129
|
+
collection_id=document_id,
|
|
130
|
+
title=title,
|
|
131
|
+
modality="text",
|
|
132
|
+
pages=normalized_pages,
|
|
133
|
+
embedding_spaces=["default_text"],
|
|
134
|
+
metadata={"source_format": source_format},
|
|
135
|
+
)
|
|
136
|
+
],
|
|
137
|
+
)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _split_page_blocks(page_text: str, *, source_format: PageIndexSourceFormat = "text") -> list[_BlockSpan]:
|
|
141
|
+
blocks: list[_BlockSpan] = []
|
|
142
|
+
cursor = 0
|
|
143
|
+
paragraph_start: int | None = None
|
|
144
|
+
paragraph_end = 0
|
|
145
|
+
saw_nonblank = False
|
|
146
|
+
|
|
147
|
+
def _flush_paragraph() -> None:
|
|
148
|
+
nonlocal paragraph_start, paragraph_end
|
|
149
|
+
if paragraph_start is None:
|
|
150
|
+
return
|
|
151
|
+
raw = page_text[paragraph_start:paragraph_end]
|
|
152
|
+
trimmed = raw.strip()
|
|
153
|
+
if trimmed:
|
|
154
|
+
relative_start = raw.find(trimmed)
|
|
155
|
+
start_char = paragraph_start + relative_start
|
|
156
|
+
end_char = start_char + len(trimmed) - 1
|
|
157
|
+
blocks.append(_classify_block(trimmed, start_char, end_char, source_format=source_format, is_first=False))
|
|
158
|
+
paragraph_start = None
|
|
159
|
+
|
|
160
|
+
for line in page_text.splitlines(keepends=True):
|
|
161
|
+
line_start = cursor
|
|
162
|
+
line_end = cursor + len(line)
|
|
163
|
+
stripped = line.strip()
|
|
164
|
+
if not stripped:
|
|
165
|
+
_flush_paragraph()
|
|
166
|
+
cursor = line_end
|
|
167
|
+
continue
|
|
168
|
+
line_block = _classify_block(
|
|
169
|
+
stripped,
|
|
170
|
+
line_start,
|
|
171
|
+
line_end - 1,
|
|
172
|
+
source_format=source_format,
|
|
173
|
+
is_first=not saw_nonblank,
|
|
174
|
+
)
|
|
175
|
+
saw_nonblank = True
|
|
176
|
+
if line_block.node_type != "PARAGRAPH":
|
|
177
|
+
_flush_paragraph()
|
|
178
|
+
blocks.append(line_block)
|
|
179
|
+
else:
|
|
180
|
+
if paragraph_start is None:
|
|
181
|
+
paragraph_start = line_start
|
|
182
|
+
paragraph_end = line_end
|
|
183
|
+
cursor = line_end
|
|
184
|
+
_flush_paragraph()
|
|
185
|
+
return blocks
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _classify_block(text: str, start_char: int, end_char: int, *, source_format: PageIndexSourceFormat = "text", is_first: bool = False) -> _BlockSpan:
|
|
189
|
+
stripped = text.strip()
|
|
190
|
+
first_line = stripped.splitlines()[0].strip()
|
|
191
|
+
md_heading = re.match(r"^(#{1,6})\s+(.*)$", first_line)
|
|
192
|
+
if source_format == "markdown" and md_heading:
|
|
193
|
+
level = len(md_heading.group(1))
|
|
194
|
+
title = md_heading.group(2).strip() or first_line.lstrip("#").strip()
|
|
195
|
+
node_type: PageIndexNodeType = "SECTION" if level <= 2 else "SUBSECTION"
|
|
196
|
+
return _BlockSpan(start_char=start_char, end_char=end_char, text=stripped, node_type=node_type, title=title, heading_level=level)
|
|
197
|
+
|
|
198
|
+
plain_heading = re.match(r"^(Section|Clause|Article|Definitions?)\b[:\s].*", first_line, flags=re.IGNORECASE)
|
|
199
|
+
numbered_section = re.match(r"^\d+(?:\.\d+)+\s+\S+", first_line)
|
|
200
|
+
term_like = re.match(r"^\s*(?:\d+[.)]|[-*+])\s+\S+", first_line)
|
|
201
|
+
all_caps_heading = (
|
|
202
|
+
len(first_line.split()) <= 8
|
|
203
|
+
and any(ch.isalpha() for ch in first_line)
|
|
204
|
+
and first_line.upper() == first_line
|
|
205
|
+
)
|
|
206
|
+
title_like_first_line = is_first and len(first_line.split()) <= 6 and not first_line.endswith((".", "!", "?"))
|
|
207
|
+
|
|
208
|
+
if plain_heading or numbered_section or all_caps_heading or title_like_first_line:
|
|
209
|
+
if title_like_first_line or (all_caps_heading and is_first):
|
|
210
|
+
level = 1
|
|
211
|
+
elif numbered_section:
|
|
212
|
+
level = max(2, first_line.count(".") + 2)
|
|
213
|
+
else:
|
|
214
|
+
level = 2
|
|
215
|
+
title = first_line.rstrip(":").strip()
|
|
216
|
+
node_type = "SECTION" if level <= 2 else "SUBSECTION"
|
|
217
|
+
return _BlockSpan(start_char=start_char, end_char=end_char, text=stripped, node_type=node_type, title=title, heading_level=level)
|
|
218
|
+
if term_like:
|
|
219
|
+
title = re.sub(r"^\s*(?:\d+[.)]|[-*+])\s+", "", first_line).strip()
|
|
220
|
+
return _BlockSpan(start_char=start_char, end_char=end_char, text=stripped, node_type="TERM", title=title or first_line, heading_level=None)
|
|
221
|
+
title = first_line[:80].rstrip()
|
|
222
|
+
return _BlockSpan(start_char=start_char, end_char=end_char, text=stripped, node_type="PARAGRAPH", title=title, heading_level=None)
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def _heuristic_page_outline(page_text: str, *, page_number: int, source_format: PageIndexSourceFormat) -> list[PageIndexBlockSpec]:
|
|
226
|
+
blocks = _split_page_blocks(page_text, source_format=source_format)
|
|
227
|
+
stack: list[tuple[int, PageIndexBlockSpec]] = []
|
|
228
|
+
roots: list[PageIndexBlockSpec] = []
|
|
229
|
+
for index, block in enumerate(blocks):
|
|
230
|
+
classified = _classify_block(block.text, block.start_char, block.end_char, source_format=source_format, is_first=index == 0)
|
|
231
|
+
spec = PageIndexBlockSpec(
|
|
232
|
+
title=classified.title,
|
|
233
|
+
node_type=classified.node_type,
|
|
234
|
+
excerpt=classified.text,
|
|
235
|
+
)
|
|
236
|
+
if classified.node_type in {"SECTION", "SUBSECTION"}:
|
|
237
|
+
while stack and stack[-1][0] >= int(classified.heading_level or 1):
|
|
238
|
+
stack.pop()
|
|
239
|
+
if stack:
|
|
240
|
+
stack[-1][1].child_nodes.append(spec)
|
|
241
|
+
else:
|
|
242
|
+
roots.append(spec)
|
|
243
|
+
stack.append((int(classified.heading_level or 1), spec))
|
|
244
|
+
continue
|
|
245
|
+
parent = stack[-1][1] if stack else None
|
|
246
|
+
if parent is None:
|
|
247
|
+
roots.append(spec)
|
|
248
|
+
else:
|
|
249
|
+
parent.child_nodes.append(spec)
|
|
250
|
+
return roots
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def _llm_page_outline(
|
|
254
|
+
*,
|
|
255
|
+
page_text: str,
|
|
256
|
+
page_number: int,
|
|
257
|
+
source_format: PageIndexSourceFormat,
|
|
258
|
+
provider_settings: WorkflowProviderSettings,
|
|
259
|
+
) -> list[PageIndexBlockSpec]:
|
|
260
|
+
chat = build_chat_model_for_role("parser", provider_settings)
|
|
261
|
+
structured = chat.with_structured_output(PageIndexBlockSpec, include_raw=True)
|
|
262
|
+
from langchain_core.messages import HumanMessage, SystemMessage
|
|
263
|
+
|
|
264
|
+
prompt = (
|
|
265
|
+
"You are a document parser for a page-index pipeline.\n"
|
|
266
|
+
"Return a hierarchy of section, subsection, paragraph, and term blocks.\n"
|
|
267
|
+
"Prefer a deeper tree when the document contains nested numbering, subclauses, or subheadings.\n"
|
|
268
|
+
"Do not flatten nested structure into one section with many children if the text supports a parent/child relationship.\n"
|
|
269
|
+
"Treat headings and numbered clauses as hierarchy cues: page title > section > subsection > paragraph > term.\n"
|
|
270
|
+
"Keep paragraphs grouped under the nearest heading, and keep terms nested under the clause or subsection they belong to.\n"
|
|
271
|
+
"Use verbatim excerpts from the supplied page text.\n"
|
|
272
|
+
f"Source format: {source_format}\n"
|
|
273
|
+
f"Page number: {page_number}\n"
|
|
274
|
+
"Do not invent content. Keep excerpts short but exact."
|
|
275
|
+
)
|
|
276
|
+
payload = structured.invoke(
|
|
277
|
+
[
|
|
278
|
+
SystemMessage(content=prompt),
|
|
279
|
+
HumanMessage(content=page_text),
|
|
280
|
+
]
|
|
281
|
+
)
|
|
282
|
+
parsed = payload.get("parsed") if isinstance(payload, dict) else payload
|
|
283
|
+
if parsed is None:
|
|
284
|
+
error = payload.get("parsing_error") if isinstance(payload, dict) else None
|
|
285
|
+
raise ValueError(f"ollama page index parse failed: {error!r}")
|
|
286
|
+
if isinstance(parsed, PageIndexBlockSpec):
|
|
287
|
+
return [parsed]
|
|
288
|
+
if isinstance(parsed, list):
|
|
289
|
+
return [PageIndexBlockSpec.model_validate(item) for item in parsed]
|
|
290
|
+
if isinstance(parsed, dict) and "child_nodes" in parsed:
|
|
291
|
+
return [PageIndexBlockSpec.model_validate(parsed)]
|
|
292
|
+
raise TypeError(f"unexpected ollama page index payload: {type(parsed)!r}")
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
def _find_page_unit(authoritative_source_map: dict[str, GroundedSourceRecord], page_number: int) -> tuple[str, str]:
|
|
296
|
+
for unit_id, record in authoritative_source_map.items():
|
|
297
|
+
if record.page_number == page_number and record.participates_in_semantic_text:
|
|
298
|
+
return unit_id, record.text
|
|
299
|
+
raise ValueError(f"missing text page for page number {page_number}")
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
def _resolve_pointer(
|
|
303
|
+
*,
|
|
304
|
+
unit_id: str,
|
|
305
|
+
page_text: str,
|
|
306
|
+
excerpt: str,
|
|
307
|
+
start_at: int = 0,
|
|
308
|
+
) -> HydratedTextPointer:
|
|
309
|
+
needle = excerpt.strip() or excerpt or page_text.strip()
|
|
310
|
+
candidate = HydratedTextPointer(
|
|
311
|
+
source_cluster_id=unit_id,
|
|
312
|
+
start_char=max(0, start_at),
|
|
313
|
+
end_char=max(0, start_at + max(len(needle), 1) - 1),
|
|
314
|
+
verbatim_text=needle,
|
|
315
|
+
)
|
|
316
|
+
resolved = correct_and_validate_pointer(candidate, {unit_id: {"text": page_text}})
|
|
317
|
+
if resolved is None:
|
|
318
|
+
raise ValueError(f"unable to resolve excerpt against page text for {unit_id!r}: {needle!r}")
|
|
319
|
+
return resolved
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def _make_semantic_node(
|
|
323
|
+
*,
|
|
324
|
+
title: str,
|
|
325
|
+
node_type: str,
|
|
326
|
+
parent_id: str | None,
|
|
327
|
+
level_from_root: int,
|
|
328
|
+
pointers: list[HydratedTextPointer],
|
|
329
|
+
) -> SemanticNode:
|
|
330
|
+
return SemanticNode(
|
|
331
|
+
title=title,
|
|
332
|
+
node_type=node_type,
|
|
333
|
+
parent_id=parent_id,
|
|
334
|
+
level_from_root=level_from_root,
|
|
335
|
+
total_content_pointers=pointers,
|
|
336
|
+
child_nodes=[],
|
|
337
|
+
)
|
|
338
|
+
|
|
339
|
+
|
|
340
|
+
def _materialize_block_tree(
|
|
341
|
+
*,
|
|
342
|
+
block_specs: list[PageIndexBlockSpec],
|
|
343
|
+
page_text: str,
|
|
344
|
+
unit_id: str,
|
|
345
|
+
parent_id: str,
|
|
346
|
+
level_from_root: int,
|
|
347
|
+
start_at: int = 0,
|
|
348
|
+
) -> tuple[list[SemanticNode], int]:
|
|
349
|
+
nodes: list[SemanticNode] = []
|
|
350
|
+
cursor = start_at
|
|
351
|
+
for spec in block_specs:
|
|
352
|
+
pointer = _resolve_pointer(unit_id=unit_id, page_text=page_text, excerpt=spec.excerpt, start_at=cursor)
|
|
353
|
+
cursor = pointer.end_char + 1
|
|
354
|
+
node = _make_semantic_node(
|
|
355
|
+
title=spec.title,
|
|
356
|
+
node_type=spec.node_type,
|
|
357
|
+
parent_id=parent_id,
|
|
358
|
+
level_from_root=level_from_root,
|
|
359
|
+
pointers=[pointer],
|
|
360
|
+
)
|
|
361
|
+
child_nodes, cursor = _materialize_block_tree(
|
|
362
|
+
block_specs=spec.child_nodes,
|
|
363
|
+
page_text=page_text,
|
|
364
|
+
unit_id=unit_id,
|
|
365
|
+
parent_id=node.node_id or parent_id,
|
|
366
|
+
level_from_root=level_from_root + 1,
|
|
367
|
+
start_at=cursor,
|
|
368
|
+
)
|
|
369
|
+
node.child_nodes.extend(child_nodes)
|
|
370
|
+
nodes.append(node)
|
|
371
|
+
return nodes, cursor
|
|
372
|
+
|
|
373
|
+
|
|
374
|
+
def parse_page_index_document(
|
|
375
|
+
*,
|
|
376
|
+
document_id: str,
|
|
377
|
+
title: str,
|
|
378
|
+
raw_text: str,
|
|
379
|
+
source_format: PageIndexSourceFormat = "text",
|
|
380
|
+
mode: PageIndexMode = "heuristic",
|
|
381
|
+
provider_settings: WorkflowProviderSettings | None = None,
|
|
382
|
+
) -> PageIndexParseResult:
|
|
383
|
+
"""Parse a plain text or Markdown document into a page-index semantic tree."""
|
|
384
|
+
|
|
385
|
+
workflow_input = build_page_index_workflow_input(
|
|
386
|
+
document_id=document_id,
|
|
387
|
+
title=title,
|
|
388
|
+
raw_text=raw_text,
|
|
389
|
+
source_format=source_format,
|
|
390
|
+
)
|
|
391
|
+
authoritative_source_map = build_authoritative_source_map(workflow_input)
|
|
392
|
+
parser_input_dict = build_parser_input_dict(workflow_input.collections[0])
|
|
393
|
+
parser_source_map = build_parser_source_map(authoritative_source_map)
|
|
394
|
+
|
|
395
|
+
page_units = sorted(
|
|
396
|
+
(
|
|
397
|
+
(unit_id, record)
|
|
398
|
+
for unit_id, record in authoritative_source_map.items()
|
|
399
|
+
if record.participates_in_semantic_text
|
|
400
|
+
),
|
|
401
|
+
key=lambda item: (item[1].page_number, item[1].cluster_number or 0, item[0]),
|
|
402
|
+
)
|
|
403
|
+
|
|
404
|
+
root_pointers: list[HydratedTextPointer] = []
|
|
405
|
+
page_nodes: list[SemanticNode] = []
|
|
406
|
+
for page_number, (unit_id, record) in enumerate(page_units, start=1):
|
|
407
|
+
root_pointers.append(
|
|
408
|
+
HydratedTextPointer(
|
|
409
|
+
source_cluster_id=unit_id,
|
|
410
|
+
start_char=0,
|
|
411
|
+
end_char=max(0, len(record.text) - 1),
|
|
412
|
+
verbatim_text=record.text,
|
|
413
|
+
)
|
|
414
|
+
)
|
|
415
|
+
page_text = record.text
|
|
416
|
+
if mode == "heuristic":
|
|
417
|
+
block_specs = _heuristic_page_outline(page_text, page_number=page_number, source_format=source_format)
|
|
418
|
+
elif mode == "ollama":
|
|
419
|
+
settings = provider_settings or WorkflowProviderSettings.from_env()
|
|
420
|
+
if settings.parser.provider != "ollama":
|
|
421
|
+
raise ValueError("ollama mode requires KG_DOC_PARSER_PROVIDER=ollama")
|
|
422
|
+
block_specs = _llm_page_outline(
|
|
423
|
+
page_text=page_text,
|
|
424
|
+
page_number=page_number,
|
|
425
|
+
source_format=source_format,
|
|
426
|
+
provider_settings=settings,
|
|
427
|
+
)
|
|
428
|
+
else: # pragma: no cover - Literal guards this in type-checked code.
|
|
429
|
+
raise ValueError(f"unsupported page index mode: {mode}")
|
|
430
|
+
|
|
431
|
+
page_node = _make_semantic_node(
|
|
432
|
+
title=f"Page {page_number}",
|
|
433
|
+
node_type="PAGE",
|
|
434
|
+
parent_id=None,
|
|
435
|
+
level_from_root=1,
|
|
436
|
+
pointers=[
|
|
437
|
+
HydratedTextPointer(
|
|
438
|
+
source_cluster_id=unit_id,
|
|
439
|
+
start_char=0,
|
|
440
|
+
end_char=max(0, len(page_text) - 1),
|
|
441
|
+
verbatim_text=page_text,
|
|
442
|
+
)
|
|
443
|
+
],
|
|
444
|
+
)
|
|
445
|
+
child_nodes, _ = _materialize_block_tree(
|
|
446
|
+
block_specs=block_specs,
|
|
447
|
+
page_text=page_text,
|
|
448
|
+
unit_id=unit_id,
|
|
449
|
+
parent_id=page_node.node_id or document_id,
|
|
450
|
+
level_from_root=2,
|
|
451
|
+
)
|
|
452
|
+
page_node.child_nodes.extend(child_nodes)
|
|
453
|
+
page_nodes.append(page_node)
|
|
454
|
+
|
|
455
|
+
semantic_tree = SemanticNode(
|
|
456
|
+
title=title,
|
|
457
|
+
node_type="DOCUMENT_ROOT",
|
|
458
|
+
parent_id=None,
|
|
459
|
+
level_from_root=0,
|
|
460
|
+
total_content_pointers=root_pointers,
|
|
461
|
+
child_nodes=page_nodes,
|
|
462
|
+
)
|
|
463
|
+
coverage = compute_pointer_coverage(semantic_tree, parser_source_map)
|
|
464
|
+
return PageIndexParseResult(
|
|
465
|
+
mode=mode,
|
|
466
|
+
source_format=source_format,
|
|
467
|
+
workflow_input=workflow_input,
|
|
468
|
+
authoritative_source_map=authoritative_source_map,
|
|
469
|
+
parser_input_dict=parser_input_dict,
|
|
470
|
+
parser_source_map=parser_source_map,
|
|
471
|
+
semantic_tree=semantic_tree,
|
|
472
|
+
coverage=coverage,
|
|
473
|
+
)
|