graph-knowledge-doc-parser 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graph_knowledge_doc_parser-0.1.0.dist-info/METADATA +326 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/RECORD +38 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/WHEEL +4 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/entry_points.txt +3 -0
- kg_doc_parser/__init__.py +9 -0
- kg_doc_parser/cast_hinting.py +19 -0
- kg_doc_parser/document_ingester_logger.py +766 -0
- kg_doc_parser/models.py +277 -0
- kg_doc_parser/ocr.py +752 -0
- kg_doc_parser/pdf2png.py +286 -0
- kg_doc_parser/semantic_document_splitting_layerwise_edits.py +3302 -0
- kg_doc_parser/text_processing_utils.py +30 -0
- kg_doc_parser/utils/__init__.py +0 -0
- kg_doc_parser/utils/bounded_threadpool_executor.py +37 -0
- kg_doc_parser/utils/file_loaders.py +405 -0
- kg_doc_parser/utils/langchain.py +220 -0
- kg_doc_parser/utils/log.py +135 -0
- kg_doc_parser/utils/version_chaining.py +1278 -0
- kg_doc_parser/workflow_ingest/__init__.py +187 -0
- kg_doc_parser/workflow_ingest/_kogwistar.py +13 -0
- kg_doc_parser/workflow_ingest/adapters.py +212 -0
- kg_doc_parser/workflow_ingest/cache.py +63 -0
- kg_doc_parser/workflow_ingest/cli.py +324 -0
- kg_doc_parser/workflow_ingest/clients.py +444 -0
- kg_doc_parser/workflow_ingest/demo_harness.py +427 -0
- kg_doc_parser/workflow_ingest/design.py +208 -0
- kg_doc_parser/workflow_ingest/handlers.py +617 -0
- kg_doc_parser/workflow_ingest/models.py +575 -0
- kg_doc_parser/workflow_ingest/ocr_pipeline.py +1581 -0
- kg_doc_parser/workflow_ingest/page_index.py +473 -0
- kg_doc_parser/workflow_ingest/parser_core.py +862 -0
- kg_doc_parser/workflow_ingest/parsing.py +249 -0
- kg_doc_parser/workflow_ingest/probe.py +164 -0
- kg_doc_parser/workflow_ingest/providers.py +412 -0
- kg_doc_parser/workflow_ingest/runners.py +546 -0
- kg_doc_parser/workflow_ingest/semantics.py +231 -0
- kg_doc_parser/workflow_ingest/service.py +112 -0
- kg_doc_parser/workflow_ingest/smoke_assets.py +62 -0
|
@@ -0,0 +1,862 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import logging
|
|
4
|
+
import re
|
|
5
|
+
from typing import Any, Callable
|
|
6
|
+
|
|
7
|
+
from .cache import WorkflowLLMCallCache
|
|
8
|
+
from .models import (
|
|
9
|
+
CurrentLayerContext,
|
|
10
|
+
CurrentLayerReview,
|
|
11
|
+
CurrentLayerResult,
|
|
12
|
+
LayerCoverageGap,
|
|
13
|
+
LayerDuplicateChildNote,
|
|
14
|
+
LayerChildCandidate,
|
|
15
|
+
LayerFrontierItem,
|
|
16
|
+
LayerSpanConflict,
|
|
17
|
+
ParseSessionState,
|
|
18
|
+
)
|
|
19
|
+
from .semantics import HydratedTextPointer, SemanticNode
|
|
20
|
+
|
|
21
|
+
_LOGGER = logging.getLogger(__name__)
|
|
22
|
+
_LEGACY_POINTER_ID_RE = re.compile(r"^p(?P<page>\d+)_c(?P<cluster>\d+)$")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def default_parse_semantic_fn(
|
|
26
|
+
*,
|
|
27
|
+
collection,
|
|
28
|
+
parser_input_dict: dict[str, Any],
|
|
29
|
+
parser_source_map: dict[str, dict[str, Any]],
|
|
30
|
+
model_names: list[str] | None = None,
|
|
31
|
+
):
|
|
32
|
+
from ..semantic_document_splitting_layerwise_edits import build_document_tree
|
|
33
|
+
|
|
34
|
+
return build_document_tree(
|
|
35
|
+
doc_id=collection.collection_id,
|
|
36
|
+
llm_input_dict=parser_input_dict,
|
|
37
|
+
source_map=parser_source_map,
|
|
38
|
+
model_names=model_names,
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _coerce_semantic_tree(tree: Any) -> SemanticNode:
|
|
43
|
+
if isinstance(tree, tuple):
|
|
44
|
+
tree = tree[0]
|
|
45
|
+
if hasattr(tree, "model_dump"):
|
|
46
|
+
tree = tree.model_dump(mode="json")
|
|
47
|
+
if isinstance(tree, dict):
|
|
48
|
+
tree = SemanticNode.model_validate(tree)
|
|
49
|
+
return tree
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _root_only(tree: SemanticNode) -> SemanticNode:
|
|
53
|
+
payload = tree.model_dump(mode="json")
|
|
54
|
+
payload["child_nodes"] = []
|
|
55
|
+
return SemanticNode.model_validate(payload)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _legacy_pointer_aliases(parser_source_map: dict[str, dict[str, Any]]) -> dict[str, str]:
|
|
59
|
+
aliases: dict[str, str] = {}
|
|
60
|
+
for unit_id, record in parser_source_map.items():
|
|
61
|
+
page_number = record.get("page_number")
|
|
62
|
+
cluster_number = record.get("cluster_number")
|
|
63
|
+
if page_number is None or cluster_number is None:
|
|
64
|
+
continue
|
|
65
|
+
alias = f"p{int(page_number)}_c{int(cluster_number)}"
|
|
66
|
+
aliases.setdefault(alias, unit_id)
|
|
67
|
+
return aliases
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _canonicalize_legacy_pointer_tree(
|
|
71
|
+
tree: SemanticNode,
|
|
72
|
+
*,
|
|
73
|
+
parser_source_map: dict[str, dict[str, Any]],
|
|
74
|
+
) -> SemanticNode:
|
|
75
|
+
aliases = _legacy_pointer_aliases(parser_source_map)
|
|
76
|
+
if not aliases:
|
|
77
|
+
return tree
|
|
78
|
+
|
|
79
|
+
def _canonicalize_source_cluster_id(source_cluster_id: str) -> str:
|
|
80
|
+
alias_match = _LEGACY_POINTER_ID_RE.match(source_cluster_id)
|
|
81
|
+
if alias_match is None:
|
|
82
|
+
return source_cluster_id
|
|
83
|
+
return aliases.get(
|
|
84
|
+
source_cluster_id,
|
|
85
|
+
source_cluster_id,
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
def _remap_pointer(pointer: HydratedTextPointer) -> HydratedTextPointer:
|
|
89
|
+
canonical_id = _canonicalize_source_cluster_id(pointer.source_cluster_id)
|
|
90
|
+
if canonical_id == pointer.source_cluster_id:
|
|
91
|
+
return pointer
|
|
92
|
+
return pointer.model_copy(update={"source_cluster_id": canonical_id})
|
|
93
|
+
|
|
94
|
+
def _remap_node(node: SemanticNode) -> SemanticNode:
|
|
95
|
+
return node.model_copy(
|
|
96
|
+
update={
|
|
97
|
+
"total_content_pointers": [_remap_pointer(pointer) for pointer in node.total_content_pointers],
|
|
98
|
+
"child_nodes": [_remap_node(child) for child in node.child_nodes],
|
|
99
|
+
}
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
return _remap_node(tree)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _pointer_end(pointer: HydratedTextPointer, source_map: dict[str, dict[str, Any]] | None = None) -> int:
|
|
106
|
+
if pointer.end_char != -1:
|
|
107
|
+
return pointer.end_char
|
|
108
|
+
if source_map is not None:
|
|
109
|
+
text = source_map.get(pointer.source_cluster_id, {}).get("text", "")
|
|
110
|
+
if text:
|
|
111
|
+
return max(0, len(text) - 1)
|
|
112
|
+
return pointer.start_char
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _normalize_text(text: str) -> str:
|
|
116
|
+
return "".join(text.split()).lower()
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _child_pointer_fingerprint(child: LayerChildCandidate) -> tuple[tuple[str, int, int, str], ...]:
|
|
120
|
+
return tuple(
|
|
121
|
+
(
|
|
122
|
+
ptr.source_cluster_id,
|
|
123
|
+
ptr.start_char,
|
|
124
|
+
ptr.end_char,
|
|
125
|
+
_normalize_text(ptr.verbatim_text),
|
|
126
|
+
)
|
|
127
|
+
for ptr in child.total_content_pointers
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _merge_intervals(intervals: list[tuple[int, int]]) -> list[tuple[int, int]]:
|
|
132
|
+
if not intervals:
|
|
133
|
+
return []
|
|
134
|
+
merged: list[tuple[int, int]] = []
|
|
135
|
+
cur_s, cur_e = sorted(intervals)[0]
|
|
136
|
+
for s, e in sorted(intervals)[1:]:
|
|
137
|
+
if s <= cur_e + 1:
|
|
138
|
+
cur_e = max(cur_e, e)
|
|
139
|
+
else:
|
|
140
|
+
merged.append((cur_s, cur_e))
|
|
141
|
+
cur_s, cur_e = s, e
|
|
142
|
+
merged.append((cur_s, cur_e))
|
|
143
|
+
return merged
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _has_meaningful_gap_text(text: str) -> bool:
|
|
147
|
+
return bool("".join(text.split()))
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def detect_layer_invariants(
|
|
151
|
+
*,
|
|
152
|
+
current_layer_context: CurrentLayerContext,
|
|
153
|
+
current_layer_result: CurrentLayerResult,
|
|
154
|
+
parser_source_map: dict[str, dict[str, Any]] | None = None,
|
|
155
|
+
) -> tuple[
|
|
156
|
+
bool,
|
|
157
|
+
bool,
|
|
158
|
+
list[LayerSpanConflict],
|
|
159
|
+
list[LayerCoverageGap],
|
|
160
|
+
list[LayerDuplicateChildNote],
|
|
161
|
+
list[str],
|
|
162
|
+
]:
|
|
163
|
+
overlap_conflicts: list[LayerSpanConflict] = []
|
|
164
|
+
coverage_gaps: list[LayerCoverageGap] = []
|
|
165
|
+
duplicate_notes: list[LayerDuplicateChildNote] = []
|
|
166
|
+
review_notes: list[str] = []
|
|
167
|
+
seen_overlap_pairs: set[tuple[str, str, str, int, int, str]] = set()
|
|
168
|
+
|
|
169
|
+
parent_pointers = current_layer_context.parent_content_pointers_by_id or {}
|
|
170
|
+
for parent_id in current_layer_context.parent_node_ids:
|
|
171
|
+
parent_children = [
|
|
172
|
+
child for child in current_layer_result.children if child.parent_node_id == parent_id
|
|
173
|
+
]
|
|
174
|
+
if not parent_children:
|
|
175
|
+
continue
|
|
176
|
+
|
|
177
|
+
seen_signatures: dict[tuple[str, tuple[tuple[str, int, int, str], ...]], str] = {}
|
|
178
|
+
for child in parent_children:
|
|
179
|
+
signature = (child.title.strip().lower(), _child_pointer_fingerprint(child))
|
|
180
|
+
if signature in seen_signatures:
|
|
181
|
+
duplicate_of = seen_signatures[signature]
|
|
182
|
+
duplicate_notes.append(
|
|
183
|
+
LayerDuplicateChildNote(
|
|
184
|
+
parent_node_id=parent_id,
|
|
185
|
+
child_node_id=child.node_id,
|
|
186
|
+
duplicate_of_child_node_id=duplicate_of,
|
|
187
|
+
reason="duplicate child proposal under the same parent",
|
|
188
|
+
)
|
|
189
|
+
)
|
|
190
|
+
review_notes.append(
|
|
191
|
+
f"duplicate child proposal under parent {parent_id}: {child.node_id} duplicates {duplicate_of}"
|
|
192
|
+
)
|
|
193
|
+
else:
|
|
194
|
+
seen_signatures[signature] = child.node_id
|
|
195
|
+
|
|
196
|
+
for left_index, left_child in enumerate(parent_children):
|
|
197
|
+
for right_child in parent_children[left_index + 1 :]:
|
|
198
|
+
for left_ptr in left_child.total_content_pointers:
|
|
199
|
+
for right_ptr in right_child.total_content_pointers:
|
|
200
|
+
if left_ptr.source_cluster_id != right_ptr.source_cluster_id:
|
|
201
|
+
continue
|
|
202
|
+
left_end = _pointer_end(left_ptr, parser_source_map)
|
|
203
|
+
right_end = _pointer_end(right_ptr, parser_source_map)
|
|
204
|
+
overlap_start = max(left_ptr.start_char, right_ptr.start_char)
|
|
205
|
+
overlap_end = min(left_end, right_end)
|
|
206
|
+
if overlap_start > overlap_end:
|
|
207
|
+
continue
|
|
208
|
+
pair_key = (
|
|
209
|
+
parent_id,
|
|
210
|
+
left_child.node_id,
|
|
211
|
+
right_child.node_id,
|
|
212
|
+
left_ptr.source_cluster_id,
|
|
213
|
+
overlap_start,
|
|
214
|
+
overlap_end,
|
|
215
|
+
"duplicate" if (
|
|
216
|
+
left_ptr.start_char == right_ptr.start_char
|
|
217
|
+
and left_end == right_end
|
|
218
|
+
and _normalize_text(left_ptr.verbatim_text) == _normalize_text(right_ptr.verbatim_text)
|
|
219
|
+
) else "overlap",
|
|
220
|
+
)
|
|
221
|
+
if pair_key in seen_overlap_pairs:
|
|
222
|
+
continue
|
|
223
|
+
seen_overlap_pairs.add(pair_key)
|
|
224
|
+
conflict_kind = pair_key[-1]
|
|
225
|
+
overlap_conflicts.append(
|
|
226
|
+
LayerSpanConflict(
|
|
227
|
+
parent_node_id=parent_id,
|
|
228
|
+
left_child_id=left_child.node_id,
|
|
229
|
+
right_child_id=right_child.node_id,
|
|
230
|
+
source_cluster_id=left_ptr.source_cluster_id,
|
|
231
|
+
left_span=left_ptr,
|
|
232
|
+
right_span=right_ptr,
|
|
233
|
+
overlap_start=overlap_start,
|
|
234
|
+
overlap_end=overlap_end,
|
|
235
|
+
conflict_kind=conflict_kind,
|
|
236
|
+
)
|
|
237
|
+
)
|
|
238
|
+
review_notes.append(
|
|
239
|
+
f"{conflict_kind} between {left_child.node_id} and {right_child.node_id} "
|
|
240
|
+
f"on {left_ptr.source_cluster_id}:{overlap_start}-{overlap_end}"
|
|
241
|
+
)
|
|
242
|
+
|
|
243
|
+
for parent_ptr in parent_pointers.get(parent_id, []):
|
|
244
|
+
parent_end = _pointer_end(parent_ptr, parser_source_map)
|
|
245
|
+
parent_start = max(parent_ptr.start_char, 0)
|
|
246
|
+
if parent_end < parent_start:
|
|
247
|
+
continue
|
|
248
|
+
child_intervals = []
|
|
249
|
+
for child in parent_children:
|
|
250
|
+
for child_ptr in child.total_content_pointers:
|
|
251
|
+
if child_ptr.source_cluster_id != parent_ptr.source_cluster_id:
|
|
252
|
+
continue
|
|
253
|
+
child_intervals.append(
|
|
254
|
+
(max(child_ptr.start_char, 0), _pointer_end(child_ptr, parser_source_map))
|
|
255
|
+
)
|
|
256
|
+
merged = _merge_intervals(child_intervals)
|
|
257
|
+
cursor = parent_start
|
|
258
|
+
cluster_text = parser_source_map.get(parent_ptr.source_cluster_id, {}).get("text", "") if parser_source_map else ""
|
|
259
|
+
for start, end in merged:
|
|
260
|
+
if start > cursor:
|
|
261
|
+
gap_start = cursor
|
|
262
|
+
gap_end = min(start - 1, parent_end)
|
|
263
|
+
if gap_start <= gap_end:
|
|
264
|
+
gap_text = cluster_text[gap_start : gap_end + 1] if cluster_text else ""
|
|
265
|
+
if not _has_meaningful_gap_text(gap_text):
|
|
266
|
+
cursor = max(cursor, gap_end + 1)
|
|
267
|
+
else:
|
|
268
|
+
coverage_gaps.append(
|
|
269
|
+
LayerCoverageGap(
|
|
270
|
+
parent_node_id=parent_id,
|
|
271
|
+
source_cluster_id=parent_ptr.source_cluster_id,
|
|
272
|
+
gap_start=gap_start,
|
|
273
|
+
gap_end=gap_end,
|
|
274
|
+
expected_text=gap_text,
|
|
275
|
+
)
|
|
276
|
+
)
|
|
277
|
+
review_notes.append(
|
|
278
|
+
f"gap in parent {parent_id} for {parent_ptr.source_cluster_id}: {gap_start}-{gap_end}"
|
|
279
|
+
)
|
|
280
|
+
cursor = max(cursor, end + 1)
|
|
281
|
+
if cursor > parent_end:
|
|
282
|
+
break
|
|
283
|
+
if cursor <= parent_end:
|
|
284
|
+
gap_text = cluster_text[cursor : parent_end + 1] if cluster_text else ""
|
|
285
|
+
if not _has_meaningful_gap_text(gap_text):
|
|
286
|
+
continue
|
|
287
|
+
coverage_gaps.append(
|
|
288
|
+
LayerCoverageGap(
|
|
289
|
+
parent_node_id=parent_id,
|
|
290
|
+
source_cluster_id=parent_ptr.source_cluster_id,
|
|
291
|
+
gap_start=cursor,
|
|
292
|
+
gap_end=parent_end,
|
|
293
|
+
expected_text=gap_text,
|
|
294
|
+
)
|
|
295
|
+
)
|
|
296
|
+
review_notes.append(
|
|
297
|
+
f"gap in parent {parent_id} for {parent_ptr.source_cluster_id}: {cursor}-{parent_end}"
|
|
298
|
+
)
|
|
299
|
+
|
|
300
|
+
coverage_ok = not coverage_gaps
|
|
301
|
+
satisfied = coverage_ok and not overlap_conflicts and not duplicate_notes
|
|
302
|
+
return coverage_ok, satisfied, overlap_conflicts, coverage_gaps, duplicate_notes, review_notes
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
def initialize_parse_session(
|
|
306
|
+
*,
|
|
307
|
+
collection,
|
|
308
|
+
parser_input_dict: dict[str, Any],
|
|
309
|
+
parser_source_map: dict[str, dict[str, Any]],
|
|
310
|
+
max_depth: int = 10,
|
|
311
|
+
allow_review: bool = True,
|
|
312
|
+
split_strategy: str = "excerpt_first",
|
|
313
|
+
fallback_split_strategy: str = "boundary_first",
|
|
314
|
+
parse_semantic_fn: Callable[..., Any] | None = None,
|
|
315
|
+
) -> tuple[ParseSessionState, list[LayerFrontierItem], SemanticNode]:
|
|
316
|
+
if parse_semantic_fn is not None:
|
|
317
|
+
full_tree = _coerce_semantic_tree(
|
|
318
|
+
parse_semantic_fn(
|
|
319
|
+
collection=collection,
|
|
320
|
+
parser_input_dict=parser_input_dict,
|
|
321
|
+
parser_source_map=parser_source_map,
|
|
322
|
+
)
|
|
323
|
+
)
|
|
324
|
+
full_tree = _canonicalize_legacy_pointer_tree(
|
|
325
|
+
full_tree,
|
|
326
|
+
parser_source_map=parser_source_map,
|
|
327
|
+
)
|
|
328
|
+
root = _root_only(full_tree)
|
|
329
|
+
session = ParseSessionState(
|
|
330
|
+
collection_id=collection.collection_id,
|
|
331
|
+
root_node_id=root.node_id,
|
|
332
|
+
current_depth=0,
|
|
333
|
+
max_depth=max_depth,
|
|
334
|
+
allow_review=allow_review,
|
|
335
|
+
split_strategy=split_strategy,
|
|
336
|
+
fallback_split_strategy=fallback_split_strategy,
|
|
337
|
+
strategy_history=[split_strategy],
|
|
338
|
+
mode="legacy_compat",
|
|
339
|
+
compat_full_tree=full_tree.model_dump(),
|
|
340
|
+
)
|
|
341
|
+
frontier = [LayerFrontierItem(parent_node_id=root.node_id, depth=0, order=0)]
|
|
342
|
+
return session, frontier, root
|
|
343
|
+
|
|
344
|
+
root = SemanticNode(
|
|
345
|
+
node_id=f"{collection.collection_id}|root",
|
|
346
|
+
title=collection.title,
|
|
347
|
+
node_type="DOCUMENT_ROOT",
|
|
348
|
+
total_content_pointers=[
|
|
349
|
+
HydratedTextPointer(
|
|
350
|
+
source_cluster_id=unit_id,
|
|
351
|
+
start_char=0,
|
|
352
|
+
end_char=-1,
|
|
353
|
+
# The canonical persistence path validates excerpts against the
|
|
354
|
+
# stored document content, so the root pointers need real text.
|
|
355
|
+
verbatim_text=str(record.get("text") or ""),
|
|
356
|
+
)
|
|
357
|
+
for unit_id, record in sorted(parser_source_map.items())
|
|
358
|
+
if record.get("participates_in_semantic_text", True)
|
|
359
|
+
],
|
|
360
|
+
child_nodes=[],
|
|
361
|
+
level_from_root=0,
|
|
362
|
+
)
|
|
363
|
+
session = ParseSessionState(
|
|
364
|
+
collection_id=collection.collection_id,
|
|
365
|
+
root_node_id=root.node_id,
|
|
366
|
+
current_depth=0,
|
|
367
|
+
max_depth=max_depth,
|
|
368
|
+
allow_review=allow_review,
|
|
369
|
+
split_strategy=split_strategy,
|
|
370
|
+
fallback_split_strategy=fallback_split_strategy,
|
|
371
|
+
strategy_history=[split_strategy],
|
|
372
|
+
mode="workflow_layered",
|
|
373
|
+
)
|
|
374
|
+
frontier = [LayerFrontierItem(parent_node_id=root.node_id, depth=0, order=0)]
|
|
375
|
+
return session, frontier, root
|
|
376
|
+
|
|
377
|
+
|
|
378
|
+
def prepare_layer_frontier(
|
|
379
|
+
*,
|
|
380
|
+
parse_session: ParseSessionState,
|
|
381
|
+
frontier_queue: list[LayerFrontierItem],
|
|
382
|
+
semantic_tree: SemanticNode,
|
|
383
|
+
max_retries: int = 3,
|
|
384
|
+
) -> tuple[CurrentLayerContext, list[LayerFrontierItem], ParseSessionState]:
|
|
385
|
+
if not frontier_queue:
|
|
386
|
+
raise ValueError("frontier queue is empty")
|
|
387
|
+
sorted_queue = sorted(frontier_queue, key=lambda item: (item.depth, item.order))
|
|
388
|
+
current_depth = sorted_queue[0].depth
|
|
389
|
+
current_items = [item for item in sorted_queue if item.depth == current_depth]
|
|
390
|
+
remaining = [item for item in sorted_queue if item.depth != current_depth]
|
|
391
|
+
parent_titles = []
|
|
392
|
+
for parent_id in [item.parent_node_id for item in current_items]:
|
|
393
|
+
node = find_semantic_node(semantic_tree, parent_id)
|
|
394
|
+
parent_titles.append(node.title if node is not None else parent_id)
|
|
395
|
+
session = parse_session.model_copy(update={"current_depth": current_depth})
|
|
396
|
+
context = CurrentLayerContext(
|
|
397
|
+
depth=current_depth,
|
|
398
|
+
parent_node_ids=[item.parent_node_id for item in current_items],
|
|
399
|
+
parent_titles=parent_titles,
|
|
400
|
+
parent_content_pointers_by_id={
|
|
401
|
+
item.parent_node_id: list(find_semantic_node(semantic_tree, item.parent_node_id).total_content_pointers)
|
|
402
|
+
if find_semantic_node(semantic_tree, item.parent_node_id) is not None
|
|
403
|
+
else []
|
|
404
|
+
for item in current_items
|
|
405
|
+
},
|
|
406
|
+
split_strategy=parse_session.split_strategy,
|
|
407
|
+
retry_count=int(parse_session.layer_attempts.get(str(current_depth), 0)),
|
|
408
|
+
max_retries=max_retries,
|
|
409
|
+
)
|
|
410
|
+
return context, remaining, session
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
def legacy_children_for_context(
|
|
414
|
+
*,
|
|
415
|
+
parse_session: ParseSessionState,
|
|
416
|
+
current_layer_context: CurrentLayerContext,
|
|
417
|
+
) -> CurrentLayerResult:
|
|
418
|
+
if parse_session.compat_full_tree is None:
|
|
419
|
+
raise ValueError("legacy compatibility tree is missing")
|
|
420
|
+
full_tree = SemanticNode.model_validate(parse_session.compat_full_tree)
|
|
421
|
+
children: list[LayerChildCandidate] = []
|
|
422
|
+
for parent_id in current_layer_context.parent_node_ids:
|
|
423
|
+
parent = find_semantic_node(full_tree, parent_id)
|
|
424
|
+
if parent is None:
|
|
425
|
+
continue
|
|
426
|
+
for child in parent.child_nodes:
|
|
427
|
+
children.append(
|
|
428
|
+
LayerChildCandidate(
|
|
429
|
+
node_id=child.node_id,
|
|
430
|
+
parent_node_id=parent_id,
|
|
431
|
+
title=child.title,
|
|
432
|
+
node_type=child.node_type,
|
|
433
|
+
total_content_pointers=list(child.total_content_pointers),
|
|
434
|
+
expandable=child.node_type != "KEY_VALUE_PAIR",
|
|
435
|
+
metadata={"source": "legacy_compat"},
|
|
436
|
+
)
|
|
437
|
+
)
|
|
438
|
+
return CurrentLayerResult(children=children, satisfied=True, reasoning_history=[])
|
|
439
|
+
|
|
440
|
+
|
|
441
|
+
def propose_layer_breakdown(
|
|
442
|
+
*,
|
|
443
|
+
collection,
|
|
444
|
+
parser_input_dict: dict[str, Any],
|
|
445
|
+
parser_source_map: dict[str, dict[str, Any]],
|
|
446
|
+
parse_session: ParseSessionState,
|
|
447
|
+
current_layer_context: CurrentLayerContext,
|
|
448
|
+
semantic_tree: SemanticNode,
|
|
449
|
+
propose_layer_fn: Callable[..., Any] | None = None,
|
|
450
|
+
llm_cache: WorkflowLLMCallCache | None = None,
|
|
451
|
+
) -> CurrentLayerResult:
|
|
452
|
+
if parse_session.mode == "legacy_compat":
|
|
453
|
+
return legacy_children_for_context(
|
|
454
|
+
parse_session=parse_session,
|
|
455
|
+
current_layer_context=current_layer_context,
|
|
456
|
+
)
|
|
457
|
+
if propose_layer_fn is None:
|
|
458
|
+
raise ValueError("workflow_layered mode requires propose_layer_fn")
|
|
459
|
+
call = lambda: propose_layer_fn(
|
|
460
|
+
collection=collection,
|
|
461
|
+
parser_input_dict=parser_input_dict,
|
|
462
|
+
parser_source_map=parser_source_map,
|
|
463
|
+
parse_session=parse_session,
|
|
464
|
+
current_layer_context=current_layer_context,
|
|
465
|
+
semantic_tree=semantic_tree,
|
|
466
|
+
split_strategy=current_layer_context.split_strategy,
|
|
467
|
+
)
|
|
468
|
+
if llm_cache is not None:
|
|
469
|
+
proposed = llm_cache.cached_call(
|
|
470
|
+
operation="propose_layer_breakdown",
|
|
471
|
+
fingerprint={
|
|
472
|
+
"collection_id": collection.collection_id,
|
|
473
|
+
"parse_session": parse_session,
|
|
474
|
+
"current_layer_context": current_layer_context,
|
|
475
|
+
"semantic_tree": semantic_tree,
|
|
476
|
+
"parser_input_dict": parser_input_dict,
|
|
477
|
+
"parser_source_map": parser_source_map,
|
|
478
|
+
},
|
|
479
|
+
fn=call,
|
|
480
|
+
)
|
|
481
|
+
else:
|
|
482
|
+
proposed = call()
|
|
483
|
+
if isinstance(proposed, CurrentLayerResult):
|
|
484
|
+
return proposed
|
|
485
|
+
if isinstance(proposed, dict):
|
|
486
|
+
return CurrentLayerResult.model_validate(proposed)
|
|
487
|
+
if isinstance(proposed, list):
|
|
488
|
+
return CurrentLayerResult(children=[_coerce_layer_child(child) for child in proposed])
|
|
489
|
+
raise TypeError("unsupported proposed layer result")
|
|
490
|
+
|
|
491
|
+
|
|
492
|
+
def review_layer(
|
|
493
|
+
*,
|
|
494
|
+
parse_session: ParseSessionState,
|
|
495
|
+
current_layer_context: CurrentLayerContext,
|
|
496
|
+
current_layer_result: CurrentLayerResult,
|
|
497
|
+
parser_source_map: dict[str, dict[str, Any]] | None = None,
|
|
498
|
+
review_layer_fn: Callable[..., Any] | None = None,
|
|
499
|
+
llm_cache: WorkflowLLMCallCache | None = None,
|
|
500
|
+
) -> tuple[CurrentLayerReview, ParseSessionState]:
|
|
501
|
+
if parse_session.mode == "legacy_compat" or not parse_session.allow_review:
|
|
502
|
+
return (
|
|
503
|
+
CurrentLayerReview(
|
|
504
|
+
updated_result=current_layer_result.model_copy(update={"satisfied": True}),
|
|
505
|
+
coverage_ok=True,
|
|
506
|
+
satisfied=True,
|
|
507
|
+
strategy_used=current_layer_context.split_strategy,
|
|
508
|
+
),
|
|
509
|
+
parse_session,
|
|
510
|
+
)
|
|
511
|
+
if review_layer_fn is None:
|
|
512
|
+
reviewed = current_layer_result
|
|
513
|
+
else:
|
|
514
|
+
call = lambda: review_layer_fn(
|
|
515
|
+
parse_session=parse_session,
|
|
516
|
+
current_layer_context=current_layer_context,
|
|
517
|
+
current_layer_result=current_layer_result,
|
|
518
|
+
split_strategy=current_layer_context.split_strategy,
|
|
519
|
+
)
|
|
520
|
+
if llm_cache is not None:
|
|
521
|
+
reviewed = llm_cache.cached_call(
|
|
522
|
+
operation=f"review_cud_proposal:{current_layer_context.split_strategy}",
|
|
523
|
+
fingerprint={
|
|
524
|
+
"parse_session": parse_session,
|
|
525
|
+
"current_layer_context": current_layer_context,
|
|
526
|
+
"current_layer_result": current_layer_result,
|
|
527
|
+
"split_strategy": current_layer_context.split_strategy,
|
|
528
|
+
},
|
|
529
|
+
fn=call,
|
|
530
|
+
)
|
|
531
|
+
else:
|
|
532
|
+
reviewed = call()
|
|
533
|
+
if isinstance(reviewed, CurrentLayerReview):
|
|
534
|
+
result = reviewed
|
|
535
|
+
elif isinstance(reviewed, CurrentLayerResult):
|
|
536
|
+
result = CurrentLayerReview(
|
|
537
|
+
updated_result=reviewed,
|
|
538
|
+
coverage_ok=reviewed.metadata.get("layer_coverage_ok"),
|
|
539
|
+
satisfied=reviewed.satisfied,
|
|
540
|
+
)
|
|
541
|
+
elif isinstance(reviewed, dict):
|
|
542
|
+
if "updated_result" in reviewed or "coverage_ok" in reviewed or "review_notes" in reviewed:
|
|
543
|
+
result = CurrentLayerReview.model_validate(reviewed)
|
|
544
|
+
else:
|
|
545
|
+
parsed = CurrentLayerResult.model_validate(reviewed)
|
|
546
|
+
result = CurrentLayerReview(
|
|
547
|
+
updated_result=parsed,
|
|
548
|
+
coverage_ok=parsed.metadata.get("layer_coverage_ok"),
|
|
549
|
+
satisfied=parsed.satisfied,
|
|
550
|
+
)
|
|
551
|
+
else:
|
|
552
|
+
raise TypeError("unsupported reviewed layer result")
|
|
553
|
+
coverage_ok, invariant_satisfied, overlap_conflicts, coverage_gaps, duplicate_notes, invariant_notes = detect_layer_invariants(
|
|
554
|
+
current_layer_context=current_layer_context,
|
|
555
|
+
current_layer_result=result.updated_result or current_layer_result,
|
|
556
|
+
parser_source_map=parser_source_map,
|
|
557
|
+
)
|
|
558
|
+
merged_notes = list(result.review_notes)
|
|
559
|
+
for note in invariant_notes:
|
|
560
|
+
if note not in merged_notes:
|
|
561
|
+
merged_notes.append(note)
|
|
562
|
+
base_satisfied = result.satisfied if result.satisfied is not None else invariant_satisfied
|
|
563
|
+
satisfied = bool(base_satisfied and not (overlap_conflicts or coverage_gaps or duplicate_notes))
|
|
564
|
+
updated_result = (result.updated_result or current_layer_result).model_copy(
|
|
565
|
+
update={
|
|
566
|
+
"satisfied": satisfied,
|
|
567
|
+
"metadata": {
|
|
568
|
+
**(result.updated_result or current_layer_result).metadata,
|
|
569
|
+
"split_strategy": current_layer_context.split_strategy,
|
|
570
|
+
"overlap_conflicts": [
|
|
571
|
+
item.model_dump(field_mode="backend", dump_format="json") for item in overlap_conflicts
|
|
572
|
+
],
|
|
573
|
+
"coverage_gaps": [
|
|
574
|
+
item.model_dump(field_mode="backend", dump_format="json") for item in coverage_gaps
|
|
575
|
+
],
|
|
576
|
+
"duplicate_child_notes": [
|
|
577
|
+
item.model_dump(field_mode="backend", dump_format="json") for item in duplicate_notes
|
|
578
|
+
],
|
|
579
|
+
},
|
|
580
|
+
}
|
|
581
|
+
)
|
|
582
|
+
result = result.model_copy(
|
|
583
|
+
update={
|
|
584
|
+
"updated_result": updated_result,
|
|
585
|
+
"coverage_ok": coverage_ok,
|
|
586
|
+
"satisfied": satisfied,
|
|
587
|
+
"strategy_used": current_layer_context.split_strategy,
|
|
588
|
+
"overlap_conflicts": overlap_conflicts,
|
|
589
|
+
"coverage_gap_notes": coverage_gaps,
|
|
590
|
+
"duplicate_child_notes": duplicate_notes,
|
|
591
|
+
"review_notes": merged_notes,
|
|
592
|
+
}
|
|
593
|
+
)
|
|
594
|
+
attempts = dict(parse_session.layer_attempts)
|
|
595
|
+
attempts[str(current_layer_context.depth)] = current_layer_context.retry_count + 1
|
|
596
|
+
return result, parse_session.model_copy(update={"layer_attempts": attempts})
|
|
597
|
+
|
|
598
|
+
|
|
599
|
+
def apply_cud_update(
|
|
600
|
+
*,
|
|
601
|
+
current_layer_result: CurrentLayerResult,
|
|
602
|
+
current_layer_review: CurrentLayerReview,
|
|
603
|
+
) -> CurrentLayerResult:
|
|
604
|
+
updated = current_layer_review.updated_result or current_layer_result
|
|
605
|
+
metadata = dict(updated.metadata)
|
|
606
|
+
metadata.update(current_layer_review.metadata)
|
|
607
|
+
metadata["split_strategy"] = current_layer_review.strategy_used
|
|
608
|
+
if current_layer_review.coverage_ok is not None:
|
|
609
|
+
metadata["layer_coverage_ok"] = current_layer_review.coverage_ok
|
|
610
|
+
if current_layer_review.review_notes:
|
|
611
|
+
metadata["review_notes"] = list(current_layer_review.review_notes)
|
|
612
|
+
if current_layer_review.overlap_conflicts:
|
|
613
|
+
metadata["overlap_conflicts"] = [
|
|
614
|
+
item.model_dump(field_mode="backend", dump_format="json")
|
|
615
|
+
for item in current_layer_review.overlap_conflicts
|
|
616
|
+
]
|
|
617
|
+
if current_layer_review.coverage_gap_notes:
|
|
618
|
+
metadata["coverage_gap_notes"] = [
|
|
619
|
+
item.model_dump(field_mode="backend", dump_format="json")
|
|
620
|
+
for item in current_layer_review.coverage_gap_notes
|
|
621
|
+
]
|
|
622
|
+
if current_layer_review.duplicate_child_notes:
|
|
623
|
+
metadata["duplicate_child_notes"] = [
|
|
624
|
+
item.model_dump(field_mode="backend", dump_format="json")
|
|
625
|
+
for item in current_layer_review.duplicate_child_notes
|
|
626
|
+
]
|
|
627
|
+
satisfied = (
|
|
628
|
+
current_layer_review.satisfied
|
|
629
|
+
if current_layer_review.satisfied is not None
|
|
630
|
+
else updated.satisfied
|
|
631
|
+
)
|
|
632
|
+
return updated.model_copy(
|
|
633
|
+
update={
|
|
634
|
+
"satisfied": satisfied,
|
|
635
|
+
"review_rounds": updated.review_rounds + 1,
|
|
636
|
+
"metadata": metadata,
|
|
637
|
+
}
|
|
638
|
+
)
|
|
639
|
+
|
|
640
|
+
|
|
641
|
+
def check_layer_coverage(
|
|
642
|
+
*,
|
|
643
|
+
current_layer_context: CurrentLayerContext,
|
|
644
|
+
current_layer_result: CurrentLayerResult,
|
|
645
|
+
current_layer_review: CurrentLayerReview | None = None,
|
|
646
|
+
) -> tuple[bool, list[str]]:
|
|
647
|
+
if current_layer_review is not None and current_layer_review.coverage_ok is not None:
|
|
648
|
+
notes = list(current_layer_review.review_notes)
|
|
649
|
+
notes.extend(
|
|
650
|
+
[
|
|
651
|
+
f"overlap conflict: {item.left_child_id} vs {item.right_child_id} @ {item.source_cluster_id}:{item.overlap_start}-{item.overlap_end}"
|
|
652
|
+
for item in current_layer_review.overlap_conflicts
|
|
653
|
+
]
|
|
654
|
+
)
|
|
655
|
+
notes.extend(
|
|
656
|
+
[
|
|
657
|
+
f"coverage gap: {item.parent_node_id} {item.source_cluster_id}:{item.gap_start}-{item.gap_end}"
|
|
658
|
+
for item in current_layer_review.coverage_gap_notes
|
|
659
|
+
]
|
|
660
|
+
)
|
|
661
|
+
notes.extend(
|
|
662
|
+
[
|
|
663
|
+
f"duplicate child: {item.child_node_id} duplicates {item.duplicate_of_child_node_id}"
|
|
664
|
+
for item in current_layer_review.duplicate_child_notes
|
|
665
|
+
]
|
|
666
|
+
)
|
|
667
|
+
return bool(current_layer_review.coverage_ok), notes
|
|
668
|
+
|
|
669
|
+
metadata_flag = current_layer_result.metadata.get("layer_coverage_ok")
|
|
670
|
+
if isinstance(metadata_flag, bool):
|
|
671
|
+
notes = current_layer_result.metadata.get("review_notes") or []
|
|
672
|
+
return metadata_flag, [str(note) for note in notes]
|
|
673
|
+
|
|
674
|
+
if current_layer_result.metadata.get("allow_empty_layer"):
|
|
675
|
+
return True, []
|
|
676
|
+
|
|
677
|
+
parent_ids = set(current_layer_context.parent_node_ids)
|
|
678
|
+
covered_parents = {
|
|
679
|
+
child.parent_node_id for child in current_layer_result.children if child.parent_node_id in parent_ids
|
|
680
|
+
}
|
|
681
|
+
missing = sorted(parent_ids - covered_parents)
|
|
682
|
+
if missing:
|
|
683
|
+
return False, [f"missing children for parent ids: {', '.join(missing)}"]
|
|
684
|
+
return True, []
|
|
685
|
+
|
|
686
|
+
|
|
687
|
+
def switch_split_strategy(
|
|
688
|
+
*,
|
|
689
|
+
parse_session: ParseSessionState,
|
|
690
|
+
current_layer_context: CurrentLayerContext,
|
|
691
|
+
) -> tuple[ParseSessionState, CurrentLayerContext]:
|
|
692
|
+
if current_layer_context.split_strategy == parse_session.fallback_split_strategy:
|
|
693
|
+
raise ValueError("fallback split strategy already exhausted")
|
|
694
|
+
next_strategy = parse_session.fallback_split_strategy
|
|
695
|
+
history = list(parse_session.strategy_history)
|
|
696
|
+
if not history or history[-1] != current_layer_context.split_strategy:
|
|
697
|
+
history.append(current_layer_context.split_strategy)
|
|
698
|
+
if history[-1] != next_strategy:
|
|
699
|
+
history.append(next_strategy)
|
|
700
|
+
updated_session = parse_session.model_copy(
|
|
701
|
+
update={
|
|
702
|
+
"split_strategy": next_strategy,
|
|
703
|
+
"strategy_history": history,
|
|
704
|
+
"strategy_switch_count": parse_session.strategy_switch_count + 1,
|
|
705
|
+
}
|
|
706
|
+
)
|
|
707
|
+
updated_context = current_layer_context.model_copy(
|
|
708
|
+
update={
|
|
709
|
+
"split_strategy": next_strategy,
|
|
710
|
+
"retry_count": 0,
|
|
711
|
+
"metadata": {
|
|
712
|
+
**current_layer_context.metadata,
|
|
713
|
+
"split_strategy_switch_from": current_layer_context.split_strategy,
|
|
714
|
+
"split_strategy_switch_to": next_strategy,
|
|
715
|
+
},
|
|
716
|
+
}
|
|
717
|
+
)
|
|
718
|
+
return updated_session, updated_context
|
|
719
|
+
|
|
720
|
+
|
|
721
|
+
def repair_layer_candidates(
|
|
722
|
+
*,
|
|
723
|
+
current_layer_result: CurrentLayerResult,
|
|
724
|
+
parser_source_map: dict[str, dict[str, Any]],
|
|
725
|
+
correct_pointer_fn: Callable[[HydratedTextPointer, dict[str, dict[str, Any]]], HydratedTextPointer | None],
|
|
726
|
+
) -> tuple[CurrentLayerResult, int]:
|
|
727
|
+
def _pointer_context(pointer: HydratedTextPointer) -> str:
|
|
728
|
+
source = parser_source_map.get(pointer.source_cluster_id, {})
|
|
729
|
+
text = str(source.get("text", ""))
|
|
730
|
+
preview = text.replace("\n", "\\n").replace("\t", "\\t")
|
|
731
|
+
if len(preview) > 120:
|
|
732
|
+
preview = preview[:117] + "..."
|
|
733
|
+
return (
|
|
734
|
+
f"source_cluster_id={pointer.source_cluster_id!r}, "
|
|
735
|
+
f"span=({pointer.start_char},{pointer.end_char}), "
|
|
736
|
+
f"verbatim_text={pointer.verbatim_text!r}, "
|
|
737
|
+
f"source_text={preview!r}"
|
|
738
|
+
)
|
|
739
|
+
|
|
740
|
+
repaired_children: list[LayerChildCandidate] = []
|
|
741
|
+
repaired_count = 0
|
|
742
|
+
for child in current_layer_result.children:
|
|
743
|
+
repaired_ptrs = []
|
|
744
|
+
for pointer in child.total_content_pointers:
|
|
745
|
+
fixed = correct_pointer_fn(pointer, parser_source_map)
|
|
746
|
+
if fixed is None:
|
|
747
|
+
message = (
|
|
748
|
+
f"unrecoverable pointer for child {child.title!r} "
|
|
749
|
+
f"(parent={child.parent_node_id!r}, node_id={child.node_id!r}); "
|
|
750
|
+
f"{_pointer_context(pointer)}"
|
|
751
|
+
)
|
|
752
|
+
_LOGGER.warning("repair_layer_candidates failed: %s", message)
|
|
753
|
+
raise ValueError(message)
|
|
754
|
+
if fixed.model_dump() != pointer.model_dump():
|
|
755
|
+
repaired_count += 1
|
|
756
|
+
repaired_ptrs.append(fixed)
|
|
757
|
+
repaired_children.append(
|
|
758
|
+
child.model_copy(update={"total_content_pointers": repaired_ptrs})
|
|
759
|
+
)
|
|
760
|
+
return current_layer_result.model_copy(update={"children": repaired_children}), repaired_count
|
|
761
|
+
|
|
762
|
+
|
|
763
|
+
def dedupe_and_filter_layer(
|
|
764
|
+
*,
|
|
765
|
+
current_layer_context: CurrentLayerContext,
|
|
766
|
+
current_layer_result: CurrentLayerResult,
|
|
767
|
+
) -> CurrentLayerResult:
|
|
768
|
+
parent_title_lookup = {
|
|
769
|
+
node_id: title for node_id, title in zip(current_layer_context.parent_node_ids, current_layer_context.parent_titles)
|
|
770
|
+
}
|
|
771
|
+
seen: set[tuple[str, str, str]] = set()
|
|
772
|
+
filtered: list[LayerChildCandidate] = []
|
|
773
|
+
for child in current_layer_result.children:
|
|
774
|
+
if child.parent_node_id not in parent_title_lookup:
|
|
775
|
+
continue
|
|
776
|
+
if child.title.strip() == parent_title_lookup[child.parent_node_id].strip():
|
|
777
|
+
continue
|
|
778
|
+
key = (child.parent_node_id, child.node_type, child.title.strip().lower())
|
|
779
|
+
if key in seen:
|
|
780
|
+
continue
|
|
781
|
+
seen.add(key)
|
|
782
|
+
filtered.append(child)
|
|
783
|
+
return current_layer_result.model_copy(update={"children": filtered})
|
|
784
|
+
|
|
785
|
+
|
|
786
|
+
def commit_layer_children(
|
|
787
|
+
*,
|
|
788
|
+
semantic_tree: SemanticNode,
|
|
789
|
+
current_layer_result: CurrentLayerResult,
|
|
790
|
+
current_depth: int,
|
|
791
|
+
) -> SemanticNode:
|
|
792
|
+
tree = SemanticNode.model_validate(semantic_tree.model_dump())
|
|
793
|
+
children_by_parent: dict[str, list[SemanticNode]] = {}
|
|
794
|
+
for child in current_layer_result.children:
|
|
795
|
+
children_by_parent.setdefault(child.parent_node_id, []).append(
|
|
796
|
+
SemanticNode(
|
|
797
|
+
node_id=child.node_id,
|
|
798
|
+
parent_id=child.parent_node_id,
|
|
799
|
+
node_type=child.node_type,
|
|
800
|
+
title=child.title,
|
|
801
|
+
total_content_pointers=list(child.total_content_pointers),
|
|
802
|
+
child_nodes=[],
|
|
803
|
+
level_from_root=current_depth + 1,
|
|
804
|
+
)
|
|
805
|
+
)
|
|
806
|
+
|
|
807
|
+
def walk(node: SemanticNode) -> SemanticNode:
|
|
808
|
+
updated_children = [walk(existing) for existing in node.child_nodes]
|
|
809
|
+
if str(node.node_id) in children_by_parent:
|
|
810
|
+
updated_children.extend(children_by_parent[str(node.node_id)])
|
|
811
|
+
payload = node.model_dump()
|
|
812
|
+
payload["child_nodes"] = [child.model_dump() for child in updated_children]
|
|
813
|
+
return SemanticNode.model_validate(payload)
|
|
814
|
+
|
|
815
|
+
return walk(tree)
|
|
816
|
+
|
|
817
|
+
|
|
818
|
+
def enqueue_next_layer_frontier(
|
|
819
|
+
*,
|
|
820
|
+
frontier_queue: list[LayerFrontierItem],
|
|
821
|
+
current_layer_context: CurrentLayerContext,
|
|
822
|
+
current_layer_result: CurrentLayerResult,
|
|
823
|
+
parse_session: ParseSessionState,
|
|
824
|
+
) -> list[LayerFrontierItem]:
|
|
825
|
+
queued = list(frontier_queue)
|
|
826
|
+
next_depth = current_layer_context.depth + 1
|
|
827
|
+
if next_depth >= parse_session.max_depth:
|
|
828
|
+
return queued
|
|
829
|
+
next_order = max([item.order for item in queued], default=-1) + 1
|
|
830
|
+
for child in current_layer_result.children:
|
|
831
|
+
if child.expandable:
|
|
832
|
+
queued.append(
|
|
833
|
+
LayerFrontierItem(
|
|
834
|
+
parent_node_id=child.node_id,
|
|
835
|
+
depth=next_depth,
|
|
836
|
+
order=next_order,
|
|
837
|
+
)
|
|
838
|
+
)
|
|
839
|
+
next_order += 1
|
|
840
|
+
return queued
|
|
841
|
+
|
|
842
|
+
|
|
843
|
+
def find_semantic_node(root: SemanticNode, node_id: str) -> SemanticNode | None:
|
|
844
|
+
if str(root.node_id) == str(node_id):
|
|
845
|
+
return root
|
|
846
|
+
for child in root.child_nodes:
|
|
847
|
+
found = find_semantic_node(child, node_id)
|
|
848
|
+
if found is not None:
|
|
849
|
+
return found
|
|
850
|
+
return None
|
|
851
|
+
|
|
852
|
+
|
|
853
|
+
def finalize_semantic_tree(semantic_tree: SemanticNode) -> SemanticNode:
|
|
854
|
+
return semantic_tree
|
|
855
|
+
|
|
856
|
+
|
|
857
|
+
def _coerce_layer_child(value: Any) -> LayerChildCandidate:
|
|
858
|
+
if isinstance(value, LayerChildCandidate):
|
|
859
|
+
return value
|
|
860
|
+
if isinstance(value, dict):
|
|
861
|
+
return LayerChildCandidate.model_validate(value)
|
|
862
|
+
raise TypeError("unsupported layer child candidate")
|