graph-knowledge-doc-parser 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graph_knowledge_doc_parser-0.1.0.dist-info/METADATA +326 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/RECORD +38 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/WHEEL +4 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/entry_points.txt +3 -0
- kg_doc_parser/__init__.py +9 -0
- kg_doc_parser/cast_hinting.py +19 -0
- kg_doc_parser/document_ingester_logger.py +766 -0
- kg_doc_parser/models.py +277 -0
- kg_doc_parser/ocr.py +752 -0
- kg_doc_parser/pdf2png.py +286 -0
- kg_doc_parser/semantic_document_splitting_layerwise_edits.py +3302 -0
- kg_doc_parser/text_processing_utils.py +30 -0
- kg_doc_parser/utils/__init__.py +0 -0
- kg_doc_parser/utils/bounded_threadpool_executor.py +37 -0
- kg_doc_parser/utils/file_loaders.py +405 -0
- kg_doc_parser/utils/langchain.py +220 -0
- kg_doc_parser/utils/log.py +135 -0
- kg_doc_parser/utils/version_chaining.py +1278 -0
- kg_doc_parser/workflow_ingest/__init__.py +187 -0
- kg_doc_parser/workflow_ingest/_kogwistar.py +13 -0
- kg_doc_parser/workflow_ingest/adapters.py +212 -0
- kg_doc_parser/workflow_ingest/cache.py +63 -0
- kg_doc_parser/workflow_ingest/cli.py +324 -0
- kg_doc_parser/workflow_ingest/clients.py +444 -0
- kg_doc_parser/workflow_ingest/demo_harness.py +427 -0
- kg_doc_parser/workflow_ingest/design.py +208 -0
- kg_doc_parser/workflow_ingest/handlers.py +617 -0
- kg_doc_parser/workflow_ingest/models.py +575 -0
- kg_doc_parser/workflow_ingest/ocr_pipeline.py +1581 -0
- kg_doc_parser/workflow_ingest/page_index.py +473 -0
- kg_doc_parser/workflow_ingest/parser_core.py +862 -0
- kg_doc_parser/workflow_ingest/parsing.py +249 -0
- kg_doc_parser/workflow_ingest/probe.py +164 -0
- kg_doc_parser/workflow_ingest/providers.py +412 -0
- kg_doc_parser/workflow_ingest/runners.py +546 -0
- kg_doc_parser/workflow_ingest/semantics.py +231 -0
- kg_doc_parser/workflow_ingest/service.py +112 -0
- kg_doc_parser/workflow_ingest/smoke_assets.py +62 -0
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
from .adapters import (
|
|
2
|
+
build_authoritative_source_map,
|
|
3
|
+
build_parser_input_dict,
|
|
4
|
+
build_parser_source_map,
|
|
5
|
+
normalize_ocr_pages,
|
|
6
|
+
)
|
|
7
|
+
from .clients import (
|
|
8
|
+
CanonicalGraphPersistenceClient,
|
|
9
|
+
DocumentTreeApiPersistenceClient,
|
|
10
|
+
DirectRuntimeIngestClient,
|
|
11
|
+
IngestExecutionClient,
|
|
12
|
+
ServerCanonicalKgClient,
|
|
13
|
+
UnsupportedClientOperation,
|
|
14
|
+
)
|
|
15
|
+
from .cache import WorkflowLLMCallCache
|
|
16
|
+
from .design import DEFAULT_WORKFLOW_ID, build_ingest_workflow_design, ensure_ingest_workflow_design
|
|
17
|
+
from .demo_harness import DemoHarnessArtifacts, DemoHarnessConfig, run_demo_harness
|
|
18
|
+
from .handlers import build_ingest_step_resolver
|
|
19
|
+
from .parsing import (
|
|
20
|
+
OCRParseRequest,
|
|
21
|
+
PageIndexParseRequest,
|
|
22
|
+
ParseMode,
|
|
23
|
+
TreeParseRequest,
|
|
24
|
+
parse_document,
|
|
25
|
+
parse_ocr_document,
|
|
26
|
+
parse_page_index_document,
|
|
27
|
+
parse_tree_document,
|
|
28
|
+
)
|
|
29
|
+
from .ocr_pipeline import (
|
|
30
|
+
OCRImagePayload,
|
|
31
|
+
OCRWorkflowArtifacts,
|
|
32
|
+
OCRWorkflowStateStore,
|
|
33
|
+
prepare_ocr_workflow_input,
|
|
34
|
+
run_ocr_ingest_workflow,
|
|
35
|
+
)
|
|
36
|
+
from .runners import (
|
|
37
|
+
LayerwiseWorkflowCommandResult,
|
|
38
|
+
OcrWorkflowCommandResult,
|
|
39
|
+
PageIndexWorkflowCommandResult,
|
|
40
|
+
WorkflowCommandResult,
|
|
41
|
+
discover_input_files,
|
|
42
|
+
run_demo_harness_workflow,
|
|
43
|
+
run_layerwise_batch_workflow,
|
|
44
|
+
run_layerwise_source_workflow,
|
|
45
|
+
run_ocr_batch_workflow,
|
|
46
|
+
run_ocr_source_workflow,
|
|
47
|
+
run_page_index_batch_workflow,
|
|
48
|
+
run_page_index_source_workflow,
|
|
49
|
+
)
|
|
50
|
+
from .smoke_assets import generate_ocr_smoke_assets
|
|
51
|
+
from .page_index import PageIndexBlockSpec, PageIndexParseResult, build_page_index_workflow_input
|
|
52
|
+
from .models import (
|
|
53
|
+
BoundingBox,
|
|
54
|
+
CanonicalGraphWriteResult,
|
|
55
|
+
CurrentLayerContext,
|
|
56
|
+
CurrentLayerReview,
|
|
57
|
+
CurrentLayerResult,
|
|
58
|
+
LayerCoverageGap,
|
|
59
|
+
LayerDuplicateChildNote,
|
|
60
|
+
GroundedSourceRecord,
|
|
61
|
+
IngestRunHandle,
|
|
62
|
+
IngestRunResult,
|
|
63
|
+
LayerChildCandidate,
|
|
64
|
+
LayerFrontierItem,
|
|
65
|
+
LayerSpanConflict,
|
|
66
|
+
NormalizedPage,
|
|
67
|
+
NormalizedSourceCollection,
|
|
68
|
+
ParseSessionState,
|
|
69
|
+
SourceUnit,
|
|
70
|
+
ValidationReport,
|
|
71
|
+
WorkflowExportBundle,
|
|
72
|
+
WorkflowIngestInput,
|
|
73
|
+
)
|
|
74
|
+
from .providers import (
|
|
75
|
+
EmbeddingProviderConfig,
|
|
76
|
+
FakeChatModel,
|
|
77
|
+
ProviderEndpointConfig,
|
|
78
|
+
WorkflowProviderSettings,
|
|
79
|
+
build_chat_model,
|
|
80
|
+
build_chat_model_for_role,
|
|
81
|
+
build_embedding_function,
|
|
82
|
+
)
|
|
83
|
+
from .probe import WorkflowProbe, emit_probe_event
|
|
84
|
+
from .parser_core import (
|
|
85
|
+
apply_cud_update,
|
|
86
|
+
check_layer_coverage,
|
|
87
|
+
default_parse_semantic_fn,
|
|
88
|
+
finalize_semantic_tree,
|
|
89
|
+
initialize_parse_session,
|
|
90
|
+
prepare_layer_frontier,
|
|
91
|
+
propose_layer_breakdown,
|
|
92
|
+
review_layer,
|
|
93
|
+
)
|
|
94
|
+
from .service import build_default_engines, build_runtime, run_ingest_workflow
|
|
95
|
+
from .semantics import HydratedTextPointer, SemanticNode
|
|
96
|
+
|
|
97
|
+
__all__ = [
|
|
98
|
+
"BoundingBox",
|
|
99
|
+
"CanonicalGraphPersistenceClient",
|
|
100
|
+
"CanonicalGraphWriteResult",
|
|
101
|
+
"CurrentLayerContext",
|
|
102
|
+
"CurrentLayerReview",
|
|
103
|
+
"CurrentLayerResult",
|
|
104
|
+
"LayerCoverageGap",
|
|
105
|
+
"LayerDuplicateChildNote",
|
|
106
|
+
"DemoHarnessArtifacts",
|
|
107
|
+
"DemoHarnessConfig",
|
|
108
|
+
"DocumentTreeApiPersistenceClient",
|
|
109
|
+
"DirectRuntimeIngestClient",
|
|
110
|
+
"EmbeddingProviderConfig",
|
|
111
|
+
"GroundedSourceRecord",
|
|
112
|
+
"HydratedTextPointer",
|
|
113
|
+
"FakeChatModel",
|
|
114
|
+
"IngestExecutionClient",
|
|
115
|
+
"IngestRunHandle",
|
|
116
|
+
"IngestRunResult",
|
|
117
|
+
"LayerChildCandidate",
|
|
118
|
+
"LayerFrontierItem",
|
|
119
|
+
"LayerSpanConflict",
|
|
120
|
+
"NormalizedPage",
|
|
121
|
+
"NormalizedSourceCollection",
|
|
122
|
+
"OCRImagePayload",
|
|
123
|
+
"OCRWorkflowArtifacts",
|
|
124
|
+
"OCRWorkflowStateStore",
|
|
125
|
+
"OCRParseRequest",
|
|
126
|
+
"ParseSessionState",
|
|
127
|
+
"ProviderEndpointConfig",
|
|
128
|
+
"PageIndexParseRequest",
|
|
129
|
+
"PageIndexBlockSpec",
|
|
130
|
+
"PageIndexParseResult",
|
|
131
|
+
"ParseMode",
|
|
132
|
+
"TreeParseRequest",
|
|
133
|
+
"SemanticNode",
|
|
134
|
+
"ServerCanonicalKgClient",
|
|
135
|
+
"SourceUnit",
|
|
136
|
+
"UnsupportedClientOperation",
|
|
137
|
+
"ValidationReport",
|
|
138
|
+
"WorkflowLLMCallCache",
|
|
139
|
+
"WorkflowProbe",
|
|
140
|
+
"WorkflowProviderSettings",
|
|
141
|
+
"WorkflowExportBundle",
|
|
142
|
+
"WorkflowIngestInput",
|
|
143
|
+
"WorkflowCommandResult",
|
|
144
|
+
"OcrWorkflowCommandResult",
|
|
145
|
+
"PageIndexWorkflowCommandResult",
|
|
146
|
+
"LayerwiseWorkflowCommandResult",
|
|
147
|
+
"DEFAULT_WORKFLOW_ID",
|
|
148
|
+
"build_chat_model",
|
|
149
|
+
"build_chat_model_for_role",
|
|
150
|
+
"build_embedding_function",
|
|
151
|
+
"apply_cud_update",
|
|
152
|
+
"check_layer_coverage",
|
|
153
|
+
"default_parse_semantic_fn",
|
|
154
|
+
"finalize_semantic_tree",
|
|
155
|
+
"initialize_parse_session",
|
|
156
|
+
"prepare_layer_frontier",
|
|
157
|
+
"propose_layer_breakdown",
|
|
158
|
+
"review_layer",
|
|
159
|
+
"build_authoritative_source_map",
|
|
160
|
+
"build_default_engines",
|
|
161
|
+
"build_ingest_workflow_design",
|
|
162
|
+
"build_ingest_step_resolver",
|
|
163
|
+
"build_parser_input_dict",
|
|
164
|
+
"build_page_index_workflow_input",
|
|
165
|
+
"build_parser_source_map",
|
|
166
|
+
"build_runtime",
|
|
167
|
+
"ensure_ingest_workflow_design",
|
|
168
|
+
"emit_probe_event",
|
|
169
|
+
"normalize_ocr_pages",
|
|
170
|
+
"parse_document",
|
|
171
|
+
"parse_ocr_document",
|
|
172
|
+
"parse_page_index_document",
|
|
173
|
+
"parse_tree_document",
|
|
174
|
+
"prepare_ocr_workflow_input",
|
|
175
|
+
"run_demo_harness",
|
|
176
|
+
"run_demo_harness_workflow",
|
|
177
|
+
"run_ingest_workflow",
|
|
178
|
+
"discover_input_files",
|
|
179
|
+
"run_layerwise_batch_workflow",
|
|
180
|
+
"run_layerwise_source_workflow",
|
|
181
|
+
"run_ocr_batch_workflow",
|
|
182
|
+
"run_ocr_source_workflow",
|
|
183
|
+
"run_page_index_batch_workflow",
|
|
184
|
+
"run_page_index_source_workflow",
|
|
185
|
+
"generate_ocr_smoke_assets",
|
|
186
|
+
"run_ocr_ingest_workflow",
|
|
187
|
+
]
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from importlib import import_module
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def ensure_kogwistar_on_path() -> None:
|
|
7
|
+
"""Compatibility shim: require a normal kogwistar installation/import.
|
|
8
|
+
|
|
9
|
+
We intentionally do not mutate ``sys.path`` here. The repo may contain a sibling
|
|
10
|
+
checkout for inspection, but runtime imports should resolve through the installed
|
|
11
|
+
package environment, including editable installs.
|
|
12
|
+
"""
|
|
13
|
+
import_module("kogwistar")
|
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from typing import Any, TypedDict
|
|
4
|
+
|
|
5
|
+
from .models import (
|
|
6
|
+
BoundingBox,
|
|
7
|
+
GroundedSourceRecord,
|
|
8
|
+
NormalizedPage,
|
|
9
|
+
NormalizedSourceCollection,
|
|
10
|
+
SourceUnit,
|
|
11
|
+
WorkflowIngestInput,
|
|
12
|
+
)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class OCRPageJSON(TypedDict, total=False):
|
|
16
|
+
"""Serialized OCR page payload produced by the OCR preparation layer."""
|
|
17
|
+
|
|
18
|
+
pdf_page_num: int
|
|
19
|
+
printed_page_number: str
|
|
20
|
+
contains_table: bool
|
|
21
|
+
OCR_text_clusters: list[dict[str, Any]]
|
|
22
|
+
non_text_objects: list[dict[str, Any]]
|
|
23
|
+
text: str
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def normalize_ocr_pages(
|
|
27
|
+
*,
|
|
28
|
+
document_id: str,
|
|
29
|
+
title: str,
|
|
30
|
+
pages: list[OCRPageJSON],
|
|
31
|
+
) -> WorkflowIngestInput:
|
|
32
|
+
"""Normalize raw OCR JSON into workflow ingest models.
|
|
33
|
+
|
|
34
|
+
OCR text clusters and non-text regions are both preserved with page and
|
|
35
|
+
cluster metadata. `embedding_space="image"` is an intent label here; it
|
|
36
|
+
does not imply a separate image embedder is already wired at runtime.
|
|
37
|
+
"""
|
|
38
|
+
normalized_pages: list[NormalizedPage] = []
|
|
39
|
+
for raw_page in pages:
|
|
40
|
+
page_number = int(raw_page["pdf_page_num"])
|
|
41
|
+
units: list[SourceUnit] = []
|
|
42
|
+
for cluster in raw_page.get("OCR_text_clusters", []):
|
|
43
|
+
bbox = BoundingBox(
|
|
44
|
+
y_min=float(cluster.get("bb_y_min", 0.0)),
|
|
45
|
+
x_min=float(cluster.get("bb_x_min", 0.0)),
|
|
46
|
+
y_max=float(cluster.get("bb_y_max", 0.0)),
|
|
47
|
+
x_max=float(cluster.get("bb_x_max", 0.0)),
|
|
48
|
+
)
|
|
49
|
+
units.append(
|
|
50
|
+
SourceUnit(
|
|
51
|
+
modality="ocr_text",
|
|
52
|
+
page_number=page_number,
|
|
53
|
+
cluster_number=cluster.get("cluster_number"),
|
|
54
|
+
text=cluster.get("text"),
|
|
55
|
+
bbox=bbox,
|
|
56
|
+
metadata={},
|
|
57
|
+
)
|
|
58
|
+
)
|
|
59
|
+
for obj in raw_page.get("non_text_objects", []):
|
|
60
|
+
bbox = BoundingBox(
|
|
61
|
+
y_min=float(obj.get("bb_y_min", 0.0)),
|
|
62
|
+
x_min=float(obj.get("bb_x_min", 0.0)),
|
|
63
|
+
y_max=float(obj.get("bb_y_max", 0.0)),
|
|
64
|
+
x_max=float(obj.get("bb_x_max", 0.0)),
|
|
65
|
+
)
|
|
66
|
+
units.append(
|
|
67
|
+
SourceUnit(
|
|
68
|
+
modality="image_region",
|
|
69
|
+
page_number=page_number,
|
|
70
|
+
cluster_number=obj.get("cluster_number"),
|
|
71
|
+
description=obj.get("description"),
|
|
72
|
+
bbox=bbox,
|
|
73
|
+
# Keep the "image" space label for future routing; the
|
|
74
|
+
# current engine still embeds through a single function.
|
|
75
|
+
embedding_space="image",
|
|
76
|
+
metadata={"participates_in_semantic_text": False},
|
|
77
|
+
)
|
|
78
|
+
)
|
|
79
|
+
normalized_pages.append(
|
|
80
|
+
NormalizedPage(
|
|
81
|
+
page_number=page_number,
|
|
82
|
+
units=units,
|
|
83
|
+
metadata={
|
|
84
|
+
"printed_page_number": raw_page.get("printed_page_number"),
|
|
85
|
+
"contains_table": raw_page.get("contains_table"),
|
|
86
|
+
},
|
|
87
|
+
)
|
|
88
|
+
)
|
|
89
|
+
embedding_spaces = ["default_text"]
|
|
90
|
+
if any(
|
|
91
|
+
unit.embedding_space == "image"
|
|
92
|
+
for page in normalized_pages
|
|
93
|
+
for unit in page.units
|
|
94
|
+
):
|
|
95
|
+
embedding_spaces.append("image")
|
|
96
|
+
return WorkflowIngestInput(
|
|
97
|
+
request_id=document_id,
|
|
98
|
+
collections=[
|
|
99
|
+
NormalizedSourceCollection(
|
|
100
|
+
collection_id=document_id,
|
|
101
|
+
title=title,
|
|
102
|
+
modality="ocr",
|
|
103
|
+
pages=normalized_pages,
|
|
104
|
+
embedding_spaces=embedding_spaces,
|
|
105
|
+
)
|
|
106
|
+
],
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def build_authoritative_source_map(
|
|
111
|
+
inp: WorkflowIngestInput,
|
|
112
|
+
) -> dict[str, GroundedSourceRecord]:
|
|
113
|
+
source_map: dict[str, GroundedSourceRecord] = {}
|
|
114
|
+
for collection in inp.collections:
|
|
115
|
+
for page in collection.pages:
|
|
116
|
+
ordinal_by_modality: dict[str, int] = {}
|
|
117
|
+
for unit in page.units:
|
|
118
|
+
modality_prefix = "t" if unit.modality in {"text", "ocr_text"} else "i"
|
|
119
|
+
original_cluster = unit.cluster_number
|
|
120
|
+
if original_cluster is None:
|
|
121
|
+
original_cluster = ordinal_by_modality.get(modality_prefix, 0)
|
|
122
|
+
ordinal_by_modality[modality_prefix] = ordinal_by_modality.get(modality_prefix, 0) + 1
|
|
123
|
+
unit_id = unit.unit_id or f"{collection.collection_id}|p{page.page_number}_{modality_prefix}{original_cluster}"
|
|
124
|
+
record = GroundedSourceRecord(
|
|
125
|
+
unit_id=unit_id,
|
|
126
|
+
collection_id=collection.collection_id,
|
|
127
|
+
modality=unit.modality,
|
|
128
|
+
page_number=page.page_number,
|
|
129
|
+
cluster_number=unit.cluster_number,
|
|
130
|
+
text=unit.text or unit.description or "",
|
|
131
|
+
parser_text=unit.parser_text,
|
|
132
|
+
source_uri=unit.source_uri,
|
|
133
|
+
embedding_space=unit.embedding_space,
|
|
134
|
+
participates_in_semantic_text=unit.modality in {"text", "ocr_text"},
|
|
135
|
+
bbox=unit.bbox,
|
|
136
|
+
metadata={
|
|
137
|
+
**unit.metadata,
|
|
138
|
+
"original_cluster_number": unit.cluster_number,
|
|
139
|
+
},
|
|
140
|
+
)
|
|
141
|
+
if unit_id in source_map:
|
|
142
|
+
raise ValueError(f"source map collision detected for {unit_id}")
|
|
143
|
+
source_map[unit_id] = record
|
|
144
|
+
return source_map
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def select_primary_collection(inp: WorkflowIngestInput) -> NormalizedSourceCollection:
|
|
148
|
+
return inp.collections[0]
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def build_parser_input_dict(
|
|
152
|
+
collection: NormalizedSourceCollection,
|
|
153
|
+
) -> dict[str, Any]:
|
|
154
|
+
pages: list[dict[str, Any]] = []
|
|
155
|
+
for page in collection.pages:
|
|
156
|
+
text_clusters = []
|
|
157
|
+
non_text_objects = []
|
|
158
|
+
next_cluster = 0
|
|
159
|
+
for unit in page.units:
|
|
160
|
+
bbox = unit.bbox
|
|
161
|
+
cluster_number = unit.cluster_number if unit.cluster_number is not None else next_cluster
|
|
162
|
+
if unit.modality in {"text", "ocr_text"}:
|
|
163
|
+
text_clusters.append(
|
|
164
|
+
{
|
|
165
|
+
"text": unit.text or "",
|
|
166
|
+
"bb_x_min": bbox.x_min if bbox else 0.0,
|
|
167
|
+
"bb_x_max": bbox.x_max if bbox else 0.0,
|
|
168
|
+
"bb_y_min": bbox.y_min if bbox else 0.0,
|
|
169
|
+
"bb_y_max": bbox.y_max if bbox else 0.0,
|
|
170
|
+
"cluster_number": cluster_number,
|
|
171
|
+
}
|
|
172
|
+
)
|
|
173
|
+
else:
|
|
174
|
+
non_text_objects.append(
|
|
175
|
+
{
|
|
176
|
+
"description": unit.description or unit.source_uri or "image-region",
|
|
177
|
+
"bb_x_min": bbox.x_min if bbox else 0.0,
|
|
178
|
+
"bb_x_max": bbox.x_max if bbox else 0.0,
|
|
179
|
+
"bb_y_min": bbox.y_min if bbox else 0.0,
|
|
180
|
+
"bb_y_max": bbox.y_max if bbox else 0.0,
|
|
181
|
+
"cluster_number": cluster_number,
|
|
182
|
+
}
|
|
183
|
+
)
|
|
184
|
+
next_cluster += 1
|
|
185
|
+
pages.append(
|
|
186
|
+
{
|
|
187
|
+
"pdf_page_num": page.page_number,
|
|
188
|
+
"printed_page_number": page.metadata.get("printed_page_number"),
|
|
189
|
+
"contains_table": page.metadata.get("contains_table", False),
|
|
190
|
+
"OCR_text_clusters": text_clusters,
|
|
191
|
+
"non_text_objects": non_text_objects,
|
|
192
|
+
}
|
|
193
|
+
)
|
|
194
|
+
return {"document_filename": collection.title, "pages": pages}
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def build_parser_source_map(
|
|
198
|
+
source_map: dict[str, GroundedSourceRecord],
|
|
199
|
+
) -> dict[str, dict[str, Any]]:
|
|
200
|
+
return {
|
|
201
|
+
unit_id: {
|
|
202
|
+
"id": record.unit_id,
|
|
203
|
+
"text": record.parser_text,
|
|
204
|
+
"modality": record.modality,
|
|
205
|
+
"page_number": record.page_number,
|
|
206
|
+
"cluster_number": record.cluster_number,
|
|
207
|
+
"embedding_space": record.embedding_space,
|
|
208
|
+
"participates_in_semantic_text": record.participates_in_semantic_text,
|
|
209
|
+
"metadata": record.metadata,
|
|
210
|
+
}
|
|
211
|
+
for unit_id, record in source_map.items()
|
|
212
|
+
}
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Any, Callable, TypeVar
|
|
6
|
+
|
|
7
|
+
from kogwistar.id_provider import stable_id
|
|
8
|
+
|
|
9
|
+
from .probe import emit_probe_event
|
|
10
|
+
|
|
11
|
+
T = TypeVar("T")
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _jsonable(value: Any) -> Any:
|
|
15
|
+
if hasattr(value, "model_dump"):
|
|
16
|
+
try:
|
|
17
|
+
return value.model_dump(field_mode="backend", dump_format="json")
|
|
18
|
+
except TypeError:
|
|
19
|
+
return value.model_dump()
|
|
20
|
+
if isinstance(value, dict):
|
|
21
|
+
return {str(k): _jsonable(v) for k, v in value.items()}
|
|
22
|
+
if isinstance(value, (list, tuple)):
|
|
23
|
+
return [_jsonable(v) for v in value]
|
|
24
|
+
return value
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class WorkflowLLMCallCache:
|
|
28
|
+
def __init__(self, cache_dir: str | Path, *, probe=None) -> None:
|
|
29
|
+
self.cache_dir = Path(cache_dir)
|
|
30
|
+
self.cache_dir.mkdir(parents=True, exist_ok=True)
|
|
31
|
+
self.probe = probe
|
|
32
|
+
|
|
33
|
+
def _cache_path(self, operation: str, fingerprint: dict[str, Any]) -> Path:
|
|
34
|
+
payload = json.dumps(_jsonable(fingerprint), sort_keys=True, ensure_ascii=False, separators=(",", ":"))
|
|
35
|
+
cache_id = stable_id("workflow_ingest.llm_call", operation, payload)
|
|
36
|
+
return self.cache_dir / f"{cache_id}.json"
|
|
37
|
+
|
|
38
|
+
def cached_call(
|
|
39
|
+
self,
|
|
40
|
+
*,
|
|
41
|
+
operation: str,
|
|
42
|
+
fingerprint: dict[str, Any],
|
|
43
|
+
fn: Callable[[], T],
|
|
44
|
+
) -> Any:
|
|
45
|
+
path = self._cache_path(operation, fingerprint)
|
|
46
|
+
if path.exists():
|
|
47
|
+
emit_probe_event(
|
|
48
|
+
self.probe,
|
|
49
|
+
"workflow.llm_cache_hit",
|
|
50
|
+
operation=operation,
|
|
51
|
+
cache_path=str(path),
|
|
52
|
+
)
|
|
53
|
+
return json.loads(path.read_text(encoding="utf-8"))
|
|
54
|
+
result = fn()
|
|
55
|
+
payload = _jsonable(result)
|
|
56
|
+
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
|
|
57
|
+
emit_probe_event(
|
|
58
|
+
self.probe,
|
|
59
|
+
"workflow.llm_cache_miss",
|
|
60
|
+
operation=operation,
|
|
61
|
+
cache_path=str(path),
|
|
62
|
+
)
|
|
63
|
+
return payload
|