graph-knowledge-doc-parser 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graph_knowledge_doc_parser-0.1.0.dist-info/METADATA +326 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/RECORD +38 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/WHEEL +4 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/entry_points.txt +3 -0
- kg_doc_parser/__init__.py +9 -0
- kg_doc_parser/cast_hinting.py +19 -0
- kg_doc_parser/document_ingester_logger.py +766 -0
- kg_doc_parser/models.py +277 -0
- kg_doc_parser/ocr.py +752 -0
- kg_doc_parser/pdf2png.py +286 -0
- kg_doc_parser/semantic_document_splitting_layerwise_edits.py +3302 -0
- kg_doc_parser/text_processing_utils.py +30 -0
- kg_doc_parser/utils/__init__.py +0 -0
- kg_doc_parser/utils/bounded_threadpool_executor.py +37 -0
- kg_doc_parser/utils/file_loaders.py +405 -0
- kg_doc_parser/utils/langchain.py +220 -0
- kg_doc_parser/utils/log.py +135 -0
- kg_doc_parser/utils/version_chaining.py +1278 -0
- kg_doc_parser/workflow_ingest/__init__.py +187 -0
- kg_doc_parser/workflow_ingest/_kogwistar.py +13 -0
- kg_doc_parser/workflow_ingest/adapters.py +212 -0
- kg_doc_parser/workflow_ingest/cache.py +63 -0
- kg_doc_parser/workflow_ingest/cli.py +324 -0
- kg_doc_parser/workflow_ingest/clients.py +444 -0
- kg_doc_parser/workflow_ingest/demo_harness.py +427 -0
- kg_doc_parser/workflow_ingest/design.py +208 -0
- kg_doc_parser/workflow_ingest/handlers.py +617 -0
- kg_doc_parser/workflow_ingest/models.py +575 -0
- kg_doc_parser/workflow_ingest/ocr_pipeline.py +1581 -0
- kg_doc_parser/workflow_ingest/page_index.py +473 -0
- kg_doc_parser/workflow_ingest/parser_core.py +862 -0
- kg_doc_parser/workflow_ingest/parsing.py +249 -0
- kg_doc_parser/workflow_ingest/probe.py +164 -0
- kg_doc_parser/workflow_ingest/providers.py +412 -0
- kg_doc_parser/workflow_ingest/runners.py +546 -0
- kg_doc_parser/workflow_ingest/semantics.py +231 -0
- kg_doc_parser/workflow_ingest/service.py +112 -0
- kg_doc_parser/workflow_ingest/smoke_assets.py +62 -0
|
@@ -0,0 +1,427 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
"""Demo harness for running workflow ingest in a controlled end-to-end setup.
|
|
4
|
+
|
|
5
|
+
The harness is intentionally explicit about its moving parts:
|
|
6
|
+
- it builds a fake or legacy parser input path
|
|
7
|
+
- it starts either an in-process or subprocess server
|
|
8
|
+
- it wires a workflow client through a canonical graph persistence adapter
|
|
9
|
+
- it emits probe events and summary artifacts so the run can be inspected later
|
|
10
|
+
|
|
11
|
+
This file is closer to a reproducible scenario runner than a minimal test helper,
|
|
12
|
+
so the extra structure helps readers understand where the runtime boundaries are.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
import importlib
|
|
16
|
+
import json
|
|
17
|
+
import os
|
|
18
|
+
import socket
|
|
19
|
+
import subprocess
|
|
20
|
+
import sys
|
|
21
|
+
import threading
|
|
22
|
+
import time
|
|
23
|
+
from contextlib import AbstractContextManager
|
|
24
|
+
from dataclasses import dataclass, field
|
|
25
|
+
from pathlib import Path
|
|
26
|
+
from typing import Any, Literal
|
|
27
|
+
|
|
28
|
+
from .cache import WorkflowLLMCallCache
|
|
29
|
+
from .clients import DocumentTreeApiPersistenceClient, ServerCanonicalKgClient
|
|
30
|
+
from .models import CurrentLayerResult, CurrentLayerReview, LayerChildCandidate, WorkflowIngestInput
|
|
31
|
+
from .providers import WorkflowProviderSettings
|
|
32
|
+
from .probe import WorkflowProbe, emit_probe_event
|
|
33
|
+
from .semantics import HydratedTextPointer
|
|
34
|
+
from .service import build_default_engines
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@dataclass
|
|
38
|
+
class DemoHarnessConfig:
|
|
39
|
+
"""Configuration knobs for a demo ingest run."""
|
|
40
|
+
|
|
41
|
+
output_dir: Path
|
|
42
|
+
document_id: str = "demo-doc"
|
|
43
|
+
title: str = "Workflow Ingest Demo"
|
|
44
|
+
text: str = "Alpha clause\nBeta clause\nGamma clause"
|
|
45
|
+
workflow_id: str = "kg_doc_parser.ingest.v1"
|
|
46
|
+
parser_mode: Literal["fake_layered", "legacy_cached"] = "fake_layered"
|
|
47
|
+
server_mode: Literal["testclient", "subprocess_http", "external_http"] = "testclient"
|
|
48
|
+
external_base_url: str | None = None
|
|
49
|
+
backend_factory: Any | None = None
|
|
50
|
+
provider_settings: WorkflowProviderSettings | None = None
|
|
51
|
+
enable_sys_monitoring: bool = True
|
|
52
|
+
probe_filename: str = "probe-events.jsonl"
|
|
53
|
+
summary_filename: str = "demo-summary.json"
|
|
54
|
+
cache_dirname: str = "llm-cache"
|
|
55
|
+
deps: dict[str, Any] = field(default_factory=dict)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@dataclass
|
|
59
|
+
class DemoHarnessArtifacts:
|
|
60
|
+
"""Outputs captured from a demo ingest run."""
|
|
61
|
+
|
|
62
|
+
output_dir: Path
|
|
63
|
+
probe_path: Path
|
|
64
|
+
summary_path: Path
|
|
65
|
+
cache_dir: Path
|
|
66
|
+
engine_dir: Path
|
|
67
|
+
server_data_dir: Path
|
|
68
|
+
run_id: str | None = None
|
|
69
|
+
status: str | None = None
|
|
70
|
+
canonical_write_confirmed: bool = False
|
|
71
|
+
persistence_mode: str | None = None
|
|
72
|
+
kg_authority: str | None = None
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
class _ServerContext(AbstractContextManager):
|
|
76
|
+
"""Wrapper for server transports so shutdown behavior stays explicit."""
|
|
77
|
+
|
|
78
|
+
def __init__(self, *, client: Any, transport: str, base_url: str = "", cleanup=None) -> None:
|
|
79
|
+
self.client = client
|
|
80
|
+
self.transport = transport
|
|
81
|
+
self.base_url = base_url
|
|
82
|
+
self._cleanup = cleanup
|
|
83
|
+
|
|
84
|
+
def __exit__(self, exc_type, exc, tb):
|
|
85
|
+
if self._cleanup is not None:
|
|
86
|
+
self._cleanup(exc_type, exc, tb)
|
|
87
|
+
return False
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _load_isolated_server_app(server_data_dir: Path) -> _ServerContext:
|
|
91
|
+
"""Start an isolated in-process server for demo runs."""
|
|
92
|
+
from fastapi.testclient import TestClient
|
|
93
|
+
|
|
94
|
+
os.environ["GKE_BACKEND"] = "chroma"
|
|
95
|
+
os.environ["GKE_PERSIST_DIRECTORY"] = str(server_data_dir)
|
|
96
|
+
os.environ["AUTH_MODE"] = "dev"
|
|
97
|
+
os.environ["ANONYMIZED_TELEMETRY"] = "FALSE"
|
|
98
|
+
os.environ["KOGWISTAR_LOG_LEVEL"] = "WARNING"
|
|
99
|
+
for module_name in (
|
|
100
|
+
"kogwistar.server.resources",
|
|
101
|
+
"kogwistar.server.bootstrap",
|
|
102
|
+
"kogwistar.server_mcp_with_admin",
|
|
103
|
+
):
|
|
104
|
+
sys.modules.pop(module_name, None)
|
|
105
|
+
server_module = importlib.import_module("kogwistar.server_mcp_with_admin")
|
|
106
|
+
client = TestClient(server_module.app)
|
|
107
|
+
client.__enter__()
|
|
108
|
+
return _ServerContext(
|
|
109
|
+
client=client,
|
|
110
|
+
transport="fastapi_testclient",
|
|
111
|
+
base_url="",
|
|
112
|
+
cleanup=client.__exit__,
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _pick_free_port() -> int:
|
|
117
|
+
"""Reserve a local TCP port for the subprocess server."""
|
|
118
|
+
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s:
|
|
119
|
+
s.bind(("127.0.0.1", 0))
|
|
120
|
+
return int(s.getsockname()[1])
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _start_subprocess_server(server_data_dir: Path) -> _ServerContext:
|
|
124
|
+
"""Launch the demo server as a subprocess and wait for health."""
|
|
125
|
+
import requests
|
|
126
|
+
|
|
127
|
+
host = "127.0.0.1"
|
|
128
|
+
port = _pick_free_port()
|
|
129
|
+
env = os.environ.copy()
|
|
130
|
+
env["GKE_BACKEND"] = "chroma"
|
|
131
|
+
env["GKE_PERSIST_DIRECTORY"] = str(server_data_dir)
|
|
132
|
+
env["AUTH_MODE"] = "dev"
|
|
133
|
+
env["ANONYMIZED_TELEMETRY"] = "FALSE"
|
|
134
|
+
env["KOGWISTAR_LOG_LEVEL"] = "WARNING"
|
|
135
|
+
cmd = [
|
|
136
|
+
sys.executable,
|
|
137
|
+
"-m",
|
|
138
|
+
"uvicorn",
|
|
139
|
+
"kogwistar.server_mcp_with_admin:app",
|
|
140
|
+
"--host",
|
|
141
|
+
host,
|
|
142
|
+
"--port",
|
|
143
|
+
str(port),
|
|
144
|
+
"--log-level",
|
|
145
|
+
"warning",
|
|
146
|
+
]
|
|
147
|
+
proc = subprocess.Popen(
|
|
148
|
+
cmd,
|
|
149
|
+
cwd=str(Path(__file__).resolve().parents[2]),
|
|
150
|
+
env=env,
|
|
151
|
+
stdout=subprocess.PIPE,
|
|
152
|
+
stderr=subprocess.STDOUT,
|
|
153
|
+
text=True,
|
|
154
|
+
)
|
|
155
|
+
lines: list[str] = []
|
|
156
|
+
|
|
157
|
+
def _reader() -> None:
|
|
158
|
+
if proc.stdout is None:
|
|
159
|
+
return
|
|
160
|
+
for line in proc.stdout:
|
|
161
|
+
lines.append(line.rstrip())
|
|
162
|
+
|
|
163
|
+
thread = threading.Thread(target=_reader, daemon=True)
|
|
164
|
+
thread.start()
|
|
165
|
+
|
|
166
|
+
base_url = f"http://{host}:{port}"
|
|
167
|
+
deadline = time.time() + 60.0
|
|
168
|
+
last_error: Exception | None = None
|
|
169
|
+
while time.time() < deadline:
|
|
170
|
+
if proc.poll() is not None:
|
|
171
|
+
raise RuntimeError(
|
|
172
|
+
"demo server exited before healthy:\n" + "\n".join(lines[-50:])
|
|
173
|
+
)
|
|
174
|
+
try:
|
|
175
|
+
response = requests.get(f"{base_url}/health", timeout=1.5)
|
|
176
|
+
if response.ok:
|
|
177
|
+
session = requests.Session()
|
|
178
|
+
return _ServerContext(
|
|
179
|
+
client=session,
|
|
180
|
+
transport="subprocess_http",
|
|
181
|
+
base_url=base_url,
|
|
182
|
+
cleanup=lambda *_args: _shutdown_subprocess_server(proc, session),
|
|
183
|
+
)
|
|
184
|
+
except Exception as exc: # noqa: BLE001
|
|
185
|
+
last_error = exc
|
|
186
|
+
time.sleep(0.2)
|
|
187
|
+
_shutdown_subprocess_server(proc, None)
|
|
188
|
+
raise RuntimeError(f"demo server did not become healthy: {last_error}")
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def _connect_external_server(base_url: str) -> _ServerContext:
|
|
192
|
+
"""Connect to an already-running external demo server."""
|
|
193
|
+
import requests
|
|
194
|
+
|
|
195
|
+
base_url = str(base_url).rstrip("/")
|
|
196
|
+
response = requests.get(f"{base_url}/health", timeout=5.0)
|
|
197
|
+
response.raise_for_status()
|
|
198
|
+
session = requests.Session()
|
|
199
|
+
return _ServerContext(
|
|
200
|
+
client=session,
|
|
201
|
+
transport="external_http",
|
|
202
|
+
base_url=base_url,
|
|
203
|
+
cleanup=lambda *_args: session.close(),
|
|
204
|
+
)
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def _shutdown_subprocess_server(proc: subprocess.Popen[str], session: Any | None) -> None:
|
|
208
|
+
"""Terminate the subprocess server and close its HTTP session."""
|
|
209
|
+
if session is not None:
|
|
210
|
+
session.close()
|
|
211
|
+
if proc.poll() is None:
|
|
212
|
+
proc.terminate()
|
|
213
|
+
try:
|
|
214
|
+
proc.wait(timeout=5)
|
|
215
|
+
except Exception: # noqa: BLE001
|
|
216
|
+
proc.kill()
|
|
217
|
+
proc.wait(timeout=5)
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def _fake_layered_deps(inp: WorkflowIngestInput) -> dict[str, Any]:
|
|
221
|
+
"""Build deterministic proposal/review hooks for the fake layered demo mode."""
|
|
222
|
+
text = inp.collections[0].pages[0].units[0].text or ""
|
|
223
|
+
unit_id = f"{inp.request_id}|p1_t0"
|
|
224
|
+
lines = [line.strip() for line in text.splitlines() if line.strip()]
|
|
225
|
+
|
|
226
|
+
def _pointer(fragment: str) -> HydratedTextPointer:
|
|
227
|
+
start = text.index(fragment)
|
|
228
|
+
end = start + len(fragment) - 1
|
|
229
|
+
return HydratedTextPointer(
|
|
230
|
+
source_cluster_id=unit_id,
|
|
231
|
+
start_char=start,
|
|
232
|
+
end_char=end,
|
|
233
|
+
verbatim_text=fragment,
|
|
234
|
+
)
|
|
235
|
+
|
|
236
|
+
def _propose_layer_fn(*, current_layer_context, **kwargs):
|
|
237
|
+
if current_layer_context.depth == 0:
|
|
238
|
+
return CurrentLayerResult(
|
|
239
|
+
children=[
|
|
240
|
+
LayerChildCandidate(
|
|
241
|
+
node_id=f"{inp.request_id}|section|overview",
|
|
242
|
+
parent_node_id=current_layer_context.parent_node_ids[0],
|
|
243
|
+
title="Overview",
|
|
244
|
+
node_type="TEXT_FLOW",
|
|
245
|
+
total_content_pointers=[_pointer(text)],
|
|
246
|
+
expandable=len(lines) > 1,
|
|
247
|
+
)
|
|
248
|
+
],
|
|
249
|
+
satisfied=True,
|
|
250
|
+
reasoning_history=[{"stage": "proposal", "depth": 0, "lines": len(lines)}],
|
|
251
|
+
)
|
|
252
|
+
return CurrentLayerResult(
|
|
253
|
+
children=[
|
|
254
|
+
LayerChildCandidate(
|
|
255
|
+
node_id=f"{inp.request_id}|clause|{idx}",
|
|
256
|
+
parent_node_id=current_layer_context.parent_node_ids[0],
|
|
257
|
+
title=line,
|
|
258
|
+
node_type="TEXT_FLOW",
|
|
259
|
+
total_content_pointers=[_pointer(line)],
|
|
260
|
+
expandable=False,
|
|
261
|
+
)
|
|
262
|
+
for idx, line in enumerate(lines)
|
|
263
|
+
],
|
|
264
|
+
satisfied=True,
|
|
265
|
+
reasoning_history=[{"stage": "proposal", "depth": current_layer_context.depth}],
|
|
266
|
+
)
|
|
267
|
+
|
|
268
|
+
def _review_layer_fn(*, current_layer_result, **kwargs):
|
|
269
|
+
return CurrentLayerReview(
|
|
270
|
+
updated_result=current_layer_result.model_copy(update={"satisfied": True}),
|
|
271
|
+
coverage_ok=True,
|
|
272
|
+
satisfied=True,
|
|
273
|
+
review_notes=["fake_review_ok"],
|
|
274
|
+
)
|
|
275
|
+
|
|
276
|
+
return {
|
|
277
|
+
"propose_layer_fn": _propose_layer_fn,
|
|
278
|
+
"review_layer_fn": _review_layer_fn,
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def _configure_legacy_cache(cache_dir: Path, probe: WorkflowProbe | None) -> None:
|
|
283
|
+
"""Initialize the legacy parser cache used by the cached demo mode."""
|
|
284
|
+
cache_dir.mkdir(parents=True, exist_ok=True)
|
|
285
|
+
os.environ["KG_DOC_PARSER_JOBLIB_CACHE_DIR"] = str(cache_dir)
|
|
286
|
+
module_name = "kg_doc_parser.semantic_document_splitting_layerwise_edits"
|
|
287
|
+
if module_name in sys.modules:
|
|
288
|
+
importlib.reload(sys.modules[module_name])
|
|
289
|
+
emit_probe_event(probe, "demo.cache_reloaded", module=module_name, cache_dir=str(cache_dir))
|
|
290
|
+
else:
|
|
291
|
+
emit_probe_event(probe, "demo.cache_configured", module=module_name, cache_dir=str(cache_dir))
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
def _seed_demo_document(server_ctx: _ServerContext, *, document_id: str, text: str) -> None:
|
|
295
|
+
"""Create the source document in the demo server before graph persistence.
|
|
296
|
+
|
|
297
|
+
The generic tree-upsert route validates span excerpts against stored document
|
|
298
|
+
content, so the demo must seed the raw document first.
|
|
299
|
+
"""
|
|
300
|
+
payload = {
|
|
301
|
+
"doc_id": document_id,
|
|
302
|
+
"doc_type": "text",
|
|
303
|
+
"insertion_method": "demo_harness",
|
|
304
|
+
"content": text,
|
|
305
|
+
}
|
|
306
|
+
response = server_ctx.client.post(f"{server_ctx.base_url}/api/document", json=payload)
|
|
307
|
+
status_code = int(getattr(response, "status_code", 500))
|
|
308
|
+
if status_code >= 400:
|
|
309
|
+
body = getattr(response, "text", "")
|
|
310
|
+
raise RuntimeError(f"demo document seed failed: {body}")
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
def run_demo_harness(config: DemoHarnessConfig) -> DemoHarnessArtifacts:
|
|
314
|
+
"""Run the ingest demo and persist probe and summary artifacts."""
|
|
315
|
+
output_dir = Path(config.output_dir)
|
|
316
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
317
|
+
probe = WorkflowProbe(output_dir / config.probe_filename)
|
|
318
|
+
if config.enable_sys_monitoring:
|
|
319
|
+
probe.enable_sys_monitoring()
|
|
320
|
+
|
|
321
|
+
artifacts = DemoHarnessArtifacts(
|
|
322
|
+
output_dir=output_dir,
|
|
323
|
+
probe_path=output_dir / config.probe_filename,
|
|
324
|
+
summary_path=output_dir / config.summary_filename,
|
|
325
|
+
cache_dir=output_dir / config.cache_dirname,
|
|
326
|
+
engine_dir=output_dir / "engines",
|
|
327
|
+
server_data_dir=output_dir / "server-data",
|
|
328
|
+
)
|
|
329
|
+
inp = WorkflowIngestInput.from_text(
|
|
330
|
+
document_id=config.document_id,
|
|
331
|
+
text=config.text,
|
|
332
|
+
title=config.title,
|
|
333
|
+
)
|
|
334
|
+
|
|
335
|
+
deps = dict(config.deps)
|
|
336
|
+
deps["llm_cache"] = WorkflowLLMCallCache(artifacts.cache_dir, probe=probe)
|
|
337
|
+
if config.parser_mode == "fake_layered":
|
|
338
|
+
deps.update(_fake_layered_deps(inp))
|
|
339
|
+
else:
|
|
340
|
+
_configure_legacy_cache(artifacts.cache_dir, probe)
|
|
341
|
+
|
|
342
|
+
deps["probe"] = probe
|
|
343
|
+
|
|
344
|
+
emit_probe_event(
|
|
345
|
+
probe,
|
|
346
|
+
"demo.started",
|
|
347
|
+
output_dir=str(output_dir),
|
|
348
|
+
parser_mode=config.parser_mode,
|
|
349
|
+
server_mode=config.server_mode,
|
|
350
|
+
document_id=config.document_id,
|
|
351
|
+
external_base_url=config.external_base_url,
|
|
352
|
+
)
|
|
353
|
+
|
|
354
|
+
if config.server_mode == "testclient":
|
|
355
|
+
server_ctx = _load_isolated_server_app(artifacts.server_data_dir)
|
|
356
|
+
elif config.server_mode == "subprocess_http":
|
|
357
|
+
server_ctx = _start_subprocess_server(artifacts.server_data_dir)
|
|
358
|
+
else:
|
|
359
|
+
if not config.external_base_url:
|
|
360
|
+
raise ValueError("external_base_url is required when server_mode='external_http'")
|
|
361
|
+
server_ctx = _connect_external_server(config.external_base_url)
|
|
362
|
+
emit_probe_event(
|
|
363
|
+
probe,
|
|
364
|
+
"demo.server_started",
|
|
365
|
+
server_mode=config.server_mode,
|
|
366
|
+
transport=server_ctx.transport,
|
|
367
|
+
server_data_dir=str(artifacts.server_data_dir),
|
|
368
|
+
base_url=server_ctx.base_url,
|
|
369
|
+
)
|
|
370
|
+
try:
|
|
371
|
+
_seed_demo_document(
|
|
372
|
+
server_ctx,
|
|
373
|
+
document_id=config.document_id,
|
|
374
|
+
text=config.text,
|
|
375
|
+
)
|
|
376
|
+
workflow_engine, conversation_engine, _knowledge_engine = build_default_engines(
|
|
377
|
+
artifacts.engine_dir,
|
|
378
|
+
backend_factory=config.backend_factory,
|
|
379
|
+
provider_settings=config.provider_settings,
|
|
380
|
+
)
|
|
381
|
+
persistence_client = DocumentTreeApiPersistenceClient(
|
|
382
|
+
client=server_ctx.client,
|
|
383
|
+
base_url=server_ctx.base_url,
|
|
384
|
+
transport=server_ctx.transport,
|
|
385
|
+
)
|
|
386
|
+
client = ServerCanonicalKgClient(
|
|
387
|
+
workflow_engine=workflow_engine,
|
|
388
|
+
conversation_engine=conversation_engine,
|
|
389
|
+
persistence_client=persistence_client,
|
|
390
|
+
)
|
|
391
|
+
result = client.run_ingest(
|
|
392
|
+
inp=inp,
|
|
393
|
+
workflow_id=config.workflow_id,
|
|
394
|
+
deps=deps,
|
|
395
|
+
)
|
|
396
|
+
artifacts.run_id = result.handle.run_id
|
|
397
|
+
artifacts.status = result.status
|
|
398
|
+
if result.bundle is not None:
|
|
399
|
+
artifacts.canonical_write_confirmed = result.bundle.canonical_write_confirmed
|
|
400
|
+
artifacts.persistence_mode = result.bundle.persistence_mode
|
|
401
|
+
artifacts.kg_authority = result.bundle.kg_authority
|
|
402
|
+
summary = {
|
|
403
|
+
"run_id": artifacts.run_id,
|
|
404
|
+
"status": artifacts.status,
|
|
405
|
+
"canonical_write_confirmed": artifacts.canonical_write_confirmed,
|
|
406
|
+
"persistence_mode": artifacts.persistence_mode,
|
|
407
|
+
"kg_authority": artifacts.kg_authority,
|
|
408
|
+
"probe_path": str(artifacts.probe_path),
|
|
409
|
+
"cache_dir": str(artifacts.cache_dir),
|
|
410
|
+
"engine_dir": str(artifacts.engine_dir),
|
|
411
|
+
"server_data_dir": str(artifacts.server_data_dir),
|
|
412
|
+
"parser_mode": config.parser_mode,
|
|
413
|
+
"server_mode": config.server_mode,
|
|
414
|
+
}
|
|
415
|
+
artifacts.summary_path.write_text(json.dumps(summary, indent=2), encoding="utf-8")
|
|
416
|
+
emit_probe_event(
|
|
417
|
+
probe,
|
|
418
|
+
"demo.finished",
|
|
419
|
+
run_id=artifacts.run_id,
|
|
420
|
+
status=artifacts.status,
|
|
421
|
+
canonical_write_confirmed=artifacts.canonical_write_confirmed,
|
|
422
|
+
)
|
|
423
|
+
return artifacts
|
|
424
|
+
finally:
|
|
425
|
+
emit_probe_event(probe, "demo.server_stopped", server_mode=config.server_mode)
|
|
426
|
+
server_ctx.__exit__(None, None, None)
|
|
427
|
+
probe.close()
|
|
@@ -0,0 +1,208 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import logging
|
|
4
|
+
from typing import Iterable
|
|
5
|
+
|
|
6
|
+
from kogwistar.engine_core.models import Grounding, Span
|
|
7
|
+
from kogwistar.runtime.models import WorkflowEdge, WorkflowNode
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
DEFAULT_WORKFLOW_ID = "kg_doc_parser.ingest.v1"
|
|
11
|
+
_LOGGER = logging.getLogger(__name__)
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _progress_bar(done: int, total: int, width: int = 20) -> str:
|
|
15
|
+
if total <= 0:
|
|
16
|
+
return "?" * width
|
|
17
|
+
filled = min(width, max(0, round((done / total) * width)))
|
|
18
|
+
return ("█" * filled) + ("░" * (width - filled))
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _grounding(workflow_id: str) -> Grounding:
|
|
22
|
+
return Grounding(spans=[Span.from_dummy_for_workflow(workflow_id)])
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _workflow_node(
|
|
26
|
+
*,
|
|
27
|
+
workflow_id: str,
|
|
28
|
+
node_id: str,
|
|
29
|
+
op: str,
|
|
30
|
+
start: bool = False,
|
|
31
|
+
terminal: bool = False,
|
|
32
|
+
) -> WorkflowNode:
|
|
33
|
+
return WorkflowNode(
|
|
34
|
+
id=node_id,
|
|
35
|
+
label=node_id.split("|")[-1],
|
|
36
|
+
type="entity",
|
|
37
|
+
doc_id=node_id,
|
|
38
|
+
summary=op,
|
|
39
|
+
properties={},
|
|
40
|
+
metadata={
|
|
41
|
+
"entity_type": "workflow_node",
|
|
42
|
+
"workflow_id": workflow_id,
|
|
43
|
+
"wf_op": op,
|
|
44
|
+
"wf_start": bool(start),
|
|
45
|
+
"wf_terminal": bool(terminal),
|
|
46
|
+
"wf_version": "v1",
|
|
47
|
+
},
|
|
48
|
+
mentions=[_grounding(workflow_id)],
|
|
49
|
+
level_from_root=0,
|
|
50
|
+
domain_id=None,
|
|
51
|
+
canonical_entity_id=None,
|
|
52
|
+
embedding=None,
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _workflow_edge(
|
|
57
|
+
*,
|
|
58
|
+
workflow_id: str,
|
|
59
|
+
edge_id: str,
|
|
60
|
+
src: str,
|
|
61
|
+
dst: str,
|
|
62
|
+
) -> WorkflowEdge:
|
|
63
|
+
dst_name = dst.split("|")[-1]
|
|
64
|
+
return WorkflowEdge(
|
|
65
|
+
id=edge_id,
|
|
66
|
+
label=dst_name,
|
|
67
|
+
type="relationship",
|
|
68
|
+
doc_id=edge_id,
|
|
69
|
+
summary=f"next:{dst_name}",
|
|
70
|
+
properties={},
|
|
71
|
+
source_ids=[src],
|
|
72
|
+
target_ids=[dst],
|
|
73
|
+
relation="wf_next",
|
|
74
|
+
source_edge_ids=[],
|
|
75
|
+
target_edge_ids=[],
|
|
76
|
+
metadata={
|
|
77
|
+
"entity_type": "workflow_edge",
|
|
78
|
+
"workflow_id": workflow_id,
|
|
79
|
+
"wf_priority": 100,
|
|
80
|
+
"wf_is_default": True,
|
|
81
|
+
"wf_multiplicity": "one",
|
|
82
|
+
"wf_version": "v1",
|
|
83
|
+
},
|
|
84
|
+
mentions=[_grounding(workflow_id)],
|
|
85
|
+
domain_id=None,
|
|
86
|
+
canonical_entity_id=None,
|
|
87
|
+
embedding=None,
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def build_ingest_workflow_design(
|
|
92
|
+
workflow_id: str = DEFAULT_WORKFLOW_ID,
|
|
93
|
+
) -> tuple[list[WorkflowNode], list[WorkflowEdge]]:
|
|
94
|
+
node_specs = [
|
|
95
|
+
("start", "start", True, False),
|
|
96
|
+
("normalize_input", "normalize_input", False, False),
|
|
97
|
+
("build_source_map", "build_source_map", False, False),
|
|
98
|
+
("init_parse_session", "init_parse_session", False, False),
|
|
99
|
+
("check_frontier_remaining", "check_frontier_remaining", False, False),
|
|
100
|
+
("prepare_layer_frontier", "prepare_layer_frontier", False, False),
|
|
101
|
+
("propose_layer_breakdown", "propose_layer_breakdown", False, False),
|
|
102
|
+
("review_cud_proposal", "review_cud_proposal", False, False),
|
|
103
|
+
("apply_cud_update", "apply_cud_update", False, False),
|
|
104
|
+
("check_layer_coverage", "check_layer_coverage", False, False),
|
|
105
|
+
("check_layer_satisfaction", "check_layer_satisfaction", False, False),
|
|
106
|
+
("switch_split_strategy", "switch_split_strategy", False, False),
|
|
107
|
+
("repair_layer_pointers", "repair_layer_pointers", False, False),
|
|
108
|
+
("dedupe_and_filter_layer", "dedupe_and_filter_layer", False, False),
|
|
109
|
+
("commit_layer_children", "commit_layer_children", False, False),
|
|
110
|
+
("check_children_expandable", "check_children_expandable", False, False),
|
|
111
|
+
("enqueue_next_layer_frontier", "enqueue_next_layer_frontier", False, False),
|
|
112
|
+
("finalize_semantic_tree", "finalize_semantic_tree", False, False),
|
|
113
|
+
("validate_tree", "validate_tree", False, False),
|
|
114
|
+
("export_graph", "export_graph", False, False),
|
|
115
|
+
("persist_canonical_graph", "persist_canonical_graph", False, False),
|
|
116
|
+
("end", "end", False, True),
|
|
117
|
+
]
|
|
118
|
+
nodes = [
|
|
119
|
+
_workflow_node(
|
|
120
|
+
workflow_id=workflow_id,
|
|
121
|
+
node_id=f"wf|{workflow_id}|{suffix}",
|
|
122
|
+
op=op,
|
|
123
|
+
start=start,
|
|
124
|
+
terminal=terminal,
|
|
125
|
+
)
|
|
126
|
+
for suffix, op, start, terminal in node_specs
|
|
127
|
+
]
|
|
128
|
+
node_by_suffix = {node.id.split("|")[-1]: node for node in nodes}
|
|
129
|
+
edge_pairs: Iterable[tuple[str, str]] = [
|
|
130
|
+
("start", "normalize_input"),
|
|
131
|
+
("normalize_input", "build_source_map"),
|
|
132
|
+
("build_source_map", "init_parse_session"),
|
|
133
|
+
("init_parse_session", "check_frontier_remaining"),
|
|
134
|
+
("check_frontier_remaining", "prepare_layer_frontier"),
|
|
135
|
+
("check_frontier_remaining", "finalize_semantic_tree"),
|
|
136
|
+
("prepare_layer_frontier", "propose_layer_breakdown"),
|
|
137
|
+
("propose_layer_breakdown", "review_cud_proposal"),
|
|
138
|
+
("review_cud_proposal", "apply_cud_update"),
|
|
139
|
+
("apply_cud_update", "check_layer_coverage"),
|
|
140
|
+
("check_layer_coverage", "check_layer_satisfaction"),
|
|
141
|
+
("check_layer_satisfaction", "propose_layer_breakdown"),
|
|
142
|
+
("check_layer_satisfaction", "switch_split_strategy"),
|
|
143
|
+
("switch_split_strategy", "propose_layer_breakdown"),
|
|
144
|
+
("check_layer_satisfaction", "repair_layer_pointers"),
|
|
145
|
+
("repair_layer_pointers", "dedupe_and_filter_layer"),
|
|
146
|
+
("dedupe_and_filter_layer", "commit_layer_children"),
|
|
147
|
+
("commit_layer_children", "check_children_expandable"),
|
|
148
|
+
("check_children_expandable", "enqueue_next_layer_frontier"),
|
|
149
|
+
("check_children_expandable", "check_frontier_remaining"),
|
|
150
|
+
("enqueue_next_layer_frontier", "check_frontier_remaining"),
|
|
151
|
+
("finalize_semantic_tree", "validate_tree"),
|
|
152
|
+
("validate_tree", "export_graph"),
|
|
153
|
+
("export_graph", "persist_canonical_graph"),
|
|
154
|
+
("persist_canonical_graph", "end"),
|
|
155
|
+
]
|
|
156
|
+
edges = [
|
|
157
|
+
_workflow_edge(
|
|
158
|
+
workflow_id=workflow_id,
|
|
159
|
+
edge_id=f"wf|{workflow_id}|e|{src}->{dst}",
|
|
160
|
+
src=node_by_suffix[src].safe_get_id(),
|
|
161
|
+
dst=node_by_suffix[dst].safe_get_id(),
|
|
162
|
+
)
|
|
163
|
+
for src, dst in edge_pairs
|
|
164
|
+
]
|
|
165
|
+
return nodes, edges
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def ensure_ingest_workflow_design(workflow_engine, workflow_id: str = DEFAULT_WORKFLOW_ID) -> None:
|
|
169
|
+
nodes, edges = build_ingest_workflow_design(workflow_id)
|
|
170
|
+
total = len(nodes) + len(edges)
|
|
171
|
+
done = 0
|
|
172
|
+
_LOGGER.info(
|
|
173
|
+
"⏳ workflow design | 0/%s | 0%% | %s | bootstrap %s nodes + %s edges",
|
|
174
|
+
total,
|
|
175
|
+
_progress_bar(0, total),
|
|
176
|
+
len(nodes),
|
|
177
|
+
len(edges),
|
|
178
|
+
)
|
|
179
|
+
for node_idx, node in enumerate(nodes, start=1):
|
|
180
|
+
node_id = node.safe_get_id()
|
|
181
|
+
if not workflow_engine.persist.exists_node(node_id):
|
|
182
|
+
workflow_engine.write.add_node(node)
|
|
183
|
+
done += 1
|
|
184
|
+
_LOGGER.info(
|
|
185
|
+
"⏳ workflow design | %s/%s | %3s%% | %s | node %s/%s | %s",
|
|
186
|
+
done,
|
|
187
|
+
total,
|
|
188
|
+
round((done / total) * 100),
|
|
189
|
+
_progress_bar(done, total),
|
|
190
|
+
node_idx,
|
|
191
|
+
len(nodes),
|
|
192
|
+
node.label,
|
|
193
|
+
)
|
|
194
|
+
for edge_idx, edge in enumerate(edges, start=1):
|
|
195
|
+
edge_id = edge.safe_get_id()
|
|
196
|
+
if not workflow_engine.persist.exists_edge(edge_id):
|
|
197
|
+
workflow_engine.write.add_edge(edge)
|
|
198
|
+
done += 1
|
|
199
|
+
_LOGGER.info(
|
|
200
|
+
"⏳ workflow design | %s/%s | %3s%% | %s | edge %s/%s | %s",
|
|
201
|
+
done,
|
|
202
|
+
total,
|
|
203
|
+
round((done / total) * 100),
|
|
204
|
+
_progress_bar(done, total),
|
|
205
|
+
edge_idx,
|
|
206
|
+
len(edges),
|
|
207
|
+
edge.label,
|
|
208
|
+
)
|