graph-knowledge-doc-parser 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. graph_knowledge_doc_parser-0.1.0.dist-info/METADATA +326 -0
  2. graph_knowledge_doc_parser-0.1.0.dist-info/RECORD +38 -0
  3. graph_knowledge_doc_parser-0.1.0.dist-info/WHEEL +4 -0
  4. graph_knowledge_doc_parser-0.1.0.dist-info/entry_points.txt +3 -0
  5. kg_doc_parser/__init__.py +9 -0
  6. kg_doc_parser/cast_hinting.py +19 -0
  7. kg_doc_parser/document_ingester_logger.py +766 -0
  8. kg_doc_parser/models.py +277 -0
  9. kg_doc_parser/ocr.py +752 -0
  10. kg_doc_parser/pdf2png.py +286 -0
  11. kg_doc_parser/semantic_document_splitting_layerwise_edits.py +3302 -0
  12. kg_doc_parser/text_processing_utils.py +30 -0
  13. kg_doc_parser/utils/__init__.py +0 -0
  14. kg_doc_parser/utils/bounded_threadpool_executor.py +37 -0
  15. kg_doc_parser/utils/file_loaders.py +405 -0
  16. kg_doc_parser/utils/langchain.py +220 -0
  17. kg_doc_parser/utils/log.py +135 -0
  18. kg_doc_parser/utils/version_chaining.py +1278 -0
  19. kg_doc_parser/workflow_ingest/__init__.py +187 -0
  20. kg_doc_parser/workflow_ingest/_kogwistar.py +13 -0
  21. kg_doc_parser/workflow_ingest/adapters.py +212 -0
  22. kg_doc_parser/workflow_ingest/cache.py +63 -0
  23. kg_doc_parser/workflow_ingest/cli.py +324 -0
  24. kg_doc_parser/workflow_ingest/clients.py +444 -0
  25. kg_doc_parser/workflow_ingest/demo_harness.py +427 -0
  26. kg_doc_parser/workflow_ingest/design.py +208 -0
  27. kg_doc_parser/workflow_ingest/handlers.py +617 -0
  28. kg_doc_parser/workflow_ingest/models.py +575 -0
  29. kg_doc_parser/workflow_ingest/ocr_pipeline.py +1581 -0
  30. kg_doc_parser/workflow_ingest/page_index.py +473 -0
  31. kg_doc_parser/workflow_ingest/parser_core.py +862 -0
  32. kg_doc_parser/workflow_ingest/parsing.py +249 -0
  33. kg_doc_parser/workflow_ingest/probe.py +164 -0
  34. kg_doc_parser/workflow_ingest/providers.py +412 -0
  35. kg_doc_parser/workflow_ingest/runners.py +546 -0
  36. kg_doc_parser/workflow_ingest/semantics.py +231 -0
  37. kg_doc_parser/workflow_ingest/service.py +112 -0
  38. kg_doc_parser/workflow_ingest/smoke_assets.py +62 -0
@@ -0,0 +1,546 @@
1
+ from __future__ import annotations
2
+
3
+ """Reusable workflow runners for CLI and higher-level orchestration.
4
+
5
+ These helpers compose the existing workflow ingest primitives without changing
6
+ their core behavior. The CLI calls into this module, but test code and other
7
+ workflow code can also reuse the same wrappers directly.
8
+ """
9
+
10
+ import json
11
+ import os
12
+ from contextlib import contextmanager
13
+ from dataclasses import dataclass, field
14
+ from pathlib import Path
15
+ from typing import Any, Iterable, Literal, Sequence
16
+
17
+ from kogwistar.engine_core.models import Edge, Node
18
+
19
+ from .demo_harness import DemoHarnessConfig, run_demo_harness
20
+ from .ocr_pipeline import OCRImagePayload, OCRWorkflowArtifacts, prepare_ocr_workflow_input, run_ocr_ingest_workflow
21
+ from .page_index import PageIndexParseResult, PageIndexSourceFormat
22
+ from .parsing import parse_ocr_document, parse_page_index_document, parse_tree_document
23
+ from .probe import WorkflowProbe, emit_probe_event
24
+ from .providers import WorkflowProviderSettings
25
+ from .parser_core import default_parse_semantic_fn
26
+ from .service import build_default_engines, run_ingest_workflow
27
+ from .semantics import HydratedTextPointer, SemanticNode
28
+
29
+ SupportedOCRInput = Literal["image", "pdf"]
30
+ SupportedPageIndexInput = Literal["text", "markdown"]
31
+
32
+ OCR_FILE_SUFFIXES = {".png", ".jpg", ".jpeg", ".webp", ".bmp", ".tif", ".tiff", ".pdf"}
33
+ PAGE_INDEX_SUFFIXES = {".txt", ".md"}
34
+
35
+
36
+ @dataclass(slots=True)
37
+ class WorkflowCommandResult:
38
+ kind: str
39
+ input_path: Path
40
+ output_dir: Path
41
+ status: str | None = None
42
+ probe_path: Path | None = None
43
+ summary_path: Path | None = None
44
+ extra: dict[str, Any] = field(default_factory=dict)
45
+
46
+
47
+ @dataclass(slots=True)
48
+ class OcrWorkflowCommandResult(WorkflowCommandResult):
49
+ artifacts: OCRWorkflowArtifacts | None = None
50
+
51
+
52
+ @dataclass(slots=True)
53
+ class PageIndexWorkflowCommandResult(WorkflowCommandResult):
54
+ result: PageIndexParseResult | None = None
55
+
56
+
57
+ @dataclass(slots=True)
58
+ class LayerwiseWorkflowCommandResult(WorkflowCommandResult):
59
+ tree: Any | None = None
60
+ source_map: dict[str, Any] | None = None
61
+ graph_payload: dict[str, Any] | None = None
62
+
63
+
64
+ def _fallback_parse_semantic_fn(*, collection, parser_input_dict: dict[str, Any], parser_source_map: dict[str, dict[str, Any]]):
65
+ root = SemanticNode(
66
+ title=collection.title,
67
+ node_type="DOCUMENT_ROOT",
68
+ total_content_pointers=[],
69
+ child_nodes=[],
70
+ level_from_root=0,
71
+ )
72
+ for page in collection.pages:
73
+ page_text_parts: list[str] = []
74
+ page_node = SemanticNode(
75
+ title=f"Page {page.page_number}",
76
+ node_type="PAGE",
77
+ total_content_pointers=[],
78
+ child_nodes=[],
79
+ level_from_root=1,
80
+ parent_id=root.node_id,
81
+ )
82
+ for unit in page.units:
83
+ if not getattr(unit, "text", None):
84
+ continue
85
+ text = str(unit.text)
86
+ page_text_parts.append(text)
87
+ page_node.child_nodes.append(
88
+ SemanticNode(
89
+ title=(text.splitlines()[0].strip()[:80] or f"Unit {unit.cluster_number or 0}"),
90
+ node_type="TEXT_FLOW",
91
+ total_content_pointers=[
92
+ HydratedTextPointer(
93
+ source_cluster_id=unit.unit_id or f"{collection.collection_id}|p{page.page_number}",
94
+ start_char=0,
95
+ end_char=max(0, len(text) - 1),
96
+ verbatim_text=text,
97
+ )
98
+ ],
99
+ child_nodes=[],
100
+ level_from_root=2,
101
+ parent_id=page_node.node_id,
102
+ )
103
+ )
104
+ if not page_node.child_nodes:
105
+ page_node.child_nodes.append(
106
+ SemanticNode(
107
+ title=f"Page {page.page_number} Empty",
108
+ node_type="TEXT_FLOW",
109
+ total_content_pointers=[],
110
+ child_nodes=[],
111
+ level_from_root=2,
112
+ parent_id=page_node.node_id,
113
+ )
114
+ )
115
+ if page_text_parts:
116
+ page_node.total_content_pointers = [
117
+ HydratedTextPointer(
118
+ source_cluster_id=page.units[0].unit_id or f"{collection.collection_id}|p{page.page_number}",
119
+ start_char=0,
120
+ end_char=max(0, len("\n".join(page_text_parts)) - 1),
121
+ verbatim_text="\n".join(page_text_parts),
122
+ )
123
+ ]
124
+ root.child_nodes.append(page_node)
125
+ return root
126
+
127
+
128
+ @contextmanager
129
+ def _temporary_env(overrides: dict[str, str | None]):
130
+ previous: dict[str, str | None] = {}
131
+ try:
132
+ for key, value in overrides.items():
133
+ previous[key] = os.environ.get(key)
134
+ if value is None:
135
+ os.environ.pop(key, None)
136
+ else:
137
+ os.environ[key] = value
138
+ yield
139
+ finally:
140
+ for key, value in previous.items():
141
+ if value is None:
142
+ os.environ.pop(key, None)
143
+ else:
144
+ os.environ[key] = value
145
+
146
+
147
+ def build_legacy_parse_semantic_fn(
148
+ *,
149
+ provider_settings: WorkflowProviderSettings,
150
+ model_names: Sequence[str] | None = None,
151
+ ):
152
+ parser_spec = provider_settings.parser
153
+ parser_model_names = list(model_names) if model_names else [parser_spec.model]
154
+
155
+ def _parse_semantic_fn(*, collection, parser_input_dict: dict[str, Any], parser_source_map: dict[str, dict[str, Any]]):
156
+ env_overrides = {
157
+ "KG_DOC_PARSER_PROVIDER": parser_spec.provider,
158
+ "KG_DOC_PARSER_MODEL": parser_spec.model,
159
+ "KG_DOC_PARSER_TEMPERATURE": str(parser_spec.temperature),
160
+ "KG_DOC_PARSER_BASE_URL": parser_spec.base_url,
161
+ "KG_DOC_PARSER_API_KEY_ENV": parser_spec.api_key_env,
162
+ "KG_DOC_PARSER_PROJECT": parser_spec.project,
163
+ "KG_DOC_PARSER_LOCATION": parser_spec.location,
164
+ "KG_DOC_PARSER_MAX_RETRIES": str(parser_spec.max_retries),
165
+ }
166
+ with _temporary_env(env_overrides):
167
+ return default_parse_semantic_fn(
168
+ collection=collection,
169
+ parser_input_dict=parser_input_dict,
170
+ parser_source_map=parser_source_map,
171
+ model_names=parser_model_names,
172
+ )
173
+
174
+ return _parse_semantic_fn
175
+
176
+
177
+ def _ensure_probe(output_dir: Path, probe: WorkflowProbe | None = None) -> WorkflowProbe:
178
+ if probe is not None:
179
+ return probe
180
+ return WorkflowProbe(output_dir / "workflow-events.jsonl")
181
+
182
+
183
+ def _emit(probe: WorkflowProbe | None, kind: str, /, **payload: Any) -> None:
184
+ emit_probe_event(probe, kind, **payload)
185
+
186
+
187
+ def discover_input_files(
188
+ paths: Sequence[str | Path],
189
+ *,
190
+ allowed_suffixes: set[str],
191
+ recursive: bool = True,
192
+ ) -> list[Path]:
193
+ files: list[Path] = []
194
+ for raw_path in paths:
195
+ path = Path(raw_path)
196
+ if path.is_dir():
197
+ iterator = path.rglob("*") if recursive else path.iterdir()
198
+ for candidate in iterator:
199
+ if candidate.is_file() and candidate.suffix.lower() in allowed_suffixes:
200
+ files.append(candidate)
201
+ continue
202
+ if path.suffix.lower() in allowed_suffixes:
203
+ files.append(path)
204
+ return sorted({p.resolve(): p for p in files}.values(), key=lambda p: str(p))
205
+
206
+
207
+ def run_ocr_source_workflow(
208
+ source_path: str | Path,
209
+ *,
210
+ output_dir: str | Path,
211
+ provider_settings: WorkflowProviderSettings | None = None,
212
+ ocr_runner=None,
213
+ pdf_rasterizer=None,
214
+ ocr_candidate_models: Sequence[str] | None = None,
215
+ workflow_engine=None,
216
+ conversation_engine=None,
217
+ knowledge_engine=None,
218
+ probe: WorkflowProbe | None = None,
219
+ deps: dict[str, Any] | None = None,
220
+ document_id: str | None = None,
221
+ title: str | None = None,
222
+ ) -> OcrWorkflowCommandResult:
223
+ source_path = Path(source_path)
224
+ output_dir = Path(output_dir)
225
+ output_dir.mkdir(parents=True, exist_ok=True)
226
+ probe = _ensure_probe(output_dir, probe)
227
+ provider_settings = provider_settings or WorkflowProviderSettings.from_env()
228
+ document_id = document_id or source_path.stem
229
+ title = title or source_path.stem
230
+
231
+ _emit(
232
+ probe,
233
+ "workflow.file_started",
234
+ workflow_kind="ocr",
235
+ source_path=str(source_path),
236
+ output_dir=str(output_dir),
237
+ document_id=document_id,
238
+ )
239
+ if source_path.suffix.lower() == ".pdf":
240
+ image_payloads = None
241
+ pdf_path = source_path
242
+ else:
243
+ pdf_path = None
244
+ image_payloads = [
245
+ OCRImagePayload(page_number=1, image_path=str(source_path)),
246
+ ]
247
+ if workflow_engine is None or conversation_engine is None:
248
+ workflow_engine, conversation_engine, default_knowledge_engine = build_default_engines(
249
+ output_dir / "engines",
250
+ provider_settings=provider_settings,
251
+ )
252
+ if knowledge_engine is None:
253
+ knowledge_engine = default_knowledge_engine
254
+ run, bundle, artifacts = run_ocr_ingest_workflow(
255
+ document_id=document_id,
256
+ title=title,
257
+ output_dir=output_dir,
258
+ workflow_engine=workflow_engine,
259
+ conversation_engine=conversation_engine,
260
+ knowledge_engine=knowledge_engine,
261
+ image_payloads=image_payloads,
262
+ pdf_path=pdf_path,
263
+ provider_settings=provider_settings,
264
+ ocr_runner=ocr_runner,
265
+ pdf_rasterizer=pdf_rasterizer,
266
+ ocr_candidate_models=ocr_candidate_models,
267
+ deps={"probe": probe, **(deps or {})},
268
+ probe=probe,
269
+ )
270
+ _emit(
271
+ probe,
272
+ "workflow.file_finished",
273
+ workflow_kind="ocr",
274
+ source_path=str(source_path),
275
+ output_dir=str(output_dir),
276
+ document_id=document_id,
277
+ status=run.status,
278
+ summary_path=str(artifacts.summary_path),
279
+ )
280
+ return OcrWorkflowCommandResult(
281
+ kind="ocr",
282
+ input_path=source_path,
283
+ output_dir=output_dir,
284
+ status=run.status,
285
+ probe_path=probe.path,
286
+ summary_path=artifacts.summary_path,
287
+ extra={
288
+ "run_id": run.run_id,
289
+ "bundle": bundle.model_dump(field_mode="backend", dump_format="json") if bundle is not None else None,
290
+ "state_db_path": str(artifacts.state_db_path),
291
+ "legacy_dir": str(artifacts.legacy_dir),
292
+ "rendered_dir": str(artifacts.rendered_dir),
293
+ },
294
+ artifacts=artifacts,
295
+ )
296
+
297
+
298
+ def run_ocr_batch_workflow(
299
+ source_paths: Sequence[str | Path],
300
+ *,
301
+ output_dir: str | Path,
302
+ provider_settings: WorkflowProviderSettings | None = None,
303
+ ocr_runner=None,
304
+ pdf_rasterizer=None,
305
+ ocr_candidate_models: Sequence[str] | None = None,
306
+ workflow_engine=None,
307
+ conversation_engine=None,
308
+ knowledge_engine=None,
309
+ probe: WorkflowProbe | None = None,
310
+ deps: dict[str, Any] | None = None,
311
+ ) -> list[OcrWorkflowCommandResult]:
312
+ output_dir = Path(output_dir)
313
+ output_dir.mkdir(parents=True, exist_ok=True)
314
+ probe = _ensure_probe(output_dir, probe)
315
+ files = discover_input_files(source_paths, allowed_suffixes=OCR_FILE_SUFFIXES, recursive=True)
316
+ _emit(probe, "workflow.batch_started", workflow_kind="ocr", file_count=len(files), output_dir=str(output_dir))
317
+ results: list[OcrWorkflowCommandResult] = []
318
+ for source_path in files:
319
+ relative_output = output_dir / source_path.stem
320
+ result = run_ocr_source_workflow(
321
+ source_path,
322
+ output_dir=relative_output,
323
+ provider_settings=provider_settings,
324
+ ocr_runner=ocr_runner,
325
+ pdf_rasterizer=pdf_rasterizer,
326
+ ocr_candidate_models=ocr_candidate_models,
327
+ workflow_engine=workflow_engine,
328
+ conversation_engine=conversation_engine,
329
+ knowledge_engine=knowledge_engine,
330
+ probe=probe,
331
+ deps=deps,
332
+ document_id=source_path.stem,
333
+ title=source_path.stem,
334
+ )
335
+ results.append(result)
336
+ _emit(probe, "workflow.batch_finished", workflow_kind="ocr", file_count=len(files), output_dir=str(output_dir))
337
+ return results
338
+
339
+
340
+ def run_page_index_source_workflow(
341
+ source_path: str | Path,
342
+ *,
343
+ output_dir: str | Path,
344
+ mode: str = "heuristic",
345
+ source_format: str = "auto",
346
+ provider_settings: WorkflowProviderSettings | None = None,
347
+ probe: WorkflowProbe | None = None,
348
+ ) -> PageIndexWorkflowCommandResult:
349
+ source_path = Path(source_path)
350
+ output_dir = Path(output_dir)
351
+ output_dir.mkdir(parents=True, exist_ok=True)
352
+ probe = _ensure_probe(output_dir, probe)
353
+ raw_text = source_path.read_text(encoding="utf-8")
354
+ inferred_format: PageIndexSourceFormat = (
355
+ "markdown" if source_format == "auto" and source_path.suffix.lower() == ".md" else "text"
356
+ )
357
+ if source_format in {"text", "markdown"}:
358
+ inferred_format = source_format # type: ignore[assignment]
359
+
360
+ _emit(
361
+ probe,
362
+ "workflow.file_started",
363
+ workflow_kind="page_index",
364
+ source_path=str(source_path),
365
+ output_dir=str(output_dir),
366
+ mode=mode,
367
+ )
368
+ result = parse_page_index_document(
369
+ document_id=source_path.stem,
370
+ title=source_path.stem,
371
+ raw_text=raw_text,
372
+ source_format=inferred_format,
373
+ mode=mode, # type: ignore[arg-type]
374
+ provider_settings=provider_settings,
375
+ )
376
+ summary = {
377
+ "kind": "page_index",
378
+ "source_path": str(source_path),
379
+ "mode": mode,
380
+ "source_format": inferred_format,
381
+ "overall_coverage": result.coverage.get("overall"),
382
+ "page_count": len(result.workflow_input.collections[0].pages),
383
+ "max_depth": max((len(page.child_nodes) for page in result.semantic_tree.child_nodes), default=0),
384
+ }
385
+ summary_path = output_dir / "page-index-summary.json"
386
+ summary_path.write_text(json.dumps(summary, indent=2), encoding="utf-8")
387
+ _emit(
388
+ probe,
389
+ "workflow.file_finished",
390
+ workflow_kind="page_index",
391
+ source_path=str(source_path),
392
+ output_dir=str(output_dir),
393
+ mode=mode,
394
+ summary_path=str(summary_path),
395
+ )
396
+ return PageIndexWorkflowCommandResult(
397
+ kind="page_index",
398
+ input_path=source_path,
399
+ output_dir=output_dir,
400
+ status="succeeded",
401
+ probe_path=probe.path,
402
+ summary_path=summary_path,
403
+ extra=summary,
404
+ result=result,
405
+ )
406
+
407
+
408
+ def run_page_index_batch_workflow(
409
+ source_paths: Sequence[str | Path],
410
+ *,
411
+ output_dir: str | Path,
412
+ mode: str = "heuristic",
413
+ source_format: str = "auto",
414
+ provider_settings: WorkflowProviderSettings | None = None,
415
+ probe: WorkflowProbe | None = None,
416
+ ) -> list[PageIndexWorkflowCommandResult]:
417
+ output_dir = Path(output_dir)
418
+ output_dir.mkdir(parents=True, exist_ok=True)
419
+ probe = _ensure_probe(output_dir, probe)
420
+ files = discover_input_files(source_paths, allowed_suffixes=PAGE_INDEX_SUFFIXES, recursive=True)
421
+ _emit(
422
+ probe,
423
+ "workflow.batch_started",
424
+ workflow_kind="page_index",
425
+ file_count=len(files),
426
+ output_dir=str(output_dir),
427
+ )
428
+ results: list[PageIndexWorkflowCommandResult] = []
429
+ for source_path in files:
430
+ result = run_page_index_source_workflow(
431
+ source_path,
432
+ output_dir=output_dir / source_path.stem,
433
+ mode=mode,
434
+ source_format=source_format,
435
+ provider_settings=provider_settings,
436
+ probe=probe,
437
+ )
438
+ results.append(result)
439
+ _emit(probe, "workflow.batch_finished", workflow_kind="page_index", file_count=len(files), output_dir=str(output_dir))
440
+ return results
441
+
442
+
443
+ def run_layerwise_source_workflow(
444
+ source_path: str | Path,
445
+ *,
446
+ output_dir: str | Path,
447
+ parsing_mode: str = "snippet",
448
+ max_depth: int = 10,
449
+ probe: WorkflowProbe | None = None,
450
+ ) -> LayerwiseWorkflowCommandResult:
451
+ source_path = Path(source_path)
452
+ output_dir = Path(output_dir)
453
+ output_dir.mkdir(parents=True, exist_ok=True)
454
+ probe = _ensure_probe(output_dir, probe)
455
+ _emit(
456
+ probe,
457
+ "workflow.file_started",
458
+ workflow_kind="layerwise",
459
+ source_path=str(source_path),
460
+ output_dir=str(output_dir),
461
+ parsing_mode=parsing_mode,
462
+ )
463
+ if not source_path.is_dir():
464
+ raise ValueError("layerwise workflow expects a directory of legacy OCR page artifacts")
465
+ from kg_doc_parser.ocr import regen_doc
466
+ from kg_doc_parser.semantic_document_splitting_layerwise_edits import (
467
+ semantic_tree_to_kge_payload as legacy_semantic_tree_to_kge_payload,
468
+ )
469
+
470
+ raw_doc = {source_path.name: regen_doc(str(source_path), use_raw=True)}
471
+ tree, source_map = parse_tree_document(
472
+ doc_id=source_path.name,
473
+ raw_doc_dict=raw_doc,
474
+ parsing_mode=parsing_mode, # type: ignore[arg-type]
475
+ max_depth=max_depth,
476
+ )
477
+ graph_payload = legacy_semantic_tree_to_kge_payload(tree, doc_id=source_path.name)
478
+ graph_path = output_dir / "layerwise-graph.json"
479
+ graph_path.write_text(json.dumps(graph_payload, indent=2), encoding="utf-8")
480
+ summary = {
481
+ "kind": "layerwise",
482
+ "source_path": str(source_path),
483
+ "node_count": len(graph_payload.get("nodes", [])),
484
+ "edge_count": len(graph_payload.get("edges", [])),
485
+ "source_count": len(source_map),
486
+ "graph_path": str(graph_path),
487
+ }
488
+ summary_path = output_dir / "layerwise-summary.json"
489
+ summary_path.write_text(json.dumps(summary, indent=2), encoding="utf-8")
490
+ _emit(
491
+ probe,
492
+ "workflow.file_finished",
493
+ workflow_kind="layerwise",
494
+ source_path=str(source_path),
495
+ output_dir=str(output_dir),
496
+ summary_path=str(summary_path),
497
+ )
498
+ return LayerwiseWorkflowCommandResult(
499
+ kind="layerwise",
500
+ input_path=source_path,
501
+ output_dir=output_dir,
502
+ status="succeeded",
503
+ probe_path=probe.path,
504
+ summary_path=summary_path,
505
+ extra=summary,
506
+ tree=tree,
507
+ source_map=source_map,
508
+ graph_payload=graph_payload,
509
+ )
510
+
511
+
512
+ def run_layerwise_batch_workflow(
513
+ source_paths: Sequence[str | Path],
514
+ *,
515
+ output_dir: str | Path,
516
+ parsing_mode: str = "snippet",
517
+ max_depth: int = 10,
518
+ probe: WorkflowProbe | None = None,
519
+ ) -> list[LayerwiseWorkflowCommandResult]:
520
+ output_dir = Path(output_dir)
521
+ output_dir.mkdir(parents=True, exist_ok=True)
522
+ probe = _ensure_probe(output_dir, probe)
523
+ dirs = [Path(path) for path in source_paths if Path(path).is_dir()]
524
+ _emit(
525
+ probe,
526
+ "workflow.batch_started",
527
+ workflow_kind="layerwise",
528
+ file_count=len(dirs),
529
+ output_dir=str(output_dir),
530
+ )
531
+ results: list[LayerwiseWorkflowCommandResult] = []
532
+ for source_path in dirs:
533
+ result = run_layerwise_source_workflow(
534
+ source_path,
535
+ output_dir=output_dir / source_path.name,
536
+ parsing_mode=parsing_mode,
537
+ max_depth=max_depth,
538
+ probe=probe,
539
+ )
540
+ results.append(result)
541
+ _emit(probe, "workflow.batch_finished", workflow_kind="layerwise", file_count=len(dirs), output_dir=str(output_dir))
542
+ return results
543
+
544
+
545
+ def run_demo_harness_workflow(config: DemoHarnessConfig):
546
+ return run_demo_harness(config)