graph-knowledge-doc-parser 0.2.4__tar.gz → 0.2.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/PKG-INFO +27 -6
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/README.md +17 -3
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/document_ingester_logger.py +6 -5
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/__init__.py +6 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/cli.py +17 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/handlers.py +52 -6
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/layerwise_llm.py +282 -69
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/models.py +31 -4
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/ocr_pipeline.py +93 -20
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/page_index.py +609 -49
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/parser_core.py +151 -34
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/parsing.py +8 -2
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/providers.py +326 -15
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/runners.py +5 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/semantics.py +25 -12
- graph_knowledge_doc_parser-0.2.5/kg_doc_parser/workflow_ingest/serialization.py +58 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/strategy.py +45 -15
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/pyproject.toml +12 -3
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/__init__.py +0 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/cast_hinting.py +0 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/document_ingest_log_config.py +0 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/llm_structured_output.py +0 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/models.py +0 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/ocr.py +0 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/pdf2png.py +0 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/semantic_document_splitting_layerwise_edits.py +0 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/text_processing_utils.py +0 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/utils/__init__.py +0 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/utils/bounded_threadpool_executor.py +0 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/utils/file_loaders.py +0 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/utils/langchain.py +0 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/utils/log.py +0 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/utils/version_chaining.py +0 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/_kogwistar.py +0 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/adapters.py +0 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/cache.py +0 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/clients.py +0 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/demo_harness.py +0 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/design.py +0 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/layered_contracts.py +0 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/probe.py +0 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/service.py +0 -0
- {graph_knowledge_doc_parser-0.2.4 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/smoke_assets.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: graph-knowledge-doc-parser
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.5
|
|
4
4
|
Summary: doc parser using llm driven engine with graph knowledge awareness
|
|
5
5
|
Author: humblemat810
|
|
6
6
|
Author-email: 67593116+humblemat810@users.noreply.github.com
|
|
@@ -14,11 +14,18 @@ Classifier: Programming Language :: Python :: 3.14
|
|
|
14
14
|
Classifier: Programming Language :: Python :: 3.15
|
|
15
15
|
Classifier: Programming Language :: Python :: 3 :: Only
|
|
16
16
|
Classifier: Topic :: Text Processing :: General
|
|
17
|
+
Provides-Extra: azure
|
|
18
|
+
Provides-Extra: cloud
|
|
19
|
+
Provides-Extra: gemini
|
|
20
|
+
Provides-Extra: openai
|
|
21
|
+
Provides-Extra: vertex
|
|
17
22
|
Requires-Dist: diskcache (>=5.6,<6.0) ; platform_python_implementation == "PyPy"
|
|
18
23
|
Requires-Dist: joblib (>=1.5.3,<2.0.0) ; platform_python_implementation == "CPython"
|
|
19
|
-
Requires-Dist: kogwistar (==0.6.
|
|
24
|
+
Requires-Dist: kogwistar (==0.6.4)
|
|
20
25
|
Requires-Dist: langchain-core (>=1.2.5,<2.0.0)
|
|
21
|
-
Requires-Dist: langchain-google-genai (>=4.1.2,<5.0.0)
|
|
26
|
+
Requires-Dist: langchain-google-genai (>=4.1.2,<5.0.0) ; extra == "gemini" or extra == "cloud"
|
|
27
|
+
Requires-Dist: langchain-google-vertexai (>=3.2.4,<4.0.0) ; extra == "vertex" or extra == "cloud"
|
|
28
|
+
Requires-Dist: langchain-openai (>=1.6.7,<2.0.0) ; extra == "openai" or extra == "azure" or extra == "cloud"
|
|
22
29
|
Requires-Dist: mcp (>=2.2.0,<3.0.0)
|
|
23
30
|
Requires-Dist: pathspec (>=0.12.1,<0.13.0)
|
|
24
31
|
Requires-Dist: pdf2image (>=1.17.0,<2.0.0)
|
|
@@ -73,7 +80,7 @@ disabled for that layer and the workflow routes through the remaining methods
|
|
|
73
80
|
before reaching explicit parse failure. PageIndex is a one-layer structural
|
|
74
81
|
fallback that preserves exact source pointers and returns expandable children
|
|
75
82
|
to normal strategy selection. See the
|
|
76
|
-
[`0.2.
|
|
83
|
+
[`0.2.5` release note](doc/release_0.2.5.md) and the
|
|
77
84
|
[progressive refinement ADR](doc/adr_progressive_refinement_strategy_arbitration.md).
|
|
78
85
|
The deterministic and text-only Bonsai fixture evaluation is recorded in the
|
|
79
86
|
[`PageIndex adversarial fixture report`](doc/page_index_adversarial_fixture_report.md).
|
|
@@ -211,7 +218,7 @@ The project currently expects or optionally uses:
|
|
|
211
218
|
- `split_raw_file_list`: optional allow-list file for PDF splitting runs
|
|
212
219
|
- `answer_export_list`: optional export list path used by local workflows
|
|
213
220
|
|
|
214
|
-
An example template is provided in [`.env.example`](
|
|
221
|
+
An example template is provided in [`.env.example`](.env.example).
|
|
215
222
|
|
|
216
223
|
## Provider Guide
|
|
217
224
|
|
|
@@ -255,6 +262,20 @@ and structured extraction.
|
|
|
255
262
|
- `KG_DOC_PARSER_PROJECT=my-project`
|
|
256
263
|
- `KG_DOC_PARSER_LOCATION=us-central1`
|
|
257
264
|
|
|
265
|
+
Gemini, OpenAI/Azure, and Vertex adapters are optional so local-only
|
|
266
|
+
installations do not pull their cloud SDK trees:
|
|
267
|
+
|
|
268
|
+
```powershell
|
|
269
|
+
poetry install -E openai -E gemini -E vertex
|
|
270
|
+
# or, after building/installing the package:
|
|
271
|
+
python -m pip install "graph-knowledge-doc-parser[openai,gemini,vertex]"
|
|
272
|
+
```
|
|
273
|
+
|
|
274
|
+
Installing an adapter does not create credentials, a cloud project, or a paid
|
|
275
|
+
service. Use the no-charge offline provider contract tests for CI validation;
|
|
276
|
+
only run a live provider smoke test when the account, endpoint, model, and
|
|
277
|
+
zero-cost allowance are explicitly confirmed.
|
|
278
|
+
|
|
258
279
|
### Recipe Parsing Example
|
|
259
280
|
|
|
260
281
|
If you are parsing a cooking recipe, one practical split is:
|
|
@@ -343,7 +364,7 @@ There is now a manual workflow-ingest demo harness that can run the end-to-end f
|
|
|
343
364
|
- an already running external Kogwistar server
|
|
344
365
|
|
|
345
366
|
The legacy semantic-smoke test in
|
|
346
|
-
[`tests/test_semantic_layerwise_doc_parsing.py`](
|
|
367
|
+
[`tests/test_semantic_layerwise_doc_parsing.py`](tests/test_semantic_layerwise_doc_parsing.py)
|
|
347
368
|
also expects a live Kogwistar server at `http://127.0.0.1:28110`. It does not
|
|
348
369
|
start that server for you, so use the VS Code server launch config or start it
|
|
349
370
|
manually before running the Ollama case.
|
|
@@ -40,7 +40,7 @@ disabled for that layer and the workflow routes through the remaining methods
|
|
|
40
40
|
before reaching explicit parse failure. PageIndex is a one-layer structural
|
|
41
41
|
fallback that preserves exact source pointers and returns expandable children
|
|
42
42
|
to normal strategy selection. See the
|
|
43
|
-
[`0.2.
|
|
43
|
+
[`0.2.5` release note](doc/release_0.2.5.md) and the
|
|
44
44
|
[progressive refinement ADR](doc/adr_progressive_refinement_strategy_arbitration.md).
|
|
45
45
|
The deterministic and text-only Bonsai fixture evaluation is recorded in the
|
|
46
46
|
[`PageIndex adversarial fixture report`](doc/page_index_adversarial_fixture_report.md).
|
|
@@ -178,7 +178,7 @@ The project currently expects or optionally uses:
|
|
|
178
178
|
- `split_raw_file_list`: optional allow-list file for PDF splitting runs
|
|
179
179
|
- `answer_export_list`: optional export list path used by local workflows
|
|
180
180
|
|
|
181
|
-
An example template is provided in [`.env.example`](
|
|
181
|
+
An example template is provided in [`.env.example`](.env.example).
|
|
182
182
|
|
|
183
183
|
## Provider Guide
|
|
184
184
|
|
|
@@ -222,6 +222,20 @@ and structured extraction.
|
|
|
222
222
|
- `KG_DOC_PARSER_PROJECT=my-project`
|
|
223
223
|
- `KG_DOC_PARSER_LOCATION=us-central1`
|
|
224
224
|
|
|
225
|
+
Gemini, OpenAI/Azure, and Vertex adapters are optional so local-only
|
|
226
|
+
installations do not pull their cloud SDK trees:
|
|
227
|
+
|
|
228
|
+
```powershell
|
|
229
|
+
poetry install -E openai -E gemini -E vertex
|
|
230
|
+
# or, after building/installing the package:
|
|
231
|
+
python -m pip install "graph-knowledge-doc-parser[openai,gemini,vertex]"
|
|
232
|
+
```
|
|
233
|
+
|
|
234
|
+
Installing an adapter does not create credentials, a cloud project, or a paid
|
|
235
|
+
service. Use the no-charge offline provider contract tests for CI validation;
|
|
236
|
+
only run a live provider smoke test when the account, endpoint, model, and
|
|
237
|
+
zero-cost allowance are explicitly confirmed.
|
|
238
|
+
|
|
225
239
|
### Recipe Parsing Example
|
|
226
240
|
|
|
227
241
|
If you are parsing a cooking recipe, one practical split is:
|
|
@@ -310,7 +324,7 @@ There is now a manual workflow-ingest demo harness that can run the end-to-end f
|
|
|
310
324
|
- an already running external Kogwistar server
|
|
311
325
|
|
|
312
326
|
The legacy semantic-smoke test in
|
|
313
|
-
[`tests/test_semantic_layerwise_doc_parsing.py`](
|
|
327
|
+
[`tests/test_semantic_layerwise_doc_parsing.py`](tests/test_semantic_layerwise_doc_parsing.py)
|
|
314
328
|
also expects a live Kogwistar server at `http://127.0.0.1:28110`. It does not
|
|
315
329
|
start that server for you, so use the VS Code server launch config or start it
|
|
316
330
|
manually before running the Ollama case.
|
|
@@ -82,7 +82,6 @@ NOTES
|
|
|
82
82
|
|
|
83
83
|
from __future__ import annotations
|
|
84
84
|
|
|
85
|
-
import json
|
|
86
85
|
import os
|
|
87
86
|
import queue
|
|
88
87
|
import sqlite3
|
|
@@ -100,6 +99,8 @@ from langchain_core.outputs.llm_result import LLMResult
|
|
|
100
99
|
from langchain_core.messages import BaseMessage
|
|
101
100
|
from uuid import UUID
|
|
102
101
|
|
|
102
|
+
from kg_doc_parser.workflow_ingest.serialization import safe_json_dumps
|
|
103
|
+
|
|
103
104
|
# ---------------------------
|
|
104
105
|
# Pricing / cost calculation
|
|
105
106
|
# ---------------------------
|
|
@@ -537,7 +538,7 @@ class DocumentIngestSQLiteCallback(BaseCallbackHandler):
|
|
|
537
538
|
token_count=0,
|
|
538
539
|
cost_usd=0.0,
|
|
539
540
|
n_try=n_try,
|
|
540
|
-
metadata_json=
|
|
541
|
+
metadata_json=safe_json_dumps(payload, ensure_ascii=False),
|
|
541
542
|
)
|
|
542
543
|
)
|
|
543
544
|
def on_llm_start(
|
|
@@ -594,7 +595,7 @@ class DocumentIngestSQLiteCallback(BaseCallbackHandler):
|
|
|
594
595
|
token_count=0,
|
|
595
596
|
cost_usd=0.0,
|
|
596
597
|
n_try = payload.get('n_try', 0),
|
|
597
|
-
metadata_json=
|
|
598
|
+
metadata_json=safe_json_dumps(payload, ensure_ascii=False),
|
|
598
599
|
)
|
|
599
600
|
)
|
|
600
601
|
|
|
@@ -710,7 +711,7 @@ class DocumentIngestSQLiteCallback(BaseCallbackHandler):
|
|
|
710
711
|
token_count=token_count,
|
|
711
712
|
cost_usd=float(cost_usd),
|
|
712
713
|
n_try=float(payload.get('n_try', 0)),
|
|
713
|
-
metadata_json=
|
|
714
|
+
metadata_json=safe_json_dumps(payload, ensure_ascii=False),
|
|
714
715
|
)
|
|
715
716
|
)
|
|
716
717
|
|
|
@@ -777,6 +778,6 @@ class DocumentIngestSQLiteCallback(BaseCallbackHandler):
|
|
|
777
778
|
token_count=0,
|
|
778
779
|
cost_usd=0.0,
|
|
779
780
|
n_try=float(payload.get('n_try', 0)),
|
|
780
|
-
metadata_json=
|
|
781
|
+
metadata_json=safe_json_dumps(payload, ensure_ascii=False),
|
|
781
782
|
)
|
|
782
783
|
)
|
|
@@ -53,6 +53,8 @@ from .page_index import (
|
|
|
53
53
|
BlockAssignment,
|
|
54
54
|
BlockAssignmentBatch,
|
|
55
55
|
CandidateBlock,
|
|
56
|
+
HierarchicalSummaryAssignment,
|
|
57
|
+
HierarchicalSummaryBatch,
|
|
56
58
|
PageIndexBlockSpec,
|
|
57
59
|
PageIndexParseResult,
|
|
58
60
|
PageIndexValidationResult,
|
|
@@ -88,6 +90,7 @@ from .providers import (
|
|
|
88
90
|
build_chat_model,
|
|
89
91
|
build_chat_model_for_role,
|
|
90
92
|
build_embedding_function,
|
|
93
|
+
provider_call_metrics_snapshot,
|
|
91
94
|
)
|
|
92
95
|
from .runners import (
|
|
93
96
|
LayerwiseWorkflowCommandResult,
|
|
@@ -137,6 +140,8 @@ __all__ = [
|
|
|
137
140
|
"EmbeddingProviderConfig",
|
|
138
141
|
"FakeChatModel",
|
|
139
142
|
"GroundedSourceRecord",
|
|
143
|
+
"HierarchicalSummaryAssignment",
|
|
144
|
+
"HierarchicalSummaryBatch",
|
|
140
145
|
"HydratedTextPointer",
|
|
141
146
|
"IngestExecutionClient",
|
|
142
147
|
"IngestRunHandle",
|
|
@@ -210,6 +215,7 @@ __all__ = [
|
|
|
210
215
|
"prepare_layer_frontier",
|
|
211
216
|
"prepare_ocr_workflow_input",
|
|
212
217
|
"propose_layer_breakdown",
|
|
218
|
+
"provider_call_metrics_snapshot",
|
|
213
219
|
"review_layer",
|
|
214
220
|
"run_demo_harness",
|
|
215
221
|
"run_demo_harness_workflow",
|
|
@@ -50,6 +50,9 @@ def _provider_settings_from_args(args: argparse.Namespace) -> WorkflowProviderSe
|
|
|
50
50
|
parse_strategy_order = getattr(args, "parse_strategy_order", None)
|
|
51
51
|
triage_enabled = getattr(args, "triage_enabled", None)
|
|
52
52
|
page_index_summary_enabled = getattr(args, "page_index_summary_enabled", None)
|
|
53
|
+
page_index_hierarchical_summary_enabled = getattr(
|
|
54
|
+
args, "page_index_hierarchical_summary_enabled", None
|
|
55
|
+
)
|
|
53
56
|
ocr_override = any(value is not None for value in ocr_values.values())
|
|
54
57
|
parser_override = any(value is not None for value in parser_values.values())
|
|
55
58
|
proposal_override = proposal_mode is not None
|
|
@@ -57,6 +60,7 @@ def _provider_settings_from_args(args: argparse.Namespace) -> WorkflowProviderSe
|
|
|
57
60
|
strategy_order_override = parse_strategy_order is not None
|
|
58
61
|
triage_override = triage_enabled is not None
|
|
59
62
|
summary_override = page_index_summary_enabled is not None
|
|
63
|
+
hierarchical_summary_override = page_index_hierarchical_summary_enabled is not None
|
|
60
64
|
if (
|
|
61
65
|
not ocr_override
|
|
62
66
|
and not parser_override
|
|
@@ -65,6 +69,7 @@ def _provider_settings_from_args(args: argparse.Namespace) -> WorkflowProviderSe
|
|
|
65
69
|
and not strategy_order_override
|
|
66
70
|
and not triage_override
|
|
67
71
|
and not summary_override
|
|
72
|
+
and not hierarchical_summary_override
|
|
68
73
|
):
|
|
69
74
|
return None
|
|
70
75
|
settings = WorkflowProviderSettings.from_env()
|
|
@@ -80,6 +85,12 @@ def _provider_settings_from_args(args: argparse.Namespace) -> WorkflowProviderSe
|
|
|
80
85
|
settings = settings.model_copy(update={"triage_enabled": triage_enabled})
|
|
81
86
|
if summary_override:
|
|
82
87
|
settings = settings.model_copy(update={"page_index_summary_enabled": page_index_summary_enabled})
|
|
88
|
+
if hierarchical_summary_override:
|
|
89
|
+
settings = settings.model_copy(
|
|
90
|
+
update={
|
|
91
|
+
"page_index_hierarchical_summary_enabled": page_index_hierarchical_summary_enabled
|
|
92
|
+
}
|
|
93
|
+
)
|
|
83
94
|
if ocr_override:
|
|
84
95
|
settings = settings.model_copy(
|
|
85
96
|
update={
|
|
@@ -125,6 +136,12 @@ def _add_provider_args(parser: argparse.ArgumentParser) -> None:
|
|
|
125
136
|
default=None,
|
|
126
137
|
help="Enable or disable summaries on PageIndex nodes for this parse call",
|
|
127
138
|
)
|
|
139
|
+
group.add_argument(
|
|
140
|
+
"--page-index-hierarchical-summary-enabled",
|
|
141
|
+
action=argparse.BooleanOptionalAction,
|
|
142
|
+
default=None,
|
|
143
|
+
help="Enable or disable the bounded parent-summary context pass",
|
|
144
|
+
)
|
|
128
145
|
group.add_argument("--ocr-provider", default=None)
|
|
129
146
|
group.add_argument("--ocr-model", default=None)
|
|
130
147
|
group.add_argument("--ocr-base-url", default=None)
|
|
@@ -2,7 +2,7 @@ from __future__ import annotations
|
|
|
2
2
|
|
|
3
3
|
import logging
|
|
4
4
|
from collections.abc import Callable, Mapping
|
|
5
|
-
from typing import Protocol, TypedDict, cast
|
|
5
|
+
from typing import Literal, Protocol, TypedDict, cast
|
|
6
6
|
|
|
7
7
|
from kogwistar.runtime import MappingStepResolver
|
|
8
8
|
from kogwistar.runtime.models import RunFailure, RunSuccess, RunSuspended, StepRunResult
|
|
@@ -28,7 +28,7 @@ from .models import (
|
|
|
28
28
|
WorkflowExportBundle,
|
|
29
29
|
WorkflowIngestInput,
|
|
30
30
|
)
|
|
31
|
-
from .page_index import parse_page_index_layer
|
|
31
|
+
from .page_index import PageIndexSourceFormat, parse_page_index_layer
|
|
32
32
|
from .parser_core import (
|
|
33
33
|
ParseSemanticFn,
|
|
34
34
|
ProposeLayerFn,
|
|
@@ -64,6 +64,21 @@ from .strategy import (
|
|
|
64
64
|
_LOGGER = logging.getLogger(__name__)
|
|
65
65
|
|
|
66
66
|
|
|
67
|
+
def _page_index_source_format(inp: WorkflowIngestInput) -> PageIndexSourceFormat:
|
|
68
|
+
"""Read the document format carried by the normalized collection metadata."""
|
|
69
|
+
|
|
70
|
+
collection = select_primary_collection(inp)
|
|
71
|
+
candidates: list[object] = [collection.metadata.get("source_format")]
|
|
72
|
+
for page in collection.pages:
|
|
73
|
+
candidates.append(page.metadata.get("source_format"))
|
|
74
|
+
for unit in page.units:
|
|
75
|
+
candidates.append(unit.metadata.get("source_format"))
|
|
76
|
+
for value in candidates:
|
|
77
|
+
if value in {"text", "markdown"}:
|
|
78
|
+
return cast(Literal["text", "markdown"], value)
|
|
79
|
+
return "text"
|
|
80
|
+
|
|
81
|
+
|
|
67
82
|
class StepHandler(Protocol):
|
|
68
83
|
"""Execute one parser workflow step against the runtime context."""
|
|
69
84
|
|
|
@@ -92,9 +107,11 @@ class WorkflowRuntimeDeps(TypedDict, total=False):
|
|
|
92
107
|
split_strategy: SplitStrategy
|
|
93
108
|
fallback_split_strategy: SplitStrategy
|
|
94
109
|
max_review_retries: int
|
|
110
|
+
layer_frontier_batch_size: int
|
|
95
111
|
coverage_threshold: float
|
|
96
112
|
provider_settings: WorkflowProviderSettings
|
|
97
113
|
triage_strategy_fn: Callable[[dict[str, object]], object]
|
|
114
|
+
provider_diagnostics_sink: Callable[[dict[str, object]], None]
|
|
98
115
|
|
|
99
116
|
|
|
100
117
|
def _build_export_bundle(
|
|
@@ -363,10 +380,22 @@ def register_layerwise_parser_steps(
|
|
|
363
380
|
if normalized_input.page_index_summary_enabled is not None
|
|
364
381
|
else (settings.page_index_summary_enabled if settings is not None else True)
|
|
365
382
|
)
|
|
383
|
+
page_index_hierarchical_summary_enabled = (
|
|
384
|
+
normalized_input.page_index_hierarchical_summary_enabled
|
|
385
|
+
if normalized_input.page_index_hierarchical_summary_enabled is not None
|
|
386
|
+
else (
|
|
387
|
+
settings.page_index_hierarchical_summary_enabled
|
|
388
|
+
if settings is not None
|
|
389
|
+
else False
|
|
390
|
+
)
|
|
391
|
+
)
|
|
366
392
|
triage_build_error: str | None = None
|
|
367
393
|
if triage_fn is None and settings is not None and triage_enabled and requested == "auto":
|
|
368
394
|
try:
|
|
369
|
-
triage_fn = build_llm_strategy_triage(
|
|
395
|
+
triage_fn = build_llm_strategy_triage(
|
|
396
|
+
settings,
|
|
397
|
+
diagnostics_sink=runtime_deps.get("provider_diagnostics_sink"),
|
|
398
|
+
)
|
|
370
399
|
except Exception as exc: # noqa: BLE001 - unavailable providers use deterministic fallback.
|
|
371
400
|
triage_build_error = f"triage provider unavailable: {type(exc).__name__}: {exc}"
|
|
372
401
|
parent_context: list[dict[str, object]] = []
|
|
@@ -434,6 +463,7 @@ def register_layerwise_parser_steps(
|
|
|
434
463
|
"parse_strategy_fallback_order": list(decision.fallback_order),
|
|
435
464
|
"disabled_strategies": sorted(disabled_strategies),
|
|
436
465
|
"page_index_summary_enabled": page_index_summary_enabled,
|
|
466
|
+
"page_index_hierarchical_summary_enabled": page_index_hierarchical_summary_enabled,
|
|
437
467
|
}
|
|
438
468
|
)
|
|
439
469
|
selected_split_strategy = (
|
|
@@ -483,6 +513,7 @@ def register_layerwise_parser_steps(
|
|
|
483
513
|
"disabled_strategies": sorted(disabled_strategies),
|
|
484
514
|
"page_index_attempted": False,
|
|
485
515
|
"page_index_summary_enabled": page_index_summary_enabled,
|
|
516
|
+
"page_index_hierarchical_summary_enabled": page_index_hierarchical_summary_enabled,
|
|
486
517
|
},
|
|
487
518
|
"retry_count": 0,
|
|
488
519
|
}
|
|
@@ -515,7 +546,9 @@ def register_layerwise_parser_steps(
|
|
|
515
546
|
@_register_step(resolver, step_name="page_index_layer", runtime_deps=runtime_deps)
|
|
516
547
|
def _page_index_layer(ctx: StepContext) -> StepRunResult:
|
|
517
548
|
current_layer_context = CurrentLayerContext.model_validate(ctx.state_view["current_layer_context"])
|
|
549
|
+
normalized_input = WorkflowIngestInput.model_validate(ctx.state_view["normalized_input"])
|
|
518
550
|
parser_source_map = ctx.state_view.get("parser_source_map") or {}
|
|
551
|
+
source_format = _page_index_source_format(normalized_input)
|
|
519
552
|
candidates = []
|
|
520
553
|
for parent_id, parent_title in zip(
|
|
521
554
|
current_layer_context.parent_node_ids,
|
|
@@ -527,7 +560,7 @@ def register_layerwise_parser_steps(
|
|
|
527
560
|
parent_title=parent_title,
|
|
528
561
|
parent_pointers=current_layer_context.parent_content_pointers_by_id.get(parent_id, []),
|
|
529
562
|
parser_source_map=parser_source_map,
|
|
530
|
-
source_format=
|
|
563
|
+
source_format=source_format,
|
|
531
564
|
summary_enabled=bool(current_layer_context.metadata.get("page_index_summary_enabled", True)),
|
|
532
565
|
)
|
|
533
566
|
)
|
|
@@ -568,6 +601,14 @@ def register_layerwise_parser_steps(
|
|
|
568
601
|
def _prepare_layer_frontier(ctx: StepContext) -> StepRunResult:
|
|
569
602
|
parse_session = ParseSessionState.model_validate(ctx.state_view["parse_session"])
|
|
570
603
|
semantic_tree = SemanticNode.model_validate(ctx.state_view["semantic_tree"])
|
|
604
|
+
configured_batch_size = runtime_deps.get("layer_frontier_batch_size")
|
|
605
|
+
if configured_batch_size is None:
|
|
606
|
+
settings = runtime_deps.get("provider_settings")
|
|
607
|
+
configured_batch_size = (
|
|
608
|
+
getattr(settings, "layer_frontier_batch_size", None)
|
|
609
|
+
if settings is not None
|
|
610
|
+
else 1
|
|
611
|
+
)
|
|
571
612
|
context, remaining, updated_session = prepare_layer_frontier(
|
|
572
613
|
parse_session=parse_session,
|
|
573
614
|
frontier_queue=[
|
|
@@ -576,6 +617,7 @@ def register_layerwise_parser_steps(
|
|
|
576
617
|
],
|
|
577
618
|
semantic_tree=semantic_tree,
|
|
578
619
|
max_retries=int(runtime_deps.get("max_review_retries", 3)),
|
|
620
|
+
max_items=(int(configured_batch_size) if configured_batch_size is not None else None),
|
|
579
621
|
)
|
|
580
622
|
with ctx.state_write as st:
|
|
581
623
|
st["parse_session"] = updated_session.model_dump(field_mode="backend", dump_format="json")
|
|
@@ -731,6 +773,7 @@ def register_layerwise_parser_steps(
|
|
|
731
773
|
parent_node_ids=list(current_layer_context.parent_node_ids),
|
|
732
774
|
attempt=int((ctx.state_view.get("strategy_attempt_counts") or {}).get(strategy, 1)),
|
|
733
775
|
event="failed",
|
|
776
|
+
failure_type="retry",
|
|
734
777
|
reasons=reasons[:12],
|
|
735
778
|
)
|
|
736
779
|
if remaining:
|
|
@@ -844,7 +887,10 @@ def register_layerwise_parser_steps(
|
|
|
844
887
|
|
|
845
888
|
@_register_step(resolver, step_name="finalize_semantic_tree", runtime_deps=runtime_deps)
|
|
846
889
|
def _finalize_semantic_tree(ctx: StepContext) -> StepRunResult:
|
|
847
|
-
tree = finalize_semantic_tree(
|
|
890
|
+
tree = finalize_semantic_tree(
|
|
891
|
+
SemanticNode.model_validate(ctx.state_view["semantic_tree"]),
|
|
892
|
+
parser_source_map=ctx.state_view.get("parser_source_map") or {},
|
|
893
|
+
)
|
|
848
894
|
with ctx.state_write as st:
|
|
849
895
|
st["semantic_tree"] = tree.model_dump()
|
|
850
896
|
return _success("validate_tree")
|
|
@@ -872,7 +918,7 @@ def register_postparse_steps(
|
|
|
872
918
|
validation_notes=[],
|
|
873
919
|
)
|
|
874
920
|
bundle = _build_export_bundle(ctx=ctx, runtime_deps=runtime_deps)
|
|
875
|
-
threshold = float(runtime_deps.get("coverage_threshold", 0
|
|
921
|
+
threshold = float(runtime_deps.get("coverage_threshold", 1.0))
|
|
876
922
|
if report.overall_text_coverage < threshold:
|
|
877
923
|
error_message = (
|
|
878
924
|
f"text coverage below threshold: {report.overall_text_coverage:.3f} < {threshold:.3f}"
|