graph-knowledge-doc-parser 0.2.2__tar.gz → 0.2.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/PKG-INFO +39 -5
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/README.md +29 -2
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/document_ingester_logger.py +6 -5
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/__init__.py +125 -93
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/cli.py +67 -1
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/design.py +27 -12
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/handlers.py +390 -35
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/layerwise_llm.py +290 -59
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/models.py +92 -5
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/ocr_pipeline.py +92 -15
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/page_index.py +836 -64
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/parser_core.py +171 -24
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/parsing.py +11 -2
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/providers.py +415 -10
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/runners.py +5 -0
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/semantics.py +40 -14
- graph_knowledge_doc_parser-0.2.5/kg_doc_parser/workflow_ingest/serialization.py +58 -0
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/service.py +74 -2
- graph_knowledge_doc_parser-0.2.5/kg_doc_parser/workflow_ingest/strategy.py +239 -0
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/pyproject.toml +12 -3
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/__init__.py +0 -0
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/cast_hinting.py +0 -0
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/document_ingest_log_config.py +0 -0
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/llm_structured_output.py +0 -0
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/models.py +0 -0
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/ocr.py +0 -0
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/pdf2png.py +0 -0
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/semantic_document_splitting_layerwise_edits.py +0 -0
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/text_processing_utils.py +0 -0
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/utils/__init__.py +0 -0
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/utils/bounded_threadpool_executor.py +0 -0
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/utils/file_loaders.py +0 -0
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/utils/langchain.py +0 -0
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/utils/log.py +0 -0
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/utils/version_chaining.py +0 -0
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/_kogwistar.py +0 -0
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/adapters.py +0 -0
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/cache.py +0 -0
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/clients.py +0 -0
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/demo_harness.py +0 -0
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/layered_contracts.py +0 -0
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/probe.py +0 -0
- {graph_knowledge_doc_parser-0.2.2 → graph_knowledge_doc_parser-0.2.5}/kg_doc_parser/workflow_ingest/smoke_assets.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: graph-knowledge-doc-parser
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.5
|
|
4
4
|
Summary: doc parser using llm driven engine with graph knowledge awareness
|
|
5
5
|
Author: humblemat810
|
|
6
6
|
Author-email: 67593116+humblemat810@users.noreply.github.com
|
|
@@ -14,11 +14,18 @@ Classifier: Programming Language :: Python :: 3.14
|
|
|
14
14
|
Classifier: Programming Language :: Python :: 3.15
|
|
15
15
|
Classifier: Programming Language :: Python :: 3 :: Only
|
|
16
16
|
Classifier: Topic :: Text Processing :: General
|
|
17
|
+
Provides-Extra: azure
|
|
18
|
+
Provides-Extra: cloud
|
|
19
|
+
Provides-Extra: gemini
|
|
20
|
+
Provides-Extra: openai
|
|
21
|
+
Provides-Extra: vertex
|
|
17
22
|
Requires-Dist: diskcache (>=5.6,<6.0) ; platform_python_implementation == "PyPy"
|
|
18
23
|
Requires-Dist: joblib (>=1.5.3,<2.0.0) ; platform_python_implementation == "CPython"
|
|
19
|
-
Requires-Dist: kogwistar (==0.6.
|
|
24
|
+
Requires-Dist: kogwistar (==0.6.4)
|
|
20
25
|
Requires-Dist: langchain-core (>=1.2.5,<2.0.0)
|
|
21
|
-
Requires-Dist: langchain-google-genai (>=4.1.2,<5.0.0)
|
|
26
|
+
Requires-Dist: langchain-google-genai (>=4.1.2,<5.0.0) ; extra == "gemini" or extra == "cloud"
|
|
27
|
+
Requires-Dist: langchain-google-vertexai (>=3.2.4,<4.0.0) ; extra == "vertex" or extra == "cloud"
|
|
28
|
+
Requires-Dist: langchain-openai (>=1.6.7,<2.0.0) ; extra == "openai" or extra == "azure" or extra == "cloud"
|
|
22
29
|
Requires-Dist: mcp (>=2.2.0,<3.0.0)
|
|
23
30
|
Requires-Dist: pathspec (>=0.12.1,<0.13.0)
|
|
24
31
|
Requires-Dist: pdf2image (>=1.17.0,<2.0.0)
|
|
@@ -65,6 +72,19 @@ The reusable helpers live under `src/workflow_ingest/` and are designed so the
|
|
|
65
72
|
same core logic can be called from tests, scripts, and higher-level workflow
|
|
66
73
|
code without duplicating orchestration.
|
|
67
74
|
|
|
75
|
+
Layerwise parsing selects a strategy independently for each frontier layer.
|
|
76
|
+
The default deterministic order is `layer_excerpt`, `layer_boundary`, then
|
|
77
|
+
`page_index`; callers may provide another complete order, and optional triage
|
|
78
|
+
may choose among the currently enabled strategies. Failed strategies are
|
|
79
|
+
disabled for that layer and the workflow routes through the remaining methods
|
|
80
|
+
before reaching explicit parse failure. PageIndex is a one-layer structural
|
|
81
|
+
fallback that preserves exact source pointers and returns expandable children
|
|
82
|
+
to normal strategy selection. See the
|
|
83
|
+
[`0.2.5` release note](doc/release_0.2.5.md) and the
|
|
84
|
+
[progressive refinement ADR](doc/adr_progressive_refinement_strategy_arbitration.md).
|
|
85
|
+
The deterministic and text-only Bonsai fixture evaluation is recorded in the
|
|
86
|
+
[`PageIndex adversarial fixture report`](doc/page_index_adversarial_fixture_report.md).
|
|
87
|
+
|
|
68
88
|
For an adoption path that starts with parser-grounded source units and later
|
|
69
89
|
adds LLM-Wiki cross-document maintenance, see
|
|
70
90
|
[`doc/progressive_adoption_guide.md`](doc/progressive_adoption_guide.md). The
|
|
@@ -198,7 +218,7 @@ The project currently expects or optionally uses:
|
|
|
198
218
|
- `split_raw_file_list`: optional allow-list file for PDF splitting runs
|
|
199
219
|
- `answer_export_list`: optional export list path used by local workflows
|
|
200
220
|
|
|
201
|
-
An example template is provided in [`.env.example`](
|
|
221
|
+
An example template is provided in [`.env.example`](.env.example).
|
|
202
222
|
|
|
203
223
|
## Provider Guide
|
|
204
224
|
|
|
@@ -242,6 +262,20 @@ and structured extraction.
|
|
|
242
262
|
- `KG_DOC_PARSER_PROJECT=my-project`
|
|
243
263
|
- `KG_DOC_PARSER_LOCATION=us-central1`
|
|
244
264
|
|
|
265
|
+
Gemini, OpenAI/Azure, and Vertex adapters are optional so local-only
|
|
266
|
+
installations do not pull their cloud SDK trees:
|
|
267
|
+
|
|
268
|
+
```powershell
|
|
269
|
+
poetry install -E openai -E gemini -E vertex
|
|
270
|
+
# or, after building/installing the package:
|
|
271
|
+
python -m pip install "graph-knowledge-doc-parser[openai,gemini,vertex]"
|
|
272
|
+
```
|
|
273
|
+
|
|
274
|
+
Installing an adapter does not create credentials, a cloud project, or a paid
|
|
275
|
+
service. Use the no-charge offline provider contract tests for CI validation;
|
|
276
|
+
only run a live provider smoke test when the account, endpoint, model, and
|
|
277
|
+
zero-cost allowance are explicitly confirmed.
|
|
278
|
+
|
|
245
279
|
### Recipe Parsing Example
|
|
246
280
|
|
|
247
281
|
If you are parsing a cooking recipe, one practical split is:
|
|
@@ -330,7 +364,7 @@ There is now a manual workflow-ingest demo harness that can run the end-to-end f
|
|
|
330
364
|
- an already running external Kogwistar server
|
|
331
365
|
|
|
332
366
|
The legacy semantic-smoke test in
|
|
333
|
-
[`tests/test_semantic_layerwise_doc_parsing.py`](
|
|
367
|
+
[`tests/test_semantic_layerwise_doc_parsing.py`](tests/test_semantic_layerwise_doc_parsing.py)
|
|
334
368
|
also expects a live Kogwistar server at `http://127.0.0.1:28110`. It does not
|
|
335
369
|
start that server for you, so use the VS Code server launch config or start it
|
|
336
370
|
manually before running the Ollama case.
|
|
@@ -32,6 +32,19 @@ The reusable helpers live under `src/workflow_ingest/` and are designed so the
|
|
|
32
32
|
same core logic can be called from tests, scripts, and higher-level workflow
|
|
33
33
|
code without duplicating orchestration.
|
|
34
34
|
|
|
35
|
+
Layerwise parsing selects a strategy independently for each frontier layer.
|
|
36
|
+
The default deterministic order is `layer_excerpt`, `layer_boundary`, then
|
|
37
|
+
`page_index`; callers may provide another complete order, and optional triage
|
|
38
|
+
may choose among the currently enabled strategies. Failed strategies are
|
|
39
|
+
disabled for that layer and the workflow routes through the remaining methods
|
|
40
|
+
before reaching explicit parse failure. PageIndex is a one-layer structural
|
|
41
|
+
fallback that preserves exact source pointers and returns expandable children
|
|
42
|
+
to normal strategy selection. See the
|
|
43
|
+
[`0.2.5` release note](doc/release_0.2.5.md) and the
|
|
44
|
+
[progressive refinement ADR](doc/adr_progressive_refinement_strategy_arbitration.md).
|
|
45
|
+
The deterministic and text-only Bonsai fixture evaluation is recorded in the
|
|
46
|
+
[`PageIndex adversarial fixture report`](doc/page_index_adversarial_fixture_report.md).
|
|
47
|
+
|
|
35
48
|
For an adoption path that starts with parser-grounded source units and later
|
|
36
49
|
adds LLM-Wiki cross-document maintenance, see
|
|
37
50
|
[`doc/progressive_adoption_guide.md`](doc/progressive_adoption_guide.md). The
|
|
@@ -165,7 +178,7 @@ The project currently expects or optionally uses:
|
|
|
165
178
|
- `split_raw_file_list`: optional allow-list file for PDF splitting runs
|
|
166
179
|
- `answer_export_list`: optional export list path used by local workflows
|
|
167
180
|
|
|
168
|
-
An example template is provided in [`.env.example`](
|
|
181
|
+
An example template is provided in [`.env.example`](.env.example).
|
|
169
182
|
|
|
170
183
|
## Provider Guide
|
|
171
184
|
|
|
@@ -209,6 +222,20 @@ and structured extraction.
|
|
|
209
222
|
- `KG_DOC_PARSER_PROJECT=my-project`
|
|
210
223
|
- `KG_DOC_PARSER_LOCATION=us-central1`
|
|
211
224
|
|
|
225
|
+
Gemini, OpenAI/Azure, and Vertex adapters are optional so local-only
|
|
226
|
+
installations do not pull their cloud SDK trees:
|
|
227
|
+
|
|
228
|
+
```powershell
|
|
229
|
+
poetry install -E openai -E gemini -E vertex
|
|
230
|
+
# or, after building/installing the package:
|
|
231
|
+
python -m pip install "graph-knowledge-doc-parser[openai,gemini,vertex]"
|
|
232
|
+
```
|
|
233
|
+
|
|
234
|
+
Installing an adapter does not create credentials, a cloud project, or a paid
|
|
235
|
+
service. Use the no-charge offline provider contract tests for CI validation;
|
|
236
|
+
only run a live provider smoke test when the account, endpoint, model, and
|
|
237
|
+
zero-cost allowance are explicitly confirmed.
|
|
238
|
+
|
|
212
239
|
### Recipe Parsing Example
|
|
213
240
|
|
|
214
241
|
If you are parsing a cooking recipe, one practical split is:
|
|
@@ -297,7 +324,7 @@ There is now a manual workflow-ingest demo harness that can run the end-to-end f
|
|
|
297
324
|
- an already running external Kogwistar server
|
|
298
325
|
|
|
299
326
|
The legacy semantic-smoke test in
|
|
300
|
-
[`tests/test_semantic_layerwise_doc_parsing.py`](
|
|
327
|
+
[`tests/test_semantic_layerwise_doc_parsing.py`](tests/test_semantic_layerwise_doc_parsing.py)
|
|
301
328
|
also expects a live Kogwistar server at `http://127.0.0.1:28110`. It does not
|
|
302
329
|
start that server for you, so use the VS Code server launch config or start it
|
|
303
330
|
manually before running the Ollama case.
|
|
@@ -82,7 +82,6 @@ NOTES
|
|
|
82
82
|
|
|
83
83
|
from __future__ import annotations
|
|
84
84
|
|
|
85
|
-
import json
|
|
86
85
|
import os
|
|
87
86
|
import queue
|
|
88
87
|
import sqlite3
|
|
@@ -100,6 +99,8 @@ from langchain_core.outputs.llm_result import LLMResult
|
|
|
100
99
|
from langchain_core.messages import BaseMessage
|
|
101
100
|
from uuid import UUID
|
|
102
101
|
|
|
102
|
+
from kg_doc_parser.workflow_ingest.serialization import safe_json_dumps
|
|
103
|
+
|
|
103
104
|
# ---------------------------
|
|
104
105
|
# Pricing / cost calculation
|
|
105
106
|
# ---------------------------
|
|
@@ -537,7 +538,7 @@ class DocumentIngestSQLiteCallback(BaseCallbackHandler):
|
|
|
537
538
|
token_count=0,
|
|
538
539
|
cost_usd=0.0,
|
|
539
540
|
n_try=n_try,
|
|
540
|
-
metadata_json=
|
|
541
|
+
metadata_json=safe_json_dumps(payload, ensure_ascii=False),
|
|
541
542
|
)
|
|
542
543
|
)
|
|
543
544
|
def on_llm_start(
|
|
@@ -594,7 +595,7 @@ class DocumentIngestSQLiteCallback(BaseCallbackHandler):
|
|
|
594
595
|
token_count=0,
|
|
595
596
|
cost_usd=0.0,
|
|
596
597
|
n_try = payload.get('n_try', 0),
|
|
597
|
-
metadata_json=
|
|
598
|
+
metadata_json=safe_json_dumps(payload, ensure_ascii=False),
|
|
598
599
|
)
|
|
599
600
|
)
|
|
600
601
|
|
|
@@ -710,7 +711,7 @@ class DocumentIngestSQLiteCallback(BaseCallbackHandler):
|
|
|
710
711
|
token_count=token_count,
|
|
711
712
|
cost_usd=float(cost_usd),
|
|
712
713
|
n_try=float(payload.get('n_try', 0)),
|
|
713
|
-
metadata_json=
|
|
714
|
+
metadata_json=safe_json_dumps(payload, ensure_ascii=False),
|
|
714
715
|
)
|
|
715
716
|
)
|
|
716
717
|
|
|
@@ -777,6 +778,6 @@ class DocumentIngestSQLiteCallback(BaseCallbackHandler):
|
|
|
777
778
|
token_count=0,
|
|
778
779
|
cost_usd=0.0,
|
|
779
780
|
n_try=float(payload.get('n_try', 0)),
|
|
780
|
-
metadata_json=
|
|
781
|
+
metadata_json=safe_json_dumps(payload, ensure_ascii=False),
|
|
781
782
|
)
|
|
782
783
|
)
|
|
@@ -4,71 +4,34 @@ from .adapters import (
|
|
|
4
4
|
build_parser_source_map,
|
|
5
5
|
normalize_ocr_pages,
|
|
6
6
|
)
|
|
7
|
+
from .cache import WorkflowLLMCallCache
|
|
7
8
|
from .clients import (
|
|
8
9
|
CanonicalGraphPersistenceClient,
|
|
9
|
-
DocumentTreeApiPersistenceClient,
|
|
10
10
|
DirectRuntimeIngestClient,
|
|
11
|
+
DocumentTreeApiPersistenceClient,
|
|
11
12
|
IngestExecutionClient,
|
|
12
13
|
ServerCanonicalKgClient,
|
|
13
14
|
UnsupportedClientOperation,
|
|
14
15
|
)
|
|
15
|
-
from .cache import WorkflowLLMCallCache
|
|
16
|
-
from .design import DEFAULT_WORKFLOW_ID, build_ingest_workflow_design, ensure_ingest_workflow_design
|
|
17
16
|
from .demo_harness import DemoHarnessArtifacts, DemoHarnessConfig, run_demo_harness
|
|
18
|
-
from .
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
ParseMode,
|
|
23
|
-
TreeParseRequest,
|
|
24
|
-
parse_document,
|
|
25
|
-
parse_ocr_document,
|
|
26
|
-
parse_page_index_document,
|
|
27
|
-
parse_tree_document,
|
|
28
|
-
)
|
|
29
|
-
from .ocr_pipeline import (
|
|
30
|
-
OCRImagePayload,
|
|
31
|
-
OCRWorkflowArtifacts,
|
|
32
|
-
OCRWorkflowStateStore,
|
|
33
|
-
prepare_ocr_workflow_input,
|
|
34
|
-
run_ocr_ingest_workflow,
|
|
35
|
-
)
|
|
36
|
-
from .runners import (
|
|
37
|
-
LayerwiseWorkflowCommandResult,
|
|
38
|
-
OcrWorkflowCommandResult,
|
|
39
|
-
PageIndexWorkflowCommandResult,
|
|
40
|
-
WorkflowCommandResult,
|
|
41
|
-
discover_input_files,
|
|
42
|
-
run_demo_harness_workflow,
|
|
43
|
-
run_layerwise_batch_workflow,
|
|
44
|
-
run_layerwise_source_workflow,
|
|
45
|
-
run_ocr_batch_workflow,
|
|
46
|
-
run_ocr_source_workflow,
|
|
47
|
-
run_page_index_batch_workflow,
|
|
48
|
-
run_page_index_source_workflow,
|
|
49
|
-
)
|
|
50
|
-
from .smoke_assets import generate_ocr_smoke_assets
|
|
51
|
-
from .page_index import (
|
|
52
|
-
BlockAssignment,
|
|
53
|
-
BlockAssignmentBatch,
|
|
54
|
-
CandidateBlock,
|
|
55
|
-
PageIndexBlockSpec,
|
|
56
|
-
PageIndexParseResult,
|
|
57
|
-
PageIndexValidationResult,
|
|
58
|
-
build_page_index_workflow_input,
|
|
17
|
+
from .design import (
|
|
18
|
+
DEFAULT_WORKFLOW_ID,
|
|
19
|
+
build_ingest_workflow_design,
|
|
20
|
+
ensure_ingest_workflow_design,
|
|
59
21
|
)
|
|
22
|
+
from .handlers import build_ingest_step_resolver
|
|
60
23
|
from .models import (
|
|
61
24
|
BoundingBox,
|
|
62
25
|
CanonicalGraphWriteResult,
|
|
63
26
|
CurrentLayerContext,
|
|
64
|
-
CurrentLayerReview,
|
|
65
27
|
CurrentLayerResult,
|
|
66
|
-
|
|
67
|
-
LayerDuplicateChildNote,
|
|
28
|
+
CurrentLayerReview,
|
|
68
29
|
GroundedSourceRecord,
|
|
69
30
|
IngestRunHandle,
|
|
70
31
|
IngestRunResult,
|
|
71
32
|
LayerChildCandidate,
|
|
33
|
+
LayerCoverageGap,
|
|
34
|
+
LayerDuplicateChildNote,
|
|
72
35
|
LayerFrontierItem,
|
|
73
36
|
LayerSpanConflict,
|
|
74
37
|
NormalizedPage,
|
|
@@ -79,16 +42,25 @@ from .models import (
|
|
|
79
42
|
WorkflowExportBundle,
|
|
80
43
|
WorkflowIngestInput,
|
|
81
44
|
)
|
|
82
|
-
from .
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
45
|
+
from .ocr_pipeline import (
|
|
46
|
+
OCRImagePayload,
|
|
47
|
+
OCRWorkflowArtifacts,
|
|
48
|
+
OCRWorkflowStateStore,
|
|
49
|
+
prepare_ocr_workflow_input,
|
|
50
|
+
run_ocr_ingest_workflow,
|
|
51
|
+
)
|
|
52
|
+
from .page_index import (
|
|
53
|
+
BlockAssignment,
|
|
54
|
+
BlockAssignmentBatch,
|
|
55
|
+
CandidateBlock,
|
|
56
|
+
HierarchicalSummaryAssignment,
|
|
57
|
+
HierarchicalSummaryBatch,
|
|
58
|
+
PageIndexBlockSpec,
|
|
59
|
+
PageIndexParseResult,
|
|
60
|
+
PageIndexValidationResult,
|
|
61
|
+
build_page_index_workflow_input,
|
|
62
|
+
parse_page_index_layer,
|
|
90
63
|
)
|
|
91
|
-
from .probe import WorkflowProbe, emit_probe_event
|
|
92
64
|
from .parser_core import (
|
|
93
65
|
apply_cud_update,
|
|
94
66
|
check_layer_coverage,
|
|
@@ -99,101 +71,161 @@ from .parser_core import (
|
|
|
99
71
|
propose_layer_breakdown,
|
|
100
72
|
review_layer,
|
|
101
73
|
)
|
|
102
|
-
from .
|
|
74
|
+
from .parsing import (
|
|
75
|
+
OCRParseRequest,
|
|
76
|
+
PageIndexParseRequest,
|
|
77
|
+
ParseMode,
|
|
78
|
+
TreeParseRequest,
|
|
79
|
+
parse_document,
|
|
80
|
+
parse_ocr_document,
|
|
81
|
+
parse_page_index_document,
|
|
82
|
+
parse_tree_document,
|
|
83
|
+
)
|
|
84
|
+
from .probe import WorkflowProbe, emit_probe_event
|
|
85
|
+
from .providers import (
|
|
86
|
+
EmbeddingProviderConfig,
|
|
87
|
+
FakeChatModel,
|
|
88
|
+
ProviderEndpointConfig,
|
|
89
|
+
WorkflowProviderSettings,
|
|
90
|
+
build_chat_model,
|
|
91
|
+
build_chat_model_for_role,
|
|
92
|
+
build_embedding_function,
|
|
93
|
+
provider_call_metrics_snapshot,
|
|
94
|
+
)
|
|
95
|
+
from .runners import (
|
|
96
|
+
LayerwiseWorkflowCommandResult,
|
|
97
|
+
OcrWorkflowCommandResult,
|
|
98
|
+
PageIndexWorkflowCommandResult,
|
|
99
|
+
WorkflowCommandResult,
|
|
100
|
+
discover_input_files,
|
|
101
|
+
run_demo_harness_workflow,
|
|
102
|
+
run_layerwise_batch_workflow,
|
|
103
|
+
run_layerwise_source_workflow,
|
|
104
|
+
run_ocr_batch_workflow,
|
|
105
|
+
run_ocr_source_workflow,
|
|
106
|
+
run_page_index_batch_workflow,
|
|
107
|
+
run_page_index_source_workflow,
|
|
108
|
+
)
|
|
103
109
|
from .semantics import HydratedTextPointer, SemanticNode
|
|
110
|
+
from .service import build_default_engines, build_runtime, run_ingest_workflow
|
|
111
|
+
from .smoke_assets import generate_ocr_smoke_assets
|
|
112
|
+
from .strategy import (
|
|
113
|
+
HARD_CODED_STRATEGY_PRIORITY,
|
|
114
|
+
ParseStrategy,
|
|
115
|
+
ParseStrategyAssessment,
|
|
116
|
+
ParseStrategyDecision,
|
|
117
|
+
ParseStrategyRequest,
|
|
118
|
+
ParseStrategyTriage,
|
|
119
|
+
build_llm_strategy_triage,
|
|
120
|
+
hardcoded_strategy,
|
|
121
|
+
select_parse_strategy,
|
|
122
|
+
)
|
|
104
123
|
|
|
105
124
|
__all__ = [
|
|
125
|
+
"DEFAULT_WORKFLOW_ID",
|
|
126
|
+
"HARD_CODED_STRATEGY_PRIORITY",
|
|
127
|
+
"BlockAssignment",
|
|
128
|
+
"BlockAssignmentBatch",
|
|
106
129
|
"BoundingBox",
|
|
130
|
+
"CandidateBlock",
|
|
107
131
|
"CanonicalGraphPersistenceClient",
|
|
108
132
|
"CanonicalGraphWriteResult",
|
|
109
133
|
"CurrentLayerContext",
|
|
110
|
-
"CurrentLayerReview",
|
|
111
134
|
"CurrentLayerResult",
|
|
112
|
-
"
|
|
113
|
-
"LayerDuplicateChildNote",
|
|
135
|
+
"CurrentLayerReview",
|
|
114
136
|
"DemoHarnessArtifacts",
|
|
115
137
|
"DemoHarnessConfig",
|
|
116
|
-
"DocumentTreeApiPersistenceClient",
|
|
117
138
|
"DirectRuntimeIngestClient",
|
|
139
|
+
"DocumentTreeApiPersistenceClient",
|
|
118
140
|
"EmbeddingProviderConfig",
|
|
141
|
+
"FakeChatModel",
|
|
119
142
|
"GroundedSourceRecord",
|
|
143
|
+
"HierarchicalSummaryAssignment",
|
|
144
|
+
"HierarchicalSummaryBatch",
|
|
120
145
|
"HydratedTextPointer",
|
|
121
|
-
"FakeChatModel",
|
|
122
146
|
"IngestExecutionClient",
|
|
123
147
|
"IngestRunHandle",
|
|
124
148
|
"IngestRunResult",
|
|
125
149
|
"LayerChildCandidate",
|
|
150
|
+
"LayerCoverageGap",
|
|
151
|
+
"LayerDuplicateChildNote",
|
|
126
152
|
"LayerFrontierItem",
|
|
127
153
|
"LayerSpanConflict",
|
|
154
|
+
"LayerwiseWorkflowCommandResult",
|
|
128
155
|
"NormalizedPage",
|
|
129
156
|
"NormalizedSourceCollection",
|
|
130
157
|
"OCRImagePayload",
|
|
158
|
+
"OCRParseRequest",
|
|
131
159
|
"OCRWorkflowArtifacts",
|
|
132
160
|
"OCRWorkflowStateStore",
|
|
133
|
-
"
|
|
134
|
-
"ParseSessionState",
|
|
135
|
-
"ProviderEndpointConfig",
|
|
136
|
-
"PageIndexParseRequest",
|
|
161
|
+
"OcrWorkflowCommandResult",
|
|
137
162
|
"PageIndexBlockSpec",
|
|
163
|
+
"PageIndexParseRequest",
|
|
138
164
|
"PageIndexParseResult",
|
|
139
165
|
"PageIndexValidationResult",
|
|
166
|
+
"PageIndexWorkflowCommandResult",
|
|
140
167
|
"ParseMode",
|
|
141
|
-
"
|
|
168
|
+
"ParseSessionState",
|
|
169
|
+
"ParseStrategy",
|
|
170
|
+
"ParseStrategyAssessment",
|
|
171
|
+
"ParseStrategyDecision",
|
|
172
|
+
"ParseStrategyRequest",
|
|
173
|
+
"ParseStrategyTriage",
|
|
174
|
+
"ProviderEndpointConfig",
|
|
142
175
|
"SemanticNode",
|
|
143
176
|
"ServerCanonicalKgClient",
|
|
144
177
|
"SourceUnit",
|
|
178
|
+
"TreeParseRequest",
|
|
145
179
|
"UnsupportedClientOperation",
|
|
146
180
|
"ValidationReport",
|
|
181
|
+
"WorkflowCommandResult",
|
|
182
|
+
"WorkflowExportBundle",
|
|
183
|
+
"WorkflowIngestInput",
|
|
147
184
|
"WorkflowLLMCallCache",
|
|
148
185
|
"WorkflowProbe",
|
|
149
186
|
"WorkflowProviderSettings",
|
|
150
|
-
"WorkflowExportBundle",
|
|
151
|
-
"WorkflowIngestInput",
|
|
152
|
-
"WorkflowCommandResult",
|
|
153
|
-
"OcrWorkflowCommandResult",
|
|
154
|
-
"PageIndexWorkflowCommandResult",
|
|
155
|
-
"LayerwiseWorkflowCommandResult",
|
|
156
|
-
"CandidateBlock",
|
|
157
|
-
"BlockAssignment",
|
|
158
|
-
"BlockAssignmentBatch",
|
|
159
|
-
"DEFAULT_WORKFLOW_ID",
|
|
160
|
-
"build_chat_model",
|
|
161
|
-
"build_chat_model_for_role",
|
|
162
|
-
"build_embedding_function",
|
|
163
187
|
"apply_cud_update",
|
|
164
|
-
"check_layer_coverage",
|
|
165
|
-
"default_parse_semantic_fn",
|
|
166
|
-
"finalize_semantic_tree",
|
|
167
|
-
"initialize_parse_session",
|
|
168
|
-
"prepare_layer_frontier",
|
|
169
|
-
"propose_layer_breakdown",
|
|
170
|
-
"review_layer",
|
|
171
188
|
"build_authoritative_source_map",
|
|
189
|
+
"build_chat_model",
|
|
190
|
+
"build_chat_model_for_role",
|
|
172
191
|
"build_default_engines",
|
|
173
|
-
"
|
|
192
|
+
"build_embedding_function",
|
|
174
193
|
"build_ingest_step_resolver",
|
|
175
|
-
"
|
|
194
|
+
"build_ingest_workflow_design",
|
|
195
|
+
"build_llm_strategy_triage",
|
|
176
196
|
"build_page_index_workflow_input",
|
|
197
|
+
"build_parser_input_dict",
|
|
177
198
|
"build_parser_source_map",
|
|
178
199
|
"build_runtime",
|
|
179
|
-
"
|
|
200
|
+
"check_layer_coverage",
|
|
201
|
+
"default_parse_semantic_fn",
|
|
202
|
+
"discover_input_files",
|
|
180
203
|
"emit_probe_event",
|
|
204
|
+
"ensure_ingest_workflow_design",
|
|
205
|
+
"finalize_semantic_tree",
|
|
206
|
+
"generate_ocr_smoke_assets",
|
|
207
|
+
"hardcoded_strategy",
|
|
208
|
+
"initialize_parse_session",
|
|
181
209
|
"normalize_ocr_pages",
|
|
182
210
|
"parse_document",
|
|
183
211
|
"parse_ocr_document",
|
|
184
212
|
"parse_page_index_document",
|
|
213
|
+
"parse_page_index_layer",
|
|
185
214
|
"parse_tree_document",
|
|
215
|
+
"prepare_layer_frontier",
|
|
186
216
|
"prepare_ocr_workflow_input",
|
|
217
|
+
"propose_layer_breakdown",
|
|
218
|
+
"provider_call_metrics_snapshot",
|
|
219
|
+
"review_layer",
|
|
187
220
|
"run_demo_harness",
|
|
188
221
|
"run_demo_harness_workflow",
|
|
189
222
|
"run_ingest_workflow",
|
|
190
|
-
"discover_input_files",
|
|
191
223
|
"run_layerwise_batch_workflow",
|
|
192
224
|
"run_layerwise_source_workflow",
|
|
193
225
|
"run_ocr_batch_workflow",
|
|
226
|
+
"run_ocr_ingest_workflow",
|
|
194
227
|
"run_ocr_source_workflow",
|
|
195
228
|
"run_page_index_batch_workflow",
|
|
196
229
|
"run_page_index_source_workflow",
|
|
197
|
-
"
|
|
198
|
-
"run_ocr_ingest_workflow",
|
|
230
|
+
"select_parse_strategy",
|
|
199
231
|
]
|
|
@@ -46,14 +46,51 @@ def _provider_settings_from_args(args: argparse.Namespace) -> WorkflowProviderSe
|
|
|
46
46
|
"api_key_env": args.parser_api_key_env,
|
|
47
47
|
}
|
|
48
48
|
proposal_mode = getattr(args, "proposal_mode", None)
|
|
49
|
+
parse_strategy = getattr(args, "parse_strategy", None)
|
|
50
|
+
parse_strategy_order = getattr(args, "parse_strategy_order", None)
|
|
51
|
+
triage_enabled = getattr(args, "triage_enabled", None)
|
|
52
|
+
page_index_summary_enabled = getattr(args, "page_index_summary_enabled", None)
|
|
53
|
+
page_index_hierarchical_summary_enabled = getattr(
|
|
54
|
+
args, "page_index_hierarchical_summary_enabled", None
|
|
55
|
+
)
|
|
49
56
|
ocr_override = any(value is not None for value in ocr_values.values())
|
|
50
57
|
parser_override = any(value is not None for value in parser_values.values())
|
|
51
58
|
proposal_override = proposal_mode is not None
|
|
52
|
-
|
|
59
|
+
strategy_override = parse_strategy is not None
|
|
60
|
+
strategy_order_override = parse_strategy_order is not None
|
|
61
|
+
triage_override = triage_enabled is not None
|
|
62
|
+
summary_override = page_index_summary_enabled is not None
|
|
63
|
+
hierarchical_summary_override = page_index_hierarchical_summary_enabled is not None
|
|
64
|
+
if (
|
|
65
|
+
not ocr_override
|
|
66
|
+
and not parser_override
|
|
67
|
+
and not proposal_override
|
|
68
|
+
and not strategy_override
|
|
69
|
+
and not strategy_order_override
|
|
70
|
+
and not triage_override
|
|
71
|
+
and not summary_override
|
|
72
|
+
and not hierarchical_summary_override
|
|
73
|
+
):
|
|
53
74
|
return None
|
|
54
75
|
settings = WorkflowProviderSettings.from_env()
|
|
55
76
|
if proposal_override:
|
|
56
77
|
settings = settings.model_copy(update={"proposal_mode": proposal_mode})
|
|
78
|
+
if strategy_override:
|
|
79
|
+
settings = settings.model_copy(update={"parse_strategy": parse_strategy})
|
|
80
|
+
if strategy_order_override:
|
|
81
|
+
settings = settings.model_copy(
|
|
82
|
+
update={"parse_strategy_order": tuple(parse_strategy_order.split(","))}
|
|
83
|
+
)
|
|
84
|
+
if triage_override:
|
|
85
|
+
settings = settings.model_copy(update={"triage_enabled": triage_enabled})
|
|
86
|
+
if summary_override:
|
|
87
|
+
settings = settings.model_copy(update={"page_index_summary_enabled": page_index_summary_enabled})
|
|
88
|
+
if hierarchical_summary_override:
|
|
89
|
+
settings = settings.model_copy(
|
|
90
|
+
update={
|
|
91
|
+
"page_index_hierarchical_summary_enabled": page_index_hierarchical_summary_enabled
|
|
92
|
+
}
|
|
93
|
+
)
|
|
57
94
|
if ocr_override:
|
|
58
95
|
settings = settings.model_copy(
|
|
59
96
|
update={
|
|
@@ -76,6 +113,35 @@ def _provider_settings_from_args(args: argparse.Namespace) -> WorkflowProviderSe
|
|
|
76
113
|
def _add_provider_args(parser: argparse.ArgumentParser) -> None:
|
|
77
114
|
group = parser.add_argument_group("provider overrides")
|
|
78
115
|
group.add_argument("--proposal-mode", choices=["children", "boundaries"], default=None)
|
|
116
|
+
group.add_argument(
|
|
117
|
+
"--parse-strategy",
|
|
118
|
+
choices=["auto", "layer_excerpt", "layer_boundary", "page_index"],
|
|
119
|
+
default=None,
|
|
120
|
+
help="Select the parser strategy for every frontier layer in this parse call",
|
|
121
|
+
)
|
|
122
|
+
group.add_argument(
|
|
123
|
+
"--parse-strategy-order",
|
|
124
|
+
default=None,
|
|
125
|
+
help="comma-separated parser order, e.g. layer_boundary,layer_excerpt,page_index",
|
|
126
|
+
)
|
|
127
|
+
group.add_argument(
|
|
128
|
+
"--triage-enabled",
|
|
129
|
+
action=argparse.BooleanOptionalAction,
|
|
130
|
+
default=None,
|
|
131
|
+
help="Enable or disable provider-backed per-layer strategy triage",
|
|
132
|
+
)
|
|
133
|
+
group.add_argument(
|
|
134
|
+
"--page-index-summary-enabled",
|
|
135
|
+
action=argparse.BooleanOptionalAction,
|
|
136
|
+
default=None,
|
|
137
|
+
help="Enable or disable summaries on PageIndex nodes for this parse call",
|
|
138
|
+
)
|
|
139
|
+
group.add_argument(
|
|
140
|
+
"--page-index-hierarchical-summary-enabled",
|
|
141
|
+
action=argparse.BooleanOptionalAction,
|
|
142
|
+
default=None,
|
|
143
|
+
help="Enable or disable the bounded parent-summary context pass",
|
|
144
|
+
)
|
|
79
145
|
group.add_argument("--ocr-provider", default=None)
|
|
80
146
|
group.add_argument("--ocr-model", default=None)
|
|
81
147
|
group.add_argument("--ocr-base-url", default=None)
|