graph-knowledge-doc-parser 0.1.0__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graph_knowledge_doc_parser-0.1.0/README.md → graph_knowledge_doc_parser-0.2.2/PKG-INFO +367 -265
- graph_knowledge_doc_parser-0.1.0/PKG-INFO → graph_knowledge_doc_parser-0.2.2/README.md +72 -36
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/cast_hinting.py +19 -19
- graph_knowledge_doc_parser-0.2.2/kg_doc_parser/document_ingest_log_config.py +26 -0
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/document_ingester_logger.py +781 -765
- graph_knowledge_doc_parser-0.2.2/kg_doc_parser/llm_structured_output.py +46 -0
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/models.py +276 -276
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/ocr.py +795 -752
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/pdf2png.py +315 -283
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/semantic_document_splitting_layerwise_edits.py +3772 -3297
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/text_processing_utils.py +30 -30
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/utils/bounded_threadpool_executor.py +42 -37
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/utils/file_loaders.py +414 -405
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/utils/langchain.py +245 -220
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/utils/log.py +140 -135
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/utils/version_chaining.py +1285 -1274
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/workflow_ingest/__init__.py +157 -145
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/workflow_ingest/_kogwistar.py +13 -13
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/workflow_ingest/adapters.py +207 -191
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/workflow_ingest/cache.py +64 -63
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/workflow_ingest/cli.py +334 -323
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/workflow_ingest/clients.py +508 -444
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/workflow_ingest/demo_harness.py +480 -426
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/workflow_ingest/design.py +207 -208
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/workflow_ingest/handlers.py +701 -617
- graph_knowledge_doc_parser-0.2.2/kg_doc_parser/workflow_ingest/layered_contracts.py +295 -0
- graph_knowledge_doc_parser-0.2.2/kg_doc_parser/workflow_ingest/layerwise_llm.py +2587 -0
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/workflow_ingest/models.py +720 -575
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/workflow_ingest/ocr_pipeline.py +1590 -1581
- graph_knowledge_doc_parser-0.2.2/kg_doc_parser/workflow_ingest/page_index.py +2005 -0
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/workflow_ingest/parser_core.py +1004 -862
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/workflow_ingest/parsing.py +51 -24
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/workflow_ingest/probe.py +164 -164
- graph_knowledge_doc_parser-0.2.2/kg_doc_parser/workflow_ingest/providers.py +720 -0
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/workflow_ingest/runners.py +560 -546
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/workflow_ingest/semantics.py +233 -231
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/workflow_ingest/service.py +133 -112
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/workflow_ingest/smoke_assets.py +62 -62
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/pyproject.toml +18 -17
- graph_knowledge_doc_parser-0.1.0/kg_doc_parser/workflow_ingest/page_index.py +0 -473
- graph_knowledge_doc_parser-0.1.0/kg_doc_parser/workflow_ingest/providers.py +0 -412
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/__init__.py +0 -0
- {graph_knowledge_doc_parser-0.1.0 → graph_knowledge_doc_parser-0.2.2}/kg_doc_parser/utils/__init__.py +0 -0
|
@@ -1,50 +1,123 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
-
|
|
18
|
-
-
|
|
19
|
-
-
|
|
20
|
-
-
|
|
21
|
-
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
-
|
|
28
|
-
-
|
|
29
|
-
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: graph-knowledge-doc-parser
|
|
3
|
+
Version: 0.2.2
|
|
4
|
+
Summary: doc parser using llm driven engine with graph knowledge awareness
|
|
5
|
+
Author: humblemat810
|
|
6
|
+
Author-email: 67593116+humblemat810@users.noreply.github.com
|
|
7
|
+
Requires-Python: >=3.12,<4.0
|
|
8
|
+
Classifier: License :: Other/Proprietary License
|
|
9
|
+
Classifier: Operating System :: OS Independent
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.15
|
|
15
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
16
|
+
Classifier: Topic :: Text Processing :: General
|
|
17
|
+
Requires-Dist: diskcache (>=5.6,<6.0) ; platform_python_implementation == "PyPy"
|
|
18
|
+
Requires-Dist: joblib (>=1.5.3,<2.0.0) ; platform_python_implementation == "CPython"
|
|
19
|
+
Requires-Dist: kogwistar (==0.6.3)
|
|
20
|
+
Requires-Dist: langchain-core (>=1.2.5,<2.0.0)
|
|
21
|
+
Requires-Dist: langchain-google-genai (>=4.1.2,<5.0.0)
|
|
22
|
+
Requires-Dist: mcp (>=2.2.0,<3.0.0)
|
|
23
|
+
Requires-Dist: pathspec (>=0.12.1,<0.13.0)
|
|
24
|
+
Requires-Dist: pdf2image (>=1.17.0,<2.0.0)
|
|
25
|
+
Requires-Dist: pikepdf (>=10.1.0,<11.0.0) ; platform_python_implementation != "PyPy"
|
|
26
|
+
Requires-Dist: pydantic (>=2.12.5,<3.0.0)
|
|
27
|
+
Requires-Dist: pydantic-extension (>=0.0.7,<0.0.8)
|
|
28
|
+
Requires-Dist: pypdf (>=6.5.0,<7.0.0)
|
|
29
|
+
Requires-Dist: rapidfuzz (>=3.14.3,<4.0.0)
|
|
30
|
+
Project-URL: Homepage, https://github.com/humblemat810/kg_doc_parser
|
|
31
|
+
Project-URL: Repository, https://github.com/humblemat810/kg_doc_parser
|
|
32
|
+
Description-Content-Type: text/markdown
|
|
33
|
+
|
|
34
|
+
# Kogwistar Graph-Knowledge Doc Pipeline
|
|
35
|
+
|
|
36
|
+
If you want the shortest path into the project, start with [QUICKSTART.md](QUICKSTART.md).
|
|
37
|
+
|
|
38
|
+
## Graph Knowledge Doc Pipeline
|
|
39
|
+
|
|
40
|
+
Utilities and experiments for document ingestion, PDF splitting, OCR, page-level parsing, and conversion into graph-knowledge artifacts that the Obsidian sink and downstream Kogwistar workflows can consume. Refactored from the kogwistar project as a standalone ingestion-to-graph pipeline.
|
|
41
|
+
|
|
42
|
+
## Status
|
|
43
|
+
|
|
44
|
+
This repository is still being refactored and should be treated as work in progress.
|
|
45
|
+
|
|
46
|
+
The document ingestion pipeline is currently being extracted and consolidated into the main `kogwistar` repository. Until that refactor is complete, this repo should be considered an active staging area for parser and ingestion-related work.
|
|
47
|
+
|
|
48
|
+
## What Is Here
|
|
49
|
+
|
|
50
|
+
- PDF splitting and image generation helpers in `src/pdf2png.py`
|
|
51
|
+
- Gemini-based OCR and page parsing flows in `src/ocr.py`
|
|
52
|
+
- SQLite-based ingestion telemetry in `src/document_ingester_logger.py`
|
|
53
|
+
- File discovery and filtering helpers in `src/utils/file_loaders.py`
|
|
54
|
+
- Experimental and regression-style tests under `tests/`
|
|
55
|
+
|
|
56
|
+
## Workflow Surface
|
|
57
|
+
|
|
58
|
+
The reusable workflow-ingest code now has three layers:
|
|
59
|
+
|
|
60
|
+
- Python APIs, which are the primary contract for tests and orchestration
|
|
61
|
+
- CLI entrypoints, which are thin wrappers around those APIs
|
|
62
|
+
- composable subworkflows for OCR, page-index parsing, recursive layerwise parsing, and graph-knowledge conversion
|
|
63
|
+
|
|
64
|
+
The reusable helpers live under `src/workflow_ingest/` and are designed so the
|
|
65
|
+
same core logic can be called from tests, scripts, and higher-level workflow
|
|
66
|
+
code without duplicating orchestration.
|
|
67
|
+
|
|
68
|
+
For an adoption path that starts with parser-grounded source units and later
|
|
69
|
+
adds LLM-Wiki cross-document maintenance, see
|
|
70
|
+
[`doc/progressive_adoption_guide.md`](doc/progressive_adoption_guide.md). The
|
|
71
|
+
guide also explains which vector-RAG and source-import responsibilities remain
|
|
72
|
+
with the integrating application.
|
|
73
|
+
|
|
74
|
+
## CLI Cheatsheet
|
|
75
|
+
|
|
76
|
+
| Command | What it does | Main tests |
|
|
77
|
+
|---|---|---|
|
|
78
|
+
| `workflow-ingest ocr <source> --output-dir <dir>` | Runs OCR over a file or folder and writes OCR artifacts | `tests/test_workflow_ingest_parsing_api.py`, `tests/test_workflow_ingest_cli.py` |
|
|
79
|
+
| `workflow-ingest page-index <source> --output-dir <dir>` | Parses page-index text into structured graph input | `tests/test_workflow_ingest_parsing_api.py`, `tests/test_workflow_ingest_cli.py` |
|
|
80
|
+
| `workflow-ingest layerwise <source> --output-dir <dir>` | Runs the recursive layerwise parser workflow | `tests/test_workflow_ingest_layerwise_parser.py`, `tests/test_workflow_ingest_cli.py` |
|
|
81
|
+
| `workflow-ingest demo --output-dir <dir>` | Runs the end-to-end demo harness and writes run artifacts | `tests/test_workflow_ingest_demo_harness.py` |
|
|
82
|
+
| `workflow-ingest ocr-smoke-assets --output-dir <dir>` | Generates local OCR smoke assets for manual or automated runs | `tests/test_workflow_ingest_cli.py` |
|
|
83
|
+
|
|
84
|
+
For a quick sanity check, `workflow-ingest --help` shows the full command tree,
|
|
85
|
+
and `workflow-ingest demo --help` shows the demo-specific options without
|
|
86
|
+
writing any artifacts.
|
|
87
|
+
|
|
88
|
+
### CLI Commands
|
|
89
|
+
|
|
37
90
|
After `poetry install`, the repo exposes a `workflow-ingest` command family:
|
|
38
|
-
|
|
39
|
-
```powershell
|
|
40
|
-
workflow-ingest --help
|
|
41
|
-
workflow-ingest ocr --help
|
|
42
|
-
workflow-ingest page-index --help
|
|
43
|
-
workflow-ingest layerwise --help
|
|
44
|
-
workflow-ingest demo --help
|
|
91
|
+
|
|
92
|
+
```powershell
|
|
93
|
+
workflow-ingest --help
|
|
94
|
+
workflow-ingest ocr --help
|
|
95
|
+
workflow-ingest page-index --help
|
|
96
|
+
workflow-ingest layerwise --help
|
|
97
|
+
workflow-ingest demo --help
|
|
45
98
|
workflow-ingest ocr-smoke-assets --help
|
|
46
99
|
```
|
|
47
100
|
|
|
101
|
+
To inspect the demo harness options:
|
|
102
|
+
|
|
103
|
+
```powershell
|
|
104
|
+
workflow-ingest demo --help
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
To actually run the demo harness, use:
|
|
108
|
+
|
|
109
|
+
```powershell
|
|
110
|
+
workflow-ingest demo --output-dir logs\workflow_ingest_demo
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
Then open the output directory:
|
|
114
|
+
|
|
115
|
+
```powershell
|
|
116
|
+
explorer logs\workflow_ingest_demo
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
That directory contains the probe trail, summary JSON, cache directory, and the local engine/server data for the run.
|
|
120
|
+
|
|
48
121
|
If you want the local checked-out `./kogwistar` subtree to win over the GitHub
|
|
49
122
|
dependency during development, run the Bash bootstrap helper after install:
|
|
50
123
|
|
|
@@ -59,178 +132,197 @@ That script does two things:
|
|
|
59
132
|
|
|
60
133
|
If you do not run the bootstrap helper, the repo keeps using the GitHub-sourced
|
|
61
134
|
`kogwistar` dependency declared in `pyproject.toml`.
|
|
62
|
-
|
|
63
|
-
Typical examples:
|
|
64
|
-
|
|
65
|
-
```powershell
|
|
66
|
-
workflow-ingest ocr tests\.tmp_workflow_ingest_ocr\generated_smoke_assets\ocr_smoke_document.pdf --output-dir logs\ocr_run
|
|
67
|
-
workflow-ingest ocr tests\.tmp_workflow_ingest_ocr\generated_smoke_assets --output-dir logs\ocr_batch
|
|
68
|
-
workflow-ingest page-index tests\fixtures\page_index\sample_page_index.txt --output-dir logs\page_index
|
|
69
|
-
workflow-ingest layerwise tests\.tmp_workflow_ingest_ocr\manual_cases\ollama\glm-ocr_latest\image\artifacts\legacy_split_pages\ocr-manual-ollama-image --output-dir logs\layerwise
|
|
70
|
-
workflow-ingest ocr-smoke-assets --output-dir tests\.tmp_workflow_ingest_ocr\generated_smoke_assets
|
|
71
|
-
```
|
|
72
|
-
|
|
73
|
-
### Reusable Artifacts
|
|
74
|
-
|
|
75
|
-
The workflow outputs are intentionally inspectable on disk:
|
|
76
|
-
|
|
77
|
-
- `workflow-events.jsonl`: readable step trail for the outer orchestration layer
|
|
78
|
-
- `ocr-state.sqlite`: authoritative OCR/render resume state
|
|
79
|
-
- `ocr-progress.json`: human-readable mirror of the current OCR state
|
|
80
|
-
- `ocr-summary.json`: final OCR run summary
|
|
81
|
-
- `legacy_split_pages/<document>/page_N.json`: legacy-compatible OCR page artifacts
|
|
82
|
-
- `rendered_pages/<document>/page_N.png`: rasterized page images
|
|
83
|
-
- `page-index-summary.json`: page-index run summary
|
|
84
|
-
- `layerwise-summary.json`: recursive layerwise parser summary
|
|
85
|
-
- `layerwise-graph.json`: legacy recursive layerwise graph payload
|
|
86
|
-
|
|
87
|
-
### Python APIs
|
|
88
|
-
|
|
89
|
-
If you want to embed the pipelines directly, the main helpers are:
|
|
90
|
-
|
|
91
|
-
- `src.workflow_ingest.run_ocr_source_workflow(...)`
|
|
92
|
-
- `src.workflow_ingest.run_ocr_batch_workflow(...)`
|
|
93
|
-
- `src.workflow_ingest.parse_page_index_document(...)`
|
|
94
|
-
- `src.workflow_ingest.run_page_index_source_workflow(...)`
|
|
95
|
-
- `src.workflow_ingest.run_layerwise_source_workflow(...)`
|
|
96
|
-
- `src.workflow_ingest.run_demo_harness_workflow(...)`
|
|
97
|
-
|
|
98
|
-
Those helpers are meant to stay stable and are what the CLI layer calls under
|
|
99
|
-
the hood.
|
|
100
|
-
|
|
101
|
-
## Setup
|
|
102
|
-
|
|
103
|
-
1. Create and activate a Python 3.13 environment.
|
|
104
|
-
2. Install dependencies with Poetry:
|
|
105
|
-
|
|
106
|
-
```powershell
|
|
107
|
-
poetry install
|
|
108
|
-
```
|
|
109
|
-
|
|
110
|
-
3. Create a local env file from the example:
|
|
111
|
-
|
|
112
|
-
```powershell
|
|
113
|
-
Copy-Item .env.example .env
|
|
114
|
-
```
|
|
115
|
-
|
|
116
|
-
4. Add your local `GOOGLE_API_KEY` and any optional file-list paths needed for your workflow.
|
|
117
|
-
|
|
118
|
-
## Environment Variables
|
|
119
|
-
|
|
120
|
-
The project currently expects or optionally uses:
|
|
121
|
-
|
|
122
|
-
- `GOOGLE_API_KEY`: required for Gemini OCR and LLM-backed parsing flows
|
|
123
|
-
- `LANGSMITH_TRACING`: optional LangSmith tracing toggle
|
|
124
|
-
- `ocr_file_list`: optional allow-list file for OCR runs
|
|
125
|
-
- `split_raw_file_list`: optional allow-list file for PDF splitting runs
|
|
126
|
-
- `answer_export_list`: optional export list path used by local workflows
|
|
127
|
-
|
|
128
|
-
An example template is provided in [`.env.example`](/c:/Users/chanh/Documents/kg_doc_parser/.env.example).
|
|
129
|
-
|
|
130
|
-
## Provider Guide
|
|
131
|
-
|
|
132
|
-
The workflow layer is vendor-neutral, but the concrete OCR, parser, and embedding
|
|
133
|
-
backends are selected by config.
|
|
134
|
-
|
|
135
|
-
### OCR Provider Examples
|
|
136
|
-
|
|
137
|
-
- Google GenAI OCR:
|
|
138
|
-
- `KG_DOC_OCR_PROVIDER=gemini`
|
|
139
|
-
- `KG_DOC_OCR_MODEL=gemini-2.5-flash`
|
|
140
|
-
- Ollama OCR or vision-capable local model:
|
|
141
|
-
- `KG_DOC_OCR_PROVIDER=ollama`
|
|
142
|
-
- `KG_DOC_OCR_MODEL=llava:latest`
|
|
143
|
-
- `KG_DOC_OCR_BASE_URL=http://127.0.0.1:11434`
|
|
144
|
-
- Vertex AI OCR:
|
|
145
|
-
- `KG_DOC_OCR_PROVIDER=vertex`
|
|
146
|
-
- `KG_DOC_OCR_MODEL=gemini-2.5-pro`
|
|
147
|
-
- `KG_DOC_OCR_PROJECT=my-project`
|
|
148
|
-
- `KG_DOC_OCR_LOCATION=us-central1`
|
|
149
|
-
|
|
150
|
-
### Parser / LLM Provider Examples
|
|
151
|
-
|
|
152
|
-
The parser provider is the chat model used for semantic parsing, layer review,
|
|
153
|
-
and structured extraction.
|
|
154
|
-
|
|
155
|
-
- LangChain Google GenAI:
|
|
156
|
-
- `KG_DOC_PARSER_PROVIDER=gemini`
|
|
157
|
-
- `KG_DOC_PARSER_MODEL=gemini-2.5-flash`
|
|
158
|
-
- ChatGPT / OpenAI REST:
|
|
159
|
-
- `KG_DOC_PARSER_PROVIDER=openai`
|
|
160
|
-
- `KG_DOC_PARSER_MODEL=gpt-4.1-mini`
|
|
161
|
-
- `KG_DOC_PARSER_API_KEY_ENV=OPENAI_API_KEY`
|
|
162
|
-
- LangChain Ollama:
|
|
163
|
-
- `KG_DOC_PARSER_PROVIDER=ollama`
|
|
164
|
-
- `KG_DOC_PARSER_MODEL=llama3.1`
|
|
165
|
-
- `KG_DOC_PARSER_BASE_URL=http://127.0.0.1:11434`
|
|
166
|
-
- LangChain Vertex AI:
|
|
167
|
-
- `KG_DOC_PARSER_PROVIDER=vertex`
|
|
168
|
-
- `KG_DOC_PARSER_MODEL=gemini-2.5-pro`
|
|
169
|
-
- `KG_DOC_PARSER_PROJECT=my-project`
|
|
170
|
-
- `KG_DOC_PARSER_LOCATION=us-central1`
|
|
171
|
-
|
|
172
|
-
### Recipe Parsing Example
|
|
173
|
-
|
|
174
|
-
If you are parsing a cooking recipe, one practical split is:
|
|
175
|
-
|
|
176
|
-
- OCR on Gemini or another vision model
|
|
177
|
-
- parser on OpenAI, Ollama, or Vertex AI
|
|
178
|
-
|
|
179
|
-
For example:
|
|
180
|
-
|
|
181
|
-
```powershell
|
|
182
|
-
KG_DOC_OCR_PROVIDER=gemini
|
|
183
|
-
KG_DOC_OCR_MODEL=gemini-2.5-flash
|
|
184
|
-
KG_DOC_PARSER_PROVIDER=openai
|
|
185
|
-
KG_DOC_PARSER_MODEL=gpt-4.1-mini
|
|
186
|
-
KG_DOC_PARSER_API_KEY_ENV=OPENAI_API_KEY
|
|
187
|
-
KG_DOC_EMBED_PROVIDER=fake
|
|
188
|
-
```
|
|
189
|
-
|
|
190
|
-
That setup can extract a recipe into structured graph data such as:
|
|
191
|
-
- ingredients
|
|
192
|
-
- steps
|
|
193
|
-
- tools
|
|
194
|
-
- timers
|
|
195
|
-
- inferred sections like `prep`, `cook`, and `serve`
|
|
196
|
-
|
|
197
|
-
### Embedding Examples
|
|
198
|
-
|
|
199
|
-
- Fake deterministic CI embedding:
|
|
200
|
-
- `KG_DOC_EMBED_PROVIDER=fake`
|
|
201
|
-
- OpenAI embeddings:
|
|
202
|
-
- `KG_DOC_EMBED_PROVIDER=openai`
|
|
203
|
-
- `KG_DOC_EMBED_MODEL=text-embedding-3-small`
|
|
204
|
-
- Vertex AI embeddings:
|
|
205
|
-
- `KG_DOC_EMBED_PROVIDER=vertex`
|
|
206
|
-
- `KG_DOC_EMBED_MODEL=text-embedding-004`
|
|
207
|
-
- Ollama embeddings:
|
|
208
|
-
- `KG_DOC_EMBED_PROVIDER=ollama`
|
|
209
|
-
- `KG_DOC_EMBED_MODEL=nomic-embed-text`
|
|
210
|
-
|
|
211
|
-
Note:
|
|
212
|
-
|
|
213
|
-
- `embedding_space` in the workflow ingest models is currently a metadata and
|
|
214
|
-
routing-intent label.
|
|
215
|
-
- It does not yet imply that the engine is using a separate embedder per space.
|
|
216
|
-
- The current engine bootstrap still wires one embedding function per engine
|
|
217
|
-
instance, while the multi-space routing proposal remains a future Kogwistar
|
|
218
|
-
core concern.
|
|
219
|
-
|
|
220
|
-
## Running Tests
|
|
221
|
-
|
|
222
|
-
Some tests are integration-style and expect local document folders and API credentials to exist. That means not every test is portable in a clean checkout.
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
135
|
+
|
|
136
|
+
Typical examples:
|
|
137
|
+
|
|
138
|
+
```powershell
|
|
139
|
+
workflow-ingest ocr tests\.tmp_workflow_ingest_ocr\generated_smoke_assets\ocr_smoke_document.pdf --output-dir logs\ocr_run
|
|
140
|
+
workflow-ingest ocr tests\.tmp_workflow_ingest_ocr\generated_smoke_assets --output-dir logs\ocr_batch
|
|
141
|
+
workflow-ingest page-index tests\fixtures\page_index\sample_page_index.txt --output-dir logs\page_index
|
|
142
|
+
workflow-ingest layerwise tests\.tmp_workflow_ingest_ocr\manual_cases\ollama\glm-ocr_latest\image\artifacts\legacy_split_pages\ocr-manual-ollama-image --output-dir logs\layerwise
|
|
143
|
+
workflow-ingest ocr-smoke-assets --output-dir tests\.tmp_workflow_ingest_ocr\generated_smoke_assets
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
### Reusable Artifacts
|
|
147
|
+
|
|
148
|
+
The workflow outputs are intentionally inspectable on disk:
|
|
149
|
+
|
|
150
|
+
- `workflow-events.jsonl`: readable step trail for the outer orchestration layer
|
|
151
|
+
- `ocr-state.sqlite`: authoritative OCR/render resume state
|
|
152
|
+
- `ocr-progress.json`: human-readable mirror of the current OCR state
|
|
153
|
+
- `ocr-summary.json`: final OCR run summary
|
|
154
|
+
- `legacy_split_pages/<document>/page_N.json`: legacy-compatible OCR page artifacts
|
|
155
|
+
- `rendered_pages/<document>/page_N.png`: rasterized page images
|
|
156
|
+
- `page-index-summary.json`: page-index run summary
|
|
157
|
+
- `layerwise-summary.json`: recursive layerwise parser summary
|
|
158
|
+
- `layerwise-graph.json`: legacy recursive layerwise graph payload
|
|
159
|
+
|
|
160
|
+
### Python APIs
|
|
161
|
+
|
|
162
|
+
If you want to embed the pipelines directly, the main helpers are:
|
|
163
|
+
|
|
164
|
+
- `src.workflow_ingest.run_ocr_source_workflow(...)`
|
|
165
|
+
- `src.workflow_ingest.run_ocr_batch_workflow(...)`
|
|
166
|
+
- `src.workflow_ingest.parse_page_index_document(...)`
|
|
167
|
+
- `src.workflow_ingest.run_page_index_source_workflow(...)`
|
|
168
|
+
- `src.workflow_ingest.run_layerwise_source_workflow(...)`
|
|
169
|
+
- `src.workflow_ingest.run_demo_harness_workflow(...)`
|
|
170
|
+
|
|
171
|
+
Those helpers are meant to stay stable and are what the CLI layer calls under
|
|
172
|
+
the hood.
|
|
173
|
+
|
|
174
|
+
## Setup
|
|
175
|
+
|
|
176
|
+
1. Create and activate a Python 3.13 environment.
|
|
177
|
+
2. Install dependencies with Poetry:
|
|
178
|
+
|
|
179
|
+
```powershell
|
|
180
|
+
poetry install
|
|
181
|
+
```
|
|
182
|
+
|
|
183
|
+
3. Create a local env file from the example:
|
|
184
|
+
|
|
185
|
+
```powershell
|
|
186
|
+
Copy-Item .env.example .env
|
|
187
|
+
```
|
|
188
|
+
|
|
189
|
+
4. Add your local `GOOGLE_API_KEY` and any optional file-list paths needed for your workflow.
|
|
190
|
+
|
|
191
|
+
## Environment Variables
|
|
192
|
+
|
|
193
|
+
The project currently expects or optionally uses:
|
|
194
|
+
|
|
195
|
+
- `GOOGLE_API_KEY`: required for Gemini OCR and LLM-backed parsing flows
|
|
196
|
+
- `LANGSMITH_TRACING`: optional LangSmith tracing toggle
|
|
197
|
+
- `ocr_file_list`: optional allow-list file for OCR runs
|
|
198
|
+
- `split_raw_file_list`: optional allow-list file for PDF splitting runs
|
|
199
|
+
- `answer_export_list`: optional export list path used by local workflows
|
|
200
|
+
|
|
201
|
+
An example template is provided in [`.env.example`](/c:/Users/chanh/Documents/kg_doc_parser/.env.example).
|
|
202
|
+
|
|
203
|
+
## Provider Guide
|
|
204
|
+
|
|
205
|
+
The workflow layer is vendor-neutral, but the concrete OCR, parser, and embedding
|
|
206
|
+
backends are selected by config.
|
|
207
|
+
|
|
208
|
+
### OCR Provider Examples
|
|
209
|
+
|
|
210
|
+
- Google GenAI OCR:
|
|
211
|
+
- `KG_DOC_OCR_PROVIDER=gemini`
|
|
212
|
+
- `KG_DOC_OCR_MODEL=gemini-2.5-flash`
|
|
213
|
+
- Ollama OCR or vision-capable local model:
|
|
214
|
+
- `KG_DOC_OCR_PROVIDER=ollama`
|
|
215
|
+
- `KG_DOC_OCR_MODEL=llava:latest`
|
|
216
|
+
- `KG_DOC_OCR_BASE_URL=http://127.0.0.1:11434`
|
|
217
|
+
- Vertex AI OCR:
|
|
218
|
+
- `KG_DOC_OCR_PROVIDER=vertex`
|
|
219
|
+
- `KG_DOC_OCR_MODEL=gemini-2.5-pro`
|
|
220
|
+
- `KG_DOC_OCR_PROJECT=my-project`
|
|
221
|
+
- `KG_DOC_OCR_LOCATION=us-central1`
|
|
222
|
+
|
|
223
|
+
### Parser / LLM Provider Examples
|
|
224
|
+
|
|
225
|
+
The parser provider is the chat model used for semantic parsing, layer review,
|
|
226
|
+
and structured extraction.
|
|
227
|
+
|
|
228
|
+
- LangChain Google GenAI:
|
|
229
|
+
- `KG_DOC_PARSER_PROVIDER=gemini`
|
|
230
|
+
- `KG_DOC_PARSER_MODEL=gemini-2.5-flash`
|
|
231
|
+
- ChatGPT / OpenAI REST:
|
|
232
|
+
- `KG_DOC_PARSER_PROVIDER=openai`
|
|
233
|
+
- `KG_DOC_PARSER_MODEL=gpt-4.1-mini`
|
|
234
|
+
- `KG_DOC_PARSER_API_KEY_ENV=OPENAI_API_KEY`
|
|
235
|
+
- LangChain Ollama:
|
|
236
|
+
- `KG_DOC_PARSER_PROVIDER=ollama`
|
|
237
|
+
- `KG_DOC_PARSER_MODEL=llama3.1`
|
|
238
|
+
- `KG_DOC_PARSER_BASE_URL=http://127.0.0.1:11434`
|
|
239
|
+
- LangChain Vertex AI:
|
|
240
|
+
- `KG_DOC_PARSER_PROVIDER=vertex`
|
|
241
|
+
- `KG_DOC_PARSER_MODEL=gemini-2.5-pro`
|
|
242
|
+
- `KG_DOC_PARSER_PROJECT=my-project`
|
|
243
|
+
- `KG_DOC_PARSER_LOCATION=us-central1`
|
|
244
|
+
|
|
245
|
+
### Recipe Parsing Example
|
|
246
|
+
|
|
247
|
+
If you are parsing a cooking recipe, one practical split is:
|
|
248
|
+
|
|
249
|
+
- OCR on Gemini or another vision model
|
|
250
|
+
- parser on OpenAI, Ollama, or Vertex AI
|
|
251
|
+
|
|
252
|
+
For example:
|
|
253
|
+
|
|
254
|
+
```powershell
|
|
255
|
+
KG_DOC_OCR_PROVIDER=gemini
|
|
256
|
+
KG_DOC_OCR_MODEL=gemini-2.5-flash
|
|
257
|
+
KG_DOC_PARSER_PROVIDER=openai
|
|
258
|
+
KG_DOC_PARSER_MODEL=gpt-4.1-mini
|
|
259
|
+
KG_DOC_PARSER_API_KEY_ENV=OPENAI_API_KEY
|
|
260
|
+
KG_DOC_EMBED_PROVIDER=fake
|
|
261
|
+
```
|
|
262
|
+
|
|
263
|
+
That setup can extract a recipe into structured graph data such as:
|
|
264
|
+
- ingredients
|
|
265
|
+
- steps
|
|
266
|
+
- tools
|
|
267
|
+
- timers
|
|
268
|
+
- inferred sections like `prep`, `cook`, and `serve`
|
|
269
|
+
|
|
270
|
+
### Embedding Examples
|
|
271
|
+
|
|
272
|
+
- Fake deterministic CI embedding:
|
|
273
|
+
- `KG_DOC_EMBED_PROVIDER=fake`
|
|
274
|
+
- OpenAI embeddings:
|
|
275
|
+
- `KG_DOC_EMBED_PROVIDER=openai`
|
|
276
|
+
- `KG_DOC_EMBED_MODEL=text-embedding-3-small`
|
|
277
|
+
- Vertex AI embeddings:
|
|
278
|
+
- `KG_DOC_EMBED_PROVIDER=vertex`
|
|
279
|
+
- `KG_DOC_EMBED_MODEL=text-embedding-004`
|
|
280
|
+
- Ollama embeddings:
|
|
281
|
+
- `KG_DOC_EMBED_PROVIDER=ollama`
|
|
282
|
+
- `KG_DOC_EMBED_MODEL=nomic-embed-text`
|
|
283
|
+
|
|
284
|
+
Note:
|
|
285
|
+
|
|
286
|
+
- `embedding_space` in the workflow ingest models is currently a metadata and
|
|
287
|
+
routing-intent label.
|
|
288
|
+
- It does not yet imply that the engine is using a separate embedder per space.
|
|
289
|
+
- The current engine bootstrap still wires one embedding function per engine
|
|
290
|
+
instance, while the multi-space routing proposal remains a future Kogwistar
|
|
291
|
+
core concern.
|
|
292
|
+
|
|
293
|
+
## Running Tests
|
|
294
|
+
|
|
295
|
+
Some tests are integration-style and expect local document folders and API credentials to exist. That means not every test is portable in a clean checkout.
|
|
296
|
+
|
|
297
|
+
Deterministic CI covers parser contracts with fake or injected model responses.
|
|
298
|
+
Model choice and output quality are parser-repository concerns, not core migration
|
|
299
|
+
gates. Any test that invokes Ollama, Gemini, OpenAI, Azure, Vertex, or another real
|
|
300
|
+
provider must use `llm_real` or `manual`; Ollama tests must also use
|
|
301
|
+
`requires_ollama`. Such tests must not carry `ci` or `ci_full`. Collection fails if
|
|
302
|
+
these marker classes are mixed, preventing accidental live-model calls in CI.
|
|
303
|
+
|
|
304
|
+
Run deterministic contract CI with:
|
|
305
|
+
|
|
306
|
+
```powershell
|
|
307
|
+
pytest -m "(ci or ci_full) and not manual and not llm_real and not requires_ollama"
|
|
308
|
+
```
|
|
309
|
+
|
|
310
|
+
Run provider/model checks only when explicitly requested, for example:
|
|
311
|
+
|
|
312
|
+
```powershell
|
|
313
|
+
pytest -m "llm_real and requires_ollama"
|
|
314
|
+
```
|
|
315
|
+
|
|
316
|
+
To run the test suite:
|
|
317
|
+
|
|
318
|
+
```powershell
|
|
319
|
+
pytest -q
|
|
320
|
+
```
|
|
321
|
+
|
|
322
|
+
If you only want to work on isolated units, review the test files first and run a narrower subset.
|
|
323
|
+
|
|
324
|
+
## Demo Harness
|
|
325
|
+
|
|
234
326
|
There is now a manual workflow-ingest demo harness that can run the end-to-end flow against:
|
|
235
327
|
|
|
236
328
|
- an in-process isolated FastAPI server
|
|
@@ -242,53 +334,63 @@ The legacy semantic-smoke test in
|
|
|
242
334
|
also expects a live Kogwistar server at `http://127.0.0.1:28110`. It does not
|
|
243
335
|
start that server for you, so use the VS Code server launch config or start it
|
|
244
336
|
manually before running the Ollama case.
|
|
245
|
-
|
|
246
|
-
Example:
|
|
247
|
-
|
|
248
|
-
```powershell
|
|
249
|
-
.venv\Scripts\python.exe scripts\run_workflow_ingest_demo.py --output-dir logs\workflow_ingest_demo
|
|
250
|
-
```
|
|
251
|
-
|
|
252
|
-
External live server example:
|
|
253
|
-
|
|
254
|
-
```powershell
|
|
255
|
-
.venv\Scripts\python.exe scripts\run_workflow_ingest_demo.py --server-mode external_http --external-base-url http://127.0.0.1:28110
|
|
256
|
-
```
|
|
257
|
-
|
|
258
|
-
Demo artifacts are written into the chosen output directory:
|
|
259
|
-
|
|
260
|
-
- `probe-events.jsonl`: demo-friendly step and lifecycle probe events
|
|
261
|
-
- `demo-summary.json`: run summary, persistence result, and artifact pointers
|
|
262
|
-
- `llm-cache/`: workflow-native cached proposal/review call results
|
|
263
|
-
- `engines/`: local workflow and conversation graph storage for the run
|
|
264
|
-
- `server-data/`: isolated server-side persistence directory when the harness boots its own server
|
|
265
|
-
|
|
266
|
-
Notes:
|
|
267
|
-
|
|
268
|
-
- The workflow-native layer proposal/review path uses deterministic file-backed caching to reduce repeated token cost and compute time.
|
|
269
|
-
- The legacy parser path
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
-
|
|
281
|
-
-
|
|
282
|
-
-
|
|
283
|
-
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
337
|
+
|
|
338
|
+
Example:
|
|
339
|
+
|
|
340
|
+
```powershell
|
|
341
|
+
.venv\Scripts\python.exe scripts\run_workflow_ingest_demo.py --output-dir logs\workflow_ingest_demo
|
|
342
|
+
```
|
|
343
|
+
|
|
344
|
+
External live server example:
|
|
345
|
+
|
|
346
|
+
```powershell
|
|
347
|
+
.venv\Scripts\python.exe scripts\run_workflow_ingest_demo.py --server-mode external_http --external-base-url http://127.0.0.1:28110
|
|
348
|
+
```
|
|
349
|
+
|
|
350
|
+
Demo artifacts are written into the chosen output directory:
|
|
351
|
+
|
|
352
|
+
- `probe-events.jsonl`: demo-friendly step and lifecycle probe events
|
|
353
|
+
- `demo-summary.json`: run summary, persistence result, and artifact pointers
|
|
354
|
+
- `llm-cache/`: workflow-native cached proposal/review call results
|
|
355
|
+
- `engines/`: local workflow and conversation graph storage for the run
|
|
356
|
+
- `server-data/`: isolated server-side persistence directory when the harness boots its own server
|
|
357
|
+
|
|
358
|
+
Notes:
|
|
359
|
+
|
|
360
|
+
- The workflow-native layer proposal/review path uses deterministic file-backed caching to reduce repeated token cost and compute time.
|
|
361
|
+
- The legacy parser path uses the shared Kogwistar cache facade. Set
|
|
362
|
+
`KG_DOC_PARSER_CACHE_BACKEND=auto|joblib|diskcache|none` and
|
|
363
|
+
`KG_DOC_PARSER_CACHE_DIR` to choose the provider and location. `auto` selects
|
|
364
|
+
DiskCache on PyPy and Joblib on CPython. The older
|
|
365
|
+
`KG_DOC_PARSER_JOBLIB_CACHE_DIR` variable remains a compatibility alias.
|
|
366
|
+
- Probe logging is separate from CDC and conversation graph traces, so demos can show a short readable event trail without digging into runtime internals.
|
|
367
|
+
|
|
368
|
+
## OCR And Parsing Workflows
|
|
369
|
+
|
|
370
|
+
The newer workflow-first paths are designed as reusable subworkflows:
|
|
371
|
+
|
|
372
|
+
- OCR image/PDF ingest
|
|
373
|
+
- resumable via `ocr-state.sqlite`
|
|
374
|
+
- emits `workflow-events.jsonl`
|
|
375
|
+
- keeps legacy OCR page artifacts on disk
|
|
376
|
+
- page-index parsing
|
|
377
|
+
- heuristic mode for deterministic structure extraction
|
|
378
|
+
- Ollama mode for local parser-backed parsing
|
|
379
|
+
- recursive layerwise parsing
|
|
380
|
+
- wraps the legacy recursive parser in a reusable workflow runner
|
|
381
|
+
|
|
382
|
+
These can be invoked from Python directly or through the `workflow-ingest`
|
|
383
|
+
CLI family, depending on whether you want reusable orchestration or a quick
|
|
384
|
+
shell command.
|
|
385
|
+
|
|
386
|
+
## Notes
|
|
387
|
+
|
|
388
|
+
- `README.md`, env handling, and ingestion boundaries are still being cleaned up as part of the ongoing refactor.
|
|
389
|
+
- Runtime outputs such as `logs/`, local `.env`, caches, and generated artifacts should remain uncommitted.
|
|
390
|
+
- If behavior diverges between this repo and `kogwistar`, prefer the direction of the ongoing migration and refactor work.
|
|
391
|
+
|
|
392
|
+
Two-stage conversation materialization is an engine-owned capability. The
|
|
393
|
+
parser writes grounded stage-one artifacts through the supplied conversation
|
|
394
|
+
engine and does not create a second embedding queue. See
|
|
395
|
+
[`doc/adr_conversation_two_stage_parser_contract.md`](doc/adr_conversation_two_stage_parser_contract.md).
|
|
396
|
+
|