graph-knowledge-doc-parser 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. graph_knowledge_doc_parser-0.1.0/PKG-INFO +326 -0
  2. graph_knowledge_doc_parser-0.1.0/README.md +294 -0
  3. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/__init__.py +9 -0
  4. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/cast_hinting.py +19 -0
  5. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/document_ingester_logger.py +766 -0
  6. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/models.py +277 -0
  7. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/ocr.py +752 -0
  8. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/pdf2png.py +286 -0
  9. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/semantic_document_splitting_layerwise_edits.py +3302 -0
  10. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/text_processing_utils.py +30 -0
  11. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/utils/__init__.py +0 -0
  12. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/utils/bounded_threadpool_executor.py +37 -0
  13. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/utils/file_loaders.py +405 -0
  14. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/utils/langchain.py +220 -0
  15. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/utils/log.py +135 -0
  16. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/utils/version_chaining.py +1278 -0
  17. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/workflow_ingest/__init__.py +187 -0
  18. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/workflow_ingest/_kogwistar.py +13 -0
  19. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/workflow_ingest/adapters.py +212 -0
  20. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/workflow_ingest/cache.py +63 -0
  21. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/workflow_ingest/cli.py +324 -0
  22. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/workflow_ingest/clients.py +444 -0
  23. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/workflow_ingest/demo_harness.py +427 -0
  24. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/workflow_ingest/design.py +208 -0
  25. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/workflow_ingest/handlers.py +617 -0
  26. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/workflow_ingest/models.py +575 -0
  27. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/workflow_ingest/ocr_pipeline.py +1581 -0
  28. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/workflow_ingest/page_index.py +473 -0
  29. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/workflow_ingest/parser_core.py +862 -0
  30. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/workflow_ingest/parsing.py +249 -0
  31. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/workflow_ingest/probe.py +164 -0
  32. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/workflow_ingest/providers.py +412 -0
  33. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/workflow_ingest/runners.py +546 -0
  34. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/workflow_ingest/semantics.py +231 -0
  35. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/workflow_ingest/service.py +112 -0
  36. graph_knowledge_doc_parser-0.1.0/kg_doc_parser/workflow_ingest/smoke_assets.py +62 -0
  37. graph_knowledge_doc_parser-0.1.0/pyproject.toml +46 -0
@@ -0,0 +1,326 @@
1
+ Metadata-Version: 2.4
2
+ Name: graph-knowledge-doc-parser
3
+ Version: 0.1.0
4
+ Summary: doc parser using llm driven engine with graph knowledge awareness
5
+ Author: humblemat810
6
+ Author-email: 67593116+humblemat810@users.noreply.github.com
7
+ Requires-Python: >=3.13,<4.0
8
+ Classifier: License :: Other/Proprietary License
9
+ Classifier: Operating System :: OS Independent
10
+ Classifier: Programming Language :: Python :: 3
11
+ Classifier: Programming Language :: Python :: 3.13
12
+ Classifier: Programming Language :: Python :: 3.14
13
+ Classifier: Programming Language :: Python :: 3 :: Only
14
+ Classifier: Topic :: Text Processing :: General
15
+ Requires-Dist: fastmcp (>=3.2.3,<4.0.0)
16
+ Requires-Dist: joblib (>=1.5.3,<2.0.0)
17
+ Requires-Dist: kogwistar (>=0.2.0,<0.3.0)
18
+ Requires-Dist: langchain-core (>=1.2.5,<2.0.0)
19
+ Requires-Dist: langchain-google-genai (>=4.1.2,<5.0.0)
20
+ Requires-Dist: mcp (>=1.25.0,<2.0.0)
21
+ Requires-Dist: pathspec (>=0.12.1,<0.13.0)
22
+ Requires-Dist: pdf2image (>=1.17.0,<2.0.0)
23
+ Requires-Dist: pikepdf (>=10.1.0,<11.0.0)
24
+ Requires-Dist: pydantic (>=2.12.5,<3.0.0)
25
+ Requires-Dist: pydantic-extension (>=0.0.7,<0.0.8)
26
+ Requires-Dist: pypdf (>=6.5.0,<7.0.0)
27
+ Requires-Dist: rapidfuzz (>=3.14.3,<4.0.0)
28
+ Project-URL: Homepage, https://github.com/humblemat810/kg_doc_parser
29
+ Project-URL: Repository, https://github.com/humblemat810/kg_doc_parser
30
+ Description-Content-Type: text/markdown
31
+
32
+ # Kogwistar-docparser
33
+
34
+ If you want the shortest path into the project, start with [QUICKSTART.md](QUICKSTART.md).
35
+
36
+ ## Graph Knowledge Doc Parser
37
+
38
+ Utilities and experiments for document ingestion, PDF splitting, OCR, and page-level parsing. Refactored from the kogwistar project as a stand alone ingestor.
39
+
40
+ ## Status
41
+
42
+ This repository is still being refactored and should be treated as work in progress.
43
+
44
+ The document ingestion pipeline is currently being extracted and consolidated into the main `kogwistar` repository. Until that refactor is complete, this repo should be considered an active staging area for parser and ingestion-related work.
45
+
46
+ ## What Is Here
47
+
48
+ - PDF splitting and image generation helpers in `src/pdf2png.py`
49
+ - Gemini-based OCR and page parsing flows in `src/ocr.py`
50
+ - SQLite-based ingestion telemetry in `src/document_ingester_logger.py`
51
+ - File discovery and filtering helpers in `src/utils/file_loaders.py`
52
+ - Experimental and regression-style tests under `tests/`
53
+
54
+ ## Workflow Surface
55
+
56
+ The reusable workflow-ingest code now has three layers:
57
+
58
+ - Python APIs, which are the primary contract for tests and orchestration
59
+ - CLI entrypoints, which are thin wrappers around those APIs
60
+ - composable subworkflows for OCR, page-index parsing, and recursive layerwise parsing
61
+
62
+ The reusable helpers live under `src/workflow_ingest/` and are designed so the
63
+ same core logic can be called from tests, scripts, and higher-level workflow
64
+ code without duplicating orchestration.
65
+
66
+ ### CLI Commands
67
+
68
+ After `poetry install`, the repo exposes a `workflow-ingest` command family:
69
+
70
+ ```powershell
71
+ workflow-ingest --help
72
+ workflow-ingest ocr --help
73
+ workflow-ingest page-index --help
74
+ workflow-ingest layerwise --help
75
+ workflow-ingest demo --help
76
+ workflow-ingest ocr-smoke-assets --help
77
+ ```
78
+
79
+ If you want the local checked-out `./kogwistar` subtree to win over the GitHub
80
+ dependency during development, run the Bash bootstrap helper after install:
81
+
82
+ ```powershell
83
+ bash ./scripts/bootstrap-dev.sh
84
+ ```
85
+
86
+ That script does two things:
87
+
88
+ - runs `poetry install`
89
+ - if `./kogwistar` exists, installs it editable into the active environment
90
+
91
+ If you do not run the bootstrap helper, the repo keeps using the GitHub-sourced
92
+ `kogwistar` dependency declared in `pyproject.toml`.
93
+
94
+ Typical examples:
95
+
96
+ ```powershell
97
+ workflow-ingest ocr tests\.tmp_workflow_ingest_ocr\generated_smoke_assets\ocr_smoke_document.pdf --output-dir logs\ocr_run
98
+ workflow-ingest ocr tests\.tmp_workflow_ingest_ocr\generated_smoke_assets --output-dir logs\ocr_batch
99
+ workflow-ingest page-index tests\fixtures\page_index\sample_page_index.txt --output-dir logs\page_index
100
+ workflow-ingest layerwise tests\.tmp_workflow_ingest_ocr\manual_cases\ollama\glm-ocr_latest\image\artifacts\legacy_split_pages\ocr-manual-ollama-image --output-dir logs\layerwise
101
+ workflow-ingest ocr-smoke-assets --output-dir tests\.tmp_workflow_ingest_ocr\generated_smoke_assets
102
+ ```
103
+
104
+ ### Reusable Artifacts
105
+
106
+ The workflow outputs are intentionally inspectable on disk:
107
+
108
+ - `workflow-events.jsonl`: readable step trail for the outer orchestration layer
109
+ - `ocr-state.sqlite`: authoritative OCR/render resume state
110
+ - `ocr-progress.json`: human-readable mirror of the current OCR state
111
+ - `ocr-summary.json`: final OCR run summary
112
+ - `legacy_split_pages/<document>/page_N.json`: legacy-compatible OCR page artifacts
113
+ - `rendered_pages/<document>/page_N.png`: rasterized page images
114
+ - `page-index-summary.json`: page-index run summary
115
+ - `layerwise-summary.json`: recursive layerwise parser summary
116
+ - `layerwise-graph.json`: legacy recursive layerwise graph payload
117
+
118
+ ### Python APIs
119
+
120
+ If you want to embed the pipelines directly, the main helpers are:
121
+
122
+ - `src.workflow_ingest.run_ocr_source_workflow(...)`
123
+ - `src.workflow_ingest.run_ocr_batch_workflow(...)`
124
+ - `src.workflow_ingest.parse_page_index_document(...)`
125
+ - `src.workflow_ingest.run_page_index_source_workflow(...)`
126
+ - `src.workflow_ingest.run_layerwise_source_workflow(...)`
127
+ - `src.workflow_ingest.run_demo_harness_workflow(...)`
128
+
129
+ Those helpers are meant to stay stable and are what the CLI layer calls under
130
+ the hood.
131
+
132
+ ## Setup
133
+
134
+ 1. Create and activate a Python 3.13 environment.
135
+ 2. Install dependencies with Poetry:
136
+
137
+ ```powershell
138
+ poetry install
139
+ ```
140
+
141
+ 3. Create a local env file from the example:
142
+
143
+ ```powershell
144
+ Copy-Item .env.example .env
145
+ ```
146
+
147
+ 4. Add your local `GOOGLE_API_KEY` and any optional file-list paths needed for your workflow.
148
+
149
+ ## Environment Variables
150
+
151
+ The project currently expects or optionally uses:
152
+
153
+ - `GOOGLE_API_KEY`: required for Gemini OCR and LLM-backed parsing flows
154
+ - `LANGSMITH_TRACING`: optional LangSmith tracing toggle
155
+ - `ocr_file_list`: optional allow-list file for OCR runs
156
+ - `split_raw_file_list`: optional allow-list file for PDF splitting runs
157
+ - `answer_export_list`: optional export list path used by local workflows
158
+
159
+ An example template is provided in [`.env.example`](/c:/Users/chanh/Documents/kg_doc_parser/.env.example).
160
+
161
+ ## Provider Guide
162
+
163
+ The workflow layer is vendor-neutral, but the concrete OCR, parser, and embedding
164
+ backends are selected by config.
165
+
166
+ ### OCR Provider Examples
167
+
168
+ - Google GenAI OCR:
169
+ - `KG_DOC_OCR_PROVIDER=gemini`
170
+ - `KG_DOC_OCR_MODEL=gemini-2.5-flash`
171
+ - Ollama OCR or vision-capable local model:
172
+ - `KG_DOC_OCR_PROVIDER=ollama`
173
+ - `KG_DOC_OCR_MODEL=llava:latest`
174
+ - `KG_DOC_OCR_BASE_URL=http://127.0.0.1:11434`
175
+ - Vertex AI OCR:
176
+ - `KG_DOC_OCR_PROVIDER=vertex`
177
+ - `KG_DOC_OCR_MODEL=gemini-2.5-pro`
178
+ - `KG_DOC_OCR_PROJECT=my-project`
179
+ - `KG_DOC_OCR_LOCATION=us-central1`
180
+
181
+ ### Parser / LLM Provider Examples
182
+
183
+ The parser provider is the chat model used for semantic parsing, layer review,
184
+ and structured extraction.
185
+
186
+ - LangChain Google GenAI:
187
+ - `KG_DOC_PARSER_PROVIDER=gemini`
188
+ - `KG_DOC_PARSER_MODEL=gemini-2.5-flash`
189
+ - ChatGPT / OpenAI REST:
190
+ - `KG_DOC_PARSER_PROVIDER=openai`
191
+ - `KG_DOC_PARSER_MODEL=gpt-4.1-mini`
192
+ - `KG_DOC_PARSER_API_KEY_ENV=OPENAI_API_KEY`
193
+ - LangChain Ollama:
194
+ - `KG_DOC_PARSER_PROVIDER=ollama`
195
+ - `KG_DOC_PARSER_MODEL=llama3.1`
196
+ - `KG_DOC_PARSER_BASE_URL=http://127.0.0.1:11434`
197
+ - LangChain Vertex AI:
198
+ - `KG_DOC_PARSER_PROVIDER=vertex`
199
+ - `KG_DOC_PARSER_MODEL=gemini-2.5-pro`
200
+ - `KG_DOC_PARSER_PROJECT=my-project`
201
+ - `KG_DOC_PARSER_LOCATION=us-central1`
202
+
203
+ ### Recipe Parsing Example
204
+
205
+ If you are parsing a cooking recipe, one practical split is:
206
+
207
+ - OCR on Gemini or another vision model
208
+ - parser on OpenAI, Ollama, or Vertex AI
209
+
210
+ For example:
211
+
212
+ ```powershell
213
+ KG_DOC_OCR_PROVIDER=gemini
214
+ KG_DOC_OCR_MODEL=gemini-2.5-flash
215
+ KG_DOC_PARSER_PROVIDER=openai
216
+ KG_DOC_PARSER_MODEL=gpt-4.1-mini
217
+ KG_DOC_PARSER_API_KEY_ENV=OPENAI_API_KEY
218
+ KG_DOC_EMBED_PROVIDER=fake
219
+ ```
220
+
221
+ That setup can extract a recipe into structured graph data such as:
222
+ - ingredients
223
+ - steps
224
+ - tools
225
+ - timers
226
+ - inferred sections like `prep`, `cook`, and `serve`
227
+
228
+ ### Embedding Examples
229
+
230
+ - Fake deterministic CI embedding:
231
+ - `KG_DOC_EMBED_PROVIDER=fake`
232
+ - OpenAI embeddings:
233
+ - `KG_DOC_EMBED_PROVIDER=openai`
234
+ - `KG_DOC_EMBED_MODEL=text-embedding-3-small`
235
+ - Vertex AI embeddings:
236
+ - `KG_DOC_EMBED_PROVIDER=vertex`
237
+ - `KG_DOC_EMBED_MODEL=text-embedding-004`
238
+ - Ollama embeddings:
239
+ - `KG_DOC_EMBED_PROVIDER=ollama`
240
+ - `KG_DOC_EMBED_MODEL=nomic-embed-text`
241
+
242
+ Note:
243
+
244
+ - `embedding_space` in the workflow ingest models is currently a metadata and
245
+ routing-intent label.
246
+ - It does not yet imply that the engine is using a separate embedder per space.
247
+ - The current engine bootstrap still wires one embedding function per engine
248
+ instance, while the multi-space routing proposal remains a future Kogwistar
249
+ core concern.
250
+
251
+ ## Running Tests
252
+
253
+ Some tests are integration-style and expect local document folders and API credentials to exist. That means not every test is portable in a clean checkout.
254
+
255
+ To run the test suite:
256
+
257
+ ```powershell
258
+ pytest -q
259
+ ```
260
+
261
+ If you only want to work on isolated units, review the test files first and run a narrower subset.
262
+
263
+ ## Demo Harness
264
+
265
+ There is now a manual workflow-ingest demo harness that can run the end-to-end flow against:
266
+
267
+ - an in-process isolated FastAPI server
268
+ - a subprocess-hosted local server
269
+ - an already running external Kogwistar server
270
+
271
+ The legacy semantic-smoke test in
272
+ [`tests/test_semantic_layerwise_doc_parsing.py`](/c:/Users/chanh/Documents/kg_doc_parser/tests/test_semantic_layerwise_doc_parsing.py)
273
+ also expects a live Kogwistar server at `http://127.0.0.1:28110`. It does not
274
+ start that server for you, so use the VS Code server launch config or start it
275
+ manually before running the Ollama case.
276
+
277
+ Example:
278
+
279
+ ```powershell
280
+ .venv\Scripts\python.exe scripts\run_workflow_ingest_demo.py --output-dir logs\workflow_ingest_demo
281
+ ```
282
+
283
+ External live server example:
284
+
285
+ ```powershell
286
+ .venv\Scripts\python.exe scripts\run_workflow_ingest_demo.py --server-mode external_http --external-base-url http://127.0.0.1:28110
287
+ ```
288
+
289
+ Demo artifacts are written into the chosen output directory:
290
+
291
+ - `probe-events.jsonl`: demo-friendly step and lifecycle probe events
292
+ - `demo-summary.json`: run summary, persistence result, and artifact pointers
293
+ - `llm-cache/`: workflow-native cached proposal/review call results
294
+ - `engines/`: local workflow and conversation graph storage for the run
295
+ - `server-data/`: isolated server-side persistence directory when the harness boots its own server
296
+
297
+ Notes:
298
+
299
+ - The workflow-native layer proposal/review path uses deterministic file-backed caching to reduce repeated token cost and compute time.
300
+ - The legacy parser path also supports a redirected `joblib` cache via `KG_DOC_PARSER_JOBLIB_CACHE_DIR`.
301
+ - Probe logging is separate from CDC and conversation graph traces, so demos can show a short readable event trail without digging into runtime internals.
302
+
303
+ ## OCR And Parsing Workflows
304
+
305
+ The newer workflow-first paths are designed as reusable subworkflows:
306
+
307
+ - OCR image/PDF ingest
308
+ - resumable via `ocr-state.sqlite`
309
+ - emits `workflow-events.jsonl`
310
+ - keeps legacy OCR page artifacts on disk
311
+ - page-index parsing
312
+ - heuristic mode for deterministic structure extraction
313
+ - Ollama mode for local parser-backed parsing
314
+ - recursive layerwise parsing
315
+ - wraps the legacy recursive parser in a reusable workflow runner
316
+
317
+ These can be invoked from Python directly or through the `workflow-ingest`
318
+ CLI family, depending on whether you want reusable orchestration or a quick
319
+ shell command.
320
+
321
+ ## Notes
322
+
323
+ - `README.md`, env handling, and ingestion boundaries are still being cleaned up as part of the ongoing refactor.
324
+ - Runtime outputs such as `logs/`, local `.env`, caches, and generated artifacts should remain uncommitted.
325
+ - If behavior diverges between this repo and `kogwistar`, prefer the direction of the ongoing migration and refactor work.
326
+
@@ -0,0 +1,294 @@
1
+ # Kogwistar-docparser
2
+
3
+ If you want the shortest path into the project, start with [QUICKSTART.md](QUICKSTART.md).
4
+
5
+ ## Graph Knowledge Doc Parser
6
+
7
+ Utilities and experiments for document ingestion, PDF splitting, OCR, and page-level parsing. Refactored from the kogwistar project as a stand alone ingestor.
8
+
9
+ ## Status
10
+
11
+ This repository is still being refactored and should be treated as work in progress.
12
+
13
+ The document ingestion pipeline is currently being extracted and consolidated into the main `kogwistar` repository. Until that refactor is complete, this repo should be considered an active staging area for parser and ingestion-related work.
14
+
15
+ ## What Is Here
16
+
17
+ - PDF splitting and image generation helpers in `src/pdf2png.py`
18
+ - Gemini-based OCR and page parsing flows in `src/ocr.py`
19
+ - SQLite-based ingestion telemetry in `src/document_ingester_logger.py`
20
+ - File discovery and filtering helpers in `src/utils/file_loaders.py`
21
+ - Experimental and regression-style tests under `tests/`
22
+
23
+ ## Workflow Surface
24
+
25
+ The reusable workflow-ingest code now has three layers:
26
+
27
+ - Python APIs, which are the primary contract for tests and orchestration
28
+ - CLI entrypoints, which are thin wrappers around those APIs
29
+ - composable subworkflows for OCR, page-index parsing, and recursive layerwise parsing
30
+
31
+ The reusable helpers live under `src/workflow_ingest/` and are designed so the
32
+ same core logic can be called from tests, scripts, and higher-level workflow
33
+ code without duplicating orchestration.
34
+
35
+ ### CLI Commands
36
+
37
+ After `poetry install`, the repo exposes a `workflow-ingest` command family:
38
+
39
+ ```powershell
40
+ workflow-ingest --help
41
+ workflow-ingest ocr --help
42
+ workflow-ingest page-index --help
43
+ workflow-ingest layerwise --help
44
+ workflow-ingest demo --help
45
+ workflow-ingest ocr-smoke-assets --help
46
+ ```
47
+
48
+ If you want the local checked-out `./kogwistar` subtree to win over the GitHub
49
+ dependency during development, run the Bash bootstrap helper after install:
50
+
51
+ ```powershell
52
+ bash ./scripts/bootstrap-dev.sh
53
+ ```
54
+
55
+ That script does two things:
56
+
57
+ - runs `poetry install`
58
+ - if `./kogwistar` exists, installs it editable into the active environment
59
+
60
+ If you do not run the bootstrap helper, the repo keeps using the GitHub-sourced
61
+ `kogwistar` dependency declared in `pyproject.toml`.
62
+
63
+ Typical examples:
64
+
65
+ ```powershell
66
+ workflow-ingest ocr tests\.tmp_workflow_ingest_ocr\generated_smoke_assets\ocr_smoke_document.pdf --output-dir logs\ocr_run
67
+ workflow-ingest ocr tests\.tmp_workflow_ingest_ocr\generated_smoke_assets --output-dir logs\ocr_batch
68
+ workflow-ingest page-index tests\fixtures\page_index\sample_page_index.txt --output-dir logs\page_index
69
+ workflow-ingest layerwise tests\.tmp_workflow_ingest_ocr\manual_cases\ollama\glm-ocr_latest\image\artifacts\legacy_split_pages\ocr-manual-ollama-image --output-dir logs\layerwise
70
+ workflow-ingest ocr-smoke-assets --output-dir tests\.tmp_workflow_ingest_ocr\generated_smoke_assets
71
+ ```
72
+
73
+ ### Reusable Artifacts
74
+
75
+ The workflow outputs are intentionally inspectable on disk:
76
+
77
+ - `workflow-events.jsonl`: readable step trail for the outer orchestration layer
78
+ - `ocr-state.sqlite`: authoritative OCR/render resume state
79
+ - `ocr-progress.json`: human-readable mirror of the current OCR state
80
+ - `ocr-summary.json`: final OCR run summary
81
+ - `legacy_split_pages/<document>/page_N.json`: legacy-compatible OCR page artifacts
82
+ - `rendered_pages/<document>/page_N.png`: rasterized page images
83
+ - `page-index-summary.json`: page-index run summary
84
+ - `layerwise-summary.json`: recursive layerwise parser summary
85
+ - `layerwise-graph.json`: legacy recursive layerwise graph payload
86
+
87
+ ### Python APIs
88
+
89
+ If you want to embed the pipelines directly, the main helpers are:
90
+
91
+ - `src.workflow_ingest.run_ocr_source_workflow(...)`
92
+ - `src.workflow_ingest.run_ocr_batch_workflow(...)`
93
+ - `src.workflow_ingest.parse_page_index_document(...)`
94
+ - `src.workflow_ingest.run_page_index_source_workflow(...)`
95
+ - `src.workflow_ingest.run_layerwise_source_workflow(...)`
96
+ - `src.workflow_ingest.run_demo_harness_workflow(...)`
97
+
98
+ Those helpers are meant to stay stable and are what the CLI layer calls under
99
+ the hood.
100
+
101
+ ## Setup
102
+
103
+ 1. Create and activate a Python 3.13 environment.
104
+ 2. Install dependencies with Poetry:
105
+
106
+ ```powershell
107
+ poetry install
108
+ ```
109
+
110
+ 3. Create a local env file from the example:
111
+
112
+ ```powershell
113
+ Copy-Item .env.example .env
114
+ ```
115
+
116
+ 4. Add your local `GOOGLE_API_KEY` and any optional file-list paths needed for your workflow.
117
+
118
+ ## Environment Variables
119
+
120
+ The project currently expects or optionally uses:
121
+
122
+ - `GOOGLE_API_KEY`: required for Gemini OCR and LLM-backed parsing flows
123
+ - `LANGSMITH_TRACING`: optional LangSmith tracing toggle
124
+ - `ocr_file_list`: optional allow-list file for OCR runs
125
+ - `split_raw_file_list`: optional allow-list file for PDF splitting runs
126
+ - `answer_export_list`: optional export list path used by local workflows
127
+
128
+ An example template is provided in [`.env.example`](/c:/Users/chanh/Documents/kg_doc_parser/.env.example).
129
+
130
+ ## Provider Guide
131
+
132
+ The workflow layer is vendor-neutral, but the concrete OCR, parser, and embedding
133
+ backends are selected by config.
134
+
135
+ ### OCR Provider Examples
136
+
137
+ - Google GenAI OCR:
138
+ - `KG_DOC_OCR_PROVIDER=gemini`
139
+ - `KG_DOC_OCR_MODEL=gemini-2.5-flash`
140
+ - Ollama OCR or vision-capable local model:
141
+ - `KG_DOC_OCR_PROVIDER=ollama`
142
+ - `KG_DOC_OCR_MODEL=llava:latest`
143
+ - `KG_DOC_OCR_BASE_URL=http://127.0.0.1:11434`
144
+ - Vertex AI OCR:
145
+ - `KG_DOC_OCR_PROVIDER=vertex`
146
+ - `KG_DOC_OCR_MODEL=gemini-2.5-pro`
147
+ - `KG_DOC_OCR_PROJECT=my-project`
148
+ - `KG_DOC_OCR_LOCATION=us-central1`
149
+
150
+ ### Parser / LLM Provider Examples
151
+
152
+ The parser provider is the chat model used for semantic parsing, layer review,
153
+ and structured extraction.
154
+
155
+ - LangChain Google GenAI:
156
+ - `KG_DOC_PARSER_PROVIDER=gemini`
157
+ - `KG_DOC_PARSER_MODEL=gemini-2.5-flash`
158
+ - ChatGPT / OpenAI REST:
159
+ - `KG_DOC_PARSER_PROVIDER=openai`
160
+ - `KG_DOC_PARSER_MODEL=gpt-4.1-mini`
161
+ - `KG_DOC_PARSER_API_KEY_ENV=OPENAI_API_KEY`
162
+ - LangChain Ollama:
163
+ - `KG_DOC_PARSER_PROVIDER=ollama`
164
+ - `KG_DOC_PARSER_MODEL=llama3.1`
165
+ - `KG_DOC_PARSER_BASE_URL=http://127.0.0.1:11434`
166
+ - LangChain Vertex AI:
167
+ - `KG_DOC_PARSER_PROVIDER=vertex`
168
+ - `KG_DOC_PARSER_MODEL=gemini-2.5-pro`
169
+ - `KG_DOC_PARSER_PROJECT=my-project`
170
+ - `KG_DOC_PARSER_LOCATION=us-central1`
171
+
172
+ ### Recipe Parsing Example
173
+
174
+ If you are parsing a cooking recipe, one practical split is:
175
+
176
+ - OCR on Gemini or another vision model
177
+ - parser on OpenAI, Ollama, or Vertex AI
178
+
179
+ For example:
180
+
181
+ ```powershell
182
+ KG_DOC_OCR_PROVIDER=gemini
183
+ KG_DOC_OCR_MODEL=gemini-2.5-flash
184
+ KG_DOC_PARSER_PROVIDER=openai
185
+ KG_DOC_PARSER_MODEL=gpt-4.1-mini
186
+ KG_DOC_PARSER_API_KEY_ENV=OPENAI_API_KEY
187
+ KG_DOC_EMBED_PROVIDER=fake
188
+ ```
189
+
190
+ That setup can extract a recipe into structured graph data such as:
191
+ - ingredients
192
+ - steps
193
+ - tools
194
+ - timers
195
+ - inferred sections like `prep`, `cook`, and `serve`
196
+
197
+ ### Embedding Examples
198
+
199
+ - Fake deterministic CI embedding:
200
+ - `KG_DOC_EMBED_PROVIDER=fake`
201
+ - OpenAI embeddings:
202
+ - `KG_DOC_EMBED_PROVIDER=openai`
203
+ - `KG_DOC_EMBED_MODEL=text-embedding-3-small`
204
+ - Vertex AI embeddings:
205
+ - `KG_DOC_EMBED_PROVIDER=vertex`
206
+ - `KG_DOC_EMBED_MODEL=text-embedding-004`
207
+ - Ollama embeddings:
208
+ - `KG_DOC_EMBED_PROVIDER=ollama`
209
+ - `KG_DOC_EMBED_MODEL=nomic-embed-text`
210
+
211
+ Note:
212
+
213
+ - `embedding_space` in the workflow ingest models is currently a metadata and
214
+ routing-intent label.
215
+ - It does not yet imply that the engine is using a separate embedder per space.
216
+ - The current engine bootstrap still wires one embedding function per engine
217
+ instance, while the multi-space routing proposal remains a future Kogwistar
218
+ core concern.
219
+
220
+ ## Running Tests
221
+
222
+ Some tests are integration-style and expect local document folders and API credentials to exist. That means not every test is portable in a clean checkout.
223
+
224
+ To run the test suite:
225
+
226
+ ```powershell
227
+ pytest -q
228
+ ```
229
+
230
+ If you only want to work on isolated units, review the test files first and run a narrower subset.
231
+
232
+ ## Demo Harness
233
+
234
+ There is now a manual workflow-ingest demo harness that can run the end-to-end flow against:
235
+
236
+ - an in-process isolated FastAPI server
237
+ - a subprocess-hosted local server
238
+ - an already running external Kogwistar server
239
+
240
+ The legacy semantic-smoke test in
241
+ [`tests/test_semantic_layerwise_doc_parsing.py`](/c:/Users/chanh/Documents/kg_doc_parser/tests/test_semantic_layerwise_doc_parsing.py)
242
+ also expects a live Kogwistar server at `http://127.0.0.1:28110`. It does not
243
+ start that server for you, so use the VS Code server launch config or start it
244
+ manually before running the Ollama case.
245
+
246
+ Example:
247
+
248
+ ```powershell
249
+ .venv\Scripts\python.exe scripts\run_workflow_ingest_demo.py --output-dir logs\workflow_ingest_demo
250
+ ```
251
+
252
+ External live server example:
253
+
254
+ ```powershell
255
+ .venv\Scripts\python.exe scripts\run_workflow_ingest_demo.py --server-mode external_http --external-base-url http://127.0.0.1:28110
256
+ ```
257
+
258
+ Demo artifacts are written into the chosen output directory:
259
+
260
+ - `probe-events.jsonl`: demo-friendly step and lifecycle probe events
261
+ - `demo-summary.json`: run summary, persistence result, and artifact pointers
262
+ - `llm-cache/`: workflow-native cached proposal/review call results
263
+ - `engines/`: local workflow and conversation graph storage for the run
264
+ - `server-data/`: isolated server-side persistence directory when the harness boots its own server
265
+
266
+ Notes:
267
+
268
+ - The workflow-native layer proposal/review path uses deterministic file-backed caching to reduce repeated token cost and compute time.
269
+ - The legacy parser path also supports a redirected `joblib` cache via `KG_DOC_PARSER_JOBLIB_CACHE_DIR`.
270
+ - Probe logging is separate from CDC and conversation graph traces, so demos can show a short readable event trail without digging into runtime internals.
271
+
272
+ ## OCR And Parsing Workflows
273
+
274
+ The newer workflow-first paths are designed as reusable subworkflows:
275
+
276
+ - OCR image/PDF ingest
277
+ - resumable via `ocr-state.sqlite`
278
+ - emits `workflow-events.jsonl`
279
+ - keeps legacy OCR page artifacts on disk
280
+ - page-index parsing
281
+ - heuristic mode for deterministic structure extraction
282
+ - Ollama mode for local parser-backed parsing
283
+ - recursive layerwise parsing
284
+ - wraps the legacy recursive parser in a reusable workflow runner
285
+
286
+ These can be invoked from Python directly or through the `workflow-ingest`
287
+ CLI family, depending on whether you want reusable orchestration or a quick
288
+ shell command.
289
+
290
+ ## Notes
291
+
292
+ - `README.md`, env handling, and ingestion boundaries are still being cleaned up as part of the ongoing refactor.
293
+ - Runtime outputs such as `logs/`, local `.env`, caches, and generated artifacts should remain uncommitted.
294
+ - If behavior diverges between this repo and `kogwistar`, prefer the direction of the ongoing migration and refactor work.
@@ -0,0 +1,9 @@
1
+ from __future__ import annotations
2
+
3
+ """Public import surface for the document parser package."""
4
+
5
+ from . import workflow_ingest
6
+ from .workflow_ingest import * # noqa: F401,F403
7
+ from .workflow_ingest import __all__ as _workflow_ingest_all
8
+
9
+ __all__ = ["workflow_ingest", *_workflow_ingest_all]
@@ -0,0 +1,19 @@
1
+ from __future__ import annotations
2
+ from typing import Callable, TypeVar, ParamSpec, cast
3
+ from joblib import Memory
4
+
5
+ P = ParamSpec("P")
6
+ R = TypeVar("R")
7
+
8
+ def cached(memory: Memory, fn: Callable[P, R]) -> Callable[P, R]:
9
+ return cast(Callable[P, R], memory.cache(fn))
10
+
11
+ memory = Memory(".cache", verbose=0)
12
+
13
+ def compute(a: int, b: float, *, mode: str = "x") -> tuple[int, float]:
14
+ return (a, b)
15
+
16
+ compute_cached = cached(memory, compute)
17
+
18
+ # Type checker should now know:
19
+ # (a: int, b: float, *, mode: str = ...) -> tuple[int, float]