graph-knowledge-doc-parser 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. graph_knowledge_doc_parser-0.1.0.dist-info/METADATA +326 -0
  2. graph_knowledge_doc_parser-0.1.0.dist-info/RECORD +38 -0
  3. graph_knowledge_doc_parser-0.1.0.dist-info/WHEEL +4 -0
  4. graph_knowledge_doc_parser-0.1.0.dist-info/entry_points.txt +3 -0
  5. kg_doc_parser/__init__.py +9 -0
  6. kg_doc_parser/cast_hinting.py +19 -0
  7. kg_doc_parser/document_ingester_logger.py +766 -0
  8. kg_doc_parser/models.py +277 -0
  9. kg_doc_parser/ocr.py +752 -0
  10. kg_doc_parser/pdf2png.py +286 -0
  11. kg_doc_parser/semantic_document_splitting_layerwise_edits.py +3302 -0
  12. kg_doc_parser/text_processing_utils.py +30 -0
  13. kg_doc_parser/utils/__init__.py +0 -0
  14. kg_doc_parser/utils/bounded_threadpool_executor.py +37 -0
  15. kg_doc_parser/utils/file_loaders.py +405 -0
  16. kg_doc_parser/utils/langchain.py +220 -0
  17. kg_doc_parser/utils/log.py +135 -0
  18. kg_doc_parser/utils/version_chaining.py +1278 -0
  19. kg_doc_parser/workflow_ingest/__init__.py +187 -0
  20. kg_doc_parser/workflow_ingest/_kogwistar.py +13 -0
  21. kg_doc_parser/workflow_ingest/adapters.py +212 -0
  22. kg_doc_parser/workflow_ingest/cache.py +63 -0
  23. kg_doc_parser/workflow_ingest/cli.py +324 -0
  24. kg_doc_parser/workflow_ingest/clients.py +444 -0
  25. kg_doc_parser/workflow_ingest/demo_harness.py +427 -0
  26. kg_doc_parser/workflow_ingest/design.py +208 -0
  27. kg_doc_parser/workflow_ingest/handlers.py +617 -0
  28. kg_doc_parser/workflow_ingest/models.py +575 -0
  29. kg_doc_parser/workflow_ingest/ocr_pipeline.py +1581 -0
  30. kg_doc_parser/workflow_ingest/page_index.py +473 -0
  31. kg_doc_parser/workflow_ingest/parser_core.py +862 -0
  32. kg_doc_parser/workflow_ingest/parsing.py +249 -0
  33. kg_doc_parser/workflow_ingest/probe.py +164 -0
  34. kg_doc_parser/workflow_ingest/providers.py +412 -0
  35. kg_doc_parser/workflow_ingest/runners.py +546 -0
  36. kg_doc_parser/workflow_ingest/semantics.py +231 -0
  37. kg_doc_parser/workflow_ingest/service.py +112 -0
  38. kg_doc_parser/workflow_ingest/smoke_assets.py +62 -0
@@ -0,0 +1,326 @@
1
+ Metadata-Version: 2.4
2
+ Name: graph-knowledge-doc-parser
3
+ Version: 0.1.0
4
+ Summary: doc parser using llm driven engine with graph knowledge awareness
5
+ Author: humblemat810
6
+ Author-email: 67593116+humblemat810@users.noreply.github.com
7
+ Requires-Python: >=3.13,<4.0
8
+ Classifier: License :: Other/Proprietary License
9
+ Classifier: Operating System :: OS Independent
10
+ Classifier: Programming Language :: Python :: 3
11
+ Classifier: Programming Language :: Python :: 3.13
12
+ Classifier: Programming Language :: Python :: 3.14
13
+ Classifier: Programming Language :: Python :: 3 :: Only
14
+ Classifier: Topic :: Text Processing :: General
15
+ Requires-Dist: fastmcp (>=3.2.3,<4.0.0)
16
+ Requires-Dist: joblib (>=1.5.3,<2.0.0)
17
+ Requires-Dist: kogwistar (>=0.2.0,<0.3.0)
18
+ Requires-Dist: langchain-core (>=1.2.5,<2.0.0)
19
+ Requires-Dist: langchain-google-genai (>=4.1.2,<5.0.0)
20
+ Requires-Dist: mcp (>=1.25.0,<2.0.0)
21
+ Requires-Dist: pathspec (>=0.12.1,<0.13.0)
22
+ Requires-Dist: pdf2image (>=1.17.0,<2.0.0)
23
+ Requires-Dist: pikepdf (>=10.1.0,<11.0.0)
24
+ Requires-Dist: pydantic (>=2.12.5,<3.0.0)
25
+ Requires-Dist: pydantic-extension (>=0.0.7,<0.0.8)
26
+ Requires-Dist: pypdf (>=6.5.0,<7.0.0)
27
+ Requires-Dist: rapidfuzz (>=3.14.3,<4.0.0)
28
+ Project-URL: Homepage, https://github.com/humblemat810/kg_doc_parser
29
+ Project-URL: Repository, https://github.com/humblemat810/kg_doc_parser
30
+ Description-Content-Type: text/markdown
31
+
32
+ # Kogwistar-docparser
33
+
34
+ If you want the shortest path into the project, start with [QUICKSTART.md](QUICKSTART.md).
35
+
36
+ ## Graph Knowledge Doc Parser
37
+
38
+ Utilities and experiments for document ingestion, PDF splitting, OCR, and page-level parsing. Refactored from the kogwistar project as a stand alone ingestor.
39
+
40
+ ## Status
41
+
42
+ This repository is still being refactored and should be treated as work in progress.
43
+
44
+ The document ingestion pipeline is currently being extracted and consolidated into the main `kogwistar` repository. Until that refactor is complete, this repo should be considered an active staging area for parser and ingestion-related work.
45
+
46
+ ## What Is Here
47
+
48
+ - PDF splitting and image generation helpers in `src/pdf2png.py`
49
+ - Gemini-based OCR and page parsing flows in `src/ocr.py`
50
+ - SQLite-based ingestion telemetry in `src/document_ingester_logger.py`
51
+ - File discovery and filtering helpers in `src/utils/file_loaders.py`
52
+ - Experimental and regression-style tests under `tests/`
53
+
54
+ ## Workflow Surface
55
+
56
+ The reusable workflow-ingest code now has three layers:
57
+
58
+ - Python APIs, which are the primary contract for tests and orchestration
59
+ - CLI entrypoints, which are thin wrappers around those APIs
60
+ - composable subworkflows for OCR, page-index parsing, and recursive layerwise parsing
61
+
62
+ The reusable helpers live under `src/workflow_ingest/` and are designed so the
63
+ same core logic can be called from tests, scripts, and higher-level workflow
64
+ code without duplicating orchestration.
65
+
66
+ ### CLI Commands
67
+
68
+ After `poetry install`, the repo exposes a `workflow-ingest` command family:
69
+
70
+ ```powershell
71
+ workflow-ingest --help
72
+ workflow-ingest ocr --help
73
+ workflow-ingest page-index --help
74
+ workflow-ingest layerwise --help
75
+ workflow-ingest demo --help
76
+ workflow-ingest ocr-smoke-assets --help
77
+ ```
78
+
79
+ If you want the local checked-out `./kogwistar` subtree to win over the GitHub
80
+ dependency during development, run the Bash bootstrap helper after install:
81
+
82
+ ```powershell
83
+ bash ./scripts/bootstrap-dev.sh
84
+ ```
85
+
86
+ That script does two things:
87
+
88
+ - runs `poetry install`
89
+ - if `./kogwistar` exists, installs it editable into the active environment
90
+
91
+ If you do not run the bootstrap helper, the repo keeps using the GitHub-sourced
92
+ `kogwistar` dependency declared in `pyproject.toml`.
93
+
94
+ Typical examples:
95
+
96
+ ```powershell
97
+ workflow-ingest ocr tests\.tmp_workflow_ingest_ocr\generated_smoke_assets\ocr_smoke_document.pdf --output-dir logs\ocr_run
98
+ workflow-ingest ocr tests\.tmp_workflow_ingest_ocr\generated_smoke_assets --output-dir logs\ocr_batch
99
+ workflow-ingest page-index tests\fixtures\page_index\sample_page_index.txt --output-dir logs\page_index
100
+ workflow-ingest layerwise tests\.tmp_workflow_ingest_ocr\manual_cases\ollama\glm-ocr_latest\image\artifacts\legacy_split_pages\ocr-manual-ollama-image --output-dir logs\layerwise
101
+ workflow-ingest ocr-smoke-assets --output-dir tests\.tmp_workflow_ingest_ocr\generated_smoke_assets
102
+ ```
103
+
104
+ ### Reusable Artifacts
105
+
106
+ The workflow outputs are intentionally inspectable on disk:
107
+
108
+ - `workflow-events.jsonl`: readable step trail for the outer orchestration layer
109
+ - `ocr-state.sqlite`: authoritative OCR/render resume state
110
+ - `ocr-progress.json`: human-readable mirror of the current OCR state
111
+ - `ocr-summary.json`: final OCR run summary
112
+ - `legacy_split_pages/<document>/page_N.json`: legacy-compatible OCR page artifacts
113
+ - `rendered_pages/<document>/page_N.png`: rasterized page images
114
+ - `page-index-summary.json`: page-index run summary
115
+ - `layerwise-summary.json`: recursive layerwise parser summary
116
+ - `layerwise-graph.json`: legacy recursive layerwise graph payload
117
+
118
+ ### Python APIs
119
+
120
+ If you want to embed the pipelines directly, the main helpers are:
121
+
122
+ - `src.workflow_ingest.run_ocr_source_workflow(...)`
123
+ - `src.workflow_ingest.run_ocr_batch_workflow(...)`
124
+ - `src.workflow_ingest.parse_page_index_document(...)`
125
+ - `src.workflow_ingest.run_page_index_source_workflow(...)`
126
+ - `src.workflow_ingest.run_layerwise_source_workflow(...)`
127
+ - `src.workflow_ingest.run_demo_harness_workflow(...)`
128
+
129
+ Those helpers are meant to stay stable and are what the CLI layer calls under
130
+ the hood.
131
+
132
+ ## Setup
133
+
134
+ 1. Create and activate a Python 3.13 environment.
135
+ 2. Install dependencies with Poetry:
136
+
137
+ ```powershell
138
+ poetry install
139
+ ```
140
+
141
+ 3. Create a local env file from the example:
142
+
143
+ ```powershell
144
+ Copy-Item .env.example .env
145
+ ```
146
+
147
+ 4. Add your local `GOOGLE_API_KEY` and any optional file-list paths needed for your workflow.
148
+
149
+ ## Environment Variables
150
+
151
+ The project currently expects or optionally uses:
152
+
153
+ - `GOOGLE_API_KEY`: required for Gemini OCR and LLM-backed parsing flows
154
+ - `LANGSMITH_TRACING`: optional LangSmith tracing toggle
155
+ - `ocr_file_list`: optional allow-list file for OCR runs
156
+ - `split_raw_file_list`: optional allow-list file for PDF splitting runs
157
+ - `answer_export_list`: optional export list path used by local workflows
158
+
159
+ An example template is provided in [`.env.example`](/c:/Users/chanh/Documents/kg_doc_parser/.env.example).
160
+
161
+ ## Provider Guide
162
+
163
+ The workflow layer is vendor-neutral, but the concrete OCR, parser, and embedding
164
+ backends are selected by config.
165
+
166
+ ### OCR Provider Examples
167
+
168
+ - Google GenAI OCR:
169
+ - `KG_DOC_OCR_PROVIDER=gemini`
170
+ - `KG_DOC_OCR_MODEL=gemini-2.5-flash`
171
+ - Ollama OCR or vision-capable local model:
172
+ - `KG_DOC_OCR_PROVIDER=ollama`
173
+ - `KG_DOC_OCR_MODEL=llava:latest`
174
+ - `KG_DOC_OCR_BASE_URL=http://127.0.0.1:11434`
175
+ - Vertex AI OCR:
176
+ - `KG_DOC_OCR_PROVIDER=vertex`
177
+ - `KG_DOC_OCR_MODEL=gemini-2.5-pro`
178
+ - `KG_DOC_OCR_PROJECT=my-project`
179
+ - `KG_DOC_OCR_LOCATION=us-central1`
180
+
181
+ ### Parser / LLM Provider Examples
182
+
183
+ The parser provider is the chat model used for semantic parsing, layer review,
184
+ and structured extraction.
185
+
186
+ - LangChain Google GenAI:
187
+ - `KG_DOC_PARSER_PROVIDER=gemini`
188
+ - `KG_DOC_PARSER_MODEL=gemini-2.5-flash`
189
+ - ChatGPT / OpenAI REST:
190
+ - `KG_DOC_PARSER_PROVIDER=openai`
191
+ - `KG_DOC_PARSER_MODEL=gpt-4.1-mini`
192
+ - `KG_DOC_PARSER_API_KEY_ENV=OPENAI_API_KEY`
193
+ - LangChain Ollama:
194
+ - `KG_DOC_PARSER_PROVIDER=ollama`
195
+ - `KG_DOC_PARSER_MODEL=llama3.1`
196
+ - `KG_DOC_PARSER_BASE_URL=http://127.0.0.1:11434`
197
+ - LangChain Vertex AI:
198
+ - `KG_DOC_PARSER_PROVIDER=vertex`
199
+ - `KG_DOC_PARSER_MODEL=gemini-2.5-pro`
200
+ - `KG_DOC_PARSER_PROJECT=my-project`
201
+ - `KG_DOC_PARSER_LOCATION=us-central1`
202
+
203
+ ### Recipe Parsing Example
204
+
205
+ If you are parsing a cooking recipe, one practical split is:
206
+
207
+ - OCR on Gemini or another vision model
208
+ - parser on OpenAI, Ollama, or Vertex AI
209
+
210
+ For example:
211
+
212
+ ```powershell
213
+ KG_DOC_OCR_PROVIDER=gemini
214
+ KG_DOC_OCR_MODEL=gemini-2.5-flash
215
+ KG_DOC_PARSER_PROVIDER=openai
216
+ KG_DOC_PARSER_MODEL=gpt-4.1-mini
217
+ KG_DOC_PARSER_API_KEY_ENV=OPENAI_API_KEY
218
+ KG_DOC_EMBED_PROVIDER=fake
219
+ ```
220
+
221
+ That setup can extract a recipe into structured graph data such as:
222
+ - ingredients
223
+ - steps
224
+ - tools
225
+ - timers
226
+ - inferred sections like `prep`, `cook`, and `serve`
227
+
228
+ ### Embedding Examples
229
+
230
+ - Fake deterministic CI embedding:
231
+ - `KG_DOC_EMBED_PROVIDER=fake`
232
+ - OpenAI embeddings:
233
+ - `KG_DOC_EMBED_PROVIDER=openai`
234
+ - `KG_DOC_EMBED_MODEL=text-embedding-3-small`
235
+ - Vertex AI embeddings:
236
+ - `KG_DOC_EMBED_PROVIDER=vertex`
237
+ - `KG_DOC_EMBED_MODEL=text-embedding-004`
238
+ - Ollama embeddings:
239
+ - `KG_DOC_EMBED_PROVIDER=ollama`
240
+ - `KG_DOC_EMBED_MODEL=nomic-embed-text`
241
+
242
+ Note:
243
+
244
+ - `embedding_space` in the workflow ingest models is currently a metadata and
245
+ routing-intent label.
246
+ - It does not yet imply that the engine is using a separate embedder per space.
247
+ - The current engine bootstrap still wires one embedding function per engine
248
+ instance, while the multi-space routing proposal remains a future Kogwistar
249
+ core concern.
250
+
251
+ ## Running Tests
252
+
253
+ Some tests are integration-style and expect local document folders and API credentials to exist. That means not every test is portable in a clean checkout.
254
+
255
+ To run the test suite:
256
+
257
+ ```powershell
258
+ pytest -q
259
+ ```
260
+
261
+ If you only want to work on isolated units, review the test files first and run a narrower subset.
262
+
263
+ ## Demo Harness
264
+
265
+ There is now a manual workflow-ingest demo harness that can run the end-to-end flow against:
266
+
267
+ - an in-process isolated FastAPI server
268
+ - a subprocess-hosted local server
269
+ - an already running external Kogwistar server
270
+
271
+ The legacy semantic-smoke test in
272
+ [`tests/test_semantic_layerwise_doc_parsing.py`](/c:/Users/chanh/Documents/kg_doc_parser/tests/test_semantic_layerwise_doc_parsing.py)
273
+ also expects a live Kogwistar server at `http://127.0.0.1:28110`. It does not
274
+ start that server for you, so use the VS Code server launch config or start it
275
+ manually before running the Ollama case.
276
+
277
+ Example:
278
+
279
+ ```powershell
280
+ .venv\Scripts\python.exe scripts\run_workflow_ingest_demo.py --output-dir logs\workflow_ingest_demo
281
+ ```
282
+
283
+ External live server example:
284
+
285
+ ```powershell
286
+ .venv\Scripts\python.exe scripts\run_workflow_ingest_demo.py --server-mode external_http --external-base-url http://127.0.0.1:28110
287
+ ```
288
+
289
+ Demo artifacts are written into the chosen output directory:
290
+
291
+ - `probe-events.jsonl`: demo-friendly step and lifecycle probe events
292
+ - `demo-summary.json`: run summary, persistence result, and artifact pointers
293
+ - `llm-cache/`: workflow-native cached proposal/review call results
294
+ - `engines/`: local workflow and conversation graph storage for the run
295
+ - `server-data/`: isolated server-side persistence directory when the harness boots its own server
296
+
297
+ Notes:
298
+
299
+ - The workflow-native layer proposal/review path uses deterministic file-backed caching to reduce repeated token cost and compute time.
300
+ - The legacy parser path also supports a redirected `joblib` cache via `KG_DOC_PARSER_JOBLIB_CACHE_DIR`.
301
+ - Probe logging is separate from CDC and conversation graph traces, so demos can show a short readable event trail without digging into runtime internals.
302
+
303
+ ## OCR And Parsing Workflows
304
+
305
+ The newer workflow-first paths are designed as reusable subworkflows:
306
+
307
+ - OCR image/PDF ingest
308
+ - resumable via `ocr-state.sqlite`
309
+ - emits `workflow-events.jsonl`
310
+ - keeps legacy OCR page artifacts on disk
311
+ - page-index parsing
312
+ - heuristic mode for deterministic structure extraction
313
+ - Ollama mode for local parser-backed parsing
314
+ - recursive layerwise parsing
315
+ - wraps the legacy recursive parser in a reusable workflow runner
316
+
317
+ These can be invoked from Python directly or through the `workflow-ingest`
318
+ CLI family, depending on whether you want reusable orchestration or a quick
319
+ shell command.
320
+
321
+ ## Notes
322
+
323
+ - `README.md`, env handling, and ingestion boundaries are still being cleaned up as part of the ongoing refactor.
324
+ - Runtime outputs such as `logs/`, local `.env`, caches, and generated artifacts should remain uncommitted.
325
+ - If behavior diverges between this repo and `kogwistar`, prefer the direction of the ongoing migration and refactor work.
326
+
@@ -0,0 +1,38 @@
1
+ kg_doc_parser/__init__.py,sha256=5XWC63GZ0t54MAYLEejv4ng3OXeTCP48mdE5QsCHzPc,293
2
+ kg_doc_parser/cast_hinting.py,sha256=z0YAcFGIh8eQ7GdMU1XN_Mt2J8E0IAqqBvWuUQgMvmY,552
3
+ kg_doc_parser/document_ingester_logger.py,sha256=D0OotXpkjLkIkszhfBY18Peu9ZKkvZHD8o2o6wVrX40,28589
4
+ kg_doc_parser/models.py,sha256=ev9MkpIEW1uWgEEoICldTi1yNtRMGUjT6aEjutbfpCE,16118
5
+ kg_doc_parser/ocr.py,sha256=8dcxufwnw20veA3ejLnuC5mYsYgV9WYfQ0q1lgl-Klg,48823
6
+ kg_doc_parser/pdf2png.py,sha256=6y_YL2kYF22yJvuO1inj4mCJqqI3nIdQ_IvMpxESoFE,11005
7
+ kg_doc_parser/semantic_document_splitting_layerwise_edits.py,sha256=tyBHJM_1irXBX0hIUBKneo9tZL1314cPpTlQFRXgtdc,144547
8
+ kg_doc_parser/text_processing_utils.py,sha256=mBIjzfKms4ipl4waL710A4bbpJO2_nztl4gyY6GB1Fw,1075
9
+ kg_doc_parser/utils/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
10
+ kg_doc_parser/utils/bounded_threadpool_executor.py,sha256=jBTlKiVFoyGzL9cryCZTkE1mDtPrVqzHP6022DYOcjY,1294
11
+ kg_doc_parser/utils/file_loaders.py,sha256=lhcsBD3lKLoqUckjG3nDTFrQULx_PzYY6RfKrEm3Xx8,18535
12
+ kg_doc_parser/utils/langchain.py,sha256=RfBJdqp6giDEV19UvYMgr00B3wsoWUyJh-vKVcQ7oww,10950
13
+ kg_doc_parser/utils/log.py,sha256=M1tBVet2wsJZtpzB2O6pZLoN5eLdifyIaWdcz_n2tSE,4441
14
+ kg_doc_parser/utils/version_chaining.py,sha256=4_-uqulqLHw_YrGI6YeVPktXbaqH5Y3x6pZpexRYE7Y,55900
15
+ kg_doc_parser/workflow_ingest/__init__.py,sha256=35jQ_qqYC5YE-E7WJHMBW3_i3-V6vnQt7WRmbItveFE,5439
16
+ kg_doc_parser/workflow_ingest/_kogwistar.py,sha256=NYCsqrY3CnNCCWVdoLhjvH0nznF4Z5PrsY3Q8xNH6IU,467
17
+ kg_doc_parser/workflow_ingest/adapters.py,sha256=VdIPQR-NHncSDLeeFdDkT2Tbm3kBoJdBCBT6eHQcoBo,8324
18
+ kg_doc_parser/workflow_ingest/cache.py,sha256=ZPVKLzUTYsM4RrvqV_2l_n_bH_zUQ_zKIuxXqZtozng,2078
19
+ kg_doc_parser/workflow_ingest/cli.py,sha256=PA0zGVRKrQh3m5NFmS-vMSY9bkIWZM6rk1MXwF1xFt8,12563
20
+ kg_doc_parser/workflow_ingest/clients.py,sha256=XstEK0CalJsg1uolIlI6aQeRaux1N0bGOk8WF0yKXVc,16666
21
+ kg_doc_parser/workflow_ingest/demo_harness.py,sha256=v9UvTa5jx7Dc0PygYqfz5eV1buRS9gCWQfYGYvBd7SE,15931
22
+ kg_doc_parser/workflow_ingest/design.py,sha256=7_87zrCL77orU7z4l4yl5EsygwcS0d-fg8eDjyMByuY,7714
23
+ kg_doc_parser/workflow_ingest/handlers.py,sha256=hUKBPdUz6q0MHvp4_t7wD8BwIn9I-J7J91GSa-AAl90,29825
24
+ kg_doc_parser/workflow_ingest/models.py,sha256=gjXp0MqwmabEcDxIr9-OJSW_0pC-_-ZwgQBB_Wr9xBg,24861
25
+ kg_doc_parser/workflow_ingest/ocr_pipeline.py,sha256=bab1UislTafQMgVqLp3bS-gT2OQmSIjlfL0wixu252c,58539
26
+ kg_doc_parser/workflow_ingest/page_index.py,sha256=N5-Y7-wbGZiV1i6cCz7_JbaB40CGI7_63rOfhEp5Z-s,18286
27
+ kg_doc_parser/workflow_ingest/parser_core.py,sha256=S9jZj2LcsIppBangYBQzvGQfYtIeSFGPqGjmM0yulVI,35628
28
+ kg_doc_parser/workflow_ingest/parsing.py,sha256=6RZqb1gnOmmNZO_VYCfeURNhNx1Qq5682Wbk-PvqQhg,7840
29
+ kg_doc_parser/workflow_ingest/probe.py,sha256=M6jkSAH4n5ckpddHnpVJgWh3z1RAGMYjgGq7jZLbht4,6404
30
+ kg_doc_parser/workflow_ingest/providers.py,sha256=Yi3xRnTPOj5nkkWZld19ZNlZdbjemcF5pFJG-jUYyqk,15552
31
+ kg_doc_parser/workflow_ingest/runners.py,sha256=MOdz5idY2A_Dyt_yW3We45IUnL5Co5T34pnH9PeAvFo,19971
32
+ kg_doc_parser/workflow_ingest/semantics.py,sha256=qvbsCyebEr21iDGaGjEi9IrYDzvQdVAhVGvB1DLYQus,8291
33
+ kg_doc_parser/workflow_ingest/service.py,sha256=_J-xd0IrESeBxNWJy7VcKIjPRxlXvjX00BUB01FYdZQ,3664
34
+ kg_doc_parser/workflow_ingest/smoke_assets.py,sha256=ucppMTXug9aHTUXkfEDtVZcp3gO1Z4n4G0I7sFLiaK0,2040
35
+ graph_knowledge_doc_parser-0.1.0.dist-info/entry_points.txt,sha256=AXIuydpwsOveH3S3HRlyJ8gphiKOznJc1c_f2rNZNV0,74
36
+ graph_knowledge_doc_parser-0.1.0.dist-info/METADATA,sha256=OufL-Ima5N6Jl3VYW5alJZaAFTQkMSMV9bdF_6-P1eA,11973
37
+ graph_knowledge_doc_parser-0.1.0.dist-info/WHEEL,sha256=zp0Cn7JsFoX2ATtOhtaFYIiE2rmFAD4OcMhtUki8W3U,88
38
+ graph_knowledge_doc_parser-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: poetry-core 2.2.1
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,3 @@
1
+ [console_scripts]
2
+ workflow-ingest=kg_doc_parser.workflow_ingest.cli:main
3
+
@@ -0,0 +1,9 @@
1
+ from __future__ import annotations
2
+
3
+ """Public import surface for the document parser package."""
4
+
5
+ from . import workflow_ingest
6
+ from .workflow_ingest import * # noqa: F401,F403
7
+ from .workflow_ingest import __all__ as _workflow_ingest_all
8
+
9
+ __all__ = ["workflow_ingest", *_workflow_ingest_all]
@@ -0,0 +1,19 @@
1
+ from __future__ import annotations
2
+ from typing import Callable, TypeVar, ParamSpec, cast
3
+ from joblib import Memory
4
+
5
+ P = ParamSpec("P")
6
+ R = TypeVar("R")
7
+
8
+ def cached(memory: Memory, fn: Callable[P, R]) -> Callable[P, R]:
9
+ return cast(Callable[P, R], memory.cache(fn))
10
+
11
+ memory = Memory(".cache", verbose=0)
12
+
13
+ def compute(a: int, b: float, *, mode: str = "x") -> tuple[int, float]:
14
+ return (a, b)
15
+
16
+ compute_cached = cached(memory, compute)
17
+
18
+ # Type checker should now know:
19
+ # (a: int, b: float, *, mode: str = ...) -> tuple[int, float]