docqwise 0.3.1__tar.gz → 0.4.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. docqwise-0.4.1/PKG-INFO +623 -0
  2. docqwise-0.4.1/README.md +535 -0
  3. docqwise-0.4.1/docqwise/_version.py +1 -0
  4. docqwise-0.4.1/docqwise/agent/__init__.py +4 -0
  5. docqwise-0.4.1/docqwise/agent/extraction_agent.py +527 -0
  6. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/engine.py +117 -89
  7. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/extractors/field_extractor.py +1 -1
  8. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/extractors/llm_extractor.py +4 -4
  9. docqwise-0.4.1/docqwise/extractors/multimodal_extractor.py +458 -0
  10. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/extractors/rag_extractor.py +7 -10
  11. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/factory.py +12 -7
  12. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/llm/hf_llm.py +5 -4
  13. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/llm/ollama.py +12 -16
  14. docqwise-0.4.1/docqwise/llm/openai_llm.py +52 -0
  15. docqwise-0.4.1/docqwise/retrieval/graphrag.py +424 -0
  16. docqwise-0.4.1/docqwise/retrieval/qa_engine.py +103 -0
  17. docqwise-0.4.1/docqwise/retrieval/rag_strategy.py +172 -0
  18. docqwise-0.4.1/docqwise/server/__main__.py +5 -0
  19. docqwise-0.4.1/docqwise/server/mcp_server.py +569 -0
  20. docqwise-0.4.1/docqwise/stores/faiss_store.py +229 -0
  21. docqwise-0.4.1/docqwise.egg-info/PKG-INFO +623 -0
  22. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise.egg-info/SOURCES.txt +2 -0
  23. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise.egg-info/entry_points.txt +1 -0
  24. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise.egg-info/requires.txt +6 -7
  25. docqwise-0.4.1/docqwise.egg-info/top_level.txt +2 -0
  26. {docqwise-0.3.1 → docqwise-0.4.1}/pyproject.toml +9 -8
  27. docqwise-0.3.1/PKG-INFO +0 -443
  28. docqwise-0.3.1/README.md +0 -355
  29. docqwise-0.3.1/docqwise/_version.py +0 -1
  30. docqwise-0.3.1/docqwise/extractors/multimodal_extractor.py +0 -589
  31. docqwise-0.3.1/docqwise/llm/openai_llm.py +0 -51
  32. docqwise-0.3.1/docqwise/retrieval/graphrag.py +0 -430
  33. docqwise-0.3.1/docqwise/retrieval/qa_engine.py +0 -265
  34. docqwise-0.3.1/docqwise/retrieval/rag_strategy.py +0 -104
  35. docqwise-0.3.1/docqwise/server/mcp_server.py +0 -26
  36. docqwise-0.3.1/docqwise/stores/faiss_store.py +0 -187
  37. docqwise-0.3.1/docqwise/templates/__init__.py +0 -0
  38. docqwise-0.3.1/docqwise.egg-info/PKG-INFO +0 -443
  39. docqwise-0.3.1/docqwise.egg-info/top_level.txt +0 -1
  40. {docqwise-0.3.1 → docqwise-0.4.1}/LICENSE +0 -0
  41. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/__init__.py +0 -0
  42. {docqwise-0.3.1/docqwise/agent → docqwise-0.4.1/docqwise/benchmark}/__init__.py +0 -0
  43. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/benchmark/runner.py +0 -0
  44. {docqwise-0.3.1/docqwise/benchmark → docqwise-0.4.1/docqwise/bridges}/__init__.py +0 -0
  45. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/bridges/adaptive_bridge.py +0 -0
  46. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/bridges/evalkit_bridge.py +0 -0
  47. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/bridges/sightrag_bridge.py +0 -0
  48. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/bridges/sonarwise_bridge.py +0 -0
  49. {docqwise-0.3.1/docqwise/bridges → docqwise-0.4.1/docqwise/chunkers}/__init__.py +0 -0
  50. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/chunkers/base.py +0 -0
  51. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/chunkers/fixed_chunker.py +0 -0
  52. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/chunkers/sentence_chunker.py +0 -0
  53. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/chunkers/structure_chunker.py +0 -0
  54. {docqwise-0.3.1/docqwise/chunkers → docqwise-0.4.1/docqwise/classifier}/__init__.py +0 -0
  55. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/classifier/zero_shot.py +0 -0
  56. {docqwise-0.3.1/docqwise/classifier → docqwise-0.4.1/docqwise/cli}/__init__.py +0 -0
  57. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/cli/main.py +0 -0
  58. {docqwise-0.3.1/docqwise/cli → docqwise-0.4.1/docqwise/comparator}/__init__.py +0 -0
  59. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/comparator/text_diff.py +0 -0
  60. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/config.py +0 -0
  61. {docqwise-0.3.1/docqwise/comparator → docqwise-0.4.1/docqwise/connectors}/__init__.py +0 -0
  62. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/connectors/base.py +0 -0
  63. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/connectors/filesystem.py +0 -0
  64. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/core/__init__.py +0 -0
  65. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/core/bases.py +0 -0
  66. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/core/chunk.py +0 -0
  67. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/core/document.py +0 -0
  68. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/core/element.py +0 -0
  69. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/core/field.py +0 -0
  70. {docqwise-0.3.1/docqwise/connectors → docqwise-0.4.1/docqwise/databases}/__init__.py +0 -0
  71. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/databases/base.py +0 -0
  72. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/databases/sqlite_db.py +0 -0
  73. {docqwise-0.3.1/docqwise/databases → docqwise-0.4.1/docqwise/embedders}/__init__.py +0 -0
  74. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/embedders/base.py +0 -0
  75. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/embedders/sentence_transformer.py +0 -0
  76. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/exceptions.py +0 -0
  77. {docqwise-0.3.1/docqwise/embedders → docqwise-0.4.1/docqwise/export}/__init__.py +0 -0
  78. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/export/csv_exporter.py +0 -0
  79. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/export/json_exporter.py +0 -0
  80. {docqwise-0.3.1/docqwise/export → docqwise-0.4.1/docqwise/extractors}/__init__.py +0 -0
  81. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/extractors/base.py +0 -0
  82. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/extractors/entity_extractor.py +0 -0
  83. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/extractors/image_extractor.py +0 -0
  84. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/extractors/metadata_extractor.py +0 -0
  85. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/extractors/table_extractor.py +0 -0
  86. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/extractors/text_extractor.py +0 -0
  87. {docqwise-0.3.1/docqwise/extractors → docqwise-0.4.1/docqwise/graph}/__init__.py +0 -0
  88. {docqwise-0.3.1/docqwise/graph → docqwise-0.4.1/docqwise/graph/backends}/__init__.py +0 -0
  89. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/graph/document_graph.py +0 -0
  90. {docqwise-0.3.1/docqwise/graph/backends → docqwise-0.4.1/docqwise/incremental}/__init__.py +0 -0
  91. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/incremental/change_detector.py +0 -0
  92. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/incremental/watcher.py +0 -0
  93. {docqwise-0.3.1/docqwise/incremental → docqwise-0.4.1/docqwise/layout}/__init__.py +0 -0
  94. {docqwise-0.3.1/docqwise/layout → docqwise-0.4.1/docqwise/learning}/__init__.py +0 -0
  95. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/learning/correction.py +0 -0
  96. {docqwise-0.3.1/docqwise/learning → docqwise-0.4.1/docqwise/llm}/__init__.py +0 -0
  97. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/llm/base.py +0 -0
  98. {docqwise-0.3.1/docqwise/llm → docqwise-0.4.1/docqwise/ocr}/__init__.py +0 -0
  99. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/ocr/base.py +0 -0
  100. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/ocr/easyocr_engine.py +0 -0
  101. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/ocr/tesseract_engine.py +0 -0
  102. {docqwise-0.3.1/docqwise/ocr → docqwise-0.4.1/docqwise/pipeline}/__init__.py +0 -0
  103. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/pipeline/pipeline.py +0 -0
  104. {docqwise-0.3.1/docqwise/pipeline → docqwise-0.4.1/docqwise/pipeline/prebuilt}/__init__.py +0 -0
  105. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/readers/__init__.py +0 -0
  106. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/readers/auto_reader.py +0 -0
  107. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/readers/base.py +0 -0
  108. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/readers/csv_reader.py +0 -0
  109. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/readers/docx_reader.py +0 -0
  110. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/readers/email_reader.py +0 -0
  111. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/readers/excel_reader.py +0 -0
  112. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/readers/html_reader.py +0 -0
  113. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/readers/image_reader.py +0 -0
  114. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/readers/json_reader.py +0 -0
  115. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/readers/parquet_reader.py +0 -0
  116. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/readers/pdf_reader.py +0 -0
  117. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/readers/pptx_reader.py +0 -0
  118. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/readers/text_reader.py +0 -0
  119. {docqwise-0.3.1/docqwise/pipeline/prebuilt → docqwise-0.4.1/docqwise/retrieval}/__init__.py +0 -0
  120. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/retrieval/hybrid_search.py +0 -0
  121. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/retrieval/query_router.py +0 -0
  122. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/retrieval/retriever.py +0 -0
  123. {docqwise-0.3.1/docqwise/retrieval → docqwise-0.4.1/docqwise/schema}/__init__.py +0 -0
  124. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/schema/detector.py +0 -0
  125. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/schema/quality.py +0 -0
  126. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/schema/validator.py +0 -0
  127. {docqwise-0.3.1/docqwise/schema → docqwise-0.4.1/docqwise/security}/__init__.py +0 -0
  128. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/security/pii_detector.py +0 -0
  129. {docqwise-0.3.1/docqwise/security → docqwise-0.4.1/docqwise/server}/__init__.py +0 -0
  130. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/server/api.py +0 -0
  131. {docqwise-0.3.1/docqwise/server → docqwise-0.4.1/docqwise/server/routes}/__init__.py +0 -0
  132. {docqwise-0.3.1/docqwise/server/routes → docqwise-0.4.1/docqwise/speed}/__init__.py +0 -0
  133. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/speed/dispatcher.py +0 -0
  134. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/speed/progress.py +0 -0
  135. {docqwise-0.3.1/docqwise/speed → docqwise-0.4.1/docqwise/stores}/__init__.py +0 -0
  136. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/stores/base.py +0 -0
  137. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/stores/sqlite_store.py +0 -0
  138. {docqwise-0.3.1/docqwise/stores → docqwise-0.4.1/docqwise/strategy}/__init__.py +0 -0
  139. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/strategy/profiler.py +0 -0
  140. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/strategy/router.py +0 -0
  141. {docqwise-0.3.1/docqwise/strategy → docqwise-0.4.1/docqwise/templates}/__init__.py +0 -0
  142. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/templates/contract.py +0 -0
  143. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/templates/invoice.py +0 -0
  144. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/templates/receipt.py +0 -0
  145. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise/templates/resume.py +0 -0
  146. {docqwise-0.3.1 → docqwise-0.4.1}/docqwise.egg-info/dependency_links.txt +0 -0
  147. {docqwise-0.3.1 → docqwise-0.4.1}/setup.cfg +0 -0
@@ -0,0 +1,623 @@
1
+ Metadata-Version: 2.4
2
+ Name: docqwise
3
+ Version: 0.4.1
4
+ Summary: Read, Extract, Retrieve: Document intelligence that adapts, accelerates, and scales.
5
+ Author: Venkatkumar Rajan
6
+ License: Apache-2.0
7
+ Project-URL: Homepage, https://github.com/VK-Ant/docqwise
8
+ Project-URL: Documentation, https://github.com/VK-Ant/docqwise#readme
9
+ Project-URL: Repository, https://github.com/VK-Ant/docqwise
10
+ Project-URL: Issues, https://github.com/VK-Ant/docqwise/issues
11
+ Keywords: document-intelligence,ocr,extraction,rag,nlp,pdf,table-extraction,field-extraction,document-ai,structured-data,unstructured-data,graphrag,mcp
12
+ Classifier: Development Status :: 4 - Beta
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Intended Audience :: Science/Research
15
+ Classifier: License :: OSI Approved :: Apache Software License
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.9
18
+ Classifier: Programming Language :: Python :: 3.10
19
+ Classifier: Programming Language :: Python :: 3.11
20
+ Classifier: Programming Language :: Python :: 3.12
21
+ Classifier: Programming Language :: Python :: 3.13
22
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
23
+ Classifier: Topic :: Text Processing :: General
24
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
25
+ Requires-Python: >=3.9
26
+ Description-Content-Type: text/markdown
27
+ License-File: LICENSE
28
+ Requires-Dist: pymupdf>=1.24.0
29
+ Requires-Dist: python-docx>=1.0.0
30
+ Requires-Dist: numpy>=1.24.0
31
+ Requires-Dist: pillow>=10.4.0
32
+ Requires-Dist: tqdm>=4.65.0
33
+ Requires-Dist: pyyaml>=6.0.1
34
+ Requires-Dist: click>=8.1.0
35
+ Provides-Extra: ml
36
+ Requires-Dist: easyocr>=1.7.0; extra == "ml"
37
+ Requires-Dist: sentence-transformers>=2.7.0; extra == "ml"
38
+ Requires-Dist: transformers>=4.40.0; extra == "ml"
39
+ Requires-Dist: torch>=2.0.0; extra == "ml"
40
+ Requires-Dist: spacy>=3.7.0; extra == "ml"
41
+ Provides-Extra: full
42
+ Requires-Dist: docqwise[ml]; extra == "full"
43
+ Requires-Dist: pandas>=2.0.0; extra == "full"
44
+ Requires-Dist: openpyxl>=3.1.0; extra == "full"
45
+ Requires-Dist: camelot-py>=0.11.0; extra == "full"
46
+ Requires-Dist: pyarrow>=14.0.0; extra == "full"
47
+ Requires-Dist: rank-bm25>=0.2.2; extra == "full"
48
+ Requires-Dist: pydantic>=2.0.0; extra == "full"
49
+ Requires-Dist: beautifulsoup4>=4.12.0; extra == "full"
50
+ Requires-Dist: python-pptx>=0.6.21; extra == "full"
51
+ Provides-Extra: api
52
+ Requires-Dist: anthropic>=0.20.0; extra == "api"
53
+ Requires-Dist: google-cloud-vision>=3.5.0; extra == "api"
54
+ Requires-Dist: boto3>=1.28.0; extra == "api"
55
+ Provides-Extra: connectors
56
+ Requires-Dist: chromadb>=0.4.0; extra == "connectors"
57
+ Requires-Dist: qdrant-client>=1.7.0; extra == "connectors"
58
+ Requires-Dist: pymilvus>=2.3.0; extra == "connectors"
59
+ Requires-Dist: faiss-cpu>=1.7.0; extra == "connectors"
60
+ Requires-Dist: psycopg2-binary>=2.9.0; extra == "connectors"
61
+ Requires-Dist: pymongo>=4.6.0; extra == "connectors"
62
+ Requires-Dist: redis>=5.0.0; extra == "connectors"
63
+ Requires-Dist: elasticsearch>=8.0.0; extra == "connectors"
64
+ Requires-Dist: lancedb>=0.4.0; extra == "connectors"
65
+ Provides-Extra: graph
66
+ Requires-Dist: networkx>=3.2.0; extra == "graph"
67
+ Requires-Dist: neo4j>=5.0.0; extra == "graph"
68
+ Provides-Extra: server
69
+ Requires-Dist: fastapi>=0.104.0; extra == "server"
70
+ Requires-Dist: uvicorn>=0.24.0; extra == "server"
71
+ Requires-Dist: websockets>=12.0; extra == "server"
72
+ Requires-Dist: python-multipart>=0.0.6; extra == "server"
73
+ Provides-Extra: all
74
+ Requires-Dist: docqwise[full]; extra == "all"
75
+ Requires-Dist: docqwise[api]; extra == "all"
76
+ Requires-Dist: docqwise[connectors]; extra == "all"
77
+ Requires-Dist: docqwise[graph]; extra == "all"
78
+ Requires-Dist: docqwise[server]; extra == "all"
79
+ Provides-Extra: dev
80
+ Requires-Dist: pytest>=7.4.0; extra == "dev"
81
+ Requires-Dist: pytest-asyncio>=0.21.0; extra == "dev"
82
+ Requires-Dist: pytest-cov>=4.1.0; extra == "dev"
83
+ Requires-Dist: black>=23.0.0; extra == "dev"
84
+ Requires-Dist: ruff>=0.1.0; extra == "dev"
85
+ Requires-Dist: mypy>=1.7.0; extra == "dev"
86
+ Requires-Dist: pre-commit>=3.5.0; extra == "dev"
87
+ Dynamic: license-file
88
+
89
+ <p align="center">
90
+ <img src="https://raw.githubusercontent.com/VK-Ant/docqwise/main/assets/hero.png" alt="DocQWise: Read, Extract, Retrieve" width="100%">
91
+ </p>
92
+
93
+ <p align="center">
94
+ <strong>Document intelligence that adapts, accelerates, and scales.</strong>
95
+ </p>
96
+
97
+ <p align="center">
98
+ <a href="https://pypi.org/project/docqwise/"><img src="https://img.shields.io/pypi/v/docqwise?color=blue" alt="PyPI"></a>
99
+ <a href="https://pypi.org/project/docqwise/"><img src="https://img.shields.io/pypi/pyversions/docqwise" alt="Python"></a>
100
+ <a href="https://github.com/VK-Ant/docqwise/blob/main/LICENSE"><img src="https://img.shields.io/badge/License-Apache%202.0-blue.svg" alt="License"></a>
101
+ <a href="https://github.com/VK-Ant/docqwise/blob/main/notebooks/docqwise_getting_started.ipynb">
102
+ <img src="https://colab.research.google.com/assets/colab-badge.svg" alt="Open In Colab"></a>
103
+ </p>
104
+
105
+ ---
106
+
107
+ ## What is DocQWise?
108
+
109
+ DocQWise is a pluggable, AI-powered document intelligence engine. It reads any document format, extracts structured data using LLMs and RAG pipelines, and retrieves information with semantic search — locally, at scale, for zero per-page cost.
110
+
111
+ ### What's New in v0.4.0
112
+
113
+ - **Agentic Extraction** — Multi-pass, self-correcting extraction agent with validate → retry → cross-check loop
114
+ - **MCP Server** — Model Context Protocol server for AI agent integration (Claude, GPT, etc.)
115
+ - **GraphRAG** — Entity graph traversal with evidence chains and knowledge graph visualization
116
+ - **FAISS Vector Store** — Production-scale similarity search alongside SQLite
117
+ - **Source Attribution** — Every answer includes source, confidence, method, and evidence
118
+ - **Vision Model Routing** — Auto-routes to Ollama vision, HuggingFace, OpenAI, or local models (.gguf, .onnx, .pt)
119
+ - **Custom Prompts** — Full control with `prompt`, `prompt_template`, and `system_prompt`
120
+ - **Python 3.9–3.13 Compatible** — Zero `requests` dependency (uses urllib), no charset_normalizer crash
121
+
122
+ ---
123
+
124
+ ## Install
125
+
126
+ ```bash
127
+ # Core (PDF reading, AI extraction)
128
+ pip install docqwise
129
+
130
+ # With ML (RAG pipeline, embeddings, OCR — recommended)
131
+ pip install docqwise[ml]
132
+
133
+ # Everything
134
+ pip install docqwise[all]
135
+
136
+ # Specific extras
137
+ pip install docqwise[graph] # GraphRAG + NetworkX visualization
138
+ pip install docqwise[connectors] # FAISS, ChromaDB, Qdrant, etc.
139
+ pip install docqwise[server] # MCP server + FastAPI
140
+ ```
141
+
142
+ ### LLM Backend (pick one)
143
+
144
+ ```bash
145
+ # Ollama — local, free, recommended
146
+ # Download from https://ollama.com then:
147
+ ollama pull nemotron-mini
148
+
149
+ # OR HuggingFace — local GPU
150
+ pip install transformers torch bitsandbytes accelerate
151
+
152
+ # OR OpenAI — cloud API (no SDK needed, docqwise uses urllib)
153
+ export OPENAI_API_KEY=your-key
154
+ ```
155
+
156
+ ---
157
+
158
+ ## Quick Start
159
+
160
+ ```python
161
+ from docqwise import Docqwise
162
+
163
+ dq = Docqwise()
164
+
165
+ # Ingest any document
166
+ dq.ingest("documents/")
167
+
168
+ # Extract fields with template
169
+ result = dq.extract_fields("invoice.pdf", template="invoice")
170
+ print(result.to_json())
171
+
172
+ # Ask questions
173
+ answer = dq.ask("What is the total amount?")
174
+ ```
175
+
176
+ ---
177
+
178
+ ## Agentic Extraction (v0.4.0)
179
+
180
+ Multi-pass, self-correcting extraction that validates, retries failed fields, and cross-checks results:
181
+
182
+ ```python
183
+ from docqwise import Docqwise
184
+
185
+ dq = Docqwise()
186
+
187
+ # Agentic extraction — autonomous multi-pass pipeline
188
+ result = dq.extract_agentic(
189
+ "invoice.pdf",
190
+ template="invoice",
191
+ max_retries=3, # retry failed fields up to 3 times
192
+ cross_check=True, # verify with cross-check
193
+ )
194
+
195
+ print(result.to_json())
196
+ print(f"Confidence: {result.confidence}")
197
+ print(f"Strategy: {result.strategy_used}") # 'agentic'
198
+ print(f"Warnings: {result.warnings}")
199
+
200
+ # Direct agent usage with full trace
201
+ from docqwise.agent import ExtractionAgent
202
+
203
+ agent = ExtractionAgent(max_retries=2, cross_check=True)
204
+ result = agent.extract("contract.pdf", template="contract")
205
+
206
+ # Full audit trail
207
+ print(agent.trace.summary())
208
+ # Agent trace: 4 passes, 2 retries, confidence=0.92
209
+ # ✓ Step 1: extract (350ms) RAG extraction: 8 fields
210
+ # ✓ Step 2: validate (5ms) 2 fields failed validation
211
+ # ✓ Step 3: retry (280ms) Attempt 1: retried 2, fixed 2
212
+ # ✓ Step 4: cross_check (120ms) Cross-checked extraction
213
+ # ✓ Step 5: merge (1ms) Final confidence: 0.92
214
+ ```
215
+
216
+ ### Custom Validation Rules
217
+
218
+ ```python
219
+ from docqwise.agent.extraction_agent import ExtractionAgent, ValidationRule
220
+
221
+ rules = [
222
+ ValidationRule("invoice_number", "required"),
223
+ ValidationRule("total", "range", {"min": 0, "max": 1000000}),
224
+ ValidationRule("email", "pattern", {"pattern": r"[\w.]+@[\w.]+"}),
225
+ ValidationRule("status", "choices", {"values": ["paid", "pending", "overdue"]}),
226
+ ]
227
+
228
+ agent = ExtractionAgent(validation_rules=rules)
229
+ result = agent.extract("invoice.pdf", schema={
230
+ "invoice_number": "string",
231
+ "total": "number",
232
+ "email": "string",
233
+ "status": "string",
234
+ })
235
+ ```
236
+
237
+ ---
238
+
239
+ ## MCP Server (v0.4.0)
240
+
241
+ Use docqwise as a tool server for any AI agent via the [Model Context Protocol](https://modelcontextprotocol.io):
242
+
243
+ ```bash
244
+ # Start MCP server (stdio transport)
245
+ python -m docqwise.server.mcp_server
246
+
247
+ # Or with HTTP transport
248
+ python -m docqwise.server.mcp_server --port 8080
249
+ ```
250
+
251
+ ### Claude Desktop Integration
252
+
253
+ Add to your `claude_desktop_config.json`:
254
+
255
+ ```json
256
+ {
257
+ "mcpServers": {
258
+ "docqwise": {
259
+ "command": "python",
260
+ "args": ["-m", "docqwise.server.mcp_server"]
261
+ }
262
+ }
263
+ }
264
+ ```
265
+
266
+ ### Available MCP Tools
267
+
268
+ | Tool | Description |
269
+ |---|---|
270
+ | `docqwise_extract` | Extract fields from any document |
271
+ | `docqwise_extract_agentic` | Multi-pass self-correcting extraction |
272
+ | `docqwise_ask` | RAG-powered Q&A over documents |
273
+ | `docqwise_ingest` | Ingest documents into vector store |
274
+ | `docqwise_classify` | Zero-shot document classification |
275
+ | `docqwise_extract_tables` | Extract tables as structured data |
276
+ | `docqwise_extract_entities` | Extract named entities |
277
+ | `docqwise_compare` | Compare two documents |
278
+ | `docqwise_detect_pii` | Detect PII in documents |
279
+ | `docqwise_retrieve` | Semantic search across corpus |
280
+
281
+ ---
282
+
283
+ ## Extraction Methods
284
+
285
+ ```python
286
+ dq = Docqwise()
287
+
288
+ # RAG (default) — chunk → embed → retrieve → LLM extract
289
+ dq.extract_fields("doc.pdf", template="invoice")
290
+
291
+ # Agentic — multi-pass, self-correcting (v0.4.0)
292
+ dq.extract_agentic("doc.pdf", template="invoice")
293
+
294
+ # Direct LLM
295
+ dq.extract_fields("doc.pdf", template="invoice", method="llm")
296
+
297
+ # Vision (scanned docs, handwriting)
298
+ dq.extract_fields("scan.jpg", method="vision", model="gpt-4o")
299
+
300
+ # Agentic (multi-pass, self-correcting)
301
+ dq.extract_agentic("doc.pdf", template="invoice")
302
+ ```
303
+
304
+ ---
305
+
306
+ ## GraphRAG (v0.4.0)
307
+
308
+ Entity graph traversal with evidence chains:
309
+
310
+ ```python
311
+ dq = Docqwise()
312
+ dq.ingest("contracts/")
313
+
314
+ # RAG with graph-enhanced retrieval
315
+ result = dq.ask_rag(
316
+ "What is the payment terms for Acme Corp?",
317
+ mode="graphrag",
318
+ sources=["contract_acme.pdf"],
319
+ )
320
+ print(result["answer"])
321
+ print(result["evidence"]) # evidence chain
322
+ print(result["confidence"]) # confidence score
323
+
324
+ # Build and visualize knowledge graph
325
+ engine = dq.build_graph(sources=["contract1.pdf", "contract2.pdf"])
326
+ print(engine.graph.stats()) # {'nodes': 42, 'edges': 87, 'documents': 2}
327
+
328
+ # Visualize
329
+ dq.visualize_graph("graph.png", sources=["contract1.pdf"])
330
+ dq.visualize_graph("graph.html", sources=["contract1.pdf"]) # interactive vis.js
331
+
332
+ # Query-aware visualization (highlights relevant nodes)
333
+ result = dq.visualize_query(
334
+ "Who signed the contract?",
335
+ output="query_graph.png",
336
+ sources=["contract1.pdf"],
337
+ )
338
+ print(result["answer"])
339
+ ```
340
+
341
+ ---
342
+
343
+ ## Source Attribution (v0.4.0)
344
+
345
+ Every answer tracks where it came from:
346
+
347
+ ```python
348
+ dq = Docqwise()
349
+ dq.ingest("invoices/")
350
+
351
+ # Answer with source tracking
352
+ answer = dq.ask_with_source("What is invoice #1234 total?", source="inv_1234.pdf")
353
+ print(answer.answer) # "$15,750.00"
354
+ print(answer.source_name) # "inv_1234.pdf"
355
+ print(answer.confidence) # 0.85
356
+ print(answer.method) # "qa_engine"
357
+
358
+ # Full RAG with source attribution
359
+ result = dq.ask_rag(
360
+ "What is the total amount?",
361
+ mode="general",
362
+ source="invoice.pdf",
363
+ system_prompt="You are an accounting assistant.",
364
+ )
365
+ print(result["answer"])
366
+ print(result["source_name"])
367
+ print(result["confidence"])
368
+ print(result["method"]) # "general", "graphrag", or "multimodal"
369
+ ```
370
+
371
+ ---
372
+
373
+ ## LLM Backends
374
+
375
+ ```python
376
+ # Ollama (local, free)
377
+ dq.extract_fields("doc.pdf", model="nemotron-mini")
378
+
379
+ # HuggingFace (local GPU)
380
+ from docqwise.llm.hf_llm import HuggingFaceLLM
381
+ llm = HuggingFaceLLM("Qwen/Qwen2.5-3B-Instruct", quantize="4bit")
382
+ dq.extract_fields("doc.pdf", llm=llm)
383
+
384
+ # OpenAI (cloud — no SDK needed)
385
+ from docqwise.llm.openai_llm import OpenAILLM
386
+ llm = OpenAILLM(model="gpt-4o-mini")
387
+ dq.extract_fields("doc.pdf", llm=llm)
388
+
389
+ # Azure / vLLM / LM Studio (any OpenAI-compatible API)
390
+ llm = OpenAILLM(model="my-model", base_url="https://my-server.com/v1")
391
+ ```
392
+
393
+ ---
394
+
395
+ ## Vision Models (v0.4.0)
396
+
397
+ Comprehensive vision model routing:
398
+
399
+ ```python
400
+ # Ollama vision models
401
+ dq.extract_fields("scan.jpg", method="vision", model="qwen2-vl")
402
+ dq.extract_fields("scan.jpg", method="vision", model="gemma3")
403
+ dq.extract_fields("scan.jpg", method="vision", model="llava")
404
+
405
+ # OpenAI / Azure vision
406
+ dq.extract_fields("scan.jpg", method="vision", model="gpt-4o")
407
+
408
+ # HuggingFace vision
409
+ dq.extract_fields("scan.jpg", method="vision", model="Qwen/Qwen2-VL-7B-Instruct")
410
+ dq.extract_fields("scan.jpg", method="vision", model="microsoft/Florence-2-large")
411
+
412
+ # Local model files
413
+ dq.extract_fields("scan.jpg", method="vision", model="model.gguf") # llama-cpp
414
+ dq.extract_fields("scan.jpg", method="vision", model="model.onnx") # ONNX Runtime
415
+ dq.extract_fields("scan.jpg", method="vision", model="model.pt") # PyTorch
416
+ ```
417
+
418
+ ---
419
+
420
+ ## Custom Prompts
421
+
422
+ Full control over extraction prompts:
423
+
424
+ ```python
425
+ dq = Docqwise()
426
+
427
+ # Your own prompt
428
+ dq.extract_fields("doc.pdf", prompt="""
429
+ You are a medical record parser.
430
+ Extract patient name, diagnosis, and prescribed medications.
431
+ Return JSON only.
432
+
433
+ Document:
434
+ {context}
435
+
436
+ JSON:
437
+ """)
438
+
439
+ # Prompt template with schema placeholder
440
+ dq.extract_fields("doc.pdf",
441
+ schema={"patient": "string", "diagnosis": "string"},
442
+ prompt_template="""
443
+ Given this schema: {schema}
444
+ Parse this document: {context}
445
+ Return JSON matching the schema exactly.
446
+ """)
447
+
448
+ # System prompt for LLM role
449
+ result = dq.ask_rag(
450
+ "Summarize the contract terms",
451
+ source="contract.pdf",
452
+ system_prompt="You are a legal analyst. Be precise and cite clause numbers.",
453
+ )
454
+ ```
455
+
456
+ ---
457
+
458
+ ## Vector Stores
459
+
460
+ ```python
461
+ # SQLite (default — zero install)
462
+ dq = Docqwise(vector_store="sqlite")
463
+
464
+ # FAISS (production scale — v0.4.0)
465
+ dq = Docqwise(vector_store="faiss")
466
+ # pip install faiss-cpu
467
+ # GPU: pip install faiss-gpu
468
+ ```
469
+
470
+ ---
471
+
472
+ ## Templates
473
+
474
+ ```python
475
+ dq.extract_fields("invoice.pdf", template="invoice")
476
+ dq.extract_fields("contract.pdf", template="contract")
477
+ dq.extract_fields("resume.pdf", template="resume")
478
+ dq.extract_fields("receipt.jpg", template="receipt")
479
+
480
+ # Custom schema
481
+ schema = {
482
+ "vendor": {"type": "string", "description": "Company name"},
483
+ "total": {"type": "number", "description": "Total amount"},
484
+ "date": {"type": "date"},
485
+ }
486
+ dq.extract_fields("doc.pdf", schema=schema)
487
+ ```
488
+
489
+ ---
490
+
491
+ ## Self-Improving Corrections
492
+
493
+ ```python
494
+ result = dq.extract_fields("invoice.pdf", template="invoice")
495
+ result.correct({"tax": 33300.00, "gst_number": "29AABCU9603R1ZM"})
496
+ # Next similar document → corrections applied automatically
497
+ ```
498
+
499
+ ---
500
+
501
+ ## Structured Data Q&A
502
+
503
+ ```python
504
+ dq.ingest("sales.xlsx")
505
+ dq.ask("What is the total amount?") # exact SUM
506
+ dq.ask("Which vendor has highest sales?") # GROUP BY + MAX
507
+ dq.ask("How many invoices are overdue?") # COUNT + WHERE
508
+ ```
509
+
510
+ ---
511
+
512
+ ## All Features
513
+
514
+ ```python
515
+ dq = Docqwise()
516
+
517
+ # Ingestion
518
+ dq.ingest("file.pdf") # single file
519
+ dq.ingest("documents/") # folder (all formats)
520
+ dq.ingest("data.csv") # structured data
521
+
522
+ # Extraction
523
+ dq.extract_fields("doc.pdf") # RAG field extraction
524
+ dq.extract_agentic("doc.pdf") # agentic multi-pass (v0.4.0)
525
+ dq.extract_tables("doc.pdf") # table extraction
526
+ dq.extract_entities("doc.pdf") # entity extraction
527
+ dq.extract_images("doc.pdf") # image extraction
528
+ dq.extract_text("doc.pdf") # text extraction
529
+ dq.auto_extract("doc.pdf") # auto-detect + extract
530
+
531
+ # Intelligence
532
+ dq.retrieve("query", top_k=5) # semantic search
533
+ dq.ask("question") # Q&A
534
+ dq.ask_with_source("question") # Q&A with source attribution
535
+ dq.ask_rag("question", mode="graphrag") # RAG with graph/multimodal
536
+ dq.classify("doc.pdf") # classification
537
+ dq.compare("v1.pdf", "v2.pdf") # comparison
538
+ dq.detect_schema("data.csv") # schema detection
539
+ dq.detect_pii("doc.pdf") # PII detection
540
+
541
+ # Knowledge Graph
542
+ dq.build_graph(sources=["a.pdf", "b.pdf"])
543
+ dq.visualize_graph("graph.png")
544
+ dq.visualize_query("Who signed?", output="query.html")
545
+ ```
546
+
547
+ ---
548
+
549
+ ## Demos
550
+
551
+ | Demo | What | Install |
552
+ |---|---|---|
553
+ | `python demo/01_quickstart.py` | All core features | `pip install docqwise` |
554
+ | `python demo/02_ollama.py` | AI extraction with Ollama | `ollama pull nemotron-mini` |
555
+ | `python demo/03_huggingface.py` | AI extraction on GPU | `pip install transformers torch bitsandbytes accelerate` |
556
+ | `python demo/04_rag.py` | Full RAG pipeline | `pip install sentence-transformers` |
557
+ | `python demo/05_agentic.py` | Agentic multi-pass extraction | `pip install docqwise` |
558
+ | `python demo/06_mcp_server.py` | MCP server for AI agents | `pip install docqwise` |
559
+ | `python demo/07_graphrag.py` | GraphRAG + visualization | `pip install docqwise[graph]` |
560
+
561
+ ---
562
+
563
+ ## Architecture
564
+
565
+ <p align="center">
566
+ <img src="https://raw.githubusercontent.com/VK-Ant/docqwise/main/assets/arc.png" alt="arc" width="100%">
567
+ </p>
568
+
569
+ ```
570
+ engine.py (stable — never changes)
571
+ └── factory.py (all component selection)
572
+ ├── ExtractorFactory → rag | llm | vision | agentic
573
+ ├── LLMFactory → ollama | huggingface | openai
574
+ ├── EmbedderFactory → sentence-transformers | any
575
+ ├── StoreFactory → sqlite | faiss | any
576
+ ├── ChunkerFactory → structure | fixed | sentence
577
+ └── TemplateFactory → invoice | contract | resume | receipt
578
+
579
+ └── agent/
580
+ └── ExtractionAgent → validate → retry → cross-check → merge
581
+
582
+ └── retrieval/
583
+ ├── RAGStrategy → general | graphrag | multimodal
584
+ ├── GraphRAGEngine → entity graph + evidence chains
585
+ └── QAEngine → structured data Q&A
586
+
587
+ └── server/
588
+ └── MCPServer → JSON-RPC over stdio (MCP protocol)
589
+ ```
590
+
591
+ ---
592
+
593
+ ## Testing
594
+
595
+ ```bash
596
+ pip install pytest
597
+ pytest -v
598
+ ```
599
+
600
+ ---
601
+
602
+ ## Docker
603
+
604
+ ```bash
605
+ docker compose up --build
606
+ ```
607
+
608
+ ---
609
+
610
+ ## License
611
+
612
+ Apache License 2.0
613
+
614
+ ## Author
615
+
616
+ **Venkatkumar Rajan**
617
+
618
+ ---
619
+
620
+ ## Ant Intelligence Ecosystem
621
+
622
+ Documentation: https://vk-ant.github.io/ant-intelligence-ecosystem/#home
623
+