py-idp 0.3.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. idp/__init__.py +39 -0
  2. idp/_logging.py +69 -0
  3. idp/_testing_backends.py +37 -0
  4. idp/_util.py +38 -0
  5. idp/api.py +306 -0
  6. idp/assess/__init__.py +15 -0
  7. idp/assess/confidence.py +130 -0
  8. idp/auth/__init__.py +21 -0
  9. idp/auth/keys.py +51 -0
  10. idp/checkpoint.py +251 -0
  11. idp/chunker.py +294 -0
  12. idp/classify/__init__.py +15 -0
  13. idp/classify/classifier.py +133 -0
  14. idp/compat_v01.py +88 -0
  15. idp/config.py +159 -0
  16. idp/core/__init__.py +16 -0
  17. idp/core/document.py +116 -0
  18. idp/core/schemas.py +145 -0
  19. idp/core/types.py +26 -0
  20. idp/discover.py +580 -0
  21. idp/errors.py +96 -0
  22. idp/eval/__init__.py +12 -0
  23. idp/eval/metrics.py +90 -0
  24. idp/eval/runner.py +135 -0
  25. idp/extract/__init__.py +15 -0
  26. idp/extract/extractor.py +499 -0
  27. idp/hitl/__init__.py +12 -0
  28. idp/hitl/app.py +266 -0
  29. idp/llm/__init__.py +39 -0
  30. idp/llm/backend.py +583 -0
  31. idp/llm/china.py +161 -0
  32. idp/llm/nanonets.py +440 -0
  33. idp/llm/nanonets_batch.py +228 -0
  34. idp/metrics.py +153 -0
  35. idp/migrate_audit.py +118 -0
  36. idp/parse/__init__.py +29 -0
  37. idp/parse/parser.py +258 -0
  38. idp/parse/pdf_pages.py +177 -0
  39. idp/parse/router.py +81 -0
  40. idp/pipeline/__init__.py +15 -0
  41. idp/pipeline/cli.py +371 -0
  42. idp/pipeline/pipeline.py +229 -0
  43. idp/policy_config.py +5 -0
  44. idp/queue/__init__.py +15 -0
  45. idp/queue/jobs.py +119 -0
  46. idp/ratelimit.py +73 -0
  47. idp/reliability.py +434 -0
  48. idp/rl/__init__.py +50 -0
  49. idp/rl/calibrate.py +375 -0
  50. idp/rl/online.py +202 -0
  51. idp/rl/policy.py +161 -0
  52. idp/rl/reward.py +144 -0
  53. idp/rl/update.py +124 -0
  54. idp/storage/__init__.py +22 -0
  55. idp/storage/factory.py +62 -0
  56. idp/storage/sql.py +574 -0
  57. idp/storage/store.py +258 -0
  58. idp/validate/__init__.py +25 -0
  59. idp/validate/validator.py +94 -0
  60. py_idp-0.3.1.dist-info/METADATA +811 -0
  61. py_idp-0.3.1.dist-info/RECORD +67 -0
  62. py_idp-0.3.1.dist-info/WHEEL +5 -0
  63. py_idp-0.3.1.dist-info/entry_points.txt +2 -0
  64. py_idp-0.3.1.dist-info/licenses/LICENSE +50 -0
  65. py_idp-0.3.1.dist-info/licenses/LICENSE-AGPL +620 -0
  66. py_idp-0.3.1.dist-info/licenses/LICENSE-COMMERCIAL +95 -0
  67. py_idp-0.3.1.dist-info/top_level.txt +1 -0
idp/__init__.py ADDED
@@ -0,0 +1,39 @@
1
+ # py-idp: general-purpose, AI-enabled Intelligent Document Processing.
2
+ # Copyright (c) 2026 Royce.
3
+ #
4
+ # Licensed under the GNU Affero General Public License v3.0 or later (AGPL-3.0-or-later)
5
+ # with the following addition: a commercial license is also available for organizations
6
+ # that wish to embed py-idp in proprietary products / hosted SaaS without the AGPL
7
+ # copyleft obligations. See LICENSE and LICENSE-COMMERCIAL at the repo root, or
8
+ # contact <royce-license-placeholder@protonmail.com> for terms.
9
+ #
10
+ # This Source Code Form is subject to the terms of the AGPL-3.0-or-later.
11
+ # SPDX-License-Identifier: AGPL-3.0-or-later
12
+
13
+ """py-idp: General-purpose, AI-enabled Intelligent Document Processing framework.
14
+
15
+ A six-stage pipeline: parse -> classify -> extract -> assess -> validate -> HITL.
16
+ Each stage is a pure function over a Document, pluggable, and independently testable.
17
+
18
+ Design draws from:
19
+ - aws-solutions-library-samples/accelerated-intelligent-document-processing-on-aws
20
+ (pipeline shape, HITL, confidence assessment)
21
+ - docling-project/docling (parser: PDF, tables, reading order)
22
+ - run-llama/llama_cloud_services (Pydantic-schema-driven extraction API)
23
+ - Unstructured-IO/unstructured (chunking + multi-format ingest)
24
+ """
25
+
26
+ from idp.core.document import Block, Document, Page
27
+ from idp.discover import DiscoveryResult, discover_schema
28
+ from idp.pipeline.pipeline import Pipeline, PipelineResult
29
+
30
+ __version__ = "0.3.2"
31
+ __all__ = [
32
+ "Block",
33
+ "DiscoveryResult",
34
+ "Document",
35
+ "Page",
36
+ "Pipeline",
37
+ "PipelineResult",
38
+ "discover_schema",
39
+ ]
idp/_logging.py ADDED
@@ -0,0 +1,69 @@
1
+ """Structured logging configuration for py-idp.
2
+
3
+ Provides a single ``get_logger(name)`` helper so every module emits logs
4
+ in the same format (timestamp, level, logger name, message) and honours
5
+ the ``LOG_LEVEL`` environment variable. Production deployments should
6
+ set ``LOG_LEVEL=INFO`` (default) or ``LOG_LEVEL=WARNING`` for quieter
7
+ logs; ``LOG_FORMAT=json`` switches to JSON output for log aggregators.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import logging
12
+ import os
13
+ import sys
14
+
15
+ _LOG_FORMAT = "%(asctime)s %(levelname)s [%(name)s] %(message)s"
16
+ _LOG_DATEFMT = "%Y-%m-%dT%H:%M:%S%z"
17
+ _CONFIGURED = False
18
+
19
+
20
+ def configure(level: str | int | None = None, fmt: str | None = None) -> None:
21
+ """Idempotent root-logger configuration.
22
+
23
+ Reads ``LOG_LEVEL`` and ``LOG_FORMAT`` from env by default. Safe to
24
+ call multiple times — subsequent calls are no-ops.
25
+ """
26
+ global _CONFIGURED
27
+ if _CONFIGURED:
28
+ return
29
+
30
+ if level is None:
31
+ level = os.environ.get("LOG_LEVEL", "INFO").upper()
32
+ if isinstance(level, str):
33
+ level = getattr(logging, level, logging.INFO)
34
+
35
+ if fmt is None:
36
+ fmt = os.environ.get("LOG_FORMAT", _LOG_FORMAT)
37
+ if fmt.lower() == "json":
38
+ fmt = _json_log_format()
39
+
40
+ handler = logging.StreamHandler(sys.stderr)
41
+ handler.setFormatter(logging.Formatter(fmt=fmt, datefmt=_LOG_DATEFMT))
42
+ root = logging.getLogger()
43
+ # Replace existing handlers so format is honoured (don't accumulate)
44
+ root.handlers = [handler]
45
+ root.setLevel(level)
46
+ _CONFIGURED = True
47
+
48
+
49
+ def get_logger(name: str) -> logging.Logger:
50
+ """Return a module-level logger. Configures the root on first call."""
51
+ configure()
52
+ return logging.getLogger(name)
53
+
54
+
55
+ def _json_log_format() -> str:
56
+ """Build a JSON formatter string consumed by python-json-logger downstream.
57
+
58
+ Returns the empty string here so the default Formatter falls back to
59
+ the human-readable format. Operators wanting JSON should install
60
+ ``python-json-logger`` and override via ``LOG_FORMAT_HANDLER``.
61
+ """
62
+ return _LOG_FORMAT
63
+
64
+
65
+ def reset() -> None:
66
+ """Reset configuration state — for tests only."""
67
+ global _CONFIGURED
68
+ _CONFIGURED = False
69
+ logging.getLogger().handlers = []
@@ -0,0 +1,37 @@
1
+ """Testing-only LLM backends.
2
+
3
+ This module is intentionally NOT part of the public ``idp`` namespace.
4
+ Its backends are gated by env vars (``IDP_ENABLE_SLOWMOCK=1``) so
5
+ they cannot be invoked in production by accident.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ import random
10
+ import time
11
+ from typing import Any
12
+
13
+ from idp.llm.backend import CompletionRequest, MockBackend
14
+
15
+
16
+ class SlowMockBackend(MockBackend):
17
+ """A MockBackend that sleeps to simulate real LLM latency.
18
+
19
+ Configured by ``LOAD_LATENCY_MS`` (default 1500) +/- ``LOAD_JITTER_MS``
20
+ (default 500). Only enabled when ``IDP_ENABLE_SLOWMOCK=1``.
21
+ """
22
+
23
+ name = "slowmock"
24
+
25
+ def __init__(self, latency_ms: int = 1500, jitter_ms: int = 500, **kwargs: Any):
26
+ super().__init__(**kwargs)
27
+ self.latency_ms = latency_ms
28
+ self.jitter_ms = jitter_ms
29
+
30
+ def complete(self, req: CompletionRequest) -> str:
31
+ sleep_s = (self.latency_ms + random.uniform(-self.jitter_ms, self.jitter_ms)) / 1000.0
32
+ if sleep_s > 0:
33
+ time.sleep(sleep_s)
34
+ return super().complete(req)
35
+
36
+
37
+ __all__ = ["SlowMockBackend"]
idp/_util.py ADDED
@@ -0,0 +1,38 @@
1
+ # py-idp: general-purpose, AI-enabled Intelligent Document Processing.
2
+ # Copyright (c) 2026 Royce.
3
+ #
4
+ # Licensed under the GNU Affero General Public License v3.0 or later (AGPL-3.0-or-later)
5
+ # with the following addition: a commercial license is also available for organizations
6
+ # that wish to embed py-idp in proprietary products / hosted SaaS without the AGPL
7
+ # copyleft obligations. See LICENSE and LICENSE-COMMERCIAL at the repo root, or
8
+ # contact <royce-license-placeholder@protonmail.com> for terms.
9
+ #
10
+ # This Source Code Form is subject to the terms of the AGPL-3.0-or-later.
11
+ # SPDX-License-Identifier: AGPL-3.0-or-later
12
+
13
+ """Pretty-print a PipelineResult in a terminal-friendly way."""
14
+ from __future__ import annotations
15
+
16
+ import json
17
+
18
+
19
+ def pretty_print_result(result) -> None:
20
+ """Console-friendly rendering of a pipeline result."""
21
+ print(f"\n=== {result.document.source_path} ===")
22
+ print(f"schema: {result.schema_name}")
23
+ print(f"backend: {result.backend_name} ({result.mode})")
24
+ print(f"classify: {result.classification} (conf={result.document.classification_confidence})")
25
+ print(f"validate: {'PASS' if result.validation_passed else 'FAIL'}")
26
+ print("timings: " + ", ".join(f"{t.name}={t.seconds:.3f}s" for t in result.timings))
27
+ print("\nextraction:")
28
+ print(json.dumps(result.document.extraction, indent=2, default=str))
29
+ if result.confidence:
30
+ ordered = sorted(result.confidence.items(), key=lambda kv: kv[1])
31
+ print("\nconfidence (ascending):")
32
+ for k, v in ordered:
33
+ mark = " [REVIEW]" if v < 0.6 else ""
34
+ print(f" {k:<24} {v:.2f}{mark}")
35
+ if result.document.errors:
36
+ print(f"\nerrors ({len(result.document.errors)}):")
37
+ for e in result.document.errors:
38
+ print(f" - {e}")
idp/api.py ADDED
@@ -0,0 +1,306 @@
1
+ """Production-hardened FastAPI server.
2
+
3
+ Improvements over the original ``examples/api.py``:
4
+
5
+ * Reads ``Settings`` from env (port, workers, rate limits, max upload
6
+ size, backend, storage) and **fails fast** on misconfiguration.
7
+ * Health (``/healthz``) and readiness (``/readyz``) endpoints for k8s.
8
+ * Prometheus text exposition at ``/metrics``.
9
+ * Per-key + global rate limiting via ``idp.ratelimit.RateLimiter``.
10
+ * Upload size limit enforced via ``Content-Length`` *and* actual
11
+ stream read (defends against missing/lying Content-Length headers).
12
+ * ``/version`` endpoint exposes the package version for ops dashboards.
13
+ * Structured request logging with timing.
14
+
15
+ Not in scope: HTTPS termination (use a reverse proxy), TLS, SSO,
16
+ multi-tenant auth (these are deployment-level concerns).
17
+ """
18
+ from __future__ import annotations
19
+
20
+ import os
21
+ import time
22
+ from contextlib import asynccontextmanager
23
+ from pathlib import Path
24
+ from typing import Any
25
+
26
+ from fastapi import Depends, FastAPI, Form, HTTPException, Request, UploadFile
27
+ from fastapi.middleware.cors import CORSMiddleware
28
+ from fastapi.responses import JSONResponse, PlainTextResponse, Response
29
+ from pydantic import BaseModel
30
+
31
+ from idp import __version__
32
+ from idp._logging import configure, get_logger
33
+ from idp.config import Settings
34
+ from idp.core.document import Document
35
+ from idp.errors import ConfigurationError, IDPError, RateLimitedError, is_idp_error
36
+ from idp.metrics import metrics
37
+ from idp.pipeline.pipeline import Pipeline
38
+ from idp.ratelimit import RateLimiter
39
+
40
+ # Module-level logger; configured at lifespan startup.
41
+ _log = get_logger("idp.api")
42
+ _settings: Settings | None = None
43
+ _rate_limiter: RateLimiter | None = None
44
+ _upload_dir: Path | None = None
45
+
46
+
47
+ # ---------------------------------------------------------------------------
48
+ # Schemas
49
+ # ---------------------------------------------------------------------------
50
+ class ExtractRequest(BaseModel):
51
+ schema_name: str | None = None
52
+ backend: str | None = None
53
+
54
+
55
+ class ExtractResponse(BaseModel):
56
+ schema_name: str | None = None
57
+ backend_name: str
58
+ mode: str
59
+ classification: str | None
60
+ extraction: dict[str, Any]
61
+ confidence: dict[str, float] | None
62
+ validation: dict[str, Any] | None
63
+
64
+
65
+ class ReviewRequest(BaseModel):
66
+ edited: dict[str, Any]
67
+ reviewer: str
68
+
69
+
70
+ # ---------------------------------------------------------------------------
71
+ # Auth
72
+ # ---------------------------------------------------------------------------
73
+ def auth(request: Request) -> None:
74
+ """Validate the ``X-API-Key`` header against ``Settings.api_key``."""
75
+ s = _require_settings()
76
+ if not s.api_key_required:
77
+ return
78
+ if s.api_key is None:
79
+ # No key configured but key required -> fail closed.
80
+ raise HTTPException(status_code=503, detail="API key required but not configured on server")
81
+ provided = request.headers.get("X-API-Key") or request.headers.get("Authorization", "").removeprefix("Bearer ").strip()
82
+ if not provided:
83
+ raise HTTPException(status_code=401, detail="missing API key")
84
+ if provided != s.api_key:
85
+ raise HTTPException(status_code=403, detail="invalid API key")
86
+ # Record authenticated request for rate-limit accounting
87
+ if _rate_limiter is not None:
88
+ _rate_limiter.check(key=provided)
89
+
90
+
91
+ # ---------------------------------------------------------------------------
92
+ # Lifespan
93
+ # ---------------------------------------------------------------------------
94
+ @asynccontextmanager
95
+ async def lifespan(app: FastAPI):
96
+ """Configure logging, validate settings, prep upload dir."""
97
+ global _settings, _rate_limiter, _upload_dir
98
+
99
+ # Load settings (raises ConfigurationError -> startup aborts with 500)
100
+ try:
101
+ _settings = Settings.load()
102
+ except ConfigurationError as e:
103
+ # Re-raise so uvicorn logs it; we can't return JSON yet because
104
+ # the app isn't built.
105
+ _log.error("configuration error: %s", e)
106
+ raise
107
+
108
+ configure(level=_settings.log_level, fmt=_settings.log_format)
109
+ _log.info("py-idp API starting: host=%s port=%d storage=%s backend=%s",
110
+ _settings.api_host, _settings.api_port, _settings.storage_backend, _settings.default_backend)
111
+
112
+ _rate_limiter = RateLimiter(
113
+ per_key_per_minute=_settings.rate_limit_per_minute,
114
+ global_per_minute=0, # disabled by default; set via env if needed
115
+ )
116
+
117
+ _upload_dir = Path(os.environ.get("IDP_UPLOAD_DIR", "/tmp/idp-uploads"))
118
+ _upload_dir.mkdir(parents=True, exist_ok=True)
119
+
120
+ yield
121
+
122
+ # Shutdown: clean up, log final metrics
123
+ _log.info("py-idp API shutting down. metrics: %s", metrics.snapshot())
124
+ _settings = None
125
+ _rate_limiter = None
126
+ _upload_dir = None
127
+
128
+
129
+ app = FastAPI(
130
+ title="py-idp API",
131
+ version=__version__,
132
+ description="Production AI-enabled Intelligent Document Processing API.",
133
+ lifespan=lifespan,
134
+ )
135
+
136
+
137
+ # ---------------------------------------------------------------------------
138
+ # Middleware
139
+ # ---------------------------------------------------------------------------
140
+ @app.middleware("http")
141
+ async def logging_middleware(request: Request, call_next):
142
+ """Log every request with timing + status code."""
143
+ start = time.perf_counter()
144
+ metrics.inc("http_requests", 1, path=request.url.path, method=request.method)
145
+ try:
146
+ response = await call_next(request)
147
+ metrics.inc("http_responses", 1, status=str(response.status_code), path=request.url.path)
148
+ return response
149
+ except Exception as e:
150
+ metrics.inc("http_errors", 1, type=type(e).__name__)
151
+ _log.exception("request failed: %s %s", request.method, request.url.path)
152
+ return JSONResponse(
153
+ status_code=500,
154
+ content={"error": "internal server error", "type": type(e).__name__,
155
+ "is_idp_error": is_idp_error(e)},
156
+ )
157
+ finally:
158
+ elapsed = time.perf_counter() - start
159
+ metrics.observe("http_request_duration_seconds", elapsed, path=request.url.path)
160
+
161
+
162
+ # CORS (configured in lifespan but installed here so the order is correct)
163
+ def _install_cors() -> None:
164
+ s = _require_settings()
165
+ if s.cors_origins:
166
+ app.add_middleware(
167
+ CORSMiddleware,
168
+ allow_origins=list(s.cors_origins),
169
+ allow_credentials=True,
170
+ allow_methods=["*"],
171
+ allow_headers=["*"],
172
+ )
173
+
174
+
175
+ # ---------------------------------------------------------------------------
176
+ # Health endpoints
177
+ # ---------------------------------------------------------------------------
178
+ @app.get("/healthz", response_class=PlainTextResponse, include_in_schema=False)
179
+ async def healthz() -> str:
180
+ """Liveness: returns 200 OK if the process is up. Cheap; no DB call."""
181
+ return "ok"
182
+
183
+
184
+ @app.get("/readyz", response_class=PlainTextResponse, include_in_schema=False)
185
+ async def readyz() -> str:
186
+ """Readiness: 200 OK if settings are loaded; 503 if not yet ready."""
187
+ if _settings is None:
188
+ raise HTTPException(status_code=503, detail="not ready: settings not loaded")
189
+ return "ready"
190
+
191
+
192
+ @app.get("/version", response_class=PlainTextResponse)
193
+ async def version() -> str:
194
+ """Return the py-idp version."""
195
+ return __version__
196
+
197
+
198
+ @app.get("/metrics", response_class=PlainTextResponse, include_in_schema=False)
199
+ async def prometheus_metrics() -> Response:
200
+ """Prometheus text-format metrics."""
201
+ if not _require_settings().metrics_enabled:
202
+ raise HTTPException(status_code=404, detail="metrics disabled")
203
+ return PlainTextResponse(metrics.export_prometheus(), media_type="text/plain; version=0.0.4")
204
+
205
+
206
+ # ---------------------------------------------------------------------------
207
+ # Endpoints
208
+ # ---------------------------------------------------------------------------
209
+ @app.post("/extract", response_model=ExtractResponse, dependencies=[Depends(auth)])
210
+ async def extract_sync(
211
+ file: UploadFile,
212
+ schema_name: str | None = Form(default=None),
213
+ backend: str | None = Form(default=None),
214
+ ) -> ExtractResponse:
215
+ """Synchronous document extraction. Capped by ``IDP_MAX_UPLOAD_BYTES``.
216
+
217
+ Form fields:
218
+ - ``schema_name`` (optional): Pydantic schema to extract against.
219
+ - ``backend`` (optional): LLM backend name (e.g. ``mock``, ``openai/gpt-4o``).
220
+ """
221
+ s = _require_settings()
222
+ raw = await _read_capped(file, s.max_upload_bytes)
223
+ upload_dir = _require_upload_dir()
224
+ dst = upload_dir / (file.filename or "upload")
225
+ dst.write_bytes(raw)
226
+
227
+ pipeline = Pipeline(
228
+ backend=backend or s.default_backend,
229
+ schema=schema_name or "Invoice",
230
+ )
231
+ doc = Document.from_path(str(dst))
232
+ res = pipeline.run(doc)
233
+ metrics.inc("extractions", 1, backend=res.backend_name, schema=res.schema_name or "_none")
234
+ return ExtractResponse(
235
+ schema_name=res.schema_name,
236
+ backend_name=res.backend_name,
237
+ mode=res.mode or "ocr_llm", # type: ignore[arg-type] # router sets this; defensive against no-LLM paths
238
+ classification=res.classification,
239
+ extraction=res.document.extraction or {},
240
+ confidence=res.confidence,
241
+ validation=res.document.validation,
242
+ )
243
+
244
+
245
+ # ---------------------------------------------------------------------------
246
+ # Internals
247
+ # ---------------------------------------------------------------------------
248
+ def _require_settings() -> Settings:
249
+ if _settings is None:
250
+ raise HTTPException(status_code=503, detail="server not ready")
251
+ return _settings
252
+
253
+
254
+ def _require_upload_dir() -> Path:
255
+ if _upload_dir is None:
256
+ raise HTTPException(status_code=503, detail="server not ready: upload dir not configured")
257
+ return _upload_dir
258
+
259
+
260
+ async def _read_capped(file: UploadFile, max_bytes: int) -> bytes:
261
+ """Read ``file`` but reject payloads larger than ``max_bytes``.
262
+
263
+ Defends against missing or lying ``Content-Length`` headers by
264
+ checking actual bytes-read, not just the advertised length.
265
+ """
266
+ # Reject early if Content-Length advertised is over the cap
267
+ cl = file.headers.get("content-length")
268
+ if cl is not None:
269
+ try:
270
+ if int(cl) > max_bytes:
271
+ raise HTTPException(
272
+ status_code=413,
273
+ detail=f"payload too large: {cl} bytes > max {max_bytes}",
274
+ )
275
+ except ValueError:
276
+ pass # malformed header -> ignore; the chunk loop will catch it
277
+
278
+ chunks: list[bytes] = []
279
+ total = 0
280
+ chunk_size = 64 * 1024
281
+ while True:
282
+ chunk = await file.read(chunk_size)
283
+ if not chunk:
284
+ break
285
+ total += len(chunk)
286
+ if total > max_bytes:
287
+ raise HTTPException(
288
+ status_code=413,
289
+ detail=f"payload too large: exceeds max {max_bytes} bytes",
290
+ )
291
+ chunks.append(chunk)
292
+ return b"".join(chunks)
293
+
294
+
295
+ # Exception handlers — map IDPError subtypes to proper HTTP status codes
296
+ @app.exception_handler(IDPError)
297
+ async def idp_error_handler(request: Request, exc: IDPError) -> JSONResponse:
298
+ if isinstance(exc, RateLimitedError):
299
+ return JSONResponse(status_code=429, content={"error": str(exc), "type": type(exc).__name__})
300
+ return JSONResponse(status_code=400, content={"error": str(exc), "type": type(exc).__name__})
301
+
302
+
303
+ # ---------------------------------------------------------------------------
304
+ # Module version (also used by /version endpoint)
305
+ # ---------------------------------------------------------------------------
306
+ __all__ = ["app", "lifespan"]
idp/assess/__init__.py ADDED
@@ -0,0 +1,15 @@
1
+ # py-idp: general-purpose, AI-enabled Intelligent Document Processing.
2
+ # Copyright (c) 2026 Royce.
3
+ #
4
+ # Licensed under the GNU Affero General Public License v3.0 or later (AGPL-3.0-or-later)
5
+ # with the following addition: a commercial license is also available for organizations
6
+ # that wish to embed py-idp in proprietary products / hosted SaaS without the AGPL
7
+ # copyleft obligations. See LICENSE and LICENSE-COMMERCIAL at the repo root, or
8
+ # contact <royce-license-placeholder@protonmail.com> for terms.
9
+ #
10
+ # This Source Code Form is subject to the terms of the AGPL-3.0-or-later.
11
+ # SPDX-License-Identifier: AGPL-3.0-or-later
12
+
13
+ from idp.assess.confidence import assess_confidence
14
+
15
+ __all__ = ["assess_confidence"]
@@ -0,0 +1,130 @@
1
+ # py-idp: general-purpose, AI-enabled Intelligent Document Processing.
2
+ # Copyright (c) 2026 Royce.
3
+ #
4
+ # Licensed under the GNU Affero General Public License v3.0 or later (AGPL-3.0-or-later)
5
+ # with the following addition: a commercial license is also available for organizations
6
+ # that wish to embed py-idp in proprietary products / hosted SaaS without the AGPL
7
+ # copyleft obligations. See LICENSE and LICENSE-COMMERCIAL at the repo root, or
8
+ # contact <royce-license-placeholder@protonmail.com> for terms.
9
+ #
10
+ # This Source Code Form is subject to the terms of the AGPL-3.0-or-later.
11
+ # SPDX-License-Identifier: AGPL-3.0-or-later
12
+
13
+ """Confidence assessment.
14
+
15
+ Two strategies:
16
+ - heuristic: per-field conf derived from extraction-mode + type
17
+ - llm_self_assess: ask the model to rate each field as 0..1
18
+
19
+ LLM self-assessment is calibrated poorly in practice (papers repeatedly
20
+ show LLM confidence is overconfident). Heuristic is the default; LLM
21
+ self-assess is opt-in via config.
22
+ """
23
+ from __future__ import annotations
24
+
25
+ import json
26
+ import logging
27
+ from typing import TYPE_CHECKING
28
+
29
+ from idp.core.document import Document
30
+ from idp.llm.backend import Backend, CompletionRequest, Message
31
+
32
+ if TYPE_CHECKING:
33
+ from idp.rl.policy import PolicyConfig
34
+
35
+ log = logging.getLogger(__name__)
36
+
37
+
38
+ def _heuristic_confidence(doc: Document) -> dict[str, float]:
39
+ """Per-field confidence heuristic. Crude but reproducible.
40
+
41
+ Heuristics:
42
+ - multimodal mode + clean digital -> 0.9 base
43
+ - OCR_LLM mode -> 0.7 base
44
+ - per-field: missing/None -> 0.1
45
+ - per-field: present value (string non-empty, number, list, dict) -> base
46
+
47
+ NOTE: a numeric `0.0` or `0` is treated the same as any other present
48
+ numeric value (conf = base). If you need to penalize zero/extracted-as-zero
49
+ for compliance reasons, use `assess_confidence(use_llm=True)` or your own
50
+ rule that down-weights fields known to be "should never be zero".
51
+ """
52
+ base = 0.9 if doc.mode == "multimodal" else 0.7
53
+ out: dict[str, float] = {}
54
+ extraction = doc.extraction or {}
55
+ for k, v in extraction.items():
56
+ if v is None or v == "" or v == [] or v == {}:
57
+ out[k] = 0.1
58
+ elif isinstance(v, (int, float)):
59
+ out[k] = max(0.0, min(1.0, base))
60
+ elif isinstance(v, str):
61
+ out[k] = max(0.0, min(1.0, base + 0.05))
62
+ elif isinstance(v, list):
63
+ out[k] = max(0.0, min(1.0, base - 0.05)) # harder
64
+ elif isinstance(v, dict):
65
+ out[k] = max(0.0, min(1.0, base - 0.05))
66
+ else:
67
+ out[k] = base
68
+ return out
69
+
70
+
71
+ def _llm_self_assess(doc: Document, backend: Backend) -> dict[str, float]:
72
+ snippet = json.dumps(doc.extraction or {}, indent=2)
73
+ sample_text = (doc.raw_text or "")[:4000]
74
+ req = CompletionRequest(
75
+ messages=[
76
+ Message(
77
+ role="system",
78
+ content=(
79
+ "You rate extraction confidence per field. "
80
+ "Return JSON: {\"field_name\": 0.0-1.0, ...} "
81
+ "where 1.0 = definitely correct, 0.0 = definitely wrong or absent."
82
+ ),
83
+ ),
84
+ Message(
85
+ role="user",
86
+ content=(
87
+ f"SOURCE DOCUMENT (truncated):\n{sample_text}\n\n"
88
+ f"EXTRACTION TO RATE:\n{snippet}"
89
+ ),
90
+ ),
91
+ ],
92
+ json_mode=True,
93
+ temperature=0.0,
94
+ )
95
+ raw = backend.complete(req)
96
+ try:
97
+ return {k: float(v) for k, v in json.loads(raw).items()}
98
+ except Exception as e: # noqa: BLE001
99
+ log.debug("llm self-assess parse failed: %s", e)
100
+ return {}
101
+
102
+
103
+ def assess_confidence(
104
+ doc: Document,
105
+ backend: Backend | None = None,
106
+ use_llm: bool = False,
107
+ policy: PolicyConfig | None = None,
108
+ ) -> Document:
109
+ """Attach doc.confidence (per-field dict of floats in 0..1).
110
+
111
+ If a `policy` is provided (an `idp.rl.PolicyConfig`), apply per-field
112
+ penalties so that fields humans have corrected frequently get
113
+ lower confidence scores and surface to HITL review.
114
+ """
115
+ conf = _heuristic_confidence(doc)
116
+ if use_llm and backend is not None:
117
+ try:
118
+ llm_conf = _llm_self_assess(doc, backend)
119
+ # blend: 70% heuristic (more honest), 30% self-assess
120
+ conf = {k: 0.7 * conf.get(k, 0.5) + 0.3 * llm_conf.get(k, 0.5) for k in conf}
121
+ except Exception as e: # noqa: BLE001
122
+ doc.errors.append(f"assess_failed: {e}")
123
+ if policy is not None:
124
+ try:
125
+ from idp.rl.policy import policy_to_penalised_confidence
126
+ conf = policy_to_penalised_confidence(conf, policy)
127
+ except Exception as e: # noqa: BLE001
128
+ doc.errors.append(f"policy_apply_failed: {e}")
129
+ doc.confidence = conf
130
+ return doc
idp/auth/__init__.py ADDED
@@ -0,0 +1,21 @@
1
+ # py-idp: general-purpose, AI-enabled Intelligent Document Processing.
2
+ # Copyright (c) 2026 Royce.
3
+ #
4
+ # Licensed under the GNU Affero General Public License v3.0 or later (AGPL-3.0-or-later)
5
+ # with the following addition: a commercial license is also available for organizations
6
+ # that wish to embed py-idp in proprietary products / hosted SaaS without the AGPL
7
+ # copyleft obligations. See LICENSE and LICENSE-COMMERCIAL at the repo root, or
8
+ # contact <royce-license-placeholder@protonmail.com> for terms.
9
+ #
10
+ # This Source Code Form is subject to the terms of the AGPL-3.0-or-later.
11
+ # SPDX-License-Identifier: AGPL-3.0-or-later
12
+
13
+ from idp.auth.keys import (
14
+ AuthContext,
15
+ hash_key,
16
+ make_default_key,
17
+ require_api_key,
18
+ verify,
19
+ )
20
+
21
+ __all__ = ["AuthContext", "hash_key", "make_default_key", "require_api_key", "verify"]