py-idp 0.3.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- idp/__init__.py +39 -0
- idp/_logging.py +69 -0
- idp/_testing_backends.py +37 -0
- idp/_util.py +38 -0
- idp/api.py +306 -0
- idp/assess/__init__.py +15 -0
- idp/assess/confidence.py +130 -0
- idp/auth/__init__.py +21 -0
- idp/auth/keys.py +51 -0
- idp/checkpoint.py +251 -0
- idp/chunker.py +294 -0
- idp/classify/__init__.py +15 -0
- idp/classify/classifier.py +133 -0
- idp/compat_v01.py +88 -0
- idp/config.py +159 -0
- idp/core/__init__.py +16 -0
- idp/core/document.py +116 -0
- idp/core/schemas.py +145 -0
- idp/core/types.py +26 -0
- idp/discover.py +580 -0
- idp/errors.py +96 -0
- idp/eval/__init__.py +12 -0
- idp/eval/metrics.py +90 -0
- idp/eval/runner.py +135 -0
- idp/extract/__init__.py +15 -0
- idp/extract/extractor.py +499 -0
- idp/hitl/__init__.py +12 -0
- idp/hitl/app.py +266 -0
- idp/llm/__init__.py +39 -0
- idp/llm/backend.py +583 -0
- idp/llm/china.py +161 -0
- idp/llm/nanonets.py +440 -0
- idp/llm/nanonets_batch.py +228 -0
- idp/metrics.py +153 -0
- idp/migrate_audit.py +118 -0
- idp/parse/__init__.py +29 -0
- idp/parse/parser.py +258 -0
- idp/parse/pdf_pages.py +177 -0
- idp/parse/router.py +81 -0
- idp/pipeline/__init__.py +15 -0
- idp/pipeline/cli.py +371 -0
- idp/pipeline/pipeline.py +229 -0
- idp/policy_config.py +5 -0
- idp/queue/__init__.py +15 -0
- idp/queue/jobs.py +119 -0
- idp/ratelimit.py +73 -0
- idp/reliability.py +434 -0
- idp/rl/__init__.py +50 -0
- idp/rl/calibrate.py +375 -0
- idp/rl/online.py +202 -0
- idp/rl/policy.py +161 -0
- idp/rl/reward.py +144 -0
- idp/rl/update.py +124 -0
- idp/storage/__init__.py +22 -0
- idp/storage/factory.py +62 -0
- idp/storage/sql.py +574 -0
- idp/storage/store.py +258 -0
- idp/validate/__init__.py +25 -0
- idp/validate/validator.py +94 -0
- py_idp-0.3.1.dist-info/METADATA +811 -0
- py_idp-0.3.1.dist-info/RECORD +67 -0
- py_idp-0.3.1.dist-info/WHEEL +5 -0
- py_idp-0.3.1.dist-info/entry_points.txt +2 -0
- py_idp-0.3.1.dist-info/licenses/LICENSE +50 -0
- py_idp-0.3.1.dist-info/licenses/LICENSE-AGPL +620 -0
- py_idp-0.3.1.dist-info/licenses/LICENSE-COMMERCIAL +95 -0
- py_idp-0.3.1.dist-info/top_level.txt +1 -0
idp/__init__.py
ADDED
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
# py-idp: general-purpose, AI-enabled Intelligent Document Processing.
|
|
2
|
+
# Copyright (c) 2026 Royce.
|
|
3
|
+
#
|
|
4
|
+
# Licensed under the GNU Affero General Public License v3.0 or later (AGPL-3.0-or-later)
|
|
5
|
+
# with the following addition: a commercial license is also available for organizations
|
|
6
|
+
# that wish to embed py-idp in proprietary products / hosted SaaS without the AGPL
|
|
7
|
+
# copyleft obligations. See LICENSE and LICENSE-COMMERCIAL at the repo root, or
|
|
8
|
+
# contact <royce-license-placeholder@protonmail.com> for terms.
|
|
9
|
+
#
|
|
10
|
+
# This Source Code Form is subject to the terms of the AGPL-3.0-or-later.
|
|
11
|
+
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
12
|
+
|
|
13
|
+
"""py-idp: General-purpose, AI-enabled Intelligent Document Processing framework.
|
|
14
|
+
|
|
15
|
+
A six-stage pipeline: parse -> classify -> extract -> assess -> validate -> HITL.
|
|
16
|
+
Each stage is a pure function over a Document, pluggable, and independently testable.
|
|
17
|
+
|
|
18
|
+
Design draws from:
|
|
19
|
+
- aws-solutions-library-samples/accelerated-intelligent-document-processing-on-aws
|
|
20
|
+
(pipeline shape, HITL, confidence assessment)
|
|
21
|
+
- docling-project/docling (parser: PDF, tables, reading order)
|
|
22
|
+
- run-llama/llama_cloud_services (Pydantic-schema-driven extraction API)
|
|
23
|
+
- Unstructured-IO/unstructured (chunking + multi-format ingest)
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from idp.core.document import Block, Document, Page
|
|
27
|
+
from idp.discover import DiscoveryResult, discover_schema
|
|
28
|
+
from idp.pipeline.pipeline import Pipeline, PipelineResult
|
|
29
|
+
|
|
30
|
+
__version__ = "0.3.2"
|
|
31
|
+
__all__ = [
|
|
32
|
+
"Block",
|
|
33
|
+
"DiscoveryResult",
|
|
34
|
+
"Document",
|
|
35
|
+
"Page",
|
|
36
|
+
"Pipeline",
|
|
37
|
+
"PipelineResult",
|
|
38
|
+
"discover_schema",
|
|
39
|
+
]
|
idp/_logging.py
ADDED
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"""Structured logging configuration for py-idp.
|
|
2
|
+
|
|
3
|
+
Provides a single ``get_logger(name)`` helper so every module emits logs
|
|
4
|
+
in the same format (timestamp, level, logger name, message) and honours
|
|
5
|
+
the ``LOG_LEVEL`` environment variable. Production deployments should
|
|
6
|
+
set ``LOG_LEVEL=INFO`` (default) or ``LOG_LEVEL=WARNING`` for quieter
|
|
7
|
+
logs; ``LOG_FORMAT=json`` switches to JSON output for log aggregators.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import logging
|
|
12
|
+
import os
|
|
13
|
+
import sys
|
|
14
|
+
|
|
15
|
+
_LOG_FORMAT = "%(asctime)s %(levelname)s [%(name)s] %(message)s"
|
|
16
|
+
_LOG_DATEFMT = "%Y-%m-%dT%H:%M:%S%z"
|
|
17
|
+
_CONFIGURED = False
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def configure(level: str | int | None = None, fmt: str | None = None) -> None:
|
|
21
|
+
"""Idempotent root-logger configuration.
|
|
22
|
+
|
|
23
|
+
Reads ``LOG_LEVEL`` and ``LOG_FORMAT`` from env by default. Safe to
|
|
24
|
+
call multiple times — subsequent calls are no-ops.
|
|
25
|
+
"""
|
|
26
|
+
global _CONFIGURED
|
|
27
|
+
if _CONFIGURED:
|
|
28
|
+
return
|
|
29
|
+
|
|
30
|
+
if level is None:
|
|
31
|
+
level = os.environ.get("LOG_LEVEL", "INFO").upper()
|
|
32
|
+
if isinstance(level, str):
|
|
33
|
+
level = getattr(logging, level, logging.INFO)
|
|
34
|
+
|
|
35
|
+
if fmt is None:
|
|
36
|
+
fmt = os.environ.get("LOG_FORMAT", _LOG_FORMAT)
|
|
37
|
+
if fmt.lower() == "json":
|
|
38
|
+
fmt = _json_log_format()
|
|
39
|
+
|
|
40
|
+
handler = logging.StreamHandler(sys.stderr)
|
|
41
|
+
handler.setFormatter(logging.Formatter(fmt=fmt, datefmt=_LOG_DATEFMT))
|
|
42
|
+
root = logging.getLogger()
|
|
43
|
+
# Replace existing handlers so format is honoured (don't accumulate)
|
|
44
|
+
root.handlers = [handler]
|
|
45
|
+
root.setLevel(level)
|
|
46
|
+
_CONFIGURED = True
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def get_logger(name: str) -> logging.Logger:
|
|
50
|
+
"""Return a module-level logger. Configures the root on first call."""
|
|
51
|
+
configure()
|
|
52
|
+
return logging.getLogger(name)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _json_log_format() -> str:
|
|
56
|
+
"""Build a JSON formatter string consumed by python-json-logger downstream.
|
|
57
|
+
|
|
58
|
+
Returns the empty string here so the default Formatter falls back to
|
|
59
|
+
the human-readable format. Operators wanting JSON should install
|
|
60
|
+
``python-json-logger`` and override via ``LOG_FORMAT_HANDLER``.
|
|
61
|
+
"""
|
|
62
|
+
return _LOG_FORMAT
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def reset() -> None:
|
|
66
|
+
"""Reset configuration state — for tests only."""
|
|
67
|
+
global _CONFIGURED
|
|
68
|
+
_CONFIGURED = False
|
|
69
|
+
logging.getLogger().handlers = []
|
idp/_testing_backends.py
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""Testing-only LLM backends.
|
|
2
|
+
|
|
3
|
+
This module is intentionally NOT part of the public ``idp`` namespace.
|
|
4
|
+
Its backends are gated by env vars (``IDP_ENABLE_SLOWMOCK=1``) so
|
|
5
|
+
they cannot be invoked in production by accident.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import random
|
|
10
|
+
import time
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from idp.llm.backend import CompletionRequest, MockBackend
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class SlowMockBackend(MockBackend):
|
|
17
|
+
"""A MockBackend that sleeps to simulate real LLM latency.
|
|
18
|
+
|
|
19
|
+
Configured by ``LOAD_LATENCY_MS`` (default 1500) +/- ``LOAD_JITTER_MS``
|
|
20
|
+
(default 500). Only enabled when ``IDP_ENABLE_SLOWMOCK=1``.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
name = "slowmock"
|
|
24
|
+
|
|
25
|
+
def __init__(self, latency_ms: int = 1500, jitter_ms: int = 500, **kwargs: Any):
|
|
26
|
+
super().__init__(**kwargs)
|
|
27
|
+
self.latency_ms = latency_ms
|
|
28
|
+
self.jitter_ms = jitter_ms
|
|
29
|
+
|
|
30
|
+
def complete(self, req: CompletionRequest) -> str:
|
|
31
|
+
sleep_s = (self.latency_ms + random.uniform(-self.jitter_ms, self.jitter_ms)) / 1000.0
|
|
32
|
+
if sleep_s > 0:
|
|
33
|
+
time.sleep(sleep_s)
|
|
34
|
+
return super().complete(req)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
__all__ = ["SlowMockBackend"]
|
idp/_util.py
ADDED
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
# py-idp: general-purpose, AI-enabled Intelligent Document Processing.
|
|
2
|
+
# Copyright (c) 2026 Royce.
|
|
3
|
+
#
|
|
4
|
+
# Licensed under the GNU Affero General Public License v3.0 or later (AGPL-3.0-or-later)
|
|
5
|
+
# with the following addition: a commercial license is also available for organizations
|
|
6
|
+
# that wish to embed py-idp in proprietary products / hosted SaaS without the AGPL
|
|
7
|
+
# copyleft obligations. See LICENSE and LICENSE-COMMERCIAL at the repo root, or
|
|
8
|
+
# contact <royce-license-placeholder@protonmail.com> for terms.
|
|
9
|
+
#
|
|
10
|
+
# This Source Code Form is subject to the terms of the AGPL-3.0-or-later.
|
|
11
|
+
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
12
|
+
|
|
13
|
+
"""Pretty-print a PipelineResult in a terminal-friendly way."""
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import json
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def pretty_print_result(result) -> None:
|
|
20
|
+
"""Console-friendly rendering of a pipeline result."""
|
|
21
|
+
print(f"\n=== {result.document.source_path} ===")
|
|
22
|
+
print(f"schema: {result.schema_name}")
|
|
23
|
+
print(f"backend: {result.backend_name} ({result.mode})")
|
|
24
|
+
print(f"classify: {result.classification} (conf={result.document.classification_confidence})")
|
|
25
|
+
print(f"validate: {'PASS' if result.validation_passed else 'FAIL'}")
|
|
26
|
+
print("timings: " + ", ".join(f"{t.name}={t.seconds:.3f}s" for t in result.timings))
|
|
27
|
+
print("\nextraction:")
|
|
28
|
+
print(json.dumps(result.document.extraction, indent=2, default=str))
|
|
29
|
+
if result.confidence:
|
|
30
|
+
ordered = sorted(result.confidence.items(), key=lambda kv: kv[1])
|
|
31
|
+
print("\nconfidence (ascending):")
|
|
32
|
+
for k, v in ordered:
|
|
33
|
+
mark = " [REVIEW]" if v < 0.6 else ""
|
|
34
|
+
print(f" {k:<24} {v:.2f}{mark}")
|
|
35
|
+
if result.document.errors:
|
|
36
|
+
print(f"\nerrors ({len(result.document.errors)}):")
|
|
37
|
+
for e in result.document.errors:
|
|
38
|
+
print(f" - {e}")
|
idp/api.py
ADDED
|
@@ -0,0 +1,306 @@
|
|
|
1
|
+
"""Production-hardened FastAPI server.
|
|
2
|
+
|
|
3
|
+
Improvements over the original ``examples/api.py``:
|
|
4
|
+
|
|
5
|
+
* Reads ``Settings`` from env (port, workers, rate limits, max upload
|
|
6
|
+
size, backend, storage) and **fails fast** on misconfiguration.
|
|
7
|
+
* Health (``/healthz``) and readiness (``/readyz``) endpoints for k8s.
|
|
8
|
+
* Prometheus text exposition at ``/metrics``.
|
|
9
|
+
* Per-key + global rate limiting via ``idp.ratelimit.RateLimiter``.
|
|
10
|
+
* Upload size limit enforced via ``Content-Length`` *and* actual
|
|
11
|
+
stream read (defends against missing/lying Content-Length headers).
|
|
12
|
+
* ``/version`` endpoint exposes the package version for ops dashboards.
|
|
13
|
+
* Structured request logging with timing.
|
|
14
|
+
|
|
15
|
+
Not in scope: HTTPS termination (use a reverse proxy), TLS, SSO,
|
|
16
|
+
multi-tenant auth (these are deployment-level concerns).
|
|
17
|
+
"""
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import os
|
|
21
|
+
import time
|
|
22
|
+
from contextlib import asynccontextmanager
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
from typing import Any
|
|
25
|
+
|
|
26
|
+
from fastapi import Depends, FastAPI, Form, HTTPException, Request, UploadFile
|
|
27
|
+
from fastapi.middleware.cors import CORSMiddleware
|
|
28
|
+
from fastapi.responses import JSONResponse, PlainTextResponse, Response
|
|
29
|
+
from pydantic import BaseModel
|
|
30
|
+
|
|
31
|
+
from idp import __version__
|
|
32
|
+
from idp._logging import configure, get_logger
|
|
33
|
+
from idp.config import Settings
|
|
34
|
+
from idp.core.document import Document
|
|
35
|
+
from idp.errors import ConfigurationError, IDPError, RateLimitedError, is_idp_error
|
|
36
|
+
from idp.metrics import metrics
|
|
37
|
+
from idp.pipeline.pipeline import Pipeline
|
|
38
|
+
from idp.ratelimit import RateLimiter
|
|
39
|
+
|
|
40
|
+
# Module-level logger; configured at lifespan startup.
|
|
41
|
+
_log = get_logger("idp.api")
|
|
42
|
+
_settings: Settings | None = None
|
|
43
|
+
_rate_limiter: RateLimiter | None = None
|
|
44
|
+
_upload_dir: Path | None = None
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
# ---------------------------------------------------------------------------
|
|
48
|
+
# Schemas
|
|
49
|
+
# ---------------------------------------------------------------------------
|
|
50
|
+
class ExtractRequest(BaseModel):
|
|
51
|
+
schema_name: str | None = None
|
|
52
|
+
backend: str | None = None
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class ExtractResponse(BaseModel):
|
|
56
|
+
schema_name: str | None = None
|
|
57
|
+
backend_name: str
|
|
58
|
+
mode: str
|
|
59
|
+
classification: str | None
|
|
60
|
+
extraction: dict[str, Any]
|
|
61
|
+
confidence: dict[str, float] | None
|
|
62
|
+
validation: dict[str, Any] | None
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class ReviewRequest(BaseModel):
|
|
66
|
+
edited: dict[str, Any]
|
|
67
|
+
reviewer: str
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
# ---------------------------------------------------------------------------
|
|
71
|
+
# Auth
|
|
72
|
+
# ---------------------------------------------------------------------------
|
|
73
|
+
def auth(request: Request) -> None:
|
|
74
|
+
"""Validate the ``X-API-Key`` header against ``Settings.api_key``."""
|
|
75
|
+
s = _require_settings()
|
|
76
|
+
if not s.api_key_required:
|
|
77
|
+
return
|
|
78
|
+
if s.api_key is None:
|
|
79
|
+
# No key configured but key required -> fail closed.
|
|
80
|
+
raise HTTPException(status_code=503, detail="API key required but not configured on server")
|
|
81
|
+
provided = request.headers.get("X-API-Key") or request.headers.get("Authorization", "").removeprefix("Bearer ").strip()
|
|
82
|
+
if not provided:
|
|
83
|
+
raise HTTPException(status_code=401, detail="missing API key")
|
|
84
|
+
if provided != s.api_key:
|
|
85
|
+
raise HTTPException(status_code=403, detail="invalid API key")
|
|
86
|
+
# Record authenticated request for rate-limit accounting
|
|
87
|
+
if _rate_limiter is not None:
|
|
88
|
+
_rate_limiter.check(key=provided)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
# ---------------------------------------------------------------------------
|
|
92
|
+
# Lifespan
|
|
93
|
+
# ---------------------------------------------------------------------------
|
|
94
|
+
@asynccontextmanager
|
|
95
|
+
async def lifespan(app: FastAPI):
|
|
96
|
+
"""Configure logging, validate settings, prep upload dir."""
|
|
97
|
+
global _settings, _rate_limiter, _upload_dir
|
|
98
|
+
|
|
99
|
+
# Load settings (raises ConfigurationError -> startup aborts with 500)
|
|
100
|
+
try:
|
|
101
|
+
_settings = Settings.load()
|
|
102
|
+
except ConfigurationError as e:
|
|
103
|
+
# Re-raise so uvicorn logs it; we can't return JSON yet because
|
|
104
|
+
# the app isn't built.
|
|
105
|
+
_log.error("configuration error: %s", e)
|
|
106
|
+
raise
|
|
107
|
+
|
|
108
|
+
configure(level=_settings.log_level, fmt=_settings.log_format)
|
|
109
|
+
_log.info("py-idp API starting: host=%s port=%d storage=%s backend=%s",
|
|
110
|
+
_settings.api_host, _settings.api_port, _settings.storage_backend, _settings.default_backend)
|
|
111
|
+
|
|
112
|
+
_rate_limiter = RateLimiter(
|
|
113
|
+
per_key_per_minute=_settings.rate_limit_per_minute,
|
|
114
|
+
global_per_minute=0, # disabled by default; set via env if needed
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
_upload_dir = Path(os.environ.get("IDP_UPLOAD_DIR", "/tmp/idp-uploads"))
|
|
118
|
+
_upload_dir.mkdir(parents=True, exist_ok=True)
|
|
119
|
+
|
|
120
|
+
yield
|
|
121
|
+
|
|
122
|
+
# Shutdown: clean up, log final metrics
|
|
123
|
+
_log.info("py-idp API shutting down. metrics: %s", metrics.snapshot())
|
|
124
|
+
_settings = None
|
|
125
|
+
_rate_limiter = None
|
|
126
|
+
_upload_dir = None
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
app = FastAPI(
|
|
130
|
+
title="py-idp API",
|
|
131
|
+
version=__version__,
|
|
132
|
+
description="Production AI-enabled Intelligent Document Processing API.",
|
|
133
|
+
lifespan=lifespan,
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
# ---------------------------------------------------------------------------
|
|
138
|
+
# Middleware
|
|
139
|
+
# ---------------------------------------------------------------------------
|
|
140
|
+
@app.middleware("http")
|
|
141
|
+
async def logging_middleware(request: Request, call_next):
|
|
142
|
+
"""Log every request with timing + status code."""
|
|
143
|
+
start = time.perf_counter()
|
|
144
|
+
metrics.inc("http_requests", 1, path=request.url.path, method=request.method)
|
|
145
|
+
try:
|
|
146
|
+
response = await call_next(request)
|
|
147
|
+
metrics.inc("http_responses", 1, status=str(response.status_code), path=request.url.path)
|
|
148
|
+
return response
|
|
149
|
+
except Exception as e:
|
|
150
|
+
metrics.inc("http_errors", 1, type=type(e).__name__)
|
|
151
|
+
_log.exception("request failed: %s %s", request.method, request.url.path)
|
|
152
|
+
return JSONResponse(
|
|
153
|
+
status_code=500,
|
|
154
|
+
content={"error": "internal server error", "type": type(e).__name__,
|
|
155
|
+
"is_idp_error": is_idp_error(e)},
|
|
156
|
+
)
|
|
157
|
+
finally:
|
|
158
|
+
elapsed = time.perf_counter() - start
|
|
159
|
+
metrics.observe("http_request_duration_seconds", elapsed, path=request.url.path)
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
# CORS (configured in lifespan but installed here so the order is correct)
|
|
163
|
+
def _install_cors() -> None:
|
|
164
|
+
s = _require_settings()
|
|
165
|
+
if s.cors_origins:
|
|
166
|
+
app.add_middleware(
|
|
167
|
+
CORSMiddleware,
|
|
168
|
+
allow_origins=list(s.cors_origins),
|
|
169
|
+
allow_credentials=True,
|
|
170
|
+
allow_methods=["*"],
|
|
171
|
+
allow_headers=["*"],
|
|
172
|
+
)
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
# ---------------------------------------------------------------------------
|
|
176
|
+
# Health endpoints
|
|
177
|
+
# ---------------------------------------------------------------------------
|
|
178
|
+
@app.get("/healthz", response_class=PlainTextResponse, include_in_schema=False)
|
|
179
|
+
async def healthz() -> str:
|
|
180
|
+
"""Liveness: returns 200 OK if the process is up. Cheap; no DB call."""
|
|
181
|
+
return "ok"
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
@app.get("/readyz", response_class=PlainTextResponse, include_in_schema=False)
|
|
185
|
+
async def readyz() -> str:
|
|
186
|
+
"""Readiness: 200 OK if settings are loaded; 503 if not yet ready."""
|
|
187
|
+
if _settings is None:
|
|
188
|
+
raise HTTPException(status_code=503, detail="not ready: settings not loaded")
|
|
189
|
+
return "ready"
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
@app.get("/version", response_class=PlainTextResponse)
|
|
193
|
+
async def version() -> str:
|
|
194
|
+
"""Return the py-idp version."""
|
|
195
|
+
return __version__
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
@app.get("/metrics", response_class=PlainTextResponse, include_in_schema=False)
|
|
199
|
+
async def prometheus_metrics() -> Response:
|
|
200
|
+
"""Prometheus text-format metrics."""
|
|
201
|
+
if not _require_settings().metrics_enabled:
|
|
202
|
+
raise HTTPException(status_code=404, detail="metrics disabled")
|
|
203
|
+
return PlainTextResponse(metrics.export_prometheus(), media_type="text/plain; version=0.0.4")
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
# ---------------------------------------------------------------------------
|
|
207
|
+
# Endpoints
|
|
208
|
+
# ---------------------------------------------------------------------------
|
|
209
|
+
@app.post("/extract", response_model=ExtractResponse, dependencies=[Depends(auth)])
|
|
210
|
+
async def extract_sync(
|
|
211
|
+
file: UploadFile,
|
|
212
|
+
schema_name: str | None = Form(default=None),
|
|
213
|
+
backend: str | None = Form(default=None),
|
|
214
|
+
) -> ExtractResponse:
|
|
215
|
+
"""Synchronous document extraction. Capped by ``IDP_MAX_UPLOAD_BYTES``.
|
|
216
|
+
|
|
217
|
+
Form fields:
|
|
218
|
+
- ``schema_name`` (optional): Pydantic schema to extract against.
|
|
219
|
+
- ``backend`` (optional): LLM backend name (e.g. ``mock``, ``openai/gpt-4o``).
|
|
220
|
+
"""
|
|
221
|
+
s = _require_settings()
|
|
222
|
+
raw = await _read_capped(file, s.max_upload_bytes)
|
|
223
|
+
upload_dir = _require_upload_dir()
|
|
224
|
+
dst = upload_dir / (file.filename or "upload")
|
|
225
|
+
dst.write_bytes(raw)
|
|
226
|
+
|
|
227
|
+
pipeline = Pipeline(
|
|
228
|
+
backend=backend or s.default_backend,
|
|
229
|
+
schema=schema_name or "Invoice",
|
|
230
|
+
)
|
|
231
|
+
doc = Document.from_path(str(dst))
|
|
232
|
+
res = pipeline.run(doc)
|
|
233
|
+
metrics.inc("extractions", 1, backend=res.backend_name, schema=res.schema_name or "_none")
|
|
234
|
+
return ExtractResponse(
|
|
235
|
+
schema_name=res.schema_name,
|
|
236
|
+
backend_name=res.backend_name,
|
|
237
|
+
mode=res.mode or "ocr_llm", # type: ignore[arg-type] # router sets this; defensive against no-LLM paths
|
|
238
|
+
classification=res.classification,
|
|
239
|
+
extraction=res.document.extraction or {},
|
|
240
|
+
confidence=res.confidence,
|
|
241
|
+
validation=res.document.validation,
|
|
242
|
+
)
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
# ---------------------------------------------------------------------------
|
|
246
|
+
# Internals
|
|
247
|
+
# ---------------------------------------------------------------------------
|
|
248
|
+
def _require_settings() -> Settings:
|
|
249
|
+
if _settings is None:
|
|
250
|
+
raise HTTPException(status_code=503, detail="server not ready")
|
|
251
|
+
return _settings
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def _require_upload_dir() -> Path:
|
|
255
|
+
if _upload_dir is None:
|
|
256
|
+
raise HTTPException(status_code=503, detail="server not ready: upload dir not configured")
|
|
257
|
+
return _upload_dir
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
async def _read_capped(file: UploadFile, max_bytes: int) -> bytes:
|
|
261
|
+
"""Read ``file`` but reject payloads larger than ``max_bytes``.
|
|
262
|
+
|
|
263
|
+
Defends against missing or lying ``Content-Length`` headers by
|
|
264
|
+
checking actual bytes-read, not just the advertised length.
|
|
265
|
+
"""
|
|
266
|
+
# Reject early if Content-Length advertised is over the cap
|
|
267
|
+
cl = file.headers.get("content-length")
|
|
268
|
+
if cl is not None:
|
|
269
|
+
try:
|
|
270
|
+
if int(cl) > max_bytes:
|
|
271
|
+
raise HTTPException(
|
|
272
|
+
status_code=413,
|
|
273
|
+
detail=f"payload too large: {cl} bytes > max {max_bytes}",
|
|
274
|
+
)
|
|
275
|
+
except ValueError:
|
|
276
|
+
pass # malformed header -> ignore; the chunk loop will catch it
|
|
277
|
+
|
|
278
|
+
chunks: list[bytes] = []
|
|
279
|
+
total = 0
|
|
280
|
+
chunk_size = 64 * 1024
|
|
281
|
+
while True:
|
|
282
|
+
chunk = await file.read(chunk_size)
|
|
283
|
+
if not chunk:
|
|
284
|
+
break
|
|
285
|
+
total += len(chunk)
|
|
286
|
+
if total > max_bytes:
|
|
287
|
+
raise HTTPException(
|
|
288
|
+
status_code=413,
|
|
289
|
+
detail=f"payload too large: exceeds max {max_bytes} bytes",
|
|
290
|
+
)
|
|
291
|
+
chunks.append(chunk)
|
|
292
|
+
return b"".join(chunks)
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
# Exception handlers — map IDPError subtypes to proper HTTP status codes
|
|
296
|
+
@app.exception_handler(IDPError)
|
|
297
|
+
async def idp_error_handler(request: Request, exc: IDPError) -> JSONResponse:
|
|
298
|
+
if isinstance(exc, RateLimitedError):
|
|
299
|
+
return JSONResponse(status_code=429, content={"error": str(exc), "type": type(exc).__name__})
|
|
300
|
+
return JSONResponse(status_code=400, content={"error": str(exc), "type": type(exc).__name__})
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
# ---------------------------------------------------------------------------
|
|
304
|
+
# Module version (also used by /version endpoint)
|
|
305
|
+
# ---------------------------------------------------------------------------
|
|
306
|
+
__all__ = ["app", "lifespan"]
|
idp/assess/__init__.py
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
# py-idp: general-purpose, AI-enabled Intelligent Document Processing.
|
|
2
|
+
# Copyright (c) 2026 Royce.
|
|
3
|
+
#
|
|
4
|
+
# Licensed under the GNU Affero General Public License v3.0 or later (AGPL-3.0-or-later)
|
|
5
|
+
# with the following addition: a commercial license is also available for organizations
|
|
6
|
+
# that wish to embed py-idp in proprietary products / hosted SaaS without the AGPL
|
|
7
|
+
# copyleft obligations. See LICENSE and LICENSE-COMMERCIAL at the repo root, or
|
|
8
|
+
# contact <royce-license-placeholder@protonmail.com> for terms.
|
|
9
|
+
#
|
|
10
|
+
# This Source Code Form is subject to the terms of the AGPL-3.0-or-later.
|
|
11
|
+
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
12
|
+
|
|
13
|
+
from idp.assess.confidence import assess_confidence
|
|
14
|
+
|
|
15
|
+
__all__ = ["assess_confidence"]
|
idp/assess/confidence.py
ADDED
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
# py-idp: general-purpose, AI-enabled Intelligent Document Processing.
|
|
2
|
+
# Copyright (c) 2026 Royce.
|
|
3
|
+
#
|
|
4
|
+
# Licensed under the GNU Affero General Public License v3.0 or later (AGPL-3.0-or-later)
|
|
5
|
+
# with the following addition: a commercial license is also available for organizations
|
|
6
|
+
# that wish to embed py-idp in proprietary products / hosted SaaS without the AGPL
|
|
7
|
+
# copyleft obligations. See LICENSE and LICENSE-COMMERCIAL at the repo root, or
|
|
8
|
+
# contact <royce-license-placeholder@protonmail.com> for terms.
|
|
9
|
+
#
|
|
10
|
+
# This Source Code Form is subject to the terms of the AGPL-3.0-or-later.
|
|
11
|
+
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
12
|
+
|
|
13
|
+
"""Confidence assessment.
|
|
14
|
+
|
|
15
|
+
Two strategies:
|
|
16
|
+
- heuristic: per-field conf derived from extraction-mode + type
|
|
17
|
+
- llm_self_assess: ask the model to rate each field as 0..1
|
|
18
|
+
|
|
19
|
+
LLM self-assessment is calibrated poorly in practice (papers repeatedly
|
|
20
|
+
show LLM confidence is overconfident). Heuristic is the default; LLM
|
|
21
|
+
self-assess is opt-in via config.
|
|
22
|
+
"""
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import json
|
|
26
|
+
import logging
|
|
27
|
+
from typing import TYPE_CHECKING
|
|
28
|
+
|
|
29
|
+
from idp.core.document import Document
|
|
30
|
+
from idp.llm.backend import Backend, CompletionRequest, Message
|
|
31
|
+
|
|
32
|
+
if TYPE_CHECKING:
|
|
33
|
+
from idp.rl.policy import PolicyConfig
|
|
34
|
+
|
|
35
|
+
log = logging.getLogger(__name__)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _heuristic_confidence(doc: Document) -> dict[str, float]:
|
|
39
|
+
"""Per-field confidence heuristic. Crude but reproducible.
|
|
40
|
+
|
|
41
|
+
Heuristics:
|
|
42
|
+
- multimodal mode + clean digital -> 0.9 base
|
|
43
|
+
- OCR_LLM mode -> 0.7 base
|
|
44
|
+
- per-field: missing/None -> 0.1
|
|
45
|
+
- per-field: present value (string non-empty, number, list, dict) -> base
|
|
46
|
+
|
|
47
|
+
NOTE: a numeric `0.0` or `0` is treated the same as any other present
|
|
48
|
+
numeric value (conf = base). If you need to penalize zero/extracted-as-zero
|
|
49
|
+
for compliance reasons, use `assess_confidence(use_llm=True)` or your own
|
|
50
|
+
rule that down-weights fields known to be "should never be zero".
|
|
51
|
+
"""
|
|
52
|
+
base = 0.9 if doc.mode == "multimodal" else 0.7
|
|
53
|
+
out: dict[str, float] = {}
|
|
54
|
+
extraction = doc.extraction or {}
|
|
55
|
+
for k, v in extraction.items():
|
|
56
|
+
if v is None or v == "" or v == [] or v == {}:
|
|
57
|
+
out[k] = 0.1
|
|
58
|
+
elif isinstance(v, (int, float)):
|
|
59
|
+
out[k] = max(0.0, min(1.0, base))
|
|
60
|
+
elif isinstance(v, str):
|
|
61
|
+
out[k] = max(0.0, min(1.0, base + 0.05))
|
|
62
|
+
elif isinstance(v, list):
|
|
63
|
+
out[k] = max(0.0, min(1.0, base - 0.05)) # harder
|
|
64
|
+
elif isinstance(v, dict):
|
|
65
|
+
out[k] = max(0.0, min(1.0, base - 0.05))
|
|
66
|
+
else:
|
|
67
|
+
out[k] = base
|
|
68
|
+
return out
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _llm_self_assess(doc: Document, backend: Backend) -> dict[str, float]:
|
|
72
|
+
snippet = json.dumps(doc.extraction or {}, indent=2)
|
|
73
|
+
sample_text = (doc.raw_text or "")[:4000]
|
|
74
|
+
req = CompletionRequest(
|
|
75
|
+
messages=[
|
|
76
|
+
Message(
|
|
77
|
+
role="system",
|
|
78
|
+
content=(
|
|
79
|
+
"You rate extraction confidence per field. "
|
|
80
|
+
"Return JSON: {\"field_name\": 0.0-1.0, ...} "
|
|
81
|
+
"where 1.0 = definitely correct, 0.0 = definitely wrong or absent."
|
|
82
|
+
),
|
|
83
|
+
),
|
|
84
|
+
Message(
|
|
85
|
+
role="user",
|
|
86
|
+
content=(
|
|
87
|
+
f"SOURCE DOCUMENT (truncated):\n{sample_text}\n\n"
|
|
88
|
+
f"EXTRACTION TO RATE:\n{snippet}"
|
|
89
|
+
),
|
|
90
|
+
),
|
|
91
|
+
],
|
|
92
|
+
json_mode=True,
|
|
93
|
+
temperature=0.0,
|
|
94
|
+
)
|
|
95
|
+
raw = backend.complete(req)
|
|
96
|
+
try:
|
|
97
|
+
return {k: float(v) for k, v in json.loads(raw).items()}
|
|
98
|
+
except Exception as e: # noqa: BLE001
|
|
99
|
+
log.debug("llm self-assess parse failed: %s", e)
|
|
100
|
+
return {}
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def assess_confidence(
|
|
104
|
+
doc: Document,
|
|
105
|
+
backend: Backend | None = None,
|
|
106
|
+
use_llm: bool = False,
|
|
107
|
+
policy: PolicyConfig | None = None,
|
|
108
|
+
) -> Document:
|
|
109
|
+
"""Attach doc.confidence (per-field dict of floats in 0..1).
|
|
110
|
+
|
|
111
|
+
If a `policy` is provided (an `idp.rl.PolicyConfig`), apply per-field
|
|
112
|
+
penalties so that fields humans have corrected frequently get
|
|
113
|
+
lower confidence scores and surface to HITL review.
|
|
114
|
+
"""
|
|
115
|
+
conf = _heuristic_confidence(doc)
|
|
116
|
+
if use_llm and backend is not None:
|
|
117
|
+
try:
|
|
118
|
+
llm_conf = _llm_self_assess(doc, backend)
|
|
119
|
+
# blend: 70% heuristic (more honest), 30% self-assess
|
|
120
|
+
conf = {k: 0.7 * conf.get(k, 0.5) + 0.3 * llm_conf.get(k, 0.5) for k in conf}
|
|
121
|
+
except Exception as e: # noqa: BLE001
|
|
122
|
+
doc.errors.append(f"assess_failed: {e}")
|
|
123
|
+
if policy is not None:
|
|
124
|
+
try:
|
|
125
|
+
from idp.rl.policy import policy_to_penalised_confidence
|
|
126
|
+
conf = policy_to_penalised_confidence(conf, policy)
|
|
127
|
+
except Exception as e: # noqa: BLE001
|
|
128
|
+
doc.errors.append(f"policy_apply_failed: {e}")
|
|
129
|
+
doc.confidence = conf
|
|
130
|
+
return doc
|
idp/auth/__init__.py
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
# py-idp: general-purpose, AI-enabled Intelligent Document Processing.
|
|
2
|
+
# Copyright (c) 2026 Royce.
|
|
3
|
+
#
|
|
4
|
+
# Licensed under the GNU Affero General Public License v3.0 or later (AGPL-3.0-or-later)
|
|
5
|
+
# with the following addition: a commercial license is also available for organizations
|
|
6
|
+
# that wish to embed py-idp in proprietary products / hosted SaaS without the AGPL
|
|
7
|
+
# copyleft obligations. See LICENSE and LICENSE-COMMERCIAL at the repo root, or
|
|
8
|
+
# contact <royce-license-placeholder@protonmail.com> for terms.
|
|
9
|
+
#
|
|
10
|
+
# This Source Code Form is subject to the terms of the AGPL-3.0-or-later.
|
|
11
|
+
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
12
|
+
|
|
13
|
+
from idp.auth.keys import (
|
|
14
|
+
AuthContext,
|
|
15
|
+
hash_key,
|
|
16
|
+
make_default_key,
|
|
17
|
+
require_api_key,
|
|
18
|
+
verify,
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
__all__ = ["AuthContext", "hash_key", "make_default_key", "require_api_key", "verify"]
|