ndi-sdk 0.7.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ndi_sdk/__init__.py +108 -0
- ndi_sdk/_base.py +47 -0
- ndi_sdk/_files.py +56 -0
- ndi_sdk/_pagination.py +58 -0
- ndi_sdk/_transport.py +506 -0
- ndi_sdk/client.py +286 -0
- ndi_sdk/errors.py +279 -0
- ndi_sdk/models/__init__.py +335 -0
- ndi_sdk/models/common.py +450 -0
- ndi_sdk/models/document_ops.py +313 -0
- ndi_sdk/models/domains.py +210 -0
- ndi_sdk/models/files.py +214 -0
- ndi_sdk/models/jobs.py +550 -0
- ndi_sdk/models/legacy.py +429 -0
- ndi_sdk/models/tools.py +473 -0
- ndi_sdk/models/workspaces.py +124 -0
- ndi_sdk/resources/__init__.py +37 -0
- ndi_sdk/resources/documents.py +568 -0
- ndi_sdk/resources/domains.py +299 -0
- ndi_sdk/resources/files.py +615 -0
- ndi_sdk/resources/ingestion.py +196 -0
- ndi_sdk/resources/jobs.py +403 -0
- ndi_sdk/resources/legacy.py +581 -0
- ndi_sdk/resources/search.py +163 -0
- ndi_sdk/resources/tools.py +449 -0
- ndi_sdk/resources/workspaces.py +239 -0
- ndi_sdk-0.7.0.dist-info/METADATA +584 -0
- ndi_sdk-0.7.0.dist-info/RECORD +30 -0
- ndi_sdk-0.7.0.dist-info/WHEEL +4 -0
- ndi_sdk-0.7.0.dist-info/licenses/LICENSE +202 -0
ndi_sdk/__init__.py
ADDED
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
"""Python SDK for NDI (Nace Document Intelligence).
|
|
2
|
+
|
|
3
|
+
Two API surfaces, one client. The platform API (``/v1``) is where workspaces,
|
|
4
|
+
ingestion, retrieval, and the stateless document operations live; the legacy
|
|
5
|
+
product API (``/api/v1``) is the older one-shot surface, kept whole behind
|
|
6
|
+
``client.legacy``.
|
|
7
|
+
|
|
8
|
+
from ndi_sdk import NdiClient, UrlSource
|
|
9
|
+
|
|
10
|
+
with NdiClient(api_key="ndi_sk_…") as client:
|
|
11
|
+
job = client.documents.extract(
|
|
12
|
+
UrlSource(url="https://example.com/invoice.pdf", file_name="invoice.pdf"),
|
|
13
|
+
json_schema={"type": "object", "properties": {"total": {"type": "number"}}},
|
|
14
|
+
)
|
|
15
|
+
result = client.jobs.wait(job.job_id).result
|
|
16
|
+
|
|
17
|
+
Every slow method returns a :class:`~ndi_sdk.models.jobs.Job` rather than
|
|
18
|
+
blocking, because the underlying work is a Temporal workflow that can outlive any
|
|
19
|
+
HTTP request. Wait for one with ``client.jobs.wait(job_id)``, follow it with
|
|
20
|
+
``client.jobs.events(job_id)``, or hand a ``wait_seconds=`` to the call itself
|
|
21
|
+
when the work is short enough to finish inline.
|
|
22
|
+
|
|
23
|
+
Wire models here are hand-written copies of the server's, so this package depends
|
|
24
|
+
only on ``httpx`` and ``pydantic`` and imports nothing else from the monorepo.
|
|
25
|
+
Responses allow unknown fields and unknown enum members (see
|
|
26
|
+
:mod:`ndi_sdk._base`), so a server-side additive change never breaks a client.
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
from ndi_sdk import models
|
|
30
|
+
from ndi_sdk._transport import RetryPolicy
|
|
31
|
+
from ndi_sdk.client import DEFAULT_BASE_URL, AsyncNdiClient, NdiClient, UploadAndIngestResult
|
|
32
|
+
from ndi_sdk.errors import (
|
|
33
|
+
AuthenticationError,
|
|
34
|
+
ConflictError,
|
|
35
|
+
ErrorBody,
|
|
36
|
+
ErrorCode,
|
|
37
|
+
InvalidRequestError,
|
|
38
|
+
JobFailedError,
|
|
39
|
+
JobTimeoutError,
|
|
40
|
+
NdiConnectionError,
|
|
41
|
+
NdiError,
|
|
42
|
+
NdiStatusError,
|
|
43
|
+
NdiTimeoutError,
|
|
44
|
+
NotFoundError,
|
|
45
|
+
NotImplementedByServerError,
|
|
46
|
+
PermissionDeniedError,
|
|
47
|
+
QuotaExceededError,
|
|
48
|
+
RateLimitError,
|
|
49
|
+
ResultExpiredError,
|
|
50
|
+
ServerError,
|
|
51
|
+
SyncWaitTimeoutError,
|
|
52
|
+
)
|
|
53
|
+
from ndi_sdk.models.common import (
|
|
54
|
+
DocumentSource,
|
|
55
|
+
IngestionStatus,
|
|
56
|
+
JobKind,
|
|
57
|
+
JobStatus,
|
|
58
|
+
Page,
|
|
59
|
+
ParseResultSource,
|
|
60
|
+
UploadSource,
|
|
61
|
+
UrlSource,
|
|
62
|
+
WorkspaceFileSource,
|
|
63
|
+
)
|
|
64
|
+
from ndi_sdk.models.jobs import Job, JobEvent
|
|
65
|
+
from ndi_sdk.models.workspaces import Workspace
|
|
66
|
+
|
|
67
|
+
__version__ = "0.7.0"
|
|
68
|
+
|
|
69
|
+
__all__ = [
|
|
70
|
+
"DEFAULT_BASE_URL",
|
|
71
|
+
"AsyncNdiClient",
|
|
72
|
+
"AuthenticationError",
|
|
73
|
+
"ConflictError",
|
|
74
|
+
"DocumentSource",
|
|
75
|
+
"ErrorBody",
|
|
76
|
+
"ErrorCode",
|
|
77
|
+
"IngestionStatus",
|
|
78
|
+
"InvalidRequestError",
|
|
79
|
+
"Job",
|
|
80
|
+
"JobEvent",
|
|
81
|
+
"JobFailedError",
|
|
82
|
+
"JobKind",
|
|
83
|
+
"JobStatus",
|
|
84
|
+
"JobTimeoutError",
|
|
85
|
+
"NdiClient",
|
|
86
|
+
"NdiConnectionError",
|
|
87
|
+
"NdiError",
|
|
88
|
+
"NdiStatusError",
|
|
89
|
+
"NdiTimeoutError",
|
|
90
|
+
"NotFoundError",
|
|
91
|
+
"NotImplementedByServerError",
|
|
92
|
+
"Page",
|
|
93
|
+
"ParseResultSource",
|
|
94
|
+
"PermissionDeniedError",
|
|
95
|
+
"QuotaExceededError",
|
|
96
|
+
"RateLimitError",
|
|
97
|
+
"ResultExpiredError",
|
|
98
|
+
"RetryPolicy",
|
|
99
|
+
"ServerError",
|
|
100
|
+
"SyncWaitTimeoutError",
|
|
101
|
+
"UploadAndIngestResult",
|
|
102
|
+
"UploadSource",
|
|
103
|
+
"UrlSource",
|
|
104
|
+
"Workspace",
|
|
105
|
+
"WorkspaceFileSource",
|
|
106
|
+
"__version__",
|
|
107
|
+
"models",
|
|
108
|
+
]
|
ndi_sdk/_base.py
ADDED
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""Base behaviour shared by every wire model in this SDK.
|
|
2
|
+
|
|
3
|
+
Two deliberate differences from the schemas the server validates against:
|
|
4
|
+
|
|
5
|
+
* ``extra="allow"`` — the server may add a field to a response at any time, and a
|
|
6
|
+
client that rejects unknown fields turns every additive change into an outage.
|
|
7
|
+
* :class:`OpenEnum` — the same reasoning for enum members. NDI documents its
|
|
8
|
+
status, kind, and error-code enums as open (spec 12 §9), so an unrecognised
|
|
9
|
+
member arrives as a usable string instead of a validation error.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from enum import StrEnum
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
from pydantic import BaseModel, ConfigDict
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class NdiModel(BaseModel):
|
|
21
|
+
"""Base for request and response models.
|
|
22
|
+
|
|
23
|
+
``serialize_by_alias`` matters for the handful of fields whose wire name is a
|
|
24
|
+
Python keyword or shadows a ``BaseModel`` attribute (``schema``, ``class``):
|
|
25
|
+
the alias is the contract, the attribute name is an implementation detail.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
model_config = ConfigDict(extra="allow", populate_by_name=True, serialize_by_alias=True)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class OpenEnum(StrEnum):
|
|
32
|
+
"""A string enum that keeps values added to the API after this SDK shipped.
|
|
33
|
+
|
|
34
|
+
An unknown value becomes a member carrying that string, so comparisons
|
|
35
|
+
against known members work as usual and ``str(value)`` is always the wire
|
|
36
|
+
value. Compare with ``==`` rather than ``is``: two pseudo-members built from
|
|
37
|
+
the same string are equal but not identical.
|
|
38
|
+
"""
|
|
39
|
+
|
|
40
|
+
@classmethod
|
|
41
|
+
def _missing_(cls, value: object) -> Any:
|
|
42
|
+
if not isinstance(value, str):
|
|
43
|
+
return None
|
|
44
|
+
unknown = str.__new__(cls, value)
|
|
45
|
+
unknown._name_ = value
|
|
46
|
+
unknown._value_ = value
|
|
47
|
+
return unknown
|
ndi_sdk/_files.py
ADDED
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
"""Turning what a caller has into what a multipart upload needs.
|
|
2
|
+
|
|
3
|
+
Upload methods accept a path, raw bytes, or an open binary file, because those
|
|
4
|
+
are the three things callers actually hold. Everything converges on the
|
|
5
|
+
``(file_name, bytes, content_type)`` triple ``httpx`` wants for its ``files=``
|
|
6
|
+
argument.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import mimetypes
|
|
12
|
+
from os import PathLike
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from typing import IO, Any, TypeAlias
|
|
15
|
+
|
|
16
|
+
#: Anything an upload method accepts as the document itself.
|
|
17
|
+
FileContent: TypeAlias = "str | PathLike[str] | bytes | IO[bytes]"
|
|
18
|
+
|
|
19
|
+
_DEFAULT_CONTENT_TYPE = "application/octet-stream"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def read_file_content(content: FileContent, *, file_name: str | None = None) -> tuple[str, bytes, str]:
|
|
23
|
+
"""Normalise an upload argument into ``(file_name, data, content_type)``.
|
|
24
|
+
|
|
25
|
+
Raises:
|
|
26
|
+
ValueError: When raw bytes or a nameless stream is passed without a
|
|
27
|
+
``file_name``. NDI infers the document format from that name, so
|
|
28
|
+
guessing one here would silently pick the wrong parser.
|
|
29
|
+
"""
|
|
30
|
+
if isinstance(content, bytes | bytearray):
|
|
31
|
+
if not file_name:
|
|
32
|
+
raise ValueError("file_name is required when uploading raw bytes: NDI infers the format from it")
|
|
33
|
+
return file_name, bytes(content), guess_content_type(file_name)
|
|
34
|
+
|
|
35
|
+
if isinstance(content, str | PathLike):
|
|
36
|
+
path = Path(content)
|
|
37
|
+
name = file_name or path.name
|
|
38
|
+
return name, path.read_bytes(), guess_content_type(name)
|
|
39
|
+
|
|
40
|
+
data = content.read()
|
|
41
|
+
if isinstance(data, str): # a text-mode handle: the bytes are already lossy
|
|
42
|
+
raise ValueError("open the file in binary mode ('rb'): a text handle re-encodes the document")
|
|
43
|
+
name = file_name or _stream_name(content)
|
|
44
|
+
if not name:
|
|
45
|
+
raise ValueError("file_name is required when uploading a stream with no name")
|
|
46
|
+
return name, data, guess_content_type(name)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def guess_content_type(file_name: str) -> str:
|
|
50
|
+
guessed, _ = mimetypes.guess_type(file_name)
|
|
51
|
+
return guessed or _DEFAULT_CONTENT_TYPE
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _stream_name(stream: Any) -> str | None:
|
|
55
|
+
name = getattr(stream, "name", None)
|
|
56
|
+
return Path(name).name if isinstance(name, str) else None
|
ndi_sdk/_pagination.py
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""Cursor pagination as iterators.
|
|
2
|
+
|
|
3
|
+
Every ``/v1`` list method returns one :class:`~ndi_sdk.models.common.Page` with a
|
|
4
|
+
``next_cursor``. The paired ``iter_*`` methods on each resource wrap that in an
|
|
5
|
+
iterator so a caller who wants every row does not write the cursor loop by hand.
|
|
6
|
+
|
|
7
|
+
Pages are still returned as-is by the plain ``list`` methods: ``total_count`` and
|
|
8
|
+
``next_cursor`` are real answers, and an iterator that hides them would make the
|
|
9
|
+
common "how many are there" question unanswerable without draining the list.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from collections.abc import AsyncIterator, Awaitable, Callable, Iterator
|
|
15
|
+
from typing import TypeVar
|
|
16
|
+
|
|
17
|
+
from ndi_sdk.models.common import Page
|
|
18
|
+
|
|
19
|
+
ItemT = TypeVar("ItemT")
|
|
20
|
+
|
|
21
|
+
#: Fetches one page given the previous page's ``next_cursor`` (``None`` = first).
|
|
22
|
+
AsyncPageFetcher = Callable[[str | None], Awaitable[Page[ItemT]]]
|
|
23
|
+
SyncPageFetcher = Callable[[str | None], Page[ItemT]]
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
async def aiter_pages(fetch: AsyncPageFetcher[ItemT]) -> AsyncIterator[Page[ItemT]]:
|
|
27
|
+
"""Yield pages until the server stops handing back a cursor."""
|
|
28
|
+
cursor: str | None = None
|
|
29
|
+
while True:
|
|
30
|
+
page = await fetch(cursor)
|
|
31
|
+
yield page
|
|
32
|
+
if page.next_cursor is None:
|
|
33
|
+
return
|
|
34
|
+
cursor = page.next_cursor
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
async def aiter_items(fetch: AsyncPageFetcher[ItemT]) -> AsyncIterator[ItemT]:
|
|
38
|
+
"""Yield every item across every page."""
|
|
39
|
+
async for page in aiter_pages(fetch):
|
|
40
|
+
for item in page.items:
|
|
41
|
+
yield item
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def iter_pages(fetch: SyncPageFetcher[ItemT]) -> Iterator[Page[ItemT]]:
|
|
45
|
+
"""Yield pages until the server stops handing back a cursor."""
|
|
46
|
+
cursor: str | None = None
|
|
47
|
+
while True:
|
|
48
|
+
page = fetch(cursor)
|
|
49
|
+
yield page
|
|
50
|
+
if page.next_cursor is None:
|
|
51
|
+
return
|
|
52
|
+
cursor = page.next_cursor
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def iter_items(fetch: SyncPageFetcher[ItemT]) -> Iterator[ItemT]:
|
|
56
|
+
"""Yield every item across every page."""
|
|
57
|
+
for page in iter_pages(fetch):
|
|
58
|
+
yield from page.items
|