ndi-sdk 0.7.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
ndi_sdk/__init__.py ADDED
@@ -0,0 +1,108 @@
1
+ """Python SDK for NDI (Nace Document Intelligence).
2
+
3
+ Two API surfaces, one client. The platform API (``/v1``) is where workspaces,
4
+ ingestion, retrieval, and the stateless document operations live; the legacy
5
+ product API (``/api/v1``) is the older one-shot surface, kept whole behind
6
+ ``client.legacy``.
7
+
8
+ from ndi_sdk import NdiClient, UrlSource
9
+
10
+ with NdiClient(api_key="ndi_sk_…") as client:
11
+ job = client.documents.extract(
12
+ UrlSource(url="https://example.com/invoice.pdf", file_name="invoice.pdf"),
13
+ json_schema={"type": "object", "properties": {"total": {"type": "number"}}},
14
+ )
15
+ result = client.jobs.wait(job.job_id).result
16
+
17
+ Every slow method returns a :class:`~ndi_sdk.models.jobs.Job` rather than
18
+ blocking, because the underlying work is a Temporal workflow that can outlive any
19
+ HTTP request. Wait for one with ``client.jobs.wait(job_id)``, follow it with
20
+ ``client.jobs.events(job_id)``, or hand a ``wait_seconds=`` to the call itself
21
+ when the work is short enough to finish inline.
22
+
23
+ Wire models here are hand-written copies of the server's, so this package depends
24
+ only on ``httpx`` and ``pydantic`` and imports nothing else from the monorepo.
25
+ Responses allow unknown fields and unknown enum members (see
26
+ :mod:`ndi_sdk._base`), so a server-side additive change never breaks a client.
27
+ """
28
+
29
+ from ndi_sdk import models
30
+ from ndi_sdk._transport import RetryPolicy
31
+ from ndi_sdk.client import DEFAULT_BASE_URL, AsyncNdiClient, NdiClient, UploadAndIngestResult
32
+ from ndi_sdk.errors import (
33
+ AuthenticationError,
34
+ ConflictError,
35
+ ErrorBody,
36
+ ErrorCode,
37
+ InvalidRequestError,
38
+ JobFailedError,
39
+ JobTimeoutError,
40
+ NdiConnectionError,
41
+ NdiError,
42
+ NdiStatusError,
43
+ NdiTimeoutError,
44
+ NotFoundError,
45
+ NotImplementedByServerError,
46
+ PermissionDeniedError,
47
+ QuotaExceededError,
48
+ RateLimitError,
49
+ ResultExpiredError,
50
+ ServerError,
51
+ SyncWaitTimeoutError,
52
+ )
53
+ from ndi_sdk.models.common import (
54
+ DocumentSource,
55
+ IngestionStatus,
56
+ JobKind,
57
+ JobStatus,
58
+ Page,
59
+ ParseResultSource,
60
+ UploadSource,
61
+ UrlSource,
62
+ WorkspaceFileSource,
63
+ )
64
+ from ndi_sdk.models.jobs import Job, JobEvent
65
+ from ndi_sdk.models.workspaces import Workspace
66
+
67
+ __version__ = "0.7.0"
68
+
69
+ __all__ = [
70
+ "DEFAULT_BASE_URL",
71
+ "AsyncNdiClient",
72
+ "AuthenticationError",
73
+ "ConflictError",
74
+ "DocumentSource",
75
+ "ErrorBody",
76
+ "ErrorCode",
77
+ "IngestionStatus",
78
+ "InvalidRequestError",
79
+ "Job",
80
+ "JobEvent",
81
+ "JobFailedError",
82
+ "JobKind",
83
+ "JobStatus",
84
+ "JobTimeoutError",
85
+ "NdiClient",
86
+ "NdiConnectionError",
87
+ "NdiError",
88
+ "NdiStatusError",
89
+ "NdiTimeoutError",
90
+ "NotFoundError",
91
+ "NotImplementedByServerError",
92
+ "Page",
93
+ "ParseResultSource",
94
+ "PermissionDeniedError",
95
+ "QuotaExceededError",
96
+ "RateLimitError",
97
+ "ResultExpiredError",
98
+ "RetryPolicy",
99
+ "ServerError",
100
+ "SyncWaitTimeoutError",
101
+ "UploadAndIngestResult",
102
+ "UploadSource",
103
+ "UrlSource",
104
+ "Workspace",
105
+ "WorkspaceFileSource",
106
+ "__version__",
107
+ "models",
108
+ ]
ndi_sdk/_base.py ADDED
@@ -0,0 +1,47 @@
1
+ """Base behaviour shared by every wire model in this SDK.
2
+
3
+ Two deliberate differences from the schemas the server validates against:
4
+
5
+ * ``extra="allow"`` — the server may add a field to a response at any time, and a
6
+ client that rejects unknown fields turns every additive change into an outage.
7
+ * :class:`OpenEnum` — the same reasoning for enum members. NDI documents its
8
+ status, kind, and error-code enums as open (spec 12 §9), so an unrecognised
9
+ member arrives as a usable string instead of a validation error.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from enum import StrEnum
15
+ from typing import Any
16
+
17
+ from pydantic import BaseModel, ConfigDict
18
+
19
+
20
+ class NdiModel(BaseModel):
21
+ """Base for request and response models.
22
+
23
+ ``serialize_by_alias`` matters for the handful of fields whose wire name is a
24
+ Python keyword or shadows a ``BaseModel`` attribute (``schema``, ``class``):
25
+ the alias is the contract, the attribute name is an implementation detail.
26
+ """
27
+
28
+ model_config = ConfigDict(extra="allow", populate_by_name=True, serialize_by_alias=True)
29
+
30
+
31
+ class OpenEnum(StrEnum):
32
+ """A string enum that keeps values added to the API after this SDK shipped.
33
+
34
+ An unknown value becomes a member carrying that string, so comparisons
35
+ against known members work as usual and ``str(value)`` is always the wire
36
+ value. Compare with ``==`` rather than ``is``: two pseudo-members built from
37
+ the same string are equal but not identical.
38
+ """
39
+
40
+ @classmethod
41
+ def _missing_(cls, value: object) -> Any:
42
+ if not isinstance(value, str):
43
+ return None
44
+ unknown = str.__new__(cls, value)
45
+ unknown._name_ = value
46
+ unknown._value_ = value
47
+ return unknown
ndi_sdk/_files.py ADDED
@@ -0,0 +1,56 @@
1
+ """Turning what a caller has into what a multipart upload needs.
2
+
3
+ Upload methods accept a path, raw bytes, or an open binary file, because those
4
+ are the three things callers actually hold. Everything converges on the
5
+ ``(file_name, bytes, content_type)`` triple ``httpx`` wants for its ``files=``
6
+ argument.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import mimetypes
12
+ from os import PathLike
13
+ from pathlib import Path
14
+ from typing import IO, Any, TypeAlias
15
+
16
+ #: Anything an upload method accepts as the document itself.
17
+ FileContent: TypeAlias = "str | PathLike[str] | bytes | IO[bytes]"
18
+
19
+ _DEFAULT_CONTENT_TYPE = "application/octet-stream"
20
+
21
+
22
+ def read_file_content(content: FileContent, *, file_name: str | None = None) -> tuple[str, bytes, str]:
23
+ """Normalise an upload argument into ``(file_name, data, content_type)``.
24
+
25
+ Raises:
26
+ ValueError: When raw bytes or a nameless stream is passed without a
27
+ ``file_name``. NDI infers the document format from that name, so
28
+ guessing one here would silently pick the wrong parser.
29
+ """
30
+ if isinstance(content, bytes | bytearray):
31
+ if not file_name:
32
+ raise ValueError("file_name is required when uploading raw bytes: NDI infers the format from it")
33
+ return file_name, bytes(content), guess_content_type(file_name)
34
+
35
+ if isinstance(content, str | PathLike):
36
+ path = Path(content)
37
+ name = file_name or path.name
38
+ return name, path.read_bytes(), guess_content_type(name)
39
+
40
+ data = content.read()
41
+ if isinstance(data, str): # a text-mode handle: the bytes are already lossy
42
+ raise ValueError("open the file in binary mode ('rb'): a text handle re-encodes the document")
43
+ name = file_name or _stream_name(content)
44
+ if not name:
45
+ raise ValueError("file_name is required when uploading a stream with no name")
46
+ return name, data, guess_content_type(name)
47
+
48
+
49
+ def guess_content_type(file_name: str) -> str:
50
+ guessed, _ = mimetypes.guess_type(file_name)
51
+ return guessed or _DEFAULT_CONTENT_TYPE
52
+
53
+
54
+ def _stream_name(stream: Any) -> str | None:
55
+ name = getattr(stream, "name", None)
56
+ return Path(name).name if isinstance(name, str) else None
ndi_sdk/_pagination.py ADDED
@@ -0,0 +1,58 @@
1
+ """Cursor pagination as iterators.
2
+
3
+ Every ``/v1`` list method returns one :class:`~ndi_sdk.models.common.Page` with a
4
+ ``next_cursor``. The paired ``iter_*`` methods on each resource wrap that in an
5
+ iterator so a caller who wants every row does not write the cursor loop by hand.
6
+
7
+ Pages are still returned as-is by the plain ``list`` methods: ``total_count`` and
8
+ ``next_cursor`` are real answers, and an iterator that hides them would make the
9
+ common "how many are there" question unanswerable without draining the list.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from collections.abc import AsyncIterator, Awaitable, Callable, Iterator
15
+ from typing import TypeVar
16
+
17
+ from ndi_sdk.models.common import Page
18
+
19
+ ItemT = TypeVar("ItemT")
20
+
21
+ #: Fetches one page given the previous page's ``next_cursor`` (``None`` = first).
22
+ AsyncPageFetcher = Callable[[str | None], Awaitable[Page[ItemT]]]
23
+ SyncPageFetcher = Callable[[str | None], Page[ItemT]]
24
+
25
+
26
+ async def aiter_pages(fetch: AsyncPageFetcher[ItemT]) -> AsyncIterator[Page[ItemT]]:
27
+ """Yield pages until the server stops handing back a cursor."""
28
+ cursor: str | None = None
29
+ while True:
30
+ page = await fetch(cursor)
31
+ yield page
32
+ if page.next_cursor is None:
33
+ return
34
+ cursor = page.next_cursor
35
+
36
+
37
+ async def aiter_items(fetch: AsyncPageFetcher[ItemT]) -> AsyncIterator[ItemT]:
38
+ """Yield every item across every page."""
39
+ async for page in aiter_pages(fetch):
40
+ for item in page.items:
41
+ yield item
42
+
43
+
44
+ def iter_pages(fetch: SyncPageFetcher[ItemT]) -> Iterator[Page[ItemT]]:
45
+ """Yield pages until the server stops handing back a cursor."""
46
+ cursor: str | None = None
47
+ while True:
48
+ page = fetch(cursor)
49
+ yield page
50
+ if page.next_cursor is None:
51
+ return
52
+ cursor = page.next_cursor
53
+
54
+
55
+ def iter_items(fetch: SyncPageFetcher[ItemT]) -> Iterator[ItemT]:
56
+ """Yield every item across every page."""
57
+ for page in iter_pages(fetch):
58
+ yield from page.items