unstructured-transform-client 0.18.17__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. unstructured_transform_client/__init__.py +28 -0
  2. unstructured_transform_client/_client.py +420 -0
  3. unstructured_transform_client/_files.py +72 -0
  4. unstructured_transform_client/_generated/__init__.py +100 -0
  5. unstructured_transform_client/_generated/api/__init__.py +8 -0
  6. unstructured_transform_client/_generated/api/extract_api.py +351 -0
  7. unstructured_transform_client/_generated/api/jobs_api.py +1212 -0
  8. unstructured_transform_client/_generated/api/parse_api.py +461 -0
  9. unstructured_transform_client/_generated/api/upload_api.py +868 -0
  10. unstructured_transform_client/_generated/api_client.py +833 -0
  11. unstructured_transform_client/_generated/api_response.py +21 -0
  12. unstructured_transform_client/_generated/configuration.py +688 -0
  13. unstructured_transform_client/_generated/exceptions.py +218 -0
  14. unstructured_transform_client/_generated/models/__init__.py +39 -0
  15. unstructured_transform_client/_generated/models/citation.py +95 -0
  16. unstructured_transform_client/_generated/models/document_metadata.py +93 -0
  17. unstructured_transform_client/_generated/models/element.py +103 -0
  18. unstructured_transform_client/_generated/models/element_metadata.py +107 -0
  19. unstructured_transform_client/_generated/models/error.py +91 -0
  20. unstructured_transform_client/_generated/models/error_code.py +58 -0
  21. unstructured_transform_client/_generated/models/extract_request.py +92 -0
  22. unstructured_transform_client/_generated/models/extraction_result.py +107 -0
  23. unstructured_transform_client/_generated/models/field_metadata.py +92 -0
  24. unstructured_transform_client/_generated/models/job_accepted.py +134 -0
  25. unstructured_transform_client/_generated/models/job_operation.py +37 -0
  26. unstructured_transform_client/_generated/models/job_page.py +102 -0
  27. unstructured_transform_client/_generated/models/job_result.py +130 -0
  28. unstructured_transform_client/_generated/models/job_result_state.py +38 -0
  29. unstructured_transform_client/_generated/models/job_status.py +41 -0
  30. unstructured_transform_client/_generated/models/job_summary.py +141 -0
  31. unstructured_transform_client/_generated/models/locator.py +99 -0
  32. unstructured_transform_client/_generated/models/output_format.py +37 -0
  33. unstructured_transform_client/_generated/models/parse_result.py +173 -0
  34. unstructured_transform_client/_generated/models/source_file.py +115 -0
  35. unstructured_transform_client/_generated/models/transform_status.py +38 -0
  36. unstructured_transform_client/_generated/models/transform_warning.py +90 -0
  37. unstructured_transform_client/_generated/models/upload_result.py +93 -0
  38. unstructured_transform_client/_generated/py.typed +0 -0
  39. unstructured_transform_client/_generated/rest.py +334 -0
  40. unstructured_transform_client/_host_headers.py +129 -0
  41. unstructured_transform_client/_multipart.py +100 -0
  42. unstructured_transform_client/_retry.py +205 -0
  43. unstructured_transform_client/_sse.py +187 -0
  44. unstructured_transform_client/_version.py +17 -0
  45. unstructured_transform_client-0.18.17.dist-info/METADATA +201 -0
  46. unstructured_transform_client-0.18.17.dist-info/RECORD +48 -0
  47. unstructured_transform_client-0.18.17.dist-info/WHEEL +4 -0
  48. unstructured_transform_client-0.18.17.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,28 @@
1
+ """Python client for the Unstructured Transform v2 API.
2
+
3
+ from unstructured_transform_client import TransformClient
4
+
5
+ with TransformClient(api_key="...") as client:
6
+ result = client.parse.run(input=open("invoice.pdf", "rb"))
7
+ print(result.markdown)
8
+
9
+ Everything importable from here is public and covered by the version policy.
10
+ A module with a leading underscore is not: `_generated` in particular is
11
+ produced from openapi.yaml at build time and its layout is the generator's
12
+ business, so importing from it directly will break without notice.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ from ._client import DEFAULT_SERVER_URL, TransformClient
18
+ from ._retry import RetryConfig
19
+ from ._sse import JobEvent
20
+ from ._version import __version__
21
+
22
+ __all__ = [
23
+ "DEFAULT_SERVER_URL",
24
+ "JobEvent",
25
+ "RetryConfig",
26
+ "TransformClient",
27
+ "__version__",
28
+ ]
@@ -0,0 +1,420 @@
1
+ """The public surface: `client.parse.run(...)`, `client.extract.run(...)`.
2
+
3
+ WHY A FACADE AT ALL
4
+ ───────────────────
5
+ openapi-generator produces `ParseApi(api_client).parse_run(...)` across four
6
+ unrelated Api classes. That is not the surface openapi.yaml says we ship — the
7
+ spec declares `tags` + `x-sdk-method-name` per operation precisely so the call
8
+ reads `client.parse.run(document)` (decision N1, API v2 Surface Proposal), and
9
+ no generator reads those. This module is where that declaration becomes real.
10
+
11
+ It also supplies three things the generator cannot:
12
+
13
+ 1. `extract.from_document(...)`. POST /api/v2/extract declares BOTH an
14
+ application/json body (ExtractRequest, the parse_id flow) and a
15
+ multipart/form-data body (ExtractDocumentRequest, the raw-document flow).
16
+ openapi-generator emits only the first and drops the second without a
17
+ warning — ExtractDocumentRequest is absent from the generated models
18
+ entirely. One of the three launch flows, silently missing, in both Python
19
+ and TypeScript. It is reconstructed here over the generated transport.
20
+ 2. `jobs.stream(...)`. The SSE progress stream the service really serves.
21
+ 3. `schema=` instead of `var_schema=`. The generator renames the field to dodge
22
+ Pydantic's `BaseModel.schema()`; the wire name stays `schema` via an alias,
23
+ so only the Python signature is ugly. Callers should never see it.
24
+
25
+ The method names here are not free-form: `tests/test_facade_matches_the_spec.py`
26
+ reads openapi.yaml and asserts this module exposes exactly the calls the
27
+ contract declares. The spec stays the source of truth; this file is held to it.
28
+ """
29
+
30
+ from __future__ import annotations
31
+
32
+ import json
33
+ import os
34
+ from typing import Any, Iterator
35
+ from urllib.parse import quote, urlencode
36
+
37
+ from . import _host_headers
38
+ from ._generated.api.extract_api import ExtractApi
39
+ from ._generated.api.jobs_api import JobsApi
40
+ from ._generated.api.parse_api import ParseApi
41
+ from ._generated.api.upload_api import UploadApi
42
+ from ._generated.api_client import ApiClient
43
+ from ._generated.configuration import Configuration
44
+ from ._files import FileInput, as_file_part
45
+ from ._generated.models.extract_request import ExtractRequest
46
+ from ._multipart import post_multipart
47
+ from ._retry import RetryConfig, call_with_retries
48
+ from ._sse import JobEvent, stream_job_events
49
+
50
+ DEFAULT_SERVER_URL = "https://platform.unstructuredapp.io"
51
+
52
+ # Matches the service's own security scheme names; see openapi.yaml's
53
+ # `securitySchemes` and the generated Configuration.auth_settings.
54
+ _API_KEY_HEADER = "unstructured-api-key"
55
+ _API_KEY_ENV_VAR = "UNSTRUCTURED_API_KEY"
56
+
57
+
58
+ class TransformClient:
59
+ """A client for the Unstructured Transform v2 API.
60
+
61
+ Pass an API key or bearer token explicitly, or leave both out to read
62
+ `UNSTRUCTURED_API_KEY`. Both credential types are accepted by the service
63
+ and whichever you send is forwarded upstream verbatim, so an API key and a
64
+ bearer token are not interchangeable in what they authorise — they are two
65
+ different identities, and sending both would leave which one is in force up
66
+ to the server's precedence rules rather than your intent.
67
+ """
68
+
69
+ def __init__(
70
+ self,
71
+ *,
72
+ api_key: str | None = None,
73
+ bearer_token: str | None = None,
74
+ server_url: str = DEFAULT_SERVER_URL,
75
+ send_host_headers: bool | None = None,
76
+ user_agent_suffix: str | None = None,
77
+ timeout_seconds: float | None = None,
78
+ retries: RetryConfig | None = RetryConfig(),
79
+ ) -> None:
80
+ # An explicit credential always wins. A bearer token is intentionally
81
+ # independent from the API-key environment variable: callers who pass
82
+ # one have made an explicit authentication choice.
83
+ if api_key is None and bearer_token is None:
84
+ api_key = os.environ.get(_API_KEY_ENV_VAR) or None
85
+
86
+ if (api_key is None) == (bearer_token is None):
87
+ raise ValueError(
88
+ "pass exactly one of api_key or bearer_token, or set "
89
+ f"{_API_KEY_ENV_VAR} — they are different identities, not "
90
+ "interchangeable spellings of one credential"
91
+ )
92
+
93
+ self._server_url = server_url.rstrip("/")
94
+ self._timeout_seconds = timeout_seconds
95
+ self._retries = retries
96
+
97
+ configuration = Configuration(host=self._server_url)
98
+ if api_key is not None:
99
+ configuration.api_key["ApiKeyAuth"] = api_key
100
+ self._auth_headers = {_API_KEY_HEADER: api_key}
101
+ else:
102
+ configuration.access_token = bearer_token
103
+ self._auth_headers = {"Authorization": f"Bearer {bearer_token}"}
104
+ # urllib3 has its own Retry layer underneath the generated client. Keep
105
+ # retry decisions in this facade so non-idempotent submit safety and
106
+ # Retry-After caps are applied consistently.
107
+ configuration.retries = False
108
+
109
+ self._api_client = ApiClient(configuration)
110
+
111
+ if timeout_seconds is not None:
112
+ self._apply_default_timeout(self._api_client, timeout_seconds)
113
+ self._apply_retries(self._api_client, retries)
114
+
115
+ # Identification headers, per TRNSFRM-91. Set as ApiClient defaults so
116
+ # every generated call carries them without each namespace remembering
117
+ # to; the SSE path is not a generated call, so it merges the same dict
118
+ # itself rather than relying on that.
119
+ self._client_headers = _host_headers.build(
120
+ send_host_headers=send_host_headers,
121
+ user_agent_suffix=user_agent_suffix,
122
+ )
123
+ for name, value in self._client_headers.items():
124
+ self._api_client.set_default_header(name, value)
125
+
126
+ self.parse = _Parse(self)
127
+ self.extract = _Extract(self)
128
+ self.jobs = _Jobs(self)
129
+ self.upload = _Upload(self)
130
+
131
+ @staticmethod
132
+ def _apply_default_timeout(api_client: ApiClient, seconds: float) -> None:
133
+ """Make `timeout_seconds` apply to every request, not just some.
134
+
135
+ The generated code has no notion of a default timeout: `Configuration`
136
+ carries no `timeout` attribute, and the rest client honours only a
137
+ per-call `_request_timeout`. So an earlier version of this client
138
+ accepted `timeout_seconds` and then silently ignored it on all fourteen
139
+ generated operations — only the two hand-written paths, which build
140
+ their own requests, ever used it. A caller who asked for a 5-second
141
+ timeout could still wait forever on an ordinary parse.
142
+
143
+ Threading the argument through fourteen call sites would work and would
144
+ be fourteen places to forget it on the fifteenth. `call_api` is the
145
+ single point every generated operation funnels through, so the default
146
+ is injected there instead — and only when the caller has not passed one
147
+ explicitly, so a per-call override still wins.
148
+ """
149
+ original = api_client.call_api
150
+
151
+ def call_api(*args: Any, **kwargs: Any) -> Any:
152
+ # Not setdefault: the generated operations always pass the keyword,
153
+ # with None when the caller supplied nothing. So the check is on the
154
+ # VALUE being absent, not the key.
155
+ if kwargs.get("_request_timeout") is None:
156
+ kwargs["_request_timeout"] = seconds
157
+ return original(*args, **kwargs)
158
+
159
+ api_client.call_api = call_api # type: ignore[method-assign]
160
+
161
+ @staticmethod
162
+ def _apply_retries(api_client: ApiClient, retries: RetryConfig | None) -> None:
163
+ """Make retries cover generated operations through the same funnel.
164
+
165
+ Generated calls do not expose a retry hook. `call_api` is the one point
166
+ every generated request reaches, while `_sse.py` and `_multipart.py`
167
+ pass through this facade explicitly because they bypass generation.
168
+ """
169
+ original = api_client.call_api
170
+
171
+ def call_api(*args: Any, **kwargs: Any) -> Any:
172
+ method = args[0] if args else kwargs["method"]
173
+ return call_with_retries(
174
+ retries=retries,
175
+ method=method,
176
+ send=lambda: original(*args, **kwargs),
177
+ )
178
+
179
+ api_client.call_api = call_api # type: ignore[method-assign]
180
+
181
+ # ── plumbing shared with the namespaces ──────────────────────────────────
182
+
183
+ def _request_headers(self) -> dict[str, str]:
184
+ """Auth plus identification, for calls that bypass the generated APIs."""
185
+ return {**self._auth_headers, **self._client_headers}
186
+
187
+ def close(self) -> None:
188
+ """Release the underlying connection pool.
189
+
190
+ The generated `ApiClient` has no `close()` and its `__exit__` is a bare
191
+ `pass`, so there is nothing to delegate to — an earlier version called
192
+ `self._api_client.close()` and raised AttributeError on every
193
+ `with` block's exit. Nothing caught it until the built package was
194
+ driven against a real socket.
195
+
196
+ `clear()` on the pool manager is what actually frees the sockets.
197
+ Guarded, because the attribute belongs to generated code: a generator
198
+ change should not turn cleanup into a crash.
199
+ """
200
+ rest_client = getattr(self._api_client, "rest_client", None)
201
+ pool_manager = getattr(rest_client, "pool_manager", None)
202
+ if pool_manager is not None:
203
+ pool_manager.clear()
204
+
205
+ def __enter__(self) -> TransformClient:
206
+ return self
207
+
208
+ def __exit__(self, *exc: object) -> None:
209
+ self.close()
210
+
211
+
212
+ class _Namespace:
213
+ def __init__(self, client: TransformClient) -> None:
214
+ self._client = client
215
+
216
+
217
+ class _Parse(_Namespace):
218
+ def run(
219
+ self,
220
+ *,
221
+ input: FileInput | None = None,
222
+ file_id: str | None = None,
223
+ schema: dict[str, Any] | str | None = None,
224
+ prompt: str | None = None,
225
+ output: str | None = None,
226
+ include: list[str] | None = None,
227
+ profile: str | None = None,
228
+ wait_seconds: int | None = None,
229
+ ) -> Any:
230
+ """Parse a document. Supplying `schema` extracts in the same job.
231
+
232
+ `schema` is accepted as a dict and serialized here. The wire format is
233
+ a JSON *string* on this operation and a JSON *object* on extract.run —
234
+ an inconsistency in the contract that callers should not have to know,
235
+ so both take a dict.
236
+ """
237
+ if (input is None) == (file_id is None):
238
+ raise ValueError("pass exactly one of input or file_id")
239
+
240
+ return ParseApi(self._client._api_client).parse_run(
241
+ prefer=_prefer(wait_seconds),
242
+ include=include,
243
+ input=as_file_part(input) if input is not None else None,
244
+ file_id=file_id,
245
+ output=output,
246
+ # `var_schema` is the generator's rename of `schema`; a str is
247
+ # passed through so a caller holding a pre-serialized schema is not
248
+ # forced to round-trip it through json.loads.
249
+ var_schema=schema if isinstance(schema, str) or schema is None else json.dumps(schema),
250
+ prompt=prompt,
251
+ profile=profile,
252
+ )
253
+
254
+
255
+ class _Extract(_Namespace):
256
+ def run(
257
+ self,
258
+ *,
259
+ parse_id: str,
260
+ schema: dict[str, Any],
261
+ prompt: str | None = None,
262
+ wait_seconds: int | None = None,
263
+ ) -> Any:
264
+ """Extract against an existing parse. The document is not parsed again."""
265
+ api = ExtractApi(self._client._api_client)
266
+ return api.extract_run(
267
+ ExtractRequest(parse_id=parse_id, var_schema=schema, prompt=prompt),
268
+ prefer=_prefer(wait_seconds),
269
+ )
270
+
271
+ def from_document(
272
+ self,
273
+ *,
274
+ input: FileInput | None = None,
275
+ file_id: str | None = None,
276
+ filename: str | None = None,
277
+ schema: dict[str, Any],
278
+ prompt: str | None = None,
279
+ profile: str | None = None,
280
+ wait_seconds: int | None = None,
281
+ ) -> Any:
282
+ """Extract directly from a document, in one call.
283
+
284
+ The multipart variant of POST /api/v2/extract. Reconstructed over the
285
+ generated transport because openapi-generator emits only the JSON body
286
+ for this path — see this module's docstring.
287
+
288
+ Takes a document OR the id of one already uploaded, exactly as
289
+ parse.run does: ExtractDocumentRequest declares both `input` and
290
+ `file_id`, so requiring the bytes would have made an already-uploaded
291
+ file unusable here and forced a needless re-upload. `filename` only
292
+ applies to the `file_id` path — a raw `input` upload already carries
293
+ its own name.
294
+ """
295
+ if (input is None) == (file_id is None):
296
+ raise ValueError("pass exactly one of input or file_id")
297
+ if input is not None and filename is not None:
298
+ raise ValueError("filename only applies to file_id; input already carries its name")
299
+
300
+ return post_multipart(
301
+ self._client,
302
+ path="/api/v2/extract",
303
+ file_field="input",
304
+ file_value=None if input is None else as_file_part(input),
305
+ fields={
306
+ "file_id": file_id,
307
+ "filename": filename,
308
+ "schema": json.dumps(schema),
309
+ "prompt": prompt,
310
+ "profile": profile,
311
+ },
312
+ wait_seconds=wait_seconds,
313
+ )
314
+
315
+
316
+ class _Jobs(_Namespace):
317
+ def get(
318
+ self,
319
+ job_id: str,
320
+ *,
321
+ output: str | None = None,
322
+ include: list[str] | None = None,
323
+ ) -> Any:
324
+ """The job's status, and its parsed content once available.
325
+
326
+ `output` and `include` select the representation, exactly as on
327
+ parse.run — omitting them here would have made the job route able to
328
+ return only the server default, so a caller who asked for elements on
329
+ the original request could not retrieve them.
330
+ """
331
+ return JobsApi(self._client._api_client).jobs_get(
332
+ job_id, output=output, include=include
333
+ )
334
+
335
+ def list(
336
+ self,
337
+ *,
338
+ cursor: str | None = None,
339
+ limit: int | None = None,
340
+ status: str | None = None,
341
+ ) -> Any:
342
+ return JobsApi(self._client._api_client).jobs_list(
343
+ cursor=cursor, limit=limit, status=status
344
+ )
345
+
346
+ def iterate(
347
+ self,
348
+ *,
349
+ cursor: str | None = None,
350
+ limit: int | None = None,
351
+ status: str | None = None,
352
+ ) -> Iterator[Any]:
353
+ while True:
354
+ page = self.list(cursor=cursor, limit=limit, status=status)
355
+ yield from page.jobs
356
+ cursor = page.next_cursor
357
+ if cursor is None:
358
+ return
359
+
360
+ def cancel(self, job_id: str) -> Any:
361
+ return JobsApi(self._client._api_client).jobs_cancel(job_id)
362
+
363
+ def delete(self, job_id: str) -> None:
364
+ JobsApi(self._client._api_client).jobs_delete(job_id)
365
+
366
+ def stream(
367
+ self,
368
+ job_id: str,
369
+ *,
370
+ output: str | None = None,
371
+ include: list[str] | None = None,
372
+ ) -> Iterator[JobEvent]:
373
+ """Progress events for a running job, as they happen.
374
+
375
+ The same resource as `get(job_id)`, selected by Accept — not a second
376
+ route. Yields `status` events on every public status change and then
377
+ exactly one terminal `result` or `error`, after which the stream ends.
378
+ Keep-alive comments are consumed here and never surface as events.
379
+ """
380
+ # The terminal `result` event carries the same representation the
381
+ # JSON path would return, so these have to reach the stream too — a
382
+ # stream that could only ever deliver the server default would be a
383
+ # second-class way to read the same resource.
384
+ query: dict[str, Any] = {}
385
+ if output is not None:
386
+ query["output"] = output
387
+ if include:
388
+ query["include"] = include
389
+
390
+ url = f"{self._client._server_url}/api/v2/jobs/{quote(job_id, safe='')}"
391
+ if query:
392
+ url = f"{url}?{urlencode(query, doseq=True)}"
393
+
394
+ return stream_job_events(
395
+ url=url,
396
+ headers=self._client._request_headers(),
397
+ timeout_seconds=self._client._timeout_seconds,
398
+ retries=self._client._retries,
399
+ )
400
+
401
+
402
+ class _Upload(_Namespace):
403
+ def run(self, input: FileInput) -> Any:
404
+ return UploadApi(self._client._api_client).upload_run(as_file_part(input))
405
+
406
+ def get(self, file_id: str) -> Any:
407
+ return UploadApi(self._client._api_client).upload_get(file_id)
408
+
409
+ def delete(self, file_id: str) -> None:
410
+ UploadApi(self._client._api_client).upload_delete(file_id)
411
+
412
+
413
+ def _prefer(wait_seconds: int | None) -> str | None:
414
+ """`Prefer: wait=N`, the contract's spelling for "block for up to N seconds".
415
+
416
+ Exposed as `wait_seconds=` rather than making callers hand-build a header
417
+ value; `wait_seconds=0` is meaningful (return the job handle immediately),
418
+ so the check is against None rather than falsiness.
419
+ """
420
+ return None if wait_seconds is None else f"wait={wait_seconds}"
@@ -0,0 +1,72 @@
1
+ """Turning what a caller has into what the request needs.
2
+
3
+ THE FILENAME IS NOT DECORATION. The service validates the upload's extension
4
+ against a published allowlist (`transform_api/document_limits.py`), so a part
5
+ sent without a filename is rejected as "Unsupported file type" — for a
6
+ perfectly good PDF. Any path that loses the filename produces a confusing
7
+ error at the far end, so none of them do.
8
+
9
+ The generated client accepts `bytes`, `str`, or `(filename, bytes)` and NOT a
10
+ file object, which is the one thing every caller reaches for first — the
11
+ obvious `client.parse.run(input=open("invoice.pdf", "rb"))` fails a pydantic
12
+ validation deep inside generated code with four union-branch errors and no
13
+ mention of files. So the accepted types are widened here and normalized to the
14
+ tuple the transport wants.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import os
20
+ from pathlib import Path
21
+ from typing import IO, Union
22
+
23
+ # What a caller may pass for a document.
24
+ FileInput = Union[str, Path, IO[bytes], bytes, tuple[str, bytes]]
25
+
26
+ # What the generated transport wants: (filename, content).
27
+ FilePart = tuple[str, bytes]
28
+
29
+
30
+ def as_file_part(value: FileInput) -> FilePart:
31
+ """Normalize any accepted file input to `(filename, content)`."""
32
+ if isinstance(value, tuple):
33
+ filename, content = value
34
+ return (filename, content)
35
+
36
+ if isinstance(value, (str, Path)):
37
+ path = Path(value)
38
+ return (path.name, path.read_bytes())
39
+
40
+ if isinstance(value, bytes):
41
+ raise TypeError(
42
+ "pass a filename with raw bytes: the service validates the upload's "
43
+ "extension against an allowlist, so bytes alone are rejected as an "
44
+ 'unsupported file type. Use ("invoice.pdf", data), a path, or an '
45
+ "open file."
46
+ )
47
+
48
+ read = getattr(value, "read", None)
49
+ if read is None:
50
+ raise TypeError(
51
+ f"cannot read a document from {type(value).__name__}; pass a path, an "
52
+ 'open binary file, or ("invoice.pdf", data)'
53
+ )
54
+
55
+ content = read()
56
+ if not isinstance(content, bytes):
57
+ raise TypeError(
58
+ "the file must be opened in binary mode — open(path, 'rb'), not open(path)"
59
+ )
60
+
61
+ # `.name` is a str for a file opened from a path, an int for one opened
62
+ # from a file descriptor, and absent for BytesIO. Only the first is a
63
+ # filename; the others have to be supplied by the caller, because guessing
64
+ # one would mean guessing the extension the service validates.
65
+ name = getattr(value, "name", None)
66
+ if not isinstance(name, str):
67
+ raise TypeError(
68
+ "this file object carries no filename, so the extension the service "
69
+ 'validates is unknown. Pass ("invoice.pdf", stream.read()) instead.'
70
+ )
71
+
72
+ return (os.path.basename(name), content)
@@ -0,0 +1,100 @@
1
+ # coding: utf-8
2
+
3
+ # flake8: noqa
4
+
5
+ """
6
+ Unstructured Transform API
7
+
8
+ One document in, the parsed document back. A single call takes a document and returns structured output — no job graph, no strategy selection, no model provider setup, and no polling loop. An optional schema switches the request from parse-only to parse-then-extract.
9
+
10
+ The version of the OpenAPI document: 0.1.0
11
+ Generated by OpenAPI Generator (https://openapi-generator.tech)
12
+
13
+ Do not edit the class manually.
14
+ """ # noqa: E501
15
+
16
+
17
+ __version__ = "0.18.17"
18
+
19
+ # Define package exports
20
+ __all__ = [
21
+ "ExtractApi",
22
+ "JobsApi",
23
+ "ParseApi",
24
+ "UploadApi",
25
+ "ApiResponse",
26
+ "ApiClient",
27
+ "Configuration",
28
+ "OpenApiException",
29
+ "ApiTypeError",
30
+ "ApiValueError",
31
+ "ApiKeyError",
32
+ "ApiAttributeError",
33
+ "ApiException",
34
+ "Citation",
35
+ "DocumentMetadata",
36
+ "Element",
37
+ "ElementMetadata",
38
+ "Error",
39
+ "ErrorCode",
40
+ "ExtractRequest",
41
+ "ExtractionResult",
42
+ "FieldMetadata",
43
+ "JobAccepted",
44
+ "JobOperation",
45
+ "JobPage",
46
+ "JobResult",
47
+ "JobResultState",
48
+ "JobStatus",
49
+ "JobSummary",
50
+ "Locator",
51
+ "OutputFormat",
52
+ "ParseResult",
53
+ "SourceFile",
54
+ "TransformStatus",
55
+ "TransformWarning",
56
+ "UploadResult",
57
+ ]
58
+
59
+ # import apis into sdk package
60
+ from unstructured_transform_client._generated.api.extract_api import ExtractApi as ExtractApi
61
+ from unstructured_transform_client._generated.api.jobs_api import JobsApi as JobsApi
62
+ from unstructured_transform_client._generated.api.parse_api import ParseApi as ParseApi
63
+ from unstructured_transform_client._generated.api.upload_api import UploadApi as UploadApi
64
+
65
+ # import ApiClient
66
+ from unstructured_transform_client._generated.api_response import ApiResponse as ApiResponse
67
+ from unstructured_transform_client._generated.api_client import ApiClient as ApiClient
68
+ from unstructured_transform_client._generated.configuration import Configuration as Configuration
69
+ from unstructured_transform_client._generated.exceptions import OpenApiException as OpenApiException
70
+ from unstructured_transform_client._generated.exceptions import ApiTypeError as ApiTypeError
71
+ from unstructured_transform_client._generated.exceptions import ApiValueError as ApiValueError
72
+ from unstructured_transform_client._generated.exceptions import ApiKeyError as ApiKeyError
73
+ from unstructured_transform_client._generated.exceptions import ApiAttributeError as ApiAttributeError
74
+ from unstructured_transform_client._generated.exceptions import ApiException as ApiException
75
+
76
+ # import models into sdk package
77
+ from unstructured_transform_client._generated.models.citation import Citation as Citation
78
+ from unstructured_transform_client._generated.models.document_metadata import DocumentMetadata as DocumentMetadata
79
+ from unstructured_transform_client._generated.models.element import Element as Element
80
+ from unstructured_transform_client._generated.models.element_metadata import ElementMetadata as ElementMetadata
81
+ from unstructured_transform_client._generated.models.error import Error as Error
82
+ from unstructured_transform_client._generated.models.error_code import ErrorCode as ErrorCode
83
+ from unstructured_transform_client._generated.models.extract_request import ExtractRequest as ExtractRequest
84
+ from unstructured_transform_client._generated.models.extraction_result import ExtractionResult as ExtractionResult
85
+ from unstructured_transform_client._generated.models.field_metadata import FieldMetadata as FieldMetadata
86
+ from unstructured_transform_client._generated.models.job_accepted import JobAccepted as JobAccepted
87
+ from unstructured_transform_client._generated.models.job_operation import JobOperation as JobOperation
88
+ from unstructured_transform_client._generated.models.job_page import JobPage as JobPage
89
+ from unstructured_transform_client._generated.models.job_result import JobResult as JobResult
90
+ from unstructured_transform_client._generated.models.job_result_state import JobResultState as JobResultState
91
+ from unstructured_transform_client._generated.models.job_status import JobStatus as JobStatus
92
+ from unstructured_transform_client._generated.models.job_summary import JobSummary as JobSummary
93
+ from unstructured_transform_client._generated.models.locator import Locator as Locator
94
+ from unstructured_transform_client._generated.models.output_format import OutputFormat as OutputFormat
95
+ from unstructured_transform_client._generated.models.parse_result import ParseResult as ParseResult
96
+ from unstructured_transform_client._generated.models.source_file import SourceFile as SourceFile
97
+ from unstructured_transform_client._generated.models.transform_status import TransformStatus as TransformStatus
98
+ from unstructured_transform_client._generated.models.transform_warning import TransformWarning as TransformWarning
99
+ from unstructured_transform_client._generated.models.upload_result import UploadResult as UploadResult
100
+
@@ -0,0 +1,8 @@
1
+ # flake8: noqa
2
+
3
+ # import apis into api package
4
+ from unstructured_transform_client._generated.api.extract_api import ExtractApi
5
+ from unstructured_transform_client._generated.api.jobs_api import JobsApi
6
+ from unstructured_transform_client._generated.api.parse_api import ParseApi
7
+ from unstructured_transform_client._generated.api.upload_api import UploadApi
8
+