pdfdancer-client-python 0.3.13__py3-none-any.whl → 3.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pdfdancer/__init__.py +79 -22
- pdfdancer/_runtime_version.py +30 -0
- pdfdancer/_version.py +24 -0
- pdfdancer/image_builder.py +23 -3
- pdfdancer/models.py +127 -470
- pdfdancer/page_builder.py +6 -17
- pdfdancer/path_builder.py +127 -6
- pdfdancer/{pdfdancer_v1.py → pdfdancer_v2.py} +850 -1316
- pdfdancer/text_editing.py +1472 -0
- pdfdancer/types.py +94 -399
- pdfdancer_client_python-3.0.0.dist-info/METADATA +521 -0
- pdfdancer_client_python-3.0.0.dist-info/RECORD +18 -0
- {pdfdancer_client_python-0.3.13.dist-info → pdfdancer_client_python-3.0.0.dist-info}/WHEEL +1 -1
- pdfdancer/paragraph_builder.py +0 -554
- pdfdancer/text_line_builder.py +0 -290
- pdfdancer_client_python-0.3.13.dist-info/METADATA +0 -685
- pdfdancer_client_python-0.3.13.dist-info/RECORD +0 -17
- {pdfdancer_client_python-0.3.13.dist-info → pdfdancer_client_python-3.0.0.dist-info}/licenses/LICENSE +0 -0
- {pdfdancer_client_python-0.3.13.dist-info → pdfdancer_client_python-3.0.0.dist-info}/licenses/NOTICE +0 -0
- {pdfdancer_client_python-0.3.13.dist-info → pdfdancer_client_python-3.0.0.dist-info}/top_level.txt +0 -0
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
"""
|
|
2
|
-
PDFDancer Python Client
|
|
2
|
+
PDFDancer Python Client V2
|
|
3
3
|
|
|
4
4
|
A Python client that closely mirrors the Java Client class structure and functionality.
|
|
5
5
|
Provides session-based PDF manipulation operations with strict validation.
|
|
@@ -10,18 +10,29 @@ from __future__ import annotations
|
|
|
10
10
|
import gzip
|
|
11
11
|
import json
|
|
12
12
|
import logging
|
|
13
|
+
import math
|
|
13
14
|
import os
|
|
14
15
|
import sys
|
|
15
16
|
import time
|
|
16
17
|
from datetime import datetime, timezone
|
|
17
|
-
from
|
|
18
|
+
from email.utils import parsedate_to_datetime
|
|
18
19
|
from pathlib import Path
|
|
19
|
-
from typing import
|
|
20
|
+
from typing import (
|
|
21
|
+
TYPE_CHECKING,
|
|
22
|
+
Any,
|
|
23
|
+
BinaryIO,
|
|
24
|
+
Callable,
|
|
25
|
+
List,
|
|
26
|
+
Mapping,
|
|
27
|
+
Optional,
|
|
28
|
+
Union,
|
|
29
|
+
cast,
|
|
30
|
+
)
|
|
20
31
|
|
|
21
32
|
import httpx
|
|
22
|
-
from dotenv import find_dotenv, load_dotenv
|
|
23
33
|
|
|
24
|
-
from . import BezierBuilder, LineBuilder,
|
|
34
|
+
from . import BezierBuilder, LineBuilder, PathBuilder
|
|
35
|
+
from ._runtime_version import resolve_package_version
|
|
25
36
|
from .exceptions import (
|
|
26
37
|
FontNotFoundException,
|
|
27
38
|
HttpClientException,
|
|
@@ -48,8 +59,6 @@ from .models import (
|
|
|
48
59
|
FormFieldRef,
|
|
49
60
|
Image,
|
|
50
61
|
ModifyPathRequest,
|
|
51
|
-
ModifyRequest,
|
|
52
|
-
ModifyTextRequest,
|
|
53
62
|
MoveRequest,
|
|
54
63
|
ObjectRef,
|
|
55
64
|
ObjectType,
|
|
@@ -58,54 +67,42 @@ from .models import (
|
|
|
58
67
|
PageRef,
|
|
59
68
|
PageSize,
|
|
60
69
|
PageSnapshot,
|
|
61
|
-
|
|
70
|
+
)
|
|
71
|
+
from .models import Path as PDFPath
|
|
72
|
+
from .models import (
|
|
73
|
+
PathGroupInfo,
|
|
62
74
|
PathObjectRef,
|
|
63
75
|
Position,
|
|
64
76
|
PositionMode,
|
|
65
|
-
RedactRequest,
|
|
66
|
-
RedactResponse,
|
|
67
|
-
RedactTarget,
|
|
68
|
-
ReflowPreset,
|
|
69
77
|
ShapeType,
|
|
70
|
-
TemplateReplacement,
|
|
71
|
-
TemplateReplaceRequest,
|
|
72
|
-
TextLine,
|
|
73
78
|
TextObjectRef,
|
|
74
79
|
)
|
|
75
80
|
from .page_builder import PageBuilder
|
|
76
|
-
from .
|
|
81
|
+
from .text_editing import (
|
|
82
|
+
TextDeleteRequest,
|
|
83
|
+
TextEditResponse,
|
|
84
|
+
TextInsertRequest,
|
|
85
|
+
TextReplaceRequest,
|
|
86
|
+
TextStyleRequest,
|
|
87
|
+
)
|
|
77
88
|
from .types import (
|
|
78
89
|
FormFieldObject,
|
|
79
90
|
FormObject,
|
|
80
91
|
ImageObject,
|
|
81
|
-
ParagraphObject,
|
|
82
92
|
PathObject,
|
|
83
|
-
PDFObjectBase,
|
|
84
|
-
TextLineObject,
|
|
85
93
|
)
|
|
86
94
|
|
|
87
95
|
if TYPE_CHECKING:
|
|
96
|
+
from .models import BoundingRect as ModelBoundingRect
|
|
88
97
|
from .models import ImageTransformRequest, PathSegment
|
|
89
98
|
from .path_builder import RectangleBuilder
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
def _load_env():
|
|
95
|
-
global _env_loaded
|
|
96
|
-
if _env_loaded:
|
|
97
|
-
return
|
|
98
|
-
load_dotenv(find_dotenv(usecwd=True))
|
|
99
|
-
_env_loaded = True
|
|
100
|
-
|
|
99
|
+
from .types import BoundingRect as GroupBoundingRect
|
|
100
|
+
from .types import PathGroupObject
|
|
101
101
|
|
|
102
102
|
# Client identifier header for all HTTP requests
|
|
103
|
-
#
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
except Exception:
|
|
107
|
-
# Fallback in case package metadata is not available
|
|
108
|
-
CLIENT_HEADER_VALUE = "python/unknown"
|
|
103
|
+
# Prefer the SCM-generated version module; fall back to installed metadata.
|
|
104
|
+
CLIENT_HEADER_VALUE = f"python/{resolve_package_version(default='unknown')}"
|
|
105
|
+
API_PATH_PREFIX = "/v2"
|
|
109
106
|
|
|
110
107
|
# Global variable to disable SSL certificate verification
|
|
111
108
|
# Set to True to skip SSL verification (useful for testing with self-signed certificates)
|
|
@@ -117,70 +114,32 @@ DEFAULT_TOLERANCE = 0.01
|
|
|
117
114
|
|
|
118
115
|
# Retry configuration for transient network errors
|
|
119
116
|
# These settings control automatic retry behavior when encountering transient network errors
|
|
120
|
-
#
|
|
121
|
-
# HTTP status errors (4xx, 5xx) are NOT retried as they are application-level errors.
|
|
117
|
+
# and transient server response statuses.
|
|
122
118
|
#
|
|
123
|
-
#
|
|
124
|
-
#
|
|
119
|
+
# PDFDANCER_MAX_ATTEMPTS: Maximum number of total attempts (default: 3).
|
|
120
|
+
# The initial request counts as one attempt, so 3 permits at most 2 retries.
|
|
125
121
|
#
|
|
126
|
-
# PDFDANCER_RETRY_BACKOFF_FACTOR:
|
|
127
|
-
# The actual delay for each retry is calculated as:
|
|
122
|
+
# PDFDANCER_RETRY_BACKOFF_FACTOR: Multiplier for exponential backoff delays (default: 2.0)
|
|
123
|
+
# The actual delay for each retry is calculated as: initial_delay * (backoff_factor ** retry_count)
|
|
128
124
|
# Examples:
|
|
129
|
-
# -
|
|
130
|
-
# -
|
|
131
|
-
|
|
132
|
-
DEFAULT_MAX_RETRIES = int(os.environ.get("PDFDANCER_MAX_RETRIES", "3"))
|
|
125
|
+
# - retry_backoff_factor=2.0: delays are 1s, 2s, 4s, 8s, ...
|
|
126
|
+
# - retry_backoff_factor=3.0: delays are 1s, 3s, 9s, ...
|
|
127
|
+
DEFAULT_MAX_ATTEMPTS = int(os.environ.get("PDFDANCER_MAX_ATTEMPTS", "3"))
|
|
133
128
|
DEFAULT_RETRY_BACKOFF_FACTOR = float(
|
|
134
129
|
os.environ.get("PDFDANCER_RETRY_BACKOFF_FACTOR", "2.0")
|
|
135
130
|
)
|
|
131
|
+
DEFAULT_RETRY_INITIAL_DELAY = 1.0
|
|
132
|
+
DEFAULT_RETRY_MAX_DELAY = 5.0
|
|
133
|
+
DEFAULT_RETRYABLE_STATUS_CODES = {408, 429, 500, 502, 503, 504, 520}
|
|
136
134
|
|
|
137
135
|
|
|
138
|
-
def
|
|
139
|
-
|
|
140
|
-
)
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
result.append(TemplateReplacement(placeholder=placeholder, text=value))
|
|
146
|
-
elif "image" in value:
|
|
147
|
-
image_source = value["image"]
|
|
148
|
-
if isinstance(image_source, Path):
|
|
149
|
-
image_data = image_source.read_bytes()
|
|
150
|
-
image_format = image_source.suffix.lstrip(".").upper()
|
|
151
|
-
if image_format == "JPG":
|
|
152
|
-
image_format = "JPEG"
|
|
153
|
-
elif isinstance(image_source, bytes):
|
|
154
|
-
image_data = image_source
|
|
155
|
-
image_format = None
|
|
156
|
-
else:
|
|
157
|
-
raise ValueError(
|
|
158
|
-
f"Unsupported image source type: {type(image_source)}. "
|
|
159
|
-
"Use a Path or bytes."
|
|
160
|
-
)
|
|
161
|
-
img = Image(
|
|
162
|
-
data=image_data,
|
|
163
|
-
format=value.get("format", image_format),
|
|
164
|
-
width=value.get("width"),
|
|
165
|
-
height=value.get("height"),
|
|
166
|
-
)
|
|
167
|
-
result.append(
|
|
168
|
-
TemplateReplacement(
|
|
169
|
-
placeholder=placeholder,
|
|
170
|
-
text=None,
|
|
171
|
-
image=img,
|
|
172
|
-
)
|
|
173
|
-
)
|
|
174
|
-
else:
|
|
175
|
-
result.append(
|
|
176
|
-
TemplateReplacement(
|
|
177
|
-
placeholder=placeholder,
|
|
178
|
-
text=value["text"],
|
|
179
|
-
font=value.get("font"),
|
|
180
|
-
color=value.get("color"),
|
|
181
|
-
)
|
|
182
|
-
)
|
|
183
|
-
return result
|
|
136
|
+
def _validate_max_attempts(max_attempts: int) -> int:
|
|
137
|
+
"""Validate that the total-attempt limit can include an initial request."""
|
|
138
|
+
if isinstance(max_attempts, bool) or not isinstance(max_attempts, int):
|
|
139
|
+
raise ValidationException("max_attempts must be an integer")
|
|
140
|
+
if max_attempts < 1:
|
|
141
|
+
raise ValidationException("max_attempts must be at least 1")
|
|
142
|
+
return max_attempts
|
|
184
143
|
|
|
185
144
|
|
|
186
145
|
def _generate_timestamp() -> str:
|
|
@@ -291,15 +250,17 @@ def _is_retryable_error(error: Exception) -> bool:
|
|
|
291
250
|
# - PoolTimeout (connection pool exhausted)
|
|
292
251
|
error_msg = str(error).lower()
|
|
293
252
|
|
|
294
|
-
if isinstance(
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
253
|
+
if isinstance(
|
|
254
|
+
error,
|
|
255
|
+
(
|
|
256
|
+
httpx.RemoteProtocolError,
|
|
257
|
+
httpx.ConnectError,
|
|
258
|
+
httpx.ConnectTimeout,
|
|
259
|
+
httpx.ReadTimeout,
|
|
260
|
+
httpx.PoolTimeout,
|
|
261
|
+
httpx.WriteTimeout,
|
|
262
|
+
),
|
|
263
|
+
):
|
|
303
264
|
return True
|
|
304
265
|
|
|
305
266
|
# Check for specific error messages that indicate transient issues
|
|
@@ -329,15 +290,167 @@ def _get_retry_after_delay(response: httpx.Response) -> Optional[int]:
|
|
|
329
290
|
return None
|
|
330
291
|
|
|
331
292
|
try:
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
return int(retry_after)
|
|
293
|
+
delay_seconds = int(retry_after)
|
|
294
|
+
return delay_seconds if delay_seconds >= 0 else None
|
|
335
295
|
except ValueError:
|
|
336
|
-
|
|
337
|
-
|
|
296
|
+
pass
|
|
297
|
+
|
|
298
|
+
try:
|
|
299
|
+
retry_at = parsedate_to_datetime(retry_after)
|
|
300
|
+
if retry_at.tzinfo is None:
|
|
301
|
+
retry_at = retry_at.replace(tzinfo=timezone.utc)
|
|
302
|
+
delay = (retry_at - datetime.now(timezone.utc)).total_seconds()
|
|
303
|
+
return max(0, math.ceil(delay))
|
|
304
|
+
except (TypeError, ValueError, OverflowError):
|
|
338
305
|
return None
|
|
339
306
|
|
|
340
307
|
|
|
308
|
+
def _calculate_retry_delay(
|
|
309
|
+
response: Optional[httpx.Response],
|
|
310
|
+
attempt: int,
|
|
311
|
+
retry_backoff_factor: float,
|
|
312
|
+
max_delay_seconds: float = DEFAULT_RETRY_MAX_DELAY,
|
|
313
|
+
) -> float:
|
|
314
|
+
"""
|
|
315
|
+
Calculate retry delay for a given attempt.
|
|
316
|
+
|
|
317
|
+
Args:
|
|
318
|
+
response: HTTP response when retrying on status code, otherwise None
|
|
319
|
+
attempt: Zero-based retry attempt index (0 for first retry)
|
|
320
|
+
retry_backoff_factor: Backoff multiplier
|
|
321
|
+
max_delay_seconds: Maximum retry delay
|
|
322
|
+
|
|
323
|
+
Returns:
|
|
324
|
+
Delay in seconds, respecting max delay and Retry-After when available.
|
|
325
|
+
"""
|
|
326
|
+
if response is not None and response.status_code == 429:
|
|
327
|
+
retry_after = _get_retry_after_delay(response)
|
|
328
|
+
if retry_after is not None:
|
|
329
|
+
return float(min(max_delay_seconds, retry_after))
|
|
330
|
+
delay = DEFAULT_RETRY_INITIAL_DELAY * (retry_backoff_factor**attempt)
|
|
331
|
+
return float(min(max_delay_seconds, delay))
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
def _execute_request_with_retries(
|
|
335
|
+
request_callable: Callable[[], httpx.Response],
|
|
336
|
+
operation: str,
|
|
337
|
+
max_attempts: int,
|
|
338
|
+
retry_backoff_factor: float,
|
|
339
|
+
retryable_status_codes: Optional[set[int]] = None,
|
|
340
|
+
pre_request_hook: Optional[Callable[[int], None]] = None,
|
|
341
|
+
) -> httpx.Response:
|
|
342
|
+
"""
|
|
343
|
+
Execute a request callable with retry handling for transient statuses and request errors.
|
|
344
|
+
|
|
345
|
+
Args:
|
|
346
|
+
request_callable: Zero-arg callable returning an httpx.Response.
|
|
347
|
+
operation: Human-readable operation name for retry logging.
|
|
348
|
+
max_attempts: Maximum number of total attempts.
|
|
349
|
+
retry_backoff_factor: Retry backoff multiplier.
|
|
350
|
+
retryable_status_codes: Status codes that should trigger retries.
|
|
351
|
+
pre_request_hook: Optional callback invoked before each request attempt.
|
|
352
|
+
|
|
353
|
+
Returns:
|
|
354
|
+
The last response returned by the request callable.
|
|
355
|
+
|
|
356
|
+
Raises:
|
|
357
|
+
httpx.RequestError: When a non-retriable request error occurs.
|
|
358
|
+
"""
|
|
359
|
+
attempts = max(1, max_attempts)
|
|
360
|
+
retry_statuses = (
|
|
361
|
+
retryable_status_codes
|
|
362
|
+
if retryable_status_codes is not None
|
|
363
|
+
else DEFAULT_RETRYABLE_STATUS_CODES
|
|
364
|
+
)
|
|
365
|
+
attempt = 0
|
|
366
|
+
last_response = None
|
|
367
|
+
|
|
368
|
+
while attempt < attempts:
|
|
369
|
+
try:
|
|
370
|
+
if pre_request_hook:
|
|
371
|
+
pre_request_hook(attempt)
|
|
372
|
+
|
|
373
|
+
response = request_callable()
|
|
374
|
+
last_response = response
|
|
375
|
+
|
|
376
|
+
if response.status_code in retry_statuses and attempt < attempts - 1:
|
|
377
|
+
delay = _calculate_retry_delay(
|
|
378
|
+
response=response,
|
|
379
|
+
attempt=attempt,
|
|
380
|
+
retry_backoff_factor=retry_backoff_factor,
|
|
381
|
+
)
|
|
382
|
+
if response.status_code == 429:
|
|
383
|
+
print(
|
|
384
|
+
f"Rate limit (429) on {operation} - retrying in {delay}s "
|
|
385
|
+
f"(attempt {attempt + 1}/{attempts})",
|
|
386
|
+
file=sys.stderr,
|
|
387
|
+
)
|
|
388
|
+
elif DEBUG:
|
|
389
|
+
print(
|
|
390
|
+
f"{time.time()}|{operation} - Retryable HTTP {response.status_code}, "
|
|
391
|
+
f"retrying in {delay}s (attempt {attempt + 1}/{attempts})"
|
|
392
|
+
)
|
|
393
|
+
|
|
394
|
+
if delay > 0:
|
|
395
|
+
time.sleep(delay)
|
|
396
|
+
attempt += 1
|
|
397
|
+
continue
|
|
398
|
+
|
|
399
|
+
return response
|
|
400
|
+
|
|
401
|
+
except httpx.HTTPStatusError as e:
|
|
402
|
+
response = e.response
|
|
403
|
+
last_response = response
|
|
404
|
+
if response.status_code in retry_statuses and attempt < attempts - 1:
|
|
405
|
+
delay = _calculate_retry_delay(
|
|
406
|
+
response=response,
|
|
407
|
+
attempt=attempt,
|
|
408
|
+
retry_backoff_factor=retry_backoff_factor,
|
|
409
|
+
)
|
|
410
|
+
if response.status_code == 429:
|
|
411
|
+
print(
|
|
412
|
+
f"Rate limit (429) on {operation} - retrying in {delay}s "
|
|
413
|
+
f"(attempt {attempt + 1}/{attempts})",
|
|
414
|
+
file=sys.stderr,
|
|
415
|
+
)
|
|
416
|
+
elif DEBUG:
|
|
417
|
+
print(
|
|
418
|
+
f"{time.time()}|{operation} - Retryable HTTP {response.status_code}, "
|
|
419
|
+
f"retrying in {delay}s (attempt {attempt + 1}/{attempts})"
|
|
420
|
+
)
|
|
421
|
+
|
|
422
|
+
if delay > 0:
|
|
423
|
+
time.sleep(delay)
|
|
424
|
+
attempt += 1
|
|
425
|
+
continue
|
|
426
|
+
raise
|
|
427
|
+
|
|
428
|
+
except httpx.RequestError as e:
|
|
429
|
+
if _is_retryable_error(e) and attempt < attempts - 1:
|
|
430
|
+
delay = _calculate_retry_delay(
|
|
431
|
+
response=None,
|
|
432
|
+
attempt=attempt,
|
|
433
|
+
retry_backoff_factor=retry_backoff_factor,
|
|
434
|
+
)
|
|
435
|
+
if DEBUG:
|
|
436
|
+
print(
|
|
437
|
+
f"{time.time()}|{operation} - Retryable error: {str(e)}, "
|
|
438
|
+
f"retrying in {delay}s (attempt {attempt + 1}/{attempts})"
|
|
439
|
+
)
|
|
440
|
+
if delay > 0:
|
|
441
|
+
time.sleep(delay)
|
|
442
|
+
attempt += 1
|
|
443
|
+
continue
|
|
444
|
+
raise
|
|
445
|
+
|
|
446
|
+
if last_response is None:
|
|
447
|
+
raise RuntimeError(
|
|
448
|
+
f"Request exhausted retries without a response for operation: {operation}"
|
|
449
|
+
)
|
|
450
|
+
|
|
451
|
+
return last_response
|
|
452
|
+
|
|
453
|
+
|
|
341
454
|
class PageClient:
|
|
342
455
|
def __init__(
|
|
343
456
|
self,
|
|
@@ -352,6 +465,7 @@ class PageClient:
|
|
|
352
465
|
self.position = Position.at_page(page_number)
|
|
353
466
|
self.internal_id = f"PAGE-{page_number}"
|
|
354
467
|
self.page_size = page_size
|
|
468
|
+
self.orientation: Optional[Union[Orientation, str]]
|
|
355
469
|
if isinstance(orientation, str):
|
|
356
470
|
normalized = orientation.strip().upper()
|
|
357
471
|
try:
|
|
@@ -368,59 +482,6 @@ class PageClient:
|
|
|
368
482
|
# noinspection PyProtectedMember
|
|
369
483
|
return self.root._to_path_objects(self.root._find_paths(position, tolerance))
|
|
370
484
|
|
|
371
|
-
def select_paragraphs(self) -> List[ParagraphObject]:
|
|
372
|
-
# noinspection PyProtectedMember
|
|
373
|
-
return self.root._to_paragraph_objects(
|
|
374
|
-
self.root._find_paragraphs(Position.at_page(self.page_number))
|
|
375
|
-
)
|
|
376
|
-
|
|
377
|
-
def select_paragraphs_starting_with(self, text: str) -> List[ParagraphObject]:
|
|
378
|
-
position = Position.at_page(self.page_number)
|
|
379
|
-
position.with_text_starts(text)
|
|
380
|
-
# noinspection PyProtectedMember
|
|
381
|
-
return self.root._to_paragraph_objects(self.root._find_paragraphs(position))
|
|
382
|
-
|
|
383
|
-
def select_paragraphs_matching(self, pattern):
|
|
384
|
-
position = Position.at_page(self.page_number)
|
|
385
|
-
position.text_pattern = pattern
|
|
386
|
-
# noinspection PyProtectedMember
|
|
387
|
-
return self.root._to_paragraph_objects(self.root._find_paragraphs(position))
|
|
388
|
-
|
|
389
|
-
def select_text_lines_matching(self, pattern: str) -> List[TextLineObject]:
|
|
390
|
-
position = Position.at_page(self.page_number)
|
|
391
|
-
position.text_pattern = pattern
|
|
392
|
-
# noinspection PyProtectedMember
|
|
393
|
-
return self.root._to_textline_objects(self.root._find_text_lines(position))
|
|
394
|
-
|
|
395
|
-
def select_paragraphs_at(
|
|
396
|
-
self, x: float, y: float, tolerance: float = DEFAULT_TOLERANCE
|
|
397
|
-
) -> List[ParagraphObject]:
|
|
398
|
-
position = Position.at_page_coordinates(self.page_number, x, y)
|
|
399
|
-
# noinspection PyProtectedMember
|
|
400
|
-
return self.root._to_paragraph_objects(
|
|
401
|
-
self.root._find_paragraphs(position, tolerance)
|
|
402
|
-
)
|
|
403
|
-
|
|
404
|
-
def select_text_lines(self) -> List[TextLineObject]:
|
|
405
|
-
position = Position.at_page(self.page_number)
|
|
406
|
-
# noinspection PyProtectedMember
|
|
407
|
-
return self.root._to_textline_objects(self.root._find_text_lines(position))
|
|
408
|
-
|
|
409
|
-
def select_text_lines_starting_with(self, text: str) -> List[TextLineObject]:
|
|
410
|
-
position = Position.at_page(self.page_number)
|
|
411
|
-
position.with_text_starts(text)
|
|
412
|
-
# noinspection PyProtectedMember
|
|
413
|
-
return self.root._to_textline_objects(self.root._find_text_lines(position))
|
|
414
|
-
|
|
415
|
-
def select_text_lines_at(
|
|
416
|
-
self, x, y, tolerance: float = DEFAULT_TOLERANCE
|
|
417
|
-
) -> List[TextLineObject]:
|
|
418
|
-
position = Position.at_page_coordinates(self.page_number, x, y)
|
|
419
|
-
# noinspection PyProtectedMember
|
|
420
|
-
return self.root._to_textline_objects(
|
|
421
|
-
self.root._find_text_lines(position, tolerance)
|
|
422
|
-
)
|
|
423
|
-
|
|
424
485
|
def select_images(self) -> List[ImageObject]:
|
|
425
486
|
# noinspection PyProtectedMember
|
|
426
487
|
return self.root._to_image_objects(
|
|
@@ -470,92 +531,6 @@ class PageClient:
|
|
|
470
531
|
|
|
471
532
|
# Singular selection methods (convenience methods returning first match or None)
|
|
472
533
|
|
|
473
|
-
def select_paragraph_at(
|
|
474
|
-
self, x: float, y: float, tolerance: float = DEFAULT_TOLERANCE
|
|
475
|
-
) -> Optional[ParagraphObject]:
|
|
476
|
-
"""
|
|
477
|
-
Select the first paragraph at the specified coordinates.
|
|
478
|
-
|
|
479
|
-
Args:
|
|
480
|
-
x: X coordinate in points
|
|
481
|
-
y: Y coordinate in points
|
|
482
|
-
tolerance: Tolerance in points for spatial matching (default: DEFAULT_TOLERANCE)
|
|
483
|
-
|
|
484
|
-
Returns:
|
|
485
|
-
First ParagraphObject at the coordinates, or None if no match
|
|
486
|
-
"""
|
|
487
|
-
results = self.select_paragraphs_at(x, y, tolerance)
|
|
488
|
-
return results[0] if results else None
|
|
489
|
-
|
|
490
|
-
def select_paragraph_starting_with(self, text: str) -> Optional[ParagraphObject]:
|
|
491
|
-
"""
|
|
492
|
-
Select the first paragraph starting with the specified text.
|
|
493
|
-
|
|
494
|
-
Args:
|
|
495
|
-
text: Text to search for at the start of paragraphs
|
|
496
|
-
|
|
497
|
-
Returns:
|
|
498
|
-
First ParagraphObject starting with the text, or None if no match
|
|
499
|
-
"""
|
|
500
|
-
results = self.select_paragraphs_starting_with(text)
|
|
501
|
-
return results[0] if results else None
|
|
502
|
-
|
|
503
|
-
def select_paragraph_matching(self, pattern: str) -> Optional[ParagraphObject]:
|
|
504
|
-
"""
|
|
505
|
-
Select the first paragraph matching the specified regex pattern.
|
|
506
|
-
|
|
507
|
-
Args:
|
|
508
|
-
pattern: Regex pattern to match against paragraph text
|
|
509
|
-
|
|
510
|
-
Returns:
|
|
511
|
-
First ParagraphObject matching the pattern, or None if no match
|
|
512
|
-
"""
|
|
513
|
-
results = self.select_paragraphs_matching(pattern)
|
|
514
|
-
return results[0] if results else None
|
|
515
|
-
|
|
516
|
-
def select_text_line_at(
|
|
517
|
-
self, x: float, y: float, tolerance: float = DEFAULT_TOLERANCE
|
|
518
|
-
) -> Optional[TextLineObject]:
|
|
519
|
-
"""
|
|
520
|
-
Select the first text line at the specified coordinates.
|
|
521
|
-
|
|
522
|
-
Args:
|
|
523
|
-
x: X coordinate in points
|
|
524
|
-
y: Y coordinate in points
|
|
525
|
-
tolerance: Tolerance in points for spatial matching (default: DEFAULT_TOLERANCE)
|
|
526
|
-
|
|
527
|
-
Returns:
|
|
528
|
-
First TextLineObject at the coordinates, or None if no match
|
|
529
|
-
"""
|
|
530
|
-
results = self.select_text_lines_at(x, y, tolerance)
|
|
531
|
-
return results[0] if results else None
|
|
532
|
-
|
|
533
|
-
def select_text_line_starting_with(self, text: str) -> Optional[TextLineObject]:
|
|
534
|
-
"""
|
|
535
|
-
Select the first text line starting with the specified text.
|
|
536
|
-
|
|
537
|
-
Args:
|
|
538
|
-
text: Text to search for at the start of text lines
|
|
539
|
-
|
|
540
|
-
Returns:
|
|
541
|
-
First TextLineObject starting with the text, or None if no match
|
|
542
|
-
"""
|
|
543
|
-
results = self.select_text_lines_starting_with(text)
|
|
544
|
-
return results[0] if results else None
|
|
545
|
-
|
|
546
|
-
def select_text_line_matching(self, pattern: str) -> Optional[TextLineObject]:
|
|
547
|
-
"""
|
|
548
|
-
Select the first text line matching the specified regex pattern.
|
|
549
|
-
|
|
550
|
-
Args:
|
|
551
|
-
pattern: Regex pattern to match against text line text
|
|
552
|
-
|
|
553
|
-
Returns:
|
|
554
|
-
First TextLineObject matching the pattern, or None if no match
|
|
555
|
-
"""
|
|
556
|
-
results = self.select_text_lines_matching(pattern)
|
|
557
|
-
return results[0] if results else None
|
|
558
|
-
|
|
559
534
|
def select_image_at(
|
|
560
535
|
self, x: float, y: float, tolerance: float = DEFAULT_TOLERANCE
|
|
561
536
|
) -> Optional[ImageObject]:
|
|
@@ -573,6 +548,10 @@ class PageClient:
|
|
|
573
548
|
results = self.select_images_at(x, y, tolerance)
|
|
574
549
|
return results[0] if results else None
|
|
575
550
|
|
|
551
|
+
def select_image(self) -> Optional[ImageObject]:
|
|
552
|
+
results = self.select_images()
|
|
553
|
+
return results[0] if results else None
|
|
554
|
+
|
|
576
555
|
def select_form_at(
|
|
577
556
|
self, x: float, y: float, tolerance: float = DEFAULT_TOLERANCE
|
|
578
557
|
) -> Optional[FormObject]:
|
|
@@ -590,6 +569,10 @@ class PageClient:
|
|
|
590
569
|
results = self.select_forms_at(x, y, tolerance)
|
|
591
570
|
return results[0] if results else None
|
|
592
571
|
|
|
572
|
+
def select_form(self) -> Optional[FormObject]:
|
|
573
|
+
results = self.select_forms()
|
|
574
|
+
return results[0] if results else None
|
|
575
|
+
|
|
593
576
|
def select_form_field_at(
|
|
594
577
|
self, x: float, y: float, tolerance: float = DEFAULT_TOLERANCE
|
|
595
578
|
) -> Optional[FormFieldObject]:
|
|
@@ -637,10 +620,15 @@ class PageClient:
|
|
|
637
620
|
results = self.select_paths_at(x, y, tolerance)
|
|
638
621
|
return results[0] if results else None
|
|
639
622
|
|
|
623
|
+
def select_path(self) -> Optional[PathObject]:
|
|
624
|
+
results = self.select_paths()
|
|
625
|
+
return results[0] if results else None
|
|
626
|
+
|
|
640
627
|
@classmethod
|
|
641
628
|
def from_ref(cls, root: "PDFDancer", page_ref: PageRef) -> "PageClient":
|
|
629
|
+
page_number = cast(int, page_ref.position.page_number)
|
|
642
630
|
page_client = PageClient(
|
|
643
|
-
page_number=
|
|
631
|
+
page_number=page_number,
|
|
644
632
|
root=root,
|
|
645
633
|
page_size=page_ref.page_size,
|
|
646
634
|
orientation=page_ref.orientation,
|
|
@@ -648,7 +636,7 @@ class PageClient:
|
|
|
648
636
|
page_client.internal_id = page_ref.internal_id
|
|
649
637
|
if page_ref.position is not None:
|
|
650
638
|
page_client.position = page_ref.position
|
|
651
|
-
page_client.page_number = page_ref.position.page_number
|
|
639
|
+
page_client.page_number = cast(int, page_ref.position.page_number)
|
|
652
640
|
return page_client
|
|
653
641
|
|
|
654
642
|
def delete(self) -> bool:
|
|
@@ -657,9 +645,9 @@ class PageClient:
|
|
|
657
645
|
|
|
658
646
|
def move_to(self, target_page_number: int) -> bool:
|
|
659
647
|
"""Move this page to a different index within the document."""
|
|
660
|
-
if target_page_number is None or target_page_number <
|
|
648
|
+
if target_page_number is None or target_page_number < 1:
|
|
661
649
|
raise ValidationException(
|
|
662
|
-
f"Target page
|
|
650
|
+
f"Target page number must be >= 1, got {target_page_number}"
|
|
663
651
|
)
|
|
664
652
|
|
|
665
653
|
# noinspection PyProtectedMember
|
|
@@ -669,54 +657,11 @@ class PageClient:
|
|
|
669
657
|
self.position = Position.at_page(target_page_number)
|
|
670
658
|
return moved
|
|
671
659
|
|
|
672
|
-
def
|
|
673
|
-
self,
|
|
674
|
-
replacements: Dict[str, Union[str, dict]],
|
|
675
|
-
reflow_preset: Optional[ReflowPreset] = None,
|
|
676
|
-
) -> bool:
|
|
677
|
-
"""
|
|
678
|
-
Replace template placeholders on this page.
|
|
679
|
-
|
|
680
|
-
Finds exact text matches for placeholders and replaces them with specified
|
|
681
|
-
content. All placeholders must be found or the operation fails atomically.
|
|
682
|
-
|
|
683
|
-
Args:
|
|
684
|
-
replacements: Dict mapping placeholder strings to replacement values.
|
|
685
|
-
- Simple: {"{{NAME}}": "John Doe"}
|
|
686
|
-
- With options: {"{{NAME}}": {"text": "John", "font": Font(...), "color": Color(...)}}
|
|
687
|
-
- With image: {"{{LOGO}}": {"image": Path("logo.png")}}
|
|
688
|
-
- With image and size: {"{{LOGO}}": {"image": Path("logo.png"), "width": 50, "height": 50}}
|
|
689
|
-
reflow_preset: Optional ReflowPreset to control text reflow behavior.
|
|
690
|
-
- BEST_EFFORT: Attempt to reflow, proceed even if imperfect
|
|
691
|
-
- FIT_OR_FAIL: Reflow must succeed or operation fails
|
|
692
|
-
- NONE: No reflow, replacement placed as-is
|
|
693
|
-
|
|
694
|
-
Returns:
|
|
695
|
-
True if all replacements were successful
|
|
696
|
-
|
|
697
|
-
Example:
|
|
698
|
-
```python
|
|
699
|
-
page.apply_replacements({
|
|
700
|
-
"{{NAME}}": "John Doe",
|
|
701
|
-
})
|
|
702
|
-
```
|
|
703
|
-
"""
|
|
704
|
-
replacement_list = _dict_to_replacements(replacements)
|
|
705
|
-
# noinspection PyProtectedMember
|
|
706
|
-
return self.root._apply_replacements(
|
|
707
|
-
replacements=replacement_list,
|
|
708
|
-
page_number=self.page_number,
|
|
709
|
-
reflow_preset=reflow_preset,
|
|
710
|
-
)
|
|
711
|
-
|
|
712
|
-
def _ref(self):
|
|
660
|
+
def _ref(self) -> ObjectRef:
|
|
713
661
|
return ObjectRef(
|
|
714
662
|
internal_id=self.internal_id, position=self.position, type=self.object_type
|
|
715
663
|
)
|
|
716
664
|
|
|
717
|
-
def new_paragraph(self) -> ParagraphBuilder:
|
|
718
|
-
return ParagraphPageBuilder(self.root, self.page_number)
|
|
719
|
-
|
|
720
665
|
def new_path(self) -> PathBuilder:
|
|
721
666
|
return PathBuilder(self.root, self.page_number)
|
|
722
667
|
|
|
@@ -734,13 +679,17 @@ class PageClient:
|
|
|
734
679
|
|
|
735
680
|
return RectangleBuilder(self.root, self.page_number)
|
|
736
681
|
|
|
737
|
-
def
|
|
682
|
+
def text(self) -> "PageTextClient":
|
|
683
|
+
"""Return selector-based text operations scoped to this one-based page."""
|
|
684
|
+
return PageTextClient(self.root, self.page_number)
|
|
685
|
+
|
|
686
|
+
def select_paths(self) -> List[PathObject]:
|
|
738
687
|
# noinspection PyProtectedMember
|
|
739
688
|
return self.root._to_path_objects(
|
|
740
689
|
self.root._find_paths(Position.at_page(self.page_number))
|
|
741
690
|
)
|
|
742
691
|
|
|
743
|
-
def group_paths(self, path_ids):
|
|
692
|
+
def group_paths(self, path_ids: List[str]) -> "PathGroupObject":
|
|
744
693
|
"""Group paths by their IDs. Returns a PathGroupObject."""
|
|
745
694
|
from .types import PathGroupObject
|
|
746
695
|
|
|
@@ -748,7 +697,7 @@ class PageClient:
|
|
|
748
697
|
info = self.root._create_path_group(page_index, path_ids=path_ids)
|
|
749
698
|
return PathGroupObject(self.root, page_index, info)
|
|
750
699
|
|
|
751
|
-
def group_paths_in_region(self, region):
|
|
700
|
+
def group_paths_in_region(self, region: "GroupBoundingRect") -> "PathGroupObject":
|
|
752
701
|
"""Group paths within a bounding region. Returns a PathGroupObject."""
|
|
753
702
|
from .types import PathGroupObject
|
|
754
703
|
|
|
@@ -756,39 +705,72 @@ class PageClient:
|
|
|
756
705
|
info = self.root._create_path_group(page_index, region=region)
|
|
757
706
|
return PathGroupObject(self.root, page_index, info)
|
|
758
707
|
|
|
759
|
-
def get_path_groups(self):
|
|
708
|
+
def get_path_groups(self) -> List["PathGroupObject"]:
|
|
760
709
|
"""List all path groups on this page."""
|
|
710
|
+
return self.root._list_path_groups(self.page_number)
|
|
761
711
|
|
|
762
|
-
|
|
763
|
-
return self.root._list_path_groups(page_index)
|
|
764
|
-
|
|
765
|
-
def select_elements(self):
|
|
712
|
+
def select_elements(self, types: Optional[str] = None) -> List[ObjectRef]:
|
|
766
713
|
"""
|
|
767
|
-
Select all
|
|
714
|
+
Select all live object-reference elements on this page.
|
|
768
715
|
|
|
769
716
|
Returns:
|
|
770
717
|
List of all PDF objects on this page
|
|
771
718
|
"""
|
|
772
|
-
|
|
773
|
-
|
|
774
|
-
|
|
775
|
-
|
|
776
|
-
result.extend(self.select_paths())
|
|
777
|
-
result.extend(self.select_forms())
|
|
778
|
-
result.extend(self.select_form_fields())
|
|
779
|
-
return result
|
|
719
|
+
return self.get_snapshot(types).elements
|
|
720
|
+
|
|
721
|
+
def get_snapshot(self, types: Optional[str] = None) -> PageSnapshot:
|
|
722
|
+
return self.root.get_page_snapshot(self.page_number, types)
|
|
780
723
|
|
|
781
724
|
@property
|
|
782
|
-
def size(self):
|
|
725
|
+
def size(self) -> Optional[PageSize]:
|
|
783
726
|
"""Property alias for page size."""
|
|
784
727
|
return self.page_size
|
|
785
728
|
|
|
786
729
|
@property
|
|
787
|
-
def page_orientation(self):
|
|
730
|
+
def page_orientation(self) -> Optional[Union[Orientation, str]]:
|
|
788
731
|
"""Property alias for orientation."""
|
|
789
732
|
return self.orientation
|
|
790
733
|
|
|
791
734
|
|
|
735
|
+
class TextClient:
|
|
736
|
+
"""Document-scoped selector-based text editing operations."""
|
|
737
|
+
|
|
738
|
+
def __init__(self, root: "PDFDancer") -> None:
|
|
739
|
+
self._root = root
|
|
740
|
+
|
|
741
|
+
def replace(self, request: TextReplaceRequest) -> TextEditResponse:
|
|
742
|
+
return self._root._edit_text("replace", request)
|
|
743
|
+
|
|
744
|
+
def delete(self, request: TextDeleteRequest) -> TextEditResponse:
|
|
745
|
+
return self._root._edit_text("delete", request)
|
|
746
|
+
|
|
747
|
+
def insert(self, request: TextInsertRequest) -> TextEditResponse:
|
|
748
|
+
return self._root._edit_text("insert", request)
|
|
749
|
+
|
|
750
|
+
def style(self, request: TextStyleRequest) -> TextEditResponse:
|
|
751
|
+
return self._root._edit_text("style", request)
|
|
752
|
+
|
|
753
|
+
|
|
754
|
+
class PageTextClient(TextClient):
|
|
755
|
+
"""Selector-based text editing operations scoped to one page."""
|
|
756
|
+
|
|
757
|
+
def __init__(self, root: "PDFDancer", page_number: int) -> None:
|
|
758
|
+
super().__init__(root)
|
|
759
|
+
self._page_number = page_number
|
|
760
|
+
|
|
761
|
+
def replace(self, request: TextReplaceRequest) -> TextEditResponse:
|
|
762
|
+
return super().replace(request.with_pages((self._page_number,)))
|
|
763
|
+
|
|
764
|
+
def delete(self, request: TextDeleteRequest) -> TextEditResponse:
|
|
765
|
+
return super().delete(request.with_pages((self._page_number,)))
|
|
766
|
+
|
|
767
|
+
def insert(self, request: TextInsertRequest) -> TextEditResponse:
|
|
768
|
+
return super().insert(request.with_pages((self._page_number,)))
|
|
769
|
+
|
|
770
|
+
def style(self, request: TextStyleRequest) -> TextEditResponse:
|
|
771
|
+
return super().style(request.with_pages((self._page_number,)))
|
|
772
|
+
|
|
773
|
+
|
|
792
774
|
class PDFDancer:
|
|
793
775
|
"""
|
|
794
776
|
REST API client for interacting with the PDFDancer PDF manipulation service.
|
|
@@ -797,6 +779,8 @@ class PDFDancer:
|
|
|
797
779
|
Handles authentication, session lifecycle, and HTTP communication transparently.
|
|
798
780
|
"""
|
|
799
781
|
|
|
782
|
+
_pdf_bytes: Optional[bytes]
|
|
783
|
+
|
|
800
784
|
# --------------------------------------------------------------
|
|
801
785
|
# CLASS METHOD ENTRY POINT
|
|
802
786
|
# --------------------------------------------------------------
|
|
@@ -807,7 +791,7 @@ class PDFDancer:
|
|
|
807
791
|
token: Optional[str] = None,
|
|
808
792
|
base_url: Optional[str] = None,
|
|
809
793
|
timeout: float = 30.0,
|
|
810
|
-
|
|
794
|
+
max_attempts: int = DEFAULT_MAX_ATTEMPTS,
|
|
811
795
|
retry_backoff_factor: float = DEFAULT_RETRY_BACKOFF_FACTOR,
|
|
812
796
|
) -> "PDFDancer":
|
|
813
797
|
"""
|
|
@@ -826,15 +810,16 @@ class PDFDancer:
|
|
|
826
810
|
base_url: Override for the API base URL; falls back to `PDFDANCER_BASE_URL`
|
|
827
811
|
or defaults to `https://api.pdfdancer.com`.
|
|
828
812
|
timeout: HTTP read timeout in seconds.
|
|
829
|
-
|
|
830
|
-
|
|
831
|
-
retry_backoff_factor: Base multiplier for exponential backoff delays (default:
|
|
832
|
-
Delay calculation:
|
|
833
|
-
Examples:
|
|
813
|
+
max_attempts: Maximum number of total attempts (default: 3).
|
|
814
|
+
The initial request counts as one attempt.
|
|
815
|
+
retry_backoff_factor: Base multiplier for exponential backoff delays (default: 2.0).
|
|
816
|
+
Delay calculation: initial_delay * (retry_backoff_factor ** attempt_number).
|
|
817
|
+
Examples: 2.0 → delays of 1s, 2s, 4s; 3.0 → delays of 1s, 3s, 9s.
|
|
834
818
|
|
|
835
819
|
Returns:
|
|
836
820
|
A ready-to-use `PDFDancer` client instance.
|
|
837
821
|
"""
|
|
822
|
+
_validate_max_attempts(max_attempts)
|
|
838
823
|
resolved_token = cls._resolve_token(token)
|
|
839
824
|
resolved_base_url = cls._resolve_base_url(base_url)
|
|
840
825
|
|
|
@@ -847,13 +832,12 @@ class PDFDancer:
|
|
|
847
832
|
pdf_data,
|
|
848
833
|
resolved_base_url,
|
|
849
834
|
timeout,
|
|
850
|
-
|
|
835
|
+
max_attempts,
|
|
851
836
|
retry_backoff_factor,
|
|
852
837
|
)
|
|
853
838
|
|
|
854
839
|
@classmethod
|
|
855
|
-
def _resolve_base_url(cls, base_url: Optional[str]) ->
|
|
856
|
-
_load_env()
|
|
840
|
+
def _resolve_base_url(cls, base_url: Optional[str]) -> str:
|
|
857
841
|
env_base_url = os.getenv("PDFDANCER_BASE_URL")
|
|
858
842
|
resolved_base_url = base_url or (
|
|
859
843
|
env_base_url.strip() if env_base_url and env_base_url.strip() else None
|
|
@@ -865,7 +849,7 @@ class PDFDancer:
|
|
|
865
849
|
@classmethod
|
|
866
850
|
def _obtain_anonymous_token(cls, base_url: str, timeout: float = 30.0) -> str:
|
|
867
851
|
"""
|
|
868
|
-
Obtain an anonymous token from the /keys/anon endpoint.
|
|
852
|
+
Obtain an anonymous token from the /v2/keys/anon endpoint.
|
|
869
853
|
|
|
870
854
|
Args:
|
|
871
855
|
base_url: Base URL of the PDFDancer API server
|
|
@@ -880,101 +864,63 @@ class PDFDancer:
|
|
|
880
864
|
"""
|
|
881
865
|
# Create temporary client without authentication
|
|
882
866
|
temp_client = httpx.Client(http2=True, verify=not DISABLE_SSL_VERIFY)
|
|
883
|
-
|
|
884
|
-
retry_backoff_factor =
|
|
867
|
+
max_attempts = DEFAULT_MAX_ATTEMPTS
|
|
868
|
+
retry_backoff_factor = DEFAULT_RETRY_BACKOFF_FACTOR
|
|
885
869
|
|
|
886
870
|
try:
|
|
887
|
-
last_error: Optional[Exception] = None
|
|
888
|
-
attempt = 0
|
|
889
871
|
|
|
890
|
-
|
|
891
|
-
|
|
892
|
-
|
|
893
|
-
|
|
894
|
-
|
|
895
|
-
|
|
896
|
-
|
|
897
|
-
|
|
898
|
-
|
|
899
|
-
|
|
900
|
-
|
|
901
|
-
)
|
|
872
|
+
def request_token() -> httpx.Response:
|
|
873
|
+
headers = {
|
|
874
|
+
"X-Fingerprint": Fingerprint.generate(),
|
|
875
|
+
"X-PDFDancer-Client": CLIENT_HEADER_VALUE,
|
|
876
|
+
"X-API-VERSION": "2",
|
|
877
|
+
}
|
|
878
|
+
return temp_client.post(
|
|
879
|
+
cls._cleanup_url_path(base_url, "/keys/anon"),
|
|
880
|
+
headers=headers,
|
|
881
|
+
timeout=timeout if timeout > 0 else None,
|
|
882
|
+
)
|
|
902
883
|
|
|
903
|
-
|
|
904
|
-
|
|
884
|
+
response = _execute_request_with_retries(
|
|
885
|
+
request_callable=request_token,
|
|
886
|
+
operation="POST /keys/anon",
|
|
887
|
+
max_attempts=max_attempts,
|
|
888
|
+
retry_backoff_factor=retry_backoff_factor,
|
|
889
|
+
)
|
|
905
890
|
|
|
906
|
-
|
|
907
|
-
|
|
908
|
-
return token_data["token"]
|
|
909
|
-
else:
|
|
910
|
-
raise HttpClientException(
|
|
911
|
-
"Invalid anonymous token response format"
|
|
912
|
-
)
|
|
891
|
+
response.raise_for_status()
|
|
892
|
+
token_data = response.json()
|
|
913
893
|
|
|
914
|
-
|
|
915
|
-
|
|
916
|
-
|
|
917
|
-
|
|
918
|
-
|
|
919
|
-
|
|
920
|
-
|
|
921
|
-
|
|
922
|
-
|
|
923
|
-
|
|
924
|
-
|
|
925
|
-
|
|
926
|
-
f"Rate limit (429) on POST /keys/anon - retrying in {delay}s "
|
|
927
|
-
f"(attempt {attempt + 1}/{max_retries})",
|
|
928
|
-
file=sys.stderr,
|
|
929
|
-
)
|
|
930
|
-
if DEBUG:
|
|
931
|
-
print(
|
|
932
|
-
f"{time.time()}|POST /keys/anon - Rate limit exceeded (429), "
|
|
933
|
-
f"retrying in {delay}s (attempt {attempt + 1}/{max_retries})"
|
|
934
|
-
)
|
|
935
|
-
time.sleep(delay)
|
|
936
|
-
attempt += 1
|
|
937
|
-
continue
|
|
938
|
-
|
|
939
|
-
# Raise RateLimitException for 429 after exhausting retries
|
|
940
|
-
if e.response.status_code == 429:
|
|
941
|
-
retry_after = _get_retry_after_delay(e.response)
|
|
942
|
-
print(
|
|
943
|
-
"Rate limit (429) on POST /keys/anon - max retries exhausted",
|
|
944
|
-
file=sys.stderr,
|
|
945
|
-
)
|
|
946
|
-
raise RateLimitException(
|
|
947
|
-
"Rate limit exceeded when obtaining anonymous token",
|
|
948
|
-
retry_after=retry_after,
|
|
949
|
-
response=e.response,
|
|
950
|
-
) from None
|
|
951
|
-
|
|
952
|
-
# Other HTTP status errors
|
|
953
|
-
raise HttpClientException(
|
|
954
|
-
f"Failed to obtain anonymous token: HTTP {e.response.status_code}",
|
|
955
|
-
response=e.response,
|
|
956
|
-
cause=e,
|
|
957
|
-
) from None
|
|
958
|
-
except httpx.RequestError as e:
|
|
959
|
-
last_error = e
|
|
960
|
-
raise HttpClientException(
|
|
961
|
-
f"Failed to obtain anonymous token: {str(e)}",
|
|
962
|
-
response=None,
|
|
963
|
-
cause=e,
|
|
964
|
-
) from None
|
|
965
|
-
|
|
966
|
-
# Should not reach here, but handle just in case
|
|
967
|
-
if last_error:
|
|
968
|
-
raise HttpClientException(
|
|
969
|
-
f"Failed to obtain anonymous token after {max_retries + 1} attempts: {str(last_error)}",
|
|
970
|
-
response=None,
|
|
971
|
-
cause=last_error,
|
|
972
|
-
) from None
|
|
973
|
-
else:
|
|
974
|
-
raise HttpClientException(
|
|
975
|
-
f"Failed to obtain anonymous token after {max_retries + 1} attempts",
|
|
976
|
-
response=None,
|
|
894
|
+
# Extract token from response (matches Java AnonTokenResponse structure)
|
|
895
|
+
if isinstance(token_data, dict) and "token" in token_data:
|
|
896
|
+
return cast(str, token_data["token"])
|
|
897
|
+
|
|
898
|
+
raise HttpClientException("Invalid anonymous token response format")
|
|
899
|
+
|
|
900
|
+
except httpx.HTTPStatusError as e:
|
|
901
|
+
if e.response.status_code == 429:
|
|
902
|
+
retry_after = _get_retry_after_delay(e.response)
|
|
903
|
+
print(
|
|
904
|
+
"Rate limit (429) on POST /keys/anon - maximum attempts exhausted",
|
|
905
|
+
file=sys.stderr,
|
|
977
906
|
)
|
|
907
|
+
raise RateLimitException(
|
|
908
|
+
"Rate limit exceeded when obtaining anonymous token",
|
|
909
|
+
retry_after=retry_after,
|
|
910
|
+
response=e.response,
|
|
911
|
+
) from None
|
|
912
|
+
|
|
913
|
+
raise HttpClientException(
|
|
914
|
+
f"Failed to obtain anonymous token: HTTP {e.response.status_code}",
|
|
915
|
+
response=e.response,
|
|
916
|
+
cause=e,
|
|
917
|
+
) from None
|
|
918
|
+
except httpx.RequestError as e:
|
|
919
|
+
raise HttpClientException(
|
|
920
|
+
f"Failed to obtain anonymous token: {str(e)}",
|
|
921
|
+
response=None,
|
|
922
|
+
cause=e,
|
|
923
|
+
) from None
|
|
978
924
|
finally:
|
|
979
925
|
temp_client.close()
|
|
980
926
|
|
|
@@ -988,7 +934,6 @@ class PDFDancer:
|
|
|
988
934
|
1. PDFDANCER_API_TOKEN (preferred)
|
|
989
935
|
2. PDFDANCER_TOKEN (legacy)
|
|
990
936
|
"""
|
|
991
|
-
_load_env()
|
|
992
937
|
resolved_token = token.strip() if token and token.strip() else None
|
|
993
938
|
if resolved_token is None:
|
|
994
939
|
# Check PDFDANCER_API_TOKEN first (preferred), then PDFDANCER_TOKEN (legacy)
|
|
@@ -1007,7 +952,7 @@ class PDFDancer:
|
|
|
1007
952
|
page_size: Optional[Union[PageSize, str, Mapping[str, Any]]] = None,
|
|
1008
953
|
orientation: Optional[Union[Orientation, str]] = None,
|
|
1009
954
|
initial_page_count: int = 1,
|
|
1010
|
-
|
|
955
|
+
max_attempts: int = DEFAULT_MAX_ATTEMPTS,
|
|
1011
956
|
retry_backoff_factor: float = DEFAULT_RETRY_BACKOFF_FACTOR,
|
|
1012
957
|
) -> "PDFDancer":
|
|
1013
958
|
"""
|
|
@@ -1029,15 +974,16 @@ class PDFDancer:
|
|
|
1029
974
|
mapping with `width`/`height` values.
|
|
1030
975
|
orientation: Page orientation (default: PORTRAIT). Can be Orientation enum or string.
|
|
1031
976
|
initial_page_count: Number of initial blank pages (default: 1).
|
|
1032
|
-
|
|
1033
|
-
|
|
1034
|
-
retry_backoff_factor: Base multiplier for exponential backoff delays (default:
|
|
1035
|
-
Delay calculation:
|
|
1036
|
-
Examples:
|
|
977
|
+
max_attempts: Maximum number of total attempts (default: 3).
|
|
978
|
+
The initial request counts as one attempt.
|
|
979
|
+
retry_backoff_factor: Base multiplier for exponential backoff delays (default: 2.0).
|
|
980
|
+
Delay calculation: initial_delay * (retry_backoff_factor ** attempt_number).
|
|
981
|
+
Examples: 2.0 → delays of 1s, 2s, 4s; 3.0 → delays of 1s, 3s, 9s.
|
|
1037
982
|
|
|
1038
983
|
Returns:
|
|
1039
984
|
A ready-to-use `PDFDancer` client instance with a blank PDF.
|
|
1040
985
|
"""
|
|
986
|
+
_validate_max_attempts(max_attempts)
|
|
1041
987
|
resolved_token = cls._resolve_token(token)
|
|
1042
988
|
resolved_base_url = cls._resolve_base_url(base_url)
|
|
1043
989
|
|
|
@@ -1055,7 +1001,7 @@ class PDFDancer:
|
|
|
1055
1001
|
instance._token = resolved_token.strip()
|
|
1056
1002
|
instance._base_url = resolved_base_url.rstrip("/")
|
|
1057
1003
|
instance._read_timeout = timeout
|
|
1058
|
-
instance.
|
|
1004
|
+
instance._max_attempts = max_attempts
|
|
1059
1005
|
instance._retry_backoff_factor = retry_backoff_factor
|
|
1060
1006
|
|
|
1061
1007
|
# Create HTTP client for connection reuse with HTTP/2 support
|
|
@@ -1064,6 +1010,7 @@ class PDFDancer:
|
|
|
1064
1010
|
headers={
|
|
1065
1011
|
"Authorization": f"Bearer {instance._token}",
|
|
1066
1012
|
"X-PDFDancer-Client": CLIENT_HEADER_VALUE,
|
|
1013
|
+
"X-API-VERSION": "2",
|
|
1067
1014
|
},
|
|
1068
1015
|
verify=not DISABLE_SSL_VERIFY,
|
|
1069
1016
|
)
|
|
@@ -1090,7 +1037,7 @@ class PDFDancer:
|
|
|
1090
1037
|
pdf_data: Union[bytes, Path, str, BinaryIO],
|
|
1091
1038
|
base_url: str,
|
|
1092
1039
|
read_timeout: float = 0,
|
|
1093
|
-
|
|
1040
|
+
max_attempts: int = DEFAULT_MAX_ATTEMPTS,
|
|
1094
1041
|
retry_backoff_factor: float = DEFAULT_RETRY_BACKOFF_FACTOR,
|
|
1095
1042
|
):
|
|
1096
1043
|
"""
|
|
@@ -1103,11 +1050,11 @@ class PDFDancer:
|
|
|
1103
1050
|
pdf_data: PDF file data as bytes, Path, filename string, or file-like object
|
|
1104
1051
|
base_url: Base URL of the PDFDancer API server
|
|
1105
1052
|
read_timeout: Timeout in seconds for HTTP requests (default: 30.0)
|
|
1106
|
-
|
|
1107
|
-
|
|
1108
|
-
retry_backoff_factor: Base multiplier for exponential backoff delays (default:
|
|
1109
|
-
Delay calculation:
|
|
1110
|
-
Examples:
|
|
1053
|
+
max_attempts: Maximum number of total attempts (default: 3).
|
|
1054
|
+
The initial request counts as one attempt.
|
|
1055
|
+
retry_backoff_factor: Base multiplier for exponential backoff delays (default: 2.0).
|
|
1056
|
+
Delay calculation: initial_delay * (retry_backoff_factor ** attempt_number).
|
|
1057
|
+
Examples: 2.0 → delays of 1s, 2s, 4s; 3.0 → delays of 1s, 3s, 9s.
|
|
1111
1058
|
|
|
1112
1059
|
Raises:
|
|
1113
1060
|
ValidationException: If token is empty or PDF data is invalid
|
|
@@ -1117,15 +1064,16 @@ class PDFDancer:
|
|
|
1117
1064
|
# Strict validation like Java client
|
|
1118
1065
|
if not token or not token.strip():
|
|
1119
1066
|
raise ValidationException("Authentication token cannot be null or empty")
|
|
1067
|
+
_validate_max_attempts(max_attempts)
|
|
1120
1068
|
|
|
1121
1069
|
self._token = token.strip()
|
|
1122
1070
|
self._base_url = base_url.rstrip("/")
|
|
1123
1071
|
self._read_timeout = read_timeout
|
|
1124
|
-
self.
|
|
1072
|
+
self._max_attempts = max_attempts
|
|
1125
1073
|
self._retry_backoff_factor = retry_backoff_factor
|
|
1126
1074
|
|
|
1127
1075
|
# Process PDF data with validation
|
|
1128
|
-
self._pdf_bytes = self._process_pdf_data(pdf_data)
|
|
1076
|
+
self._pdf_bytes: Optional[bytes] = self._process_pdf_data(pdf_data)
|
|
1129
1077
|
|
|
1130
1078
|
# Create HTTP client for connection reuse with HTTP/2 support
|
|
1131
1079
|
self._client = httpx.Client(
|
|
@@ -1133,6 +1081,7 @@ class PDFDancer:
|
|
|
1133
1081
|
headers={
|
|
1134
1082
|
"Authorization": f"Bearer {self._token}",
|
|
1135
1083
|
"X-PDFDancer-Client": CLIENT_HEADER_VALUE,
|
|
1084
|
+
"X-API-VERSION": "2",
|
|
1136
1085
|
},
|
|
1137
1086
|
verify=not DISABLE_SSL_VERIFY,
|
|
1138
1087
|
)
|
|
@@ -1214,7 +1163,7 @@ class PDFDancer:
|
|
|
1214
1163
|
|
|
1215
1164
|
# Check for top-level message
|
|
1216
1165
|
if "message" in error_data:
|
|
1217
|
-
return error_data["message"]
|
|
1166
|
+
return cast(str, error_data["message"])
|
|
1218
1167
|
|
|
1219
1168
|
# Fallback to response content
|
|
1220
1169
|
return response.text or f"HTTP {response.status_code}"
|
|
@@ -1242,18 +1191,24 @@ class PDFDancer:
|
|
|
1242
1191
|
@staticmethod
|
|
1243
1192
|
def _cleanup_url_path(base_url: str, path: str) -> str:
|
|
1244
1193
|
"""
|
|
1245
|
-
Combine base_url and path, ensuring no double slashes.
|
|
1194
|
+
Combine base_url and API path, ensuring no double slashes.
|
|
1246
1195
|
|
|
1247
1196
|
Args:
|
|
1248
1197
|
base_url: Base URL (may or may not have trailing slash)
|
|
1249
1198
|
path: Path segment (may or may not have leading slash)
|
|
1250
1199
|
|
|
1251
1200
|
Returns:
|
|
1252
|
-
Combined URL with no double slashes
|
|
1201
|
+
Combined URL with the v2 API prefix and no double slashes
|
|
1253
1202
|
"""
|
|
1254
1203
|
base = base_url.rstrip("/")
|
|
1255
1204
|
path = path.lstrip("/")
|
|
1256
|
-
|
|
1205
|
+
|
|
1206
|
+
if base.endswith(API_PATH_PREFIX) or path.startswith(
|
|
1207
|
+
API_PATH_PREFIX.strip("/") + "/"
|
|
1208
|
+
):
|
|
1209
|
+
return f"{base}/{path}"
|
|
1210
|
+
|
|
1211
|
+
return f"{base}{API_PATH_PREFIX}/{path}"
|
|
1257
1212
|
|
|
1258
1213
|
def _create_session(self) -> str:
|
|
1259
1214
|
"""
|
|
@@ -1261,163 +1216,109 @@ class PDFDancer:
|
|
|
1261
1216
|
"""
|
|
1262
1217
|
import uuid
|
|
1263
1218
|
|
|
1264
|
-
|
|
1265
|
-
|
|
1219
|
+
# Build multipart body manually to avoid base64 encoding and enable compression
|
|
1220
|
+
# httpx by default may add Content-Transfer-Encoding: base64 which the server rejects
|
|
1221
|
+
boundary = uuid.uuid4().hex
|
|
1266
1222
|
|
|
1267
|
-
|
|
1268
|
-
|
|
1269
|
-
|
|
1270
|
-
|
|
1271
|
-
|
|
1272
|
-
|
|
1273
|
-
|
|
1274
|
-
|
|
1275
|
-
|
|
1276
|
-
|
|
1277
|
-
|
|
1223
|
+
# Build multipart body with binary (not base64) encoding
|
|
1224
|
+
body_parts = []
|
|
1225
|
+
body_parts.append(f"--{boundary}\r\n".encode("utf-8"))
|
|
1226
|
+
body_parts.append(
|
|
1227
|
+
b'Content-Disposition: form-data; name="pdf"; filename="document.pdf"\r\n'
|
|
1228
|
+
)
|
|
1229
|
+
body_parts.append(b"Content-Type: application/pdf\r\n")
|
|
1230
|
+
body_parts.append(b"\r\n") # End of headers, no Content-Transfer-Encoding
|
|
1231
|
+
body_parts.append(cast(bytes, self._pdf_bytes))
|
|
1232
|
+
body_parts.append(b"\r\n")
|
|
1233
|
+
body_parts.append(f"--{boundary}--\r\n".encode("utf-8"))
|
|
1234
|
+
|
|
1235
|
+
uncompressed_body = b"".join(body_parts)
|
|
1236
|
+
compressed_body = gzip.compress(uncompressed_body)
|
|
1237
|
+
|
|
1238
|
+
original_size = len(uncompressed_body)
|
|
1239
|
+
compressed_size = len(compressed_body)
|
|
1240
|
+
compression_ratio = (
|
|
1241
|
+
(1 - compressed_size / original_size) * 100 if original_size > 0 else 0
|
|
1242
|
+
)
|
|
1243
|
+
|
|
1244
|
+
def log_create_attempt(attempt: int) -> None:
|
|
1245
|
+
if DEBUG:
|
|
1246
|
+
retry_info = (
|
|
1247
|
+
f" (attempt {attempt + 1}/{self._max_attempts})"
|
|
1248
|
+
if attempt > 0
|
|
1249
|
+
else ""
|
|
1278
1250
|
)
|
|
1279
|
-
|
|
1280
|
-
|
|
1281
|
-
|
|
1282
|
-
|
|
1283
|
-
body_parts.append(self._pdf_bytes)
|
|
1284
|
-
body_parts.append(b"\r\n")
|
|
1285
|
-
body_parts.append(f"--{boundary}--\r\n".encode("utf-8"))
|
|
1286
|
-
|
|
1287
|
-
uncompressed_body = b"".join(body_parts)
|
|
1288
|
-
|
|
1289
|
-
# Compress entire request body using gzip
|
|
1290
|
-
compressed_body = gzip.compress(uncompressed_body)
|
|
1291
|
-
|
|
1292
|
-
original_size = len(uncompressed_body)
|
|
1293
|
-
compressed_size = len(compressed_body)
|
|
1294
|
-
compression_ratio = (
|
|
1295
|
-
(1 - compressed_size / original_size) * 100
|
|
1296
|
-
if original_size > 0
|
|
1297
|
-
else 0
|
|
1251
|
+
print(
|
|
1252
|
+
f"{time.time()}|POST /session/create{retry_info} - original size: {original_size} bytes, "
|
|
1253
|
+
f"compressed size: {compressed_size} bytes, "
|
|
1254
|
+
f"compression: {compression_ratio:.1f}%"
|
|
1298
1255
|
)
|
|
1299
1256
|
|
|
1300
|
-
|
|
1301
|
-
|
|
1302
|
-
|
|
1303
|
-
|
|
1304
|
-
|
|
1305
|
-
|
|
1306
|
-
print(
|
|
1307
|
-
f"{time.time()}|POST /session/create{retry_info} - original size: {original_size} bytes, "
|
|
1308
|
-
f"compressed size: {compressed_size} bytes, "
|
|
1309
|
-
f"compression: {compression_ratio:.1f}%"
|
|
1310
|
-
)
|
|
1311
|
-
|
|
1312
|
-
headers = {
|
|
1313
|
-
"X-Generated-At": _generate_timestamp(),
|
|
1314
|
-
"Content-Type": f"multipart/form-data; boundary={boundary}",
|
|
1315
|
-
"Content-Encoding": "gzip",
|
|
1316
|
-
}
|
|
1317
|
-
|
|
1318
|
-
response = self._client.post(
|
|
1319
|
-
self._cleanup_url_path(self._base_url, "/session/create"),
|
|
1320
|
-
content=compressed_body,
|
|
1321
|
-
headers=headers,
|
|
1322
|
-
timeout=self._read_timeout if self._read_timeout > 0 else None,
|
|
1323
|
-
)
|
|
1257
|
+
def request_session() -> httpx.Response:
|
|
1258
|
+
headers = {
|
|
1259
|
+
"X-Generated-At": _generate_timestamp(),
|
|
1260
|
+
"Content-Type": f"multipart/form-data; boundary={boundary}",
|
|
1261
|
+
"Content-Encoding": "gzip",
|
|
1262
|
+
}
|
|
1324
1263
|
|
|
1325
|
-
|
|
1326
|
-
|
|
1327
|
-
|
|
1328
|
-
|
|
1329
|
-
|
|
1264
|
+
return self._client.post(
|
|
1265
|
+
self._cleanup_url_path(self._base_url, "/session/create"),
|
|
1266
|
+
content=compressed_body,
|
|
1267
|
+
headers=headers,
|
|
1268
|
+
timeout=self._read_timeout if self._read_timeout > 0 else None,
|
|
1269
|
+
)
|
|
1330
1270
|
|
|
1331
|
-
|
|
1332
|
-
|
|
1333
|
-
|
|
1334
|
-
|
|
1271
|
+
try:
|
|
1272
|
+
response = _execute_request_with_retries(
|
|
1273
|
+
request_callable=request_session,
|
|
1274
|
+
operation="POST /session/create",
|
|
1275
|
+
max_attempts=self._max_attempts,
|
|
1276
|
+
retry_backoff_factor=self._retry_backoff_factor,
|
|
1277
|
+
pre_request_hook=log_create_attempt,
|
|
1278
|
+
)
|
|
1335
1279
|
|
|
1336
|
-
|
|
1337
|
-
|
|
1280
|
+
response_size = len(response.content)
|
|
1281
|
+
if DEBUG:
|
|
1282
|
+
print(
|
|
1283
|
+
f"{time.time()}|POST /session/create - response size: {response_size} bytes"
|
|
1284
|
+
)
|
|
1338
1285
|
|
|
1339
|
-
|
|
1286
|
+
_log_generated_at_header(response, "POST", "/session/create")
|
|
1287
|
+
self._handle_authentication_error(response)
|
|
1288
|
+
response.raise_for_status()
|
|
1289
|
+
session_id = response.text.strip()
|
|
1340
1290
|
|
|
1341
|
-
|
|
1342
|
-
|
|
1343
|
-
if e.response.status_code == 429 and attempt < self._max_retries:
|
|
1344
|
-
retry_after = _get_retry_after_delay(e.response)
|
|
1345
|
-
if retry_after is not None:
|
|
1346
|
-
delay = retry_after
|
|
1347
|
-
else:
|
|
1348
|
-
# Use exponential backoff if no Retry-After header
|
|
1349
|
-
delay = self._retry_backoff_factor * (2**attempt)
|
|
1291
|
+
if not session_id:
|
|
1292
|
+
raise SessionException("Server returned empty session ID")
|
|
1350
1293
|
|
|
1351
|
-
|
|
1352
|
-
print(
|
|
1353
|
-
f"Rate limit (429) on POST /session/create - retrying in {delay}s "
|
|
1354
|
-
f"(attempt {attempt + 1}/{self._max_retries})",
|
|
1355
|
-
file=sys.stderr,
|
|
1356
|
-
)
|
|
1357
|
-
if DEBUG:
|
|
1358
|
-
print(
|
|
1359
|
-
f"{time.time()}|POST /session/create - Rate limit exceeded (429), "
|
|
1360
|
-
f"retrying in {delay}s (attempt {attempt + 1}/{self._max_retries})"
|
|
1361
|
-
)
|
|
1362
|
-
time.sleep(delay)
|
|
1363
|
-
attempt += 1
|
|
1364
|
-
continue
|
|
1294
|
+
return session_id
|
|
1365
1295
|
|
|
1366
|
-
|
|
1367
|
-
|
|
1368
|
-
|
|
1296
|
+
except httpx.HTTPStatusError as e:
|
|
1297
|
+
self._handle_authentication_error(e.response)
|
|
1298
|
+
error_message = self._extract_error_message(e.response)
|
|
1369
1299
|
|
|
1370
|
-
|
|
1371
|
-
|
|
1372
|
-
|
|
1373
|
-
|
|
1374
|
-
|
|
1375
|
-
|
|
1376
|
-
|
|
1377
|
-
|
|
1378
|
-
|
|
1379
|
-
|
|
1380
|
-
response=e.response,
|
|
1381
|
-
) from None
|
|
1382
|
-
|
|
1383
|
-
raise HttpClientException(
|
|
1384
|
-
f"Failed to create session: {error_message}",
|
|
1300
|
+
# Raise RateLimitException for 429 after retry attempts are exhausted
|
|
1301
|
+
if e.response.status_code == 429:
|
|
1302
|
+
retry_after = _get_retry_after_delay(e.response)
|
|
1303
|
+
print(
|
|
1304
|
+
"Rate limit (429) on POST /session/create - maximum attempts exhausted",
|
|
1305
|
+
file=sys.stderr,
|
|
1306
|
+
)
|
|
1307
|
+
raise RateLimitException(
|
|
1308
|
+
f"Rate limit exceeded: {error_message}",
|
|
1309
|
+
retry_after=retry_after,
|
|
1385
1310
|
response=e.response,
|
|
1386
|
-
cause=e,
|
|
1387
1311
|
) from None
|
|
1388
|
-
except httpx.RequestError as e:
|
|
1389
|
-
last_error = e
|
|
1390
|
-
|
|
1391
|
-
# Check if this is a retryable error
|
|
1392
|
-
if _is_retryable_error(e) and attempt < self._max_retries:
|
|
1393
|
-
# Calculate exponential backoff delay
|
|
1394
|
-
delay = self._retry_backoff_factor * (2**attempt)
|
|
1395
|
-
if DEBUG:
|
|
1396
|
-
print(
|
|
1397
|
-
f"{time.time()}|POST /session/create - Retryable error: {str(e)}, "
|
|
1398
|
-
f"retrying in {delay}s (attempt {attempt + 1}/{self._max_retries})"
|
|
1399
|
-
)
|
|
1400
|
-
time.sleep(delay)
|
|
1401
|
-
attempt += 1
|
|
1402
|
-
continue
|
|
1403
|
-
else:
|
|
1404
|
-
# Non-retryable error or exhausted retries
|
|
1405
|
-
raise HttpClientException(
|
|
1406
|
-
f"Failed to create session: {str(e)}", response=None, cause=e
|
|
1407
|
-
) from None
|
|
1408
1312
|
|
|
1409
|
-
# Should not reach here, but handle just in case
|
|
1410
|
-
if last_error:
|
|
1411
1313
|
raise HttpClientException(
|
|
1412
|
-
f"Failed to create session
|
|
1413
|
-
response=
|
|
1414
|
-
cause=
|
|
1314
|
+
f"Failed to create session: {error_message}",
|
|
1315
|
+
response=e.response,
|
|
1316
|
+
cause=e,
|
|
1415
1317
|
) from None
|
|
1416
|
-
|
|
1318
|
+
except httpx.RequestError as e:
|
|
1417
1319
|
raise HttpClientException(
|
|
1418
|
-
f"Failed to create session
|
|
1419
|
-
|
|
1420
|
-
)
|
|
1320
|
+
f"Failed to create session: {str(e)}", response=None, cause=e
|
|
1321
|
+
) from None
|
|
1421
1322
|
|
|
1422
1323
|
def _create_blank_pdf_session(
|
|
1423
1324
|
self,
|
|
@@ -1442,7 +1343,7 @@ class PDFDancer:
|
|
|
1442
1343
|
HttpClientException: If HTTP communication fails
|
|
1443
1344
|
"""
|
|
1444
1345
|
# Build request payload (outside retry loop since validation should only happen once)
|
|
1445
|
-
request_data = {}
|
|
1346
|
+
request_data: dict[str, Any] = {}
|
|
1446
1347
|
|
|
1447
1348
|
# Handle page_size - convert to type-safe object with dimensions
|
|
1448
1349
|
if page_size is not None:
|
|
@@ -1469,286 +1370,192 @@ class PDFDancer:
|
|
|
1469
1370
|
raise ValidationException(
|
|
1470
1371
|
f"Initial page count must be at least 1, got {initial_page_count}"
|
|
1471
1372
|
)
|
|
1472
|
-
request_data["initialPageCount"] = initial_page_count
|
|
1473
|
-
|
|
1474
|
-
|
|
1475
|
-
attempt = 0
|
|
1476
|
-
|
|
1477
|
-
while attempt <= self._max_retries:
|
|
1478
|
-
try:
|
|
1479
|
-
request_body = json.dumps(request_data)
|
|
1480
|
-
request_size = len(request_body.encode("utf-8"))
|
|
1481
|
-
if DEBUG:
|
|
1482
|
-
retry_info = (
|
|
1483
|
-
f" (attempt {attempt + 1}/{self._max_retries + 1})"
|
|
1484
|
-
if attempt > 0
|
|
1485
|
-
else ""
|
|
1486
|
-
)
|
|
1487
|
-
print(
|
|
1488
|
-
f"{time.time()}|POST /session/new{retry_info} - request size: {request_size} bytes"
|
|
1489
|
-
)
|
|
1373
|
+
request_data["initialPageCount"] = int(initial_page_count)
|
|
1374
|
+
request_body = json.dumps(request_data)
|
|
1375
|
+
request_size = len(request_body.encode("utf-8"))
|
|
1490
1376
|
|
|
1491
|
-
|
|
1492
|
-
|
|
1493
|
-
|
|
1494
|
-
|
|
1495
|
-
|
|
1496
|
-
|
|
1497
|
-
|
|
1498
|
-
|
|
1499
|
-
|
|
1377
|
+
def log_blank_pdf_attempt(attempt: int) -> None:
|
|
1378
|
+
if DEBUG:
|
|
1379
|
+
retry_info = (
|
|
1380
|
+
f" (attempt {attempt + 1}/{self._max_attempts})"
|
|
1381
|
+
if attempt > 0
|
|
1382
|
+
else ""
|
|
1383
|
+
)
|
|
1384
|
+
print(
|
|
1385
|
+
f"{time.time()}|POST /session/new{retry_info} - request size: {request_size} bytes"
|
|
1500
1386
|
)
|
|
1501
1387
|
|
|
1502
|
-
|
|
1503
|
-
|
|
1504
|
-
|
|
1505
|
-
|
|
1506
|
-
|
|
1388
|
+
def request_blank_pdf() -> httpx.Response:
|
|
1389
|
+
headers = {
|
|
1390
|
+
"Content-Type": "application/json",
|
|
1391
|
+
"X-Generated-At": _generate_timestamp(),
|
|
1392
|
+
}
|
|
1393
|
+
return self._client.post(
|
|
1394
|
+
self._cleanup_url_path(self._base_url, "/session/new"),
|
|
1395
|
+
json=request_data,
|
|
1396
|
+
headers=headers,
|
|
1397
|
+
timeout=self._read_timeout if self._read_timeout > 0 else None,
|
|
1398
|
+
)
|
|
1507
1399
|
|
|
1508
|
-
|
|
1509
|
-
|
|
1510
|
-
|
|
1511
|
-
|
|
1400
|
+
try:
|
|
1401
|
+
response = _execute_request_with_retries(
|
|
1402
|
+
request_callable=request_blank_pdf,
|
|
1403
|
+
operation="POST /session/new",
|
|
1404
|
+
max_attempts=self._max_attempts,
|
|
1405
|
+
retry_backoff_factor=self._retry_backoff_factor,
|
|
1406
|
+
pre_request_hook=log_blank_pdf_attempt,
|
|
1407
|
+
)
|
|
1512
1408
|
|
|
1513
|
-
|
|
1514
|
-
|
|
1409
|
+
response_size = len(response.content)
|
|
1410
|
+
if DEBUG:
|
|
1411
|
+
print(
|
|
1412
|
+
f"{time.time()}|POST /session/new - response size: {response_size} bytes"
|
|
1413
|
+
)
|
|
1515
1414
|
|
|
1516
|
-
|
|
1415
|
+
_log_generated_at_header(response, "POST", "/session/new")
|
|
1416
|
+
self._handle_authentication_error(response)
|
|
1417
|
+
response.raise_for_status()
|
|
1418
|
+
session_id = response.text.strip()
|
|
1517
1419
|
|
|
1518
|
-
|
|
1519
|
-
|
|
1520
|
-
if e.response.status_code == 429 and attempt < self._max_retries:
|
|
1521
|
-
retry_after = _get_retry_after_delay(e.response)
|
|
1522
|
-
if retry_after is not None:
|
|
1523
|
-
delay = retry_after
|
|
1524
|
-
else:
|
|
1525
|
-
# Use exponential backoff if no Retry-After header
|
|
1526
|
-
delay = self._retry_backoff_factor * (2**attempt)
|
|
1420
|
+
if not session_id:
|
|
1421
|
+
raise SessionException("Server returned empty session ID")
|
|
1527
1422
|
|
|
1528
|
-
|
|
1529
|
-
print(
|
|
1530
|
-
f"Rate limit (429) on POST /session/new - retrying in {delay}s "
|
|
1531
|
-
f"(attempt {attempt + 1}/{self._max_retries})",
|
|
1532
|
-
file=sys.stderr,
|
|
1533
|
-
)
|
|
1534
|
-
if DEBUG:
|
|
1535
|
-
print(
|
|
1536
|
-
f"{time.time()}|POST /session/new - Rate limit exceeded (429), "
|
|
1537
|
-
f"retrying in {delay}s (attempt {attempt + 1}/{self._max_retries})"
|
|
1538
|
-
)
|
|
1539
|
-
time.sleep(delay)
|
|
1540
|
-
attempt += 1
|
|
1541
|
-
continue
|
|
1423
|
+
return session_id
|
|
1542
1424
|
|
|
1543
|
-
|
|
1544
|
-
|
|
1545
|
-
|
|
1425
|
+
except httpx.HTTPStatusError as e:
|
|
1426
|
+
self._handle_authentication_error(e.response)
|
|
1427
|
+
error_message = self._extract_error_message(e.response)
|
|
1546
1428
|
|
|
1547
|
-
|
|
1548
|
-
|
|
1549
|
-
|
|
1550
|
-
|
|
1551
|
-
|
|
1552
|
-
|
|
1553
|
-
|
|
1554
|
-
|
|
1555
|
-
|
|
1556
|
-
|
|
1557
|
-
response=e.response,
|
|
1558
|
-
) from None
|
|
1559
|
-
|
|
1560
|
-
raise HttpClientException(
|
|
1561
|
-
f"Failed to create blank PDF session: {error_message}",
|
|
1429
|
+
# Raise RateLimitException for 429 after retry attempts are exhausted
|
|
1430
|
+
if e.response.status_code == 429:
|
|
1431
|
+
retry_after = _get_retry_after_delay(e.response)
|
|
1432
|
+
print(
|
|
1433
|
+
"Rate limit (429) on POST /session/new - maximum attempts exhausted",
|
|
1434
|
+
file=sys.stderr,
|
|
1435
|
+
)
|
|
1436
|
+
raise RateLimitException(
|
|
1437
|
+
f"Rate limit exceeded: {error_message}",
|
|
1438
|
+
retry_after=retry_after,
|
|
1562
1439
|
response=e.response,
|
|
1563
|
-
cause=e,
|
|
1564
1440
|
) from None
|
|
1565
|
-
|
|
1566
|
-
last_error = e
|
|
1567
|
-
|
|
1568
|
-
# Check if this is a retryable error
|
|
1569
|
-
if _is_retryable_error(e) and attempt < self._max_retries:
|
|
1570
|
-
# Calculate exponential backoff delay
|
|
1571
|
-
delay = self._retry_backoff_factor * (2**attempt)
|
|
1572
|
-
if DEBUG:
|
|
1573
|
-
print(
|
|
1574
|
-
f"{time.time()}|POST /session/new - Retryable error: {str(e)}, "
|
|
1575
|
-
f"retrying in {delay}s (attempt {attempt + 1}/{self._max_retries})"
|
|
1576
|
-
)
|
|
1577
|
-
time.sleep(delay)
|
|
1578
|
-
attempt += 1
|
|
1579
|
-
continue
|
|
1580
|
-
else:
|
|
1581
|
-
# Non-retryable error or exhausted retries
|
|
1582
|
-
raise HttpClientException(
|
|
1583
|
-
f"Failed to create blank PDF session: {str(e)}",
|
|
1584
|
-
response=None,
|
|
1585
|
-
cause=e,
|
|
1586
|
-
) from None
|
|
1587
|
-
|
|
1588
|
-
# Should not reach here, but handle just in case
|
|
1589
|
-
if last_error:
|
|
1441
|
+
|
|
1590
1442
|
raise HttpClientException(
|
|
1591
|
-
f"Failed to create blank PDF session
|
|
1592
|
-
response=
|
|
1593
|
-
cause=
|
|
1443
|
+
f"Failed to create blank PDF session: {error_message}",
|
|
1444
|
+
response=e.response,
|
|
1445
|
+
cause=e,
|
|
1594
1446
|
) from None
|
|
1595
|
-
|
|
1447
|
+
except httpx.RequestError as e:
|
|
1596
1448
|
raise HttpClientException(
|
|
1597
|
-
f"Failed to create blank PDF session
|
|
1449
|
+
f"Failed to create blank PDF session: {str(e)}",
|
|
1598
1450
|
response=None,
|
|
1599
|
-
|
|
1451
|
+
cause=e,
|
|
1452
|
+
) from None
|
|
1600
1453
|
|
|
1601
1454
|
def _make_request(
|
|
1602
1455
|
self,
|
|
1603
1456
|
method: str,
|
|
1604
1457
|
path: str,
|
|
1605
|
-
data: Optional[dict] = None,
|
|
1606
|
-
params: Optional[dict] = None,
|
|
1458
|
+
data: Optional[dict[str, Any]] = None,
|
|
1459
|
+
params: Optional[dict[str, Any]] = None,
|
|
1607
1460
|
) -> httpx.Response:
|
|
1608
1461
|
"""
|
|
1609
1462
|
Make HTTP request with session headers, error handling, and automatic retry for transient errors.
|
|
1610
1463
|
"""
|
|
1611
1464
|
headers = {
|
|
1612
1465
|
"X-Session-Id": self._session_id,
|
|
1466
|
+
"X-API-VERSION": "2",
|
|
1613
1467
|
"Content-Type": "application/json",
|
|
1614
1468
|
"X-Generated-At": _generate_timestamp(),
|
|
1615
1469
|
"X-Fingerprint": Fingerprint.generate(),
|
|
1616
|
-
"X-API-VERSION": "1",
|
|
1617
1470
|
}
|
|
1618
1471
|
|
|
1619
|
-
|
|
1620
|
-
|
|
1472
|
+
request_body = json.dumps(data) if data is not None else None
|
|
1473
|
+
request_size = (
|
|
1474
|
+
len(request_body.encode("utf-8")) if request_body is not None else 0
|
|
1475
|
+
)
|
|
1621
1476
|
|
|
1622
|
-
|
|
1623
|
-
|
|
1624
|
-
|
|
1625
|
-
|
|
1626
|
-
|
|
1627
|
-
|
|
1628
|
-
|
|
1629
|
-
|
|
1630
|
-
|
|
1631
|
-
|
|
1632
|
-
else ""
|
|
1633
|
-
)
|
|
1634
|
-
print(
|
|
1635
|
-
f"{time.time()}|{method} {path}{retry_info} - request size: {request_size} bytes"
|
|
1636
|
-
)
|
|
1477
|
+
def log_attempt(attempt: int) -> None:
|
|
1478
|
+
if DEBUG:
|
|
1479
|
+
retry_info = (
|
|
1480
|
+
f" (attempt {attempt + 1}/{self._max_attempts})"
|
|
1481
|
+
if attempt > 0
|
|
1482
|
+
else ""
|
|
1483
|
+
)
|
|
1484
|
+
print(
|
|
1485
|
+
f"{time.time()}|{method} {path}{retry_info} - request size: {request_size} bytes"
|
|
1486
|
+
)
|
|
1637
1487
|
|
|
1638
|
-
|
|
1639
|
-
|
|
1640
|
-
|
|
1641
|
-
|
|
1642
|
-
|
|
1643
|
-
|
|
1644
|
-
|
|
1488
|
+
def request_api() -> httpx.Response:
|
|
1489
|
+
return self._client.request(
|
|
1490
|
+
method=method,
|
|
1491
|
+
url=self._cleanup_url_path(self._base_url, path),
|
|
1492
|
+
json=data,
|
|
1493
|
+
params=params,
|
|
1494
|
+
headers=headers,
|
|
1495
|
+
timeout=self._read_timeout if self._read_timeout > 0 else None,
|
|
1496
|
+
)
|
|
1497
|
+
|
|
1498
|
+
try:
|
|
1499
|
+
response = _execute_request_with_retries(
|
|
1500
|
+
request_callable=request_api,
|
|
1501
|
+
operation=f"{method} {path}",
|
|
1502
|
+
max_attempts=self._max_attempts,
|
|
1503
|
+
retry_backoff_factor=self._retry_backoff_factor,
|
|
1504
|
+
pre_request_hook=log_attempt,
|
|
1505
|
+
)
|
|
1506
|
+
|
|
1507
|
+
response_size = len(response.content)
|
|
1508
|
+
if DEBUG:
|
|
1509
|
+
print(
|
|
1510
|
+
f"{time.time()}|{method} {path} - response size: {response_size} bytes"
|
|
1645
1511
|
)
|
|
1646
1512
|
|
|
1647
|
-
|
|
1648
|
-
if DEBUG:
|
|
1649
|
-
print(
|
|
1650
|
-
f"{time.time()}|{method} {path} - response size: {response_size} bytes"
|
|
1651
|
-
)
|
|
1513
|
+
_log_generated_at_header(response, method, path)
|
|
1652
1514
|
|
|
1653
|
-
|
|
1654
|
-
|
|
1655
|
-
|
|
1656
|
-
|
|
1657
|
-
|
|
1658
|
-
|
|
1659
|
-
|
|
1660
|
-
raise FontNotFoundException(
|
|
1661
|
-
error_data.get("message", "Font not found")
|
|
1662
|
-
)
|
|
1663
|
-
if error_data.get("error") == "SessionNotFoundException":
|
|
1664
|
-
raise SessionNotFoundException(
|
|
1665
|
-
error_data.get("message", "Session not found")
|
|
1666
|
-
)
|
|
1667
|
-
except (json.JSONDecodeError, KeyError):
|
|
1668
|
-
pass
|
|
1669
|
-
|
|
1670
|
-
self._handle_authentication_error(response)
|
|
1671
|
-
response.raise_for_status()
|
|
1672
|
-
return response
|
|
1673
|
-
|
|
1674
|
-
except httpx.HTTPStatusError as e:
|
|
1675
|
-
# Handle 429 (rate limit) with retry
|
|
1676
|
-
if e.response.status_code == 429 and attempt < self._max_retries:
|
|
1677
|
-
retry_after = _get_retry_after_delay(e.response)
|
|
1678
|
-
if retry_after is not None:
|
|
1679
|
-
delay = retry_after
|
|
1680
|
-
else:
|
|
1681
|
-
# Use exponential backoff if no Retry-After header
|
|
1682
|
-
delay = self._retry_backoff_factor * (2**attempt)
|
|
1683
|
-
|
|
1684
|
-
# Always log 429 to stderr for visibility
|
|
1685
|
-
print(
|
|
1686
|
-
f"Rate limit (429) on {method} {path} - retrying in {delay}s "
|
|
1687
|
-
f"(attempt {attempt + 1}/{self._max_retries})",
|
|
1688
|
-
file=sys.stderr,
|
|
1689
|
-
)
|
|
1690
|
-
if DEBUG:
|
|
1691
|
-
print(
|
|
1692
|
-
f"{time.time()}|{method} {path} - Rate limit exceeded (429), "
|
|
1693
|
-
f"retrying in {delay}s (attempt {attempt + 1}/{self._max_retries})"
|
|
1515
|
+
# Handle 404 errors
|
|
1516
|
+
if response.status_code == 404:
|
|
1517
|
+
try:
|
|
1518
|
+
error_data = response.json()
|
|
1519
|
+
if error_data.get("error") == "FontNotFoundException":
|
|
1520
|
+
raise FontNotFoundException(
|
|
1521
|
+
error_data.get("message", "Font not found")
|
|
1694
1522
|
)
|
|
1695
|
-
|
|
1696
|
-
|
|
1697
|
-
|
|
1523
|
+
if error_data.get("error") == "SessionNotFoundException":
|
|
1524
|
+
raise SessionNotFoundException(
|
|
1525
|
+
error_data.get("message", "Session not found")
|
|
1526
|
+
)
|
|
1527
|
+
except (json.JSONDecodeError, KeyError):
|
|
1528
|
+
pass
|
|
1529
|
+
|
|
1530
|
+
self._handle_authentication_error(response)
|
|
1531
|
+
response.raise_for_status()
|
|
1532
|
+
return response
|
|
1698
1533
|
|
|
1699
|
-
|
|
1700
|
-
|
|
1701
|
-
|
|
1534
|
+
except httpx.HTTPStatusError as e:
|
|
1535
|
+
# Other HTTP status errors are not retried (these are application-level errors)
|
|
1536
|
+
self._handle_authentication_error(e.response)
|
|
1537
|
+
error_message = self._extract_error_message(e.response)
|
|
1702
1538
|
|
|
1703
|
-
|
|
1704
|
-
|
|
1705
|
-
|
|
1706
|
-
|
|
1707
|
-
|
|
1708
|
-
|
|
1709
|
-
|
|
1710
|
-
|
|
1711
|
-
|
|
1712
|
-
|
|
1713
|
-
|
|
1714
|
-
) from None
|
|
1715
|
-
|
|
1716
|
-
raise HttpClientException(
|
|
1717
|
-
f"API request failed: {error_message}", response=e.response, cause=e
|
|
1539
|
+
# Raise RateLimitException for 429 after retry attempts are exhausted
|
|
1540
|
+
if e.response.status_code == 429:
|
|
1541
|
+
retry_after = _get_retry_after_delay(e.response)
|
|
1542
|
+
print(
|
|
1543
|
+
f"Rate limit (429) on {method} {path} - maximum attempts exhausted",
|
|
1544
|
+
file=sys.stderr,
|
|
1545
|
+
)
|
|
1546
|
+
raise RateLimitException(
|
|
1547
|
+
f"Rate limit exceeded: {error_message}",
|
|
1548
|
+
retry_after=retry_after,
|
|
1549
|
+
response=e.response,
|
|
1718
1550
|
) from None
|
|
1719
|
-
except httpx.RequestError as e:
|
|
1720
|
-
last_error = e
|
|
1721
|
-
|
|
1722
|
-
# Check if this is a retryable error
|
|
1723
|
-
if _is_retryable_error(e) and attempt < self._max_retries:
|
|
1724
|
-
# Calculate exponential backoff delay
|
|
1725
|
-
delay = self._retry_backoff_factor * (2**attempt)
|
|
1726
|
-
if DEBUG:
|
|
1727
|
-
print(
|
|
1728
|
-
f"{time.time()}|{method} {path} - Retryable error: {str(e)}, "
|
|
1729
|
-
f"retrying in {delay}s (attempt {attempt + 1}/{self._max_retries})"
|
|
1730
|
-
)
|
|
1731
|
-
time.sleep(delay)
|
|
1732
|
-
attempt += 1
|
|
1733
|
-
continue
|
|
1734
|
-
else:
|
|
1735
|
-
# Non-retryable error or exhausted retries
|
|
1736
|
-
raise HttpClientException(
|
|
1737
|
-
f"API request failed: {str(e)}", response=None, cause=e
|
|
1738
|
-
) from None
|
|
1739
1551
|
|
|
1740
|
-
# Should not reach here, but handle just in case
|
|
1741
|
-
if last_error:
|
|
1742
1552
|
raise HttpClientException(
|
|
1743
|
-
f"API request failed
|
|
1744
|
-
response=None,
|
|
1745
|
-
cause=last_error,
|
|
1553
|
+
f"API request failed: {error_message}", response=e.response, cause=e
|
|
1746
1554
|
) from None
|
|
1747
|
-
|
|
1555
|
+
except httpx.RequestError as e:
|
|
1748
1556
|
raise HttpClientException(
|
|
1749
|
-
f"API request failed
|
|
1750
|
-
|
|
1751
|
-
)
|
|
1557
|
+
f"API request failed: {str(e)}", response=None, cause=e
|
|
1558
|
+
) from None
|
|
1752
1559
|
|
|
1753
1560
|
def _find(
|
|
1754
1561
|
self,
|
|
@@ -1777,74 +1584,19 @@ class PDFDancer:
|
|
|
1777
1584
|
|
|
1778
1585
|
# Use snapshot for all other queries
|
|
1779
1586
|
if position and position.page_number is not None:
|
|
1780
|
-
|
|
1587
|
+
page_snapshot = self._get_or_fetch_page_snapshot(position.page_number)
|
|
1781
1588
|
return self._filter_snapshot_elements(
|
|
1782
|
-
|
|
1589
|
+
page_snapshot.elements, object_type, position, tolerance
|
|
1783
1590
|
)
|
|
1784
1591
|
else:
|
|
1785
|
-
|
|
1786
|
-
all_elements = []
|
|
1787
|
-
for page_snap in
|
|
1592
|
+
document_snapshot = self._get_or_fetch_document_snapshot()
|
|
1593
|
+
all_elements: List[ObjectRef] = []
|
|
1594
|
+
for page_snap in document_snapshot.pages:
|
|
1788
1595
|
all_elements.extend(page_snap.elements)
|
|
1789
1596
|
return self._filter_snapshot_elements(
|
|
1790
1597
|
all_elements, object_type, position, tolerance
|
|
1791
1598
|
)
|
|
1792
1599
|
|
|
1793
|
-
def select_paragraphs(self) -> List[ParagraphObject]:
|
|
1794
|
-
"""
|
|
1795
|
-
Searches for paragraph objects returning ParagraphObject instances.
|
|
1796
|
-
"""
|
|
1797
|
-
return self._to_paragraph_objects(self._find_paragraphs(None))
|
|
1798
|
-
|
|
1799
|
-
def select_paragraphs_matching(self, pattern: str) -> List[ParagraphObject]:
|
|
1800
|
-
"""
|
|
1801
|
-
Searches for paragraph objects matching a regex pattern.
|
|
1802
|
-
|
|
1803
|
-
Args:
|
|
1804
|
-
pattern: Regex pattern to match against paragraph text
|
|
1805
|
-
|
|
1806
|
-
Returns:
|
|
1807
|
-
List of ParagraphObject instances matching the pattern
|
|
1808
|
-
"""
|
|
1809
|
-
position = Position()
|
|
1810
|
-
position.text_pattern = pattern
|
|
1811
|
-
return self._to_paragraph_objects(self._find_paragraphs(position))
|
|
1812
|
-
|
|
1813
|
-
def select_paragraph_matching(self, pattern: str) -> Optional[ParagraphObject]:
|
|
1814
|
-
"""
|
|
1815
|
-
Select the first paragraph matching the specified regex pattern.
|
|
1816
|
-
|
|
1817
|
-
Args:
|
|
1818
|
-
pattern: Regex pattern to match against paragraph text
|
|
1819
|
-
|
|
1820
|
-
Returns:
|
|
1821
|
-
First ParagraphObject matching the pattern, or None if no match
|
|
1822
|
-
"""
|
|
1823
|
-
results = self.select_paragraphs_matching(pattern)
|
|
1824
|
-
return results[0] if results else None
|
|
1825
|
-
|
|
1826
|
-
def _find_paragraphs(
|
|
1827
|
-
self, position: Optional[Position] = None, tolerance: float = DEFAULT_TOLERANCE
|
|
1828
|
-
) -> List[TextObjectRef]:
|
|
1829
|
-
"""
|
|
1830
|
-
Searches for paragraph objects returning TextObjectRef with hierarchical structure.
|
|
1831
|
-
Uses snapshot cache for all queries.
|
|
1832
|
-
"""
|
|
1833
|
-
# Use snapshot for all queries (including spatial)
|
|
1834
|
-
if position and position.page_number is not None:
|
|
1835
|
-
snapshot = self._get_or_fetch_page_snapshot(position.page_number)
|
|
1836
|
-
return self._filter_snapshot_elements(
|
|
1837
|
-
snapshot.elements, ObjectType.PARAGRAPH, position, tolerance
|
|
1838
|
-
)
|
|
1839
|
-
else:
|
|
1840
|
-
snapshot = self._get_or_fetch_document_snapshot()
|
|
1841
|
-
all_elements = []
|
|
1842
|
-
for page_snap in snapshot.pages:
|
|
1843
|
-
all_elements.extend(page_snap.elements)
|
|
1844
|
-
return self._filter_snapshot_elements(
|
|
1845
|
-
all_elements, ObjectType.PARAGRAPH, position, tolerance
|
|
1846
|
-
)
|
|
1847
|
-
|
|
1848
1600
|
def _find_images(
|
|
1849
1601
|
self, position: Optional[Position] = None, tolerance: float = DEFAULT_TOLERANCE
|
|
1850
1602
|
) -> List[ObjectRef]:
|
|
@@ -1854,14 +1606,14 @@ class PDFDancer:
|
|
|
1854
1606
|
"""
|
|
1855
1607
|
# Use snapshot for all queries (including spatial)
|
|
1856
1608
|
if position and position.page_number is not None:
|
|
1857
|
-
|
|
1609
|
+
page_snapshot = self._get_or_fetch_page_snapshot(position.page_number)
|
|
1858
1610
|
return self._filter_snapshot_elements(
|
|
1859
|
-
|
|
1611
|
+
page_snapshot.elements, ObjectType.IMAGE, position, tolerance
|
|
1860
1612
|
)
|
|
1861
1613
|
else:
|
|
1862
|
-
|
|
1863
|
-
all_elements = []
|
|
1864
|
-
for page_snap in
|
|
1614
|
+
document_snapshot = self._get_or_fetch_document_snapshot()
|
|
1615
|
+
all_elements: List[ObjectRef] = []
|
|
1616
|
+
for page_snap in document_snapshot.pages:
|
|
1865
1617
|
all_elements.extend(page_snap.elements)
|
|
1866
1618
|
return self._filter_snapshot_elements(
|
|
1867
1619
|
all_elements, ObjectType.IMAGE, position, tolerance
|
|
@@ -1888,14 +1640,14 @@ class PDFDancer:
|
|
|
1888
1640
|
"""
|
|
1889
1641
|
# Use snapshot for all queries (including spatial)
|
|
1890
1642
|
if position and position.page_number is not None:
|
|
1891
|
-
|
|
1643
|
+
page_snapshot = self._get_or_fetch_page_snapshot(position.page_number)
|
|
1892
1644
|
return self._filter_snapshot_elements(
|
|
1893
|
-
|
|
1645
|
+
page_snapshot.elements, ObjectType.FORM_X_OBJECT, position, tolerance
|
|
1894
1646
|
)
|
|
1895
1647
|
else:
|
|
1896
|
-
|
|
1897
|
-
all_elements = []
|
|
1898
|
-
for page_snap in
|
|
1648
|
+
document_snapshot = self._get_or_fetch_document_snapshot()
|
|
1649
|
+
all_elements: List[ObjectRef] = []
|
|
1650
|
+
for page_snap in document_snapshot.pages:
|
|
1899
1651
|
all_elements.extend(page_snap.elements)
|
|
1900
1652
|
return self._filter_snapshot_elements(
|
|
1901
1653
|
all_elements, ObjectType.FORM_X_OBJECT, position, tolerance
|
|
@@ -1938,17 +1690,23 @@ class PDFDancer:
|
|
|
1938
1690
|
"""
|
|
1939
1691
|
# Use snapshot for all queries (including name and spatial)
|
|
1940
1692
|
if position and position.page_number is not None:
|
|
1941
|
-
|
|
1942
|
-
return
|
|
1943
|
-
|
|
1693
|
+
page_snapshot = self._get_or_fetch_page_snapshot(position.page_number)
|
|
1694
|
+
return cast(
|
|
1695
|
+
List[FormFieldRef],
|
|
1696
|
+
self._filter_snapshot_elements(
|
|
1697
|
+
page_snapshot.elements, ObjectType.FORM_FIELD, position, tolerance
|
|
1698
|
+
),
|
|
1944
1699
|
)
|
|
1945
1700
|
else:
|
|
1946
|
-
|
|
1947
|
-
all_elements = []
|
|
1948
|
-
for page_snap in
|
|
1701
|
+
document_snapshot = self._get_or_fetch_document_snapshot()
|
|
1702
|
+
all_elements: List[ObjectRef] = []
|
|
1703
|
+
for page_snap in document_snapshot.pages:
|
|
1949
1704
|
all_elements.extend(page_snap.elements)
|
|
1950
|
-
return
|
|
1951
|
-
|
|
1705
|
+
return cast(
|
|
1706
|
+
List[FormFieldRef],
|
|
1707
|
+
self._filter_snapshot_elements(
|
|
1708
|
+
all_elements, ObjectType.FORM_FIELD, position, tolerance
|
|
1709
|
+
),
|
|
1952
1710
|
)
|
|
1953
1711
|
|
|
1954
1712
|
def _change_form_field(self, form_field_ref: FormFieldRef, new_value: str) -> bool:
|
|
@@ -1963,7 +1721,7 @@ class PDFDancer:
|
|
|
1963
1721
|
response = self._make_request(
|
|
1964
1722
|
"PUT", "/pdf/modify/formField", data=request_data
|
|
1965
1723
|
)
|
|
1966
|
-
return response.json()
|
|
1724
|
+
return cast(bool, response.json())
|
|
1967
1725
|
finally:
|
|
1968
1726
|
self._invalidate_snapshots()
|
|
1969
1727
|
|
|
@@ -1988,75 +1746,20 @@ class PDFDancer:
|
|
|
1988
1746
|
|
|
1989
1747
|
# For simple page-level "all paths" queries, use snapshot
|
|
1990
1748
|
if position and position.page_number is not None:
|
|
1991
|
-
|
|
1749
|
+
page_snapshot = self._get_or_fetch_page_snapshot(position.page_number)
|
|
1992
1750
|
return self._filter_snapshot_elements(
|
|
1993
|
-
|
|
1751
|
+
page_snapshot.elements, ObjectType.PATH, position, tolerance
|
|
1994
1752
|
)
|
|
1995
1753
|
else:
|
|
1996
1754
|
# Document-level query - use document snapshot
|
|
1997
|
-
|
|
1998
|
-
all_elements = []
|
|
1999
|
-
for page_snap in
|
|
1755
|
+
document_snapshot = self._get_or_fetch_document_snapshot()
|
|
1756
|
+
all_elements: List[ObjectRef] = []
|
|
1757
|
+
for page_snap in document_snapshot.pages:
|
|
2000
1758
|
all_elements.extend(page_snap.elements)
|
|
2001
1759
|
return self._filter_snapshot_elements(
|
|
2002
1760
|
all_elements, ObjectType.PATH, position, tolerance
|
|
2003
1761
|
)
|
|
2004
1762
|
|
|
2005
|
-
def _find_text_lines(
|
|
2006
|
-
self, position: Optional[Position] = None, tolerance: float = DEFAULT_TOLERANCE
|
|
2007
|
-
) -> List[TextObjectRef]:
|
|
2008
|
-
"""
|
|
2009
|
-
Searches for text line objects returning TextObjectRef with hierarchical structure.
|
|
2010
|
-
Uses snapshot cache for all queries.
|
|
2011
|
-
"""
|
|
2012
|
-
# Use snapshot for all queries (including spatial)
|
|
2013
|
-
if position and position.page_number is not None:
|
|
2014
|
-
snapshot = self._get_or_fetch_page_snapshot(position.page_number)
|
|
2015
|
-
return self._filter_snapshot_elements(
|
|
2016
|
-
snapshot.elements, ObjectType.TEXT_LINE, position, tolerance
|
|
2017
|
-
)
|
|
2018
|
-
else:
|
|
2019
|
-
snapshot = self._get_or_fetch_document_snapshot()
|
|
2020
|
-
all_elements = []
|
|
2021
|
-
for page_snap in snapshot.pages:
|
|
2022
|
-
all_elements.extend(page_snap.elements)
|
|
2023
|
-
return self._filter_snapshot_elements(
|
|
2024
|
-
all_elements, ObjectType.TEXT_LINE, position, tolerance
|
|
2025
|
-
)
|
|
2026
|
-
|
|
2027
|
-
def select_text_lines(self) -> List[TextLineObject]:
|
|
2028
|
-
"""
|
|
2029
|
-
Searches for text line objects returning TextLineObject wrappers.
|
|
2030
|
-
"""
|
|
2031
|
-
return self._to_textline_objects(self._find_text_lines(None))
|
|
2032
|
-
|
|
2033
|
-
def select_text_lines_matching(self, pattern: str) -> List[TextLineObject]:
|
|
2034
|
-
"""
|
|
2035
|
-
Searches for text line objects matching a regex pattern.
|
|
2036
|
-
|
|
2037
|
-
Args:
|
|
2038
|
-
pattern: Regex pattern to match against text line text
|
|
2039
|
-
|
|
2040
|
-
Returns:
|
|
2041
|
-
List of TextLineObject instances matching the pattern
|
|
2042
|
-
"""
|
|
2043
|
-
position = Position()
|
|
2044
|
-
position.text_pattern = pattern
|
|
2045
|
-
return self._to_textline_objects(self._find_text_lines(position))
|
|
2046
|
-
|
|
2047
|
-
def select_text_line_matching(self, pattern: str) -> Optional[TextLineObject]:
|
|
2048
|
-
"""
|
|
2049
|
-
Select the first text line matching the specified regex pattern.
|
|
2050
|
-
|
|
2051
|
-
Args:
|
|
2052
|
-
pattern: Regex pattern to match against text line text
|
|
2053
|
-
|
|
2054
|
-
Returns:
|
|
2055
|
-
First TextLineObject matching the pattern, or None if no match
|
|
2056
|
-
"""
|
|
2057
|
-
results = self.select_text_lines_matching(pattern)
|
|
2058
|
-
return results[0] if results else None
|
|
2059
|
-
|
|
2060
1763
|
def page(self, page_number: int) -> PageClient:
|
|
2061
1764
|
"""
|
|
2062
1765
|
Get a specific page by page number, using snapshot cache when available.
|
|
@@ -2146,7 +1849,7 @@ class PDFDancer:
|
|
|
2146
1849
|
if result:
|
|
2147
1850
|
self._invalidate_snapshots()
|
|
2148
1851
|
|
|
2149
|
-
return result
|
|
1852
|
+
return cast(bool, result)
|
|
2150
1853
|
|
|
2151
1854
|
def move_page(self, from_page: int, to_page: int) -> bool:
|
|
2152
1855
|
"""
|
|
@@ -2193,6 +1896,39 @@ class PDFDancer:
|
|
|
2193
1896
|
|
|
2194
1897
|
# Manipulation Operations
|
|
2195
1898
|
|
|
1899
|
+
def _edit_text(
|
|
1900
|
+
self,
|
|
1901
|
+
operation: str,
|
|
1902
|
+
request: Union[
|
|
1903
|
+
TextReplaceRequest,
|
|
1904
|
+
TextDeleteRequest,
|
|
1905
|
+
TextInsertRequest,
|
|
1906
|
+
TextStyleRequest,
|
|
1907
|
+
],
|
|
1908
|
+
) -> TextEditResponse:
|
|
1909
|
+
"""Execute one validated selector-based text mutation."""
|
|
1910
|
+
expected_types = {
|
|
1911
|
+
"replace": TextReplaceRequest,
|
|
1912
|
+
"delete": TextDeleteRequest,
|
|
1913
|
+
"insert": TextInsertRequest,
|
|
1914
|
+
"style": TextStyleRequest,
|
|
1915
|
+
}
|
|
1916
|
+
expected_type = expected_types.get(operation)
|
|
1917
|
+
if expected_type is None:
|
|
1918
|
+
raise ValidationException(f"Unsupported text operation: {operation}")
|
|
1919
|
+
if not isinstance(request, expected_type):
|
|
1920
|
+
raise ValidationException(
|
|
1921
|
+
f"{operation} requires {expected_type.__name__}, "
|
|
1922
|
+
f"got {type(request).__name__}"
|
|
1923
|
+
)
|
|
1924
|
+
|
|
1925
|
+
response = self._make_request(
|
|
1926
|
+
"POST", f"/pdf/text/{operation}", data=request.to_dict()
|
|
1927
|
+
)
|
|
1928
|
+
result = TextEditResponse.from_dict(response.json())
|
|
1929
|
+
self._invalidate_snapshots()
|
|
1930
|
+
return result
|
|
1931
|
+
|
|
2196
1932
|
def _delete(self, object_ref: ObjectRef) -> bool:
|
|
2197
1933
|
"""
|
|
2198
1934
|
Deletes the specified PDF object from the document.
|
|
@@ -2214,7 +1950,7 @@ class PDFDancer:
|
|
|
2214
1950
|
if result:
|
|
2215
1951
|
self._invalidate_snapshots()
|
|
2216
1952
|
|
|
2217
|
-
return result
|
|
1953
|
+
return cast(bool, result)
|
|
2218
1954
|
|
|
2219
1955
|
def _move(self, object_ref: ObjectRef, position: Position) -> bool:
|
|
2220
1956
|
"""
|
|
@@ -2240,59 +1976,7 @@ class PDFDancer:
|
|
|
2240
1976
|
if result:
|
|
2241
1977
|
self._invalidate_snapshots()
|
|
2242
1978
|
|
|
2243
|
-
return result
|
|
2244
|
-
|
|
2245
|
-
def _redact(
|
|
2246
|
-
self,
|
|
2247
|
-
targets: List["RedactTarget"],
|
|
2248
|
-
default_replacement: str = "[REDACTED]",
|
|
2249
|
-
placeholder_color: Optional[Color] = None,
|
|
2250
|
-
) -> "RedactResponse":
|
|
2251
|
-
"""
|
|
2252
|
-
Redacts specified objects from the PDF document.
|
|
2253
|
-
|
|
2254
|
-
Args:
|
|
2255
|
-
targets: List of RedactTarget objects identifying what to redact
|
|
2256
|
-
default_replacement: Default replacement text for redacted content
|
|
2257
|
-
placeholder_color: Color for image/path placeholder rectangles
|
|
2258
|
-
|
|
2259
|
-
Returns:
|
|
2260
|
-
RedactResponse with count, success status, and any warnings
|
|
2261
|
-
"""
|
|
2262
|
-
if not targets:
|
|
2263
|
-
raise ValidationException("At least one redaction target is required")
|
|
2264
|
-
|
|
2265
|
-
if placeholder_color is None:
|
|
2266
|
-
placeholder_color = Color(0, 0, 0)
|
|
2267
|
-
|
|
2268
|
-
request = RedactRequest(targets, default_replacement, placeholder_color)
|
|
2269
|
-
response = self._make_request("POST", "/pdf/redact", data=request.to_dict())
|
|
2270
|
-
result = RedactResponse.from_dict(response.json())
|
|
2271
|
-
|
|
2272
|
-
if result.success:
|
|
2273
|
-
self._invalidate_snapshots()
|
|
2274
|
-
|
|
2275
|
-
return result
|
|
2276
|
-
|
|
2277
|
-
def redact(
|
|
2278
|
-
self,
|
|
2279
|
-
objects: List["PDFObjectBase"],
|
|
2280
|
-
replacement: str = "[REDACTED]",
|
|
2281
|
-
placeholder_color: Optional[Color] = None,
|
|
2282
|
-
) -> "RedactResponse":
|
|
2283
|
-
"""
|
|
2284
|
-
Redacts multiple objects from the PDF document.
|
|
2285
|
-
|
|
2286
|
-
Args:
|
|
2287
|
-
objects: List of PDF objects to redact
|
|
2288
|
-
replacement: Replacement text for all redacted content
|
|
2289
|
-
placeholder_color: Color for image/path placeholder rectangles
|
|
2290
|
-
|
|
2291
|
-
Returns:
|
|
2292
|
-
RedactResponse with count, success status, and any warnings
|
|
2293
|
-
"""
|
|
2294
|
-
targets = [RedactTarget(obj.internal_id, replacement) for obj in objects]
|
|
2295
|
-
return self._redact(targets, replacement, placeholder_color)
|
|
1979
|
+
return cast(bool, result)
|
|
2296
1980
|
|
|
2297
1981
|
def clear_clipping(self, object_ref: ObjectRef) -> bool:
|
|
2298
1982
|
"""
|
|
@@ -2325,88 +2009,6 @@ class PDFDancer:
|
|
|
2325
2009
|
|
|
2326
2010
|
return result
|
|
2327
2011
|
|
|
2328
|
-
# Template Replacement Operations
|
|
2329
|
-
|
|
2330
|
-
def _apply_replacements(
|
|
2331
|
-
self,
|
|
2332
|
-
replacements: List[TemplateReplacement],
|
|
2333
|
-
page_number: Optional[int] = None,
|
|
2334
|
-
reflow_preset: Optional[ReflowPreset] = None,
|
|
2335
|
-
) -> bool:
|
|
2336
|
-
"""
|
|
2337
|
-
Internal method to replace template placeholders in the PDF.
|
|
2338
|
-
|
|
2339
|
-
Args:
|
|
2340
|
-
replacements: List of TemplateReplacement objects
|
|
2341
|
-
page_number: Optional 1-based page number. If None, applies to all pages.
|
|
2342
|
-
reflow_preset: Optional reflow behavior preset
|
|
2343
|
-
|
|
2344
|
-
Returns:
|
|
2345
|
-
True if replacement was successful
|
|
2346
|
-
"""
|
|
2347
|
-
if not replacements:
|
|
2348
|
-
raise ValidationException("At least one replacement is required")
|
|
2349
|
-
|
|
2350
|
-
# Convert 1-based page_number to 0-based page_index for API
|
|
2351
|
-
page_index = None
|
|
2352
|
-
if page_number is not None:
|
|
2353
|
-
page_index = page_number - 1
|
|
2354
|
-
|
|
2355
|
-
request = TemplateReplaceRequest(
|
|
2356
|
-
replacements=replacements,
|
|
2357
|
-
page_index=page_index,
|
|
2358
|
-
reflow_preset=reflow_preset,
|
|
2359
|
-
)
|
|
2360
|
-
response = self._make_request(
|
|
2361
|
-
"POST", "/template/replace", data=request.to_dict()
|
|
2362
|
-
)
|
|
2363
|
-
result = response.json()
|
|
2364
|
-
|
|
2365
|
-
if result:
|
|
2366
|
-
self._invalidate_snapshots()
|
|
2367
|
-
|
|
2368
|
-
return result
|
|
2369
|
-
|
|
2370
|
-
def apply_replacements(
|
|
2371
|
-
self,
|
|
2372
|
-
replacements: Dict[str, Union[str, dict]],
|
|
2373
|
-
reflow_preset: Optional[ReflowPreset] = None,
|
|
2374
|
-
) -> bool:
|
|
2375
|
-
"""
|
|
2376
|
-
Replace template placeholders in the PDF document.
|
|
2377
|
-
|
|
2378
|
-
Finds exact text matches for placeholders and replaces them with specified
|
|
2379
|
-
content. All placeholders must be found or the operation fails atomically.
|
|
2380
|
-
|
|
2381
|
-
Args:
|
|
2382
|
-
replacements: Dict mapping placeholder strings to replacement values.
|
|
2383
|
-
- Simple: {"{{NAME}}": "John Doe"}
|
|
2384
|
-
- With options: {"{{NAME}}": {"text": "John", "font": Font(...), "color": Color(...)}}
|
|
2385
|
-
- With image: {"{{LOGO}}": {"image": Path("logo.png")}}
|
|
2386
|
-
- With image and size: {"{{LOGO}}": {"image": Path("logo.png"), "width": 50, "height": 50}}
|
|
2387
|
-
reflow_preset: Optional ReflowPreset to control text reflow behavior.
|
|
2388
|
-
- BEST_EFFORT: Attempt to reflow, proceed even if imperfect
|
|
2389
|
-
- FIT_OR_FAIL: Reflow must succeed or operation fails
|
|
2390
|
-
- NONE: No reflow, replacement placed as-is
|
|
2391
|
-
|
|
2392
|
-
Returns:
|
|
2393
|
-
True if all replacements were successful
|
|
2394
|
-
|
|
2395
|
-
Example:
|
|
2396
|
-
```python
|
|
2397
|
-
pdf.apply_replacements({
|
|
2398
|
-
"{{NAME}}": "John Doe",
|
|
2399
|
-
"{{DATE}}": "2025-01-15",
|
|
2400
|
-
})
|
|
2401
|
-
```
|
|
2402
|
-
"""
|
|
2403
|
-
replacement_list = _dict_to_replacements(replacements)
|
|
2404
|
-
return self._apply_replacements(
|
|
2405
|
-
replacements=replacement_list,
|
|
2406
|
-
page_number=None,
|
|
2407
|
-
reflow_preset=reflow_preset,
|
|
2408
|
-
)
|
|
2409
|
-
|
|
2410
2012
|
# Add Operations
|
|
2411
2013
|
|
|
2412
2014
|
def _add_image(self, image: Image, position: Optional[Position] = None) -> bool:
|
|
@@ -2431,46 +2033,27 @@ class PDFDancer:
|
|
|
2431
2033
|
|
|
2432
2034
|
return self._add_object(image)
|
|
2433
2035
|
|
|
2434
|
-
def
|
|
2435
|
-
"""
|
|
2436
|
-
Adds a paragraph to the PDF document.
|
|
2437
|
-
|
|
2438
|
-
Args:
|
|
2439
|
-
paragraph: The paragraph object to add
|
|
2440
|
-
|
|
2441
|
-
Returns:
|
|
2442
|
-
True if the paragraph was successfully added
|
|
2443
|
-
"""
|
|
2444
|
-
if paragraph is None:
|
|
2445
|
-
raise ValidationException("Paragraph cannot be null")
|
|
2446
|
-
if paragraph.get_position() is None:
|
|
2447
|
-
raise ValidationException("Paragraph position is null")
|
|
2448
|
-
if paragraph.get_position().page_number is None:
|
|
2449
|
-
raise ValidationException("Paragraph position page number is null")
|
|
2450
|
-
if paragraph.get_position().page_number < 1:
|
|
2451
|
-
raise ValidationException("Paragraph position page number is less than 1")
|
|
2452
|
-
|
|
2453
|
-
return self._add_object(paragraph)
|
|
2454
|
-
|
|
2455
|
-
def _add_path(self, path: "Path") -> bool:
|
|
2036
|
+
def _add_path(self, path: PDFPath) -> bool:
|
|
2456
2037
|
"""
|
|
2457
2038
|
Internal method to add a path to the document after validation.
|
|
2458
2039
|
"""
|
|
2459
2040
|
|
|
2460
2041
|
if path is None:
|
|
2461
2042
|
raise ValidationException("Path cannot be null")
|
|
2462
|
-
|
|
2043
|
+
position = path.get_position()
|
|
2044
|
+
if position is None:
|
|
2463
2045
|
raise ValidationException("Path position is null")
|
|
2464
|
-
if
|
|
2046
|
+
if position.page_number is None:
|
|
2465
2047
|
raise ValidationException("Path position page number is null")
|
|
2466
|
-
if
|
|
2048
|
+
if position.page_number < 1:
|
|
2467
2049
|
raise ValidationException("Path position page number is less than 1")
|
|
2468
|
-
|
|
2050
|
+
path_segments = path.get_path_segments()
|
|
2051
|
+
if not path_segments:
|
|
2469
2052
|
raise ValidationException("Path must have at least one segment")
|
|
2470
2053
|
|
|
2471
2054
|
return self._add_object(path)
|
|
2472
2055
|
|
|
2473
|
-
def _add_object(self, pdf_object) -> bool:
|
|
2056
|
+
def _add_object(self, pdf_object: Any) -> bool:
|
|
2474
2057
|
"""
|
|
2475
2058
|
Internal method to add any PDF object.
|
|
2476
2059
|
"""
|
|
@@ -2482,7 +2065,7 @@ class PDFDancer:
|
|
|
2482
2065
|
if result:
|
|
2483
2066
|
self._invalidate_snapshots()
|
|
2484
2067
|
|
|
2485
|
-
return result
|
|
2068
|
+
return cast(bool, result)
|
|
2486
2069
|
|
|
2487
2070
|
def _add_page(self, request: Optional[AddPageRequest]) -> PageRef:
|
|
2488
2071
|
"""
|
|
@@ -2527,14 +2110,17 @@ class PDFDancer:
|
|
|
2527
2110
|
return result
|
|
2528
2111
|
|
|
2529
2112
|
# Path Group Operations (internal, 0-based page_index)
|
|
2530
|
-
def _create_path_group(
|
|
2531
|
-
|
|
2532
|
-
|
|
2113
|
+
def _create_path_group(
|
|
2114
|
+
self,
|
|
2115
|
+
page_index: int,
|
|
2116
|
+
path_ids: Optional[List[str]] = None,
|
|
2117
|
+
region: Optional["GroupBoundingRect"] = None,
|
|
2118
|
+
) -> PathGroupInfo:
|
|
2533
2119
|
if path_ids is not None:
|
|
2534
2120
|
if not isinstance(path_ids, list) or len(path_ids) == 0:
|
|
2535
2121
|
raise ValidationException("path_ids must be a non-empty list")
|
|
2536
2122
|
|
|
2537
|
-
data = {"pageIndex": page_index}
|
|
2123
|
+
data: dict[str, Any] = {"pageIndex": page_index}
|
|
2538
2124
|
if path_ids is not None:
|
|
2539
2125
|
data["pathIds"] = path_ids
|
|
2540
2126
|
if region is not None:
|
|
@@ -2548,7 +2134,9 @@ class PDFDancer:
|
|
|
2548
2134
|
self._invalidate_snapshots()
|
|
2549
2135
|
return PathGroupInfo.from_dict(response.json())
|
|
2550
2136
|
|
|
2551
|
-
def _move_path_group(
|
|
2137
|
+
def _move_path_group(
|
|
2138
|
+
self, page_index: int, group_id: str, x: float, y: float
|
|
2139
|
+
) -> bool:
|
|
2552
2140
|
data = {
|
|
2553
2141
|
"pageIndex": page_index,
|
|
2554
2142
|
"groupId": group_id,
|
|
@@ -2557,10 +2145,16 @@ class PDFDancer:
|
|
|
2557
2145
|
}
|
|
2558
2146
|
response = self._make_request("PUT", "/pdf/path-group/move", data=data)
|
|
2559
2147
|
self._invalidate_snapshots()
|
|
2560
|
-
return response.json()
|
|
2148
|
+
return cast(bool, response.json())
|
|
2561
2149
|
|
|
2562
|
-
def _transform_path_group(
|
|
2563
|
-
|
|
2150
|
+
def _transform_path_group(
|
|
2151
|
+
self,
|
|
2152
|
+
page_index: int,
|
|
2153
|
+
group_id: str,
|
|
2154
|
+
transform_type: str,
|
|
2155
|
+
**kwargs: Any,
|
|
2156
|
+
) -> bool:
|
|
2157
|
+
data: dict[str, Any] = {
|
|
2564
2158
|
"pageIndex": page_index,
|
|
2565
2159
|
"groupId": group_id,
|
|
2566
2160
|
"transformType": transform_type,
|
|
@@ -2568,21 +2162,25 @@ class PDFDancer:
|
|
|
2568
2162
|
data.update({k: v for k, v in kwargs.items() if v is not None})
|
|
2569
2163
|
response = self._make_request("PUT", "/pdf/path-group/transform", data=data)
|
|
2570
2164
|
self._invalidate_snapshots()
|
|
2571
|
-
return response.json()
|
|
2165
|
+
return cast(bool, response.json())
|
|
2572
2166
|
|
|
2573
|
-
def _scale_path_group(self, page_index, group_id, factor):
|
|
2167
|
+
def _scale_path_group(self, page_index: int, group_id: str, factor: float) -> bool:
|
|
2574
2168
|
if factor <= 0:
|
|
2575
2169
|
raise ValidationException("Scale factor must be positive")
|
|
2576
2170
|
return self._transform_path_group(
|
|
2577
2171
|
page_index, group_id, "SCALE", scaleFactor=factor
|
|
2578
2172
|
)
|
|
2579
2173
|
|
|
2580
|
-
def _rotate_path_group(
|
|
2174
|
+
def _rotate_path_group(
|
|
2175
|
+
self, page_index: int, group_id: str, degrees: float
|
|
2176
|
+
) -> bool:
|
|
2581
2177
|
return self._transform_path_group(
|
|
2582
2178
|
page_index, group_id, "ROTATE", rotationAngle=degrees
|
|
2583
2179
|
)
|
|
2584
2180
|
|
|
2585
|
-
def _resize_path_group(
|
|
2181
|
+
def _resize_path_group(
|
|
2182
|
+
self, page_index: int, group_id: str, width: float, height: float
|
|
2183
|
+
) -> bool:
|
|
2586
2184
|
if width <= 0 or height <= 0:
|
|
2587
2185
|
raise ValidationException("Width and height must be positive")
|
|
2588
2186
|
return self._transform_path_group(
|
|
@@ -2593,18 +2191,18 @@ class PDFDancer:
|
|
|
2593
2191
|
targetHeight=height,
|
|
2594
2192
|
)
|
|
2595
2193
|
|
|
2596
|
-
def _remove_path_group(self, page_index, group_id):
|
|
2194
|
+
def _remove_path_group(self, page_index: int, group_id: str) -> bool:
|
|
2597
2195
|
data = {"pageIndex": page_index, "groupId": group_id}
|
|
2598
2196
|
response = self._make_request("DELETE", "/pdf/path-group/remove", data=data)
|
|
2599
2197
|
self._invalidate_snapshots()
|
|
2600
|
-
return response.json()
|
|
2198
|
+
return cast(bool, response.json())
|
|
2601
2199
|
|
|
2602
|
-
def _list_path_groups(self,
|
|
2603
|
-
from .models import PathGroupInfo
|
|
2200
|
+
def _list_path_groups(self, page_number: int) -> List["PathGroupObject"]:
|
|
2604
2201
|
from .types import PathGroupObject
|
|
2605
2202
|
|
|
2606
|
-
response = self._make_request("GET", f"/pdf/page/{
|
|
2203
|
+
response = self._make_request("GET", f"/pdf/page/{page_number}/path-groups")
|
|
2607
2204
|
infos = [PathGroupInfo.from_dict(d) for d in response.json()]
|
|
2205
|
+
page_index = page_number - 1
|
|
2608
2206
|
return [PathGroupObject(self, page_index, info) for info in infos]
|
|
2609
2207
|
|
|
2610
2208
|
def clear_path_group_clipping(self, page_number: int, group_id: str) -> bool:
|
|
@@ -2622,7 +2220,7 @@ class PDFDancer:
|
|
|
2622
2220
|
|
|
2623
2221
|
def _clear_path_group_clipping(self, page_number: int, group_id: str) -> bool:
|
|
2624
2222
|
"""
|
|
2625
|
-
Internal helper to clear clipping from a path group using the
|
|
2223
|
+
Internal helper to clear clipping from a path group using the V2 API.
|
|
2626
2224
|
"""
|
|
2627
2225
|
if page_number is None:
|
|
2628
2226
|
raise ValidationException("page_number cannot be null")
|
|
@@ -2649,11 +2247,14 @@ class PDFDancer:
|
|
|
2649
2247
|
|
|
2650
2248
|
return result
|
|
2651
2249
|
|
|
2652
|
-
def
|
|
2653
|
-
|
|
2250
|
+
def text(self) -> TextClient:
|
|
2251
|
+
"""Return document-scoped selector-based text editing operations."""
|
|
2252
|
+
return TextClient(self)
|
|
2654
2253
|
|
|
2655
2254
|
def new_page(
|
|
2656
|
-
self,
|
|
2255
|
+
self,
|
|
2256
|
+
orientation: Optional[Union[Orientation, str]] = Orientation.PORTRAIT,
|
|
2257
|
+
size: Optional[Union[PageSize, str, Mapping[str, Any]]] = PageSize.A4,
|
|
2657
2258
|
) -> PageBuilder:
|
|
2658
2259
|
builder = PageBuilder(self)
|
|
2659
2260
|
if orientation is not None:
|
|
@@ -2665,98 +2266,23 @@ class PDFDancer:
|
|
|
2665
2266
|
def new_image(self) -> ImageBuilder:
|
|
2666
2267
|
return ImageBuilder(self)
|
|
2667
2268
|
|
|
2668
|
-
def new_path(self) -> "PathBuilder":
|
|
2269
|
+
def new_path(self, page_number: int) -> "PathBuilder":
|
|
2669
2270
|
from .path_builder import PathBuilder
|
|
2670
2271
|
|
|
2671
|
-
return PathBuilder(self)
|
|
2672
|
-
|
|
2673
|
-
# Modify Operations
|
|
2674
|
-
def _modify_paragraph(
|
|
2675
|
-
self, object_ref: ObjectRef, new_paragraph: Union[Paragraph, str]
|
|
2676
|
-
) -> CommandResult:
|
|
2677
|
-
"""
|
|
2678
|
-
Modifies a paragraph object or its text content.
|
|
2679
|
-
|
|
2680
|
-
Args:
|
|
2681
|
-
object_ref: Reference to the paragraph to modify
|
|
2682
|
-
new_paragraph: New paragraph object or text string
|
|
2683
|
-
|
|
2684
|
-
Returns:
|
|
2685
|
-
True if the paragraph was successfully modified
|
|
2686
|
-
"""
|
|
2687
|
-
if object_ref is None:
|
|
2688
|
-
raise ValidationException("Object reference cannot be null")
|
|
2689
|
-
if new_paragraph is None:
|
|
2690
|
-
return CommandResult.empty("ModifyParagraph", object_ref.internal_id)
|
|
2691
|
-
|
|
2692
|
-
if isinstance(new_paragraph, str):
|
|
2693
|
-
# Text modification - returns CommandResult
|
|
2694
|
-
request_data = ModifyTextRequest(object_ref, new_paragraph).to_dict()
|
|
2695
|
-
response = self._make_request(
|
|
2696
|
-
"PUT", "/pdf/text/paragraph", data=request_data
|
|
2697
|
-
)
|
|
2698
|
-
result = CommandResult.from_dict(response.json())
|
|
2699
|
-
else:
|
|
2700
|
-
# Object modification
|
|
2701
|
-
request_data = ModifyRequest(object_ref, new_paragraph).to_dict()
|
|
2702
|
-
response = self._make_request("PUT", "/pdf/modify", data=request_data)
|
|
2703
|
-
result = CommandResult.from_dict(response.json())
|
|
2704
|
-
|
|
2705
|
-
# Invalidate snapshot caches after mutation
|
|
2706
|
-
self._invalidate_snapshots()
|
|
2707
|
-
return result
|
|
2708
|
-
|
|
2709
|
-
def _modify_text_line(self, object_ref: ObjectRef, new_text: str) -> CommandResult:
|
|
2710
|
-
"""
|
|
2711
|
-
Modifies a text line object.
|
|
2712
|
-
|
|
2713
|
-
Args:
|
|
2714
|
-
object_ref: Reference to the text line to modify
|
|
2715
|
-
new_text: New text content
|
|
2716
|
-
|
|
2717
|
-
Returns:
|
|
2718
|
-
True if the text line was successfully modified
|
|
2719
|
-
"""
|
|
2720
|
-
if object_ref is None:
|
|
2721
|
-
raise ValidationException("Object reference cannot be null")
|
|
2722
|
-
if new_text is None:
|
|
2723
|
-
raise ValidationException("New text cannot be null")
|
|
2724
|
-
|
|
2725
|
-
request_data = ModifyTextRequest(object_ref, new_text).to_dict()
|
|
2726
|
-
response = self._make_request("PUT", "/pdf/text/line", data=request_data)
|
|
2727
|
-
result = CommandResult.from_dict(response.json())
|
|
2272
|
+
return PathBuilder(self, page_number)
|
|
2728
2273
|
|
|
2729
|
-
|
|
2730
|
-
self
|
|
2731
|
-
return result
|
|
2732
|
-
|
|
2733
|
-
def _modify_text_line_full(
|
|
2734
|
-
self, object_ref: ObjectRef, new_text_line: TextLine
|
|
2735
|
-
) -> CommandResult:
|
|
2736
|
-
"""
|
|
2737
|
-
Modifies a text line object with full styling (font, color, position).
|
|
2738
|
-
|
|
2739
|
-
Args:
|
|
2740
|
-
object_ref: Reference to the text line to modify
|
|
2741
|
-
new_text_line: New text line object with styling
|
|
2274
|
+
def new_line(self, page_number: int) -> LineBuilder:
|
|
2275
|
+
return LineBuilder(self, page_number)
|
|
2742
2276
|
|
|
2743
|
-
|
|
2744
|
-
|
|
2745
|
-
"""
|
|
2746
|
-
if object_ref is None:
|
|
2747
|
-
raise ValidationException("Object reference cannot be null")
|
|
2748
|
-
if new_text_line is None:
|
|
2749
|
-
raise ValidationException("New text line cannot be null")
|
|
2277
|
+
def new_bezier(self, page_number: int) -> BezierBuilder:
|
|
2278
|
+
return BezierBuilder(self, page_number)
|
|
2750
2279
|
|
|
2751
|
-
|
|
2752
|
-
|
|
2753
|
-
response = self._make_request("PUT", "/pdf/modify", data=request_data)
|
|
2754
|
-
result = CommandResult.from_dict(response.json())
|
|
2280
|
+
def new_rectangle(self, page_number: int) -> "RectangleBuilder":
|
|
2281
|
+
from .path_builder import RectangleBuilder
|
|
2755
2282
|
|
|
2756
|
-
|
|
2757
|
-
self._invalidate_snapshots()
|
|
2758
|
-
return result
|
|
2283
|
+
return RectangleBuilder(self, page_number)
|
|
2759
2284
|
|
|
2285
|
+
# Modify Operations
|
|
2760
2286
|
def _modify_path(
|
|
2761
2287
|
self,
|
|
2762
2288
|
object_ref: ObjectRef,
|
|
@@ -2879,11 +2405,32 @@ class PDFDancer:
|
|
|
2879
2405
|
"X-Session-Id": self._session_id,
|
|
2880
2406
|
"X-Generated-At": _generate_timestamp(),
|
|
2881
2407
|
}
|
|
2882
|
-
|
|
2883
|
-
|
|
2884
|
-
|
|
2885
|
-
|
|
2886
|
-
|
|
2408
|
+
|
|
2409
|
+
def log_font_register_attempt(attempt: int) -> None:
|
|
2410
|
+
if DEBUG:
|
|
2411
|
+
retry_info = (
|
|
2412
|
+
f" (attempt {attempt + 1}/{self._max_attempts})"
|
|
2413
|
+
if attempt > 0
|
|
2414
|
+
else ""
|
|
2415
|
+
)
|
|
2416
|
+
print(
|
|
2417
|
+
f"{time.time()}|POST /font/register{retry_info} - request size: {request_size} bytes"
|
|
2418
|
+
)
|
|
2419
|
+
|
|
2420
|
+
def request_font_register() -> httpx.Response:
|
|
2421
|
+
return self._client.post(
|
|
2422
|
+
self._cleanup_url_path(self._base_url, "/font/register"),
|
|
2423
|
+
files=files,
|
|
2424
|
+
headers=headers,
|
|
2425
|
+
timeout=30,
|
|
2426
|
+
)
|
|
2427
|
+
|
|
2428
|
+
response = _execute_request_with_retries(
|
|
2429
|
+
request_callable=request_font_register,
|
|
2430
|
+
operation="POST /font/register",
|
|
2431
|
+
max_attempts=self._max_attempts,
|
|
2432
|
+
retry_backoff_factor=self._retry_backoff_factor,
|
|
2433
|
+
pre_request_hook=log_font_register_attempt,
|
|
2887
2434
|
)
|
|
2888
2435
|
|
|
2889
2436
|
response_size = len(response.content)
|
|
@@ -2919,7 +2466,7 @@ class PDFDancer:
|
|
|
2919
2466
|
Retrieve a snapshot of the entire document with all pages and elements.
|
|
2920
2467
|
|
|
2921
2468
|
Args:
|
|
2922
|
-
types: Optional comma-separated string of object types to filter (e.g., "
|
|
2469
|
+
types: Optional comma-separated string of object types to filter (e.g., "TEXT_LINE,IMAGE")
|
|
2923
2470
|
|
|
2924
2471
|
Returns:
|
|
2925
2472
|
DocumentSnapshot containing page count, fonts, and all page snapshots
|
|
@@ -2941,7 +2488,7 @@ class PDFDancer:
|
|
|
2941
2488
|
|
|
2942
2489
|
Args:
|
|
2943
2490
|
page_number: The page number to snapshot (1-based, page 1 is first page)
|
|
2944
|
-
types: Optional comma-separated string of object types to filter (e.g., "
|
|
2491
|
+
types: Optional comma-separated string of object types to filter (e.g., "TEXT_LINE,IMAGE")
|
|
2945
2492
|
|
|
2946
2493
|
Returns:
|
|
2947
2494
|
PageSnapshot containing page reference and all elements on that page
|
|
@@ -3014,11 +2561,11 @@ class PDFDancer:
|
|
|
3014
2561
|
|
|
3015
2562
|
def _filter_snapshot_elements(
|
|
3016
2563
|
self,
|
|
3017
|
-
elements: List,
|
|
3018
|
-
object_type: ObjectType,
|
|
2564
|
+
elements: List[ObjectRef],
|
|
2565
|
+
object_type: Optional[ObjectType],
|
|
3019
2566
|
position: Optional[Position] = None,
|
|
3020
2567
|
tolerance: float = DEFAULT_TOLERANCE,
|
|
3021
|
-
) -> List:
|
|
2568
|
+
) -> List[ObjectRef]:
|
|
3022
2569
|
"""
|
|
3023
2570
|
Filter snapshot elements client-side based on object type and position criteria.
|
|
3024
2571
|
|
|
@@ -3034,12 +2581,14 @@ class PDFDancer:
|
|
|
3034
2581
|
import re
|
|
3035
2582
|
|
|
3036
2583
|
# Filter by object type (handle form field subtypes)
|
|
3037
|
-
if object_type
|
|
3038
|
-
|
|
2584
|
+
if object_type is None:
|
|
2585
|
+
filtered = list(elements)
|
|
2586
|
+
elif object_type == ObjectType.FORM_FIELD:
|
|
2587
|
+
# Form fields include TEXT_FIELD, CHECKBOX, RADIO_BUTTON, BUTTON, DROPDOWN
|
|
3039
2588
|
form_field_types = {
|
|
3040
2589
|
ObjectType.FORM_FIELD,
|
|
3041
2590
|
ObjectType.TEXT_FIELD,
|
|
3042
|
-
ObjectType.
|
|
2591
|
+
ObjectType.CHECKBOX,
|
|
3043
2592
|
ObjectType.RADIO_BUTTON,
|
|
3044
2593
|
ObjectType.BUTTON,
|
|
3045
2594
|
ObjectType.DROPDOWN,
|
|
@@ -3098,7 +2647,11 @@ class PDFDancer:
|
|
|
3098
2647
|
return result
|
|
3099
2648
|
|
|
3100
2649
|
@staticmethod
|
|
3101
|
-
def _rects_intersect(
|
|
2650
|
+
def _rects_intersect(
|
|
2651
|
+
rect1: "ModelBoundingRect",
|
|
2652
|
+
rect2: "ModelBoundingRect",
|
|
2653
|
+
tolerance: float = DEFAULT_TOLERANCE,
|
|
2654
|
+
) -> bool:
|
|
3102
2655
|
"""
|
|
3103
2656
|
Check if two bounding rectangles intersect or are very close.
|
|
3104
2657
|
Handles point queries (width/height = 0) with tolerance.
|
|
@@ -3165,7 +2718,7 @@ class PDFDancer:
|
|
|
3165
2718
|
|
|
3166
2719
|
# Utility Methods
|
|
3167
2720
|
|
|
3168
|
-
def _parse_object_ref(self, obj_data: dict) -> ObjectRef:
|
|
2721
|
+
def _parse_object_ref(self, obj_data: dict[str, Any]) -> ObjectRef:
|
|
3169
2722
|
"""Parse JSON object data into ObjectRef instance."""
|
|
3170
2723
|
position_data = obj_data.get("position", {})
|
|
3171
2724
|
position = self._parse_position(position_data) if position_data else None
|
|
@@ -3173,12 +2726,12 @@ class PDFDancer:
|
|
|
3173
2726
|
object_type = ObjectType(obj_data["type"])
|
|
3174
2727
|
|
|
3175
2728
|
return ObjectRef(
|
|
3176
|
-
internal_id=obj_data
|
|
3177
|
-
position=position,
|
|
2729
|
+
internal_id=cast(str, obj_data.get("internalId")),
|
|
2730
|
+
position=cast(Position, position),
|
|
3178
2731
|
type=object_type,
|
|
3179
2732
|
)
|
|
3180
2733
|
|
|
3181
|
-
def _parse_form_field_ref(self, obj_data: dict) -> FormFieldRef:
|
|
2734
|
+
def _parse_form_field_ref(self, obj_data: dict[str, Any]) -> FormFieldRef:
|
|
3182
2735
|
"""Parse JSON object data into ObjectRef instance."""
|
|
3183
2736
|
position_data = obj_data.get("position", {})
|
|
3184
2737
|
position = self._parse_position(position_data) if position_data else None
|
|
@@ -3186,14 +2739,14 @@ class PDFDancer:
|
|
|
3186
2739
|
object_type = ObjectType(obj_data["type"])
|
|
3187
2740
|
|
|
3188
2741
|
return FormFieldRef(
|
|
3189
|
-
internal_id=obj_data
|
|
3190
|
-
position=position,
|
|
2742
|
+
internal_id=cast(str, obj_data.get("internalId")),
|
|
2743
|
+
position=cast(Position, position),
|
|
3191
2744
|
type=object_type,
|
|
3192
2745
|
name=obj_data["name"] if "name" in obj_data else None,
|
|
3193
2746
|
value=obj_data["value"] if "value" in obj_data else None,
|
|
3194
2747
|
)
|
|
3195
2748
|
|
|
3196
|
-
def _parse_path_object_ref(self, obj_data: dict) -> PathObjectRef:
|
|
2749
|
+
def _parse_path_object_ref(self, obj_data: dict[str, Any]) -> PathObjectRef:
|
|
3197
2750
|
"""Parse JSON object data into PathObjectRef instance with color information."""
|
|
3198
2751
|
position_data = obj_data.get("position", {})
|
|
3199
2752
|
position = self._parse_position(position_data) if position_data else None
|
|
@@ -3209,7 +2762,12 @@ class PDFDancer:
|
|
|
3209
2762
|
blue = stroke_color_data.get("blue")
|
|
3210
2763
|
alpha = stroke_color_data.get("alpha", 255)
|
|
3211
2764
|
if all(isinstance(v, int) for v in [red, green, blue]):
|
|
3212
|
-
stroke_color = Color(
|
|
2765
|
+
stroke_color = Color(
|
|
2766
|
+
cast(int, red),
|
|
2767
|
+
cast(int, green),
|
|
2768
|
+
cast(int, blue),
|
|
2769
|
+
cast(int, alpha),
|
|
2770
|
+
)
|
|
3213
2771
|
|
|
3214
2772
|
# Parse fill color if present
|
|
3215
2773
|
fill_color = None
|
|
@@ -3220,22 +2778,28 @@ class PDFDancer:
|
|
|
3220
2778
|
blue = fill_color_data.get("blue")
|
|
3221
2779
|
alpha = fill_color_data.get("alpha", 255)
|
|
3222
2780
|
if all(isinstance(v, int) for v in [red, green, blue]):
|
|
3223
|
-
fill_color = Color(
|
|
2781
|
+
fill_color = Color(
|
|
2782
|
+
cast(int, red),
|
|
2783
|
+
cast(int, green),
|
|
2784
|
+
cast(int, blue),
|
|
2785
|
+
cast(int, alpha),
|
|
2786
|
+
)
|
|
3224
2787
|
|
|
3225
2788
|
return PathObjectRef(
|
|
3226
|
-
internal_id=obj_data
|
|
3227
|
-
position=position,
|
|
2789
|
+
internal_id=cast(str, obj_data.get("internalId")),
|
|
2790
|
+
position=cast(Position, position),
|
|
3228
2791
|
object_type=object_type,
|
|
3229
2792
|
stroke_color=stroke_color,
|
|
3230
2793
|
fill_color=fill_color,
|
|
3231
2794
|
)
|
|
3232
2795
|
|
|
3233
2796
|
@staticmethod
|
|
3234
|
-
def _parse_position(pos_data: dict) -> Position:
|
|
2797
|
+
def _parse_position(pos_data: dict[str, Any]) -> Position:
|
|
3235
2798
|
"""Parse JSON position data into Position instance."""
|
|
3236
2799
|
position = Position()
|
|
3237
2800
|
position.page_number = pos_data.get("pageNumber")
|
|
3238
2801
|
position.text_starts_with = pos_data.get("textStartsWith")
|
|
2802
|
+
position.text_pattern = pos_data.get("textPattern")
|
|
3239
2803
|
|
|
3240
2804
|
if "shape" in pos_data:
|
|
3241
2805
|
position.shape = ShapeType(pos_data["shape"])
|
|
@@ -3256,7 +2820,7 @@ class PDFDancer:
|
|
|
3256
2820
|
return position
|
|
3257
2821
|
|
|
3258
2822
|
def _parse_text_object_ref(
|
|
3259
|
-
self, obj_data: dict, fallback_id: Optional[str] = None
|
|
2823
|
+
self, obj_data: dict[str, Any], fallback_id: Optional[str] = None
|
|
3260
2824
|
) -> TextObjectRef:
|
|
3261
2825
|
"""Parse JSON object data into TextObjectRef instance with hierarchical structure."""
|
|
3262
2826
|
position_data = obj_data.get("position", {})
|
|
@@ -3280,7 +2844,12 @@ class PDFDancer:
|
|
|
3280
2844
|
blue = color_data.get("blue")
|
|
3281
2845
|
alpha = color_data.get("alpha", 255)
|
|
3282
2846
|
if all(isinstance(v, int) for v in [red, green, blue]):
|
|
3283
|
-
color = Color(
|
|
2847
|
+
color = Color(
|
|
2848
|
+
cast(int, red),
|
|
2849
|
+
cast(int, green),
|
|
2850
|
+
cast(int, blue),
|
|
2851
|
+
cast(int, alpha),
|
|
2852
|
+
)
|
|
3284
2853
|
|
|
3285
2854
|
# Parse status if present
|
|
3286
2855
|
status = None
|
|
@@ -3306,7 +2875,7 @@ class PDFDancer:
|
|
|
3306
2875
|
)
|
|
3307
2876
|
|
|
3308
2877
|
text_object = TextObjectRef(
|
|
3309
|
-
internal_id=internal_id,
|
|
2878
|
+
internal_id=cast(str, internal_id),
|
|
3310
2879
|
position=position,
|
|
3311
2880
|
object_type=object_type,
|
|
3312
2881
|
text=(
|
|
@@ -3343,7 +2912,7 @@ class PDFDancer:
|
|
|
3343
2912
|
|
|
3344
2913
|
return text_object
|
|
3345
2914
|
|
|
3346
|
-
def _parse_page_ref(self, obj_data: dict) -> PageRef:
|
|
2915
|
+
def _parse_page_ref(self, obj_data: dict[str, Any]) -> PageRef:
|
|
3347
2916
|
"""Parse JSON object data into PageRef instance with page-specific properties."""
|
|
3348
2917
|
position_data = obj_data.get("position", {})
|
|
3349
2918
|
position = self._parse_position(position_data) if position_data else None
|
|
@@ -3372,14 +2941,14 @@ class PDFDancer:
|
|
|
3372
2941
|
orientation = orientation_value
|
|
3373
2942
|
|
|
3374
2943
|
return PageRef(
|
|
3375
|
-
internal_id=obj_data.get("internalId"),
|
|
3376
|
-
position=position,
|
|
2944
|
+
internal_id=cast(str, obj_data.get("internalId")),
|
|
2945
|
+
position=cast(Position, position),
|
|
3377
2946
|
type=object_type,
|
|
3378
2947
|
page_size=page_size,
|
|
3379
2948
|
orientation=orientation,
|
|
3380
2949
|
)
|
|
3381
2950
|
|
|
3382
|
-
def _parse_path_segment(self, segment_data: dict) -> "PathSegment":
|
|
2951
|
+
def _parse_path_segment(self, segment_data: dict[str, Any]) -> "PathSegment":
|
|
3383
2952
|
"""Parse JSON data into PathSegment instance (Line or Bezier)."""
|
|
3384
2953
|
from .models import Bezier, Color, Line, PathSegment, Point
|
|
3385
2954
|
|
|
@@ -3473,7 +3042,7 @@ class PDFDancer:
|
|
|
3473
3042
|
dash_phase=dash_phase,
|
|
3474
3043
|
)
|
|
3475
3044
|
|
|
3476
|
-
def _parse_path(self, obj_data: dict) ->
|
|
3045
|
+
def _parse_path(self, obj_data: dict[str, Any]) -> PDFPath:
|
|
3477
3046
|
"""Parse JSON data into Path instance with path segments."""
|
|
3478
3047
|
from .models import Path
|
|
3479
3048
|
|
|
@@ -3496,7 +3065,7 @@ class PDFDancer:
|
|
|
3496
3065
|
even_odd_fill=even_odd_fill,
|
|
3497
3066
|
)
|
|
3498
3067
|
|
|
3499
|
-
def _parse_document_font_info(self, data: dict) -> FontRecommendation:
|
|
3068
|
+
def _parse_document_font_info(self, data: dict[str, Any]) -> FontRecommendation:
|
|
3500
3069
|
"""Parse JSON data into FontRecommendation instance."""
|
|
3501
3070
|
font_type_str = data.get("fontType", "SYSTEM")
|
|
3502
3071
|
font_type = FontType(font_type_str)
|
|
@@ -3507,37 +3076,28 @@ class PDFDancer:
|
|
|
3507
3076
|
similarity_score=data.get("similarityScore", 0.0),
|
|
3508
3077
|
)
|
|
3509
3078
|
|
|
3510
|
-
def _parse_page_snapshot(self, data: dict) -> PageSnapshot:
|
|
3079
|
+
def _parse_page_snapshot(self, data: dict[str, Any]) -> PageSnapshot:
|
|
3511
3080
|
"""Parse JSON data into PageSnapshot instance with proper type handling."""
|
|
3512
3081
|
page_ref = self._parse_page_ref(data.get("pageRef", {}))
|
|
3513
3082
|
|
|
3514
3083
|
# Parse elements using appropriate parser based on type
|
|
3515
|
-
elements = []
|
|
3084
|
+
elements: List[ObjectRef] = []
|
|
3516
3085
|
for elem_data in data.get("elements", []):
|
|
3517
3086
|
elem_type_str = elem_data.get("type")
|
|
3518
3087
|
if not elem_type_str:
|
|
3519
3088
|
continue
|
|
3520
3089
|
|
|
3521
3090
|
try:
|
|
3522
|
-
# Normalize type string (API returns "CHECKBOX" but enum is "CHECK_BOX")
|
|
3523
|
-
if elem_type_str == "CHECKBOX":
|
|
3524
|
-
elem_type_str = "CHECK_BOX"
|
|
3525
|
-
# Deep copy to avoid modifying original
|
|
3526
|
-
import copy
|
|
3527
|
-
|
|
3528
|
-
elem_data = copy.deepcopy(elem_data)
|
|
3529
|
-
elem_data["type"] = elem_type_str # Update type in data
|
|
3530
|
-
|
|
3531
3091
|
elem_type = ObjectType(elem_type_str)
|
|
3532
3092
|
|
|
3533
3093
|
# Use appropriate parser based on element type
|
|
3534
|
-
if elem_type
|
|
3094
|
+
if elem_type == ObjectType.TEXT_LINE:
|
|
3535
3095
|
# Parse as TextObjectRef to capture text, font, color, children
|
|
3536
3096
|
elements.append(self._parse_text_object_ref(elem_data))
|
|
3537
3097
|
elif elem_type in (
|
|
3538
3098
|
ObjectType.FORM_FIELD,
|
|
3539
3099
|
ObjectType.TEXT_FIELD,
|
|
3540
|
-
ObjectType.
|
|
3100
|
+
ObjectType.CHECKBOX,
|
|
3541
3101
|
ObjectType.RADIO_BUTTON,
|
|
3542
3102
|
ObjectType.BUTTON,
|
|
3543
3103
|
ObjectType.DROPDOWN,
|
|
@@ -3556,7 +3116,7 @@ class PDFDancer:
|
|
|
3556
3116
|
|
|
3557
3117
|
return PageSnapshot(page_ref=page_ref, elements=elements)
|
|
3558
3118
|
|
|
3559
|
-
def _parse_document_snapshot(self, data: dict) -> DocumentSnapshot:
|
|
3119
|
+
def _parse_document_snapshot(self, data: dict[str, Any]) -> DocumentSnapshot:
|
|
3560
3120
|
"""Parse JSON data into DocumentSnapshot instance."""
|
|
3561
3121
|
page_count = data.get("pageCount", 0)
|
|
3562
3122
|
fonts = [
|
|
@@ -3569,24 +3129,17 @@ class PDFDancer:
|
|
|
3569
3129
|
|
|
3570
3130
|
return DocumentSnapshot(page_count=page_count, fonts=fonts, pages=pages)
|
|
3571
3131
|
|
|
3572
|
-
# Builder Pattern Support
|
|
3573
|
-
|
|
3574
|
-
def _paragraph_builder(self) -> "ParagraphBuilder":
|
|
3575
|
-
"""
|
|
3576
|
-
Creates a new ParagraphBuilder for fluent paragraph construction.
|
|
3577
|
-
Returns:
|
|
3578
|
-
A new ParagraphBuilder instance
|
|
3579
|
-
"""
|
|
3580
|
-
from .paragraph_builder import ParagraphBuilder
|
|
3581
|
-
|
|
3582
|
-
return ParagraphBuilder(self)
|
|
3583
|
-
|
|
3584
3132
|
# Context Manager Support (Python enhancement)
|
|
3585
|
-
def __enter__(self):
|
|
3133
|
+
def __enter__(self) -> "PDFDancer":
|
|
3586
3134
|
"""Context manager entry."""
|
|
3587
3135
|
return self
|
|
3588
3136
|
|
|
3589
|
-
def __exit__(
|
|
3137
|
+
def __exit__(
|
|
3138
|
+
self,
|
|
3139
|
+
exc_type: Optional[type[BaseException]],
|
|
3140
|
+
exc_val: Optional[BaseException],
|
|
3141
|
+
exc_tb: Any,
|
|
3142
|
+
) -> None:
|
|
3590
3143
|
"""Context manager exit - cleanup if needed."""
|
|
3591
3144
|
# Close the HTTP client to free resources
|
|
3592
3145
|
if hasattr(self, "_client"):
|
|
@@ -3594,7 +3147,7 @@ class PDFDancer:
|
|
|
3594
3147
|
# TODO Could add session cleanup here if API supports it. Cleanup on the server
|
|
3595
3148
|
pass
|
|
3596
3149
|
|
|
3597
|
-
def close(self):
|
|
3150
|
+
def close(self) -> None:
|
|
3598
3151
|
"""Close the HTTP client and free resources."""
|
|
3599
3152
|
if hasattr(self, "_client"):
|
|
3600
3153
|
self._client.close()
|
|
@@ -3602,12 +3155,6 @@ class PDFDancer:
|
|
|
3602
3155
|
def _to_path_objects(self, refs: List[ObjectRef]) -> List[PathObject]:
|
|
3603
3156
|
return [PathObject(self, ref) for ref in refs]
|
|
3604
3157
|
|
|
3605
|
-
def _to_paragraph_objects(self, refs: List[TextObjectRef]) -> List[ParagraphObject]:
|
|
3606
|
-
return [ParagraphObject(self, ref) for ref in refs]
|
|
3607
|
-
|
|
3608
|
-
def _to_textline_objects(self, refs: List[TextObjectRef]) -> List[TextLineObject]:
|
|
3609
|
-
return [TextLineObject(self, ref) for ref in refs]
|
|
3610
|
-
|
|
3611
3158
|
def _to_image_objects(self, refs: List[ObjectRef]) -> List[ImageObject]:
|
|
3612
3159
|
return [
|
|
3613
3160
|
ImageObject(self, ref.internal_id, ref.type, ref.position) for ref in refs
|
|
@@ -3621,7 +3168,12 @@ class PDFDancer:
|
|
|
3621
3168
|
def _to_form_field_objects(self, refs: List[FormFieldRef]) -> List[FormFieldObject]:
|
|
3622
3169
|
return [
|
|
3623
3170
|
FormFieldObject(
|
|
3624
|
-
self,
|
|
3171
|
+
self,
|
|
3172
|
+
ref.internal_id,
|
|
3173
|
+
ref.type,
|
|
3174
|
+
ref.position,
|
|
3175
|
+
cast(str, ref.name),
|
|
3176
|
+
cast(str, ref.value),
|
|
3625
3177
|
)
|
|
3626
3178
|
for ref in refs
|
|
3627
3179
|
]
|
|
@@ -3632,33 +3184,21 @@ class PDFDancer:
|
|
|
3632
3184
|
def _to_page_object(self, ref: PageRef) -> PageClient:
|
|
3633
3185
|
return PageClient.from_ref(self, ref)
|
|
3634
3186
|
|
|
3635
|
-
def _to_mixed_objects(
|
|
3187
|
+
def _to_mixed_objects(
|
|
3188
|
+
self, refs: List[ObjectRef]
|
|
3189
|
+
) -> List[Union[ImageObject, PathObject, FormObject, FormFieldObject]]:
|
|
3636
3190
|
"""
|
|
3637
3191
|
Convert a list of ObjectRefs to their appropriate object types.
|
|
3638
3192
|
Handles mixed object types by checking the type of each ref.
|
|
3639
3193
|
"""
|
|
3640
|
-
result = []
|
|
3194
|
+
result: List[Union[ImageObject, PathObject, FormObject, FormFieldObject]] = []
|
|
3641
3195
|
for ref in refs:
|
|
3642
|
-
if ref.type == ObjectType.
|
|
3643
|
-
# Need to convert to TextObjectRef first
|
|
3644
|
-
if isinstance(ref, TextObjectRef):
|
|
3645
|
-
result.append(ParagraphObject(self, ref))
|
|
3646
|
-
else:
|
|
3647
|
-
# Re-fetch with proper type
|
|
3648
|
-
text_refs = self._find_paragraphs(ref.position)
|
|
3649
|
-
result.extend(self._to_paragraph_objects(text_refs))
|
|
3650
|
-
elif ref.type == ObjectType.TEXT_LINE:
|
|
3651
|
-
if isinstance(ref, TextObjectRef):
|
|
3652
|
-
result.append(TextLineObject(self, ref))
|
|
3653
|
-
else:
|
|
3654
|
-
text_refs = self._find_text_lines(ref.position)
|
|
3655
|
-
result.extend(self._to_textline_objects(text_refs))
|
|
3656
|
-
elif ref.type == ObjectType.IMAGE:
|
|
3196
|
+
if ref.type == ObjectType.IMAGE:
|
|
3657
3197
|
result.append(
|
|
3658
3198
|
ImageObject(self, ref.internal_id, ref.type, ref.position)
|
|
3659
3199
|
)
|
|
3660
3200
|
elif ref.type == ObjectType.PATH:
|
|
3661
|
-
result.append(PathObject(self, ref
|
|
3201
|
+
result.append(PathObject(self, ref))
|
|
3662
3202
|
elif ref.type == ObjectType.FORM_X_OBJECT:
|
|
3663
3203
|
result.append(FormObject(self, ref.internal_id, ref.type, ref.position))
|
|
3664
3204
|
elif ref.type == ObjectType.FORM_FIELD:
|
|
@@ -3669,8 +3209,8 @@ class PDFDancer:
|
|
|
3669
3209
|
ref.internal_id,
|
|
3670
3210
|
ref.type,
|
|
3671
3211
|
ref.position,
|
|
3672
|
-
ref.name,
|
|
3673
|
-
ref.value,
|
|
3212
|
+
cast(str, ref.name),
|
|
3213
|
+
cast(str, ref.value),
|
|
3674
3214
|
)
|
|
3675
3215
|
)
|
|
3676
3216
|
else:
|
|
@@ -3678,18 +3218,12 @@ class PDFDancer:
|
|
|
3678
3218
|
result.extend(self._to_form_field_objects(form_refs))
|
|
3679
3219
|
return result
|
|
3680
3220
|
|
|
3681
|
-
def select_elements(self):
|
|
3221
|
+
def select_elements(self) -> List[ObjectRef]:
|
|
3682
3222
|
"""
|
|
3683
|
-
Select all
|
|
3223
|
+
Select all live object-reference elements in the document.
|
|
3684
3224
|
|
|
3685
3225
|
Returns:
|
|
3686
3226
|
List of all PDF objects in the document
|
|
3687
3227
|
"""
|
|
3688
|
-
|
|
3689
|
-
|
|
3690
|
-
result.extend(self.select_text_lines())
|
|
3691
|
-
result.extend(self.select_images())
|
|
3692
|
-
result.extend(self.select_paths())
|
|
3693
|
-
result.extend(self.select_forms())
|
|
3694
|
-
result.extend(self.select_form_fields())
|
|
3695
|
-
return result
|
|
3228
|
+
snapshot = self._get_or_fetch_document_snapshot()
|
|
3229
|
+
return [element for page in snapshot.pages for element in page.elements]
|