pdfdancer-client-python 0.3.14__py3-none-any.whl → 3.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pdfdancer/__init__.py +77 -22
- pdfdancer/_runtime_version.py +8 -3
- pdfdancer/_version.py +2 -2
- pdfdancer/image_builder.py +23 -3
- pdfdancer/models.py +127 -470
- pdfdancer/page_builder.py +6 -17
- pdfdancer/path_builder.py +127 -6
- pdfdancer/{pdfdancer_v1.py → pdfdancer_v2.py} +848 -1310
- pdfdancer/text_editing.py +1472 -0
- pdfdancer/types.py +94 -399
- {pdfdancer_client_python-0.3.14.dist-info → pdfdancer_client_python-3.0.1.dist-info}/METADATA +181 -137
- pdfdancer_client_python-3.0.1.dist-info/RECORD +18 -0
- {pdfdancer_client_python-0.3.14.dist-info → pdfdancer_client_python-3.0.1.dist-info}/WHEEL +1 -1
- pdfdancer/paragraph_builder.py +0 -554
- pdfdancer/text_line_builder.py +0 -290
- pdfdancer_client_python-0.3.14.dist-info/RECORD +0 -19
- {pdfdancer_client_python-0.3.14.dist-info → pdfdancer_client_python-3.0.1.dist-info}/licenses/LICENSE +0 -0
- {pdfdancer_client_python-0.3.14.dist-info → pdfdancer_client_python-3.0.1.dist-info}/licenses/NOTICE +0 -0
- {pdfdancer_client_python-0.3.14.dist-info → pdfdancer_client_python-3.0.1.dist-info}/top_level.txt +0 -0
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
"""
|
|
2
|
-
PDFDancer Python Client
|
|
2
|
+
PDFDancer Python Client V2
|
|
3
3
|
|
|
4
4
|
A Python client that closely mirrors the Java Client class structure and functionality.
|
|
5
5
|
Provides session-based PDF manipulation operations with strict validation.
|
|
@@ -10,17 +10,29 @@ from __future__ import annotations
|
|
|
10
10
|
import gzip
|
|
11
11
|
import json
|
|
12
12
|
import logging
|
|
13
|
+
import math
|
|
13
14
|
import os
|
|
14
15
|
import sys
|
|
15
16
|
import time
|
|
16
17
|
from datetime import datetime, timezone
|
|
18
|
+
from email.utils import parsedate_to_datetime
|
|
17
19
|
from pathlib import Path
|
|
18
|
-
from typing import
|
|
20
|
+
from typing import (
|
|
21
|
+
TYPE_CHECKING,
|
|
22
|
+
Any,
|
|
23
|
+
BinaryIO,
|
|
24
|
+
Callable,
|
|
25
|
+
List,
|
|
26
|
+
Mapping,
|
|
27
|
+
Optional,
|
|
28
|
+
Union,
|
|
29
|
+
cast,
|
|
30
|
+
)
|
|
19
31
|
|
|
20
32
|
import httpx
|
|
21
|
-
from dotenv import find_dotenv, load_dotenv
|
|
22
33
|
|
|
23
|
-
from . import BezierBuilder, LineBuilder,
|
|
34
|
+
from . import BezierBuilder, LineBuilder, PathBuilder
|
|
35
|
+
from ._runtime_version import resolve_package_version
|
|
24
36
|
from .exceptions import (
|
|
25
37
|
FontNotFoundException,
|
|
26
38
|
HttpClientException,
|
|
@@ -47,8 +59,6 @@ from .models import (
|
|
|
47
59
|
FormFieldRef,
|
|
48
60
|
Image,
|
|
49
61
|
ModifyPathRequest,
|
|
50
|
-
ModifyRequest,
|
|
51
|
-
ModifyTextRequest,
|
|
52
62
|
MoveRequest,
|
|
53
63
|
ObjectRef,
|
|
54
64
|
ObjectType,
|
|
@@ -57,51 +67,42 @@ from .models import (
|
|
|
57
67
|
PageRef,
|
|
58
68
|
PageSize,
|
|
59
69
|
PageSnapshot,
|
|
60
|
-
|
|
70
|
+
)
|
|
71
|
+
from .models import Path as PDFPath
|
|
72
|
+
from .models import (
|
|
73
|
+
PathGroupInfo,
|
|
61
74
|
PathObjectRef,
|
|
62
75
|
Position,
|
|
63
76
|
PositionMode,
|
|
64
|
-
RedactRequest,
|
|
65
|
-
RedactResponse,
|
|
66
|
-
RedactTarget,
|
|
67
|
-
ReflowPreset,
|
|
68
77
|
ShapeType,
|
|
69
|
-
TemplateReplacement,
|
|
70
|
-
TemplateReplaceRequest,
|
|
71
|
-
TextLine,
|
|
72
78
|
TextObjectRef,
|
|
73
79
|
)
|
|
74
80
|
from .page_builder import PageBuilder
|
|
75
|
-
from .
|
|
76
|
-
|
|
81
|
+
from .text_editing import (
|
|
82
|
+
TextDeleteRequest,
|
|
83
|
+
TextEditResponse,
|
|
84
|
+
TextInsertRequest,
|
|
85
|
+
TextReplaceRequest,
|
|
86
|
+
TextStyleRequest,
|
|
87
|
+
)
|
|
77
88
|
from .types import (
|
|
78
89
|
FormFieldObject,
|
|
79
90
|
FormObject,
|
|
80
91
|
ImageObject,
|
|
81
|
-
ParagraphObject,
|
|
82
92
|
PathObject,
|
|
83
|
-
PDFObjectBase,
|
|
84
|
-
TextLineObject,
|
|
85
93
|
)
|
|
86
94
|
|
|
87
95
|
if TYPE_CHECKING:
|
|
96
|
+
from .models import BoundingRect as ModelBoundingRect
|
|
88
97
|
from .models import ImageTransformRequest, PathSegment
|
|
89
98
|
from .path_builder import RectangleBuilder
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
def _load_env():
|
|
95
|
-
global _env_loaded
|
|
96
|
-
if _env_loaded:
|
|
97
|
-
return
|
|
98
|
-
load_dotenv(find_dotenv(usecwd=True))
|
|
99
|
-
_env_loaded = True
|
|
100
|
-
|
|
99
|
+
from .types import BoundingRect as GroupBoundingRect
|
|
100
|
+
from .types import PathGroupObject
|
|
101
101
|
|
|
102
102
|
# Client identifier header for all HTTP requests
|
|
103
103
|
# Prefer the SCM-generated version module; fall back to installed metadata.
|
|
104
104
|
CLIENT_HEADER_VALUE = f"python/{resolve_package_version(default='unknown')}"
|
|
105
|
+
API_PATH_PREFIX = "/v2"
|
|
105
106
|
|
|
106
107
|
# Global variable to disable SSL certificate verification
|
|
107
108
|
# Set to True to skip SSL verification (useful for testing with self-signed certificates)
|
|
@@ -113,70 +114,32 @@ DEFAULT_TOLERANCE = 0.01
|
|
|
113
114
|
|
|
114
115
|
# Retry configuration for transient network errors
|
|
115
116
|
# These settings control automatic retry behavior when encountering transient network errors
|
|
116
|
-
#
|
|
117
|
-
# HTTP status errors (4xx, 5xx) are NOT retried as they are application-level errors.
|
|
117
|
+
# and transient server response statuses.
|
|
118
118
|
#
|
|
119
|
-
#
|
|
120
|
-
#
|
|
119
|
+
# PDFDANCER_MAX_ATTEMPTS: Maximum number of total attempts (default: 3).
|
|
120
|
+
# The initial request counts as one attempt, so 3 permits at most 2 retries.
|
|
121
121
|
#
|
|
122
|
-
# PDFDANCER_RETRY_BACKOFF_FACTOR:
|
|
123
|
-
# The actual delay for each retry is calculated as:
|
|
122
|
+
# PDFDANCER_RETRY_BACKOFF_FACTOR: Multiplier for exponential backoff delays (default: 2.0)
|
|
123
|
+
# The actual delay for each retry is calculated as: initial_delay * (backoff_factor ** retry_count)
|
|
124
124
|
# Examples:
|
|
125
|
-
# -
|
|
126
|
-
# -
|
|
127
|
-
|
|
128
|
-
DEFAULT_MAX_RETRIES = int(os.environ.get("PDFDANCER_MAX_RETRIES", "3"))
|
|
125
|
+
# - retry_backoff_factor=2.0: delays are 1s, 2s, 4s, 8s, ...
|
|
126
|
+
# - retry_backoff_factor=3.0: delays are 1s, 3s, 9s, ...
|
|
127
|
+
DEFAULT_MAX_ATTEMPTS = int(os.environ.get("PDFDANCER_MAX_ATTEMPTS", "3"))
|
|
129
128
|
DEFAULT_RETRY_BACKOFF_FACTOR = float(
|
|
130
129
|
os.environ.get("PDFDANCER_RETRY_BACKOFF_FACTOR", "2.0")
|
|
131
130
|
)
|
|
131
|
+
DEFAULT_RETRY_INITIAL_DELAY = 1.0
|
|
132
|
+
DEFAULT_RETRY_MAX_DELAY = 5.0
|
|
133
|
+
DEFAULT_RETRYABLE_STATUS_CODES = {408, 429, 500, 502, 503, 504, 520}
|
|
132
134
|
|
|
133
135
|
|
|
134
|
-
def
|
|
135
|
-
|
|
136
|
-
)
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
result.append(TemplateReplacement(placeholder=placeholder, text=value))
|
|
142
|
-
elif "image" in value:
|
|
143
|
-
image_source = value["image"]
|
|
144
|
-
if isinstance(image_source, Path):
|
|
145
|
-
image_data = image_source.read_bytes()
|
|
146
|
-
image_format = image_source.suffix.lstrip(".").upper()
|
|
147
|
-
if image_format == "JPG":
|
|
148
|
-
image_format = "JPEG"
|
|
149
|
-
elif isinstance(image_source, bytes):
|
|
150
|
-
image_data = image_source
|
|
151
|
-
image_format = None
|
|
152
|
-
else:
|
|
153
|
-
raise ValueError(
|
|
154
|
-
f"Unsupported image source type: {type(image_source)}. "
|
|
155
|
-
"Use a Path or bytes."
|
|
156
|
-
)
|
|
157
|
-
img = Image(
|
|
158
|
-
data=image_data,
|
|
159
|
-
format=value.get("format", image_format),
|
|
160
|
-
width=value.get("width"),
|
|
161
|
-
height=value.get("height"),
|
|
162
|
-
)
|
|
163
|
-
result.append(
|
|
164
|
-
TemplateReplacement(
|
|
165
|
-
placeholder=placeholder,
|
|
166
|
-
text=None,
|
|
167
|
-
image=img,
|
|
168
|
-
)
|
|
169
|
-
)
|
|
170
|
-
else:
|
|
171
|
-
result.append(
|
|
172
|
-
TemplateReplacement(
|
|
173
|
-
placeholder=placeholder,
|
|
174
|
-
text=value["text"],
|
|
175
|
-
font=value.get("font"),
|
|
176
|
-
color=value.get("color"),
|
|
177
|
-
)
|
|
178
|
-
)
|
|
179
|
-
return result
|
|
136
|
+
def _validate_max_attempts(max_attempts: int) -> int:
|
|
137
|
+
"""Validate that the total-attempt limit can include an initial request."""
|
|
138
|
+
if isinstance(max_attempts, bool) or not isinstance(max_attempts, int):
|
|
139
|
+
raise ValidationException("max_attempts must be an integer")
|
|
140
|
+
if max_attempts < 1:
|
|
141
|
+
raise ValidationException("max_attempts must be at least 1")
|
|
142
|
+
return max_attempts
|
|
180
143
|
|
|
181
144
|
|
|
182
145
|
def _generate_timestamp() -> str:
|
|
@@ -287,15 +250,17 @@ def _is_retryable_error(error: Exception) -> bool:
|
|
|
287
250
|
# - PoolTimeout (connection pool exhausted)
|
|
288
251
|
error_msg = str(error).lower()
|
|
289
252
|
|
|
290
|
-
if isinstance(
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
253
|
+
if isinstance(
|
|
254
|
+
error,
|
|
255
|
+
(
|
|
256
|
+
httpx.RemoteProtocolError,
|
|
257
|
+
httpx.ConnectError,
|
|
258
|
+
httpx.ConnectTimeout,
|
|
259
|
+
httpx.ReadTimeout,
|
|
260
|
+
httpx.PoolTimeout,
|
|
261
|
+
httpx.WriteTimeout,
|
|
262
|
+
),
|
|
263
|
+
):
|
|
299
264
|
return True
|
|
300
265
|
|
|
301
266
|
# Check for specific error messages that indicate transient issues
|
|
@@ -325,15 +290,167 @@ def _get_retry_after_delay(response: httpx.Response) -> Optional[int]:
|
|
|
325
290
|
return None
|
|
326
291
|
|
|
327
292
|
try:
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
return int(retry_after)
|
|
293
|
+
delay_seconds = int(retry_after)
|
|
294
|
+
return delay_seconds if delay_seconds >= 0 else None
|
|
331
295
|
except ValueError:
|
|
332
|
-
|
|
333
|
-
|
|
296
|
+
pass
|
|
297
|
+
|
|
298
|
+
try:
|
|
299
|
+
retry_at = parsedate_to_datetime(retry_after)
|
|
300
|
+
if retry_at.tzinfo is None:
|
|
301
|
+
retry_at = retry_at.replace(tzinfo=timezone.utc)
|
|
302
|
+
delay = (retry_at - datetime.now(timezone.utc)).total_seconds()
|
|
303
|
+
return max(0, math.ceil(delay))
|
|
304
|
+
except (TypeError, ValueError, OverflowError):
|
|
334
305
|
return None
|
|
335
306
|
|
|
336
307
|
|
|
308
|
+
def _calculate_retry_delay(
|
|
309
|
+
response: Optional[httpx.Response],
|
|
310
|
+
attempt: int,
|
|
311
|
+
retry_backoff_factor: float,
|
|
312
|
+
max_delay_seconds: float = DEFAULT_RETRY_MAX_DELAY,
|
|
313
|
+
) -> float:
|
|
314
|
+
"""
|
|
315
|
+
Calculate retry delay for a given attempt.
|
|
316
|
+
|
|
317
|
+
Args:
|
|
318
|
+
response: HTTP response when retrying on status code, otherwise None
|
|
319
|
+
attempt: Zero-based retry attempt index (0 for first retry)
|
|
320
|
+
retry_backoff_factor: Backoff multiplier
|
|
321
|
+
max_delay_seconds: Maximum retry delay
|
|
322
|
+
|
|
323
|
+
Returns:
|
|
324
|
+
Delay in seconds, respecting max delay and Retry-After when available.
|
|
325
|
+
"""
|
|
326
|
+
if response is not None and response.status_code == 429:
|
|
327
|
+
retry_after = _get_retry_after_delay(response)
|
|
328
|
+
if retry_after is not None:
|
|
329
|
+
return float(min(max_delay_seconds, retry_after))
|
|
330
|
+
delay = DEFAULT_RETRY_INITIAL_DELAY * (retry_backoff_factor**attempt)
|
|
331
|
+
return float(min(max_delay_seconds, delay))
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
def _execute_request_with_retries(
|
|
335
|
+
request_callable: Callable[[], httpx.Response],
|
|
336
|
+
operation: str,
|
|
337
|
+
max_attempts: int,
|
|
338
|
+
retry_backoff_factor: float,
|
|
339
|
+
retryable_status_codes: Optional[set[int]] = None,
|
|
340
|
+
pre_request_hook: Optional[Callable[[int], None]] = None,
|
|
341
|
+
) -> httpx.Response:
|
|
342
|
+
"""
|
|
343
|
+
Execute a request callable with retry handling for transient statuses and request errors.
|
|
344
|
+
|
|
345
|
+
Args:
|
|
346
|
+
request_callable: Zero-arg callable returning an httpx.Response.
|
|
347
|
+
operation: Human-readable operation name for retry logging.
|
|
348
|
+
max_attempts: Maximum number of total attempts.
|
|
349
|
+
retry_backoff_factor: Retry backoff multiplier.
|
|
350
|
+
retryable_status_codes: Status codes that should trigger retries.
|
|
351
|
+
pre_request_hook: Optional callback invoked before each request attempt.
|
|
352
|
+
|
|
353
|
+
Returns:
|
|
354
|
+
The last response returned by the request callable.
|
|
355
|
+
|
|
356
|
+
Raises:
|
|
357
|
+
httpx.RequestError: When a non-retriable request error occurs.
|
|
358
|
+
"""
|
|
359
|
+
attempts = max(1, max_attempts)
|
|
360
|
+
retry_statuses = (
|
|
361
|
+
retryable_status_codes
|
|
362
|
+
if retryable_status_codes is not None
|
|
363
|
+
else DEFAULT_RETRYABLE_STATUS_CODES
|
|
364
|
+
)
|
|
365
|
+
attempt = 0
|
|
366
|
+
last_response = None
|
|
367
|
+
|
|
368
|
+
while attempt < attempts:
|
|
369
|
+
try:
|
|
370
|
+
if pre_request_hook:
|
|
371
|
+
pre_request_hook(attempt)
|
|
372
|
+
|
|
373
|
+
response = request_callable()
|
|
374
|
+
last_response = response
|
|
375
|
+
|
|
376
|
+
if response.status_code in retry_statuses and attempt < attempts - 1:
|
|
377
|
+
delay = _calculate_retry_delay(
|
|
378
|
+
response=response,
|
|
379
|
+
attempt=attempt,
|
|
380
|
+
retry_backoff_factor=retry_backoff_factor,
|
|
381
|
+
)
|
|
382
|
+
if response.status_code == 429:
|
|
383
|
+
print(
|
|
384
|
+
f"Rate limit (429) on {operation} - retrying in {delay}s "
|
|
385
|
+
f"(attempt {attempt + 1}/{attempts})",
|
|
386
|
+
file=sys.stderr,
|
|
387
|
+
)
|
|
388
|
+
elif DEBUG:
|
|
389
|
+
print(
|
|
390
|
+
f"{time.time()}|{operation} - Retryable HTTP {response.status_code}, "
|
|
391
|
+
f"retrying in {delay}s (attempt {attempt + 1}/{attempts})"
|
|
392
|
+
)
|
|
393
|
+
|
|
394
|
+
if delay > 0:
|
|
395
|
+
time.sleep(delay)
|
|
396
|
+
attempt += 1
|
|
397
|
+
continue
|
|
398
|
+
|
|
399
|
+
return response
|
|
400
|
+
|
|
401
|
+
except httpx.HTTPStatusError as e:
|
|
402
|
+
response = e.response
|
|
403
|
+
last_response = response
|
|
404
|
+
if response.status_code in retry_statuses and attempt < attempts - 1:
|
|
405
|
+
delay = _calculate_retry_delay(
|
|
406
|
+
response=response,
|
|
407
|
+
attempt=attempt,
|
|
408
|
+
retry_backoff_factor=retry_backoff_factor,
|
|
409
|
+
)
|
|
410
|
+
if response.status_code == 429:
|
|
411
|
+
print(
|
|
412
|
+
f"Rate limit (429) on {operation} - retrying in {delay}s "
|
|
413
|
+
f"(attempt {attempt + 1}/{attempts})",
|
|
414
|
+
file=sys.stderr,
|
|
415
|
+
)
|
|
416
|
+
elif DEBUG:
|
|
417
|
+
print(
|
|
418
|
+
f"{time.time()}|{operation} - Retryable HTTP {response.status_code}, "
|
|
419
|
+
f"retrying in {delay}s (attempt {attempt + 1}/{attempts})"
|
|
420
|
+
)
|
|
421
|
+
|
|
422
|
+
if delay > 0:
|
|
423
|
+
time.sleep(delay)
|
|
424
|
+
attempt += 1
|
|
425
|
+
continue
|
|
426
|
+
raise
|
|
427
|
+
|
|
428
|
+
except httpx.RequestError as e:
|
|
429
|
+
if _is_retryable_error(e) and attempt < attempts - 1:
|
|
430
|
+
delay = _calculate_retry_delay(
|
|
431
|
+
response=None,
|
|
432
|
+
attempt=attempt,
|
|
433
|
+
retry_backoff_factor=retry_backoff_factor,
|
|
434
|
+
)
|
|
435
|
+
if DEBUG:
|
|
436
|
+
print(
|
|
437
|
+
f"{time.time()}|{operation} - Retryable error: {str(e)}, "
|
|
438
|
+
f"retrying in {delay}s (attempt {attempt + 1}/{attempts})"
|
|
439
|
+
)
|
|
440
|
+
if delay > 0:
|
|
441
|
+
time.sleep(delay)
|
|
442
|
+
attempt += 1
|
|
443
|
+
continue
|
|
444
|
+
raise
|
|
445
|
+
|
|
446
|
+
if last_response is None:
|
|
447
|
+
raise RuntimeError(
|
|
448
|
+
f"Request exhausted retries without a response for operation: {operation}"
|
|
449
|
+
)
|
|
450
|
+
|
|
451
|
+
return last_response
|
|
452
|
+
|
|
453
|
+
|
|
337
454
|
class PageClient:
|
|
338
455
|
def __init__(
|
|
339
456
|
self,
|
|
@@ -348,6 +465,7 @@ class PageClient:
|
|
|
348
465
|
self.position = Position.at_page(page_number)
|
|
349
466
|
self.internal_id = f"PAGE-{page_number}"
|
|
350
467
|
self.page_size = page_size
|
|
468
|
+
self.orientation: Optional[Union[Orientation, str]]
|
|
351
469
|
if isinstance(orientation, str):
|
|
352
470
|
normalized = orientation.strip().upper()
|
|
353
471
|
try:
|
|
@@ -364,59 +482,6 @@ class PageClient:
|
|
|
364
482
|
# noinspection PyProtectedMember
|
|
365
483
|
return self.root._to_path_objects(self.root._find_paths(position, tolerance))
|
|
366
484
|
|
|
367
|
-
def select_paragraphs(self) -> List[ParagraphObject]:
|
|
368
|
-
# noinspection PyProtectedMember
|
|
369
|
-
return self.root._to_paragraph_objects(
|
|
370
|
-
self.root._find_paragraphs(Position.at_page(self.page_number))
|
|
371
|
-
)
|
|
372
|
-
|
|
373
|
-
def select_paragraphs_starting_with(self, text: str) -> List[ParagraphObject]:
|
|
374
|
-
position = Position.at_page(self.page_number)
|
|
375
|
-
position.with_text_starts(text)
|
|
376
|
-
# noinspection PyProtectedMember
|
|
377
|
-
return self.root._to_paragraph_objects(self.root._find_paragraphs(position))
|
|
378
|
-
|
|
379
|
-
def select_paragraphs_matching(self, pattern):
|
|
380
|
-
position = Position.at_page(self.page_number)
|
|
381
|
-
position.text_pattern = pattern
|
|
382
|
-
# noinspection PyProtectedMember
|
|
383
|
-
return self.root._to_paragraph_objects(self.root._find_paragraphs(position))
|
|
384
|
-
|
|
385
|
-
def select_text_lines_matching(self, pattern: str) -> List[TextLineObject]:
|
|
386
|
-
position = Position.at_page(self.page_number)
|
|
387
|
-
position.text_pattern = pattern
|
|
388
|
-
# noinspection PyProtectedMember
|
|
389
|
-
return self.root._to_textline_objects(self.root._find_text_lines(position))
|
|
390
|
-
|
|
391
|
-
def select_paragraphs_at(
|
|
392
|
-
self, x: float, y: float, tolerance: float = DEFAULT_TOLERANCE
|
|
393
|
-
) -> List[ParagraphObject]:
|
|
394
|
-
position = Position.at_page_coordinates(self.page_number, x, y)
|
|
395
|
-
# noinspection PyProtectedMember
|
|
396
|
-
return self.root._to_paragraph_objects(
|
|
397
|
-
self.root._find_paragraphs(position, tolerance)
|
|
398
|
-
)
|
|
399
|
-
|
|
400
|
-
def select_text_lines(self) -> List[TextLineObject]:
|
|
401
|
-
position = Position.at_page(self.page_number)
|
|
402
|
-
# noinspection PyProtectedMember
|
|
403
|
-
return self.root._to_textline_objects(self.root._find_text_lines(position))
|
|
404
|
-
|
|
405
|
-
def select_text_lines_starting_with(self, text: str) -> List[TextLineObject]:
|
|
406
|
-
position = Position.at_page(self.page_number)
|
|
407
|
-
position.with_text_starts(text)
|
|
408
|
-
# noinspection PyProtectedMember
|
|
409
|
-
return self.root._to_textline_objects(self.root._find_text_lines(position))
|
|
410
|
-
|
|
411
|
-
def select_text_lines_at(
|
|
412
|
-
self, x, y, tolerance: float = DEFAULT_TOLERANCE
|
|
413
|
-
) -> List[TextLineObject]:
|
|
414
|
-
position = Position.at_page_coordinates(self.page_number, x, y)
|
|
415
|
-
# noinspection PyProtectedMember
|
|
416
|
-
return self.root._to_textline_objects(
|
|
417
|
-
self.root._find_text_lines(position, tolerance)
|
|
418
|
-
)
|
|
419
|
-
|
|
420
485
|
def select_images(self) -> List[ImageObject]:
|
|
421
486
|
# noinspection PyProtectedMember
|
|
422
487
|
return self.root._to_image_objects(
|
|
@@ -466,92 +531,6 @@ class PageClient:
|
|
|
466
531
|
|
|
467
532
|
# Singular selection methods (convenience methods returning first match or None)
|
|
468
533
|
|
|
469
|
-
def select_paragraph_at(
|
|
470
|
-
self, x: float, y: float, tolerance: float = DEFAULT_TOLERANCE
|
|
471
|
-
) -> Optional[ParagraphObject]:
|
|
472
|
-
"""
|
|
473
|
-
Select the first paragraph at the specified coordinates.
|
|
474
|
-
|
|
475
|
-
Args:
|
|
476
|
-
x: X coordinate in points
|
|
477
|
-
y: Y coordinate in points
|
|
478
|
-
tolerance: Tolerance in points for spatial matching (default: DEFAULT_TOLERANCE)
|
|
479
|
-
|
|
480
|
-
Returns:
|
|
481
|
-
First ParagraphObject at the coordinates, or None if no match
|
|
482
|
-
"""
|
|
483
|
-
results = self.select_paragraphs_at(x, y, tolerance)
|
|
484
|
-
return results[0] if results else None
|
|
485
|
-
|
|
486
|
-
def select_paragraph_starting_with(self, text: str) -> Optional[ParagraphObject]:
|
|
487
|
-
"""
|
|
488
|
-
Select the first paragraph starting with the specified text.
|
|
489
|
-
|
|
490
|
-
Args:
|
|
491
|
-
text: Text to search for at the start of paragraphs
|
|
492
|
-
|
|
493
|
-
Returns:
|
|
494
|
-
First ParagraphObject starting with the text, or None if no match
|
|
495
|
-
"""
|
|
496
|
-
results = self.select_paragraphs_starting_with(text)
|
|
497
|
-
return results[0] if results else None
|
|
498
|
-
|
|
499
|
-
def select_paragraph_matching(self, pattern: str) -> Optional[ParagraphObject]:
|
|
500
|
-
"""
|
|
501
|
-
Select the first paragraph matching the specified regex pattern.
|
|
502
|
-
|
|
503
|
-
Args:
|
|
504
|
-
pattern: Regex pattern to match against paragraph text
|
|
505
|
-
|
|
506
|
-
Returns:
|
|
507
|
-
First ParagraphObject matching the pattern, or None if no match
|
|
508
|
-
"""
|
|
509
|
-
results = self.select_paragraphs_matching(pattern)
|
|
510
|
-
return results[0] if results else None
|
|
511
|
-
|
|
512
|
-
def select_text_line_at(
|
|
513
|
-
self, x: float, y: float, tolerance: float = DEFAULT_TOLERANCE
|
|
514
|
-
) -> Optional[TextLineObject]:
|
|
515
|
-
"""
|
|
516
|
-
Select the first text line at the specified coordinates.
|
|
517
|
-
|
|
518
|
-
Args:
|
|
519
|
-
x: X coordinate in points
|
|
520
|
-
y: Y coordinate in points
|
|
521
|
-
tolerance: Tolerance in points for spatial matching (default: DEFAULT_TOLERANCE)
|
|
522
|
-
|
|
523
|
-
Returns:
|
|
524
|
-
First TextLineObject at the coordinates, or None if no match
|
|
525
|
-
"""
|
|
526
|
-
results = self.select_text_lines_at(x, y, tolerance)
|
|
527
|
-
return results[0] if results else None
|
|
528
|
-
|
|
529
|
-
def select_text_line_starting_with(self, text: str) -> Optional[TextLineObject]:
|
|
530
|
-
"""
|
|
531
|
-
Select the first text line starting with the specified text.
|
|
532
|
-
|
|
533
|
-
Args:
|
|
534
|
-
text: Text to search for at the start of text lines
|
|
535
|
-
|
|
536
|
-
Returns:
|
|
537
|
-
First TextLineObject starting with the text, or None if no match
|
|
538
|
-
"""
|
|
539
|
-
results = self.select_text_lines_starting_with(text)
|
|
540
|
-
return results[0] if results else None
|
|
541
|
-
|
|
542
|
-
def select_text_line_matching(self, pattern: str) -> Optional[TextLineObject]:
|
|
543
|
-
"""
|
|
544
|
-
Select the first text line matching the specified regex pattern.
|
|
545
|
-
|
|
546
|
-
Args:
|
|
547
|
-
pattern: Regex pattern to match against text line text
|
|
548
|
-
|
|
549
|
-
Returns:
|
|
550
|
-
First TextLineObject matching the pattern, or None if no match
|
|
551
|
-
"""
|
|
552
|
-
results = self.select_text_lines_matching(pattern)
|
|
553
|
-
return results[0] if results else None
|
|
554
|
-
|
|
555
534
|
def select_image_at(
|
|
556
535
|
self, x: float, y: float, tolerance: float = DEFAULT_TOLERANCE
|
|
557
536
|
) -> Optional[ImageObject]:
|
|
@@ -569,6 +548,10 @@ class PageClient:
|
|
|
569
548
|
results = self.select_images_at(x, y, tolerance)
|
|
570
549
|
return results[0] if results else None
|
|
571
550
|
|
|
551
|
+
def select_image(self) -> Optional[ImageObject]:
|
|
552
|
+
results = self.select_images()
|
|
553
|
+
return results[0] if results else None
|
|
554
|
+
|
|
572
555
|
def select_form_at(
|
|
573
556
|
self, x: float, y: float, tolerance: float = DEFAULT_TOLERANCE
|
|
574
557
|
) -> Optional[FormObject]:
|
|
@@ -586,6 +569,10 @@ class PageClient:
|
|
|
586
569
|
results = self.select_forms_at(x, y, tolerance)
|
|
587
570
|
return results[0] if results else None
|
|
588
571
|
|
|
572
|
+
def select_form(self) -> Optional[FormObject]:
|
|
573
|
+
results = self.select_forms()
|
|
574
|
+
return results[0] if results else None
|
|
575
|
+
|
|
589
576
|
def select_form_field_at(
|
|
590
577
|
self, x: float, y: float, tolerance: float = DEFAULT_TOLERANCE
|
|
591
578
|
) -> Optional[FormFieldObject]:
|
|
@@ -633,10 +620,15 @@ class PageClient:
|
|
|
633
620
|
results = self.select_paths_at(x, y, tolerance)
|
|
634
621
|
return results[0] if results else None
|
|
635
622
|
|
|
623
|
+
def select_path(self) -> Optional[PathObject]:
|
|
624
|
+
results = self.select_paths()
|
|
625
|
+
return results[0] if results else None
|
|
626
|
+
|
|
636
627
|
@classmethod
|
|
637
628
|
def from_ref(cls, root: "PDFDancer", page_ref: PageRef) -> "PageClient":
|
|
629
|
+
page_number = cast(int, page_ref.position.page_number)
|
|
638
630
|
page_client = PageClient(
|
|
639
|
-
page_number=
|
|
631
|
+
page_number=page_number,
|
|
640
632
|
root=root,
|
|
641
633
|
page_size=page_ref.page_size,
|
|
642
634
|
orientation=page_ref.orientation,
|
|
@@ -644,7 +636,7 @@ class PageClient:
|
|
|
644
636
|
page_client.internal_id = page_ref.internal_id
|
|
645
637
|
if page_ref.position is not None:
|
|
646
638
|
page_client.position = page_ref.position
|
|
647
|
-
page_client.page_number = page_ref.position.page_number
|
|
639
|
+
page_client.page_number = cast(int, page_ref.position.page_number)
|
|
648
640
|
return page_client
|
|
649
641
|
|
|
650
642
|
def delete(self) -> bool:
|
|
@@ -653,9 +645,9 @@ class PageClient:
|
|
|
653
645
|
|
|
654
646
|
def move_to(self, target_page_number: int) -> bool:
|
|
655
647
|
"""Move this page to a different index within the document."""
|
|
656
|
-
if target_page_number is None or target_page_number <
|
|
648
|
+
if target_page_number is None or target_page_number < 1:
|
|
657
649
|
raise ValidationException(
|
|
658
|
-
f"Target page
|
|
650
|
+
f"Target page number must be >= 1, got {target_page_number}"
|
|
659
651
|
)
|
|
660
652
|
|
|
661
653
|
# noinspection PyProtectedMember
|
|
@@ -665,54 +657,11 @@ class PageClient:
|
|
|
665
657
|
self.position = Position.at_page(target_page_number)
|
|
666
658
|
return moved
|
|
667
659
|
|
|
668
|
-
def
|
|
669
|
-
self,
|
|
670
|
-
replacements: Dict[str, Union[str, dict]],
|
|
671
|
-
reflow_preset: Optional[ReflowPreset] = None,
|
|
672
|
-
) -> bool:
|
|
673
|
-
"""
|
|
674
|
-
Replace template placeholders on this page.
|
|
675
|
-
|
|
676
|
-
Finds exact text matches for placeholders and replaces them with specified
|
|
677
|
-
content. All placeholders must be found or the operation fails atomically.
|
|
678
|
-
|
|
679
|
-
Args:
|
|
680
|
-
replacements: Dict mapping placeholder strings to replacement values.
|
|
681
|
-
- Simple: {"{{NAME}}": "John Doe"}
|
|
682
|
-
- With options: {"{{NAME}}": {"text": "John", "font": Font(...), "color": Color(...)}}
|
|
683
|
-
- With image: {"{{LOGO}}": {"image": Path("logo.png")}}
|
|
684
|
-
- With image and size: {"{{LOGO}}": {"image": Path("logo.png"), "width": 50, "height": 50}}
|
|
685
|
-
reflow_preset: Optional ReflowPreset to control text reflow behavior.
|
|
686
|
-
- BEST_EFFORT: Attempt to reflow, proceed even if imperfect
|
|
687
|
-
- FIT_OR_FAIL: Reflow must succeed or operation fails
|
|
688
|
-
- NONE: No reflow, replacement placed as-is
|
|
689
|
-
|
|
690
|
-
Returns:
|
|
691
|
-
True if all replacements were successful
|
|
692
|
-
|
|
693
|
-
Example:
|
|
694
|
-
```python
|
|
695
|
-
page.apply_replacements({
|
|
696
|
-
"{{NAME}}": "John Doe",
|
|
697
|
-
})
|
|
698
|
-
```
|
|
699
|
-
"""
|
|
700
|
-
replacement_list = _dict_to_replacements(replacements)
|
|
701
|
-
# noinspection PyProtectedMember
|
|
702
|
-
return self.root._apply_replacements(
|
|
703
|
-
replacements=replacement_list,
|
|
704
|
-
page_number=self.page_number,
|
|
705
|
-
reflow_preset=reflow_preset,
|
|
706
|
-
)
|
|
707
|
-
|
|
708
|
-
def _ref(self):
|
|
660
|
+
def _ref(self) -> ObjectRef:
|
|
709
661
|
return ObjectRef(
|
|
710
662
|
internal_id=self.internal_id, position=self.position, type=self.object_type
|
|
711
663
|
)
|
|
712
664
|
|
|
713
|
-
def new_paragraph(self) -> ParagraphBuilder:
|
|
714
|
-
return ParagraphPageBuilder(self.root, self.page_number)
|
|
715
|
-
|
|
716
665
|
def new_path(self) -> PathBuilder:
|
|
717
666
|
return PathBuilder(self.root, self.page_number)
|
|
718
667
|
|
|
@@ -730,13 +679,17 @@ class PageClient:
|
|
|
730
679
|
|
|
731
680
|
return RectangleBuilder(self.root, self.page_number)
|
|
732
681
|
|
|
733
|
-
def
|
|
682
|
+
def text(self) -> "PageTextClient":
|
|
683
|
+
"""Return selector-based text operations scoped to this one-based page."""
|
|
684
|
+
return PageTextClient(self.root, self.page_number)
|
|
685
|
+
|
|
686
|
+
def select_paths(self) -> List[PathObject]:
|
|
734
687
|
# noinspection PyProtectedMember
|
|
735
688
|
return self.root._to_path_objects(
|
|
736
689
|
self.root._find_paths(Position.at_page(self.page_number))
|
|
737
690
|
)
|
|
738
691
|
|
|
739
|
-
def group_paths(self, path_ids):
|
|
692
|
+
def group_paths(self, path_ids: List[str]) -> "PathGroupObject":
|
|
740
693
|
"""Group paths by their IDs. Returns a PathGroupObject."""
|
|
741
694
|
from .types import PathGroupObject
|
|
742
695
|
|
|
@@ -744,7 +697,7 @@ class PageClient:
|
|
|
744
697
|
info = self.root._create_path_group(page_index, path_ids=path_ids)
|
|
745
698
|
return PathGroupObject(self.root, page_index, info)
|
|
746
699
|
|
|
747
|
-
def group_paths_in_region(self, region):
|
|
700
|
+
def group_paths_in_region(self, region: "GroupBoundingRect") -> "PathGroupObject":
|
|
748
701
|
"""Group paths within a bounding region. Returns a PathGroupObject."""
|
|
749
702
|
from .types import PathGroupObject
|
|
750
703
|
|
|
@@ -752,39 +705,72 @@ class PageClient:
|
|
|
752
705
|
info = self.root._create_path_group(page_index, region=region)
|
|
753
706
|
return PathGroupObject(self.root, page_index, info)
|
|
754
707
|
|
|
755
|
-
def get_path_groups(self):
|
|
708
|
+
def get_path_groups(self) -> List["PathGroupObject"]:
|
|
756
709
|
"""List all path groups on this page."""
|
|
710
|
+
return self.root._list_path_groups(self.page_number)
|
|
757
711
|
|
|
758
|
-
|
|
759
|
-
return self.root._list_path_groups(page_index)
|
|
760
|
-
|
|
761
|
-
def select_elements(self):
|
|
712
|
+
def select_elements(self, types: Optional[str] = None) -> List[ObjectRef]:
|
|
762
713
|
"""
|
|
763
|
-
Select all
|
|
714
|
+
Select all live object-reference elements on this page.
|
|
764
715
|
|
|
765
716
|
Returns:
|
|
766
717
|
List of all PDF objects on this page
|
|
767
718
|
"""
|
|
768
|
-
|
|
769
|
-
|
|
770
|
-
|
|
771
|
-
|
|
772
|
-
result.extend(self.select_paths())
|
|
773
|
-
result.extend(self.select_forms())
|
|
774
|
-
result.extend(self.select_form_fields())
|
|
775
|
-
return result
|
|
719
|
+
return self.get_snapshot(types).elements
|
|
720
|
+
|
|
721
|
+
def get_snapshot(self, types: Optional[str] = None) -> PageSnapshot:
|
|
722
|
+
return self.root.get_page_snapshot(self.page_number, types)
|
|
776
723
|
|
|
777
724
|
@property
|
|
778
|
-
def size(self):
|
|
725
|
+
def size(self) -> Optional[PageSize]:
|
|
779
726
|
"""Property alias for page size."""
|
|
780
727
|
return self.page_size
|
|
781
728
|
|
|
782
729
|
@property
|
|
783
|
-
def page_orientation(self):
|
|
730
|
+
def page_orientation(self) -> Optional[Union[Orientation, str]]:
|
|
784
731
|
"""Property alias for orientation."""
|
|
785
732
|
return self.orientation
|
|
786
733
|
|
|
787
734
|
|
|
735
|
+
class TextClient:
|
|
736
|
+
"""Document-scoped selector-based text editing operations."""
|
|
737
|
+
|
|
738
|
+
def __init__(self, root: "PDFDancer") -> None:
|
|
739
|
+
self._root = root
|
|
740
|
+
|
|
741
|
+
def replace(self, request: TextReplaceRequest) -> TextEditResponse:
|
|
742
|
+
return self._root._edit_text("replace", request)
|
|
743
|
+
|
|
744
|
+
def delete(self, request: TextDeleteRequest) -> TextEditResponse:
|
|
745
|
+
return self._root._edit_text("delete", request)
|
|
746
|
+
|
|
747
|
+
def insert(self, request: TextInsertRequest) -> TextEditResponse:
|
|
748
|
+
return self._root._edit_text("insert", request)
|
|
749
|
+
|
|
750
|
+
def style(self, request: TextStyleRequest) -> TextEditResponse:
|
|
751
|
+
return self._root._edit_text("style", request)
|
|
752
|
+
|
|
753
|
+
|
|
754
|
+
class PageTextClient(TextClient):
|
|
755
|
+
"""Selector-based text editing operations scoped to one page."""
|
|
756
|
+
|
|
757
|
+
def __init__(self, root: "PDFDancer", page_number: int) -> None:
|
|
758
|
+
super().__init__(root)
|
|
759
|
+
self._page_number = page_number
|
|
760
|
+
|
|
761
|
+
def replace(self, request: TextReplaceRequest) -> TextEditResponse:
|
|
762
|
+
return super().replace(request.with_pages((self._page_number,)))
|
|
763
|
+
|
|
764
|
+
def delete(self, request: TextDeleteRequest) -> TextEditResponse:
|
|
765
|
+
return super().delete(request.with_pages((self._page_number,)))
|
|
766
|
+
|
|
767
|
+
def insert(self, request: TextInsertRequest) -> TextEditResponse:
|
|
768
|
+
return super().insert(request.with_pages((self._page_number,)))
|
|
769
|
+
|
|
770
|
+
def style(self, request: TextStyleRequest) -> TextEditResponse:
|
|
771
|
+
return super().style(request.with_pages((self._page_number,)))
|
|
772
|
+
|
|
773
|
+
|
|
788
774
|
class PDFDancer:
|
|
789
775
|
"""
|
|
790
776
|
REST API client for interacting with the PDFDancer PDF manipulation service.
|
|
@@ -793,6 +779,8 @@ class PDFDancer:
|
|
|
793
779
|
Handles authentication, session lifecycle, and HTTP communication transparently.
|
|
794
780
|
"""
|
|
795
781
|
|
|
782
|
+
_pdf_bytes: Optional[bytes]
|
|
783
|
+
|
|
796
784
|
# --------------------------------------------------------------
|
|
797
785
|
# CLASS METHOD ENTRY POINT
|
|
798
786
|
# --------------------------------------------------------------
|
|
@@ -803,7 +791,7 @@ class PDFDancer:
|
|
|
803
791
|
token: Optional[str] = None,
|
|
804
792
|
base_url: Optional[str] = None,
|
|
805
793
|
timeout: float = 30.0,
|
|
806
|
-
|
|
794
|
+
max_attempts: int = DEFAULT_MAX_ATTEMPTS,
|
|
807
795
|
retry_backoff_factor: float = DEFAULT_RETRY_BACKOFF_FACTOR,
|
|
808
796
|
) -> "PDFDancer":
|
|
809
797
|
"""
|
|
@@ -822,15 +810,16 @@ class PDFDancer:
|
|
|
822
810
|
base_url: Override for the API base URL; falls back to `PDFDANCER_BASE_URL`
|
|
823
811
|
or defaults to `https://api.pdfdancer.com`.
|
|
824
812
|
timeout: HTTP read timeout in seconds.
|
|
825
|
-
|
|
826
|
-
|
|
827
|
-
retry_backoff_factor: Base multiplier for exponential backoff delays (default:
|
|
828
|
-
Delay calculation:
|
|
829
|
-
Examples:
|
|
813
|
+
max_attempts: Maximum number of total attempts (default: 3).
|
|
814
|
+
The initial request counts as one attempt.
|
|
815
|
+
retry_backoff_factor: Base multiplier for exponential backoff delays (default: 2.0).
|
|
816
|
+
Delay calculation: initial_delay * (retry_backoff_factor ** attempt_number).
|
|
817
|
+
Examples: 2.0 → delays of 1s, 2s, 4s; 3.0 → delays of 1s, 3s, 9s.
|
|
830
818
|
|
|
831
819
|
Returns:
|
|
832
820
|
A ready-to-use `PDFDancer` client instance.
|
|
833
821
|
"""
|
|
822
|
+
_validate_max_attempts(max_attempts)
|
|
834
823
|
resolved_token = cls._resolve_token(token)
|
|
835
824
|
resolved_base_url = cls._resolve_base_url(base_url)
|
|
836
825
|
|
|
@@ -843,13 +832,12 @@ class PDFDancer:
|
|
|
843
832
|
pdf_data,
|
|
844
833
|
resolved_base_url,
|
|
845
834
|
timeout,
|
|
846
|
-
|
|
835
|
+
max_attempts,
|
|
847
836
|
retry_backoff_factor,
|
|
848
837
|
)
|
|
849
838
|
|
|
850
839
|
@classmethod
|
|
851
|
-
def _resolve_base_url(cls, base_url: Optional[str]) ->
|
|
852
|
-
_load_env()
|
|
840
|
+
def _resolve_base_url(cls, base_url: Optional[str]) -> str:
|
|
853
841
|
env_base_url = os.getenv("PDFDANCER_BASE_URL")
|
|
854
842
|
resolved_base_url = base_url or (
|
|
855
843
|
env_base_url.strip() if env_base_url and env_base_url.strip() else None
|
|
@@ -861,7 +849,7 @@ class PDFDancer:
|
|
|
861
849
|
@classmethod
|
|
862
850
|
def _obtain_anonymous_token(cls, base_url: str, timeout: float = 30.0) -> str:
|
|
863
851
|
"""
|
|
864
|
-
Obtain an anonymous token from the /keys/anon endpoint.
|
|
852
|
+
Obtain an anonymous token from the /v2/keys/anon endpoint.
|
|
865
853
|
|
|
866
854
|
Args:
|
|
867
855
|
base_url: Base URL of the PDFDancer API server
|
|
@@ -876,101 +864,63 @@ class PDFDancer:
|
|
|
876
864
|
"""
|
|
877
865
|
# Create temporary client without authentication
|
|
878
866
|
temp_client = httpx.Client(http2=True, verify=not DISABLE_SSL_VERIFY)
|
|
879
|
-
|
|
880
|
-
retry_backoff_factor =
|
|
867
|
+
max_attempts = DEFAULT_MAX_ATTEMPTS
|
|
868
|
+
retry_backoff_factor = DEFAULT_RETRY_BACKOFF_FACTOR
|
|
881
869
|
|
|
882
870
|
try:
|
|
883
|
-
last_error: Optional[Exception] = None
|
|
884
|
-
attempt = 0
|
|
885
871
|
|
|
886
|
-
|
|
887
|
-
|
|
888
|
-
|
|
889
|
-
|
|
890
|
-
|
|
891
|
-
|
|
892
|
-
|
|
893
|
-
|
|
894
|
-
|
|
895
|
-
|
|
896
|
-
|
|
897
|
-
)
|
|
872
|
+
def request_token() -> httpx.Response:
|
|
873
|
+
headers = {
|
|
874
|
+
"X-Fingerprint": Fingerprint.generate(),
|
|
875
|
+
"X-PDFDancer-Client": CLIENT_HEADER_VALUE,
|
|
876
|
+
"X-API-VERSION": "2",
|
|
877
|
+
}
|
|
878
|
+
return temp_client.post(
|
|
879
|
+
cls._cleanup_url_path(base_url, "/keys/anon"),
|
|
880
|
+
headers=headers,
|
|
881
|
+
timeout=timeout if timeout > 0 else None,
|
|
882
|
+
)
|
|
898
883
|
|
|
899
|
-
|
|
900
|
-
|
|
884
|
+
response = _execute_request_with_retries(
|
|
885
|
+
request_callable=request_token,
|
|
886
|
+
operation="POST /keys/anon",
|
|
887
|
+
max_attempts=max_attempts,
|
|
888
|
+
retry_backoff_factor=retry_backoff_factor,
|
|
889
|
+
)
|
|
901
890
|
|
|
902
|
-
|
|
903
|
-
|
|
904
|
-
return token_data["token"]
|
|
905
|
-
else:
|
|
906
|
-
raise HttpClientException(
|
|
907
|
-
"Invalid anonymous token response format"
|
|
908
|
-
)
|
|
891
|
+
response.raise_for_status()
|
|
892
|
+
token_data = response.json()
|
|
909
893
|
|
|
910
|
-
|
|
911
|
-
|
|
912
|
-
|
|
913
|
-
|
|
914
|
-
|
|
915
|
-
|
|
916
|
-
|
|
917
|
-
|
|
918
|
-
|
|
919
|
-
|
|
920
|
-
|
|
921
|
-
|
|
922
|
-
f"Rate limit (429) on POST /keys/anon - retrying in {delay}s "
|
|
923
|
-
f"(attempt {attempt + 1}/{max_retries})",
|
|
924
|
-
file=sys.stderr,
|
|
925
|
-
)
|
|
926
|
-
if DEBUG:
|
|
927
|
-
print(
|
|
928
|
-
f"{time.time()}|POST /keys/anon - Rate limit exceeded (429), "
|
|
929
|
-
f"retrying in {delay}s (attempt {attempt + 1}/{max_retries})"
|
|
930
|
-
)
|
|
931
|
-
time.sleep(delay)
|
|
932
|
-
attempt += 1
|
|
933
|
-
continue
|
|
934
|
-
|
|
935
|
-
# Raise RateLimitException for 429 after exhausting retries
|
|
936
|
-
if e.response.status_code == 429:
|
|
937
|
-
retry_after = _get_retry_after_delay(e.response)
|
|
938
|
-
print(
|
|
939
|
-
"Rate limit (429) on POST /keys/anon - max retries exhausted",
|
|
940
|
-
file=sys.stderr,
|
|
941
|
-
)
|
|
942
|
-
raise RateLimitException(
|
|
943
|
-
"Rate limit exceeded when obtaining anonymous token",
|
|
944
|
-
retry_after=retry_after,
|
|
945
|
-
response=e.response,
|
|
946
|
-
) from None
|
|
947
|
-
|
|
948
|
-
# Other HTTP status errors
|
|
949
|
-
raise HttpClientException(
|
|
950
|
-
f"Failed to obtain anonymous token: HTTP {e.response.status_code}",
|
|
951
|
-
response=e.response,
|
|
952
|
-
cause=e,
|
|
953
|
-
) from None
|
|
954
|
-
except httpx.RequestError as e:
|
|
955
|
-
last_error = e
|
|
956
|
-
raise HttpClientException(
|
|
957
|
-
f"Failed to obtain anonymous token: {str(e)}",
|
|
958
|
-
response=None,
|
|
959
|
-
cause=e,
|
|
960
|
-
) from None
|
|
961
|
-
|
|
962
|
-
# Should not reach here, but handle just in case
|
|
963
|
-
if last_error:
|
|
964
|
-
raise HttpClientException(
|
|
965
|
-
f"Failed to obtain anonymous token after {max_retries + 1} attempts: {str(last_error)}",
|
|
966
|
-
response=None,
|
|
967
|
-
cause=last_error,
|
|
968
|
-
) from None
|
|
969
|
-
else:
|
|
970
|
-
raise HttpClientException(
|
|
971
|
-
f"Failed to obtain anonymous token after {max_retries + 1} attempts",
|
|
972
|
-
response=None,
|
|
894
|
+
# Extract token from response (matches Java AnonTokenResponse structure)
|
|
895
|
+
if isinstance(token_data, dict) and "token" in token_data:
|
|
896
|
+
return cast(str, token_data["token"])
|
|
897
|
+
|
|
898
|
+
raise HttpClientException("Invalid anonymous token response format")
|
|
899
|
+
|
|
900
|
+
except httpx.HTTPStatusError as e:
|
|
901
|
+
if e.response.status_code == 429:
|
|
902
|
+
retry_after = _get_retry_after_delay(e.response)
|
|
903
|
+
print(
|
|
904
|
+
"Rate limit (429) on POST /keys/anon - maximum attempts exhausted",
|
|
905
|
+
file=sys.stderr,
|
|
973
906
|
)
|
|
907
|
+
raise RateLimitException(
|
|
908
|
+
"Rate limit exceeded when obtaining anonymous token",
|
|
909
|
+
retry_after=retry_after,
|
|
910
|
+
response=e.response,
|
|
911
|
+
) from None
|
|
912
|
+
|
|
913
|
+
raise HttpClientException(
|
|
914
|
+
f"Failed to obtain anonymous token: HTTP {e.response.status_code}",
|
|
915
|
+
response=e.response,
|
|
916
|
+
cause=e,
|
|
917
|
+
) from None
|
|
918
|
+
except httpx.RequestError as e:
|
|
919
|
+
raise HttpClientException(
|
|
920
|
+
f"Failed to obtain anonymous token: {str(e)}",
|
|
921
|
+
response=None,
|
|
922
|
+
cause=e,
|
|
923
|
+
) from None
|
|
974
924
|
finally:
|
|
975
925
|
temp_client.close()
|
|
976
926
|
|
|
@@ -984,7 +934,6 @@ class PDFDancer:
|
|
|
984
934
|
1. PDFDANCER_API_TOKEN (preferred)
|
|
985
935
|
2. PDFDANCER_TOKEN (legacy)
|
|
986
936
|
"""
|
|
987
|
-
_load_env()
|
|
988
937
|
resolved_token = token.strip() if token and token.strip() else None
|
|
989
938
|
if resolved_token is None:
|
|
990
939
|
# Check PDFDANCER_API_TOKEN first (preferred), then PDFDANCER_TOKEN (legacy)
|
|
@@ -1003,7 +952,7 @@ class PDFDancer:
|
|
|
1003
952
|
page_size: Optional[Union[PageSize, str, Mapping[str, Any]]] = None,
|
|
1004
953
|
orientation: Optional[Union[Orientation, str]] = None,
|
|
1005
954
|
initial_page_count: int = 1,
|
|
1006
|
-
|
|
955
|
+
max_attempts: int = DEFAULT_MAX_ATTEMPTS,
|
|
1007
956
|
retry_backoff_factor: float = DEFAULT_RETRY_BACKOFF_FACTOR,
|
|
1008
957
|
) -> "PDFDancer":
|
|
1009
958
|
"""
|
|
@@ -1025,15 +974,16 @@ class PDFDancer:
|
|
|
1025
974
|
mapping with `width`/`height` values.
|
|
1026
975
|
orientation: Page orientation (default: PORTRAIT). Can be Orientation enum or string.
|
|
1027
976
|
initial_page_count: Number of initial blank pages (default: 1).
|
|
1028
|
-
|
|
1029
|
-
|
|
1030
|
-
retry_backoff_factor: Base multiplier for exponential backoff delays (default:
|
|
1031
|
-
Delay calculation:
|
|
1032
|
-
Examples:
|
|
977
|
+
max_attempts: Maximum number of total attempts (default: 3).
|
|
978
|
+
The initial request counts as one attempt.
|
|
979
|
+
retry_backoff_factor: Base multiplier for exponential backoff delays (default: 2.0).
|
|
980
|
+
Delay calculation: initial_delay * (retry_backoff_factor ** attempt_number).
|
|
981
|
+
Examples: 2.0 → delays of 1s, 2s, 4s; 3.0 → delays of 1s, 3s, 9s.
|
|
1033
982
|
|
|
1034
983
|
Returns:
|
|
1035
984
|
A ready-to-use `PDFDancer` client instance with a blank PDF.
|
|
1036
985
|
"""
|
|
986
|
+
_validate_max_attempts(max_attempts)
|
|
1037
987
|
resolved_token = cls._resolve_token(token)
|
|
1038
988
|
resolved_base_url = cls._resolve_base_url(base_url)
|
|
1039
989
|
|
|
@@ -1051,7 +1001,7 @@ class PDFDancer:
|
|
|
1051
1001
|
instance._token = resolved_token.strip()
|
|
1052
1002
|
instance._base_url = resolved_base_url.rstrip("/")
|
|
1053
1003
|
instance._read_timeout = timeout
|
|
1054
|
-
instance.
|
|
1004
|
+
instance._max_attempts = max_attempts
|
|
1055
1005
|
instance._retry_backoff_factor = retry_backoff_factor
|
|
1056
1006
|
|
|
1057
1007
|
# Create HTTP client for connection reuse with HTTP/2 support
|
|
@@ -1060,6 +1010,7 @@ class PDFDancer:
|
|
|
1060
1010
|
headers={
|
|
1061
1011
|
"Authorization": f"Bearer {instance._token}",
|
|
1062
1012
|
"X-PDFDancer-Client": CLIENT_HEADER_VALUE,
|
|
1013
|
+
"X-API-VERSION": "2",
|
|
1063
1014
|
},
|
|
1064
1015
|
verify=not DISABLE_SSL_VERIFY,
|
|
1065
1016
|
)
|
|
@@ -1086,7 +1037,7 @@ class PDFDancer:
|
|
|
1086
1037
|
pdf_data: Union[bytes, Path, str, BinaryIO],
|
|
1087
1038
|
base_url: str,
|
|
1088
1039
|
read_timeout: float = 0,
|
|
1089
|
-
|
|
1040
|
+
max_attempts: int = DEFAULT_MAX_ATTEMPTS,
|
|
1090
1041
|
retry_backoff_factor: float = DEFAULT_RETRY_BACKOFF_FACTOR,
|
|
1091
1042
|
):
|
|
1092
1043
|
"""
|
|
@@ -1099,11 +1050,11 @@ class PDFDancer:
|
|
|
1099
1050
|
pdf_data: PDF file data as bytes, Path, filename string, or file-like object
|
|
1100
1051
|
base_url: Base URL of the PDFDancer API server
|
|
1101
1052
|
read_timeout: Timeout in seconds for HTTP requests (default: 30.0)
|
|
1102
|
-
|
|
1103
|
-
|
|
1104
|
-
retry_backoff_factor: Base multiplier for exponential backoff delays (default:
|
|
1105
|
-
Delay calculation:
|
|
1106
|
-
Examples:
|
|
1053
|
+
max_attempts: Maximum number of total attempts (default: 3).
|
|
1054
|
+
The initial request counts as one attempt.
|
|
1055
|
+
retry_backoff_factor: Base multiplier for exponential backoff delays (default: 2.0).
|
|
1056
|
+
Delay calculation: initial_delay * (retry_backoff_factor ** attempt_number).
|
|
1057
|
+
Examples: 2.0 → delays of 1s, 2s, 4s; 3.0 → delays of 1s, 3s, 9s.
|
|
1107
1058
|
|
|
1108
1059
|
Raises:
|
|
1109
1060
|
ValidationException: If token is empty or PDF data is invalid
|
|
@@ -1113,15 +1064,16 @@ class PDFDancer:
|
|
|
1113
1064
|
# Strict validation like Java client
|
|
1114
1065
|
if not token or not token.strip():
|
|
1115
1066
|
raise ValidationException("Authentication token cannot be null or empty")
|
|
1067
|
+
_validate_max_attempts(max_attempts)
|
|
1116
1068
|
|
|
1117
1069
|
self._token = token.strip()
|
|
1118
1070
|
self._base_url = base_url.rstrip("/")
|
|
1119
1071
|
self._read_timeout = read_timeout
|
|
1120
|
-
self.
|
|
1072
|
+
self._max_attempts = max_attempts
|
|
1121
1073
|
self._retry_backoff_factor = retry_backoff_factor
|
|
1122
1074
|
|
|
1123
1075
|
# Process PDF data with validation
|
|
1124
|
-
self._pdf_bytes = self._process_pdf_data(pdf_data)
|
|
1076
|
+
self._pdf_bytes: Optional[bytes] = self._process_pdf_data(pdf_data)
|
|
1125
1077
|
|
|
1126
1078
|
# Create HTTP client for connection reuse with HTTP/2 support
|
|
1127
1079
|
self._client = httpx.Client(
|
|
@@ -1129,6 +1081,7 @@ class PDFDancer:
|
|
|
1129
1081
|
headers={
|
|
1130
1082
|
"Authorization": f"Bearer {self._token}",
|
|
1131
1083
|
"X-PDFDancer-Client": CLIENT_HEADER_VALUE,
|
|
1084
|
+
"X-API-VERSION": "2",
|
|
1132
1085
|
},
|
|
1133
1086
|
verify=not DISABLE_SSL_VERIFY,
|
|
1134
1087
|
)
|
|
@@ -1210,7 +1163,7 @@ class PDFDancer:
|
|
|
1210
1163
|
|
|
1211
1164
|
# Check for top-level message
|
|
1212
1165
|
if "message" in error_data:
|
|
1213
|
-
return error_data["message"]
|
|
1166
|
+
return cast(str, error_data["message"])
|
|
1214
1167
|
|
|
1215
1168
|
# Fallback to response content
|
|
1216
1169
|
return response.text or f"HTTP {response.status_code}"
|
|
@@ -1238,18 +1191,24 @@ class PDFDancer:
|
|
|
1238
1191
|
@staticmethod
|
|
1239
1192
|
def _cleanup_url_path(base_url: str, path: str) -> str:
|
|
1240
1193
|
"""
|
|
1241
|
-
Combine base_url and path, ensuring no double slashes.
|
|
1194
|
+
Combine base_url and API path, ensuring no double slashes.
|
|
1242
1195
|
|
|
1243
1196
|
Args:
|
|
1244
1197
|
base_url: Base URL (may or may not have trailing slash)
|
|
1245
1198
|
path: Path segment (may or may not have leading slash)
|
|
1246
1199
|
|
|
1247
1200
|
Returns:
|
|
1248
|
-
Combined URL with no double slashes
|
|
1201
|
+
Combined URL with the v2 API prefix and no double slashes
|
|
1249
1202
|
"""
|
|
1250
1203
|
base = base_url.rstrip("/")
|
|
1251
1204
|
path = path.lstrip("/")
|
|
1252
|
-
|
|
1205
|
+
|
|
1206
|
+
if base.endswith(API_PATH_PREFIX) or path.startswith(
|
|
1207
|
+
API_PATH_PREFIX.strip("/") + "/"
|
|
1208
|
+
):
|
|
1209
|
+
return f"{base}/{path}"
|
|
1210
|
+
|
|
1211
|
+
return f"{base}{API_PATH_PREFIX}/{path}"
|
|
1253
1212
|
|
|
1254
1213
|
def _create_session(self) -> str:
|
|
1255
1214
|
"""
|
|
@@ -1257,163 +1216,109 @@ class PDFDancer:
|
|
|
1257
1216
|
"""
|
|
1258
1217
|
import uuid
|
|
1259
1218
|
|
|
1260
|
-
|
|
1261
|
-
|
|
1219
|
+
# Build multipart body manually to avoid base64 encoding and enable compression
|
|
1220
|
+
# httpx by default may add Content-Transfer-Encoding: base64 which the server rejects
|
|
1221
|
+
boundary = uuid.uuid4().hex
|
|
1262
1222
|
|
|
1263
|
-
|
|
1264
|
-
|
|
1265
|
-
|
|
1266
|
-
|
|
1267
|
-
|
|
1268
|
-
|
|
1269
|
-
|
|
1270
|
-
|
|
1271
|
-
|
|
1272
|
-
|
|
1273
|
-
|
|
1223
|
+
# Build multipart body with binary (not base64) encoding
|
|
1224
|
+
body_parts = []
|
|
1225
|
+
body_parts.append(f"--{boundary}\r\n".encode("utf-8"))
|
|
1226
|
+
body_parts.append(
|
|
1227
|
+
b'Content-Disposition: form-data; name="pdf"; filename="document.pdf"\r\n'
|
|
1228
|
+
)
|
|
1229
|
+
body_parts.append(b"Content-Type: application/pdf\r\n")
|
|
1230
|
+
body_parts.append(b"\r\n") # End of headers, no Content-Transfer-Encoding
|
|
1231
|
+
body_parts.append(cast(bytes, self._pdf_bytes))
|
|
1232
|
+
body_parts.append(b"\r\n")
|
|
1233
|
+
body_parts.append(f"--{boundary}--\r\n".encode("utf-8"))
|
|
1234
|
+
|
|
1235
|
+
uncompressed_body = b"".join(body_parts)
|
|
1236
|
+
compressed_body = gzip.compress(uncompressed_body)
|
|
1237
|
+
|
|
1238
|
+
original_size = len(uncompressed_body)
|
|
1239
|
+
compressed_size = len(compressed_body)
|
|
1240
|
+
compression_ratio = (
|
|
1241
|
+
(1 - compressed_size / original_size) * 100 if original_size > 0 else 0
|
|
1242
|
+
)
|
|
1243
|
+
|
|
1244
|
+
def log_create_attempt(attempt: int) -> None:
|
|
1245
|
+
if DEBUG:
|
|
1246
|
+
retry_info = (
|
|
1247
|
+
f" (attempt {attempt + 1}/{self._max_attempts})"
|
|
1248
|
+
if attempt > 0
|
|
1249
|
+
else ""
|
|
1274
1250
|
)
|
|
1275
|
-
|
|
1276
|
-
|
|
1277
|
-
|
|
1278
|
-
|
|
1279
|
-
body_parts.append(self._pdf_bytes)
|
|
1280
|
-
body_parts.append(b"\r\n")
|
|
1281
|
-
body_parts.append(f"--{boundary}--\r\n".encode("utf-8"))
|
|
1282
|
-
|
|
1283
|
-
uncompressed_body = b"".join(body_parts)
|
|
1284
|
-
|
|
1285
|
-
# Compress entire request body using gzip
|
|
1286
|
-
compressed_body = gzip.compress(uncompressed_body)
|
|
1287
|
-
|
|
1288
|
-
original_size = len(uncompressed_body)
|
|
1289
|
-
compressed_size = len(compressed_body)
|
|
1290
|
-
compression_ratio = (
|
|
1291
|
-
(1 - compressed_size / original_size) * 100
|
|
1292
|
-
if original_size > 0
|
|
1293
|
-
else 0
|
|
1251
|
+
print(
|
|
1252
|
+
f"{time.time()}|POST /session/create{retry_info} - original size: {original_size} bytes, "
|
|
1253
|
+
f"compressed size: {compressed_size} bytes, "
|
|
1254
|
+
f"compression: {compression_ratio:.1f}%"
|
|
1294
1255
|
)
|
|
1295
1256
|
|
|
1296
|
-
|
|
1297
|
-
|
|
1298
|
-
|
|
1299
|
-
|
|
1300
|
-
|
|
1301
|
-
|
|
1302
|
-
print(
|
|
1303
|
-
f"{time.time()}|POST /session/create{retry_info} - original size: {original_size} bytes, "
|
|
1304
|
-
f"compressed size: {compressed_size} bytes, "
|
|
1305
|
-
f"compression: {compression_ratio:.1f}%"
|
|
1306
|
-
)
|
|
1307
|
-
|
|
1308
|
-
headers = {
|
|
1309
|
-
"X-Generated-At": _generate_timestamp(),
|
|
1310
|
-
"Content-Type": f"multipart/form-data; boundary={boundary}",
|
|
1311
|
-
"Content-Encoding": "gzip",
|
|
1312
|
-
}
|
|
1313
|
-
|
|
1314
|
-
response = self._client.post(
|
|
1315
|
-
self._cleanup_url_path(self._base_url, "/session/create"),
|
|
1316
|
-
content=compressed_body,
|
|
1317
|
-
headers=headers,
|
|
1318
|
-
timeout=self._read_timeout if self._read_timeout > 0 else None,
|
|
1319
|
-
)
|
|
1257
|
+
def request_session() -> httpx.Response:
|
|
1258
|
+
headers = {
|
|
1259
|
+
"X-Generated-At": _generate_timestamp(),
|
|
1260
|
+
"Content-Type": f"multipart/form-data; boundary={boundary}",
|
|
1261
|
+
"Content-Encoding": "gzip",
|
|
1262
|
+
}
|
|
1320
1263
|
|
|
1321
|
-
|
|
1322
|
-
|
|
1323
|
-
|
|
1324
|
-
|
|
1325
|
-
|
|
1264
|
+
return self._client.post(
|
|
1265
|
+
self._cleanup_url_path(self._base_url, "/session/create"),
|
|
1266
|
+
content=compressed_body,
|
|
1267
|
+
headers=headers,
|
|
1268
|
+
timeout=self._read_timeout if self._read_timeout > 0 else None,
|
|
1269
|
+
)
|
|
1326
1270
|
|
|
1327
|
-
|
|
1328
|
-
|
|
1329
|
-
|
|
1330
|
-
|
|
1271
|
+
try:
|
|
1272
|
+
response = _execute_request_with_retries(
|
|
1273
|
+
request_callable=request_session,
|
|
1274
|
+
operation="POST /session/create",
|
|
1275
|
+
max_attempts=self._max_attempts,
|
|
1276
|
+
retry_backoff_factor=self._retry_backoff_factor,
|
|
1277
|
+
pre_request_hook=log_create_attempt,
|
|
1278
|
+
)
|
|
1331
1279
|
|
|
1332
|
-
|
|
1333
|
-
|
|
1280
|
+
response_size = len(response.content)
|
|
1281
|
+
if DEBUG:
|
|
1282
|
+
print(
|
|
1283
|
+
f"{time.time()}|POST /session/create - response size: {response_size} bytes"
|
|
1284
|
+
)
|
|
1334
1285
|
|
|
1335
|
-
|
|
1286
|
+
_log_generated_at_header(response, "POST", "/session/create")
|
|
1287
|
+
self._handle_authentication_error(response)
|
|
1288
|
+
response.raise_for_status()
|
|
1289
|
+
session_id = response.text.strip()
|
|
1336
1290
|
|
|
1337
|
-
|
|
1338
|
-
|
|
1339
|
-
if e.response.status_code == 429 and attempt < self._max_retries:
|
|
1340
|
-
retry_after = _get_retry_after_delay(e.response)
|
|
1341
|
-
if retry_after is not None:
|
|
1342
|
-
delay = retry_after
|
|
1343
|
-
else:
|
|
1344
|
-
# Use exponential backoff if no Retry-After header
|
|
1345
|
-
delay = self._retry_backoff_factor * (2**attempt)
|
|
1291
|
+
if not session_id:
|
|
1292
|
+
raise SessionException("Server returned empty session ID")
|
|
1346
1293
|
|
|
1347
|
-
|
|
1348
|
-
print(
|
|
1349
|
-
f"Rate limit (429) on POST /session/create - retrying in {delay}s "
|
|
1350
|
-
f"(attempt {attempt + 1}/{self._max_retries})",
|
|
1351
|
-
file=sys.stderr,
|
|
1352
|
-
)
|
|
1353
|
-
if DEBUG:
|
|
1354
|
-
print(
|
|
1355
|
-
f"{time.time()}|POST /session/create - Rate limit exceeded (429), "
|
|
1356
|
-
f"retrying in {delay}s (attempt {attempt + 1}/{self._max_retries})"
|
|
1357
|
-
)
|
|
1358
|
-
time.sleep(delay)
|
|
1359
|
-
attempt += 1
|
|
1360
|
-
continue
|
|
1294
|
+
return session_id
|
|
1361
1295
|
|
|
1362
|
-
|
|
1363
|
-
|
|
1364
|
-
|
|
1296
|
+
except httpx.HTTPStatusError as e:
|
|
1297
|
+
self._handle_authentication_error(e.response)
|
|
1298
|
+
error_message = self._extract_error_message(e.response)
|
|
1365
1299
|
|
|
1366
|
-
|
|
1367
|
-
|
|
1368
|
-
|
|
1369
|
-
|
|
1370
|
-
|
|
1371
|
-
|
|
1372
|
-
|
|
1373
|
-
|
|
1374
|
-
|
|
1375
|
-
|
|
1376
|
-
response=e.response,
|
|
1377
|
-
) from None
|
|
1378
|
-
|
|
1379
|
-
raise HttpClientException(
|
|
1380
|
-
f"Failed to create session: {error_message}",
|
|
1300
|
+
# Raise RateLimitException for 429 after retry attempts are exhausted
|
|
1301
|
+
if e.response.status_code == 429:
|
|
1302
|
+
retry_after = _get_retry_after_delay(e.response)
|
|
1303
|
+
print(
|
|
1304
|
+
"Rate limit (429) on POST /session/create - maximum attempts exhausted",
|
|
1305
|
+
file=sys.stderr,
|
|
1306
|
+
)
|
|
1307
|
+
raise RateLimitException(
|
|
1308
|
+
f"Rate limit exceeded: {error_message}",
|
|
1309
|
+
retry_after=retry_after,
|
|
1381
1310
|
response=e.response,
|
|
1382
|
-
cause=e,
|
|
1383
1311
|
) from None
|
|
1384
|
-
except httpx.RequestError as e:
|
|
1385
|
-
last_error = e
|
|
1386
|
-
|
|
1387
|
-
# Check if this is a retryable error
|
|
1388
|
-
if _is_retryable_error(e) and attempt < self._max_retries:
|
|
1389
|
-
# Calculate exponential backoff delay
|
|
1390
|
-
delay = self._retry_backoff_factor * (2**attempt)
|
|
1391
|
-
if DEBUG:
|
|
1392
|
-
print(
|
|
1393
|
-
f"{time.time()}|POST /session/create - Retryable error: {str(e)}, "
|
|
1394
|
-
f"retrying in {delay}s (attempt {attempt + 1}/{self._max_retries})"
|
|
1395
|
-
)
|
|
1396
|
-
time.sleep(delay)
|
|
1397
|
-
attempt += 1
|
|
1398
|
-
continue
|
|
1399
|
-
else:
|
|
1400
|
-
# Non-retryable error or exhausted retries
|
|
1401
|
-
raise HttpClientException(
|
|
1402
|
-
f"Failed to create session: {str(e)}", response=None, cause=e
|
|
1403
|
-
) from None
|
|
1404
1312
|
|
|
1405
|
-
# Should not reach here, but handle just in case
|
|
1406
|
-
if last_error:
|
|
1407
1313
|
raise HttpClientException(
|
|
1408
|
-
f"Failed to create session
|
|
1409
|
-
response=
|
|
1410
|
-
cause=
|
|
1314
|
+
f"Failed to create session: {error_message}",
|
|
1315
|
+
response=e.response,
|
|
1316
|
+
cause=e,
|
|
1411
1317
|
) from None
|
|
1412
|
-
|
|
1318
|
+
except httpx.RequestError as e:
|
|
1413
1319
|
raise HttpClientException(
|
|
1414
|
-
f"Failed to create session
|
|
1415
|
-
|
|
1416
|
-
)
|
|
1320
|
+
f"Failed to create session: {str(e)}", response=None, cause=e
|
|
1321
|
+
) from None
|
|
1417
1322
|
|
|
1418
1323
|
def _create_blank_pdf_session(
|
|
1419
1324
|
self,
|
|
@@ -1438,7 +1343,7 @@ class PDFDancer:
|
|
|
1438
1343
|
HttpClientException: If HTTP communication fails
|
|
1439
1344
|
"""
|
|
1440
1345
|
# Build request payload (outside retry loop since validation should only happen once)
|
|
1441
|
-
request_data = {}
|
|
1346
|
+
request_data: dict[str, Any] = {}
|
|
1442
1347
|
|
|
1443
1348
|
# Handle page_size - convert to type-safe object with dimensions
|
|
1444
1349
|
if page_size is not None:
|
|
@@ -1465,286 +1370,192 @@ class PDFDancer:
|
|
|
1465
1370
|
raise ValidationException(
|
|
1466
1371
|
f"Initial page count must be at least 1, got {initial_page_count}"
|
|
1467
1372
|
)
|
|
1468
|
-
request_data["initialPageCount"] = initial_page_count
|
|
1469
|
-
|
|
1470
|
-
|
|
1471
|
-
attempt = 0
|
|
1472
|
-
|
|
1473
|
-
while attempt <= self._max_retries:
|
|
1474
|
-
try:
|
|
1475
|
-
request_body = json.dumps(request_data)
|
|
1476
|
-
request_size = len(request_body.encode("utf-8"))
|
|
1477
|
-
if DEBUG:
|
|
1478
|
-
retry_info = (
|
|
1479
|
-
f" (attempt {attempt + 1}/{self._max_retries + 1})"
|
|
1480
|
-
if attempt > 0
|
|
1481
|
-
else ""
|
|
1482
|
-
)
|
|
1483
|
-
print(
|
|
1484
|
-
f"{time.time()}|POST /session/new{retry_info} - request size: {request_size} bytes"
|
|
1485
|
-
)
|
|
1373
|
+
request_data["initialPageCount"] = int(initial_page_count)
|
|
1374
|
+
request_body = json.dumps(request_data)
|
|
1375
|
+
request_size = len(request_body.encode("utf-8"))
|
|
1486
1376
|
|
|
1487
|
-
|
|
1488
|
-
|
|
1489
|
-
|
|
1490
|
-
|
|
1491
|
-
|
|
1492
|
-
|
|
1493
|
-
|
|
1494
|
-
|
|
1495
|
-
|
|
1377
|
+
def log_blank_pdf_attempt(attempt: int) -> None:
|
|
1378
|
+
if DEBUG:
|
|
1379
|
+
retry_info = (
|
|
1380
|
+
f" (attempt {attempt + 1}/{self._max_attempts})"
|
|
1381
|
+
if attempt > 0
|
|
1382
|
+
else ""
|
|
1383
|
+
)
|
|
1384
|
+
print(
|
|
1385
|
+
f"{time.time()}|POST /session/new{retry_info} - request size: {request_size} bytes"
|
|
1496
1386
|
)
|
|
1497
1387
|
|
|
1498
|
-
|
|
1499
|
-
|
|
1500
|
-
|
|
1501
|
-
|
|
1502
|
-
|
|
1388
|
+
def request_blank_pdf() -> httpx.Response:
|
|
1389
|
+
headers = {
|
|
1390
|
+
"Content-Type": "application/json",
|
|
1391
|
+
"X-Generated-At": _generate_timestamp(),
|
|
1392
|
+
}
|
|
1393
|
+
return self._client.post(
|
|
1394
|
+
self._cleanup_url_path(self._base_url, "/session/new"),
|
|
1395
|
+
json=request_data,
|
|
1396
|
+
headers=headers,
|
|
1397
|
+
timeout=self._read_timeout if self._read_timeout > 0 else None,
|
|
1398
|
+
)
|
|
1503
1399
|
|
|
1504
|
-
|
|
1505
|
-
|
|
1506
|
-
|
|
1507
|
-
|
|
1400
|
+
try:
|
|
1401
|
+
response = _execute_request_with_retries(
|
|
1402
|
+
request_callable=request_blank_pdf,
|
|
1403
|
+
operation="POST /session/new",
|
|
1404
|
+
max_attempts=self._max_attempts,
|
|
1405
|
+
retry_backoff_factor=self._retry_backoff_factor,
|
|
1406
|
+
pre_request_hook=log_blank_pdf_attempt,
|
|
1407
|
+
)
|
|
1508
1408
|
|
|
1509
|
-
|
|
1510
|
-
|
|
1409
|
+
response_size = len(response.content)
|
|
1410
|
+
if DEBUG:
|
|
1411
|
+
print(
|
|
1412
|
+
f"{time.time()}|POST /session/new - response size: {response_size} bytes"
|
|
1413
|
+
)
|
|
1511
1414
|
|
|
1512
|
-
|
|
1415
|
+
_log_generated_at_header(response, "POST", "/session/new")
|
|
1416
|
+
self._handle_authentication_error(response)
|
|
1417
|
+
response.raise_for_status()
|
|
1418
|
+
session_id = response.text.strip()
|
|
1513
1419
|
|
|
1514
|
-
|
|
1515
|
-
|
|
1516
|
-
if e.response.status_code == 429 and attempt < self._max_retries:
|
|
1517
|
-
retry_after = _get_retry_after_delay(e.response)
|
|
1518
|
-
if retry_after is not None:
|
|
1519
|
-
delay = retry_after
|
|
1520
|
-
else:
|
|
1521
|
-
# Use exponential backoff if no Retry-After header
|
|
1522
|
-
delay = self._retry_backoff_factor * (2**attempt)
|
|
1420
|
+
if not session_id:
|
|
1421
|
+
raise SessionException("Server returned empty session ID")
|
|
1523
1422
|
|
|
1524
|
-
|
|
1525
|
-
print(
|
|
1526
|
-
f"Rate limit (429) on POST /session/new - retrying in {delay}s "
|
|
1527
|
-
f"(attempt {attempt + 1}/{self._max_retries})",
|
|
1528
|
-
file=sys.stderr,
|
|
1529
|
-
)
|
|
1530
|
-
if DEBUG:
|
|
1531
|
-
print(
|
|
1532
|
-
f"{time.time()}|POST /session/new - Rate limit exceeded (429), "
|
|
1533
|
-
f"retrying in {delay}s (attempt {attempt + 1}/{self._max_retries})"
|
|
1534
|
-
)
|
|
1535
|
-
time.sleep(delay)
|
|
1536
|
-
attempt += 1
|
|
1537
|
-
continue
|
|
1423
|
+
return session_id
|
|
1538
1424
|
|
|
1539
|
-
|
|
1540
|
-
|
|
1541
|
-
|
|
1425
|
+
except httpx.HTTPStatusError as e:
|
|
1426
|
+
self._handle_authentication_error(e.response)
|
|
1427
|
+
error_message = self._extract_error_message(e.response)
|
|
1542
1428
|
|
|
1543
|
-
|
|
1544
|
-
|
|
1545
|
-
|
|
1546
|
-
|
|
1547
|
-
|
|
1548
|
-
|
|
1549
|
-
|
|
1550
|
-
|
|
1551
|
-
|
|
1552
|
-
|
|
1553
|
-
response=e.response,
|
|
1554
|
-
) from None
|
|
1555
|
-
|
|
1556
|
-
raise HttpClientException(
|
|
1557
|
-
f"Failed to create blank PDF session: {error_message}",
|
|
1429
|
+
# Raise RateLimitException for 429 after retry attempts are exhausted
|
|
1430
|
+
if e.response.status_code == 429:
|
|
1431
|
+
retry_after = _get_retry_after_delay(e.response)
|
|
1432
|
+
print(
|
|
1433
|
+
"Rate limit (429) on POST /session/new - maximum attempts exhausted",
|
|
1434
|
+
file=sys.stderr,
|
|
1435
|
+
)
|
|
1436
|
+
raise RateLimitException(
|
|
1437
|
+
f"Rate limit exceeded: {error_message}",
|
|
1438
|
+
retry_after=retry_after,
|
|
1558
1439
|
response=e.response,
|
|
1559
|
-
cause=e,
|
|
1560
1440
|
) from None
|
|
1561
|
-
|
|
1562
|
-
last_error = e
|
|
1563
|
-
|
|
1564
|
-
# Check if this is a retryable error
|
|
1565
|
-
if _is_retryable_error(e) and attempt < self._max_retries:
|
|
1566
|
-
# Calculate exponential backoff delay
|
|
1567
|
-
delay = self._retry_backoff_factor * (2**attempt)
|
|
1568
|
-
if DEBUG:
|
|
1569
|
-
print(
|
|
1570
|
-
f"{time.time()}|POST /session/new - Retryable error: {str(e)}, "
|
|
1571
|
-
f"retrying in {delay}s (attempt {attempt + 1}/{self._max_retries})"
|
|
1572
|
-
)
|
|
1573
|
-
time.sleep(delay)
|
|
1574
|
-
attempt += 1
|
|
1575
|
-
continue
|
|
1576
|
-
else:
|
|
1577
|
-
# Non-retryable error or exhausted retries
|
|
1578
|
-
raise HttpClientException(
|
|
1579
|
-
f"Failed to create blank PDF session: {str(e)}",
|
|
1580
|
-
response=None,
|
|
1581
|
-
cause=e,
|
|
1582
|
-
) from None
|
|
1583
|
-
|
|
1584
|
-
# Should not reach here, but handle just in case
|
|
1585
|
-
if last_error:
|
|
1441
|
+
|
|
1586
1442
|
raise HttpClientException(
|
|
1587
|
-
f"Failed to create blank PDF session
|
|
1588
|
-
response=
|
|
1589
|
-
cause=
|
|
1443
|
+
f"Failed to create blank PDF session: {error_message}",
|
|
1444
|
+
response=e.response,
|
|
1445
|
+
cause=e,
|
|
1590
1446
|
) from None
|
|
1591
|
-
|
|
1447
|
+
except httpx.RequestError as e:
|
|
1592
1448
|
raise HttpClientException(
|
|
1593
|
-
f"Failed to create blank PDF session
|
|
1449
|
+
f"Failed to create blank PDF session: {str(e)}",
|
|
1594
1450
|
response=None,
|
|
1595
|
-
|
|
1451
|
+
cause=e,
|
|
1452
|
+
) from None
|
|
1596
1453
|
|
|
1597
1454
|
def _make_request(
|
|
1598
1455
|
self,
|
|
1599
1456
|
method: str,
|
|
1600
1457
|
path: str,
|
|
1601
|
-
data: Optional[dict] = None,
|
|
1602
|
-
params: Optional[dict] = None,
|
|
1458
|
+
data: Optional[dict[str, Any]] = None,
|
|
1459
|
+
params: Optional[dict[str, Any]] = None,
|
|
1603
1460
|
) -> httpx.Response:
|
|
1604
1461
|
"""
|
|
1605
1462
|
Make HTTP request with session headers, error handling, and automatic retry for transient errors.
|
|
1606
1463
|
"""
|
|
1607
1464
|
headers = {
|
|
1608
1465
|
"X-Session-Id": self._session_id,
|
|
1466
|
+
"X-API-VERSION": "2",
|
|
1609
1467
|
"Content-Type": "application/json",
|
|
1610
1468
|
"X-Generated-At": _generate_timestamp(),
|
|
1611
1469
|
"X-Fingerprint": Fingerprint.generate(),
|
|
1612
|
-
"X-API-VERSION": "1",
|
|
1613
1470
|
}
|
|
1614
1471
|
|
|
1615
|
-
|
|
1616
|
-
|
|
1472
|
+
request_body = json.dumps(data) if data is not None else None
|
|
1473
|
+
request_size = (
|
|
1474
|
+
len(request_body.encode("utf-8")) if request_body is not None else 0
|
|
1475
|
+
)
|
|
1617
1476
|
|
|
1618
|
-
|
|
1619
|
-
|
|
1620
|
-
|
|
1621
|
-
|
|
1622
|
-
|
|
1623
|
-
|
|
1624
|
-
|
|
1625
|
-
|
|
1626
|
-
|
|
1627
|
-
|
|
1628
|
-
else ""
|
|
1629
|
-
)
|
|
1630
|
-
print(
|
|
1631
|
-
f"{time.time()}|{method} {path}{retry_info} - request size: {request_size} bytes"
|
|
1632
|
-
)
|
|
1477
|
+
def log_attempt(attempt: int) -> None:
|
|
1478
|
+
if DEBUG:
|
|
1479
|
+
retry_info = (
|
|
1480
|
+
f" (attempt {attempt + 1}/{self._max_attempts})"
|
|
1481
|
+
if attempt > 0
|
|
1482
|
+
else ""
|
|
1483
|
+
)
|
|
1484
|
+
print(
|
|
1485
|
+
f"{time.time()}|{method} {path}{retry_info} - request size: {request_size} bytes"
|
|
1486
|
+
)
|
|
1633
1487
|
|
|
1634
|
-
|
|
1635
|
-
|
|
1636
|
-
|
|
1637
|
-
|
|
1638
|
-
|
|
1639
|
-
|
|
1640
|
-
|
|
1488
|
+
def request_api() -> httpx.Response:
|
|
1489
|
+
return self._client.request(
|
|
1490
|
+
method=method,
|
|
1491
|
+
url=self._cleanup_url_path(self._base_url, path),
|
|
1492
|
+
json=data,
|
|
1493
|
+
params=params,
|
|
1494
|
+
headers=headers,
|
|
1495
|
+
timeout=self._read_timeout if self._read_timeout > 0 else None,
|
|
1496
|
+
)
|
|
1497
|
+
|
|
1498
|
+
try:
|
|
1499
|
+
response = _execute_request_with_retries(
|
|
1500
|
+
request_callable=request_api,
|
|
1501
|
+
operation=f"{method} {path}",
|
|
1502
|
+
max_attempts=self._max_attempts,
|
|
1503
|
+
retry_backoff_factor=self._retry_backoff_factor,
|
|
1504
|
+
pre_request_hook=log_attempt,
|
|
1505
|
+
)
|
|
1506
|
+
|
|
1507
|
+
response_size = len(response.content)
|
|
1508
|
+
if DEBUG:
|
|
1509
|
+
print(
|
|
1510
|
+
f"{time.time()}|{method} {path} - response size: {response_size} bytes"
|
|
1641
1511
|
)
|
|
1642
1512
|
|
|
1643
|
-
|
|
1644
|
-
if DEBUG:
|
|
1645
|
-
print(
|
|
1646
|
-
f"{time.time()}|{method} {path} - response size: {response_size} bytes"
|
|
1647
|
-
)
|
|
1513
|
+
_log_generated_at_header(response, method, path)
|
|
1648
1514
|
|
|
1649
|
-
|
|
1650
|
-
|
|
1651
|
-
|
|
1652
|
-
|
|
1653
|
-
|
|
1654
|
-
|
|
1655
|
-
|
|
1656
|
-
raise FontNotFoundException(
|
|
1657
|
-
error_data.get("message", "Font not found")
|
|
1658
|
-
)
|
|
1659
|
-
if error_data.get("error") == "SessionNotFoundException":
|
|
1660
|
-
raise SessionNotFoundException(
|
|
1661
|
-
error_data.get("message", "Session not found")
|
|
1662
|
-
)
|
|
1663
|
-
except (json.JSONDecodeError, KeyError):
|
|
1664
|
-
pass
|
|
1665
|
-
|
|
1666
|
-
self._handle_authentication_error(response)
|
|
1667
|
-
response.raise_for_status()
|
|
1668
|
-
return response
|
|
1669
|
-
|
|
1670
|
-
except httpx.HTTPStatusError as e:
|
|
1671
|
-
# Handle 429 (rate limit) with retry
|
|
1672
|
-
if e.response.status_code == 429 and attempt < self._max_retries:
|
|
1673
|
-
retry_after = _get_retry_after_delay(e.response)
|
|
1674
|
-
if retry_after is not None:
|
|
1675
|
-
delay = retry_after
|
|
1676
|
-
else:
|
|
1677
|
-
# Use exponential backoff if no Retry-After header
|
|
1678
|
-
delay = self._retry_backoff_factor * (2**attempt)
|
|
1679
|
-
|
|
1680
|
-
# Always log 429 to stderr for visibility
|
|
1681
|
-
print(
|
|
1682
|
-
f"Rate limit (429) on {method} {path} - retrying in {delay}s "
|
|
1683
|
-
f"(attempt {attempt + 1}/{self._max_retries})",
|
|
1684
|
-
file=sys.stderr,
|
|
1685
|
-
)
|
|
1686
|
-
if DEBUG:
|
|
1687
|
-
print(
|
|
1688
|
-
f"{time.time()}|{method} {path} - Rate limit exceeded (429), "
|
|
1689
|
-
f"retrying in {delay}s (attempt {attempt + 1}/{self._max_retries})"
|
|
1515
|
+
# Handle 404 errors
|
|
1516
|
+
if response.status_code == 404:
|
|
1517
|
+
try:
|
|
1518
|
+
error_data = response.json()
|
|
1519
|
+
if error_data.get("error") == "FontNotFoundException":
|
|
1520
|
+
raise FontNotFoundException(
|
|
1521
|
+
error_data.get("message", "Font not found")
|
|
1690
1522
|
)
|
|
1691
|
-
|
|
1692
|
-
|
|
1693
|
-
|
|
1523
|
+
if error_data.get("error") == "SessionNotFoundException":
|
|
1524
|
+
raise SessionNotFoundException(
|
|
1525
|
+
error_data.get("message", "Session not found")
|
|
1526
|
+
)
|
|
1527
|
+
except (json.JSONDecodeError, KeyError):
|
|
1528
|
+
pass
|
|
1529
|
+
|
|
1530
|
+
self._handle_authentication_error(response)
|
|
1531
|
+
response.raise_for_status()
|
|
1532
|
+
return response
|
|
1694
1533
|
|
|
1695
|
-
|
|
1696
|
-
|
|
1697
|
-
|
|
1534
|
+
except httpx.HTTPStatusError as e:
|
|
1535
|
+
# Other HTTP status errors are not retried (these are application-level errors)
|
|
1536
|
+
self._handle_authentication_error(e.response)
|
|
1537
|
+
error_message = self._extract_error_message(e.response)
|
|
1698
1538
|
|
|
1699
|
-
|
|
1700
|
-
|
|
1701
|
-
|
|
1702
|
-
|
|
1703
|
-
|
|
1704
|
-
|
|
1705
|
-
|
|
1706
|
-
|
|
1707
|
-
|
|
1708
|
-
|
|
1709
|
-
|
|
1710
|
-
) from None
|
|
1711
|
-
|
|
1712
|
-
raise HttpClientException(
|
|
1713
|
-
f"API request failed: {error_message}", response=e.response, cause=e
|
|
1539
|
+
# Raise RateLimitException for 429 after retry attempts are exhausted
|
|
1540
|
+
if e.response.status_code == 429:
|
|
1541
|
+
retry_after = _get_retry_after_delay(e.response)
|
|
1542
|
+
print(
|
|
1543
|
+
f"Rate limit (429) on {method} {path} - maximum attempts exhausted",
|
|
1544
|
+
file=sys.stderr,
|
|
1545
|
+
)
|
|
1546
|
+
raise RateLimitException(
|
|
1547
|
+
f"Rate limit exceeded: {error_message}",
|
|
1548
|
+
retry_after=retry_after,
|
|
1549
|
+
response=e.response,
|
|
1714
1550
|
) from None
|
|
1715
|
-
except httpx.RequestError as e:
|
|
1716
|
-
last_error = e
|
|
1717
|
-
|
|
1718
|
-
# Check if this is a retryable error
|
|
1719
|
-
if _is_retryable_error(e) and attempt < self._max_retries:
|
|
1720
|
-
# Calculate exponential backoff delay
|
|
1721
|
-
delay = self._retry_backoff_factor * (2**attempt)
|
|
1722
|
-
if DEBUG:
|
|
1723
|
-
print(
|
|
1724
|
-
f"{time.time()}|{method} {path} - Retryable error: {str(e)}, "
|
|
1725
|
-
f"retrying in {delay}s (attempt {attempt + 1}/{self._max_retries})"
|
|
1726
|
-
)
|
|
1727
|
-
time.sleep(delay)
|
|
1728
|
-
attempt += 1
|
|
1729
|
-
continue
|
|
1730
|
-
else:
|
|
1731
|
-
# Non-retryable error or exhausted retries
|
|
1732
|
-
raise HttpClientException(
|
|
1733
|
-
f"API request failed: {str(e)}", response=None, cause=e
|
|
1734
|
-
) from None
|
|
1735
1551
|
|
|
1736
|
-
# Should not reach here, but handle just in case
|
|
1737
|
-
if last_error:
|
|
1738
1552
|
raise HttpClientException(
|
|
1739
|
-
f"API request failed
|
|
1740
|
-
response=None,
|
|
1741
|
-
cause=last_error,
|
|
1553
|
+
f"API request failed: {error_message}", response=e.response, cause=e
|
|
1742
1554
|
) from None
|
|
1743
|
-
|
|
1555
|
+
except httpx.RequestError as e:
|
|
1744
1556
|
raise HttpClientException(
|
|
1745
|
-
f"API request failed
|
|
1746
|
-
|
|
1747
|
-
)
|
|
1557
|
+
f"API request failed: {str(e)}", response=None, cause=e
|
|
1558
|
+
) from None
|
|
1748
1559
|
|
|
1749
1560
|
def _find(
|
|
1750
1561
|
self,
|
|
@@ -1773,74 +1584,19 @@ class PDFDancer:
|
|
|
1773
1584
|
|
|
1774
1585
|
# Use snapshot for all other queries
|
|
1775
1586
|
if position and position.page_number is not None:
|
|
1776
|
-
|
|
1587
|
+
page_snapshot = self._get_or_fetch_page_snapshot(position.page_number)
|
|
1777
1588
|
return self._filter_snapshot_elements(
|
|
1778
|
-
|
|
1589
|
+
page_snapshot.elements, object_type, position, tolerance
|
|
1779
1590
|
)
|
|
1780
1591
|
else:
|
|
1781
|
-
|
|
1782
|
-
all_elements = []
|
|
1783
|
-
for page_snap in
|
|
1592
|
+
document_snapshot = self._get_or_fetch_document_snapshot()
|
|
1593
|
+
all_elements: List[ObjectRef] = []
|
|
1594
|
+
for page_snap in document_snapshot.pages:
|
|
1784
1595
|
all_elements.extend(page_snap.elements)
|
|
1785
1596
|
return self._filter_snapshot_elements(
|
|
1786
1597
|
all_elements, object_type, position, tolerance
|
|
1787
1598
|
)
|
|
1788
1599
|
|
|
1789
|
-
def select_paragraphs(self) -> List[ParagraphObject]:
|
|
1790
|
-
"""
|
|
1791
|
-
Searches for paragraph objects returning ParagraphObject instances.
|
|
1792
|
-
"""
|
|
1793
|
-
return self._to_paragraph_objects(self._find_paragraphs(None))
|
|
1794
|
-
|
|
1795
|
-
def select_paragraphs_matching(self, pattern: str) -> List[ParagraphObject]:
|
|
1796
|
-
"""
|
|
1797
|
-
Searches for paragraph objects matching a regex pattern.
|
|
1798
|
-
|
|
1799
|
-
Args:
|
|
1800
|
-
pattern: Regex pattern to match against paragraph text
|
|
1801
|
-
|
|
1802
|
-
Returns:
|
|
1803
|
-
List of ParagraphObject instances matching the pattern
|
|
1804
|
-
"""
|
|
1805
|
-
position = Position()
|
|
1806
|
-
position.text_pattern = pattern
|
|
1807
|
-
return self._to_paragraph_objects(self._find_paragraphs(position))
|
|
1808
|
-
|
|
1809
|
-
def select_paragraph_matching(self, pattern: str) -> Optional[ParagraphObject]:
|
|
1810
|
-
"""
|
|
1811
|
-
Select the first paragraph matching the specified regex pattern.
|
|
1812
|
-
|
|
1813
|
-
Args:
|
|
1814
|
-
pattern: Regex pattern to match against paragraph text
|
|
1815
|
-
|
|
1816
|
-
Returns:
|
|
1817
|
-
First ParagraphObject matching the pattern, or None if no match
|
|
1818
|
-
"""
|
|
1819
|
-
results = self.select_paragraphs_matching(pattern)
|
|
1820
|
-
return results[0] if results else None
|
|
1821
|
-
|
|
1822
|
-
def _find_paragraphs(
|
|
1823
|
-
self, position: Optional[Position] = None, tolerance: float = DEFAULT_TOLERANCE
|
|
1824
|
-
) -> List[TextObjectRef]:
|
|
1825
|
-
"""
|
|
1826
|
-
Searches for paragraph objects returning TextObjectRef with hierarchical structure.
|
|
1827
|
-
Uses snapshot cache for all queries.
|
|
1828
|
-
"""
|
|
1829
|
-
# Use snapshot for all queries (including spatial)
|
|
1830
|
-
if position and position.page_number is not None:
|
|
1831
|
-
snapshot = self._get_or_fetch_page_snapshot(position.page_number)
|
|
1832
|
-
return self._filter_snapshot_elements(
|
|
1833
|
-
snapshot.elements, ObjectType.PARAGRAPH, position, tolerance
|
|
1834
|
-
)
|
|
1835
|
-
else:
|
|
1836
|
-
snapshot = self._get_or_fetch_document_snapshot()
|
|
1837
|
-
all_elements = []
|
|
1838
|
-
for page_snap in snapshot.pages:
|
|
1839
|
-
all_elements.extend(page_snap.elements)
|
|
1840
|
-
return self._filter_snapshot_elements(
|
|
1841
|
-
all_elements, ObjectType.PARAGRAPH, position, tolerance
|
|
1842
|
-
)
|
|
1843
|
-
|
|
1844
1600
|
def _find_images(
|
|
1845
1601
|
self, position: Optional[Position] = None, tolerance: float = DEFAULT_TOLERANCE
|
|
1846
1602
|
) -> List[ObjectRef]:
|
|
@@ -1850,14 +1606,14 @@ class PDFDancer:
|
|
|
1850
1606
|
"""
|
|
1851
1607
|
# Use snapshot for all queries (including spatial)
|
|
1852
1608
|
if position and position.page_number is not None:
|
|
1853
|
-
|
|
1609
|
+
page_snapshot = self._get_or_fetch_page_snapshot(position.page_number)
|
|
1854
1610
|
return self._filter_snapshot_elements(
|
|
1855
|
-
|
|
1611
|
+
page_snapshot.elements, ObjectType.IMAGE, position, tolerance
|
|
1856
1612
|
)
|
|
1857
1613
|
else:
|
|
1858
|
-
|
|
1859
|
-
all_elements = []
|
|
1860
|
-
for page_snap in
|
|
1614
|
+
document_snapshot = self._get_or_fetch_document_snapshot()
|
|
1615
|
+
all_elements: List[ObjectRef] = []
|
|
1616
|
+
for page_snap in document_snapshot.pages:
|
|
1861
1617
|
all_elements.extend(page_snap.elements)
|
|
1862
1618
|
return self._filter_snapshot_elements(
|
|
1863
1619
|
all_elements, ObjectType.IMAGE, position, tolerance
|
|
@@ -1884,14 +1640,14 @@ class PDFDancer:
|
|
|
1884
1640
|
"""
|
|
1885
1641
|
# Use snapshot for all queries (including spatial)
|
|
1886
1642
|
if position and position.page_number is not None:
|
|
1887
|
-
|
|
1643
|
+
page_snapshot = self._get_or_fetch_page_snapshot(position.page_number)
|
|
1888
1644
|
return self._filter_snapshot_elements(
|
|
1889
|
-
|
|
1645
|
+
page_snapshot.elements, ObjectType.FORM_X_OBJECT, position, tolerance
|
|
1890
1646
|
)
|
|
1891
1647
|
else:
|
|
1892
|
-
|
|
1893
|
-
all_elements = []
|
|
1894
|
-
for page_snap in
|
|
1648
|
+
document_snapshot = self._get_or_fetch_document_snapshot()
|
|
1649
|
+
all_elements: List[ObjectRef] = []
|
|
1650
|
+
for page_snap in document_snapshot.pages:
|
|
1895
1651
|
all_elements.extend(page_snap.elements)
|
|
1896
1652
|
return self._filter_snapshot_elements(
|
|
1897
1653
|
all_elements, ObjectType.FORM_X_OBJECT, position, tolerance
|
|
@@ -1934,17 +1690,23 @@ class PDFDancer:
|
|
|
1934
1690
|
"""
|
|
1935
1691
|
# Use snapshot for all queries (including name and spatial)
|
|
1936
1692
|
if position and position.page_number is not None:
|
|
1937
|
-
|
|
1938
|
-
return
|
|
1939
|
-
|
|
1693
|
+
page_snapshot = self._get_or_fetch_page_snapshot(position.page_number)
|
|
1694
|
+
return cast(
|
|
1695
|
+
List[FormFieldRef],
|
|
1696
|
+
self._filter_snapshot_elements(
|
|
1697
|
+
page_snapshot.elements, ObjectType.FORM_FIELD, position, tolerance
|
|
1698
|
+
),
|
|
1940
1699
|
)
|
|
1941
1700
|
else:
|
|
1942
|
-
|
|
1943
|
-
all_elements = []
|
|
1944
|
-
for page_snap in
|
|
1701
|
+
document_snapshot = self._get_or_fetch_document_snapshot()
|
|
1702
|
+
all_elements: List[ObjectRef] = []
|
|
1703
|
+
for page_snap in document_snapshot.pages:
|
|
1945
1704
|
all_elements.extend(page_snap.elements)
|
|
1946
|
-
return
|
|
1947
|
-
|
|
1705
|
+
return cast(
|
|
1706
|
+
List[FormFieldRef],
|
|
1707
|
+
self._filter_snapshot_elements(
|
|
1708
|
+
all_elements, ObjectType.FORM_FIELD, position, tolerance
|
|
1709
|
+
),
|
|
1948
1710
|
)
|
|
1949
1711
|
|
|
1950
1712
|
def _change_form_field(self, form_field_ref: FormFieldRef, new_value: str) -> bool:
|
|
@@ -1959,7 +1721,7 @@ class PDFDancer:
|
|
|
1959
1721
|
response = self._make_request(
|
|
1960
1722
|
"PUT", "/pdf/modify/formField", data=request_data
|
|
1961
1723
|
)
|
|
1962
|
-
return response.json()
|
|
1724
|
+
return cast(bool, response.json())
|
|
1963
1725
|
finally:
|
|
1964
1726
|
self._invalidate_snapshots()
|
|
1965
1727
|
|
|
@@ -1984,75 +1746,20 @@ class PDFDancer:
|
|
|
1984
1746
|
|
|
1985
1747
|
# For simple page-level "all paths" queries, use snapshot
|
|
1986
1748
|
if position and position.page_number is not None:
|
|
1987
|
-
|
|
1749
|
+
page_snapshot = self._get_or_fetch_page_snapshot(position.page_number)
|
|
1988
1750
|
return self._filter_snapshot_elements(
|
|
1989
|
-
|
|
1751
|
+
page_snapshot.elements, ObjectType.PATH, position, tolerance
|
|
1990
1752
|
)
|
|
1991
1753
|
else:
|
|
1992
1754
|
# Document-level query - use document snapshot
|
|
1993
|
-
|
|
1994
|
-
all_elements = []
|
|
1995
|
-
for page_snap in
|
|
1755
|
+
document_snapshot = self._get_or_fetch_document_snapshot()
|
|
1756
|
+
all_elements: List[ObjectRef] = []
|
|
1757
|
+
for page_snap in document_snapshot.pages:
|
|
1996
1758
|
all_elements.extend(page_snap.elements)
|
|
1997
1759
|
return self._filter_snapshot_elements(
|
|
1998
1760
|
all_elements, ObjectType.PATH, position, tolerance
|
|
1999
1761
|
)
|
|
2000
1762
|
|
|
2001
|
-
def _find_text_lines(
|
|
2002
|
-
self, position: Optional[Position] = None, tolerance: float = DEFAULT_TOLERANCE
|
|
2003
|
-
) -> List[TextObjectRef]:
|
|
2004
|
-
"""
|
|
2005
|
-
Searches for text line objects returning TextObjectRef with hierarchical structure.
|
|
2006
|
-
Uses snapshot cache for all queries.
|
|
2007
|
-
"""
|
|
2008
|
-
# Use snapshot for all queries (including spatial)
|
|
2009
|
-
if position and position.page_number is not None:
|
|
2010
|
-
snapshot = self._get_or_fetch_page_snapshot(position.page_number)
|
|
2011
|
-
return self._filter_snapshot_elements(
|
|
2012
|
-
snapshot.elements, ObjectType.TEXT_LINE, position, tolerance
|
|
2013
|
-
)
|
|
2014
|
-
else:
|
|
2015
|
-
snapshot = self._get_or_fetch_document_snapshot()
|
|
2016
|
-
all_elements = []
|
|
2017
|
-
for page_snap in snapshot.pages:
|
|
2018
|
-
all_elements.extend(page_snap.elements)
|
|
2019
|
-
return self._filter_snapshot_elements(
|
|
2020
|
-
all_elements, ObjectType.TEXT_LINE, position, tolerance
|
|
2021
|
-
)
|
|
2022
|
-
|
|
2023
|
-
def select_text_lines(self) -> List[TextLineObject]:
|
|
2024
|
-
"""
|
|
2025
|
-
Searches for text line objects returning TextLineObject wrappers.
|
|
2026
|
-
"""
|
|
2027
|
-
return self._to_textline_objects(self._find_text_lines(None))
|
|
2028
|
-
|
|
2029
|
-
def select_text_lines_matching(self, pattern: str) -> List[TextLineObject]:
|
|
2030
|
-
"""
|
|
2031
|
-
Searches for text line objects matching a regex pattern.
|
|
2032
|
-
|
|
2033
|
-
Args:
|
|
2034
|
-
pattern: Regex pattern to match against text line text
|
|
2035
|
-
|
|
2036
|
-
Returns:
|
|
2037
|
-
List of TextLineObject instances matching the pattern
|
|
2038
|
-
"""
|
|
2039
|
-
position = Position()
|
|
2040
|
-
position.text_pattern = pattern
|
|
2041
|
-
return self._to_textline_objects(self._find_text_lines(position))
|
|
2042
|
-
|
|
2043
|
-
def select_text_line_matching(self, pattern: str) -> Optional[TextLineObject]:
|
|
2044
|
-
"""
|
|
2045
|
-
Select the first text line matching the specified regex pattern.
|
|
2046
|
-
|
|
2047
|
-
Args:
|
|
2048
|
-
pattern: Regex pattern to match against text line text
|
|
2049
|
-
|
|
2050
|
-
Returns:
|
|
2051
|
-
First TextLineObject matching the pattern, or None if no match
|
|
2052
|
-
"""
|
|
2053
|
-
results = self.select_text_lines_matching(pattern)
|
|
2054
|
-
return results[0] if results else None
|
|
2055
|
-
|
|
2056
1763
|
def page(self, page_number: int) -> PageClient:
|
|
2057
1764
|
"""
|
|
2058
1765
|
Get a specific page by page number, using snapshot cache when available.
|
|
@@ -2142,7 +1849,7 @@ class PDFDancer:
|
|
|
2142
1849
|
if result:
|
|
2143
1850
|
self._invalidate_snapshots()
|
|
2144
1851
|
|
|
2145
|
-
return result
|
|
1852
|
+
return cast(bool, result)
|
|
2146
1853
|
|
|
2147
1854
|
def move_page(self, from_page: int, to_page: int) -> bool:
|
|
2148
1855
|
"""
|
|
@@ -2189,6 +1896,39 @@ class PDFDancer:
|
|
|
2189
1896
|
|
|
2190
1897
|
# Manipulation Operations
|
|
2191
1898
|
|
|
1899
|
+
def _edit_text(
|
|
1900
|
+
self,
|
|
1901
|
+
operation: str,
|
|
1902
|
+
request: Union[
|
|
1903
|
+
TextReplaceRequest,
|
|
1904
|
+
TextDeleteRequest,
|
|
1905
|
+
TextInsertRequest,
|
|
1906
|
+
TextStyleRequest,
|
|
1907
|
+
],
|
|
1908
|
+
) -> TextEditResponse:
|
|
1909
|
+
"""Execute one validated selector-based text mutation."""
|
|
1910
|
+
expected_types = {
|
|
1911
|
+
"replace": TextReplaceRequest,
|
|
1912
|
+
"delete": TextDeleteRequest,
|
|
1913
|
+
"insert": TextInsertRequest,
|
|
1914
|
+
"style": TextStyleRequest,
|
|
1915
|
+
}
|
|
1916
|
+
expected_type = expected_types.get(operation)
|
|
1917
|
+
if expected_type is None:
|
|
1918
|
+
raise ValidationException(f"Unsupported text operation: {operation}")
|
|
1919
|
+
if not isinstance(request, expected_type):
|
|
1920
|
+
raise ValidationException(
|
|
1921
|
+
f"{operation} requires {expected_type.__name__}, "
|
|
1922
|
+
f"got {type(request).__name__}"
|
|
1923
|
+
)
|
|
1924
|
+
|
|
1925
|
+
response = self._make_request(
|
|
1926
|
+
"POST", f"/pdf/text/{operation}", data=request.to_dict()
|
|
1927
|
+
)
|
|
1928
|
+
result = TextEditResponse.from_dict(response.json())
|
|
1929
|
+
self._invalidate_snapshots()
|
|
1930
|
+
return result
|
|
1931
|
+
|
|
2192
1932
|
def _delete(self, object_ref: ObjectRef) -> bool:
|
|
2193
1933
|
"""
|
|
2194
1934
|
Deletes the specified PDF object from the document.
|
|
@@ -2210,7 +1950,7 @@ class PDFDancer:
|
|
|
2210
1950
|
if result:
|
|
2211
1951
|
self._invalidate_snapshots()
|
|
2212
1952
|
|
|
2213
|
-
return result
|
|
1953
|
+
return cast(bool, result)
|
|
2214
1954
|
|
|
2215
1955
|
def _move(self, object_ref: ObjectRef, position: Position) -> bool:
|
|
2216
1956
|
"""
|
|
@@ -2236,59 +1976,7 @@ class PDFDancer:
|
|
|
2236
1976
|
if result:
|
|
2237
1977
|
self._invalidate_snapshots()
|
|
2238
1978
|
|
|
2239
|
-
return result
|
|
2240
|
-
|
|
2241
|
-
def _redact(
|
|
2242
|
-
self,
|
|
2243
|
-
targets: List["RedactTarget"],
|
|
2244
|
-
default_replacement: str = "[REDACTED]",
|
|
2245
|
-
placeholder_color: Optional[Color] = None,
|
|
2246
|
-
) -> "RedactResponse":
|
|
2247
|
-
"""
|
|
2248
|
-
Redacts specified objects from the PDF document.
|
|
2249
|
-
|
|
2250
|
-
Args:
|
|
2251
|
-
targets: List of RedactTarget objects identifying what to redact
|
|
2252
|
-
default_replacement: Default replacement text for redacted content
|
|
2253
|
-
placeholder_color: Color for image/path placeholder rectangles
|
|
2254
|
-
|
|
2255
|
-
Returns:
|
|
2256
|
-
RedactResponse with count, success status, and any warnings
|
|
2257
|
-
"""
|
|
2258
|
-
if not targets:
|
|
2259
|
-
raise ValidationException("At least one redaction target is required")
|
|
2260
|
-
|
|
2261
|
-
if placeholder_color is None:
|
|
2262
|
-
placeholder_color = Color(0, 0, 0)
|
|
2263
|
-
|
|
2264
|
-
request = RedactRequest(targets, default_replacement, placeholder_color)
|
|
2265
|
-
response = self._make_request("POST", "/pdf/redact", data=request.to_dict())
|
|
2266
|
-
result = RedactResponse.from_dict(response.json())
|
|
2267
|
-
|
|
2268
|
-
if result.success:
|
|
2269
|
-
self._invalidate_snapshots()
|
|
2270
|
-
|
|
2271
|
-
return result
|
|
2272
|
-
|
|
2273
|
-
def redact(
|
|
2274
|
-
self,
|
|
2275
|
-
objects: List["PDFObjectBase"],
|
|
2276
|
-
replacement: str = "[REDACTED]",
|
|
2277
|
-
placeholder_color: Optional[Color] = None,
|
|
2278
|
-
) -> "RedactResponse":
|
|
2279
|
-
"""
|
|
2280
|
-
Redacts multiple objects from the PDF document.
|
|
2281
|
-
|
|
2282
|
-
Args:
|
|
2283
|
-
objects: List of PDF objects to redact
|
|
2284
|
-
replacement: Replacement text for all redacted content
|
|
2285
|
-
placeholder_color: Color for image/path placeholder rectangles
|
|
2286
|
-
|
|
2287
|
-
Returns:
|
|
2288
|
-
RedactResponse with count, success status, and any warnings
|
|
2289
|
-
"""
|
|
2290
|
-
targets = [RedactTarget(obj.internal_id, replacement) for obj in objects]
|
|
2291
|
-
return self._redact(targets, replacement, placeholder_color)
|
|
1979
|
+
return cast(bool, result)
|
|
2292
1980
|
|
|
2293
1981
|
def clear_clipping(self, object_ref: ObjectRef) -> bool:
|
|
2294
1982
|
"""
|
|
@@ -2321,88 +2009,6 @@ class PDFDancer:
|
|
|
2321
2009
|
|
|
2322
2010
|
return result
|
|
2323
2011
|
|
|
2324
|
-
# Template Replacement Operations
|
|
2325
|
-
|
|
2326
|
-
def _apply_replacements(
|
|
2327
|
-
self,
|
|
2328
|
-
replacements: List[TemplateReplacement],
|
|
2329
|
-
page_number: Optional[int] = None,
|
|
2330
|
-
reflow_preset: Optional[ReflowPreset] = None,
|
|
2331
|
-
) -> bool:
|
|
2332
|
-
"""
|
|
2333
|
-
Internal method to replace template placeholders in the PDF.
|
|
2334
|
-
|
|
2335
|
-
Args:
|
|
2336
|
-
replacements: List of TemplateReplacement objects
|
|
2337
|
-
page_number: Optional 1-based page number. If None, applies to all pages.
|
|
2338
|
-
reflow_preset: Optional reflow behavior preset
|
|
2339
|
-
|
|
2340
|
-
Returns:
|
|
2341
|
-
True if replacement was successful
|
|
2342
|
-
"""
|
|
2343
|
-
if not replacements:
|
|
2344
|
-
raise ValidationException("At least one replacement is required")
|
|
2345
|
-
|
|
2346
|
-
# Convert 1-based page_number to 0-based page_index for API
|
|
2347
|
-
page_index = None
|
|
2348
|
-
if page_number is not None:
|
|
2349
|
-
page_index = page_number - 1
|
|
2350
|
-
|
|
2351
|
-
request = TemplateReplaceRequest(
|
|
2352
|
-
replacements=replacements,
|
|
2353
|
-
page_index=page_index,
|
|
2354
|
-
reflow_preset=reflow_preset,
|
|
2355
|
-
)
|
|
2356
|
-
response = self._make_request(
|
|
2357
|
-
"POST", "/template/replace", data=request.to_dict()
|
|
2358
|
-
)
|
|
2359
|
-
result = response.json()
|
|
2360
|
-
|
|
2361
|
-
if result:
|
|
2362
|
-
self._invalidate_snapshots()
|
|
2363
|
-
|
|
2364
|
-
return result
|
|
2365
|
-
|
|
2366
|
-
def apply_replacements(
|
|
2367
|
-
self,
|
|
2368
|
-
replacements: Dict[str, Union[str, dict]],
|
|
2369
|
-
reflow_preset: Optional[ReflowPreset] = None,
|
|
2370
|
-
) -> bool:
|
|
2371
|
-
"""
|
|
2372
|
-
Replace template placeholders in the PDF document.
|
|
2373
|
-
|
|
2374
|
-
Finds exact text matches for placeholders and replaces them with specified
|
|
2375
|
-
content. All placeholders must be found or the operation fails atomically.
|
|
2376
|
-
|
|
2377
|
-
Args:
|
|
2378
|
-
replacements: Dict mapping placeholder strings to replacement values.
|
|
2379
|
-
- Simple: {"{{NAME}}": "John Doe"}
|
|
2380
|
-
- With options: {"{{NAME}}": {"text": "John", "font": Font(...), "color": Color(...)}}
|
|
2381
|
-
- With image: {"{{LOGO}}": {"image": Path("logo.png")}}
|
|
2382
|
-
- With image and size: {"{{LOGO}}": {"image": Path("logo.png"), "width": 50, "height": 50}}
|
|
2383
|
-
reflow_preset: Optional ReflowPreset to control text reflow behavior.
|
|
2384
|
-
- BEST_EFFORT: Attempt to reflow, proceed even if imperfect
|
|
2385
|
-
- FIT_OR_FAIL: Reflow must succeed or operation fails
|
|
2386
|
-
- NONE: No reflow, replacement placed as-is
|
|
2387
|
-
|
|
2388
|
-
Returns:
|
|
2389
|
-
True if all replacements were successful
|
|
2390
|
-
|
|
2391
|
-
Example:
|
|
2392
|
-
```python
|
|
2393
|
-
pdf.apply_replacements({
|
|
2394
|
-
"{{NAME}}": "John Doe",
|
|
2395
|
-
"{{DATE}}": "2025-01-15",
|
|
2396
|
-
})
|
|
2397
|
-
```
|
|
2398
|
-
"""
|
|
2399
|
-
replacement_list = _dict_to_replacements(replacements)
|
|
2400
|
-
return self._apply_replacements(
|
|
2401
|
-
replacements=replacement_list,
|
|
2402
|
-
page_number=None,
|
|
2403
|
-
reflow_preset=reflow_preset,
|
|
2404
|
-
)
|
|
2405
|
-
|
|
2406
2012
|
# Add Operations
|
|
2407
2013
|
|
|
2408
2014
|
def _add_image(self, image: Image, position: Optional[Position] = None) -> bool:
|
|
@@ -2427,46 +2033,27 @@ class PDFDancer:
|
|
|
2427
2033
|
|
|
2428
2034
|
return self._add_object(image)
|
|
2429
2035
|
|
|
2430
|
-
def
|
|
2431
|
-
"""
|
|
2432
|
-
Adds a paragraph to the PDF document.
|
|
2433
|
-
|
|
2434
|
-
Args:
|
|
2435
|
-
paragraph: The paragraph object to add
|
|
2436
|
-
|
|
2437
|
-
Returns:
|
|
2438
|
-
True if the paragraph was successfully added
|
|
2439
|
-
"""
|
|
2440
|
-
if paragraph is None:
|
|
2441
|
-
raise ValidationException("Paragraph cannot be null")
|
|
2442
|
-
if paragraph.get_position() is None:
|
|
2443
|
-
raise ValidationException("Paragraph position is null")
|
|
2444
|
-
if paragraph.get_position().page_number is None:
|
|
2445
|
-
raise ValidationException("Paragraph position page number is null")
|
|
2446
|
-
if paragraph.get_position().page_number < 1:
|
|
2447
|
-
raise ValidationException("Paragraph position page number is less than 1")
|
|
2448
|
-
|
|
2449
|
-
return self._add_object(paragraph)
|
|
2450
|
-
|
|
2451
|
-
def _add_path(self, path: "Path") -> bool:
|
|
2036
|
+
def _add_path(self, path: PDFPath) -> bool:
|
|
2452
2037
|
"""
|
|
2453
2038
|
Internal method to add a path to the document after validation.
|
|
2454
2039
|
"""
|
|
2455
2040
|
|
|
2456
2041
|
if path is None:
|
|
2457
2042
|
raise ValidationException("Path cannot be null")
|
|
2458
|
-
|
|
2043
|
+
position = path.get_position()
|
|
2044
|
+
if position is None:
|
|
2459
2045
|
raise ValidationException("Path position is null")
|
|
2460
|
-
if
|
|
2046
|
+
if position.page_number is None:
|
|
2461
2047
|
raise ValidationException("Path position page number is null")
|
|
2462
|
-
if
|
|
2048
|
+
if position.page_number < 1:
|
|
2463
2049
|
raise ValidationException("Path position page number is less than 1")
|
|
2464
|
-
|
|
2050
|
+
path_segments = path.get_path_segments()
|
|
2051
|
+
if not path_segments:
|
|
2465
2052
|
raise ValidationException("Path must have at least one segment")
|
|
2466
2053
|
|
|
2467
2054
|
return self._add_object(path)
|
|
2468
2055
|
|
|
2469
|
-
def _add_object(self, pdf_object) -> bool:
|
|
2056
|
+
def _add_object(self, pdf_object: Any) -> bool:
|
|
2470
2057
|
"""
|
|
2471
2058
|
Internal method to add any PDF object.
|
|
2472
2059
|
"""
|
|
@@ -2478,7 +2065,7 @@ class PDFDancer:
|
|
|
2478
2065
|
if result:
|
|
2479
2066
|
self._invalidate_snapshots()
|
|
2480
2067
|
|
|
2481
|
-
return result
|
|
2068
|
+
return cast(bool, result)
|
|
2482
2069
|
|
|
2483
2070
|
def _add_page(self, request: Optional[AddPageRequest]) -> PageRef:
|
|
2484
2071
|
"""
|
|
@@ -2523,14 +2110,17 @@ class PDFDancer:
|
|
|
2523
2110
|
return result
|
|
2524
2111
|
|
|
2525
2112
|
# Path Group Operations (internal, 0-based page_index)
|
|
2526
|
-
def _create_path_group(
|
|
2527
|
-
|
|
2528
|
-
|
|
2113
|
+
def _create_path_group(
|
|
2114
|
+
self,
|
|
2115
|
+
page_index: int,
|
|
2116
|
+
path_ids: Optional[List[str]] = None,
|
|
2117
|
+
region: Optional["GroupBoundingRect"] = None,
|
|
2118
|
+
) -> PathGroupInfo:
|
|
2529
2119
|
if path_ids is not None:
|
|
2530
2120
|
if not isinstance(path_ids, list) or len(path_ids) == 0:
|
|
2531
2121
|
raise ValidationException("path_ids must be a non-empty list")
|
|
2532
2122
|
|
|
2533
|
-
data = {"pageIndex": page_index}
|
|
2123
|
+
data: dict[str, Any] = {"pageIndex": page_index}
|
|
2534
2124
|
if path_ids is not None:
|
|
2535
2125
|
data["pathIds"] = path_ids
|
|
2536
2126
|
if region is not None:
|
|
@@ -2544,7 +2134,9 @@ class PDFDancer:
|
|
|
2544
2134
|
self._invalidate_snapshots()
|
|
2545
2135
|
return PathGroupInfo.from_dict(response.json())
|
|
2546
2136
|
|
|
2547
|
-
def _move_path_group(
|
|
2137
|
+
def _move_path_group(
|
|
2138
|
+
self, page_index: int, group_id: str, x: float, y: float
|
|
2139
|
+
) -> bool:
|
|
2548
2140
|
data = {
|
|
2549
2141
|
"pageIndex": page_index,
|
|
2550
2142
|
"groupId": group_id,
|
|
@@ -2553,10 +2145,16 @@ class PDFDancer:
|
|
|
2553
2145
|
}
|
|
2554
2146
|
response = self._make_request("PUT", "/pdf/path-group/move", data=data)
|
|
2555
2147
|
self._invalidate_snapshots()
|
|
2556
|
-
return response.json()
|
|
2148
|
+
return cast(bool, response.json())
|
|
2557
2149
|
|
|
2558
|
-
def _transform_path_group(
|
|
2559
|
-
|
|
2150
|
+
def _transform_path_group(
|
|
2151
|
+
self,
|
|
2152
|
+
page_index: int,
|
|
2153
|
+
group_id: str,
|
|
2154
|
+
transform_type: str,
|
|
2155
|
+
**kwargs: Any,
|
|
2156
|
+
) -> bool:
|
|
2157
|
+
data: dict[str, Any] = {
|
|
2560
2158
|
"pageIndex": page_index,
|
|
2561
2159
|
"groupId": group_id,
|
|
2562
2160
|
"transformType": transform_type,
|
|
@@ -2564,21 +2162,25 @@ class PDFDancer:
|
|
|
2564
2162
|
data.update({k: v for k, v in kwargs.items() if v is not None})
|
|
2565
2163
|
response = self._make_request("PUT", "/pdf/path-group/transform", data=data)
|
|
2566
2164
|
self._invalidate_snapshots()
|
|
2567
|
-
return response.json()
|
|
2165
|
+
return cast(bool, response.json())
|
|
2568
2166
|
|
|
2569
|
-
def _scale_path_group(self, page_index, group_id, factor):
|
|
2167
|
+
def _scale_path_group(self, page_index: int, group_id: str, factor: float) -> bool:
|
|
2570
2168
|
if factor <= 0:
|
|
2571
2169
|
raise ValidationException("Scale factor must be positive")
|
|
2572
2170
|
return self._transform_path_group(
|
|
2573
2171
|
page_index, group_id, "SCALE", scaleFactor=factor
|
|
2574
2172
|
)
|
|
2575
2173
|
|
|
2576
|
-
def _rotate_path_group(
|
|
2174
|
+
def _rotate_path_group(
|
|
2175
|
+
self, page_index: int, group_id: str, degrees: float
|
|
2176
|
+
) -> bool:
|
|
2577
2177
|
return self._transform_path_group(
|
|
2578
2178
|
page_index, group_id, "ROTATE", rotationAngle=degrees
|
|
2579
2179
|
)
|
|
2580
2180
|
|
|
2581
|
-
def _resize_path_group(
|
|
2181
|
+
def _resize_path_group(
|
|
2182
|
+
self, page_index: int, group_id: str, width: float, height: float
|
|
2183
|
+
) -> bool:
|
|
2582
2184
|
if width <= 0 or height <= 0:
|
|
2583
2185
|
raise ValidationException("Width and height must be positive")
|
|
2584
2186
|
return self._transform_path_group(
|
|
@@ -2589,18 +2191,18 @@ class PDFDancer:
|
|
|
2589
2191
|
targetHeight=height,
|
|
2590
2192
|
)
|
|
2591
2193
|
|
|
2592
|
-
def _remove_path_group(self, page_index, group_id):
|
|
2194
|
+
def _remove_path_group(self, page_index: int, group_id: str) -> bool:
|
|
2593
2195
|
data = {"pageIndex": page_index, "groupId": group_id}
|
|
2594
2196
|
response = self._make_request("DELETE", "/pdf/path-group/remove", data=data)
|
|
2595
2197
|
self._invalidate_snapshots()
|
|
2596
|
-
return response.json()
|
|
2198
|
+
return cast(bool, response.json())
|
|
2597
2199
|
|
|
2598
|
-
def _list_path_groups(self,
|
|
2599
|
-
from .models import PathGroupInfo
|
|
2200
|
+
def _list_path_groups(self, page_number: int) -> List["PathGroupObject"]:
|
|
2600
2201
|
from .types import PathGroupObject
|
|
2601
2202
|
|
|
2602
|
-
response = self._make_request("GET", f"/pdf/page/{
|
|
2203
|
+
response = self._make_request("GET", f"/pdf/page/{page_number}/path-groups")
|
|
2603
2204
|
infos = [PathGroupInfo.from_dict(d) for d in response.json()]
|
|
2205
|
+
page_index = page_number - 1
|
|
2604
2206
|
return [PathGroupObject(self, page_index, info) for info in infos]
|
|
2605
2207
|
|
|
2606
2208
|
def clear_path_group_clipping(self, page_number: int, group_id: str) -> bool:
|
|
@@ -2618,7 +2220,7 @@ class PDFDancer:
|
|
|
2618
2220
|
|
|
2619
2221
|
def _clear_path_group_clipping(self, page_number: int, group_id: str) -> bool:
|
|
2620
2222
|
"""
|
|
2621
|
-
Internal helper to clear clipping from a path group using the
|
|
2223
|
+
Internal helper to clear clipping from a path group using the V2 API.
|
|
2622
2224
|
"""
|
|
2623
2225
|
if page_number is None:
|
|
2624
2226
|
raise ValidationException("page_number cannot be null")
|
|
@@ -2645,11 +2247,14 @@ class PDFDancer:
|
|
|
2645
2247
|
|
|
2646
2248
|
return result
|
|
2647
2249
|
|
|
2648
|
-
def
|
|
2649
|
-
|
|
2250
|
+
def text(self) -> TextClient:
|
|
2251
|
+
"""Return document-scoped selector-based text editing operations."""
|
|
2252
|
+
return TextClient(self)
|
|
2650
2253
|
|
|
2651
2254
|
def new_page(
|
|
2652
|
-
self,
|
|
2255
|
+
self,
|
|
2256
|
+
orientation: Optional[Union[Orientation, str]] = Orientation.PORTRAIT,
|
|
2257
|
+
size: Optional[Union[PageSize, str, Mapping[str, Any]]] = PageSize.A4,
|
|
2653
2258
|
) -> PageBuilder:
|
|
2654
2259
|
builder = PageBuilder(self)
|
|
2655
2260
|
if orientation is not None:
|
|
@@ -2661,98 +2266,23 @@ class PDFDancer:
|
|
|
2661
2266
|
def new_image(self) -> ImageBuilder:
|
|
2662
2267
|
return ImageBuilder(self)
|
|
2663
2268
|
|
|
2664
|
-
def new_path(self) -> "PathBuilder":
|
|
2269
|
+
def new_path(self, page_number: int) -> "PathBuilder":
|
|
2665
2270
|
from .path_builder import PathBuilder
|
|
2666
2271
|
|
|
2667
|
-
return PathBuilder(self)
|
|
2668
|
-
|
|
2669
|
-
# Modify Operations
|
|
2670
|
-
def _modify_paragraph(
|
|
2671
|
-
self, object_ref: ObjectRef, new_paragraph: Union[Paragraph, str]
|
|
2672
|
-
) -> CommandResult:
|
|
2673
|
-
"""
|
|
2674
|
-
Modifies a paragraph object or its text content.
|
|
2675
|
-
|
|
2676
|
-
Args:
|
|
2677
|
-
object_ref: Reference to the paragraph to modify
|
|
2678
|
-
new_paragraph: New paragraph object or text string
|
|
2679
|
-
|
|
2680
|
-
Returns:
|
|
2681
|
-
True if the paragraph was successfully modified
|
|
2682
|
-
"""
|
|
2683
|
-
if object_ref is None:
|
|
2684
|
-
raise ValidationException("Object reference cannot be null")
|
|
2685
|
-
if new_paragraph is None:
|
|
2686
|
-
return CommandResult.empty("ModifyParagraph", object_ref.internal_id)
|
|
2687
|
-
|
|
2688
|
-
if isinstance(new_paragraph, str):
|
|
2689
|
-
# Text modification - returns CommandResult
|
|
2690
|
-
request_data = ModifyTextRequest(object_ref, new_paragraph).to_dict()
|
|
2691
|
-
response = self._make_request(
|
|
2692
|
-
"PUT", "/pdf/text/paragraph", data=request_data
|
|
2693
|
-
)
|
|
2694
|
-
result = CommandResult.from_dict(response.json())
|
|
2695
|
-
else:
|
|
2696
|
-
# Object modification
|
|
2697
|
-
request_data = ModifyRequest(object_ref, new_paragraph).to_dict()
|
|
2698
|
-
response = self._make_request("PUT", "/pdf/modify", data=request_data)
|
|
2699
|
-
result = CommandResult.from_dict(response.json())
|
|
2700
|
-
|
|
2701
|
-
# Invalidate snapshot caches after mutation
|
|
2702
|
-
self._invalidate_snapshots()
|
|
2703
|
-
return result
|
|
2704
|
-
|
|
2705
|
-
def _modify_text_line(self, object_ref: ObjectRef, new_text: str) -> CommandResult:
|
|
2706
|
-
"""
|
|
2707
|
-
Modifies a text line object.
|
|
2708
|
-
|
|
2709
|
-
Args:
|
|
2710
|
-
object_ref: Reference to the text line to modify
|
|
2711
|
-
new_text: New text content
|
|
2712
|
-
|
|
2713
|
-
Returns:
|
|
2714
|
-
True if the text line was successfully modified
|
|
2715
|
-
"""
|
|
2716
|
-
if object_ref is None:
|
|
2717
|
-
raise ValidationException("Object reference cannot be null")
|
|
2718
|
-
if new_text is None:
|
|
2719
|
-
raise ValidationException("New text cannot be null")
|
|
2720
|
-
|
|
2721
|
-
request_data = ModifyTextRequest(object_ref, new_text).to_dict()
|
|
2722
|
-
response = self._make_request("PUT", "/pdf/text/line", data=request_data)
|
|
2723
|
-
result = CommandResult.from_dict(response.json())
|
|
2272
|
+
return PathBuilder(self, page_number)
|
|
2724
2273
|
|
|
2725
|
-
|
|
2726
|
-
self
|
|
2727
|
-
return result
|
|
2728
|
-
|
|
2729
|
-
def _modify_text_line_full(
|
|
2730
|
-
self, object_ref: ObjectRef, new_text_line: TextLine
|
|
2731
|
-
) -> CommandResult:
|
|
2732
|
-
"""
|
|
2733
|
-
Modifies a text line object with full styling (font, color, position).
|
|
2734
|
-
|
|
2735
|
-
Args:
|
|
2736
|
-
object_ref: Reference to the text line to modify
|
|
2737
|
-
new_text_line: New text line object with styling
|
|
2274
|
+
def new_line(self, page_number: int) -> LineBuilder:
|
|
2275
|
+
return LineBuilder(self, page_number)
|
|
2738
2276
|
|
|
2739
|
-
|
|
2740
|
-
|
|
2741
|
-
"""
|
|
2742
|
-
if object_ref is None:
|
|
2743
|
-
raise ValidationException("Object reference cannot be null")
|
|
2744
|
-
if new_text_line is None:
|
|
2745
|
-
raise ValidationException("New text line cannot be null")
|
|
2277
|
+
def new_bezier(self, page_number: int) -> BezierBuilder:
|
|
2278
|
+
return BezierBuilder(self, page_number)
|
|
2746
2279
|
|
|
2747
|
-
|
|
2748
|
-
|
|
2749
|
-
response = self._make_request("PUT", "/pdf/modify", data=request_data)
|
|
2750
|
-
result = CommandResult.from_dict(response.json())
|
|
2280
|
+
def new_rectangle(self, page_number: int) -> "RectangleBuilder":
|
|
2281
|
+
from .path_builder import RectangleBuilder
|
|
2751
2282
|
|
|
2752
|
-
|
|
2753
|
-
self._invalidate_snapshots()
|
|
2754
|
-
return result
|
|
2283
|
+
return RectangleBuilder(self, page_number)
|
|
2755
2284
|
|
|
2285
|
+
# Modify Operations
|
|
2756
2286
|
def _modify_path(
|
|
2757
2287
|
self,
|
|
2758
2288
|
object_ref: ObjectRef,
|
|
@@ -2875,11 +2405,32 @@ class PDFDancer:
|
|
|
2875
2405
|
"X-Session-Id": self._session_id,
|
|
2876
2406
|
"X-Generated-At": _generate_timestamp(),
|
|
2877
2407
|
}
|
|
2878
|
-
|
|
2879
|
-
|
|
2880
|
-
|
|
2881
|
-
|
|
2882
|
-
|
|
2408
|
+
|
|
2409
|
+
def log_font_register_attempt(attempt: int) -> None:
|
|
2410
|
+
if DEBUG:
|
|
2411
|
+
retry_info = (
|
|
2412
|
+
f" (attempt {attempt + 1}/{self._max_attempts})"
|
|
2413
|
+
if attempt > 0
|
|
2414
|
+
else ""
|
|
2415
|
+
)
|
|
2416
|
+
print(
|
|
2417
|
+
f"{time.time()}|POST /font/register{retry_info} - request size: {request_size} bytes"
|
|
2418
|
+
)
|
|
2419
|
+
|
|
2420
|
+
def request_font_register() -> httpx.Response:
|
|
2421
|
+
return self._client.post(
|
|
2422
|
+
self._cleanup_url_path(self._base_url, "/font/register"),
|
|
2423
|
+
files=files,
|
|
2424
|
+
headers=headers,
|
|
2425
|
+
timeout=30,
|
|
2426
|
+
)
|
|
2427
|
+
|
|
2428
|
+
response = _execute_request_with_retries(
|
|
2429
|
+
request_callable=request_font_register,
|
|
2430
|
+
operation="POST /font/register",
|
|
2431
|
+
max_attempts=self._max_attempts,
|
|
2432
|
+
retry_backoff_factor=self._retry_backoff_factor,
|
|
2433
|
+
pre_request_hook=log_font_register_attempt,
|
|
2883
2434
|
)
|
|
2884
2435
|
|
|
2885
2436
|
response_size = len(response.content)
|
|
@@ -2915,7 +2466,7 @@ class PDFDancer:
|
|
|
2915
2466
|
Retrieve a snapshot of the entire document with all pages and elements.
|
|
2916
2467
|
|
|
2917
2468
|
Args:
|
|
2918
|
-
types: Optional comma-separated string of object types to filter (e.g., "
|
|
2469
|
+
types: Optional comma-separated string of object types to filter (e.g., "TEXT_LINE,IMAGE")
|
|
2919
2470
|
|
|
2920
2471
|
Returns:
|
|
2921
2472
|
DocumentSnapshot containing page count, fonts, and all page snapshots
|
|
@@ -2937,7 +2488,7 @@ class PDFDancer:
|
|
|
2937
2488
|
|
|
2938
2489
|
Args:
|
|
2939
2490
|
page_number: The page number to snapshot (1-based, page 1 is first page)
|
|
2940
|
-
types: Optional comma-separated string of object types to filter (e.g., "
|
|
2491
|
+
types: Optional comma-separated string of object types to filter (e.g., "TEXT_LINE,IMAGE")
|
|
2941
2492
|
|
|
2942
2493
|
Returns:
|
|
2943
2494
|
PageSnapshot containing page reference and all elements on that page
|
|
@@ -3010,11 +2561,11 @@ class PDFDancer:
|
|
|
3010
2561
|
|
|
3011
2562
|
def _filter_snapshot_elements(
|
|
3012
2563
|
self,
|
|
3013
|
-
elements: List,
|
|
3014
|
-
object_type: ObjectType,
|
|
2564
|
+
elements: List[ObjectRef],
|
|
2565
|
+
object_type: Optional[ObjectType],
|
|
3015
2566
|
position: Optional[Position] = None,
|
|
3016
2567
|
tolerance: float = DEFAULT_TOLERANCE,
|
|
3017
|
-
) -> List:
|
|
2568
|
+
) -> List[ObjectRef]:
|
|
3018
2569
|
"""
|
|
3019
2570
|
Filter snapshot elements client-side based on object type and position criteria.
|
|
3020
2571
|
|
|
@@ -3030,12 +2581,14 @@ class PDFDancer:
|
|
|
3030
2581
|
import re
|
|
3031
2582
|
|
|
3032
2583
|
# Filter by object type (handle form field subtypes)
|
|
3033
|
-
if object_type
|
|
3034
|
-
|
|
2584
|
+
if object_type is None:
|
|
2585
|
+
filtered = list(elements)
|
|
2586
|
+
elif object_type == ObjectType.FORM_FIELD:
|
|
2587
|
+
# Form fields include TEXT_FIELD, CHECKBOX, RADIO_BUTTON, BUTTON, DROPDOWN
|
|
3035
2588
|
form_field_types = {
|
|
3036
2589
|
ObjectType.FORM_FIELD,
|
|
3037
2590
|
ObjectType.TEXT_FIELD,
|
|
3038
|
-
ObjectType.
|
|
2591
|
+
ObjectType.CHECKBOX,
|
|
3039
2592
|
ObjectType.RADIO_BUTTON,
|
|
3040
2593
|
ObjectType.BUTTON,
|
|
3041
2594
|
ObjectType.DROPDOWN,
|
|
@@ -3094,7 +2647,11 @@ class PDFDancer:
|
|
|
3094
2647
|
return result
|
|
3095
2648
|
|
|
3096
2649
|
@staticmethod
|
|
3097
|
-
def _rects_intersect(
|
|
2650
|
+
def _rects_intersect(
|
|
2651
|
+
rect1: "ModelBoundingRect",
|
|
2652
|
+
rect2: "ModelBoundingRect",
|
|
2653
|
+
tolerance: float = DEFAULT_TOLERANCE,
|
|
2654
|
+
) -> bool:
|
|
3098
2655
|
"""
|
|
3099
2656
|
Check if two bounding rectangles intersect or are very close.
|
|
3100
2657
|
Handles point queries (width/height = 0) with tolerance.
|
|
@@ -3161,7 +2718,7 @@ class PDFDancer:
|
|
|
3161
2718
|
|
|
3162
2719
|
# Utility Methods
|
|
3163
2720
|
|
|
3164
|
-
def _parse_object_ref(self, obj_data: dict) -> ObjectRef:
|
|
2721
|
+
def _parse_object_ref(self, obj_data: dict[str, Any]) -> ObjectRef:
|
|
3165
2722
|
"""Parse JSON object data into ObjectRef instance."""
|
|
3166
2723
|
position_data = obj_data.get("position", {})
|
|
3167
2724
|
position = self._parse_position(position_data) if position_data else None
|
|
@@ -3169,12 +2726,12 @@ class PDFDancer:
|
|
|
3169
2726
|
object_type = ObjectType(obj_data["type"])
|
|
3170
2727
|
|
|
3171
2728
|
return ObjectRef(
|
|
3172
|
-
internal_id=obj_data
|
|
3173
|
-
position=position,
|
|
2729
|
+
internal_id=cast(str, obj_data.get("internalId")),
|
|
2730
|
+
position=cast(Position, position),
|
|
3174
2731
|
type=object_type,
|
|
3175
2732
|
)
|
|
3176
2733
|
|
|
3177
|
-
def _parse_form_field_ref(self, obj_data: dict) -> FormFieldRef:
|
|
2734
|
+
def _parse_form_field_ref(self, obj_data: dict[str, Any]) -> FormFieldRef:
|
|
3178
2735
|
"""Parse JSON object data into ObjectRef instance."""
|
|
3179
2736
|
position_data = obj_data.get("position", {})
|
|
3180
2737
|
position = self._parse_position(position_data) if position_data else None
|
|
@@ -3182,14 +2739,14 @@ class PDFDancer:
|
|
|
3182
2739
|
object_type = ObjectType(obj_data["type"])
|
|
3183
2740
|
|
|
3184
2741
|
return FormFieldRef(
|
|
3185
|
-
internal_id=obj_data
|
|
3186
|
-
position=position,
|
|
2742
|
+
internal_id=cast(str, obj_data.get("internalId")),
|
|
2743
|
+
position=cast(Position, position),
|
|
3187
2744
|
type=object_type,
|
|
3188
2745
|
name=obj_data["name"] if "name" in obj_data else None,
|
|
3189
2746
|
value=obj_data["value"] if "value" in obj_data else None,
|
|
3190
2747
|
)
|
|
3191
2748
|
|
|
3192
|
-
def _parse_path_object_ref(self, obj_data: dict) -> PathObjectRef:
|
|
2749
|
+
def _parse_path_object_ref(self, obj_data: dict[str, Any]) -> PathObjectRef:
|
|
3193
2750
|
"""Parse JSON object data into PathObjectRef instance with color information."""
|
|
3194
2751
|
position_data = obj_data.get("position", {})
|
|
3195
2752
|
position = self._parse_position(position_data) if position_data else None
|
|
@@ -3205,7 +2762,12 @@ class PDFDancer:
|
|
|
3205
2762
|
blue = stroke_color_data.get("blue")
|
|
3206
2763
|
alpha = stroke_color_data.get("alpha", 255)
|
|
3207
2764
|
if all(isinstance(v, int) for v in [red, green, blue]):
|
|
3208
|
-
stroke_color = Color(
|
|
2765
|
+
stroke_color = Color(
|
|
2766
|
+
cast(int, red),
|
|
2767
|
+
cast(int, green),
|
|
2768
|
+
cast(int, blue),
|
|
2769
|
+
cast(int, alpha),
|
|
2770
|
+
)
|
|
3209
2771
|
|
|
3210
2772
|
# Parse fill color if present
|
|
3211
2773
|
fill_color = None
|
|
@@ -3216,22 +2778,28 @@ class PDFDancer:
|
|
|
3216
2778
|
blue = fill_color_data.get("blue")
|
|
3217
2779
|
alpha = fill_color_data.get("alpha", 255)
|
|
3218
2780
|
if all(isinstance(v, int) for v in [red, green, blue]):
|
|
3219
|
-
fill_color = Color(
|
|
2781
|
+
fill_color = Color(
|
|
2782
|
+
cast(int, red),
|
|
2783
|
+
cast(int, green),
|
|
2784
|
+
cast(int, blue),
|
|
2785
|
+
cast(int, alpha),
|
|
2786
|
+
)
|
|
3220
2787
|
|
|
3221
2788
|
return PathObjectRef(
|
|
3222
|
-
internal_id=obj_data
|
|
3223
|
-
position=position,
|
|
2789
|
+
internal_id=cast(str, obj_data.get("internalId")),
|
|
2790
|
+
position=cast(Position, position),
|
|
3224
2791
|
object_type=object_type,
|
|
3225
2792
|
stroke_color=stroke_color,
|
|
3226
2793
|
fill_color=fill_color,
|
|
3227
2794
|
)
|
|
3228
2795
|
|
|
3229
2796
|
@staticmethod
|
|
3230
|
-
def _parse_position(pos_data: dict) -> Position:
|
|
2797
|
+
def _parse_position(pos_data: dict[str, Any]) -> Position:
|
|
3231
2798
|
"""Parse JSON position data into Position instance."""
|
|
3232
2799
|
position = Position()
|
|
3233
2800
|
position.page_number = pos_data.get("pageNumber")
|
|
3234
2801
|
position.text_starts_with = pos_data.get("textStartsWith")
|
|
2802
|
+
position.text_pattern = pos_data.get("textPattern")
|
|
3235
2803
|
|
|
3236
2804
|
if "shape" in pos_data:
|
|
3237
2805
|
position.shape = ShapeType(pos_data["shape"])
|
|
@@ -3252,7 +2820,7 @@ class PDFDancer:
|
|
|
3252
2820
|
return position
|
|
3253
2821
|
|
|
3254
2822
|
def _parse_text_object_ref(
|
|
3255
|
-
self, obj_data: dict, fallback_id: Optional[str] = None
|
|
2823
|
+
self, obj_data: dict[str, Any], fallback_id: Optional[str] = None
|
|
3256
2824
|
) -> TextObjectRef:
|
|
3257
2825
|
"""Parse JSON object data into TextObjectRef instance with hierarchical structure."""
|
|
3258
2826
|
position_data = obj_data.get("position", {})
|
|
@@ -3276,7 +2844,12 @@ class PDFDancer:
|
|
|
3276
2844
|
blue = color_data.get("blue")
|
|
3277
2845
|
alpha = color_data.get("alpha", 255)
|
|
3278
2846
|
if all(isinstance(v, int) for v in [red, green, blue]):
|
|
3279
|
-
color = Color(
|
|
2847
|
+
color = Color(
|
|
2848
|
+
cast(int, red),
|
|
2849
|
+
cast(int, green),
|
|
2850
|
+
cast(int, blue),
|
|
2851
|
+
cast(int, alpha),
|
|
2852
|
+
)
|
|
3280
2853
|
|
|
3281
2854
|
# Parse status if present
|
|
3282
2855
|
status = None
|
|
@@ -3302,7 +2875,7 @@ class PDFDancer:
|
|
|
3302
2875
|
)
|
|
3303
2876
|
|
|
3304
2877
|
text_object = TextObjectRef(
|
|
3305
|
-
internal_id=internal_id,
|
|
2878
|
+
internal_id=cast(str, internal_id),
|
|
3306
2879
|
position=position,
|
|
3307
2880
|
object_type=object_type,
|
|
3308
2881
|
text=(
|
|
@@ -3339,7 +2912,7 @@ class PDFDancer:
|
|
|
3339
2912
|
|
|
3340
2913
|
return text_object
|
|
3341
2914
|
|
|
3342
|
-
def _parse_page_ref(self, obj_data: dict) -> PageRef:
|
|
2915
|
+
def _parse_page_ref(self, obj_data: dict[str, Any]) -> PageRef:
|
|
3343
2916
|
"""Parse JSON object data into PageRef instance with page-specific properties."""
|
|
3344
2917
|
position_data = obj_data.get("position", {})
|
|
3345
2918
|
position = self._parse_position(position_data) if position_data else None
|
|
@@ -3368,14 +2941,14 @@ class PDFDancer:
|
|
|
3368
2941
|
orientation = orientation_value
|
|
3369
2942
|
|
|
3370
2943
|
return PageRef(
|
|
3371
|
-
internal_id=obj_data.get("internalId"),
|
|
3372
|
-
position=position,
|
|
2944
|
+
internal_id=cast(str, obj_data.get("internalId")),
|
|
2945
|
+
position=cast(Position, position),
|
|
3373
2946
|
type=object_type,
|
|
3374
2947
|
page_size=page_size,
|
|
3375
2948
|
orientation=orientation,
|
|
3376
2949
|
)
|
|
3377
2950
|
|
|
3378
|
-
def _parse_path_segment(self, segment_data: dict) -> "PathSegment":
|
|
2951
|
+
def _parse_path_segment(self, segment_data: dict[str, Any]) -> "PathSegment":
|
|
3379
2952
|
"""Parse JSON data into PathSegment instance (Line or Bezier)."""
|
|
3380
2953
|
from .models import Bezier, Color, Line, PathSegment, Point
|
|
3381
2954
|
|
|
@@ -3469,7 +3042,7 @@ class PDFDancer:
|
|
|
3469
3042
|
dash_phase=dash_phase,
|
|
3470
3043
|
)
|
|
3471
3044
|
|
|
3472
|
-
def _parse_path(self, obj_data: dict) ->
|
|
3045
|
+
def _parse_path(self, obj_data: dict[str, Any]) -> PDFPath:
|
|
3473
3046
|
"""Parse JSON data into Path instance with path segments."""
|
|
3474
3047
|
from .models import Path
|
|
3475
3048
|
|
|
@@ -3492,7 +3065,7 @@ class PDFDancer:
|
|
|
3492
3065
|
even_odd_fill=even_odd_fill,
|
|
3493
3066
|
)
|
|
3494
3067
|
|
|
3495
|
-
def _parse_document_font_info(self, data: dict) -> FontRecommendation:
|
|
3068
|
+
def _parse_document_font_info(self, data: dict[str, Any]) -> FontRecommendation:
|
|
3496
3069
|
"""Parse JSON data into FontRecommendation instance."""
|
|
3497
3070
|
font_type_str = data.get("fontType", "SYSTEM")
|
|
3498
3071
|
font_type = FontType(font_type_str)
|
|
@@ -3503,37 +3076,28 @@ class PDFDancer:
|
|
|
3503
3076
|
similarity_score=data.get("similarityScore", 0.0),
|
|
3504
3077
|
)
|
|
3505
3078
|
|
|
3506
|
-
def _parse_page_snapshot(self, data: dict) -> PageSnapshot:
|
|
3079
|
+
def _parse_page_snapshot(self, data: dict[str, Any]) -> PageSnapshot:
|
|
3507
3080
|
"""Parse JSON data into PageSnapshot instance with proper type handling."""
|
|
3508
3081
|
page_ref = self._parse_page_ref(data.get("pageRef", {}))
|
|
3509
3082
|
|
|
3510
3083
|
# Parse elements using appropriate parser based on type
|
|
3511
|
-
elements = []
|
|
3084
|
+
elements: List[ObjectRef] = []
|
|
3512
3085
|
for elem_data in data.get("elements", []):
|
|
3513
3086
|
elem_type_str = elem_data.get("type")
|
|
3514
3087
|
if not elem_type_str:
|
|
3515
3088
|
continue
|
|
3516
3089
|
|
|
3517
3090
|
try:
|
|
3518
|
-
# Normalize type string (API returns "CHECKBOX" but enum is "CHECK_BOX")
|
|
3519
|
-
if elem_type_str == "CHECKBOX":
|
|
3520
|
-
elem_type_str = "CHECK_BOX"
|
|
3521
|
-
# Deep copy to avoid modifying original
|
|
3522
|
-
import copy
|
|
3523
|
-
|
|
3524
|
-
elem_data = copy.deepcopy(elem_data)
|
|
3525
|
-
elem_data["type"] = elem_type_str # Update type in data
|
|
3526
|
-
|
|
3527
3091
|
elem_type = ObjectType(elem_type_str)
|
|
3528
3092
|
|
|
3529
3093
|
# Use appropriate parser based on element type
|
|
3530
|
-
if elem_type
|
|
3094
|
+
if elem_type == ObjectType.TEXT_LINE:
|
|
3531
3095
|
# Parse as TextObjectRef to capture text, font, color, children
|
|
3532
3096
|
elements.append(self._parse_text_object_ref(elem_data))
|
|
3533
3097
|
elif elem_type in (
|
|
3534
3098
|
ObjectType.FORM_FIELD,
|
|
3535
3099
|
ObjectType.TEXT_FIELD,
|
|
3536
|
-
ObjectType.
|
|
3100
|
+
ObjectType.CHECKBOX,
|
|
3537
3101
|
ObjectType.RADIO_BUTTON,
|
|
3538
3102
|
ObjectType.BUTTON,
|
|
3539
3103
|
ObjectType.DROPDOWN,
|
|
@@ -3552,7 +3116,7 @@ class PDFDancer:
|
|
|
3552
3116
|
|
|
3553
3117
|
return PageSnapshot(page_ref=page_ref, elements=elements)
|
|
3554
3118
|
|
|
3555
|
-
def _parse_document_snapshot(self, data: dict) -> DocumentSnapshot:
|
|
3119
|
+
def _parse_document_snapshot(self, data: dict[str, Any]) -> DocumentSnapshot:
|
|
3556
3120
|
"""Parse JSON data into DocumentSnapshot instance."""
|
|
3557
3121
|
page_count = data.get("pageCount", 0)
|
|
3558
3122
|
fonts = [
|
|
@@ -3565,24 +3129,17 @@ class PDFDancer:
|
|
|
3565
3129
|
|
|
3566
3130
|
return DocumentSnapshot(page_count=page_count, fonts=fonts, pages=pages)
|
|
3567
3131
|
|
|
3568
|
-
# Builder Pattern Support
|
|
3569
|
-
|
|
3570
|
-
def _paragraph_builder(self) -> "ParagraphBuilder":
|
|
3571
|
-
"""
|
|
3572
|
-
Creates a new ParagraphBuilder for fluent paragraph construction.
|
|
3573
|
-
Returns:
|
|
3574
|
-
A new ParagraphBuilder instance
|
|
3575
|
-
"""
|
|
3576
|
-
from .paragraph_builder import ParagraphBuilder
|
|
3577
|
-
|
|
3578
|
-
return ParagraphBuilder(self)
|
|
3579
|
-
|
|
3580
3132
|
# Context Manager Support (Python enhancement)
|
|
3581
|
-
def __enter__(self):
|
|
3133
|
+
def __enter__(self) -> "PDFDancer":
|
|
3582
3134
|
"""Context manager entry."""
|
|
3583
3135
|
return self
|
|
3584
3136
|
|
|
3585
|
-
def __exit__(
|
|
3137
|
+
def __exit__(
|
|
3138
|
+
self,
|
|
3139
|
+
exc_type: Optional[type[BaseException]],
|
|
3140
|
+
exc_val: Optional[BaseException],
|
|
3141
|
+
exc_tb: Any,
|
|
3142
|
+
) -> None:
|
|
3586
3143
|
"""Context manager exit - cleanup if needed."""
|
|
3587
3144
|
# Close the HTTP client to free resources
|
|
3588
3145
|
if hasattr(self, "_client"):
|
|
@@ -3590,7 +3147,7 @@ class PDFDancer:
|
|
|
3590
3147
|
# TODO Could add session cleanup here if API supports it. Cleanup on the server
|
|
3591
3148
|
pass
|
|
3592
3149
|
|
|
3593
|
-
def close(self):
|
|
3150
|
+
def close(self) -> None:
|
|
3594
3151
|
"""Close the HTTP client and free resources."""
|
|
3595
3152
|
if hasattr(self, "_client"):
|
|
3596
3153
|
self._client.close()
|
|
@@ -3598,12 +3155,6 @@ class PDFDancer:
|
|
|
3598
3155
|
def _to_path_objects(self, refs: List[ObjectRef]) -> List[PathObject]:
|
|
3599
3156
|
return [PathObject(self, ref) for ref in refs]
|
|
3600
3157
|
|
|
3601
|
-
def _to_paragraph_objects(self, refs: List[TextObjectRef]) -> List[ParagraphObject]:
|
|
3602
|
-
return [ParagraphObject(self, ref) for ref in refs]
|
|
3603
|
-
|
|
3604
|
-
def _to_textline_objects(self, refs: List[TextObjectRef]) -> List[TextLineObject]:
|
|
3605
|
-
return [TextLineObject(self, ref) for ref in refs]
|
|
3606
|
-
|
|
3607
3158
|
def _to_image_objects(self, refs: List[ObjectRef]) -> List[ImageObject]:
|
|
3608
3159
|
return [
|
|
3609
3160
|
ImageObject(self, ref.internal_id, ref.type, ref.position) for ref in refs
|
|
@@ -3617,7 +3168,12 @@ class PDFDancer:
|
|
|
3617
3168
|
def _to_form_field_objects(self, refs: List[FormFieldRef]) -> List[FormFieldObject]:
|
|
3618
3169
|
return [
|
|
3619
3170
|
FormFieldObject(
|
|
3620
|
-
self,
|
|
3171
|
+
self,
|
|
3172
|
+
ref.internal_id,
|
|
3173
|
+
ref.type,
|
|
3174
|
+
ref.position,
|
|
3175
|
+
cast(str, ref.name),
|
|
3176
|
+
cast(str, ref.value),
|
|
3621
3177
|
)
|
|
3622
3178
|
for ref in refs
|
|
3623
3179
|
]
|
|
@@ -3628,33 +3184,21 @@ class PDFDancer:
|
|
|
3628
3184
|
def _to_page_object(self, ref: PageRef) -> PageClient:
|
|
3629
3185
|
return PageClient.from_ref(self, ref)
|
|
3630
3186
|
|
|
3631
|
-
def _to_mixed_objects(
|
|
3187
|
+
def _to_mixed_objects(
|
|
3188
|
+
self, refs: List[ObjectRef]
|
|
3189
|
+
) -> List[Union[ImageObject, PathObject, FormObject, FormFieldObject]]:
|
|
3632
3190
|
"""
|
|
3633
3191
|
Convert a list of ObjectRefs to their appropriate object types.
|
|
3634
3192
|
Handles mixed object types by checking the type of each ref.
|
|
3635
3193
|
"""
|
|
3636
|
-
result = []
|
|
3194
|
+
result: List[Union[ImageObject, PathObject, FormObject, FormFieldObject]] = []
|
|
3637
3195
|
for ref in refs:
|
|
3638
|
-
if ref.type == ObjectType.
|
|
3639
|
-
# Need to convert to TextObjectRef first
|
|
3640
|
-
if isinstance(ref, TextObjectRef):
|
|
3641
|
-
result.append(ParagraphObject(self, ref))
|
|
3642
|
-
else:
|
|
3643
|
-
# Re-fetch with proper type
|
|
3644
|
-
text_refs = self._find_paragraphs(ref.position)
|
|
3645
|
-
result.extend(self._to_paragraph_objects(text_refs))
|
|
3646
|
-
elif ref.type == ObjectType.TEXT_LINE:
|
|
3647
|
-
if isinstance(ref, TextObjectRef):
|
|
3648
|
-
result.append(TextLineObject(self, ref))
|
|
3649
|
-
else:
|
|
3650
|
-
text_refs = self._find_text_lines(ref.position)
|
|
3651
|
-
result.extend(self._to_textline_objects(text_refs))
|
|
3652
|
-
elif ref.type == ObjectType.IMAGE:
|
|
3196
|
+
if ref.type == ObjectType.IMAGE:
|
|
3653
3197
|
result.append(
|
|
3654
3198
|
ImageObject(self, ref.internal_id, ref.type, ref.position)
|
|
3655
3199
|
)
|
|
3656
3200
|
elif ref.type == ObjectType.PATH:
|
|
3657
|
-
result.append(PathObject(self, ref
|
|
3201
|
+
result.append(PathObject(self, ref))
|
|
3658
3202
|
elif ref.type == ObjectType.FORM_X_OBJECT:
|
|
3659
3203
|
result.append(FormObject(self, ref.internal_id, ref.type, ref.position))
|
|
3660
3204
|
elif ref.type == ObjectType.FORM_FIELD:
|
|
@@ -3665,8 +3209,8 @@ class PDFDancer:
|
|
|
3665
3209
|
ref.internal_id,
|
|
3666
3210
|
ref.type,
|
|
3667
3211
|
ref.position,
|
|
3668
|
-
ref.name,
|
|
3669
|
-
ref.value,
|
|
3212
|
+
cast(str, ref.name),
|
|
3213
|
+
cast(str, ref.value),
|
|
3670
3214
|
)
|
|
3671
3215
|
)
|
|
3672
3216
|
else:
|
|
@@ -3674,18 +3218,12 @@ class PDFDancer:
|
|
|
3674
3218
|
result.extend(self._to_form_field_objects(form_refs))
|
|
3675
3219
|
return result
|
|
3676
3220
|
|
|
3677
|
-
def select_elements(self):
|
|
3221
|
+
def select_elements(self) -> List[ObjectRef]:
|
|
3678
3222
|
"""
|
|
3679
|
-
Select all
|
|
3223
|
+
Select all live object-reference elements in the document.
|
|
3680
3224
|
|
|
3681
3225
|
Returns:
|
|
3682
3226
|
List of all PDF objects in the document
|
|
3683
3227
|
"""
|
|
3684
|
-
|
|
3685
|
-
|
|
3686
|
-
result.extend(self.select_text_lines())
|
|
3687
|
-
result.extend(self.select_images())
|
|
3688
|
-
result.extend(self.select_paths())
|
|
3689
|
-
result.extend(self.select_forms())
|
|
3690
|
-
result.extend(self.select_form_fields())
|
|
3691
|
-
return result
|
|
3228
|
+
snapshot = self._get_or_fetch_document_snapshot()
|
|
3229
|
+
return [element for page in snapshot.pages for element in page.elements]
|