pam-python 0.2.2__py3-none-any.whl → 0.2.4__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pam/__init__.py +1 -1
- pam/cli.py +1 -1
- pam/result_batch_uploader.py +251 -66
- pam/task_manager.py +18 -19
- pam/templates/init/AGENT.md +89 -22
- pam/templates/service/pandas/service.test.tmpl +4 -0
- pam/templates/service/pandas/service_class.tmpl +18 -5
- pam/templates/service/polars/service.test.tmpl +6 -2
- pam/templates/service/polars/service_class.tmpl +25 -6
- pam/tester_task.py +2 -4
- {pam_python-0.2.2.dist-info → pam_python-0.2.4.dist-info}/METADATA +16 -13
- {pam_python-0.2.2.dist-info → pam_python-0.2.4.dist-info}/RECORD +16 -16
- {pam_python-0.2.2.dist-info → pam_python-0.2.4.dist-info}/WHEEL +0 -0
- {pam_python-0.2.2.dist-info → pam_python-0.2.4.dist-info}/entry_points.txt +0 -0
- {pam_python-0.2.2.dist-info → pam_python-0.2.4.dist-info}/licenses/LICENSE.txt +0 -0
- {pam_python-0.2.2.dist-info → pam_python-0.2.4.dist-info}/top_level.txt +0 -0
pam/__init__.py
CHANGED
pam/cli.py
CHANGED
pam/result_batch_uploader.py
CHANGED
|
@@ -1,13 +1,29 @@
|
|
|
1
|
-
"""
|
|
1
|
+
"""Asynchronous, bounded result batching for PAM CSV uploads."""
|
|
2
2
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
5
|
from dataclasses import dataclass
|
|
6
|
+
import queue
|
|
6
7
|
import threading
|
|
8
|
+
import time
|
|
7
9
|
from typing import Any, Dict, Optional, Tuple
|
|
8
10
|
|
|
11
|
+
from pam.utils import log
|
|
12
|
+
|
|
9
13
|
|
|
10
14
|
DEFAULT_BATCH_SIZE = 50000
|
|
15
|
+
DEFAULT_QUEUE_SIZE = 4
|
|
16
|
+
DEFAULT_MAX_RETRIES = 3
|
|
17
|
+
DEFAULT_RETRY_DELAY_SECONDS = 2.0
|
|
18
|
+
DEFAULT_RETRY_MAX_DELAY_SECONDS = 30.0
|
|
19
|
+
_FLUSH = object()
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass
|
|
23
|
+
class _ResultInput:
|
|
24
|
+
dataframe: Any
|
|
25
|
+
name: str
|
|
26
|
+
options: Optional[dict]
|
|
11
27
|
|
|
12
28
|
|
|
13
29
|
@dataclass
|
|
@@ -18,20 +34,56 @@ class _ResultStream:
|
|
|
18
34
|
buffer: Any
|
|
19
35
|
uploaded_rows: int = 0
|
|
20
36
|
uploaded_batches: int = 0
|
|
37
|
+
failed_rows: int = 0
|
|
38
|
+
failed_batches: int = 0
|
|
39
|
+
retried_uploads: int = 0
|
|
21
40
|
|
|
22
41
|
|
|
23
42
|
class ResultBatchUploader:
|
|
24
|
-
"""
|
|
43
|
+
"""Queue result frames and upload bounded CSV batches in one worker.
|
|
25
44
|
|
|
26
|
-
|
|
27
|
-
|
|
45
|
+
Plugin computation and intermediate storage are outside this class. Calls to
|
|
46
|
+
``upload`` submit final result frames without waiting for network I/O. A
|
|
47
|
+
bounded queue applies backpressure when producers outrun the uploader.
|
|
48
|
+
|
|
49
|
+
``flush`` closes input and places a FIFO completion marker. The worker first
|
|
50
|
+
consumes every earlier result, uploads all complete batches, then uploads all
|
|
51
|
+
remaining rows. ``wait_for_uploader`` is the completion barrier that must
|
|
52
|
+
return before a service exits.
|
|
28
53
|
"""
|
|
29
54
|
|
|
30
55
|
def __init__(self, service, batch_size: int = DEFAULT_BATCH_SIZE):
|
|
31
56
|
self._service = service
|
|
32
|
-
self.batch_size = self.
|
|
57
|
+
self.batch_size = self._resolve_positive_int("batch_size", batch_size)
|
|
58
|
+
self.queue_size = self._resolve_positive_int(
|
|
59
|
+
"upload_queue_size", DEFAULT_QUEUE_SIZE
|
|
60
|
+
)
|
|
61
|
+
self.max_retries = self._resolve_non_negative_int(
|
|
62
|
+
"upload_max_retries", DEFAULT_MAX_RETRIES
|
|
63
|
+
)
|
|
64
|
+
self.retry_delay_seconds = self._resolve_non_negative_float(
|
|
65
|
+
"upload_retry_delay_seconds", DEFAULT_RETRY_DELAY_SECONDS
|
|
66
|
+
)
|
|
67
|
+
self.retry_max_delay_seconds = self._resolve_non_negative_float(
|
|
68
|
+
"upload_retry_max_delay_seconds", DEFAULT_RETRY_MAX_DELAY_SECONDS
|
|
69
|
+
)
|
|
70
|
+
|
|
33
71
|
self._streams: Dict[str, _ResultStream] = {}
|
|
34
|
-
self.
|
|
72
|
+
self._queue: queue.Queue = queue.Queue(maxsize=self.queue_size)
|
|
73
|
+
self._submission_lock = threading.Lock()
|
|
74
|
+
self._state_lock = threading.RLock()
|
|
75
|
+
self._completion = threading.Event()
|
|
76
|
+
self._accepting = True
|
|
77
|
+
self._flush_requested = False
|
|
78
|
+
self._fatal_error: Optional[BaseException] = None
|
|
79
|
+
self._state = "open"
|
|
80
|
+
|
|
81
|
+
self._worker = threading.Thread(
|
|
82
|
+
target=self._run_worker,
|
|
83
|
+
name=f"{self.__class__.__name__}-worker",
|
|
84
|
+
daemon=True,
|
|
85
|
+
)
|
|
86
|
+
self._worker.start()
|
|
35
87
|
|
|
36
88
|
def upload(
|
|
37
89
|
self,
|
|
@@ -39,17 +91,107 @@ class ResultBatchUploader:
|
|
|
39
91
|
name: str = "default",
|
|
40
92
|
options: Optional[dict] = None,
|
|
41
93
|
) -> None:
|
|
42
|
-
|
|
94
|
+
"""Submit a final result without waiting for CSV/network upload."""
|
|
43
95
|
stream_name = self._normalize_name(name)
|
|
44
96
|
normalized_options = self._normalize_options(options)
|
|
97
|
+
submitted = self._clone_submitted_dataframe(dataframe)
|
|
45
98
|
|
|
46
|
-
with self.
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
)
|
|
99
|
+
with self._submission_lock:
|
|
100
|
+
with self._state_lock:
|
|
101
|
+
if not self._accepting:
|
|
102
|
+
raise RuntimeError("ResultBatchUploader no longer accepts results")
|
|
103
|
+
if self._fatal_error is not None:
|
|
104
|
+
raise RuntimeError("ResultBatchUploader worker failed") from self._fatal_error
|
|
105
|
+
self._queue.put(_ResultInput(submitted, stream_name, normalized_options))
|
|
106
|
+
|
|
107
|
+
def flush(self) -> None:
|
|
108
|
+
"""Close input and enqueue one terminal flush marker without blocking."""
|
|
109
|
+
with self._submission_lock:
|
|
110
|
+
with self._state_lock:
|
|
111
|
+
if self._flush_requested:
|
|
112
|
+
return
|
|
113
|
+
self._accepting = False
|
|
114
|
+
self._flush_requested = True
|
|
115
|
+
self._state = "flush_requested"
|
|
116
|
+
self._queue.put(_FLUSH)
|
|
117
|
+
|
|
118
|
+
def wait_for_uploader(self) -> None:
|
|
119
|
+
"""Wait for all queued/retried uploads and the final remainder."""
|
|
120
|
+
with self._state_lock:
|
|
121
|
+
if not self._flush_requested:
|
|
122
|
+
raise RuntimeError("flush() must be called before wait_for_uploader()")
|
|
123
|
+
self._completion.wait()
|
|
124
|
+
with self._state_lock:
|
|
125
|
+
if self._fatal_error is not None:
|
|
126
|
+
raise RuntimeError("ResultBatchUploader worker failed") from self._fatal_error
|
|
127
|
+
|
|
128
|
+
def get_status(self) -> Dict[str, dict]:
|
|
129
|
+
with self._state_lock:
|
|
130
|
+
return {
|
|
131
|
+
name: {
|
|
132
|
+
"buffered_rows": len(stream.buffer),
|
|
133
|
+
"uploaded_rows": stream.uploaded_rows,
|
|
134
|
+
"uploaded_batches": stream.uploaded_batches,
|
|
135
|
+
"failed_rows": stream.failed_rows,
|
|
136
|
+
"failed_batches": stream.failed_batches,
|
|
137
|
+
"retried_uploads": stream.retried_uploads,
|
|
138
|
+
}
|
|
139
|
+
for name, stream in self._streams.items()
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
@property
|
|
143
|
+
def state(self) -> str:
|
|
144
|
+
with self._state_lock:
|
|
145
|
+
return self._state
|
|
146
|
+
|
|
147
|
+
def _run_worker(self) -> None:
|
|
148
|
+
try:
|
|
149
|
+
while True:
|
|
150
|
+
item = self._queue.get()
|
|
151
|
+
try:
|
|
152
|
+
if item is _FLUSH:
|
|
153
|
+
if self._fatal_error is None:
|
|
154
|
+
self._flush_all_streams()
|
|
155
|
+
with self._state_lock:
|
|
156
|
+
self._state = (
|
|
157
|
+
"failed" if self._fatal_error is not None else "completed"
|
|
158
|
+
)
|
|
159
|
+
return
|
|
160
|
+
if self._fatal_error is None:
|
|
161
|
+
self._consume_input(item)
|
|
162
|
+
except BaseException as exc: # keep worker alive until the flush barrier
|
|
163
|
+
self._record_fatal_error(exc)
|
|
164
|
+
finally:
|
|
165
|
+
self._queue.task_done()
|
|
166
|
+
finally:
|
|
167
|
+
with self._state_lock:
|
|
168
|
+
self._accepting = False
|
|
169
|
+
if self._state not in {"completed", "failed"}:
|
|
170
|
+
self._state = "failed"
|
|
171
|
+
self._completion.set()
|
|
172
|
+
|
|
173
|
+
def _consume_input(self, item: _ResultInput) -> None:
|
|
174
|
+
dataframe = item.dataframe
|
|
175
|
+
if self._is_polars_lazyframe(dataframe):
|
|
176
|
+
for batch in dataframe.collect_batches(
|
|
177
|
+
chunk_size=self.batch_size,
|
|
178
|
+
maintain_order=True,
|
|
179
|
+
engine="streaming",
|
|
180
|
+
):
|
|
181
|
+
self._consume_eager(batch, item.name, item.options)
|
|
182
|
+
return
|
|
183
|
+
self._consume_eager(dataframe, item.name, item.options)
|
|
184
|
+
|
|
185
|
+
def _consume_eager(
|
|
186
|
+
self,
|
|
187
|
+
dataframe: Any,
|
|
188
|
+
name: str,
|
|
189
|
+
options: Optional[dict],
|
|
190
|
+
) -> None:
|
|
191
|
+
engine, normalized = self._normalize_eager_dataframe(dataframe)
|
|
192
|
+
batches = []
|
|
193
|
+
with self._state_lock:
|
|
194
|
+
stream = self._get_or_create_stream(name, engine, normalized, options)
|
|
53
195
|
if len(normalized) == 0:
|
|
54
196
|
return
|
|
55
197
|
|
|
@@ -68,12 +210,11 @@ class ResultBatchUploader:
|
|
|
68
210
|
len(remaining) - rows_needed,
|
|
69
211
|
)
|
|
70
212
|
if len(stream.buffer) == self.batch_size:
|
|
71
|
-
self.
|
|
213
|
+
batches.append(self._clone(engine, stream.buffer))
|
|
72
214
|
stream.buffer = self._empty_frame(engine, stream.buffer)
|
|
73
215
|
|
|
74
216
|
while len(remaining) >= self.batch_size:
|
|
75
|
-
|
|
76
|
-
self._upload_batch(stream, batch)
|
|
217
|
+
batches.append(self._slice(engine, remaining, 0, self.batch_size))
|
|
77
218
|
remaining = self._slice(
|
|
78
219
|
engine,
|
|
79
220
|
remaining,
|
|
@@ -84,30 +225,64 @@ class ResultBatchUploader:
|
|
|
84
225
|
if len(remaining) > 0:
|
|
85
226
|
stream.buffer = self._clone(engine, remaining)
|
|
86
227
|
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
)
|
|
94
|
-
|
|
95
|
-
stream = self._streams.get(stream_name)
|
|
96
|
-
if stream is None or len(stream.buffer) == 0:
|
|
228
|
+
for batch in batches:
|
|
229
|
+
self._upload_batch(name, stream, batch)
|
|
230
|
+
|
|
231
|
+
def _flush_all_streams(self) -> None:
|
|
232
|
+
pending = []
|
|
233
|
+
with self._state_lock:
|
|
234
|
+
for name, stream in self._streams.items():
|
|
235
|
+
if len(stream.buffer) == 0:
|
|
97
236
|
continue
|
|
98
|
-
self.
|
|
237
|
+
pending.append((name, stream, self._clone(stream.engine, stream.buffer)))
|
|
99
238
|
stream.buffer = self._empty_frame(stream.engine, stream.buffer)
|
|
239
|
+
for name, stream, batch in pending:
|
|
240
|
+
self._upload_batch(name, stream, batch)
|
|
100
241
|
|
|
101
|
-
def
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
242
|
+
def _upload_batch(self, name: str, stream: _ResultStream, dataframe: Any) -> None:
|
|
243
|
+
attempts = self.max_retries + 1
|
|
244
|
+
for attempt in range(1, attempts + 1):
|
|
245
|
+
try:
|
|
246
|
+
self._service._upload_result(
|
|
247
|
+
self._clone(stream.engine, dataframe),
|
|
248
|
+
stream.options,
|
|
249
|
+
)
|
|
250
|
+
with self._state_lock:
|
|
251
|
+
stream.uploaded_rows += len(dataframe)
|
|
252
|
+
stream.uploaded_batches += 1
|
|
253
|
+
return
|
|
254
|
+
except Exception as exc: # network/upload failures are retryable
|
|
255
|
+
if attempt < attempts:
|
|
256
|
+
with self._state_lock:
|
|
257
|
+
stream.retried_uploads += 1
|
|
258
|
+
delay = min(
|
|
259
|
+
self.retry_delay_seconds * (2 ** (attempt - 1)),
|
|
260
|
+
self.retry_max_delay_seconds,
|
|
261
|
+
)
|
|
262
|
+
log(
|
|
263
|
+
f"Result upload retry stream={name} rows={len(dataframe)} "
|
|
264
|
+
f"attempt={attempt + 1}/{attempts} delay={delay}s error={exc}",
|
|
265
|
+
level="WARNING",
|
|
266
|
+
)
|
|
267
|
+
if delay > 0:
|
|
268
|
+
time.sleep(delay)
|
|
269
|
+
continue
|
|
270
|
+
|
|
271
|
+
with self._state_lock:
|
|
272
|
+
stream.failed_rows += len(dataframe)
|
|
273
|
+
stream.failed_batches += 1
|
|
274
|
+
log(
|
|
275
|
+
f"Result upload skipped after retries stream={name} "
|
|
276
|
+
f"rows={len(dataframe)} attempts={attempts} error={exc}",
|
|
277
|
+
level="ERROR",
|
|
278
|
+
)
|
|
279
|
+
|
|
280
|
+
def _record_fatal_error(self, exc: BaseException) -> None:
|
|
281
|
+
with self._state_lock:
|
|
282
|
+
if self._fatal_error is None:
|
|
283
|
+
self._fatal_error = exc
|
|
284
|
+
self._accepting = False
|
|
285
|
+
log(f"Result uploader worker failed: {exc}", level="ERROR")
|
|
111
286
|
|
|
112
287
|
def _get_or_create_stream(
|
|
113
288
|
self,
|
|
@@ -138,46 +313,56 @@ class ResultBatchUploader:
|
|
|
138
313
|
raise ValueError(f"Result stream '{name}' upload options changed")
|
|
139
314
|
return stream
|
|
140
315
|
|
|
141
|
-
def
|
|
142
|
-
self._service.
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
def _resolve_batch_size(self, configured_default: int) -> int:
|
|
150
|
-
fallback = (
|
|
151
|
-
configured_default
|
|
152
|
-
if isinstance(configured_default, int)
|
|
153
|
-
and not isinstance(configured_default, bool)
|
|
154
|
-
and configured_default > 0
|
|
155
|
-
else DEFAULT_BATCH_SIZE
|
|
156
|
-
)
|
|
157
|
-
raw_value = getattr(self._service.request, "runtime_parameters", {}).get(
|
|
158
|
-
"batch_size"
|
|
159
|
-
)
|
|
160
|
-
if raw_value is None:
|
|
316
|
+
def _runtime_parameters(self) -> dict:
|
|
317
|
+
return getattr(self._service.request, "runtime_parameters", {})
|
|
318
|
+
|
|
319
|
+
def _resolve_positive_int(self, key: str, fallback: int) -> int:
|
|
320
|
+
try:
|
|
321
|
+
value = int(self._runtime_parameters().get(key, fallback))
|
|
322
|
+
except (TypeError, ValueError):
|
|
161
323
|
return fallback
|
|
324
|
+
return value if value > 0 else fallback
|
|
325
|
+
|
|
326
|
+
def _resolve_non_negative_int(self, key: str, fallback: int) -> int:
|
|
162
327
|
try:
|
|
163
|
-
|
|
328
|
+
value = int(self._runtime_parameters().get(key, fallback))
|
|
164
329
|
except (TypeError, ValueError):
|
|
165
330
|
return fallback
|
|
166
|
-
return
|
|
331
|
+
return value if value >= 0 else fallback
|
|
332
|
+
|
|
333
|
+
def _resolve_non_negative_float(self, key: str, fallback: float) -> float:
|
|
334
|
+
try:
|
|
335
|
+
value = float(self._runtime_parameters().get(key, fallback))
|
|
336
|
+
except (TypeError, ValueError):
|
|
337
|
+
return fallback
|
|
338
|
+
return value if value >= 0 else fallback
|
|
339
|
+
|
|
340
|
+
@classmethod
|
|
341
|
+
def _clone_submitted_dataframe(cls, dataframe: Any):
|
|
342
|
+
if cls._is_polars_lazyframe(dataframe):
|
|
343
|
+
return dataframe.clone()
|
|
344
|
+
_, normalized = cls._normalize_eager_dataframe(dataframe)
|
|
345
|
+
return normalized
|
|
167
346
|
|
|
168
347
|
@staticmethod
|
|
169
|
-
def
|
|
348
|
+
def _normalize_eager_dataframe(dataframe: Any) -> tuple[str, Any]:
|
|
170
349
|
module_name = type(dataframe).__module__.split(".", 1)[0]
|
|
171
350
|
class_name = type(dataframe).__name__
|
|
172
|
-
if module_name == "polars" and class_name == "LazyFrame":
|
|
173
|
-
raise TypeError(
|
|
174
|
-
"Polars LazyFrame must be uploaded directly with Service._upload_result"
|
|
175
|
-
)
|
|
176
351
|
if module_name == "polars" and class_name == "DataFrame":
|
|
177
352
|
return "polars", dataframe.clone()
|
|
178
353
|
if module_name == "pandas" and class_name == "DataFrame":
|
|
179
354
|
return "pandas", dataframe.reset_index(drop=True).copy()
|
|
180
|
-
raise TypeError(
|
|
355
|
+
raise TypeError(
|
|
356
|
+
"dataframe must be a Polars LazyFrame, Polars DataFrame, "
|
|
357
|
+
"or Pandas DataFrame"
|
|
358
|
+
)
|
|
359
|
+
|
|
360
|
+
@staticmethod
|
|
361
|
+
def _is_polars_lazyframe(dataframe: Any) -> bool:
|
|
362
|
+
return (
|
|
363
|
+
type(dataframe).__module__.split(".", 1)[0] == "polars"
|
|
364
|
+
and type(dataframe).__name__ == "LazyFrame"
|
|
365
|
+
)
|
|
181
366
|
|
|
182
367
|
@staticmethod
|
|
183
368
|
def _clone(engine: str, dataframe: Any):
|
pam/task_manager.py
CHANGED
|
@@ -305,7 +305,10 @@ class TaskManager(ITaskManager):
|
|
|
305
305
|
|
|
306
306
|
def service_upload_result(self, service: Service, file_path, options=None):
|
|
307
307
|
"""
|
|
308
|
-
Uploads
|
|
308
|
+
Uploads one result file synchronously.
|
|
309
|
+
|
|
310
|
+
ResultBatchUploader owns background concurrency and invokes this method
|
|
311
|
+
from its worker, so failures must propagate for retry handling.
|
|
309
312
|
"""
|
|
310
313
|
endpoint = service.request.response_api
|
|
311
314
|
payload = None
|
|
@@ -318,25 +321,21 @@ class TaskManager(ITaskManager):
|
|
|
318
321
|
is_priority = options.get("is_priority")
|
|
319
322
|
payload["is_priority"] = str(is_priority).lower() if isinstance(is_priority, bool) else is_priority
|
|
320
323
|
|
|
321
|
-
def handle_upload_response(response):
|
|
322
|
-
"""Logs the response after the upload completes."""
|
|
323
|
-
if response is None:
|
|
324
|
-
log(f"Response from upload to {endpoint}: None")
|
|
325
|
-
return
|
|
326
|
-
try:
|
|
327
|
-
response_data = response.json()
|
|
328
|
-
except ValueError:
|
|
329
|
-
response_data = response.text
|
|
330
|
-
log(f"Response from upload to {endpoint}: {response_data}")
|
|
331
|
-
|
|
332
|
-
def upload_wrapper():
|
|
333
|
-
"""Wrapper for the upload to handle response logging."""
|
|
334
|
-
response = self.api.http_upload(endpoint, file_path, payload)
|
|
335
|
-
handle_upload_response(response)
|
|
336
|
-
|
|
337
324
|
log(f"Uploading Result to: {endpoint}")
|
|
338
|
-
|
|
339
|
-
|
|
325
|
+
response = self.api.http_upload(endpoint, file_path, payload)
|
|
326
|
+
if response is None:
|
|
327
|
+
raise RuntimeError(f"No response from result upload to {endpoint}")
|
|
328
|
+
if not response.ok:
|
|
329
|
+
raise RuntimeError(
|
|
330
|
+
f"Result upload failed status={response.status_code} "
|
|
331
|
+
f"endpoint={endpoint}"
|
|
332
|
+
)
|
|
333
|
+
try:
|
|
334
|
+
response_data = response.json()
|
|
335
|
+
except ValueError:
|
|
336
|
+
response_data = response.text
|
|
337
|
+
log(f"Response from upload to {endpoint}: {response_data}")
|
|
338
|
+
return response
|
|
340
339
|
|
|
341
340
|
def service_upload_report(self, service: Service, file_path):
|
|
342
341
|
"""
|
pam/templates/init/AGENT.md
CHANGED
|
@@ -117,15 +117,17 @@ input data. CDP returns up to 5 Parquet files (each file is a separate event).
|
|
|
117
117
|
|
|
118
118
|
- File order is guaranteed per agreement, so you can index by position.
|
|
119
119
|
- Column names and record counts are defined by CDP configuration.
|
|
120
|
+
- `input_files` is not a list of interchangeable files. Each position represents
|
|
121
|
+
a different configured data source and must be bound to a meaningful domain name.
|
|
122
|
+
- Never process `input_files` with `for file_path in req.input_files`.
|
|
120
123
|
|
|
121
124
|
Example file order usage:
|
|
122
125
|
|
|
123
126
|
```python
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
use_points_path = input_files[1]
|
|
127
|
+
if len(req.input_files) != 2:
|
|
128
|
+
raise ValueError("Expected purchase_event and point_received_event inputs")
|
|
129
|
+
|
|
130
|
+
purchase_event_file, point_received_event_file = req.input_files
|
|
129
131
|
```
|
|
130
132
|
|
|
131
133
|
Column meaning
|
|
@@ -148,6 +150,11 @@ When CDP finishes collecting data, it calls `on_data_input`.
|
|
|
148
150
|
Example:
|
|
149
151
|
|
|
150
152
|
```python
|
|
153
|
+
def __init__(self, task_manager, req):
|
|
154
|
+
super().__init__(task_manager, req)
|
|
155
|
+
self.batch_uploader = ResultBatchUploader(self, batch_size=50000)
|
|
156
|
+
|
|
157
|
+
|
|
151
158
|
def on_data_input(self, req: RequestCommand):
|
|
152
159
|
log(f"on_data_input req.is_end = {req.is_end}")
|
|
153
160
|
|
|
@@ -163,26 +170,40 @@ def __run_process_data_in_thread(self, req: RequestCommand):
|
|
|
163
170
|
if req.file_format is not RequestFileFormat.PARQUET:
|
|
164
171
|
raise ValueError(f"Expected parquet, received {req.file_format.value}")
|
|
165
172
|
|
|
166
|
-
|
|
167
|
-
|
|
173
|
+
if len(req.input_files) != 2:
|
|
174
|
+
raise ValueError("Expected purchase_event and point_received_event inputs")
|
|
175
|
+
|
|
176
|
+
purchase_event_file, point_received_event_file = req.input_files
|
|
177
|
+
result = transform(
|
|
178
|
+
pl.scan_parquet(purchase_event_file),
|
|
179
|
+
pl.scan_parquet(point_received_event_file),
|
|
180
|
+
)
|
|
181
|
+
self.batch_uploader.upload(result, name="main")
|
|
168
182
|
|
|
169
183
|
if not req.is_end:
|
|
170
|
-
self._request_data(
|
|
184
|
+
self._request_data(
|
|
185
|
+
page=req.next,
|
|
186
|
+
file_format=RequestFileFormat.PARQUET,
|
|
187
|
+
)
|
|
171
188
|
else:
|
|
189
|
+
self.batch_uploader.flush()
|
|
190
|
+
self.batch_uploader.wait_for_uploader()
|
|
172
191
|
self._exit()
|
|
173
192
|
```
|
|
174
193
|
|
|
175
194
|
Why a thread: to end the HTTP request quickly and avoid timeouts.
|
|
176
195
|
|
|
177
|
-
`req.input_files` is the ordered
|
|
178
|
-
typed metadata sent by PAM and is the source of truth;
|
|
179
|
-
an additional consistency check.
|
|
196
|
+
`req.input_files` is the ordered positional contract of Parquet paths.
|
|
197
|
+
`req.file_format` is the typed metadata sent by PAM and is the source of truth;
|
|
198
|
+
validate file extensions as an additional consistency check.
|
|
180
199
|
|
|
181
200
|
---
|
|
182
201
|
|
|
183
202
|
## Upload Results to CDP
|
|
184
203
|
|
|
185
|
-
After processing,
|
|
204
|
+
After processing, submit final dataframe results through `ResultBatchUploader`.
|
|
205
|
+
The uploader creates bounded CSV files and sends them through the PAM API.
|
|
206
|
+
Example output:
|
|
186
207
|
|
|
187
208
|
```csv
|
|
188
209
|
id,data_x,rfm
|
|
@@ -258,21 +279,32 @@ CDP paging is **customer-based**, not row-based.
|
|
|
258
279
|
|
|
259
280
|
When requesting data page by page:
|
|
260
281
|
|
|
261
|
-
-
|
|
262
|
-
-
|
|
263
|
-
-
|
|
282
|
+
- PAM first selects a cohort of approximately **10,000 unique customers per page**.
|
|
283
|
+
- PAM then queries **every configured dataset** with that same customer-ID cohort.
|
|
284
|
+
- Dataset 1 through dataset N in one callback therefore belong to the same group
|
|
285
|
+
of customers, while retaining their configured positional meanings.
|
|
286
|
+
- Each dataset has its own Parquet file and may have a different row count because
|
|
287
|
+
event density differs and some customers may have no rows in a dataset.
|
|
288
|
+
- A customer's relevant data is kept within one logical page instead of being
|
|
289
|
+
split across customer pages.
|
|
290
|
+
- `req.next` advances the customer-cohort cursor, not an event-row offset.
|
|
264
291
|
|
|
265
292
|
This means:
|
|
266
293
|
|
|
267
294
|
- Each page represents a logical customer group, not a fixed number of records.
|
|
268
295
|
- A page may contain a small or very large number of rows depending on behavior density.
|
|
296
|
+
- Files at different input positions must be joined or compared by the agreed
|
|
297
|
+
customer key; row positions across files do not correspond.
|
|
269
298
|
|
|
270
299
|
### Execution Implications
|
|
271
300
|
|
|
272
301
|
Because paging is based on customers:
|
|
273
302
|
|
|
274
303
|
- In many cases, a service can process **one page (≈10,000 customers)** independently.
|
|
275
|
-
- If the logic allows, you may process and upload results immediately because
|
|
304
|
+
- If the logic allows, you may process and upload results immediately because all
|
|
305
|
+
configured datasets for that customer cohort arrive together.
|
|
306
|
+
- Do not create row-offset pagination or carry partial state because one dataset
|
|
307
|
+
happens to have more event rows than another.
|
|
276
308
|
|
|
277
309
|
However, upload behavior must balance two constraints:
|
|
278
310
|
|
|
@@ -368,6 +400,11 @@ Report types outside that closed contract require a newer framework and CMS rele
|
|
|
368
400
|
When uploading large DataFrames, always upload in batches to avoid memory/network spikes.
|
|
369
401
|
Batch size MUST be configurable via `request.runtime_parameters["batch_size"]`.
|
|
370
402
|
|
|
403
|
+
Computation is a developer-owned black box. The service may calculate within one
|
|
404
|
+
page, retain state across pages, or use any intermediate store such as SQLite or
|
|
405
|
+
DuckDB. The framework does not choose that strategy. Call the uploader only when
|
|
406
|
+
the produced rows are ready for the PAM result contract.
|
|
407
|
+
|
|
371
408
|
## Required behavior
|
|
372
409
|
|
|
373
410
|
- Read batch size from `self.request.runtime_parameters.get("batch_size", "50000")`
|
|
@@ -379,10 +416,17 @@ Batch size MUST be configurable via `request.runtime_parameters["batch_size"]`.
|
|
|
379
416
|
|
|
380
417
|
## Reference implementation pattern
|
|
381
418
|
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
`runtime_parameters["batch_size"]`, keeps named result streams separate,
|
|
385
|
-
complete batches, and
|
|
419
|
+
`ResultBatchUploader` accepts Polars `LazyFrame`, Polars `DataFrame`, and Pandas
|
|
420
|
+
`DataFrame`. It evaluates lazy results as bounded streaming chunks, reads
|
|
421
|
+
`runtime_parameters["batch_size"]`, keeps named result streams separate, converts
|
|
422
|
+
complete batches to CSV, and uploads through the PAM API. Incomplete batches may
|
|
423
|
+
remain buffered across pages or calls until more final rows arrive or `flush()` is
|
|
424
|
+
called.
|
|
425
|
+
|
|
426
|
+
`upload()` queues work for one background uploader and normally returns without
|
|
427
|
+
waiting for network I/O, allowing the service to request and process the next PAM
|
|
428
|
+
page. The queue is bounded, so `upload()` may apply backpressure if uploads are
|
|
429
|
+
slower than result production.
|
|
386
430
|
|
|
387
431
|
```python
|
|
388
432
|
from pam.result_batch_uploader import ResultBatchUploader
|
|
@@ -390,13 +434,36 @@ from pam.result_batch_uploader import ResultBatchUploader
|
|
|
390
434
|
self.batch_uploader = ResultBatchUploader(self, batch_size=50000)
|
|
391
435
|
self.batch_uploader.upload(result_df, name="main")
|
|
392
436
|
self.batch_uploader.flush()
|
|
437
|
+
self.batch_uploader.wait_for_uploader()
|
|
438
|
+
self._exit()
|
|
393
439
|
```
|
|
394
440
|
|
|
441
|
+
`flush()` is both a terminal state change and a FIFO marker. It stops new result
|
|
442
|
+
submissions but does not wait. The worker processes every result submitted before
|
|
443
|
+
the marker, preserves `batch_size`, uploads the final remainder, and then completes.
|
|
444
|
+
`wait_for_uploader()` blocks until queued and in-flight network uploads finish.
|
|
445
|
+
It intentionally has no timeout: timeout each HTTP upload attempt instead, then
|
|
446
|
+
retry within the configured limit or skip the permanently failed batch. A timeout
|
|
447
|
+
on the completion barrier could let the plugin exit while uploads are still active.
|
|
448
|
+
|
|
449
|
+
Upload retry settings are optional runtime parameters:
|
|
450
|
+
|
|
451
|
+
- `upload_queue_size` (default `4`): maximum queued result objects before backpressure
|
|
452
|
+
- `upload_max_retries` (default `3`): retries after the first upload attempt
|
|
453
|
+
- `upload_retry_delay_seconds` (default `2`): initial exponential-backoff delay
|
|
454
|
+
- `upload_retry_max_delay_seconds` (default `30`): backoff delay cap
|
|
455
|
+
|
|
456
|
+
After retries are exhausted, the uploader logs the failed batch, records
|
|
457
|
+
`failed_rows`/`failed_batches`, skips it, and continues with the next batch.
|
|
458
|
+
`wait_for_uploader()` still waits for the whole queue before returning.
|
|
459
|
+
|
|
395
460
|
## Notes
|
|
396
461
|
|
|
397
462
|
- Do NOT hardcode a fixed batch size inside upload loops.
|
|
398
463
|
- Do NOT read environment variables for batch sizing unless explicitly required; use `runtime_parameters` only.
|
|
399
|
-
- Do NOT
|
|
464
|
+
- Do NOT call `_upload_result(result)` directly from plugin code; use `ResultBatchUploader`.
|
|
465
|
+
- Call `flush()` only after all intended final rows have been submitted.
|
|
466
|
+
- Always call `wait_for_uploader()` after `flush()` and before `_exit()`.
|
|
400
467
|
|
|
401
468
|
---
|
|
402
469
|
|
|
@@ -419,7 +486,7 @@ Prefer streaming/chunked/incremental processing whenever possible.
|
|
|
419
486
|
|
|
420
487
|
- Start with `pl.scan_parquet(...)`, not `pl.read_parquet(...)`.
|
|
421
488
|
- Select required columns and filter rows early for projection/predicate pushdown.
|
|
422
|
-
- Keep transformations as a `LazyFrame` through `
|
|
489
|
+
- Keep transformations as a `LazyFrame` through `ResultBatchUploader.upload(...)`.
|
|
423
490
|
|
|
424
491
|
### 2. DB-first computation
|
|
425
492
|
|
|
@@ -18,6 +18,7 @@ from #MODULE_NAME#.#CLASS_NAME# import #CLASS_NAME#
|
|
|
18
18
|
TOTAL_PAGES = 3
|
|
19
19
|
MOCK_DIR = "./TEMP_TEST_DATA"
|
|
20
20
|
UPLOADED_RESULTS = []
|
|
21
|
+
UPLOADED_ROW_COUNTS = []
|
|
21
22
|
REQUESTED_FORMATS = []
|
|
22
23
|
TempfileUtils.temp_base_path = f"{MOCK_DIR}/app/data"
|
|
23
24
|
TempfileUtils.temp_datasource_path = f"{MOCK_DIR}/app/data/data_sources"
|
|
@@ -68,6 +69,7 @@ def on_publish_report_pointer(report_key, pointer):
|
|
|
68
69
|
|
|
69
70
|
def on_upload_result(result_file):
|
|
70
71
|
UPLOADED_RESULTS.append(result_file)
|
|
72
|
+
UPLOADED_ROW_COUNTS.append(len(pd.read_csv(result_file)))
|
|
71
73
|
|
|
72
74
|
|
|
73
75
|
class Test#CLASS_NAME#(unittest.TestCase):
|
|
@@ -77,6 +79,7 @@ class Test#CLASS_NAME#(unittest.TestCase):
|
|
|
77
79
|
os.makedirs(MOCK_DIR)
|
|
78
80
|
create_mock_parquet(TOTAL_PAGES, MOCK_DIR)
|
|
79
81
|
UPLOADED_RESULTS.clear()
|
|
82
|
+
UPLOADED_ROW_COUNTS.clear()
|
|
80
83
|
REQUESTED_FORMATS.clear()
|
|
81
84
|
|
|
82
85
|
self.tester_task = TesterTask()
|
|
@@ -112,6 +115,7 @@ class Test#CLASS_NAME#(unittest.TestCase):
|
|
|
112
115
|
|
|
113
116
|
self.assertTrue(self.tester_task.is_exit)
|
|
114
117
|
self.assertEqual(2, len(UPLOADED_RESULTS))
|
|
118
|
+
self.assertEqual([15, 15], UPLOADED_ROW_COUNTS)
|
|
115
119
|
self.assertEqual(
|
|
116
120
|
[RequestFileFormat.PARQUET] * TOTAL_PAGES,
|
|
117
121
|
REQUESTED_FORMATS,
|
|
@@ -13,6 +13,7 @@ from #MODULE_NAME#.functions import load_input_file, transform
|
|
|
13
13
|
|
|
14
14
|
class #CLASS_NAME#(Service):
|
|
15
15
|
INPUT_FILE_FORMAT = RequestFileFormat.PARQUET
|
|
16
|
+
EXPECTED_DATASET_COUNT = 1
|
|
16
17
|
|
|
17
18
|
def __init__(self, task_manager, req):
|
|
18
19
|
super().__init__(task_manager, req)
|
|
@@ -40,11 +41,22 @@ class #CLASS_NAME#(Service):
|
|
|
40
41
|
f"received {req.file_format.value}"
|
|
41
42
|
)
|
|
42
43
|
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
44
|
+
if len(req.input_files) != self.EXPECTED_DATASET_COUNT:
|
|
45
|
+
raise ValueError(
|
|
46
|
+
f"Expected {self.EXPECTED_DATASET_COUNT} configured dataset, "
|
|
47
|
+
f"received {len(req.input_files)}"
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
# Bind every position to the data source name configured in PAM.
|
|
51
|
+
# Add more named positions here when the service contract has more datasets.
|
|
52
|
+
(input_dataset_file,) = req.input_files
|
|
53
|
+
if not input_dataset_file.lower().endswith(".parquet"):
|
|
54
|
+
raise ValueError(
|
|
55
|
+
f"Expected a .parquet input file: {input_dataset_file}"
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
result = transform(load_input_file(input_dataset_file))
|
|
59
|
+
self.batch_uploader.upload(result, name="main")
|
|
48
60
|
|
|
49
61
|
if not req.is_end:
|
|
50
62
|
self._request_data(
|
|
@@ -53,6 +65,7 @@ class #CLASS_NAME#(Service):
|
|
|
53
65
|
)
|
|
54
66
|
else:
|
|
55
67
|
self.batch_uploader.flush()
|
|
68
|
+
self.batch_uploader.wait_for_uploader()
|
|
56
69
|
self._exit()
|
|
57
70
|
|
|
58
71
|
def on_destroy(self):
|
|
@@ -18,6 +18,7 @@ from #MODULE_NAME#.#CLASS_NAME# import #CLASS_NAME#
|
|
|
18
18
|
TOTAL_PAGES = 3
|
|
19
19
|
MOCK_DIR = "./TEMP_TEST_DATA"
|
|
20
20
|
UPLOADED_RESULTS = []
|
|
21
|
+
UPLOADED_ROW_COUNTS = []
|
|
21
22
|
REQUESTED_FORMATS = []
|
|
22
23
|
TempfileUtils.temp_base_path = f"{MOCK_DIR}/app/data"
|
|
23
24
|
TempfileUtils.temp_datasource_path = f"{MOCK_DIR}/app/data/data_sources"
|
|
@@ -68,6 +69,7 @@ def on_publish_report_pointer(report_key, pointer):
|
|
|
68
69
|
|
|
69
70
|
def on_upload_result(result_file):
|
|
70
71
|
UPLOADED_RESULTS.append(result_file)
|
|
72
|
+
UPLOADED_ROW_COUNTS.append(pl.read_csv(result_file).height)
|
|
71
73
|
|
|
72
74
|
|
|
73
75
|
class Test#CLASS_NAME#(unittest.TestCase):
|
|
@@ -77,6 +79,7 @@ class Test#CLASS_NAME#(unittest.TestCase):
|
|
|
77
79
|
os.makedirs(MOCK_DIR)
|
|
78
80
|
create_mock_parquet(TOTAL_PAGES, MOCK_DIR)
|
|
79
81
|
UPLOADED_RESULTS.clear()
|
|
82
|
+
UPLOADED_ROW_COUNTS.clear()
|
|
80
83
|
REQUESTED_FORMATS.clear()
|
|
81
84
|
|
|
82
85
|
self.tester_task = TesterTask()
|
|
@@ -96,7 +99,7 @@ class Test#CLASS_NAME#(unittest.TestCase):
|
|
|
96
99
|
sqlite_download="",
|
|
97
100
|
sqlite_download_v2="",
|
|
98
101
|
sqlite_upload="",
|
|
99
|
-
runtime_parameters={"some_param": "some_value"},
|
|
102
|
+
runtime_parameters={"some_param": "some_value", "batch_size": "15"},
|
|
100
103
|
token=uuid.uuid4().hex,
|
|
101
104
|
cmd="start",
|
|
102
105
|
data_api="",
|
|
@@ -111,7 +114,8 @@ class Test#CLASS_NAME#(unittest.TestCase):
|
|
|
111
114
|
self.tester_task.wait_for_task_done()
|
|
112
115
|
|
|
113
116
|
self.assertTrue(self.tester_task.is_exit)
|
|
114
|
-
self.assertEqual(
|
|
117
|
+
self.assertEqual(2, len(UPLOADED_RESULTS))
|
|
118
|
+
self.assertEqual([15, 15], UPLOADED_ROW_COUNTS)
|
|
115
119
|
self.assertEqual(
|
|
116
120
|
[RequestFileFormat.PARQUET] * TOTAL_PAGES,
|
|
117
121
|
REQUESTED_FORMATS,
|
|
@@ -3,6 +3,7 @@ import threading
|
|
|
3
3
|
|
|
4
4
|
from pam.models.request_command import RequestCommand
|
|
5
5
|
from pam.request_file_format import RequestFileFormat
|
|
6
|
+
from pam.result_batch_uploader import ResultBatchUploader
|
|
6
7
|
from pam.service import Service
|
|
7
8
|
from pam.utils import log
|
|
8
9
|
|
|
@@ -12,6 +13,11 @@ from #MODULE_NAME#.functions import load_input_file, transform
|
|
|
12
13
|
|
|
13
14
|
class #CLASS_NAME#(Service):
|
|
14
15
|
INPUT_FILE_FORMAT = RequestFileFormat.PARQUET
|
|
16
|
+
EXPECTED_DATASET_COUNT = 1
|
|
17
|
+
|
|
18
|
+
def __init__(self, task_manager, req):
|
|
19
|
+
super().__init__(task_manager, req)
|
|
20
|
+
self.batch_uploader = ResultBatchUploader(self, batch_size=50000)
|
|
15
21
|
|
|
16
22
|
def on_start(self):
|
|
17
23
|
log("on_start")
|
|
@@ -35,11 +41,22 @@ class #CLASS_NAME#(Service):
|
|
|
35
41
|
f"received {req.file_format.value}"
|
|
36
42
|
)
|
|
37
43
|
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
44
|
+
if len(req.input_files) != self.EXPECTED_DATASET_COUNT:
|
|
45
|
+
raise ValueError(
|
|
46
|
+
f"Expected {self.EXPECTED_DATASET_COUNT} configured dataset, "
|
|
47
|
+
f"received {len(req.input_files)}"
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
# Bind every position to the data source name configured in PAM.
|
|
51
|
+
# Add more named positions here when the service contract has more datasets.
|
|
52
|
+
(input_dataset_file,) = req.input_files
|
|
53
|
+
if not input_dataset_file.lower().endswith(".parquet"):
|
|
54
|
+
raise ValueError(
|
|
55
|
+
f"Expected a .parquet input file: {input_dataset_file}"
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
result = transform(load_input_file(input_dataset_file))
|
|
59
|
+
self.batch_uploader.upload(result, name="main")
|
|
43
60
|
|
|
44
61
|
if not req.is_end:
|
|
45
62
|
self._request_data(
|
|
@@ -47,6 +64,8 @@ class #CLASS_NAME#(Service):
|
|
|
47
64
|
file_format=self.INPUT_FILE_FORMAT,
|
|
48
65
|
)
|
|
49
66
|
else:
|
|
67
|
+
self.batch_uploader.flush()
|
|
68
|
+
self.batch_uploader.wait_for_uploader()
|
|
50
69
|
self._exit()
|
|
51
70
|
|
|
52
71
|
def on_destroy(self):
|
|
@@ -56,4 +75,4 @@ class #CLASS_NAME#(Service):
|
|
|
56
75
|
log("on_terminate")
|
|
57
76
|
|
|
58
77
|
def get_status(self):
|
|
59
|
-
return json.dumps(
|
|
78
|
+
return json.dumps(self.batch_uploader.get_status(), sort_keys=True)
|
pam/tester_task.py
CHANGED
|
@@ -207,12 +207,10 @@ class TesterTask(ITaskManager):
|
|
|
207
207
|
:param options: Optional upload options (e.g. is_realtime, is_priority).
|
|
208
208
|
"""
|
|
209
209
|
if self.upload_result_callback is not None:
|
|
210
|
-
|
|
211
|
-
self.upload_result_callback(file_path)
|
|
212
|
-
except Exception as e:
|
|
213
|
-
print(f"Error in upload_result_callback: {e}")
|
|
210
|
+
return self.upload_result_callback(file_path)
|
|
214
211
|
else:
|
|
215
212
|
print("Upload result callback is not set.")
|
|
213
|
+
return None
|
|
216
214
|
|
|
217
215
|
def service_upload_report(self, service: Service, file_path: str):
|
|
218
216
|
"""
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pam-python
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.4
|
|
4
4
|
Summary: Pam Python Library
|
|
5
5
|
Author-email: Narongrit Kanhanoi <narongrit@pams.ai>
|
|
6
6
|
Project-URL: Homepage, https://github.com/heart/pam-python
|
|
@@ -122,13 +122,13 @@ The runtime calls your service in two main phases.
|
|
|
122
122
|
2. `on_data_input`
|
|
123
123
|
|
|
124
124
|
- Called when CDP sends input files
|
|
125
|
-
- `req.input_files`
|
|
125
|
+
- `req.input_files` is the ordered positional data-source contract configured in PAM
|
|
126
126
|
- `req.file_format` is the typed format sent by PAM
|
|
127
127
|
- Should also return quickly (use a thread if needed)
|
|
128
128
|
|
|
129
129
|
When your service is done:
|
|
130
130
|
|
|
131
|
-
- Upload ordinary CDP result rows through `ResultBatchUploader`
|
|
131
|
+
- Upload ordinary CDP result rows through `ResultBatchUploader`
|
|
132
132
|
- Publish managed reports through `self.reports`; do not build report JSON manually
|
|
133
133
|
- Call `self._exit()` to signal completion
|
|
134
134
|
|
|
@@ -151,11 +151,11 @@ Notes:
|
|
|
151
151
|
---
|
|
152
152
|
|
|
153
153
|
**Uploading Results in Batches**
|
|
154
|
-
Polars
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
154
|
+
Polars `LazyFrame`/`DataFrame` and Pandas `DataFrame` results all pass through
|
|
155
|
+
`ResultBatchUploader`. It streams lazy results in bounded chunks, buffers incomplete
|
|
156
|
+
batches across calls, writes complete batches as CSV, and uploads them through PAM.
|
|
157
|
+
The framework does not prescribe page-local/global computation or intermediate
|
|
158
|
+
storage; submit rows when your business result is ready.
|
|
159
159
|
|
|
160
160
|
Recommended usage:
|
|
161
161
|
|
|
@@ -165,6 +165,7 @@ from pam.result_batch_uploader import ResultBatchUploader
|
|
|
165
165
|
batch_uploader = ResultBatchUploader(self, batch_size=50000)
|
|
166
166
|
batch_uploader.upload(df, name="main")
|
|
167
167
|
batch_uploader.flush()
|
|
168
|
+
batch_uploader.wait_for_uploader()
|
|
168
169
|
status = batch_uploader.get_status()
|
|
169
170
|
```
|
|
170
171
|
|
|
@@ -173,9 +174,12 @@ Notes:
|
|
|
173
174
|
- `request.runtime_parameters["batch_size"]` overrides the constructor default when valid.
|
|
174
175
|
- `name` separates result streams and enforces one stable column schema per stream.
|
|
175
176
|
- `options` may be passed to `upload(...)` and are forwarded to `_upload_result`.
|
|
176
|
-
-
|
|
177
|
-
- `
|
|
178
|
-
- `
|
|
177
|
+
- Uploads run in a bounded background queue so processing can continue while PAM receives CSV batches.
|
|
178
|
+
- `flush()` closes input and queues the final remainder; it does not wait for network completion.
|
|
179
|
+
- Always call `wait_for_uploader()` after `flush()` and before `_exit()`.
|
|
180
|
+
- Failed uploads retry with exponential backoff, then log/record the failed batch and continue.
|
|
181
|
+
- `get_status()` includes buffered, uploaded, failed, and retried counts per stream.
|
|
182
|
+
- Polars `LazyFrame` is supported without collecting the complete result at once.
|
|
179
183
|
|
|
180
184
|
---
|
|
181
185
|
|
|
@@ -304,6 +308,5 @@ After `pam init` and one service:
|
|
|
304
308
|
- Implement your logic in `functions.py`.
|
|
305
309
|
- Wire it into `on_start` and `on_data_input` in your service class.
|
|
306
310
|
- Use the temp utilities to write intermediate files.
|
|
307
|
-
- Keep Polars transformations lazy and
|
|
308
|
-
- Use `ResultBatchUploader` only for eager Polars/Pandas frames that need row batching.
|
|
311
|
+
- Keep Polars transformations lazy and submit final results through `ResultBatchUploader`.
|
|
309
312
|
- Read `REPORTS.md` before implementing a managed report.
|
|
@@ -1,16 +1,16 @@
|
|
|
1
|
-
pam/__init__.py,sha256=
|
|
1
|
+
pam/__init__.py,sha256=cSLmJntln2OVNNJ07_9Z2eKFLloAZc33rsRlnRTj0wk,56
|
|
2
2
|
pam/api.py,sha256=4t64ey_Jc4gC3SvfI-mtedEGzkX2cPTt0cdZgP_huLk,7231
|
|
3
|
-
pam/cli.py,sha256=
|
|
3
|
+
pam/cli.py,sha256=Kbl5K4xS5KtW1cU3QRZOi8eAdKzUaJqe2JDGuhy4wQA,14131
|
|
4
4
|
pam/interface_task_manager.py,sha256=moKUjSpAeNF9C2dluDFPAdU4IyvwXckB0RK-jINTLm4,1665
|
|
5
5
|
pam/logger.py,sha256=JYPfIpaQZy6QrfVKrzEMrxOgTzWIEhoegvIZgHTGWoc,3021
|
|
6
6
|
pam/request_file_format.py,sha256=fSRlJagbvVGqdIEqhINSeqBNUTNQdm8GL5JfqgX9bxU,317
|
|
7
|
-
pam/result_batch_uploader.py,sha256=
|
|
7
|
+
pam/result_batch_uploader.py,sha256=xw4b-q7i52CoCjvQtQ8EgfBrrqAl1zEl6sTO-wIvFPc,15424
|
|
8
8
|
pam/server.py,sha256=ZRukc3uLJXxJdWIQDcPqrxjd_LnsGTxPJQtLX0bMcRc,7594
|
|
9
9
|
pam/service.py,sha256=EymzqpHwjmCpSC3e3QchdnrE7L39_hxtDVB3kCJkfr4,7760
|
|
10
10
|
pam/sqlite.py,sha256=0bEUnb17V_Z6N6u843MYyGQnigaAuMx6FUckeWifn8I,2585
|
|
11
|
-
pam/task_manager.py,sha256=
|
|
11
|
+
pam/task_manager.py,sha256=DC2pZ1iJU-GnCl8mawuC6Ym105ywoZ6rs6neXUexHZ4,15797
|
|
12
12
|
pam/temp_file_utils.py,sha256=lVkRJuNMdmqbl22BeP4tGo2YAdYy4C_SMzcR4vSQOb4,6762
|
|
13
|
-
pam/tester_task.py,sha256=
|
|
13
|
+
pam/tester_task.py,sha256=s6BelOUMOahUh_5-lRN5yUEO72caR-ERQS8b3ZpfQgw,9213
|
|
14
14
|
pam/utils.py,sha256=-pfnQ00Ib-IXvrMz4S5lTlO36PtSaJoQOhwXpwet8h4,1744
|
|
15
15
|
pam/models/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
16
16
|
pam/models/request_command.py,sha256=tGzPuP1TQI0CrNjiigW929lJ99Fvz-WfcUh02eOVFK4,6621
|
|
@@ -48,7 +48,7 @@ pam/reports/transport.py,sha256=t3oEVptS9msnNafia1G8vmTUCdhxvV5L2-0AY-8SKFk,2062
|
|
|
48
48
|
pam/reports/treemap.py,sha256=W_VwxbiRMSvkNpY2z1HfbZtGKWXNr6FdbVdI0grVesw,4251
|
|
49
49
|
pam/templates/buildcmd/pamb,sha256=58GCN59ysvHGNhPvTD3nkF4aLg-kM1WfWmH7CL3eQVo,530
|
|
50
50
|
pam/templates/buildcmd/pamb-base.sh,sha256=Q0DkbeHtPx33PIjc68EeI7ZIwBOLlwcSmPhpxALXxVY,518
|
|
51
|
-
pam/templates/init/AGENT.md,sha256=
|
|
51
|
+
pam/templates/init/AGENT.md,sha256=uEvHox6K43Em2vWQ8H_5a2NswLUdlSmfve3cWEwLDes,18524
|
|
52
52
|
pam/templates/init/REPORTS.md,sha256=rg9oi0_xDYr0BZBluJh0vPF8lSx7mCRFpDCpUgNl68A,42851
|
|
53
53
|
pam/templates/init/dockerignore.tmpl,sha256=Ht1eCgQLqalEwpUjih0FQavzhYJbT2jH-FP4btM3960,52
|
|
54
54
|
pam/templates/init/gitignore.tmpl,sha256=U59ynPRu_Kw6evqQzOOTAhWKdIXbf1uBOF2V8223tU8,3510
|
|
@@ -64,14 +64,14 @@ pam/templates/project/uv/pyproject.toml,sha256=30a2hJn6oN1AGvmgEBdxLETOBlHY8L-II
|
|
|
64
64
|
pam/templates/project/uv/python-version,sha256=e1X45ntWI8S-8_ppEojalDfXnTq6FW3kjUgdsyrH0W0,5
|
|
65
65
|
pam/templates/service/common/service.yaml,sha256=ceyFswXjTh07vB4hrJEu7pj3mjlq3NPD4GPYTmvGwmE,44
|
|
66
66
|
pam/templates/service/pandas/functions.tmpl,sha256=ziHgPaK7i8ESY-TxPMuKbqAFILjzdgn1p5FdPf3pYuw,360
|
|
67
|
-
pam/templates/service/pandas/service.test.tmpl,sha256=
|
|
68
|
-
pam/templates/service/pandas/service_class.tmpl,sha256=
|
|
67
|
+
pam/templates/service/pandas/service.test.tmpl,sha256=kK86qW79eFtJCGSqUe0p0dckQlo9DnGzAcTkGB97WKo,3939
|
|
68
|
+
pam/templates/service/pandas/service_class.tmpl,sha256=Pzw2Wcv1FZMDbWAcfakM2V0QhxKjq31mX5bfv5ekeTI,2666
|
|
69
69
|
pam/templates/service/polars/functions.tmpl,sha256=nMKFC5GqIm2Q1dSg2zDOscSwQaOpzS2jzqjQV7xhPRk,382
|
|
70
|
-
pam/templates/service/polars/service.test.tmpl,sha256=
|
|
71
|
-
pam/templates/service/polars/service_class.tmpl,sha256=
|
|
72
|
-
pam_python-0.2.
|
|
73
|
-
pam_python-0.2.
|
|
74
|
-
pam_python-0.2.
|
|
75
|
-
pam_python-0.2.
|
|
76
|
-
pam_python-0.2.
|
|
77
|
-
pam_python-0.2.
|
|
70
|
+
pam/templates/service/polars/service.test.tmpl,sha256=2jduY9a2yILkSzKg9ViPo2_VQsVziXS1KdqFp_U6PUU,3937
|
|
71
|
+
pam/templates/service/polars/service_class.tmpl,sha256=Pzw2Wcv1FZMDbWAcfakM2V0QhxKjq31mX5bfv5ekeTI,2666
|
|
72
|
+
pam_python-0.2.4.dist-info/licenses/LICENSE.txt,sha256=yeHD2-gzrkkET2ZkMe22WyUGpDjlwTrqA4POb0FoZi4,1138
|
|
73
|
+
pam_python-0.2.4.dist-info/METADATA,sha256=7i-8B4rbtCG13UGNa8Y3SNc-v7HwWQManVP9dw5kYjw,9025
|
|
74
|
+
pam_python-0.2.4.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
75
|
+
pam_python-0.2.4.dist-info/entry_points.txt,sha256=xqYpsIqvNlO0j8elZc02bDwR-12dLK9FU2B1ACsLYxw,37
|
|
76
|
+
pam_python-0.2.4.dist-info/top_level.txt,sha256=0EOjbyc3hQyzjhn6iyMgsEseqA66Xz0p27iBN7G7W1w,4
|
|
77
|
+
pam_python-0.2.4.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|