pam-python 0.2.3__py3-none-any.whl → 0.2.4__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pam/__init__.py CHANGED
@@ -1,3 +1,3 @@
1
1
  """PAM Data Plugin framework."""
2
2
 
3
- __version__ = "0.2.3"
3
+ __version__ = "0.2.4"
pam/cli.py CHANGED
@@ -55,7 +55,7 @@ SERVICE_TEMPLATES = {
55
55
  }
56
56
 
57
57
  ENGINE_DEPENDENCIES = {
58
- "polars": ["polars>=1.0.0"],
58
+ "polars": ["polars>=1.43.0"],
59
59
  "pandas": ["pandas>=2.2.3", "pyarrow>=19.0.1"],
60
60
  }
61
61
 
@@ -1,13 +1,29 @@
1
- """Bounded, named eager DataFrame batching for plugin result uploads."""
1
+ """Asynchronous, bounded result batching for PAM CSV uploads."""
2
2
 
3
3
  from __future__ import annotations
4
4
 
5
5
  from dataclasses import dataclass
6
+ import queue
6
7
  import threading
8
+ import time
7
9
  from typing import Any, Dict, Optional, Tuple
8
10
 
11
+ from pam.utils import log
12
+
9
13
 
10
14
  DEFAULT_BATCH_SIZE = 50000
15
+ DEFAULT_QUEUE_SIZE = 4
16
+ DEFAULT_MAX_RETRIES = 3
17
+ DEFAULT_RETRY_DELAY_SECONDS = 2.0
18
+ DEFAULT_RETRY_MAX_DELAY_SECONDS = 30.0
19
+ _FLUSH = object()
20
+
21
+
22
+ @dataclass
23
+ class _ResultInput:
24
+ dataframe: Any
25
+ name: str
26
+ options: Optional[dict]
11
27
 
12
28
 
13
29
  @dataclass
@@ -18,20 +34,56 @@ class _ResultStream:
18
34
  buffer: Any
19
35
  uploaded_rows: int = 0
20
36
  uploaded_batches: int = 0
37
+ failed_rows: int = 0
38
+ failed_batches: int = 0
39
+ retried_uploads: int = 0
21
40
 
22
41
 
23
42
  class ResultBatchUploader:
24
- """Accumulate eager Polars or Pandas frames into bounded upload batches.
43
+ """Queue result frames and upload bounded CSV batches in one worker.
25
44
 
26
- LazyFrames should be passed directly to ``Service._upload_result`` so Polars
27
- can stream the query result to CSV without first collecting every row.
45
+ Plugin computation and intermediate storage are outside this class. Calls to
46
+ ``upload`` submit final result frames without waiting for network I/O. A
47
+ bounded queue applies backpressure when producers outrun the uploader.
48
+
49
+ ``flush`` closes input and places a FIFO completion marker. The worker first
50
+ consumes every earlier result, uploads all complete batches, then uploads all
51
+ remaining rows. ``wait_for_uploader`` is the completion barrier that must
52
+ return before a service exits.
28
53
  """
29
54
 
30
55
  def __init__(self, service, batch_size: int = DEFAULT_BATCH_SIZE):
31
56
  self._service = service
32
- self.batch_size = self._resolve_batch_size(batch_size)
57
+ self.batch_size = self._resolve_positive_int("batch_size", batch_size)
58
+ self.queue_size = self._resolve_positive_int(
59
+ "upload_queue_size", DEFAULT_QUEUE_SIZE
60
+ )
61
+ self.max_retries = self._resolve_non_negative_int(
62
+ "upload_max_retries", DEFAULT_MAX_RETRIES
63
+ )
64
+ self.retry_delay_seconds = self._resolve_non_negative_float(
65
+ "upload_retry_delay_seconds", DEFAULT_RETRY_DELAY_SECONDS
66
+ )
67
+ self.retry_max_delay_seconds = self._resolve_non_negative_float(
68
+ "upload_retry_max_delay_seconds", DEFAULT_RETRY_MAX_DELAY_SECONDS
69
+ )
70
+
33
71
  self._streams: Dict[str, _ResultStream] = {}
34
- self._lock = threading.RLock()
72
+ self._queue: queue.Queue = queue.Queue(maxsize=self.queue_size)
73
+ self._submission_lock = threading.Lock()
74
+ self._state_lock = threading.RLock()
75
+ self._completion = threading.Event()
76
+ self._accepting = True
77
+ self._flush_requested = False
78
+ self._fatal_error: Optional[BaseException] = None
79
+ self._state = "open"
80
+
81
+ self._worker = threading.Thread(
82
+ target=self._run_worker,
83
+ name=f"{self.__class__.__name__}-worker",
84
+ daemon=True,
85
+ )
86
+ self._worker.start()
35
87
 
36
88
  def upload(
37
89
  self,
@@ -39,17 +91,107 @@ class ResultBatchUploader:
39
91
  name: str = "default",
40
92
  options: Optional[dict] = None,
41
93
  ) -> None:
42
- engine, normalized = self._normalize_dataframe(dataframe)
94
+ """Submit a final result without waiting for CSV/network upload."""
43
95
  stream_name = self._normalize_name(name)
44
96
  normalized_options = self._normalize_options(options)
97
+ submitted = self._clone_submitted_dataframe(dataframe)
45
98
 
46
- with self._lock:
47
- stream = self._get_or_create_stream(
48
- stream_name,
49
- engine,
50
- normalized,
51
- normalized_options,
52
- )
99
+ with self._submission_lock:
100
+ with self._state_lock:
101
+ if not self._accepting:
102
+ raise RuntimeError("ResultBatchUploader no longer accepts results")
103
+ if self._fatal_error is not None:
104
+ raise RuntimeError("ResultBatchUploader worker failed") from self._fatal_error
105
+ self._queue.put(_ResultInput(submitted, stream_name, normalized_options))
106
+
107
+ def flush(self) -> None:
108
+ """Close input and enqueue one terminal flush marker without blocking."""
109
+ with self._submission_lock:
110
+ with self._state_lock:
111
+ if self._flush_requested:
112
+ return
113
+ self._accepting = False
114
+ self._flush_requested = True
115
+ self._state = "flush_requested"
116
+ self._queue.put(_FLUSH)
117
+
118
+ def wait_for_uploader(self) -> None:
119
+ """Wait for all queued/retried uploads and the final remainder."""
120
+ with self._state_lock:
121
+ if not self._flush_requested:
122
+ raise RuntimeError("flush() must be called before wait_for_uploader()")
123
+ self._completion.wait()
124
+ with self._state_lock:
125
+ if self._fatal_error is not None:
126
+ raise RuntimeError("ResultBatchUploader worker failed") from self._fatal_error
127
+
128
+ def get_status(self) -> Dict[str, dict]:
129
+ with self._state_lock:
130
+ return {
131
+ name: {
132
+ "buffered_rows": len(stream.buffer),
133
+ "uploaded_rows": stream.uploaded_rows,
134
+ "uploaded_batches": stream.uploaded_batches,
135
+ "failed_rows": stream.failed_rows,
136
+ "failed_batches": stream.failed_batches,
137
+ "retried_uploads": stream.retried_uploads,
138
+ }
139
+ for name, stream in self._streams.items()
140
+ }
141
+
142
+ @property
143
+ def state(self) -> str:
144
+ with self._state_lock:
145
+ return self._state
146
+
147
+ def _run_worker(self) -> None:
148
+ try:
149
+ while True:
150
+ item = self._queue.get()
151
+ try:
152
+ if item is _FLUSH:
153
+ if self._fatal_error is None:
154
+ self._flush_all_streams()
155
+ with self._state_lock:
156
+ self._state = (
157
+ "failed" if self._fatal_error is not None else "completed"
158
+ )
159
+ return
160
+ if self._fatal_error is None:
161
+ self._consume_input(item)
162
+ except BaseException as exc: # keep worker alive until the flush barrier
163
+ self._record_fatal_error(exc)
164
+ finally:
165
+ self._queue.task_done()
166
+ finally:
167
+ with self._state_lock:
168
+ self._accepting = False
169
+ if self._state not in {"completed", "failed"}:
170
+ self._state = "failed"
171
+ self._completion.set()
172
+
173
+ def _consume_input(self, item: _ResultInput) -> None:
174
+ dataframe = item.dataframe
175
+ if self._is_polars_lazyframe(dataframe):
176
+ for batch in dataframe.collect_batches(
177
+ chunk_size=self.batch_size,
178
+ maintain_order=True,
179
+ engine="streaming",
180
+ ):
181
+ self._consume_eager(batch, item.name, item.options)
182
+ return
183
+ self._consume_eager(dataframe, item.name, item.options)
184
+
185
+ def _consume_eager(
186
+ self,
187
+ dataframe: Any,
188
+ name: str,
189
+ options: Optional[dict],
190
+ ) -> None:
191
+ engine, normalized = self._normalize_eager_dataframe(dataframe)
192
+ batches = []
193
+ with self._state_lock:
194
+ stream = self._get_or_create_stream(name, engine, normalized, options)
53
195
  if len(normalized) == 0:
54
196
  return
55
197
 
@@ -68,12 +210,11 @@ class ResultBatchUploader:
68
210
  len(remaining) - rows_needed,
69
211
  )
70
212
  if len(stream.buffer) == self.batch_size:
71
- self._upload_batch(stream, stream.buffer)
213
+ batches.append(self._clone(engine, stream.buffer))
72
214
  stream.buffer = self._empty_frame(engine, stream.buffer)
73
215
 
74
216
  while len(remaining) >= self.batch_size:
75
- batch = self._slice(engine, remaining, 0, self.batch_size)
76
- self._upload_batch(stream, batch)
217
+ batches.append(self._slice(engine, remaining, 0, self.batch_size))
77
218
  remaining = self._slice(
78
219
  engine,
79
220
  remaining,
@@ -84,30 +225,64 @@ class ResultBatchUploader:
84
225
  if len(remaining) > 0:
85
226
  stream.buffer = self._clone(engine, remaining)
86
227
 
87
- def flush(self, name: Optional[str] = None) -> None:
88
- with self._lock:
89
- stream_names = (
90
- list(self._streams)
91
- if name is None
92
- else [self._normalize_name(name)]
93
- )
94
- for stream_name in stream_names:
95
- stream = self._streams.get(stream_name)
96
- if stream is None or len(stream.buffer) == 0:
228
+ for batch in batches:
229
+ self._upload_batch(name, stream, batch)
230
+
231
+ def _flush_all_streams(self) -> None:
232
+ pending = []
233
+ with self._state_lock:
234
+ for name, stream in self._streams.items():
235
+ if len(stream.buffer) == 0:
97
236
  continue
98
- self._upload_batch(stream, stream.buffer)
237
+ pending.append((name, stream, self._clone(stream.engine, stream.buffer)))
99
238
  stream.buffer = self._empty_frame(stream.engine, stream.buffer)
239
+ for name, stream, batch in pending:
240
+ self._upload_batch(name, stream, batch)
100
241
 
101
- def get_status(self) -> Dict[str, dict]:
102
- with self._lock:
103
- return {
104
- name: {
105
- "buffered_rows": len(stream.buffer),
106
- "uploaded_rows": stream.uploaded_rows,
107
- "uploaded_batches": stream.uploaded_batches,
108
- }
109
- for name, stream in self._streams.items()
110
- }
242
+ def _upload_batch(self, name: str, stream: _ResultStream, dataframe: Any) -> None:
243
+ attempts = self.max_retries + 1
244
+ for attempt in range(1, attempts + 1):
245
+ try:
246
+ self._service._upload_result(
247
+ self._clone(stream.engine, dataframe),
248
+ stream.options,
249
+ )
250
+ with self._state_lock:
251
+ stream.uploaded_rows += len(dataframe)
252
+ stream.uploaded_batches += 1
253
+ return
254
+ except Exception as exc: # network/upload failures are retryable
255
+ if attempt < attempts:
256
+ with self._state_lock:
257
+ stream.retried_uploads += 1
258
+ delay = min(
259
+ self.retry_delay_seconds * (2 ** (attempt - 1)),
260
+ self.retry_max_delay_seconds,
261
+ )
262
+ log(
263
+ f"Result upload retry stream={name} rows={len(dataframe)} "
264
+ f"attempt={attempt + 1}/{attempts} delay={delay}s error={exc}",
265
+ level="WARNING",
266
+ )
267
+ if delay > 0:
268
+ time.sleep(delay)
269
+ continue
270
+
271
+ with self._state_lock:
272
+ stream.failed_rows += len(dataframe)
273
+ stream.failed_batches += 1
274
+ log(
275
+ f"Result upload skipped after retries stream={name} "
276
+ f"rows={len(dataframe)} attempts={attempts} error={exc}",
277
+ level="ERROR",
278
+ )
279
+
280
+ def _record_fatal_error(self, exc: BaseException) -> None:
281
+ with self._state_lock:
282
+ if self._fatal_error is None:
283
+ self._fatal_error = exc
284
+ self._accepting = False
285
+ log(f"Result uploader worker failed: {exc}", level="ERROR")
111
286
 
112
287
  def _get_or_create_stream(
113
288
  self,
@@ -138,46 +313,56 @@ class ResultBatchUploader:
138
313
  raise ValueError(f"Result stream '{name}' upload options changed")
139
314
  return stream
140
315
 
141
- def _upload_batch(self, stream: _ResultStream, dataframe: Any) -> None:
142
- self._service._upload_result(
143
- self._clone(stream.engine, dataframe),
144
- stream.options,
145
- )
146
- stream.uploaded_rows += len(dataframe)
147
- stream.uploaded_batches += 1
148
-
149
- def _resolve_batch_size(self, configured_default: int) -> int:
150
- fallback = (
151
- configured_default
152
- if isinstance(configured_default, int)
153
- and not isinstance(configured_default, bool)
154
- and configured_default > 0
155
- else DEFAULT_BATCH_SIZE
156
- )
157
- raw_value = getattr(self._service.request, "runtime_parameters", {}).get(
158
- "batch_size"
159
- )
160
- if raw_value is None:
316
+ def _runtime_parameters(self) -> dict:
317
+ return getattr(self._service.request, "runtime_parameters", {})
318
+
319
+ def _resolve_positive_int(self, key: str, fallback: int) -> int:
320
+ try:
321
+ value = int(self._runtime_parameters().get(key, fallback))
322
+ except (TypeError, ValueError):
161
323
  return fallback
324
+ return value if value > 0 else fallback
325
+
326
+ def _resolve_non_negative_int(self, key: str, fallback: int) -> int:
162
327
  try:
163
- parsed = int(raw_value)
328
+ value = int(self._runtime_parameters().get(key, fallback))
164
329
  except (TypeError, ValueError):
165
330
  return fallback
166
- return parsed if parsed > 0 else fallback
331
+ return value if value >= 0 else fallback
332
+
333
+ def _resolve_non_negative_float(self, key: str, fallback: float) -> float:
334
+ try:
335
+ value = float(self._runtime_parameters().get(key, fallback))
336
+ except (TypeError, ValueError):
337
+ return fallback
338
+ return value if value >= 0 else fallback
339
+
340
+ @classmethod
341
+ def _clone_submitted_dataframe(cls, dataframe: Any):
342
+ if cls._is_polars_lazyframe(dataframe):
343
+ return dataframe.clone()
344
+ _, normalized = cls._normalize_eager_dataframe(dataframe)
345
+ return normalized
167
346
 
168
347
  @staticmethod
169
- def _normalize_dataframe(dataframe: Any) -> tuple[str, Any]:
348
+ def _normalize_eager_dataframe(dataframe: Any) -> tuple[str, Any]:
170
349
  module_name = type(dataframe).__module__.split(".", 1)[0]
171
350
  class_name = type(dataframe).__name__
172
- if module_name == "polars" and class_name == "LazyFrame":
173
- raise TypeError(
174
- "Polars LazyFrame must be uploaded directly with Service._upload_result"
175
- )
176
351
  if module_name == "polars" and class_name == "DataFrame":
177
352
  return "polars", dataframe.clone()
178
353
  if module_name == "pandas" and class_name == "DataFrame":
179
354
  return "pandas", dataframe.reset_index(drop=True).copy()
180
- raise TypeError("dataframe must be a Polars or Pandas DataFrame")
355
+ raise TypeError(
356
+ "dataframe must be a Polars LazyFrame, Polars DataFrame, "
357
+ "or Pandas DataFrame"
358
+ )
359
+
360
+ @staticmethod
361
+ def _is_polars_lazyframe(dataframe: Any) -> bool:
362
+ return (
363
+ type(dataframe).__module__.split(".", 1)[0] == "polars"
364
+ and type(dataframe).__name__ == "LazyFrame"
365
+ )
181
366
 
182
367
  @staticmethod
183
368
  def _clone(engine: str, dataframe: Any):
pam/task_manager.py CHANGED
@@ -305,7 +305,10 @@ class TaskManager(ITaskManager):
305
305
 
306
306
  def service_upload_result(self, service: Service, file_path, options=None):
307
307
  """
308
- Uploads a result file asynchronously and logs the response.
308
+ Uploads one result file synchronously.
309
+
310
+ ResultBatchUploader owns background concurrency and invokes this method
311
+ from its worker, so failures must propagate for retry handling.
309
312
  """
310
313
  endpoint = service.request.response_api
311
314
  payload = None
@@ -318,25 +321,21 @@ class TaskManager(ITaskManager):
318
321
  is_priority = options.get("is_priority")
319
322
  payload["is_priority"] = str(is_priority).lower() if isinstance(is_priority, bool) else is_priority
320
323
 
321
- def handle_upload_response(response):
322
- """Logs the response after the upload completes."""
323
- if response is None:
324
- log(f"Response from upload to {endpoint}: None")
325
- return
326
- try:
327
- response_data = response.json()
328
- except ValueError:
329
- response_data = response.text
330
- log(f"Response from upload to {endpoint}: {response_data}")
331
-
332
- def upload_wrapper():
333
- """Wrapper for the upload to handle response logging."""
334
- response = self.api.http_upload(endpoint, file_path, payload)
335
- handle_upload_response(response)
336
-
337
324
  log(f"Uploading Result to: {endpoint}")
338
- http_thread = threading.Thread(target=upload_wrapper)
339
- http_thread.start()
325
+ response = self.api.http_upload(endpoint, file_path, payload)
326
+ if response is None:
327
+ raise RuntimeError(f"No response from result upload to {endpoint}")
328
+ if not response.ok:
329
+ raise RuntimeError(
330
+ f"Result upload failed status={response.status_code} "
331
+ f"endpoint={endpoint}"
332
+ )
333
+ try:
334
+ response_data = response.json()
335
+ except ValueError:
336
+ response_data = response.text
337
+ log(f"Response from upload to {endpoint}: {response_data}")
338
+ return response
340
339
 
341
340
  def service_upload_report(self, service: Service, file_path):
342
341
  """
@@ -150,6 +150,11 @@ When CDP finishes collecting data, it calls `on_data_input`.
150
150
  Example:
151
151
 
152
152
  ```python
153
+ def __init__(self, task_manager, req):
154
+ super().__init__(task_manager, req)
155
+ self.batch_uploader = ResultBatchUploader(self, batch_size=50000)
156
+
157
+
153
158
  def on_data_input(self, req: RequestCommand):
154
159
  log(f"on_data_input req.is_end = {req.is_end}")
155
160
 
@@ -173,7 +178,7 @@ def __run_process_data_in_thread(self, req: RequestCommand):
173
178
  pl.scan_parquet(purchase_event_file),
174
179
  pl.scan_parquet(point_received_event_file),
175
180
  )
176
- self._upload_result(result)
181
+ self.batch_uploader.upload(result, name="main")
177
182
 
178
183
  if not req.is_end:
179
184
  self._request_data(
@@ -181,6 +186,8 @@ def __run_process_data_in_thread(self, req: RequestCommand):
181
186
  file_format=RequestFileFormat.PARQUET,
182
187
  )
183
188
  else:
189
+ self.batch_uploader.flush()
190
+ self.batch_uploader.wait_for_uploader()
184
191
  self._exit()
185
192
  ```
186
193
 
@@ -194,7 +201,9 @@ validate file extensions as an additional consistency check.
194
201
 
195
202
  ## Upload Results to CDP
196
203
 
197
- After processing, call `_upload_result` with a dataframe. Example output:
204
+ After processing, submit final dataframe results through `ResultBatchUploader`.
205
+ The uploader creates bounded CSV files and sends them through the PAM API.
206
+ Example output:
198
207
 
199
208
  ```csv
200
209
  id,data_x,rfm
@@ -391,6 +400,11 @@ Report types outside that closed contract require a newer framework and CMS rele
391
400
  When uploading large DataFrames, always upload in batches to avoid memory/network spikes.
392
401
  Batch size MUST be configurable via `request.runtime_parameters["batch_size"]`.
393
402
 
403
+ Computation is a developer-owned black box. The service may calculate within one
404
+ page, retain state across pages, or use any intermediate store such as SQLite or
405
+ DuckDB. The framework does not choose that strategy. Call the uploader only when
406
+ the produced rows are ready for the PAM result contract.
407
+
394
408
  ## Required behavior
395
409
 
396
410
  - Read batch size from `self.request.runtime_parameters.get("batch_size", "50000")`
@@ -402,10 +416,17 @@ Batch size MUST be configurable via `request.runtime_parameters["batch_size"]`.
402
416
 
403
417
  ## Reference implementation pattern
404
418
 
405
- For a Polars `LazyFrame`, call `_upload_result(lazy_frame)` directly so the runtime
406
- can stream it to CSV. For eager Polars/Pandas frames, `ResultBatchUploader` reads
407
- `runtime_parameters["batch_size"]`, keeps named result streams separate, uploads
408
- complete batches, and flushes remaining rows when successful processing finishes.
419
+ `ResultBatchUploader` accepts Polars `LazyFrame`, Polars `DataFrame`, and Pandas
420
+ `DataFrame`. It evaluates lazy results as bounded streaming chunks, reads
421
+ `runtime_parameters["batch_size"]`, keeps named result streams separate, converts
422
+ complete batches to CSV, and uploads through the PAM API. Incomplete batches may
423
+ remain buffered across pages or calls until more final rows arrive or `flush()` is
424
+ called.
425
+
426
+ `upload()` queues work for one background uploader and normally returns without
427
+ waiting for network I/O, allowing the service to request and process the next PAM
428
+ page. The queue is bounded, so `upload()` may apply backpressure if uploads are
429
+ slower than result production.
409
430
 
410
431
  ```python
411
432
  from pam.result_batch_uploader import ResultBatchUploader
@@ -413,13 +434,36 @@ from pam.result_batch_uploader import ResultBatchUploader
413
434
  self.batch_uploader = ResultBatchUploader(self, batch_size=50000)
414
435
  self.batch_uploader.upload(result_df, name="main")
415
436
  self.batch_uploader.flush()
437
+ self.batch_uploader.wait_for_uploader()
438
+ self._exit()
416
439
  ```
417
440
 
441
+ `flush()` is both a terminal state change and a FIFO marker. It stops new result
442
+ submissions but does not wait. The worker processes every result submitted before
443
+ the marker, preserves `batch_size`, uploads the final remainder, and then completes.
444
+ `wait_for_uploader()` blocks until queued and in-flight network uploads finish.
445
+ It intentionally has no timeout: timeout each HTTP upload attempt instead, then
446
+ retry within the configured limit or skip the permanently failed batch. A timeout
447
+ on the completion barrier could let the plugin exit while uploads are still active.
448
+
449
+ Upload retry settings are optional runtime parameters:
450
+
451
+ - `upload_queue_size` (default `4`): maximum queued result objects before backpressure
452
+ - `upload_max_retries` (default `3`): retries after the first upload attempt
453
+ - `upload_retry_delay_seconds` (default `2`): initial exponential-backoff delay
454
+ - `upload_retry_max_delay_seconds` (default `30`): backoff delay cap
455
+
456
+ After retries are exhausted, the uploader logs the failed batch, records
457
+ `failed_rows`/`failed_batches`, skips it, and continues with the next batch.
458
+ `wait_for_uploader()` still waits for the whole queue before returning.
459
+
418
460
  ## Notes
419
461
 
420
462
  - Do NOT hardcode a fixed batch size inside upload loops.
421
463
  - Do NOT read environment variables for batch sizing unless explicitly required; use `runtime_parameters` only.
422
- - Do NOT pass a Polars `LazyFrame` to `ResultBatchUploader`.
464
+ - Do NOT call `_upload_result(result)` directly from plugin code; use `ResultBatchUploader`.
465
+ - Call `flush()` only after all intended final rows have been submitted.
466
+ - Always call `wait_for_uploader()` after `flush()` and before `_exit()`.
423
467
 
424
468
  ---
425
469
 
@@ -442,7 +486,7 @@ Prefer streaming/chunked/incremental processing whenever possible.
442
486
 
443
487
  - Start with `pl.scan_parquet(...)`, not `pl.read_parquet(...)`.
444
488
  - Select required columns and filter rows early for projection/predicate pushdown.
445
- - Keep transformations as a `LazyFrame` through `_upload_result(...)`.
489
+ - Keep transformations as a `LazyFrame` through `ResultBatchUploader.upload(...)`.
446
490
 
447
491
  ### 2. DB-first computation
448
492
 
@@ -18,6 +18,7 @@ from #MODULE_NAME#.#CLASS_NAME# import #CLASS_NAME#
18
18
  TOTAL_PAGES = 3
19
19
  MOCK_DIR = "./TEMP_TEST_DATA"
20
20
  UPLOADED_RESULTS = []
21
+ UPLOADED_ROW_COUNTS = []
21
22
  REQUESTED_FORMATS = []
22
23
  TempfileUtils.temp_base_path = f"{MOCK_DIR}/app/data"
23
24
  TempfileUtils.temp_datasource_path = f"{MOCK_DIR}/app/data/data_sources"
@@ -68,6 +69,7 @@ def on_publish_report_pointer(report_key, pointer):
68
69
 
69
70
  def on_upload_result(result_file):
70
71
  UPLOADED_RESULTS.append(result_file)
72
+ UPLOADED_ROW_COUNTS.append(len(pd.read_csv(result_file)))
71
73
 
72
74
 
73
75
  class Test#CLASS_NAME#(unittest.TestCase):
@@ -77,6 +79,7 @@ class Test#CLASS_NAME#(unittest.TestCase):
77
79
  os.makedirs(MOCK_DIR)
78
80
  create_mock_parquet(TOTAL_PAGES, MOCK_DIR)
79
81
  UPLOADED_RESULTS.clear()
82
+ UPLOADED_ROW_COUNTS.clear()
80
83
  REQUESTED_FORMATS.clear()
81
84
 
82
85
  self.tester_task = TesterTask()
@@ -112,6 +115,7 @@ class Test#CLASS_NAME#(unittest.TestCase):
112
115
 
113
116
  self.assertTrue(self.tester_task.is_exit)
114
117
  self.assertEqual(2, len(UPLOADED_RESULTS))
118
+ self.assertEqual([15, 15], UPLOADED_ROW_COUNTS)
115
119
  self.assertEqual(
116
120
  [RequestFileFormat.PARQUET] * TOTAL_PAGES,
117
121
  REQUESTED_FORMATS,
@@ -65,6 +65,7 @@ class #CLASS_NAME#(Service):
65
65
  )
66
66
  else:
67
67
  self.batch_uploader.flush()
68
+ self.batch_uploader.wait_for_uploader()
68
69
  self._exit()
69
70
 
70
71
  def on_destroy(self):
@@ -18,6 +18,7 @@ from #MODULE_NAME#.#CLASS_NAME# import #CLASS_NAME#
18
18
  TOTAL_PAGES = 3
19
19
  MOCK_DIR = "./TEMP_TEST_DATA"
20
20
  UPLOADED_RESULTS = []
21
+ UPLOADED_ROW_COUNTS = []
21
22
  REQUESTED_FORMATS = []
22
23
  TempfileUtils.temp_base_path = f"{MOCK_DIR}/app/data"
23
24
  TempfileUtils.temp_datasource_path = f"{MOCK_DIR}/app/data/data_sources"
@@ -68,6 +69,7 @@ def on_publish_report_pointer(report_key, pointer):
68
69
 
69
70
  def on_upload_result(result_file):
70
71
  UPLOADED_RESULTS.append(result_file)
72
+ UPLOADED_ROW_COUNTS.append(pl.read_csv(result_file).height)
71
73
 
72
74
 
73
75
  class Test#CLASS_NAME#(unittest.TestCase):
@@ -77,6 +79,7 @@ class Test#CLASS_NAME#(unittest.TestCase):
77
79
  os.makedirs(MOCK_DIR)
78
80
  create_mock_parquet(TOTAL_PAGES, MOCK_DIR)
79
81
  UPLOADED_RESULTS.clear()
82
+ UPLOADED_ROW_COUNTS.clear()
80
83
  REQUESTED_FORMATS.clear()
81
84
 
82
85
  self.tester_task = TesterTask()
@@ -96,7 +99,7 @@ class Test#CLASS_NAME#(unittest.TestCase):
96
99
  sqlite_download="",
97
100
  sqlite_download_v2="",
98
101
  sqlite_upload="",
99
- runtime_parameters={"some_param": "some_value"},
102
+ runtime_parameters={"some_param": "some_value", "batch_size": "15"},
100
103
  token=uuid.uuid4().hex,
101
104
  cmd="start",
102
105
  data_api="",
@@ -111,7 +114,8 @@ class Test#CLASS_NAME#(unittest.TestCase):
111
114
  self.tester_task.wait_for_task_done()
112
115
 
113
116
  self.assertTrue(self.tester_task.is_exit)
114
- self.assertEqual(TOTAL_PAGES, len(UPLOADED_RESULTS))
117
+ self.assertEqual(2, len(UPLOADED_RESULTS))
118
+ self.assertEqual([15, 15], UPLOADED_ROW_COUNTS)
115
119
  self.assertEqual(
116
120
  [RequestFileFormat.PARQUET] * TOTAL_PAGES,
117
121
  REQUESTED_FORMATS,
@@ -3,6 +3,7 @@ import threading
3
3
 
4
4
  from pam.models.request_command import RequestCommand
5
5
  from pam.request_file_format import RequestFileFormat
6
+ from pam.result_batch_uploader import ResultBatchUploader
6
7
  from pam.service import Service
7
8
  from pam.utils import log
8
9
 
@@ -14,6 +15,10 @@ class #CLASS_NAME#(Service):
14
15
  INPUT_FILE_FORMAT = RequestFileFormat.PARQUET
15
16
  EXPECTED_DATASET_COUNT = 1
16
17
 
18
+ def __init__(self, task_manager, req):
19
+ super().__init__(task_manager, req)
20
+ self.batch_uploader = ResultBatchUploader(self, batch_size=50000)
21
+
17
22
  def on_start(self):
18
23
  log("on_start")
19
24
  some_param = self.request.runtime_parameters.get("some_param", "")
@@ -51,7 +56,7 @@ class #CLASS_NAME#(Service):
51
56
  )
52
57
 
53
58
  result = transform(load_input_file(input_dataset_file))
54
- self._upload_result(result)
59
+ self.batch_uploader.upload(result, name="main")
55
60
 
56
61
  if not req.is_end:
57
62
  self._request_data(
@@ -59,6 +64,8 @@ class #CLASS_NAME#(Service):
59
64
  file_format=self.INPUT_FILE_FORMAT,
60
65
  )
61
66
  else:
67
+ self.batch_uploader.flush()
68
+ self.batch_uploader.wait_for_uploader()
62
69
  self._exit()
63
70
 
64
71
  def on_destroy(self):
@@ -68,4 +75,4 @@ class #CLASS_NAME#(Service):
68
75
  log("on_terminate")
69
76
 
70
77
  def get_status(self):
71
- return json.dumps({"engine": "polars"})
78
+ return json.dumps(self.batch_uploader.get_status(), sort_keys=True)
pam/tester_task.py CHANGED
@@ -207,12 +207,10 @@ class TesterTask(ITaskManager):
207
207
  :param options: Optional upload options (e.g. is_realtime, is_priority).
208
208
  """
209
209
  if self.upload_result_callback is not None:
210
- try:
211
- self.upload_result_callback(file_path)
212
- except Exception as e:
213
- print(f"Error in upload_result_callback: {e}")
210
+ return self.upload_result_callback(file_path)
214
211
  else:
215
212
  print("Upload result callback is not set.")
213
+ return None
216
214
 
217
215
  def service_upload_report(self, service: Service, file_path: str):
218
216
  """
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pam-python
3
- Version: 0.2.3
3
+ Version: 0.2.4
4
4
  Summary: Pam Python Library
5
5
  Author-email: Narongrit Kanhanoi <narongrit@pams.ai>
6
6
  Project-URL: Homepage, https://github.com/heart/pam-python
@@ -128,7 +128,7 @@ The runtime calls your service in two main phases.
128
128
 
129
129
  When your service is done:
130
130
 
131
- - Upload ordinary CDP result rows through `ResultBatchUploader` or `_upload_result(...)`
131
+ - Upload ordinary CDP result rows through `ResultBatchUploader`
132
132
  - Publish managed reports through `self.reports`; do not build report JSON manually
133
133
  - Call `self._exit()` to signal completion
134
134
 
@@ -151,11 +151,11 @@ Notes:
151
151
  ---
152
152
 
153
153
  **Uploading Results in Batches**
154
- Polars templates pass `LazyFrame` results directly to `_upload_result(...)`, which
155
- streams them to the CSV upload boundary. Do not collect or convert to Pandas first.
156
-
157
- For eager Polars or Pandas DataFrames, use the batch uploader to handle chunking
158
- and flushing automatically.
154
+ Polars `LazyFrame`/`DataFrame` and Pandas `DataFrame` results all pass through
155
+ `ResultBatchUploader`. It streams lazy results in bounded chunks, buffers incomplete
156
+ batches across calls, writes complete batches as CSV, and uploads them through PAM.
157
+ The framework does not prescribe page-local/global computation or intermediate
158
+ storage; submit rows when your business result is ready.
159
159
 
160
160
  Recommended usage:
161
161
 
@@ -165,6 +165,7 @@ from pam.result_batch_uploader import ResultBatchUploader
165
165
  batch_uploader = ResultBatchUploader(self, batch_size=50000)
166
166
  batch_uploader.upload(df, name="main")
167
167
  batch_uploader.flush()
168
+ batch_uploader.wait_for_uploader()
168
169
  status = batch_uploader.get_status()
169
170
  ```
170
171
 
@@ -173,9 +174,12 @@ Notes:
173
174
  - `request.runtime_parameters["batch_size"]` overrides the constructor default when valid.
174
175
  - `name` separates result streams and enforces one stable column schema per stream.
175
176
  - `options` may be passed to `upload(...)` and are forwarded to `_upload_result`.
176
- - Complete batches upload immediately; `flush()` sends the remaining rows.
177
- - `get_status()` returns buffered rows, uploaded rows, and uploaded batch counts per stream.
178
- - `ResultBatchUploader` intentionally rejects Polars `LazyFrame`; upload it directly.
177
+ - Uploads run in a bounded background queue so processing can continue while PAM receives CSV batches.
178
+ - `flush()` closes input and queues the final remainder; it does not wait for network completion.
179
+ - Always call `wait_for_uploader()` after `flush()` and before `_exit()`.
180
+ - Failed uploads retry with exponential backoff, then log/record the failed batch and continue.
181
+ - `get_status()` includes buffered, uploaded, failed, and retried counts per stream.
182
+ - Polars `LazyFrame` is supported without collecting the complete result at once.
179
183
 
180
184
  ---
181
185
 
@@ -304,6 +308,5 @@ After `pam init` and one service:
304
308
  - Implement your logic in `functions.py`.
305
309
  - Wire it into `on_start` and `on_data_input` in your service class.
306
310
  - Use the temp utilities to write intermediate files.
307
- - Keep Polars transformations lazy and upload the resulting `LazyFrame` directly.
308
- - Use `ResultBatchUploader` only for eager Polars/Pandas frames that need row batching.
311
+ - Keep Polars transformations lazy and submit final results through `ResultBatchUploader`.
309
312
  - Read `REPORTS.md` before implementing a managed report.
@@ -1,16 +1,16 @@
1
- pam/__init__.py,sha256=_my-Ddw7odQ1s6OaAir2Dk7rSmUS9FIHUgcNaBBy6mw,56
1
+ pam/__init__.py,sha256=cSLmJntln2OVNNJ07_9Z2eKFLloAZc33rsRlnRTj0wk,56
2
2
  pam/api.py,sha256=4t64ey_Jc4gC3SvfI-mtedEGzkX2cPTt0cdZgP_huLk,7231
3
- pam/cli.py,sha256=GbTyHD2Gc2WLTY54bZ9rtofjVc7bV2JjqXThR9JCQf4,14130
3
+ pam/cli.py,sha256=Kbl5K4xS5KtW1cU3QRZOi8eAdKzUaJqe2JDGuhy4wQA,14131
4
4
  pam/interface_task_manager.py,sha256=moKUjSpAeNF9C2dluDFPAdU4IyvwXckB0RK-jINTLm4,1665
5
5
  pam/logger.py,sha256=JYPfIpaQZy6QrfVKrzEMrxOgTzWIEhoegvIZgHTGWoc,3021
6
6
  pam/request_file_format.py,sha256=fSRlJagbvVGqdIEqhINSeqBNUTNQdm8GL5JfqgX9bxU,317
7
- pam/result_batch_uploader.py,sha256=RVnaE-Itfacx1yoBJfhWwqaXk7ytEKyvLM3ZNlj6Pug,7997
7
+ pam/result_batch_uploader.py,sha256=xw4b-q7i52CoCjvQtQ8EgfBrrqAl1zEl6sTO-wIvFPc,15424
8
8
  pam/server.py,sha256=ZRukc3uLJXxJdWIQDcPqrxjd_LnsGTxPJQtLX0bMcRc,7594
9
9
  pam/service.py,sha256=EymzqpHwjmCpSC3e3QchdnrE7L39_hxtDVB3kCJkfr4,7760
10
10
  pam/sqlite.py,sha256=0bEUnb17V_Z6N6u843MYyGQnigaAuMx6FUckeWifn8I,2585
11
- pam/task_manager.py,sha256=tgxf0pkK8zn-BcPNHvzkgHAZ4oNGuhoCyqL4Q69R9dY,15839
11
+ pam/task_manager.py,sha256=DC2pZ1iJU-GnCl8mawuC6Ym105ywoZ6rs6neXUexHZ4,15797
12
12
  pam/temp_file_utils.py,sha256=lVkRJuNMdmqbl22BeP4tGo2YAdYy4C_SMzcR4vSQOb4,6762
13
- pam/tester_task.py,sha256=RhYCh9ia_-bkYdHjRkZmFyJectmW8wcOr1EjPHWnJ9o,9305
13
+ pam/tester_task.py,sha256=s6BelOUMOahUh_5-lRN5yUEO72caR-ERQS8b3ZpfQgw,9213
14
14
  pam/utils.py,sha256=-pfnQ00Ib-IXvrMz4S5lTlO36PtSaJoQOhwXpwet8h4,1744
15
15
  pam/models/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
16
16
  pam/models/request_command.py,sha256=tGzPuP1TQI0CrNjiigW929lJ99Fvz-WfcUh02eOVFK4,6621
@@ -48,7 +48,7 @@ pam/reports/transport.py,sha256=t3oEVptS9msnNafia1G8vmTUCdhxvV5L2-0AY-8SKFk,2062
48
48
  pam/reports/treemap.py,sha256=W_VwxbiRMSvkNpY2z1HfbZtGKWXNr6FdbVdI0grVesw,4251
49
49
  pam/templates/buildcmd/pamb,sha256=58GCN59ysvHGNhPvTD3nkF4aLg-kM1WfWmH7CL3eQVo,530
50
50
  pam/templates/buildcmd/pamb-base.sh,sha256=Q0DkbeHtPx33PIjc68EeI7ZIwBOLlwcSmPhpxALXxVY,518
51
- pam/templates/init/AGENT.md,sha256=PWZ-qp1bfyU5D0H5hnFLR6LHnv4efNLIglAGhsAfb1A,16105
51
+ pam/templates/init/AGENT.md,sha256=uEvHox6K43Em2vWQ8H_5a2NswLUdlSmfve3cWEwLDes,18524
52
52
  pam/templates/init/REPORTS.md,sha256=rg9oi0_xDYr0BZBluJh0vPF8lSx7mCRFpDCpUgNl68A,42851
53
53
  pam/templates/init/dockerignore.tmpl,sha256=Ht1eCgQLqalEwpUjih0FQavzhYJbT2jH-FP4btM3960,52
54
54
  pam/templates/init/gitignore.tmpl,sha256=U59ynPRu_Kw6evqQzOOTAhWKdIXbf1uBOF2V8223tU8,3510
@@ -64,14 +64,14 @@ pam/templates/project/uv/pyproject.toml,sha256=30a2hJn6oN1AGvmgEBdxLETOBlHY8L-II
64
64
  pam/templates/project/uv/python-version,sha256=e1X45ntWI8S-8_ppEojalDfXnTq6FW3kjUgdsyrH0W0,5
65
65
  pam/templates/service/common/service.yaml,sha256=ceyFswXjTh07vB4hrJEu7pj3mjlq3NPD4GPYTmvGwmE,44
66
66
  pam/templates/service/pandas/functions.tmpl,sha256=ziHgPaK7i8ESY-TxPMuKbqAFILjzdgn1p5FdPf3pYuw,360
67
- pam/templates/service/pandas/service.test.tmpl,sha256=PybHu_c5BMhUmvkc9ycQH7vJbPWt7isJAH2r-ElQ9dE,3760
68
- pam/templates/service/pandas/service_class.tmpl,sha256=629kX2qJAmNzj5MrGdg6FLqgijEML8D3Ddo7k27LzbM,2614
67
+ pam/templates/service/pandas/service.test.tmpl,sha256=kK86qW79eFtJCGSqUe0p0dckQlo9DnGzAcTkGB97WKo,3939
68
+ pam/templates/service/pandas/service_class.tmpl,sha256=Pzw2Wcv1FZMDbWAcfakM2V0QhxKjq31mX5bfv5ekeTI,2666
69
69
  pam/templates/service/polars/functions.tmpl,sha256=nMKFC5GqIm2Q1dSg2zDOscSwQaOpzS2jzqjQV7xhPRk,382
70
- pam/templates/service/polars/service.test.tmpl,sha256=58x_MJYwp7BvDtuSsETAt9Q6v5KwOI8MaEdjNciAQDk,3746
71
- pam/templates/service/polars/service_class.tmpl,sha256=tWfEv6dBeL0wm41VfL2YI5JA6EUF08rH_nnsmk25A84,2306
72
- pam_python-0.2.3.dist-info/licenses/LICENSE.txt,sha256=yeHD2-gzrkkET2ZkMe22WyUGpDjlwTrqA4POb0FoZi4,1138
73
- pam_python-0.2.3.dist-info/METADATA,sha256=Ce02r8gxDTde0fNWk6WGCIwxBz6538-pJBTcy8HR0Do,8703
74
- pam_python-0.2.3.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
75
- pam_python-0.2.3.dist-info/entry_points.txt,sha256=xqYpsIqvNlO0j8elZc02bDwR-12dLK9FU2B1ACsLYxw,37
76
- pam_python-0.2.3.dist-info/top_level.txt,sha256=0EOjbyc3hQyzjhn6iyMgsEseqA66Xz0p27iBN7G7W1w,4
77
- pam_python-0.2.3.dist-info/RECORD,,
70
+ pam/templates/service/polars/service.test.tmpl,sha256=2jduY9a2yILkSzKg9ViPo2_VQsVziXS1KdqFp_U6PUU,3937
71
+ pam/templates/service/polars/service_class.tmpl,sha256=Pzw2Wcv1FZMDbWAcfakM2V0QhxKjq31mX5bfv5ekeTI,2666
72
+ pam_python-0.2.4.dist-info/licenses/LICENSE.txt,sha256=yeHD2-gzrkkET2ZkMe22WyUGpDjlwTrqA4POb0FoZi4,1138
73
+ pam_python-0.2.4.dist-info/METADATA,sha256=7i-8B4rbtCG13UGNa8Y3SNc-v7HwWQManVP9dw5kYjw,9025
74
+ pam_python-0.2.4.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
75
+ pam_python-0.2.4.dist-info/entry_points.txt,sha256=xqYpsIqvNlO0j8elZc02bDwR-12dLK9FU2B1ACsLYxw,37
76
+ pam_python-0.2.4.dist-info/top_level.txt,sha256=0EOjbyc3hQyzjhn6iyMgsEseqA66Xz0p27iBN7G7W1w,4
77
+ pam_python-0.2.4.dist-info/RECORD,,