rapidsegment 1.3.post3__tar.gz → 1.3.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/PKG-INFO +33 -26
  2. {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/README.md +32 -25
  3. {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/pyproject.toml +1 -1
  4. {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/pyproject.toml.orig +1 -1
  5. {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/__init__.py +2 -2
  6. {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/scorer.py +24 -7
  7. {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/pages/1_Data_Loader.py +0 -0
  8. rapidsegment-1.3.1/src/rapidsegment/utils/__init__.py +3 -0
  9. rapidsegment-1.3.1/src/rapidsegment/utils/data_loader.py +538 -0
  10. rapidsegment-1.3.post3/src/rapidsegment/utils/__init__.py +0 -3
  11. rapidsegment-1.3.post3/src/rapidsegment/utils/data_loader.py +0 -320
  12. {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/LICENSE +0 -0
  13. {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/builder.py +0 -0
  14. {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/py.typed +0 -0
  15. {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/.streamlit/config.toml +0 -0
  16. {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/__init__.py +0 -0
  17. {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/_exit.py +0 -0
  18. {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/_theme.py +0 -0
  19. {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/app.py +0 -0
  20. {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/pages/2_Workbench.py +0 -0
  21. {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/pages/3_Execution_Console.py +0 -0
  22. {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/pages/4_Results_Dashboard.py +0 -0
  23. {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/pages/5_Leaderboard.py +0 -0
  24. {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/pages/6_Arena.py +0 -0
  25. {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/pages/__init__.py +0 -0
  26. {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/utils/on_gcp_feature_selection.py +0 -0
  27. {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/utils/undersampler.sql +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rapidsegment
3
- Version: 1.3.post3
3
+ Version: 1.3.1
4
4
  Summary: A fast, multi-segment population scoring and combinatorial heuristic rule extraction engine.
5
5
  Author: Bishwarup Biswas
6
6
  Author-email: Bishwarup Biswas <bishwarup1429@gmail.com>
@@ -331,12 +331,15 @@ flowchart TD
331
331
 
332
332
  I --> J["Validate on residual current_df: COUNT + SUM WHERE sql_filter"]
333
333
 
334
- J --> K{"Meets min_sample_size, min_events, and min_lift?"}
334
+ J --> K{"Volume floors (growth): count ≥ min_sample_size AND events ≥ min_events?"}
335
335
 
336
336
  K -- No --> L["Reject and try next candidate, or stop"]
337
337
  L --> C
338
338
 
339
- K -- Yes --> M["Store segment with actual counts from residual"]
339
+ K -- Yes --> K2{"Acceptance floor: lift ≥ min_lift?"}
340
+
341
+ K2 -- No --> L
342
+ K2 -- Yes --> M["Store segment with actual counts from residual"]
340
343
 
341
344
  M --> N["Update feature usage tracker"]
342
345
 
@@ -387,13 +390,13 @@ The engine evaluates combinations in a layered fashion:
387
390
 
388
391
  ```mermaid
389
392
  flowchart LR
390
- A[Top 20 Features] --> B[1‑Way Checks]
391
- B -->|Only successful features| C[2‑Way Combinations]
392
- C -->|Only pairs that passed| D[3‑Way Combinations]
393
+ A[Top-N Features<br/>top_n_vars] --> B[1‑Way Checks]
394
+ B -->|Only features with bins clearing volume floors| C[2‑Way Combinations]
395
+ C -->|Only variable pairs that cleared volume floors| D[3‑Way Combinations]
393
396
  D --> E[Final Candidate Rules]
394
397
  ```
395
398
 
396
- If a 1‑way rule fails the thresholds, all higher‑order combinations containing that feature are pruned – drastically reducing the search space.
399
+ The pruning trigger at each layer is **volume only** — a rule clears the floor when `count ≥ min_sample_size` **and** `events ≥ min_events`. `min_lift` is deliberately **not** part of pruning: it is a hard *acceptance* floor applied to candidate rules afterwards, at the grid shortlist and the final raw-SQL validation (see [How 1-Way → 2-Way → 3-Way Segment Search Works](#how-1-way--2-way--3-way-segment-search-works)). A feature leaves the search only when **none of its bins** clears the volume floors. That keeps the search small while still allowing a 3-way rule to carry more lift than any of the parts it was grown from.
397
400
 
398
401
  ## How 1-Way → 2-Way → 3-Way Segment Search Works
399
402
 
@@ -401,34 +404,34 @@ RapidSegment builds candidate segments in layers: it tests single features first
401
404
 
402
405
  ### Worked example — from 1-way to 3-way on real-looking data
403
406
 
404
- Say the target is `churned` (1 = customer left), the overall base rate is **20%** (2,000 of 10,000 customers churned), `min_lift = 1.5`, and `min_sample_size = 300`. Three binned features are in play: `tenure_bin`, `plan_type`, `support_tickets_bin`.
407
+ Say the target is `churned` (1 = customer left), the overall base rate is **20%** (2,000 of 10,000 customers churned), and the floors are `min_sample_size = 300`, `min_events = 30`, `min_lift = 1.5`. Three binned features are in play: `tenure_bin`, `plan_type`, `support_tickets_bin`.
405
408
 
406
409
  #### Step 1 — 1-way: test each bin of each feature alone
407
410
 
408
- Every individual bin is checked against the base rate. A rule only survives if its `count ≥ min_sample_size` and `lift ≥ min_lift` (`lift = segment_rate / base_rate`):
409
-
410
- | Rule (1-way) | Count | Churn rate | Lift | Survives? |
411
- |---|---|---|---|:---:|
412
- | `tenure_bin = [0-3mo]` | 1,200 | 42% | 2.1x | ✅ |
413
- | `plan_type = [Basic]` | 900 | 35% | 1.75x | ✅ |
414
- | `support_tickets_bin = [3+]` | 600 | 55% | 2.75x | ✅ |
415
- | `plan_type = [Premium]` | 800 | 8% | 0.4x | ❌ (below 1.0, protective not risky) |
416
- | `tenure_bin = [12mo+]` | 3,000 | 6% | 0.3x | ❌ |
411
+ Every individual bin is checked against the base rate. To **survive the pruning gate** a rule must clear the *volume* floors — `count ≥ min_sample_size` **and** `events ≥ min_events`. Rows and events are anti‑monotone (adding a condition can only shrink the population), so pruning on them is safe. `lift = segment_rate / base_rate` is a separate **acceptance** floor (`min_lift`) applied later — surviving the pruning gate does not by itself make a rule a segment:
412
+
413
+ | Rule (1-way) | Count | Churn rate | Lift | Pruning gate (count+events) | Accepted as segment (lift)? |
414
+ |---|---|---|---|:---:|:---:|
415
+ | `tenure_bin = [0-3mo]` | 1,200 | 42% | 2.1x | ✅ count 1,200 ≥ 300, events ≥ 30 | ✅ |
416
+ | `plan_type = [Basic]` | 900 | 35% | 1.75x | ✅ count 900 ≥ 300, events ≥ 30 | ✅ |
417
+ | `support_tickets_bin = [3+]` | 600 | 55% | 2.75x | ✅ count 600 ≥ 300, events ≥ 30 | ✅ |
418
+ | `plan_type = [Premium]` | 800 | 8% | 0.4x | ✅ count 800 ≥ 300, events ≥ 30 | ❌ lift 0.4x < 1.5 (protective, not risky) |
419
+ | `tenure_bin = [12mo+]` | 3,000 | 6% | 0.3x | ✅ count 3,000 ≥ 300, events ≥ 30 | ❌ lift 0.3x < 1.5 |
417
420
 
418
- Only bins that pass move forward. Say the survivors are `{[0-3mo], [Basic], [3+ tickets]}` — call them **A**, **B**, **C** for short. Anything that failed (like `[Premium]` or `[12mo+]`) is now completely dropped: it will never be tried in any pair or triplet, because pairing a bad bin with anything can't undo the fact that alone it wasn't predictive enough at the volume required.
421
+ Every bin above clears the volume gate, so all three features (`tenure_bin`, `plan_type`, `support_tickets_bin`) stay in the search. Only `[0-3mo]`, `[Basic]`, and `[3+ tickets]` — call them **A**, **B**, **C** for short — also clear the 1‑way lift floor. The bins that failed lift (`[Premium]`, `[12mo+]`) are **not** dropped from the search: Apriori pruning here works per **feature**, never per bin. Those bin values are still aggregated into later 2‑way / 3‑way combinations and must themselves clear the volume floors and the lift floor to be accepted — but an individual 1‑way lift shortfall never prunes the feature.
419
422
 
420
423
  #### Step 2 — 2-way: pair up only the survivors
421
424
 
422
425
  With 3 survivors there are `C(3,2) = 3` possible pairs: `A+B`, `A+C`, `B+C`. Each pair is aggregated as its own joint segment:
423
426
 
424
- | Rule (2-way) | Count | Churn rate | Lift | Survives? |
425
- |---|---|---|---|:---:|
426
- | `A+B` = `[0-3mo] AND [Basic]` | 420 | 51% | 2.55x | ✅ |
427
- | `A+C` = `[0-3mo] AND [3+ tickets]` | 310 | 58% | 2.9x | ✅ |
428
- | `B+C` = `[Basic] AND [3+ tickets]` | 180 | 60% | 3.0x | ❌ — count 180 < min_sample_size 300 |
429
-
430
- Notice `B+C` actually has the *highest* churn rate and lift of the three pairs — but it's still rejected, because too few customers (180) fall into that exact overlap to trust the number. This is the key trade-off: **survival is about count AND lift together, not lift alone.**
427
+ | Rule (2-way) | Count | Churn rate | Lift | Passes volume gate? | Meets lift floor? |
428
+ |---|---|---|---|:---:|:---:|
429
+ | `A+B` = `[0-3mo] AND [Basic]` | 420 | 51% | 2.55x | ✅ | ✅ |
430
+ | `A+C` = `[0-3mo] AND [3+ tickets]` | 310 | 58% | 2.9x | ✅ | ✅ |
431
+ | `B+C` = `[Basic] AND [3+ tickets]` | 180 | 60% | 3.0x | ❌ — count 180 < min_sample_size 300 | — |
431
432
 
433
+ Notice `B+C` actually has the *highest* churn rate and lift of the three pairs — but it's still rejected, because too few customers (180) fall into that exact overlap to trust the number. This is the key trade-off inside the volume gate: **a rule can fail pruning on count alone, even with the best lift in the room.** `B+C` is pruned before `min_lift` even gets a say.
434
+
432
435
  Survivors: `valid_2way_sets = { {A,B}, {A,C} }`.
433
436
 
434
437
  #### Step 3 — 3-way: only try triplets where every pair inside them already passed
@@ -442,15 +445,18 @@ With 3 bins there's only one possible triplet: `A+B+C`. Before RapidSegment even
442
445
  | `{B,C}` | ❌ (rejected in Step 2 for low count) |
443
446
 
444
447
  Because `{B,C}` never passed, the triplet `A+B+C` is **skipped entirely** — it is never even aggregated, no matter how strong its true joint churn rate might be. This is the pruning payoff: instead of testing every possible triplet from scratch, the engine only tests triplets whose *every* pairwise sub-relationship already proved itself statistically solid on its own.
448
+
449
+ > **Granularity note:** the engine keys `valid_2way_sets` on **variable pairs**, not bin pairs. A variable pair qualifies for 3-way growth as soon as *any* joint bin combination of those two variables clears the volume floors. In this example `plan_type` and `tenure_bin` qualify (their `A+B` overlap passes), while `tenure_bin` and `support_tickets_bin` never produce a passing overlap (the `B+C` case at 180 rows), so the triplet isn't grown. The story above shows that same idea at bin-pair level for readability.
445
450
 
446
451
  ### Why prune this way instead of just testing every triplet directly?
447
452
 
448
453
  - **Speed:** with `top_n_vars = 15`, testing all triplets directly is `C(15,3) = 455` SQL aggregations. Pruning by pairwise survival first can cut that dramatically, since most triplets get eliminated before ever touching the data.
449
454
  - **The cost:** a genuinely strong 3-way interaction can be missed if one of its underlying pairs happened to fall just under `min_sample_size` (as `{B,C}` did above at count 180) — even if the full triplet would have had a healthy count. This is the same trade-off classic Apriori pruning makes in market-basket analysis: cheap, scalable, but not exhaustive.
455
+ - **Why not prune on `min_lift` too?** Because lift can *rise* when you add a condition — a 3-way can beat every pair it was grown from. Pruning on lift would throw away exactly those strong interactions. Rows and events never rise when a rule narrows, so they're what pruning uses; `min_lift` is applied only afterward, as an acceptance check (grid shortlist + final raw validation).
450
456
  ---
451
457
 
452
458
  ### 3. Grid Search
453
- For each iteration, the engine sweeps over a user‑defined grid of `(min_sample_size, min_lift)` values. Each grid point produces a candidate champion. After all grid points are evaluated, the global champion is chosen by sorting on `(lift, count, rate)`.
459
+ For each iteration, the engine sweeps over a user‑defined grid of `(min_sample_size, min_lift)` values. Each grid point keeps the rules that clear its `count` and `lift` floors, and the top rule for that config (by `sort_priority`) becomes a candidate champion. The champions are ordered and the first to pass the raw‑residual validation (next section) becomes the iteration's champion.
454
460
 
455
461
  ### 4. Champion Validation & Extraction
456
462
  The champion’s SQL filter is validated against the **raw residual** to ensure it meets the absolute hard constraints. Only then is it accepted.
@@ -651,3 +657,4 @@ _Independent, open‑source, and ready for production._
651
657
 
652
658
 
653
659
 
660
+
@@ -293,12 +293,15 @@ flowchart TD
293
293
 
294
294
  I --> J["Validate on residual current_df: COUNT + SUM WHERE sql_filter"]
295
295
 
296
- J --> K{"Meets min_sample_size, min_events, and min_lift?"}
296
+ J --> K{"Volume floors (growth): count ≥ min_sample_size AND events ≥ min_events?"}
297
297
 
298
298
  K -- No --> L["Reject and try next candidate, or stop"]
299
299
  L --> C
300
300
 
301
- K -- Yes --> M["Store segment with actual counts from residual"]
301
+ K -- Yes --> K2{"Acceptance floor: lift ≥ min_lift?"}
302
+
303
+ K2 -- No --> L
304
+ K2 -- Yes --> M["Store segment with actual counts from residual"]
302
305
 
303
306
  M --> N["Update feature usage tracker"]
304
307
 
@@ -349,13 +352,13 @@ The engine evaluates combinations in a layered fashion:
349
352
 
350
353
  ```mermaid
351
354
  flowchart LR
352
- A[Top 20 Features] --> B[1‑Way Checks]
353
- B -->|Only successful features| C[2‑Way Combinations]
354
- C -->|Only pairs that passed| D[3‑Way Combinations]
355
+ A[Top-N Features<br/>top_n_vars] --> B[1‑Way Checks]
356
+ B -->|Only features with bins clearing volume floors| C[2‑Way Combinations]
357
+ C -->|Only variable pairs that cleared volume floors| D[3‑Way Combinations]
355
358
  D --> E[Final Candidate Rules]
356
359
  ```
357
360
 
358
- If a 1‑way rule fails the thresholds, all higher‑order combinations containing that feature are pruned – drastically reducing the search space.
361
+ The pruning trigger at each layer is **volume only** — a rule clears the floor when `count ≥ min_sample_size` **and** `events ≥ min_events`. `min_lift` is deliberately **not** part of pruning: it is a hard *acceptance* floor applied to candidate rules afterwards, at the grid shortlist and the final raw-SQL validation (see [How 1-Way → 2-Way → 3-Way Segment Search Works](#how-1-way--2-way--3-way-segment-search-works)). A feature leaves the search only when **none of its bins** clears the volume floors. That keeps the search small while still allowing a 3-way rule to carry more lift than any of the parts it was grown from.
359
362
 
360
363
  ## How 1-Way → 2-Way → 3-Way Segment Search Works
361
364
 
@@ -363,34 +366,34 @@ RapidSegment builds candidate segments in layers: it tests single features first
363
366
 
364
367
  ### Worked example — from 1-way to 3-way on real-looking data
365
368
 
366
- Say the target is `churned` (1 = customer left), the overall base rate is **20%** (2,000 of 10,000 customers churned), `min_lift = 1.5`, and `min_sample_size = 300`. Three binned features are in play: `tenure_bin`, `plan_type`, `support_tickets_bin`.
369
+ Say the target is `churned` (1 = customer left), the overall base rate is **20%** (2,000 of 10,000 customers churned), and the floors are `min_sample_size = 300`, `min_events = 30`, `min_lift = 1.5`. Three binned features are in play: `tenure_bin`, `plan_type`, `support_tickets_bin`.
367
370
 
368
371
  #### Step 1 — 1-way: test each bin of each feature alone
369
372
 
370
- Every individual bin is checked against the base rate. A rule only survives if its `count ≥ min_sample_size` and `lift ≥ min_lift` (`lift = segment_rate / base_rate`):
371
-
372
- | Rule (1-way) | Count | Churn rate | Lift | Survives? |
373
- |---|---|---|---|:---:|
374
- | `tenure_bin = [0-3mo]` | 1,200 | 42% | 2.1x | ✅ |
375
- | `plan_type = [Basic]` | 900 | 35% | 1.75x | ✅ |
376
- | `support_tickets_bin = [3+]` | 600 | 55% | 2.75x | ✅ |
377
- | `plan_type = [Premium]` | 800 | 8% | 0.4x | ❌ (below 1.0, protective not risky) |
378
- | `tenure_bin = [12mo+]` | 3,000 | 6% | 0.3x | ❌ |
373
+ Every individual bin is checked against the base rate. To **survive the pruning gate** a rule must clear the *volume* floors — `count ≥ min_sample_size` **and** `events ≥ min_events`. Rows and events are anti‑monotone (adding a condition can only shrink the population), so pruning on them is safe. `lift = segment_rate / base_rate` is a separate **acceptance** floor (`min_lift`) applied later — surviving the pruning gate does not by itself make a rule a segment:
374
+
375
+ | Rule (1-way) | Count | Churn rate | Lift | Pruning gate (count+events) | Accepted as segment (lift)? |
376
+ |---|---|---|---|:---:|:---:|
377
+ | `tenure_bin = [0-3mo]` | 1,200 | 42% | 2.1x | ✅ count 1,200 ≥ 300, events ≥ 30 | ✅ |
378
+ | `plan_type = [Basic]` | 900 | 35% | 1.75x | ✅ count 900 ≥ 300, events ≥ 30 | ✅ |
379
+ | `support_tickets_bin = [3+]` | 600 | 55% | 2.75x | ✅ count 600 ≥ 300, events ≥ 30 | ✅ |
380
+ | `plan_type = [Premium]` | 800 | 8% | 0.4x | ✅ count 800 ≥ 300, events ≥ 30 | ❌ lift 0.4x < 1.5 (protective, not risky) |
381
+ | `tenure_bin = [12mo+]` | 3,000 | 6% | 0.3x | ✅ count 3,000 ≥ 300, events ≥ 30 | ❌ lift 0.3x < 1.5 |
379
382
 
380
- Only bins that pass move forward. Say the survivors are `{[0-3mo], [Basic], [3+ tickets]}` — call them **A**, **B**, **C** for short. Anything that failed (like `[Premium]` or `[12mo+]`) is now completely dropped: it will never be tried in any pair or triplet, because pairing a bad bin with anything can't undo the fact that alone it wasn't predictive enough at the volume required.
383
+ Every bin above clears the volume gate, so all three features (`tenure_bin`, `plan_type`, `support_tickets_bin`) stay in the search. Only `[0-3mo]`, `[Basic]`, and `[3+ tickets]` — call them **A**, **B**, **C** for short — also clear the 1‑way lift floor. The bins that failed lift (`[Premium]`, `[12mo+]`) are **not** dropped from the search: Apriori pruning here works per **feature**, never per bin. Those bin values are still aggregated into later 2‑way / 3‑way combinations and must themselves clear the volume floors and the lift floor to be accepted — but an individual 1‑way lift shortfall never prunes the feature.
381
384
 
382
385
  #### Step 2 — 2-way: pair up only the survivors
383
386
 
384
387
  With 3 survivors there are `C(3,2) = 3` possible pairs: `A+B`, `A+C`, `B+C`. Each pair is aggregated as its own joint segment:
385
388
 
386
- | Rule (2-way) | Count | Churn rate | Lift | Survives? |
387
- |---|---|---|---|:---:|
388
- | `A+B` = `[0-3mo] AND [Basic]` | 420 | 51% | 2.55x | ✅ |
389
- | `A+C` = `[0-3mo] AND [3+ tickets]` | 310 | 58% | 2.9x | ✅ |
390
- | `B+C` = `[Basic] AND [3+ tickets]` | 180 | 60% | 3.0x | ❌ — count 180 < min_sample_size 300 |
391
-
392
- Notice `B+C` actually has the *highest* churn rate and lift of the three pairs — but it's still rejected, because too few customers (180) fall into that exact overlap to trust the number. This is the key trade-off: **survival is about count AND lift together, not lift alone.**
389
+ | Rule (2-way) | Count | Churn rate | Lift | Passes volume gate? | Meets lift floor? |
390
+ |---|---|---|---|:---:|:---:|
391
+ | `A+B` = `[0-3mo] AND [Basic]` | 420 | 51% | 2.55x | ✅ | ✅ |
392
+ | `A+C` = `[0-3mo] AND [3+ tickets]` | 310 | 58% | 2.9x | ✅ | ✅ |
393
+ | `B+C` = `[Basic] AND [3+ tickets]` | 180 | 60% | 3.0x | ❌ — count 180 < min_sample_size 300 | — |
393
394
 
395
+ Notice `B+C` actually has the *highest* churn rate and lift of the three pairs — but it's still rejected, because too few customers (180) fall into that exact overlap to trust the number. This is the key trade-off inside the volume gate: **a rule can fail pruning on count alone, even with the best lift in the room.** `B+C` is pruned before `min_lift` even gets a say.
396
+
394
397
  Survivors: `valid_2way_sets = { {A,B}, {A,C} }`.
395
398
 
396
399
  #### Step 3 — 3-way: only try triplets where every pair inside them already passed
@@ -404,15 +407,18 @@ With 3 bins there's only one possible triplet: `A+B+C`. Before RapidSegment even
404
407
  | `{B,C}` | ❌ (rejected in Step 2 for low count) |
405
408
 
406
409
  Because `{B,C}` never passed, the triplet `A+B+C` is **skipped entirely** — it is never even aggregated, no matter how strong its true joint churn rate might be. This is the pruning payoff: instead of testing every possible triplet from scratch, the engine only tests triplets whose *every* pairwise sub-relationship already proved itself statistically solid on its own.
410
+
411
+ > **Granularity note:** the engine keys `valid_2way_sets` on **variable pairs**, not bin pairs. A variable pair qualifies for 3-way growth as soon as *any* joint bin combination of those two variables clears the volume floors. In this example `plan_type` and `tenure_bin` qualify (their `A+B` overlap passes), while `tenure_bin` and `support_tickets_bin` never produce a passing overlap (the `B+C` case at 180 rows), so the triplet isn't grown. The story above shows that same idea at bin-pair level for readability.
407
412
 
408
413
  ### Why prune this way instead of just testing every triplet directly?
409
414
 
410
415
  - **Speed:** with `top_n_vars = 15`, testing all triplets directly is `C(15,3) = 455` SQL aggregations. Pruning by pairwise survival first can cut that dramatically, since most triplets get eliminated before ever touching the data.
411
416
  - **The cost:** a genuinely strong 3-way interaction can be missed if one of its underlying pairs happened to fall just under `min_sample_size` (as `{B,C}` did above at count 180) — even if the full triplet would have had a healthy count. This is the same trade-off classic Apriori pruning makes in market-basket analysis: cheap, scalable, but not exhaustive.
417
+ - **Why not prune on `min_lift` too?** Because lift can *rise* when you add a condition — a 3-way can beat every pair it was grown from. Pruning on lift would throw away exactly those strong interactions. Rows and events never rise when a rule narrows, so they're what pruning uses; `min_lift` is applied only afterward, as an acceptance check (grid shortlist + final raw validation).
412
418
  ---
413
419
 
414
420
  ### 3. Grid Search
415
- For each iteration, the engine sweeps over a user‑defined grid of `(min_sample_size, min_lift)` values. Each grid point produces a candidate champion. After all grid points are evaluated, the global champion is chosen by sorting on `(lift, count, rate)`.
421
+ For each iteration, the engine sweeps over a user‑defined grid of `(min_sample_size, min_lift)` values. Each grid point keeps the rules that clear its `count` and `lift` floors, and the top rule for that config (by `sort_priority`) becomes a candidate champion. The champions are ordered and the first to pass the raw‑residual validation (next section) becomes the iteration's champion.
416
422
 
417
423
  ### 4. Champion Validation & Extraction
418
424
  The champion’s SQL filter is validated against the **raw residual** to ensure it meets the absolute hard constraints. Only then is it accepted.
@@ -613,3 +619,4 @@ _Independent, open‑source, and ready for production._
613
619
 
614
620
 
615
621
 
622
+
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "rapidsegment"
3
- version = "1.3.post3"
3
+ version = "1.3.1"
4
4
  description = "A fast, multi-segment population scoring and combinatorial heuristic rule extraction engine."
5
5
  readme = "README.md"
6
6
  license = "MIT"
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "rapidsegment"
3
- version = "1.3.post3"
3
+ version = "1.3.1"
4
4
  description = "A fast, multi-segment population scoring and combinatorial heuristic rule extraction engine."
5
5
  readme = "README.md"
6
6
  authors = [
@@ -8,8 +8,8 @@ except PackageNotFoundError:
8
8
 
9
9
  __author__ = "Bishwarup Biswas <bishwarup1429@gmail.com>"
10
10
 
11
- from .utils import UniversalDataLoader
11
+ from .utils import UniversalDataLoader, duckdb_to_arrow
12
12
  from .builder import StrategicSegmentBuilder
13
13
  from .scorer import StrategicSegmentScore
14
14
 
15
- __all__ = ["UniversalDataLoader", "StrategicSegmentBuilder", "StrategicSegmentScore"]
15
+ __all__ = ["UniversalDataLoader", "duckdb_to_arrow", "StrategicSegmentBuilder", "StrategicSegmentScore"]
@@ -11,6 +11,7 @@ Python Version: 3.11+
11
11
  import json
12
12
  import logging
13
13
  import os
14
+ import re
14
15
  import tempfile
15
16
  import uuid
16
17
  from datetime import datetime
@@ -60,23 +61,37 @@ class StrategicSegmentScore:
60
61
  data: Any,
61
62
  export_path: str = f"scored_experiment_{timestamp}.json",
62
63
  db_path: Optional[str] = None,
64
+ table_name: str = "df",
63
65
  ) -> Dict[str, Any]:
64
66
  """
65
67
  Calculates harmonic weights and derives decile boundaries via vectorised execution.
66
68
 
67
69
  Args:
68
- data: Input data (will be loaded into DuckDB).
70
+ data: Input data. Either the path to a DuckDB database file containing
71
+ ``table_name`` (zero-copy ATTACH), or any DuckDB-registrable object
72
+ (pandas DataFrame, PyArrow Table, duckdb Relation, ...).
69
73
  export_path: File path to save the model artifact JSON.
70
- db_path: Optional path to a persistent DuckDB file/connection to reuse as the
71
- single data artifact (e.g. the builder's ``db_path``). When omitted, a
72
- temporary file-backed DB is created under the system temp dir and removed
73
- after the run, avoiding CWD pollution.
74
+ db_path: Optional path to a persistent DuckDB file/connection to reuse
75
+ as the single data artifact (e.g. the builder's ``db_path``). When
76
+ omitted, a temporary file-backed DB is created under the system
77
+ temp dir and removed after the run, avoiding CWD pollution.
78
+ table_name: Table or view to read from the file handed in ``data``
79
+ (default 'df'). Point it at any table you prepared, e.g.
80
+ `table_name='predicted'` after creating a scored table named
81
+ 'predicted' inside the database.
74
82
 
75
83
  Returns:
76
84
  Dictionary containing model metadata, segment weights, and decile thresholds.
77
85
  """
78
86
  logger.info("🚀 Initialising out‑of‑core DuckDB scorecard engine...")
79
87
 
88
+ if not isinstance(table_name, str) or not re.fullmatch(
89
+ r"[A-Za-z_][A-Za-z0-9_]*", table_name
90
+ ):
91
+ raise ValueError(
92
+ f"table_name must be a valid SQL identifier; got {table_name!r}"
93
+ )
94
+
80
95
  # Use file‑backed storage for large datasets. Reuse a caller-supplied DB when
81
96
  # provided; otherwise create a unique temp file and clean it up afterwards so we
82
97
  # never leak a `score_experiment_*.db` into the current working directory.
@@ -86,7 +101,7 @@ class StrategicSegmentScore:
86
101
  db_path = os.path.join(
87
102
  tempfile.gettempdir(), f"rapidsegment_score_{unique_id}.duckdb"
88
103
  )
89
- if os.path.exists(db_path):
104
+ if own_db and os.path.exists(db_path):
90
105
  os.remove(db_path)
91
106
  ctx = duckdb.connect(db_path)
92
107
  if isinstance(data, str):
@@ -95,7 +110,9 @@ class StrategicSegmentScore:
95
110
  # Attach it read-only instead of materialising it into Python.
96
111
  src_path = data.replace("\\", "/")
97
112
  ctx.execute(f"ATTACH '{src_path}' AS __rs_src (READ_ONLY)")
98
- ctx.execute("CREATE OR REPLACE TABLE df AS SELECT * FROM __rs_src.df")
113
+ ctx.execute(
114
+ f'CREATE OR REPLACE TABLE df AS SELECT * FROM __rs_src."{table_name}"'
115
+ )
99
116
  else:
100
117
  ctx.execute("CREATE OR REPLACE TABLE df AS SELECT * FROM data")
101
118
 
@@ -0,0 +1,3 @@
1
+ from .data_loader import UniversalDataLoader, duckdb_to_arrow
2
+
3
+ __all__ = ["UniversalDataLoader", "duckdb_to_arrow"]
@@ -0,0 +1,538 @@
1
+ """
2
+ Unified Data Ingestion Layer
3
+ ============================
4
+ Multi‑format data loader supporting Local Files (CSV, Parquet, Arrow, Excel),
5
+ In‑Memory PyArrow Tables, and Google Cloud BigQuery Storage API streams.
6
+
7
+ Author: Bishwarup Biswas + Gemini + DeepSeek
8
+ Python Version: 3.9+
9
+ """
10
+
11
+ import logging
12
+ import os
13
+ import re
14
+ import tempfile
15
+ import uuid
16
+ from typing import Any, Optional, Union
17
+
18
+ import duckdb
19
+ import pyarrow as pa
20
+ import pyarrow.compute as pc
21
+ import pyarrow.csv as pa_csv
22
+ import pyarrow.parquet as pa_pq
23
+
24
+ logger = logging.getLogger("StrategicEngine.DataLoader")
25
+
26
+
27
+ class UniversalDataLoader:
28
+ """
29
+ Handles multi‑source data ingestion, normalising inputs into highly optimised
30
+ in‑memory PyArrow Tables suitable for vectorised downstream compute engines.
31
+
32
+ The loader automatically detects the source type based on constructor arguments.
33
+ If a `fallback_data` object is passed to `load()`, it takes precedence.
34
+
35
+ Args:
36
+ project_id: (Optional) GCP project ID for BigQuery.
37
+ dataset_id: (Optional) BigQuery dataset ID.
38
+ table_id: (Optional) BigQuery table ID.
39
+ file_path: (Optional) Local file path (CSV, Parquet, Arrow/Feather, Excel).
40
+ """
41
+
42
+ def __init__(
43
+ self,
44
+ project_id: Optional[str] = None,
45
+ dataset_id: Optional[str] = None,
46
+ table_id: Optional[str] = None,
47
+ file_path: Optional[str] = None,
48
+ ) -> None:
49
+ self.project_id = project_id
50
+ self.dataset_id = dataset_id
51
+ self.table_id = table_id
52
+ self.file_path = file_path
53
+
54
+ def load(self, fallback_data: Optional[Any] = None) -> Union[pa.Table, str]:
55
+ """
56
+ Auto‑detects the source configuration and loads the dataset.
57
+
58
+ Priority order:
59
+ 1. If `fallback_data` is provided, it is returned (with type normalisation).
60
+ 2. If BigQuery identifiers are set, load from BigQuery.
61
+ 3. If a local `file_path` is provided, load from file.
62
+
63
+ Args:
64
+ fallback_data: Optional pre‑loaded data (e.g., a PyArrow Table or any
65
+ object that can be passed to DuckDB directly).
66
+
67
+ Returns:
68
+ A PyArrow Table, or a DuckDB scan macro string (when BigQuery client
69
+ is not available and fallback is not provided).
70
+ """
71
+ # Scenario 1: Direct in‑memory object
72
+ if fallback_data is not None:
73
+ if isinstance(fallback_data, pa.Table):
74
+ logger.info("📥 Ingesting directly provided in‑memory PyArrow Table.")
75
+ return self._cast_table_numerics_to_float(fallback_data)
76
+ logger.info("📥 Using provided fallback data (non‑Arrow) as‑is.")
77
+ return fallback_data
78
+
79
+ # Scenario 2: BigQuery
80
+ if self.dataset_id and self.table_id:
81
+ return self._load_from_bigquery()
82
+
83
+ # Scenario 3: Local file
84
+ if self.file_path:
85
+ return self._load_from_file()
86
+
87
+ raise ValueError(
88
+ "Invalid Configuration: You must provide either a valid `file_path`, "
89
+ "BigQuery identifiers, or pass an explicit `fallback_data` object."
90
+ )
91
+
92
+ @staticmethod
93
+ def _cast_table_numerics_to_float(table: pa.Table) -> pa.Table:
94
+ """
95
+ Casts all numeric columns in a PyArrow Table to float64.
96
+
97
+ This ensures consistent numerical precision across downstream operations.
98
+
99
+ Args:
100
+ table: Input PyArrow Table.
101
+
102
+ Returns:
103
+ A new table with numeric columns cast to float64.
104
+ """
105
+ if not isinstance(table, pa.Table):
106
+ return table
107
+
108
+ new_columns = []
109
+ new_fields = []
110
+
111
+ for i, field in enumerate(table.schema):
112
+ # Check if the type is integer, floating, or decimal
113
+ if (
114
+ pa.types.is_integer(field.type)
115
+ or pa.types.is_floating(field.type)
116
+ or pa.types.is_decimal(field.type)
117
+ ):
118
+ try:
119
+ casted_col = pc.cast(
120
+ table.column(i), pa.float64(), safe=False
121
+ )
122
+ new_columns.append(casted_col)
123
+ new_fields.append(
124
+ pa.field(field.name, pa.float64(), nullable=field.nullable)
125
+ )
126
+ except Exception as e:
127
+ logger.warning(
128
+ f"⚠️ Failed to cast column {field.name} to float64. Reason: {e}"
129
+ )
130
+ new_columns.append(table.column(i))
131
+ new_fields.append(field)
132
+ else:
133
+ new_columns.append(table.column(i))
134
+ new_fields.append(field)
135
+
136
+ return pa.Table.from_arrays(new_columns, schema=pa.schema(new_fields))
137
+
138
+ def _load_from_file(self) -> pa.Table:
139
+ """
140
+ Parses a local file using high‑performance C++ Arrow readers.
141
+
142
+ Supports: .parquet, .csv, .arrow / .feather, .xlsx / .xls.
143
+
144
+ Returns:
145
+ PyArrow Table with numeric columns cast to float64.
146
+ """
147
+ if not os.path.exists(self.file_path):
148
+ raise FileNotFoundError(f"Data file not found at: {self.file_path}")
149
+
150
+ ext = os.path.splitext(self.file_path)[-1].lower()
151
+ logger.info(f"📂 Loading file: {self.file_path} (extension: {ext})")
152
+
153
+ if ext == ".parquet":
154
+ table = pa_pq.read_table(self.file_path)
155
+ elif ext == ".csv":
156
+ table = pa_csv.read_csv(self.file_path)
157
+ elif ext in (".arrow", ".feather"):
158
+ with pa.memory_map(self.file_path, "r") as source:
159
+ table = pa.ipc.open_file(source).read_all()
160
+ elif ext in (".xlsx", ".xls"):
161
+ table = self._load_excel_to_arrow()
162
+ else:
163
+ raise ValueError(f"Unsupported file format: '{ext}'.")
164
+
165
+ return self._cast_table_numerics_to_float(table)
166
+
167
+ def _load_excel_to_arrow(self) -> pa.Table:
168
+ """
169
+ Parses an Excel file using openpyxl with positional column tracking.
170
+
171
+ Returns:
172
+ PyArrow Table.
173
+ """
174
+ logger.info("📊 Parsing Excel spreadsheet via positional column tracking...")
175
+ try:
176
+ import openpyxl
177
+
178
+ wb = openpyxl.load_workbook(
179
+ self.file_path, data_only=True, read_only=True
180
+ )
181
+ sheet = wb.active
182
+ rows = sheet.iter_rows(values_only=True)
183
+
184
+ headers = next(rows)
185
+ if not headers:
186
+ raise ValueError("The Excel file appears to be empty.")
187
+
188
+ # Use column indices to prevent header‑shift corruption
189
+ column_names = [
190
+ f"{h}" if h is not None else f"_col_{i}"
191
+ for i, h in enumerate(headers)
192
+ ]
193
+ data_columns = {name: [] for name in column_names}
194
+
195
+ for row in rows:
196
+ for i, name in enumerate(column_names):
197
+ val = row[i] if i < len(row) else None
198
+ data_columns[name].append(val)
199
+
200
+ wb.close()
201
+ return pa.Table.from_pydict(data_columns)
202
+
203
+ except ImportError:
204
+ raise ImportError(
205
+ "Dependency missing: `pip install openpyxl` required for Excel files."
206
+ )
207
+
208
+ def _load_from_bigquery(self) -> Union[pa.Table, str]:
209
+ """
210
+ Resolves BigQuery ingestion using cost‑optimised metadata inspection.
211
+
212
+ If the `google‑cloud‑bigquery` library is available, streams the table
213
+ as a PyArrow Table. Otherwise, returns a DuckDB scan macro string for
214
+ later execution (requires DuckDB's BigQuery extension).
215
+
216
+ Returns:
217
+ PyArrow Table or a DuckDB macro string.
218
+ """
219
+ full_bq_path = (
220
+ f"{self.project_id}.{self.dataset_id}.{self.table_id}"
221
+ if self.project_id
222
+ else f"{self.dataset_id}.{self.table_id}"
223
+ )
224
+ logger.info(f"☁️ Initialising BigQuery client for: {full_bq_path}")
225
+
226
+ try:
227
+ from google.cloud import bigquery
228
+
229
+ bq_client = bigquery.Client(project=self.project_id)
230
+ full_table_ref = (
231
+ f"{self.project_id or bq_client.project}."
232
+ f"{self.dataset_id}.{self.table_id}"
233
+ )
234
+
235
+ # Fetch schema via get_table (cheaper than INFORMATION_SCHEMA)
236
+ table = bq_client.get_table(full_table_ref)
237
+
238
+ select_clauses = []
239
+ for field in table.schema:
240
+ # Cast numeric types to FLOAT64 for consistency
241
+ if field.field_type in (
242
+ "NUMERIC",
243
+ "BIGNUMERIC",
244
+ "DECIMAL",
245
+ "INTEGER",
246
+ "INT64",
247
+ "FLOAT",
248
+ "FLOAT64",
249
+ ):
250
+ select_clauses.append(
251
+ f"SAFE_CAST(`{field.name}` AS FLOAT64) AS `{field.name}`"
252
+ )
253
+ else:
254
+ select_clauses.append(f"`{field.name}`")
255
+
256
+ query = f"SELECT {', '.join(select_clauses)} FROM `{full_table_ref}`"
257
+ arrow_table = bq_client.query(query).to_arrow()
258
+ logger.info(f"✅ Loaded {len(arrow_table):,} rows from BigQuery.")
259
+ return arrow_table
260
+
261
+ except ImportError:
262
+ logger.warning(
263
+ "⚠️ google‑cloud‑bigquery not found. Returning DuckDB scan macro. "
264
+ "This macro requires the BigQuery extension to be loaded in DuckDB."
265
+ )
266
+ # Return a string that DuckDB can interpret as a table reference
267
+ # (assuming the extension is loaded)
268
+ return f"bigquery_scan('{self.project_id or 'default'}', '{self.dataset_id}', '{self.table_id}')"
269
+
270
+ @staticmethod
271
+ def _validate_identifier(name: str, arg: str) -> str:
272
+ """Ensures a caller-supplied table name is a safe SQL identifier."""
273
+ if not isinstance(name, str) or not re.fullmatch(r"[A-Za-z_][A-Za-z0-9_]*", name):
274
+ raise ValueError(f"{arg} must be a valid SQL identifier; got {name!r}")
275
+ return name
276
+
277
+ @staticmethod
278
+ def _drop_object_if_exists(con: duckdb.DuckDBPyConnection, name: str) -> None:
279
+ """Drops a table or view by its exact catalog kind so creation is idempotent."""
280
+ rows = con.execute(
281
+ "SELECT table_type FROM information_schema.tables WHERE table_name = ?",
282
+ [name],
283
+ ).fetchall()
284
+ for (table_type,) in rows:
285
+ if table_type.upper() == "VIEW":
286
+ con.execute(f'DROP VIEW IF EXISTS "{name}"')
287
+ else:
288
+ con.execute(f'DROP TABLE IF EXISTS "{name}"')
289
+
290
+ @staticmethod
291
+ def _cast_numeric_columns_to_double(
292
+ con: duckdb.DuckDBPyConnection, table_name: str
293
+ ) -> None:
294
+ """Memory-light on-disk cast of every numeric column to DOUBLE (float64)."""
295
+ cols = con.execute(
296
+ "SELECT column_name, data_type FROM information_schema.columns "
297
+ "WHERE table_name = ?",
298
+ [table_name],
299
+ ).fetchall()
300
+ for name, dtype in cols:
301
+ up = dtype.upper()
302
+ if (
303
+ any(
304
+ k in up
305
+ for k in ("INT", "FLOAT", "DECIMAL", "REAL", "DOUBLE", "NUMERIC")
306
+ )
307
+ and "DOUBLE" not in up
308
+ ):
309
+ con.execute(
310
+ f'ALTER TABLE "{table_name}" ALTER COLUMN "{name}" TYPE DOUBLE'
311
+ )
312
+
313
+ def stream_to_duckdb(
314
+ self,
315
+ db_path: Optional[str] = None,
316
+ path: Optional[str] = None,
317
+ encoding: Optional[str] = None,
318
+ data: Optional[Any] = None,
319
+ table_name: str = "udl_data",
320
+ scorer_view_name: str = "df",
321
+ ) -> str:
322
+ """Stream a data source straight into a persisted DuckDB database file.
323
+
324
+ Handles local files (read straight from disk, memory-light), in-memory
325
+ objects (PyArrow Table, pandas DataFrame, or any DuckDB-registrable
326
+ relation), and BigQuery tables. Numeric columns are cast to DOUBLE on-disk
327
+ to match the loader's float64 convention, and a ``df`` view is added over
328
+ the data table so the returned path can be handed directly to the engine's
329
+ builder, evaluator, and scorer.
330
+
331
+ Supported files: .csv/.tsv (read_csv_auto), .parquet/.pq (read_parquet),
332
+ .arrow/.feather (read_ipc). Excel is not streamed directly — call
333
+ ``UniversalDataLoader(file_path=...).load()`` and pass the result as ``data=``.
334
+
335
+ Args:
336
+ db_path: Optional. Database name or file path to write to. A ``.duckdb``
337
+ extension is appended automatically when missing and the parent
338
+ directory is created as needed. When omitted, a unique database is
339
+ auto-created under the system temp directory.
340
+ path: Source file path (overrides the constructor's ``file_path`` when
341
+ both are set).
342
+ encoding: Optional encoding hint ("Latin-1" is honored for CSV/TSV).
343
+ data: In-memory source object (PyArrow Table, pandas DataFrame, or any
344
+ DuckDB-registrable relation) used instead of a file.
345
+ table_name: Name of the persisted data table (default 'udl_data'). A
346
+ custom name is fine — a ``udl_data`` view is always added as an
347
+ alias, so `StrategicSegmentBuilder.extract_segments` and
348
+ `evaluate_final_coverage` keep working when handed the returned
349
+ path (both resolve 'udl_data' by that fixed name).
350
+ scorer_view_name: Optional view name aliasing the data table so
351
+ `StrategicSegmentScore.calculate_and_export_weights` can consume
352
+ the same file (default 'df'). Pass None/'' to skip creating it.
353
+
354
+ Returns:
355
+ Absolute path to the created .duckdb file, usable directly by
356
+ ``extract_segments``, ``evaluate_final_coverage``, and
357
+ ``calculate_and_export_weights``.
358
+
359
+ Examples:
360
+ >>> from rapidsegment.utils.data_loader import UniversalDataLoader
361
+ >>> out = UniversalDataLoader(file_path="bank_train.csv").stream_to_duckdb("bank_data")
362
+ >>> builder.extract_segments(out) # reads table/view 'udl_data'
363
+ >>> builder.evaluate_final_coverage(out) # reads table/view 'udl_data'
364
+ >>> scorer.calculate_and_export_weights(out, "w.json") # reads table/view 'df'
365
+ """
366
+ if not db_path:
367
+ unique_id = uuid.uuid4().hex[:8]
368
+ db_path = os.path.join(
369
+ tempfile.gettempdir(), f"rapidsegment_udl_{unique_id}.duckdb"
370
+ )
371
+ logger.info(f"🗄️ No db_path given — created default database at: {db_path}")
372
+ table_name = self._validate_identifier(table_name, "table_name")
373
+ if scorer_view_name:
374
+ scorer_view_name = self._validate_identifier(
375
+ scorer_view_name, "scorer_view_name"
376
+ )
377
+
378
+ # Resolve the source, mirroring load()'s priority order.
379
+ source_kind: str
380
+ source_path: Optional[str] = None
381
+ if data is not None:
382
+ source_kind = "object"
383
+ elif path is not None:
384
+ source_kind = "file"
385
+ source_path = os.path.abspath(path)
386
+ elif self.file_path is not None:
387
+ source_kind = "file"
388
+ source_path = os.path.abspath(self.file_path)
389
+ elif self.dataset_id and self.table_id:
390
+ bq_result = self._load_from_bigquery()
391
+ if isinstance(bq_result, str):
392
+ raise ValueError(
393
+ "stream_to_duckdb needs the 'google-cloud-bigquery' library to "
394
+ "materialise BigQuery data; install it and retry."
395
+ )
396
+ data = bq_result
397
+ source_kind = "object"
398
+ else:
399
+ raise ValueError(
400
+ "No data source configured. Pass `path=`, constructor `file_path`, "
401
+ "BigQuery identifiers (project_id/dataset_id/table_id), or in-memory `data=`."
402
+ )
403
+
404
+ if source_kind == "file" and source_path:
405
+ ext = os.path.splitext(source_path)[1].lower()
406
+ if ext not in (".csv", ".tsv", ".parquet", ".pq", ".arrow", ".feather"):
407
+ raise ValueError(
408
+ f"stream_to_duckdb does not support '{ext}' files directly; "
409
+ f"call UniversalDataLoader(file_path='{source_path}').load() and "
410
+ "pass the result as `data=` instead."
411
+ )
412
+
413
+ # Normalize the target: absolute path, always ending in .duckdb.
414
+ db_name = str(db_path)
415
+ if not db_name.lower().endswith(".duckdb"):
416
+ db_name = f"{db_name}.duckdb"
417
+ db_abs = os.path.abspath(db_name)
418
+ parent = os.path.dirname(db_abs)
419
+ if parent:
420
+ os.makedirs(parent, exist_ok=True)
421
+
422
+ con = duckdb.connect(db_abs)
423
+ try:
424
+ self._drop_object_if_exists(con, table_name)
425
+ if source_kind == "file" and source_path:
426
+ if ext in (".csv", ".tsv"):
427
+ opts = "header=true, sample_size=-1"
428
+ if encoding == "Latin-1":
429
+ opts += ", encoding='LATIN-1'"
430
+ con.execute(
431
+ f'CREATE TABLE "{table_name}" AS SELECT * FROM read_csv_auto(?, {opts})',
432
+ [source_path],
433
+ )
434
+ elif ext in (".parquet", ".pq"):
435
+ con.execute(
436
+ f'CREATE TABLE "{table_name}" AS SELECT * FROM read_parquet(?)',
437
+ [source_path],
438
+ )
439
+ else: # .arrow / .feather
440
+ con.execute(
441
+ f'CREATE TABLE "{table_name}" AS SELECT * FROM read_ipc(?)',
442
+ [source_path],
443
+ )
444
+ else:
445
+ con.register("__rs_stream_src", data)
446
+ con.execute(
447
+ f'CREATE TABLE "{table_name}" AS SELECT * FROM __rs_stream_src'
448
+ )
449
+ # float64 convention, memory-light (on-disk ALTER per numeric column)
450
+ self._cast_numeric_columns_to_double(con, table_name)
451
+ # Engine-compatibility aliases: keep a `udl_data` view (builder /
452
+ # evaluator resolve it by that fixed name) and a `df` view (scorer).
453
+ # Views are catalog metadata only — a few bytes, never a data copy.
454
+ engine_base = "udl_data"
455
+ if table_name != engine_base:
456
+ self._drop_object_if_exists(con, engine_base)
457
+ con.execute(
458
+ f'CREATE VIEW "{engine_base}" AS SELECT * FROM "{table_name}"'
459
+ )
460
+ if scorer_view_name and scorer_view_name != table_name:
461
+ self._drop_object_if_exists(con, scorer_view_name)
462
+ con.execute(
463
+ f'CREATE VIEW "{scorer_view_name}" AS SELECT * FROM "{table_name}"'
464
+ )
465
+ finally:
466
+ con.close()
467
+ return db_abs
468
+
469
+ @staticmethod
470
+ def duckdb_to_arrow(
471
+ db: Union[str, os.PathLike, duckdb.DuckDBPyConnection],
472
+ table_name: str = "udl_data",
473
+ ) -> pa.Table:
474
+ """Read a table from a DuckDB file (or an open DuckDB connection) into PyArrow.
475
+
476
+ Args:
477
+ db: A DuckDB database file path or an already-open ``duckdb`` connection.
478
+ table_name: Table or view to read (default 'udl_data'; artifacts written
479
+ by ``stream_to_duckdb`` also carry the 'df' scorer view).
480
+
481
+ Returns:
482
+ PyArrow Table of the requested table.
483
+
484
+ Raises:
485
+ ValueError: If ``table_name`` does not exist in the DuckDB source.
486
+ """
487
+ owned = False
488
+ if isinstance(db, duckdb.DuckDBPyConnection):
489
+ con = db
490
+ else:
491
+ db_file = str(db)
492
+ try:
493
+ con = duckdb.connect(db_file, read_only=True)
494
+ except duckdb.Error:
495
+ # A same-process connection with a different (read-write) config
496
+ # may already hold the file; match it. Never create a missing DB.
497
+ if not os.path.exists(db_file):
498
+ raise
499
+ con = duckdb.connect(db_file)
500
+ owned = True
501
+ try:
502
+ exists = con.execute(
503
+ "SELECT 1 FROM information_schema.tables WHERE table_name = ?",
504
+ [table_name],
505
+ ).fetchone()
506
+ if not exists:
507
+ available = ", ".join(
508
+ r[0]
509
+ for r in con.execute(
510
+ "SELECT table_name FROM information_schema.tables ORDER BY 1"
511
+ ).fetchall()
512
+ )
513
+ raise ValueError(
514
+ f"Table '{table_name}' not found in DuckDB source. "
515
+ f"Available tables: {available or '(none)'}"
516
+ )
517
+ return UniversalDataLoader._fetch_arrow_table(
518
+ con.execute(f'SELECT * FROM "{table_name}"')
519
+ )
520
+ finally:
521
+ if owned:
522
+ con.close()
523
+
524
+ @staticmethod
525
+ def _fetch_arrow_table(result: Any) -> pa.Table:
526
+ """Normalises duckdb's varying result hand-off into a concrete pa.Table."""
527
+ try:
528
+ return result.fetch_arrow_table()
529
+ except AttributeError: # duckdb < 1.x exposes .arrow() -> RecordBatchReader
530
+ return result.arrow().read_all()
531
+
532
+
533
+ def duckdb_to_arrow(
534
+ db: Union[str, os.PathLike, duckdb.DuckDBPyConnection],
535
+ table_name: str = "udl_data",
536
+ ) -> pa.Table:
537
+ """Module-level convenience wrapper for ``UniversalDataLoader.duckdb_to_arrow``."""
538
+ return UniversalDataLoader.duckdb_to_arrow(db, table_name=table_name)
@@ -1,3 +0,0 @@
1
- from .data_loader import UniversalDataLoader
2
-
3
- __all__ = ["UniversalDataLoader"]
@@ -1,320 +0,0 @@
1
- """
2
- Unified Data Ingestion Layer
3
- ============================
4
- Multi‑format data loader supporting Local Files (CSV, Parquet, Arrow, Excel),
5
- In‑Memory PyArrow Tables, and Google Cloud BigQuery Storage API streams.
6
-
7
- Author: Bishwarup Biswas + Gemini + DeepSeek
8
- Python Version: 3.9+
9
- """
10
-
11
- import logging
12
- import os
13
- from typing import Any, Optional, Union
14
-
15
- import duckdb
16
- import pyarrow as pa
17
- import pyarrow.compute as pc
18
- import pyarrow.csv as pa_csv
19
- import pyarrow.parquet as pa_pq
20
-
21
- logger = logging.getLogger("StrategicEngine.DataLoader")
22
-
23
-
24
- class UniversalDataLoader:
25
- """
26
- Handles multi‑source data ingestion, normalising inputs into highly optimised
27
- in‑memory PyArrow Tables suitable for vectorised downstream compute engines.
28
-
29
- The loader automatically detects the source type based on constructor arguments.
30
- If a `fallback_data` object is passed to `load()`, it takes precedence.
31
-
32
- Args:
33
- project_id: (Optional) GCP project ID for BigQuery.
34
- dataset_id: (Optional) BigQuery dataset ID.
35
- table_id: (Optional) BigQuery table ID.
36
- file_path: (Optional) Local file path (CSV, Parquet, Arrow/Feather, Excel).
37
- """
38
-
39
- def __init__(
40
- self,
41
- project_id: Optional[str] = None,
42
- dataset_id: Optional[str] = None,
43
- table_id: Optional[str] = None,
44
- file_path: Optional[str] = None,
45
- ) -> None:
46
- self.project_id = project_id
47
- self.dataset_id = dataset_id
48
- self.table_id = table_id
49
- self.file_path = file_path
50
-
51
- def load(self, fallback_data: Optional[Any] = None) -> Union[pa.Table, str]:
52
- """
53
- Auto‑detects the source configuration and loads the dataset.
54
-
55
- Priority order:
56
- 1. If `fallback_data` is provided, it is returned (with type normalisation).
57
- 2. If BigQuery identifiers are set, load from BigQuery.
58
- 3. If a local `file_path` is provided, load from file.
59
-
60
- Args:
61
- fallback_data: Optional pre‑loaded data (e.g., a PyArrow Table or any
62
- object that can be passed to DuckDB directly).
63
-
64
- Returns:
65
- A PyArrow Table, or a DuckDB scan macro string (when BigQuery client
66
- is not available and fallback is not provided).
67
- """
68
- # Scenario 1: Direct in‑memory object
69
- if fallback_data is not None:
70
- if isinstance(fallback_data, pa.Table):
71
- logger.info("📥 Ingesting directly provided in‑memory PyArrow Table.")
72
- return self._cast_table_numerics_to_float(fallback_data)
73
- logger.info("📥 Using provided fallback data (non‑Arrow) as‑is.")
74
- return fallback_data
75
-
76
- # Scenario 2: BigQuery
77
- if self.dataset_id and self.table_id:
78
- return self._load_from_bigquery()
79
-
80
- # Scenario 3: Local file
81
- if self.file_path:
82
- return self._load_from_file()
83
-
84
- raise ValueError(
85
- "Invalid Configuration: You must provide either a valid `file_path`, "
86
- "BigQuery identifiers, or pass an explicit `fallback_data` object."
87
- )
88
-
89
- @staticmethod
90
- def _cast_table_numerics_to_float(table: pa.Table) -> pa.Table:
91
- """
92
- Casts all numeric columns in a PyArrow Table to float64.
93
-
94
- This ensures consistent numerical precision across downstream operations.
95
-
96
- Args:
97
- table: Input PyArrow Table.
98
-
99
- Returns:
100
- A new table with numeric columns cast to float64.
101
- """
102
- if not isinstance(table, pa.Table):
103
- return table
104
-
105
- new_columns = []
106
- new_fields = []
107
-
108
- for i, field in enumerate(table.schema):
109
- # Check if the type is integer, floating, or decimal
110
- if (
111
- pa.types.is_integer(field.type)
112
- or pa.types.is_floating(field.type)
113
- or pa.types.is_decimal(field.type)
114
- ):
115
- try:
116
- casted_col = pc.cast(
117
- table.column(i), pa.float64(), safe=False
118
- )
119
- new_columns.append(casted_col)
120
- new_fields.append(
121
- pa.field(field.name, pa.float64(), nullable=field.nullable)
122
- )
123
- except Exception as e:
124
- logger.warning(
125
- f"⚠️ Failed to cast column {field.name} to float64. Reason: {e}"
126
- )
127
- new_columns.append(table.column(i))
128
- new_fields.append(field)
129
- else:
130
- new_columns.append(table.column(i))
131
- new_fields.append(field)
132
-
133
- return pa.Table.from_arrays(new_columns, schema=pa.schema(new_fields))
134
-
135
- def _load_from_file(self) -> pa.Table:
136
- """
137
- Parses a local file using high‑performance C++ Arrow readers.
138
-
139
- Supports: .parquet, .csv, .arrow / .feather, .xlsx / .xls.
140
-
141
- Returns:
142
- PyArrow Table with numeric columns cast to float64.
143
- """
144
- if not os.path.exists(self.file_path):
145
- raise FileNotFoundError(f"Data file not found at: {self.file_path}")
146
-
147
- ext = os.path.splitext(self.file_path)[-1].lower()
148
- logger.info(f"📂 Loading file: {self.file_path} (extension: {ext})")
149
-
150
- if ext == ".parquet":
151
- table = pa_pq.read_table(self.file_path)
152
- elif ext == ".csv":
153
- table = pa_csv.read_csv(self.file_path)
154
- elif ext in (".arrow", ".feather"):
155
- with pa.memory_map(self.file_path, "r") as source:
156
- table = pa.ipc.open_file(source).read_all()
157
- elif ext in (".xlsx", ".xls"):
158
- table = self._load_excel_to_arrow()
159
- else:
160
- raise ValueError(f"Unsupported file format: '{ext}'.")
161
-
162
- return self._cast_table_numerics_to_float(table)
163
-
164
- def _load_excel_to_arrow(self) -> pa.Table:
165
- """
166
- Parses an Excel file using openpyxl with positional column tracking.
167
-
168
- Returns:
169
- PyArrow Table.
170
- """
171
- logger.info("📊 Parsing Excel spreadsheet via positional column tracking...")
172
- try:
173
- import openpyxl
174
-
175
- wb = openpyxl.load_workbook(
176
- self.file_path, data_only=True, read_only=True
177
- )
178
- sheet = wb.active
179
- rows = sheet.iter_rows(values_only=True)
180
-
181
- headers = next(rows)
182
- if not headers:
183
- raise ValueError("The Excel file appears to be empty.")
184
-
185
- # Use column indices to prevent header‑shift corruption
186
- column_names = [
187
- f"{h}" if h is not None else f"_col_{i}"
188
- for i, h in enumerate(headers)
189
- ]
190
- data_columns = {name: [] for name in column_names}
191
-
192
- for row in rows:
193
- for i, name in enumerate(column_names):
194
- val = row[i] if i < len(row) else None
195
- data_columns[name].append(val)
196
-
197
- wb.close()
198
- return pa.Table.from_pydict(data_columns)
199
-
200
- except ImportError:
201
- raise ImportError(
202
- "Dependency missing: `pip install openpyxl` required for Excel files."
203
- )
204
-
205
- def _load_from_bigquery(self) -> Union[pa.Table, str]:
206
- """
207
- Resolves BigQuery ingestion using cost‑optimised metadata inspection.
208
-
209
- If the `google‑cloud‑bigquery` library is available, streams the table
210
- as a PyArrow Table. Otherwise, returns a DuckDB scan macro string for
211
- later execution (requires DuckDB's BigQuery extension).
212
-
213
- Returns:
214
- PyArrow Table or a DuckDB macro string.
215
- """
216
- full_bq_path = (
217
- f"{self.project_id}.{self.dataset_id}.{self.table_id}"
218
- if self.project_id
219
- else f"{self.dataset_id}.{self.table_id}"
220
- )
221
- logger.info(f"☁️ Initialising BigQuery client for: {full_bq_path}")
222
-
223
- try:
224
- from google.cloud import bigquery
225
-
226
- bq_client = bigquery.Client(project=self.project_id)
227
- full_table_ref = (
228
- f"{self.project_id or bq_client.project}."
229
- f"{self.dataset_id}.{self.table_id}"
230
- )
231
-
232
- # Fetch schema via get_table (cheaper than INFORMATION_SCHEMA)
233
- table = bq_client.get_table(full_table_ref)
234
-
235
- select_clauses = []
236
- for field in table.schema:
237
- # Cast numeric types to FLOAT64 for consistency
238
- if field.field_type in (
239
- "NUMERIC",
240
- "BIGNUMERIC",
241
- "DECIMAL",
242
- "INTEGER",
243
- "INT64",
244
- "FLOAT",
245
- "FLOAT64",
246
- ):
247
- select_clauses.append(
248
- f"SAFE_CAST(`{field.name}` AS FLOAT64) AS `{field.name}`"
249
- )
250
- else:
251
- select_clauses.append(f"`{field.name}`")
252
-
253
- query = f"SELECT {', '.join(select_clauses)} FROM `{full_table_ref}`"
254
- arrow_table = bq_client.query(query).to_arrow()
255
- logger.info(f"✅ Loaded {len(arrow_table):,} rows from BigQuery.")
256
- return arrow_table
257
-
258
- except ImportError:
259
- logger.warning(
260
- "⚠️ google‑cloud‑bigquery not found. Returning DuckDB scan macro. "
261
- "This macro requires the BigQuery extension to be loaded in DuckDB."
262
- )
263
- # Return a string that DuckDB can interpret as a table reference
264
- # (assuming the extension is loaded)
265
- return f"bigquery_scan('{self.project_id or 'default'}', '{self.dataset_id}', '{self.table_id}')"
266
-
267
- def stream_to_duckdb(self, db_path: str, path: str, encoding: Optional[str] = None) -> None:
268
- """Stream a local file straight into a DuckDB table `udl_data` at ``db_path``.
269
-
270
- Avoids materialising a full in-RAM PyArrow table (unlike the arrow-based
271
- load path) so multi-GB files stay memory-light — DuckDB reads directly
272
- from disk. Numeric columns are cast to DOUBLE on-disk to match the
273
- loader's float64 convention.
274
-
275
- Supported: .csv/.tsv (read_csv_auto), .parquet/.pq (read_parquet),
276
- .arrow/.feather (read_ipc).
277
- """
278
- ext = os.path.splitext(path)[1].lower()
279
- if ext not in (".csv", ".tsv", ".parquet", ".pq", ".arrow", ".feather"):
280
- raise ValueError(
281
- f"stream_to_duckdb does not support '{ext}' (use the arrow/Excel path)."
282
- )
283
- con = duckdb.connect(db_path)
284
- try:
285
- con.execute("DROP TABLE IF EXISTS udl_data")
286
- if ext in (".csv", ".tsv"):
287
- opts = "header=true, sample_size=-1"
288
- if encoding == "Latin-1":
289
- opts += ", encoding='LATIN-1'"
290
- con.execute(
291
- f"CREATE TABLE udl_data AS SELECT * FROM read_csv_auto(?, {opts})",
292
- [path],
293
- )
294
- elif ext in (".parquet", ".pq"):
295
- con.execute(
296
- "CREATE TABLE udl_data AS SELECT * FROM read_parquet(?)", [path]
297
- )
298
- else: # .arrow / .feather
299
- con.execute(
300
- "CREATE TABLE udl_data AS SELECT * FROM read_ipc(?)", [path]
301
- )
302
- # float64 convention, memory-light (on-disk ALTER per numeric column)
303
- cols = con.execute(
304
- "SELECT column_name, data_type FROM information_schema.columns "
305
- "WHERE table_name='udl_data'"
306
- ).fetchall()
307
- for name, dtype in cols:
308
- up = dtype.upper()
309
- if (
310
- any(
311
- k in up
312
- for k in ("INT", "FLOAT", "DECIMAL", "REAL", "DOUBLE", "NUMERIC")
313
- )
314
- and "DOUBLE" not in up
315
- ):
316
- con.execute(
317
- f'ALTER TABLE udl_data ALTER COLUMN "{name}" TYPE DOUBLE'
318
- )
319
- finally:
320
- con.close()
File without changes