rapidsegment 1.3.post3__tar.gz → 1.3.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/PKG-INFO +33 -26
- {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/README.md +32 -25
- {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/pyproject.toml +1 -1
- {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/pyproject.toml.orig +1 -1
- {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/__init__.py +2 -2
- {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/scorer.py +24 -7
- {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/pages/1_Data_Loader.py +0 -0
- rapidsegment-1.3.1/src/rapidsegment/utils/__init__.py +3 -0
- rapidsegment-1.3.1/src/rapidsegment/utils/data_loader.py +538 -0
- rapidsegment-1.3.post3/src/rapidsegment/utils/__init__.py +0 -3
- rapidsegment-1.3.post3/src/rapidsegment/utils/data_loader.py +0 -320
- {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/LICENSE +0 -0
- {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/builder.py +0 -0
- {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/py.typed +0 -0
- {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/.streamlit/config.toml +0 -0
- {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/__init__.py +0 -0
- {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/_exit.py +0 -0
- {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/_theme.py +0 -0
- {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/app.py +0 -0
- {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/pages/2_Workbench.py +0 -0
- {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/pages/3_Execution_Console.py +0 -0
- {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/pages/4_Results_Dashboard.py +0 -0
- {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/pages/5_Leaderboard.py +0 -0
- {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/pages/6_Arena.py +0 -0
- {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/pages/__init__.py +0 -0
- {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/utils/on_gcp_feature_selection.py +0 -0
- {rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/utils/undersampler.sql +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rapidsegment
|
|
3
|
-
Version: 1.3.
|
|
3
|
+
Version: 1.3.1
|
|
4
4
|
Summary: A fast, multi-segment population scoring and combinatorial heuristic rule extraction engine.
|
|
5
5
|
Author: Bishwarup Biswas
|
|
6
6
|
Author-email: Bishwarup Biswas <bishwarup1429@gmail.com>
|
|
@@ -331,12 +331,15 @@ flowchart TD
|
|
|
331
331
|
|
|
332
332
|
I --> J["Validate on residual current_df: COUNT + SUM WHERE sql_filter"]
|
|
333
333
|
|
|
334
|
-
J --> K{"
|
|
334
|
+
J --> K{"Volume floors (growth): count ≥ min_sample_size AND events ≥ min_events?"}
|
|
335
335
|
|
|
336
336
|
K -- No --> L["Reject and try next candidate, or stop"]
|
|
337
337
|
L --> C
|
|
338
338
|
|
|
339
|
-
K -- Yes -->
|
|
339
|
+
K -- Yes --> K2{"Acceptance floor: lift ≥ min_lift?"}
|
|
340
|
+
|
|
341
|
+
K2 -- No --> L
|
|
342
|
+
K2 -- Yes --> M["Store segment with actual counts from residual"]
|
|
340
343
|
|
|
341
344
|
M --> N["Update feature usage tracker"]
|
|
342
345
|
|
|
@@ -387,13 +390,13 @@ The engine evaluates combinations in a layered fashion:
|
|
|
387
390
|
|
|
388
391
|
```mermaid
|
|
389
392
|
flowchart LR
|
|
390
|
-
A[Top
|
|
391
|
-
B -->|Only
|
|
392
|
-
C -->|Only pairs that
|
|
393
|
+
A[Top-N Features<br/>top_n_vars] --> B[1‑Way Checks]
|
|
394
|
+
B -->|Only features with bins clearing volume floors| C[2‑Way Combinations]
|
|
395
|
+
C -->|Only variable pairs that cleared volume floors| D[3‑Way Combinations]
|
|
393
396
|
D --> E[Final Candidate Rules]
|
|
394
397
|
```
|
|
395
398
|
|
|
396
|
-
|
|
399
|
+
The pruning trigger at each layer is **volume only** — a rule clears the floor when `count ≥ min_sample_size` **and** `events ≥ min_events`. `min_lift` is deliberately **not** part of pruning: it is a hard *acceptance* floor applied to candidate rules afterwards, at the grid shortlist and the final raw-SQL validation (see [How 1-Way → 2-Way → 3-Way Segment Search Works](#how-1-way--2-way--3-way-segment-search-works)). A feature leaves the search only when **none of its bins** clears the volume floors. That keeps the search small while still allowing a 3-way rule to carry more lift than any of the parts it was grown from.
|
|
397
400
|
|
|
398
401
|
## How 1-Way → 2-Way → 3-Way Segment Search Works
|
|
399
402
|
|
|
@@ -401,34 +404,34 @@ RapidSegment builds candidate segments in layers: it tests single features first
|
|
|
401
404
|
|
|
402
405
|
### Worked example — from 1-way to 3-way on real-looking data
|
|
403
406
|
|
|
404
|
-
Say the target is `churned` (1 = customer left), the overall base rate is **20%** (2,000 of 10,000 customers churned), `
|
|
407
|
+
Say the target is `churned` (1 = customer left), the overall base rate is **20%** (2,000 of 10,000 customers churned), and the floors are `min_sample_size = 300`, `min_events = 30`, `min_lift = 1.5`. Three binned features are in play: `tenure_bin`, `plan_type`, `support_tickets_bin`.
|
|
405
408
|
|
|
406
409
|
#### Step 1 — 1-way: test each bin of each feature alone
|
|
407
410
|
|
|
408
|
-
Every individual bin is checked against the base rate.
|
|
409
|
-
|
|
410
|
-
| Rule (1-way) | Count | Churn rate | Lift |
|
|
411
|
-
|
|
412
|
-
| `tenure_bin = [0-3mo]` | 1,200 | 42% | 2.1x | ✅ |
|
|
413
|
-
| `plan_type = [Basic]` | 900 | 35% | 1.75x | ✅ |
|
|
414
|
-
| `support_tickets_bin = [3+]` | 600 | 55% | 2.75x | ✅ |
|
|
415
|
-
| `plan_type = [Premium]` | 800 | 8% | 0.4x | ❌
|
|
416
|
-
| `tenure_bin = [12mo+]` | 3,000 | 6% | 0.3x | ❌ |
|
|
411
|
+
Every individual bin is checked against the base rate. To **survive the pruning gate** a rule must clear the *volume* floors — `count ≥ min_sample_size` **and** `events ≥ min_events`. Rows and events are anti‑monotone (adding a condition can only shrink the population), so pruning on them is safe. `lift = segment_rate / base_rate` is a separate **acceptance** floor (`min_lift`) applied later — surviving the pruning gate does not by itself make a rule a segment:
|
|
412
|
+
|
|
413
|
+
| Rule (1-way) | Count | Churn rate | Lift | Pruning gate (count+events) | Accepted as segment (lift)? |
|
|
414
|
+
|---|---|---|---|:---:|:---:|
|
|
415
|
+
| `tenure_bin = [0-3mo]` | 1,200 | 42% | 2.1x | ✅ count 1,200 ≥ 300, events ≥ 30 | ✅ |
|
|
416
|
+
| `plan_type = [Basic]` | 900 | 35% | 1.75x | ✅ count 900 ≥ 300, events ≥ 30 | ✅ |
|
|
417
|
+
| `support_tickets_bin = [3+]` | 600 | 55% | 2.75x | ✅ count 600 ≥ 300, events ≥ 30 | ✅ |
|
|
418
|
+
| `plan_type = [Premium]` | 800 | 8% | 0.4x | ✅ count 800 ≥ 300, events ≥ 30 | ❌ lift 0.4x < 1.5 (protective, not risky) |
|
|
419
|
+
| `tenure_bin = [12mo+]` | 3,000 | 6% | 0.3x | ✅ count 3,000 ≥ 300, events ≥ 30 | ❌ lift 0.3x < 1.5 |
|
|
417
420
|
|
|
418
|
-
|
|
421
|
+
Every bin above clears the volume gate, so all three features (`tenure_bin`, `plan_type`, `support_tickets_bin`) stay in the search. Only `[0-3mo]`, `[Basic]`, and `[3+ tickets]` — call them **A**, **B**, **C** for short — also clear the 1‑way lift floor. The bins that failed lift (`[Premium]`, `[12mo+]`) are **not** dropped from the search: Apriori pruning here works per **feature**, never per bin. Those bin values are still aggregated into later 2‑way / 3‑way combinations and must themselves clear the volume floors and the lift floor to be accepted — but an individual 1‑way lift shortfall never prunes the feature.
|
|
419
422
|
|
|
420
423
|
#### Step 2 — 2-way: pair up only the survivors
|
|
421
424
|
|
|
422
425
|
With 3 survivors there are `C(3,2) = 3` possible pairs: `A+B`, `A+C`, `B+C`. Each pair is aggregated as its own joint segment:
|
|
423
426
|
|
|
424
|
-
| Rule (2-way) | Count | Churn rate | Lift |
|
|
425
|
-
|
|
426
|
-
| `A+B` = `[0-3mo] AND [Basic]` | 420 | 51% | 2.55x | ✅ |
|
|
427
|
-
| `A+C` = `[0-3mo] AND [3+ tickets]` | 310 | 58% | 2.9x | ✅ |
|
|
428
|
-
| `B+C` = `[Basic] AND [3+ tickets]` | 180 | 60% | 3.0x | ❌ — count 180 < min_sample_size 300 |
|
|
429
|
-
|
|
430
|
-
Notice `B+C` actually has the *highest* churn rate and lift of the three pairs — but it's still rejected, because too few customers (180) fall into that exact overlap to trust the number. This is the key trade-off: **survival is about count AND lift together, not lift alone.**
|
|
427
|
+
| Rule (2-way) | Count | Churn rate | Lift | Passes volume gate? | Meets lift floor? |
|
|
428
|
+
|---|---|---|---|:---:|:---:|
|
|
429
|
+
| `A+B` = `[0-3mo] AND [Basic]` | 420 | 51% | 2.55x | ✅ | ✅ |
|
|
430
|
+
| `A+C` = `[0-3mo] AND [3+ tickets]` | 310 | 58% | 2.9x | ✅ | ✅ |
|
|
431
|
+
| `B+C` = `[Basic] AND [3+ tickets]` | 180 | 60% | 3.0x | ❌ — count 180 < min_sample_size 300 | — |
|
|
431
432
|
|
|
433
|
+
Notice `B+C` actually has the *highest* churn rate and lift of the three pairs — but it's still rejected, because too few customers (180) fall into that exact overlap to trust the number. This is the key trade-off inside the volume gate: **a rule can fail pruning on count alone, even with the best lift in the room.** `B+C` is pruned before `min_lift` even gets a say.
|
|
434
|
+
|
|
432
435
|
Survivors: `valid_2way_sets = { {A,B}, {A,C} }`.
|
|
433
436
|
|
|
434
437
|
#### Step 3 — 3-way: only try triplets where every pair inside them already passed
|
|
@@ -442,15 +445,18 @@ With 3 bins there's only one possible triplet: `A+B+C`. Before RapidSegment even
|
|
|
442
445
|
| `{B,C}` | ❌ (rejected in Step 2 for low count) |
|
|
443
446
|
|
|
444
447
|
Because `{B,C}` never passed, the triplet `A+B+C` is **skipped entirely** — it is never even aggregated, no matter how strong its true joint churn rate might be. This is the pruning payoff: instead of testing every possible triplet from scratch, the engine only tests triplets whose *every* pairwise sub-relationship already proved itself statistically solid on its own.
|
|
448
|
+
|
|
449
|
+
> **Granularity note:** the engine keys `valid_2way_sets` on **variable pairs**, not bin pairs. A variable pair qualifies for 3-way growth as soon as *any* joint bin combination of those two variables clears the volume floors. In this example `plan_type` and `tenure_bin` qualify (their `A+B` overlap passes), while `tenure_bin` and `support_tickets_bin` never produce a passing overlap (the `B+C` case at 180 rows), so the triplet isn't grown. The story above shows that same idea at bin-pair level for readability.
|
|
445
450
|
|
|
446
451
|
### Why prune this way instead of just testing every triplet directly?
|
|
447
452
|
|
|
448
453
|
- **Speed:** with `top_n_vars = 15`, testing all triplets directly is `C(15,3) = 455` SQL aggregations. Pruning by pairwise survival first can cut that dramatically, since most triplets get eliminated before ever touching the data.
|
|
449
454
|
- **The cost:** a genuinely strong 3-way interaction can be missed if one of its underlying pairs happened to fall just under `min_sample_size` (as `{B,C}` did above at count 180) — even if the full triplet would have had a healthy count. This is the same trade-off classic Apriori pruning makes in market-basket analysis: cheap, scalable, but not exhaustive.
|
|
455
|
+
- **Why not prune on `min_lift` too?** Because lift can *rise* when you add a condition — a 3-way can beat every pair it was grown from. Pruning on lift would throw away exactly those strong interactions. Rows and events never rise when a rule narrows, so they're what pruning uses; `min_lift` is applied only afterward, as an acceptance check (grid shortlist + final raw validation).
|
|
450
456
|
---
|
|
451
457
|
|
|
452
458
|
### 3. Grid Search
|
|
453
|
-
For each iteration, the engine sweeps over a user‑defined grid of `(min_sample_size, min_lift)` values. Each grid point
|
|
459
|
+
For each iteration, the engine sweeps over a user‑defined grid of `(min_sample_size, min_lift)` values. Each grid point keeps the rules that clear its `count` and `lift` floors, and the top rule for that config (by `sort_priority`) becomes a candidate champion. The champions are ordered and the first to pass the raw‑residual validation (next section) becomes the iteration's champion.
|
|
454
460
|
|
|
455
461
|
### 4. Champion Validation & Extraction
|
|
456
462
|
The champion’s SQL filter is validated against the **raw residual** to ensure it meets the absolute hard constraints. Only then is it accepted.
|
|
@@ -651,3 +657,4 @@ _Independent, open‑source, and ready for production._
|
|
|
651
657
|
|
|
652
658
|
|
|
653
659
|
|
|
660
|
+
|
|
@@ -293,12 +293,15 @@ flowchart TD
|
|
|
293
293
|
|
|
294
294
|
I --> J["Validate on residual current_df: COUNT + SUM WHERE sql_filter"]
|
|
295
295
|
|
|
296
|
-
J --> K{"
|
|
296
|
+
J --> K{"Volume floors (growth): count ≥ min_sample_size AND events ≥ min_events?"}
|
|
297
297
|
|
|
298
298
|
K -- No --> L["Reject and try next candidate, or stop"]
|
|
299
299
|
L --> C
|
|
300
300
|
|
|
301
|
-
K -- Yes -->
|
|
301
|
+
K -- Yes --> K2{"Acceptance floor: lift ≥ min_lift?"}
|
|
302
|
+
|
|
303
|
+
K2 -- No --> L
|
|
304
|
+
K2 -- Yes --> M["Store segment with actual counts from residual"]
|
|
302
305
|
|
|
303
306
|
M --> N["Update feature usage tracker"]
|
|
304
307
|
|
|
@@ -349,13 +352,13 @@ The engine evaluates combinations in a layered fashion:
|
|
|
349
352
|
|
|
350
353
|
```mermaid
|
|
351
354
|
flowchart LR
|
|
352
|
-
A[Top
|
|
353
|
-
B -->|Only
|
|
354
|
-
C -->|Only pairs that
|
|
355
|
+
A[Top-N Features<br/>top_n_vars] --> B[1‑Way Checks]
|
|
356
|
+
B -->|Only features with bins clearing volume floors| C[2‑Way Combinations]
|
|
357
|
+
C -->|Only variable pairs that cleared volume floors| D[3‑Way Combinations]
|
|
355
358
|
D --> E[Final Candidate Rules]
|
|
356
359
|
```
|
|
357
360
|
|
|
358
|
-
|
|
361
|
+
The pruning trigger at each layer is **volume only** — a rule clears the floor when `count ≥ min_sample_size` **and** `events ≥ min_events`. `min_lift` is deliberately **not** part of pruning: it is a hard *acceptance* floor applied to candidate rules afterwards, at the grid shortlist and the final raw-SQL validation (see [How 1-Way → 2-Way → 3-Way Segment Search Works](#how-1-way--2-way--3-way-segment-search-works)). A feature leaves the search only when **none of its bins** clears the volume floors. That keeps the search small while still allowing a 3-way rule to carry more lift than any of the parts it was grown from.
|
|
359
362
|
|
|
360
363
|
## How 1-Way → 2-Way → 3-Way Segment Search Works
|
|
361
364
|
|
|
@@ -363,34 +366,34 @@ RapidSegment builds candidate segments in layers: it tests single features first
|
|
|
363
366
|
|
|
364
367
|
### Worked example — from 1-way to 3-way on real-looking data
|
|
365
368
|
|
|
366
|
-
Say the target is `churned` (1 = customer left), the overall base rate is **20%** (2,000 of 10,000 customers churned), `
|
|
369
|
+
Say the target is `churned` (1 = customer left), the overall base rate is **20%** (2,000 of 10,000 customers churned), and the floors are `min_sample_size = 300`, `min_events = 30`, `min_lift = 1.5`. Three binned features are in play: `tenure_bin`, `plan_type`, `support_tickets_bin`.
|
|
367
370
|
|
|
368
371
|
#### Step 1 — 1-way: test each bin of each feature alone
|
|
369
372
|
|
|
370
|
-
Every individual bin is checked against the base rate.
|
|
371
|
-
|
|
372
|
-
| Rule (1-way) | Count | Churn rate | Lift |
|
|
373
|
-
|
|
374
|
-
| `tenure_bin = [0-3mo]` | 1,200 | 42% | 2.1x | ✅ |
|
|
375
|
-
| `plan_type = [Basic]` | 900 | 35% | 1.75x | ✅ |
|
|
376
|
-
| `support_tickets_bin = [3+]` | 600 | 55% | 2.75x | ✅ |
|
|
377
|
-
| `plan_type = [Premium]` | 800 | 8% | 0.4x | ❌
|
|
378
|
-
| `tenure_bin = [12mo+]` | 3,000 | 6% | 0.3x | ❌ |
|
|
373
|
+
Every individual bin is checked against the base rate. To **survive the pruning gate** a rule must clear the *volume* floors — `count ≥ min_sample_size` **and** `events ≥ min_events`. Rows and events are anti‑monotone (adding a condition can only shrink the population), so pruning on them is safe. `lift = segment_rate / base_rate` is a separate **acceptance** floor (`min_lift`) applied later — surviving the pruning gate does not by itself make a rule a segment:
|
|
374
|
+
|
|
375
|
+
| Rule (1-way) | Count | Churn rate | Lift | Pruning gate (count+events) | Accepted as segment (lift)? |
|
|
376
|
+
|---|---|---|---|:---:|:---:|
|
|
377
|
+
| `tenure_bin = [0-3mo]` | 1,200 | 42% | 2.1x | ✅ count 1,200 ≥ 300, events ≥ 30 | ✅ |
|
|
378
|
+
| `plan_type = [Basic]` | 900 | 35% | 1.75x | ✅ count 900 ≥ 300, events ≥ 30 | ✅ |
|
|
379
|
+
| `support_tickets_bin = [3+]` | 600 | 55% | 2.75x | ✅ count 600 ≥ 300, events ≥ 30 | ✅ |
|
|
380
|
+
| `plan_type = [Premium]` | 800 | 8% | 0.4x | ✅ count 800 ≥ 300, events ≥ 30 | ❌ lift 0.4x < 1.5 (protective, not risky) |
|
|
381
|
+
| `tenure_bin = [12mo+]` | 3,000 | 6% | 0.3x | ✅ count 3,000 ≥ 300, events ≥ 30 | ❌ lift 0.3x < 1.5 |
|
|
379
382
|
|
|
380
|
-
|
|
383
|
+
Every bin above clears the volume gate, so all three features (`tenure_bin`, `plan_type`, `support_tickets_bin`) stay in the search. Only `[0-3mo]`, `[Basic]`, and `[3+ tickets]` — call them **A**, **B**, **C** for short — also clear the 1‑way lift floor. The bins that failed lift (`[Premium]`, `[12mo+]`) are **not** dropped from the search: Apriori pruning here works per **feature**, never per bin. Those bin values are still aggregated into later 2‑way / 3‑way combinations and must themselves clear the volume floors and the lift floor to be accepted — but an individual 1‑way lift shortfall never prunes the feature.
|
|
381
384
|
|
|
382
385
|
#### Step 2 — 2-way: pair up only the survivors
|
|
383
386
|
|
|
384
387
|
With 3 survivors there are `C(3,2) = 3` possible pairs: `A+B`, `A+C`, `B+C`. Each pair is aggregated as its own joint segment:
|
|
385
388
|
|
|
386
|
-
| Rule (2-way) | Count | Churn rate | Lift |
|
|
387
|
-
|
|
388
|
-
| `A+B` = `[0-3mo] AND [Basic]` | 420 | 51% | 2.55x | ✅ |
|
|
389
|
-
| `A+C` = `[0-3mo] AND [3+ tickets]` | 310 | 58% | 2.9x | ✅ |
|
|
390
|
-
| `B+C` = `[Basic] AND [3+ tickets]` | 180 | 60% | 3.0x | ❌ — count 180 < min_sample_size 300 |
|
|
391
|
-
|
|
392
|
-
Notice `B+C` actually has the *highest* churn rate and lift of the three pairs — but it's still rejected, because too few customers (180) fall into that exact overlap to trust the number. This is the key trade-off: **survival is about count AND lift together, not lift alone.**
|
|
389
|
+
| Rule (2-way) | Count | Churn rate | Lift | Passes volume gate? | Meets lift floor? |
|
|
390
|
+
|---|---|---|---|:---:|:---:|
|
|
391
|
+
| `A+B` = `[0-3mo] AND [Basic]` | 420 | 51% | 2.55x | ✅ | ✅ |
|
|
392
|
+
| `A+C` = `[0-3mo] AND [3+ tickets]` | 310 | 58% | 2.9x | ✅ | ✅ |
|
|
393
|
+
| `B+C` = `[Basic] AND [3+ tickets]` | 180 | 60% | 3.0x | ❌ — count 180 < min_sample_size 300 | — |
|
|
393
394
|
|
|
395
|
+
Notice `B+C` actually has the *highest* churn rate and lift of the three pairs — but it's still rejected, because too few customers (180) fall into that exact overlap to trust the number. This is the key trade-off inside the volume gate: **a rule can fail pruning on count alone, even with the best lift in the room.** `B+C` is pruned before `min_lift` even gets a say.
|
|
396
|
+
|
|
394
397
|
Survivors: `valid_2way_sets = { {A,B}, {A,C} }`.
|
|
395
398
|
|
|
396
399
|
#### Step 3 — 3-way: only try triplets where every pair inside them already passed
|
|
@@ -404,15 +407,18 @@ With 3 bins there's only one possible triplet: `A+B+C`. Before RapidSegment even
|
|
|
404
407
|
| `{B,C}` | ❌ (rejected in Step 2 for low count) |
|
|
405
408
|
|
|
406
409
|
Because `{B,C}` never passed, the triplet `A+B+C` is **skipped entirely** — it is never even aggregated, no matter how strong its true joint churn rate might be. This is the pruning payoff: instead of testing every possible triplet from scratch, the engine only tests triplets whose *every* pairwise sub-relationship already proved itself statistically solid on its own.
|
|
410
|
+
|
|
411
|
+
> **Granularity note:** the engine keys `valid_2way_sets` on **variable pairs**, not bin pairs. A variable pair qualifies for 3-way growth as soon as *any* joint bin combination of those two variables clears the volume floors. In this example `plan_type` and `tenure_bin` qualify (their `A+B` overlap passes), while `tenure_bin` and `support_tickets_bin` never produce a passing overlap (the `B+C` case at 180 rows), so the triplet isn't grown. The story above shows that same idea at bin-pair level for readability.
|
|
407
412
|
|
|
408
413
|
### Why prune this way instead of just testing every triplet directly?
|
|
409
414
|
|
|
410
415
|
- **Speed:** with `top_n_vars = 15`, testing all triplets directly is `C(15,3) = 455` SQL aggregations. Pruning by pairwise survival first can cut that dramatically, since most triplets get eliminated before ever touching the data.
|
|
411
416
|
- **The cost:** a genuinely strong 3-way interaction can be missed if one of its underlying pairs happened to fall just under `min_sample_size` (as `{B,C}` did above at count 180) — even if the full triplet would have had a healthy count. This is the same trade-off classic Apriori pruning makes in market-basket analysis: cheap, scalable, but not exhaustive.
|
|
417
|
+
- **Why not prune on `min_lift` too?** Because lift can *rise* when you add a condition — a 3-way can beat every pair it was grown from. Pruning on lift would throw away exactly those strong interactions. Rows and events never rise when a rule narrows, so they're what pruning uses; `min_lift` is applied only afterward, as an acceptance check (grid shortlist + final raw validation).
|
|
412
418
|
---
|
|
413
419
|
|
|
414
420
|
### 3. Grid Search
|
|
415
|
-
For each iteration, the engine sweeps over a user‑defined grid of `(min_sample_size, min_lift)` values. Each grid point
|
|
421
|
+
For each iteration, the engine sweeps over a user‑defined grid of `(min_sample_size, min_lift)` values. Each grid point keeps the rules that clear its `count` and `lift` floors, and the top rule for that config (by `sort_priority`) becomes a candidate champion. The champions are ordered and the first to pass the raw‑residual validation (next section) becomes the iteration's champion.
|
|
416
422
|
|
|
417
423
|
### 4. Champion Validation & Extraction
|
|
418
424
|
The champion’s SQL filter is validated against the **raw residual** to ensure it meets the absolute hard constraints. Only then is it accepted.
|
|
@@ -613,3 +619,4 @@ _Independent, open‑source, and ready for production._
|
|
|
613
619
|
|
|
614
620
|
|
|
615
621
|
|
|
622
|
+
|
|
@@ -8,8 +8,8 @@ except PackageNotFoundError:
|
|
|
8
8
|
|
|
9
9
|
__author__ = "Bishwarup Biswas <bishwarup1429@gmail.com>"
|
|
10
10
|
|
|
11
|
-
from .utils import UniversalDataLoader
|
|
11
|
+
from .utils import UniversalDataLoader, duckdb_to_arrow
|
|
12
12
|
from .builder import StrategicSegmentBuilder
|
|
13
13
|
from .scorer import StrategicSegmentScore
|
|
14
14
|
|
|
15
|
-
__all__ = ["UniversalDataLoader", "StrategicSegmentBuilder", "StrategicSegmentScore"]
|
|
15
|
+
__all__ = ["UniversalDataLoader", "duckdb_to_arrow", "StrategicSegmentBuilder", "StrategicSegmentScore"]
|
|
@@ -11,6 +11,7 @@ Python Version: 3.11+
|
|
|
11
11
|
import json
|
|
12
12
|
import logging
|
|
13
13
|
import os
|
|
14
|
+
import re
|
|
14
15
|
import tempfile
|
|
15
16
|
import uuid
|
|
16
17
|
from datetime import datetime
|
|
@@ -60,23 +61,37 @@ class StrategicSegmentScore:
|
|
|
60
61
|
data: Any,
|
|
61
62
|
export_path: str = f"scored_experiment_{timestamp}.json",
|
|
62
63
|
db_path: Optional[str] = None,
|
|
64
|
+
table_name: str = "df",
|
|
63
65
|
) -> Dict[str, Any]:
|
|
64
66
|
"""
|
|
65
67
|
Calculates harmonic weights and derives decile boundaries via vectorised execution.
|
|
66
68
|
|
|
67
69
|
Args:
|
|
68
|
-
data: Input data
|
|
70
|
+
data: Input data. Either the path to a DuckDB database file containing
|
|
71
|
+
``table_name`` (zero-copy ATTACH), or any DuckDB-registrable object
|
|
72
|
+
(pandas DataFrame, PyArrow Table, duckdb Relation, ...).
|
|
69
73
|
export_path: File path to save the model artifact JSON.
|
|
70
|
-
db_path: Optional path to a persistent DuckDB file/connection to reuse
|
|
71
|
-
single data artifact (e.g. the builder's ``db_path``). When
|
|
72
|
-
temporary file-backed DB is created under the system
|
|
73
|
-
after the run, avoiding CWD pollution.
|
|
74
|
+
db_path: Optional path to a persistent DuckDB file/connection to reuse
|
|
75
|
+
as the single data artifact (e.g. the builder's ``db_path``). When
|
|
76
|
+
omitted, a temporary file-backed DB is created under the system
|
|
77
|
+
temp dir and removed after the run, avoiding CWD pollution.
|
|
78
|
+
table_name: Table or view to read from the file handed in ``data``
|
|
79
|
+
(default 'df'). Point it at any table you prepared, e.g.
|
|
80
|
+
`table_name='predicted'` after creating a scored table named
|
|
81
|
+
'predicted' inside the database.
|
|
74
82
|
|
|
75
83
|
Returns:
|
|
76
84
|
Dictionary containing model metadata, segment weights, and decile thresholds.
|
|
77
85
|
"""
|
|
78
86
|
logger.info("🚀 Initialising out‑of‑core DuckDB scorecard engine...")
|
|
79
87
|
|
|
88
|
+
if not isinstance(table_name, str) or not re.fullmatch(
|
|
89
|
+
r"[A-Za-z_][A-Za-z0-9_]*", table_name
|
|
90
|
+
):
|
|
91
|
+
raise ValueError(
|
|
92
|
+
f"table_name must be a valid SQL identifier; got {table_name!r}"
|
|
93
|
+
)
|
|
94
|
+
|
|
80
95
|
# Use file‑backed storage for large datasets. Reuse a caller-supplied DB when
|
|
81
96
|
# provided; otherwise create a unique temp file and clean it up afterwards so we
|
|
82
97
|
# never leak a `score_experiment_*.db` into the current working directory.
|
|
@@ -86,7 +101,7 @@ class StrategicSegmentScore:
|
|
|
86
101
|
db_path = os.path.join(
|
|
87
102
|
tempfile.gettempdir(), f"rapidsegment_score_{unique_id}.duckdb"
|
|
88
103
|
)
|
|
89
|
-
if os.path.exists(db_path):
|
|
104
|
+
if own_db and os.path.exists(db_path):
|
|
90
105
|
os.remove(db_path)
|
|
91
106
|
ctx = duckdb.connect(db_path)
|
|
92
107
|
if isinstance(data, str):
|
|
@@ -95,7 +110,9 @@ class StrategicSegmentScore:
|
|
|
95
110
|
# Attach it read-only instead of materialising it into Python.
|
|
96
111
|
src_path = data.replace("\\", "/")
|
|
97
112
|
ctx.execute(f"ATTACH '{src_path}' AS __rs_src (READ_ONLY)")
|
|
98
|
-
ctx.execute(
|
|
113
|
+
ctx.execute(
|
|
114
|
+
f'CREATE OR REPLACE TABLE df AS SELECT * FROM __rs_src."{table_name}"'
|
|
115
|
+
)
|
|
99
116
|
else:
|
|
100
117
|
ctx.execute("CREATE OR REPLACE TABLE df AS SELECT * FROM data")
|
|
101
118
|
|
|
File without changes
|
|
@@ -0,0 +1,538 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Unified Data Ingestion Layer
|
|
3
|
+
============================
|
|
4
|
+
Multi‑format data loader supporting Local Files (CSV, Parquet, Arrow, Excel),
|
|
5
|
+
In‑Memory PyArrow Tables, and Google Cloud BigQuery Storage API streams.
|
|
6
|
+
|
|
7
|
+
Author: Bishwarup Biswas + Gemini + DeepSeek
|
|
8
|
+
Python Version: 3.9+
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
import logging
|
|
12
|
+
import os
|
|
13
|
+
import re
|
|
14
|
+
import tempfile
|
|
15
|
+
import uuid
|
|
16
|
+
from typing import Any, Optional, Union
|
|
17
|
+
|
|
18
|
+
import duckdb
|
|
19
|
+
import pyarrow as pa
|
|
20
|
+
import pyarrow.compute as pc
|
|
21
|
+
import pyarrow.csv as pa_csv
|
|
22
|
+
import pyarrow.parquet as pa_pq
|
|
23
|
+
|
|
24
|
+
logger = logging.getLogger("StrategicEngine.DataLoader")
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class UniversalDataLoader:
|
|
28
|
+
"""
|
|
29
|
+
Handles multi‑source data ingestion, normalising inputs into highly optimised
|
|
30
|
+
in‑memory PyArrow Tables suitable for vectorised downstream compute engines.
|
|
31
|
+
|
|
32
|
+
The loader automatically detects the source type based on constructor arguments.
|
|
33
|
+
If a `fallback_data` object is passed to `load()`, it takes precedence.
|
|
34
|
+
|
|
35
|
+
Args:
|
|
36
|
+
project_id: (Optional) GCP project ID for BigQuery.
|
|
37
|
+
dataset_id: (Optional) BigQuery dataset ID.
|
|
38
|
+
table_id: (Optional) BigQuery table ID.
|
|
39
|
+
file_path: (Optional) Local file path (CSV, Parquet, Arrow/Feather, Excel).
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
def __init__(
|
|
43
|
+
self,
|
|
44
|
+
project_id: Optional[str] = None,
|
|
45
|
+
dataset_id: Optional[str] = None,
|
|
46
|
+
table_id: Optional[str] = None,
|
|
47
|
+
file_path: Optional[str] = None,
|
|
48
|
+
) -> None:
|
|
49
|
+
self.project_id = project_id
|
|
50
|
+
self.dataset_id = dataset_id
|
|
51
|
+
self.table_id = table_id
|
|
52
|
+
self.file_path = file_path
|
|
53
|
+
|
|
54
|
+
def load(self, fallback_data: Optional[Any] = None) -> Union[pa.Table, str]:
|
|
55
|
+
"""
|
|
56
|
+
Auto‑detects the source configuration and loads the dataset.
|
|
57
|
+
|
|
58
|
+
Priority order:
|
|
59
|
+
1. If `fallback_data` is provided, it is returned (with type normalisation).
|
|
60
|
+
2. If BigQuery identifiers are set, load from BigQuery.
|
|
61
|
+
3. If a local `file_path` is provided, load from file.
|
|
62
|
+
|
|
63
|
+
Args:
|
|
64
|
+
fallback_data: Optional pre‑loaded data (e.g., a PyArrow Table or any
|
|
65
|
+
object that can be passed to DuckDB directly).
|
|
66
|
+
|
|
67
|
+
Returns:
|
|
68
|
+
A PyArrow Table, or a DuckDB scan macro string (when BigQuery client
|
|
69
|
+
is not available and fallback is not provided).
|
|
70
|
+
"""
|
|
71
|
+
# Scenario 1: Direct in‑memory object
|
|
72
|
+
if fallback_data is not None:
|
|
73
|
+
if isinstance(fallback_data, pa.Table):
|
|
74
|
+
logger.info("📥 Ingesting directly provided in‑memory PyArrow Table.")
|
|
75
|
+
return self._cast_table_numerics_to_float(fallback_data)
|
|
76
|
+
logger.info("📥 Using provided fallback data (non‑Arrow) as‑is.")
|
|
77
|
+
return fallback_data
|
|
78
|
+
|
|
79
|
+
# Scenario 2: BigQuery
|
|
80
|
+
if self.dataset_id and self.table_id:
|
|
81
|
+
return self._load_from_bigquery()
|
|
82
|
+
|
|
83
|
+
# Scenario 3: Local file
|
|
84
|
+
if self.file_path:
|
|
85
|
+
return self._load_from_file()
|
|
86
|
+
|
|
87
|
+
raise ValueError(
|
|
88
|
+
"Invalid Configuration: You must provide either a valid `file_path`, "
|
|
89
|
+
"BigQuery identifiers, or pass an explicit `fallback_data` object."
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
@staticmethod
|
|
93
|
+
def _cast_table_numerics_to_float(table: pa.Table) -> pa.Table:
|
|
94
|
+
"""
|
|
95
|
+
Casts all numeric columns in a PyArrow Table to float64.
|
|
96
|
+
|
|
97
|
+
This ensures consistent numerical precision across downstream operations.
|
|
98
|
+
|
|
99
|
+
Args:
|
|
100
|
+
table: Input PyArrow Table.
|
|
101
|
+
|
|
102
|
+
Returns:
|
|
103
|
+
A new table with numeric columns cast to float64.
|
|
104
|
+
"""
|
|
105
|
+
if not isinstance(table, pa.Table):
|
|
106
|
+
return table
|
|
107
|
+
|
|
108
|
+
new_columns = []
|
|
109
|
+
new_fields = []
|
|
110
|
+
|
|
111
|
+
for i, field in enumerate(table.schema):
|
|
112
|
+
# Check if the type is integer, floating, or decimal
|
|
113
|
+
if (
|
|
114
|
+
pa.types.is_integer(field.type)
|
|
115
|
+
or pa.types.is_floating(field.type)
|
|
116
|
+
or pa.types.is_decimal(field.type)
|
|
117
|
+
):
|
|
118
|
+
try:
|
|
119
|
+
casted_col = pc.cast(
|
|
120
|
+
table.column(i), pa.float64(), safe=False
|
|
121
|
+
)
|
|
122
|
+
new_columns.append(casted_col)
|
|
123
|
+
new_fields.append(
|
|
124
|
+
pa.field(field.name, pa.float64(), nullable=field.nullable)
|
|
125
|
+
)
|
|
126
|
+
except Exception as e:
|
|
127
|
+
logger.warning(
|
|
128
|
+
f"⚠️ Failed to cast column {field.name} to float64. Reason: {e}"
|
|
129
|
+
)
|
|
130
|
+
new_columns.append(table.column(i))
|
|
131
|
+
new_fields.append(field)
|
|
132
|
+
else:
|
|
133
|
+
new_columns.append(table.column(i))
|
|
134
|
+
new_fields.append(field)
|
|
135
|
+
|
|
136
|
+
return pa.Table.from_arrays(new_columns, schema=pa.schema(new_fields))
|
|
137
|
+
|
|
138
|
+
def _load_from_file(self) -> pa.Table:
|
|
139
|
+
"""
|
|
140
|
+
Parses a local file using high‑performance C++ Arrow readers.
|
|
141
|
+
|
|
142
|
+
Supports: .parquet, .csv, .arrow / .feather, .xlsx / .xls.
|
|
143
|
+
|
|
144
|
+
Returns:
|
|
145
|
+
PyArrow Table with numeric columns cast to float64.
|
|
146
|
+
"""
|
|
147
|
+
if not os.path.exists(self.file_path):
|
|
148
|
+
raise FileNotFoundError(f"Data file not found at: {self.file_path}")
|
|
149
|
+
|
|
150
|
+
ext = os.path.splitext(self.file_path)[-1].lower()
|
|
151
|
+
logger.info(f"📂 Loading file: {self.file_path} (extension: {ext})")
|
|
152
|
+
|
|
153
|
+
if ext == ".parquet":
|
|
154
|
+
table = pa_pq.read_table(self.file_path)
|
|
155
|
+
elif ext == ".csv":
|
|
156
|
+
table = pa_csv.read_csv(self.file_path)
|
|
157
|
+
elif ext in (".arrow", ".feather"):
|
|
158
|
+
with pa.memory_map(self.file_path, "r") as source:
|
|
159
|
+
table = pa.ipc.open_file(source).read_all()
|
|
160
|
+
elif ext in (".xlsx", ".xls"):
|
|
161
|
+
table = self._load_excel_to_arrow()
|
|
162
|
+
else:
|
|
163
|
+
raise ValueError(f"Unsupported file format: '{ext}'.")
|
|
164
|
+
|
|
165
|
+
return self._cast_table_numerics_to_float(table)
|
|
166
|
+
|
|
167
|
+
def _load_excel_to_arrow(self) -> pa.Table:
|
|
168
|
+
"""
|
|
169
|
+
Parses an Excel file using openpyxl with positional column tracking.
|
|
170
|
+
|
|
171
|
+
Returns:
|
|
172
|
+
PyArrow Table.
|
|
173
|
+
"""
|
|
174
|
+
logger.info("📊 Parsing Excel spreadsheet via positional column tracking...")
|
|
175
|
+
try:
|
|
176
|
+
import openpyxl
|
|
177
|
+
|
|
178
|
+
wb = openpyxl.load_workbook(
|
|
179
|
+
self.file_path, data_only=True, read_only=True
|
|
180
|
+
)
|
|
181
|
+
sheet = wb.active
|
|
182
|
+
rows = sheet.iter_rows(values_only=True)
|
|
183
|
+
|
|
184
|
+
headers = next(rows)
|
|
185
|
+
if not headers:
|
|
186
|
+
raise ValueError("The Excel file appears to be empty.")
|
|
187
|
+
|
|
188
|
+
# Use column indices to prevent header‑shift corruption
|
|
189
|
+
column_names = [
|
|
190
|
+
f"{h}" if h is not None else f"_col_{i}"
|
|
191
|
+
for i, h in enumerate(headers)
|
|
192
|
+
]
|
|
193
|
+
data_columns = {name: [] for name in column_names}
|
|
194
|
+
|
|
195
|
+
for row in rows:
|
|
196
|
+
for i, name in enumerate(column_names):
|
|
197
|
+
val = row[i] if i < len(row) else None
|
|
198
|
+
data_columns[name].append(val)
|
|
199
|
+
|
|
200
|
+
wb.close()
|
|
201
|
+
return pa.Table.from_pydict(data_columns)
|
|
202
|
+
|
|
203
|
+
except ImportError:
|
|
204
|
+
raise ImportError(
|
|
205
|
+
"Dependency missing: `pip install openpyxl` required for Excel files."
|
|
206
|
+
)
|
|
207
|
+
|
|
208
|
+
def _load_from_bigquery(self) -> Union[pa.Table, str]:
|
|
209
|
+
"""
|
|
210
|
+
Resolves BigQuery ingestion using cost‑optimised metadata inspection.
|
|
211
|
+
|
|
212
|
+
If the `google‑cloud‑bigquery` library is available, streams the table
|
|
213
|
+
as a PyArrow Table. Otherwise, returns a DuckDB scan macro string for
|
|
214
|
+
later execution (requires DuckDB's BigQuery extension).
|
|
215
|
+
|
|
216
|
+
Returns:
|
|
217
|
+
PyArrow Table or a DuckDB macro string.
|
|
218
|
+
"""
|
|
219
|
+
full_bq_path = (
|
|
220
|
+
f"{self.project_id}.{self.dataset_id}.{self.table_id}"
|
|
221
|
+
if self.project_id
|
|
222
|
+
else f"{self.dataset_id}.{self.table_id}"
|
|
223
|
+
)
|
|
224
|
+
logger.info(f"☁️ Initialising BigQuery client for: {full_bq_path}")
|
|
225
|
+
|
|
226
|
+
try:
|
|
227
|
+
from google.cloud import bigquery
|
|
228
|
+
|
|
229
|
+
bq_client = bigquery.Client(project=self.project_id)
|
|
230
|
+
full_table_ref = (
|
|
231
|
+
f"{self.project_id or bq_client.project}."
|
|
232
|
+
f"{self.dataset_id}.{self.table_id}"
|
|
233
|
+
)
|
|
234
|
+
|
|
235
|
+
# Fetch schema via get_table (cheaper than INFORMATION_SCHEMA)
|
|
236
|
+
table = bq_client.get_table(full_table_ref)
|
|
237
|
+
|
|
238
|
+
select_clauses = []
|
|
239
|
+
for field in table.schema:
|
|
240
|
+
# Cast numeric types to FLOAT64 for consistency
|
|
241
|
+
if field.field_type in (
|
|
242
|
+
"NUMERIC",
|
|
243
|
+
"BIGNUMERIC",
|
|
244
|
+
"DECIMAL",
|
|
245
|
+
"INTEGER",
|
|
246
|
+
"INT64",
|
|
247
|
+
"FLOAT",
|
|
248
|
+
"FLOAT64",
|
|
249
|
+
):
|
|
250
|
+
select_clauses.append(
|
|
251
|
+
f"SAFE_CAST(`{field.name}` AS FLOAT64) AS `{field.name}`"
|
|
252
|
+
)
|
|
253
|
+
else:
|
|
254
|
+
select_clauses.append(f"`{field.name}`")
|
|
255
|
+
|
|
256
|
+
query = f"SELECT {', '.join(select_clauses)} FROM `{full_table_ref}`"
|
|
257
|
+
arrow_table = bq_client.query(query).to_arrow()
|
|
258
|
+
logger.info(f"✅ Loaded {len(arrow_table):,} rows from BigQuery.")
|
|
259
|
+
return arrow_table
|
|
260
|
+
|
|
261
|
+
except ImportError:
|
|
262
|
+
logger.warning(
|
|
263
|
+
"⚠️ google‑cloud‑bigquery not found. Returning DuckDB scan macro. "
|
|
264
|
+
"This macro requires the BigQuery extension to be loaded in DuckDB."
|
|
265
|
+
)
|
|
266
|
+
# Return a string that DuckDB can interpret as a table reference
|
|
267
|
+
# (assuming the extension is loaded)
|
|
268
|
+
return f"bigquery_scan('{self.project_id or 'default'}', '{self.dataset_id}', '{self.table_id}')"
|
|
269
|
+
|
|
270
|
+
@staticmethod
|
|
271
|
+
def _validate_identifier(name: str, arg: str) -> str:
|
|
272
|
+
"""Ensures a caller-supplied table name is a safe SQL identifier."""
|
|
273
|
+
if not isinstance(name, str) or not re.fullmatch(r"[A-Za-z_][A-Za-z0-9_]*", name):
|
|
274
|
+
raise ValueError(f"{arg} must be a valid SQL identifier; got {name!r}")
|
|
275
|
+
return name
|
|
276
|
+
|
|
277
|
+
@staticmethod
|
|
278
|
+
def _drop_object_if_exists(con: duckdb.DuckDBPyConnection, name: str) -> None:
|
|
279
|
+
"""Drops a table or view by its exact catalog kind so creation is idempotent."""
|
|
280
|
+
rows = con.execute(
|
|
281
|
+
"SELECT table_type FROM information_schema.tables WHERE table_name = ?",
|
|
282
|
+
[name],
|
|
283
|
+
).fetchall()
|
|
284
|
+
for (table_type,) in rows:
|
|
285
|
+
if table_type.upper() == "VIEW":
|
|
286
|
+
con.execute(f'DROP VIEW IF EXISTS "{name}"')
|
|
287
|
+
else:
|
|
288
|
+
con.execute(f'DROP TABLE IF EXISTS "{name}"')
|
|
289
|
+
|
|
290
|
+
@staticmethod
|
|
291
|
+
def _cast_numeric_columns_to_double(
|
|
292
|
+
con: duckdb.DuckDBPyConnection, table_name: str
|
|
293
|
+
) -> None:
|
|
294
|
+
"""Memory-light on-disk cast of every numeric column to DOUBLE (float64)."""
|
|
295
|
+
cols = con.execute(
|
|
296
|
+
"SELECT column_name, data_type FROM information_schema.columns "
|
|
297
|
+
"WHERE table_name = ?",
|
|
298
|
+
[table_name],
|
|
299
|
+
).fetchall()
|
|
300
|
+
for name, dtype in cols:
|
|
301
|
+
up = dtype.upper()
|
|
302
|
+
if (
|
|
303
|
+
any(
|
|
304
|
+
k in up
|
|
305
|
+
for k in ("INT", "FLOAT", "DECIMAL", "REAL", "DOUBLE", "NUMERIC")
|
|
306
|
+
)
|
|
307
|
+
and "DOUBLE" not in up
|
|
308
|
+
):
|
|
309
|
+
con.execute(
|
|
310
|
+
f'ALTER TABLE "{table_name}" ALTER COLUMN "{name}" TYPE DOUBLE'
|
|
311
|
+
)
|
|
312
|
+
|
|
313
|
+
def stream_to_duckdb(
|
|
314
|
+
self,
|
|
315
|
+
db_path: Optional[str] = None,
|
|
316
|
+
path: Optional[str] = None,
|
|
317
|
+
encoding: Optional[str] = None,
|
|
318
|
+
data: Optional[Any] = None,
|
|
319
|
+
table_name: str = "udl_data",
|
|
320
|
+
scorer_view_name: str = "df",
|
|
321
|
+
) -> str:
|
|
322
|
+
"""Stream a data source straight into a persisted DuckDB database file.
|
|
323
|
+
|
|
324
|
+
Handles local files (read straight from disk, memory-light), in-memory
|
|
325
|
+
objects (PyArrow Table, pandas DataFrame, or any DuckDB-registrable
|
|
326
|
+
relation), and BigQuery tables. Numeric columns are cast to DOUBLE on-disk
|
|
327
|
+
to match the loader's float64 convention, and a ``df`` view is added over
|
|
328
|
+
the data table so the returned path can be handed directly to the engine's
|
|
329
|
+
builder, evaluator, and scorer.
|
|
330
|
+
|
|
331
|
+
Supported files: .csv/.tsv (read_csv_auto), .parquet/.pq (read_parquet),
|
|
332
|
+
.arrow/.feather (read_ipc). Excel is not streamed directly — call
|
|
333
|
+
``UniversalDataLoader(file_path=...).load()`` and pass the result as ``data=``.
|
|
334
|
+
|
|
335
|
+
Args:
|
|
336
|
+
db_path: Optional. Database name or file path to write to. A ``.duckdb``
|
|
337
|
+
extension is appended automatically when missing and the parent
|
|
338
|
+
directory is created as needed. When omitted, a unique database is
|
|
339
|
+
auto-created under the system temp directory.
|
|
340
|
+
path: Source file path (overrides the constructor's ``file_path`` when
|
|
341
|
+
both are set).
|
|
342
|
+
encoding: Optional encoding hint ("Latin-1" is honored for CSV/TSV).
|
|
343
|
+
data: In-memory source object (PyArrow Table, pandas DataFrame, or any
|
|
344
|
+
DuckDB-registrable relation) used instead of a file.
|
|
345
|
+
table_name: Name of the persisted data table (default 'udl_data'). A
|
|
346
|
+
custom name is fine — a ``udl_data`` view is always added as an
|
|
347
|
+
alias, so `StrategicSegmentBuilder.extract_segments` and
|
|
348
|
+
`evaluate_final_coverage` keep working when handed the returned
|
|
349
|
+
path (both resolve 'udl_data' by that fixed name).
|
|
350
|
+
scorer_view_name: Optional view name aliasing the data table so
|
|
351
|
+
`StrategicSegmentScore.calculate_and_export_weights` can consume
|
|
352
|
+
the same file (default 'df'). Pass None/'' to skip creating it.
|
|
353
|
+
|
|
354
|
+
Returns:
|
|
355
|
+
Absolute path to the created .duckdb file, usable directly by
|
|
356
|
+
``extract_segments``, ``evaluate_final_coverage``, and
|
|
357
|
+
``calculate_and_export_weights``.
|
|
358
|
+
|
|
359
|
+
Examples:
|
|
360
|
+
>>> from rapidsegment.utils.data_loader import UniversalDataLoader
|
|
361
|
+
>>> out = UniversalDataLoader(file_path="bank_train.csv").stream_to_duckdb("bank_data")
|
|
362
|
+
>>> builder.extract_segments(out) # reads table/view 'udl_data'
|
|
363
|
+
>>> builder.evaluate_final_coverage(out) # reads table/view 'udl_data'
|
|
364
|
+
>>> scorer.calculate_and_export_weights(out, "w.json") # reads table/view 'df'
|
|
365
|
+
"""
|
|
366
|
+
if not db_path:
|
|
367
|
+
unique_id = uuid.uuid4().hex[:8]
|
|
368
|
+
db_path = os.path.join(
|
|
369
|
+
tempfile.gettempdir(), f"rapidsegment_udl_{unique_id}.duckdb"
|
|
370
|
+
)
|
|
371
|
+
logger.info(f"🗄️ No db_path given — created default database at: {db_path}")
|
|
372
|
+
table_name = self._validate_identifier(table_name, "table_name")
|
|
373
|
+
if scorer_view_name:
|
|
374
|
+
scorer_view_name = self._validate_identifier(
|
|
375
|
+
scorer_view_name, "scorer_view_name"
|
|
376
|
+
)
|
|
377
|
+
|
|
378
|
+
# Resolve the source, mirroring load()'s priority order.
|
|
379
|
+
source_kind: str
|
|
380
|
+
source_path: Optional[str] = None
|
|
381
|
+
if data is not None:
|
|
382
|
+
source_kind = "object"
|
|
383
|
+
elif path is not None:
|
|
384
|
+
source_kind = "file"
|
|
385
|
+
source_path = os.path.abspath(path)
|
|
386
|
+
elif self.file_path is not None:
|
|
387
|
+
source_kind = "file"
|
|
388
|
+
source_path = os.path.abspath(self.file_path)
|
|
389
|
+
elif self.dataset_id and self.table_id:
|
|
390
|
+
bq_result = self._load_from_bigquery()
|
|
391
|
+
if isinstance(bq_result, str):
|
|
392
|
+
raise ValueError(
|
|
393
|
+
"stream_to_duckdb needs the 'google-cloud-bigquery' library to "
|
|
394
|
+
"materialise BigQuery data; install it and retry."
|
|
395
|
+
)
|
|
396
|
+
data = bq_result
|
|
397
|
+
source_kind = "object"
|
|
398
|
+
else:
|
|
399
|
+
raise ValueError(
|
|
400
|
+
"No data source configured. Pass `path=`, constructor `file_path`, "
|
|
401
|
+
"BigQuery identifiers (project_id/dataset_id/table_id), or in-memory `data=`."
|
|
402
|
+
)
|
|
403
|
+
|
|
404
|
+
if source_kind == "file" and source_path:
|
|
405
|
+
ext = os.path.splitext(source_path)[1].lower()
|
|
406
|
+
if ext not in (".csv", ".tsv", ".parquet", ".pq", ".arrow", ".feather"):
|
|
407
|
+
raise ValueError(
|
|
408
|
+
f"stream_to_duckdb does not support '{ext}' files directly; "
|
|
409
|
+
f"call UniversalDataLoader(file_path='{source_path}').load() and "
|
|
410
|
+
"pass the result as `data=` instead."
|
|
411
|
+
)
|
|
412
|
+
|
|
413
|
+
# Normalize the target: absolute path, always ending in .duckdb.
|
|
414
|
+
db_name = str(db_path)
|
|
415
|
+
if not db_name.lower().endswith(".duckdb"):
|
|
416
|
+
db_name = f"{db_name}.duckdb"
|
|
417
|
+
db_abs = os.path.abspath(db_name)
|
|
418
|
+
parent = os.path.dirname(db_abs)
|
|
419
|
+
if parent:
|
|
420
|
+
os.makedirs(parent, exist_ok=True)
|
|
421
|
+
|
|
422
|
+
con = duckdb.connect(db_abs)
|
|
423
|
+
try:
|
|
424
|
+
self._drop_object_if_exists(con, table_name)
|
|
425
|
+
if source_kind == "file" and source_path:
|
|
426
|
+
if ext in (".csv", ".tsv"):
|
|
427
|
+
opts = "header=true, sample_size=-1"
|
|
428
|
+
if encoding == "Latin-1":
|
|
429
|
+
opts += ", encoding='LATIN-1'"
|
|
430
|
+
con.execute(
|
|
431
|
+
f'CREATE TABLE "{table_name}" AS SELECT * FROM read_csv_auto(?, {opts})',
|
|
432
|
+
[source_path],
|
|
433
|
+
)
|
|
434
|
+
elif ext in (".parquet", ".pq"):
|
|
435
|
+
con.execute(
|
|
436
|
+
f'CREATE TABLE "{table_name}" AS SELECT * FROM read_parquet(?)',
|
|
437
|
+
[source_path],
|
|
438
|
+
)
|
|
439
|
+
else: # .arrow / .feather
|
|
440
|
+
con.execute(
|
|
441
|
+
f'CREATE TABLE "{table_name}" AS SELECT * FROM read_ipc(?)',
|
|
442
|
+
[source_path],
|
|
443
|
+
)
|
|
444
|
+
else:
|
|
445
|
+
con.register("__rs_stream_src", data)
|
|
446
|
+
con.execute(
|
|
447
|
+
f'CREATE TABLE "{table_name}" AS SELECT * FROM __rs_stream_src'
|
|
448
|
+
)
|
|
449
|
+
# float64 convention, memory-light (on-disk ALTER per numeric column)
|
|
450
|
+
self._cast_numeric_columns_to_double(con, table_name)
|
|
451
|
+
# Engine-compatibility aliases: keep a `udl_data` view (builder /
|
|
452
|
+
# evaluator resolve it by that fixed name) and a `df` view (scorer).
|
|
453
|
+
# Views are catalog metadata only — a few bytes, never a data copy.
|
|
454
|
+
engine_base = "udl_data"
|
|
455
|
+
if table_name != engine_base:
|
|
456
|
+
self._drop_object_if_exists(con, engine_base)
|
|
457
|
+
con.execute(
|
|
458
|
+
f'CREATE VIEW "{engine_base}" AS SELECT * FROM "{table_name}"'
|
|
459
|
+
)
|
|
460
|
+
if scorer_view_name and scorer_view_name != table_name:
|
|
461
|
+
self._drop_object_if_exists(con, scorer_view_name)
|
|
462
|
+
con.execute(
|
|
463
|
+
f'CREATE VIEW "{scorer_view_name}" AS SELECT * FROM "{table_name}"'
|
|
464
|
+
)
|
|
465
|
+
finally:
|
|
466
|
+
con.close()
|
|
467
|
+
return db_abs
|
|
468
|
+
|
|
469
|
+
@staticmethod
|
|
470
|
+
def duckdb_to_arrow(
|
|
471
|
+
db: Union[str, os.PathLike, duckdb.DuckDBPyConnection],
|
|
472
|
+
table_name: str = "udl_data",
|
|
473
|
+
) -> pa.Table:
|
|
474
|
+
"""Read a table from a DuckDB file (or an open DuckDB connection) into PyArrow.
|
|
475
|
+
|
|
476
|
+
Args:
|
|
477
|
+
db: A DuckDB database file path or an already-open ``duckdb`` connection.
|
|
478
|
+
table_name: Table or view to read (default 'udl_data'; artifacts written
|
|
479
|
+
by ``stream_to_duckdb`` also carry the 'df' scorer view).
|
|
480
|
+
|
|
481
|
+
Returns:
|
|
482
|
+
PyArrow Table of the requested table.
|
|
483
|
+
|
|
484
|
+
Raises:
|
|
485
|
+
ValueError: If ``table_name`` does not exist in the DuckDB source.
|
|
486
|
+
"""
|
|
487
|
+
owned = False
|
|
488
|
+
if isinstance(db, duckdb.DuckDBPyConnection):
|
|
489
|
+
con = db
|
|
490
|
+
else:
|
|
491
|
+
db_file = str(db)
|
|
492
|
+
try:
|
|
493
|
+
con = duckdb.connect(db_file, read_only=True)
|
|
494
|
+
except duckdb.Error:
|
|
495
|
+
# A same-process connection with a different (read-write) config
|
|
496
|
+
# may already hold the file; match it. Never create a missing DB.
|
|
497
|
+
if not os.path.exists(db_file):
|
|
498
|
+
raise
|
|
499
|
+
con = duckdb.connect(db_file)
|
|
500
|
+
owned = True
|
|
501
|
+
try:
|
|
502
|
+
exists = con.execute(
|
|
503
|
+
"SELECT 1 FROM information_schema.tables WHERE table_name = ?",
|
|
504
|
+
[table_name],
|
|
505
|
+
).fetchone()
|
|
506
|
+
if not exists:
|
|
507
|
+
available = ", ".join(
|
|
508
|
+
r[0]
|
|
509
|
+
for r in con.execute(
|
|
510
|
+
"SELECT table_name FROM information_schema.tables ORDER BY 1"
|
|
511
|
+
).fetchall()
|
|
512
|
+
)
|
|
513
|
+
raise ValueError(
|
|
514
|
+
f"Table '{table_name}' not found in DuckDB source. "
|
|
515
|
+
f"Available tables: {available or '(none)'}"
|
|
516
|
+
)
|
|
517
|
+
return UniversalDataLoader._fetch_arrow_table(
|
|
518
|
+
con.execute(f'SELECT * FROM "{table_name}"')
|
|
519
|
+
)
|
|
520
|
+
finally:
|
|
521
|
+
if owned:
|
|
522
|
+
con.close()
|
|
523
|
+
|
|
524
|
+
@staticmethod
|
|
525
|
+
def _fetch_arrow_table(result: Any) -> pa.Table:
|
|
526
|
+
"""Normalises duckdb's varying result hand-off into a concrete pa.Table."""
|
|
527
|
+
try:
|
|
528
|
+
return result.fetch_arrow_table()
|
|
529
|
+
except AttributeError: # duckdb < 1.x exposes .arrow() -> RecordBatchReader
|
|
530
|
+
return result.arrow().read_all()
|
|
531
|
+
|
|
532
|
+
|
|
533
|
+
def duckdb_to_arrow(
|
|
534
|
+
db: Union[str, os.PathLike, duckdb.DuckDBPyConnection],
|
|
535
|
+
table_name: str = "udl_data",
|
|
536
|
+
) -> pa.Table:
|
|
537
|
+
"""Module-level convenience wrapper for ``UniversalDataLoader.duckdb_to_arrow``."""
|
|
538
|
+
return UniversalDataLoader.duckdb_to_arrow(db, table_name=table_name)
|
|
@@ -1,320 +0,0 @@
|
|
|
1
|
-
"""
|
|
2
|
-
Unified Data Ingestion Layer
|
|
3
|
-
============================
|
|
4
|
-
Multi‑format data loader supporting Local Files (CSV, Parquet, Arrow, Excel),
|
|
5
|
-
In‑Memory PyArrow Tables, and Google Cloud BigQuery Storage API streams.
|
|
6
|
-
|
|
7
|
-
Author: Bishwarup Biswas + Gemini + DeepSeek
|
|
8
|
-
Python Version: 3.9+
|
|
9
|
-
"""
|
|
10
|
-
|
|
11
|
-
import logging
|
|
12
|
-
import os
|
|
13
|
-
from typing import Any, Optional, Union
|
|
14
|
-
|
|
15
|
-
import duckdb
|
|
16
|
-
import pyarrow as pa
|
|
17
|
-
import pyarrow.compute as pc
|
|
18
|
-
import pyarrow.csv as pa_csv
|
|
19
|
-
import pyarrow.parquet as pa_pq
|
|
20
|
-
|
|
21
|
-
logger = logging.getLogger("StrategicEngine.DataLoader")
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
class UniversalDataLoader:
|
|
25
|
-
"""
|
|
26
|
-
Handles multi‑source data ingestion, normalising inputs into highly optimised
|
|
27
|
-
in‑memory PyArrow Tables suitable for vectorised downstream compute engines.
|
|
28
|
-
|
|
29
|
-
The loader automatically detects the source type based on constructor arguments.
|
|
30
|
-
If a `fallback_data` object is passed to `load()`, it takes precedence.
|
|
31
|
-
|
|
32
|
-
Args:
|
|
33
|
-
project_id: (Optional) GCP project ID for BigQuery.
|
|
34
|
-
dataset_id: (Optional) BigQuery dataset ID.
|
|
35
|
-
table_id: (Optional) BigQuery table ID.
|
|
36
|
-
file_path: (Optional) Local file path (CSV, Parquet, Arrow/Feather, Excel).
|
|
37
|
-
"""
|
|
38
|
-
|
|
39
|
-
def __init__(
|
|
40
|
-
self,
|
|
41
|
-
project_id: Optional[str] = None,
|
|
42
|
-
dataset_id: Optional[str] = None,
|
|
43
|
-
table_id: Optional[str] = None,
|
|
44
|
-
file_path: Optional[str] = None,
|
|
45
|
-
) -> None:
|
|
46
|
-
self.project_id = project_id
|
|
47
|
-
self.dataset_id = dataset_id
|
|
48
|
-
self.table_id = table_id
|
|
49
|
-
self.file_path = file_path
|
|
50
|
-
|
|
51
|
-
def load(self, fallback_data: Optional[Any] = None) -> Union[pa.Table, str]:
|
|
52
|
-
"""
|
|
53
|
-
Auto‑detects the source configuration and loads the dataset.
|
|
54
|
-
|
|
55
|
-
Priority order:
|
|
56
|
-
1. If `fallback_data` is provided, it is returned (with type normalisation).
|
|
57
|
-
2. If BigQuery identifiers are set, load from BigQuery.
|
|
58
|
-
3. If a local `file_path` is provided, load from file.
|
|
59
|
-
|
|
60
|
-
Args:
|
|
61
|
-
fallback_data: Optional pre‑loaded data (e.g., a PyArrow Table or any
|
|
62
|
-
object that can be passed to DuckDB directly).
|
|
63
|
-
|
|
64
|
-
Returns:
|
|
65
|
-
A PyArrow Table, or a DuckDB scan macro string (when BigQuery client
|
|
66
|
-
is not available and fallback is not provided).
|
|
67
|
-
"""
|
|
68
|
-
# Scenario 1: Direct in‑memory object
|
|
69
|
-
if fallback_data is not None:
|
|
70
|
-
if isinstance(fallback_data, pa.Table):
|
|
71
|
-
logger.info("📥 Ingesting directly provided in‑memory PyArrow Table.")
|
|
72
|
-
return self._cast_table_numerics_to_float(fallback_data)
|
|
73
|
-
logger.info("📥 Using provided fallback data (non‑Arrow) as‑is.")
|
|
74
|
-
return fallback_data
|
|
75
|
-
|
|
76
|
-
# Scenario 2: BigQuery
|
|
77
|
-
if self.dataset_id and self.table_id:
|
|
78
|
-
return self._load_from_bigquery()
|
|
79
|
-
|
|
80
|
-
# Scenario 3: Local file
|
|
81
|
-
if self.file_path:
|
|
82
|
-
return self._load_from_file()
|
|
83
|
-
|
|
84
|
-
raise ValueError(
|
|
85
|
-
"Invalid Configuration: You must provide either a valid `file_path`, "
|
|
86
|
-
"BigQuery identifiers, or pass an explicit `fallback_data` object."
|
|
87
|
-
)
|
|
88
|
-
|
|
89
|
-
@staticmethod
|
|
90
|
-
def _cast_table_numerics_to_float(table: pa.Table) -> pa.Table:
|
|
91
|
-
"""
|
|
92
|
-
Casts all numeric columns in a PyArrow Table to float64.
|
|
93
|
-
|
|
94
|
-
This ensures consistent numerical precision across downstream operations.
|
|
95
|
-
|
|
96
|
-
Args:
|
|
97
|
-
table: Input PyArrow Table.
|
|
98
|
-
|
|
99
|
-
Returns:
|
|
100
|
-
A new table with numeric columns cast to float64.
|
|
101
|
-
"""
|
|
102
|
-
if not isinstance(table, pa.Table):
|
|
103
|
-
return table
|
|
104
|
-
|
|
105
|
-
new_columns = []
|
|
106
|
-
new_fields = []
|
|
107
|
-
|
|
108
|
-
for i, field in enumerate(table.schema):
|
|
109
|
-
# Check if the type is integer, floating, or decimal
|
|
110
|
-
if (
|
|
111
|
-
pa.types.is_integer(field.type)
|
|
112
|
-
or pa.types.is_floating(field.type)
|
|
113
|
-
or pa.types.is_decimal(field.type)
|
|
114
|
-
):
|
|
115
|
-
try:
|
|
116
|
-
casted_col = pc.cast(
|
|
117
|
-
table.column(i), pa.float64(), safe=False
|
|
118
|
-
)
|
|
119
|
-
new_columns.append(casted_col)
|
|
120
|
-
new_fields.append(
|
|
121
|
-
pa.field(field.name, pa.float64(), nullable=field.nullable)
|
|
122
|
-
)
|
|
123
|
-
except Exception as e:
|
|
124
|
-
logger.warning(
|
|
125
|
-
f"⚠️ Failed to cast column {field.name} to float64. Reason: {e}"
|
|
126
|
-
)
|
|
127
|
-
new_columns.append(table.column(i))
|
|
128
|
-
new_fields.append(field)
|
|
129
|
-
else:
|
|
130
|
-
new_columns.append(table.column(i))
|
|
131
|
-
new_fields.append(field)
|
|
132
|
-
|
|
133
|
-
return pa.Table.from_arrays(new_columns, schema=pa.schema(new_fields))
|
|
134
|
-
|
|
135
|
-
def _load_from_file(self) -> pa.Table:
|
|
136
|
-
"""
|
|
137
|
-
Parses a local file using high‑performance C++ Arrow readers.
|
|
138
|
-
|
|
139
|
-
Supports: .parquet, .csv, .arrow / .feather, .xlsx / .xls.
|
|
140
|
-
|
|
141
|
-
Returns:
|
|
142
|
-
PyArrow Table with numeric columns cast to float64.
|
|
143
|
-
"""
|
|
144
|
-
if not os.path.exists(self.file_path):
|
|
145
|
-
raise FileNotFoundError(f"Data file not found at: {self.file_path}")
|
|
146
|
-
|
|
147
|
-
ext = os.path.splitext(self.file_path)[-1].lower()
|
|
148
|
-
logger.info(f"📂 Loading file: {self.file_path} (extension: {ext})")
|
|
149
|
-
|
|
150
|
-
if ext == ".parquet":
|
|
151
|
-
table = pa_pq.read_table(self.file_path)
|
|
152
|
-
elif ext == ".csv":
|
|
153
|
-
table = pa_csv.read_csv(self.file_path)
|
|
154
|
-
elif ext in (".arrow", ".feather"):
|
|
155
|
-
with pa.memory_map(self.file_path, "r") as source:
|
|
156
|
-
table = pa.ipc.open_file(source).read_all()
|
|
157
|
-
elif ext in (".xlsx", ".xls"):
|
|
158
|
-
table = self._load_excel_to_arrow()
|
|
159
|
-
else:
|
|
160
|
-
raise ValueError(f"Unsupported file format: '{ext}'.")
|
|
161
|
-
|
|
162
|
-
return self._cast_table_numerics_to_float(table)
|
|
163
|
-
|
|
164
|
-
def _load_excel_to_arrow(self) -> pa.Table:
|
|
165
|
-
"""
|
|
166
|
-
Parses an Excel file using openpyxl with positional column tracking.
|
|
167
|
-
|
|
168
|
-
Returns:
|
|
169
|
-
PyArrow Table.
|
|
170
|
-
"""
|
|
171
|
-
logger.info("📊 Parsing Excel spreadsheet via positional column tracking...")
|
|
172
|
-
try:
|
|
173
|
-
import openpyxl
|
|
174
|
-
|
|
175
|
-
wb = openpyxl.load_workbook(
|
|
176
|
-
self.file_path, data_only=True, read_only=True
|
|
177
|
-
)
|
|
178
|
-
sheet = wb.active
|
|
179
|
-
rows = sheet.iter_rows(values_only=True)
|
|
180
|
-
|
|
181
|
-
headers = next(rows)
|
|
182
|
-
if not headers:
|
|
183
|
-
raise ValueError("The Excel file appears to be empty.")
|
|
184
|
-
|
|
185
|
-
# Use column indices to prevent header‑shift corruption
|
|
186
|
-
column_names = [
|
|
187
|
-
f"{h}" if h is not None else f"_col_{i}"
|
|
188
|
-
for i, h in enumerate(headers)
|
|
189
|
-
]
|
|
190
|
-
data_columns = {name: [] for name in column_names}
|
|
191
|
-
|
|
192
|
-
for row in rows:
|
|
193
|
-
for i, name in enumerate(column_names):
|
|
194
|
-
val = row[i] if i < len(row) else None
|
|
195
|
-
data_columns[name].append(val)
|
|
196
|
-
|
|
197
|
-
wb.close()
|
|
198
|
-
return pa.Table.from_pydict(data_columns)
|
|
199
|
-
|
|
200
|
-
except ImportError:
|
|
201
|
-
raise ImportError(
|
|
202
|
-
"Dependency missing: `pip install openpyxl` required for Excel files."
|
|
203
|
-
)
|
|
204
|
-
|
|
205
|
-
def _load_from_bigquery(self) -> Union[pa.Table, str]:
|
|
206
|
-
"""
|
|
207
|
-
Resolves BigQuery ingestion using cost‑optimised metadata inspection.
|
|
208
|
-
|
|
209
|
-
If the `google‑cloud‑bigquery` library is available, streams the table
|
|
210
|
-
as a PyArrow Table. Otherwise, returns a DuckDB scan macro string for
|
|
211
|
-
later execution (requires DuckDB's BigQuery extension).
|
|
212
|
-
|
|
213
|
-
Returns:
|
|
214
|
-
PyArrow Table or a DuckDB macro string.
|
|
215
|
-
"""
|
|
216
|
-
full_bq_path = (
|
|
217
|
-
f"{self.project_id}.{self.dataset_id}.{self.table_id}"
|
|
218
|
-
if self.project_id
|
|
219
|
-
else f"{self.dataset_id}.{self.table_id}"
|
|
220
|
-
)
|
|
221
|
-
logger.info(f"☁️ Initialising BigQuery client for: {full_bq_path}")
|
|
222
|
-
|
|
223
|
-
try:
|
|
224
|
-
from google.cloud import bigquery
|
|
225
|
-
|
|
226
|
-
bq_client = bigquery.Client(project=self.project_id)
|
|
227
|
-
full_table_ref = (
|
|
228
|
-
f"{self.project_id or bq_client.project}."
|
|
229
|
-
f"{self.dataset_id}.{self.table_id}"
|
|
230
|
-
)
|
|
231
|
-
|
|
232
|
-
# Fetch schema via get_table (cheaper than INFORMATION_SCHEMA)
|
|
233
|
-
table = bq_client.get_table(full_table_ref)
|
|
234
|
-
|
|
235
|
-
select_clauses = []
|
|
236
|
-
for field in table.schema:
|
|
237
|
-
# Cast numeric types to FLOAT64 for consistency
|
|
238
|
-
if field.field_type in (
|
|
239
|
-
"NUMERIC",
|
|
240
|
-
"BIGNUMERIC",
|
|
241
|
-
"DECIMAL",
|
|
242
|
-
"INTEGER",
|
|
243
|
-
"INT64",
|
|
244
|
-
"FLOAT",
|
|
245
|
-
"FLOAT64",
|
|
246
|
-
):
|
|
247
|
-
select_clauses.append(
|
|
248
|
-
f"SAFE_CAST(`{field.name}` AS FLOAT64) AS `{field.name}`"
|
|
249
|
-
)
|
|
250
|
-
else:
|
|
251
|
-
select_clauses.append(f"`{field.name}`")
|
|
252
|
-
|
|
253
|
-
query = f"SELECT {', '.join(select_clauses)} FROM `{full_table_ref}`"
|
|
254
|
-
arrow_table = bq_client.query(query).to_arrow()
|
|
255
|
-
logger.info(f"✅ Loaded {len(arrow_table):,} rows from BigQuery.")
|
|
256
|
-
return arrow_table
|
|
257
|
-
|
|
258
|
-
except ImportError:
|
|
259
|
-
logger.warning(
|
|
260
|
-
"⚠️ google‑cloud‑bigquery not found. Returning DuckDB scan macro. "
|
|
261
|
-
"This macro requires the BigQuery extension to be loaded in DuckDB."
|
|
262
|
-
)
|
|
263
|
-
# Return a string that DuckDB can interpret as a table reference
|
|
264
|
-
# (assuming the extension is loaded)
|
|
265
|
-
return f"bigquery_scan('{self.project_id or 'default'}', '{self.dataset_id}', '{self.table_id}')"
|
|
266
|
-
|
|
267
|
-
def stream_to_duckdb(self, db_path: str, path: str, encoding: Optional[str] = None) -> None:
|
|
268
|
-
"""Stream a local file straight into a DuckDB table `udl_data` at ``db_path``.
|
|
269
|
-
|
|
270
|
-
Avoids materialising a full in-RAM PyArrow table (unlike the arrow-based
|
|
271
|
-
load path) so multi-GB files stay memory-light — DuckDB reads directly
|
|
272
|
-
from disk. Numeric columns are cast to DOUBLE on-disk to match the
|
|
273
|
-
loader's float64 convention.
|
|
274
|
-
|
|
275
|
-
Supported: .csv/.tsv (read_csv_auto), .parquet/.pq (read_parquet),
|
|
276
|
-
.arrow/.feather (read_ipc).
|
|
277
|
-
"""
|
|
278
|
-
ext = os.path.splitext(path)[1].lower()
|
|
279
|
-
if ext not in (".csv", ".tsv", ".parquet", ".pq", ".arrow", ".feather"):
|
|
280
|
-
raise ValueError(
|
|
281
|
-
f"stream_to_duckdb does not support '{ext}' (use the arrow/Excel path)."
|
|
282
|
-
)
|
|
283
|
-
con = duckdb.connect(db_path)
|
|
284
|
-
try:
|
|
285
|
-
con.execute("DROP TABLE IF EXISTS udl_data")
|
|
286
|
-
if ext in (".csv", ".tsv"):
|
|
287
|
-
opts = "header=true, sample_size=-1"
|
|
288
|
-
if encoding == "Latin-1":
|
|
289
|
-
opts += ", encoding='LATIN-1'"
|
|
290
|
-
con.execute(
|
|
291
|
-
f"CREATE TABLE udl_data AS SELECT * FROM read_csv_auto(?, {opts})",
|
|
292
|
-
[path],
|
|
293
|
-
)
|
|
294
|
-
elif ext in (".parquet", ".pq"):
|
|
295
|
-
con.execute(
|
|
296
|
-
"CREATE TABLE udl_data AS SELECT * FROM read_parquet(?)", [path]
|
|
297
|
-
)
|
|
298
|
-
else: # .arrow / .feather
|
|
299
|
-
con.execute(
|
|
300
|
-
"CREATE TABLE udl_data AS SELECT * FROM read_ipc(?)", [path]
|
|
301
|
-
)
|
|
302
|
-
# float64 convention, memory-light (on-disk ALTER per numeric column)
|
|
303
|
-
cols = con.execute(
|
|
304
|
-
"SELECT column_name, data_type FROM information_schema.columns "
|
|
305
|
-
"WHERE table_name='udl_data'"
|
|
306
|
-
).fetchall()
|
|
307
|
-
for name, dtype in cols:
|
|
308
|
-
up = dtype.upper()
|
|
309
|
-
if (
|
|
310
|
-
any(
|
|
311
|
-
k in up
|
|
312
|
-
for k in ("INT", "FLOAT", "DECIMAL", "REAL", "DOUBLE", "NUMERIC")
|
|
313
|
-
)
|
|
314
|
-
and "DOUBLE" not in up
|
|
315
|
-
):
|
|
316
|
-
con.execute(
|
|
317
|
-
f'ALTER TABLE udl_data ALTER COLUMN "{name}" TYPE DOUBLE'
|
|
318
|
-
)
|
|
319
|
-
finally:
|
|
320
|
-
con.close()
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/pages/3_Execution_Console.py
RENAMED
|
File without changes
|
{rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/ui/pages/4_Results_Dashboard.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{rapidsegment-1.3.post3 → rapidsegment-1.3.1}/src/rapidsegment/utils/on_gcp_feature_selection.py
RENAMED
|
File without changes
|
|
File without changes
|