typedframes 0.8.0__tar.gz → 0.9.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {typedframes-0.8.0 → typedframes-0.9.0}/PKG-INFO +12 -12
- {typedframes-0.8.0 → typedframes-0.9.0}/README.md +11 -11
- {typedframes-0.8.0 → typedframes-0.9.0}/pyproject.toml +1 -1
- {typedframes-0.8.0 → typedframes-0.9.0}/rust/Cargo.lock +1 -1
- {typedframes-0.8.0 → typedframes-0.9.0}/rust/Cargo.toml +1 -1
- {typedframes-0.8.0 → typedframes-0.9.0}/rust/src/ast_extract.rs +25 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/rust/src/linter.rs +563 -10
- {typedframes-0.8.0 → typedframes-0.9.0}/src/typedframes/__init__.py +1 -1
- {typedframes-0.8.0 → typedframes-0.9.0}/src/typedframes/cli.py +84 -10
- {typedframes-0.8.0 → typedframes-0.9.0}/rust/.cargo/config.toml +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/rust/README.md +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/rust/benches/parser_bench.rs +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/rust/src/config.rs +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/rust/src/constants.rs +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/rust/src/contract.rs +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/rust/src/errors.rs +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/rust/src/frame_ops.rs +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/rust/src/index.rs +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/rust/src/lib.rs +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/rust/src/main.rs +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/rust/src/notebook.rs +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/rust/src/pyapi.rs +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/rust/src/sql.rs +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/rust/src/typo.rs +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/rust/tests/integration_test.rs +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/src/typedframes/_rust_checker.pyi +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/src/typedframes/base_schema.py +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/src/typedframes/column.py +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/src/typedframes/column_group.py +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/src/typedframes/column_group_error.py +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/src/typedframes/column_set.py +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/src/typedframes/missing_dependency_error.py +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/src/typedframes/mypy.py +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/src/typedframes/pandas.py +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/src/typedframes/pandera.py +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/src/typedframes/polars.py +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/src/typedframes/py.typed +0 -0
- {typedframes-0.8.0 → typedframes-0.9.0}/src/typedframes/schema_algebra.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: typedframes
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.9.0
|
|
4
4
|
Classifier: Development Status :: 3 - Alpha
|
|
5
5
|
Classifier: Intended Audience :: Developers
|
|
6
6
|
Classifier: License :: OSI Approved :: MIT License
|
|
@@ -42,7 +42,7 @@ Project-URL: Repository, https://github.com/w-martin/typedframes
|
|
|
42
42
|
|
|
43
43
|
> ⚠️ **Project Status: Proof of Concept**
|
|
44
44
|
>
|
|
45
|
-
> `typedframes` (v0.
|
|
45
|
+
> `typedframes` (v0.9.0) is currently an experimental proof-of-concept. The core static analysis and mypy/Rust
|
|
46
46
|
> integrations work, but expect rough edges. The codebase prioritizes demonstrating the viability of static DataFrame
|
|
47
47
|
> column checking over production-grade stability.
|
|
48
48
|
>
|
|
@@ -504,7 +504,7 @@ repos:
|
|
|
504
504
|
name: typedframes check
|
|
505
505
|
entry: typedframes check . --strict
|
|
506
506
|
language: python
|
|
507
|
-
additional_dependencies: ["typedframes==0.
|
|
507
|
+
additional_dependencies: ["typedframes==0.9.0"]
|
|
508
508
|
types_or: [python, jupyter]
|
|
509
509
|
pass_filenames: false
|
|
510
510
|
```
|
|
@@ -553,7 +553,7 @@ The action installs the PyPI wheel into a throwaway virtualenv and runs the chec
|
|
|
553
553
|
| Input | Default | |
|
|
554
554
|
|-------|---------|-|
|
|
555
555
|
| `path` | `.` | File or directory to check |
|
|
556
|
-
| `version` | `latest` | PyPI version to install, e.g. `"0.
|
|
556
|
+
| `version` | `latest` | PyPI version to install, e.g. `"0.9.0"` |
|
|
557
557
|
| `strict` | `true` | Fail the step on errors — `typedframes check` exits 0 without it |
|
|
558
558
|
| `coverage-fail-under` | *(unset)* | Minimum DataFrame schema coverage, e.g. `"90"` |
|
|
559
559
|
| `coverage-detail` | `summary` | Or `term-missing` for the per-file breakdown, `explain` to diagnose a lower-than-expected total |
|
|
@@ -667,17 +667,17 @@ plus the requested features, so both halves get checked normally. See
|
|
|
667
667
|
Fast feedback reduces development time. The typedframes Rust binary provides near-instant column checking.
|
|
668
668
|
|
|
669
669
|
**Benchmark results** (20 runs, 3 warmup, caches cleared between runs):
|
|
670
|
-
*2026-09-
|
|
670
|
+
*2026-09-30 · Darwin 27.0.0 · arm · CPython 3.14.4 · 64GiB RAM · Great Expectations pinned @ 1.20.0*
|
|
671
671
|
|
|
672
672
|
| Tool | Version | What it does | typedframes (13 files) | great_expectations (485 files) |
|
|
673
673
|
|------|---------|--------------|------------------------|--------------------------------|
|
|
674
|
-
| typedframes | 0.
|
|
675
|
-
| ruff | 0.16.3 | Linter (no type checking) |
|
|
676
|
-
| ty | 0.0.72 | Type checker |
|
|
677
|
-
| pyrefly | 1.2.0 | Type checker | 98ms ±1ms (IQR
|
|
678
|
-
| mypy | 2.3.1 | Type checker (no plugin) | 2.
|
|
679
|
-
| mypy + typedframes | 2.3.1 | Type checker + column checker | 2.
|
|
680
|
-
| pyright | 1.1.411 | Type checker |
|
|
674
|
+
| typedframes | 0.9.0 | DataFrame column checker | 51ms ±613µs (IQR 997µs) | 268ms ±4ms (IQR 5ms) |
|
|
675
|
+
| ruff | 0.16.3 | Linter (no type checking) | 31ms ±735µs (IQR 882µs) | 237ms ±4ms (IQR 6ms) |
|
|
676
|
+
| ty | 0.0.72 | Type checker | 74ms ±2ms (IQR 3ms) | 782ms ±12ms (IQR 17ms) |
|
|
677
|
+
| pyrefly | 1.2.0 | Type checker | 98ms ±1ms (IQR 2ms) | 276ms ±8ms (IQR 14ms) |
|
|
678
|
+
| mypy | 2.3.1 | Type checker (no plugin) | 2.73s ±37ms (IQR 43ms) | 4.30s ±72ms (IQR 106ms) |
|
|
679
|
+
| mypy + typedframes | 2.3.1 | Type checker + column checker | 2.72s ±19ms (IQR 23ms) | 4.61s ±22ms (IQR 23ms) |
|
|
680
|
+
| pyright | 1.1.411 | Type checker | 826ms ±11ms (IQR 11ms) | 3.34s ±26ms (IQR 40ms) |
|
|
681
681
|
|
|
682
682
|
*Run `uv run python benchmarks/benchmark_checkers.py` to reproduce.*
|
|
683
683
|
|
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
|
|
9
9
|
> ⚠️ **Project Status: Proof of Concept**
|
|
10
10
|
>
|
|
11
|
-
> `typedframes` (v0.
|
|
11
|
+
> `typedframes` (v0.9.0) is currently an experimental proof-of-concept. The core static analysis and mypy/Rust
|
|
12
12
|
> integrations work, but expect rough edges. The codebase prioritizes demonstrating the viability of static DataFrame
|
|
13
13
|
> column checking over production-grade stability.
|
|
14
14
|
>
|
|
@@ -470,7 +470,7 @@ repos:
|
|
|
470
470
|
name: typedframes check
|
|
471
471
|
entry: typedframes check . --strict
|
|
472
472
|
language: python
|
|
473
|
-
additional_dependencies: ["typedframes==0.
|
|
473
|
+
additional_dependencies: ["typedframes==0.9.0"]
|
|
474
474
|
types_or: [python, jupyter]
|
|
475
475
|
pass_filenames: false
|
|
476
476
|
```
|
|
@@ -519,7 +519,7 @@ The action installs the PyPI wheel into a throwaway virtualenv and runs the chec
|
|
|
519
519
|
| Input | Default | |
|
|
520
520
|
|-------|---------|-|
|
|
521
521
|
| `path` | `.` | File or directory to check |
|
|
522
|
-
| `version` | `latest` | PyPI version to install, e.g. `"0.
|
|
522
|
+
| `version` | `latest` | PyPI version to install, e.g. `"0.9.0"` |
|
|
523
523
|
| `strict` | `true` | Fail the step on errors — `typedframes check` exits 0 without it |
|
|
524
524
|
| `coverage-fail-under` | *(unset)* | Minimum DataFrame schema coverage, e.g. `"90"` |
|
|
525
525
|
| `coverage-detail` | `summary` | Or `term-missing` for the per-file breakdown, `explain` to diagnose a lower-than-expected total |
|
|
@@ -633,17 +633,17 @@ plus the requested features, so both halves get checked normally. See
|
|
|
633
633
|
Fast feedback reduces development time. The typedframes Rust binary provides near-instant column checking.
|
|
634
634
|
|
|
635
635
|
**Benchmark results** (20 runs, 3 warmup, caches cleared between runs):
|
|
636
|
-
*2026-09-
|
|
636
|
+
*2026-09-30 · Darwin 27.0.0 · arm · CPython 3.14.4 · 64GiB RAM · Great Expectations pinned @ 1.20.0*
|
|
637
637
|
|
|
638
638
|
| Tool | Version | What it does | typedframes (13 files) | great_expectations (485 files) |
|
|
639
639
|
|------|---------|--------------|------------------------|--------------------------------|
|
|
640
|
-
| typedframes | 0.
|
|
641
|
-
| ruff | 0.16.3 | Linter (no type checking) |
|
|
642
|
-
| ty | 0.0.72 | Type checker |
|
|
643
|
-
| pyrefly | 1.2.0 | Type checker | 98ms ±1ms (IQR
|
|
644
|
-
| mypy | 2.3.1 | Type checker (no plugin) | 2.
|
|
645
|
-
| mypy + typedframes | 2.3.1 | Type checker + column checker | 2.
|
|
646
|
-
| pyright | 1.1.411 | Type checker |
|
|
640
|
+
| typedframes | 0.9.0 | DataFrame column checker | 51ms ±613µs (IQR 997µs) | 268ms ±4ms (IQR 5ms) |
|
|
641
|
+
| ruff | 0.16.3 | Linter (no type checking) | 31ms ±735µs (IQR 882µs) | 237ms ±4ms (IQR 6ms) |
|
|
642
|
+
| ty | 0.0.72 | Type checker | 74ms ±2ms (IQR 3ms) | 782ms ±12ms (IQR 17ms) |
|
|
643
|
+
| pyrefly | 1.2.0 | Type checker | 98ms ±1ms (IQR 2ms) | 276ms ±8ms (IQR 14ms) |
|
|
644
|
+
| mypy | 2.3.1 | Type checker (no plugin) | 2.73s ±37ms (IQR 43ms) | 4.30s ±72ms (IQR 106ms) |
|
|
645
|
+
| mypy + typedframes | 2.3.1 | Type checker + column checker | 2.72s ±19ms (IQR 23ms) | 4.61s ±22ms (IQR 23ms) |
|
|
646
|
+
| pyright | 1.1.411 | Type checker | 826ms ±11ms (IQR 11ms) | 3.34s ±26ms (IQR 40ms) |
|
|
647
647
|
|
|
648
648
|
*Run `uv run python benchmarks/benchmark_checkers.py` to reproduce.*
|
|
649
649
|
|
|
@@ -4,7 +4,7 @@ build-backend = "maturin"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "typedframes"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.9.0"
|
|
8
8
|
description = "Static analysis for pandas and polars DataFrames. Catch column errors at lint-time, not runtime."
|
|
9
9
|
keywords = ["pandas", "polars", "type-checking", "static-analysis", "dataframe", "linter", "mypy-plugin"]
|
|
10
10
|
readme = "README.md"
|
|
@@ -29,6 +29,31 @@ pub(crate) fn decorator_name(expr: &Expr) -> Option<&str> {
|
|
|
29
29
|
}
|
|
30
30
|
}
|
|
31
31
|
|
|
32
|
+
// Column names for `pd.DataFrame([{...}, {...}])` -- a literal list of per-row dict
|
|
33
|
+
// literals (pandas' "records" orientation). Every element must be a dict literal with
|
|
34
|
+
// only string-literal keys; the result is the union of every row's keys, in
|
|
35
|
+
// first-seen order (matching how pandas itself orders the resulting columns) -- a row
|
|
36
|
+
// missing a key another row has doesn't invalidate the inference, it's just NaN for
|
|
37
|
+
// that row.
|
|
38
|
+
pub(crate) fn extract_records_list_columns(list: &ast::ExprList) -> Option<Vec<String>> {
|
|
39
|
+
if list.elts.is_empty() {
|
|
40
|
+
return None;
|
|
41
|
+
}
|
|
42
|
+
let mut columns = Vec::new();
|
|
43
|
+
for elt in &list.elts {
|
|
44
|
+
let Expr::Dict(dict) = elt else {
|
|
45
|
+
return None;
|
|
46
|
+
};
|
|
47
|
+
for item in &dict.items {
|
|
48
|
+
let key = item.key.as_ref().and_then(|k| extract_string_literal(k))?;
|
|
49
|
+
if !columns.iter().any(|c: &String| c == key) {
|
|
50
|
+
columns.push(key.to_string());
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
Some(columns)
|
|
55
|
+
}
|
|
56
|
+
|
|
32
57
|
pub(crate) fn extract_string_literal(expr: &Expr) -> Option<&str> {
|
|
33
58
|
if let Expr::StringLiteral(s) = expr {
|
|
34
59
|
Some(s.value.to_str())
|
|
@@ -293,6 +293,19 @@ pub struct Linter {
|
|
|
293
293
|
// Names bound exactly once in the whole file, to a `True`/`False` literal -- the
|
|
294
294
|
// boolean counterpart of `string_var_candidates`, for `inplace=flag`.
|
|
295
295
|
pub(crate) bool_var_candidates: HashMap<String, bool>,
|
|
296
|
+
// Names initialized as `name = []` and thereafter only ever added to via
|
|
297
|
+
// `name.append({...})` with a literal dict of string-literal keys, right up to
|
|
298
|
+
// being read (as the first positional argument) by a `DataFrame(name)` call --
|
|
299
|
+
// the accumulator counterpart of `pd.DataFrame([{...}, {...}])`. Maps to the
|
|
300
|
+
// union of every append's keys, in first-seen order. Populated by a
|
|
301
|
+
// `DictListBindingCollector` pre-pass alongside `string_var_candidates`; poisoned
|
|
302
|
+
// (absent from this map) by a second `= []`, any other reassignment, any
|
|
303
|
+
// `.append()` call whose argument isn't such a dict literal, or any other bare
|
|
304
|
+
// reference to the name at all (passed to a function, checked for truthiness,
|
|
305
|
+
// indexed, ...) -- same "give up rather than guess" policy as
|
|
306
|
+
// `string_var_candidates`, deliberately conservative since a missed mutation here
|
|
307
|
+
// would mean inferring a schema real code could add columns past.
|
|
308
|
+
pub(crate) dict_list_var_candidates: HashMap<String, Vec<String>>,
|
|
296
309
|
// Names bound to a resolved SQLAlchemy Core `select(...)` column list — e.g.
|
|
297
310
|
// `stmt = select(Order.id, Order.amount)`. Populated inline during the main
|
|
298
311
|
// top-to-bottom statement walk (unlike `string_var_candidates`, which needs a
|
|
@@ -729,6 +742,232 @@ impl<'a> Visitor<'a> for StringBindingCollector<'a> {
|
|
|
729
742
|
}
|
|
730
743
|
}
|
|
731
744
|
|
|
745
|
+
// One `name = []` candidate's state -- see `Linter::dict_list_var_candidates`.
|
|
746
|
+
enum DictListBinding {
|
|
747
|
+
// The union of every recognized `.append({...})` call's keys, in first-seen
|
|
748
|
+
// order.
|
|
749
|
+
Building(Vec<String>),
|
|
750
|
+
Poisoned,
|
|
751
|
+
}
|
|
752
|
+
|
|
753
|
+
// AST visitor backing `Linter::dict_list_var_candidates` -- see that field's doc
|
|
754
|
+
// comment for the exact policy. `accounted_for` holds the source range of every Name
|
|
755
|
+
// node already explained by a recognized `.append({...})` receiver or `DataFrame(name)`
|
|
756
|
+
// argument, so the generic "any other appearance poisons it" rule in `visit_expr`
|
|
757
|
+
// doesn't also fire for those two approved occurrences.
|
|
758
|
+
struct DictListBindingCollector {
|
|
759
|
+
bindings: HashMap<String, DictListBinding>,
|
|
760
|
+
accounted_for: std::collections::HashSet<ruff_text_size::TextRange>,
|
|
761
|
+
}
|
|
762
|
+
|
|
763
|
+
impl DictListBindingCollector {
|
|
764
|
+
fn poison(&mut self, name: &str) {
|
|
765
|
+
self.bindings
|
|
766
|
+
.insert(name.to_string(), DictListBinding::Poisoned);
|
|
767
|
+
}
|
|
768
|
+
|
|
769
|
+
// Poison every bare name an assignment/for/with/del target touches -- mirrors
|
|
770
|
+
// `StringBindingCollector::record_target_names`.
|
|
771
|
+
fn record_target_names(&mut self, target: &Expr) {
|
|
772
|
+
match target {
|
|
773
|
+
Expr::Name(n) => self.poison(n.id.as_str()),
|
|
774
|
+
Expr::Tuple(t) => t.elts.iter().for_each(|e| self.record_target_names(e)),
|
|
775
|
+
Expr::List(l) => l.elts.iter().for_each(|e| self.record_target_names(e)),
|
|
776
|
+
Expr::Starred(s) => self.record_target_names(&s.value),
|
|
777
|
+
_ => {}
|
|
778
|
+
}
|
|
779
|
+
}
|
|
780
|
+
|
|
781
|
+
// `name.append(arg)`: extends the candidate with `arg`'s keys if it's a dict
|
|
782
|
+
// literal of string-literal keys, else poisons it. A no-op for a name that was
|
|
783
|
+
// never initialized as `= []` in the first place.
|
|
784
|
+
fn record_append(&mut self, name: &str, arg: Option<&Expr>) {
|
|
785
|
+
if !self.bindings.contains_key(name) {
|
|
786
|
+
return;
|
|
787
|
+
}
|
|
788
|
+
let extracted = match arg {
|
|
789
|
+
Some(Expr::Dict(dict)) => dict
|
|
790
|
+
.items
|
|
791
|
+
.iter()
|
|
792
|
+
.map(|item| {
|
|
793
|
+
item.key
|
|
794
|
+
.as_ref()
|
|
795
|
+
.and_then(|k| ast_extract::extract_string_literal(k))
|
|
796
|
+
.map(str::to_string)
|
|
797
|
+
})
|
|
798
|
+
.collect::<Option<Vec<String>>>(),
|
|
799
|
+
_ => None,
|
|
800
|
+
};
|
|
801
|
+
let Some(binding) = self.bindings.get_mut(name) else {
|
|
802
|
+
return;
|
|
803
|
+
};
|
|
804
|
+
match (binding, extracted) {
|
|
805
|
+
(DictListBinding::Poisoned, _) => {}
|
|
806
|
+
(binding, None) => *binding = DictListBinding::Poisoned,
|
|
807
|
+
(DictListBinding::Building(keys), Some(new_keys)) => {
|
|
808
|
+
for key in new_keys {
|
|
809
|
+
if !keys.contains(&key) {
|
|
810
|
+
keys.push(key);
|
|
811
|
+
}
|
|
812
|
+
}
|
|
813
|
+
}
|
|
814
|
+
}
|
|
815
|
+
}
|
|
816
|
+
}
|
|
817
|
+
|
|
818
|
+
impl<'a> Visitor<'a> for DictListBindingCollector {
|
|
819
|
+
fn visit_stmt(&mut self, stmt: &'a Stmt) {
|
|
820
|
+
match stmt {
|
|
821
|
+
Stmt::Assign(assign) => {
|
|
822
|
+
if let [Expr::Name(target)] = assign.targets.as_slice() {
|
|
823
|
+
let name = target.id.as_str();
|
|
824
|
+
if matches!(&*assign.value, Expr::List(list) if list.elts.is_empty()) {
|
|
825
|
+
if self.bindings.contains_key(name) {
|
|
826
|
+
self.poison(name);
|
|
827
|
+
} else {
|
|
828
|
+
self.bindings
|
|
829
|
+
.insert(name.to_string(), DictListBinding::Building(Vec::new()));
|
|
830
|
+
// The defining target itself is a bare Name reference too --
|
|
831
|
+
// exempt it from the generic poison-on-any-appearance rule
|
|
832
|
+
// below, the same way an `.append` receiver or a
|
|
833
|
+
// `DataFrame(name)` argument is.
|
|
834
|
+
self.accounted_for.insert(target.range());
|
|
835
|
+
}
|
|
836
|
+
} else {
|
|
837
|
+
self.poison(name);
|
|
838
|
+
}
|
|
839
|
+
} else {
|
|
840
|
+
for target in &assign.targets {
|
|
841
|
+
self.record_target_names(target);
|
|
842
|
+
}
|
|
843
|
+
}
|
|
844
|
+
}
|
|
845
|
+
Stmt::AnnAssign(ann) => {
|
|
846
|
+
if let Expr::Name(n) = &*ann.target {
|
|
847
|
+
self.poison(n.id.as_str());
|
|
848
|
+
}
|
|
849
|
+
}
|
|
850
|
+
Stmt::AugAssign(aug) => self.record_target_names(&aug.target),
|
|
851
|
+
Stmt::For(for_stmt) => self.record_target_names(&for_stmt.target),
|
|
852
|
+
Stmt::Global(g) => {
|
|
853
|
+
for name in &g.names {
|
|
854
|
+
self.poison(name.as_str());
|
|
855
|
+
}
|
|
856
|
+
}
|
|
857
|
+
Stmt::Nonlocal(nl) => {
|
|
858
|
+
for name in &nl.names {
|
|
859
|
+
self.poison(name.as_str());
|
|
860
|
+
}
|
|
861
|
+
}
|
|
862
|
+
Stmt::Delete(del) => {
|
|
863
|
+
for target in &del.targets {
|
|
864
|
+
self.record_target_names(target);
|
|
865
|
+
}
|
|
866
|
+
}
|
|
867
|
+
Stmt::Import(imp) => {
|
|
868
|
+
for alias in &imp.names {
|
|
869
|
+
let bound = alias.asname.as_ref().unwrap_or(&alias.name);
|
|
870
|
+
self.poison(bound.as_str());
|
|
871
|
+
}
|
|
872
|
+
}
|
|
873
|
+
Stmt::ImportFrom(imp) => {
|
|
874
|
+
for alias in &imp.names {
|
|
875
|
+
let bound = alias.asname.as_ref().unwrap_or(&alias.name);
|
|
876
|
+
self.poison(bound.as_str());
|
|
877
|
+
}
|
|
878
|
+
}
|
|
879
|
+
_ => {}
|
|
880
|
+
}
|
|
881
|
+
ast_visitor::walk_stmt(self, stmt);
|
|
882
|
+
}
|
|
883
|
+
|
|
884
|
+
fn visit_except_handler(&mut self, handler: &'a ast::ExceptHandler) {
|
|
885
|
+
let ast::ExceptHandler::ExceptHandler(h) = handler;
|
|
886
|
+
if let Some(name) = &h.name {
|
|
887
|
+
self.poison(name.as_str());
|
|
888
|
+
}
|
|
889
|
+
ast_visitor::walk_except_handler(self, handler);
|
|
890
|
+
}
|
|
891
|
+
|
|
892
|
+
fn visit_parameter(&mut self, parameter: &'a ast::Parameter) {
|
|
893
|
+
self.poison(parameter.name.as_str());
|
|
894
|
+
ast_visitor::walk_parameter(self, parameter);
|
|
895
|
+
}
|
|
896
|
+
|
|
897
|
+
fn visit_with_item(&mut self, item: &'a ast::WithItem) {
|
|
898
|
+
if let Some(vars) = &item.optional_vars {
|
|
899
|
+
self.record_target_names(vars);
|
|
900
|
+
}
|
|
901
|
+
ast_visitor::walk_with_item(self, item);
|
|
902
|
+
}
|
|
903
|
+
|
|
904
|
+
fn visit_comprehension(&mut self, comp: &'a ast::Comprehension) {
|
|
905
|
+
self.record_target_names(&comp.target);
|
|
906
|
+
ast_visitor::walk_comprehension(self, comp);
|
|
907
|
+
}
|
|
908
|
+
|
|
909
|
+
fn visit_expr(&mut self, expr: &'a Expr) {
|
|
910
|
+
if let Expr::Call(call) = expr {
|
|
911
|
+
if let Expr::Attribute(attr) = &*call.func {
|
|
912
|
+
if attr.attr.as_str() == "append" {
|
|
913
|
+
if let Expr::Name(recv) = &*attr.value {
|
|
914
|
+
self.accounted_for.insert(attr.value.range());
|
|
915
|
+
self.record_append(recv.id.as_str(), call.arguments.args.first());
|
|
916
|
+
}
|
|
917
|
+
}
|
|
918
|
+
}
|
|
919
|
+
if is_dataframe_constructor_callee(&call.func) {
|
|
920
|
+
if let Some(first @ Expr::Name(_)) = call.arguments.args.first() {
|
|
921
|
+
self.accounted_for.insert(first.range());
|
|
922
|
+
}
|
|
923
|
+
}
|
|
924
|
+
}
|
|
925
|
+
if let Expr::Named(named) = expr {
|
|
926
|
+
if let Expr::Name(n) = &*named.target {
|
|
927
|
+
self.poison(n.id.as_str());
|
|
928
|
+
}
|
|
929
|
+
}
|
|
930
|
+
if let Expr::Name(name) = expr {
|
|
931
|
+
if !self.accounted_for.contains(&expr.range()) {
|
|
932
|
+
if let Some(DictListBinding::Building(_)) = self.bindings.get(name.id.as_str()) {
|
|
933
|
+
self.poison(name.id.as_str());
|
|
934
|
+
}
|
|
935
|
+
}
|
|
936
|
+
}
|
|
937
|
+
ast_visitor::walk_expr(self, expr);
|
|
938
|
+
}
|
|
939
|
+
}
|
|
940
|
+
|
|
941
|
+
// `DataFrame(...)` or `<anything>.DataFrame(...)` -- the constructor shape
|
|
942
|
+
// `dict_list_var_candidates`/`extract_load_columns` both recognize, matching how the
|
|
943
|
+
// rest of the checker treats a `DataFrame` call regardless of its receiver (`pd.`,
|
|
944
|
+
// `pl.`, an aliased import, ...).
|
|
945
|
+
fn is_dataframe_constructor_callee(func: &Expr) -> bool {
|
|
946
|
+
match func {
|
|
947
|
+
Expr::Name(name) => name.id.as_str() == "DataFrame",
|
|
948
|
+
Expr::Attribute(attr) => attr.attr.as_str() == "DataFrame",
|
|
949
|
+
_ => false,
|
|
950
|
+
}
|
|
951
|
+
}
|
|
952
|
+
|
|
953
|
+
fn collect_dict_list_var_candidates(body: &[Stmt]) -> HashMap<String, Vec<String>> {
|
|
954
|
+
let mut collector = DictListBindingCollector {
|
|
955
|
+
bindings: HashMap::new(),
|
|
956
|
+
accounted_for: std::collections::HashSet::new(),
|
|
957
|
+
};
|
|
958
|
+
for stmt in body {
|
|
959
|
+
collector.visit_stmt(stmt);
|
|
960
|
+
}
|
|
961
|
+
collector
|
|
962
|
+
.bindings
|
|
963
|
+
.into_iter()
|
|
964
|
+
.filter_map(|(name, binding)| match binding {
|
|
965
|
+
DictListBinding::Building(keys) if !keys.is_empty() => Some((name, keys)),
|
|
966
|
+
_ => None,
|
|
967
|
+
})
|
|
968
|
+
.collect()
|
|
969
|
+
}
|
|
970
|
+
|
|
732
971
|
// The row indexer of a `df.loc[rows, cols]` target.
|
|
733
972
|
fn loc_row_indexer(target: &Expr) -> Option<&Expr> {
|
|
734
973
|
let Expr::Subscript(subscript) = target else {
|
|
@@ -1148,6 +1387,7 @@ impl Linter {
|
|
|
1148
1387
|
sql_dialect: sql::SqlDialect::Generic,
|
|
1149
1388
|
project_root: None,
|
|
1150
1389
|
string_var_candidates: HashMap::new(),
|
|
1390
|
+
dict_list_var_candidates: HashMap::new(),
|
|
1151
1391
|
bool_var_candidates: HashMap::new(),
|
|
1152
1392
|
stmt_var_candidates: HashMap::new(),
|
|
1153
1393
|
retrieval_jobs: HashMap::new(),
|
|
@@ -1271,6 +1511,7 @@ impl Linter {
|
|
|
1271
1511
|
self.line_index = Some(LineIndex::from_source_text(source));
|
|
1272
1512
|
(self.string_var_candidates, self.bool_var_candidates) =
|
|
1273
1513
|
self.collect_literal_var_candidates(&module.body, path);
|
|
1514
|
+
self.dict_list_var_candidates = collect_dict_list_var_candidates(&module.body);
|
|
1274
1515
|
let mut df_usage_collector = DataFrameShapedUsageCollector {
|
|
1275
1516
|
names: std::collections::HashSet::new(),
|
|
1276
1517
|
};
|
|
@@ -1724,18 +1965,35 @@ impl Linter {
|
|
|
1724
1965
|
// never fires for read_csv/read_json/etc., whose first positional argument is
|
|
1725
1966
|
// a path or buffer, not column data — a dict there means something else
|
|
1726
1967
|
// entirely and inferring columns from it would be wrong.
|
|
1968
|
+
// `DataFrame([{...}, {...}])` -- a literal list of per-row dict literals
|
|
1969
|
+
// (records orientation), or `DataFrame(records)` where `records` is a name
|
|
1970
|
+
// this file's own pre-pass traced back to such a list built via `.append()`
|
|
1971
|
+
// in a loop -- see `Linter::dict_list_var_candidates`.
|
|
1727
1972
|
if func_name == "DataFrame" {
|
|
1728
|
-
|
|
1729
|
-
|
|
1730
|
-
|
|
1731
|
-
|
|
1732
|
-
|
|
1733
|
-
|
|
1734
|
-
|
|
1735
|
-
|
|
1736
|
-
|
|
1737
|
-
|
|
1973
|
+
match call.arguments.args.first() {
|
|
1974
|
+
Some(Expr::Dict(dict)) => {
|
|
1975
|
+
let keys: Vec<String> = dict
|
|
1976
|
+
.items
|
|
1977
|
+
.iter()
|
|
1978
|
+
.filter_map(|item| item.key.as_ref())
|
|
1979
|
+
.filter_map(|k| ast_extract::extract_string_literal(k))
|
|
1980
|
+
.map(|s| s.to_string())
|
|
1981
|
+
.collect();
|
|
1982
|
+
if !keys.is_empty() {
|
|
1983
|
+
return (Some(keys), LoadKind::File);
|
|
1984
|
+
}
|
|
1985
|
+
}
|
|
1986
|
+
Some(Expr::List(list)) => {
|
|
1987
|
+
if let Some(cols) = ast_extract::extract_records_list_columns(list) {
|
|
1988
|
+
return (Some(cols), LoadKind::File);
|
|
1989
|
+
}
|
|
1990
|
+
}
|
|
1991
|
+
Some(Expr::Name(name)) => {
|
|
1992
|
+
if let Some(cols) = self.dict_list_var_candidates.get(name.id.as_str()) {
|
|
1993
|
+
return (Some(cols.clone()), LoadKind::File);
|
|
1994
|
+
}
|
|
1738
1995
|
}
|
|
1996
|
+
_ => {}
|
|
1739
1997
|
}
|
|
1740
1998
|
}
|
|
1741
1999
|
|
|
@@ -8429,6 +8687,301 @@ print(df["a"])
|
|
|
8429
8687
|
assert_eq!(linter.dataframes_typed, 1);
|
|
8430
8688
|
}
|
|
8431
8689
|
|
|
8690
|
+
#[test]
|
|
8691
|
+
fn test_should_infer_columns_from_a_literal_list_of_dict_literals() {
|
|
8692
|
+
// arrange: pandas' "records" orientation -- one dict literal per row, passed
|
|
8693
|
+
// directly as a list literal. Columns are the union of every row's keys.
|
|
8694
|
+
let source = r#"
|
|
8695
|
+
import pandas as pd
|
|
8696
|
+
|
|
8697
|
+
df = pd.DataFrame([{"a": 1, "b": 2}, {"a": 3, "c": 4}])
|
|
8698
|
+
print(df["a"])
|
|
8699
|
+
print(df["c"])
|
|
8700
|
+
print(df["nope"])
|
|
8701
|
+
"#;
|
|
8702
|
+
let mut linter = Linter::new();
|
|
8703
|
+
|
|
8704
|
+
// act
|
|
8705
|
+
let errors = linter
|
|
8706
|
+
.check_file_internal(source, Path::new("test.py"))
|
|
8707
|
+
.unwrap();
|
|
8708
|
+
|
|
8709
|
+
// assert: union is {a, b, c}, in first-seen order; only the genuine typo flags.
|
|
8710
|
+
assert_eq!(linter.dataframes_total, 1);
|
|
8711
|
+
assert_eq!(linter.dataframes_typed, 1);
|
|
8712
|
+
assert_eq!(errors.len(), 1, "errors: {errors:?}");
|
|
8713
|
+
assert!(errors[0].message.contains("'nope'"));
|
|
8714
|
+
assert!(errors[0].message.contains("{a, b, c}"));
|
|
8715
|
+
}
|
|
8716
|
+
|
|
8717
|
+
#[test]
|
|
8718
|
+
fn test_should_leave_a_list_of_non_dict_elements_untracked() {
|
|
8719
|
+
// arrange: not every element is a dict literal -- can't infer columns from it.
|
|
8720
|
+
let source = r#"
|
|
8721
|
+
import pandas as pd
|
|
8722
|
+
|
|
8723
|
+
df = pd.DataFrame([{"a": 1}, "not a dict"])
|
|
8724
|
+
print(df["a"])
|
|
8725
|
+
"#;
|
|
8726
|
+
let mut linter = Linter::new();
|
|
8727
|
+
|
|
8728
|
+
// act
|
|
8729
|
+
let errors = linter
|
|
8730
|
+
.check_file_internal(source, Path::new("test.py"))
|
|
8731
|
+
.unwrap();
|
|
8732
|
+
|
|
8733
|
+
// assert: recognized as a DataFrame origin, but not typed -- no false claim.
|
|
8734
|
+
assert_eq!(linter.dataframes_total, 1);
|
|
8735
|
+
assert_eq!(linter.dataframes_typed, 0);
|
|
8736
|
+
assert!(
|
|
8737
|
+
errors
|
|
8738
|
+
.iter()
|
|
8739
|
+
.all(|e| e.code != CODE_UNKNOWN_COLUMN && e.code != CODE_UNVERIFIABLE_COLUMN),
|
|
8740
|
+
"errors: {errors:?}"
|
|
8741
|
+
);
|
|
8742
|
+
}
|
|
8743
|
+
|
|
8744
|
+
#[test]
|
|
8745
|
+
fn test_should_infer_columns_from_an_append_loop_accumulator() {
|
|
8746
|
+
// arrange: `records = []` then `.append({literal dict})` calls in a loop --
|
|
8747
|
+
// the shape a `pd.DataFrame([{...}])` literal can't cover on its own.
|
|
8748
|
+
let source = r#"
|
|
8749
|
+
import pandas as pd
|
|
8750
|
+
|
|
8751
|
+
records = []
|
|
8752
|
+
for row in rows:
|
|
8753
|
+
records.append({"a": row.a, "b": row.b})
|
|
8754
|
+
|
|
8755
|
+
df = pd.DataFrame(records)
|
|
8756
|
+
print(df["a"])
|
|
8757
|
+
print(df["nope"])
|
|
8758
|
+
"#;
|
|
8759
|
+
let mut linter = Linter::new();
|
|
8760
|
+
|
|
8761
|
+
// act
|
|
8762
|
+
let errors = linter
|
|
8763
|
+
.check_file_internal(source, Path::new("test.py"))
|
|
8764
|
+
.unwrap();
|
|
8765
|
+
|
|
8766
|
+
// assert
|
|
8767
|
+
assert_eq!(linter.dataframes_total, 1);
|
|
8768
|
+
assert_eq!(linter.dataframes_typed, 1);
|
|
8769
|
+
assert_eq!(errors.len(), 1, "errors: {errors:?}");
|
|
8770
|
+
assert!(errors[0].message.contains("'nope'"));
|
|
8771
|
+
assert!(errors[0].message.contains("{a, b}"));
|
|
8772
|
+
}
|
|
8773
|
+
|
|
8774
|
+
#[test]
|
|
8775
|
+
fn test_should_union_keys_across_nested_loop_appends_in_first_seen_order() {
|
|
8776
|
+
// arrange: a nested loop, appending the same shape of dict every iteration --
|
|
8777
|
+
// the pattern this feature exists for.
|
|
8778
|
+
let source = r#"
|
|
8779
|
+
import pandas as pd
|
|
8780
|
+
|
|
8781
|
+
records = []
|
|
8782
|
+
for x in xs:
|
|
8783
|
+
for y in ys:
|
|
8784
|
+
records.append({"sku": x, "date": y, "price": 1})
|
|
8785
|
+
|
|
8786
|
+
df = pd.DataFrame(records)
|
|
8787
|
+
print(df["sku"])
|
|
8788
|
+
print(df["price"])
|
|
8789
|
+
print(df["nope"])
|
|
8790
|
+
"#;
|
|
8791
|
+
let mut linter = Linter::new();
|
|
8792
|
+
|
|
8793
|
+
// act
|
|
8794
|
+
let errors = linter
|
|
8795
|
+
.check_file_internal(source, Path::new("test.py"))
|
|
8796
|
+
.unwrap();
|
|
8797
|
+
|
|
8798
|
+
// assert
|
|
8799
|
+
assert_eq!(errors.len(), 1, "errors: {errors:?}");
|
|
8800
|
+
assert!(errors[0].message.contains("'nope'"));
|
|
8801
|
+
assert!(errors[0].message.contains("{sku, date, price}"));
|
|
8802
|
+
}
|
|
8803
|
+
|
|
8804
|
+
#[test]
|
|
8805
|
+
fn test_should_leave_an_accumulator_untracked_when_mutated_by_another_function() {
|
|
8806
|
+
// arrange: `helper` could append anything -- a call we can't see into must not
|
|
8807
|
+
// be silently assumed harmless.
|
|
8808
|
+
let source = r#"
|
|
8809
|
+
import pandas as pd
|
|
8810
|
+
|
|
8811
|
+
def helper(records):
|
|
8812
|
+
records.append({"weird_key": 1})
|
|
8813
|
+
|
|
8814
|
+
records = []
|
|
8815
|
+
records.append({"a": 1})
|
|
8816
|
+
helper(records)
|
|
8817
|
+
df = pd.DataFrame(records)
|
|
8818
|
+
print(df["a"])
|
|
8819
|
+
"#;
|
|
8820
|
+
let mut linter = Linter::new();
|
|
8821
|
+
|
|
8822
|
+
// act
|
|
8823
|
+
let errors = linter
|
|
8824
|
+
.check_file_internal(source, Path::new("test.py"))
|
|
8825
|
+
.unwrap();
|
|
8826
|
+
|
|
8827
|
+
// assert: no false unknown-column against an incomplete {a}-only schema.
|
|
8828
|
+
assert_eq!(linter.dataframes_typed, 0);
|
|
8829
|
+
assert!(
|
|
8830
|
+
errors
|
|
8831
|
+
.iter()
|
|
8832
|
+
.all(|e| e.code != CODE_UNKNOWN_COLUMN && e.code != CODE_UNVERIFIABLE_COLUMN),
|
|
8833
|
+
"errors: {errors:?}"
|
|
8834
|
+
);
|
|
8835
|
+
}
|
|
8836
|
+
|
|
8837
|
+
#[test]
|
|
8838
|
+
fn test_should_leave_an_accumulator_untracked_after_a_non_append_mutation() {
|
|
8839
|
+
// arrange
|
|
8840
|
+
let source = r#"
|
|
8841
|
+
import pandas as pd
|
|
8842
|
+
|
|
8843
|
+
records = []
|
|
8844
|
+
records.append({"a": 1})
|
|
8845
|
+
records.extend([{"b": 2}])
|
|
8846
|
+
df = pd.DataFrame(records)
|
|
8847
|
+
print(df["a"])
|
|
8848
|
+
"#;
|
|
8849
|
+
let mut linter = Linter::new();
|
|
8850
|
+
|
|
8851
|
+
// act
|
|
8852
|
+
let errors = linter
|
|
8853
|
+
.check_file_internal(source, Path::new("test.py"))
|
|
8854
|
+
.unwrap();
|
|
8855
|
+
|
|
8856
|
+
// assert
|
|
8857
|
+
assert_eq!(linter.dataframes_typed, 0);
|
|
8858
|
+
assert!(
|
|
8859
|
+
errors
|
|
8860
|
+
.iter()
|
|
8861
|
+
.all(|e| e.code != CODE_UNKNOWN_COLUMN && e.code != CODE_UNVERIFIABLE_COLUMN),
|
|
8862
|
+
"errors: {errors:?}"
|
|
8863
|
+
);
|
|
8864
|
+
}
|
|
8865
|
+
|
|
8866
|
+
#[test]
|
|
8867
|
+
fn test_should_leave_an_accumulator_untracked_after_an_append_with_a_dynamic_key() {
|
|
8868
|
+
// arrange
|
|
8869
|
+
let source = r#"
|
|
8870
|
+
import pandas as pd
|
|
8871
|
+
|
|
8872
|
+
records = []
|
|
8873
|
+
key = "a"
|
|
8874
|
+
records.append({key: 1})
|
|
8875
|
+
df = pd.DataFrame(records)
|
|
8876
|
+
print(df["a"])
|
|
8877
|
+
"#;
|
|
8878
|
+
let mut linter = Linter::new();
|
|
8879
|
+
|
|
8880
|
+
// act
|
|
8881
|
+
let errors = linter
|
|
8882
|
+
.check_file_internal(source, Path::new("test.py"))
|
|
8883
|
+
.unwrap();
|
|
8884
|
+
|
|
8885
|
+
// assert
|
|
8886
|
+
assert_eq!(linter.dataframes_typed, 0);
|
|
8887
|
+
assert!(
|
|
8888
|
+
errors
|
|
8889
|
+
.iter()
|
|
8890
|
+
.all(|e| e.code != CODE_UNKNOWN_COLUMN && e.code != CODE_UNVERIFIABLE_COLUMN),
|
|
8891
|
+
"errors: {errors:?}"
|
|
8892
|
+
);
|
|
8893
|
+
}
|
|
8894
|
+
|
|
8895
|
+
#[test]
|
|
8896
|
+
fn test_should_leave_an_accumulator_untracked_after_a_read_only_reference() {
|
|
8897
|
+
// arrange: even a harmless-looking `len(records)` before construction poisons
|
|
8898
|
+
// it -- deliberately conservative, since distinguishing "definitely read-only"
|
|
8899
|
+
// calls from ones that might mutate isn't attempted.
|
|
8900
|
+
let source = r#"
|
|
8901
|
+
import pandas as pd
|
|
8902
|
+
|
|
8903
|
+
records = []
|
|
8904
|
+
records.append({"a": 1})
|
|
8905
|
+
print(len(records))
|
|
8906
|
+
df = pd.DataFrame(records)
|
|
8907
|
+
print(df["a"])
|
|
8908
|
+
"#;
|
|
8909
|
+
let mut linter = Linter::new();
|
|
8910
|
+
|
|
8911
|
+
// act
|
|
8912
|
+
let errors = linter
|
|
8913
|
+
.check_file_internal(source, Path::new("test.py"))
|
|
8914
|
+
.unwrap();
|
|
8915
|
+
|
|
8916
|
+
// assert
|
|
8917
|
+
assert_eq!(linter.dataframes_typed, 0);
|
|
8918
|
+
assert!(
|
|
8919
|
+
errors
|
|
8920
|
+
.iter()
|
|
8921
|
+
.all(|e| e.code != CODE_UNKNOWN_COLUMN && e.code != CODE_UNVERIFIABLE_COLUMN),
|
|
8922
|
+
"errors: {errors:?}"
|
|
8923
|
+
);
|
|
8924
|
+
}
|
|
8925
|
+
|
|
8926
|
+
#[test]
|
|
8927
|
+
fn test_should_leave_an_accumulator_untracked_after_being_reinitialized() {
|
|
8928
|
+
// arrange: two separate `records = []` bindings -- ambiguous which loop's
|
|
8929
|
+
// appends belong to "the" list by the time DataFrame(records) is reached.
|
|
8930
|
+
let source = r#"
|
|
8931
|
+
import pandas as pd
|
|
8932
|
+
|
|
8933
|
+
records = []
|
|
8934
|
+
records.append({"a": 1})
|
|
8935
|
+
records = []
|
|
8936
|
+
records.append({"b": 2})
|
|
8937
|
+
df = pd.DataFrame(records)
|
|
8938
|
+
print(df["b"])
|
|
8939
|
+
"#;
|
|
8940
|
+
let mut linter = Linter::new();
|
|
8941
|
+
|
|
8942
|
+
// act
|
|
8943
|
+
let errors = linter
|
|
8944
|
+
.check_file_internal(source, Path::new("test.py"))
|
|
8945
|
+
.unwrap();
|
|
8946
|
+
|
|
8947
|
+
// assert
|
|
8948
|
+
assert_eq!(linter.dataframes_typed, 0);
|
|
8949
|
+
assert!(
|
|
8950
|
+
errors
|
|
8951
|
+
.iter()
|
|
8952
|
+
.all(|e| e.code != CODE_UNKNOWN_COLUMN && e.code != CODE_UNVERIFIABLE_COLUMN),
|
|
8953
|
+
"errors: {errors:?}"
|
|
8954
|
+
);
|
|
8955
|
+
}
|
|
8956
|
+
|
|
8957
|
+
#[test]
|
|
8958
|
+
fn test_should_not_infer_columns_for_an_unrelated_name_that_happens_to_hold_a_list() {
|
|
8959
|
+
// arrange: `items` is never initialized as `[]`, so it's simply never a
|
|
8960
|
+
// candidate -- not poisoned, just never tracked in the first place.
|
|
8961
|
+
let source = r#"
|
|
8962
|
+
import pandas as pd
|
|
8963
|
+
|
|
8964
|
+
items = get_items()
|
|
8965
|
+
df = pd.DataFrame(items)
|
|
8966
|
+
print(df["anything"])
|
|
8967
|
+
"#;
|
|
8968
|
+
let mut linter = Linter::new();
|
|
8969
|
+
|
|
8970
|
+
// act
|
|
8971
|
+
let errors = linter
|
|
8972
|
+
.check_file_internal(source, Path::new("test.py"))
|
|
8973
|
+
.unwrap();
|
|
8974
|
+
|
|
8975
|
+
// assert
|
|
8976
|
+
assert_eq!(linter.dataframes_typed, 0);
|
|
8977
|
+
assert!(
|
|
8978
|
+
errors
|
|
8979
|
+
.iter()
|
|
8980
|
+
.all(|e| e.code != CODE_UNKNOWN_COLUMN && e.code != CODE_UNVERIFIABLE_COLUMN),
|
|
8981
|
+
"errors: {errors:?}"
|
|
8982
|
+
);
|
|
8983
|
+
}
|
|
8984
|
+
|
|
8432
8985
|
#[test]
|
|
8433
8986
|
fn test_should_infer_columns_from_dataframe_constructor_dict_literal() {
|
|
8434
8987
|
// arrange: a dict literal passed directly to DataFrame() names its columns as
|
|
@@ -671,6 +671,80 @@ class RunStats:
|
|
|
671
671
|
dataframes_typed: int
|
|
672
672
|
|
|
673
673
|
|
|
674
|
+
def _split_errors_by_confidence(errors_only: list[dict]) -> tuple[list[dict], list[dict]]:
|
|
675
|
+
"""Split error-severity diagnostics into confirmed and unverifiable ones.
|
|
676
|
+
|
|
677
|
+
"Confirmed" means definitely wrong; "unverifiable" means the checker couldn't
|
|
678
|
+
check the access at all, not "checked and it's wrong" -- the summary line
|
|
679
|
+
counts them separately so "1 error" never quietly means "1 access the checker
|
|
680
|
+
had no basis to judge".
|
|
681
|
+
"""
|
|
682
|
+
confirmed = [e for e in errors_only if e.get("code") != "unverifiable-column"]
|
|
683
|
+
unverifiable = [e for e in errors_only if e.get("code") == "unverifiable-column"]
|
|
684
|
+
return confirmed, unverifiable
|
|
685
|
+
|
|
686
|
+
|
|
687
|
+
def _error_summary_parts(errors_only: list[dict], warnings: list[dict]) -> list[str]:
|
|
688
|
+
"""The comma-joined pieces of the "Found ..." summary line.
|
|
689
|
+
|
|
690
|
+
Confirmed errors, unverifiable ones, and warnings, each omitted when there are
|
|
691
|
+
none of that kind. Confirmed and unverifiable are only split into two phrases
|
|
692
|
+
when at least one unverifiable-column diagnostic is present -- with none, this
|
|
693
|
+
reads exactly as it always has ("N errors"), so the common case is unchanged.
|
|
694
|
+
"""
|
|
695
|
+
confirmed, unverifiable = _split_errors_by_confidence(errors_only)
|
|
696
|
+
parts = []
|
|
697
|
+
if confirmed or not unverifiable:
|
|
698
|
+
label = "error" if len(confirmed) == 1 else "errors"
|
|
699
|
+
parts.append(f"{len(confirmed)} {label}")
|
|
700
|
+
if unverifiable:
|
|
701
|
+
label = "unverifiable error" if len(unverifiable) == 1 else "unverifiable errors"
|
|
702
|
+
parts.append(f"{len(unverifiable)} {label}")
|
|
703
|
+
if warnings:
|
|
704
|
+
label = "warning" if len(warnings) == 1 else "warnings"
|
|
705
|
+
parts.append(f"{len(warnings)} {label}")
|
|
706
|
+
return parts
|
|
707
|
+
|
|
708
|
+
|
|
709
|
+
# Below this fraction of tracked DataFrames having a concrete column list, the
|
|
710
|
+
# error/warning summary leads with a coverage caveat instead of trailing it -- see
|
|
711
|
+
# `_low_coverage_caveat`.
|
|
712
|
+
_LOW_COVERAGE_THRESHOLD = 0.5
|
|
713
|
+
|
|
714
|
+
|
|
715
|
+
def _low_coverage_caveat(stats: RunStats, all_errors: list[dict]) -> str | None:
|
|
716
|
+
"""A coverage-ratio line noting that the diagnostics below only reflect what little could be tracked.
|
|
717
|
+
|
|
718
|
+
Printed FIRST, ahead of the diagnostics themselves, whenever under half the
|
|
719
|
+
DataFrames the checker saw had a concrete column list. Below that threshold a
|
|
720
|
+
small "N errors" count reads as reassuring ("only N problems") when the more
|
|
721
|
+
important fact is that most of the run was never actually checked;
|
|
722
|
+
`_coverage_message` carries the same caveat in its usual, easy-to-miss spot at
|
|
723
|
+
the very end of the output, for everything above the threshold.
|
|
724
|
+
|
|
725
|
+
`all_errors` must be the exact list about to be printed (post `--no-warnings`/
|
|
726
|
+
`--no-info`/`--lenient-ingest` filtering, done by `_apply_diagnostic_policy`
|
|
727
|
+
before this is ever called) -- not `errors_only`/`warnings`, which exclude
|
|
728
|
+
info-severity diagnostics (e.g. `--lenient-ingest`'s untracked-dataframe
|
|
729
|
+
downgrade) that still print in the body, and would otherwise undercount. With
|
|
730
|
+
`all_errors` empty -- genuinely nothing found, or everything filtered out by one
|
|
731
|
+
of those flags -- this returns `None` rather than a "0 findings below" caveat
|
|
732
|
+
with nothing actually below it; `_coverage_message`'s own plain wording already
|
|
733
|
+
covers that case correctly, with no claim about what's printed alongside it.
|
|
734
|
+
"""
|
|
735
|
+
if not all_errors:
|
|
736
|
+
return None
|
|
737
|
+
if stats.dataframes_total == 0 or stats.dataframes_typed / stats.dataframes_total >= _LOW_COVERAGE_THRESHOLD:
|
|
738
|
+
return None
|
|
739
|
+
pct = round(100 * stats.dataframes_typed / stats.dataframes_total)
|
|
740
|
+
findings = len(all_errors)
|
|
741
|
+
noun, verb = ("finding", "covers") if findings == 1 else ("findings", "cover")
|
|
742
|
+
return (
|
|
743
|
+
f"\u2139 {stats.dataframes_typed}/{stats.dataframes_total} DataFrames had column info "
|
|
744
|
+
f"({pct}%) \u2014 the {findings} {noun} below only {verb} what little could be tracked"
|
|
745
|
+
)
|
|
746
|
+
|
|
747
|
+
|
|
674
748
|
def _coverage_message(stats: RunStats) -> str:
|
|
675
749
|
"""Build the low-key DataFrame schema coverage summary line.
|
|
676
750
|
|
|
@@ -1151,12 +1225,19 @@ def _print_results(
|
|
|
1151
1225
|
use_color = output_format == "text" and hasattr(sys.stdout, "isatty") and sys.stdout.isatty()
|
|
1152
1226
|
|
|
1153
1227
|
if output_format == "github":
|
|
1228
|
+
caveat = _low_coverage_caveat(stats, all_errors) if show_info else None
|
|
1229
|
+
if caveat:
|
|
1230
|
+
print(f"::notice title=typedframes DataFrame schema coverage::{caveat[2:]}")
|
|
1154
1231
|
if all_errors:
|
|
1155
1232
|
print(_format_github(all_errors))
|
|
1156
|
-
if show_info:
|
|
1233
|
+
if show_info and not caveat:
|
|
1157
1234
|
print(f"::notice title=typedframes DataFrame schema coverage::{_coverage_message(stats)[2:]}")
|
|
1158
1235
|
return
|
|
1159
1236
|
|
|
1237
|
+
caveat = _low_coverage_caveat(stats, all_errors) if show_info else None
|
|
1238
|
+
if caveat:
|
|
1239
|
+
print(f"{_DIM}{caveat}{_RESET}" if use_color else caveat)
|
|
1240
|
+
|
|
1160
1241
|
# text format
|
|
1161
1242
|
if all_errors:
|
|
1162
1243
|
print(_format_text(all_errors, color=use_color))
|
|
@@ -1164,21 +1245,14 @@ def _print_results(
|
|
|
1164
1245
|
|
|
1165
1246
|
file_label = "file" if len(files) == 1 else "files"
|
|
1166
1247
|
if errors_only or warnings:
|
|
1167
|
-
|
|
1168
|
-
if errors_only:
|
|
1169
|
-
error_label = "error" if len(errors_only) == 1 else "errors"
|
|
1170
|
-
parts.append(f"{len(errors_only)} {error_label}")
|
|
1171
|
-
if warnings:
|
|
1172
|
-
warn_label = "warning" if len(warnings) == 1 else "warnings"
|
|
1173
|
-
parts.append(f"{len(warnings)} {warn_label}")
|
|
1174
|
-
summary = ", ".join(parts)
|
|
1248
|
+
summary = ", ".join(_error_summary_parts(errors_only, warnings))
|
|
1175
1249
|
msg = f"\u2717 Found {summary} in {len(files)} {file_label} ({stats.elapsed:.1f}s)"
|
|
1176
1250
|
print(f"{_BOLD_RED}{msg}{_RESET}" if use_color else msg)
|
|
1177
1251
|
else:
|
|
1178
1252
|
msg = f"\u2713 Checked {len(files)} {file_label} in {stats.elapsed:.1f}s"
|
|
1179
1253
|
print(f"{_BOLD_GREEN}{msg}{_RESET}" if use_color else msg)
|
|
1180
1254
|
|
|
1181
|
-
if show_info:
|
|
1255
|
+
if show_info and not caveat:
|
|
1182
1256
|
coverage_msg = _coverage_message(stats)
|
|
1183
1257
|
print(f"{_DIM}{coverage_msg}{_RESET}" if use_color else coverage_msg)
|
|
1184
1258
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|