whyvalue 0.1.0__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- whyvalue-0.2.0/PKG-INFO +242 -0
- whyvalue-0.2.0/README.md +214 -0
- {whyvalue-0.1.0 → whyvalue-0.2.0}/pyproject.toml +3 -2
- whyvalue-0.2.0/src/whyvalue/__init__.py +31 -0
- whyvalue-0.2.0/src/whyvalue/adapters/csv_adapter.py +93 -0
- whyvalue-0.2.0/src/whyvalue/adapters/http_adapter.py +220 -0
- whyvalue-0.2.0/src/whyvalue/adapters/json_adapter.py +114 -0
- {whyvalue-0.1.0 → whyvalue-0.2.0}/src/whyvalue/adapters/pandas.py +96 -12
- whyvalue-0.2.0/src/whyvalue/adapters/python_dict.py +168 -0
- whyvalue-0.2.0/src/whyvalue/adapters/python_list.py +165 -0
- whyvalue-0.2.0/src/whyvalue/adapters/txt_adapter.py +60 -0
- whyvalue-0.2.0/src/whyvalue/core.py +1944 -0
- {whyvalue-0.1.0 → whyvalue-0.2.0}/src/whyvalue/history.py +5 -0
- whyvalue-0.2.0/src/whyvalue/provenance.py +120 -0
- whyvalue-0.2.0/src/whyvalue.egg-info/PKG-INFO +242 -0
- whyvalue-0.2.0/src/whyvalue.egg-info/SOURCES.txt +33 -0
- whyvalue-0.2.0/src/whyvalue.egg-info/requires.txt +2 -0
- whyvalue-0.2.0/tests/test_audit.py +62 -0
- whyvalue-0.2.0/tests/test_context_manager.py +143 -0
- whyvalue-0.2.0/tests/test_csv.py +217 -0
- whyvalue-0.2.0/tests/test_explain_modes.py +132 -0
- whyvalue-0.2.0/tests/test_http.py +320 -0
- whyvalue-0.2.0/tests/test_json.py +233 -0
- whyvalue-0.2.0/tests/test_provenance.py +102 -0
- whyvalue-0.2.0/tests/test_python_dict.py +287 -0
- whyvalue-0.2.0/tests/test_python_list.py +174 -0
- whyvalue-0.2.0/tests/test_snapshots.py +93 -0
- whyvalue-0.2.0/tests/test_to_dataframe.py +184 -0
- whyvalue-0.2.0/tests/test_txt.py +252 -0
- whyvalue-0.1.0/PKG-INFO +0 -337
- whyvalue-0.1.0/README.md +0 -310
- whyvalue-0.1.0/src/whyvalue/__init__.py +0 -17
- whyvalue-0.1.0/src/whyvalue/core.py +0 -638
- whyvalue-0.1.0/src/whyvalue.egg-info/PKG-INFO +0 -337
- whyvalue-0.1.0/src/whyvalue.egg-info/SOURCES.txt +0 -14
- whyvalue-0.1.0/src/whyvalue.egg-info/requires.txt +0 -1
- {whyvalue-0.1.0 → whyvalue-0.2.0}/LICENSE +0 -0
- {whyvalue-0.1.0 → whyvalue-0.2.0}/setup.cfg +0 -0
- {whyvalue-0.1.0 → whyvalue-0.2.0}/src/whyvalue/adapters/__init__.py +0 -0
- {whyvalue-0.1.0 → whyvalue-0.2.0}/src/whyvalue.egg-info/dependency_links.txt +0 -0
- {whyvalue-0.1.0 → whyvalue-0.2.0}/src/whyvalue.egg-info/top_level.txt +0 -0
- {whyvalue-0.1.0 → whyvalue-0.2.0}/tests/test_pandas.py +0 -0
whyvalue-0.2.0/PKG-INFO
ADDED
|
@@ -0,0 +1,242 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: whyvalue
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Ask your data why.
|
|
5
|
+
Author: Muktar Yakub
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/Muktaryy/whyvalue
|
|
8
|
+
Project-URL: Repository, https://github.com/Muktaryy/whyvalue
|
|
9
|
+
Project-URL: Issues, https://github.com/Muktaryy/whyvalue/issues
|
|
10
|
+
Keywords: pandas,data-debugging,data-lineage,provenance,data-engineering,debugging
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Programming Language :: Python
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
20
|
+
Classifier: Topic :: Software Development :: Debuggers
|
|
21
|
+
Classifier: Topic :: Scientific/Engineering :: Information Analysis
|
|
22
|
+
Requires-Python: >=3.10
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
License-File: LICENSE
|
|
25
|
+
Requires-Dist: pandas>=2.0
|
|
26
|
+
Requires-Dist: requests>=2.31
|
|
27
|
+
Dynamic: license-file
|
|
28
|
+
|
|
29
|
+
<p align="center">
|
|
30
|
+
<a href="https://muktaryy.github.io/whyvalue/">
|
|
31
|
+
<img src="https://raw.githubusercontent.com/Muktaryy/whyvalue/main/docs/assets/whyvalue-readme.png" alt="WhyValue Logo" width="400" />
|
|
32
|
+
</a>
|
|
33
|
+
</p>
|
|
34
|
+
|
|
35
|
+
<h3 align="center">Ask Your Data Why</h3>
|
|
36
|
+
|
|
37
|
+
<p align="center">
|
|
38
|
+
<em>Transparent data lineage and value provenance for Python.</em>
|
|
39
|
+
</p>
|
|
40
|
+
|
|
41
|
+
<p align="center">
|
|
42
|
+
<a href="https://pypi.org/project/whyvalue/"><img src="https://img.shields.io/pypi/v/whyvalue.svg" alt="PyPI Version"></a>
|
|
43
|
+
<a href="https://pypi.org/project/whyvalue/"><img src="https://img.shields.io/pypi/pyversions/whyvalue.svg" alt="Python Versions"></a>
|
|
44
|
+
<a href="https://github.com/Muktaryy/whyvalue/blob/main/LICENSE"><img src="https://img.shields.io/github/license/Muktaryy/whyvalue.svg" alt="License"></a>
|
|
45
|
+
<a href="https://muktaryy.github.io/whyvalue/"><img src="https://img.shields.io/badge/docs-live-blue.svg" alt="Documentation"></a>
|
|
46
|
+
</p>
|
|
47
|
+
|
|
48
|
+
---
|
|
49
|
+
|
|
50
|
+
## What is WhyValue?
|
|
51
|
+
|
|
52
|
+
**WhyValue** is a developer-first Python library for data lineage, mutation tracking, and value provenance.
|
|
53
|
+
|
|
54
|
+
When a pipeline yields an unexpected calculation, a missing field, or an invalid metric, traditional debugging requires stepping through code line-by-line or placing manual print statements across modules. **WhyValue** eliminates this guesswork by capturing the history of your data as it flows from file and network sources through Python objects and pandas DataFrames.
|
|
55
|
+
|
|
56
|
+
Whenever you ask **"Why does this variable have this value?"**, WhyValue delivers a human-readable explanation of its origin, file boundaries, mathematical operations, and mutation sequence.
|
|
57
|
+
|
|
58
|
+
---
|
|
59
|
+
|
|
60
|
+
## Core Mental Model
|
|
61
|
+
|
|
62
|
+
WhyValue operates as a runtime observer:
|
|
63
|
+
|
|
64
|
+
```
|
|
65
|
+
┌─────────────────┐ ┌──────────────────────┐ ┌─────────────────┐ ┌──────────────────────┐
|
|
66
|
+
│ Source Data │ ────► │ Transformations │ ────► │ Target Value │ ────► │ Lineage & Provenance│
|
|
67
|
+
│ CSV, JSON, HTTP │ │ Math, Mutates, pandas│ │ Dict, DF, List │ │ why.explain(target) │
|
|
68
|
+
└─────────────────┘ └──────────────────────┘ └─────────────────┘ └──────────────────────┘
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
1. **Watch**: Wrap execution blocks with `with why.watch():` to activate observation.
|
|
72
|
+
2. **Track**: Ingest file/HTTP sources (`why.load_csv`, `why.get`) or wrap native structures (`why.track()`).
|
|
73
|
+
3. **Explain**: Query any value or cell at any point using `why.explain()` to inspect its transformation sequence.
|
|
74
|
+
|
|
75
|
+
---
|
|
76
|
+
|
|
77
|
+
## Key Features
|
|
78
|
+
|
|
79
|
+
- 🔍 **Non-Invasive Tracing**: Transparently observes object mutations without altering standard Python semantics.
|
|
80
|
+
- 🌐 **Cross-Source Lineage**: Tracks data across CSV files, JSON documents, text files, and HTTP API responses.
|
|
81
|
+
- 🐼 **Pandas Integration**: Observes DataFrame column creation, element-wise arithmetic, filtering, and assignment.
|
|
82
|
+
- 📜 **Transformation History**: Explains origins down to file names, source line/row indices, and math formulas.
|
|
83
|
+
- 📸 **Snapshot Auditing**: Optional before-and-after state capturing for auditing using `why.watch(snapshot=True)`.
|
|
84
|
+
- ⚡ **Lightweight & Fast**: Pure Python core with minimal dependency requirements for native tracking.
|
|
85
|
+
|
|
86
|
+
---
|
|
87
|
+
|
|
88
|
+
## Installation
|
|
89
|
+
|
|
90
|
+
Install WhyValue via PyPI using `pip`:
|
|
91
|
+
|
|
92
|
+
```bash
|
|
93
|
+
pip install whyvalue
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
*Requirements: Python 3.10+*
|
|
97
|
+
|
|
98
|
+
---
|
|
99
|
+
|
|
100
|
+
## Quick Start
|
|
101
|
+
|
|
102
|
+
### 1. Tracking Native Python Objects
|
|
103
|
+
|
|
104
|
+
Track mutations on standard dictionaries and lists:
|
|
105
|
+
|
|
106
|
+
```python
|
|
107
|
+
import whyvalue as why
|
|
108
|
+
|
|
109
|
+
with why.watch():
|
|
110
|
+
user = why.track({"name": "Ali", "age": 21})
|
|
111
|
+
user["age"] = 22
|
|
112
|
+
user["status"] = "active"
|
|
113
|
+
|
|
114
|
+
# Ask WhyValue why user['age'] equals 22
|
|
115
|
+
why.explain(user, key="age")
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
**Output:**
|
|
119
|
+
|
|
120
|
+
```text
|
|
121
|
+
Why is age = 22?
|
|
122
|
+
|
|
123
|
+
Transformation history:
|
|
124
|
+
|
|
125
|
+
1. Original: age = 21
|
|
126
|
+
2. set age = 22
|
|
127
|
+
|
|
128
|
+
Final:
|
|
129
|
+
age = 22
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
---
|
|
133
|
+
|
|
134
|
+
### 2. Cross-Source Lineage: CSV to Pandas
|
|
135
|
+
|
|
136
|
+
Track data as it flows from external files into pandas DataFrames and derived columns:
|
|
137
|
+
|
|
138
|
+
```python
|
|
139
|
+
import pandas as pd
|
|
140
|
+
import whyvalue as why
|
|
141
|
+
|
|
142
|
+
with why.watch():
|
|
143
|
+
# Load and track raw CSV data
|
|
144
|
+
rows = why.load_csv("sales.csv")
|
|
145
|
+
|
|
146
|
+
# Convert tracked rows into a tracked DataFrame
|
|
147
|
+
df = why.to_dataframe(rows)
|
|
148
|
+
|
|
149
|
+
# Perform pandas calculations
|
|
150
|
+
df["total"] = df["price"].astype(float) * df["quantity"].astype(int)
|
|
151
|
+
|
|
152
|
+
# Explain the lineage of a calculated cell
|
|
153
|
+
why.explain(df, row=0, column="total")
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
**Output:**
|
|
157
|
+
|
|
158
|
+
```text
|
|
159
|
+
Why is total = 59.97?
|
|
160
|
+
|
|
161
|
+
Source:
|
|
162
|
+
sales.csv
|
|
163
|
+
|
|
164
|
+
Format:
|
|
165
|
+
CSV
|
|
166
|
+
|
|
167
|
+
Source row:
|
|
168
|
+
1
|
|
169
|
+
|
|
170
|
+
Transformations:
|
|
171
|
+
|
|
172
|
+
1. price × quantity → total
|
|
173
|
+
|
|
174
|
+
Final:
|
|
175
|
+
total = 59.97
|
|
176
|
+
```
|
|
177
|
+
|
|
178
|
+
---
|
|
179
|
+
|
|
180
|
+
## Feature Matrix
|
|
181
|
+
|
|
182
|
+
| Feature / Domain | Supported Operations | Description |
|
|
183
|
+
| :--- | :--- | :--- |
|
|
184
|
+
| **Native Python** | `dict`, `list`, primitive scalars | Tracks key assignments, item appends, deletions, and updates. |
|
|
185
|
+
| **File I/O** | `why.load_csv()`, `why.load_json()`, `why.load_txt()` | Ingests files while binding line/row indices for origin tracking. |
|
|
186
|
+
| **HTTP Requests** | `why.get(url)` | Records API URLs, HTTP response metadata, and payload structures. |
|
|
187
|
+
| **Pandas DataFrames**| `why.to_dataframe()`, column math, assignments | Tracks column derivations (`df['a'] * df['b']`), fills, and splits. |
|
|
188
|
+
| **Lineage & Inspection** | `why.explain()`, `why.explain_removed()`, `why.trace()` | Generates step-by-step human-readable transformation histories. |
|
|
189
|
+
| **State Snapshots** | `why.watch(snapshot=True)` | Captures state copies before and after operations for auditing. |
|
|
190
|
+
|
|
191
|
+
---
|
|
192
|
+
|
|
193
|
+
## Main API Overview
|
|
194
|
+
|
|
195
|
+
| Function | Signature | Description |
|
|
196
|
+
| :--- | :--- | :--- |
|
|
197
|
+
| `why.watch()` | `watch(snapshot=False)` | Context manager to begin observing data operations. |
|
|
198
|
+
| `why.track()` | `track(obj)` | Wraps a native `dict` or `list` for mutation tracking. |
|
|
199
|
+
| `why.explain()` | `explain(target, key=None, row=None, column=None)` | Prints the lineage and transformation history of a value. |
|
|
200
|
+
| `why.explain_removed()` | `explain_removed(target, key=None)` | Explains why a key or item was removed from a collection. |
|
|
201
|
+
| `why.trace()` | `trace(df)` | Prints a summary of all column transformations on a DataFrame. |
|
|
202
|
+
| `why.load_csv()` | `load_csv(path_or_str)` | Loads a CSV file into tracked dictionary records. |
|
|
203
|
+
| `why.load_json()` | `load_json(path_or_str)` | Loads a JSON document into tracked nested structures. |
|
|
204
|
+
| `why.load_txt()` | `load_txt(path_or_str)` | Loads a text file into tracked line lists. |
|
|
205
|
+
| `why.get()` | `get(url, **kwargs)` | Fetches HTTP API endpoints and tracks JSON response bodies. |
|
|
206
|
+
| `why.to_dataframe()` | `to_dataframe(tracked_data)` | Converts tracked records into a lineage-aware pandas DataFrame. |
|
|
207
|
+
|
|
208
|
+
---
|
|
209
|
+
|
|
210
|
+
## Performance & Best Practices
|
|
211
|
+
|
|
212
|
+
WhyValue is engineered for development, data pipeline auditing, debugging, and automated tests.
|
|
213
|
+
|
|
214
|
+
- **No Overhead when Inactive**: Calling code outside of `with why.watch():` incurs no tracking overhead.
|
|
215
|
+
- **In-Memory Tracking**: Event history is maintained in memory during a watch session and released upon completion.
|
|
216
|
+
- **Snapshot Mode**: Use `snapshot=True` only when full state auditing is required for complex transformations.
|
|
217
|
+
|
|
218
|
+
---
|
|
219
|
+
|
|
220
|
+
## What's New in v0.2
|
|
221
|
+
|
|
222
|
+
- 🚀 **Cross-Source Tracking**: Provenance across CSV, JSON, TXT files, and HTTP APIs.
|
|
223
|
+
- 🐼 **Full Pandas Integration**: Column math, element-wise transformations, and DataFrame lineage.
|
|
224
|
+
- 📊 **Enhanced Explanations**: Improved transformation outputs with file names, line numbers, and math symbols.
|
|
225
|
+
- ⚡ **Streamlined API**: Dedicated file loaders (`load_csv`, `load_json`, `load_txt`) and HTTP wrappers.
|
|
226
|
+
|
|
227
|
+
---
|
|
228
|
+
|
|
229
|
+
## Documentation & Resources
|
|
230
|
+
|
|
231
|
+
- 📖 **Live Documentation Website**: [https://muktaryy.github.io/whyvalue/](https://muktaryy.github.io/whyvalue/)
|
|
232
|
+
- 📦 **PyPI Package**: [https://pypi.org/project/whyvalue/](https://pypi.org/project/whyvalue/)
|
|
233
|
+
- 💻 **GitHub Repository**: [https://github.com/Muktaryy/whyvalue](https://github.com/Muktaryy/whyvalue)
|
|
234
|
+
- 🐛 **Issue Tracker**: [https://github.com/Muktaryy/whyvalue/issues](https://github.com/Muktaryy/whyvalue/issues)
|
|
235
|
+
|
|
236
|
+
---
|
|
237
|
+
|
|
238
|
+
## License
|
|
239
|
+
|
|
240
|
+
WhyValue is released under the [MIT License](LICENSE).
|
|
241
|
+
|
|
242
|
+
Developed and maintained by **Muktar Yakub**.
|
whyvalue-0.2.0/README.md
ADDED
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
<p align="center">
|
|
2
|
+
<a href="https://muktaryy.github.io/whyvalue/">
|
|
3
|
+
<img src="https://raw.githubusercontent.com/Muktaryy/whyvalue/main/docs/assets/whyvalue-readme.png" alt="WhyValue Logo" width="400" />
|
|
4
|
+
</a>
|
|
5
|
+
</p>
|
|
6
|
+
|
|
7
|
+
<h3 align="center">Ask Your Data Why</h3>
|
|
8
|
+
|
|
9
|
+
<p align="center">
|
|
10
|
+
<em>Transparent data lineage and value provenance for Python.</em>
|
|
11
|
+
</p>
|
|
12
|
+
|
|
13
|
+
<p align="center">
|
|
14
|
+
<a href="https://pypi.org/project/whyvalue/"><img src="https://img.shields.io/pypi/v/whyvalue.svg" alt="PyPI Version"></a>
|
|
15
|
+
<a href="https://pypi.org/project/whyvalue/"><img src="https://img.shields.io/pypi/pyversions/whyvalue.svg" alt="Python Versions"></a>
|
|
16
|
+
<a href="https://github.com/Muktaryy/whyvalue/blob/main/LICENSE"><img src="https://img.shields.io/github/license/Muktaryy/whyvalue.svg" alt="License"></a>
|
|
17
|
+
<a href="https://muktaryy.github.io/whyvalue/"><img src="https://img.shields.io/badge/docs-live-blue.svg" alt="Documentation"></a>
|
|
18
|
+
</p>
|
|
19
|
+
|
|
20
|
+
---
|
|
21
|
+
|
|
22
|
+
## What is WhyValue?
|
|
23
|
+
|
|
24
|
+
**WhyValue** is a developer-first Python library for data lineage, mutation tracking, and value provenance.
|
|
25
|
+
|
|
26
|
+
When a pipeline yields an unexpected calculation, a missing field, or an invalid metric, traditional debugging requires stepping through code line-by-line or placing manual print statements across modules. **WhyValue** eliminates this guesswork by capturing the history of your data as it flows from file and network sources through Python objects and pandas DataFrames.
|
|
27
|
+
|
|
28
|
+
Whenever you ask **"Why does this variable have this value?"**, WhyValue delivers a human-readable explanation of its origin, file boundaries, mathematical operations, and mutation sequence.
|
|
29
|
+
|
|
30
|
+
---
|
|
31
|
+
|
|
32
|
+
## Core Mental Model
|
|
33
|
+
|
|
34
|
+
WhyValue operates as a runtime observer:
|
|
35
|
+
|
|
36
|
+
```
|
|
37
|
+
┌─────────────────┐ ┌──────────────────────┐ ┌─────────────────┐ ┌──────────────────────┐
|
|
38
|
+
│ Source Data │ ────► │ Transformations │ ────► │ Target Value │ ────► │ Lineage & Provenance│
|
|
39
|
+
│ CSV, JSON, HTTP │ │ Math, Mutates, pandas│ │ Dict, DF, List │ │ why.explain(target) │
|
|
40
|
+
└─────────────────┘ └──────────────────────┘ └─────────────────┘ └──────────────────────┘
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
1. **Watch**: Wrap execution blocks with `with why.watch():` to activate observation.
|
|
44
|
+
2. **Track**: Ingest file/HTTP sources (`why.load_csv`, `why.get`) or wrap native structures (`why.track()`).
|
|
45
|
+
3. **Explain**: Query any value or cell at any point using `why.explain()` to inspect its transformation sequence.
|
|
46
|
+
|
|
47
|
+
---
|
|
48
|
+
|
|
49
|
+
## Key Features
|
|
50
|
+
|
|
51
|
+
- 🔍 **Non-Invasive Tracing**: Transparently observes object mutations without altering standard Python semantics.
|
|
52
|
+
- 🌐 **Cross-Source Lineage**: Tracks data across CSV files, JSON documents, text files, and HTTP API responses.
|
|
53
|
+
- 🐼 **Pandas Integration**: Observes DataFrame column creation, element-wise arithmetic, filtering, and assignment.
|
|
54
|
+
- 📜 **Transformation History**: Explains origins down to file names, source line/row indices, and math formulas.
|
|
55
|
+
- 📸 **Snapshot Auditing**: Optional before-and-after state capturing for auditing using `why.watch(snapshot=True)`.
|
|
56
|
+
- ⚡ **Lightweight & Fast**: Pure Python core with minimal dependency requirements for native tracking.
|
|
57
|
+
|
|
58
|
+
---
|
|
59
|
+
|
|
60
|
+
## Installation
|
|
61
|
+
|
|
62
|
+
Install WhyValue via PyPI using `pip`:
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
pip install whyvalue
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
*Requirements: Python 3.10+*
|
|
69
|
+
|
|
70
|
+
---
|
|
71
|
+
|
|
72
|
+
## Quick Start
|
|
73
|
+
|
|
74
|
+
### 1. Tracking Native Python Objects
|
|
75
|
+
|
|
76
|
+
Track mutations on standard dictionaries and lists:
|
|
77
|
+
|
|
78
|
+
```python
|
|
79
|
+
import whyvalue as why
|
|
80
|
+
|
|
81
|
+
with why.watch():
|
|
82
|
+
user = why.track({"name": "Ali", "age": 21})
|
|
83
|
+
user["age"] = 22
|
|
84
|
+
user["status"] = "active"
|
|
85
|
+
|
|
86
|
+
# Ask WhyValue why user['age'] equals 22
|
|
87
|
+
why.explain(user, key="age")
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
**Output:**
|
|
91
|
+
|
|
92
|
+
```text
|
|
93
|
+
Why is age = 22?
|
|
94
|
+
|
|
95
|
+
Transformation history:
|
|
96
|
+
|
|
97
|
+
1. Original: age = 21
|
|
98
|
+
2. set age = 22
|
|
99
|
+
|
|
100
|
+
Final:
|
|
101
|
+
age = 22
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
---
|
|
105
|
+
|
|
106
|
+
### 2. Cross-Source Lineage: CSV to Pandas
|
|
107
|
+
|
|
108
|
+
Track data as it flows from external files into pandas DataFrames and derived columns:
|
|
109
|
+
|
|
110
|
+
```python
|
|
111
|
+
import pandas as pd
|
|
112
|
+
import whyvalue as why
|
|
113
|
+
|
|
114
|
+
with why.watch():
|
|
115
|
+
# Load and track raw CSV data
|
|
116
|
+
rows = why.load_csv("sales.csv")
|
|
117
|
+
|
|
118
|
+
# Convert tracked rows into a tracked DataFrame
|
|
119
|
+
df = why.to_dataframe(rows)
|
|
120
|
+
|
|
121
|
+
# Perform pandas calculations
|
|
122
|
+
df["total"] = df["price"].astype(float) * df["quantity"].astype(int)
|
|
123
|
+
|
|
124
|
+
# Explain the lineage of a calculated cell
|
|
125
|
+
why.explain(df, row=0, column="total")
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
**Output:**
|
|
129
|
+
|
|
130
|
+
```text
|
|
131
|
+
Why is total = 59.97?
|
|
132
|
+
|
|
133
|
+
Source:
|
|
134
|
+
sales.csv
|
|
135
|
+
|
|
136
|
+
Format:
|
|
137
|
+
CSV
|
|
138
|
+
|
|
139
|
+
Source row:
|
|
140
|
+
1
|
|
141
|
+
|
|
142
|
+
Transformations:
|
|
143
|
+
|
|
144
|
+
1. price × quantity → total
|
|
145
|
+
|
|
146
|
+
Final:
|
|
147
|
+
total = 59.97
|
|
148
|
+
```
|
|
149
|
+
|
|
150
|
+
---
|
|
151
|
+
|
|
152
|
+
## Feature Matrix
|
|
153
|
+
|
|
154
|
+
| Feature / Domain | Supported Operations | Description |
|
|
155
|
+
| :--- | :--- | :--- |
|
|
156
|
+
| **Native Python** | `dict`, `list`, primitive scalars | Tracks key assignments, item appends, deletions, and updates. |
|
|
157
|
+
| **File I/O** | `why.load_csv()`, `why.load_json()`, `why.load_txt()` | Ingests files while binding line/row indices for origin tracking. |
|
|
158
|
+
| **HTTP Requests** | `why.get(url)` | Records API URLs, HTTP response metadata, and payload structures. |
|
|
159
|
+
| **Pandas DataFrames**| `why.to_dataframe()`, column math, assignments | Tracks column derivations (`df['a'] * df['b']`), fills, and splits. |
|
|
160
|
+
| **Lineage & Inspection** | `why.explain()`, `why.explain_removed()`, `why.trace()` | Generates step-by-step human-readable transformation histories. |
|
|
161
|
+
| **State Snapshots** | `why.watch(snapshot=True)` | Captures state copies before and after operations for auditing. |
|
|
162
|
+
|
|
163
|
+
---
|
|
164
|
+
|
|
165
|
+
## Main API Overview
|
|
166
|
+
|
|
167
|
+
| Function | Signature | Description |
|
|
168
|
+
| :--- | :--- | :--- |
|
|
169
|
+
| `why.watch()` | `watch(snapshot=False)` | Context manager to begin observing data operations. |
|
|
170
|
+
| `why.track()` | `track(obj)` | Wraps a native `dict` or `list` for mutation tracking. |
|
|
171
|
+
| `why.explain()` | `explain(target, key=None, row=None, column=None)` | Prints the lineage and transformation history of a value. |
|
|
172
|
+
| `why.explain_removed()` | `explain_removed(target, key=None)` | Explains why a key or item was removed from a collection. |
|
|
173
|
+
| `why.trace()` | `trace(df)` | Prints a summary of all column transformations on a DataFrame. |
|
|
174
|
+
| `why.load_csv()` | `load_csv(path_or_str)` | Loads a CSV file into tracked dictionary records. |
|
|
175
|
+
| `why.load_json()` | `load_json(path_or_str)` | Loads a JSON document into tracked nested structures. |
|
|
176
|
+
| `why.load_txt()` | `load_txt(path_or_str)` | Loads a text file into tracked line lists. |
|
|
177
|
+
| `why.get()` | `get(url, **kwargs)` | Fetches HTTP API endpoints and tracks JSON response bodies. |
|
|
178
|
+
| `why.to_dataframe()` | `to_dataframe(tracked_data)` | Converts tracked records into a lineage-aware pandas DataFrame. |
|
|
179
|
+
|
|
180
|
+
---
|
|
181
|
+
|
|
182
|
+
## Performance & Best Practices
|
|
183
|
+
|
|
184
|
+
WhyValue is engineered for development, data pipeline auditing, debugging, and automated tests.
|
|
185
|
+
|
|
186
|
+
- **No Overhead when Inactive**: Calling code outside of `with why.watch():` incurs no tracking overhead.
|
|
187
|
+
- **In-Memory Tracking**: Event history is maintained in memory during a watch session and released upon completion.
|
|
188
|
+
- **Snapshot Mode**: Use `snapshot=True` only when full state auditing is required for complex transformations.
|
|
189
|
+
|
|
190
|
+
---
|
|
191
|
+
|
|
192
|
+
## What's New in v0.2
|
|
193
|
+
|
|
194
|
+
- 🚀 **Cross-Source Tracking**: Provenance across CSV, JSON, TXT files, and HTTP APIs.
|
|
195
|
+
- 🐼 **Full Pandas Integration**: Column math, element-wise transformations, and DataFrame lineage.
|
|
196
|
+
- 📊 **Enhanced Explanations**: Improved transformation outputs with file names, line numbers, and math symbols.
|
|
197
|
+
- ⚡ **Streamlined API**: Dedicated file loaders (`load_csv`, `load_json`, `load_txt`) and HTTP wrappers.
|
|
198
|
+
|
|
199
|
+
---
|
|
200
|
+
|
|
201
|
+
## Documentation & Resources
|
|
202
|
+
|
|
203
|
+
- 📖 **Live Documentation Website**: [https://muktaryy.github.io/whyvalue/](https://muktaryy.github.io/whyvalue/)
|
|
204
|
+
- 📦 **PyPI Package**: [https://pypi.org/project/whyvalue/](https://pypi.org/project/whyvalue/)
|
|
205
|
+
- 💻 **GitHub Repository**: [https://github.com/Muktaryy/whyvalue](https://github.com/Muktaryy/whyvalue)
|
|
206
|
+
- 🐛 **Issue Tracker**: [https://github.com/Muktaryy/whyvalue/issues](https://github.com/Muktaryy/whyvalue/issues)
|
|
207
|
+
|
|
208
|
+
---
|
|
209
|
+
|
|
210
|
+
## License
|
|
211
|
+
|
|
212
|
+
WhyValue is released under the [MIT License](LICENSE).
|
|
213
|
+
|
|
214
|
+
Developed and maintained by **Muktar Yakub**.
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "whyvalue"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.2.0"
|
|
8
8
|
description = "Ask your data why."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "MIT"
|
|
@@ -13,7 +13,8 @@ authors = [
|
|
|
13
13
|
]
|
|
14
14
|
requires-python = ">=3.10"
|
|
15
15
|
dependencies = [
|
|
16
|
-
"pandas>=2.0"
|
|
16
|
+
"pandas>=2.0",
|
|
17
|
+
"requests>=2.31",
|
|
17
18
|
]
|
|
18
19
|
keywords = [
|
|
19
20
|
"pandas",
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
from .core import (
|
|
2
|
+
watch,
|
|
3
|
+
stop,
|
|
4
|
+
is_watching,
|
|
5
|
+
trace,
|
|
6
|
+
explain,
|
|
7
|
+
explain_removed,
|
|
8
|
+
WatchSession,
|
|
9
|
+
track,
|
|
10
|
+
load_json,
|
|
11
|
+
load_csv,
|
|
12
|
+
load_txt,
|
|
13
|
+
get,
|
|
14
|
+
to_dataframe,
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
__all__ = [
|
|
18
|
+
"watch",
|
|
19
|
+
"stop",
|
|
20
|
+
"is_watching",
|
|
21
|
+
"trace",
|
|
22
|
+
"explain",
|
|
23
|
+
"explain_removed",
|
|
24
|
+
"WatchSession",
|
|
25
|
+
"track",
|
|
26
|
+
"load_json",
|
|
27
|
+
"load_csv",
|
|
28
|
+
"load_txt",
|
|
29
|
+
"get",
|
|
30
|
+
"to_dataframe",
|
|
31
|
+
]
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
import copy
|
|
2
|
+
import csv
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
from .python_dict import TrackedDict
|
|
6
|
+
from .python_list import TrackedList
|
|
7
|
+
from ..history import history
|
|
8
|
+
from ..provenance import ProvenanceEvent
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def load_csv(path, encoding="utf-8", delimiter=","):
|
|
12
|
+
"""Load CSV from a file path into WhyValue tracked containers (TrackedList of TrackedDict)."""
|
|
13
|
+
from ..core import is_watching, is_snapshot_enabled
|
|
14
|
+
|
|
15
|
+
if not is_watching():
|
|
16
|
+
raise RuntimeError("why.load_csv() requires an active WhyValue watch session.")
|
|
17
|
+
|
|
18
|
+
file_path = Path(path)
|
|
19
|
+
is_snapshot = is_snapshot_enabled()
|
|
20
|
+
|
|
21
|
+
with open(file_path, mode="r", encoding=encoding, newline="") as f:
|
|
22
|
+
reader = csv.DictReader(f, delimiter=delimiter)
|
|
23
|
+
|
|
24
|
+
rows_list = TrackedList()
|
|
25
|
+
rows_list._whyvalue_source = {
|
|
26
|
+
"source_type": "csv",
|
|
27
|
+
"source_path": str(path),
|
|
28
|
+
"delimiter": delimiter,
|
|
29
|
+
"encoding": encoding,
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
row_idx = 1
|
|
33
|
+
for row in reader:
|
|
34
|
+
row_dict = TrackedDict()
|
|
35
|
+
row_dict._whyvalue_source = {
|
|
36
|
+
"source_type": "csv",
|
|
37
|
+
"source_path": str(path),
|
|
38
|
+
"csv_row": row_idx,
|
|
39
|
+
"delimiter": delimiter,
|
|
40
|
+
"encoding": encoding,
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
for k, v in row.items():
|
|
44
|
+
super(TrackedDict, row_dict).__setitem__(k, v)
|
|
45
|
+
|
|
46
|
+
history.add(
|
|
47
|
+
ProvenanceEvent(
|
|
48
|
+
event_type="csv_loaded",
|
|
49
|
+
object_id=row_dict._whyvalue_id,
|
|
50
|
+
after_value=copy.deepcopy(dict(row_dict)) if is_snapshot else None,
|
|
51
|
+
inputs={
|
|
52
|
+
"source_path": str(path),
|
|
53
|
+
"csv_row": row_idx,
|
|
54
|
+
"initial_value": copy.deepcopy(dict(row_dict)),
|
|
55
|
+
},
|
|
56
|
+
metadata={
|
|
57
|
+
"operation": "csv_loaded",
|
|
58
|
+
"source_type": "csv",
|
|
59
|
+
"source_path": str(path),
|
|
60
|
+
"csv_row": row_idx,
|
|
61
|
+
"delimiter": delimiter,
|
|
62
|
+
"encoding": encoding,
|
|
63
|
+
"initial_value": copy.deepcopy(dict(row_dict)),
|
|
64
|
+
},
|
|
65
|
+
)
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
super(TrackedList, rows_list).append(row_dict)
|
|
69
|
+
row_idx += 1
|
|
70
|
+
|
|
71
|
+
history.add(
|
|
72
|
+
ProvenanceEvent(
|
|
73
|
+
event_type="csv_loaded",
|
|
74
|
+
object_id=rows_list._whyvalue_id,
|
|
75
|
+
after_value=copy.deepcopy(list(rows_list)) if is_snapshot else None,
|
|
76
|
+
inputs={
|
|
77
|
+
"source_path": str(path),
|
|
78
|
+
"total_rows": row_idx - 1,
|
|
79
|
+
"initial_value": copy.deepcopy(list(rows_list)),
|
|
80
|
+
},
|
|
81
|
+
metadata={
|
|
82
|
+
"operation": "csv_loaded",
|
|
83
|
+
"source_type": "csv",
|
|
84
|
+
"source_path": str(path),
|
|
85
|
+
"total_rows": row_idx - 1,
|
|
86
|
+
"delimiter": delimiter,
|
|
87
|
+
"encoding": encoding,
|
|
88
|
+
"initial_value": copy.deepcopy(list(rows_list)),
|
|
89
|
+
},
|
|
90
|
+
)
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
return rows_list
|