memframe 0.2.0__tar.gz → 0.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (249) hide show
  1. {memframe-0.2.0 → memframe-0.2.2}/.github/workflows/release-testpypi.yml +8 -1
  2. {memframe-0.2.0 → memframe-0.2.2}/.gitignore +1 -0
  3. memframe-0.2.2/.opencode/plans/corr_cov_single_scan.md +154 -0
  4. {memframe-0.2.0 → memframe-0.2.2}/CHANGELOG.md +20 -0
  5. {memframe-0.2.0 → memframe-0.2.2}/CONTRIBUTING.md +6 -3
  6. {memframe-0.2.0 → memframe-0.2.2}/PKG-INFO +3 -1
  7. {memframe-0.2.0 → memframe-0.2.2}/README.md +2 -0
  8. {memframe-0.2.0 → memframe-0.2.2}/docs/api/cleaning.md +0 -70
  9. {memframe-0.2.0 → memframe-0.2.2}/docs/api/inspect.md +58 -50
  10. {memframe-0.2.0 → memframe-0.2.2}/docs/api/selection.md +24 -104
  11. {memframe-0.2.0 → memframe-0.2.2}/docs/api/stats.md +11 -44
  12. {memframe-0.2.0 → memframe-0.2.2}/docs/getting-started.md +0 -10
  13. {memframe-0.2.0 → memframe-0.2.2}/docs/index.md +1 -0
  14. {memframe-0.2.0 → memframe-0.2.2}/pyproject.toml +1 -1
  15. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/analytix/arithmetic.py +81 -53
  16. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/analytix/cleaning.py +0 -96
  17. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/analytix/inspection.py +157 -178
  18. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/analytix/selection.py +57 -219
  19. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/analytix/stats.py +286 -449
  20. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/orchestrator/analytix/cleaning.py +12 -32
  21. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/orchestrator/analytix/inspection.py +26 -31
  22. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/orchestrator/analytix/selection.py +14 -107
  23. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/orchestrator/analytix/stats.py +34 -40
  24. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/main.py +1 -0
  25. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/analytix/cleaning.py +12 -73
  26. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/analytix/cleaning.pyi +0 -39
  27. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/analytix/inspection.py +68 -58
  28. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/analytix/inspection.pyi +39 -14
  29. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/analytix/selection.py +4 -94
  30. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/analytix/selection.pyi +2 -42
  31. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/analytix/stats.py +14 -74
  32. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/analytix/stats.pyi +0 -44
  33. {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/tools/clean.py +0 -27
  34. {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/tools/inspect.py +22 -27
  35. {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/tools/select.py +6 -26
  36. {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/tools/stats.py +3 -21
  37. {memframe-0.2.0 → memframe-0.2.2}/tests/integration/ops/test_arithmetic.py +92 -0
  38. {memframe-0.2.0 → memframe-0.2.2}/tests/integration/ops/test_cleaning.py +0 -50
  39. {memframe-0.2.0 → memframe-0.2.2}/tests/integration/ops/test_inspect.py +38 -38
  40. {memframe-0.2.0 → memframe-0.2.2}/tests/integration/ops/test_selection.py +3 -75
  41. {memframe-0.2.0 → memframe-0.2.2}/tests/integration/ops/test_stats.py +25 -59
  42. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_clean_tools.py +1 -1
  43. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_inspect_tools.py +1 -1
  44. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_select_tools.py +1 -1
  45. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_stats_tools.py +1 -1
  46. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_tools.py +0 -9
  47. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_errors.py +12 -12
  48. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_inspection_response.py +1 -1
  49. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_public_results.py +1 -1
  50. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_selection_response.py +4 -5
  51. {memframe-0.2.0 → memframe-0.2.2}/uv.lock +1 -1
  52. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/building-pydantic-ai-agents/SKILL.md +0 -0
  53. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/building-pydantic-ai-agents/references/AGENTS-CORE.md +0 -0
  54. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/building-pydantic-ai-agents/references/ARCHITECTURE.md +0 -0
  55. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/building-pydantic-ai-agents/references/CAPABILITIES-AND-HOOKS.md +0 -0
  56. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/building-pydantic-ai-agents/references/COMMON-TASKS.md +0 -0
  57. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/building-pydantic-ai-agents/references/INPUT-AND-HISTORY.md +0 -0
  58. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/building-pydantic-ai-agents/references/NATIVE-TOOLS.md +0 -0
  59. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/building-pydantic-ai-agents/references/ON-DEMAND-CAPABILITIES.md +0 -0
  60. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/building-pydantic-ai-agents/references/ORCHESTRATION-AND-INTEGRATIONS.md +0 -0
  61. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/building-pydantic-ai-agents/references/TESTING-AND-DEBUGGING.md +0 -0
  62. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/building-pydantic-ai-agents/references/TOOLS-ADVANCED.md +0 -0
  63. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/building-pydantic-ai-agents/references/TOOLS-CORE.md +0 -0
  64. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/SKILL.md +0 -0
  65. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/collector/host-and-infra-metrics.md +0 -0
  66. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/javascript/ai-sdk.md +0 -0
  67. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/javascript/cloudflare-and-deno.md +0 -0
  68. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/javascript/frameworks.md +0 -0
  69. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/javascript/installation-and-env.md +0 -0
  70. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/javascript/nextjs.md +0 -0
  71. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/javascript/node-runtime.md +0 -0
  72. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/javascript/patterns.md +0 -0
  73. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/javascript/project-detection.md +0 -0
  74. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/javascript/react-browser.md +0 -0
  75. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/javascript/verification-troubleshooting.md +0 -0
  76. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/python/integrations.md +0 -0
  77. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/python/logging-patterns.md +0 -0
  78. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/rust/patterns.md +0 -0
  79. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-query/SKILL.md +0 -0
  80. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-query/references/client-usage.md +0 -0
  81. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-query/references/schema.md +0 -0
  82. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-ui/SKILL.md +0 -0
  83. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/pydantic/SKILL.md +0 -0
  84. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/pydantic-ai-harness/SKILL.md +0 -0
  85. {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/pydantic-ai-harness/references/CODE-MODE.md +0 -0
  86. {memframe-0.2.0 → memframe-0.2.2}/.env.test.example +0 -0
  87. {memframe-0.2.0 → memframe-0.2.2}/.githooks/pre-commit +0 -0
  88. {memframe-0.2.0 → memframe-0.2.2}/.githooks/precommit.sh +0 -0
  89. {memframe-0.2.0 → memframe-0.2.2}/.github/workflows/ci.yml +0 -0
  90. {memframe-0.2.0 → memframe-0.2.2}/.github/workflows/docs.yml +0 -0
  91. {memframe-0.2.0 → memframe-0.2.2}/.github/workflows/release-prod.yml +0 -0
  92. {memframe-0.2.0 → memframe-0.2.2}/.github/workflows/tox.yml +0 -0
  93. {memframe-0.2.0 → memframe-0.2.2}/.opencode/opencode.json +0 -0
  94. {memframe-0.2.0 → memframe-0.2.2}/.python-version +0 -0
  95. {memframe-0.2.0 → memframe-0.2.2}/CODE_OF_CONDUCT.md +0 -0
  96. {memframe-0.2.0 → memframe-0.2.2}/LICENSE +0 -0
  97. {memframe-0.2.0 → memframe-0.2.2}/docs/CODE_OF_CONDUCT.md +0 -0
  98. {memframe-0.2.0 → memframe-0.2.2}/docs/CONTRIBUTING.md +0 -0
  99. {memframe-0.2.0 → memframe-0.2.2}/docs/api/agent.md +0 -0
  100. {memframe-0.2.0 → memframe-0.2.2}/docs/api/arithmetic.md +0 -0
  101. {memframe-0.2.0 → memframe-0.2.2}/docs/api/bar.md +0 -0
  102. {memframe-0.2.0 → memframe-0.2.2}/docs/api/bar_polar.md +0 -0
  103. {memframe-0.2.0 → memframe-0.2.2}/docs/api/caching.md +0 -0
  104. {memframe-0.2.0 → memframe-0.2.2}/docs/api/connector.md +0 -0
  105. {memframe-0.2.0 → memframe-0.2.2}/docs/api/database.md +0 -0
  106. {memframe-0.2.0 → memframe-0.2.2}/docs/api/line.md +0 -0
  107. {memframe-0.2.0 → memframe-0.2.2}/docs/api/pie.md +0 -0
  108. {memframe-0.2.0 → memframe-0.2.2}/docs/api/scatter.md +0 -0
  109. {memframe-0.2.0 → memframe-0.2.2}/docs/api/scatter3d.md +0 -0
  110. {memframe-0.2.0 → memframe-0.2.2}/docs/api/upload-manager.md +0 -0
  111. {memframe-0.2.0 → memframe-0.2.2}/docs/assets/logo-nav.png +0 -0
  112. {memframe-0.2.0 → memframe-0.2.2}/docs/assets/memframe-logo-full.png +0 -0
  113. {memframe-0.2.0 → memframe-0.2.2}/docs/stylesheets/extra.css +0 -0
  114. {memframe-0.2.0 → memframe-0.2.2}/docs/testing.md +0 -0
  115. {memframe-0.2.0 → memframe-0.2.2}/mkdocs.yml +0 -0
  116. {memframe-0.2.0 → memframe-0.2.2}/scripts/release.sh +0 -0
  117. {memframe-0.2.0 → memframe-0.2.2}/scripts/run-commit-checks.sh +0 -0
  118. {memframe-0.2.0 → memframe-0.2.2}/skills-lock.json +0 -0
  119. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/__init__.py +0 -0
  120. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/cache/__init__.py +0 -0
  121. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/cache/cache_manager.py +0 -0
  122. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/__init__.py +0 -0
  123. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/analytix/_response.py +0 -0
  124. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/ingestion/datatype_detector.py +0 -0
  125. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/ingestion/upload/__init__.py +0 -0
  126. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/ingestion/upload/base.py +0 -0
  127. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/ingestion/upload/clickhouse.py +0 -0
  128. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/ingestion/upload/duckdb.py +0 -0
  129. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/ingestion/upload/postgres.py +0 -0
  130. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/ingestion/upload/strategies/__init__.py +0 -0
  131. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/ingestion/upload/strategies/base.py +0 -0
  132. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/ingestion/upload/strategies/csv.py +0 -0
  133. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/ingestion/upload/strategies/df.py +0 -0
  134. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/ingestion/upload/strategies/parquet.py +0 -0
  135. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/orchestrator/analytix/arithmetic.py +0 -0
  136. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/orchestrator/plots/bar.py +0 -0
  137. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/orchestrator/plots/bar_polar.py +0 -0
  138. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/orchestrator/plots/line.py +0 -0
  139. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/orchestrator/plots/pie.py +0 -0
  140. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/orchestrator/plots/scatter.py +0 -0
  141. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/orchestrator/plots/scatter_3d.py +0 -0
  142. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/plots/bar.py +0 -0
  143. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/plots/bar_polar.py +0 -0
  144. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/plots/line.py +0 -0
  145. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/plots/pie.py +0 -0
  146. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/plots/scatter.py +0 -0
  147. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/plots/scatter_3d.py +0 -0
  148. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/__init__.py +0 -0
  149. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/adapters/base.py +0 -0
  150. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/adapters/clickhouse.py +0 -0
  151. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/adapters/duckdb.py +0 -0
  152. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/adapters/factory.py +0 -0
  153. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/adapters/postgresql.py +0 -0
  154. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/connection/__init__.py +0 -0
  155. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/connection/connector.py +0 -0
  156. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/connection/pool.py +0 -0
  157. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/context.py +0 -0
  158. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/context.pyi +0 -0
  159. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/ops.py +0 -0
  160. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/setup/__init__.py +0 -0
  161. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/setup/base.py +0 -0
  162. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/setup/clickhouse.py +0 -0
  163. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/setup/duckdb.py +0 -0
  164. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/setup/postgres.py +0 -0
  165. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/exceptions.py +0 -0
  166. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/main.pyi +0 -0
  167. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/py.typed +0 -0
  168. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/utils/__init__.py +0 -0
  169. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/utils/async_sync.py +0 -0
  170. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/utils/helper.py +0 -0
  171. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/utils/plot_renderer.py +0 -0
  172. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/__init__.py +0 -0
  173. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/analytix/__init__.py +0 -0
  174. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/analytix/arithmetic.py +0 -0
  175. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/analytix/arithmetic.pyi +0 -0
  176. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/base.py +0 -0
  177. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/__init__.py +0 -0
  178. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/bar.py +0 -0
  179. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/bar.pyi +0 -0
  180. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/bar_polar.py +0 -0
  181. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/bar_polar.pyi +0 -0
  182. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/line.py +0 -0
  183. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/line.pyi +0 -0
  184. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/pie.py +0 -0
  185. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/pie.pyi +0 -0
  186. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/scatter.py +0 -0
  187. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/scatter.pyi +0 -0
  188. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/scatter3d.py +0 -0
  189. {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/scatter3d.pyi +0 -0
  190. {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/__init__.py +0 -0
  191. {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/agents/__init__.py +0 -0
  192. {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/agents/analytics.py +0 -0
  193. {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/agents/planning.py +0 -0
  194. {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/config.py +0 -0
  195. {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/domain.py +0 -0
  196. {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/entrypoints.py +0 -0
  197. {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/gateway.py +0 -0
  198. {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/observe.py +0 -0
  199. {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/sessions.py +0 -0
  200. {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/tools/__init__.py +0 -0
  201. {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/tools/_helpers.py +0 -0
  202. {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/tools/arithmetic.py +0 -0
  203. {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/tools/context.py +0 -0
  204. {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/tools/plot.py +0 -0
  205. {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/tools/upload.py +0 -0
  206. {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/wrappers.py +0 -0
  207. {memframe-0.2.0 → memframe-0.2.2}/tests/conftest.py +0 -0
  208. {memframe-0.2.0 → memframe-0.2.2}/tests/datasets/sample.csv +0 -0
  209. {memframe-0.2.0 → memframe-0.2.2}/tests/datasets/sample.parquet +0 -0
  210. {memframe-0.2.0 → memframe-0.2.2}/tests/db_test_utils.py +0 -0
  211. {memframe-0.2.0 → memframe-0.2.2}/tests/integration/conftest.py +0 -0
  212. {memframe-0.2.0 → memframe-0.2.2}/tests/integration/ops/test_bar.py +0 -0
  213. {memframe-0.2.0 → memframe-0.2.2}/tests/integration/ops/test_bar_polar.py +0 -0
  214. {memframe-0.2.0 → memframe-0.2.2}/tests/integration/ops/test_line.py +0 -0
  215. {memframe-0.2.0 → memframe-0.2.2}/tests/integration/ops/test_pie.py +0 -0
  216. {memframe-0.2.0 → memframe-0.2.2}/tests/integration/ops/test_scatter.py +0 -0
  217. {memframe-0.2.0 → memframe-0.2.2}/tests/integration/ops/test_scatter3d.py +0 -0
  218. {memframe-0.2.0 → memframe-0.2.2}/tests/integration/ops/test_upload.py +0 -0
  219. {memframe-0.2.0 → memframe-0.2.2}/tests/integration/test_concurrency.py +0 -0
  220. {memframe-0.2.0 → memframe-0.2.2}/tests/integration/test_lifecycle.py +0 -0
  221. {memframe-0.2.0 → memframe-0.2.2}/tests/integration/test_parity.py +0 -0
  222. {memframe-0.2.0 → memframe-0.2.2}/tests/run_tests.py +0 -0
  223. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_arithmetic_tools.py +0 -0
  224. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_domain.py +0 -0
  225. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_entrypoints.py +0 -0
  226. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_fleet.py +0 -0
  227. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_gateway.py +0 -0
  228. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_observe.py +0 -0
  229. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_planning.py +0 -0
  230. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_sessions.py +0 -0
  231. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_arithmetic_response.py +0 -0
  232. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_async_sync.py +0 -0
  233. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_cleaning_orchestrator_response.py +0 -0
  234. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_cleaning_response.py +0 -0
  235. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_connector.py +0 -0
  236. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_context.py +0 -0
  237. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_csv_invalid_values.py +0 -0
  238. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_datatype_detector.py +0 -0
  239. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_exceptions.py +0 -0
  240. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_factory.py +0 -0
  241. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_helper.py +0 -0
  242. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_inspection_orchestrator_response.py +0 -0
  243. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_ops_mixin.py +0 -0
  244. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_plot_renderer.py +0 -0
  245. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_pool.py +0 -0
  246. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_response_unwrap.py +0 -0
  247. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_setup.py +0 -0
  248. {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_stats_response.py +0 -0
  249. {memframe-0.2.0 → memframe-0.2.2}/tox.ini +0 -0
@@ -20,7 +20,14 @@ jobs:
20
20
  env:
21
21
  UV_PUBLISH_TOKEN: ${{ secrets.TEST_PYPI_TOKEN }}
22
22
  run: uv publish --publish-url https://test.pypi.org/legacy/
23
+ - name: Extract release notes from CHANGELOG
24
+ env:
25
+ VERSION: ${{ github.ref_name }}
26
+ run: |
27
+ ver="${VERSION#v}"
28
+ awk -v v="## [$ver]" '$0 ~ "^## \\[" ver "\\]"{f=1;next} /^## \[/{if(f)exit} f' CHANGELOG.md > release_notes.md
29
+ cat release_notes.md
23
30
  - name: Create GitHub release
24
31
  uses: softprops/action-gh-release@v2
25
32
  with:
26
- generate_release_notes: true
33
+ body_path: release_notes.md
@@ -86,3 +86,4 @@ Thumbs.db
86
86
  plot.html
87
87
  *.html
88
88
 
89
+ src/memframe/test_main.py
@@ -0,0 +1,154 @@
1
+ # Plan: Pure in-DB single-scan correlation / covariance (Tier-1 + Tier-2 ClickHouse)
2
+
3
+ Status: APPLICATION BLOCKED — an active permission rule denies `edit` on source files
4
+ (`edit * -> deny`), consistent with read-only plan mode. The code below is ready to
5
+ apply once edit permission / plan mode is lifted.
6
+
7
+ ## Goal
8
+ Replace the `UNION ALL` of per-pair `CORR()`/`COVAR_SAMP()` (one full table scan per
9
+ pair ≈ n²/2 scans) with batches of B aggregate calls inside ONE `SELECT` (one scan per
10
+ batch). The DB engine evaluates all aggregates in a single pass over the data. Keep the
11
+ DB-native `CORR`/`COVAR_SAMP` so results still match `pandas.corr()/.cov()` within
12
+ `rtol=1e-5` (existing tests pass). Drop the intermediate output table.
13
+
14
+ ## Tier-2 (ClickHouse, adaptive single-scan, no UDF)
15
+ Batch size `B = min(total_pairs, 250)` for ClickHouse; when the whole pair set fits the
16
+ expression budget the loop issues a *single* `SELECT` (one scan). Wider tables fall back
17
+ to batching (still one scan per batch). DuckDB/Postgres use `B = min(total_pairs, 500)`.
18
+
19
+ ## File: src/memframe/core/analytix/stats.py
20
+
21
+ ### 1) Replace `_exec_union_batched` (lines 809-819) with this helper
22
+
23
+ ```python
24
+ async def _multi_column_assoc_matrix(
25
+ self,
26
+ table: str,
27
+ schema: str,
28
+ columns: List[str],
29
+ agg_fn: str,
30
+ backend=None,
31
+ data_id: Optional[str] = None,
32
+ new_table: Optional[str] = None,
33
+ ) -> Dict[str, Any]:
34
+ """Compute an n x n correlation/covariance matrix purely in the database.
35
+
36
+ All CORR/COVAR_SAMP calls are batched into single SELECT statements so the
37
+ engine scans the table once per batch, not once per pair. ClickHouse uses a
38
+ tighter batch so a single scan covers the table whenever the pair count fits
39
+ its expression budget (Tier-2); wider tables fall back to batching.
40
+ """
41
+ try:
42
+ if not isinstance(self.db, (PostgresAdapter, DuckDBAdapter, ClickHouseAdapter)):
43
+ raise self._unsupported_backend_error()
44
+
45
+ q = self._qualified_table(table, schema)
46
+ cols = [SQLIdentifierSanitizer.sanitize(c) for c in columns]
47
+ n = len(cols)
48
+ if n < 1:
49
+ return self._error_response("No columns provided", columns)
50
+
51
+ def _cast(c: str) -> str:
52
+ if isinstance(self.db, PostgresAdapter):
53
+ return f'CAST(NULLIF("{c}"::text, \'\') AS DOUBLE PRECISION)'
54
+ if isinstance(self.db, DuckDBAdapter):
55
+ return f'TRY_CAST("{c}" AS DOUBLE)'
56
+ return f'toFloat64OrNull(toString("{c}"))'
57
+
58
+ alias = {c: f"a{idx}" for idx, c in enumerate(cols)}
59
+ pairs = [(i, j) for i in range(n) for j in range(i, n)]
60
+
61
+ # ponytail: batch size chosen under each backend's expression-count limit
62
+ if isinstance(self.db, ClickHouseAdapter):
63
+ B = min(len(pairs), 250)
64
+ else:
65
+ B = min(len(pairs), 500)
66
+
67
+ results: Dict[tuple, Any] = {}
68
+ for start in range(0, len(pairs), B):
69
+ batch = pairs[start:start + B]
70
+ used = sorted(
71
+ {cols[i] for (i, j) in batch} | {cols[j] for (i, j) in batch},
72
+ key=lambda c: cols.index(c),
73
+ )
74
+ cte = ", ".join(f"{_cast(c)} AS {alias[c]}" for c in used)
75
+ sel = []
76
+ for k, (i, j) in enumerate(batch):
77
+ ai, aj = alias[cols[i]], alias[cols[j]]
78
+ cond = f"{ai} IS NOT NULL AND {aj} IS NOT NULL"
79
+ sel.append(
80
+ f"{agg_fn}(CASE WHEN {cond} THEN {ai} END, "
81
+ f"CASE WHEN {cond} THEN {aj} END) AS v_{k}"
82
+ )
83
+ sql = f"WITH q AS (SELECT {cte} FROM {q}) SELECT " + ", ".join(sel) + " FROM q"
84
+ rows = await self._fetch(sql)
85
+ row = rows[0] if rows else {}
86
+ for k, (i, j) in enumerate(batch):
87
+ val = row.get(f"v_{k}") if row else None
88
+ results[(i, j)] = val if val is not None else float("nan")
89
+
90
+ mat = [[results.get((min(i, j), max(i, j))) for j in range(n)] for i in range(n)]
91
+ df = pd.DataFrame(mat, index=columns, columns=columns)
92
+ msg = f"Computed {agg_fn} matrix for {n} columns"
93
+ return self._success_response(msg, columns, result=df)
94
+
95
+ except Exception as e:
96
+ return self._error_response(
97
+ f"multi_column_assoc_matrix error: {str(e)}\n{traceback.format_exc()}",
98
+ columns,
99
+ )
100
+ ```
101
+
102
+ ### 2) Replace `numeric_multi_column_correlation` (lines 821-889) with
103
+
104
+ ```python
105
+ async def numeric_multi_column_correlation(
106
+ self,
107
+ table: str,
108
+ schema: str,
109
+ columns: List[str],
110
+ backend=None,
111
+ data_id: Optional[str] = None,
112
+ new_table: Optional[str] = None,
113
+ ) -> Dict[str, Any]:
114
+ agg = "CORR" if isinstance(self.db, (PostgresAdapter, DuckDBAdapter)) else "corr"
115
+ return await self._multi_column_assoc_matrix(
116
+ table, schema, columns, agg, backend=backend, data_id=data_id, new_table=new_table
117
+ )
118
+ ```
119
+
120
+ ### 3) Replace `numeric_multi_column_covariance` (lines 891-952) with
121
+
122
+ ```python
123
+ async def numeric_multi_column_covariance(
124
+ self,
125
+ table: str,
126
+ schema: str,
127
+ columns: List[str],
128
+ backend=None,
129
+ data_id: Optional[str] = None,
130
+ new_table: Optional[str] = None,
131
+ ) -> Dict[str, Any]:
132
+ agg = "COVAR_SAMP" if isinstance(self.db, (PostgresAdapter, DuckDBAdapter)) else "covarSamp"
133
+ return await self._multi_column_assoc_matrix(
134
+ table, schema, columns, agg, backend=backend, data_id=data_id, new_table=new_table
135
+ )
136
+ ```
137
+
138
+ ## Why it is correct
139
+ - One `SELECT` with B `CORR`/`COVAR_SAMP` = one table scan (SQL engines aggregate in a
140
+ single pass). Scans drop from n(n+1)/2 to ceil(n(n+1)/2 / B).
141
+ - Per-pair `CASE WHEN ai IS NOT NULL AND aj IS NOT NULL THEN ai END` keeps pairwise
142
+ deletion; cast yields NULL on empty-string/non-numeric, so old behaviour is preserved.
143
+ - Uses the DB's native `CORR`/`COVAR_SAMP` (sample, ddof=1) → matches pandas within rtol.
144
+ - `new_table`/output table removed; cache auto-persists the result DataFrame.
145
+
146
+ ## Removed
147
+ - `_exec_union_batched` (was only used by these two functions).
148
+
149
+ ## Validation
150
+ 1. `tests/integration/ops/test_inspect.py::test_corr`
151
+ `tests/integration/ops/test_stats.py::test_corr` / `test_cov` / `test_autocorr`
152
+ on DuckDB + Postgres + ClickHouse → green.
153
+ 2. `/tmp/opencode/covid_sample_test.py` (56 cols × 3 backends) vs `df.corr()/.cov()` (rtol=1e-5).
154
+ 3. Timing: 61×350k DuckDB before (~65s) vs after (seconds).
@@ -6,6 +6,26 @@ to [Semantic Versioning](https://semver.org/).
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## [0.2.2] - 2026-08-19
10
+
11
+ ### Changed
12
+ - Data Quality Reports (missing-values, completeness, numeric summary, profile report) relocated from the cleaning module to the inspect module.
13
+ - Removed bivariate association methods (chi_square, cramers_v, theil_u, mutual_information) and their categorical wrappers.
14
+ - CONTRIBUTING.md: added a distinct `[refactor]` commit tag (pure restructuring, no behavior/public-API change) separate from `[upgrade]`.
15
+ - README and docs: documented that every call compiles to backend-native SQL; added Open-in-Colab badge.
16
+
17
+ ## [0.2.1] - 2026-08-17
18
+
19
+ ### Fixed
20
+ - Postgres `.corr()`/`.cov()` now compute via streaming in-memory numpy for wide
21
+ feature sets, removing the Postgres aggregate-explosion hang and greatly
22
+ speeding up wide correlation/covariance matrices.
23
+ - DuckDB `.corr()` recursion-depth failure fixed for large column-pair counts.
24
+ - Arithmetic `add`/`subtract`/`multiply`/`divide` now correctly handle
25
+ vector-scalar operands (column ± scalar, scalar ± column, negative and float
26
+ scalars); scalar-scalar operands are now rejected with a clear `OperationError`.
27
+ - Fixed `clip` date parsing bug.
28
+
9
29
  ## [0.2.0] - 2026-08-16
10
30
 
11
31
  ### Changed
@@ -48,7 +48,8 @@ tidy. The full set:
48
48
  | ----------- | ----------------------------------------------------------------------------------------- |
49
49
  | `[feat]` | New feature addition. **Fork the repo and open a PR from a feature branch** — never commit `[feat]` directly to `main`. |
50
50
  | `[fix]` | Any bug fix — incorrect behavior, broken test, wrong SQL, etc. |
51
- | `[upgrade]` | Existing feature rewritten, refactored, or extended into a new shape (e.g. expanding the cleaning tool surface). |
51
+ | `[upgrade]` | Existing feature rewritten or extended into a new shape — **behavior or public-API change** (e.g. expanding the cleaning tool surface, moving a public method between classes). |
52
+ | `[refactor]` | Pure restructuring with **no behavior or public-API change** — moving private helpers, renaming internals, reorganizing files/modules. Can land directly on `main`. |
52
53
  | `[ci]` | CI / GitHub Actions / pre-commit / hooks / workflow changes only. |
53
54
  | `[docs]` | Documentation-only changes (README, `docs/`, docstrings). |
54
55
  | `[add]` | Adding a supporting file that does not introduce a feature (fixture, config, isolated helper). |
@@ -64,7 +65,7 @@ tidy. The full set:
64
65
 
65
66
  - Tag is lowercase, bracketed, single space, then the summary.
66
67
  - Summary is imperative mood, ≤72 chars, no trailing period.
67
- - The seven tags above are exhaustive. If a change genuinely fits none of them, prefer the closest tag and explain in the body. `[update]` is **not** a valid tag — use `[upgrade]` or `[docs]` instead.
68
+ - The eight tags above are exhaustive. If a change genuinely fits none of them, prefer the closest tag and explain in the body. `[update]` is **not** a valid tag — use `[upgrade]` or `[docs]` instead.
68
69
 
69
70
  ### Examples
70
71
 
@@ -72,6 +73,7 @@ tidy. The full set:
72
73
  git commit -m "[feat] Add ClickHouse adapter for cache reload"
73
74
  git commit -m "[fix] Avoid double COUNT(*) on empty select_dtypes result"
74
75
  git commit -m "[upgrade] Stats tool surface to full wrapper parity"
76
+ git commit -m "[refactor] Move data-quality reports into the inspect module"
75
77
  git commit -m "[ci] Remove integration job from ci workflow"
76
78
  git commit -m "[docs] Slim README and add agent section"
77
79
  git commit -m "[add] tests/datasets/sample.parquet fixture"
@@ -99,7 +101,8 @@ branch + pull request — never commit directly to `main`):
99
101
  Direct commits to `main` are reserved for the maintainer's small,
100
102
  self-contained changes that don't need another pair of eyes: single-file
101
103
  `[fix]`, single-file `[docs]`, single-file `[add]`, single-file
102
- `[remove]`, and small single-workflow `[ci]` tweaks.
104
+ `[remove]`, small `[refactor]` (no public-API change), and small
105
+ single-workflow `[ci]` tweaks.
103
106
 
104
107
  When in doubt, open a branch.
105
108
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: memframe
3
- Version: 0.2.0
3
+ Version: 0.2.2
4
4
  Summary: memFrame is a Python package for working with database-backed DataFrame operations. It lets you upload CSV files, Parquet files, and pandas DataFrames into a local DuckDB database or a remote PostgreSQL/Clickhouse database, then run DataFrame-style inspection, selection, cleaning, and table-management operations through one consistent API.
5
5
  Project-URL: Homepage, https://github.com/Debojit95/memFrame
6
6
  Project-URL: Documentation, https://debojit95.github.io/memFrame/
@@ -55,12 +55,14 @@ Description-Content-Type: text/markdown
55
55
  [![Tox](https://img.shields.io/github/actions/workflow/status/Debojit95/memFrame/tox.yml?label=tox)](https://github.com/Debojit95/memFrame/actions/workflows/tox.yml)
56
56
  [![License - AGPL-3.0](https://img.shields.io/github/license/Debojit95/memFrame)](https://github.com/Debojit95/memFrame/blob/main/LICENSE)
57
57
  [![PyPI - Downloads](https://img.shields.io/pypi/dm/memframe)](https://pypi.org/project/memframe/)
58
+ [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/drive/1UtoPjzTmia4Cr6y2btDmV8mINgXaBPGP?usp=sharing)
58
59
 
59
60
  > *memFrame brings a pandas-like DataFrame API to DuckDB, PostgreSQL, and ClickHouse — async-first, with an optional AI agent for natural-language data work.*
60
61
 
61
62
  ## Features
62
63
 
63
64
  - Database-backed DataFrame API across DuckDB, PostgreSQL, and ClickHouse.
65
+ - Compiles every pandas-style call to backend-native SQL (DuckDB / PostgreSQL / ClickHouse) and runs it in-engine — your data never leaves the database.
64
66
  - Async-first surface with sync equivalents for every operation.
65
67
  - Upload from CSV, Parquet, or pandas DataFrame.
66
68
  - Inspection, selection, cleaning, statistics, arithmetic, Plotly charts.
@@ -10,12 +10,14 @@
10
10
  [![Tox](https://img.shields.io/github/actions/workflow/status/Debojit95/memFrame/tox.yml?label=tox)](https://github.com/Debojit95/memFrame/actions/workflows/tox.yml)
11
11
  [![License - AGPL-3.0](https://img.shields.io/github/license/Debojit95/memFrame)](https://github.com/Debojit95/memFrame/blob/main/LICENSE)
12
12
  [![PyPI - Downloads](https://img.shields.io/pypi/dm/memframe)](https://pypi.org/project/memframe/)
13
+ [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/drive/1UtoPjzTmia4Cr6y2btDmV8mINgXaBPGP?usp=sharing)
13
14
 
14
15
  > *memFrame brings a pandas-like DataFrame API to DuckDB, PostgreSQL, and ClickHouse — async-first, with an optional AI agent for natural-language data work.*
15
16
 
16
17
  ## Features
17
18
 
18
19
  - Database-backed DataFrame API across DuckDB, PostgreSQL, and ClickHouse.
20
+ - Compiles every pandas-style call to backend-native SQL (DuckDB / PostgreSQL / ClickHouse) and runs it in-engine — your data never leaves the database.
19
21
  - Async-first surface with sync equivalents for every operation.
20
22
  - Upload from CSV, Parquet, or pandas DataFrame.
21
23
  - Inspection, selection, cleaning, statistics, arithmetic, Plotly charts.
@@ -46,10 +46,6 @@ Every cleaning operation has synchronous and asynchronous forms:
46
46
  | `isna()` | `await aisna()` | Boolean null mask |
47
47
  | `notna()` | `await anotna()` | Boolean non-null mask |
48
48
  | `drop_duplicates(subset=None, keep="first")` | `await adrop_duplicates(...)` | Remove duplicate rows |
49
- | `data_quality_missing_values(columns)` | `await adata_quality_missing_values(...)` | Missing-value counts |
50
- | `data_quality_completeness_score(columns)` | `await adata_quality_completeness_score(...)` | Completeness percentages |
51
- | `comprehensive_numeric_summary(columns)` | `await acomprehensive_numeric_summary(...)` | Numeric summary report |
52
- | `statistical_profile_report(columns)` | `await astatistical_profile_report(...)` | Combined profile report |
53
49
 
54
50
  Public methods return the operation value directly: usually a DataFrame, report
55
51
  dictionary, mask, or `None`. Invalid operations raise `OperationError`.
@@ -403,64 +399,6 @@ Parameters:
403
399
  | `subset` | `list[str]` or `None` | Columns used to identify duplicates. Defaults to all columns. |
404
400
  | `keep` | `"first"`, `"last"`, or `False` | Which duplicate row to keep. `False` keeps only rows with no duplicates. |
405
401
 
406
- ## Data Quality Reports
407
-
408
- ### `data_quality_missing_values`
409
-
410
- `data_quality_missing_values` returns per-column `total`, `non_null`,
411
- `missing`, and `missing_pct` values.
412
-
413
- ```python
414
- result = dataset.data_quality_missing_values(
415
- columns=["salary", "department"],
416
- )
417
- ```
418
-
419
- ```python
420
- result = await dataset.adata_quality_missing_values(
421
- columns=["salary", "department"],
422
- )
423
- ```
424
-
425
- ### `data_quality_completeness_score`
426
-
427
- `data_quality_completeness_score` returns completeness percentages for each
428
- requested column.
429
-
430
- ```python
431
- result = dataset.data_quality_completeness_score(columns=["salary"])
432
- ```
433
-
434
- ```python
435
- result = await dataset.adata_quality_completeness_score(columns=["salary"])
436
- ```
437
-
438
- ### `comprehensive_numeric_summary`
439
-
440
- `comprehensive_numeric_summary` generates numeric summaries for up to the first
441
- 20 requested columns.
442
-
443
- ```python
444
- result = dataset.comprehensive_numeric_summary(columns=["salary", "bonus"])
445
- ```
446
-
447
- ```python
448
- result = await dataset.acomprehensive_numeric_summary(columns=["salary", "bonus"])
449
- ```
450
-
451
- ### `statistical_profile_report`
452
-
453
- `statistical_profile_report` combines completeness scoring and numeric summary
454
- output into one response.
455
-
456
- ```python
457
- result = dataset.statistical_profile_report(columns=["salary", "bonus"])
458
- ```
459
-
460
- ```python
461
- result = await dataset.astatistical_profile_report(columns=["salary", "bonus"])
462
- ```
463
-
464
402
  ## Return Values and Errors
465
403
 
466
404
  Public cleaning methods return the underlying DataFrame, dictionary, mask, or
@@ -545,11 +483,3 @@ Cleaning methods raise `OperationError` for invalid input or backend failures.
545
483
  - notna
546
484
  - adrop_duplicates
547
485
  - drop_duplicates
548
- - adata_quality_missing_values
549
- - data_quality_missing_values
550
- - adata_quality_completeness_score
551
- - data_quality_completeness_score
552
- - acomprehensive_numeric_summary
553
- - comprehensive_numeric_summary
554
- - astatistical_profile_report
555
- - statistical_profile_report
@@ -38,24 +38,21 @@ Every inspect operation has synchronous and asynchronous forms:
38
38
  | `info()` | `await ainfo()` | Per-column table information |
39
39
  | `describe(columns=None)` | `await adescribe(...)` | Numeric descriptive statistics |
40
40
  | `null_analysis(columns=None)` | `await anull_analysis(...)` | Null distribution by column |
41
- | `corr(columns=None, method="pearson")` | `await acorr(...)` | Numeric correlation matrix |
41
+ | `data_quality_missing_values(columns)` | `await adata_quality_missing_values(...)` | Missing-value counts |
42
+ | `data_quality_completeness_score(columns)` | `await adata_quality_completeness_score(...)` | Completeness percentages |
43
+ | `comprehensive_numeric_summary(columns)` | `await acomprehensive_numeric_summary(...)` | Numeric summary report |
44
+ | `statistical_profile_report(columns)` | `await astatistical_profile_report(...)` | Combined profile report |
42
45
  | `full_table(columns=None, chunk_size=None)` | `await afull_table(...)` | Full table or chunk iterator |
43
46
  | `astype(columns=None, dtypes=None, dtype_map=None)` | `await aastype(...)` | Cast selected columns |
44
47
  | `insert(column, value)` | `await ainsert(...)` | Add a column from a value list |
45
48
  | `map(func, na_action=None, columns=None, datetime_action="skip")` | `await amap(...)` | Apply SQL expression to values |
46
49
  | `rename(columns)` | `await arename(...)` | Rename columns |
47
50
  | `set_index(columns)` | `await aset_index(...)` | Add a primary-key index |
48
- | `reset_index()` | `await areset_index()` | Recreate an `id` index column |
49
51
  | `update(on, other_table, other_schema="upload", overwrite=True, errors="ignore")` | `await aupdate(...)` | Update from another table |
50
52
  | `resample(time_column, rule, agg="COUNT", value_column=None, label="left", closed="left")` | `await aresample(...)` | Time-series aggregation |
51
- | `axes()` | `await aaxes()` | Row and column axes |
52
53
  | `columns()` | `await acolumns()` | Column labels |
53
54
  | `dtypes()` | `await adtypes()` | Column database types |
54
- | `first_valid_index()` | `await afirst_valid_index()` | First row with non-null values |
55
- | `memory_usage()` | `await amemory_usage()` | Backend table memory usage |
56
- | `ndim()` | `await andim()` | Number of dimensions |
57
55
  | `shape()` | `await ashape()` | Row and column count |
58
- | `size()` | `await asize()` | Total element count |
59
56
  | `values()` | `await avalues()` | Table values as nested lists |
60
57
  | `items()` | `await aitems()` | Column/value iterator result |
61
58
  | `iterrows()` | `await aiterrows()` | Row iterator result |
@@ -241,27 +238,64 @@ Parameters:
241
238
  The result DataFrame is indexed by column name when data is available and
242
239
  contains `contains_null` and `percent_missing` columns.
243
240
 
244
- ### `corr`
245
241
 
246
- `corr` computes a numeric correlation matrix.
242
+ ## Data Quality Reports
243
+
244
+ ### `data_quality_missing_values`
245
+
246
+ `data_quality_missing_values` returns per-column `total`, `non_null`,
247
+ `missing`, and `missing_pct` values.
247
248
 
248
249
  ```python
249
- result = dataset.corr(columns=["amount", "score"])
250
+ result = dataset.data_quality_missing_values(
251
+ columns=["salary", "department"],
252
+ )
250
253
  ```
251
254
 
252
255
  ```python
253
- result = await dataset.acorr()
256
+ result = await dataset.adata_quality_missing_values(
257
+ columns=["salary", "department"],
258
+ )
254
259
  ```
255
260
 
256
- Parameters:
261
+ ### `data_quality_completeness_score`
257
262
 
258
- | Parameter | Type | Description |
259
- | --- | --- | --- |
260
- | `columns` | `list[str]`, `"*"`, or `None` | Numeric columns to include. Defaults to all numeric columns. |
261
- | `method` | `str` | Accepted by the wrapper. The current core implementation computes Pearson correlation with SQL `CORR`. |
263
+ `data_quality_completeness_score` returns completeness percentages for each
264
+ requested column.
265
+
266
+ ```python
267
+ result = dataset.data_quality_completeness_score(columns=["salary"])
268
+ ```
269
+
270
+ ```python
271
+ result = await dataset.adata_quality_completeness_score(columns=["salary"])
272
+ ```
273
+
274
+ ### `comprehensive_numeric_summary`
262
275
 
263
- At least two numeric columns are required. Metadata includes the analyzed
264
- columns and up to ten strong correlations.
276
+ `comprehensive_numeric_summary` generates numeric summaries for up to the first
277
+ 20 requested columns.
278
+
279
+ ```python
280
+ result = dataset.comprehensive_numeric_summary(columns=["salary", "bonus"])
281
+ ```
282
+
283
+ ```python
284
+ result = await dataset.acomprehensive_numeric_summary(columns=["salary", "bonus"])
285
+ ```
286
+
287
+ ### `statistical_profile_report`
288
+
289
+ `statistical_profile_report` combines completeness scoring and numeric summary
290
+ output into one response.
291
+
292
+ ```python
293
+ result = dataset.statistical_profile_report(columns=["salary", "bonus"])
294
+ ```
295
+
296
+ ```python
297
+ result = await dataset.astatistical_profile_report(columns=["salary", "bonus"])
298
+ ```
265
299
 
266
300
  ## Table Utilities
267
301
 
@@ -388,19 +422,14 @@ Parameters:
388
422
  | --- | --- | --- |
389
423
  | `columns` | `dict[str, str]` | Mapping of old column names to new names. |
390
424
 
391
- ### `set_index` and `reset_index`
425
+ ### `set_index`
392
426
 
393
- `set_index` adds a primary-key constraint over selected columns. `reset_index`
394
- recreates an `id` index column.
427
+ `set_index` adds a primary-key constraint over selected columns.
395
428
 
396
429
  ```python
397
430
  result = dataset.set_index(columns=["customer_id"])
398
431
  ```
399
432
 
400
- ```python
401
- result = await dataset.areset_index()
402
- ```
403
-
404
433
  `set_index` parameters:
405
434
 
406
435
  | Parameter | Type | Description |
@@ -476,14 +505,9 @@ These methods return compact dictionary or list payloads directly:
476
505
 
477
506
  | Method | Async | Public result |
478
507
  | --- | --- | --- |
479
- | `axes()` | `aaxes()` | `[row_index, columns]` |
480
508
  | `columns()` | `acolumns()` | `[...]` |
481
509
  | `dtypes()` | `adtypes()` | `{column: db_type}` |
482
- | `first_valid_index()` | `afirst_valid_index()` | `{"first_valid_index": 0}` or `None` |
483
- | `memory_usage()` | `amemory_usage()` | `{"memory_bytes": value}` |
484
- | `ndim()` | `andim()` | `{"ndim": 2}` |
485
510
  | `shape()` | `ashape()` | `{"shape": (rows, columns)}` |
486
- | `size()` | `asize()` | `{"size": rows * columns}` |
487
511
  | `values()` | `avalues()` | `{"values": [[...], ...]}` |
488
512
 
489
513
  ```python
@@ -536,7 +560,7 @@ Failed operations raise `OperationError`.
536
560
 
537
561
  Some inspect methods create generated summary tables internally when method-call
538
562
  logging receives backend context. This is most visible for `info`, `describe`,
539
- `null_analysis`, and `corr`.
563
+ `null_analysis`, and `sample`.
540
564
 
541
565
  Preview methods such as `head`, `tail`, `sample`, and unchunked `full_table`
542
566
  are read-oriented and return a DataFrame sample directly.
@@ -548,10 +572,9 @@ Inspect supports DuckDB and PostgreSQL adapters:
548
572
  - Both backends use quoted identifiers and schema-aware table names.
549
573
  - Schema discovery is delegated to the active database adapter.
550
574
  - `describe` uses backend-specific percentile functions.
551
- - `corr` uses SQL `CORR` over numeric columns.
552
575
  - `sample` uses backend `RANDOM()` ordering.
553
- - PostgreSQL can report relation memory usage; DuckDB currently returns `None`
554
- for `memory_usage`.
576
+ - PostgreSQL can report relation memory usage through `info`; DuckDB currently
577
+ returns `None` for `memory_usage`.
555
578
  - Column and table identifiers are sanitized before SQL is generated.
556
579
 
557
580
  ## Errors
@@ -560,14 +583,13 @@ Inspect methods raise `OperationError` for invalid input or backend failures.
560
583
 
561
584
  - Invalid `chunk_size` in `full_table` raises `OperationError`.
562
585
  - `describe` raises an error when no numeric columns are available.
563
- - `corr` raises an error when fewer than two numeric columns are available.
564
586
  - `astype` raises an error for missing columns or unsupported dtype aliases.
565
587
  - `astype` requires either `dtype_map` or matching `columns` and `dtypes`.
566
588
  - `insert` raises an error when `value` is not a list or its length does not
567
589
  match the row count.
568
590
  - `map` raises an error when `func` is not a SQL expression string or when no
569
591
  selected columns are compatible with the expression.
570
- - `rename`, `set_index`, `reset_index`, `update`, and `resample` can raise
592
+ - `rename`, `set_index`, `update`, and `resample` can raise
571
593
  backend SQL errors when identifiers or constraints are invalid.
572
594
 
573
595
  ## API Reference
@@ -589,8 +611,6 @@ Inspect methods raise `OperationError` for invalid input or backend failures.
589
611
  - describe
590
612
  - anull_analysis
591
613
  - null_analysis
592
- - acorr
593
- - corr
594
614
  - afull_table
595
615
  - full_table
596
616
  - aastype
@@ -603,28 +623,16 @@ Inspect methods raise `OperationError` for invalid input or backend failures.
603
623
  - rename
604
624
  - aset_index
605
625
  - set_index
606
- - areset_index
607
- - reset_index
608
626
  - aupdate
609
627
  - update
610
628
  - aresample
611
629
  - resample
612
- - aaxes
613
- - axes
614
630
  - acolumns
615
631
  - columns
616
632
  - adtypes
617
633
  - dtypes
618
- - afirst_valid_index
619
- - first_valid_index
620
- - amemory_usage
621
- - memory_usage
622
- - andim
623
- - ndim
624
634
  - ashape
625
635
  - shape
626
- - asize
627
- - size
628
636
  - avalues
629
637
  - values
630
638
  - aitems