memframe 0.2.0__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {memframe-0.2.0 → memframe-0.2.2}/.github/workflows/release-testpypi.yml +8 -1
- {memframe-0.2.0 → memframe-0.2.2}/.gitignore +1 -0
- memframe-0.2.2/.opencode/plans/corr_cov_single_scan.md +154 -0
- {memframe-0.2.0 → memframe-0.2.2}/CHANGELOG.md +20 -0
- {memframe-0.2.0 → memframe-0.2.2}/CONTRIBUTING.md +6 -3
- {memframe-0.2.0 → memframe-0.2.2}/PKG-INFO +3 -1
- {memframe-0.2.0 → memframe-0.2.2}/README.md +2 -0
- {memframe-0.2.0 → memframe-0.2.2}/docs/api/cleaning.md +0 -70
- {memframe-0.2.0 → memframe-0.2.2}/docs/api/inspect.md +58 -50
- {memframe-0.2.0 → memframe-0.2.2}/docs/api/selection.md +24 -104
- {memframe-0.2.0 → memframe-0.2.2}/docs/api/stats.md +11 -44
- {memframe-0.2.0 → memframe-0.2.2}/docs/getting-started.md +0 -10
- {memframe-0.2.0 → memframe-0.2.2}/docs/index.md +1 -0
- {memframe-0.2.0 → memframe-0.2.2}/pyproject.toml +1 -1
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/analytix/arithmetic.py +81 -53
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/analytix/cleaning.py +0 -96
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/analytix/inspection.py +157 -178
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/analytix/selection.py +57 -219
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/analytix/stats.py +286 -449
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/orchestrator/analytix/cleaning.py +12 -32
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/orchestrator/analytix/inspection.py +26 -31
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/orchestrator/analytix/selection.py +14 -107
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/orchestrator/analytix/stats.py +34 -40
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/main.py +1 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/analytix/cleaning.py +12 -73
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/analytix/cleaning.pyi +0 -39
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/analytix/inspection.py +68 -58
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/analytix/inspection.pyi +39 -14
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/analytix/selection.py +4 -94
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/analytix/selection.pyi +2 -42
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/analytix/stats.py +14 -74
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/analytix/stats.pyi +0 -44
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/tools/clean.py +0 -27
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/tools/inspect.py +22 -27
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/tools/select.py +6 -26
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/tools/stats.py +3 -21
- {memframe-0.2.0 → memframe-0.2.2}/tests/integration/ops/test_arithmetic.py +92 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/integration/ops/test_cleaning.py +0 -50
- {memframe-0.2.0 → memframe-0.2.2}/tests/integration/ops/test_inspect.py +38 -38
- {memframe-0.2.0 → memframe-0.2.2}/tests/integration/ops/test_selection.py +3 -75
- {memframe-0.2.0 → memframe-0.2.2}/tests/integration/ops/test_stats.py +25 -59
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_clean_tools.py +1 -1
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_inspect_tools.py +1 -1
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_select_tools.py +1 -1
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_stats_tools.py +1 -1
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_tools.py +0 -9
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_errors.py +12 -12
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_inspection_response.py +1 -1
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_public_results.py +1 -1
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_selection_response.py +4 -5
- {memframe-0.2.0 → memframe-0.2.2}/uv.lock +1 -1
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/building-pydantic-ai-agents/SKILL.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/building-pydantic-ai-agents/references/AGENTS-CORE.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/building-pydantic-ai-agents/references/ARCHITECTURE.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/building-pydantic-ai-agents/references/CAPABILITIES-AND-HOOKS.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/building-pydantic-ai-agents/references/COMMON-TASKS.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/building-pydantic-ai-agents/references/INPUT-AND-HISTORY.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/building-pydantic-ai-agents/references/NATIVE-TOOLS.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/building-pydantic-ai-agents/references/ON-DEMAND-CAPABILITIES.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/building-pydantic-ai-agents/references/ORCHESTRATION-AND-INTEGRATIONS.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/building-pydantic-ai-agents/references/TESTING-AND-DEBUGGING.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/building-pydantic-ai-agents/references/TOOLS-ADVANCED.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/building-pydantic-ai-agents/references/TOOLS-CORE.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/SKILL.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/collector/host-and-infra-metrics.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/javascript/ai-sdk.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/javascript/cloudflare-and-deno.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/javascript/frameworks.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/javascript/installation-and-env.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/javascript/nextjs.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/javascript/node-runtime.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/javascript/patterns.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/javascript/project-detection.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/javascript/react-browser.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/javascript/verification-troubleshooting.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/python/integrations.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/python/logging-patterns.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-instrumentation/references/rust/patterns.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-query/SKILL.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-query/references/client-usage.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-query/references/schema.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/logfire-ui/SKILL.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/pydantic/SKILL.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/pydantic-ai-harness/SKILL.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.agents/skills/pydantic-ai-harness/references/CODE-MODE.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.env.test.example +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.githooks/pre-commit +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.githooks/precommit.sh +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.github/workflows/ci.yml +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.github/workflows/docs.yml +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.github/workflows/release-prod.yml +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.github/workflows/tox.yml +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.opencode/opencode.json +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/.python-version +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/CODE_OF_CONDUCT.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/LICENSE +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/docs/CODE_OF_CONDUCT.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/docs/CONTRIBUTING.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/docs/api/agent.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/docs/api/arithmetic.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/docs/api/bar.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/docs/api/bar_polar.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/docs/api/caching.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/docs/api/connector.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/docs/api/database.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/docs/api/line.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/docs/api/pie.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/docs/api/scatter.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/docs/api/scatter3d.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/docs/api/upload-manager.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/docs/assets/logo-nav.png +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/docs/assets/memframe-logo-full.png +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/docs/stylesheets/extra.css +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/docs/testing.md +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/mkdocs.yml +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/scripts/release.sh +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/scripts/run-commit-checks.sh +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/skills-lock.json +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/__init__.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/cache/__init__.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/cache/cache_manager.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/__init__.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/analytix/_response.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/ingestion/datatype_detector.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/ingestion/upload/__init__.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/ingestion/upload/base.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/ingestion/upload/clickhouse.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/ingestion/upload/duckdb.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/ingestion/upload/postgres.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/ingestion/upload/strategies/__init__.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/ingestion/upload/strategies/base.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/ingestion/upload/strategies/csv.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/ingestion/upload/strategies/df.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/ingestion/upload/strategies/parquet.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/orchestrator/analytix/arithmetic.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/orchestrator/plots/bar.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/orchestrator/plots/bar_polar.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/orchestrator/plots/line.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/orchestrator/plots/pie.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/orchestrator/plots/scatter.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/orchestrator/plots/scatter_3d.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/plots/bar.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/plots/bar_polar.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/plots/line.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/plots/pie.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/plots/scatter.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/core/plots/scatter_3d.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/__init__.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/adapters/base.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/adapters/clickhouse.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/adapters/duckdb.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/adapters/factory.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/adapters/postgresql.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/connection/__init__.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/connection/connector.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/connection/pool.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/context.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/context.pyi +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/ops.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/setup/__init__.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/setup/base.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/setup/clickhouse.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/setup/duckdb.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/db_manager/setup/postgres.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/exceptions.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/main.pyi +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/py.typed +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/utils/__init__.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/utils/async_sync.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/utils/helper.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/utils/plot_renderer.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/__init__.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/analytix/__init__.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/analytix/arithmetic.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/analytix/arithmetic.pyi +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/base.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/__init__.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/bar.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/bar.pyi +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/bar_polar.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/bar_polar.pyi +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/line.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/line.pyi +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/pie.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/pie.pyi +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/scatter.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/scatter.pyi +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/scatter3d.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe/wrappers/plots/scatter3d.pyi +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/__init__.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/agents/__init__.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/agents/analytics.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/agents/planning.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/config.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/domain.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/entrypoints.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/gateway.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/observe.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/sessions.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/tools/__init__.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/tools/_helpers.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/tools/arithmetic.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/tools/context.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/tools/plot.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/tools/upload.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/src/memframe_ai/wrappers.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/conftest.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/datasets/sample.csv +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/datasets/sample.parquet +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/db_test_utils.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/integration/conftest.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/integration/ops/test_bar.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/integration/ops/test_bar_polar.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/integration/ops/test_line.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/integration/ops/test_pie.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/integration/ops/test_scatter.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/integration/ops/test_scatter3d.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/integration/ops/test_upload.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/integration/test_concurrency.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/integration/test_lifecycle.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/integration/test_parity.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/run_tests.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_arithmetic_tools.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_domain.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_entrypoints.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_fleet.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_gateway.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_observe.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_planning.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/ai/test_sessions.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_arithmetic_response.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_async_sync.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_cleaning_orchestrator_response.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_cleaning_response.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_connector.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_context.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_csv_invalid_values.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_datatype_detector.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_exceptions.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_factory.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_helper.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_inspection_orchestrator_response.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_ops_mixin.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_plot_renderer.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_pool.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_response_unwrap.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_setup.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tests/unit/test_stats_response.py +0 -0
- {memframe-0.2.0 → memframe-0.2.2}/tox.ini +0 -0
|
@@ -20,7 +20,14 @@ jobs:
|
|
|
20
20
|
env:
|
|
21
21
|
UV_PUBLISH_TOKEN: ${{ secrets.TEST_PYPI_TOKEN }}
|
|
22
22
|
run: uv publish --publish-url https://test.pypi.org/legacy/
|
|
23
|
+
- name: Extract release notes from CHANGELOG
|
|
24
|
+
env:
|
|
25
|
+
VERSION: ${{ github.ref_name }}
|
|
26
|
+
run: |
|
|
27
|
+
ver="${VERSION#v}"
|
|
28
|
+
awk -v v="## [$ver]" '$0 ~ "^## \\[" ver "\\]"{f=1;next} /^## \[/{if(f)exit} f' CHANGELOG.md > release_notes.md
|
|
29
|
+
cat release_notes.md
|
|
23
30
|
- name: Create GitHub release
|
|
24
31
|
uses: softprops/action-gh-release@v2
|
|
25
32
|
with:
|
|
26
|
-
|
|
33
|
+
body_path: release_notes.md
|
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
# Plan: Pure in-DB single-scan correlation / covariance (Tier-1 + Tier-2 ClickHouse)
|
|
2
|
+
|
|
3
|
+
Status: APPLICATION BLOCKED — an active permission rule denies `edit` on source files
|
|
4
|
+
(`edit * -> deny`), consistent with read-only plan mode. The code below is ready to
|
|
5
|
+
apply once edit permission / plan mode is lifted.
|
|
6
|
+
|
|
7
|
+
## Goal
|
|
8
|
+
Replace the `UNION ALL` of per-pair `CORR()`/`COVAR_SAMP()` (one full table scan per
|
|
9
|
+
pair ≈ n²/2 scans) with batches of B aggregate calls inside ONE `SELECT` (one scan per
|
|
10
|
+
batch). The DB engine evaluates all aggregates in a single pass over the data. Keep the
|
|
11
|
+
DB-native `CORR`/`COVAR_SAMP` so results still match `pandas.corr()/.cov()` within
|
|
12
|
+
`rtol=1e-5` (existing tests pass). Drop the intermediate output table.
|
|
13
|
+
|
|
14
|
+
## Tier-2 (ClickHouse, adaptive single-scan, no UDF)
|
|
15
|
+
Batch size `B = min(total_pairs, 250)` for ClickHouse; when the whole pair set fits the
|
|
16
|
+
expression budget the loop issues a *single* `SELECT` (one scan). Wider tables fall back
|
|
17
|
+
to batching (still one scan per batch). DuckDB/Postgres use `B = min(total_pairs, 500)`.
|
|
18
|
+
|
|
19
|
+
## File: src/memframe/core/analytix/stats.py
|
|
20
|
+
|
|
21
|
+
### 1) Replace `_exec_union_batched` (lines 809-819) with this helper
|
|
22
|
+
|
|
23
|
+
```python
|
|
24
|
+
async def _multi_column_assoc_matrix(
|
|
25
|
+
self,
|
|
26
|
+
table: str,
|
|
27
|
+
schema: str,
|
|
28
|
+
columns: List[str],
|
|
29
|
+
agg_fn: str,
|
|
30
|
+
backend=None,
|
|
31
|
+
data_id: Optional[str] = None,
|
|
32
|
+
new_table: Optional[str] = None,
|
|
33
|
+
) -> Dict[str, Any]:
|
|
34
|
+
"""Compute an n x n correlation/covariance matrix purely in the database.
|
|
35
|
+
|
|
36
|
+
All CORR/COVAR_SAMP calls are batched into single SELECT statements so the
|
|
37
|
+
engine scans the table once per batch, not once per pair. ClickHouse uses a
|
|
38
|
+
tighter batch so a single scan covers the table whenever the pair count fits
|
|
39
|
+
its expression budget (Tier-2); wider tables fall back to batching.
|
|
40
|
+
"""
|
|
41
|
+
try:
|
|
42
|
+
if not isinstance(self.db, (PostgresAdapter, DuckDBAdapter, ClickHouseAdapter)):
|
|
43
|
+
raise self._unsupported_backend_error()
|
|
44
|
+
|
|
45
|
+
q = self._qualified_table(table, schema)
|
|
46
|
+
cols = [SQLIdentifierSanitizer.sanitize(c) for c in columns]
|
|
47
|
+
n = len(cols)
|
|
48
|
+
if n < 1:
|
|
49
|
+
return self._error_response("No columns provided", columns)
|
|
50
|
+
|
|
51
|
+
def _cast(c: str) -> str:
|
|
52
|
+
if isinstance(self.db, PostgresAdapter):
|
|
53
|
+
return f'CAST(NULLIF("{c}"::text, \'\') AS DOUBLE PRECISION)'
|
|
54
|
+
if isinstance(self.db, DuckDBAdapter):
|
|
55
|
+
return f'TRY_CAST("{c}" AS DOUBLE)'
|
|
56
|
+
return f'toFloat64OrNull(toString("{c}"))'
|
|
57
|
+
|
|
58
|
+
alias = {c: f"a{idx}" for idx, c in enumerate(cols)}
|
|
59
|
+
pairs = [(i, j) for i in range(n) for j in range(i, n)]
|
|
60
|
+
|
|
61
|
+
# ponytail: batch size chosen under each backend's expression-count limit
|
|
62
|
+
if isinstance(self.db, ClickHouseAdapter):
|
|
63
|
+
B = min(len(pairs), 250)
|
|
64
|
+
else:
|
|
65
|
+
B = min(len(pairs), 500)
|
|
66
|
+
|
|
67
|
+
results: Dict[tuple, Any] = {}
|
|
68
|
+
for start in range(0, len(pairs), B):
|
|
69
|
+
batch = pairs[start:start + B]
|
|
70
|
+
used = sorted(
|
|
71
|
+
{cols[i] for (i, j) in batch} | {cols[j] for (i, j) in batch},
|
|
72
|
+
key=lambda c: cols.index(c),
|
|
73
|
+
)
|
|
74
|
+
cte = ", ".join(f"{_cast(c)} AS {alias[c]}" for c in used)
|
|
75
|
+
sel = []
|
|
76
|
+
for k, (i, j) in enumerate(batch):
|
|
77
|
+
ai, aj = alias[cols[i]], alias[cols[j]]
|
|
78
|
+
cond = f"{ai} IS NOT NULL AND {aj} IS NOT NULL"
|
|
79
|
+
sel.append(
|
|
80
|
+
f"{agg_fn}(CASE WHEN {cond} THEN {ai} END, "
|
|
81
|
+
f"CASE WHEN {cond} THEN {aj} END) AS v_{k}"
|
|
82
|
+
)
|
|
83
|
+
sql = f"WITH q AS (SELECT {cte} FROM {q}) SELECT " + ", ".join(sel) + " FROM q"
|
|
84
|
+
rows = await self._fetch(sql)
|
|
85
|
+
row = rows[0] if rows else {}
|
|
86
|
+
for k, (i, j) in enumerate(batch):
|
|
87
|
+
val = row.get(f"v_{k}") if row else None
|
|
88
|
+
results[(i, j)] = val if val is not None else float("nan")
|
|
89
|
+
|
|
90
|
+
mat = [[results.get((min(i, j), max(i, j))) for j in range(n)] for i in range(n)]
|
|
91
|
+
df = pd.DataFrame(mat, index=columns, columns=columns)
|
|
92
|
+
msg = f"Computed {agg_fn} matrix for {n} columns"
|
|
93
|
+
return self._success_response(msg, columns, result=df)
|
|
94
|
+
|
|
95
|
+
except Exception as e:
|
|
96
|
+
return self._error_response(
|
|
97
|
+
f"multi_column_assoc_matrix error: {str(e)}\n{traceback.format_exc()}",
|
|
98
|
+
columns,
|
|
99
|
+
)
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
### 2) Replace `numeric_multi_column_correlation` (lines 821-889) with
|
|
103
|
+
|
|
104
|
+
```python
|
|
105
|
+
async def numeric_multi_column_correlation(
|
|
106
|
+
self,
|
|
107
|
+
table: str,
|
|
108
|
+
schema: str,
|
|
109
|
+
columns: List[str],
|
|
110
|
+
backend=None,
|
|
111
|
+
data_id: Optional[str] = None,
|
|
112
|
+
new_table: Optional[str] = None,
|
|
113
|
+
) -> Dict[str, Any]:
|
|
114
|
+
agg = "CORR" if isinstance(self.db, (PostgresAdapter, DuckDBAdapter)) else "corr"
|
|
115
|
+
return await self._multi_column_assoc_matrix(
|
|
116
|
+
table, schema, columns, agg, backend=backend, data_id=data_id, new_table=new_table
|
|
117
|
+
)
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
### 3) Replace `numeric_multi_column_covariance` (lines 891-952) with
|
|
121
|
+
|
|
122
|
+
```python
|
|
123
|
+
async def numeric_multi_column_covariance(
|
|
124
|
+
self,
|
|
125
|
+
table: str,
|
|
126
|
+
schema: str,
|
|
127
|
+
columns: List[str],
|
|
128
|
+
backend=None,
|
|
129
|
+
data_id: Optional[str] = None,
|
|
130
|
+
new_table: Optional[str] = None,
|
|
131
|
+
) -> Dict[str, Any]:
|
|
132
|
+
agg = "COVAR_SAMP" if isinstance(self.db, (PostgresAdapter, DuckDBAdapter)) else "covarSamp"
|
|
133
|
+
return await self._multi_column_assoc_matrix(
|
|
134
|
+
table, schema, columns, agg, backend=backend, data_id=data_id, new_table=new_table
|
|
135
|
+
)
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
## Why it is correct
|
|
139
|
+
- One `SELECT` with B `CORR`/`COVAR_SAMP` = one table scan (SQL engines aggregate in a
|
|
140
|
+
single pass). Scans drop from n(n+1)/2 to ceil(n(n+1)/2 / B).
|
|
141
|
+
- Per-pair `CASE WHEN ai IS NOT NULL AND aj IS NOT NULL THEN ai END` keeps pairwise
|
|
142
|
+
deletion; cast yields NULL on empty-string/non-numeric, so old behaviour is preserved.
|
|
143
|
+
- Uses the DB's native `CORR`/`COVAR_SAMP` (sample, ddof=1) → matches pandas within rtol.
|
|
144
|
+
- `new_table`/output table removed; cache auto-persists the result DataFrame.
|
|
145
|
+
|
|
146
|
+
## Removed
|
|
147
|
+
- `_exec_union_batched` (was only used by these two functions).
|
|
148
|
+
|
|
149
|
+
## Validation
|
|
150
|
+
1. `tests/integration/ops/test_inspect.py::test_corr`
|
|
151
|
+
`tests/integration/ops/test_stats.py::test_corr` / `test_cov` / `test_autocorr`
|
|
152
|
+
on DuckDB + Postgres + ClickHouse → green.
|
|
153
|
+
2. `/tmp/opencode/covid_sample_test.py` (56 cols × 3 backends) vs `df.corr()/.cov()` (rtol=1e-5).
|
|
154
|
+
3. Timing: 61×350k DuckDB before (~65s) vs after (seconds).
|
|
@@ -6,6 +6,26 @@ to [Semantic Versioning](https://semver.org/).
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.2.2] - 2026-08-19
|
|
10
|
+
|
|
11
|
+
### Changed
|
|
12
|
+
- Data Quality Reports (missing-values, completeness, numeric summary, profile report) relocated from the cleaning module to the inspect module.
|
|
13
|
+
- Removed bivariate association methods (chi_square, cramers_v, theil_u, mutual_information) and their categorical wrappers.
|
|
14
|
+
- CONTRIBUTING.md: added a distinct `[refactor]` commit tag (pure restructuring, no behavior/public-API change) separate from `[upgrade]`.
|
|
15
|
+
- README and docs: documented that every call compiles to backend-native SQL; added Open-in-Colab badge.
|
|
16
|
+
|
|
17
|
+
## [0.2.1] - 2026-08-17
|
|
18
|
+
|
|
19
|
+
### Fixed
|
|
20
|
+
- Postgres `.corr()`/`.cov()` now compute via streaming in-memory numpy for wide
|
|
21
|
+
feature sets, removing the Postgres aggregate-explosion hang and greatly
|
|
22
|
+
speeding up wide correlation/covariance matrices.
|
|
23
|
+
- DuckDB `.corr()` recursion-depth failure fixed for large column-pair counts.
|
|
24
|
+
- Arithmetic `add`/`subtract`/`multiply`/`divide` now correctly handle
|
|
25
|
+
vector-scalar operands (column ± scalar, scalar ± column, negative and float
|
|
26
|
+
scalars); scalar-scalar operands are now rejected with a clear `OperationError`.
|
|
27
|
+
- Fixed `clip` date parsing bug.
|
|
28
|
+
|
|
9
29
|
## [0.2.0] - 2026-08-16
|
|
10
30
|
|
|
11
31
|
### Changed
|
|
@@ -48,7 +48,8 @@ tidy. The full set:
|
|
|
48
48
|
| ----------- | ----------------------------------------------------------------------------------------- |
|
|
49
49
|
| `[feat]` | New feature addition. **Fork the repo and open a PR from a feature branch** — never commit `[feat]` directly to `main`. |
|
|
50
50
|
| `[fix]` | Any bug fix — incorrect behavior, broken test, wrong SQL, etc. |
|
|
51
|
-
| `[upgrade]` | Existing feature rewritten
|
|
51
|
+
| `[upgrade]` | Existing feature rewritten or extended into a new shape — **behavior or public-API change** (e.g. expanding the cleaning tool surface, moving a public method between classes). |
|
|
52
|
+
| `[refactor]` | Pure restructuring with **no behavior or public-API change** — moving private helpers, renaming internals, reorganizing files/modules. Can land directly on `main`. |
|
|
52
53
|
| `[ci]` | CI / GitHub Actions / pre-commit / hooks / workflow changes only. |
|
|
53
54
|
| `[docs]` | Documentation-only changes (README, `docs/`, docstrings). |
|
|
54
55
|
| `[add]` | Adding a supporting file that does not introduce a feature (fixture, config, isolated helper). |
|
|
@@ -64,7 +65,7 @@ tidy. The full set:
|
|
|
64
65
|
|
|
65
66
|
- Tag is lowercase, bracketed, single space, then the summary.
|
|
66
67
|
- Summary is imperative mood, ≤72 chars, no trailing period.
|
|
67
|
-
- The
|
|
68
|
+
- The eight tags above are exhaustive. If a change genuinely fits none of them, prefer the closest tag and explain in the body. `[update]` is **not** a valid tag — use `[upgrade]` or `[docs]` instead.
|
|
68
69
|
|
|
69
70
|
### Examples
|
|
70
71
|
|
|
@@ -72,6 +73,7 @@ tidy. The full set:
|
|
|
72
73
|
git commit -m "[feat] Add ClickHouse adapter for cache reload"
|
|
73
74
|
git commit -m "[fix] Avoid double COUNT(*) on empty select_dtypes result"
|
|
74
75
|
git commit -m "[upgrade] Stats tool surface to full wrapper parity"
|
|
76
|
+
git commit -m "[refactor] Move data-quality reports into the inspect module"
|
|
75
77
|
git commit -m "[ci] Remove integration job from ci workflow"
|
|
76
78
|
git commit -m "[docs] Slim README and add agent section"
|
|
77
79
|
git commit -m "[add] tests/datasets/sample.parquet fixture"
|
|
@@ -99,7 +101,8 @@ branch + pull request — never commit directly to `main`):
|
|
|
99
101
|
Direct commits to `main` are reserved for the maintainer's small,
|
|
100
102
|
self-contained changes that don't need another pair of eyes: single-file
|
|
101
103
|
`[fix]`, single-file `[docs]`, single-file `[add]`, single-file
|
|
102
|
-
`[remove]`,
|
|
104
|
+
`[remove]`, small `[refactor]` (no public-API change), and small
|
|
105
|
+
single-workflow `[ci]` tweaks.
|
|
103
106
|
|
|
104
107
|
When in doubt, open a branch.
|
|
105
108
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: memframe
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.2
|
|
4
4
|
Summary: memFrame is a Python package for working with database-backed DataFrame operations. It lets you upload CSV files, Parquet files, and pandas DataFrames into a local DuckDB database or a remote PostgreSQL/Clickhouse database, then run DataFrame-style inspection, selection, cleaning, and table-management operations through one consistent API.
|
|
5
5
|
Project-URL: Homepage, https://github.com/Debojit95/memFrame
|
|
6
6
|
Project-URL: Documentation, https://debojit95.github.io/memFrame/
|
|
@@ -55,12 +55,14 @@ Description-Content-Type: text/markdown
|
|
|
55
55
|
[](https://github.com/Debojit95/memFrame/actions/workflows/tox.yml)
|
|
56
56
|
[](https://github.com/Debojit95/memFrame/blob/main/LICENSE)
|
|
57
57
|
[](https://pypi.org/project/memframe/)
|
|
58
|
+
[](https://colab.research.google.com/drive/1UtoPjzTmia4Cr6y2btDmV8mINgXaBPGP?usp=sharing)
|
|
58
59
|
|
|
59
60
|
> *memFrame brings a pandas-like DataFrame API to DuckDB, PostgreSQL, and ClickHouse — async-first, with an optional AI agent for natural-language data work.*
|
|
60
61
|
|
|
61
62
|
## Features
|
|
62
63
|
|
|
63
64
|
- Database-backed DataFrame API across DuckDB, PostgreSQL, and ClickHouse.
|
|
65
|
+
- Compiles every pandas-style call to backend-native SQL (DuckDB / PostgreSQL / ClickHouse) and runs it in-engine — your data never leaves the database.
|
|
64
66
|
- Async-first surface with sync equivalents for every operation.
|
|
65
67
|
- Upload from CSV, Parquet, or pandas DataFrame.
|
|
66
68
|
- Inspection, selection, cleaning, statistics, arithmetic, Plotly charts.
|
|
@@ -10,12 +10,14 @@
|
|
|
10
10
|
[](https://github.com/Debojit95/memFrame/actions/workflows/tox.yml)
|
|
11
11
|
[](https://github.com/Debojit95/memFrame/blob/main/LICENSE)
|
|
12
12
|
[](https://pypi.org/project/memframe/)
|
|
13
|
+
[](https://colab.research.google.com/drive/1UtoPjzTmia4Cr6y2btDmV8mINgXaBPGP?usp=sharing)
|
|
13
14
|
|
|
14
15
|
> *memFrame brings a pandas-like DataFrame API to DuckDB, PostgreSQL, and ClickHouse — async-first, with an optional AI agent for natural-language data work.*
|
|
15
16
|
|
|
16
17
|
## Features
|
|
17
18
|
|
|
18
19
|
- Database-backed DataFrame API across DuckDB, PostgreSQL, and ClickHouse.
|
|
20
|
+
- Compiles every pandas-style call to backend-native SQL (DuckDB / PostgreSQL / ClickHouse) and runs it in-engine — your data never leaves the database.
|
|
19
21
|
- Async-first surface with sync equivalents for every operation.
|
|
20
22
|
- Upload from CSV, Parquet, or pandas DataFrame.
|
|
21
23
|
- Inspection, selection, cleaning, statistics, arithmetic, Plotly charts.
|
|
@@ -46,10 +46,6 @@ Every cleaning operation has synchronous and asynchronous forms:
|
|
|
46
46
|
| `isna()` | `await aisna()` | Boolean null mask |
|
|
47
47
|
| `notna()` | `await anotna()` | Boolean non-null mask |
|
|
48
48
|
| `drop_duplicates(subset=None, keep="first")` | `await adrop_duplicates(...)` | Remove duplicate rows |
|
|
49
|
-
| `data_quality_missing_values(columns)` | `await adata_quality_missing_values(...)` | Missing-value counts |
|
|
50
|
-
| `data_quality_completeness_score(columns)` | `await adata_quality_completeness_score(...)` | Completeness percentages |
|
|
51
|
-
| `comprehensive_numeric_summary(columns)` | `await acomprehensive_numeric_summary(...)` | Numeric summary report |
|
|
52
|
-
| `statistical_profile_report(columns)` | `await astatistical_profile_report(...)` | Combined profile report |
|
|
53
49
|
|
|
54
50
|
Public methods return the operation value directly: usually a DataFrame, report
|
|
55
51
|
dictionary, mask, or `None`. Invalid operations raise `OperationError`.
|
|
@@ -403,64 +399,6 @@ Parameters:
|
|
|
403
399
|
| `subset` | `list[str]` or `None` | Columns used to identify duplicates. Defaults to all columns. |
|
|
404
400
|
| `keep` | `"first"`, `"last"`, or `False` | Which duplicate row to keep. `False` keeps only rows with no duplicates. |
|
|
405
401
|
|
|
406
|
-
## Data Quality Reports
|
|
407
|
-
|
|
408
|
-
### `data_quality_missing_values`
|
|
409
|
-
|
|
410
|
-
`data_quality_missing_values` returns per-column `total`, `non_null`,
|
|
411
|
-
`missing`, and `missing_pct` values.
|
|
412
|
-
|
|
413
|
-
```python
|
|
414
|
-
result = dataset.data_quality_missing_values(
|
|
415
|
-
columns=["salary", "department"],
|
|
416
|
-
)
|
|
417
|
-
```
|
|
418
|
-
|
|
419
|
-
```python
|
|
420
|
-
result = await dataset.adata_quality_missing_values(
|
|
421
|
-
columns=["salary", "department"],
|
|
422
|
-
)
|
|
423
|
-
```
|
|
424
|
-
|
|
425
|
-
### `data_quality_completeness_score`
|
|
426
|
-
|
|
427
|
-
`data_quality_completeness_score` returns completeness percentages for each
|
|
428
|
-
requested column.
|
|
429
|
-
|
|
430
|
-
```python
|
|
431
|
-
result = dataset.data_quality_completeness_score(columns=["salary"])
|
|
432
|
-
```
|
|
433
|
-
|
|
434
|
-
```python
|
|
435
|
-
result = await dataset.adata_quality_completeness_score(columns=["salary"])
|
|
436
|
-
```
|
|
437
|
-
|
|
438
|
-
### `comprehensive_numeric_summary`
|
|
439
|
-
|
|
440
|
-
`comprehensive_numeric_summary` generates numeric summaries for up to the first
|
|
441
|
-
20 requested columns.
|
|
442
|
-
|
|
443
|
-
```python
|
|
444
|
-
result = dataset.comprehensive_numeric_summary(columns=["salary", "bonus"])
|
|
445
|
-
```
|
|
446
|
-
|
|
447
|
-
```python
|
|
448
|
-
result = await dataset.acomprehensive_numeric_summary(columns=["salary", "bonus"])
|
|
449
|
-
```
|
|
450
|
-
|
|
451
|
-
### `statistical_profile_report`
|
|
452
|
-
|
|
453
|
-
`statistical_profile_report` combines completeness scoring and numeric summary
|
|
454
|
-
output into one response.
|
|
455
|
-
|
|
456
|
-
```python
|
|
457
|
-
result = dataset.statistical_profile_report(columns=["salary", "bonus"])
|
|
458
|
-
```
|
|
459
|
-
|
|
460
|
-
```python
|
|
461
|
-
result = await dataset.astatistical_profile_report(columns=["salary", "bonus"])
|
|
462
|
-
```
|
|
463
|
-
|
|
464
402
|
## Return Values and Errors
|
|
465
403
|
|
|
466
404
|
Public cleaning methods return the underlying DataFrame, dictionary, mask, or
|
|
@@ -545,11 +483,3 @@ Cleaning methods raise `OperationError` for invalid input or backend failures.
|
|
|
545
483
|
- notna
|
|
546
484
|
- adrop_duplicates
|
|
547
485
|
- drop_duplicates
|
|
548
|
-
- adata_quality_missing_values
|
|
549
|
-
- data_quality_missing_values
|
|
550
|
-
- adata_quality_completeness_score
|
|
551
|
-
- data_quality_completeness_score
|
|
552
|
-
- acomprehensive_numeric_summary
|
|
553
|
-
- comprehensive_numeric_summary
|
|
554
|
-
- astatistical_profile_report
|
|
555
|
-
- statistical_profile_report
|
|
@@ -38,24 +38,21 @@ Every inspect operation has synchronous and asynchronous forms:
|
|
|
38
38
|
| `info()` | `await ainfo()` | Per-column table information |
|
|
39
39
|
| `describe(columns=None)` | `await adescribe(...)` | Numeric descriptive statistics |
|
|
40
40
|
| `null_analysis(columns=None)` | `await anull_analysis(...)` | Null distribution by column |
|
|
41
|
-
| `
|
|
41
|
+
| `data_quality_missing_values(columns)` | `await adata_quality_missing_values(...)` | Missing-value counts |
|
|
42
|
+
| `data_quality_completeness_score(columns)` | `await adata_quality_completeness_score(...)` | Completeness percentages |
|
|
43
|
+
| `comprehensive_numeric_summary(columns)` | `await acomprehensive_numeric_summary(...)` | Numeric summary report |
|
|
44
|
+
| `statistical_profile_report(columns)` | `await astatistical_profile_report(...)` | Combined profile report |
|
|
42
45
|
| `full_table(columns=None, chunk_size=None)` | `await afull_table(...)` | Full table or chunk iterator |
|
|
43
46
|
| `astype(columns=None, dtypes=None, dtype_map=None)` | `await aastype(...)` | Cast selected columns |
|
|
44
47
|
| `insert(column, value)` | `await ainsert(...)` | Add a column from a value list |
|
|
45
48
|
| `map(func, na_action=None, columns=None, datetime_action="skip")` | `await amap(...)` | Apply SQL expression to values |
|
|
46
49
|
| `rename(columns)` | `await arename(...)` | Rename columns |
|
|
47
50
|
| `set_index(columns)` | `await aset_index(...)` | Add a primary-key index |
|
|
48
|
-
| `reset_index()` | `await areset_index()` | Recreate an `id` index column |
|
|
49
51
|
| `update(on, other_table, other_schema="upload", overwrite=True, errors="ignore")` | `await aupdate(...)` | Update from another table |
|
|
50
52
|
| `resample(time_column, rule, agg="COUNT", value_column=None, label="left", closed="left")` | `await aresample(...)` | Time-series aggregation |
|
|
51
|
-
| `axes()` | `await aaxes()` | Row and column axes |
|
|
52
53
|
| `columns()` | `await acolumns()` | Column labels |
|
|
53
54
|
| `dtypes()` | `await adtypes()` | Column database types |
|
|
54
|
-
| `first_valid_index()` | `await afirst_valid_index()` | First row with non-null values |
|
|
55
|
-
| `memory_usage()` | `await amemory_usage()` | Backend table memory usage |
|
|
56
|
-
| `ndim()` | `await andim()` | Number of dimensions |
|
|
57
55
|
| `shape()` | `await ashape()` | Row and column count |
|
|
58
|
-
| `size()` | `await asize()` | Total element count |
|
|
59
56
|
| `values()` | `await avalues()` | Table values as nested lists |
|
|
60
57
|
| `items()` | `await aitems()` | Column/value iterator result |
|
|
61
58
|
| `iterrows()` | `await aiterrows()` | Row iterator result |
|
|
@@ -241,27 +238,64 @@ Parameters:
|
|
|
241
238
|
The result DataFrame is indexed by column name when data is available and
|
|
242
239
|
contains `contains_null` and `percent_missing` columns.
|
|
243
240
|
|
|
244
|
-
### `corr`
|
|
245
241
|
|
|
246
|
-
|
|
242
|
+
## Data Quality Reports
|
|
243
|
+
|
|
244
|
+
### `data_quality_missing_values`
|
|
245
|
+
|
|
246
|
+
`data_quality_missing_values` returns per-column `total`, `non_null`,
|
|
247
|
+
`missing`, and `missing_pct` values.
|
|
247
248
|
|
|
248
249
|
```python
|
|
249
|
-
result = dataset.
|
|
250
|
+
result = dataset.data_quality_missing_values(
|
|
251
|
+
columns=["salary", "department"],
|
|
252
|
+
)
|
|
250
253
|
```
|
|
251
254
|
|
|
252
255
|
```python
|
|
253
|
-
result = await dataset.
|
|
256
|
+
result = await dataset.adata_quality_missing_values(
|
|
257
|
+
columns=["salary", "department"],
|
|
258
|
+
)
|
|
254
259
|
```
|
|
255
260
|
|
|
256
|
-
|
|
261
|
+
### `data_quality_completeness_score`
|
|
257
262
|
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
263
|
+
`data_quality_completeness_score` returns completeness percentages for each
|
|
264
|
+
requested column.
|
|
265
|
+
|
|
266
|
+
```python
|
|
267
|
+
result = dataset.data_quality_completeness_score(columns=["salary"])
|
|
268
|
+
```
|
|
269
|
+
|
|
270
|
+
```python
|
|
271
|
+
result = await dataset.adata_quality_completeness_score(columns=["salary"])
|
|
272
|
+
```
|
|
273
|
+
|
|
274
|
+
### `comprehensive_numeric_summary`
|
|
262
275
|
|
|
263
|
-
|
|
264
|
-
|
|
276
|
+
`comprehensive_numeric_summary` generates numeric summaries for up to the first
|
|
277
|
+
20 requested columns.
|
|
278
|
+
|
|
279
|
+
```python
|
|
280
|
+
result = dataset.comprehensive_numeric_summary(columns=["salary", "bonus"])
|
|
281
|
+
```
|
|
282
|
+
|
|
283
|
+
```python
|
|
284
|
+
result = await dataset.acomprehensive_numeric_summary(columns=["salary", "bonus"])
|
|
285
|
+
```
|
|
286
|
+
|
|
287
|
+
### `statistical_profile_report`
|
|
288
|
+
|
|
289
|
+
`statistical_profile_report` combines completeness scoring and numeric summary
|
|
290
|
+
output into one response.
|
|
291
|
+
|
|
292
|
+
```python
|
|
293
|
+
result = dataset.statistical_profile_report(columns=["salary", "bonus"])
|
|
294
|
+
```
|
|
295
|
+
|
|
296
|
+
```python
|
|
297
|
+
result = await dataset.astatistical_profile_report(columns=["salary", "bonus"])
|
|
298
|
+
```
|
|
265
299
|
|
|
266
300
|
## Table Utilities
|
|
267
301
|
|
|
@@ -388,19 +422,14 @@ Parameters:
|
|
|
388
422
|
| --- | --- | --- |
|
|
389
423
|
| `columns` | `dict[str, str]` | Mapping of old column names to new names. |
|
|
390
424
|
|
|
391
|
-
### `set_index`
|
|
425
|
+
### `set_index`
|
|
392
426
|
|
|
393
|
-
`set_index` adds a primary-key constraint over selected columns.
|
|
394
|
-
recreates an `id` index column.
|
|
427
|
+
`set_index` adds a primary-key constraint over selected columns.
|
|
395
428
|
|
|
396
429
|
```python
|
|
397
430
|
result = dataset.set_index(columns=["customer_id"])
|
|
398
431
|
```
|
|
399
432
|
|
|
400
|
-
```python
|
|
401
|
-
result = await dataset.areset_index()
|
|
402
|
-
```
|
|
403
|
-
|
|
404
433
|
`set_index` parameters:
|
|
405
434
|
|
|
406
435
|
| Parameter | Type | Description |
|
|
@@ -476,14 +505,9 @@ These methods return compact dictionary or list payloads directly:
|
|
|
476
505
|
|
|
477
506
|
| Method | Async | Public result |
|
|
478
507
|
| --- | --- | --- |
|
|
479
|
-
| `axes()` | `aaxes()` | `[row_index, columns]` |
|
|
480
508
|
| `columns()` | `acolumns()` | `[...]` |
|
|
481
509
|
| `dtypes()` | `adtypes()` | `{column: db_type}` |
|
|
482
|
-
| `first_valid_index()` | `afirst_valid_index()` | `{"first_valid_index": 0}` or `None` |
|
|
483
|
-
| `memory_usage()` | `amemory_usage()` | `{"memory_bytes": value}` |
|
|
484
|
-
| `ndim()` | `andim()` | `{"ndim": 2}` |
|
|
485
510
|
| `shape()` | `ashape()` | `{"shape": (rows, columns)}` |
|
|
486
|
-
| `size()` | `asize()` | `{"size": rows * columns}` |
|
|
487
511
|
| `values()` | `avalues()` | `{"values": [[...], ...]}` |
|
|
488
512
|
|
|
489
513
|
```python
|
|
@@ -536,7 +560,7 @@ Failed operations raise `OperationError`.
|
|
|
536
560
|
|
|
537
561
|
Some inspect methods create generated summary tables internally when method-call
|
|
538
562
|
logging receives backend context. This is most visible for `info`, `describe`,
|
|
539
|
-
`null_analysis`, and `
|
|
563
|
+
`null_analysis`, and `sample`.
|
|
540
564
|
|
|
541
565
|
Preview methods such as `head`, `tail`, `sample`, and unchunked `full_table`
|
|
542
566
|
are read-oriented and return a DataFrame sample directly.
|
|
@@ -548,10 +572,9 @@ Inspect supports DuckDB and PostgreSQL adapters:
|
|
|
548
572
|
- Both backends use quoted identifiers and schema-aware table names.
|
|
549
573
|
- Schema discovery is delegated to the active database adapter.
|
|
550
574
|
- `describe` uses backend-specific percentile functions.
|
|
551
|
-
- `corr` uses SQL `CORR` over numeric columns.
|
|
552
575
|
- `sample` uses backend `RANDOM()` ordering.
|
|
553
|
-
- PostgreSQL can report relation memory usage
|
|
554
|
-
for `memory_usage`.
|
|
576
|
+
- PostgreSQL can report relation memory usage through `info`; DuckDB currently
|
|
577
|
+
returns `None` for `memory_usage`.
|
|
555
578
|
- Column and table identifiers are sanitized before SQL is generated.
|
|
556
579
|
|
|
557
580
|
## Errors
|
|
@@ -560,14 +583,13 @@ Inspect methods raise `OperationError` for invalid input or backend failures.
|
|
|
560
583
|
|
|
561
584
|
- Invalid `chunk_size` in `full_table` raises `OperationError`.
|
|
562
585
|
- `describe` raises an error when no numeric columns are available.
|
|
563
|
-
- `corr` raises an error when fewer than two numeric columns are available.
|
|
564
586
|
- `astype` raises an error for missing columns or unsupported dtype aliases.
|
|
565
587
|
- `astype` requires either `dtype_map` or matching `columns` and `dtypes`.
|
|
566
588
|
- `insert` raises an error when `value` is not a list or its length does not
|
|
567
589
|
match the row count.
|
|
568
590
|
- `map` raises an error when `func` is not a SQL expression string or when no
|
|
569
591
|
selected columns are compatible with the expression.
|
|
570
|
-
- `rename`, `set_index`, `
|
|
592
|
+
- `rename`, `set_index`, `update`, and `resample` can raise
|
|
571
593
|
backend SQL errors when identifiers or constraints are invalid.
|
|
572
594
|
|
|
573
595
|
## API Reference
|
|
@@ -589,8 +611,6 @@ Inspect methods raise `OperationError` for invalid input or backend failures.
|
|
|
589
611
|
- describe
|
|
590
612
|
- anull_analysis
|
|
591
613
|
- null_analysis
|
|
592
|
-
- acorr
|
|
593
|
-
- corr
|
|
594
614
|
- afull_table
|
|
595
615
|
- full_table
|
|
596
616
|
- aastype
|
|
@@ -603,28 +623,16 @@ Inspect methods raise `OperationError` for invalid input or backend failures.
|
|
|
603
623
|
- rename
|
|
604
624
|
- aset_index
|
|
605
625
|
- set_index
|
|
606
|
-
- areset_index
|
|
607
|
-
- reset_index
|
|
608
626
|
- aupdate
|
|
609
627
|
- update
|
|
610
628
|
- aresample
|
|
611
629
|
- resample
|
|
612
|
-
- aaxes
|
|
613
|
-
- axes
|
|
614
630
|
- acolumns
|
|
615
631
|
- columns
|
|
616
632
|
- adtypes
|
|
617
633
|
- dtypes
|
|
618
|
-
- afirst_valid_index
|
|
619
|
-
- first_valid_index
|
|
620
|
-
- amemory_usage
|
|
621
|
-
- memory_usage
|
|
622
|
-
- andim
|
|
623
|
-
- ndim
|
|
624
634
|
- ashape
|
|
625
635
|
- shape
|
|
626
|
-
- asize
|
|
627
|
-
- size
|
|
628
636
|
- avalues
|
|
629
637
|
- values
|
|
630
638
|
- aitems
|