oh-my-opencode 4.19.3 → 5.0.0-beta.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/command/get-unpublished-changes.md +2 -0
- package/.agents/command/omomomo.md +1 -1
- package/.agents/command/publish.md +7 -0
- package/.agents/skills/get-unpublished-changes/SKILL.md +2 -0
- package/.agents/skills/hyperplan/SKILL.md +3 -3
- package/.agents/skills/omomomo/SKILL.md +1 -1
- package/.agents/skills/publish/SKILL.md +7 -0
- package/.opencode/command/get-unpublished-changes.md +2 -0
- package/.opencode/command/omomomo.md +1 -1
- package/.opencode/command/publish.md +7 -0
- package/.opencode/skills/hyperplan/SKILL.md +3 -3
- package/README.ja.md +1 -1
- package/README.ko.md +1 -1
- package/README.md +6 -5
- package/README.ru.md +1 -1
- package/README.zh-cn.md +1 -1
- package/bin/oh-my-opencode.js +14 -1
- package/bin/oh-my-opencode.test.ts +21 -0
- package/dist/agents/sisyphus-junior/agent.d.ts +1 -1
- package/dist/cli/doctor/checks/deprecated-reasoning-keys.d.ts +2 -0
- package/dist/cli/index.js +2925 -1919
- package/dist/cli-node/index.js +2925 -1919
- package/dist/config/schema/agent-overrides.d.ts +1343 -15
- package/dist/config/schema/categories.d.ts +132 -0
- package/dist/config/schema/fallback-models.d.ts +50 -0
- package/dist/config/schema/oh-my-opencode-config.d.ts +1322 -11
- package/dist/config-migration/index.d.ts +1 -0
- package/dist/config-migration/migration-plans.d.ts +1 -0
- package/dist/config-migration/reasoning-unification.d.ts +3 -0
- package/dist/features/monitor/batcher.d.ts +3 -1
- package/dist/features/monitor/manager-internals.d.ts +1 -0
- package/dist/features/monitor/output-injector-types.d.ts +2 -0
- package/dist/features/monitor/output-injector.d.ts +6 -0
- package/dist/features/team-mode/tools/lifecycle-test-fixture.d.ts +2 -0
- package/dist/hooks/codegraph-bootstrap/command-runner.d.ts +1 -0
- package/dist/hooks/model-fallback/next-fallback.d.ts +1 -0
- package/dist/hooks/runtime-fallback/constants.d.ts +1 -1
- package/dist/hooks/todo-continuation-enforcer/types.d.ts +1 -0
- package/dist/hooks/todo-continuation-enforcer/unrecoverable-request-error.d.ts +9 -0
- package/dist/hooks/tool-pair-validator/hook.test-support.d.ts +29 -0
- package/dist/hooks/tool-pair-validator/tool-part-ids.d.ts +14 -5
- package/dist/hooks/tool-pair-validator/tool-result-repair.d.ts +4 -3
- package/dist/hooks/tool-pair-validator/types.d.ts +5 -22
- package/dist/index.js +4864 -4011
- package/dist/mcp/lsp.d.ts +1 -0
- package/dist/oh-my-opencode.schema.json +3221 -222
- package/dist/plugin-handlers/prometheus-agent-config-builder.d.ts +1 -0
- package/dist/shared/agent-variant.d.ts +11 -0
- package/dist/shared/session-prompt-params-helpers.d.ts +6 -1
- package/dist/shared/tmux/constants.d.ts +1 -1
- package/dist/skills/ast-grep/SOURCE +1 -1
- package/dist/skills/ast-grep/install.ps1 +2 -2
- package/dist/skills/ast-grep/install.sh +1 -1
- package/dist/skills/ast-grep/references/install.md +2 -2
- package/dist/skills/ast-grep/tests/smoke.sh +1 -1
- package/dist/skills/coding-agent-sessions/SKILL.md +4 -3
- package/dist/skills/coding-agent-sessions/references/all-platforms.md +3 -1
- package/dist/skills/coding-agent-sessions/scripts/agent_sessions/aside_scanner.py +140 -0
- package/dist/skills/coding-agent-sessions/scripts/agent_sessions/scanners.py +3 -0
- package/dist/skills/data-scientist/SKILL.md +243 -0
- package/dist/skills/data-scientist/references/common-scenarios.md +176 -0
- package/dist/skills/data-scientist/references/execution-templates.md +197 -0
- package/dist/skills/data-scientist/references/integration-patterns.md +153 -0
- package/dist/skills/data-scientist/references/performance-benchmarks.md +37 -0
- package/dist/skills/data-scientist/references/uv-setup.md +78 -0
- package/dist/skills/data-scientist/scripts/quick-query.py +111 -0
- package/dist/skills/data-scientist/scripts/setup-uv.ps1 +53 -0
- package/dist/skills/data-scientist/scripts/setup-uv.sh +60 -0
- package/dist/skills/debugging/SKILL.md +1 -1
- package/dist/skills/programming/SKILL.md +1 -2
- package/dist/skills/start-work/SKILL.md +54 -9
- package/dist/skills/ultimate-browsing/SKILL.md +2 -2
- package/dist/skills/ultimate-browsing/engine/__main__.py +8 -1
- package/dist/skills/ultimate-browsing/engine/bias_check.py +11 -0
- package/dist/skills/ultimate-browsing/engine/fetch_chain.py +90 -52
- package/dist/skills/ultimate-browsing/engine/result_schema.py +10 -1
- package/dist/skills/ultimate-browsing/engine/surrogate.py +214 -0
- package/dist/skills/ultimate-browsing/engine/surrogates.yaml +60 -0
- package/dist/skills/ultimate-browsing/engine/tests/fixtures/amp_redirect_stub.html +7 -0
- package/dist/skills/ultimate-browsing/engine/tests/fixtures/search_interstitial.html +19 -0
- package/dist/skills/ultimate-browsing/engine/tests/fixtures/wayback_available.json +1 -0
- package/dist/skills/ultimate-browsing/engine/tests/fixtures/wayback_snapshot.html +1128 -0
- package/dist/skills/ultimate-browsing/engine/tests/test_surrogate.py +252 -0
- package/dist/skills/ultimate-browsing/engine/tests/test_surrogate_validators.py +78 -0
- package/dist/skills/ultimate-browsing/engine/validators.py +46 -0
- package/dist/skills/ultimate-browsing/engine/waf_detector.py +1 -1
- package/dist/skills/ultimate-browsing/engine/waf_profiles.yaml +10 -5
- package/dist/skills/ultimate-browsing/references/agent-reach/social.md +1 -1
- package/dist/skills/ultimate-browsing/references/chrome-stealth.md +13 -11
- package/dist/skills/ultimate-browsing/references/insane-search/README.md +4 -4
- package/dist/skills/ultimate-browsing/references/insane-search/cache-archive.md +51 -50
- package/dist/skills/ultimate-browsing/references/insane-search/fallback.md +1 -1
- package/dist/skills/ultimate-browsing/references/insane-search/jina.md +8 -2
- package/dist/skills/ultimate-browsing/references/insane-search/naver.md +1 -1
- package/dist/skills/ultimate-browsing/references/insane-search/twitter.md +3 -3
- package/dist/skills/ulw-plan/SKILL.md +2 -2
- package/dist/skills/ulw-plan/references/full-workflow.md +3 -3
- package/dist/skills/ulw-plan/scripts/scaffold-plan.mjs +1 -1
- package/dist/skills/ulw-research/SKILL.md +128 -12
- package/dist/tools/delegate-task/builtin-categories.d.ts +1 -0
- package/dist/tools/delegate-task/builtin-category-definition.d.ts +1 -0
- package/dist/tools/delegate-task/constants.d.ts +1 -1
- package/dist/tui.js +1271 -1108
- package/docs/reference/web-terminal-visual-qa.md +1 -1
- package/package.json +27 -19
- package/packages/lsp-core/src/lsp/client-diagnostics-freshness.integration.test.ts +2 -2
- package/packages/lsp-core/src/lsp/connection.ts +1 -1
- package/packages/lsp-daemon/dist/cli.js +27 -13
- package/packages/lsp-daemon/dist/client.js +49 -35
- package/packages/lsp-daemon/dist/ensure-daemon.d.ts +1 -0
- package/packages/lsp-daemon/dist/ensure-daemon.js +18 -5
- package/packages/lsp-daemon/dist/index.js +34 -20
- package/packages/lsp-tools-mcp/dist/cli.js +1 -1
- package/packages/lsp-tools-mcp/dist/lsp/manager.js +1 -1
- package/packages/lsp-tools-mcp/dist/mcp.js +1 -1
- package/packages/lsp-tools-mcp/dist/tools.js +1 -1
- package/packages/omo-codex/plugin/.codex-plugin/plugin.json +1 -1
- package/packages/omo-codex/plugin/components/bootstrap/dist/cli.js +289 -95
- package/packages/omo-codex/plugin/components/bootstrap/hooks/hooks.json +1 -1
- package/packages/omo-codex/plugin/components/bootstrap/package.json +1 -1
- package/packages/omo-codex/plugin/components/bootstrap/src/setup.ts +7 -7
- package/packages/omo-codex/plugin/components/bootstrap/src/worker.ts +3 -0
- package/packages/omo-codex/plugin/components/codegraph/dist/cli.js +385 -64
- package/packages/omo-codex/plugin/components/codegraph/dist/serve.js +336 -47
- package/packages/omo-codex/plugin/components/codegraph/package.json +1 -1
- package/packages/omo-codex/plugin/components/codegraph/test/hook.test.ts +6 -0
- package/packages/omo-codex/plugin/components/comment-checker/hooks/hooks.json +1 -1
- package/packages/omo-codex/plugin/components/comment-checker/package.json +1 -1
- package/packages/omo-codex/plugin/components/git-bash/hooks/hooks.json +2 -2
- package/packages/omo-codex/plugin/components/git-bash/package.json +1 -1
- package/packages/omo-codex/plugin/components/lazycodex-executor-verify/hooks/hooks.json +1 -1
- package/packages/omo-codex/plugin/components/lazycodex-executor-verify/package.json +1 -1
- package/packages/omo-codex/plugin/components/lsp/dist/.omo-runtime-manifest.json +3 -3
- package/packages/omo-codex/plugin/components/lsp/dist/cli.js +56 -42
- package/packages/omo-codex/plugin/components/lsp/hooks/hooks.json +2 -2
- package/packages/omo-codex/plugin/components/lsp/package.json +1 -1
- package/packages/omo-codex/plugin/components/rules/dist/cli.js +1 -1
- package/packages/omo-codex/plugin/components/rules/hooks/hooks.json +4 -4
- package/packages/omo-codex/plugin/components/rules/package.json +1 -1
- package/packages/omo-codex/plugin/components/rules/src/post-compact-budget.ts +1 -1
- package/packages/omo-codex/plugin/components/start-work-continuation/AGENTS.md +1 -0
- package/packages/omo-codex/plugin/components/start-work-continuation/README.md +5 -1
- package/packages/omo-codex/plugin/components/start-work-continuation/directive.md +2 -1
- package/packages/omo-codex/plugin/components/start-work-continuation/dist/cli.js +18 -0
- package/packages/omo-codex/plugin/components/start-work-continuation/hooks/hooks.json +2 -2
- package/packages/omo-codex/plugin/components/start-work-continuation/package.json +1 -1
- package/packages/omo-codex/plugin/components/start-work-continuation/src/codex-hook.ts +21 -0
- package/packages/omo-codex/plugin/components/start-work-continuation/test/codex-hook.test.ts +105 -0
- package/packages/omo-codex/plugin/components/teammode/hooks/hooks.json +1 -1
- package/packages/omo-codex/plugin/components/teammode/package.json +1 -1
- package/packages/omo-codex/plugin/components/telemetry/hooks/hooks.json +1 -1
- package/packages/omo-codex/plugin/components/telemetry/package.json +1 -1
- package/packages/omo-codex/plugin/components/ultrawork/CHANGELOG.md +2 -0
- package/packages/omo-codex/plugin/components/ultrawork/agents/lazycodex-code-reviewer.toml +1 -1
- package/packages/omo-codex/plugin/components/ultrawork/agents/lazycodex-gate-reviewer.toml +1 -1
- package/packages/omo-codex/plugin/components/ultrawork/agents/lazycodex-qa-executor.toml +1 -1
- package/packages/omo-codex/plugin/components/ultrawork/agents/lazycodex-worker-high.toml +1 -1
- package/packages/omo-codex/plugin/components/ultrawork/agents/lazycodex-worker-low.toml +1 -1
- package/packages/omo-codex/plugin/components/ultrawork/agents/lazycodex-worker-medium.toml +1 -1
- package/packages/omo-codex/plugin/components/ultrawork/agents/plan.toml +2 -2
- package/packages/omo-codex/plugin/components/ultrawork/directive.md +3 -2
- package/packages/omo-codex/plugin/components/ultrawork/hooks/hooks.json +1 -1
- package/packages/omo-codex/plugin/components/ultrawork/package.json +1 -1
- package/packages/omo-codex/plugin/components/ultrawork/skills/ultrawork/SKILL.md +3 -2
- package/packages/omo-codex/plugin/components/ultrawork/skills/ulw-plan/SKILL.md +2 -2
- package/packages/omo-codex/plugin/components/ultrawork/skills/ulw-plan/references/full-workflow.md +3 -3
- package/packages/omo-codex/plugin/components/ultrawork/skills/ulw-plan/scripts/scaffold-plan.mjs +1 -1
- package/packages/omo-codex/plugin/components/ulw-loop/AGENTS.md +1 -1
- package/packages/omo-codex/plugin/components/ulw-loop/CHANGELOG.md +2 -0
- package/packages/omo-codex/plugin/components/ulw-loop/README.md +11 -11
- package/packages/omo-codex/plugin/components/ulw-loop/directive.md +3 -2
- package/packages/omo-codex/plugin/components/ulw-loop/dist/checkpoint-reconciliation.js +1 -1
- package/packages/omo-codex/plugin/components/ulw-loop/dist/cli-output.d.ts +1 -1
- package/packages/omo-codex/plugin/components/ulw-loop/dist/cli-output.js +9 -9
- package/packages/omo-codex/plugin/components/ulw-loop/dist/cli-steering.js +1 -1
- package/packages/omo-codex/plugin/components/ulw-loop/dist/cli.js +66 -66
- package/packages/omo-codex/plugin/components/ulw-loop/dist/codex-goal-instruction.js +4 -4
- package/packages/omo-codex/plugin/components/ulw-loop/dist/codex-hook.js +1 -1
- package/packages/omo-codex/plugin/components/ulw-loop/dist/plan-crud.js +1 -1
- package/packages/omo-codex/plugin/components/ulw-loop/dist/plan-io.js +1 -1
- package/packages/omo-codex/plugin/components/ulw-loop/dist/steering.js +1 -1
- package/packages/omo-codex/plugin/components/ulw-loop/dist/stop-resume-hook.js +2 -2
- package/packages/omo-codex/plugin/components/ulw-loop/hooks/hooks.json +4 -4
- package/packages/omo-codex/plugin/components/ulw-loop/package.json +1 -1
- package/packages/omo-codex/plugin/components/ulw-loop/skills/ulw-loop/SKILL.md +3 -3
- package/packages/omo-codex/plugin/components/ulw-loop/skills/ulw-loop/references/full-workflow.md +22 -25
- package/packages/omo-codex/plugin/components/ulw-loop/src/checkpoint-reconciliation.ts +1 -1
- package/packages/omo-codex/plugin/components/ulw-loop/src/cli-output.ts +9 -9
- package/packages/omo-codex/plugin/components/ulw-loop/src/cli-steering.ts +1 -1
- package/packages/omo-codex/plugin/components/ulw-loop/src/cli.ts +1 -1
- package/packages/omo-codex/plugin/components/ulw-loop/src/codex-goal-instruction.ts +4 -4
- package/packages/omo-codex/plugin/components/ulw-loop/src/codex-hook.ts +1 -1
- package/packages/omo-codex/plugin/components/ulw-loop/src/plan-crud.ts +1 -1
- package/packages/omo-codex/plugin/components/ulw-loop/src/plan-io.ts +1 -1
- package/packages/omo-codex/plugin/components/ulw-loop/src/steering.ts +1 -1
- package/packages/omo-codex/plugin/components/ulw-loop/src/stop-resume-hook.ts +2 -2
- package/packages/omo-codex/plugin/components/ulw-loop/test/cli-commands.test.ts +2 -2
- package/packages/omo-codex/plugin/components/ulw-loop/test/cli-entrypoint.test.ts +1 -1
- package/packages/omo-codex/plugin/components/ulw-loop/test/cli-helpers.test.ts +2 -2
- package/packages/omo-codex/plugin/components/ulw-loop/test/cli-steering-kind-guidance.test.ts +1 -1
- package/packages/omo-codex/plugin/components/ulw-loop/test/codex-hook.test.ts +2 -2
- package/packages/omo-codex/plugin/components/ulw-loop/test/fixtures/quality-gate-builder.ts +1 -1
- package/packages/omo-codex/plugin/components/ulw-loop/test/package-smoke.test.ts +5 -5
- package/packages/omo-codex/plugin/components/ulw-loop/test/plan-io.test.ts +1 -1
- package/packages/omo-codex/plugin/components/ulw-loop/test/quality-gate-roles.test.ts +1 -1
- package/packages/omo-codex/plugin/components/ulw-loop/test/steering.test.ts +1 -1
- package/packages/omo-codex/plugin/components/ulw-loop/test/stop-resume-hook.test.ts +1 -1
- package/packages/omo-codex/plugin/hooks/post-compact-resetting-git-bash-mcp-reminder.json +1 -1
- package/packages/omo-codex/plugin/hooks/post-compact-resetting-lsp-diagnostics-cache.json +1 -1
- package/packages/omo-codex/plugin/hooks/post-compact-resetting-project-rule-cache.json +1 -1
- package/packages/omo-codex/plugin/hooks/post-tool-use-checking-codegraph-init-guidance.json +1 -1
- package/packages/omo-codex/plugin/hooks/post-tool-use-checking-comments.json +1 -1
- package/packages/omo-codex/plugin/hooks/post-tool-use-checking-lsp-diagnostics.json +1 -1
- package/packages/omo-codex/plugin/hooks/post-tool-use-checking-thread-title-hygiene.json +1 -1
- package/packages/omo-codex/plugin/hooks/post-tool-use-matching-project-rules.json +1 -1
- package/packages/omo-codex/plugin/hooks/pre-tool-use-enforcing-unlimited-goal-budget.json +1 -1
- package/packages/omo-codex/plugin/hooks/pre-tool-use-guarding-ulw-loop-spawns.json +1 -1
- package/packages/omo-codex/plugin/hooks/pre-tool-use-recommending-git-bash-mcp.json +1 -1
- package/packages/omo-codex/plugin/hooks/session-start-checking-auto-update.json +1 -1
- package/packages/omo-codex/plugin/hooks/session-start-checking-bootstrap-provisioning.json +1 -1
- package/packages/omo-codex/plugin/hooks/session-start-checking-codegraph-bootstrap.json +1 -1
- package/packages/omo-codex/plugin/hooks/session-start-loading-project-rules.json +1 -1
- package/packages/omo-codex/plugin/hooks/session-start-recording-session-telemetry.json +1 -1
- package/packages/omo-codex/plugin/hooks/stop-checking-start-work-continuation.json +1 -1
- package/packages/omo-codex/plugin/hooks/stop-checking-ulw-loop-resume.json +1 -1
- package/packages/omo-codex/plugin/hooks/subagent-stop-checking-start-work-continuation.json +1 -1
- package/packages/omo-codex/plugin/hooks/subagent-stop-verifying-lazycodex-executor-evidence.json +1 -1
- package/packages/omo-codex/plugin/hooks/user-prompt-submit-checking-ultrawork-trigger.json +1 -1
- package/packages/omo-codex/plugin/hooks/user-prompt-submit-checking-ulw-loop-steering.json +1 -1
- package/packages/omo-codex/plugin/hooks/user-prompt-submit-loading-project-rules.json +1 -1
- package/packages/omo-codex/plugin/package-lock.json +13 -13
- package/packages/omo-codex/plugin/package.json +1 -1
- package/packages/omo-codex/plugin/scripts/sync-skills.mjs +8 -2
- package/packages/omo-codex/plugin/skills/ast-grep/SOURCE +1 -1
- package/packages/omo-codex/plugin/skills/ast-grep/install.ps1 +2 -2
- package/packages/omo-codex/plugin/skills/ast-grep/install.sh +1 -1
- package/packages/omo-codex/plugin/skills/ast-grep/references/install.md +2 -2
- package/packages/omo-codex/plugin/skills/ast-grep/tests/smoke.sh +1 -1
- package/packages/omo-codex/plugin/skills/coding-agent-sessions/SKILL.md +4 -3
- package/packages/omo-codex/plugin/skills/coding-agent-sessions/references/all-platforms.md +3 -1
- package/packages/omo-codex/plugin/skills/coding-agent-sessions/scripts/agent_sessions/aside_scanner.py +140 -0
- package/packages/omo-codex/plugin/skills/coding-agent-sessions/scripts/agent_sessions/scanners.py +3 -0
- package/packages/omo-codex/plugin/skills/data-scientist/SKILL.md +243 -0
- package/packages/omo-codex/plugin/skills/data-scientist/agents/openai.yaml +2 -0
- package/packages/omo-codex/plugin/skills/data-scientist/references/common-scenarios.md +176 -0
- package/packages/omo-codex/plugin/skills/data-scientist/references/execution-templates.md +197 -0
- package/packages/omo-codex/plugin/skills/data-scientist/references/integration-patterns.md +153 -0
- package/packages/omo-codex/plugin/skills/data-scientist/references/performance-benchmarks.md +37 -0
- package/packages/omo-codex/plugin/skills/data-scientist/references/uv-setup.md +78 -0
- package/packages/omo-codex/plugin/skills/data-scientist/scripts/quick-query.py +111 -0
- package/packages/omo-codex/plugin/skills/data-scientist/scripts/setup-uv.ps1 +53 -0
- package/packages/omo-codex/plugin/skills/data-scientist/scripts/setup-uv.sh +60 -0
- package/packages/omo-codex/plugin/skills/debugging/SKILL.md +1 -1
- package/packages/omo-codex/plugin/skills/programming/SKILL.md +1 -2
- package/packages/omo-codex/plugin/skills/start-work/SKILL.md +54 -9
- package/packages/omo-codex/plugin/skills/ultimate-browsing/SKILL.md +2 -2
- package/packages/omo-codex/plugin/skills/ultimate-browsing/engine/__main__.py +8 -1
- package/packages/omo-codex/plugin/skills/ultimate-browsing/engine/bias_check.py +11 -0
- package/packages/omo-codex/plugin/skills/ultimate-browsing/engine/fetch_chain.py +90 -52
- package/packages/omo-codex/plugin/skills/ultimate-browsing/engine/result_schema.py +10 -1
- package/packages/omo-codex/plugin/skills/ultimate-browsing/engine/surrogate.py +214 -0
- package/packages/omo-codex/plugin/skills/ultimate-browsing/engine/surrogates.yaml +60 -0
- package/packages/omo-codex/plugin/skills/ultimate-browsing/engine/tests/fixtures/amp_redirect_stub.html +7 -0
- package/packages/omo-codex/plugin/skills/ultimate-browsing/engine/tests/fixtures/search_interstitial.html +19 -0
- package/packages/omo-codex/plugin/skills/ultimate-browsing/engine/tests/fixtures/wayback_available.json +1 -0
- package/packages/omo-codex/plugin/skills/ultimate-browsing/engine/tests/fixtures/wayback_snapshot.html +1128 -0
- package/packages/omo-codex/plugin/skills/ultimate-browsing/engine/tests/test_surrogate.py +252 -0
- package/packages/omo-codex/plugin/skills/ultimate-browsing/engine/tests/test_surrogate_validators.py +78 -0
- package/packages/omo-codex/plugin/skills/ultimate-browsing/engine/validators.py +46 -0
- package/packages/omo-codex/plugin/skills/ultimate-browsing/engine/waf_detector.py +1 -1
- package/packages/omo-codex/plugin/skills/ultimate-browsing/engine/waf_profiles.yaml +10 -5
- package/packages/omo-codex/plugin/skills/ultimate-browsing/references/agent-reach/social.md +1 -1
- package/packages/omo-codex/plugin/skills/ultimate-browsing/references/chrome-stealth.md +13 -11
- package/packages/omo-codex/plugin/skills/ultimate-browsing/references/insane-search/README.md +4 -4
- package/packages/omo-codex/plugin/skills/ultimate-browsing/references/insane-search/cache-archive.md +51 -50
- package/packages/omo-codex/plugin/skills/ultimate-browsing/references/insane-search/fallback.md +1 -1
- package/packages/omo-codex/plugin/skills/ultimate-browsing/references/insane-search/jina.md +8 -2
- package/packages/omo-codex/plugin/skills/ultimate-browsing/references/insane-search/naver.md +1 -1
- package/packages/omo-codex/plugin/skills/ultimate-browsing/references/insane-search/twitter.md +3 -3
- package/packages/omo-codex/plugin/skills/ultrawork/SKILL.md +3 -2
- package/packages/omo-codex/plugin/skills/ulw-loop/SKILL.md +3 -3
- package/packages/omo-codex/plugin/skills/ulw-loop/references/full-workflow.md +22 -25
- package/packages/omo-codex/plugin/skills/ulw-plan/SKILL.md +2 -2
- package/packages/omo-codex/plugin/skills/ulw-plan/references/full-workflow.md +3 -3
- package/packages/omo-codex/plugin/skills/ulw-plan/scripts/scaffold-plan.mjs +1 -1
- package/packages/omo-codex/plugin/skills/ulw-research/SKILL.md +127 -12
- package/packages/omo-codex/plugin/test/bootstrap-binlinks.test.mjs +12 -12
- package/packages/omo-codex/plugin/test/bootstrap-orchestration.test.mjs +36 -4
- package/packages/omo-codex/plugin/test/sync-skills-orchestration.test.mjs +1 -1
- package/packages/omo-codex/plugin/test/sync-skills-test-support.mjs +9 -2
- package/packages/omo-codex/plugin/test/sync-skills.test.mjs +12 -0
- package/packages/omo-codex/scripts/install-bin-links.test.mjs +56 -2
- package/packages/omo-codex/scripts/install-delegated-command.test.mjs +6 -6
- package/packages/omo-codex/scripts/install-dist/install-local.mjs +138 -65
- package/packages/omo-codex/scripts/install-local-entrypoint.test.mjs +4 -4
- package/packages/omo-codex/scripts/install-local.test.mjs +5 -2
- package/packages/shared-skills/index.mjs +19 -1
- package/packages/shared-skills/skills/ast-grep/SOURCE +1 -1
- package/packages/shared-skills/skills/ast-grep/install.ps1 +2 -2
- package/packages/shared-skills/skills/ast-grep/install.sh +1 -1
- package/packages/shared-skills/skills/ast-grep/references/install.md +2 -2
- package/packages/shared-skills/skills/ast-grep/tests/smoke.sh +1 -1
- package/packages/shared-skills/skills/coding-agent-sessions/SKILL.md +4 -3
- package/packages/shared-skills/skills/coding-agent-sessions/references/all-platforms.md +3 -1
- package/packages/shared-skills/skills/coding-agent-sessions/scripts/agent_sessions/aside_scanner.py +140 -0
- package/packages/shared-skills/skills/coding-agent-sessions/scripts/agent_sessions/scanners.py +3 -0
- package/packages/shared-skills/skills/data-scientist/SKILL.md +243 -0
- package/packages/shared-skills/skills/data-scientist/references/common-scenarios.md +176 -0
- package/packages/shared-skills/skills/data-scientist/references/execution-templates.md +197 -0
- package/packages/shared-skills/skills/data-scientist/references/integration-patterns.md +153 -0
- package/packages/shared-skills/skills/data-scientist/references/performance-benchmarks.md +37 -0
- package/packages/shared-skills/skills/data-scientist/references/uv-setup.md +78 -0
- package/packages/shared-skills/skills/data-scientist/scripts/quick-query.py +111 -0
- package/packages/shared-skills/skills/data-scientist/scripts/setup-uv.ps1 +53 -0
- package/packages/shared-skills/skills/data-scientist/scripts/setup-uv.sh +60 -0
- package/packages/shared-skills/skills/debugging/SKILL.md +1 -1
- package/packages/shared-skills/skills/programming/SKILL.md +1 -2
- package/packages/shared-skills/skills/start-work/SKILL.md +54 -9
- package/packages/shared-skills/skills/ultimate-browsing/SKILL.md +2 -2
- package/packages/shared-skills/skills/ultimate-browsing/engine/__main__.py +8 -1
- package/packages/shared-skills/skills/ultimate-browsing/engine/bias_check.py +11 -0
- package/packages/shared-skills/skills/ultimate-browsing/engine/fetch_chain.py +90 -52
- package/packages/shared-skills/skills/ultimate-browsing/engine/result_schema.py +10 -1
- package/packages/shared-skills/skills/ultimate-browsing/engine/surrogate.py +214 -0
- package/packages/shared-skills/skills/ultimate-browsing/engine/surrogates.yaml +60 -0
- package/packages/shared-skills/skills/ultimate-browsing/engine/tests/fixtures/amp_redirect_stub.html +7 -0
- package/packages/shared-skills/skills/ultimate-browsing/engine/tests/fixtures/search_interstitial.html +19 -0
- package/packages/shared-skills/skills/ultimate-browsing/engine/tests/fixtures/wayback_available.json +1 -0
- package/packages/shared-skills/skills/ultimate-browsing/engine/tests/fixtures/wayback_snapshot.html +1128 -0
- package/packages/shared-skills/skills/ultimate-browsing/engine/tests/test_surrogate.py +252 -0
- package/packages/shared-skills/skills/ultimate-browsing/engine/tests/test_surrogate_validators.py +78 -0
- package/packages/shared-skills/skills/ultimate-browsing/engine/validators.py +46 -0
- package/packages/shared-skills/skills/ultimate-browsing/engine/waf_detector.py +1 -1
- package/packages/shared-skills/skills/ultimate-browsing/engine/waf_profiles.yaml +10 -5
- package/packages/shared-skills/skills/ultimate-browsing/references/agent-reach/social.md +1 -1
- package/packages/shared-skills/skills/ultimate-browsing/references/chrome-stealth.md +13 -11
- package/packages/shared-skills/skills/ultimate-browsing/references/insane-search/README.md +4 -4
- package/packages/shared-skills/skills/ultimate-browsing/references/insane-search/cache-archive.md +51 -50
- package/packages/shared-skills/skills/ultimate-browsing/references/insane-search/fallback.md +1 -1
- package/packages/shared-skills/skills/ultimate-browsing/references/insane-search/jina.md +8 -2
- package/packages/shared-skills/skills/ultimate-browsing/references/insane-search/naver.md +1 -1
- package/packages/shared-skills/skills/ultimate-browsing/references/insane-search/twitter.md +3 -3
- package/packages/shared-skills/skills/ulw-plan/SKILL.md +2 -2
- package/packages/shared-skills/skills/ulw-plan/references/full-workflow.md +3 -3
- package/packages/shared-skills/skills/ulw-plan/scripts/scaffold-plan.mjs +1 -1
- package/packages/shared-skills/skills/ulw-research/SKILL.md +128 -12
- package/postinstall.mjs +6 -0
|
@@ -0,0 +1,243 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: data-scientist
|
|
3
|
+
description: "Expert data processing specialist with intelligent DuckDB/Polars selection for maximum performance. Always includes numpy, never uses pandas, runs everything through uv. Triggers: 'analyze the data', 'analyze this file', 'what is in this CSV/parquet/json', 'summarize this', 'group by', 'filter rows', 'sort by', 'join these files', 'merge datasets', 'time series trend', 'last 30 days data', 'compare yesterday and today', 'distribution/histogram', 'correlation', 'clean duplicates', 'handle missing values', 'dataset larger than RAM', 'SQL query on files', 'DataFrame operations', 'chart/plot this data', DuckDB vs Polars selection, quick data exploration CLI. NOT for plain text/code inspection, configs, or tiny inline math."
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Data Scientist: High-Performance Data Processing Expert
|
|
7
|
+
|
|
8
|
+
## Role & Expertise
|
|
9
|
+
|
|
10
|
+
Performance-obsessed data scientist with expertise in:
|
|
11
|
+
- Intelligent tool selection: DuckDB vs Polars based on operation characteristics
|
|
12
|
+
- Zero-copy data interchange via Apache Arrow
|
|
13
|
+
- Memory-efficient processing for datasets exceeding RAM
|
|
14
|
+
- SQL and DataFrame API mastery for analytical workloads
|
|
15
|
+
|
|
16
|
+
## Environment Setup
|
|
17
|
+
|
|
18
|
+
Everything runs through **uv**. If `uv` is not on PATH, set it up first — pick the path that matches the system and run it, no manual guesswork:
|
|
19
|
+
|
|
20
|
+
```bash
|
|
21
|
+
bash scripts/setup-uv.sh # macOS / Linux / WSL / Git Bash — auto-detects OS + arch, installs or updates uv to latest
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
```powershell
|
|
25
|
+
powershell -ExecutionPolicy Bypass -File scripts/setup-uv.ps1 # native Windows — installs or updates uv to latest
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
Both scripts detect the platform, install uv when missing (official installer first, Homebrew/winget as fallback), upgrade it when present (`uv self update`), put it on PATH for the current shell, and verify with `uv --version`. The full per-platform matrix, PATH notes, and CI usage live in [references/uv-setup.md](references/uv-setup.md). Verify: `uv --version`.
|
|
29
|
+
|
|
30
|
+
## Core Principles
|
|
31
|
+
|
|
32
|
+
### ABSOLUTE RULES
|
|
33
|
+
|
|
34
|
+
1. **ALWAYS include numpy** in all data processing operations (`uv run --with numpy ...`)
|
|
35
|
+
2. **NEVER use pandas** - Polars and DuckDB beat it decisively on every operation; the entire skill assumes pandas is absent
|
|
36
|
+
3. **ALWAYS use Python via `uv run`** for calculations and data processing
|
|
37
|
+
4. **Intelligent tool selection**: Choose DuckDB or Polars based on operation types, NOT arbitrarily
|
|
38
|
+
5. **Zero-copy conversions**: hand data across DuckDB and Polars through Arrow — `duckdb.sql(...).pl()`. Never call `.df()` (returns a pandas frame; crashes without pandas). Keep `pyarrow` in the package set or `.pl()` raises `ModuleNotFoundError`
|
|
39
|
+
6. **Lazy evaluation**: Prefer `scan_csv`/`scan_parquet` and `.collect()` only when needed
|
|
40
|
+
7. **Direct file queries**: Let DuckDB query files directly instead of loading to memory when possible
|
|
41
|
+
|
|
42
|
+
### Standard Package Pattern
|
|
43
|
+
|
|
44
|
+
```bash
|
|
45
|
+
# Default for data tasks (numpy + pyarrow are mandatory parts of the set)
|
|
46
|
+
uv run --with numpy --with duckdb --with polars --with pyarrow python -c "{code}"
|
|
47
|
+
|
|
48
|
+
# With visualization (RECOMMENDED for most analysis requests)
|
|
49
|
+
uv run --with numpy --with duckdb --with polars --with pyarrow --with matplotlib python -c "{code}"
|
|
50
|
+
|
|
51
|
+
# Pure Polars
|
|
52
|
+
uv run --with numpy --with polars python -c "{code}"
|
|
53
|
+
|
|
54
|
+
# Pure DuckDB (with the Arrow handoff available)
|
|
55
|
+
uv run --with numpy --with duckdb --with pyarrow python -c "{code}"
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
**When to include matplotlib:**
|
|
59
|
+
- User requests visualization: "graph", "chart", "plot", "show me"
|
|
60
|
+
- Exploratory data analysis (EDA): "analyze", "trends", "patterns"
|
|
61
|
+
- Time-series analysis: "over time", "daily", "trends"
|
|
62
|
+
- Distribution analysis: "distribution", "histogram", "statistics"
|
|
63
|
+
- Comparison tasks: "compare", visual comparison implied
|
|
64
|
+
- **Default to including matplotlib** when in doubt - overhead is minimal
|
|
65
|
+
|
|
66
|
+
## Tool Selection Logic
|
|
67
|
+
|
|
68
|
+
### Decision Tree (Apply in Order)
|
|
69
|
+
|
|
70
|
+
1. **Is it a `.duckdb` file?** → **USE DUCKDB** (native format, optimal performance)
|
|
71
|
+
2. **Simple one-off query without needing full data in memory?** → **USE DUCKDB** (direct file query, zero memory load)
|
|
72
|
+
3. **Very heavy complex SQL query (multi-table joins, window functions)?** → **USE DUCKDB** (superior SQL optimizer)
|
|
73
|
+
4. **Main operation is FILTERING?** → **USE POLARS** (typically the fastest by a wide margin — see benchmarks)
|
|
74
|
+
5. **Main operation is SORTING?** → **USE POLARS** (typically the fastest)
|
|
75
|
+
6. **Complex SQL JOINS needed?** → **USE DUCKDB** (stronger join engine, more join types)
|
|
76
|
+
7. **Heavy GROUP BY AGGREGATIONS?** → **USE DUCKDB** (typically faster on large datasets)
|
|
77
|
+
8. **Window functions with partitioning?** → **POLARS** (typically faster)
|
|
78
|
+
9. **Complex TRANSFORMATIONS (pivot, melt, string ops)?** → **USE POLARS**
|
|
79
|
+
10. **Dataset larger than available RAM?** → **USE POLARS** (streaming support) or **DUCKDB** (out-of-core)
|
|
80
|
+
11. **Mixed operations?** → **USE HYBRID APPROACH** (leverage strengths of both)
|
|
81
|
+
|
|
82
|
+
### Quick Reference
|
|
83
|
+
|
|
84
|
+
```
|
|
85
|
+
Simple query → DuckDB
|
|
86
|
+
Heavy complex query → DuckDB
|
|
87
|
+
Filter → Polars
|
|
88
|
+
Sort → Polars
|
|
89
|
+
Join → DuckDB
|
|
90
|
+
Aggregate → DuckDB
|
|
91
|
+
Window → Polars
|
|
92
|
+
Transform → Polars
|
|
93
|
+
Too large for RAM → Polars streaming
|
|
94
|
+
Mixed operations → Hybrid
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
The exact multipliers these heuristics distill (with sources and caveats — routing heuristics, not guarantees) live in [performance-benchmarks.md](references/performance-benchmarks.md).
|
|
98
|
+
|
|
99
|
+
## Essential Patterns
|
|
100
|
+
|
|
101
|
+
### DuckDB Direct File Query
|
|
102
|
+
|
|
103
|
+
```python
|
|
104
|
+
import duckdb
|
|
105
|
+
# Query file directly - no memory load
|
|
106
|
+
result = duckdb.sql("""
|
|
107
|
+
SELECT category, SUM(amount) as total
|
|
108
|
+
FROM 'data.csv'
|
|
109
|
+
GROUP BY category
|
|
110
|
+
""").pl() # .pl() -> Polars via Arrow. Requires pyarrow. Never .df() (pandas).
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
### Polars Lazy Evaluation
|
|
114
|
+
|
|
115
|
+
```python
|
|
116
|
+
import polars as pl
|
|
117
|
+
# Lazy scan - optimizes and executes once
|
|
118
|
+
result = (
|
|
119
|
+
pl.scan_csv('data.csv')
|
|
120
|
+
.filter(pl.col('value') > 100)
|
|
121
|
+
.sort('value', descending=True)
|
|
122
|
+
.collect()
|
|
123
|
+
)
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
### Zero-Copy DuckDB → Polars
|
|
127
|
+
|
|
128
|
+
```python
|
|
129
|
+
import duckdb
|
|
130
|
+
# Direct conversion via Arrow (pyarrow required in the package set)
|
|
131
|
+
df_polars = duckdb.sql("SELECT * FROM 'data.csv'").pl()
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
### Hybrid Approach
|
|
135
|
+
|
|
136
|
+
```python
|
|
137
|
+
import duckdb
|
|
138
|
+
import polars as pl
|
|
139
|
+
|
|
140
|
+
# Phase 1: DuckDB for joins
|
|
141
|
+
joined = duckdb.sql(
|
|
142
|
+
"SELECT * FROM 'orders.csv' o "
|
|
143
|
+
"JOIN 'customers.csv' c ON o.customer_id = c.customer_id"
|
|
144
|
+
).pl()
|
|
145
|
+
|
|
146
|
+
# Phase 2: Polars for filtering
|
|
147
|
+
filtered = joined.filter(pl.col('amount') > 100)
|
|
148
|
+
|
|
149
|
+
# Phase 3: Back to DuckDB for aggregation
|
|
150
|
+
duckdb.register('filtered_data', filtered)
|
|
151
|
+
final = duckdb.sql('SELECT category, SUM(amount) FROM filtered_data GROUP BY category').pl()
|
|
152
|
+
```
|
|
153
|
+
|
|
154
|
+
## Quick Query CLI
|
|
155
|
+
|
|
156
|
+
For ad-hoc data exploration, use the built-in query runner:
|
|
157
|
+
|
|
158
|
+
```bash
|
|
159
|
+
# SQL query (uses DuckDB)
|
|
160
|
+
uv run scripts/quick-query.py data.csv "SELECT category, COUNT(*) FROM data GROUP BY category"
|
|
161
|
+
|
|
162
|
+
# Filter expression — Polars SQL syntax, e.g. "amount > 100" (NOT Python: never passes through eval)
|
|
163
|
+
uv run scripts/quick-query.py data.csv --filter "amount > 100"
|
|
164
|
+
|
|
165
|
+
# Auto-describe (schema + stats)
|
|
166
|
+
uv run scripts/quick-query.py data.parquet --describe
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
Supports CSV, Parquet, JSON, NDJSON. Cross-platform (macOS, Linux, Windows). Excel files are not read directly — export to CSV or Parquet first.
|
|
170
|
+
|
|
171
|
+
## Reference Documentation
|
|
172
|
+
|
|
173
|
+
For detailed guidance, consult these reference files:
|
|
174
|
+
|
|
175
|
+
- **Environment setup per platform**: See [uv-setup.md](references/uv-setup.md) — install/update uv on macOS, Linux, Windows, WSL, CI; PATH fixes; `scripts/setup-uv.sh` / `scripts/setup-uv.ps1` automate it.
|
|
176
|
+
- **Performance benchmarks and operation detection**: See [performance-benchmarks.md](references/performance-benchmarks.md)
|
|
177
|
+
- **Integration patterns and best practices**: See [integration-patterns.md](references/integration-patterns.md)
|
|
178
|
+
- **Execution templates**: See [execution-templates.md](references/execution-templates.md)
|
|
179
|
+
- **Common scenarios**: See [common-scenarios.md](references/common-scenarios.md)
|
|
180
|
+
|
|
181
|
+
## Quality Assurance Process
|
|
182
|
+
|
|
183
|
+
### Before Execution
|
|
184
|
+
1. **Analyze request** → Detect operation types (filter, join, aggregate, etc.)
|
|
185
|
+
2. **Select optimal tool** → Apply decision tree based on detected operations
|
|
186
|
+
3. **Verify approach** → Confirm tool selection matches the benchmark heuristics
|
|
187
|
+
4. **Check package list** → Ensure numpy AND pyarrow are included
|
|
188
|
+
|
|
189
|
+
### During Execution
|
|
190
|
+
1. **Use lazy evaluation** when possible (Polars `scan_*`, DuckDB direct queries)
|
|
191
|
+
2. **Monitor for errors** and have fallback strategy ready
|
|
192
|
+
3. **Provide progress updates** for long operations
|
|
193
|
+
|
|
194
|
+
### After Execution
|
|
195
|
+
1. **Report performance** → Show processing time and row counts
|
|
196
|
+
2. **Validate results** → Confirm output matches expectations
|
|
197
|
+
3. **Document tool choice** → Explain why specific tool was selected
|
|
198
|
+
|
|
199
|
+
## Activation Context
|
|
200
|
+
|
|
201
|
+
**Automatic activation triggers:**
|
|
202
|
+
|
|
203
|
+
### Exploratory Questions
|
|
204
|
+
- "Analyze the data" / "What's in the data" / "What's in this file"
|
|
205
|
+
- "Show me the data" / "Take a look at this file" / "Check the file contents"
|
|
206
|
+
|
|
207
|
+
### Temporal/Historical Analysis
|
|
208
|
+
- "What happened in the past N days?" / "How's last week's data?"
|
|
209
|
+
- "What's the trend for the last 30 days?" / "Compare yesterday and today"
|
|
210
|
+
|
|
211
|
+
### Aggregation/Summary Requests
|
|
212
|
+
- "Summarize this" / "What's the total?" / "What's the average?"
|
|
213
|
+
- "Show by category" / "Show statistics" / "How many?"
|
|
214
|
+
|
|
215
|
+
### Filtering/Search Patterns
|
|
216
|
+
- "Show only above 100" / "Find specific conditions" / "Top 10"
|
|
217
|
+
|
|
218
|
+
### Comparison/Correlation
|
|
219
|
+
- "Compare A and B" / "What's the difference?" / "Is there a correlation?" / "Merge two files"
|
|
220
|
+
|
|
221
|
+
### Transformation/Cleaning
|
|
222
|
+
- "Clean this up" / "Remove duplicates" / "Handle missing values" / "Convert format"
|
|
223
|
+
|
|
224
|
+
### Technical Patterns
|
|
225
|
+
- Working with CSV, Parquet, JSON, NDJSON, or `.duckdb` files
|
|
226
|
+
- File paths ending in `.csv`, `.parquet`, `.json`, `.jsonl`, `.ndjson`, `.tsv`, `.duckdb`
|
|
227
|
+
- Requests involving calculations or aggregations
|
|
228
|
+
- Joining, filtering, sorting, or transforming datasets
|
|
229
|
+
- Processing large datasets that may exceed memory
|
|
230
|
+
- Comparing or analyzing data from multiple sources
|
|
231
|
+
- Performance-critical data operations
|
|
232
|
+
- SQL queries or DataFrame operations mentioned
|
|
233
|
+
|
|
234
|
+
### When NOT to Activate
|
|
235
|
+
- Simple file reading for text/code inspection (use the harness's file-read surface)
|
|
236
|
+
- Non-data files (images, videos, binaries)
|
|
237
|
+
- Configuration files (YAML, TOML, JSON configs) unless specifically for data analysis
|
|
238
|
+
- Small inline calculations (run them directly)
|
|
239
|
+
- Excel files — convert to CSV/Parquet first
|
|
240
|
+
|
|
241
|
+
---
|
|
242
|
+
|
|
243
|
+
**Core execution principle:** Always apply intelligent tool selection based on operation characteristics, never use pandas, and always include numpy and pyarrow in the execution environment.
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
# Common Scenarios with Tool Selection
|
|
2
|
+
|
|
3
|
+
## Scenario 1: Simple Calculation
|
|
4
|
+
|
|
5
|
+
**Decision: Python (always)**
|
|
6
|
+
|
|
7
|
+
```python
|
|
8
|
+
uv run --with numpy python -c "
|
|
9
|
+
import numpy as np
|
|
10
|
+
result = np.sum([1, 2, 3, 4, 5])
|
|
11
|
+
print(f'Result: {result}')
|
|
12
|
+
"
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
## Scenario 2: CSV Quick Analysis
|
|
16
|
+
|
|
17
|
+
**Decision: DuckDB (direct query, no memory load)**
|
|
18
|
+
|
|
19
|
+
```python
|
|
20
|
+
uv run --with numpy --with duckdb python -c "
|
|
21
|
+
import duckdb
|
|
22
|
+
result = duckdb.sql('''
|
|
23
|
+
SELECT * FROM 'data.csv'
|
|
24
|
+
LIMIT 10
|
|
25
|
+
''').pl()
|
|
26
|
+
print(result)
|
|
27
|
+
"
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
## Scenario 3: Filter + Sort on Large Dataset
|
|
31
|
+
|
|
32
|
+
**Decision: Polars (128x faster filtering, 12x faster sorting)**
|
|
33
|
+
|
|
34
|
+
```python
|
|
35
|
+
uv run --with numpy --with polars python -c "
|
|
36
|
+
import polars as pl
|
|
37
|
+
result = (
|
|
38
|
+
pl.scan_csv('large.csv')
|
|
39
|
+
.filter(pl.col('value') > 1000)
|
|
40
|
+
.sort('value', descending=True)
|
|
41
|
+
.head(100)
|
|
42
|
+
.collect()
|
|
43
|
+
)
|
|
44
|
+
print(result)
|
|
45
|
+
"
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
## Scenario 4: Multi-Table Join + Aggregation
|
|
49
|
+
|
|
50
|
+
**Decision: DuckDB (best for joins and aggregations)**
|
|
51
|
+
|
|
52
|
+
```python
|
|
53
|
+
uv run --with numpy --with duckdb python -c "
|
|
54
|
+
import duckdb
|
|
55
|
+
result = duckdb.sql('''
|
|
56
|
+
SELECT
|
|
57
|
+
a.category,
|
|
58
|
+
COUNT(*) as count,
|
|
59
|
+
SUM(b.amount) as total
|
|
60
|
+
FROM 'table1.csv' a
|
|
61
|
+
JOIN 'table2.csv' b ON a.id = b.id
|
|
62
|
+
GROUP BY a.category
|
|
63
|
+
ORDER BY total DESC
|
|
64
|
+
''').pl()
|
|
65
|
+
print(result)
|
|
66
|
+
"
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
## Scenario 5: Data Exploration
|
|
70
|
+
|
|
71
|
+
**Decision: DuckDB for quick exploration**
|
|
72
|
+
|
|
73
|
+
```python
|
|
74
|
+
uv run --with numpy --with duckdb python -c "
|
|
75
|
+
import duckdb
|
|
76
|
+
|
|
77
|
+
# Show first few rows
|
|
78
|
+
print('**Sample Data**')
|
|
79
|
+
print(duckdb.sql('SELECT * FROM \"data.csv\" LIMIT 5').pl())
|
|
80
|
+
|
|
81
|
+
# Show summary statistics
|
|
82
|
+
print('\\n**Summary Statistics**')
|
|
83
|
+
print(duckdb.sql('DESCRIBE SELECT * FROM \"data.csv\"').pl())
|
|
84
|
+
|
|
85
|
+
# Show row count
|
|
86
|
+
print('\\n**Row Count**')
|
|
87
|
+
print(duckdb.sql('SELECT COUNT(*) as total_rows FROM \"data.csv\"').pl())
|
|
88
|
+
"
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
## Scenario 6: Time-Series Analysis
|
|
92
|
+
|
|
93
|
+
**Decision: DuckDB for aggregation + matplotlib for visualization**
|
|
94
|
+
|
|
95
|
+
```python
|
|
96
|
+
uv run --with numpy --with duckdb --with pyarrow --with matplotlib python -c "
|
|
97
|
+
import duckdb
|
|
98
|
+
import matplotlib.pyplot as plt
|
|
99
|
+
|
|
100
|
+
# Aggregate by date
|
|
101
|
+
result = duckdb.sql('''
|
|
102
|
+
SELECT
|
|
103
|
+
DATE_TRUNC('day', timestamp) as date,
|
|
104
|
+
COUNT(*) as count,
|
|
105
|
+
AVG(value) as avg_value
|
|
106
|
+
FROM 'timeseries.csv'
|
|
107
|
+
GROUP BY date
|
|
108
|
+
ORDER BY date
|
|
109
|
+
''').pl()
|
|
110
|
+
|
|
111
|
+
# Plot
|
|
112
|
+
plt.figure(figsize=(12, 6))
|
|
113
|
+
plt.subplot(2, 1, 1)
|
|
114
|
+
plt.plot(result['date'], result['count'])
|
|
115
|
+
plt.title('Daily Count')
|
|
116
|
+
|
|
117
|
+
plt.subplot(2, 1, 2)
|
|
118
|
+
plt.plot(result['date'], result['avg_value'])
|
|
119
|
+
plt.title('Daily Average Value')
|
|
120
|
+
|
|
121
|
+
plt.tight_layout()
|
|
122
|
+
plt.savefig('timeseries.png')
|
|
123
|
+
print('Saved to timeseries.png')
|
|
124
|
+
"
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
## Scenario 7: Complex Transformation
|
|
128
|
+
|
|
129
|
+
**Decision: Polars for efficient transformations**
|
|
130
|
+
|
|
131
|
+
```python
|
|
132
|
+
uv run --with numpy --with polars python -c "
|
|
133
|
+
import polars as pl
|
|
134
|
+
|
|
135
|
+
result = (
|
|
136
|
+
pl.scan_csv('data.csv')
|
|
137
|
+
.with_columns([
|
|
138
|
+
# Create new calculated columns
|
|
139
|
+
(pl.col('price') * pl.col('quantity')).alias('total'),
|
|
140
|
+
pl.col('date').str.strptime(pl.Date, '%Y-%m-%d').alias('parsed_date'),
|
|
141
|
+
pl.col('name').str.to_uppercase().alias('upper_name'),
|
|
142
|
+
])
|
|
143
|
+
.filter(pl.col('total') > 100)
|
|
144
|
+
.select(['parsed_date', 'upper_name', 'total'])
|
|
145
|
+
.collect()
|
|
146
|
+
)
|
|
147
|
+
|
|
148
|
+
print(result)
|
|
149
|
+
"
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
## Scenario 8: Large File Processing
|
|
153
|
+
|
|
154
|
+
**Decision: Polars streaming mode**
|
|
155
|
+
|
|
156
|
+
```python
|
|
157
|
+
uv run --with numpy --with polars python -c "
|
|
158
|
+
import polars as pl
|
|
159
|
+
|
|
160
|
+
# Process file larger than RAM
|
|
161
|
+
result = (
|
|
162
|
+
pl.scan_csv('huge_file.csv')
|
|
163
|
+
.filter(pl.col('active') == True)
|
|
164
|
+
.groupby('category')
|
|
165
|
+
.agg([
|
|
166
|
+
pl.count().alias('count'),
|
|
167
|
+
pl.sum('amount').alias('total'),
|
|
168
|
+
pl.mean('amount').alias('average'),
|
|
169
|
+
])
|
|
170
|
+
.collect(streaming=True) # Streaming mode
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
print(result)
|
|
174
|
+
print(f'\\nProcessed {result[\"count\"].sum():,} rows')
|
|
175
|
+
"
|
|
176
|
+
```
|
|
@@ -0,0 +1,197 @@
|
|
|
1
|
+
# Execution Templates
|
|
2
|
+
|
|
3
|
+
## Template 1: DuckDB for Simple/Complex SQL Queries
|
|
4
|
+
|
|
5
|
+
**Use when:**
|
|
6
|
+
- Simple aggregation on single file
|
|
7
|
+
- Complex multi-table joins
|
|
8
|
+
- Heavy GROUP BY operations
|
|
9
|
+
- Window functions with complex SQL logic
|
|
10
|
+
- Ad-hoc exploration queries
|
|
11
|
+
|
|
12
|
+
```python
|
|
13
|
+
uv run --with numpy --with duckdb python -c "
|
|
14
|
+
import duckdb
|
|
15
|
+
|
|
16
|
+
# Simple query - direct file access
|
|
17
|
+
result = duckdb.sql('''
|
|
18
|
+
SELECT
|
|
19
|
+
category,
|
|
20
|
+
COUNT(*) as count,
|
|
21
|
+
AVG(amount) as avg_amount,
|
|
22
|
+
SUM(amount) as total
|
|
23
|
+
FROM 'data.csv'
|
|
24
|
+
WHERE date >= '2024-01-01'
|
|
25
|
+
GROUP BY category
|
|
26
|
+
ORDER BY total DESC
|
|
27
|
+
LIMIT 10
|
|
28
|
+
''').pl()
|
|
29
|
+
|
|
30
|
+
print('**Results**')
|
|
31
|
+
print(result)
|
|
32
|
+
print(f'\\nProcessed {len(result)} categories')
|
|
33
|
+
"
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
## Template 2: Polars for Filtering & Sorting
|
|
37
|
+
|
|
38
|
+
**Use when:**
|
|
39
|
+
- Primary operation is filtering large dataset
|
|
40
|
+
- Sorting required
|
|
41
|
+
- Chain transformations
|
|
42
|
+
- Memory-efficient processing needed
|
|
43
|
+
|
|
44
|
+
```python
|
|
45
|
+
uv run --with numpy --with polars python -c "
|
|
46
|
+
import polars as pl
|
|
47
|
+
|
|
48
|
+
# Lazy evaluation for optimal performance
|
|
49
|
+
result = (
|
|
50
|
+
pl.scan_csv('data.csv') # Lazy scan
|
|
51
|
+
.filter(
|
|
52
|
+
(pl.col('amount') > 1000) &
|
|
53
|
+
(pl.col('status') == 'active') &
|
|
54
|
+
(pl.col('date') >= '2024-01-01')
|
|
55
|
+
)
|
|
56
|
+
.sort('amount', descending=True)
|
|
57
|
+
.head(100)
|
|
58
|
+
.collect() # Execute optimized plan
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
print('**Filtered and Sorted Results**')
|
|
62
|
+
print(result)
|
|
63
|
+
print(f'\\nFound {len(result)} matching rows')
|
|
64
|
+
"
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
## Template 3: Hybrid Approach (Best of Both)
|
|
68
|
+
|
|
69
|
+
**Use when:**
|
|
70
|
+
- Need joins AND heavy filtering
|
|
71
|
+
- Complex SQL followed by transformations
|
|
72
|
+
- Optimize different operation stages
|
|
73
|
+
|
|
74
|
+
```python
|
|
75
|
+
uv run --with numpy --with duckdb --with polars --with pyarrow python -c "
|
|
76
|
+
import duckdb
|
|
77
|
+
import polars as pl
|
|
78
|
+
|
|
79
|
+
print('Phase 1: DuckDB for complex join (3x faster)')
|
|
80
|
+
# DuckDB excels at joins
|
|
81
|
+
joined = duckdb.sql('''
|
|
82
|
+
SELECT
|
|
83
|
+
o.order_id,
|
|
84
|
+
o.amount,
|
|
85
|
+
c.customer_id,
|
|
86
|
+
c.region,
|
|
87
|
+
p.category
|
|
88
|
+
FROM 'orders.csv' o
|
|
89
|
+
JOIN 'customers.csv' c ON o.customer_id = c.customer_id
|
|
90
|
+
JOIN 'products.csv' p ON o.product_id = p.product_id
|
|
91
|
+
WHERE o.date >= '2024-01-01'
|
|
92
|
+
''').pl() # Convert to Polars
|
|
93
|
+
|
|
94
|
+
print(f'Joined {len(joined):,} rows')
|
|
95
|
+
|
|
96
|
+
print('\\nPhase 2: Polars for ultra-fast filtering (128x faster)')
|
|
97
|
+
# Polars excels at filtering
|
|
98
|
+
filtered = (
|
|
99
|
+
joined
|
|
100
|
+
.filter(
|
|
101
|
+
(pl.col('amount') > 100) &
|
|
102
|
+
(pl.col('region').is_in(['North', 'South', 'East']))
|
|
103
|
+
)
|
|
104
|
+
.with_columns([
|
|
105
|
+
(pl.col('amount') * 1.1).alias('amount_with_tax')
|
|
106
|
+
])
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
print(f'Filtered to {len(filtered):,} rows')
|
|
110
|
+
|
|
111
|
+
print('\\nPhase 3: DuckDB for final aggregation (4x faster)')
|
|
112
|
+
# Back to DuckDB for aggregation
|
|
113
|
+
duckdb.register('filtered_data', filtered)
|
|
114
|
+
final = duckdb.sql('''
|
|
115
|
+
SELECT
|
|
116
|
+
region,
|
|
117
|
+
category,
|
|
118
|
+
COUNT(DISTINCT customer_id) as customers,
|
|
119
|
+
SUM(amount_with_tax) as total_revenue,
|
|
120
|
+
AVG(amount_with_tax) as avg_transaction
|
|
121
|
+
FROM filtered_data
|
|
122
|
+
GROUP BY region, category
|
|
123
|
+
HAVING total_revenue > 10000
|
|
124
|
+
ORDER BY total_revenue DESC
|
|
125
|
+
''').pl()
|
|
126
|
+
|
|
127
|
+
print('\\n**Final Results**')
|
|
128
|
+
print(final)
|
|
129
|
+
"
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
## Template 4: Polars Streaming for Large Files
|
|
133
|
+
|
|
134
|
+
**Use when:**
|
|
135
|
+
- Dataset larger than available RAM
|
|
136
|
+
- Need to process data in batches
|
|
137
|
+
- Memory constraints
|
|
138
|
+
|
|
139
|
+
```python
|
|
140
|
+
uv run --with numpy --with polars python -c "
|
|
141
|
+
import polars as pl
|
|
142
|
+
|
|
143
|
+
# Streaming mode - processes data in chunks
|
|
144
|
+
result = (
|
|
145
|
+
pl.scan_csv('huge_file.csv')
|
|
146
|
+
.filter(pl.col('status') == 'active')
|
|
147
|
+
.with_columns([
|
|
148
|
+
(pl.col('amount') * 1.1).alias('adjusted_amount')
|
|
149
|
+
])
|
|
150
|
+
.groupby('category')
|
|
151
|
+
.agg([
|
|
152
|
+
pl.sum('adjusted_amount').alias('total'),
|
|
153
|
+
pl.count().alias('count')
|
|
154
|
+
])
|
|
155
|
+
.collect(streaming=True) # Streaming mode for large data
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
print('**Streaming Results**')
|
|
159
|
+
print(result)
|
|
160
|
+
print(f'\\nProcessed {result[\"count\"].sum():,} total rows')
|
|
161
|
+
"
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
## Template 5: Visualization with Matplotlib
|
|
165
|
+
|
|
166
|
+
**Use when:**
|
|
167
|
+
- User requests charts, graphs, or plots
|
|
168
|
+
- Exploratory data analysis (EDA)
|
|
169
|
+
- Time-series or distribution analysis
|
|
170
|
+
|
|
171
|
+
```python
|
|
172
|
+
uv run --with numpy --with duckdb --with polars --with pyarrow --with matplotlib python -c "
|
|
173
|
+
import duckdb
|
|
174
|
+
import matplotlib.pyplot as plt
|
|
175
|
+
|
|
176
|
+
# Query data
|
|
177
|
+
result = duckdb.sql('''
|
|
178
|
+
SELECT
|
|
179
|
+
date,
|
|
180
|
+
SUM(amount) as total
|
|
181
|
+
FROM 'data.csv'
|
|
182
|
+
GROUP BY date
|
|
183
|
+
ORDER BY date
|
|
184
|
+
''').pl()
|
|
185
|
+
|
|
186
|
+
# Create visualization
|
|
187
|
+
plt.figure(figsize=(10, 6))
|
|
188
|
+
plt.plot(result['date'], result['total'])
|
|
189
|
+
plt.xlabel('Date')
|
|
190
|
+
plt.ylabel('Total Amount')
|
|
191
|
+
plt.title('Daily Total Trends')
|
|
192
|
+
plt.xticks(rotation=45)
|
|
193
|
+
plt.tight_layout()
|
|
194
|
+
plt.savefig('output.png')
|
|
195
|
+
print('Chart saved to output.png')
|
|
196
|
+
"
|
|
197
|
+
```
|