oh-my-opencode 4.19.3 → 4.19.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (150) hide show
  1. package/.agents/command/get-unpublished-changes.md +2 -0
  2. package/.agents/command/omomomo.md +1 -1
  3. package/.agents/command/publish.md +7 -0
  4. package/.agents/skills/get-unpublished-changes/SKILL.md +2 -0
  5. package/.agents/skills/hyperplan/SKILL.md +3 -3
  6. package/.agents/skills/omomomo/SKILL.md +1 -1
  7. package/.agents/skills/publish/SKILL.md +7 -0
  8. package/.opencode/command/get-unpublished-changes.md +2 -0
  9. package/.opencode/command/omomomo.md +1 -1
  10. package/.opencode/command/publish.md +7 -0
  11. package/.opencode/skills/hyperplan/SKILL.md +3 -3
  12. package/dist/agents/sisyphus-junior/agent.d.ts +1 -1
  13. package/dist/cli/doctor/checks/deprecated-reasoning-keys.d.ts +2 -0
  14. package/dist/cli/index.js +1625 -1368
  15. package/dist/cli-node/index.js +1625 -1368
  16. package/dist/config/schema/agent-overrides.d.ts +800 -0
  17. package/dist/config/schema/categories.d.ts +132 -0
  18. package/dist/config/schema/fallback-models.d.ts +50 -0
  19. package/dist/config/schema/oh-my-opencode-config.d.ts +818 -2
  20. package/dist/config-migration/index.d.ts +1 -0
  21. package/dist/config-migration/migration-plans.d.ts +1 -0
  22. package/dist/config-migration/reasoning-unification.d.ts +3 -0
  23. package/dist/features/team-mode/tools/lifecycle-test-fixture.d.ts +2 -0
  24. package/dist/hooks/codegraph-bootstrap/command-runner.d.ts +1 -0
  25. package/dist/hooks/model-fallback/next-fallback.d.ts +1 -0
  26. package/dist/hooks/runtime-fallback/constants.d.ts +1 -1
  27. package/dist/index.js +3558 -3241
  28. package/dist/oh-my-opencode.schema.json +1672 -8
  29. package/dist/plugin-handlers/prometheus-agent-config-builder.d.ts +1 -0
  30. package/dist/shared/agent-variant.d.ts +11 -0
  31. package/dist/shared/session-prompt-params-helpers.d.ts +6 -1
  32. package/dist/skills/coding-agent-sessions/SKILL.md +4 -3
  33. package/dist/skills/coding-agent-sessions/references/all-platforms.md +3 -1
  34. package/dist/skills/coding-agent-sessions/scripts/agent_sessions/aside_scanner.py +140 -0
  35. package/dist/skills/coding-agent-sessions/scripts/agent_sessions/scanners.py +3 -0
  36. package/dist/skills/data-scientist/SKILL.md +243 -0
  37. package/dist/skills/data-scientist/references/common-scenarios.md +176 -0
  38. package/dist/skills/data-scientist/references/execution-templates.md +197 -0
  39. package/dist/skills/data-scientist/references/integration-patterns.md +153 -0
  40. package/dist/skills/data-scientist/references/performance-benchmarks.md +37 -0
  41. package/dist/skills/data-scientist/references/uv-setup.md +78 -0
  42. package/dist/skills/data-scientist/scripts/quick-query.py +111 -0
  43. package/dist/skills/data-scientist/scripts/setup-uv.ps1 +53 -0
  44. package/dist/skills/data-scientist/scripts/setup-uv.sh +60 -0
  45. package/dist/skills/debugging/SKILL.md +1 -1
  46. package/dist/skills/programming/SKILL.md +1 -2
  47. package/dist/skills/ulw-plan/SKILL.md +1 -1
  48. package/dist/skills/ulw-research/SKILL.md +122 -11
  49. package/dist/tools/delegate-task/builtin-categories.d.ts +1 -0
  50. package/dist/tools/delegate-task/builtin-category-definition.d.ts +1 -0
  51. package/dist/tools/delegate-task/constants.d.ts +1 -1
  52. package/dist/tui.js +974 -1090
  53. package/package.json +13 -13
  54. package/packages/lsp-core/src/lsp/client-diagnostics-freshness.integration.test.ts +2 -2
  55. package/packages/omo-codex/plugin/.codex-plugin/plugin.json +1 -1
  56. package/packages/omo-codex/plugin/components/bootstrap/dist/cli.js +21 -3
  57. package/packages/omo-codex/plugin/components/bootstrap/hooks/hooks.json +1 -1
  58. package/packages/omo-codex/plugin/components/bootstrap/package.json +1 -1
  59. package/packages/omo-codex/plugin/components/codegraph/dist/cli.js +279 -58
  60. package/packages/omo-codex/plugin/components/codegraph/dist/serve.js +230 -41
  61. package/packages/omo-codex/plugin/components/codegraph/package.json +1 -1
  62. package/packages/omo-codex/plugin/components/codegraph/test/hook.test.ts +6 -0
  63. package/packages/omo-codex/plugin/components/comment-checker/hooks/hooks.json +1 -1
  64. package/packages/omo-codex/plugin/components/comment-checker/package.json +1 -1
  65. package/packages/omo-codex/plugin/components/git-bash/hooks/hooks.json +2 -2
  66. package/packages/omo-codex/plugin/components/git-bash/package.json +1 -1
  67. package/packages/omo-codex/plugin/components/lazycodex-executor-verify/hooks/hooks.json +1 -1
  68. package/packages/omo-codex/plugin/components/lazycodex-executor-verify/package.json +1 -1
  69. package/packages/omo-codex/plugin/components/lsp/dist/.omo-runtime-manifest.json +2 -2
  70. package/packages/omo-codex/plugin/components/lsp/hooks/hooks.json +2 -2
  71. package/packages/omo-codex/plugin/components/lsp/package.json +1 -1
  72. package/packages/omo-codex/plugin/components/rules/dist/cli.js +1 -1
  73. package/packages/omo-codex/plugin/components/rules/hooks/hooks.json +4 -4
  74. package/packages/omo-codex/plugin/components/rules/package.json +1 -1
  75. package/packages/omo-codex/plugin/components/rules/src/post-compact-budget.ts +1 -1
  76. package/packages/omo-codex/plugin/components/start-work-continuation/hooks/hooks.json +2 -2
  77. package/packages/omo-codex/plugin/components/start-work-continuation/package.json +1 -1
  78. package/packages/omo-codex/plugin/components/teammode/hooks/hooks.json +1 -1
  79. package/packages/omo-codex/plugin/components/teammode/package.json +1 -1
  80. package/packages/omo-codex/plugin/components/telemetry/hooks/hooks.json +1 -1
  81. package/packages/omo-codex/plugin/components/telemetry/package.json +1 -1
  82. package/packages/omo-codex/plugin/components/ultrawork/hooks/hooks.json +1 -1
  83. package/packages/omo-codex/plugin/components/ultrawork/package.json +1 -1
  84. package/packages/omo-codex/plugin/components/ultrawork/skills/ulw-plan/SKILL.md +1 -1
  85. package/packages/omo-codex/plugin/components/ulw-loop/hooks/hooks.json +4 -4
  86. package/packages/omo-codex/plugin/components/ulw-loop/package.json +1 -1
  87. package/packages/omo-codex/plugin/hooks/post-compact-resetting-git-bash-mcp-reminder.json +1 -1
  88. package/packages/omo-codex/plugin/hooks/post-compact-resetting-lsp-diagnostics-cache.json +1 -1
  89. package/packages/omo-codex/plugin/hooks/post-compact-resetting-project-rule-cache.json +1 -1
  90. package/packages/omo-codex/plugin/hooks/post-tool-use-checking-codegraph-init-guidance.json +1 -1
  91. package/packages/omo-codex/plugin/hooks/post-tool-use-checking-comments.json +1 -1
  92. package/packages/omo-codex/plugin/hooks/post-tool-use-checking-lsp-diagnostics.json +1 -1
  93. package/packages/omo-codex/plugin/hooks/post-tool-use-checking-thread-title-hygiene.json +1 -1
  94. package/packages/omo-codex/plugin/hooks/post-tool-use-matching-project-rules.json +1 -1
  95. package/packages/omo-codex/plugin/hooks/pre-tool-use-enforcing-unlimited-goal-budget.json +1 -1
  96. package/packages/omo-codex/plugin/hooks/pre-tool-use-guarding-ulw-loop-spawns.json +1 -1
  97. package/packages/omo-codex/plugin/hooks/pre-tool-use-recommending-git-bash-mcp.json +1 -1
  98. package/packages/omo-codex/plugin/hooks/session-start-checking-auto-update.json +1 -1
  99. package/packages/omo-codex/plugin/hooks/session-start-checking-bootstrap-provisioning.json +1 -1
  100. package/packages/omo-codex/plugin/hooks/session-start-checking-codegraph-bootstrap.json +1 -1
  101. package/packages/omo-codex/plugin/hooks/session-start-loading-project-rules.json +1 -1
  102. package/packages/omo-codex/plugin/hooks/session-start-recording-session-telemetry.json +1 -1
  103. package/packages/omo-codex/plugin/hooks/stop-checking-start-work-continuation.json +1 -1
  104. package/packages/omo-codex/plugin/hooks/stop-checking-ulw-loop-resume.json +1 -1
  105. package/packages/omo-codex/plugin/hooks/subagent-stop-checking-start-work-continuation.json +1 -1
  106. package/packages/omo-codex/plugin/hooks/subagent-stop-verifying-lazycodex-executor-evidence.json +1 -1
  107. package/packages/omo-codex/plugin/hooks/user-prompt-submit-checking-ultrawork-trigger.json +1 -1
  108. package/packages/omo-codex/plugin/hooks/user-prompt-submit-checking-ulw-loop-steering.json +1 -1
  109. package/packages/omo-codex/plugin/hooks/user-prompt-submit-loading-project-rules.json +1 -1
  110. package/packages/omo-codex/plugin/package-lock.json +13 -13
  111. package/packages/omo-codex/plugin/package.json +1 -1
  112. package/packages/omo-codex/plugin/scripts/sync-skills.mjs +6 -0
  113. package/packages/omo-codex/plugin/skills/coding-agent-sessions/SKILL.md +4 -3
  114. package/packages/omo-codex/plugin/skills/coding-agent-sessions/references/all-platforms.md +3 -1
  115. package/packages/omo-codex/plugin/skills/coding-agent-sessions/scripts/agent_sessions/aside_scanner.py +140 -0
  116. package/packages/omo-codex/plugin/skills/coding-agent-sessions/scripts/agent_sessions/scanners.py +3 -0
  117. package/packages/omo-codex/plugin/skills/data-scientist/SKILL.md +243 -0
  118. package/packages/omo-codex/plugin/skills/data-scientist/agents/openai.yaml +2 -0
  119. package/packages/omo-codex/plugin/skills/data-scientist/references/common-scenarios.md +176 -0
  120. package/packages/omo-codex/plugin/skills/data-scientist/references/execution-templates.md +197 -0
  121. package/packages/omo-codex/plugin/skills/data-scientist/references/integration-patterns.md +153 -0
  122. package/packages/omo-codex/plugin/skills/data-scientist/references/performance-benchmarks.md +37 -0
  123. package/packages/omo-codex/plugin/skills/data-scientist/references/uv-setup.md +78 -0
  124. package/packages/omo-codex/plugin/skills/data-scientist/scripts/quick-query.py +111 -0
  125. package/packages/omo-codex/plugin/skills/data-scientist/scripts/setup-uv.ps1 +53 -0
  126. package/packages/omo-codex/plugin/skills/data-scientist/scripts/setup-uv.sh +60 -0
  127. package/packages/omo-codex/plugin/skills/debugging/SKILL.md +1 -1
  128. package/packages/omo-codex/plugin/skills/programming/SKILL.md +1 -2
  129. package/packages/omo-codex/plugin/skills/ulw-plan/SKILL.md +1 -1
  130. package/packages/omo-codex/plugin/skills/ulw-research/SKILL.md +121 -11
  131. package/packages/omo-codex/plugin/test/sync-skills-test-support.mjs +7 -0
  132. package/packages/omo-codex/plugin/test/sync-skills.test.mjs +12 -0
  133. package/packages/omo-codex/scripts/install-dist/install-local.mjs +24 -5
  134. package/packages/shared-skills/skills/coding-agent-sessions/SKILL.md +4 -3
  135. package/packages/shared-skills/skills/coding-agent-sessions/references/all-platforms.md +3 -1
  136. package/packages/shared-skills/skills/coding-agent-sessions/scripts/agent_sessions/aside_scanner.py +140 -0
  137. package/packages/shared-skills/skills/coding-agent-sessions/scripts/agent_sessions/scanners.py +3 -0
  138. package/packages/shared-skills/skills/data-scientist/SKILL.md +243 -0
  139. package/packages/shared-skills/skills/data-scientist/references/common-scenarios.md +176 -0
  140. package/packages/shared-skills/skills/data-scientist/references/execution-templates.md +197 -0
  141. package/packages/shared-skills/skills/data-scientist/references/integration-patterns.md +153 -0
  142. package/packages/shared-skills/skills/data-scientist/references/performance-benchmarks.md +37 -0
  143. package/packages/shared-skills/skills/data-scientist/references/uv-setup.md +78 -0
  144. package/packages/shared-skills/skills/data-scientist/scripts/quick-query.py +111 -0
  145. package/packages/shared-skills/skills/data-scientist/scripts/setup-uv.ps1 +53 -0
  146. package/packages/shared-skills/skills/data-scientist/scripts/setup-uv.sh +60 -0
  147. package/packages/shared-skills/skills/debugging/SKILL.md +1 -1
  148. package/packages/shared-skills/skills/programming/SKILL.md +1 -2
  149. package/packages/shared-skills/skills/ulw-plan/SKILL.md +1 -1
  150. package/packages/shared-skills/skills/ulw-research/SKILL.md +122 -11
@@ -4,7 +4,7 @@
4
4
 
5
5
  Search these first, then add user-supplied roots with `--root`:
6
6
 
7
- Registered platform keys: `codex`, `claude`, `senpi`, `oh-my-pi`, `gajae-code`, `opencode`, `openclaw`, `droid`, `amp`, `gemini`, `kimi`, `qwen`, `codebuff`, `roo-code`, `kilo-code`, `cline`, `kodu`, `cursor-cli`, `aider`, `kilo-cli`, `hermes`, `goose`, `crush`, `zed`, `kiro`.
7
+ Registered platform keys: `codex`, `claude`, `senpi`, `oh-my-pi`, `gajae-code`, `opencode`, `openclaw`, `droid`, `amp`, `gemini`, `kimi`, `qwen`, `codebuff`, `roo-code`, `kilo-code`, `cline`, `kodu`, `cursor-cli`, `aider`, `kilo-cli`, `hermes`, `goose`, `crush`, `zed`, `kiro`, `aside`.
8
8
 
9
9
  | Platform | Unix/macOS | Windows |
10
10
  |---|---|---|
@@ -25,6 +25,7 @@ Registered platform keys: `codex`, `claude`, `senpi`, `oh-my-pi`, `gajae-code`,
25
25
  | Aider | bounded project roots containing `.aider.chat.history.md`; use `--root` for other repos | pass `--root` |
26
26
  | Kilo CLI (`kilo-cli`) / Hermes / Goose / Crush / Zed | Known SQLite roots are probed cheaply; unsupported schemas return empty | pass `--root` |
27
27
  | Kiro | `~/.kiro/sessions/cli/*.json` plus paired `*.jsonl` prompt events | pass `--root` |
28
+ | Aside (browser agent) | `~/.aside/u/*/sessions/<date>_<id>/messages.jsonl`, joined with the per-user `state.db` `sessions` table (title, parent_id, cwd, model, timestamps). `~/.aside/u/*/agents/*/sessions/` is a hardlink mirror of `sessions/` and is intentionally not scanned | pass `--root` |
28
29
 
29
30
  ## Excluded usage-only sources
30
31
 
@@ -52,6 +53,7 @@ Do not add these as default transcript platforms without a separate prompt-recon
52
53
  | Claude | `projects/<proj>/<sid>/subagents/**/agent-<agentId>.jsonl` | directory `<sid>` | `agent-*.meta.json` `agentType` |
53
54
  | Codex | thread row + own rollout JSONL | `thread_spawn_edges` / `source.subagent.thread_spawn.parent_thread_id` | `agent_nickname (agent_role)` |
54
55
  | OpenCode | `session` table row / `storage/session/**.json` | `parent_id` column / `parentID` field | `agent` column |
56
+ | Aside | own `sessions/<date>_<id>/messages.jsonl` transcript | `parent_id` column in `state.db` `sessions` | child session's `title` column (task description) |
55
57
 
56
58
  ## Parallelism
57
59
 
@@ -0,0 +1,140 @@
1
+ from __future__ import annotations
2
+
3
+ import sqlite3
4
+ from dataclasses import dataclass
5
+ from pathlib import Path
6
+ from typing import TypeAlias
7
+
8
+ from .jsonio import as_map, iter_jsonl, parse_json_text, text
9
+ from .timeparse import file_time, unix_millis, unix_seconds
10
+ from .transcript import content_text, existing, flat_parallel, merge_usage, recent
11
+ from .types import JsonMap, Session
12
+
13
+ SqlValue: TypeAlias = str | int | float | bytes | None
14
+ SqlRow: TypeAlias = tuple[SqlValue, ...]
15
+
16
+
17
+ @dataclass(frozen=True, slots=True)
18
+ class _StateRow:
19
+ parent_id: str | None
20
+ title: str | None
21
+ cwd: str | None
22
+ provider: str | None
23
+ model: str | None
24
+ created_at: str | None
25
+ updated_at: str | None
26
+
27
+
28
+ def scan_aside(extra_roots: tuple[Path, ...], workers: int) -> list[Session]:
29
+ roots = _roots([Path.home() / ".aside"], extra_roots, (".aside",))
30
+ sessions: list[Session] = []
31
+ for user_dir in _user_dirs(roots):
32
+ index = _state_index(user_dir / "state.db")
33
+ paths = list((user_dir / "sessions").glob("*/messages.jsonl"))
34
+ sessions.extend(flat_parallel(recent(paths), workers, lambda path, rows=index: [_aside_session(path, rows)]))
35
+ return sessions
36
+
37
+
38
+ def _user_dirs(roots: list[Path]) -> list[Path]:
39
+ dirs: list[Path] = []
40
+ for root in roots:
41
+ users = sorted(path for path in (root / "u").glob("*") if path.is_dir())
42
+ if users:
43
+ dirs.extend(users)
44
+ elif (root / "sessions").exists():
45
+ dirs.append(root)
46
+ return dirs
47
+
48
+
49
+ def _state_index(db_path: Path) -> dict[str, _StateRow]:
50
+ if not db_path.exists():
51
+ return {}
52
+ try:
53
+ with sqlite3.connect(f"file:{db_path}?mode=ro", uri=True) as conn:
54
+ rows: list[SqlRow] = conn.execute("select id, parent_id, title, cwd, model, created_at, updated_at from sessions").fetchall()
55
+ except sqlite3.Error:
56
+ return {}
57
+ index: dict[str, _StateRow] = {}
58
+ for row in rows:
59
+ sid = _cell_text(row, 0)
60
+ if sid is None:
61
+ continue
62
+ provider, model = _model_fields(_cell_text(row, 4))
63
+ index[sid] = _StateRow(
64
+ _cell_text(row, 1),
65
+ _cell_text(row, 2),
66
+ _cell_text(row, 3),
67
+ provider,
68
+ model,
69
+ unix_seconds(_cell_number(row, 5)),
70
+ unix_seconds(_cell_number(row, 6)),
71
+ )
72
+ return index
73
+
74
+
75
+ def _cell_text(row: SqlRow, index: int) -> str | None:
76
+ value = row[index] if index < len(row) else None
77
+ return value if isinstance(value, str) else None
78
+
79
+
80
+ def _cell_number(row: SqlRow, index: int) -> int | float | None:
81
+ value = row[index] if index < len(row) else None
82
+ return value if isinstance(value, int | float) else None
83
+
84
+
85
+ def _model_fields(model_json: str | None) -> tuple[str | None, str | None]:
86
+ data = as_map(parse_json_text(model_json)) if model_json else None
87
+ if data is None:
88
+ return None, None
89
+ return text(data.get("provider")), text(data.get("modelId")) or text(data.get("model"))
90
+
91
+
92
+ def _aside_session(path: Path, index: dict[str, _StateRow]) -> Session:
93
+ sid = _dir_session_id(path.parent.name)
94
+ row = index.get(sid)
95
+ first_user = last_user = ""
96
+ provider = model = None
97
+ created = updated = None
98
+ usage: JsonMap = {}
99
+ for data in iter_jsonl(path):
100
+ stamp = data.get("timestamp")
101
+ moment = unix_millis(int(stamp)) if isinstance(stamp, int | float) else None
102
+ created = created or moment
103
+ updated = moment or updated
104
+ role = text(data.get("role"))
105
+ if role == "user":
106
+ prompt = content_text(data.get("content"))
107
+ if prompt:
108
+ first_user = first_user or prompt
109
+ last_user = prompt
110
+ elif role == "assistant":
111
+ provider = provider or text(data.get("provider"))
112
+ model = model or text(data.get("model"))
113
+ merge_usage(usage, as_map(data.get("usage")))
114
+ return Session(
115
+ "aside",
116
+ sid,
117
+ str(path),
118
+ row.cwd if row is not None else None,
119
+ (row.created_at if row is not None else None) or created or file_time(path),
120
+ (row.updated_at if row is not None else None) or updated or created or file_time(path),
121
+ (row.provider if row is not None else None) or provider,
122
+ (row.model if row is not None else None) or model,
123
+ first_user,
124
+ usage,
125
+ row.parent_id if row is not None else None,
126
+ row.title if row is not None and row.parent_id is not None else None,
127
+ last_user,
128
+ )
129
+
130
+
131
+ def _dir_session_id(name: str) -> str:
132
+ return name.split("_", 1)[-1]
133
+
134
+
135
+ def _roots(defaults: list[Path], extra_roots: tuple[Path, ...], children: tuple[str, ...]) -> list[Path]:
136
+ candidates = [*defaults]
137
+ for root in extra_roots:
138
+ candidates.append(root)
139
+ candidates.extend(root / child for child in children)
140
+ return existing(candidates)
@@ -4,6 +4,7 @@ from concurrent.futures import ThreadPoolExecutor, as_completed
4
4
  from pathlib import Path
5
5
  from typing import Callable, TypeAlias
6
6
 
7
+ from .aside_scanner import scan_aside
7
8
  from .claude import scan_claude
8
9
  from .codex import scan_codex
9
10
  from .file_scanners import (
@@ -54,6 +55,7 @@ PLATFORM_SCANNERS: dict[str, Scanner] = {
54
55
  "crush": scan_crush,
55
56
  "zed": scan_zed,
56
57
  "kiro": scan_kiro,
58
+ "aside": scan_aside,
57
59
  }
58
60
  DEFAULT_PLATFORMS = frozenset(PLATFORM_SCANNERS)
59
61
  PLATFORM_ALIASES = {
@@ -69,6 +71,7 @@ PLATFORM_ALIASES = {
69
71
  "gjc": "gajae-code",
70
72
  "gajae": "gajae-code",
71
73
  "gajaecode": "gajae-code",
74
+ "aside-browser": "aside",
72
75
  }
73
76
 
74
77
  __all__ = ["DEFAULT_PLATFORMS", "PLATFORM_SCANNERS", "scan", "scan_claude", "scan_codex", "scan_gajae_code", "scan_oh_my_pi", "scan_opencode", "scan_senpi"]
@@ -0,0 +1,243 @@
1
+ ---
2
+ name: data-scientist
3
+ description: "Expert data processing specialist with intelligent DuckDB/Polars selection for maximum performance. Always includes numpy, never uses pandas, runs everything through uv. Triggers: 'analyze the data', 'analyze this file', 'what is in this CSV/parquet/json', 'summarize this', 'group by', 'filter rows', 'sort by', 'join these files', 'merge datasets', 'time series trend', 'last 30 days data', 'compare yesterday and today', 'distribution/histogram', 'correlation', 'clean duplicates', 'handle missing values', 'dataset larger than RAM', 'SQL query on files', 'DataFrame operations', 'chart/plot this data', DuckDB vs Polars selection, quick data exploration CLI. NOT for plain text/code inspection, configs, or tiny inline math."
4
+ ---
5
+
6
+ # Data Scientist: High-Performance Data Processing Expert
7
+
8
+ ## Role & Expertise
9
+
10
+ Performance-obsessed data scientist with expertise in:
11
+ - Intelligent tool selection: DuckDB vs Polars based on operation characteristics
12
+ - Zero-copy data interchange via Apache Arrow
13
+ - Memory-efficient processing for datasets exceeding RAM
14
+ - SQL and DataFrame API mastery for analytical workloads
15
+
16
+ ## Environment Setup
17
+
18
+ Everything runs through **uv**. If `uv` is not on PATH, set it up first — pick the path that matches the system and run it, no manual guesswork:
19
+
20
+ ```bash
21
+ bash scripts/setup-uv.sh # macOS / Linux / WSL / Git Bash — auto-detects OS + arch, installs or updates uv to latest
22
+ ```
23
+
24
+ ```powershell
25
+ powershell -ExecutionPolicy Bypass -File scripts/setup-uv.ps1 # native Windows — installs or updates uv to latest
26
+ ```
27
+
28
+ Both scripts detect the platform, install uv when missing (official installer first, Homebrew/winget as fallback), upgrade it when present (`uv self update`), put it on PATH for the current shell, and verify with `uv --version`. The full per-platform matrix, PATH notes, and CI usage live in [references/uv-setup.md](references/uv-setup.md). Verify: `uv --version`.
29
+
30
+ ## Core Principles
31
+
32
+ ### ABSOLUTE RULES
33
+
34
+ 1. **ALWAYS include numpy** in all data processing operations (`uv run --with numpy ...`)
35
+ 2. **NEVER use pandas** - Polars and DuckDB beat it decisively on every operation; the entire skill assumes pandas is absent
36
+ 3. **ALWAYS use Python via `uv run`** for calculations and data processing
37
+ 4. **Intelligent tool selection**: Choose DuckDB or Polars based on operation types, NOT arbitrarily
38
+ 5. **Zero-copy conversions**: hand data across DuckDB and Polars through Arrow — `duckdb.sql(...).pl()`. Never call `.df()` (returns a pandas frame; crashes without pandas). Keep `pyarrow` in the package set or `.pl()` raises `ModuleNotFoundError`
39
+ 6. **Lazy evaluation**: Prefer `scan_csv`/`scan_parquet` and `.collect()` only when needed
40
+ 7. **Direct file queries**: Let DuckDB query files directly instead of loading to memory when possible
41
+
42
+ ### Standard Package Pattern
43
+
44
+ ```bash
45
+ # Default for data tasks (numpy + pyarrow are mandatory parts of the set)
46
+ uv run --with numpy --with duckdb --with polars --with pyarrow python -c "{code}"
47
+
48
+ # With visualization (RECOMMENDED for most analysis requests)
49
+ uv run --with numpy --with duckdb --with polars --with pyarrow --with matplotlib python -c "{code}"
50
+
51
+ # Pure Polars
52
+ uv run --with numpy --with polars python -c "{code}"
53
+
54
+ # Pure DuckDB (with the Arrow handoff available)
55
+ uv run --with numpy --with duckdb --with pyarrow python -c "{code}"
56
+ ```
57
+
58
+ **When to include matplotlib:**
59
+ - User requests visualization: "graph", "chart", "plot", "show me"
60
+ - Exploratory data analysis (EDA): "analyze", "trends", "patterns"
61
+ - Time-series analysis: "over time", "daily", "trends"
62
+ - Distribution analysis: "distribution", "histogram", "statistics"
63
+ - Comparison tasks: "compare", visual comparison implied
64
+ - **Default to including matplotlib** when in doubt - overhead is minimal
65
+
66
+ ## Tool Selection Logic
67
+
68
+ ### Decision Tree (Apply in Order)
69
+
70
+ 1. **Is it a `.duckdb` file?** → **USE DUCKDB** (native format, optimal performance)
71
+ 2. **Simple one-off query without needing full data in memory?** → **USE DUCKDB** (direct file query, zero memory load)
72
+ 3. **Very heavy complex SQL query (multi-table joins, window functions)?** → **USE DUCKDB** (superior SQL optimizer)
73
+ 4. **Main operation is FILTERING?** → **USE POLARS** (typically the fastest by a wide margin — see benchmarks)
74
+ 5. **Main operation is SORTING?** → **USE POLARS** (typically the fastest)
75
+ 6. **Complex SQL JOINS needed?** → **USE DUCKDB** (stronger join engine, more join types)
76
+ 7. **Heavy GROUP BY AGGREGATIONS?** → **USE DUCKDB** (typically faster on large datasets)
77
+ 8. **Window functions with partitioning?** → **POLARS** (typically faster)
78
+ 9. **Complex TRANSFORMATIONS (pivot, melt, string ops)?** → **USE POLARS**
79
+ 10. **Dataset larger than available RAM?** → **USE POLARS** (streaming support) or **DUCKDB** (out-of-core)
80
+ 11. **Mixed operations?** → **USE HYBRID APPROACH** (leverage strengths of both)
81
+
82
+ ### Quick Reference
83
+
84
+ ```
85
+ Simple query → DuckDB
86
+ Heavy complex query → DuckDB
87
+ Filter → Polars
88
+ Sort → Polars
89
+ Join → DuckDB
90
+ Aggregate → DuckDB
91
+ Window → Polars
92
+ Transform → Polars
93
+ Too large for RAM → Polars streaming
94
+ Mixed operations → Hybrid
95
+ ```
96
+
97
+ The exact multipliers these heuristics distill (with sources and caveats — routing heuristics, not guarantees) live in [performance-benchmarks.md](references/performance-benchmarks.md).
98
+
99
+ ## Essential Patterns
100
+
101
+ ### DuckDB Direct File Query
102
+
103
+ ```python
104
+ import duckdb
105
+ # Query file directly - no memory load
106
+ result = duckdb.sql("""
107
+ SELECT category, SUM(amount) as total
108
+ FROM 'data.csv'
109
+ GROUP BY category
110
+ """).pl() # .pl() -> Polars via Arrow. Requires pyarrow. Never .df() (pandas).
111
+ ```
112
+
113
+ ### Polars Lazy Evaluation
114
+
115
+ ```python
116
+ import polars as pl
117
+ # Lazy scan - optimizes and executes once
118
+ result = (
119
+ pl.scan_csv('data.csv')
120
+ .filter(pl.col('value') > 100)
121
+ .sort('value', descending=True)
122
+ .collect()
123
+ )
124
+ ```
125
+
126
+ ### Zero-Copy DuckDB → Polars
127
+
128
+ ```python
129
+ import duckdb
130
+ # Direct conversion via Arrow (pyarrow required in the package set)
131
+ df_polars = duckdb.sql("SELECT * FROM 'data.csv'").pl()
132
+ ```
133
+
134
+ ### Hybrid Approach
135
+
136
+ ```python
137
+ import duckdb
138
+ import polars as pl
139
+
140
+ # Phase 1: DuckDB for joins
141
+ joined = duckdb.sql(
142
+ "SELECT * FROM 'orders.csv' o "
143
+ "JOIN 'customers.csv' c ON o.customer_id = c.customer_id"
144
+ ).pl()
145
+
146
+ # Phase 2: Polars for filtering
147
+ filtered = joined.filter(pl.col('amount') > 100)
148
+
149
+ # Phase 3: Back to DuckDB for aggregation
150
+ duckdb.register('filtered_data', filtered)
151
+ final = duckdb.sql('SELECT category, SUM(amount) FROM filtered_data GROUP BY category').pl()
152
+ ```
153
+
154
+ ## Quick Query CLI
155
+
156
+ For ad-hoc data exploration, use the built-in query runner:
157
+
158
+ ```bash
159
+ # SQL query (uses DuckDB)
160
+ uv run scripts/quick-query.py data.csv "SELECT category, COUNT(*) FROM data GROUP BY category"
161
+
162
+ # Filter expression — Polars SQL syntax, e.g. "amount > 100" (NOT Python: never passes through eval)
163
+ uv run scripts/quick-query.py data.csv --filter "amount > 100"
164
+
165
+ # Auto-describe (schema + stats)
166
+ uv run scripts/quick-query.py data.parquet --describe
167
+ ```
168
+
169
+ Supports CSV, Parquet, JSON, NDJSON. Cross-platform (macOS, Linux, Windows). Excel files are not read directly — export to CSV or Parquet first.
170
+
171
+ ## Reference Documentation
172
+
173
+ For detailed guidance, consult these reference files:
174
+
175
+ - **Environment setup per platform**: See [uv-setup.md](references/uv-setup.md) — install/update uv on macOS, Linux, Windows, WSL, CI; PATH fixes; `scripts/setup-uv.sh` / `scripts/setup-uv.ps1` automate it.
176
+ - **Performance benchmarks and operation detection**: See [performance-benchmarks.md](references/performance-benchmarks.md)
177
+ - **Integration patterns and best practices**: See [integration-patterns.md](references/integration-patterns.md)
178
+ - **Execution templates**: See [execution-templates.md](references/execution-templates.md)
179
+ - **Common scenarios**: See [common-scenarios.md](references/common-scenarios.md)
180
+
181
+ ## Quality Assurance Process
182
+
183
+ ### Before Execution
184
+ 1. **Analyze request** → Detect operation types (filter, join, aggregate, etc.)
185
+ 2. **Select optimal tool** → Apply decision tree based on detected operations
186
+ 3. **Verify approach** → Confirm tool selection matches the benchmark heuristics
187
+ 4. **Check package list** → Ensure numpy AND pyarrow are included
188
+
189
+ ### During Execution
190
+ 1. **Use lazy evaluation** when possible (Polars `scan_*`, DuckDB direct queries)
191
+ 2. **Monitor for errors** and have fallback strategy ready
192
+ 3. **Provide progress updates** for long operations
193
+
194
+ ### After Execution
195
+ 1. **Report performance** → Show processing time and row counts
196
+ 2. **Validate results** → Confirm output matches expectations
197
+ 3. **Document tool choice** → Explain why specific tool was selected
198
+
199
+ ## Activation Context
200
+
201
+ **Automatic activation triggers:**
202
+
203
+ ### Exploratory Questions
204
+ - "Analyze the data" / "What's in the data" / "What's in this file"
205
+ - "Show me the data" / "Take a look at this file" / "Check the file contents"
206
+
207
+ ### Temporal/Historical Analysis
208
+ - "What happened in the past N days?" / "How's last week's data?"
209
+ - "What's the trend for the last 30 days?" / "Compare yesterday and today"
210
+
211
+ ### Aggregation/Summary Requests
212
+ - "Summarize this" / "What's the total?" / "What's the average?"
213
+ - "Show by category" / "Show statistics" / "How many?"
214
+
215
+ ### Filtering/Search Patterns
216
+ - "Show only above 100" / "Find specific conditions" / "Top 10"
217
+
218
+ ### Comparison/Correlation
219
+ - "Compare A and B" / "What's the difference?" / "Is there a correlation?" / "Merge two files"
220
+
221
+ ### Transformation/Cleaning
222
+ - "Clean this up" / "Remove duplicates" / "Handle missing values" / "Convert format"
223
+
224
+ ### Technical Patterns
225
+ - Working with CSV, Parquet, JSON, NDJSON, or `.duckdb` files
226
+ - File paths ending in `.csv`, `.parquet`, `.json`, `.jsonl`, `.ndjson`, `.tsv`, `.duckdb`
227
+ - Requests involving calculations or aggregations
228
+ - Joining, filtering, sorting, or transforming datasets
229
+ - Processing large datasets that may exceed memory
230
+ - Comparing or analyzing data from multiple sources
231
+ - Performance-critical data operations
232
+ - SQL queries or DataFrame operations mentioned
233
+
234
+ ### When NOT to Activate
235
+ - Simple file reading for text/code inspection (use the harness's file-read surface)
236
+ - Non-data files (images, videos, binaries)
237
+ - Configuration files (YAML, TOML, JSON configs) unless specifically for data analysis
238
+ - Small inline calculations (run them directly)
239
+ - Excel files — convert to CSV/Parquet first
240
+
241
+ ---
242
+
243
+ **Core execution principle:** Always apply intelligent tool selection based on operation characteristics, never use pandas, and always include numpy and pyarrow in the execution environment.
@@ -0,0 +1,2 @@
1
+ interface:
2
+ display_name: "(OmO) data-scientist"
@@ -0,0 +1,176 @@
1
+ # Common Scenarios with Tool Selection
2
+
3
+ ## Scenario 1: Simple Calculation
4
+
5
+ **Decision: Python (always)**
6
+
7
+ ```python
8
+ uv run --with numpy python -c "
9
+ import numpy as np
10
+ result = np.sum([1, 2, 3, 4, 5])
11
+ print(f'Result: {result}')
12
+ "
13
+ ```
14
+
15
+ ## Scenario 2: CSV Quick Analysis
16
+
17
+ **Decision: DuckDB (direct query, no memory load)**
18
+
19
+ ```python
20
+ uv run --with numpy --with duckdb python -c "
21
+ import duckdb
22
+ result = duckdb.sql('''
23
+ SELECT * FROM 'data.csv'
24
+ LIMIT 10
25
+ ''').pl()
26
+ print(result)
27
+ "
28
+ ```
29
+
30
+ ## Scenario 3: Filter + Sort on Large Dataset
31
+
32
+ **Decision: Polars (128x faster filtering, 12x faster sorting)**
33
+
34
+ ```python
35
+ uv run --with numpy --with polars python -c "
36
+ import polars as pl
37
+ result = (
38
+ pl.scan_csv('large.csv')
39
+ .filter(pl.col('value') > 1000)
40
+ .sort('value', descending=True)
41
+ .head(100)
42
+ .collect()
43
+ )
44
+ print(result)
45
+ "
46
+ ```
47
+
48
+ ## Scenario 4: Multi-Table Join + Aggregation
49
+
50
+ **Decision: DuckDB (best for joins and aggregations)**
51
+
52
+ ```python
53
+ uv run --with numpy --with duckdb python -c "
54
+ import duckdb
55
+ result = duckdb.sql('''
56
+ SELECT
57
+ a.category,
58
+ COUNT(*) as count,
59
+ SUM(b.amount) as total
60
+ FROM 'table1.csv' a
61
+ JOIN 'table2.csv' b ON a.id = b.id
62
+ GROUP BY a.category
63
+ ORDER BY total DESC
64
+ ''').pl()
65
+ print(result)
66
+ "
67
+ ```
68
+
69
+ ## Scenario 5: Data Exploration
70
+
71
+ **Decision: DuckDB for quick exploration**
72
+
73
+ ```python
74
+ uv run --with numpy --with duckdb python -c "
75
+ import duckdb
76
+
77
+ # Show first few rows
78
+ print('**Sample Data**')
79
+ print(duckdb.sql('SELECT * FROM \"data.csv\" LIMIT 5').pl())
80
+
81
+ # Show summary statistics
82
+ print('\\n**Summary Statistics**')
83
+ print(duckdb.sql('DESCRIBE SELECT * FROM \"data.csv\"').pl())
84
+
85
+ # Show row count
86
+ print('\\n**Row Count**')
87
+ print(duckdb.sql('SELECT COUNT(*) as total_rows FROM \"data.csv\"').pl())
88
+ "
89
+ ```
90
+
91
+ ## Scenario 6: Time-Series Analysis
92
+
93
+ **Decision: DuckDB for aggregation + matplotlib for visualization**
94
+
95
+ ```python
96
+ uv run --with numpy --with duckdb --with pyarrow --with matplotlib python -c "
97
+ import duckdb
98
+ import matplotlib.pyplot as plt
99
+
100
+ # Aggregate by date
101
+ result = duckdb.sql('''
102
+ SELECT
103
+ DATE_TRUNC('day', timestamp) as date,
104
+ COUNT(*) as count,
105
+ AVG(value) as avg_value
106
+ FROM 'timeseries.csv'
107
+ GROUP BY date
108
+ ORDER BY date
109
+ ''').pl()
110
+
111
+ # Plot
112
+ plt.figure(figsize=(12, 6))
113
+ plt.subplot(2, 1, 1)
114
+ plt.plot(result['date'], result['count'])
115
+ plt.title('Daily Count')
116
+
117
+ plt.subplot(2, 1, 2)
118
+ plt.plot(result['date'], result['avg_value'])
119
+ plt.title('Daily Average Value')
120
+
121
+ plt.tight_layout()
122
+ plt.savefig('timeseries.png')
123
+ print('Saved to timeseries.png')
124
+ "
125
+ ```
126
+
127
+ ## Scenario 7: Complex Transformation
128
+
129
+ **Decision: Polars for efficient transformations**
130
+
131
+ ```python
132
+ uv run --with numpy --with polars python -c "
133
+ import polars as pl
134
+
135
+ result = (
136
+ pl.scan_csv('data.csv')
137
+ .with_columns([
138
+ # Create new calculated columns
139
+ (pl.col('price') * pl.col('quantity')).alias('total'),
140
+ pl.col('date').str.strptime(pl.Date, '%Y-%m-%d').alias('parsed_date'),
141
+ pl.col('name').str.to_uppercase().alias('upper_name'),
142
+ ])
143
+ .filter(pl.col('total') > 100)
144
+ .select(['parsed_date', 'upper_name', 'total'])
145
+ .collect()
146
+ )
147
+
148
+ print(result)
149
+ "
150
+ ```
151
+
152
+ ## Scenario 8: Large File Processing
153
+
154
+ **Decision: Polars streaming mode**
155
+
156
+ ```python
157
+ uv run --with numpy --with polars python -c "
158
+ import polars as pl
159
+
160
+ # Process file larger than RAM
161
+ result = (
162
+ pl.scan_csv('huge_file.csv')
163
+ .filter(pl.col('active') == True)
164
+ .groupby('category')
165
+ .agg([
166
+ pl.count().alias('count'),
167
+ pl.sum('amount').alias('total'),
168
+ pl.mean('amount').alias('average'),
169
+ ])
170
+ .collect(streaming=True) # Streaming mode
171
+ )
172
+
173
+ print(result)
174
+ print(f'\\nProcessed {result[\"count\"].sum():,} rows')
175
+ "
176
+ ```