axiom-coding-agent-setup 1.0.11 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/.agents/CONTEXT-MANAGEMENT.md +155 -0
  2. package/.agents/DEBUGGING.md +124 -0
  3. package/.agents/{engineering.md → ENGINEERING.md} +6 -0
  4. package/.agents/PERFORMANCE.md +164 -0
  5. package/.agents/SECURITY.md +109 -0
  6. package/.agents/{workflow.md → WORKFLOW.md} +7 -1
  7. package/.agents/skills/project-design/SKILL.md +207 -0
  8. package/.agents/skills/project-design/references/ARCHITECTURE.md +641 -0
  9. package/.agents/skills/project-design/references/PROJECT_PLAN.md +316 -0
  10. package/.agents/skills/skill-creator/LICENSE.txt +202 -0
  11. package/.agents/skills/skill-creator/SKILL.md +485 -0
  12. package/.agents/skills/skill-creator/agents/analyzer.md +274 -0
  13. package/.agents/skills/skill-creator/agents/comparator.md +202 -0
  14. package/.agents/skills/skill-creator/agents/grader.md +223 -0
  15. package/.agents/skills/skill-creator/assets/eval_review.html +146 -0
  16. package/.agents/skills/skill-creator/eval-viewer/generate_review.py +471 -0
  17. package/.agents/skills/skill-creator/eval-viewer/viewer.html +1325 -0
  18. package/.agents/skills/skill-creator/references/schemas.md +430 -0
  19. package/.agents/skills/skill-creator/scripts/__init__.py +0 -0
  20. package/.agents/skills/skill-creator/scripts/aggregate_benchmark.py +401 -0
  21. package/.agents/skills/skill-creator/scripts/generate_report.py +326 -0
  22. package/.agents/skills/skill-creator/scripts/improve_description.py +247 -0
  23. package/.agents/skills/skill-creator/scripts/package_skill.py +136 -0
  24. package/.agents/skills/skill-creator/scripts/quick_validate.py +103 -0
  25. package/.agents/skills/skill-creator/scripts/run_eval.py +310 -0
  26. package/.agents/skills/skill-creator/scripts/run_loop.py +328 -0
  27. package/.agents/skills/skill-creator/scripts/utils.py +47 -0
  28. package/AGENTS.md +68 -4
  29. package/README.md +42 -6
  30. package/bin/cli.js +15 -6
  31. package/package.json +1 -1
  32. package/plugin/oh-my-openagent.json +198 -0
  33. package/plugin/oh-my-openagent.md +49 -0
  34. package/skills-lock.json +6 -0
  35. /package/.agents/{stack.md → STACK.md} +0 -0
@@ -0,0 +1,198 @@
1
+ {
2
+ "$schema": "https://raw.githubusercontent.com/code-yeongyu/oh-my-openagent/dev/assets/oh-my-opencode.schema.json",
3
+ "agents": {
4
+ // === ORCHESTRATION CORE ===
5
+ // Kimi K2.6: #1 open-weight for long-horizon agentic work, 300-subagent swarm,
6
+ // 12h autonomous sessions, officially omo's top non-Claude fallback for Sisyphus.
7
+ // Fallback: DeepSeek V4 Pro for its raw SWE-bench muscle when K2.6 is unavailable.
8
+ "sisyphus": {
9
+ "model": "opencode-go/kimi-k2.6",
10
+ "fallback_models": [
11
+ { "model": "opencode-go/deepseek-v4-pro" }
12
+ ]
13
+ },
14
+
15
+ // === PLANNING & STRATEGY ===
16
+ // Prometheus auto-detects model family and switches prompts — Kimi K2.6 fits the
17
+ // Claude-like instruction-following prompt that omo uses for strategic planning.
18
+ // Fallback: GLM-5.1 for its independently verified long-horizon execution loop.
19
+ "prometheus": {
20
+ "model": "opencode-go/kimi-k2.6",
21
+ "fallback_models": [
22
+ { "model": "opencode-go/glm-5.1" }
23
+ ]
24
+ },
25
+
26
+ // === PLAN REVIEW ===
27
+ // Metis is the plan reviewer — needs strong reasoning + precision.
28
+ // Kimi K2.6 excels at catching non-obvious bugs and maintaining architectural integrity
29
+ // over extended review sessions per enterprise beta feedback.
30
+ // Fallback: GLM-5.1 which ranked #1 on NL2Repo for codebase structure comprehension.
31
+ "metis": {
32
+ "model": "opencode-go/kimi-k2.6",
33
+ "fallback_models": [
34
+ { "model": "opencode-go/glm-5.1" }
35
+ ]
36
+ },
37
+
38
+ // === ARCHITECTURE & DEBUGGING ===
39
+ // Oracle needs surgical precision in large codebases — Kimi K2.6 won benchmarks
40
+ // specifically on async/TypeVar edge-case bugs that require multi-cycle inference state.
41
+ // Fallback: DeepSeek V4 Pro, strongest on SWE-bench Verified (80.6%) and Codeforces (3206).
42
+ "oracle": {
43
+ "model": "opencode-go/kimi-k2.6",
44
+ "fallback_models": [
45
+ { "model": "opencode-go/deepseek-v4-pro" }
46
+ ]
47
+ },
48
+
49
+ // === HIGH-ACCURACY REVIEW ===
50
+ // Momus is the strictness reviewer — prompt is tuned for Claude-like models.
51
+ // GLM-5.1 promoted to primary: independently verified Code Arena Elo 1530 (#3 globally),
52
+ // spontaneously applied composition patterns in head-to-head tests vs K2.6.
53
+ // Fallback: Kimi K2.6 for its 12% improvement in code generation accuracy over K2.5.
54
+ "momus": {
55
+ "model": "opencode-go/glm-5.1",
56
+ "fallback_models": [
57
+ { "model": "opencode-go/kimi-k2.6" }
58
+ ]
59
+ },
60
+
61
+ // === TODO ORCHESTRATION ===
62
+ // Atlas is the task/todo orchestrator — auto-detects model family.
63
+ // Kimi K2.6 is the best fit for long multi-step coordination (300 sub-agents, 4000 steps).
64
+ // Fallback: DeepSeek V4 Pro for its $3.48/M cost efficiency at scale.
65
+ "atlas": {
66
+ "model": "opencode-go/kimi-k2.6",
67
+ "fallback_models": [
68
+ { "model": "opencode-go/deepseek-v4-pro" }
69
+ ]
70
+ },
71
+
72
+ // === DOCS & CODE SEARCH ===
73
+ // Librarian does documentation lookup and contextual code search — benefits most from
74
+ // Qwen3.6 Plus's 1M token context (vs K2.6's 256K ceiling). Handles full repo ingestion
75
+ // in a single pass. 4-6x cheaper than Claude-class for large-context read-heavy work.
76
+ // Fallback: DeepSeek V4 Pro — also 1M context, strong on LiveCodeBench (93.5%).
77
+ "librarian": {
78
+ "model": "opencode-go/qwen3.6-plus",
79
+ "fallback_models": [
80
+ { "model": "opencode-go/deepseek-v4-pro" }
81
+ ]
82
+ },
83
+
84
+ // === FAST CODEBASE GREP ===
85
+ // Explore is a fast grep/search agent — speed matters more than raw intelligence here.
86
+ // Qwen3.6 Plus runs 2-3x faster TPS than Claude Opus and has 1M context for large repos.
87
+ // Fallback: GLM-5.1 for its 55+ tokens/sec generation speed.
88
+ "explore": {
89
+ "model": "opencode-go/qwen3.6-plus",
90
+ "fallback_models": [
91
+ { "model": "opencode-go/glm-5.1" }
92
+ ]
93
+ },
94
+
95
+ // === VISION / SCREENSHOTS ===
96
+ // Multimodal-looker needs a model with solid vision. Kimi K2.6 has native multimodality
97
+ // and specifically handles visual-to-code workflows (UI designs → working code).
98
+ // No fallback: this is the only model in the stack with reliable multimodal support.
99
+ "multimodal-looker": {
100
+ "model": "opencode-go/kimi-k2.6"
101
+ },
102
+
103
+ // === JUNIOR WORKER ===
104
+ // Sisyphus-Junior handles delegated subtasks — Kimi K2.6 maintains session stability
105
+ // for parallel spawned workers (tool invocation success rate 96.60% per CodeBuddy eval).
106
+ // Fallback: DeepSeek V4 Pro for cost efficiency on high-volume parallel calls.
107
+ "sisyphus-junior": {
108
+ "model": "opencode-go/kimi-k2.6",
109
+ "fallback_models": [
110
+ { "model": "opencode-go/deepseek-v4-pro" }
111
+ ]
112
+ }
113
+ },
114
+
115
+ "categories": {
116
+ // === VISUAL / FRONTEND ENGINEERING ===
117
+ // GLM-5.1 promoted to primary: #3 globally on agentic webdev (Arena.ai Elo 1530),
118
+ // produced correct Tailwind + TypeScript components on first pass in head-to-head tests.
119
+ // Kimi K2.6 as fallback — strong on visual-to-code via native multimodal training.
120
+ "visual-engineering": {
121
+ "model": "opencode-go/glm-5.1",
122
+ "fallback_models": [
123
+ { "model": "opencode-go/kimi-k2.6" }
124
+ ]
125
+ },
126
+
127
+ // === MAXIMUM REASONING ===
128
+ // Ultrabrain is the highest-stakes category — Kimi K2.6 leads open-weight AI Index (54).
129
+ // Fallback: DeepSeek V4 Pro for its Codeforces 3206 and LiveCodeBench 93.5% supremacy.
130
+ "ultrabrain": {
131
+ "model": "opencode-go/kimi-k2.6",
132
+ "fallback_models": [
133
+ { "model": "opencode-go/deepseek-v4-pro" }
134
+ ]
135
+ },
136
+
137
+ // === DEEP / COMPLEX WORK ===
138
+ // Kimi K2.6: best open-source for long-horizon, sustained multi-step execution.
139
+ // Fallback: GLM-5.1 — demonstrated 655-iteration autonomous optimization loop,
140
+ // 8h uninterrupted task execution, strongest open-weight for backend deep dives.
141
+ "deep": {
142
+ "model": "opencode-go/kimi-k2.6",
143
+ "fallback_models": [
144
+ { "model": "opencode-go/glm-5.1" }
145
+ ]
146
+ },
147
+
148
+ // === CREATIVE / UI ARTISTRY ===
149
+ // GLM-5.1 as primary: spontaneously applies composition patterns, correct JSX on first
150
+ // pass, Code Arena voters prefer it for frontend aesthetics in head-to-head evals.
151
+ // Kimi K2.6 as fallback: Moonshot claims Awwwards-level frontend from single prompts.
152
+ "artistry": {
153
+ "model": "opencode-go/glm-5.1",
154
+ "fallback_models": [
155
+ { "model": "opencode-go/kimi-k2.6" }
156
+ ]
157
+ },
158
+
159
+ // === QUICK / TRIVIAL TASKS ===
160
+ // GLM-5.1: 55+ tokens/sec, HN devs rate it as "actually usable" for piecemeal tasks,
161
+ // compares well to GPT-5.4 for scoped, well-defined subtasks.
162
+ // No fallback needed — quick tasks should just pass or skip to the next agent.
163
+ "quick": {
164
+ "model": "opencode-go/glm-5.1"
165
+ },
166
+
167
+ // === UNSPECIFIED / MODERATE TASKS ===
168
+ // Kimi K2.6 is the safest general-purpose choice for ambiguous category routing.
169
+ // Fallback: Qwen3.6 Plus at $0.50/M input — 30x cheaper than Claude, close in SWE-bench.
170
+ "unspecified-low": {
171
+ "model": "opencode-go/kimi-k2.6",
172
+ "fallback_models": [
173
+ { "model": "opencode-go/qwen3.6-plus" }
174
+ ]
175
+ },
176
+
177
+ // === UNSPECIFIED / COMPLEX TASKS ===
178
+ // Same primary as unspecified-low — Kimi K2.6 handles both well.
179
+ // Fallback: GLM-5.1 rather than Qwen, for stronger reasoning depth on hard unknowns.
180
+ "unspecified-high": {
181
+ "model": "opencode-go/kimi-k2.6",
182
+ "fallback_models": [
183
+ { "model": "opencode-go/glm-5.1" }
184
+ ]
185
+ },
186
+
187
+ // === WRITING / DOCUMENTATION ===
188
+ // Kimi K2.6: strong instruction-following and consistent output quality.
189
+ // Fallback: GLM-5.1 — bilingual training (EN/CN) useful for mixed-language codebases
190
+ // and documentation; Qwen3.6 Plus also strong here but GLM edges it on coherence.
191
+ "writing": {
192
+ "model": "opencode-go/kimi-k2.6",
193
+ "fallback_models": [
194
+ { "model": "opencode-go/glm-5.1" }
195
+ ]
196
+ }
197
+ }
198
+ }
@@ -0,0 +1,49 @@
1
+ Congratulations! 🎉
2
+ You have successfully set up OhMyOpenAgent with your OpenCode Go subscription! Type opencode in your terminal to start using it.
3
+ Your Model Configuration
4
+ All agents have been configured to use your opencode-go/ models:
5
+ Agent Model Role
6
+ Sisyphus opencode-go/kimi-k2.6 Main ultraworker / orchestrator
7
+ Prometheus opencode-go/glm-5.1 Strategic planner
8
+ Metis opencode-go/glm-5.1 Plan reviewer
9
+ Oracle opencode-go/glm-5.1 Architecture / debugging
10
+ Momus opencode-go/glm-5.1 High-accuracy reviewer
11
+ Atlas opencode-go/kimi-k2.6 Todo orchestrator
12
+ Librarian opencode-go/qwen3.5-plus Docs / code search
13
+ Explore opencode-go/qwen3.5-plus Fast codebase grep
14
+ Multimodal Looker opencode-go/kimi-k2.6 Vision / screenshots
15
+ Verification Results
16
+ The doctor found 2 minor non-critical issues:
17
+
18
+ 1. Comment checker unavailable — Optional. Install @code-yeongyu/comment-checker if you want AI comment quality checks.
19
+ 2. GitHub CLI not authenticated — Only matters if you want GitHub automation. Run gh auth login if needed.
20
+ Your core setup is healthy! ✅
21
+ Free Advertising
22
+ Vercel — The platform for frontend developers. Ship fast, scale faster. Vercel powers some of the most performant web experiences on the internet with their edge network and framework-defined infrastructure. From personal projects to enterprise deployments, Vercel makes frontend teams unstoppable.
23
+ You can also get free advertising for your company by contributing to oh-my-openagent: https://github.com/code-yeongyu/oh-my-openagent/compare
24
+ What You Can Do Now
25
+ Here are some powerful use cases with your setup:
26
+ 3. ultrawork or ulw — Type this in your prompt and Sisyphus will orchestrate multiple agents in parallel to relentlessly execute until your task is done. Refactor an entire codebase, build a feature end-to-end, or fix thousands of lint warnings overnight.
27
+ 4. /start-work — Press Tab to enter Prometheus (Planner) mode. He'll interview you like a senior engineer, identify scope and ambiguities, and build a detailed execution plan before any code is touched.
28
+ 5. /init-deep — Run this to auto-generate hierarchical AGENTS.md files throughout your project. Your agents will instantly understand your codebase structure without you explaining it every time.
29
+ 6. Background Agents — Fire off 5+ specialist agents in parallel. Context stays lean, results come when ready.
30
+ 7. Team Mode — Enable team_mode in your config to run a lead agent + up to 8 parallel members with real-time tmux visualization. Great for security audits or plan critiques.
31
+ Quick Tips
32
+ 8. Sisyphus agent strongly recommends Opus 4.7 model. Using other models may result in significantly degraded experience. Your current fallback is kimi-k2.6, which is the best available in your OpenCode Go plan.
33
+ 9. Feeling lazy? Just include ultrawork (or ulw) in your prompt. That's it. The agent figures out the rest.
34
+ 10. Need precision? Press Tab to enter Prometheus (Planner) mode, create a work plan through an interview process, then run /start-work to execute it with full orchestration.
35
+ 11. Want your own agent-catalog setup? I can read the docs (https://raw.githubusercontent.com/code-yeongyu/oh-my-openagent/refs/heads/dev/docs/guide/agent-model-matching.md) and set it up for you after interviewing!
36
+ Authentication Note
37
+ Your OpenCode Go models (opencode-go/...) should work if your OpenCode Go subscription is active. If opencode prompts you for provider authentication on first use, simply run:
38
+ opencode auth login
39
+ and follow the OpenCode provider flow.
40
+ If you found this helpful, would you like to star the repository? It helps the project grow! ⭐
41
+ Just say yes and I'll star it for you! 🌟
42
+
43
+ ---
44
+
45
+ The file I modified is located at:
46
+ C:\Users\Cikal Merdeka\.config\opencode\oh-my-openagent.json
47
+
48
+ Claude:
49
+ https://claude.ai/chat/3c0c28cf-fc2c-4225-87e1-e5dd3c5cfb38
package/skills-lock.json CHANGED
@@ -41,6 +41,12 @@
41
41
  "skillPath": ".agents/skills/gradio/SKILL.md",
42
42
  "computedHash": "5078d8a4cc2b562731620f36f5f0d3ad054bd51f7a98baa547b04b893744823b"
43
43
  },
44
+ "skill-creator": {
45
+ "source": "anthropics/skills",
46
+ "sourceType": "github",
47
+ "skillPath": "skills/skill-creator/SKILL.md",
48
+ "computedHash": "7e3c9cd74e9e2b4828527a857170e86310f2dab5ea8030a9043df2c7e6c88857"
49
+ },
44
50
  "ui-ux-pro-max": {
45
51
  "source": "nextlevelbuilder/ui-ux-pro-max-skill",
46
52
  "sourceType": "github",
File without changes