axiom-coding-agent-setup 1.0.11 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/CONTEXT-MANAGEMENT.md +155 -0
- package/.agents/DEBUGGING.md +124 -0
- package/.agents/{engineering.md → ENGINEERING.md} +6 -0
- package/.agents/PERFORMANCE.md +164 -0
- package/.agents/SECURITY.md +109 -0
- package/.agents/{workflow.md → WORKFLOW.md} +7 -1
- package/.agents/skills/project-design/SKILL.md +207 -0
- package/.agents/skills/project-design/references/ARCHITECTURE.md +641 -0
- package/.agents/skills/project-design/references/PROJECT_PLAN.md +316 -0
- package/.agents/skills/skill-creator/LICENSE.txt +202 -0
- package/.agents/skills/skill-creator/SKILL.md +485 -0
- package/.agents/skills/skill-creator/agents/analyzer.md +274 -0
- package/.agents/skills/skill-creator/agents/comparator.md +202 -0
- package/.agents/skills/skill-creator/agents/grader.md +223 -0
- package/.agents/skills/skill-creator/assets/eval_review.html +146 -0
- package/.agents/skills/skill-creator/eval-viewer/generate_review.py +471 -0
- package/.agents/skills/skill-creator/eval-viewer/viewer.html +1325 -0
- package/.agents/skills/skill-creator/references/schemas.md +430 -0
- package/.agents/skills/skill-creator/scripts/__init__.py +0 -0
- package/.agents/skills/skill-creator/scripts/aggregate_benchmark.py +401 -0
- package/.agents/skills/skill-creator/scripts/generate_report.py +326 -0
- package/.agents/skills/skill-creator/scripts/improve_description.py +247 -0
- package/.agents/skills/skill-creator/scripts/package_skill.py +136 -0
- package/.agents/skills/skill-creator/scripts/quick_validate.py +103 -0
- package/.agents/skills/skill-creator/scripts/run_eval.py +310 -0
- package/.agents/skills/skill-creator/scripts/run_loop.py +328 -0
- package/.agents/skills/skill-creator/scripts/utils.py +47 -0
- package/AGENTS.md +68 -4
- package/README.md +42 -6
- package/bin/cli.js +15 -6
- package/package.json +1 -1
- package/plugin/oh-my-openagent.json +198 -0
- package/plugin/oh-my-openagent.md +49 -0
- package/skills-lock.json +6 -0
- /package/.agents/{stack.md → STACK.md} +0 -0
|
@@ -0,0 +1,328 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Run the eval + improve loop until all pass or max iterations reached.
|
|
3
|
+
|
|
4
|
+
Combines run_eval.py and improve_description.py in a loop, tracking history
|
|
5
|
+
and returning the best description found. Supports train/test split to prevent
|
|
6
|
+
overfitting.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import argparse
|
|
10
|
+
import json
|
|
11
|
+
import random
|
|
12
|
+
import sys
|
|
13
|
+
import tempfile
|
|
14
|
+
import time
|
|
15
|
+
import webbrowser
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
from scripts.generate_report import generate_html
|
|
19
|
+
from scripts.improve_description import improve_description
|
|
20
|
+
from scripts.run_eval import find_project_root, run_eval
|
|
21
|
+
from scripts.utils import parse_skill_md
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def split_eval_set(eval_set: list[dict], holdout: float, seed: int = 42) -> tuple[list[dict], list[dict]]:
|
|
25
|
+
"""Split eval set into train and test sets, stratified by should_trigger."""
|
|
26
|
+
random.seed(seed)
|
|
27
|
+
|
|
28
|
+
# Separate by should_trigger
|
|
29
|
+
trigger = [e for e in eval_set if e["should_trigger"]]
|
|
30
|
+
no_trigger = [e for e in eval_set if not e["should_trigger"]]
|
|
31
|
+
|
|
32
|
+
# Shuffle each group
|
|
33
|
+
random.shuffle(trigger)
|
|
34
|
+
random.shuffle(no_trigger)
|
|
35
|
+
|
|
36
|
+
# Calculate split points
|
|
37
|
+
n_trigger_test = max(1, int(len(trigger) * holdout))
|
|
38
|
+
n_no_trigger_test = max(1, int(len(no_trigger) * holdout))
|
|
39
|
+
|
|
40
|
+
# Split
|
|
41
|
+
test_set = trigger[:n_trigger_test] + no_trigger[:n_no_trigger_test]
|
|
42
|
+
train_set = trigger[n_trigger_test:] + no_trigger[n_no_trigger_test:]
|
|
43
|
+
|
|
44
|
+
return train_set, test_set
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def run_loop(
|
|
48
|
+
eval_set: list[dict],
|
|
49
|
+
skill_path: Path,
|
|
50
|
+
description_override: str | None,
|
|
51
|
+
num_workers: int,
|
|
52
|
+
timeout: int,
|
|
53
|
+
max_iterations: int,
|
|
54
|
+
runs_per_query: int,
|
|
55
|
+
trigger_threshold: float,
|
|
56
|
+
holdout: float,
|
|
57
|
+
model: str,
|
|
58
|
+
verbose: bool,
|
|
59
|
+
live_report_path: Path | None = None,
|
|
60
|
+
log_dir: Path | None = None,
|
|
61
|
+
) -> dict:
|
|
62
|
+
"""Run the eval + improvement loop."""
|
|
63
|
+
project_root = find_project_root()
|
|
64
|
+
name, original_description, content = parse_skill_md(skill_path)
|
|
65
|
+
current_description = description_override or original_description
|
|
66
|
+
|
|
67
|
+
# Split into train/test if holdout > 0
|
|
68
|
+
if holdout > 0:
|
|
69
|
+
train_set, test_set = split_eval_set(eval_set, holdout)
|
|
70
|
+
if verbose:
|
|
71
|
+
print(f"Split: {len(train_set)} train, {len(test_set)} test (holdout={holdout})", file=sys.stderr)
|
|
72
|
+
else:
|
|
73
|
+
train_set = eval_set
|
|
74
|
+
test_set = []
|
|
75
|
+
|
|
76
|
+
history = []
|
|
77
|
+
exit_reason = "unknown"
|
|
78
|
+
|
|
79
|
+
for iteration in range(1, max_iterations + 1):
|
|
80
|
+
if verbose:
|
|
81
|
+
print(f"\n{'='*60}", file=sys.stderr)
|
|
82
|
+
print(f"Iteration {iteration}/{max_iterations}", file=sys.stderr)
|
|
83
|
+
print(f"Description: {current_description}", file=sys.stderr)
|
|
84
|
+
print(f"{'='*60}", file=sys.stderr)
|
|
85
|
+
|
|
86
|
+
# Evaluate train + test together in one batch for parallelism
|
|
87
|
+
all_queries = train_set + test_set
|
|
88
|
+
t0 = time.time()
|
|
89
|
+
all_results = run_eval(
|
|
90
|
+
eval_set=all_queries,
|
|
91
|
+
skill_name=name,
|
|
92
|
+
description=current_description,
|
|
93
|
+
num_workers=num_workers,
|
|
94
|
+
timeout=timeout,
|
|
95
|
+
project_root=project_root,
|
|
96
|
+
runs_per_query=runs_per_query,
|
|
97
|
+
trigger_threshold=trigger_threshold,
|
|
98
|
+
model=model,
|
|
99
|
+
)
|
|
100
|
+
eval_elapsed = time.time() - t0
|
|
101
|
+
|
|
102
|
+
# Split results back into train/test by matching queries
|
|
103
|
+
train_queries_set = {q["query"] for q in train_set}
|
|
104
|
+
train_result_list = [r for r in all_results["results"] if r["query"] in train_queries_set]
|
|
105
|
+
test_result_list = [r for r in all_results["results"] if r["query"] not in train_queries_set]
|
|
106
|
+
|
|
107
|
+
train_passed = sum(1 for r in train_result_list if r["pass"])
|
|
108
|
+
train_total = len(train_result_list)
|
|
109
|
+
train_summary = {"passed": train_passed, "failed": train_total - train_passed, "total": train_total}
|
|
110
|
+
train_results = {"results": train_result_list, "summary": train_summary}
|
|
111
|
+
|
|
112
|
+
if test_set:
|
|
113
|
+
test_passed = sum(1 for r in test_result_list if r["pass"])
|
|
114
|
+
test_total = len(test_result_list)
|
|
115
|
+
test_summary = {"passed": test_passed, "failed": test_total - test_passed, "total": test_total}
|
|
116
|
+
test_results = {"results": test_result_list, "summary": test_summary}
|
|
117
|
+
else:
|
|
118
|
+
test_results = None
|
|
119
|
+
test_summary = None
|
|
120
|
+
|
|
121
|
+
history.append({
|
|
122
|
+
"iteration": iteration,
|
|
123
|
+
"description": current_description,
|
|
124
|
+
"train_passed": train_summary["passed"],
|
|
125
|
+
"train_failed": train_summary["failed"],
|
|
126
|
+
"train_total": train_summary["total"],
|
|
127
|
+
"train_results": train_results["results"],
|
|
128
|
+
"test_passed": test_summary["passed"] if test_summary else None,
|
|
129
|
+
"test_failed": test_summary["failed"] if test_summary else None,
|
|
130
|
+
"test_total": test_summary["total"] if test_summary else None,
|
|
131
|
+
"test_results": test_results["results"] if test_results else None,
|
|
132
|
+
# For backward compat with report generator
|
|
133
|
+
"passed": train_summary["passed"],
|
|
134
|
+
"failed": train_summary["failed"],
|
|
135
|
+
"total": train_summary["total"],
|
|
136
|
+
"results": train_results["results"],
|
|
137
|
+
})
|
|
138
|
+
|
|
139
|
+
# Write live report if path provided
|
|
140
|
+
if live_report_path:
|
|
141
|
+
partial_output = {
|
|
142
|
+
"original_description": original_description,
|
|
143
|
+
"best_description": current_description,
|
|
144
|
+
"best_score": "in progress",
|
|
145
|
+
"iterations_run": len(history),
|
|
146
|
+
"holdout": holdout,
|
|
147
|
+
"train_size": len(train_set),
|
|
148
|
+
"test_size": len(test_set),
|
|
149
|
+
"history": history,
|
|
150
|
+
}
|
|
151
|
+
live_report_path.write_text(generate_html(partial_output, auto_refresh=True, skill_name=name))
|
|
152
|
+
|
|
153
|
+
if verbose:
|
|
154
|
+
def print_eval_stats(label, results, elapsed):
|
|
155
|
+
pos = [r for r in results if r["should_trigger"]]
|
|
156
|
+
neg = [r for r in results if not r["should_trigger"]]
|
|
157
|
+
tp = sum(r["triggers"] for r in pos)
|
|
158
|
+
pos_runs = sum(r["runs"] for r in pos)
|
|
159
|
+
fn = pos_runs - tp
|
|
160
|
+
fp = sum(r["triggers"] for r in neg)
|
|
161
|
+
neg_runs = sum(r["runs"] for r in neg)
|
|
162
|
+
tn = neg_runs - fp
|
|
163
|
+
total = tp + tn + fp + fn
|
|
164
|
+
precision = tp / (tp + fp) if (tp + fp) > 0 else 1.0
|
|
165
|
+
recall = tp / (tp + fn) if (tp + fn) > 0 else 1.0
|
|
166
|
+
accuracy = (tp + tn) / total if total > 0 else 0.0
|
|
167
|
+
print(f"{label}: {tp+tn}/{total} correct, precision={precision:.0%} recall={recall:.0%} accuracy={accuracy:.0%} ({elapsed:.1f}s)", file=sys.stderr)
|
|
168
|
+
for r in results:
|
|
169
|
+
status = "PASS" if r["pass"] else "FAIL"
|
|
170
|
+
rate_str = f"{r['triggers']}/{r['runs']}"
|
|
171
|
+
print(f" [{status}] rate={rate_str} expected={r['should_trigger']}: {r['query'][:60]}", file=sys.stderr)
|
|
172
|
+
|
|
173
|
+
print_eval_stats("Train", train_results["results"], eval_elapsed)
|
|
174
|
+
if test_summary:
|
|
175
|
+
print_eval_stats("Test ", test_results["results"], 0)
|
|
176
|
+
|
|
177
|
+
if train_summary["failed"] == 0:
|
|
178
|
+
exit_reason = f"all_passed (iteration {iteration})"
|
|
179
|
+
if verbose:
|
|
180
|
+
print(f"\nAll train queries passed on iteration {iteration}!", file=sys.stderr)
|
|
181
|
+
break
|
|
182
|
+
|
|
183
|
+
if iteration == max_iterations:
|
|
184
|
+
exit_reason = f"max_iterations ({max_iterations})"
|
|
185
|
+
if verbose:
|
|
186
|
+
print(f"\nMax iterations reached ({max_iterations}).", file=sys.stderr)
|
|
187
|
+
break
|
|
188
|
+
|
|
189
|
+
# Improve the description based on train results
|
|
190
|
+
if verbose:
|
|
191
|
+
print(f"\nImproving description...", file=sys.stderr)
|
|
192
|
+
|
|
193
|
+
t0 = time.time()
|
|
194
|
+
# Strip test scores from history so improvement model can't see them
|
|
195
|
+
blinded_history = [
|
|
196
|
+
{k: v for k, v in h.items() if not k.startswith("test_")}
|
|
197
|
+
for h in history
|
|
198
|
+
]
|
|
199
|
+
new_description = improve_description(
|
|
200
|
+
skill_name=name,
|
|
201
|
+
skill_content=content,
|
|
202
|
+
current_description=current_description,
|
|
203
|
+
eval_results=train_results,
|
|
204
|
+
history=blinded_history,
|
|
205
|
+
model=model,
|
|
206
|
+
log_dir=log_dir,
|
|
207
|
+
iteration=iteration,
|
|
208
|
+
)
|
|
209
|
+
improve_elapsed = time.time() - t0
|
|
210
|
+
|
|
211
|
+
if verbose:
|
|
212
|
+
print(f"Proposed ({improve_elapsed:.1f}s): {new_description}", file=sys.stderr)
|
|
213
|
+
|
|
214
|
+
current_description = new_description
|
|
215
|
+
|
|
216
|
+
# Find the best iteration by TEST score (or train if no test set)
|
|
217
|
+
if test_set:
|
|
218
|
+
best = max(history, key=lambda h: h["test_passed"] or 0)
|
|
219
|
+
best_score = f"{best['test_passed']}/{best['test_total']}"
|
|
220
|
+
else:
|
|
221
|
+
best = max(history, key=lambda h: h["train_passed"])
|
|
222
|
+
best_score = f"{best['train_passed']}/{best['train_total']}"
|
|
223
|
+
|
|
224
|
+
if verbose:
|
|
225
|
+
print(f"\nExit reason: {exit_reason}", file=sys.stderr)
|
|
226
|
+
print(f"Best score: {best_score} (iteration {best['iteration']})", file=sys.stderr)
|
|
227
|
+
|
|
228
|
+
return {
|
|
229
|
+
"exit_reason": exit_reason,
|
|
230
|
+
"original_description": original_description,
|
|
231
|
+
"best_description": best["description"],
|
|
232
|
+
"best_score": best_score,
|
|
233
|
+
"best_train_score": f"{best['train_passed']}/{best['train_total']}",
|
|
234
|
+
"best_test_score": f"{best['test_passed']}/{best['test_total']}" if test_set else None,
|
|
235
|
+
"final_description": current_description,
|
|
236
|
+
"iterations_run": len(history),
|
|
237
|
+
"holdout": holdout,
|
|
238
|
+
"train_size": len(train_set),
|
|
239
|
+
"test_size": len(test_set),
|
|
240
|
+
"history": history,
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def main():
|
|
245
|
+
parser = argparse.ArgumentParser(description="Run eval + improve loop")
|
|
246
|
+
parser.add_argument("--eval-set", required=True, help="Path to eval set JSON file")
|
|
247
|
+
parser.add_argument("--skill-path", required=True, help="Path to skill directory")
|
|
248
|
+
parser.add_argument("--description", default=None, help="Override starting description")
|
|
249
|
+
parser.add_argument("--num-workers", type=int, default=10, help="Number of parallel workers")
|
|
250
|
+
parser.add_argument("--timeout", type=int, default=30, help="Timeout per query in seconds")
|
|
251
|
+
parser.add_argument("--max-iterations", type=int, default=5, help="Max improvement iterations")
|
|
252
|
+
parser.add_argument("--runs-per-query", type=int, default=3, help="Number of runs per query")
|
|
253
|
+
parser.add_argument("--trigger-threshold", type=float, default=0.5, help="Trigger rate threshold")
|
|
254
|
+
parser.add_argument("--holdout", type=float, default=0.4, help="Fraction of eval set to hold out for testing (0 to disable)")
|
|
255
|
+
parser.add_argument("--model", required=True, help="Model for improvement")
|
|
256
|
+
parser.add_argument("--verbose", action="store_true", help="Print progress to stderr")
|
|
257
|
+
parser.add_argument("--report", default="auto", help="Generate HTML report at this path (default: 'auto' for temp file, 'none' to disable)")
|
|
258
|
+
parser.add_argument("--results-dir", default=None, help="Save all outputs (results.json, report.html, log.txt) to a timestamped subdirectory here")
|
|
259
|
+
args = parser.parse_args()
|
|
260
|
+
|
|
261
|
+
eval_set = json.loads(Path(args.eval_set).read_text())
|
|
262
|
+
skill_path = Path(args.skill_path)
|
|
263
|
+
|
|
264
|
+
if not (skill_path / "SKILL.md").exists():
|
|
265
|
+
print(f"Error: No SKILL.md found at {skill_path}", file=sys.stderr)
|
|
266
|
+
sys.exit(1)
|
|
267
|
+
|
|
268
|
+
name, _, _ = parse_skill_md(skill_path)
|
|
269
|
+
|
|
270
|
+
# Set up live report path
|
|
271
|
+
if args.report != "none":
|
|
272
|
+
if args.report == "auto":
|
|
273
|
+
timestamp = time.strftime("%Y%m%d_%H%M%S")
|
|
274
|
+
live_report_path = Path(tempfile.gettempdir()) / f"skill_description_report_{skill_path.name}_{timestamp}.html"
|
|
275
|
+
else:
|
|
276
|
+
live_report_path = Path(args.report)
|
|
277
|
+
# Open the report immediately so the user can watch
|
|
278
|
+
live_report_path.write_text("<html><body><h1>Starting optimization loop...</h1><meta http-equiv='refresh' content='5'></body></html>")
|
|
279
|
+
webbrowser.open(str(live_report_path))
|
|
280
|
+
else:
|
|
281
|
+
live_report_path = None
|
|
282
|
+
|
|
283
|
+
# Determine output directory (create before run_loop so logs can be written)
|
|
284
|
+
if args.results_dir:
|
|
285
|
+
timestamp = time.strftime("%Y-%m-%d_%H%M%S")
|
|
286
|
+
results_dir = Path(args.results_dir) / timestamp
|
|
287
|
+
results_dir.mkdir(parents=True, exist_ok=True)
|
|
288
|
+
else:
|
|
289
|
+
results_dir = None
|
|
290
|
+
|
|
291
|
+
log_dir = results_dir / "logs" if results_dir else None
|
|
292
|
+
|
|
293
|
+
output = run_loop(
|
|
294
|
+
eval_set=eval_set,
|
|
295
|
+
skill_path=skill_path,
|
|
296
|
+
description_override=args.description,
|
|
297
|
+
num_workers=args.num_workers,
|
|
298
|
+
timeout=args.timeout,
|
|
299
|
+
max_iterations=args.max_iterations,
|
|
300
|
+
runs_per_query=args.runs_per_query,
|
|
301
|
+
trigger_threshold=args.trigger_threshold,
|
|
302
|
+
holdout=args.holdout,
|
|
303
|
+
model=args.model,
|
|
304
|
+
verbose=args.verbose,
|
|
305
|
+
live_report_path=live_report_path,
|
|
306
|
+
log_dir=log_dir,
|
|
307
|
+
)
|
|
308
|
+
|
|
309
|
+
# Save JSON output
|
|
310
|
+
json_output = json.dumps(output, indent=2)
|
|
311
|
+
print(json_output)
|
|
312
|
+
if results_dir:
|
|
313
|
+
(results_dir / "results.json").write_text(json_output)
|
|
314
|
+
|
|
315
|
+
# Write final HTML report (without auto-refresh)
|
|
316
|
+
if live_report_path:
|
|
317
|
+
live_report_path.write_text(generate_html(output, auto_refresh=False, skill_name=name))
|
|
318
|
+
print(f"\nReport: {live_report_path}", file=sys.stderr)
|
|
319
|
+
|
|
320
|
+
if results_dir and live_report_path:
|
|
321
|
+
(results_dir / "report.html").write_text(generate_html(output, auto_refresh=False, skill_name=name))
|
|
322
|
+
|
|
323
|
+
if results_dir:
|
|
324
|
+
print(f"Results saved to: {results_dir}", file=sys.stderr)
|
|
325
|
+
|
|
326
|
+
|
|
327
|
+
if __name__ == "__main__":
|
|
328
|
+
main()
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""Shared utilities for skill-creator scripts."""
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def parse_skill_md(skill_path: Path) -> tuple[str, str, str]:
|
|
8
|
+
"""Parse a SKILL.md file, returning (name, description, full_content)."""
|
|
9
|
+
content = (skill_path / "SKILL.md").read_text()
|
|
10
|
+
lines = content.split("\n")
|
|
11
|
+
|
|
12
|
+
if lines[0].strip() != "---":
|
|
13
|
+
raise ValueError("SKILL.md missing frontmatter (no opening ---)")
|
|
14
|
+
|
|
15
|
+
end_idx = None
|
|
16
|
+
for i, line in enumerate(lines[1:], start=1):
|
|
17
|
+
if line.strip() == "---":
|
|
18
|
+
end_idx = i
|
|
19
|
+
break
|
|
20
|
+
|
|
21
|
+
if end_idx is None:
|
|
22
|
+
raise ValueError("SKILL.md missing frontmatter (no closing ---)")
|
|
23
|
+
|
|
24
|
+
name = ""
|
|
25
|
+
description = ""
|
|
26
|
+
frontmatter_lines = lines[1:end_idx]
|
|
27
|
+
i = 0
|
|
28
|
+
while i < len(frontmatter_lines):
|
|
29
|
+
line = frontmatter_lines[i]
|
|
30
|
+
if line.startswith("name:"):
|
|
31
|
+
name = line[len("name:"):].strip().strip('"').strip("'")
|
|
32
|
+
elif line.startswith("description:"):
|
|
33
|
+
value = line[len("description:"):].strip()
|
|
34
|
+
# Handle YAML multiline indicators (>, |, >-, |-)
|
|
35
|
+
if value in (">", "|", ">-", "|-"):
|
|
36
|
+
continuation_lines: list[str] = []
|
|
37
|
+
i += 1
|
|
38
|
+
while i < len(frontmatter_lines) and (frontmatter_lines[i].startswith(" ") or frontmatter_lines[i].startswith("\t")):
|
|
39
|
+
continuation_lines.append(frontmatter_lines[i].strip())
|
|
40
|
+
i += 1
|
|
41
|
+
description = " ".join(continuation_lines)
|
|
42
|
+
continue
|
|
43
|
+
else:
|
|
44
|
+
description = value.strip('"').strip("'")
|
|
45
|
+
i += 1
|
|
46
|
+
|
|
47
|
+
return name, description, content
|
package/AGENTS.md
CHANGED
|
@@ -12,9 +12,13 @@ I write code that survives contact with reality.
|
|
|
12
12
|
|
|
13
13
|
## Core Documents
|
|
14
14
|
|
|
15
|
-
- @/.agents/
|
|
16
|
-
- @/.agents/
|
|
17
|
-
- @/.agents/
|
|
15
|
+
- @/.agents/ENGINEERING.md — Principles, decision framework, anti-patterns, code standards
|
|
16
|
+
- @/.agents/STACK.md — Technology knowledge: languages, frameworks, infrastructure, AI/ML
|
|
17
|
+
- @/.agents/WORKFLOW.md — Work protocol, verification rules, git discipline, communication
|
|
18
|
+
- @/.agents/SECURITY.md — Security-first principles, checklist, attack vectors
|
|
19
|
+
- @/.agents/DEBUGGING.md — Systematic debugging methodology and anti-patterns
|
|
20
|
+
- @/.agents/PERFORMANCE.md — Performance awareness, measurement, optimization hierarchy
|
|
21
|
+
- @/.agents/CONTEXT-MANAGEMENT.md — Context budget, compaction strategy, session discipline
|
|
18
22
|
- @/.agents/templates/ — Project-type specific conventions and setup guides
|
|
19
23
|
- @/.agents/skills/ — Domain-specific skills for specialized tasks
|
|
20
24
|
|
|
@@ -36,5 +40,65 @@ I write code that survives contact with reality.
|
|
|
36
40
|
- NEVER declare "it works" without verification
|
|
37
41
|
- NEVER add dependencies without checking if built-ins suffice
|
|
38
42
|
- NEVER over-engineer for a scale that doesn't exist yet
|
|
39
|
-
- ALWAYS ask: "Does this solve the actual problem?"
|
|
43
|
+
- ALWAYS ask: "Does this solve the actual problem?".
|
|
40
44
|
- **ALWAYS reference the current date/time**: Today is {Month} {Year}. You search the current date in your operations and always remember that. When performing web searches or any time-sensitive queries, explicitly use the current year to avoid retrieving outdated results from previous years like 2024 or 2025.
|
|
45
|
+
- **ALWAYS analyze the current terminal/shell before running commands** — never mix syntax from different shells (e.g., `set` with `&&` in PowerShell, or `$env:` in CMD). See Terminal Awareness below.
|
|
46
|
+
- **NEVER retry the same failed command blindly** — if a command fails, stop, analyze the error, fix the root cause, then retry. Never enter infinite retry loops.
|
|
47
|
+
- **ALWAYS stop and ask when blocked** — if you've spent 3+ attempts on the same problem without progress, escalate to the human with: what you tried, what failed, and what you need.
|
|
48
|
+
- **NEVER suppress type errors or lint warnings** — `as any`, `@ts-ignore`, `@ts-expect-error`, and empty catch blocks are forbidden. Fix the root cause instead.
|
|
49
|
+
- **ALWAYS verify empirically** — read files before claiming contents, run tests before declaring success, observe before describing. Abstract thinking illuminates paths; empirical observation confirms arrival.
|
|
50
|
+
- **NEVER modify security-critical code without explicit approval** — authentication, authorization, secret handling, encryption. Stop and ask.
|
|
51
|
+
- **ALWAYS think before coding** — for any non-trivial change, pause and reason through: what does the user actually want? What could go wrong? What's the simplest correct approach?
|
|
52
|
+
- **NEVER leave code in a broken state** — if you can't finish, revert to last known working state and explain what's blocked.
|
|
53
|
+
- **ALWAYS match existing patterns** — read 2–3 similar files in the codebase before writing new code. Consistency > novelty.
|
|
54
|
+
- **NEVER delete failing tests to "pass"** — a deleted test is a hidden bug. Fix the code or the test, never delete to green.
|
|
55
|
+
|
|
56
|
+
---
|
|
57
|
+
|
|
58
|
+
## Terminal Awareness
|
|
59
|
+
|
|
60
|
+
Before executing any shell command, identify the active terminal and use its correct syntax. **Never assume** — check the environment context. Mixing shell syntax produces cryptic errors and wasted retry loops.
|
|
61
|
+
|
|
62
|
+
### Common Shells & Their Syntax
|
|
63
|
+
|
|
64
|
+
| Shell | Environment Variables | Command Chaining | Example |
|
|
65
|
+
| ------------------------ | --------------------- | ---------------------- | ----------------------------------- |
|
|
66
|
+
| **PowerShell** | `$env:VAR = "value"` | `;` (or `&&` in PS 7+) | `$env:CI="true"; git diff --stat` |
|
|
67
|
+
| **CMD / Command Prompt** | `set VAR=value` | `&&` | `set CI=true && git diff --stat` |
|
|
68
|
+
| **Bash / Sh / Zsh** | `export VAR=value` | `&&` or `;` | `export CI=true && git diff --stat` |
|
|
69
|
+
| **Fish** | `set -x VAR value` | `;` or `and` | `set -x CI true; git diff --stat` |
|
|
70
|
+
|
|
71
|
+
### Why This Matters
|
|
72
|
+
|
|
73
|
+
- **PowerShell** does not recognize `set` or `&&` from CMD. Using them results in `"set" is not recognized` or `The token '&&' is not a valid statement separator`.
|
|
74
|
+
- **CMD** does not recognize `$env:` syntax. Using it results in `'$env:' is not recognized`.
|
|
75
|
+
- **Bash/Sh** use `export`, not `set` (which is a built-in with different behavior) and not `$env:`.
|
|
76
|
+
|
|
77
|
+
### Practical Rule
|
|
78
|
+
|
|
79
|
+
1. **Detect the shell** before constructing a command string.
|
|
80
|
+
2. **Use the correct syntax** for that shell exclusively.
|
|
81
|
+
3. **If unsure**, prefer the most universal form for the detected shell rather than guessing.
|
|
82
|
+
4. **Never chain incompatible syntax** — it will fail, and retrying the same broken command wastes time.
|
|
83
|
+
|
|
84
|
+
### Example: What NOT to Do
|
|
85
|
+
|
|
86
|
+
```powershell
|
|
87
|
+
# WRONG: Mixing CMD 'set' and '&&' in PowerShell
|
|
88
|
+
$ set CI="true" && set GIT_TERMINAL_PROMPT="0" && git diff --stat .
|
|
89
|
+
# Result: "set" is not recognized... && is not valid...
|
|
90
|
+
|
|
91
|
+
# CORRECT: Pure PowerShell syntax
|
|
92
|
+
$ $env:CI="true"; $env:GIT_TERMINAL_PROMPT="0"; git diff --stat
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
```bash
|
|
96
|
+
# WRONG: Using PowerShell syntax in Bash
|
|
97
|
+
$ $env:CI="true"; git diff --stat
|
|
98
|
+
# Result: command not found: $env:CI=true
|
|
99
|
+
|
|
100
|
+
# CORRECT: Pure Bash syntax
|
|
101
|
+
$ export CI="true" && git diff --stat
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
---
|
package/README.md
CHANGED
|
@@ -23,9 +23,13 @@ This command downloads the following files from the [axiom-coding-agent-setup](h
|
|
|
23
23
|
- `AGENTS.md` — Main agent instructions
|
|
24
24
|
- `opencode.json` — OpenCode IDE configuration (MCP servers, plugins)
|
|
25
25
|
- `.env.axiom` — Environment variables template for AXIOM credentials
|
|
26
|
-
- `.agents/
|
|
27
|
-
- `.agents/
|
|
28
|
-
- `.agents/
|
|
26
|
+
- `.agents/ENGINEERING.md` — Engineering principles & code standards
|
|
27
|
+
- `.agents/STACK.md` — Technology stack knowledge
|
|
28
|
+
- `.agents/WORKFLOW.md` — Workflow guidelines & verification protocol
|
|
29
|
+
- `.agents/SECURITY.md` — Security-first principles & attack vector checklist
|
|
30
|
+
- `.agents/DEBUGGING.md` — Systematic debugging methodology & anti-patterns
|
|
31
|
+
- `.agents/PERFORMANCE.md` — Performance awareness & optimization hierarchy
|
|
32
|
+
- `.agents/CONTEXT-MANAGEMENT.md` — Context budget & session discipline
|
|
29
33
|
- `.agents/templates/` — Project-type specific conventions
|
|
30
34
|
- `.agents/skills/` — Domain-specific skills for specialized tasks
|
|
31
35
|
|
|
@@ -40,7 +44,7 @@ OpenCode IDE configuration including:
|
|
|
40
44
|
- Plugin configuration
|
|
41
45
|
- Environment variable references for secure credential management
|
|
42
46
|
|
|
43
|
-
### .agents/
|
|
47
|
+
### .agents/ENGINEERING.md
|
|
44
48
|
Core engineering principles including:
|
|
45
49
|
- KISS, YAGNI, DRY principles
|
|
46
50
|
- Decision framework for code reviews
|
|
@@ -48,7 +52,7 @@ Core engineering principles including:
|
|
|
48
52
|
- Anti-patterns to avoid
|
|
49
53
|
- AI-assisted development ground rules
|
|
50
54
|
|
|
51
|
-
### .agents/
|
|
55
|
+
### .agents/STACK.md
|
|
52
56
|
Technology stack knowledge covering:
|
|
53
57
|
- Languages (TypeScript, Python, Go, Rust, SQL)
|
|
54
58
|
- Frontend (React, Next.js, Tailwind, shadcn/ui)
|
|
@@ -57,13 +61,44 @@ Technology stack knowledge covering:
|
|
|
57
61
|
- AI/ML stack (LLM APIs, orchestration, observability)
|
|
58
62
|
- Infrastructure & DevOps
|
|
59
63
|
|
|
60
|
-
### .agents/
|
|
64
|
+
### .agents/WORKFLOW.md
|
|
61
65
|
Workflow guidelines including:
|
|
62
66
|
- Verification protocol (read files before claiming, test before declaring done)
|
|
63
67
|
- Git discipline
|
|
64
68
|
- Communication style
|
|
65
69
|
- Code review stance
|
|
66
70
|
- Context management for agentic sessions
|
|
71
|
+
- Error recovery & anti-loop patterns
|
|
72
|
+
|
|
73
|
+
### .agents/SECURITY.md
|
|
74
|
+
Security-first principles including:
|
|
75
|
+
- Input validation & secrets management checklist
|
|
76
|
+
- Authentication & authorization patterns
|
|
77
|
+
- Common attack vectors & prevention
|
|
78
|
+
- When to escalate security decisions to humans
|
|
79
|
+
|
|
80
|
+
### .agents/DEBUGGING.md
|
|
81
|
+
Systematic debugging methodology including:
|
|
82
|
+
- The 4-phase debugging protocol (Reproduction → Observation → Hypothesis → Fix)
|
|
83
|
+
- Debugging techniques (binary search, git bisect, rubber duck)
|
|
84
|
+
- Common bug categories & symptoms
|
|
85
|
+
- Anti-patterns to avoid (shotgun debugging, print-driven development)
|
|
86
|
+
|
|
87
|
+
### .agents/PERFORMANCE.md
|
|
88
|
+
Performance awareness including:
|
|
89
|
+
- The performance hierarchy (algorithm → database → I/O → memory → micro)
|
|
90
|
+
- Caching strategies & when (not) to cache
|
|
91
|
+
- Database query optimization
|
|
92
|
+
- Frontend Core Web Vitals
|
|
93
|
+
- Profiling & measurement tools
|
|
94
|
+
|
|
95
|
+
### .agents/CONTEXT-MANAGEMENT.md
|
|
96
|
+
Context management discipline including:
|
|
97
|
+
- The 50% rule for context compaction
|
|
98
|
+
- Session lifecycle & handoff documentation
|
|
99
|
+
- Parallel execution & context isolation
|
|
100
|
+
- Codebase navigation without context bloat
|
|
101
|
+
- Context anti-patterns
|
|
67
102
|
|
|
68
103
|
### .agents/templates/
|
|
69
104
|
Project-type specific convention files:
|
|
@@ -83,6 +118,7 @@ Domain-specific skills that can be loaded on-demand:
|
|
|
83
118
|
- `gradio/` — Gradio UI framework guides
|
|
84
119
|
- `mcp-builder/` — MCP server development guide
|
|
85
120
|
- `n8n-patterns/` — n8n workflow automation patterns
|
|
121
|
+
- `project-design/` — Project planning & architecture documentation
|
|
86
122
|
- `ui-ux-pro-max/` — Advanced UI/UX design skill
|
|
87
123
|
|
|
88
124
|
## Development
|
package/bin/cli.js
CHANGED
|
@@ -23,9 +23,13 @@ const FILES_TO_DOWNLOAD = [
|
|
|
23
23
|
// Environment variables template
|
|
24
24
|
'.env.axiom',
|
|
25
25
|
// Core agent documents
|
|
26
|
-
'.agents/
|
|
27
|
-
'.agents/
|
|
28
|
-
'.agents/
|
|
26
|
+
'.agents/ENGINEERING.md',
|
|
27
|
+
'.agents/STACK.md',
|
|
28
|
+
'.agents/WORKFLOW.md',
|
|
29
|
+
'.agents/SECURITY.md',
|
|
30
|
+
'.agents/DEBUGGING.md',
|
|
31
|
+
'.agents/PERFORMANCE.md',
|
|
32
|
+
'.agents/CONTEXT-MANAGEMENT.md',
|
|
29
33
|
// Templates (project-type conventions)
|
|
30
34
|
'.agents/templates/ai-engineering-python.md',
|
|
31
35
|
'.agents/templates/fullstack-ai-nextjs.md',
|
|
@@ -41,6 +45,7 @@ const FILES_TO_DOWNLOAD = [
|
|
|
41
45
|
'.agents/skills/gradio/SKILL.md',
|
|
42
46
|
'.agents/skills/mcp-builder/SKILL.md',
|
|
43
47
|
'.agents/skills/n8n-patterns/SKILL.md',
|
|
48
|
+
'.agents/skills/project-design/SKILL.md',
|
|
44
49
|
'.agents/skills/ui-ux-pro-max/SKILL.md'
|
|
45
50
|
];
|
|
46
51
|
|
|
@@ -137,9 +142,13 @@ async function main() {
|
|
|
137
142
|
log(' - AGENTS.md → Main agent instructions', 'cyan');
|
|
138
143
|
log(' - opencode.json → OpenCode IDE configuration', 'cyan');
|
|
139
144
|
log(' - .env.axiom → Environment variables template', 'cyan');
|
|
140
|
-
log(' - .agents/
|
|
141
|
-
log(' - .agents/
|
|
142
|
-
log(' - .agents/
|
|
145
|
+
log(' - .agents/ENGINEERING.md → Engineering principles', 'cyan');
|
|
146
|
+
log(' - .agents/STACK.md → Tech stack knowledge', 'cyan');
|
|
147
|
+
log(' - .agents/WORKFLOW.md → Workflow guidelines', 'cyan');
|
|
148
|
+
log(' - .agents/SECURITY.md → Security principles & checklist', 'cyan');
|
|
149
|
+
log(' - .agents/DEBUGGING.md → Systematic debugging methodology', 'cyan');
|
|
150
|
+
log(' - .agents/PERFORMANCE.md → Performance awareness & optimization', 'cyan');
|
|
151
|
+
log(' - .agents/CONTEXT-MANAGEMENT.md → Context budget & session discipline', 'cyan');
|
|
143
152
|
log(' - .agents/templates/ → Project-type conventions', 'cyan');
|
|
144
153
|
log(' - .agents/skills/ → Domain-specific skills\n', 'cyan');
|
|
145
154
|
}
|