forge-workflow 0.0.4 → 0.0.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/commands/dev.md +345 -340
- package/.claude/commands/plan.md +566 -521
- package/.claude/commands/premerge.md +186 -176
- package/.claude/commands/research.md +42 -42
- package/.claude/commands/review.md +448 -442
- package/.claude/commands/rollback.md +721 -721
- package/.claude/commands/ship.md +212 -164
- package/.claude/commands/sonarcloud.md +152 -152
- package/.claude/commands/status.md +90 -48
- package/.claude/commands/validate.md +288 -282
- package/.claude/commands/verify.md +269 -221
- package/.claude/rules/greptile-review-process.md +285 -285
- package/.claude/rules/workflow.md +121 -105
- package/.claude/scripts/greptile-resolve.sh +558 -526
- package/.claude/scripts/load-env.sh +32 -32
- package/.cline/workflows/dev.md +342 -337
- package/.cline/workflows/plan.md +563 -518
- package/.cline/workflows/premerge.md +183 -173
- package/.cline/workflows/research.md +39 -39
- package/.cline/workflows/review.md +445 -439
- package/.cline/workflows/rollback.md +718 -718
- package/.cline/workflows/ship.md +209 -161
- package/.cline/workflows/sonarcloud.md +146 -146
- package/.cline/workflows/status.md +87 -45
- package/.cline/workflows/validate.md +285 -279
- package/.cline/workflows/verify.md +266 -218
- package/.codex/config.toml +11 -11
- package/.codex/skills/dev/SKILL.md +345 -340
- package/.codex/skills/plan/SKILL.md +566 -521
- package/.codex/skills/premerge/SKILL.md +186 -176
- package/.codex/skills/research/SKILL.md +42 -42
- package/.codex/skills/review/SKILL.md +448 -442
- package/.codex/skills/rollback/SKILL.md +721 -721
- package/.codex/skills/ship/SKILL.md +212 -164
- package/.codex/skills/sonarcloud/SKILL.md +149 -149
- package/.codex/skills/status/SKILL.md +90 -48
- package/.codex/skills/validate/SKILL.md +288 -282
- package/.codex/skills/verify/SKILL.md +269 -221
- package/.cursor/commands/dev.md +342 -337
- package/.cursor/commands/plan.md +563 -518
- package/.cursor/commands/premerge.md +183 -173
- package/.cursor/commands/research.md +39 -39
- package/.cursor/commands/review.md +445 -439
- package/.cursor/commands/rollback.md +718 -718
- package/.cursor/commands/ship.md +209 -161
- package/.cursor/commands/sonarcloud.md +146 -146
- package/.cursor/commands/status.md +87 -45
- package/.cursor/commands/validate.md +285 -279
- package/.cursor/commands/verify.md +266 -218
- package/.cursor/rules/permissions-guidance.mdc +37 -37
- package/.forge/hooks/check-tdd.js +240 -240
- package/.github/PLUGIN_TEMPLATE.json +32 -32
- package/.github/prompts/dev.prompt.md +347 -342
- package/.github/prompts/plan.prompt.md +568 -523
- package/.github/prompts/premerge.prompt.md +188 -178
- package/.github/prompts/research.prompt.md +44 -44
- package/.github/prompts/review.prompt.md +450 -444
- package/.github/prompts/rollback.prompt.md +723 -723
- package/.github/prompts/ship.prompt.md +214 -166
- package/.github/prompts/sonarcloud.prompt.md +151 -151
- package/.github/prompts/status.prompt.md +92 -50
- package/.github/prompts/validate.prompt.md +290 -284
- package/.github/prompts/verify.prompt.md +271 -223
- package/.github/workflows/beads-to-github.yml +56 -0
- package/.github/workflows/github-to-beads.yml +97 -0
- package/.kilocode/workflows/dev.md +346 -341
- package/.kilocode/workflows/plan.md +567 -522
- package/.kilocode/workflows/premerge.md +187 -177
- package/.kilocode/workflows/research.md +43 -43
- package/.kilocode/workflows/review.md +449 -443
- package/.kilocode/workflows/rollback.md +722 -722
- package/.kilocode/workflows/ship.md +213 -165
- package/.kilocode/workflows/sonarcloud.md +150 -150
- package/.kilocode/workflows/status.md +91 -49
- package/.kilocode/workflows/validate.md +289 -283
- package/.kilocode/workflows/verify.md +270 -222
- package/.mcp.json.example +12 -12
- package/.opencode/commands/dev.md +345 -340
- package/.opencode/commands/plan.md +566 -521
- package/.opencode/commands/premerge.md +186 -176
- package/.opencode/commands/research.md +42 -42
- package/.opencode/commands/review.md +448 -442
- package/.opencode/commands/rollback.md +721 -721
- package/.opencode/commands/ship.md +212 -164
- package/.opencode/commands/sonarcloud.md +149 -149
- package/.opencode/commands/status.md +90 -48
- package/.opencode/commands/validate.md +288 -282
- package/.opencode/commands/verify.md +269 -221
- package/.roo/commands/dev.md +346 -341
- package/.roo/commands/plan.md +567 -522
- package/.roo/commands/premerge.md +187 -177
- package/.roo/commands/research.md +43 -43
- package/.roo/commands/review.md +449 -443
- package/.roo/commands/rollback.md +722 -722
- package/.roo/commands/ship.md +213 -165
- package/.roo/commands/sonarcloud.md +150 -150
- package/.roo/commands/status.md +91 -49
- package/.roo/commands/validate.md +289 -283
- package/.roo/commands/verify.md +270 -222
- package/AGENTS.md +272 -175
- package/CLAUDE.md +110 -100
- package/README.md +429 -416
- package/bin/forge-cmd.js +317 -313
- package/bin/forge-preflight.js +322 -309
- package/bin/forge.js +4765 -4303
- package/docs/AGENT_INSTALL_PROMPT.md +342 -342
- package/docs/BEADS_GITHUB_SYNC.md +251 -251
- package/docs/ENHANCED_ONBOARDING.md +612 -602
- package/docs/EXAMPLES.md +482 -482
- package/docs/GREPTILE_SETUP.md +400 -400
- package/docs/MANUAL_REVIEW_GUIDE.md +106 -106
- package/docs/ROADMAP.md +359 -359
- package/docs/SETUP.md +663 -631
- package/docs/TOOLCHAIN.md +653 -630
- package/docs/VALIDATION.md +363 -363
- package/install.sh +40 -1056
- package/lefthook.yml +50 -39
- package/lib/agents/README.md +198 -198
- package/lib/agents/claude.plugin.json +28 -28
- package/lib/agents/cline.plugin.json +22 -22
- package/lib/agents/codex.plugin.json +19 -19
- package/lib/agents/copilot.plugin.json +24 -24
- package/lib/agents/cursor.plugin.json +25 -25
- package/lib/agents/kilocode.plugin.json +22 -22
- package/lib/agents/opencode.plugin.json +20 -20
- package/lib/agents/roo.plugin.json +23 -23
- package/lib/agents-config.js +2112 -2112
- package/lib/beads-health-check.js +143 -0
- package/lib/beads-setup.js +341 -0
- package/lib/beads-sync-scaffold.js +260 -0
- package/lib/commands/_registry.js +134 -0
- package/lib/commands/clean.js +181 -0
- package/lib/commands/dev.js +571 -513
- package/lib/commands/plan.js +692 -692
- package/lib/commands/push.js +196 -0
- package/lib/commands/recommend.js +119 -119
- package/lib/commands/ship.js +377 -377
- package/lib/commands/status.js +378 -378
- package/lib/commands/sync.js +55 -0
- package/lib/commands/team.js +37 -0
- package/lib/commands/test.js +207 -0
- package/lib/commands/validate.js +602 -602
- package/lib/commands/worktree.js +310 -0
- package/lib/context-merge.js +359 -359
- package/lib/dep-guard/analyzer.js +294 -294
- package/lib/dep-guard/behavior-detector.js +98 -98
- package/lib/dep-guard/contract-detector.js +162 -162
- package/lib/dep-guard/import-detector.js +498 -498
- package/lib/dep-guard/path-utils.js +13 -13
- package/lib/dep-guard/rubric.js +120 -120
- package/lib/dep-guard/task-parser.js +318 -318
- package/lib/detect-agent.js +191 -191
- package/lib/detect-worktree.js +47 -47
- package/lib/docs-command.js +51 -0
- package/lib/docs-copy.js +50 -0
- package/lib/file-hash.js +26 -26
- package/lib/freshness-token.js +148 -0
- package/lib/greptile-match.js +80 -0
- package/lib/husky-migration.js +450 -0
- package/lib/lefthook-check.js +65 -0
- package/lib/pat-setup.js +207 -0
- package/lib/plugin-catalog.js +350 -350
- package/lib/plugin-manager.js +166 -166
- package/lib/plugin-recommender.js +141 -141
- package/lib/project-discovery.js +491 -491
- package/lib/reset.js +309 -0
- package/lib/setup-action-log.js +139 -139
- package/lib/setup-summary-renderer.js +106 -106
- package/lib/setup-utils.js +96 -0
- package/lib/setup.js +192 -192
- package/lib/smart-merge.js +64 -0
- package/lib/symlink-utils.js +81 -0
- package/lib/task-ownership.js +117 -0
- package/lib/workflow-profiles.js +197 -197
- package/package.json +131 -128
- package/scripts/beads-context.sh +426 -0
- package/scripts/beads-context.test.js +567 -0
- package/scripts/behavioral-judge.sh +378 -0
- package/scripts/benchmark.js +85 -0
- package/scripts/branch-protection.js +183 -0
- package/scripts/check-agents.js +172 -0
- package/scripts/check-forge-token.js +98 -0
- package/scripts/commitlint.js +42 -0
- package/scripts/conflict-detect.sh +323 -0
- package/scripts/dep-guard-analyze.js +71 -0
- package/scripts/dep-guard.sh +789 -0
- package/scripts/eval_win.py +249 -0
- package/scripts/file-index.sh +493 -0
- package/scripts/forge-team/index.sh +86 -0
- package/scripts/forge-team/lib/agent-prompt.sh +52 -0
- package/scripts/forge-team/lib/claim.sh +256 -0
- package/scripts/forge-team/lib/dashboard.sh +341 -0
- package/scripts/forge-team/lib/epic.sh +332 -0
- package/scripts/forge-team/lib/hooks.sh +253 -0
- package/scripts/forge-team/lib/identity.sh +235 -0
- package/scripts/forge-team/lib/sync-github.sh +317 -0
- package/scripts/forge-team/lib/verify.sh +284 -0
- package/scripts/forge-team/lib/workload.sh +296 -0
- package/scripts/forge-team/tests/agent-prompt.test.sh +72 -0
- package/scripts/forge-team/tests/claim.test.sh +179 -0
- package/scripts/forge-team/tests/dashboard.test.sh +170 -0
- package/scripts/forge-team/tests/dispatcher.test.sh +79 -0
- package/scripts/forge-team/tests/epic.test.sh +176 -0
- package/scripts/forge-team/tests/hooks.test.sh +239 -0
- package/scripts/forge-team/tests/identity.test.sh +176 -0
- package/scripts/forge-team/tests/integration.test.sh +371 -0
- package/scripts/forge-team/tests/sync-github.test.sh +209 -0
- package/scripts/forge-team/tests/verify.test.sh +314 -0
- package/scripts/forge-team/tests/workflow-integration.test.sh +43 -0
- package/scripts/forge-team/tests/workload.test.sh +209 -0
- package/scripts/github-beads-sync/comment.mjs +64 -0
- package/scripts/github-beads-sync/config.mjs +148 -0
- package/scripts/github-beads-sync/github-api.mjs +131 -0
- package/scripts/github-beads-sync/index.mjs +332 -0
- package/scripts/github-beads-sync/label-mapper.mjs +54 -0
- package/scripts/github-beads-sync/mapping.mjs +78 -0
- package/scripts/github-beads-sync/reverse-sync-cli.mjs +31 -0
- package/scripts/github-beads-sync/reverse-sync.mjs +138 -0
- package/scripts/github-beads-sync/run-bd.mjs +159 -0
- package/scripts/github-beads-sync/sanitize.mjs +121 -0
- package/scripts/github-beads-sync.config.json +26 -0
- package/scripts/improve-command.js +375 -0
- package/scripts/lib/eval-runner.js +268 -0
- package/scripts/lib/eval-schema.js +135 -0
- package/scripts/lib/eval-storage.js +78 -0
- package/scripts/lib/grading.js +203 -0
- package/scripts/lib/jsonl-lock.sh +48 -0
- package/scripts/lib/sanitize.sh +116 -0
- package/scripts/lib/transcript-parser.js +63 -0
- package/scripts/lint.js +47 -0
- package/scripts/migrate-to-bun-test.js +412 -0
- package/scripts/pr-coordinator.sh +706 -0
- package/scripts/run-command-eval.js +236 -0
- package/scripts/smart-status.sh +809 -0
- package/scripts/sync-commands.js +571 -0
- package/scripts/sync-utils.sh +455 -0
- package/scripts/test-dashboard.js +123 -0
- package/scripts/test.js +46 -0
- package/scripts/validate.sh +94 -0
- package/skills/parallel-deep-research/SKILL.md +108 -108
- package/skills/parallel-deep-research/evals/README.md +27 -27
- package/skills/parallel-deep-research/evals/evals.json +62 -62
- package/skills/sonarcloud-analysis/SKILL.md +171 -171
- package/skills/sonarcloud-analysis/evals/README.md +27 -27
- package/skills/sonarcloud-analysis/evals/evals.json +50 -50
- package/skills/sonarcloud-analysis/references/api-reference.md +466 -466
- package/.cursor/hooks/state/continual-learning-index.json +0 -19
- package/.cursor/hooks/state/continual-learning.json +0 -8
|
@@ -0,0 +1,249 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Windows-compatible trigger evaluation for skill descriptions.
|
|
3
|
+
|
|
4
|
+
Tests whether a skill's description causes Claude to trigger (invoke the Skill
|
|
5
|
+
tool for) the REAL skill when processing a query. Skills must be discoverable
|
|
6
|
+
in .claude/skills/ for this to work.
|
|
7
|
+
|
|
8
|
+
Avoids select.select() (Unix-only on pipes) and ProcessPoolExecutor (crashes
|
|
9
|
+
on Windows with paging file errors). Runs queries sequentially using
|
|
10
|
+
subprocess.communicate().
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import argparse
|
|
14
|
+
import json
|
|
15
|
+
import os
|
|
16
|
+
import subprocess
|
|
17
|
+
import sys
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def parse_skill_md(skill_path: Path) -> tuple:
|
|
22
|
+
"""Parse a SKILL.md file, returning (name, description, full_content)."""
|
|
23
|
+
content = (skill_path / "SKILL.md").read_text(encoding="utf-8")
|
|
24
|
+
lines = content.split("\n")
|
|
25
|
+
|
|
26
|
+
if lines[0].strip() != "---":
|
|
27
|
+
raise ValueError("SKILL.md missing frontmatter (no opening ---)")
|
|
28
|
+
|
|
29
|
+
end_idx = None
|
|
30
|
+
for i, line in enumerate(lines[1:], start=1):
|
|
31
|
+
if line.strip() == "---":
|
|
32
|
+
end_idx = i
|
|
33
|
+
break
|
|
34
|
+
|
|
35
|
+
if end_idx is None:
|
|
36
|
+
raise ValueError("SKILL.md missing frontmatter (no closing ---)")
|
|
37
|
+
|
|
38
|
+
name = ""
|
|
39
|
+
description = ""
|
|
40
|
+
frontmatter_lines = lines[1:end_idx]
|
|
41
|
+
i = 0
|
|
42
|
+
while i < len(frontmatter_lines):
|
|
43
|
+
line = frontmatter_lines[i]
|
|
44
|
+
if line.startswith("name:"):
|
|
45
|
+
name = line[len("name:"):].strip().strip('"').strip("'")
|
|
46
|
+
elif line.startswith("description:"):
|
|
47
|
+
value = line[len("description:"):].strip()
|
|
48
|
+
if value in (">", "|", ">-", "|-"):
|
|
49
|
+
continuation_lines = []
|
|
50
|
+
i += 1
|
|
51
|
+
while i < len(frontmatter_lines) and (
|
|
52
|
+
frontmatter_lines[i].startswith(" ")
|
|
53
|
+
or frontmatter_lines[i].startswith("\t")
|
|
54
|
+
):
|
|
55
|
+
continuation_lines.append(frontmatter_lines[i].strip())
|
|
56
|
+
i += 1
|
|
57
|
+
description = " ".join(continuation_lines)
|
|
58
|
+
continue
|
|
59
|
+
else:
|
|
60
|
+
description = value.strip('"').strip("'")
|
|
61
|
+
i += 1
|
|
62
|
+
|
|
63
|
+
return name, description, content
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def find_project_root() -> Path:
|
|
67
|
+
"""Find the project root by walking up from cwd looking for .claude/."""
|
|
68
|
+
current = Path.cwd()
|
|
69
|
+
for parent in [current, *current.parents]:
|
|
70
|
+
if (parent / ".claude").is_dir():
|
|
71
|
+
return parent
|
|
72
|
+
return current
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
# Known alias pairs: legacy command name → current skill name
|
|
76
|
+
_KNOWN_ALIASES = {
|
|
77
|
+
"sonarcloud-analysis": "sonarcloud",
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def run_single_query(
|
|
82
|
+
query: str,
|
|
83
|
+
skill_name: str,
|
|
84
|
+
timeout: int,
|
|
85
|
+
project_root: str,
|
|
86
|
+
model=None,
|
|
87
|
+
) -> bool:
|
|
88
|
+
"""Run a single query and return whether the REAL skill was triggered.
|
|
89
|
+
|
|
90
|
+
Checks if Claude invokes the Skill tool with the skill's name.
|
|
91
|
+
No temp command files — relies on .claude/skills/ discovery.
|
|
92
|
+
"""
|
|
93
|
+
cmd = [
|
|
94
|
+
"claude",
|
|
95
|
+
"-p", query,
|
|
96
|
+
"--output-format", "stream-json",
|
|
97
|
+
"--verbose",
|
|
98
|
+
]
|
|
99
|
+
if model:
|
|
100
|
+
cmd.extend(["--model", model])
|
|
101
|
+
|
|
102
|
+
env = {k: v for k, v in os.environ.items() if k != "CLAUDECODE"}
|
|
103
|
+
|
|
104
|
+
try:
|
|
105
|
+
result = subprocess.run(
|
|
106
|
+
cmd,
|
|
107
|
+
capture_output=True,
|
|
108
|
+
timeout=timeout,
|
|
109
|
+
cwd=project_root,
|
|
110
|
+
env=env,
|
|
111
|
+
)
|
|
112
|
+
output = result.stdout.decode("utf-8", errors="replace")
|
|
113
|
+
except subprocess.TimeoutExpired:
|
|
114
|
+
return False
|
|
115
|
+
|
|
116
|
+
# Parse output for skill triggering — scan ALL assistant messages
|
|
117
|
+
for line in output.split("\n"):
|
|
118
|
+
line = line.strip()
|
|
119
|
+
if not line:
|
|
120
|
+
continue
|
|
121
|
+
try:
|
|
122
|
+
event = json.loads(line)
|
|
123
|
+
except (json.JSONDecodeError, UnicodeDecodeError):
|
|
124
|
+
continue
|
|
125
|
+
|
|
126
|
+
if event.get("type") == "assistant":
|
|
127
|
+
message = event.get("message", {})
|
|
128
|
+
for content_item in message.get("content", []):
|
|
129
|
+
if content_item.get("type") != "tool_use":
|
|
130
|
+
continue
|
|
131
|
+
tool_name = content_item.get("name", "")
|
|
132
|
+
tool_input = content_item.get("input", {})
|
|
133
|
+
if tool_name == "Skill":
|
|
134
|
+
invoked_skill = tool_input.get("skill", "")
|
|
135
|
+
if skill_name == invoked_skill:
|
|
136
|
+
return True
|
|
137
|
+
if invoked_skill == _KNOWN_ALIASES.get(skill_name):
|
|
138
|
+
return True
|
|
139
|
+
|
|
140
|
+
return False
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def main():
|
|
144
|
+
parser = argparse.ArgumentParser(
|
|
145
|
+
description="Windows-compatible trigger eval for skill descriptions"
|
|
146
|
+
)
|
|
147
|
+
parser.add_argument("--eval-set", required=True, help="Path to eval set JSON")
|
|
148
|
+
parser.add_argument("--skill-path", required=True, help="Path to skill directory")
|
|
149
|
+
parser.add_argument("--timeout", type=int, default=60, help="Timeout per query (s)")
|
|
150
|
+
parser.add_argument("--runs-per-query", type=int, default=1, help="Runs per query")
|
|
151
|
+
parser.add_argument(
|
|
152
|
+
"--trigger-threshold", type=float, default=0.5, help="Trigger rate threshold"
|
|
153
|
+
)
|
|
154
|
+
parser.add_argument("--model", default=None, help="Model override")
|
|
155
|
+
parser.add_argument("--verbose", action="store_true", help="Print progress")
|
|
156
|
+
args = parser.parse_args()
|
|
157
|
+
|
|
158
|
+
eval_set = json.loads(Path(args.eval_set).read_text(encoding="utf-8"))
|
|
159
|
+
skill_path = Path(args.skill_path)
|
|
160
|
+
|
|
161
|
+
if not (skill_path / "SKILL.md").exists():
|
|
162
|
+
print(f"Error: No SKILL.md found at {skill_path}", file=sys.stderr)
|
|
163
|
+
sys.exit(1)
|
|
164
|
+
|
|
165
|
+
name, description, _ = parse_skill_md(skill_path)
|
|
166
|
+
project_root = find_project_root()
|
|
167
|
+
|
|
168
|
+
# Warn if .claude/skills/ symlink is absent — skills won't be discoverable
|
|
169
|
+
skill_link = project_root / ".claude" / "skills" / name
|
|
170
|
+
if not skill_link.exists():
|
|
171
|
+
print(
|
|
172
|
+
f"Warning: .claude/skills/{name} not found — "
|
|
173
|
+
"skills not symlinked. Trigger rates will be 0%.\n"
|
|
174
|
+
"Run 'bunx skills sync' or create the symlink manually.",
|
|
175
|
+
file=sys.stderr,
|
|
176
|
+
)
|
|
177
|
+
|
|
178
|
+
if args.verbose:
|
|
179
|
+
print(f"Skill: {name}", file=sys.stderr)
|
|
180
|
+
print(f"Description: {description[:100]}...", file=sys.stderr)
|
|
181
|
+
print(f"Project root: {project_root}", file=sys.stderr)
|
|
182
|
+
print(f"Queries: {len(eval_set)}, Runs: {args.runs_per_query}", file=sys.stderr)
|
|
183
|
+
print("", file=sys.stderr)
|
|
184
|
+
|
|
185
|
+
results = []
|
|
186
|
+
for idx, item in enumerate(eval_set):
|
|
187
|
+
query = item["query"]
|
|
188
|
+
should_trigger = item["should_trigger"]
|
|
189
|
+
triggers = []
|
|
190
|
+
|
|
191
|
+
for run_idx in range(args.runs_per_query):
|
|
192
|
+
if args.verbose:
|
|
193
|
+
print(
|
|
194
|
+
f" [{idx+1}/{len(eval_set)}] run {run_idx+1}/{args.runs_per_query}: "
|
|
195
|
+
f"{query[:60]}...",
|
|
196
|
+
file=sys.stderr,
|
|
197
|
+
)
|
|
198
|
+
try:
|
|
199
|
+
triggered = run_single_query(
|
|
200
|
+
query, name, args.timeout, str(project_root), args.model
|
|
201
|
+
)
|
|
202
|
+
except Exception as e:
|
|
203
|
+
print(f" Warning: {e}", file=sys.stderr)
|
|
204
|
+
triggered = False
|
|
205
|
+
triggers.append(triggered)
|
|
206
|
+
|
|
207
|
+
trigger_rate = sum(triggers) / len(triggers) if triggers else 0.0
|
|
208
|
+
if should_trigger:
|
|
209
|
+
did_pass = trigger_rate >= args.trigger_threshold
|
|
210
|
+
else:
|
|
211
|
+
did_pass = trigger_rate < args.trigger_threshold
|
|
212
|
+
|
|
213
|
+
results.append(
|
|
214
|
+
{
|
|
215
|
+
"query": query,
|
|
216
|
+
"should_trigger": should_trigger,
|
|
217
|
+
"trigger_rate": trigger_rate,
|
|
218
|
+
"triggers": sum(triggers),
|
|
219
|
+
"runs": len(triggers),
|
|
220
|
+
"pass": did_pass,
|
|
221
|
+
}
|
|
222
|
+
)
|
|
223
|
+
|
|
224
|
+
if args.verbose:
|
|
225
|
+
status = "PASS" if did_pass else "FAIL"
|
|
226
|
+
print(
|
|
227
|
+
f" [{status}] rate={sum(triggers)}/{len(triggers)} "
|
|
228
|
+
f"expected={should_trigger}: {query[:70]}",
|
|
229
|
+
file=sys.stderr,
|
|
230
|
+
)
|
|
231
|
+
|
|
232
|
+
passed = sum(1 for r in results if r["pass"])
|
|
233
|
+
total = len(results)
|
|
234
|
+
|
|
235
|
+
output = {
|
|
236
|
+
"skill_name": name,
|
|
237
|
+
"description": description,
|
|
238
|
+
"results": results,
|
|
239
|
+
"summary": {"total": total, "passed": passed, "failed": total - passed},
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
if args.verbose:
|
|
243
|
+
print(f"\nResults: {passed}/{total} passed", file=sys.stderr)
|
|
244
|
+
|
|
245
|
+
print(json.dumps(output, indent=2))
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
if __name__ == "__main__":
|
|
249
|
+
main()
|