@mrciphersmith/keryx 0.3.2 → 0.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. package/dist/cli.js +4634 -2445
  2. package/dist/core.js +66 -10
  3. package/package.json +1 -1
  4. package/src/gdskills/bundled/install-manifest.json +349 -2
  5. package/src/gdskills/bundled/rules/core/model-selection.mdc +18 -0
  6. package/src/gdskills/bundled/skills/orchestration/job-orchestrator/SKILL.md +1 -1
  7. package/src/gdskills/bundled/skills/planning/brainstorm/SKILL.md +1 -1
  8. package/src/gdskills/bundled/skills/planning/interviewer/SKILL.md +1 -1
  9. package/src/gdskills/bundled/skills/quality/deploy/SKILL.md +1 -1
  10. package/src/gdskills/bundled/skills/review/review-jev-contract/SKILL.md +193 -0
  11. package/src/gdskills/bundled/skills/review/review-orchestrator/SKILL.detail.md +81 -21
  12. package/src/gdskills/bundled/skills/review/review-orchestrator/SKILL.md +4 -4
  13. package/src/gdskills/bundled/stacks/c-cpp/agent-refs.json +4 -0
  14. package/src/gdskills/bundled/stacks/c-cpp/governance/eval.json +1777 -0
  15. package/src/gdskills/bundled/stacks/c-cpp/governance/scout.json +31 -0
  16. package/src/gdskills/bundled/stacks/c-cpp/pack.json +42 -0
  17. package/src/gdskills/bundled/stacks/c-cpp/rules/coding-style.mdc +80 -0
  18. package/src/gdskills/bundled/stacks/c-cpp/rules/patterns.mdc +87 -0
  19. package/src/gdskills/bundled/stacks/c-cpp/rules/security.mdc +90 -0
  20. package/src/gdskills/bundled/stacks/c-cpp/rules/testing.mdc +83 -0
  21. package/src/gdskills/bundled/stacks/c-cpp/skills/c-cpp-build-fix/SKILL.md +153 -0
  22. package/src/gdskills/bundled/stacks/c-cpp/skills/c-cpp-build-fix/evals.json +74 -0
  23. package/src/gdskills/bundled/stacks/c-cpp/skills/c-cpp-code-review/SKILL.md +132 -0
  24. package/src/gdskills/bundled/stacks/c-cpp/skills/c-cpp-code-review/evals.json +73 -0
  25. package/src/gdskills/bundled/stacks/c-cpp/skills/c-cpp-implementation/SKILL.md +151 -0
  26. package/src/gdskills/bundled/stacks/c-cpp/skills/c-cpp-implementation/evals.json +74 -0
  27. package/src/gdskills/bundled/stacks/c-cpp/skills/c-cpp-testing/SKILL.md +152 -0
  28. package/src/gdskills/bundled/stacks/c-cpp/skills/c-cpp-testing/evals.json +74 -0
  29. package/src/gdskills/bundled/stacks/ci-github-gitlab/agent-refs.json +4 -0
  30. package/src/gdskills/bundled/stacks/ci-github-gitlab/governance/eval.json +1295 -0
  31. package/src/gdskills/bundled/stacks/ci-github-gitlab/governance/scout.json +26 -0
  32. package/src/gdskills/bundled/stacks/ci-github-gitlab/pack.json +41 -0
  33. package/src/gdskills/bundled/stacks/ci-github-gitlab/rules/patterns.mdc +77 -0
  34. package/src/gdskills/bundled/stacks/ci-github-gitlab/rules/security.mdc +144 -0
  35. package/src/gdskills/bundled/stacks/ci-github-gitlab/skills/ci-pipeline-build-fix/SKILL.md +121 -0
  36. package/src/gdskills/bundled/stacks/ci-github-gitlab/skills/ci-pipeline-build-fix/evals.json +73 -0
  37. package/src/gdskills/bundled/stacks/ci-github-gitlab/skills/ci-pipeline-code-review/SKILL.md +139 -0
  38. package/src/gdskills/bundled/stacks/ci-github-gitlab/skills/ci-pipeline-code-review/evals.json +73 -0
  39. package/src/gdskills/bundled/stacks/ci-github-gitlab/skills/ci-pipeline-implementation/SKILL.md +147 -0
  40. package/src/gdskills/bundled/stacks/ci-github-gitlab/skills/ci-pipeline-implementation/evals.json +74 -0
  41. package/src/gdskills/bundled/stacks/docker-k8s-terraform/agent-refs.json +4 -0
  42. package/src/gdskills/bundled/stacks/docker-k8s-terraform/governance/eval.json +865 -0
  43. package/src/gdskills/bundled/stacks/docker-k8s-terraform/governance/scout.json +16 -0
  44. package/src/gdskills/bundled/stacks/docker-k8s-terraform/pack.json +46 -0
  45. package/src/gdskills/bundled/stacks/docker-k8s-terraform/rules/coding-style.mdc +74 -0
  46. package/src/gdskills/bundled/stacks/docker-k8s-terraform/rules/patterns.mdc +81 -0
  47. package/src/gdskills/bundled/stacks/docker-k8s-terraform/rules/security.mdc +146 -0
  48. package/src/gdskills/bundled/stacks/docker-k8s-terraform/rules/testing.mdc +61 -0
  49. package/src/gdskills/bundled/stacks/docker-k8s-terraform/skills/docker-k8s-terraform-build-fix/SKILL.md +151 -0
  50. package/src/gdskills/bundled/stacks/docker-k8s-terraform/skills/docker-k8s-terraform-build-fix/evals.json +74 -0
  51. package/src/gdskills/bundled/stacks/docker-k8s-terraform/skills/docker-k8s-terraform-review/SKILL.md +135 -0
  52. package/src/gdskills/bundled/stacks/docker-k8s-terraform/skills/docker-k8s-terraform-review/evals.json +76 -0
  53. package/src/gdskills/bundled/stacks/php-laravel/agent-refs.json +4 -0
  54. package/src/gdskills/bundled/stacks/php-laravel/governance/eval.json +1829 -0
  55. package/src/gdskills/bundled/stacks/php-laravel/governance/scout.json +33 -0
  56. package/src/gdskills/bundled/stacks/php-laravel/pack.json +41 -0
  57. package/src/gdskills/bundled/stacks/php-laravel/rules/coding-style.mdc +82 -0
  58. package/src/gdskills/bundled/stacks/php-laravel/rules/patterns.mdc +80 -0
  59. package/src/gdskills/bundled/stacks/php-laravel/rules/security.mdc +80 -0
  60. package/src/gdskills/bundled/stacks/php-laravel/rules/testing.mdc +82 -0
  61. package/src/gdskills/bundled/stacks/php-laravel/skills/php-laravel-build-fix/SKILL.md +143 -0
  62. package/src/gdskills/bundled/stacks/php-laravel/skills/php-laravel-build-fix/evals.json +74 -0
  63. package/src/gdskills/bundled/stacks/php-laravel/skills/php-laravel-code-review/SKILL.md +126 -0
  64. package/src/gdskills/bundled/stacks/php-laravel/skills/php-laravel-code-review/evals.json +76 -0
  65. package/src/gdskills/bundled/stacks/php-laravel/skills/php-laravel-implementation/SKILL.md +140 -0
  66. package/src/gdskills/bundled/stacks/php-laravel/skills/php-laravel-implementation/evals.json +75 -0
  67. package/src/gdskills/bundled/stacks/php-laravel/skills/php-laravel-testing/SKILL.md +124 -0
  68. package/src/gdskills/bundled/stacks/php-laravel/skills/php-laravel-testing/evals.json +74 -0
  69. package/src/gdskills/bundled/stacks/ruby-rails/agent-refs.json +4 -0
  70. package/src/gdskills/bundled/stacks/ruby-rails/governance/eval.json +1673 -0
  71. package/src/gdskills/bundled/stacks/ruby-rails/governance/scout.json +33 -0
  72. package/src/gdskills/bundled/stacks/ruby-rails/pack.json +42 -0
  73. package/src/gdskills/bundled/stacks/ruby-rails/rules/coding-style.mdc +69 -0
  74. package/src/gdskills/bundled/stacks/ruby-rails/rules/patterns.mdc +93 -0
  75. package/src/gdskills/bundled/stacks/ruby-rails/rules/security.mdc +90 -0
  76. package/src/gdskills/bundled/stacks/ruby-rails/rules/testing.mdc +89 -0
  77. package/src/gdskills/bundled/stacks/ruby-rails/skills/ruby-rails-build-fix/SKILL.md +143 -0
  78. package/src/gdskills/bundled/stacks/ruby-rails/skills/ruby-rails-build-fix/evals.json +73 -0
  79. package/src/gdskills/bundled/stacks/ruby-rails/skills/ruby-rails-code-review/SKILL.md +134 -0
  80. package/src/gdskills/bundled/stacks/ruby-rails/skills/ruby-rails-code-review/evals.json +71 -0
  81. package/src/gdskills/bundled/stacks/ruby-rails/skills/ruby-rails-implementation/SKILL.md +141 -0
  82. package/src/gdskills/bundled/stacks/ruby-rails/skills/ruby-rails-implementation/evals.json +72 -0
  83. package/src/gdskills/bundled/stacks/ruby-rails/skills/ruby-rails-testing/SKILL.md +125 -0
  84. package/src/gdskills/bundled/stacks/ruby-rails/skills/ruby-rails-testing/evals.json +72 -0
  85. package/src/gdskills/bundled/stacks/sql-db/agent-refs.json +4 -0
  86. package/src/gdskills/bundled/stacks/sql-db/governance/eval.json +1829 -0
  87. package/src/gdskills/bundled/stacks/sql-db/governance/scout.json +30 -0
  88. package/src/gdskills/bundled/stacks/sql-db/pack.json +40 -0
  89. package/src/gdskills/bundled/stacks/sql-db/rules/coding-style.mdc +69 -0
  90. package/src/gdskills/bundled/stacks/sql-db/rules/patterns.mdc +134 -0
  91. package/src/gdskills/bundled/stacks/sql-db/rules/security.mdc +74 -0
  92. package/src/gdskills/bundled/stacks/sql-db/rules/testing.mdc +83 -0
  93. package/src/gdskills/bundled/stacks/sql-db/skills/sql-db-build-fix/SKILL.md +147 -0
  94. package/src/gdskills/bundled/stacks/sql-db/skills/sql-db-build-fix/evals.json +72 -0
  95. package/src/gdskills/bundled/stacks/sql-db/skills/sql-db-code-review/SKILL.md +132 -0
  96. package/src/gdskills/bundled/stacks/sql-db/skills/sql-db-code-review/evals.json +73 -0
  97. package/src/gdskills/bundled/stacks/sql-db/skills/sql-db-implementation/SKILL.md +153 -0
  98. package/src/gdskills/bundled/stacks/sql-db/skills/sql-db-implementation/evals.json +77 -0
  99. package/src/gdskills/bundled/stacks/sql-db/skills/sql-db-testing/SKILL.md +129 -0
  100. package/src/gdskills/bundled/stacks/sql-db/skills/sql-db-testing/evals.json +73 -0
@@ -0,0 +1,129 @@
1
+ ---
2
+ name: sql-db-testing
3
+ description: "Use when writing or fixing tests for SQL migrations and queries at the schema/engine level -- transactional test wrapping with rollback, seed-data design for a migration or query test, testing a migration's up/down path on representative row counts, and asserting on actual query results and query counts rather than just 'no error'. Not for a language/ORM-specific test-fixture convention such as a pytest fixture, an RSpec factory, or a Laravel model factory (use that stack's own testing skill)."
4
+ triggers:
5
+ - "write a test for this migration's rollback"
6
+ - "add a test that this backfill updates every row exactly once"
7
+ - "test that this query returns the right rows in the right order"
8
+ - "write a regression test for this N+1 fix"
9
+ - "set up transactional isolation so each migration test rolls back automatically"
10
+ - "seed representative row counts for this migration's up/down test"
11
+ metadata:
12
+ origin: authored
13
+ category: test
14
+ version: "1.0.0"
15
+ compatible_harnesses: "claude,codex,cursor,zed,opencode"
16
+ license: "MIT"
17
+ ---
18
+
19
+ # SQL / database testing (Postgres, MySQL, generic SQL)
20
+
21
+ Write, extend, or fix tests for `.sql` migrations and hand-written
22
+ queries: transactional isolation, fixture/seed design, migration up/down
23
+ testing at representative scale, and result-based query assertions.
24
+ `rules/testing.mdc` carries the full rule set this skill's checklist is
25
+ built from — read it, not just this summary, before writing tests.
26
+
27
+ ## Workflow
28
+
29
+ ### Step 1: Discover the project's test conventions
30
+
31
+ 1. Find how the project already isolates database tests: a
32
+ transaction-per-test wrapper, a truncate-and-reseed helper, or a fresh
33
+ container/schema per run. Match whichever is already there rather than
34
+ introducing a second isolation strategy.
35
+ 2. Read 1-2 existing database tests for fixture/factory helper
36
+ conventions, naming style, and whether the suite already asserts on
37
+ query counts anywhere (a query-count assertion helper is worth
38
+ reusing if one exists).
39
+ 3. Identify the migration runner's test hooks, if any — some runners
40
+ expose a way to run a migration's `up` then `down` in a test harness;
41
+ use it instead of hand-rolling migration invocation in the test.
42
+
43
+ ### Step 2: Plan test cases
44
+
45
+ **Queries:** the happy-path result set (values and ordering when the
46
+ query specifies `ORDER BY`), the empty-result case (a filter that matches
47
+ nothing), and the boundary case for any range condition — these are where
48
+ an off-by-one or a `JOIN` that should have been a `LEFT JOIN` surfaces.
49
+
50
+ **N+1 regression tests:** alongside the result assertion, assert the
51
+ query count didn't regress back to one-per-row — most ORMs/test harnesses
52
+ expose a query counter or a query log to assert against.
53
+
54
+ **Migrations:** the `up` path against a representative row count (not an
55
+ empty table), the `down` path when the migration states one (running
56
+ `up` then `down` should leave the schema equivalent to before), and
57
+ idempotency when the migration was written to tolerate re-application.
58
+
59
+ **Batched backfills:** a row count large enough to exercise at least two
60
+ batches, asserting every row was updated exactly once — no row skipped at
61
+ a batch boundary, none double-applied in a way that would corrupt a
62
+ non-idempotent update.
63
+
64
+ ### Step 3: Write
65
+
66
+ 1. Wrap each test in the project's transactional-rollback isolation (or
67
+ its existing equivalent) per `rules/testing.mdc`; reserve a dedicated
68
+ torn-down schema/database only for cases a transaction can't roll back
69
+ cleanly (`CREATE INDEX CONCURRENTLY`, cross-transaction concurrency).
70
+ 2. Build fixtures through the project's existing factory/helper, seeding
71
+ only what the test needs, named descriptively
72
+ (`user_with_expired_token`, not `test_user_3`).
73
+ 3. Construct any "now"-relative fixture value from a fixed, injected
74
+ reference time — never a live `NOW()`/`CURRENT_TIMESTAMP` evaluated at
75
+ test-run time for a boundary case.
76
+ 4. Assert on the query's actual rows/values/ordering, and on the query
77
+ count for an N+1 regression test — never assert only "ran without
78
+ error."
79
+
80
+ ### Step 4: Run and fix
81
+
82
+ Run the project's own database test command. Fix failing tests (max 3
83
+ iterations) — fix the test or fixture, not the migration/query under
84
+ test, unless the test itself correctly caught a real bug (say so in the
85
+ report rather than silently changing the schema/query to make the test
86
+ pass).
87
+
88
+ ### Step 5: Report
89
+
90
+ ```
91
+ Generated: tests/migrations/test_add_status_backfill.py
92
+ - Up/down path tested against a 50k-row fixture, 3 batch boundaries
93
+ - Asserted every row updated exactly once, no row skipped or doubled
94
+ ```
95
+
96
+ ## Rules
97
+
98
+ - ALWAYS match the project's existing isolation and fixture conventions
99
+ found in Step 1, not a different project's style.
100
+ - NEVER modify the migration/query under test — only test files and
101
+ fixtures — unless the test caught a real bug (state that explicitly).
102
+ - NEVER assert only that a query "ran without error" — assert on its
103
+ actual result rows/values/ordering.
104
+ - Test a migration's `up` path against a representative row count, not
105
+ only an empty table.
106
+
107
+ ## Red Flags
108
+
109
+ | Rationalization | Why it is wrong |
110
+ |---|---|
111
+ | "The test just checks the migration doesn't throw, that's enough" | A migration that silently backfills the wrong value, or skips rows at a batch boundary, throws no error — only a result assertion catches that |
112
+ | "I'll seed with `NOW()` for the expiry fixture, it's close enough" | A live clock value makes boundary cases (a row that just expired) flaky depending on exact test-run timing; use a fixed reference time |
113
+ | "This N+1 fix already returns the right rows, that's the test" | A correct result set today doesn't prove the query count didn't regress back to one-per-row after a later change; assert the count too |
114
+ | "I'll test the migration against an empty table, it's faster" | An empty-table test can't catch a lock/lastingness/batch-boundary issue that only shows up at realistic row counts |
115
+
116
+ ## Verification
117
+
118
+ Do not report the work done until all of the following hold:
119
+
120
+ - Every new/changed test runs isolated via the project's transactional
121
+ (or equivalent) rollback strategy — no test leaves state for the next
122
+ test to trip over.
123
+ - Every query test asserts on actual result rows/values/ordering, not
124
+ merely "no error thrown."
125
+ - A migration test covers both the `up` path at a representative row
126
+ count and, when the migration states one, the `down` path.
127
+ - `git status`/the diff shows only test and fixture files changed, unless
128
+ a real bug in the migration/query was found and fixed (stated in the
129
+ report).
@@ -0,0 +1,73 @@
1
+ {
2
+ "triggers": {
3
+ "positive": [
4
+ "Write a test that confirms this migration's rollback leaves the schema the way it was before",
5
+ "Add a test fixture for a user whose invite token expired two days ago",
6
+ "How should I isolate these database tests from each other so one doesn't affect the next?",
7
+ "Write a regression test proving this order-listing query no longer issues one query per row",
8
+ "Write a test proving the batched backfill script for orders.legacy_status touches every row with no row skipped or double-updated",
9
+ "Add test cases for this query's empty-result and boundary conditions",
10
+ "Set up a test that seeds fifty thousand rows to check this migration at realistic scale"
11
+ ],
12
+ "negative": [
13
+ "Write pytest fixtures for this Django model's test suite",
14
+ "Add RSpec tests for this Rails ActiveRecord scope",
15
+ "Write Jest tests for this React component's rendering",
16
+ "Fix this failing Go test that uses testify assertions",
17
+ "Set up a mock server for this API integration test",
18
+ "Write end-to-end tests for this checkout flow using Playwright",
19
+ "Add unit tests for this Python function that formats currency"
20
+ ]
21
+ },
22
+ "scenarios": [
23
+ {
24
+ "id": "fixed-clock-fixture",
25
+ "prompt": "I'm writing a test fixture for a token that should already be expired. How should I set the expiry timestamp?",
26
+ "strictness": "high",
27
+ "expected_behavior": [
28
+ {
29
+ "grader": "judge",
30
+ "rubric": "A correct answer builds the expiry timestamp relative to a fixed, injected reference time controlled by the test (a constant, a frozen/injected clock) rather than a live NOW()/CURRENT_TIMESTAMP evaluated at the moment the test runs, because a live-clock-derived boundary value makes the test's outcome depend on exact timing and can flake.",
31
+ "pass_criteria": [
32
+ "Uses a fixed, test-controlled reference time (a constant timestamp, or a frozen/injected clock value) as the basis for the expiry fixture, not a live NOW()/CURRENT_TIMESTAMP.",
33
+ "Explains that deriving the boundary value from a live clock makes the test's pass/fail depend on when it happens to run, which is why a fixed reference time is used instead."
34
+ ],
35
+ "fail_criteria": [
36
+ "Recommends setting the fixture's expiry using NOW()/CURRENT_TIMESTAMP (or an equivalent live-clock call) evaluated at test-run time as the basis for the expired boundary, whether directly or via a small offset from it."
37
+ ]
38
+ }
39
+ ],
40
+ "calibration": {
41
+ "known_right": "Pick a fixed reference timestamp for the test -- a constant like `2026-01-01T00:00:00Z` -- and set the token's `expires_at` to something clearly before it, e.g. one day earlier. If the code under test reads the current time itself, inject or freeze that same fixed reference time into it too (a test clock, a frozen-time helper) so 'now' and the fixture's notion of 'already expired' agree deterministically. Don't derive the expiry from a live NOW()/CURRENT_TIMESTAMP call evaluated when the test actually runs -- that ties the test's pass/fail to real wall-clock timing, which is exactly the kind of thing that makes a boundary case (a token that 'just' expired) flaky depending on when CI happens to execute it.",
42
+ "known_wrong": "Just set it relative to right now: `expires_at = NOW() - INTERVAL '1 day'` (or your language's equivalent of 'now minus a day') -- that guarantees it's in the past relative to whenever the test happens to run, so you don't need to hardcode any timestamp or set up a fake clock for what's a pretty simple fixture.",
43
+ "vague": "Make sure the expiry is set to a time that's clearly in the past so the token reads as expired.",
44
+ "subtle_wrong": "Use a fixed reference constant for readability, something like `const REFERENCE = '2026-01-01T00:00:00Z'`, but derive the actual fixture value from `NOW() - INTERVAL '1 day'` at insert time rather than from the constant itself -- the constant documents the intent in the test file, and NOW() minus an offset gets you an expired timestamp either way without needing to wire a frozen clock into the code under test."
45
+ },
46
+ "anti_patterns": ["NOW()"]
47
+ },
48
+ {
49
+ "id": "n1-regression-query-count",
50
+ "prompt": "I fixed an N+1 query in the order-listing endpoint by switching to a single JOIN. How should I write the regression test so it doesn't silently regress back to N+1 later?",
51
+ "strictness": "high",
52
+ "expected_behavior": [
53
+ {
54
+ "grader": "judge",
55
+ "rubric": "A correct answer asserts both the query's actual result (rows/values) and the number of queries issued (via the test harness's or ORM's query counter/log), so a later regression back to one query per row fails the test even if the final result set still happens to be correct -- asserting the result alone, without a query-count check, is not sufficient.",
56
+ "pass_criteria": [
57
+ "Asserts the query's actual result rows/values are correct.",
58
+ "Additionally asserts the number of queries issued (a query count via the test harness's/ORM's counter or captured query log), not just that the final result is correct."
59
+ ],
60
+ "fail_criteria": [
61
+ "Tests only the final result set's correctness with no assertion on the number of queries issued, leaving a regression back to one-query-per-row undetected as long as the final rows still happen to be right."
62
+ ]
63
+ }
64
+ ],
65
+ "calibration": {
66
+ "known_right": "Assert two things, not one: the result rows come back correct (right orders, right associated user data, right ordering), and the number of queries the test harness observed is exactly what the single-JOIN version should issue -- most test setups expose a query counter or let you capture the executed query log around the call, so assert something like `assert query_count == 1` (or whatever the JOIN version's actual expected count is) alongside the row assertions. The result-only assertion would still pass if someone later reverted to a per-row fetch loop that happens to reconstruct the same final rows -- the query-count assertion is what actually proves the fix is still in effect, since it fails the moment the query pattern regresses back to one-per-row even though the output would still look identical.",
67
+ "known_wrong": "Just assert the endpoint still returns the right orders with the right user data attached -- that's the behavior that matters to callers. If the result rows are correct, the fix is working; you don't need to separately track how many queries got issued to get that, since a wrong query count with a correct final result isn't actually a bug anyone would notice.",
68
+ "vague": "Make sure the test proves the fix actually worked and would catch it if the query pattern ever changed back.",
69
+ "subtle_wrong": "Assert the result rows are correct, and add a comment above the test noting that this relies on the single-JOIN implementation so a reviewer should double check the query pattern by eye if this test is ever touched -- that's a lighter-weight way to flag the regression risk than wiring up a query counter, and it still documents the intent for whoever reads the test next."
70
+ }
71
+ }
72
+ ]
73
+ }