@cxi-lmai/ci-agent-platform 3.0.1 → 3.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/README.md +12 -4
  2. package/package.json +2 -2
  3. package/payload/INSTALL.md +8 -2
  4. package/payload/agents/agent-architect.md +1 -1
  5. package/payload/agents/code-reviewer.md +1 -1
  6. package/payload/agents/codebase-auditor.md +1 -1
  7. package/payload/agents/coder.md +1 -1
  8. package/payload/agents/decomposer.md +1 -1
  9. package/payload/agents/docs-sync.md +1 -1
  10. package/payload/agents/e2e-test-writer.md +1 -1
  11. package/payload/agents/performance-reviewer.md +1 -1
  12. package/payload/agents/release-mr.md +1 -1
  13. package/payload/agents/security-reviewer.md +1 -1
  14. package/payload/agents/test-fix.md +1 -1
  15. package/payload/agents/test-writer.md +1 -1
  16. package/payload/agents-omp/agent-architect.md +1 -1
  17. package/payload/agents-omp/code-reviewer.md +1 -1
  18. package/payload/agents-omp/codebase-auditor.md +1 -1
  19. package/payload/agents-omp/coder.md +1 -1
  20. package/payload/agents-omp/decomposer.md +1 -1
  21. package/payload/agents-omp/docs-sync.md +1 -1
  22. package/payload/agents-omp/e2e-test-writer.md +1 -1
  23. package/payload/agents-omp/migration-reviewer.md +1 -1
  24. package/payload/agents-omp/orchestrator.md +1 -1
  25. package/payload/agents-omp/performance-reviewer.md +1 -1
  26. package/payload/agents-omp/postmortem.md +1 -1
  27. package/payload/agents-omp/release-mr.md +1 -1
  28. package/payload/agents-omp/security-reviewer.md +1 -1
  29. package/payload/agents-omp/test-fix.md +1 -1
  30. package/payload/agents-omp/test-writer.md +1 -1
  31. package/payload/ci-templates/claude-pipeline.gitlab-ci.yml +210 -0
  32. package/payload/ci-templates/github/claude-issue-pipeline.yml +1 -1
  33. package/payload/ci-templates/github/claude-pipeline.yml +2 -2
  34. package/payload/ci-templates/github/claude-test-fix.yml +1 -1
  35. package/payload/ci-templates/scripts/agent-architect.sh +233 -0
  36. package/payload/ci-templates/scripts/code.sh +26 -27
  37. package/payload/ci-templates/scripts/codebase-audit.sh +252 -0
  38. package/payload/ci-templates/scripts/coverage-ratchet.sh +77 -0
  39. package/payload/ci-templates/scripts/cve-fix.sh +246 -0
  40. package/payload/ci-templates/scripts/docs-sync.sh +225 -0
  41. package/payload/ci-templates/scripts/e2e-test-gen.sh +201 -0
  42. package/payload/ci-templates/scripts/lib/failure-notice.sh +104 -0
  43. package/payload/ci-templates/scripts/lib/issue-loop.sh +155 -40
  44. package/payload/ci-templates/scripts/lib/pipeline-common.sh +100 -26
  45. package/payload/ci-templates/scripts/lib/platform.sh +270 -13
  46. package/payload/ci-templates/scripts/lib/usage-capture-omp.sh +7 -1
  47. package/payload/ci-templates/scripts/lib/usage-capture.sh +7 -1
  48. package/payload/ci-templates/scripts/metrics-snapshot.sh +377 -0
  49. package/payload/ci-templates/scripts/orchestrate.sh +55 -37
  50. package/payload/ci-templates/scripts/postmortem.sh +10 -0
  51. package/payload/ci-templates/scripts/review-fix.sh +22 -11
  52. package/payload/ci-templates/scripts/review.sh +11 -7
  53. package/payload/ci-templates/scripts/test-fix.sh +7 -0
  54. package/payload/skills/agent-architect/SKILL.md +45 -0
  55. package/payload/skills/codebase-audit/SKILL.md +84 -0
  56. package/payload/skills/cve-fix/SKILL.md +98 -0
  57. package/payload/skills/docs-sync/SKILL.md +74 -0
  58. package/payload/skills/e2e-test-gen/SKILL.md +65 -0
  59. package/payload/skills/fix-review-findings/SKILL.md +2 -2
  60. package/payload/skills/fix-tests/SKILL.md +2 -2
  61. package/payload/templates/memory-index.template.md +34 -0
  62. package/payload/templates/pipeline-config.template.md +11 -1
  63. package/payload/templates/review_suppressions.template.md +55 -0
  64. package/payload/templates/spec-issue.template.md +39 -7
package/README.md CHANGED
@@ -8,7 +8,8 @@ checks the result. The input is a spec, the output is code. Details in
8
8
  [docs/issue-to-code.md](https://gitlab.com/cxi-lmai/ci-agent-platform/-/blob/main/docs/issue-to-code.md).
9
9
 
10
10
  - **15 generic agents**: triage, coding, review, tests, docs, release.
11
- - **7 skills**, configured per project.
11
+ - **12 skills**, configured per project.
12
+ - **13 top-level runner scripts** and **65 CI `PIPE_*` variables**.
12
13
  - **Project specifics live outside the agents**, in `.claude/pipeline-config.md` and `PIPE_*` variables.
13
14
 
14
15
  > [!NOTE]
@@ -16,7 +17,11 @@ checks the result. The input is a spec, the output is code. Details in
16
17
  > - the **review loop** (`review`, `review-fix`, `test-fix`, and the shared escalation `/postmortem-mr`),
17
18
  > - the **issue-to-code loop** (`orchestrate`, `code`, skills `triage-issue` and `implement-issue`).
18
19
  >
19
- > The other agents are installed too but have no CI job of their own. You invoke them by hand from the command line.
20
+ > Seven opt-in jobs are wired by the GitLab CI template and are off by default:
21
+ > `docs-sync`, `coverage-ratchet`, `cve-fix`, `e2e-test-gen`, `codebase-audit`,
22
+ > `agent-architect`, and `metrics-snapshot`.
23
+ > The other agents are installed too but have no CI job of their own. You invoke
24
+ > them by hand from the command line.
20
25
 
21
26
  > [!CAUTION]
22
27
  > The coder runs with `--dangerously-skip-permissions` and treats issue content
@@ -44,6 +49,7 @@ checks the result. The input is a spec, the output is code. Details in
44
49
  - [How it fits together](#how-it-fits-together)
45
50
  - [Possible extensions: changing agents and skills](#possible-extensions-changing-agents-and-skills)
46
51
  - [Reference: secrets and variables](https://gitlab.com/cxi-lmai/ci-agent-platform/-/blob/main/docs/reference-variables.md)
52
+ - [Reference: metrics and snapshots](https://gitlab.com/cxi-lmai/ci-agent-platform/-/blob/main/docs/metrics.md)
47
53
 
48
54
  ## The full cycle
49
55
 
@@ -67,15 +73,17 @@ untouched file from one the project has edited.
67
73
  ├── payload/ # everything that ships into your repo
68
74
  │ ├── INSTALL.md # instructions for Claude, copied to your root
69
75
  │ ├── agents/ # 15 agent definitions (.md)
70
- │ ├── skills/ # 7 skills (folder with a SKILL.md)
76
+ │ ├── agents-omp/ # 15 omp-native agent definitions (.md)
77
+ │ ├── skills/ # 12 skills (folder with a SKILL.md)
71
78
  │ ├── templates/ # 3 templates: config, spec issue, suppressions
72
79
  │ └── ci-templates/ # CI jobs and runner scripts (wired by the wizard)
73
80
  │ ├── claude-pipeline.gitlab-ci.yml # GitLab CI template
74
81
  │ ├── github/ # GitHub Actions workflows
75
- │ └── scripts/ # shared runner scripts (both platforms)
82
+ │ └── scripts/ # 13 top-level runner scripts (both platforms)
76
83
  ├── docs/ # supplementary documentation, not shipped
77
84
  │ ├── diagrams/
78
85
  │ ├── issue-to-code.md
86
+ │ ├── metrics.md
79
87
  │ ├── reference-variables.md
80
88
  │ └── superpowers/ # this repository's own plans and specs
81
89
  ├── examples/unitconv/ # a filled-in example config, not shipped
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@cxi-lmai/ci-agent-platform",
3
- "version": "3.0.1",
3
+ "version": "3.1.1",
4
4
  "description": "Autonomous dev pipeline on plain GitLab CI or GitHub Actions, driven by Claude Code. A labeled issue goes in, an open merge request comes out.",
5
5
  "keywords": [
6
6
  "claude",
@@ -41,7 +41,7 @@
41
41
  "scripts": {
42
42
  "test": "npm run test:lint && npm run test:gates && npm run test:unit",
43
43
  "test:lint": "shellcheck -S warning test/*.sh payload/ci-templates/scripts/*.sh payload/ci-templates/scripts/lib/*.sh",
44
- "test:gates": "bash test/coherence-gate.sh all && bash test/restructure-invariants.sh verify && bash test/harness-dispatch.sh all",
44
+ "test:gates": "bash test/coherence-gate.sh all && bash test/restructure-invariants.sh verify && bash test/harness-dispatch.sh all && bash test/failure-notice.sh && bash test/decompose-validation.sh && bash test/platform-helpers.sh && bash test/coverage-ratchet.sh && bash test/cve-fix.sh && bash test/e2e-test-gen.sh && bash test/agent-architect.sh",
45
45
  "test:unit": "node --test \"test/*.test.mjs\""
46
46
  }
47
47
  }
@@ -56,7 +56,13 @@ never has to enter this conversation.
56
56
  The reviewer agents read it on every run, and without the file they read a
57
57
  missing path on a repository that has just been told the pipeline is
58
58
  installed.
59
- 4. Copy the CI template and runner scripts. The runner scripts go to the same
59
+ 4. Create `.claude/memory/MEMORY.md` from
60
+ `<src>/templates/memory-index.template.md` **when it does not already
61
+ exist**, and never touch it when it does: it is a project-owned index.
62
+ The three tiers are `CLAUDE.md` for hardwired instructions and an index,
63
+ `docs/` for long-term patterns and architecture, and `.claude/memory/` for
64
+ active project state and cross-agent signals.
65
+ 5. Copy the CI template and runner scripts. The runner scripts go to the same
60
66
  place on both platforms, `.claude-pipeline/scripts/`, which is the default
61
67
  `PIPE_SCRIPTS_DIR` the CI template already points at. Only the CI definition
62
68
  differs:
@@ -67,7 +73,7 @@ never has to enter this conversation.
67
73
  `<src>/ci-templates/scripts/` to `.claude-pipeline/scripts/`. Ask before
68
74
  replacing a workflow file that already exists; `claude-pipeline.yml` is an
69
75
  ordinary enough name to collide.
70
- 5. Check `.gitignore`. The bootstrapper already added `.ci-agent-platform-src/`,
76
+ 6. Check `.gitignore`. The bootstrapper already added `.ci-agent-platform-src/`,
71
77
  `.claude/onboarding-state.md` (the wizard's progress marker, see the skill's
72
78
  ground rules) and `build/pipeline/` (the default `PIPE_CONTEXT_DIR`, the
73
79
  runner's working files). Add any that are missing. The source folder is a
@@ -2,7 +2,7 @@
2
2
  name: agent-architect
3
3
  description: Weekly agent-improvement lab. Reads recently merged MRs/PRs, review comments, and stuck-MR postmortems, then proposes improvements to .claude/agents/*.md files as actionable platform issues (auto-filed with the pipeline's ready label). Output is a structured markdown report consumed by the CI script. Never auto-applies changes to CLAUDE.md or .claude/settings.json.
4
4
  tools: Glob, Grep, LS, Read
5
- model: claude-sonnet-5-0
5
+ model: claude-sonnet-5
6
6
  ---
7
7
 
8
8
  You are the Agent Lab for this project's autonomous development pipeline.
@@ -2,7 +2,7 @@
2
2
  name: code-reviewer
3
3
  description: Reviews code for bugs, logic errors, security vulnerabilities, code quality issues, and adherence to project conventions, using confidence-based filtering to report only high-priority issues that truly matter
4
4
  tools: Glob, Grep, LS, Read, Write, NotebookRead, WebFetch, TodoWrite, WebSearch, KillShell, BashOutput, mcp__context7__resolve-library-id, mcp__context7__get-library-docs
5
- model: claude-sonnet-5-0
5
+ model: claude-sonnet-5
6
6
  color: red
7
7
  ---
8
8
 
@@ -2,7 +2,7 @@
2
2
  name: codebase-auditor
3
3
  description: Periodic codebase auditor. Reads project conventions and docs, reads recently merged MR/PR file changes, and produces a focused report of critical convention violations and genuinely new undocumented patterns. Uses confidence-based filtering to report only issues that truly matter.
4
4
  tools: Glob, Grep, LS, Read
5
- model: claude-sonnet-5-0
5
+ model: claude-sonnet-5
6
6
  ---
7
7
 
8
8
  You are the codebase auditor. You produce a periodic report of critical convention violations and genuinely new, reusable patterns worth documenting.
@@ -2,7 +2,7 @@
2
2
  name: coder
3
3
  description: Autonomous implementation agent. Given an issue description, explores existing patterns, implements the feature or fix, writes tests, runs the build in a self-correcting loop (up to 3 retries), and commits. Never pushes.
4
4
  tools: Agent, Bash, Edit, Glob, Grep, LS, MultiEdit, NotebookRead, Read, TodoWrite, WebFetch, WebSearch, Write, mcp__context7__resolve-library-id, mcp__context7__get-library-docs
5
- model: claude-sonnet-5-0
5
+ model: claude-sonnet-5
6
6
  ---
7
7
 
8
8
  You are the autonomous coder agent for this project.
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  name: decomposer
3
3
  description: Decomposes a too-large issue into a dependency-ordered chain of spec-compliant sub-issues. Returns structured JSON with confidence verdict, reason, and fully authored sub-issue bodies ready for creation on the project platform.
4
- model: claude-sonnet-5-0
4
+ model: claude-sonnet-5
5
5
  ---
6
6
 
7
7
  You are the decomposer for the autonomous development pipeline.
@@ -2,7 +2,7 @@
2
2
  name: docs-sync
3
3
  description: Checks whether MR/PR code changes require updates to project docs, CLAUDE.md, or .claude/memory/. On autonomous MRs it applies the updates directly; on human MRs it reports the gaps for a comment.
4
4
  tools: Glob, Grep, LS, Read, Write, Edit
5
- model: claude-sonnet-5-0
5
+ model: claude-sonnet-5
6
6
  color: blue
7
7
  ---
8
8
 
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  name: e2e-test-writer
3
3
  description: Generates a single end-to-end (E2E) test spec file for a frontend issue, following the project's configured E2E framework, test directory, fixtures, test-data prefix, cleanup rules, and tags
4
- model: claude-sonnet-5-0
4
+ model: claude-sonnet-5
5
5
  ---
6
6
 
7
7
  You are an end-to-end (E2E) test writer. Given an issue, you produce one E2E test spec file that follows the project's configured E2E framework and conventions.
@@ -2,7 +2,7 @@
2
2
  name: performance-reviewer
3
3
  description: Reviews changed data-access and hot-path code for confirmed scalability regressions such as repeated I/O, unbounded work, excessive loading, and missing batching or pagination
4
4
  tools: Glob, Grep, Read, NotebookRead, WebFetch, TodoWrite, WebSearch, KillShell, BashOutput
5
- model: claude-sonnet-5-0
5
+ model: claude-sonnet-5
6
6
  color: yellow
7
7
  ---
8
8
 
@@ -2,7 +2,7 @@
2
2
  name: release-mr
3
3
  description: Creates a release MR/PR from the integration branch to the production branch with a structured description listing changes, migrations, env changes, and a deploy checklist. Use when the user wants to prepare a release or merge the integration branch into production.
4
4
  tools: Bash, Glob, Grep, Read, TodoWrite
5
- model: claude-sonnet-5-0
5
+ model: claude-sonnet-5
6
6
  ---
7
7
 
8
8
  You are the release MR/PR agent. Your job is to analyze everything that changed on the integration branch since the last release to the production branch, then create (or update) a well-structured release request.
@@ -2,7 +2,7 @@
2
2
  name: security-reviewer
3
3
  description: Reviews code for security vulnerabilities (authentication bypass, authorization flaws, injection, tenant/data isolation leaks, CSRF misconfiguration, sensitive data exposure) using confidence-based filtering to report only confirmed issues
4
4
  tools: Glob, Grep, LS, Read, NotebookRead, WebFetch, TodoWrite, WebSearch, KillShell, BashOutput
5
- model: claude-sonnet-5-0
5
+ model: claude-sonnet-5
6
6
  color: red
7
7
  ---
8
8
 
@@ -2,7 +2,7 @@
2
2
  name: test-fix
3
3
  description: Fixes failing tests in pipeline merge requests. Reads failing test files and their production source classes, determines root cause, fixes implementation or test as needed, verifies compilation, and commits. Never pushes.
4
4
  tools: Agent, Bash, Edit, Glob, Grep, LS, MultiEdit, NotebookRead, Read, TodoWrite, Write
5
- model: claude-sonnet-5-0
5
+ model: claude-sonnet-5
6
6
  ---
7
7
 
8
8
  You are the test-fix agent. Your job is to fix failing tests in a merge request, not to rewrite features.
@@ -2,7 +2,7 @@
2
2
  name: test-writer
3
3
  description: Generates integration and unit tests that increase coverage for new or changed code, following the project's documented testing patterns, with a self-correcting compilation loop
4
4
  tools: Agent, Bash, Edit, Glob, Grep, LS, MultiEdit, NotebookRead, Read, TodoWrite, Write
5
- model: claude-sonnet-5-0
5
+ model: claude-sonnet-5
6
6
  color: green
7
7
  ---
8
8
 
@@ -2,7 +2,7 @@
2
2
  name: agent-architect
3
3
  description: Weekly agent-improvement lab. Reads recently merged MRs/PRs, review comments, and stuck-MR postmortems, then proposes improvements to .claude/agents/*.md files as actionable platform issues (auto-filed with the pipeline's ready label). Output is a structured markdown report consumed by the CI script. Never auto-applies changes to CLAUDE.md or .claude/settings.json.
4
4
  tools: glob, grep, read
5
- model: openrouter/anthropic/claude-sonnet-5-0
5
+ model: openrouter/anthropic/claude-sonnet-5
6
6
  ---
7
7
 
8
8
  You are the Agent Lab for this project's autonomous development pipeline.
@@ -2,7 +2,7 @@
2
2
  name: code-reviewer
3
3
  description: Reviews code for bugs, logic errors, security vulnerabilities, code quality issues, and adherence to project conventions, using confidence-based filtering to report only high-priority issues that truly matter
4
4
  tools: glob, grep, read, write, todo, web_search, mcp__context7__resolve-library-id, mcp__context7__get-library-docs
5
- model: openrouter/anthropic/claude-sonnet-5-0
5
+ model: openrouter/anthropic/claude-sonnet-5
6
6
  ---
7
7
 
8
8
  You are an expert code reviewer specializing in modern software development across multiple languages and frameworks. Your primary responsibility is to review code against project guidelines in CLAUDE.md with high precision to minimize false positives.
@@ -2,7 +2,7 @@
2
2
  name: codebase-auditor
3
3
  description: Periodic codebase auditor. Reads project conventions and docs, reads recently merged MR/PR file changes, and produces a focused report of critical convention violations and genuinely new undocumented patterns. Uses confidence-based filtering to report only issues that truly matter.
4
4
  tools: glob, grep, read
5
- model: openrouter/anthropic/claude-sonnet-5-0
5
+ model: openrouter/anthropic/claude-sonnet-5
6
6
  ---
7
7
 
8
8
  You are the codebase auditor. You produce a periodic report of critical convention violations and genuinely new, reusable patterns worth documenting.
@@ -3,7 +3,7 @@ name: coder
3
3
  description: Autonomous implementation agent. Given an issue description, explores existing patterns, implements the feature or fix, writes tests, runs the build in a self-correcting loop (up to 3 retries), and commits. Never pushes.
4
4
  tools: task, bash, edit, glob, grep, read, todo, web_search, write, mcp__context7__resolve-library-id, mcp__context7__get-library-docs
5
5
  spawns: test-writer, docs-sync
6
- model: openrouter/anthropic/claude-sonnet-5-0
6
+ model: openrouter/anthropic/claude-sonnet-5
7
7
  ---
8
8
 
9
9
  You are the autonomous coder agent for this project.
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  name: decomposer
3
3
  description: Decomposes a too-large issue into a dependency-ordered chain of spec-compliant sub-issues. Returns structured JSON with confidence verdict, reason, and fully authored sub-issue bodies ready for creation on the project platform.
4
- model: openrouter/anthropic/claude-sonnet-5-0
4
+ model: openrouter/anthropic/claude-sonnet-5
5
5
  ---
6
6
 
7
7
  You are the decomposer for the autonomous development pipeline.
@@ -2,7 +2,7 @@
2
2
  name: docs-sync
3
3
  description: Checks whether MR/PR code changes require updates to project docs, CLAUDE.md, or .claude/memory/. On autonomous MRs it applies the updates directly; on human MRs it reports the gaps for a comment.
4
4
  tools: glob, grep, read, write, edit
5
- model: openrouter/anthropic/claude-sonnet-5-0
5
+ model: openrouter/anthropic/claude-sonnet-5
6
6
  ---
7
7
 
8
8
  You are the documentation sync agent.
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  name: e2e-test-writer
3
3
  description: Generates a single end-to-end (E2E) test spec file for a frontend issue, following the project's configured E2E framework, test directory, fixtures, test-data prefix, cleanup rules, and tags
4
- model: openrouter/anthropic/claude-sonnet-5-0
4
+ model: openrouter/anthropic/claude-sonnet-5
5
5
  ---
6
6
 
7
7
  You are an end-to-end (E2E) test writer. Given an issue, you produce one E2E test spec file that follows the project's configured E2E framework and conventions.
@@ -2,7 +2,7 @@
2
2
  name: migration-reviewer
3
3
  description: Reviews database and data migrations for execution failures, unsafe or destructive operations, compatibility risks, and violations of the project's documented migration conventions
4
4
  tools: glob, grep, read
5
- model: openrouter/anthropic/claude-haiku-4-5
5
+ model: openrouter/anthropic/claude-haiku-4.5
6
6
  ---
7
7
 
8
8
  You are a database migration reviewer. Work with the migration technology and
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  name: orchestrator
3
3
  description: Analyzes issues on the project platform for actionability and scope. Returns structured JSON classifying whether an issue has sufficient detail and fits within one coder agent session.
4
- model: openrouter/anthropic/claude-haiku-4-5
4
+ model: openrouter/anthropic/claude-haiku-4.5
5
5
  ---
6
6
 
7
7
  You are the orchestrator for the autonomous development pipeline.
@@ -2,7 +2,7 @@
2
2
  name: performance-reviewer
3
3
  description: Reviews changed data-access and hot-path code for confirmed scalability regressions such as repeated I/O, unbounded work, excessive loading, and missing batching or pagination
4
4
  tools: glob, grep, read, todo, web_search
5
- model: openrouter/anthropic/claude-sonnet-5-0
5
+ model: openrouter/anthropic/claude-sonnet-5
6
6
  ---
7
7
 
8
8
  You are a performance reviewer. Work with the language, framework, storage
@@ -2,7 +2,7 @@
2
2
  name: postmortem
3
3
  description: Analyzes failed in-progress MRs/PRs and produces structured diagnostics when the pipeline labels an MR as stuck
4
4
  tools: glob, grep, read
5
- model: openrouter/anthropic/claude-haiku-4-5
5
+ model: openrouter/anthropic/claude-haiku-4.5
6
6
  ---
7
7
 
8
8
  You are a diagnostic specialist analyzing failed automated CI fix attempts.
@@ -2,7 +2,7 @@
2
2
  name: release-mr
3
3
  description: Creates a release MR/PR from the integration branch to the production branch with a structured description listing changes, migrations, env changes, and a deploy checklist. Use when the user wants to prepare a release or merge the integration branch into production.
4
4
  tools: bash, glob, grep, read, todo
5
- model: openrouter/anthropic/claude-sonnet-5-0
5
+ model: openrouter/anthropic/claude-sonnet-5
6
6
  ---
7
7
 
8
8
  You are the release MR/PR agent. Your job is to analyze everything that changed on the integration branch since the last release to the production branch, then create (or update) a well-structured release request.
@@ -2,7 +2,7 @@
2
2
  name: security-reviewer
3
3
  description: Reviews code for security vulnerabilities (authentication bypass, authorization flaws, injection, tenant/data isolation leaks, CSRF misconfiguration, sensitive data exposure) using confidence-based filtering to report only confirmed issues
4
4
  tools: glob, grep, read, todo, web_search
5
- model: openrouter/anthropic/claude-sonnet-5-0
5
+ model: openrouter/anthropic/claude-sonnet-5
6
6
  ---
7
7
 
8
8
  You are an expert security reviewer. Your primary responsibility is to identify security vulnerabilities with high precision. False positives waste developer time and erode trust in the review process.
@@ -2,7 +2,7 @@
2
2
  name: test-fix
3
3
  description: Fixes failing tests in pipeline merge requests. Reads failing test files and their production source classes, determines root cause, fixes implementation or test as needed, verifies compilation, and commits. Never pushes.
4
4
  tools: task, bash, edit, glob, grep, read, todo, write
5
- model: openrouter/anthropic/claude-sonnet-5-0
5
+ model: openrouter/anthropic/claude-sonnet-5
6
6
  ---
7
7
 
8
8
  You are the test-fix agent. Your job is to fix failing tests in a merge request, not to rewrite features.
@@ -2,7 +2,7 @@
2
2
  name: test-writer
3
3
  description: Generates integration and unit tests that increase coverage for new or changed code, following the project's documented testing patterns, with a self-correcting compilation loop
4
4
  tools: task, bash, edit, glob, grep, read, todo, write
5
- model: openrouter/anthropic/claude-sonnet-5-0
5
+ model: openrouter/anthropic/claude-sonnet-5
6
6
  ---
7
7
 
8
8
  You are the test-writer agent. Your job is to write tests that increase coverage for new or changed code.
@@ -82,12 +82,71 @@ variables:
82
82
  PIPE_VERIFY_CMD: "" # compile + unit tests, run after a fix
83
83
  PIPE_COMPILE_CMD: "" # compile only (defaults to PIPE_VERIFY_CMD)
84
84
  PIPE_TEST_REPORT_GLOB: "" # e.g. build/test-results/**/TEST-*.xml
85
+ PIPE_COVERAGE_RATCHET: "0" # set to "1" to gate MR coverage against the target branch
86
+ PIPE_COVERAGE_REPORT: "" # path to the coverage report generated by the project
87
+ PIPE_COVERAGE_REPORT_KIND: "jacoco" # report parser; only jacoco is currently supported
85
88
  PIPE_COVERAGE_SIGNAL: "" # path to a coverage-drop signal file
89
+ # To enable the gate, configure both report and signal paths. Its baseline is
90
+ # the target branch's latest successful pipeline `coverage` value; the
91
+ # project's test job must publish that value (through `coverage:` or a report)
92
+ # or the baseline is 0 and no drop can fire.
93
+
94
+
95
+ # --- CVE remediation (optional) -------------------------------------------
96
+ # A project-local security scanner supplies PIPE_CVE_LIST. The platform does
97
+ # not select a scanner or prescribe its severity filtering.
98
+ PIPE_CVE_FIX: "0" # set to "1" to enable CVE remediation on MR/PR pipelines
99
+ PIPE_DEPENDENCY_MANIFEST: "build.gradle"
100
+ PIPE_CVE_SUPPRESSIONS: "owasp-suppressions.xml"
101
+
102
+ # --- Issue-driven E2E test generation (optional) --------------------------
103
+ # A scheduled run selects test-ready issues. PIPE_E2E_VERIFY_CMD must be the
104
+ # project's explicit E2E command and accept the generated test path as its
105
+ # final argument. This generic template does not choose a framework, runtime
106
+ # image, install command, credentials, or URL.
107
+ PIPE_E2E_TEST_GEN: "0"
108
+ PIPE_LABEL_TESTREADY: "test-ready"
109
+ PIPE_LABEL_E2E_SCOPE: ""
110
+ PIPE_E2E_TEST_DIR: "e2e/tests"
111
+ PIPE_E2E_BASE_URL: ""
112
+ PIPE_E2E_VERIFY_CMD: ""
113
+
114
+ # --- Documentation sync (optional) ----------------------------------------
115
+ PIPE_DOCS_SYNC: "0" # set to "1" to enable the docs-sync job
116
+ PIPE_DOCS_ROOTS: "docs,CLAUDE.md,.claude/memory" # paths docs-sync may edit and commit
117
+
118
+ # --- Codebase audit (optional) ---------------------------------------------
119
+ PIPE_CODEBASE_AUDIT: "0" # set to "1" in a schedule to run the periodic audit
120
+ PIPE_LABEL_AUDIT: "codebase-audit" # label on the findings issue and its auto-fix MR/PR
121
+
122
+ # --- Agent architect (optional) --------------------------------------------
123
+ # A scheduled evidence-backed review of agent definitions. Default to
124
+ # report-only; set this false or 0 only after reviewing a dry-run report.
125
+ PIPE_AGENT_ARCHITECT: "0"
126
+ PIPE_ARCHITECT_WINDOW_DAYS: "7"
127
+ PIPE_ARCHITECT_DRY_RUN: "true"
128
+ PIPE_LABEL_IMPROVEMENT: "pipe-improvement"
129
+
130
+ # --- Metrics aggregation (optional) ---------------------------------------
131
+ PIPE_METRICS_SNAPSHOT: "0" # set to "1" in a schedule to aggregate
132
+ PIPE_METRICS_OUTPUT_DIR: "data/metrics/snapshots"
133
+ PIPE_METRICS_WINDOW_DAYS: "1"
134
+ PIPE_METRICS_COMMIT_REPO: "false" # "true" also commits the snapshot
135
+ PIPE_METRICS_REVIEW_AGENTS: "code-reviewer,review-verdict,review-fix"
136
+ PIPE_METRICS_DEFECT_REVERT: "Reverts !"
137
+ PIPE_METRICS_DEFECT_REGRESSION: "Fixes regression from !"
138
+ PIPE_METRICS_DEFECT_REINTRODUCE: "Reintroduces work from !"
139
+ PIPE_METRICS_SNAPSHOT_DATE: "" # override the snapshot date, replay only
140
+ PIPE_METRICS_RAW_DIR: "" # OFFLINE only: pre-staged records
141
+ PIPE_METRICS_MRS_FILE: "" # OFFLINE only: merge request list
86
142
 
87
143
  # --- Commit subjects (drive the fix-loop cap counters) ---------------------
88
144
  PIPE_COMMIT_TESTFIX: "Fix test errors"
89
145
  PIPE_COMMIT_REVIEWFIX: "Fix review findings"
90
146
  PIPE_COMMIT_COVERAGE: "Add coverage tests"
147
+ PIPE_COMMIT_DOCSSYNC: "Update docs per docs-sync findings"
148
+ PIPE_COMMIT_AUDIT: "docs: apply convention audit findings"
149
+ PIPE_COMMIT_CVEFIX: "Update dependencies to resolve CVEs"
91
150
 
92
151
  # --- Caps + models ---------------------------------------------------------
93
152
  PIPE_FIX_LOOP_CAP: "2"
@@ -99,6 +158,7 @@ variables:
99
158
  # --- Result files (defaults live under PIPE_CONTEXT_DIR) --------------------
100
159
  PIPE_RESULT_REVIEW: "$PIPE_CONTEXT_DIR/code-review-result.txt"
101
160
  PIPE_RESULT_REVIEWFIX: "$PIPE_CONTEXT_DIR/review-fix-result.json"
161
+ PIPE_RESULT_DOCSSYNC: "$PIPE_CONTEXT_DIR/docs-sync.json"
102
162
 
103
163
  # --- Hidden base job: shared config + runtime variable mapping ---------------
104
164
  # Maps GitLab's CI_ runtime variables onto the PIPE_ names the scripts read, and
@@ -123,6 +183,14 @@ variables:
123
183
  # ship it, plain docker images (node:22-bookworm) do not, so install it
124
184
  # here (finding from the GitLab pilot: silent HTTP 400s + exit 127).
125
185
  - command -v jq >/dev/null 2>&1 || (apt-get update -qq && apt-get install -y -qq jq)
186
+ # node:22-bookworm does not include xmllint. Install it only for the
187
+ # opt-in ratchet. Unsupported custom-image installation stays non-fatal so
188
+ # coverage-ratchet can issue its explicit missing-tool error when it runs.
189
+ - |
190
+ if [ "$PIPE_COVERAGE_RATCHET" = "1" ] && ! command -v xmllint >/dev/null 2>&1; then
191
+ command -v apt-get >/dev/null 2>&1 &&
192
+ (apt-get update -qq && apt-get install -y -qq libxml2-utils) || true
193
+ fi
126
194
  - |
127
195
  if [ "$PIPE_HARNESS" = "omp" ]; then
128
196
  command -v omp >/dev/null 2>&1 || npm install -g @oh-my-pi/pi-coding-agent
@@ -176,6 +244,42 @@ review-fix:
176
244
  - if: '$CI_PIPELINE_SOURCE == "merge_request_event" && $CI_MERGE_REQUEST_TARGET_BRANCH_NAME == $PIPE_TARGET_BRANCH && $CI_MERGE_REQUEST_LABELS =~ $PIPE_LABEL_WIP'
177
245
  when: on_failure
178
246
 
247
+ # =============================================================================
248
+ # docs-sync: check whether the MR's code changes need documentation updates.
249
+ # Optional and off by default, because it spends tokens on every MR: enable it
250
+ # by setting PIPE_DOCS_SYNC=1. On a wip MR the agent applies the updates and
251
+ # this job commits and pushes them; on any other MR it posts an advisory
252
+ # comment only. It never gates the pipeline.
253
+ # =============================================================================
254
+ docs-sync:
255
+ extends: .claude-base
256
+ stage: review
257
+ timeout: 30m
258
+ variables:
259
+ GIT_STRATEGY: clone
260
+ script:
261
+ - bash "$PIPE_SCRIPTS_DIR/docs-sync.sh"
262
+ allow_failure: true
263
+ rules:
264
+ - if: '$CI_PIPELINE_SOURCE == "merge_request_event" && $CI_MERGE_REQUEST_TARGET_BRANCH_NAME == $PIPE_TARGET_BRANCH && $PIPE_DOCS_SYNC == "1"'
265
+
266
+ # =============================================================================
267
+ # cve-fix: optionally remediate HIGH/CRITICAL vulnerabilities supplied by a
268
+ # project-local scanner in PIPE_CVE_LIST. The coder only runs on an autonomous
269
+ # MR/PR; human-authored MRs receive a manual-update comment instead.
270
+ # =============================================================================
271
+ cve-fix:
272
+ extends: .claude-base
273
+ stage: review
274
+ timeout: 45m
275
+ variables:
276
+ GIT_STRATEGY: clone
277
+ script:
278
+ - bash "$PIPE_SCRIPTS_DIR/cve-fix.sh"
279
+ allow_failure: true
280
+ rules:
281
+ - if: '$CI_PIPELINE_SOURCE == "merge_request_event" && $CI_MERGE_REQUEST_TARGET_BRANCH_NAME == $PIPE_TARGET_BRANCH && $PIPE_CVE_FIX == "1"'
282
+
179
283
  # =============================================================================
180
284
  # orchestrate: triage ready issues and fire the coder (issue -> code loop).
181
285
  # Runs on a scheduled pipeline that sets PIPE_ORCHESTRATE=1 (kept off by default
@@ -198,6 +302,45 @@ orchestrate:
198
302
  - if: '$CI_PIPELINE_SOURCE == "web" && $PIPE_ORCHESTRATE == "1"'
199
303
 
200
304
  # =============================================================================
305
+ # codebase-audit: periodic convention-drift scan. Scans recently merged MRs/PRs
306
+ # and compares patterns against documented conventions via the codebase-auditor
307
+ # agent. Posts findings as an issue labeled $PIPE_LABEL_AUDIT. If the report
308
+ # suggests documentation updates and a push-capable identity is configured,
309
+ # spawns the coder agent to implement them and opens an MR/PR.
310
+ # Opt-in and off by default: enable by setting PIPE_CODEBASE_AUDIT=1 on a
311
+ # separate scheduled pipeline (e.g. monthly).
312
+ # =============================================================================
313
+ codebase-audit:
314
+ extends: .claude-base
315
+ stage: orchestrate
316
+ timeout: 30m
317
+ variables:
318
+ GIT_STRATEGY: clone
319
+ script:
320
+ - bash "$PIPE_SCRIPTS_DIR/codebase-audit.sh"
321
+ allow_failure: true
322
+ rules:
323
+ - if: '$CI_PIPELINE_SOURCE == "schedule" && $PIPE_CODEBASE_AUDIT == "1"'
324
+
325
+
326
+ # =============================================================================
327
+ # agent-architect: weekly, evidence-backed improvement proposals for shipped
328
+ # agent definitions. Schedule-only and opt-in because it spends model tokens;
329
+ # set DRY_RUN=true for an initial report-only trial.
330
+ # =============================================================================
331
+ agent-architect:
332
+ extends: .claude-base
333
+ stage: review
334
+ timeout: 30m
335
+ variables:
336
+ GIT_STRATEGY: clone
337
+ DRY_RUN: "$PIPE_ARCHITECT_DRY_RUN"
338
+ script:
339
+ - bash "$PIPE_SCRIPTS_DIR/agent-architect.sh"
340
+ allow_failure: true
341
+ rules:
342
+ - if: '$CI_PIPELINE_SOURCE == "schedule" && $PIPE_AGENT_ARCHITECT == "1"'
343
+ # =============================================================================
201
344
  # code: implement one issue and open the MR (issue -> code loop).
202
345
  # Triggered by orchestrate via the Pipeline Trigger API with CODER_ISSUE and
203
346
  # CODER_BRANCH set. The coder agent runs with --dangerously-skip-permissions on
@@ -214,6 +357,48 @@ code:
214
357
  rules:
215
358
  - if: '$CODER_ISSUE && $CODER_BRANCH'
216
359
 
360
+ # =============================================================================
361
+ # e2e-test-gen: scheduled, opt-in generation for issues carrying the configured
362
+ # test-ready label (and optional E2E scope label). The project supplies the
363
+ # verification command and any framework/runtime setup through its own CI.
364
+ # =============================================================================
365
+ e2e-test-gen:
366
+ extends: .claude-base
367
+ stage: test
368
+ timeout: 45m
369
+ variables:
370
+ GIT_STRATEGY: clone
371
+ script:
372
+ - bash "$PIPE_SCRIPTS_DIR/e2e-test-gen.sh"
373
+ allow_failure: true
374
+ rules:
375
+ - if: '$CI_PIPELINE_SOURCE == "schedule" && $PIPE_E2E_TEST_GEN == "1"'
376
+
377
+ # =============================================================================
378
+ # coverage-ratchet: optional MR coverage gate. Reads the project's coverage
379
+ # artifact and, for a wip coverage drop, fails into test-fix with its signal.
380
+ # =============================================================================
381
+ coverage-ratchet:
382
+ extends: .claude-base
383
+ stage: test
384
+ needs:
385
+ - job: test
386
+ artifacts: true
387
+ optional: true
388
+ variables:
389
+ GIT_STRATEGY: clone
390
+ script:
391
+ - bash "$PIPE_SCRIPTS_DIR/coverage-ratchet.sh"
392
+ artifacts:
393
+ paths:
394
+ - $PIPE_CONTEXT_DIR/metrics/
395
+ - $PIPE_COVERAGE_SIGNAL
396
+ when: always
397
+ expire_in: 60 days
398
+ rules:
399
+ - if: '$CI_PIPELINE_SOURCE == "merge_request_event" && $CI_MERGE_REQUEST_TARGET_BRANCH_NAME == $PIPE_TARGET_BRANCH && $PIPE_COVERAGE_RATCHET == "1"'
400
+ when: always
401
+
217
402
  # =============================================================================
218
403
  # test-fix: on failure of YOUR project's `test` job, on a wip MR, fix tests /
219
404
  # compilation / coverage. Consumes the test job's artifacts via needs.
@@ -226,6 +411,9 @@ test-fix:
226
411
  - job: test
227
412
  artifacts: true
228
413
  optional: true
414
+ - job: coverage-ratchet
415
+ artifacts: true
416
+ optional: true
229
417
  variables:
230
418
  GIT_STRATEGY: clone
231
419
  script:
@@ -234,3 +422,25 @@ test-fix:
234
422
  rules:
235
423
  - if: '$CI_PIPELINE_SOURCE == "merge_request_event" && $CI_MERGE_REQUEST_TARGET_BRANCH_NAME == $PIPE_TARGET_BRANCH && $CI_MERGE_REQUEST_LABELS =~ $PIPE_LABEL_WIP'
236
424
  when: on_failure
425
+
426
+ # =============================================================================
427
+ # metrics-snapshot: aggregates one day of pipeline metrics into a JSON snapshot.
428
+ # Optional and off by default. Enable by setting PIPE_METRICS_SNAPSHOT=1 on a
429
+ # scheduled pipeline. Writes an artifact; set PIPE_METRICS_COMMIT_REPO=true to
430
+ # also commit the snapshot to the target branch.
431
+ # =============================================================================
432
+ metrics-snapshot:
433
+ extends: .claude-base
434
+ stage: orchestrate
435
+ timeout: 30m
436
+ variables:
437
+ GIT_DEPTH: "0"
438
+ script:
439
+ - bash "$PIPE_SCRIPTS_DIR/metrics-snapshot.sh"
440
+ artifacts:
441
+ paths:
442
+ - $PIPE_METRICS_OUTPUT_DIR/
443
+ when: always
444
+ expire_in: 90 days
445
+ rules:
446
+ - if: '$CI_PIPELINE_SOURCE == "schedule" && $PIPE_METRICS_SNAPSHOT == "1"'