@hybridlabor-api/aos 4.2.0-beta.0 → 4.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/graph.md +3 -1
- package/.agents/nodes.json +4 -2
- package/.claude/workflows/startcycle-dispatch.mjs +18 -5
- package/.claude/workflows/teamwork-dispatch.mjs +287 -0
- package/CLAUDE.md +37 -84
- package/CODEX.md +31 -12
- package/GEMINI.md +51 -58
- package/README.de.md +13 -12
- package/README.md +13 -12
- package/README.pt.md +13 -12
- package/THIRD_PARTY_NOTICES.md +104 -0
- package/docs/skills_table.md +1 -0
- package/installer.js +27 -10
- package/package.json +6 -1
- package/scripts/validate-skills.mjs +402 -0
- package/skills/basic/bdbmediastorm/SKILL.md +7 -5
- package/skills/basic/godmode-engineering/SKILL.md +1 -1
- package/skills/basic/godmode-shipping/SKILL.md +1 -1
- package/skills/basic/startcycle/SKILL.md +3 -1
- package/skills/basic/startcycle-graph/SKILL.md +18 -2
- package/skills/basic/startcycle-graph-user/SKILL.md +3 -1
- package/skills/basic/teamwork-preview/SKILL.md +209 -0
- package/skills/bdbrainstorm/SKILL.md +4 -3
- package/skills/github-repo/SKILL.md +1 -0
- package/skills/global_config/agent-tool-builder/SKILL.md +5 -4
- package/skills/global_config/ai-product/SKILL.md +3 -2
- package/skills/global_config/ask-tim/SKILL.md +75 -6
- package/skills/global_config/{bdb-adobe-suite-mcp.md → bdb-adobe-suite-mcp/SKILL.md} +1 -0
- package/skills/global_config/{bdb-after-effects-mcp.md → bdb-after-effects-mcp/SKILL.md} +1 -0
- package/skills/global_config/{bdb-blender-mcp.md → bdb-blender-mcp/SKILL.md} +1 -0
- package/skills/global_config/{bdb-computer-use-mcp.md → bdb-computer-use-mcp/SKILL.md} +1 -0
- package/skills/global_config/{bdb-davinci-mcp.md → bdb-davinci-mcp/SKILL.md} +1 -0
- package/skills/global_config/bdb-ecosystem-health/SKILL.md +3 -3
- package/skills/global_config/{bdb-grandma3-mcp.md → bdb-grandma3-mcp/SKILL.md} +1 -0
- package/skills/global_config/{bdb-memb-mcp.md → bdb-memb-mcp/SKILL.md} +10 -0
- package/skills/global_config/{bdb-resolume-mcp.md → bdb-resolume-mcp/SKILL.md} +1 -0
- package/skills/global_config/{bdb-rhino-mcp.md → bdb-rhino-mcp/SKILL.md} +1 -0
- package/skills/global_config/{bdb-touchdesigner-mcp.md → bdb-touchdesigner-mcp/SKILL.md} +1 -0
- package/skills/global_config/{bdb-unreal-mcp.md → bdb-unreal-mcp/SKILL.md} +1 -0
- package/skills/global_config/{bdb-vectorworks-mcp.md → bdb-vectorworks-mcp/SKILL.md} +1 -0
- package/skills/global_config/bdbresilience/SKILL.md +216 -0
- package/skills/global_config/bdbresilience/contracts/nodes-integration.md +225 -0
- package/skills/global_config/bdbresilience/references/cicd-triage.md +179 -0
- package/skills/global_config/bdbresilience/references/distributed-locking.md +235 -0
- package/skills/global_config/bdbresilience/references/error-recovery.md +210 -0
- package/skills/global_config/bdbresilience/references/two-phase-go-gate.md +151 -0
- package/skills/global_config/bdbsaashost/SKILL.md +50 -50
- package/skills/global_config/browser-automation/SKILL.md +4 -4
- package/skills/global_config/crewai/SKILL.md +3 -2
- package/skills/global_config/debugger/SKILL.md +3 -6
- package/skills/global_config/domain-modeling/ADR-FORMAT.md +47 -0
- package/skills/global_config/domain-modeling/CONTEXT-FORMAT.md +60 -0
- package/skills/global_config/domain-modeling/SKILL.md +77 -0
- package/skills/global_config/git-advanced-workflows/SKILL.md +0 -1
- package/skills/global_config/github-actions-templates/SKILL.md +0 -2
- package/skills/global_config/google-sheets-automation/SKILL.md +2 -2
- package/skills/global_config/grill-me/SKILL.md +14 -0
- package/skills/global_config/grill-with-docs/SKILL.md +24 -0
- package/skills/global_config/grilling/SKILL.md +42 -0
- package/skills/global_config/neon-postgres/SKILL.md +3 -2
- package/skills/global_config/openwiki-skill/scripts/install_daemon.sh +55 -12
- package/skills/global_config/playwright-skill/SKILL.md +1 -1
- package/skills/global_config/posix-shell-pro/SKILL.md +0 -1
- package/skills/global_config/postgres-best-practices/SKILL.md +1 -1
- package/skills/global_config/prompt-engineering-patterns/SKILL.md +0 -1
- package/skills/global_config/rag-engineer/SKILL.md +4 -3
- package/skills/global_config/react-best-practices/SKILL.md +1 -1
- package/skills/global_config/remotion/SKILL.md +0 -1
- package/skills/global_config/seo/SKILL.md +6 -41
- package/skills/global_config/systematic-debugging/CREATION-LOG.md +1 -1
- package/skills/global_config/systematic-debugging/root-cause-tracing.md +1 -1
- package/skills/global_config/turborepo-caching/SKILL.md +0 -1
- package/skills/global_config/using-neon/SKILL.md +1 -47
- package/skills/global_config/vector-database-engineer/SKILL.md +0 -1
- package/skills/global_config/web-artifacts-builder/LICENSE.txt +1 -1
- package/skills/global_config/web-artifacts-builder/SKILL.md +1 -1
- package/skills/global_config/webapp-testing/LICENSE.txt +1 -1
- package/skills/global_config/webapp-testing/SKILL.md +1 -1
- package/.agents/skills/firecrawl/SKILL.md +0 -149
- package/.agents/skills/firecrawl/rules/install.md +0 -82
- package/.agents/skills/firecrawl/rules/security.md +0 -26
- package/.agents/skills/firecrawl-agent/SKILL.md +0 -58
- package/.agents/skills/firecrawl-build/SKILL.md +0 -39
- package/.agents/skills/firecrawl-build-interact/SKILL.md +0 -68
- package/.agents/skills/firecrawl-build-onboarding/SKILL.md +0 -103
- package/.agents/skills/firecrawl-build-onboarding/references/auth-flow.md +0 -39
- package/.agents/skills/firecrawl-build-onboarding/references/project-setup.md +0 -20
- package/.agents/skills/firecrawl-build-onboarding/references/sdk-installation.md +0 -17
- package/.agents/skills/firecrawl-build-scrape/SKILL.md +0 -69
- package/.agents/skills/firecrawl-build-search/SKILL.md +0 -69
- package/.agents/skills/firecrawl-crawl/SKILL.md +0 -59
- package/.agents/skills/firecrawl-download/SKILL.md +0 -70
- package/.agents/skills/firecrawl-interact/SKILL.md +0 -84
- package/.agents/skills/firecrawl-map/SKILL.md +0 -51
- package/.agents/skills/firecrawl-scrape/SKILL.md +0 -69
- package/.agents/skills/firecrawl-search/SKILL.md +0 -60
- package/mcps/RhinoMCP/cc-plugin/.claude/settings.json +0 -10
- package/mcps/after-effects-mcp/build/index.js +0 -840
- package/mcps/after-effects-mcp/build/scripts/applyEffect.jsx +0 -153
- package/mcps/after-effects-mcp/build/scripts/applyEffectTemplate.jsx +0 -218
- package/mcps/after-effects-mcp/build/scripts/createComposition.jsx +0 -71
- package/mcps/after-effects-mcp/build/scripts/createShapeLayer.jsx +0 -147
- package/mcps/after-effects-mcp/build/scripts/createSolidLayer.jsx +0 -114
- package/mcps/after-effects-mcp/build/scripts/createTextLayer.jsx +0 -115
- package/mcps/after-effects-mcp/build/scripts/getLayerInfo.jsx +0 -192
- package/mcps/after-effects-mcp/build/scripts/getProjectInfo.jsx +0 -90
- package/mcps/after-effects-mcp/build/scripts/listCompositions.jsx +0 -50
- package/mcps/after-effects-mcp/build/scripts/mcp-bridge-auto.jsx +0 -1773
- package/mcps/after-effects-mcp/build/scripts/setLayerProperties.jsx +0 -160
- package/mcps/bdb-remoteos-mcp/queue.db +0 -0
- package/mcps/bdb-remoteos-mcp/src/bdb_remoteos_mcp/__pycache__/__init__.cpython-312.pyc +0 -0
- package/mcps/bdb-remoteos-mcp/src/bdb_remoteos_mcp/__pycache__/incus_client.cpython-312.pyc +0 -0
- package/mcps/bdb-remoteos-mcp/src/bdb_remoteos_mcp/__pycache__/main.cpython-312.pyc +0 -0
- package/mcps/bdb-remoteos-mcp/src/bdb_remoteos_mcp/__pycache__/queue.cpython-312.pyc +0 -0
- package/mcps/bdb-remoteos-mcp/src/bdb_remoteos_mcp/__pycache__/schemas.cpython-312.pyc +0 -0
- package/mcps/bdb-remoteos-mcp/src/bdb_remoteos_mcp/__pycache__/server.cpython-312.pyc +0 -0
- package/mcps/bdb-remoteos-mcp/src/bdb_remoteos_mcp/__pycache__/webhook.cpython-312.pyc +0 -0
- package/mcps/bdb-remoteos-mcp/tests/__pycache__/__init__.cpython-312.pyc +0 -0
- package/mcps/bdb-remoteos-mcp/tests/__pycache__/mock_incus.cpython-312.pyc +0 -0
- package/mcps/bdb-remoteos-mcp/tests/__pycache__/test_mcp_server.cpython-312-pytest-9.1.1.pyc +0 -0
- package/mcps/bdb-remoteos-mcp/tests/__pycache__/test_security_redteam.cpython-312-pytest-9.1.1.pyc +0 -0
- package/mcps/bdb-remoteos-mcp/tests/__pycache__/test_webhook.cpython-312-pytest-9.1.1.pyc +0 -0
- package/mcps/computer-use-mcp/dist/client.d.ts +0 -150
- package/mcps/computer-use-mcp/dist/client.js +0 -136
- package/mcps/computer-use-mcp/dist/entrypoint.d.ts +0 -16
- package/mcps/computer-use-mcp/dist/entrypoint.js +0 -26
- package/mcps/computer-use-mcp/dist/native.d.ts +0 -212
- package/mcps/computer-use-mcp/dist/native.js +0 -50
- package/mcps/computer-use-mcp/dist/server.d.ts +0 -32
- package/mcps/computer-use-mcp/dist/server.js +0 -342
- package/mcps/computer-use-mcp/dist/session.d.ts +0 -101
- package/mcps/computer-use-mcp/dist/session.js +0 -2372
- package/skills/bdbsaastraining/scripts/__pycache__/build_profile.cpython-314.pyc +0 -0
|
@@ -0,0 +1,216 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: bdbresilience
|
|
3
|
+
description: Autonomous CI/CD error recovery, distributed file-based locking, diagnostic triage, and two-phase GO gate resilience engine for BDB Agent OS multi-agent pipelines.
|
|
4
|
+
category: bdb-core
|
|
5
|
+
triggers:
|
|
6
|
+
- error recovery
|
|
7
|
+
- transient error
|
|
8
|
+
- rate limit 429
|
|
9
|
+
- exponential backoff
|
|
10
|
+
- distributed lock
|
|
11
|
+
- lockfile
|
|
12
|
+
- concurrency control
|
|
13
|
+
- state race
|
|
14
|
+
- ci/cd triage
|
|
15
|
+
- log parser
|
|
16
|
+
- flaky test
|
|
17
|
+
- self-healing
|
|
18
|
+
- verification gate
|
|
19
|
+
- go gate
|
|
20
|
+
- two-phase gate
|
|
21
|
+
capabilities:
|
|
22
|
+
- monitoring
|
|
23
|
+
- security
|
|
24
|
+
- context-management
|
|
25
|
+
- testing
|
|
26
|
+
- automation
|
|
27
|
+
disable-model-invocation: false
|
|
28
|
+
user-invocable: true
|
|
29
|
+
---
|
|
30
|
+
|
|
31
|
+
# 🛡️ BDB Resilience & Concurrency Engine (`/bdbresilience`)
|
|
32
|
+
|
|
33
|
+
The `/bdbresilience` master skill suite provides deterministic runtime fault tolerance, concurrency coordination, diagnostic log triage, and pre-tool deployment gating for autonomous multi-agent pipelines operating across the BDB Agent OS ecosystem.
|
|
34
|
+
|
|
35
|
+
---
|
|
36
|
+
|
|
37
|
+
## 1. Core Tenets
|
|
38
|
+
|
|
39
|
+
1. **Fail Deterministically, Recover Gracefully**: Never swallow errors silently or crash abruptly. Intercept all tool, network, and runtime faults through a 4-tier taxonomy (`transient`, `tool_level_fault`, `auth_credential`, `unrecoverable`). Execute structured recovery before attempting escalation.
|
|
40
|
+
2. **Zero-Race Concurrency**: Shared mutable state—including `production_artifacts/state.json`, git branches, and worktrees—must never suffer lost updates or dirty reads. Enforce mutual exclusion through atomic POSIX/APFS file locking (`O_CREAT | O_EXCL`) or fragment isolation (`state.d/<nodeId>.json`).
|
|
41
|
+
3. **No Hallucinated Triage**: Parse raw CI/CD logs directly (Vitest, Jest, tsc, ESLint, GitHub Actions). Extract verified file coordinates, line numbers, failure diffs, and exact root causes into structured JSON reports. Never synthesize speculative fixes without log verification.
|
|
42
|
+
4. **Strict Two-Phase Gate Precedence**: High-consequence actions (`git push`, `npm publish`, `npm version`, recursive `rm`) require an uncompromised human verification gate. Enforce the BDB Pre-Tool Gate protocol: lock execution in strict read-only mode until a literal, isolated human token `"GO"` is validated.
|
|
43
|
+
|
|
44
|
+
---
|
|
45
|
+
|
|
46
|
+
## 2. When to Use vs. When to Exclude
|
|
47
|
+
|
|
48
|
+
### ✅ When to Use
|
|
49
|
+
- **External API & Network Flakiness**: Handling HTTP 429 rate limits, socket timeouts (`ETIMEDOUT`), connection resets (`ECONNRESET`), and temporary provider dropouts.
|
|
50
|
+
- **Tool Failures with Fallbacks**: Automatically rerouting failed tool invocations to secondary providers (e.g., primary MCP down → secondary CLI fallback) with immutable diversion audit trails.
|
|
51
|
+
- **Concurrent Agent Execution**: Protecting shared files (`state.json`), databases, or worktrees during parallel build node execution (Engineering, UI/UX, Media).
|
|
52
|
+
- **CI/CD Pipeline Failures**: Ingesting build and test runner failure logs, distinguishing transient infrastructure errors from code regressions, and generating actionable repair proposals.
|
|
53
|
+
- **Release Verification & Deployment**: Enforcing the two-phase approval gate prior to running destructive or outward-facing operations.
|
|
54
|
+
|
|
55
|
+
### ❌ When to Exclude
|
|
56
|
+
- **Single-Agent Read-Only Probes**: Reading static documentation, viewing local files, or querying local git status where no concurrency or network calls occur.
|
|
57
|
+
- **Internal Synchronous Transforms**: Pure CPU operations, in-memory string formatting, or deterministic array manipulations without side effects.
|
|
58
|
+
- **Explicit User-Directed Interrupts**: Direct manual cancellation or kill signals received from the operator.
|
|
59
|
+
|
|
60
|
+
---
|
|
61
|
+
|
|
62
|
+
## 3. Resilience Architecture & Operational Playbooks
|
|
63
|
+
|
|
64
|
+
```
|
|
65
|
+
+---------------------------------------------------------------------------------------+
|
|
66
|
+
| /bdbresilience Skill Suite |
|
|
67
|
+
| Master Operational Architecture |
|
|
68
|
+
+-------------------------------------------+-------------------------------------------+
|
|
69
|
+
|
|
|
70
|
+
+-----------------------------------+-----------------------------------+
|
|
71
|
+
| |
|
|
72
|
+
v v
|
|
73
|
+
+---------------------------------------+ +-----------------------------------------+
|
|
74
|
+
| Pattern 1: Error Recovery | | Pattern 2: Distributed Locking |
|
|
75
|
+
| - 4-Tier Failure Classification | | - POSIX Atomic Creation (O_CREAT|EXCL) |
|
|
76
|
+
| - Full Jitter Exponential Backoff | | - Monotonic Fencing Tokens |
|
|
77
|
+
| - Secondary Tool Router & Diversion | | - Background Heartbeat Renewal |
|
|
78
|
+
| - Fail-Closed Human Escalation | | - Atomic Stale / Orphan Eviction |
|
|
79
|
+
| [references/error-recovery.md] | | [references/distributed-locking.md] |
|
|
80
|
+
+---------------------------------------+ +-----------------------------------------+
|
|
81
|
+
| |
|
|
82
|
+
+-----------------------------------+-----------------------------------+
|
|
83
|
+
|
|
|
84
|
+
+-----------------------------------+-----------------------------------+
|
|
85
|
+
| |
|
|
86
|
+
v v
|
|
87
|
+
+---------------------------------------+ +-----------------------------------------+
|
|
88
|
+
| Pattern 3: CI/CD Triage | | Pattern 4: Two-Phase GO Gate |
|
|
89
|
+
| - Multi-Runner ANSI Normalization | | - Phase 1: Strict Read-Only Planning |
|
|
90
|
+
| - Flaky vs Deterministic Classifier | | - Phase 2: Literal Token Execution |
|
|
91
|
+
| - JSON Diagnostic Report Schema | | - Fail-Closed Transcript Scanner |
|
|
92
|
+
| - isCleanRun() Pass Predicate | | - Caller-Written approvals Ledger |
|
|
93
|
+
| [references/cicd-triage.md] | | [references/two-phase-go-gate.md] |
|
|
94
|
+
+---------------------------------------+ +-----------------------------------------+
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
### Specialized Reference Guides
|
|
98
|
+
- 📘 **[Error Recovery & Retry Routing](references/error-recovery.md)**: Taxonomy classification, Full Jitter backoff formula, diversion logging, and zero-hallucination escalation payloads.
|
|
99
|
+
- 📘 **[Distributed Locking & Concurrency](references/distributed-locking.md)**: Atomic lockfile creation, lease TTLs, heartbeat renewal, atomic rename break eviction, and `state.json` protection.
|
|
100
|
+
- 📘 **[CI/CD Self-Healing Triage](references/cicd-triage.md)**: Multi-runner log parsers, flaky vs. deterministic classifier, and structured JSON diagnostics.
|
|
101
|
+
- 📘 **[Two-Phase Pre-Tool GO Gate](references/two-phase-go-gate.md)**: PreToolUse hook specifications, transcript token verification, and fail-closed gate mechanics.
|
|
102
|
+
- 📘 **[Node Integration Contracts](contracts/nodes-integration.md)**: **PROPOSAL, not applied.** How Reviewer and Shipping *would* be equipped in `nodes.json`, plus hooks for `/startcycle-graph` and `/bdbrainstorm`. No AOS node carries `bdbresilience` today.
|
|
103
|
+
|
|
104
|
+
---
|
|
105
|
+
|
|
106
|
+
## 4. Operational Commands & Procedures
|
|
107
|
+
|
|
108
|
+
### `/bdbresilience recover`
|
|
109
|
+
Invokes the automated error classification and recovery pipeline for a failed operation.
|
|
110
|
+
```typescript
|
|
111
|
+
import { classifyError, withRetry, executeWithFallback } from 'bdb-cicd-resilience/recovery/index.js';
|
|
112
|
+
|
|
113
|
+
// 1. Classify error
|
|
114
|
+
const classification = classifyError(error, { toolName: 'api_fetch', attempt: 1 });
|
|
115
|
+
|
|
116
|
+
// 2. Retry with full jitter if transient
|
|
117
|
+
// (withRetry classifies internally too, and rethrows at once on canRetry: false)
|
|
118
|
+
if (classification.category === 'transient') {
|
|
119
|
+
const result = await withRetry(
|
|
120
|
+
() => callApi(),
|
|
121
|
+
{ maxRetries: 3, baseDelayMs: 500, maxDelayMs: 10000 }
|
|
122
|
+
);
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
// 3. Fallback route if tool fault. The fallback is a tool NAME; your execute
|
|
126
|
+
// function does the dispatch, and a canRetry:false classification suppresses
|
|
127
|
+
// the diversion instead of re-sending the call to a second provider.
|
|
128
|
+
if (classification.category === 'tool_level_fault') {
|
|
129
|
+
const fallbackResult = await executeWithFallback({
|
|
130
|
+
originalTool: 'mcp_primary',
|
|
131
|
+
fallbackTool: 'cli_fallback',
|
|
132
|
+
execute: (toolName) => invokeTool(toolName, args),
|
|
133
|
+
auditFilePath: 'production_artifacts/diversions.jsonl',
|
|
134
|
+
});
|
|
135
|
+
}
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
### `/bdbresilience lock`
|
|
139
|
+
Coordinates exclusive access to a shared resource using deterministic atomic locking.
|
|
140
|
+
```typescript
|
|
141
|
+
import { withStateLock, withDistributedLock } from 'bdb-cicd-resilience/locking/index.js';
|
|
142
|
+
|
|
143
|
+
// Guard state.json mutations
|
|
144
|
+
await withStateLock('production_artifacts/state.json', (state) => {
|
|
145
|
+
state.artifacts.backend = 'production_artifacts/02_backend_schema.md';
|
|
146
|
+
state.findings.push({ id: 'F-ENG-01', status: 'fixed' });
|
|
147
|
+
return state;
|
|
148
|
+
}, { ttlMs: 15000, acquireTimeoutMs: 10000 });
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
### `/bdbresilience triage`
|
|
152
|
+
Parses build or test logs to isolate failures and propose fixes.
|
|
153
|
+
```typescript
|
|
154
|
+
import { generateDiagnosticReport, isCleanRun } from 'bdb-cicd-resilience/triage/index.js';
|
|
155
|
+
|
|
156
|
+
const report = generateDiagnosticReport(rawBuildOutput, 'vitest');
|
|
157
|
+
console.log(`Failures: ${report.summary.totalFailures} (Transient: ${report.summary.transientCount})`);
|
|
158
|
+
if (isCleanRun(report)) {
|
|
159
|
+
// The ONLY sanctioned "it passed". `totalFailures === 0` alone is also true
|
|
160
|
+
// for an empty or unrecognised log (summary.parseStatus 'empty' / 'unparsed').
|
|
161
|
+
} else if (report.summary.canAutoRetry) {
|
|
162
|
+
// Safe infrastructure retry
|
|
163
|
+
} else {
|
|
164
|
+
// Escalate exact deterministic failure coordinates to engineer
|
|
165
|
+
}
|
|
166
|
+
```
|
|
167
|
+
|
|
168
|
+
### `/bdbresilience gate`
|
|
169
|
+
Evaluates pre-tool execution authorization against the conversation transcript.
|
|
170
|
+
```typescript
|
|
171
|
+
import { verifyGoGate, checkPreToolGate } from 'bdb-cicd-resilience/triage/index.js';
|
|
172
|
+
|
|
173
|
+
const gate = checkPreToolGate('git push origin main', transcriptPath);
|
|
174
|
+
if (!gate.allowed) {
|
|
175
|
+
throw new Error(`Gate Closed: ${gate.reason}. Reply with GO to unlock.`);
|
|
176
|
+
}
|
|
177
|
+
```
|
|
178
|
+
|
|
179
|
+
---
|
|
180
|
+
|
|
181
|
+
## 5. Common Rationalizations vs. Reality
|
|
182
|
+
|
|
183
|
+
| Rationalization | Engineering Reality | Resilience Protocol |
|
|
184
|
+
|-----------------|---------------------|---------------------|
|
|
185
|
+
| *"A simple `setTimeout(1000)` retry is fine without jitter."* | Synchronous fixed backoffs cause thundering herds on rate-limited endpoints. | **Mandatory Full Jitter**: Sleep for $T = \text{random}(0, \min(T_{\max}, T_{\text{base}} \cdot 2^{\text{attempt}}))$. |
|
|
186
|
+
| *"State file collisions won't happen because agents finish quickly."* | Parallel agent fan-out runs in separate processes; uncoordinated writes cause lost updates. | **Atomic POSIX Lock**: Use `open(..., 'wx')` with fencing tokens or isolated `state.d/<nodeId>.json` fragments. |
|
|
187
|
+
| *"The test failed due to an intermittent CI glitch, let's ignore it."* | Masking real assertion failures as "flaky" leads to shipping broken code to production. | **Deterministic Classifier**: Only classify network/OOM/timeout as flaky; assertion and type errors must never be bypassed. |
|
|
188
|
+
| *"The user said 'starte jetzt', so that counts as approval."* | Compound action verbs violate the two-phase safety protocol; user intent may be unconfirmed. | **Strict Literal Token**: Gate unlocks ONLY if the single trimmed word is `"GO"`. Fail-closed on everything else. |
|
|
189
|
+
|
|
190
|
+
---
|
|
191
|
+
|
|
192
|
+
## 6. Red Flags Checklist
|
|
193
|
+
|
|
194
|
+
- [ ] **Hardcoded Delays**: Retrying without random jitter or exponential backoff.
|
|
195
|
+
- [ ] **Missing Finally Block**: Acquiring a distributed lock without releasing in a `finally` block or relying on process exit.
|
|
196
|
+
- [ ] **Unlink Without Verification**: Deleting a stale lockfile without checking process liveness (`process.kill(pid, 0)`) or using the atomic rename break protocol.
|
|
197
|
+
- [ ] **Synthetic Error Hallucination**: Fabricating an error root cause when log parsing failed or was inconclusive.
|
|
198
|
+
- [ ] **Soft Gate Bypass**: Proceeding with deployment when gate evaluation returned `allowed: false` or when transcript was missing.
|
|
199
|
+
- [ ] **Sidechain Approval**: Permitting an automated subagent message (`isSidechain: true`) to satisfy human approval.
|
|
200
|
+
|
|
201
|
+
---
|
|
202
|
+
|
|
203
|
+
## 7. Verification & Testing Protocol
|
|
204
|
+
|
|
205
|
+
To verify the `/bdbresilience` suite and its underlying TypeScript engine:
|
|
206
|
+
|
|
207
|
+
```bash
|
|
208
|
+
# 1. Typecheck TypeScript implementation
|
|
209
|
+
npm run typecheck
|
|
210
|
+
|
|
211
|
+
# 2. Compile to dist/ distribution
|
|
212
|
+
npm run build
|
|
213
|
+
|
|
214
|
+
# 3. Execute full unit, integration, and E2E verification
|
|
215
|
+
npm test
|
|
216
|
+
```
|
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
# 🤝 Node & Pipeline Integration Contracts
|
|
2
|
+
|
|
3
|
+
> **Status: partially applied. Read the split below before relying on anything here.**
|
|
4
|
+
>
|
|
5
|
+
> **Applied (2026-09-09).** This skill ships inside AOS at
|
|
6
|
+
> `skills/global_config/bdbresilience/`, and `.agents/nodes.json` lists
|
|
7
|
+
> `bdbresilience` in the `skills` array of two nodes: **`shipping`** (it runs
|
|
8
|
+
> lint/typecheck/tests, which is exactly what the triage parsers read) and
|
|
9
|
+
> **`engineering`** (the error taxonomy and backoff guidance). The dispatcher
|
|
10
|
+
> injects each node's `skills` list into its prompt, so those two nodes are told
|
|
11
|
+
> to reach for this skill. That is the whole of the integration: **guidance, not
|
|
12
|
+
> code.** AOS does not depend on the `bdb-cicd-resilience` package — it has two
|
|
13
|
+
> runtime dependencies and neither is this one.
|
|
14
|
+
>
|
|
15
|
+
> **Not applied, and not currently possible.** Every code-level integration below.
|
|
16
|
+
> `startcycle-dispatch.mjs` contains no `withStateLock` call and cannot contain one:
|
|
17
|
+
> the Workflow runtime gives the dispatcher script **no filesystem access**, which is
|
|
18
|
+
> precisely why AOS solves its own concurrent-write race with single-writer fragments
|
|
19
|
+
> (`production_artifacts/state.d/<node>.json`) plus a merge step instead of a lock.
|
|
20
|
+
> A lock needs a filesystem; the dispatcher does not have one. Do not wire one in.
|
|
21
|
+
>
|
|
22
|
+
> **Where a lock would genuinely fit**, if this is revisited: `installer.js`, which
|
|
23
|
+
> does a real read-modify-write on `~/.agents/.bdb-install-manifest.json` and can be
|
|
24
|
+
> run twice concurrently. That is the only place in AOS with both filesystem access
|
|
25
|
+
> and real contention. Before doing it, note that this library's Windows paths
|
|
26
|
+
> (`EPERM`/`EBUSY` handling, close-before-unlink ordering) are **written but never
|
|
27
|
+
> executed** — two tests skip on non-Windows — and AOS ships to Windows users.
|
|
28
|
+
>
|
|
29
|
+
> §6 documents the `withStateLock()` signature as it actually exists in
|
|
30
|
+
> `src/locking/resource-lock.ts`.
|
|
31
|
+
|
|
32
|
+
**Deliverable**: Requirement R4 Integration Contracts (proposal)
|
|
33
|
+
**Target Architecture**: BDB Agent OS (`.agents/nodes.json`, `/startcycle-graph`, `/bdbrainstorm`)
|
|
34
|
+
**Specification Reference**: `survey_miner_1/survey_report.md`
|
|
35
|
+
|
|
36
|
+
---
|
|
37
|
+
|
|
38
|
+
## 1. Proposed Architectural Role
|
|
39
|
+
|
|
40
|
+
The proposal is to wire the `/bdbresilience` skill suite into the BDB Agent OS multi-agent dispatcher graph. Equipping the **Reviewer** and **Shipping** nodes with resilience capabilities would give the ecosystem automated adversarial auditing of network and concurrency boundaries, and deterministic self-healing quality gates.
|
|
41
|
+
|
|
42
|
+
Resilience principles would additionally be seeded upstream in **`/bdbrainstorm`** during concept ideation, so that downstream implementation nodes (Engineering, UI/UX, Media) receive explicit reliability requirements.
|
|
43
|
+
|
|
44
|
+
Applying this proposal is out of scope for the current cycle. It belongs to an AOS integration cycle, which is also where several deferred design questions get their answers — fencing-token verification, the default `ttlMs`, and retry idempotency.
|
|
45
|
+
|
|
46
|
+
---
|
|
47
|
+
|
|
48
|
+
## 2. Not adopted: the Reviewer node
|
|
49
|
+
|
|
50
|
+
> **This proposal was considered and declined on 2026-09-09.** The live
|
|
51
|
+
> `nodes.json` does **not** carry `bdbresilience` in `reviewer.skills`, and that is
|
|
52
|
+
> deliberate, not an oversight.
|
|
53
|
+
>
|
|
54
|
+
> Reviewer reads build artifacts and the plan's contract and argues about
|
|
55
|
+
> correctness. It does not run CI, does not read tool logs, and does not classify
|
|
56
|
+
> retryable failures — the three things this skill is for. A skill allowlist that
|
|
57
|
+
> lists everything guides nothing, so the registration went to `shipping` (which
|
|
58
|
+
> runs the gates whose output the triage parsers read) and `engineering` (which
|
|
59
|
+
> owns the error taxonomy) and stopped there.
|
|
60
|
+
>
|
|
61
|
+
> The section is kept because the reasoning is worth having on record, and because
|
|
62
|
+
> the shape of the entry is a useful template if a future node genuinely needs it.
|
|
63
|
+
|
|
64
|
+
The **Reviewer** node executes adversarial verification of build-node outputs against the execution plan contract using the *doubt-driven development* discipline. The JSON below is the entry that **would** be added, had this been adopted.
|
|
65
|
+
|
|
66
|
+
### Declarative Configuration (`.agents/nodes.json`)
|
|
67
|
+
```json
|
|
68
|
+
{
|
|
69
|
+
"reviewer": {
|
|
70
|
+
"label": "Reviewer",
|
|
71
|
+
"agentType": "reviewer",
|
|
72
|
+
"personaFile": ".claude/agents/reviewer.md",
|
|
73
|
+
"model": "sonnet",
|
|
74
|
+
"role": "review",
|
|
75
|
+
"artifactKey": "review",
|
|
76
|
+
"writes": null,
|
|
77
|
+
"optional": false,
|
|
78
|
+
"skills": [
|
|
79
|
+
"ui-review",
|
|
80
|
+
"ux-audit",
|
|
81
|
+
"architect-review",
|
|
82
|
+
"systematic-debugging",
|
|
83
|
+
"bdbresilience"
|
|
84
|
+
],
|
|
85
|
+
"instructions": null
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
### Proposed Reviewer Resilience Audit Discipline
|
|
91
|
+
Once equipped with `bdbresilience`, Reviewer would evaluate build outputs (`state.artifacts.{frontend,backend,media}`) against strict resilience checks:
|
|
92
|
+
|
|
93
|
+
1. **Network & Tool Call Audits**:
|
|
94
|
+
- Verifies that all external HTTP, API, database, and MCP calls implement structured error classification and Full Jitter exponential backoff.
|
|
95
|
+
- Flags bare `catch (e) {}` blocks, unhandled Promise rejections, and silent error swallows as:
|
|
96
|
+
`{ "id": "F-RES-NET-01", "severity": "blocking", "node": "engineering", "status": "open" }`.
|
|
97
|
+
2. **Concurrency & Resource Access Audits**:
|
|
98
|
+
- Verifies that parallel operations accessing shared files, worktrees, or databases either write isolated fragments (`production_artifacts/state.d/<nodeId>.json`) or acquire a distributed file lock.
|
|
99
|
+
- Flags missing `finally { await handle.release(); }` blocks as `severity: "blocking"`.
|
|
100
|
+
3. **Ownership Attribution & Stable ID Generation**:
|
|
101
|
+
- Findings must be attributed strictly to the owning build node (`engineering`, `ui_ux`, or `media`).
|
|
102
|
+
- Finding IDs must remain stable across cycles (e.g. `F-ENG-01`) so the dispatcher's **No-Progress Guard** can detect stalled repair loops and escalate immediately to human review.
|
|
103
|
+
|
|
104
|
+
---
|
|
105
|
+
|
|
106
|
+
## 3. Applied: the Shipping node (`.agents/nodes.json`)
|
|
107
|
+
|
|
108
|
+
> **Adopted 2026-09-09.** `shipping.skills` now ends with `"bdbresilience"`, and
|
|
109
|
+
> `engineering.skills` likewise. Those two are the whole of the applied
|
|
110
|
+
> integration — see the status block at the top of this file.
|
|
111
|
+
|
|
112
|
+
The **Shipping** node acts as the release gatekeeper, running mechanical verification gates (lint, typecheck, tests, a11y, seo) after Reviewer findings are cleared. It is the natural home for this skill: the triage parsers read exactly the tsc, ESLint and Jest/Vitest output those gates produce, and `isCleanRun()` is the predicate that decides whether an unrecognised log counts as passing (it does not).
|
|
113
|
+
|
|
114
|
+
### Declarative Configuration (`.agents/nodes.json`)
|
|
115
|
+
```json
|
|
116
|
+
{
|
|
117
|
+
"shipping": {
|
|
118
|
+
"label": "Godmode_Shipping",
|
|
119
|
+
"agentType": "godmode-shipping",
|
|
120
|
+
"personaFile": ".claude/agents/godmode-shipping.md",
|
|
121
|
+
"model": "sonnet",
|
|
122
|
+
"role": "gate",
|
|
123
|
+
"artifactKey": "report",
|
|
124
|
+
"writes": null,
|
|
125
|
+
"optional": false,
|
|
126
|
+
"skills": [
|
|
127
|
+
"godmode-shipping",
|
|
128
|
+
"webapp-testing",
|
|
129
|
+
"seo-audit",
|
|
130
|
+
"wcag-audit-patterns",
|
|
131
|
+
"github-repo",
|
|
132
|
+
"clean-code",
|
|
133
|
+
"bdbresilience"
|
|
134
|
+
],
|
|
135
|
+
"instructions": null
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
### Proposed Shipping Quality Gate & Verification Discipline
|
|
141
|
+
Once equipped with `bdbresilience`, Shipping would execute automated diagnostic triage and gating:
|
|
142
|
+
|
|
143
|
+
1. **Automated Diagnostic Triage Execution**:
|
|
144
|
+
- Runs mechanical verification commands: `npm run typecheck`, `npm run build`, `npm test`, `npm run lint`.
|
|
145
|
+
- If a command fails, Shipping pipes stderr/stdout directly into `generateDiagnosticReport(log, runner)`:
|
|
146
|
+
- A run counts as clean **only if `isCleanRun(report)` is true**, never on `totalFailures === 0` alone — an empty or unparseable log also has zero diagnostics, and reading that as a pass is a fail-open gate.
|
|
147
|
+
- If all failures are classified as `transient_infra` (`canAutoRetry: true`): Shipping automatically retries the command up to 2 times with Full Jitter before recording a failure.
|
|
148
|
+
- If any failure is classified as `deterministic_code_regression`: Shipping generates a structured failure section in `production_artifacts/04_release_report.md` detailing the exact failed assertion, target file, line coordinate, and remediation command.
|
|
149
|
+
2. **Gate Population & Node Attribution**:
|
|
150
|
+
- Updates `state.gate`:
|
|
151
|
+
```json
|
|
152
|
+
{
|
|
153
|
+
"lint": "pass",
|
|
154
|
+
"typecheck": "fail",
|
|
155
|
+
"tests": "pass",
|
|
156
|
+
"a11y": "skip",
|
|
157
|
+
"seo": "skip",
|
|
158
|
+
"blockingNodes": ["engineering"]
|
|
159
|
+
}
|
|
160
|
+
```
|
|
161
|
+
- Instructs the dispatcher to re-invoke only the failing node (`engineering`) rather than re-running all build nodes.
|
|
162
|
+
3. **Pre-Tool GO Gate Release Protocol**:
|
|
163
|
+
- When all checks pass, Shipping sets `state.phase = "ready_to_ship"`.
|
|
164
|
+
- Concludes turn by outputting the Release Report and instructing the operator:
|
|
165
|
+
> *"All quality gates passed. Reply with the literal word GO to authorize release deployment."*
|
|
166
|
+
- Strictly prohibits calling `git push` or `npm publish` without human approval verified via `verifyGoGate()`.
|
|
167
|
+
|
|
168
|
+
---
|
|
169
|
+
|
|
170
|
+
## 4. Proposed: Integration with `/startcycle-graph`
|
|
171
|
+
|
|
172
|
+
`~/.claude/workflows/startcycle-dispatch.mjs` and `~/.claude/hooks/graph-gate.mjs` both exist, but neither references this library. The proposal is that `bdbresilience` would power core graph infrastructure:
|
|
173
|
+
|
|
174
|
+
1. **Parallel Build Fan-Out Isolation**:
|
|
175
|
+
- When `ui_ux`, `engineering`, and `media` execute concurrently, each node would write exclusively to:
|
|
176
|
+
`production_artifacts/state.d/<nodeId>.json`
|
|
177
|
+
- A dedicated barrier folding agent would consolidate fragments into `production_artifacts/state.json` under `withStateLock()`. **This call does not exist in the dispatcher today.**
|
|
178
|
+
2. **Loop Retention Hook (`graph-gate.mjs`)**:
|
|
179
|
+
- Would intercept turn completion if `state.gate` contains failing checks and `iteration < max_iterations`, blocking session turn-end with exit code 2 to force repair execution. Whether the existing hook already does this independently of `bdbresilience` was not verified for this document.
|
|
180
|
+
3. **No-Progress Guard**:
|
|
181
|
+
- If Reviewer reports the exact same set of blocking finding IDs across successive iterations, the dispatcher halts immediately (`state.phase = "escalated"`, `state.needs_human = true`). This guard is part of the AOS graph contract, not of this library.
|
|
182
|
+
|
|
183
|
+
---
|
|
184
|
+
|
|
185
|
+
## 5. Proposed: Integration with `/bdbrainstorm`
|
|
186
|
+
|
|
187
|
+
The proposal is that `~/.agents/skills/bdbrainstorm/SKILL.md` would embed resilience requirements into concept planning. Its current content was not inspected for this document:
|
|
188
|
+
|
|
189
|
+
1. **Pillar 2: `/grill-me` Interactive Inquiries**:
|
|
190
|
+
- Grills the user on SLA thresholds, API rate limits, failure blast radius, and recovery procedures:
|
|
191
|
+
- *"What are the rate limits and fallback providers for external dependency X?"*
|
|
192
|
+
- *"How will concurrent writes to shared database entities be serialized?"*
|
|
193
|
+
2. **Pillar 4: Engineering Godmode (DDD & Clean Architecture)**:
|
|
194
|
+
- Mandates that domain models represent failure states explicitly as Discriminated Unions (e.g. `Result<T, ClassifiedError>`).
|
|
195
|
+
- Requires Architecture Decision Records (ADRs) for locking mechanisms and retry backoff strategies.
|
|
196
|
+
3. **Pillar 6: Shipping Godmode & Pipeline Hand-off**:
|
|
197
|
+
- Packages resilience specifications directly into `state.goal` so downstream Architect and TechLead nodes include them in `production_artifacts/00_execution_plan.md`.
|
|
198
|
+
|
|
199
|
+
---
|
|
200
|
+
|
|
201
|
+
## 6. `withStateLock()` — the one part of this document that is real
|
|
202
|
+
|
|
203
|
+
This function exists today in `src/locking/resource-lock.ts` and behaves as described. It is what a barrier folding agent in §4 would call.
|
|
204
|
+
|
|
205
|
+
```typescript
|
|
206
|
+
import { withStateLock } from 'bdb-cicd-resilience/locking/index.js';
|
|
207
|
+
|
|
208
|
+
const merged = await withStateLock(
|
|
209
|
+
'production_artifacts/state.json',
|
|
210
|
+
async (state) => {
|
|
211
|
+
state.findings.push({ id: 'F-ENG-01', status: 'fixed' });
|
|
212
|
+
return state;
|
|
213
|
+
},
|
|
214
|
+
{ ttlMs: 15000, acquireTimeoutMs: 10000 }
|
|
215
|
+
);
|
|
216
|
+
```
|
|
217
|
+
|
|
218
|
+
- Signature: `withStateLock<T>(statePath, updater, options?) => Promise<T>`.
|
|
219
|
+
- The lock resource key is `state:<resolved absolute path>`, so two callers passing different relative paths to the same file still serialize.
|
|
220
|
+
- `acquireTimeoutMs` defaults to `10000` here (not `0` as in `acquireLock`), so callers queue rather than fail on first contention.
|
|
221
|
+
- A missing state file is read as `{}`; any other read error propagates.
|
|
222
|
+
- The updater's return value is written; returning `undefined` writes the state object as mutated.
|
|
223
|
+
- **The write is refused if the lease was lost while the updater ran** — `handle.isExpired()` is checked after the updater returns and before the temp+rename write, and throws rather than clobber a successor's state file.
|
|
224
|
+
- The state file itself *is* written temp+rename. That is unrelated to the lockfile, which is never renamed into place (see `references/distributed-locking.md` §2).
|
|
225
|
+
- The lock is released in a `finally`, and release errors are swallowed.
|
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
# 🩺 Reference Guide: CI/CD Self-Healing Triage & Diagnostic Reporting
|
|
2
|
+
|
|
3
|
+
**Pattern**: Pattern 3 — Build, Test & Lint Failure Triage
|
|
4
|
+
**Module**: `bdb-cicd-resilience/triage`
|
|
5
|
+
**Authoritative Source**: BDB Agent OS CI/CD Triage Specification
|
|
6
|
+
|
|
7
|
+
---
|
|
8
|
+
|
|
9
|
+
## 1. Overview & Problem Statement
|
|
10
|
+
|
|
11
|
+
When automated CI/CD pipelines fail (during `npm test`, `tsc --noEmit`, `eslint`, or GitHub Actions runs), agents frequently hallucinate root causes by reading unstructured terminal output or trying random edits without identifying the real failure. Alternatively, agents waste execution turns retrying deterministic code bugs, or conversely halt prematurely on transient infrastructure hiccups (such as an npm registry timeout or port bind collision).
|
|
12
|
+
|
|
13
|
+
The CI/CD Triage Engine ingests raw terminal logs, normalizes ANSI codes and runner annotations, parses failure coordinates across 5 major runners, classifies failures into transient infrastructure versus deterministic regressions, and produces verified JSON diagnostic summaries with exact remediation commands.
|
|
14
|
+
|
|
15
|
+
---
|
|
16
|
+
|
|
17
|
+
## 2. Multi-Runner Log Ingestion & Normalization
|
|
18
|
+
|
|
19
|
+
Raw build logs contain noisy ANSI color sequences, carriage return overwrites (`\r\n`), and runner-specific group wrappers (`##[group]`, `::error::`).
|
|
20
|
+
|
|
21
|
+
### ANSI Normalization Pipeline
|
|
22
|
+
1. **Control Sequence Stripping**: Removes terminal escapes (`\x1b[[0-9;]*[mGKF]`).
|
|
23
|
+
2. **CRLF Normalization**: Converts Windows line endings to standard Unix newlines (`\r\n` → `\n`).
|
|
24
|
+
3. **CI Annotation Scrubbing**: Normalizes GitHub Actions workflow commands:
|
|
25
|
+
- `##[group]...##[endgroup]`
|
|
26
|
+
- `##[error]...`
|
|
27
|
+
- `::error file={name},line={line}::{message}`
|
|
28
|
+
|
|
29
|
+
---
|
|
30
|
+
|
|
31
|
+
## 3. Runner-Specific Log Parsers
|
|
32
|
+
|
|
33
|
+
The engine incorporates dedicated parsers for each standard ecosystem tool:
|
|
34
|
+
|
|
35
|
+
| Runner | Detected Signatures | Extracted Data Coordinates | Remediation Command Generated |
|
|
36
|
+
|--------|---------------------|----------------------------|-------------------------------|
|
|
37
|
+
| **Vitest** | `FAIL tests/...`<br>`AssertionError:`<br>`expected ... to be ...` | Test file path, line, column, assertion diff snippet | `npx vitest run <file> -t "<test>"` |
|
|
38
|
+
| **Jest** | `● <describe> › <test>`<br>`Expected: ... Received: ...`<br>`at ... (<file>:<line>:<col>)` | Test spec file, stack frame line/col (prioritizing user code over `node_modules`) | `npx jest <file> -t "<test>"` |
|
|
39
|
+
| **TypeScript (`tsc`)** | `src/auth.ts(42,15): error TS2322`<br>`src/auth.ts:42:15 - error TS2322` | Source file, line, column, TS error code, type mismatch explanation | `npx tsc --noEmit` |
|
|
40
|
+
| **ESLint** | `<file>:<line>:<col>: <msg> [<rule>]`<br>Tabular/Stylish reporter lines | Target source file, line, col, rule ID, severity | `npx eslint --fix <file>` |
|
|
41
|
+
| **GitHub Actions** | `##[error]Process completed with exit code 137`<br>`The operation was canceled`<br>`Request timeout after 30000ms` | Runner step, exit code, memory or timeout fault | Automated infrastructure retry with jitter |
|
|
42
|
+
|
|
43
|
+
---
|
|
44
|
+
|
|
45
|
+
## 4. Flaky Infrastructure vs. Deterministic Regression Classifier
|
|
46
|
+
|
|
47
|
+
The core intelligence distinguishes between errors that can be safely retried and code bugs requiring source modification:
|
|
48
|
+
|
|
49
|
+
```
|
|
50
|
+
[ Parsed Diagnostic Error ]
|
|
51
|
+
│
|
|
52
|
+
▼
|
|
53
|
+
┌───────────────────────────────────────────────┐
|
|
54
|
+
│ classifyDiagnostic(error, context) │
|
|
55
|
+
└───────────────────────┬───────────────────────┘
|
|
56
|
+
│
|
|
57
|
+
┌───────────────────────┴───────────────────────┐
|
|
58
|
+
▼ ▼
|
|
59
|
+
[ TRANSIENT INFRASTRUCTURE ] [ DETERMINISTIC REGRESSION ]
|
|
60
|
+
- HTTP 429 Throttling - AssertionError (Expected X, got Y)
|
|
61
|
+
- Socket ETIMEDOUT / ECONNRESET - TypeScript Type Error (TS2322)
|
|
62
|
+
- Runner Exit Code 137 (OOM) - SyntaxError / Parsing Failure
|
|
63
|
+
- EADDRINUSE (Port collision) - ESLint Rule Violation
|
|
64
|
+
- Registry 503 / DNS Blip - Missing Export / ReferenceError
|
|
65
|
+
│ │
|
|
66
|
+
▼ ▼
|
|
67
|
+
canAutoRetry: true canAutoRetry: false
|
|
68
|
+
(Retry step with Full Jitter) (Requires source code repair)
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
### Classification Heuristics
|
|
72
|
+
- **Transient Infrastructure (`transient_infra`)**:
|
|
73
|
+
- `canAutoRetry: true`
|
|
74
|
+
- Does NOT count against agent repair loop limits.
|
|
75
|
+
- Automatically triggered up to 2 times before escalating.
|
|
76
|
+
- **Deterministic Code Regression (`deterministic_code_regression`)**:
|
|
77
|
+
- `canAutoRetry: false`
|
|
78
|
+
- Attributed to the responsible build node (Engineering, UI/UX).
|
|
79
|
+
- Feeds into `production_artifacts/review_findings.md` or `state.findings`.
|
|
80
|
+
- **Unrecognised text** falls to `deterministic_code_regression`, i.e. this classifier **fails closed**. That is the opposite of `classifyError`'s unknown default in the recovery module, and deliberately so — see `error-recovery.md` §7 for why neither should be changed to match the other.
|
|
81
|
+
|
|
82
|
+
---
|
|
83
|
+
|
|
84
|
+
## 5. Was Anything Understood? — `parseStatus` and `isCleanRun`
|
|
85
|
+
|
|
86
|
+
`totalFailures === 0` is **not** the answer to "did this run pass". A log that was empty, and a log that no parser recognised, both produce zero diagnostics. Reading either as a clean gate is a fail-open defect, so the report carries a third axis:
|
|
87
|
+
|
|
88
|
+
| `parseStatus` | When | `totalFailures` | `canAutoRetry` | `requiresHumanIntervention` |
|
|
89
|
+
|---|---|---|---|---|
|
|
90
|
+
| `parsed` | a runner signature matched, or a parser extracted ≥1 diagnostic | as extracted | `n > 0 && transient === n` | `deterministic > 0` |
|
|
91
|
+
| `empty` | the log is empty or pure ANSI after stripping | `0` | `false` | `true` |
|
|
92
|
+
| `unparsed` | nothing matched — e.g. a Go, pytest or Maven failure | `0` | `false` | `true` |
|
|
93
|
+
|
|
94
|
+
`empty` is deliberately **not** clean: a build step that produced no output is not evidence that it passed. Neither status ever carries a synthetic diagnostic — inventing one would inflate `totalFailures` and lie to every consumer counting failures.
|
|
95
|
+
|
|
96
|
+
The predicate is exported so no consumer has to reconstruct the rule:
|
|
97
|
+
|
|
98
|
+
```typescript
|
|
99
|
+
import { isCleanRun } from 'bdb-cicd-resilience/triage/index.js';
|
|
100
|
+
|
|
101
|
+
if (isCleanRun(report)) { /* the only sanctioned "it passed" */ }
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
`isCleanRun(report) === (report.summary.parseStatus === 'parsed' && report.summary.totalFailures === 0)`. Any AOS integration must call it rather than reading `totalFailures` directly.
|
|
105
|
+
|
|
106
|
+
---
|
|
107
|
+
|
|
108
|
+
## 6. Structured JSON Diagnostic Schema
|
|
109
|
+
|
|
110
|
+
`dominantCategory` is `undefined` whenever `totalFailures === 0`; it is not defaulted to a classification.
|
|
111
|
+
|
|
112
|
+
```json
|
|
113
|
+
{
|
|
114
|
+
"timestamp": "2026-09-05T14:45:00.000Z",
|
|
115
|
+
"runner": "vitest",
|
|
116
|
+
"summary": {
|
|
117
|
+
"totalFailures": 2,
|
|
118
|
+
"transientCount": 1,
|
|
119
|
+
"deterministicCount": 1,
|
|
120
|
+
"parseStatus": "parsed",
|
|
121
|
+
"dominantCategory": "deterministic_code_regression",
|
|
122
|
+
"canAutoRetry": false,
|
|
123
|
+
"requiresHumanIntervention": true
|
|
124
|
+
},
|
|
125
|
+
"diagnostics": [
|
|
126
|
+
{
|
|
127
|
+
"file": "tests/unit/auth.test.ts",
|
|
128
|
+
"line": 42,
|
|
129
|
+
"column": 14,
|
|
130
|
+
"assertionSnippet": "expect(user.isAuthenticated).toBe(true)",
|
|
131
|
+
"classification": "deterministic_code_regression",
|
|
132
|
+
"rootCause": "AssertionError: expected false to be true // Received user object without auth token",
|
|
133
|
+
"remediationCommand": "npx vitest run tests/unit/auth.test.ts -t \"verifies authenticated user\"",
|
|
134
|
+
"canAutoRetry": false
|
|
135
|
+
},
|
|
136
|
+
{
|
|
137
|
+
"file": "src/services/api.ts",
|
|
138
|
+
"classification": "transient_infra",
|
|
139
|
+
"rootCause": "FetchError: request to https://registry.npmjs.org timed out after 30000ms (ETIMEDOUT)",
|
|
140
|
+
"remediationCommand": "npm cache clean --force && npm install",
|
|
141
|
+
"canAutoRetry": true
|
|
142
|
+
}
|
|
143
|
+
]
|
|
144
|
+
}
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
---
|
|
148
|
+
|
|
149
|
+
## 7. TypeScript API Usage Example
|
|
150
|
+
|
|
151
|
+
```typescript
|
|
152
|
+
import {
|
|
153
|
+
generateDiagnosticReport,
|
|
154
|
+
classifyDiagnostic,
|
|
155
|
+
isCleanRun,
|
|
156
|
+
stripAnsi
|
|
157
|
+
} from 'bdb-cicd-resilience/triage/index.js';
|
|
158
|
+
|
|
159
|
+
// Clean and parse raw output
|
|
160
|
+
const cleanLog = stripAnsi(rawStderr);
|
|
161
|
+
const report = generateDiagnosticReport(cleanLog);
|
|
162
|
+
|
|
163
|
+
if (isCleanRun(report)) {
|
|
164
|
+
console.log('✅ All checks passed clean.');
|
|
165
|
+
} else if (report.summary.parseStatus !== 'parsed') {
|
|
166
|
+
console.error(`❌ Log was ${report.summary.parseStatus} — nothing was understood, not a pass.`);
|
|
167
|
+
} else if (report.summary.canAutoRetry && report.summary.deterministicCount === 0) {
|
|
168
|
+
console.log('⚠️ Transient infrastructure failure detected. Executing auto-retry...');
|
|
169
|
+
await executeRetryCommand();
|
|
170
|
+
} else {
|
|
171
|
+
console.error('❌ Deterministic code regression detected:');
|
|
172
|
+
for (const diag of report.diagnostics) {
|
|
173
|
+
if (diag.classification === 'deterministic_code_regression') {
|
|
174
|
+
console.error(` - ${diag.file}:${diag.line} -> ${diag.rootCause}`);
|
|
175
|
+
console.error(` Suggested fix command: ${diag.remediationCommand}`);
|
|
176
|
+
}
|
|
177
|
+
}
|
|
178
|
+
}
|
|
179
|
+
```
|