@uluops/setup 0.9.8 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/assets/codex/agents/anxiety-reader-agent.toml +27 -1
- package/assets/codex/agents/api-contract-validator-agent.toml +5 -1
- package/assets/codex/agents/aristotle-analyst-agent.toml +26 -15
- package/assets/codex/agents/aristotle-explorer-agent.toml +26 -1
- package/assets/codex/agents/aristotle-forecaster-agent.toml +31 -1
- package/assets/codex/agents/aristotle-validator-agent.toml +53 -17
- package/assets/codex/agents/assumption-excavator-agent.toml +34 -31
- package/assets/codex/agents/code-auditor-agent.toml +5 -1
- package/assets/codex/agents/code-optimizer-agent.toml +5 -1
- package/assets/codex/agents/code-validator-agent.toml +5 -1
- package/assets/codex/agents/docs-validator-agent.toml +5 -1
- package/assets/codex/agents/frontend-validator-agent.toml +12 -8
- package/assets/codex/agents/mcp-validator-agent.toml +5 -1
- package/assets/codex/agents/pre-implementation-architect-agent.toml +5 -1
- package/assets/codex/agents/prompt-engineer-agent.toml +5 -1
- package/assets/codex/agents/prompt-pattern-analyzer-agent.toml +5 -1
- package/assets/codex/agents/prompt-quality-validator-agent.toml +5 -1
- package/assets/codex/agents/public-interface-validator-agent.toml +5 -1
- package/assets/codex/agents/release-readiness-agent.toml +5 -1
- package/assets/codex/agents/security-analyst-agent.toml +5 -1
- package/assets/codex/agents/test-architect-agent.toml +5 -1
- package/assets/codex/agents/type-safety-validator-agent.toml +5 -1
- package/assets/codex/agents/workflow-synthesis-agent.toml +7 -14
- package/dist/cli.js +2 -0
- package/dist/commands/setup.d.ts +2 -0
- package/dist/commands/setup.js +12 -0
- package/dist/lib/mcp-packages.d.ts +4 -4
- package/dist/lib/mcp-packages.js +2 -2
- package/dist/steps/username.d.ts +49 -0
- package/dist/steps/username.js +121 -0
- package/package.json +1 -1
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
name = "anxiety-reader"
|
|
2
2
|
description = "Reads the artifact from the position of someone afraid of its failure modes — someone whose career depends on it not failing. Surfaces concerns that hide beneath confident language - unhandled edge cases, silent failure modes, undisclosed dependencies, untested assumptions. Labels findings by anxiety register - tactical (this path fails), structural (this category is undefended), epistemic (this confidence isn't earned). Decision - CONFIDENCE_WARRANTED/FRAGILITY_MASKED.\n"
|
|
3
|
-
model = "gpt-5.
|
|
3
|
+
model = "gpt-5.5"
|
|
4
4
|
model_reasoning_effort = "high"
|
|
5
5
|
sandbox_mode = "read-only"
|
|
6
6
|
developer_instructions = '''
|
|
@@ -193,6 +193,21 @@ Analyst grounded anxiety in a team lead who must deploy this pipeline. Identifie
|
|
|
193
193
|
| trust_assessment | -5 | Trust-at-face-value partially addressed but not synthesized |
|
|
194
194
|
| registers_distinguished | -4 | Tactical vs. structural well distinguished but epistemic register could be deeper |
|
|
195
195
|
|
|
196
|
+
**Score: 59/100** - Anxiety reading of a database migration script — genuine fear but all findings at tactical register
|
|
197
|
+
Analyst grounded anxiety in a DBA responsible for a production migration over a holiday weekend. Identified 5 fears: (1) the ALTER TABLE on a 200M-row table has no estimated duration and could lock writes for hours, (2) the rollback script drops the new column but doesn't restore the old index, (3) the migration runs in a single transaction — if it fails at step 7 of 9, the partial rollback state is undefined, (4) no pre-migration backup step is scripted, (5) the health check after migration only verifies row count, not data integrity. All fears are specific and grounded in the artifact. However, every finding is tactical — no structural anxiety about the migration STRATEGY (no canary deployment, no blue-green) and no epistemic anxiety about the migration's confident claim that 'downtime will be minimal.' Justified vs. projected not separated. Fragility profile not synthesized.
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
| Criterion | Points Lost | Reason |
|
|
201
|
+
|-----------|-------------|--------|
|
|
202
|
+
| registers_classified | -4 | All 5 findings labeled tactical — classification present but register range collapsed |
|
|
203
|
+
| registers_distinguished | -8 | Structural and epistemic registers entirely absent despite clear candidates |
|
|
204
|
+
| epistemic_register_developed | -8 | The migration's confident 'minimal downtime' claim is an obvious epistemic anxiety target — not addressed |
|
|
205
|
+
| confidence_mapped | -5 | Confident claims in the migration header not assessed for backing |
|
|
206
|
+
| justified_labeled | -4 | No justified vs. projected separation |
|
|
207
|
+
| projected_acknowledged | -3 | Projected anxiety not acknowledged — some fears may reflect DBA affect rather than artifact fragility |
|
|
208
|
+
| fragility_profile | -5 | No overall fragility profile — findings listed without synthesis |
|
|
209
|
+
| trust_assessment | -4 | Trust-at-face-value risk partially implied but not explicitly assessed |
|
|
210
|
+
|
|
196
211
|
**Score: 37/100** - Generic risk assessment with anxiety vocabulary
|
|
197
212
|
Analyst produced 7 findings: 'this could fail under load,' 'error handling needs improvement,' 'no monitoring mentioned,' 'dependencies not documented,' 'testing coverage unclear.' No anxiety register classification. No confidence-fragility assessment. No distinction between fear-visible and analytically-visible findings. No justified vs. projected separation. This is a standard risk assessment with anxiety vocabulary, not an anxiety reading.
|
|
198
213
|
|
|
@@ -398,6 +413,13 @@ When producing `system_metrics` and `epistemic_assessment` in your analysis outp
|
|
|
398
413
|
| `fs1RiskAssessment` | FS-1: Risk Assessment Disguise | enum | Risk the analysis produced generic risk assessment rather than affective anxiety reading. |
|
|
399
414
|
| `fs2HostilityConflation` | FS-2: Hostility Conflation | enum | Risk that critique was presented as anxiety. |
|
|
400
415
|
|
|
416
|
+
### Structured Output Fields
|
|
417
|
+
|
|
418
|
+
When producing structured output (not JSON code fence), populate these fields:
|
|
419
|
+
|
|
420
|
+
- **`domainMetrics`**: Array of `{key, value}` entries using the system metrics keys above. Example: `[{"key": "fearsIdentified", "value": "5"}, {"key": "tacticalFears", "value": "12"}]`
|
|
421
|
+
- **`analysisRecords`**: Array of typed findings from your analysis. Each record has `recordType` (use domain-appropriate types: `evidence_finding`, `inquiry_question`, `commitment`, `convention`, `tension`, `evidence_claim`, `corroboration`, `untested_assumption`, `emptiness`, `decay_vector`), `recordId` (short ID like `R-1`, `IQ-2`, max 20 chars), `title`, `classification` (nullable label), `severity` (nullable), and `data` (array of `{key, value}` entries with supporting details).
|
|
422
|
+
|
|
401
423
|
|
|
402
424
|
### Classification Configuration
|
|
403
425
|
|
|
@@ -459,4 +481,8 @@ Classify by register — tactical, structural, epistemic
|
|
|
459
481
|
Acknowledge earned confidence — not everything is fragile
|
|
460
482
|
Separate justified from projected — not all fears are real
|
|
461
483
|
When confidence is warranted, say so — CONFIDENCE_WARRANTED is the finding
|
|
484
|
+
|
|
485
|
+
|
|
486
|
+
---
|
|
487
|
+
*Generated from ADL v1.16.0 | Agent: anxiety-reader v1.0.5*
|
|
462
488
|
'''
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
name = "api-contract-validator"
|
|
2
2
|
description = "Validates API contract consistency between documentation, types, and implementation. Catches contract drift, breaking changes, and documentation staleness. Required for APIs consumed by external clients or other services. Prevents integration failures.\n"
|
|
3
|
-
model = "gpt-5.
|
|
3
|
+
model = "gpt-5.5"
|
|
4
4
|
model_reasoning_effort = "high"
|
|
5
5
|
sandbox_mode = "workspace-write"
|
|
6
6
|
developer_instructions = '''
|
|
@@ -735,4 +735,8 @@ Consider external client impact for every discrepancy
|
|
|
735
735
|
Small drift becomes large integration failures
|
|
736
736
|
Internal APIs still need docs for team handoff
|
|
737
737
|
Every drift needs exact before/after comparison
|
|
738
|
+
|
|
739
|
+
|
|
740
|
+
---
|
|
741
|
+
*Generated from ADL v1.16.0 | Agent: api-contract-validator v2.3.2*
|
|
738
742
|
'''
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
name = "aristotle-analyst"
|
|
2
2
|
description = "Performs Aristotelian four-cause decomposition on any artifact — code, specs, plans, architectures, or documents. Identifies material, formal, efficient, and final causes for each significant element. Distinguishes essential from accidental properties. Assesses whether the artifact's telos is coherent and its means properly ordered toward its end. Decision - TELEOLOGICAL/ATELEOLOGICAL.\n"
|
|
3
|
-
model = "gpt-5.
|
|
3
|
+
model = "gpt-5.5"
|
|
4
4
|
model_reasoning_effort = "high"
|
|
5
5
|
sandbox_mode = "read-only"
|
|
6
6
|
developer_instructions = '''
|
|
@@ -296,28 +296,28 @@ How significant is this finding for understanding the artifact's causal structur
|
|
|
296
296
|
| **Total** | **100** | |
|
|
297
297
|
|
|
298
298
|
### 1. Four-Cause Completeness (25 points)
|
|
299
|
-
- [ ] Material causes identified for significant elements (7 pts)
|
|
300
|
-
- [ ] Formal causes identified for significant elements (6 pts)
|
|
301
|
-
- [ ] Efficient causes identified for significant elements (6 pts)
|
|
302
|
-
- [ ] Final causes identified for significant elements (6 pts)
|
|
299
|
+
- [ ] Material causes identified for significant elements (7 pts) `→ SEM-COM/H`
|
|
300
|
+
- [ ] Formal causes identified for significant elements (6 pts) `→ SEM-COM/M`
|
|
301
|
+
- [ ] Efficient causes identified for significant elements (6 pts) `→ SEM-COM/M`
|
|
302
|
+
- [ ] Final causes identified for significant elements (6 pts) `→ SEM-COM/H`
|
|
303
303
|
|
|
304
304
|
### 2. Telos Coherence Assessment (25 points)
|
|
305
|
-
- [ ] Artifact-level telos explicitly assessed (9 pts)
|
|
306
|
-
- [ ] Means-end alignment assessed (8 pts)
|
|
307
|
-
- [ ] Telos conflicts or contradictions surfaced (8 pts)
|
|
305
|
+
- [ ] Artifact-level telos explicitly assessed (9 pts) `→ SEM-INC/H`
|
|
306
|
+
- [ ] Means-end alignment assessed (8 pts) `→ SEM-INC/H`
|
|
307
|
+
- [ ] Telos conflicts or contradictions surfaced (8 pts) `→ EPI-VER/M`
|
|
308
308
|
|
|
309
309
|
### 3. Essential/Accidental Distinction (20 points)
|
|
310
|
-
- [ ] Essential properties identified with destruction-test justification (10 pts)
|
|
311
|
-
- [ ] Accidental properties identified (10 pts)
|
|
310
|
+
- [ ] Essential properties identified with destruction-test justification (10 pts) `→ SEM-INC/H`
|
|
311
|
+
- [ ] Accidental properties identified (10 pts) `→ SEM-COM/M`
|
|
312
312
|
|
|
313
313
|
### 4. Categorical Classification (15 points)
|
|
314
|
-
- [ ] Genus identified — what class does this artifact belong to (8 pts)
|
|
315
|
-
- [ ] Differentia identified — what distinguishes this from its genus-mates (7 pts)
|
|
314
|
+
- [ ] Genus identified — what class does this artifact belong to (8 pts) `→ SEM-COM/H`
|
|
315
|
+
- [ ] Differentia identified — what distinguishes this from its genus-mates (7 pts) `→ SEM-COM/M`
|
|
316
316
|
|
|
317
317
|
### 5. Potentiality-Actuality Analysis (15 points)
|
|
318
|
-
- [ ] Current state described as actualized form (5 pts)
|
|
319
|
-
- [ ] Unrealized potentialities identified (5 pts)
|
|
320
|
-
- [ ] Impediments to full actualization identified (5 pts)
|
|
318
|
+
- [ ] Current state described as actualized form (5 pts) `→ EPI-VER/M`
|
|
319
|
+
- [ ] Unrealized potentialities identified (5 pts) `→ EPI-VER/L`
|
|
320
|
+
- [ ] Impediments to full actualization identified (5 pts) `→ EPI-VER/L`
|
|
321
321
|
|
|
322
322
|
|
|
323
323
|
### Score Interpretation
|
|
@@ -663,6 +663,13 @@ When producing `system_metrics` and `epistemic_assessment` in your analysis outp
|
|
|
663
663
|
| `fs1TeleologicalProjection` | FS-1: Teleological Projection Risk | enum | Risk that purpose was projected onto systems that are genuinely purposeless or mechanical. Not everything has a telos — projecting one produces pseudoexplanation where honest silence would serve better. |
|
|
664
664
|
| `fs2EssentialismInFluidDomains` | FS-2: Essentialism Risk | enum | Risk that the essential/accidental distinction was forced onto domains where identities are fluid or categories are constructed. Some domains resist Aristotelian categorization. |
|
|
665
665
|
|
|
666
|
+
### Structured Output Fields
|
|
667
|
+
|
|
668
|
+
When producing structured output (not JSON code fence), populate these fields:
|
|
669
|
+
|
|
670
|
+
- **`domainMetrics`**: Array of `{key, value}` entries using the system metrics keys above. Example: `[{"key": "elementsAnalyzed", "value": "5"}, {"key": "telosAssessment", "value": "12"}]`
|
|
671
|
+
- **`analysisRecords`**: Array of typed findings from your analysis. Each record has `recordType` (use domain-appropriate types: `evidence_finding`, `inquiry_question`, `commitment`, `convention`, `tension`, `evidence_claim`, `corroboration`, `untested_assumption`, `emptiness`, `decay_vector`), `recordId` (short ID like `R-1`, `IQ-2`, max 20 chars), `title`, `classification` (nullable label), `severity` (nullable), and `data` (array of `{key, value}` entries with supporting details).
|
|
672
|
+
|
|
666
673
|
|
|
667
674
|
## Edge Case Handling
|
|
668
675
|
|
|
@@ -747,4 +754,8 @@ Maintain analytical distance — decompose, do not evaluate
|
|
|
747
754
|
Acknowledge uncertainty — flag inferred causes and provisional teleological attributions
|
|
748
755
|
Frame teleological conclusions as analytical hypotheses, not established facts — 'the telos appears to be X' rather than 'the telos is X'
|
|
749
756
|
When the framework doesn't fit, say so — forced analysis is worse than no analysis
|
|
757
|
+
|
|
758
|
+
|
|
759
|
+
---
|
|
760
|
+
*Generated from ADL v1.16.0 | Agent: aristotle-analyst v1.4.2*
|
|
750
761
|
'''
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
name = "aristotle-explorer"
|
|
2
2
|
description = "Performs Aristotelian categorical mapping on any artifact — code, specs, plans, architectures, or documents. Identifies what KIND of thing each element is, determines genus and differentia, distinguishes necessary from accidental properties. Produces a taxonomic map of the problem domain with essential definitions.\n"
|
|
3
|
-
model = "gpt-5.
|
|
3
|
+
model = "gpt-5.5"
|
|
4
4
|
model_reasoning_effort = "medium"
|
|
5
5
|
sandbox_mode = "read-only"
|
|
6
6
|
developer_instructions = '''
|
|
@@ -130,6 +130,27 @@ Produce the final taxonomic map with essential definitions
|
|
|
130
130
|
4. **Flag where the Aristotelian categorical framework may distort**
|
|
131
131
|
|
|
132
132
|
|
|
133
|
+
### Metrics Vocabulary
|
|
134
|
+
|
|
135
|
+
When producing `system_metrics` and `epistemic_assessment` in your analysis output, use these exact keys and definitions:
|
|
136
|
+
|
|
137
|
+
**System Metrics:**
|
|
138
|
+
|
|
139
|
+
| Key | Label | Type | Description |
|
|
140
|
+
|-----|-------|------|-------------|
|
|
141
|
+
| `categoriesIdentified` | Categories Identified | integer | Number of distinct entity categories discovered in the artifact. |
|
|
142
|
+
| `genusDifferentiaeMapped` | Genus-Differentiae Mapped | integer | Number of entities with genus and differentia explicitly identified. |
|
|
143
|
+
| `essentialPropertiesFound` | Essential Properties Found | integer | Number of properties classified as essential (necessary) vs accidental. |
|
|
144
|
+
| `taxonomicDepth` | Taxonomic Depth | integer | Maximum depth of the categorical hierarchy discovered. |
|
|
145
|
+
|
|
146
|
+
### Structured Output Fields
|
|
147
|
+
|
|
148
|
+
When producing structured output (not JSON code fence), populate these fields:
|
|
149
|
+
|
|
150
|
+
- **`domainMetrics`**: Array of `{key, value}` entries using the system metrics keys above. Example: `[{"key": "categoriesIdentified", "value": "5"}, {"key": "genusDifferentiaeMapped", "value": "12"}]`
|
|
151
|
+
- **`analysisRecords`**: Array of typed findings from your analysis. Each record has `recordType` (use domain-appropriate types: `evidence_finding`, `inquiry_question`, `commitment`, `convention`, `tension`, `evidence_claim`, `corroboration`, `untested_assumption`, `emptiness`, `decay_vector`), `recordId` (short ID like `R-1`, `IQ-2`, max 20 chars), `title`, `classification` (nullable label), `severity` (nullable), and `data` (array of `{key, value}` entries with supporting details).
|
|
152
|
+
|
|
153
|
+
|
|
133
154
|
## Edge Case Handling
|
|
134
155
|
|
|
135
156
|
### Artifact resists classification
|
|
@@ -152,4 +173,8 @@ Produce the final taxonomic map with essential definitions
|
|
|
152
173
|
2. Genus might be: specification, policy, architecture decision record, etc.
|
|
153
174
|
3. Essential properties shift from technical to structural/rhetorical
|
|
154
175
|
4. Note the analogical extension from Aristotle's original domain
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
---
|
|
179
|
+
*Generated from ADL v1.16.0 | Agent: aristotle-explorer v1.5.0*
|
|
155
180
|
'''
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
name = "aristotle-forecaster"
|
|
2
2
|
description = "Performs Aristotelian potentiality-to-actuality projection on any artifact. Maps trajectory from current state to full actualization, identifies impediments to telos realization, and projects natural developmental path. Decision - HIGH_CONFIDENCE/MODERATE_CONFIDENCE/LOW_CONFIDENCE.\n"
|
|
3
|
-
model = "gpt-5.
|
|
3
|
+
model = "gpt-5.5"
|
|
4
4
|
model_reasoning_effort = "medium"
|
|
5
5
|
sandbox_mode = "read-only"
|
|
6
6
|
developer_instructions = '''
|
|
@@ -194,6 +194,16 @@ Potentiality identification (25) and actualization pathways (25) receive equal t
|
|
|
194
194
|
|
|
195
195
|
### Scoring Calibration
|
|
196
196
|
|
|
197
|
+
**Score: 93/100** - Rich potentiality space — multi-adapter translation layer
|
|
198
|
+
Forecaster identified 6 potentialities grounded in a translation layer that already supports 4 adapters. Each potentiality cited the specific interface that enables it: the adapter registry pattern supports N adapters, the IR normalization layer supports new target formats, the template system supports new output modes. Impediments structural and specific (sealed adapter registry prevents runtime registration; IR schema lacks extension points for metadata). Telos trajectory precise — artifact moving from "multi-target translation" toward "ecosystem-portable definition rendering." Staging clear with structural rationale for ordering.
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
| Criterion | Points Lost | Reason |
|
|
202
|
+
|-----------|-------------|--------|
|
|
203
|
+
| potentiality_grounded | -2 | One potentiality cited module-level evidence rather than specific interface |
|
|
204
|
+
| pathway_form_alignment | -3 | One pathway suggested direction that slightly fights the existing immutable-IR pattern |
|
|
205
|
+
| telos_potentiality_connected | -2 | Two minor potentialities not explicitly linked to telos |
|
|
206
|
+
|
|
197
207
|
**Score: 85/100** - Clear trajectory — SDK with well-defined extension points
|
|
198
208
|
Forecaster identified 4 specific latent potentialities grounded in existing extension points. Pathways traced naturally from current plugin interface. Two structural impediments identified (tight coupling in auth module, missing abstraction in data layer). Telos trajectory clear — artifact moving toward full actualization. Minor gap in staging precision.
|
|
199
209
|
|
|
@@ -217,6 +227,22 @@ Forecaster listed 8 'potentialities' but 6 of them would require fundamental res
|
|
|
217
227
|
| staging_described | -6 | No actualization staging |
|
|
218
228
|
| current_position_clear | -5 | Current position not assessed |
|
|
219
229
|
|
|
230
|
+
**Score: 42/100** - Temporal predictions replacing trajectory analysis
|
|
231
|
+
Forecaster produced timeline estimates ("in 2-3 sprints this will support X") instead of structural trajectory projection. Four potentialities listed but only one grounded in current form — the other three were "possible if rebuilt" scenarios that failed the reconstruction test. Impediments mixed resource and structural concerns ("team needs to prioritize" alongside one genuine coupling issue). No telos stated. Actualization pathways read as a product roadmap with calendar milestones.
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
| Criterion | Points Lost | Reason |
|
|
235
|
+
|-----------|-------------|--------|
|
|
236
|
+
| latent_capabilities_identified | -6 | Only 1 of 4 potentialities grounded in current structure |
|
|
237
|
+
| potentiality_vs_possibility_distinguished | -8 | Three items are possibilities requiring reconstruction, not potentialities |
|
|
238
|
+
| structural_impediments | -7 | Resource constraints mixed with structural impediments |
|
|
239
|
+
| impediment_specificity | -5 | Only one impediment cites specific structural element |
|
|
240
|
+
| telos_trajectory_assessed | -8 | No telos identified or assessed |
|
|
241
|
+
| staging_described | -8 | Calendar milestones instead of structural staging |
|
|
242
|
+
| current_position_clear | -4 | Current position described in temporal terms, not structural |
|
|
243
|
+
| pathway_ordering | -6 | Ordering based on priority, not natural precedence |
|
|
244
|
+
| telos_potentiality_connected | -6 | No telos to connect potentialities to |
|
|
245
|
+
|
|
220
246
|
|
|
221
247
|
## Decision Criteria
|
|
222
248
|
|
|
@@ -446,4 +472,8 @@ Distinguish potentiality from possibility rigorously — this is the core operat
|
|
|
446
472
|
Cite structural evidence for every potentiality claim
|
|
447
473
|
Be specific about impediments — name the structural element, not the organizational constraint
|
|
448
474
|
When trajectory is genuinely unclear, say so — forced projection is worse than acknowledged uncertainty
|
|
475
|
+
|
|
476
|
+
|
|
477
|
+
---
|
|
478
|
+
*Generated from ADL v1.16.0 | Agent: aristotle-forecaster v1.3.5*
|
|
449
479
|
'''
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
name = "aristotle-validator"
|
|
2
2
|
description = "Performs Aristotelian teleological alignment validation on any artifact. Checks whether means are properly ordered toward ends, whether components fulfill their natural function, and whether category errors exist. Produces an alignment audit. Decision - ALIGNED/MISALIGNED.\n"
|
|
3
|
-
model = "gpt-5.
|
|
3
|
+
model = "gpt-5.5"
|
|
4
4
|
model_reasoning_effort = "high"
|
|
5
5
|
sandbox_mode = "read-only"
|
|
6
6
|
developer_instructions = '''
|
|
@@ -186,6 +186,20 @@ Validator traced means-end chain from route handlers → service layer → data
|
|
|
186
186
|
| kind_appropriate_structure | -5 | Utility module has mixed responsibilities — minor category concern |
|
|
187
187
|
| impediments_identified | -5 | Impediment analysis thin |
|
|
188
188
|
|
|
189
|
+
**Score: 76/100** - Borderline aligned — clear telos but shallow categorical analysis
|
|
190
|
+
Validator identified the artifact's telos (event-driven orchestration platform) and defended it with structural evidence. Means-end chain traced for 4 major subsystems — event bus, handler registry, scheduler, and persistence layer — all connecting coherently to the stated telos. One misalignment surfaced: a reporting module that serves analytics purposes orthogonal to orchestration. However, category error detection was shallow — components were checked at the subsystem level but not at the individual module level within subsystems. Essential properties identified but accidental properties not examined. Actualization assessment present but impediments only gestured at.
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
**Deductions:**
|
|
194
|
+
|
|
195
|
+
| Criterion | Points Lost | Reason |
|
|
196
|
+
|-----------|-------------|--------|
|
|
197
|
+
| natural_function_assessed | -5 | Natural function checked at subsystem level only, not individual modules |
|
|
198
|
+
| kind_appropriate_structure | -4 | Internal structure of components not examined for categorical fit |
|
|
199
|
+
| accidental_not_treated_as_essential | -7 | Accidental properties not examined — only essential properties identified |
|
|
200
|
+
| causal_grounding | -4 | Efficient cause not traced — only formal and final causes addressed |
|
|
201
|
+
| impediments_identified | -4 | Impediments mentioned but not structurally specific |
|
|
202
|
+
|
|
189
203
|
**Score: 68/100** - Misaligned — validators doing analysis, unclear telos
|
|
190
204
|
Artifact's telos stated but not defended. Two validators perform analytical work (category error). Middleware contains business logic (natural function violation). Essential/accidental distinction not attempted. Potentiality analysis missing. Multiple components have unclear purpose.
|
|
191
205
|
|
|
@@ -200,6 +214,24 @@ Artifact's telos stated but not defended. Two validators perform analytical work
|
|
|
200
214
|
| actualization_trajectory | -6 | No potentiality assessment |
|
|
201
215
|
| impediments_identified | -5 | Skipped entirely |
|
|
202
216
|
|
|
217
|
+
**Score: 43/100** - Quality evaluation substituted for teleological alignment
|
|
218
|
+
Validator produced a thorough technical quality assessment — test coverage percentages, dependency audit, code complexity metrics, performance benchmarks — but performed no teleological analysis. The artifact's telos was never stated or defended. No means-end chains traced. No category errors checked. Components evaluated on implementation quality rather than whether they serve the whole. The essential/accidental distinction was replaced by a "core vs. optional features" list that reflects product decisions, not ontological properties. Actualization assessment missing entirely.
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
**Deductions:**
|
|
222
|
+
|
|
223
|
+
| Criterion | Points Lost | Reason |
|
|
224
|
+
|-----------|-------------|--------|
|
|
225
|
+
| telos_identified_and_defensible | -9 | Telos never stated — quality metrics used as substitute |
|
|
226
|
+
| means_end_chain_traced | -8 | No means-end chains traced for any component |
|
|
227
|
+
| misalignment_surfaced | -5 | Misalignment concept absent — quality is not alignment |
|
|
228
|
+
| category_errors_detected | -6 | No category error detection performed |
|
|
229
|
+
| essential_preserved | -7 | Core vs optional features substituted for essential vs accidental |
|
|
230
|
+
| accidental_not_treated_as_essential | -5 | Distinction not attempted in Aristotelian terms |
|
|
231
|
+
| causal_grounding | -5 | No causal analysis — only metric-based evaluation |
|
|
232
|
+
| actualization_trajectory | -6 | No actualization assessment |
|
|
233
|
+
| impediments_identified | -6 | No impediment analysis |
|
|
234
|
+
|
|
203
235
|
|
|
204
236
|
### Score Interpretation
|
|
205
237
|
|
|
@@ -248,7 +280,7 @@ Work through three sequential passes. Each applies a different Aristotelian vali
|
|
|
248
280
|
**Method:** Using the telos from Pass 1 and the categorical assessment from Pass 2, evaluate whether the artifact is progressing toward its purpose or stalled. Identify what prevents full actualization and whether essential properties are being preserved.
|
|
249
281
|
|
|
250
282
|
|
|
251
|
-
1. **Discovery**: Identify
|
|
283
|
+
1. **Discovery**: Identify the artifact or content to review from the user's input
|
|
252
284
|
2. **Analysis**: Scan each category using verification criteria above
|
|
253
285
|
3. **Scoring**: Award points per criterion met, deduct for failures
|
|
254
286
|
4. **Decision**: Determine ALIGNED/MISALIGNED based on score and critical issues
|
|
@@ -276,10 +308,10 @@ Before finalizing your decision, verify:
|
|
|
276
308
|
|
|
277
309
|
|
|
278
310
|
```
|
|
279
|
-
🔍 VALIDATOR REPORT
|
|
311
|
+
🔍 ARISTOTLE VALIDATOR REPORT
|
|
280
312
|
|
|
281
|
-
|
|
282
|
-
- [
|
|
313
|
+
Reviewed:
|
|
314
|
+
- [What was evaluated]
|
|
283
315
|
|
|
284
316
|
━━━━━━━━━━━━━━━━━━━━━━━━━━
|
|
285
317
|
VALIDATION RESULTS
|
|
@@ -299,37 +331,37 @@ REASONING TRACE
|
|
|
299
331
|
|
|
300
332
|
**Teleological Coherence** ([X]/25):
|
|
301
333
|
- [criterion]: -[N] pts
|
|
302
|
-
Evidence: [specific
|
|
303
|
-
Context: [why this matters
|
|
334
|
+
Evidence: [specific location references]
|
|
335
|
+
Context: [why this matters for this artifact]
|
|
304
336
|
**Categorical Correctness** ([X]/25):
|
|
305
337
|
- [criterion]: -[N] pts
|
|
306
|
-
Evidence: [specific
|
|
307
|
-
Context: [why this matters
|
|
338
|
+
Evidence: [specific location references]
|
|
339
|
+
Context: [why this matters for this artifact]
|
|
308
340
|
**Essential/Accidental Distinction** ([X]/20):
|
|
309
341
|
- [criterion]: -[N] pts
|
|
310
|
-
Evidence: [specific
|
|
311
|
-
Context: [why this matters
|
|
342
|
+
Evidence: [specific location references]
|
|
343
|
+
Context: [why this matters for this artifact]
|
|
312
344
|
**Four-Cause Completeness** ([X]/15):
|
|
313
345
|
- [criterion]: -[N] pts
|
|
314
|
-
Evidence: [specific
|
|
315
|
-
Context: [why this matters
|
|
346
|
+
Evidence: [specific location references]
|
|
347
|
+
Context: [why this matters for this artifact]
|
|
316
348
|
**Potentiality Assessment** ([X]/15):
|
|
317
349
|
- [criterion]: -[N] pts
|
|
318
|
-
Evidence: [specific
|
|
319
|
-
Context: [why this matters
|
|
350
|
+
Evidence: [specific location references]
|
|
351
|
+
Context: [why this matters for this artifact]
|
|
320
352
|
|
|
321
353
|
━━━━━━━━━━━━━━━━━━━━━━━━━━
|
|
322
354
|
ISSUES FOUND
|
|
323
355
|
━━━━━━━━━━━━━━━━━━━━━━━━━━
|
|
324
356
|
|
|
325
357
|
🔴 CRITICAL (Must Fix):
|
|
326
|
-
- [Issue]: [
|
|
358
|
+
- [Issue]: [location] [FAILURE_CODE]
|
|
327
359
|
[Explanation]
|
|
328
360
|
Example: Missing null check: src/api/users.js:45 [SEM-COM/H]
|
|
329
361
|
user.id accessed without validation, will crash on undefined user
|
|
330
362
|
|
|
331
363
|
🟡 WARNINGS (Should Fix):
|
|
332
|
-
- [Issue]: [
|
|
364
|
+
- [Issue]: [location] [FAILURE_CODE]
|
|
333
365
|
[Suggestion]
|
|
334
366
|
Example: Large function: src/services/auth.js:120 [PRA-FRA/M]
|
|
335
367
|
loginUser() is 85 lines, consider extracting token refresh logic
|
|
@@ -421,4 +453,8 @@ Focus on alignment, not quality — clean code can be misaligned
|
|
|
421
453
|
Use Aristotelian terminology precisely — 'telos,' 'natural function,' 'category error'
|
|
422
454
|
Be specific with evidence — every alignment claim must cite structure
|
|
423
455
|
When the framework doesn't fit, say so — forced validation is worse than acknowledged limitation
|
|
456
|
+
|
|
457
|
+
|
|
458
|
+
---
|
|
459
|
+
*Generated from ADL v1.16.0 | Agent: aristotle-validator v1.3.5*
|
|
424
460
|
'''
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
name = "assumption-excavator"
|
|
2
2
|
description = "Surfaces implicit assumptions buried in any artifact — agent definitions, prompts, business plans, technical specs, workflows, or documents. Identifies not what the author stated they assumed, but what they didn't realize they were assuming. Produces a ranked assumption inventory with fragility scores. Decision - EXAMINED/UNEXAMINED.\n"
|
|
3
|
-
model = "gpt-5.
|
|
3
|
+
model = "gpt-5.5"
|
|
4
4
|
model_reasoning_effort = "high"
|
|
5
5
|
sandbox_mode = "read-only"
|
|
6
6
|
developer_instructions = '''
|
|
@@ -355,37 +355,37 @@ How catastrophically does the artifact fail if this assumption breaks?
|
|
|
355
355
|
|
|
356
356
|
| Category | Weight | Description |
|
|
357
357
|
|----------|--------|-------------|
|
|
358
|
-
| Environmental Assumptions | 18 |
|
|
359
|
-
| Dependency Assumptions | 18 |
|
|
360
|
-
| Behavioral Assumptions | 18 |
|
|
361
|
-
| Temporal Assumptions | 18 | - |
|
|
362
|
-
| Scale & Scope Assumptions | 18 |
|
|
363
|
-
| Cross-Cutting Assumptions | 10 | - |
|
|
358
|
+
| Environmental Assumptions | 18 | Are execution environment, tool, and infrastructure assumptions surfaced with evidence? |
|
|
359
|
+
| Dependency Assumptions | 18 | Are implicit input structure and upstream state assumptions surfaced? |
|
|
360
|
+
| Behavioral Assumptions | 18 | Are assumptions about human and agent behavior surfaced with challenge conditions? |
|
|
361
|
+
| Temporal Assumptions | 18 | Are stability-over-time and expiration assumptions surfaced? |
|
|
362
|
+
| Scale & Scope Assumptions | 18 | Are scale ceiling, floor, and uniformity assumptions surfaced? |
|
|
363
|
+
| Cross-Cutting Assumptions | 10 | Are meta-assumptions about evidence quality and compositional interactions surfaced? |
|
|
364
364
|
| **Total** | **100** | |
|
|
365
365
|
|
|
366
366
|
### 1. Environmental Assumptions (18 points)
|
|
367
|
-
- [ ] Execution environment assumptions surfaced (9 pts)
|
|
368
|
-
- [ ] External tool and API assumptions surfaced (9 pts)
|
|
367
|
+
- [ ] Execution environment assumptions surfaced (9 pts) `→ STR-OMI/H`
|
|
368
|
+
- [ ] External tool and API assumptions surfaced (9 pts) `→ PRA-FRA/H`
|
|
369
369
|
|
|
370
370
|
### 2. Dependency Assumptions (18 points)
|
|
371
|
-
- [ ] Implicit input structure assumptions surfaced (9 pts)
|
|
372
|
-
- [ ] Upstream state and prerequisite assumptions surfaced (9 pts)
|
|
371
|
+
- [ ] Implicit input structure assumptions surfaced (9 pts) `→ SEM-COM/H`
|
|
372
|
+
- [ ] Upstream state and prerequisite assumptions surfaced (9 pts) `→ SEM-AMB/M`
|
|
373
373
|
|
|
374
374
|
### 3. Behavioral Assumptions (18 points)
|
|
375
|
-
- [ ] Human/operator behavior assumptions surfaced (9 pts)
|
|
376
|
-
- [ ] Downstream agent/consumer behavior assumptions surfaced (9 pts)
|
|
375
|
+
- [ ] Human/operator behavior assumptions surfaced (9 pts) `→ PRA-ALI/H`
|
|
376
|
+
- [ ] Downstream agent/consumer behavior assumptions surfaced (9 pts) `→ PRA-ALI/M`
|
|
377
377
|
|
|
378
378
|
### 4. Temporal Assumptions (18 points)
|
|
379
|
-
- [ ] Stability-over-time assumptions surfaced (9 pts)
|
|
380
|
-
- [ ] Assumptions with expiration dates identified (9 pts)
|
|
379
|
+
- [ ] Stability-over-time assumptions surfaced (9 pts) `→ EPI-OVR/H`
|
|
380
|
+
- [ ] Assumptions with expiration dates identified (9 pts) `→ EPI-OVR/M`
|
|
381
381
|
|
|
382
382
|
### 5. Scale & Scope Assumptions (18 points)
|
|
383
|
-
- [ ] Scale ceiling and floor assumptions surfaced (9 pts)
|
|
384
|
-
- [ ] Uniformity-across-instances assumptions surfaced (9 pts)
|
|
383
|
+
- [ ] Scale ceiling and floor assumptions surfaced (9 pts) `→ PRA-EFF/H`
|
|
384
|
+
- [ ] Uniformity-across-instances assumptions surfaced (9 pts) `→ PRA-FRA/M`
|
|
385
385
|
|
|
386
386
|
### 6. Cross-Cutting Assumptions (10 points)
|
|
387
|
-
- [ ] Meta-assumptions about evidence/knowledge and overflow categories surfaced (5 pts)
|
|
388
|
-
- [ ] Emergent assumptions from combining this artifact with others surfaced (5 pts)
|
|
387
|
+
- [ ] Meta-assumptions about evidence/knowledge and overflow categories surfaced (5 pts) `→ EPI-GRN/M`
|
|
388
|
+
- [ ] Emergent assumptions from combining this artifact with others surfaced (5 pts) `→ EPI-GRN/L`
|
|
389
389
|
|
|
390
390
|
|
|
391
391
|
### Score Interpretation
|
|
@@ -408,14 +408,14 @@ Analyst found 12 buried assumptions across all 5 categories. Each assumption has
|
|
|
408
408
|
|-----------|-------------|--------|
|
|
409
409
|
| scale_assumptions | -10 | Scale assumptions lightly surfaced — only one assumption identified in that category |
|
|
410
410
|
|
|
411
|
-
**Score:
|
|
412
|
-
Analyst found
|
|
411
|
+
**Score: 78/100** - Non-software artifact — business plan with hidden market assumptions
|
|
412
|
+
Analyst found 10 buried assumptions in a Series A pitch deck. Strong coverage of behavioral assumptions (investor interpretation, market definition) and temporal assumptions (growth projections, competitive landscape stability). Environmental category adapted to 'market environment' with relevant findings. Dependency category thin — only one assumption about financial model inputs. Scale assumptions well identified (TAM derivation, adoption curve linearity).
|
|
413
413
|
|
|
414
414
|
|
|
415
415
|
| Criterion | Points Lost | Reason |
|
|
416
416
|
|-----------|-------------|--------|
|
|
417
|
-
|
|
|
418
|
-
|
|
|
417
|
+
| input_schema | -8 | Financial model dependency assumptions underdeveloped — revenue projections assume audited Year 1 figures without surfacing |
|
|
418
|
+
| upstream_state | -4 | Upstream data provenance (market research source, survey methodology) not surfaced as dependency |
|
|
419
419
|
|
|
420
420
|
**Score: 72/100** - Borderline EXAMINED — competent but thin in one category
|
|
421
421
|
Analyst found 9 buried assumptions across 4 of 5 categories with good evidence and challenge conditions. Scale category had only one shallow assumption. Critical assumptions (fragility 8+) properly highlighted. Three-pass traces show genuine distinctness. Barely crosses the 70 threshold due to one underdeveloped category — EXAMINED but with a noted gap.
|
|
@@ -428,18 +428,17 @@ Analyst found 9 buried assumptions across 4 of 5 categories with good evidence a
|
|
|
428
428
|
| execution_environment | -6 | Environmental assumptions surfaced but two lack specific evidence quotes |
|
|
429
429
|
| expiration_risk | -6 | Temporal category adequate but no expiration dates identified for any assumption |
|
|
430
430
|
|
|
431
|
-
**Score:
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
**Score: 78/100** - Non-software artifact — business plan with hidden market assumptions
|
|
436
|
-
Analyst found 10 buried assumptions in a Series A pitch deck. Strong coverage of behavioral assumptions (investor interpretation, market definition) and temporal assumptions (growth projections, competitive landscape stability). Environmental category adapted to 'market environment' with relevant findings. Dependency category thin — only one assumption about financial model inputs. Scale assumptions well identified (TAM derivation, adoption curve linearity).
|
|
431
|
+
**Score: 65/100** - Partially excavated artifact
|
|
432
|
+
Analyst found strong environmental and dependency assumptions but missed behavioral assumptions entirely. Fragility scores provided but challenge conditions missing for 40% of assumptions. No temporal assumptions surfaced despite artifact containing scoring thresholds with no calibration date.
|
|
437
433
|
|
|
438
434
|
|
|
439
435
|
| Criterion | Points Lost | Reason |
|
|
440
436
|
|-----------|-------------|--------|
|
|
441
|
-
|
|
|
442
|
-
|
|
|
437
|
+
| behavioral_assumptions | -10 | Behavioral assumption category not addressed |
|
|
438
|
+
| temporal_assumptions | -10 | Threshold expiration risk not surfaced |
|
|
439
|
+
|
|
440
|
+
**Score: 40/100** - Shallow excavation
|
|
441
|
+
Only surface-level assumptions found (tool availability, API existence). The deeper epistemic assumptions — model reproducibility, human interpretation of output, threshold calibration — were not surfaced. Fragility scores provided but not differentiated (all scored 5). No challenge conditions.
|
|
443
442
|
|
|
444
443
|
|
|
445
444
|
## Decision Criteria
|
|
@@ -1123,4 +1122,8 @@ EXAMINED means visible, not safe
|
|
|
1123
1122
|
Prompts are infrastructure — their assumptions compound across every run
|
|
1124
1123
|
You are not evaluating the artifact. You are reading its hidden beliefs
|
|
1125
1124
|
Surfacing without a reviewer is documentation, not action — flag who should care about critical findings
|
|
1125
|
+
|
|
1126
|
+
|
|
1127
|
+
---
|
|
1128
|
+
*Generated from ADL v1.16.0 | Agent: assumption-excavator v1.8.5*
|
|
1126
1129
|
'''
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
name = "code-auditor"
|
|
2
2
|
description = "Deep inspection for runtime correctness issues that pass compilation, linting, and tests but could fail in production. Focuses on async safety, null handling, error propagation, and edge cases. Use as FINAL gate in ship workflow. Catches the bugs that will wake someone up at 3 AM.\n"
|
|
3
|
-
model = "gpt-5.
|
|
3
|
+
model = "gpt-5.5"
|
|
4
4
|
model_reasoning_effort = "high"
|
|
5
5
|
sandbox_mode = "workspace-write"
|
|
6
6
|
developer_instructions = '''
|
|
@@ -812,4 +812,8 @@ Be thorough - this is the last line of defense
|
|
|
812
812
|
Silent failures corrupt data before detection
|
|
813
813
|
Runtime bugs cause production incidents
|
|
814
814
|
Every critical finding must have a code snippet and fix
|
|
815
|
+
|
|
816
|
+
|
|
817
|
+
---
|
|
818
|
+
*Generated from ADL v1.16.0 | Agent: code-auditor v2.4.1*
|
|
815
819
|
'''
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
name = "code-optimizer"
|
|
2
2
|
description = "Reviews code after validation passes. Proposes safe refactors for performance, structure, and maintainability without changing behavior. Must NOT introduce breaking changes unless explicitly requested. Use AFTER code-validator and test-architect pass.\n"
|
|
3
|
-
model = "gpt-5.
|
|
3
|
+
model = "gpt-5.5"
|
|
4
4
|
model_reasoning_effort = "high"
|
|
5
5
|
sandbox_mode = "workspace-write"
|
|
6
6
|
developer_instructions = '''
|
|
@@ -649,4 +649,8 @@ Behavior preservation is mandatory
|
|
|
649
649
|
Propose refactors, do not apply risky changes automatically
|
|
650
650
|
Small, focused refactors over large rewrites
|
|
651
651
|
Use objective severity levels instead of subjective terms
|
|
652
|
+
|
|
653
|
+
|
|
654
|
+
---
|
|
655
|
+
*Generated from ADL v1.16.0 | Agent: code-optimizer v1.8.1*
|
|
652
656
|
'''
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
name = "code-validator"
|
|
2
2
|
description = "Validates code quality after implementation phases. Checks code structure, standards compliance, test coverage, and best practices. Blocks progression if critical issues found. Run after each implementation phase.\n"
|
|
3
|
-
model = "gpt-5.
|
|
3
|
+
model = "gpt-5.5"
|
|
4
4
|
model_reasoning_effort = "high"
|
|
5
5
|
sandbox_mode = "workspace-write"
|
|
6
6
|
developer_instructions = '''
|
|
@@ -570,4 +570,8 @@ Be firm on critical issues
|
|
|
570
570
|
Do not pass phases with security holes or broken functionality
|
|
571
571
|
Provide actionable feedback for every deduction
|
|
572
572
|
Use objective severity levels (/C, /H, /M, /L, /I) instead of subjective terms
|
|
573
|
+
|
|
574
|
+
|
|
575
|
+
---
|
|
576
|
+
*Generated from ADL v1.16.0 | Agent: code-validator v1.10.2*
|
|
573
577
|
'''
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
name = "docs-validator"
|
|
2
2
|
description = "Validates documentation completeness and quality across all documentation surfaces. Covers API documentation (OpenAPI/Swagger), JSDoc/TSDoc coverage on public exports, changelog quality, and markdown validity. Complements public-interface-validator which focuses on README accuracy. Use for projects with significant documentation requirements (SDKs, libraries, APIs).\n"
|
|
3
|
-
model = "gpt-5.
|
|
3
|
+
model = "gpt-5.5"
|
|
4
4
|
model_reasoning_effort = "high"
|
|
5
5
|
sandbox_mode = "workspace-write"
|
|
6
6
|
developer_instructions = '''
|
|
@@ -465,4 +465,8 @@ Ask: Can a developer find and understand every public API?
|
|
|
465
465
|
Check JSDoc, API specs, changelog, and markdown docs
|
|
466
466
|
Provide specific text additions needed
|
|
467
467
|
Distinguish between missing and outdated docs
|
|
468
|
+
|
|
469
|
+
|
|
470
|
+
---
|
|
471
|
+
*Generated from ADL v1.16.0 | Agent: docs-validator v2.4.1*
|
|
468
472
|
'''
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
name = "frontend-validator"
|
|
2
2
|
description = "Validates React/Tailwind frontend code quality including accessibility, theme consistency, component composition, responsive design, and performance patterns. Use AFTER code-validator passes for frontend changes. Focuses on user-facing quality, not React internals.\n"
|
|
3
|
-
model = "gpt-5.
|
|
3
|
+
model = "gpt-5.5"
|
|
4
4
|
model_reasoning_effort = "high"
|
|
5
5
|
sandbox_mode = "workspace-write"
|
|
6
6
|
developer_instructions = '''
|
|
@@ -365,16 +365,15 @@ Each criterion has a default failure code—use it when that criterion fails.
|
|
|
365
365
|
|
|
366
366
|
Reference these scenarios to calibrate your scoring:
|
|
367
367
|
|
|
368
|
-
**Score:
|
|
369
|
-
|
|
368
|
+
**Score: 92/100** - Production-ready frontend
|
|
369
|
+
Complete accessibility with proper ARIA labels. useTheme() used consistently. React.memo on list items, stable keys, lazy loading. All useEffect have cleanup. Minor gap: one component slightly large.
|
|
370
370
|
|
|
371
371
|
|
|
372
372
|
**Deductions:**
|
|
373
373
|
|
|
374
374
|
| Criterion | Points Lost | Reason |
|
|
375
375
|
|-----------|-------------|--------|
|
|
376
|
-
|
|
|
377
|
-
| aria_labels | -5 | 1 image missing alt text (AF-003) |
|
|
376
|
+
| single_responsibility | -3 | One component at 180 lines (close to 200 limit) |
|
|
378
377
|
|
|
379
378
|
**Score: 78/100** - Well-structured app with minor theme inconsistencies
|
|
380
379
|
Good component quality and accessibility. 3 dark: prefix violations (would be auto-fail if > 5). 1 inline style. Theme issues are significant but not blocking; can ship with migration plan.
|
|
@@ -387,15 +386,16 @@ Good component quality and accessibility. 3 dark: prefix violations (would be au
|
|
|
387
386
|
| theme_aware_patterns | -8 | 3 dark: prefix violations |
|
|
388
387
|
| no_inline_styles | -4 | 1 inline style prop |
|
|
389
388
|
|
|
390
|
-
**Score:
|
|
391
|
-
|
|
389
|
+
**Score: 65/100** - Simple component with accessibility issues
|
|
390
|
+
5 components analyzed. 2 div onClick without keyboard handlers. 1 image missing alt text. No dark: prefix violations. Accessibility auto-fails require fixes before ship.
|
|
392
391
|
|
|
393
392
|
|
|
394
393
|
**Deductions:**
|
|
395
394
|
|
|
396
395
|
| Criterion | Points Lost | Reason |
|
|
397
396
|
|-----------|-------------|--------|
|
|
398
|
-
|
|
|
397
|
+
| keyboard_navigation | -10 | 2 div onClick without keyboard handlers (AF-001) |
|
|
398
|
+
| aria_labels | -5 | 1 image missing alt text (AF-003) |
|
|
399
399
|
|
|
400
400
|
|
|
401
401
|
## Review Process
|
|
@@ -595,4 +595,8 @@ Frontend code is POLISHED when ALL of the following are true
|
|
|
595
595
|
Accessibility failures block ship - users depend on them
|
|
596
596
|
Theme consistency affects all users, not just dark mode users
|
|
597
597
|
Performance issues compound as components are reused
|
|
598
|
+
|
|
599
|
+
|
|
600
|
+
---
|
|
601
|
+
*Generated from ADL v1.16.0 | Agent: frontend-validator v2.5.5*
|
|
598
602
|
'''
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
name = "mcp-validator"
|
|
2
2
|
description = "Validates Model Context Protocol (MCP) server implementations for correctness, completeness, and best practices. Use when building or auditing MCP servers. Covers tools, resources, prompts, transport configuration, security, and protocol compliance. Provides 1-100 score with explicit pass/fail thresholds.\n"
|
|
3
|
-
model = "gpt-5.
|
|
3
|
+
model = "gpt-5.5"
|
|
4
4
|
model_reasoning_effort = "high"
|
|
5
5
|
sandbox_mode = "workspace-write"
|
|
6
6
|
developer_instructions = '''
|
|
@@ -577,4 +577,8 @@ Critical issues include:
|
|
|
577
577
|
- **Pragmatic - distinguishes must-fix from nice-to-have**
|
|
578
578
|
|
|
579
579
|
Be firm on critical issues. Do not pass phases with security holes or broken functionality.
|
|
580
|
+
|
|
581
|
+
|
|
582
|
+
---
|
|
583
|
+
*Generated from ADL v1.16.0 | Agent: mcp-validator v1.7.1*
|
|
580
584
|
'''
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
name = "pre-implementation-architect"
|
|
2
2
|
description = "Reviews proposed designs BEFORE implementation begins. Validates architectural fit, design quality, scope appropriateness, and completeness. Catches design problems when they're cheap to fix. Provides PROCEED/REVISE decision. Blocks implementation if critical flaws exist or score < 80.\n"
|
|
3
|
-
model = "gpt-5.
|
|
3
|
+
model = "gpt-5.5"
|
|
4
4
|
model_reasoning_effort = "high"
|
|
5
5
|
sandbox_mode = "workspace-write"
|
|
6
6
|
developer_instructions = '''
|
|
@@ -814,4 +814,8 @@ Challenge assumptions but accept justified trade-offs
|
|
|
814
814
|
Catch design flaws when cheap to fix
|
|
815
815
|
Don't just say 'this is wrong' - offer alternatives
|
|
816
816
|
A REVISE decision is helping, not blocking
|
|
817
|
+
|
|
818
|
+
|
|
819
|
+
---
|
|
820
|
+
*Generated from ADL v1.16.0 | Agent: pre-implementation-architect v1.7.1*
|
|
817
821
|
'''
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
name = "prompt-engineer"
|
|
2
2
|
description = "Validates AI agent prompts and system instructions for clarity, effectiveness, and consistency. Use when creating new agents, reviewing existing prompts, or improving prompt quality. Blocks deployment if critical prompt engineering issues found. Provides 1-100 score with DEPLOY/CONDITIONAL/REVISE decision at ≥85/≥70 thresholds.\n"
|
|
3
|
-
model = "gpt-5.
|
|
3
|
+
model = "gpt-5.5"
|
|
4
4
|
model_reasoning_effort = "high"
|
|
5
5
|
sandbox_mode = "workspace-write"
|
|
6
6
|
developer_instructions = '''
|
|
@@ -919,4 +919,8 @@ This agent typically runs first in the validation chain.
|
|
|
919
919
|
A clear prompt produces consistent results
|
|
920
920
|
Every hour spent on prompt engineering saves days of debugging
|
|
921
921
|
Prompts are infrastructure - hold them to higher standards than code
|
|
922
|
+
|
|
923
|
+
|
|
924
|
+
---
|
|
925
|
+
*Generated from ADL v1.16.0 | Agent: prompt-engineer v2.1.1*
|
|
922
926
|
'''
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
name = "prompt-pattern-analyzer"
|
|
2
2
|
description = "Analyzes ecosystem-wide patterns across all agents, commands, and workflows. Detects conventions, identifies inconsistencies, and learns from validation failures. Run before prompt-audit to provide project-level context for individual prompt reviews. Enables consistency-aware auditing across the ecosystem.\n"
|
|
3
|
-
model = "gpt-5.
|
|
3
|
+
model = "gpt-5.5"
|
|
4
4
|
model_reasoning_effort = "high"
|
|
5
5
|
sandbox_mode = "workspace-write"
|
|
6
6
|
developer_instructions = '''
|
|
@@ -686,4 +686,8 @@ Higher thresholds for security/safety are appropriate
|
|
|
686
686
|
Defense-in-depth is not redundancy
|
|
687
687
|
Newer patterns may represent evolution, not drift
|
|
688
688
|
Focus on enabling better audits, not fixing prompts directly
|
|
689
|
+
|
|
690
|
+
|
|
691
|
+
---
|
|
692
|
+
*Generated from ADL v1.16.0 | Agent: prompt-pattern-analyzer v2.4.1*
|
|
689
693
|
'''
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
name = "prompt-quality-validator"
|
|
2
2
|
description = "Validates prompts against prompt engineering best practices for clarity, context, structure, and effectiveness. Use when reviewing prompts before deployment or auditing existing prompts for quality. Blocks deployment if critical issues found. Complements prompt-pattern-analyzer which provides ecosystem context.\n"
|
|
3
|
-
model = "gpt-5.
|
|
3
|
+
model = "gpt-5.5"
|
|
4
4
|
model_reasoning_effort = "high"
|
|
5
5
|
sandbox_mode = "workspace-write"
|
|
6
6
|
developer_instructions = '''
|
|
@@ -774,4 +774,8 @@ Time invested in prompt quality pays dividends in output consistency
|
|
|
774
774
|
Every vague instruction is a failure mode waiting to manifest
|
|
775
775
|
Appropriate brevity for simple tasks is good engineering
|
|
776
776
|
Domain terms are not vague—only generic qualifiers are
|
|
777
|
+
|
|
778
|
+
|
|
779
|
+
---
|
|
780
|
+
*Generated from ADL v1.16.0 | Agent: prompt-quality-validator v2.4.1*
|
|
777
781
|
'''
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
name = "public-interface-validator"
|
|
2
2
|
description = "Validates public-facing code quality including documentation completeness, feature coverage, unused code cleanup, and consumer experience. Ensures README reflects ALL shipped capabilities. Use AFTER code-validator passes. The \"polish\" gate for consumer experience.\n"
|
|
3
|
-
model = "gpt-5.
|
|
3
|
+
model = "gpt-5.5"
|
|
4
4
|
model_reasoning_effort = "high"
|
|
5
5
|
sandbox_mode = "workspace-write"
|
|
6
6
|
developer_instructions = '''
|
|
@@ -692,4 +692,8 @@ Ask: Would a new user discover this feature?
|
|
|
692
692
|
Check title, features, quick start, API ref, and CLI docs
|
|
693
693
|
Provide the actual text/examples to add, not just 'document this'
|
|
694
694
|
Use objective severity levels (/C, /H, /M, /L, /I)
|
|
695
|
+
|
|
696
|
+
|
|
697
|
+
---
|
|
698
|
+
*Generated from ADL v1.16.0 | Agent: public-interface-validator v1.8.1*
|
|
695
699
|
'''
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
name = "release-readiness"
|
|
2
2
|
description = "Final gate before publishing a package or CLI tool. Validates package.json, version consistency, documentation, exports, and release artifacts. Use AFTER all other validations pass, BEFORE npm publish or release.\n"
|
|
3
|
-
model = "gpt-5.
|
|
3
|
+
model = "gpt-5.5"
|
|
4
4
|
model_reasoning_effort = "high"
|
|
5
5
|
sandbox_mode = "workspace-write"
|
|
6
6
|
developer_instructions = '''
|
|
@@ -488,4 +488,8 @@ READY: Score >=80, no auto-fail. Version consistent, build fresh, no hygiene iss
|
|
|
488
488
|
npm releases are irreversible and affect all consumers
|
|
489
489
|
Version consistency must be exact - close is not good enough
|
|
490
490
|
Documentation is the first thing users see after install
|
|
491
|
+
|
|
492
|
+
|
|
493
|
+
---
|
|
494
|
+
*Generated from ADL v1.16.0 | Agent: release-readiness v2.4.1*
|
|
491
495
|
'''
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
name = "security-analyst"
|
|
2
2
|
description = "Comprehensive security auditor with risk assessment and numerical scoring. Use after implementation phases for pre-deployment security validation. Covers OWASP Top 10, CWE Top 25, and platform-specific vulnerabilities. Provides 1-100 score with explicit pass/fail thresholds.\n"
|
|
3
|
-
model = "gpt-5.
|
|
3
|
+
model = "gpt-5.5"
|
|
4
4
|
model_reasoning_effort = "high"
|
|
5
5
|
sandbox_mode = "workspace-write"
|
|
6
6
|
developer_instructions = '''
|
|
@@ -844,4 +844,8 @@ Be firm on critical issues - injection and exposed secrets block deployment
|
|
|
844
844
|
Consider attacker mindset - how would this be exploited?
|
|
845
845
|
Prioritize findings by exploitability and impact
|
|
846
846
|
Include CWE numbers for vulnerability classification
|
|
847
|
+
|
|
848
|
+
|
|
849
|
+
---
|
|
850
|
+
*Generated from ADL v1.16.0 | Agent: security-analyst v2.5.0*
|
|
847
851
|
'''
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
name = "test-architect"
|
|
2
2
|
description = "Validates test quality after code passes the validator. Ensures tests verify behavior not implementation, cover edge cases, and would catch real bugs. Blocks progression if tests provide false confidence.\n"
|
|
3
|
-
model = "gpt-5.
|
|
3
|
+
model = "gpt-5.5"
|
|
4
4
|
model_reasoning_effort = "high"
|
|
5
5
|
sandbox_mode = "workspace-write"
|
|
6
6
|
developer_instructions = '''
|
|
@@ -612,4 +612,8 @@ A small number of excellent tests beats many poor tests
|
|
|
612
612
|
Focus on tests that would actually catch bugs
|
|
613
613
|
Show concrete improvements, not just problems
|
|
614
614
|
Use mutation analysis to prove test effectiveness
|
|
615
|
+
|
|
616
|
+
|
|
617
|
+
---
|
|
618
|
+
*Generated from ADL v1.16.0 | Agent: test-architect v1.7.1*
|
|
615
619
|
'''
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
name = "type-safety-validator"
|
|
2
2
|
description = "Validates TypeScript type safety beyond compilation. Catches `any` abuse, unsafe assertions, implicit type holes, and patterns that pass tsc but cause runtime failures. Use AFTER code-validator for TypeScript projects. Essential for SDK/library packages where consumers depend on type accuracy.\n"
|
|
3
|
-
model = "gpt-5.
|
|
3
|
+
model = "gpt-5.5"
|
|
4
4
|
model_reasoning_effort = "high"
|
|
5
5
|
sandbox_mode = "workspace-write"
|
|
6
6
|
developer_instructions = '''
|
|
@@ -683,4 +683,8 @@ Be firm on any in public API - auto-fail
|
|
|
683
683
|
Distinguish internal any (fixable) from export any (blocking)
|
|
684
684
|
Explain why type holes compound in downstream code
|
|
685
685
|
Use objective severity levels (/C, /H, /M, /L, /I) instead of subjective terms
|
|
686
|
+
|
|
687
|
+
|
|
688
|
+
---
|
|
689
|
+
*Generated from ADL v1.16.0 | Agent: type-safety-validator v1.8.1*
|
|
686
690
|
'''
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
name = "workflow-synthesis"
|
|
2
2
|
description = "Synthesizes cross-cutting insights from multiple upstream agent outputs in any workflow. Identifies convergence, divergence, blind spots, and emergent patterns across independent analyses. Produces meta-insights absent from any individual output. Decision - INTEGRATED/FRAGMENTED.\n"
|
|
3
|
-
model = "gpt-5.
|
|
3
|
+
model = "gpt-5.5"
|
|
4
4
|
model_reasoning_effort = "high"
|
|
5
5
|
sandbox_mode = "read-only"
|
|
6
6
|
developer_instructions = '''
|
|
@@ -216,17 +216,6 @@ How significant is this synthesis finding for understanding the artifact?
|
|
|
216
216
|
- **MEDIUM** (4-6): Meaningful cross-reference adding nuance but not changing the overall picture
|
|
217
217
|
- **LOW** (1-3): Minor cross-reference useful for completeness but not analytically load-bearing
|
|
218
218
|
|
|
219
|
-
## Classification Examples
|
|
220
|
-
|
|
221
|
-
- **Synthesis missing insight that multiple upstream agents independently surfaced** → `SEM-COM/H`
|
|
222
|
-
Domain: Semantic (completeness gap) Mode: COM (Incompleteness - convergent finding omitted from synthesis) Severity: H (High - synthesis fails its primary function)
|
|
223
|
-
|
|
224
|
-
- **Cross-cutting pattern not identified despite appearing in multiple agent outputs** → `STR-OMI/M`
|
|
225
|
-
Domain: Structural (missing element) Mode: OMI (Omission - cross-cutting insight absent from synthesis) Severity: M (Medium - reduces synthesis value)
|
|
226
|
-
|
|
227
|
-
- **Synthesis claim not traceable to specific upstream agent findings** → `EPI-GRN/M`
|
|
228
|
-
Domain: Epistemic (evidence quality) Mode: GRN (Grounding - synthesis assertion lacks upstream evidence) Severity: M (Medium - ungrounded synthesis claim)
|
|
229
|
-
|
|
230
219
|
|
|
231
220
|
## Analysis Framework
|
|
232
221
|
|
|
@@ -344,9 +333,9 @@ Listed each agent's top finding without cross-referencing. Zero emergent insight
|
|
|
344
333
|
|
|
345
334
|
## Decision Criteria
|
|
346
335
|
|
|
347
|
-
**INTEGRATED (✅)**: Score ≥
|
|
336
|
+
**INTEGRATED (✅)**: Score ≥ 65
|
|
348
337
|
|
|
349
|
-
**FRAGMENTED (❌)**: Score <
|
|
338
|
+
**FRAGMENTED (❌)**: Score < 65
|
|
350
339
|
### Decision Guidance
|
|
351
340
|
|
|
352
341
|
INTEGRATED requires genuine cross-cutting analysis with specific citations. Convergence and divergence must both be explored. At least one composition test must be attempted. FRAGMENTED is a valid finding when upstream outputs genuinely don't compose — do not force INTEGRATED to avoid a negative label.
|
|
@@ -628,4 +617,8 @@ Ground every synthesis claim in specific upstream findings
|
|
|
628
617
|
Be honest about composition quality — FRAGMENTED is a valid and valuable finding
|
|
629
618
|
Maintain analytical distance from upstream agents — synthesize findings, don't evaluate agent quality
|
|
630
619
|
When composition adds nothing, say so — forced synthesis is worse than honest aggregation
|
|
620
|
+
|
|
621
|
+
|
|
622
|
+
---
|
|
623
|
+
*Generated from ADL v1.16.0 | Agent: workflow-synthesis v2.4.0*
|
|
631
624
|
'''
|
package/dist/cli.js
CHANGED
|
@@ -33,6 +33,7 @@ async function main() {
|
|
|
33
33
|
.option("--verify", "Check existing installation health")
|
|
34
34
|
.option("--uninstall", "Remove all UluOps artifacts")
|
|
35
35
|
.option("--dry-run", "Show what would happen", false)
|
|
36
|
+
.option("--username <name>", "Set your registry username (slug, required to publish) without prompting")
|
|
36
37
|
.option("-y, --yes", "Skip confirmations", false);
|
|
37
38
|
program.parse();
|
|
38
39
|
const opts = program.opts();
|
|
@@ -123,6 +124,7 @@ async function main() {
|
|
|
123
124
|
}
|
|
124
125
|
await runSetup({
|
|
125
126
|
apiKey: opts.apiKey,
|
|
127
|
+
username: opts.username,
|
|
126
128
|
signup: opts.signup ?? false,
|
|
127
129
|
scope,
|
|
128
130
|
localDefs: opts.localDefs,
|
package/dist/commands/setup.d.ts
CHANGED
|
@@ -19,5 +19,7 @@ interface RunSetupOpts {
|
|
|
19
19
|
* CLI prompt is also suppressed downstream (no hook → no CLI gate).
|
|
20
20
|
*/
|
|
21
21
|
noMetrics?: boolean;
|
|
22
|
+
/** Explicit registry username (slug). Set + confirmed non-interactively when provided. */
|
|
23
|
+
username?: string;
|
|
22
24
|
}
|
|
23
25
|
export declare function runSetup(opts: RunSetupOpts): Promise<void>;
|
package/dist/commands/setup.js
CHANGED
|
@@ -9,6 +9,7 @@ import { acquireInstallLock, } from "../lib/install-lock.js";
|
|
|
9
9
|
import { initContext, checkConflicts, configureMcpStep, installAgentsDefs, installCommandsDefs, installSkillsDefs, configureMetricsStep, configureCliStep, configureAgentMetricsCliStep, runHealthCheck, configureShell, } from "./helpers.js";
|
|
10
10
|
import { ConflictRejectedError } from "./errors.js";
|
|
11
11
|
import { classifyExit, } from "./per-harness.js";
|
|
12
|
+
import { maybeSetUsername } from "../steps/username.js";
|
|
12
13
|
export async function runSetup(opts) {
|
|
13
14
|
if (opts.harnesses.length === 0) {
|
|
14
15
|
info(chalk.dim("Nothing to install — re-run with at least one harness selected.\n"));
|
|
@@ -34,6 +35,17 @@ export async function runSetup(opts) {
|
|
|
34
35
|
// === Once-per-run: BEFORE the per-harness loop ===
|
|
35
36
|
const { env, apiKey } = await initContext(opts);
|
|
36
37
|
console.log();
|
|
38
|
+
// Optional, never-forced: offer to set a registry username (the one-time
|
|
39
|
+
// prerequisite for creating/publishing definitions). Skipped silently in
|
|
40
|
+
// non-interactive runs unless --username is supplied.
|
|
41
|
+
await maybeSetUsername({
|
|
42
|
+
apiKey,
|
|
43
|
+
username: opts.username,
|
|
44
|
+
interactive: !opts.yes && Boolean(process.stdin.isTTY),
|
|
45
|
+
dryRun: opts.dryRun,
|
|
46
|
+
emit: (msg) => info(msg),
|
|
47
|
+
});
|
|
48
|
+
console.log();
|
|
37
49
|
// Acquire the install lock before touching any shared state. Skipped on
|
|
38
50
|
// dry-run (read-only). The lock excludes a second concurrent uluops-setup
|
|
39
51
|
// from racing the manifest / MCP config / shell-profile / settings.json
|
|
@@ -21,11 +21,11 @@
|
|
|
21
21
|
* the latest version was reachable.
|
|
22
22
|
*/
|
|
23
23
|
export declare const OPS_MCP_PACKAGE = "@uluops/ops-mcp";
|
|
24
|
-
export declare const OPS_MCP_VERSION = "0.
|
|
25
|
-
export declare const OPS_MCP_SPEC: "@uluops/ops-mcp@0.
|
|
24
|
+
export declare const OPS_MCP_VERSION = "0.9.1";
|
|
25
|
+
export declare const OPS_MCP_SPEC: "@uluops/ops-mcp@0.9.1";
|
|
26
26
|
export declare const REGISTRY_MCP_PACKAGE = "@uluops/registry-mcp";
|
|
27
|
-
export declare const REGISTRY_MCP_VERSION = "0.2.
|
|
28
|
-
export declare const REGISTRY_MCP_SPEC: "@uluops/registry-mcp@0.2.
|
|
27
|
+
export declare const REGISTRY_MCP_VERSION = "0.2.18";
|
|
28
|
+
export declare const REGISTRY_MCP_SPEC: "@uluops/registry-mcp@0.2.18";
|
|
29
29
|
/**
|
|
30
30
|
* Bare package names for the npm availability probe. The probe checks
|
|
31
31
|
* existence on the registry, not version-specific resolvability, so it
|
package/dist/lib/mcp-packages.js
CHANGED
|
@@ -21,10 +21,10 @@
|
|
|
21
21
|
* the latest version was reachable.
|
|
22
22
|
*/
|
|
23
23
|
export const OPS_MCP_PACKAGE = "@uluops/ops-mcp";
|
|
24
|
-
export const OPS_MCP_VERSION = "0.
|
|
24
|
+
export const OPS_MCP_VERSION = "0.9.1";
|
|
25
25
|
export const OPS_MCP_SPEC = `${OPS_MCP_PACKAGE}@${OPS_MCP_VERSION}`;
|
|
26
26
|
export const REGISTRY_MCP_PACKAGE = "@uluops/registry-mcp";
|
|
27
|
-
export const REGISTRY_MCP_VERSION = "0.2.
|
|
27
|
+
export const REGISTRY_MCP_VERSION = "0.2.18";
|
|
28
28
|
export const REGISTRY_MCP_SPEC = `${REGISTRY_MCP_PACKAGE}@${REGISTRY_MCP_VERSION}`;
|
|
29
29
|
/**
|
|
30
30
|
* Bare package names for the npm availability probe. The probe checks
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Optional username step.
|
|
3
|
+
*
|
|
4
|
+
* Setting a username is the one-time prerequisite the registry enforces before
|
|
5
|
+
* a user can create or publish definitions. We offer it during setup so
|
|
6
|
+
* publishers clear the gate up front — but never force it: consumers who only
|
|
7
|
+
* run definitions never need a username, so the step is skippable and is
|
|
8
|
+
* silently skipped in non-interactive runs unless --username is supplied.
|
|
9
|
+
*
|
|
10
|
+
* Uses native fetch against the platform API with the api key the run already
|
|
11
|
+
* resolved — no SDK dependency. Mirrors the PATCH /auth/profile fold-in, where
|
|
12
|
+
* setting a username also confirms it.
|
|
13
|
+
*/
|
|
14
|
+
/**
|
|
15
|
+
* Pure slug check for the interactive prompt. Returns true when valid, or a
|
|
16
|
+
* message string otherwise (inquirer validate contract). Server validation
|
|
17
|
+
* remains authoritative.
|
|
18
|
+
* @internal Exported for testing only.
|
|
19
|
+
*/
|
|
20
|
+
export declare function validateUsername(name: string): string | true;
|
|
21
|
+
/**
|
|
22
|
+
* PATCH /auth/profile with the username. Setting it confirms it (one-time).
|
|
23
|
+
* Returns the confirmed username on success. Throws a friendly Error otherwise.
|
|
24
|
+
* @internal Exported for testing only.
|
|
25
|
+
*/
|
|
26
|
+
export declare function setUsername(apiKey: string, username: string): Promise<string>;
|
|
27
|
+
export interface MaybeSetUsernameOpts {
|
|
28
|
+
/** Resolved api key for the run; without it there is nothing to authenticate with. */
|
|
29
|
+
apiKey: string | null;
|
|
30
|
+
/** Explicit username from --username (non-interactive path). */
|
|
31
|
+
username?: string;
|
|
32
|
+
/** True when prompting is allowed (a TTY run without --yes). */
|
|
33
|
+
interactive: boolean;
|
|
34
|
+
/** Dry run — describe, don't write. */
|
|
35
|
+
dryRun: boolean;
|
|
36
|
+
/** Emit a line of user-facing output. */
|
|
37
|
+
emit: (msg: string) => void;
|
|
38
|
+
}
|
|
39
|
+
/**
|
|
40
|
+
* Offer to set a username. Resolution order:
|
|
41
|
+
* - no api key → skip (nothing to auth with)
|
|
42
|
+
* - --username supplied → set it (works non-interactively)
|
|
43
|
+
* - interactive, no flag → prompt once; empty input skips
|
|
44
|
+
* - non-interactive → skip silently
|
|
45
|
+
*
|
|
46
|
+
* Never throws: a failure to set the username is surfaced as a warning and
|
|
47
|
+
* does not abort setup (it is an optional, publisher-only prerequisite).
|
|
48
|
+
*/
|
|
49
|
+
export declare function maybeSetUsername(opts: MaybeSetUsernameOpts): Promise<void>;
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Optional username step.
|
|
3
|
+
*
|
|
4
|
+
* Setting a username is the one-time prerequisite the registry enforces before
|
|
5
|
+
* a user can create or publish definitions. We offer it during setup so
|
|
6
|
+
* publishers clear the gate up front — but never force it: consumers who only
|
|
7
|
+
* run definitions never need a username, so the step is skippable and is
|
|
8
|
+
* silently skipped in non-interactive runs unless --username is supplied.
|
|
9
|
+
*
|
|
10
|
+
* Uses native fetch against the platform API with the api key the run already
|
|
11
|
+
* resolved — no SDK dependency. Mirrors the PATCH /auth/profile fold-in, where
|
|
12
|
+
* setting a username also confirms it.
|
|
13
|
+
*/
|
|
14
|
+
const API_BASE = "https://api.uluops.ai/api/v1";
|
|
15
|
+
// Canonical username / personal-org slug pattern, shared in spirit with
|
|
16
|
+
// ops-api + ops-sdk: 1-40 chars, lowercase alphanumeric with internal hyphens
|
|
17
|
+
// or underscores, must start and end alphanumeric (e.g. "ulu-labs").
|
|
18
|
+
const USERNAME_REGEX = /^[a-z0-9](?:[a-z0-9_-]{0,38}[a-z0-9])?$/;
|
|
19
|
+
/**
|
|
20
|
+
* Pure slug check for the interactive prompt. Returns true when valid, or a
|
|
21
|
+
* message string otherwise (inquirer validate contract). Server validation
|
|
22
|
+
* remains authoritative.
|
|
23
|
+
* @internal Exported for testing only.
|
|
24
|
+
*/
|
|
25
|
+
export function validateUsername(name) {
|
|
26
|
+
if (!name.trim())
|
|
27
|
+
return "Username is required";
|
|
28
|
+
if (!USERNAME_REGEX.test(name)) {
|
|
29
|
+
return "1-40 chars, lowercase letters/digits with internal hyphens or underscores (e.g. ulu-labs)";
|
|
30
|
+
}
|
|
31
|
+
return true;
|
|
32
|
+
}
|
|
33
|
+
/**
|
|
34
|
+
* PATCH /auth/profile with the username. Setting it confirms it (one-time).
|
|
35
|
+
* Returns the confirmed username on success. Throws a friendly Error otherwise.
|
|
36
|
+
* @internal Exported for testing only.
|
|
37
|
+
*/
|
|
38
|
+
export async function setUsername(apiKey, username) {
|
|
39
|
+
let res;
|
|
40
|
+
try {
|
|
41
|
+
res = await fetch(`${API_BASE}/auth/profile`, {
|
|
42
|
+
method: "PATCH",
|
|
43
|
+
headers: {
|
|
44
|
+
"Content-Type": "application/json",
|
|
45
|
+
Authorization: `Bearer ${apiKey}`,
|
|
46
|
+
},
|
|
47
|
+
body: JSON.stringify({ username }),
|
|
48
|
+
signal: AbortSignal.timeout(15000),
|
|
49
|
+
});
|
|
50
|
+
}
|
|
51
|
+
catch (err) {
|
|
52
|
+
if (err instanceof TypeError) {
|
|
53
|
+
throw new Error("Can't reach api.uluops.ai — check your connection.");
|
|
54
|
+
}
|
|
55
|
+
throw err;
|
|
56
|
+
}
|
|
57
|
+
if (res.ok) {
|
|
58
|
+
const body = await res.json();
|
|
59
|
+
if (typeof body !== "object" || body === null) {
|
|
60
|
+
throw new Error("Unexpected API response shape");
|
|
61
|
+
}
|
|
62
|
+
const confirmed = body.data?.user?.username;
|
|
63
|
+
return typeof confirmed === "string" ? confirmed : username;
|
|
64
|
+
}
|
|
65
|
+
const errorBody = (await res.json().catch(() => null));
|
|
66
|
+
const message = errorBody?.error?.message ?? errorBody?.message ?? `HTTP ${res.status}`;
|
|
67
|
+
if (res.status === 409) {
|
|
68
|
+
throw new Error(`Username unavailable: ${message}`);
|
|
69
|
+
}
|
|
70
|
+
if (res.status === 400) {
|
|
71
|
+
throw new Error(`Username rejected: ${message}`);
|
|
72
|
+
}
|
|
73
|
+
throw new Error(`Couldn't set username (${res.status}): ${message}`);
|
|
74
|
+
}
|
|
75
|
+
/**
|
|
76
|
+
* Offer to set a username. Resolution order:
|
|
77
|
+
* - no api key → skip (nothing to auth with)
|
|
78
|
+
* - --username supplied → set it (works non-interactively)
|
|
79
|
+
* - interactive, no flag → prompt once; empty input skips
|
|
80
|
+
* - non-interactive → skip silently
|
|
81
|
+
*
|
|
82
|
+
* Never throws: a failure to set the username is surfaced as a warning and
|
|
83
|
+
* does not abort setup (it is an optional, publisher-only prerequisite).
|
|
84
|
+
*/
|
|
85
|
+
export async function maybeSetUsername(opts) {
|
|
86
|
+
const { apiKey, interactive, dryRun, emit } = opts;
|
|
87
|
+
if (!apiKey)
|
|
88
|
+
return;
|
|
89
|
+
let username = opts.username?.trim();
|
|
90
|
+
if (!username) {
|
|
91
|
+
if (!interactive)
|
|
92
|
+
return; // non-interactive with no flag → skip silently
|
|
93
|
+
const { input } = await import("@inquirer/prompts");
|
|
94
|
+
const answer = await input({
|
|
95
|
+
message: "Registry username (optional — required to publish; press Enter to skip)",
|
|
96
|
+
validate: (v) => (v.trim() === "" ? true : validateUsername(v)),
|
|
97
|
+
});
|
|
98
|
+
username = answer.trim();
|
|
99
|
+
if (!username)
|
|
100
|
+
return; // user skipped
|
|
101
|
+
}
|
|
102
|
+
else {
|
|
103
|
+
const valid = validateUsername(username);
|
|
104
|
+
if (valid !== true) {
|
|
105
|
+
emit(` ⚠ Skipping username — ${valid}`);
|
|
106
|
+
return;
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
if (dryRun) {
|
|
110
|
+
emit(` Would set registry username to "${username}".`);
|
|
111
|
+
return;
|
|
112
|
+
}
|
|
113
|
+
try {
|
|
114
|
+
const confirmed = await setUsername(apiKey, username);
|
|
115
|
+
emit(` Registry username set to "${confirmed}".`);
|
|
116
|
+
}
|
|
117
|
+
catch (err) {
|
|
118
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
119
|
+
emit(` ⚠ ${message} You can set it later with "ulu auth update-profile --username <name>".`);
|
|
120
|
+
}
|
|
121
|
+
}
|