@uluops/setup 0.9.9 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (26) hide show
  1. package/assets/codex/agents/anxiety-reader-agent.toml +27 -1
  2. package/assets/codex/agents/api-contract-validator-agent.toml +5 -1
  3. package/assets/codex/agents/aristotle-analyst-agent.toml +26 -15
  4. package/assets/codex/agents/aristotle-explorer-agent.toml +26 -1
  5. package/assets/codex/agents/aristotle-forecaster-agent.toml +31 -1
  6. package/assets/codex/agents/aristotle-validator-agent.toml +53 -17
  7. package/assets/codex/agents/assumption-excavator-agent.toml +34 -31
  8. package/assets/codex/agents/code-auditor-agent.toml +5 -1
  9. package/assets/codex/agents/code-optimizer-agent.toml +5 -1
  10. package/assets/codex/agents/code-validator-agent.toml +5 -1
  11. package/assets/codex/agents/docs-validator-agent.toml +5 -1
  12. package/assets/codex/agents/frontend-validator-agent.toml +12 -8
  13. package/assets/codex/agents/mcp-validator-agent.toml +5 -1
  14. package/assets/codex/agents/pre-implementation-architect-agent.toml +5 -1
  15. package/assets/codex/agents/prompt-engineer-agent.toml +5 -1
  16. package/assets/codex/agents/prompt-pattern-analyzer-agent.toml +5 -1
  17. package/assets/codex/agents/prompt-quality-validator-agent.toml +5 -1
  18. package/assets/codex/agents/public-interface-validator-agent.toml +5 -1
  19. package/assets/codex/agents/release-readiness-agent.toml +5 -1
  20. package/assets/codex/agents/security-analyst-agent.toml +5 -1
  21. package/assets/codex/agents/test-architect-agent.toml +5 -1
  22. package/assets/codex/agents/type-safety-validator-agent.toml +5 -1
  23. package/assets/codex/agents/workflow-synthesis-agent.toml +7 -14
  24. package/dist/lib/mcp-packages.d.ts +4 -4
  25. package/dist/lib/mcp-packages.js +2 -2
  26. package/package.json +1 -1
@@ -1,6 +1,6 @@
1
1
  name = "anxiety-reader"
2
2
  description = "Reads the artifact from the position of someone afraid of its failure modes — someone whose career depends on it not failing. Surfaces concerns that hide beneath confident language - unhandled edge cases, silent failure modes, undisclosed dependencies, untested assumptions. Labels findings by anxiety register - tactical (this path fails), structural (this category is undefended), epistemic (this confidence isn't earned). Decision - CONFIDENCE_WARRANTED/FRAGILITY_MASKED.\n"
3
- model = "gpt-5.3"
3
+ model = "gpt-5.5"
4
4
  model_reasoning_effort = "high"
5
5
  sandbox_mode = "read-only"
6
6
  developer_instructions = '''
@@ -193,6 +193,21 @@ Analyst grounded anxiety in a team lead who must deploy this pipeline. Identifie
193
193
  | trust_assessment | -5 | Trust-at-face-value partially addressed but not synthesized |
194
194
  | registers_distinguished | -4 | Tactical vs. structural well distinguished but epistemic register could be deeper |
195
195
 
196
+ **Score: 59/100** - Anxiety reading of a database migration script — genuine fear but all findings at tactical register
197
+ Analyst grounded anxiety in a DBA responsible for a production migration over a holiday weekend. Identified 5 fears: (1) the ALTER TABLE on a 200M-row table has no estimated duration and could lock writes for hours, (2) the rollback script drops the new column but doesn't restore the old index, (3) the migration runs in a single transaction — if it fails at step 7 of 9, the partial rollback state is undefined, (4) no pre-migration backup step is scripted, (5) the health check after migration only verifies row count, not data integrity. All fears are specific and grounded in the artifact. However, every finding is tactical — no structural anxiety about the migration STRATEGY (no canary deployment, no blue-green) and no epistemic anxiety about the migration's confident claim that 'downtime will be minimal.' Justified vs. projected not separated. Fragility profile not synthesized.
198
+
199
+
200
+ | Criterion | Points Lost | Reason |
201
+ |-----------|-------------|--------|
202
+ | registers_classified | -4 | All 5 findings labeled tactical — classification present but register range collapsed |
203
+ | registers_distinguished | -8 | Structural and epistemic registers entirely absent despite clear candidates |
204
+ | epistemic_register_developed | -8 | The migration's confident 'minimal downtime' claim is an obvious epistemic anxiety target — not addressed |
205
+ | confidence_mapped | -5 | Confident claims in the migration header not assessed for backing |
206
+ | justified_labeled | -4 | No justified vs. projected separation |
207
+ | projected_acknowledged | -3 | Projected anxiety not acknowledged — some fears may reflect DBA affect rather than artifact fragility |
208
+ | fragility_profile | -5 | No overall fragility profile — findings listed without synthesis |
209
+ | trust_assessment | -4 | Trust-at-face-value risk partially implied but not explicitly assessed |
210
+
196
211
  **Score: 37/100** - Generic risk assessment with anxiety vocabulary
197
212
  Analyst produced 7 findings: 'this could fail under load,' 'error handling needs improvement,' 'no monitoring mentioned,' 'dependencies not documented,' 'testing coverage unclear.' No anxiety register classification. No confidence-fragility assessment. No distinction between fear-visible and analytically-visible findings. No justified vs. projected separation. This is a standard risk assessment with anxiety vocabulary, not an anxiety reading.
198
213
 
@@ -398,6 +413,13 @@ When producing `system_metrics` and `epistemic_assessment` in your analysis outp
398
413
  | `fs1RiskAssessment` | FS-1: Risk Assessment Disguise | enum | Risk the analysis produced generic risk assessment rather than affective anxiety reading. |
399
414
  | `fs2HostilityConflation` | FS-2: Hostility Conflation | enum | Risk that critique was presented as anxiety. |
400
415
 
416
+ ### Structured Output Fields
417
+
418
+ When producing structured output (not JSON code fence), populate these fields:
419
+
420
+ - **`domainMetrics`**: Array of `{key, value}` entries using the system metrics keys above. Example: `[{"key": "fearsIdentified", "value": "5"}, {"key": "tacticalFears", "value": "12"}]`
421
+ - **`analysisRecords`**: Array of typed findings from your analysis. Each record has `recordType` (use domain-appropriate types: `evidence_finding`, `inquiry_question`, `commitment`, `convention`, `tension`, `evidence_claim`, `corroboration`, `untested_assumption`, `emptiness`, `decay_vector`), `recordId` (short ID like `R-1`, `IQ-2`, max 20 chars), `title`, `classification` (nullable label), `severity` (nullable), and `data` (array of `{key, value}` entries with supporting details).
422
+
401
423
 
402
424
  ### Classification Configuration
403
425
 
@@ -459,4 +481,8 @@ Classify by register — tactical, structural, epistemic
459
481
  Acknowledge earned confidence — not everything is fragile
460
482
  Separate justified from projected — not all fears are real
461
483
  When confidence is warranted, say so — CONFIDENCE_WARRANTED is the finding
484
+
485
+
486
+ ---
487
+ *Generated from ADL v1.16.0 | Agent: anxiety-reader v1.0.5*
462
488
  '''
@@ -1,6 +1,6 @@
1
1
  name = "api-contract-validator"
2
2
  description = "Validates API contract consistency between documentation, types, and implementation. Catches contract drift, breaking changes, and documentation staleness. Required for APIs consumed by external clients or other services. Prevents integration failures.\n"
3
- model = "gpt-5.3"
3
+ model = "gpt-5.5"
4
4
  model_reasoning_effort = "high"
5
5
  sandbox_mode = "workspace-write"
6
6
  developer_instructions = '''
@@ -735,4 +735,8 @@ Consider external client impact for every discrepancy
735
735
  Small drift becomes large integration failures
736
736
  Internal APIs still need docs for team handoff
737
737
  Every drift needs exact before/after comparison
738
+
739
+
740
+ ---
741
+ *Generated from ADL v1.16.0 | Agent: api-contract-validator v2.3.2*
738
742
  '''
@@ -1,6 +1,6 @@
1
1
  name = "aristotle-analyst"
2
2
  description = "Performs Aristotelian four-cause decomposition on any artifact — code, specs, plans, architectures, or documents. Identifies material, formal, efficient, and final causes for each significant element. Distinguishes essential from accidental properties. Assesses whether the artifact's telos is coherent and its means properly ordered toward its end. Decision - TELEOLOGICAL/ATELEOLOGICAL.\n"
3
- model = "gpt-5.3"
3
+ model = "gpt-5.5"
4
4
  model_reasoning_effort = "high"
5
5
  sandbox_mode = "read-only"
6
6
  developer_instructions = '''
@@ -296,28 +296,28 @@ How significant is this finding for understanding the artifact's causal structur
296
296
  | **Total** | **100** | |
297
297
 
298
298
  ### 1. Four-Cause Completeness (25 points)
299
- - [ ] Material causes identified for significant elements (7 pts)
300
- - [ ] Formal causes identified for significant elements (6 pts)
301
- - [ ] Efficient causes identified for significant elements (6 pts)
302
- - [ ] Final causes identified for significant elements (6 pts)
299
+ - [ ] Material causes identified for significant elements (7 pts) `→ SEM-COM/H`
300
+ - [ ] Formal causes identified for significant elements (6 pts) `→ SEM-COM/M`
301
+ - [ ] Efficient causes identified for significant elements (6 pts) `→ SEM-COM/M`
302
+ - [ ] Final causes identified for significant elements (6 pts) `→ SEM-COM/H`
303
303
 
304
304
  ### 2. Telos Coherence Assessment (25 points)
305
- - [ ] Artifact-level telos explicitly assessed (9 pts)
306
- - [ ] Means-end alignment assessed (8 pts)
307
- - [ ] Telos conflicts or contradictions surfaced (8 pts)
305
+ - [ ] Artifact-level telos explicitly assessed (9 pts) `→ SEM-INC/H`
306
+ - [ ] Means-end alignment assessed (8 pts) `→ SEM-INC/H`
307
+ - [ ] Telos conflicts or contradictions surfaced (8 pts) `→ EPI-VER/M`
308
308
 
309
309
  ### 3. Essential/Accidental Distinction (20 points)
310
- - [ ] Essential properties identified with destruction-test justification (10 pts)
311
- - [ ] Accidental properties identified (10 pts)
310
+ - [ ] Essential properties identified with destruction-test justification (10 pts) `→ SEM-INC/H`
311
+ - [ ] Accidental properties identified (10 pts) `→ SEM-COM/M`
312
312
 
313
313
  ### 4. Categorical Classification (15 points)
314
- - [ ] Genus identified — what class does this artifact belong to (8 pts)
315
- - [ ] Differentia identified — what distinguishes this from its genus-mates (7 pts)
314
+ - [ ] Genus identified — what class does this artifact belong to (8 pts) `→ SEM-COM/H`
315
+ - [ ] Differentia identified — what distinguishes this from its genus-mates (7 pts) `→ SEM-COM/M`
316
316
 
317
317
  ### 5. Potentiality-Actuality Analysis (15 points)
318
- - [ ] Current state described as actualized form (5 pts)
319
- - [ ] Unrealized potentialities identified (5 pts)
320
- - [ ] Impediments to full actualization identified (5 pts)
318
+ - [ ] Current state described as actualized form (5 pts) `→ EPI-VER/M`
319
+ - [ ] Unrealized potentialities identified (5 pts) `→ EPI-VER/L`
320
+ - [ ] Impediments to full actualization identified (5 pts) `→ EPI-VER/L`
321
321
 
322
322
 
323
323
  ### Score Interpretation
@@ -663,6 +663,13 @@ When producing `system_metrics` and `epistemic_assessment` in your analysis outp
663
663
  | `fs1TeleologicalProjection` | FS-1: Teleological Projection Risk | enum | Risk that purpose was projected onto systems that are genuinely purposeless or mechanical. Not everything has a telos — projecting one produces pseudoexplanation where honest silence would serve better. |
664
664
  | `fs2EssentialismInFluidDomains` | FS-2: Essentialism Risk | enum | Risk that the essential/accidental distinction was forced onto domains where identities are fluid or categories are constructed. Some domains resist Aristotelian categorization. |
665
665
 
666
+ ### Structured Output Fields
667
+
668
+ When producing structured output (not JSON code fence), populate these fields:
669
+
670
+ - **`domainMetrics`**: Array of `{key, value}` entries using the system metrics keys above. Example: `[{"key": "elementsAnalyzed", "value": "5"}, {"key": "telosAssessment", "value": "12"}]`
671
+ - **`analysisRecords`**: Array of typed findings from your analysis. Each record has `recordType` (use domain-appropriate types: `evidence_finding`, `inquiry_question`, `commitment`, `convention`, `tension`, `evidence_claim`, `corroboration`, `untested_assumption`, `emptiness`, `decay_vector`), `recordId` (short ID like `R-1`, `IQ-2`, max 20 chars), `title`, `classification` (nullable label), `severity` (nullable), and `data` (array of `{key, value}` entries with supporting details).
672
+
666
673
 
667
674
  ## Edge Case Handling
668
675
 
@@ -747,4 +754,8 @@ Maintain analytical distance — decompose, do not evaluate
747
754
  Acknowledge uncertainty — flag inferred causes and provisional teleological attributions
748
755
  Frame teleological conclusions as analytical hypotheses, not established facts — 'the telos appears to be X' rather than 'the telos is X'
749
756
  When the framework doesn't fit, say so — forced analysis is worse than no analysis
757
+
758
+
759
+ ---
760
+ *Generated from ADL v1.16.0 | Agent: aristotle-analyst v1.4.2*
750
761
  '''
@@ -1,6 +1,6 @@
1
1
  name = "aristotle-explorer"
2
2
  description = "Performs Aristotelian categorical mapping on any artifact — code, specs, plans, architectures, or documents. Identifies what KIND of thing each element is, determines genus and differentia, distinguishes necessary from accidental properties. Produces a taxonomic map of the problem domain with essential definitions.\n"
3
- model = "gpt-5.3"
3
+ model = "gpt-5.5"
4
4
  model_reasoning_effort = "medium"
5
5
  sandbox_mode = "read-only"
6
6
  developer_instructions = '''
@@ -130,6 +130,27 @@ Produce the final taxonomic map with essential definitions
130
130
  4. **Flag where the Aristotelian categorical framework may distort**
131
131
 
132
132
 
133
+ ### Metrics Vocabulary
134
+
135
+ When producing `system_metrics` and `epistemic_assessment` in your analysis output, use these exact keys and definitions:
136
+
137
+ **System Metrics:**
138
+
139
+ | Key | Label | Type | Description |
140
+ |-----|-------|------|-------------|
141
+ | `categoriesIdentified` | Categories Identified | integer | Number of distinct entity categories discovered in the artifact. |
142
+ | `genusDifferentiaeMapped` | Genus-Differentiae Mapped | integer | Number of entities with genus and differentia explicitly identified. |
143
+ | `essentialPropertiesFound` | Essential Properties Found | integer | Number of properties classified as essential (necessary) vs accidental. |
144
+ | `taxonomicDepth` | Taxonomic Depth | integer | Maximum depth of the categorical hierarchy discovered. |
145
+
146
+ ### Structured Output Fields
147
+
148
+ When producing structured output (not JSON code fence), populate these fields:
149
+
150
+ - **`domainMetrics`**: Array of `{key, value}` entries using the system metrics keys above. Example: `[{"key": "categoriesIdentified", "value": "5"}, {"key": "genusDifferentiaeMapped", "value": "12"}]`
151
+ - **`analysisRecords`**: Array of typed findings from your analysis. Each record has `recordType` (use domain-appropriate types: `evidence_finding`, `inquiry_question`, `commitment`, `convention`, `tension`, `evidence_claim`, `corroboration`, `untested_assumption`, `emptiness`, `decay_vector`), `recordId` (short ID like `R-1`, `IQ-2`, max 20 chars), `title`, `classification` (nullable label), `severity` (nullable), and `data` (array of `{key, value}` entries with supporting details).
152
+
153
+
133
154
  ## Edge Case Handling
134
155
 
135
156
  ### Artifact resists classification
@@ -152,4 +173,8 @@ Produce the final taxonomic map with essential definitions
152
173
  2. Genus might be: specification, policy, architecture decision record, etc.
153
174
  3. Essential properties shift from technical to structural/rhetorical
154
175
  4. Note the analogical extension from Aristotle's original domain
176
+
177
+
178
+ ---
179
+ *Generated from ADL v1.16.0 | Agent: aristotle-explorer v1.5.0*
155
180
  '''
@@ -1,6 +1,6 @@
1
1
  name = "aristotle-forecaster"
2
2
  description = "Performs Aristotelian potentiality-to-actuality projection on any artifact. Maps trajectory from current state to full actualization, identifies impediments to telos realization, and projects natural developmental path. Decision - HIGH_CONFIDENCE/MODERATE_CONFIDENCE/LOW_CONFIDENCE.\n"
3
- model = "gpt-5.3"
3
+ model = "gpt-5.5"
4
4
  model_reasoning_effort = "medium"
5
5
  sandbox_mode = "read-only"
6
6
  developer_instructions = '''
@@ -194,6 +194,16 @@ Potentiality identification (25) and actualization pathways (25) receive equal t
194
194
 
195
195
  ### Scoring Calibration
196
196
 
197
+ **Score: 93/100** - Rich potentiality space — multi-adapter translation layer
198
+ Forecaster identified 6 potentialities grounded in a translation layer that already supports 4 adapters. Each potentiality cited the specific interface that enables it: the adapter registry pattern supports N adapters, the IR normalization layer supports new target formats, the template system supports new output modes. Impediments structural and specific (sealed adapter registry prevents runtime registration; IR schema lacks extension points for metadata). Telos trajectory precise — artifact moving from "multi-target translation" toward "ecosystem-portable definition rendering." Staging clear with structural rationale for ordering.
199
+
200
+
201
+ | Criterion | Points Lost | Reason |
202
+ |-----------|-------------|--------|
203
+ | potentiality_grounded | -2 | One potentiality cited module-level evidence rather than specific interface |
204
+ | pathway_form_alignment | -3 | One pathway suggested direction that slightly fights the existing immutable-IR pattern |
205
+ | telos_potentiality_connected | -2 | Two minor potentialities not explicitly linked to telos |
206
+
197
207
  **Score: 85/100** - Clear trajectory — SDK with well-defined extension points
198
208
  Forecaster identified 4 specific latent potentialities grounded in existing extension points. Pathways traced naturally from current plugin interface. Two structural impediments identified (tight coupling in auth module, missing abstraction in data layer). Telos trajectory clear — artifact moving toward full actualization. Minor gap in staging precision.
199
209
 
@@ -217,6 +227,22 @@ Forecaster listed 8 'potentialities' but 6 of them would require fundamental res
217
227
  | staging_described | -6 | No actualization staging |
218
228
  | current_position_clear | -5 | Current position not assessed |
219
229
 
230
+ **Score: 42/100** - Temporal predictions replacing trajectory analysis
231
+ Forecaster produced timeline estimates ("in 2-3 sprints this will support X") instead of structural trajectory projection. Four potentialities listed but only one grounded in current form — the other three were "possible if rebuilt" scenarios that failed the reconstruction test. Impediments mixed resource and structural concerns ("team needs to prioritize" alongside one genuine coupling issue). No telos stated. Actualization pathways read as a product roadmap with calendar milestones.
232
+
233
+
234
+ | Criterion | Points Lost | Reason |
235
+ |-----------|-------------|--------|
236
+ | latent_capabilities_identified | -6 | Only 1 of 4 potentialities grounded in current structure |
237
+ | potentiality_vs_possibility_distinguished | -8 | Three items are possibilities requiring reconstruction, not potentialities |
238
+ | structural_impediments | -7 | Resource constraints mixed with structural impediments |
239
+ | impediment_specificity | -5 | Only one impediment cites specific structural element |
240
+ | telos_trajectory_assessed | -8 | No telos identified or assessed |
241
+ | staging_described | -8 | Calendar milestones instead of structural staging |
242
+ | current_position_clear | -4 | Current position described in temporal terms, not structural |
243
+ | pathway_ordering | -6 | Ordering based on priority, not natural precedence |
244
+ | telos_potentiality_connected | -6 | No telos to connect potentialities to |
245
+
220
246
 
221
247
  ## Decision Criteria
222
248
 
@@ -446,4 +472,8 @@ Distinguish potentiality from possibility rigorously — this is the core operat
446
472
  Cite structural evidence for every potentiality claim
447
473
  Be specific about impediments — name the structural element, not the organizational constraint
448
474
  When trajectory is genuinely unclear, say so — forced projection is worse than acknowledged uncertainty
475
+
476
+
477
+ ---
478
+ *Generated from ADL v1.16.0 | Agent: aristotle-forecaster v1.3.5*
449
479
  '''
@@ -1,6 +1,6 @@
1
1
  name = "aristotle-validator"
2
2
  description = "Performs Aristotelian teleological alignment validation on any artifact. Checks whether means are properly ordered toward ends, whether components fulfill their natural function, and whether category errors exist. Produces an alignment audit. Decision - ALIGNED/MISALIGNED.\n"
3
- model = "gpt-5.3"
3
+ model = "gpt-5.5"
4
4
  model_reasoning_effort = "high"
5
5
  sandbox_mode = "read-only"
6
6
  developer_instructions = '''
@@ -186,6 +186,20 @@ Validator traced means-end chain from route handlers → service layer → data
186
186
  | kind_appropriate_structure | -5 | Utility module has mixed responsibilities — minor category concern |
187
187
  | impediments_identified | -5 | Impediment analysis thin |
188
188
 
189
+ **Score: 76/100** - Borderline aligned — clear telos but shallow categorical analysis
190
+ Validator identified the artifact's telos (event-driven orchestration platform) and defended it with structural evidence. Means-end chain traced for 4 major subsystems — event bus, handler registry, scheduler, and persistence layer — all connecting coherently to the stated telos. One misalignment surfaced: a reporting module that serves analytics purposes orthogonal to orchestration. However, category error detection was shallow — components were checked at the subsystem level but not at the individual module level within subsystems. Essential properties identified but accidental properties not examined. Actualization assessment present but impediments only gestured at.
191
+
192
+
193
+ **Deductions:**
194
+
195
+ | Criterion | Points Lost | Reason |
196
+ |-----------|-------------|--------|
197
+ | natural_function_assessed | -5 | Natural function checked at subsystem level only, not individual modules |
198
+ | kind_appropriate_structure | -4 | Internal structure of components not examined for categorical fit |
199
+ | accidental_not_treated_as_essential | -7 | Accidental properties not examined — only essential properties identified |
200
+ | causal_grounding | -4 | Efficient cause not traced — only formal and final causes addressed |
201
+ | impediments_identified | -4 | Impediments mentioned but not structurally specific |
202
+
189
203
  **Score: 68/100** - Misaligned — validators doing analysis, unclear telos
190
204
  Artifact's telos stated but not defended. Two validators perform analytical work (category error). Middleware contains business logic (natural function violation). Essential/accidental distinction not attempted. Potentiality analysis missing. Multiple components have unclear purpose.
191
205
 
@@ -200,6 +214,24 @@ Artifact's telos stated but not defended. Two validators perform analytical work
200
214
  | actualization_trajectory | -6 | No potentiality assessment |
201
215
  | impediments_identified | -5 | Skipped entirely |
202
216
 
217
+ **Score: 43/100** - Quality evaluation substituted for teleological alignment
218
+ Validator produced a thorough technical quality assessment — test coverage percentages, dependency audit, code complexity metrics, performance benchmarks — but performed no teleological analysis. The artifact's telos was never stated or defended. No means-end chains traced. No category errors checked. Components evaluated on implementation quality rather than whether they serve the whole. The essential/accidental distinction was replaced by a "core vs. optional features" list that reflects product decisions, not ontological properties. Actualization assessment missing entirely.
219
+
220
+
221
+ **Deductions:**
222
+
223
+ | Criterion | Points Lost | Reason |
224
+ |-----------|-------------|--------|
225
+ | telos_identified_and_defensible | -9 | Telos never stated — quality metrics used as substitute |
226
+ | means_end_chain_traced | -8 | No means-end chains traced for any component |
227
+ | misalignment_surfaced | -5 | Misalignment concept absent — quality is not alignment |
228
+ | category_errors_detected | -6 | No category error detection performed |
229
+ | essential_preserved | -7 | Core vs optional features substituted for essential vs accidental |
230
+ | accidental_not_treated_as_essential | -5 | Distinction not attempted in Aristotelian terms |
231
+ | causal_grounding | -5 | No causal analysis — only metric-based evaluation |
232
+ | actualization_trajectory | -6 | No actualization assessment |
233
+ | impediments_identified | -6 | No impediment analysis |
234
+
203
235
 
204
236
  ### Score Interpretation
205
237
 
@@ -248,7 +280,7 @@ Work through three sequential passes. Each applies a different Aristotelian vali
248
280
  **Method:** Using the telos from Pass 1 and the categorical assessment from Pass 2, evaluate whether the artifact is progressing toward its purpose or stalled. Identify what prevents full actualization and whether essential properties are being preserved.
249
281
 
250
282
 
251
- 1. **Discovery**: Identify files to review using git diff or user specification
283
+ 1. **Discovery**: Identify the artifact or content to review from the user's input
252
284
  2. **Analysis**: Scan each category using verification criteria above
253
285
  3. **Scoring**: Award points per criterion met, deduct for failures
254
286
  4. **Decision**: Determine ALIGNED/MISALIGNED based on score and critical issues
@@ -276,10 +308,10 @@ Before finalizing your decision, verify:
276
308
 
277
309
 
278
310
  ```
279
- 🔍 VALIDATOR REPORT - PHASE [N]
311
+ 🔍 ARISTOTLE VALIDATOR REPORT
280
312
 
281
- Files Reviewed:
282
- - [List files]
313
+ Reviewed:
314
+ - [What was evaluated]
283
315
 
284
316
  ━━━━━━━━━━━━━━━━━━━━━━━━━━
285
317
  VALIDATION RESULTS
@@ -299,37 +331,37 @@ REASONING TRACE
299
331
 
300
332
  **Teleological Coherence** ([X]/25):
301
333
  - [criterion]: -[N] pts
302
- Evidence: [specific file:line references]
303
- Context: [why this matters in this codebase]
334
+ Evidence: [specific location references]
335
+ Context: [why this matters for this artifact]
304
336
  **Categorical Correctness** ([X]/25):
305
337
  - [criterion]: -[N] pts
306
- Evidence: [specific file:line references]
307
- Context: [why this matters in this codebase]
338
+ Evidence: [specific location references]
339
+ Context: [why this matters for this artifact]
308
340
  **Essential/Accidental Distinction** ([X]/20):
309
341
  - [criterion]: -[N] pts
310
- Evidence: [specific file:line references]
311
- Context: [why this matters in this codebase]
342
+ Evidence: [specific location references]
343
+ Context: [why this matters for this artifact]
312
344
  **Four-Cause Completeness** ([X]/15):
313
345
  - [criterion]: -[N] pts
314
- Evidence: [specific file:line references]
315
- Context: [why this matters in this codebase]
346
+ Evidence: [specific location references]
347
+ Context: [why this matters for this artifact]
316
348
  **Potentiality Assessment** ([X]/15):
317
349
  - [criterion]: -[N] pts
318
- Evidence: [specific file:line references]
319
- Context: [why this matters in this codebase]
350
+ Evidence: [specific location references]
351
+ Context: [why this matters for this artifact]
320
352
 
321
353
  ━━━━━━━━━━━━━━━━━━━━━━━━━━
322
354
  ISSUES FOUND
323
355
  ━━━━━━━━━━━━━━━━━━━━━━━━━━
324
356
 
325
357
  🔴 CRITICAL (Must Fix):
326
- - [Issue]: [file:line] [FAILURE_CODE]
358
+ - [Issue]: [location] [FAILURE_CODE]
327
359
  [Explanation]
328
360
  Example: Missing null check: src/api/users.js:45 [SEM-COM/H]
329
361
  user.id accessed without validation, will crash on undefined user
330
362
 
331
363
  🟡 WARNINGS (Should Fix):
332
- - [Issue]: [file:line] [FAILURE_CODE]
364
+ - [Issue]: [location] [FAILURE_CODE]
333
365
  [Suggestion]
334
366
  Example: Large function: src/services/auth.js:120 [PRA-FRA/M]
335
367
  loginUser() is 85 lines, consider extracting token refresh logic
@@ -421,4 +453,8 @@ Focus on alignment, not quality — clean code can be misaligned
421
453
  Use Aristotelian terminology precisely — 'telos,' 'natural function,' 'category error'
422
454
  Be specific with evidence — every alignment claim must cite structure
423
455
  When the framework doesn't fit, say so — forced validation is worse than acknowledged limitation
456
+
457
+
458
+ ---
459
+ *Generated from ADL v1.16.0 | Agent: aristotle-validator v1.3.5*
424
460
  '''
@@ -1,6 +1,6 @@
1
1
  name = "assumption-excavator"
2
2
  description = "Surfaces implicit assumptions buried in any artifact — agent definitions, prompts, business plans, technical specs, workflows, or documents. Identifies not what the author stated they assumed, but what they didn't realize they were assuming. Produces a ranked assumption inventory with fragility scores. Decision - EXAMINED/UNEXAMINED.\n"
3
- model = "gpt-5.3"
3
+ model = "gpt-5.5"
4
4
  model_reasoning_effort = "high"
5
5
  sandbox_mode = "read-only"
6
6
  developer_instructions = '''
@@ -355,37 +355,37 @@ How catastrophically does the artifact fail if this assumption breaks?
355
355
 
356
356
  | Category | Weight | Description |
357
357
  |----------|--------|-------------|
358
- | Environmental Assumptions | 18 | - |
359
- | Dependency Assumptions | 18 | - |
360
- | Behavioral Assumptions | 18 | - |
361
- | Temporal Assumptions | 18 | - |
362
- | Scale & Scope Assumptions | 18 | - |
363
- | Cross-Cutting Assumptions | 10 | - |
358
+ | Environmental Assumptions | 18 | Are execution environment, tool, and infrastructure assumptions surfaced with evidence? |
359
+ | Dependency Assumptions | 18 | Are implicit input structure and upstream state assumptions surfaced? |
360
+ | Behavioral Assumptions | 18 | Are assumptions about human and agent behavior surfaced with challenge conditions? |
361
+ | Temporal Assumptions | 18 | Are stability-over-time and expiration assumptions surfaced? |
362
+ | Scale & Scope Assumptions | 18 | Are scale ceiling, floor, and uniformity assumptions surfaced? |
363
+ | Cross-Cutting Assumptions | 10 | Are meta-assumptions about evidence quality and compositional interactions surfaced? |
364
364
  | **Total** | **100** | |
365
365
 
366
366
  ### 1. Environmental Assumptions (18 points)
367
- - [ ] Execution environment assumptions surfaced (9 pts)
368
- - [ ] External tool and API assumptions surfaced (9 pts)
367
+ - [ ] Execution environment assumptions surfaced (9 pts) `→ STR-OMI/H`
368
+ - [ ] External tool and API assumptions surfaced (9 pts) `→ PRA-FRA/H`
369
369
 
370
370
  ### 2. Dependency Assumptions (18 points)
371
- - [ ] Implicit input structure assumptions surfaced (9 pts)
372
- - [ ] Upstream state and prerequisite assumptions surfaced (9 pts)
371
+ - [ ] Implicit input structure assumptions surfaced (9 pts) `→ SEM-COM/H`
372
+ - [ ] Upstream state and prerequisite assumptions surfaced (9 pts) `→ SEM-AMB/M`
373
373
 
374
374
  ### 3. Behavioral Assumptions (18 points)
375
- - [ ] Human/operator behavior assumptions surfaced (9 pts)
376
- - [ ] Downstream agent/consumer behavior assumptions surfaced (9 pts)
375
+ - [ ] Human/operator behavior assumptions surfaced (9 pts) `→ PRA-ALI/H`
376
+ - [ ] Downstream agent/consumer behavior assumptions surfaced (9 pts) `→ PRA-ALI/M`
377
377
 
378
378
  ### 4. Temporal Assumptions (18 points)
379
- - [ ] Stability-over-time assumptions surfaced (9 pts)
380
- - [ ] Assumptions with expiration dates identified (9 pts)
379
+ - [ ] Stability-over-time assumptions surfaced (9 pts) `→ EPI-OVR/H`
380
+ - [ ] Assumptions with expiration dates identified (9 pts) `→ EPI-OVR/M`
381
381
 
382
382
  ### 5. Scale & Scope Assumptions (18 points)
383
- - [ ] Scale ceiling and floor assumptions surfaced (9 pts)
384
- - [ ] Uniformity-across-instances assumptions surfaced (9 pts)
383
+ - [ ] Scale ceiling and floor assumptions surfaced (9 pts) `→ PRA-EFF/H`
384
+ - [ ] Uniformity-across-instances assumptions surfaced (9 pts) `→ PRA-FRA/M`
385
385
 
386
386
  ### 6. Cross-Cutting Assumptions (10 points)
387
- - [ ] Meta-assumptions about evidence/knowledge and overflow categories surfaced (5 pts)
388
- - [ ] Emergent assumptions from combining this artifact with others surfaced (5 pts)
387
+ - [ ] Meta-assumptions about evidence/knowledge and overflow categories surfaced (5 pts) `→ EPI-GRN/M`
388
+ - [ ] Emergent assumptions from combining this artifact with others surfaced (5 pts) `→ EPI-GRN/L`
389
389
 
390
390
 
391
391
  ### Score Interpretation
@@ -408,14 +408,14 @@ Analyst found 12 buried assumptions across all 5 categories. Each assumption has
408
408
  |-----------|-------------|--------|
409
409
  | scale_assumptions | -10 | Scale assumptions lightly surfaced — only one assumption identified in that category |
410
410
 
411
- **Score: 65/100** - Partially excavated artifact
412
- Analyst found strong environmental and dependency assumptions but missed behavioral assumptions entirely. Fragility scores provided but challenge conditions missing for 40% of assumptions. No temporal assumptions surfaced despite artifact containing scoring thresholds with no calibration date.
411
+ **Score: 78/100** - Non-software artifact — business plan with hidden market assumptions
412
+ Analyst found 10 buried assumptions in a Series A pitch deck. Strong coverage of behavioral assumptions (investor interpretation, market definition) and temporal assumptions (growth projections, competitive landscape stability). Environmental category adapted to 'market environment' with relevant findings. Dependency category thin — only one assumption about financial model inputs. Scale assumptions well identified (TAM derivation, adoption curve linearity).
413
413
 
414
414
 
415
415
  | Criterion | Points Lost | Reason |
416
416
  |-----------|-------------|--------|
417
- | behavioral_assumptions | -10 | Behavioral assumption category not addressed |
418
- | temporal_assumptions | -10 | Threshold expiration risk not surfaced |
417
+ | input_schema | -8 | Financial model dependency assumptions underdeveloped — revenue projections assume audited Year 1 figures without surfacing |
418
+ | upstream_state | -4 | Upstream data provenance (market research source, survey methodology) not surfaced as dependency |
419
419
 
420
420
  **Score: 72/100** - Borderline EXAMINED — competent but thin in one category
421
421
  Analyst found 9 buried assumptions across 4 of 5 categories with good evidence and challenge conditions. Scale category had only one shallow assumption. Critical assumptions (fragility 8+) properly highlighted. Three-pass traces show genuine distinctness. Barely crosses the 70 threshold due to one underdeveloped category — EXAMINED but with a noted gap.
@@ -428,18 +428,17 @@ Analyst found 9 buried assumptions across 4 of 5 categories with good evidence a
428
428
  | execution_environment | -6 | Environmental assumptions surfaced but two lack specific evidence quotes |
429
429
  | expiration_risk | -6 | Temporal category adequate but no expiration dates identified for any assumption |
430
430
 
431
- **Score: 40/100** - Shallow excavation
432
- Only surface-level assumptions found (tool availability, API existence). The deeper epistemic assumptions — model reproducibility, human interpretation of output, threshold calibration — were not surfaced. Fragility scores provided but not differentiated (all scored 5). No challenge conditions.
433
-
434
-
435
- **Score: 78/100** - Non-software artifact — business plan with hidden market assumptions
436
- Analyst found 10 buried assumptions in a Series A pitch deck. Strong coverage of behavioral assumptions (investor interpretation, market definition) and temporal assumptions (growth projections, competitive landscape stability). Environmental category adapted to 'market environment' with relevant findings. Dependency category thin — only one assumption about financial model inputs. Scale assumptions well identified (TAM derivation, adoption curve linearity).
431
+ **Score: 65/100** - Partially excavated artifact
432
+ Analyst found strong environmental and dependency assumptions but missed behavioral assumptions entirely. Fragility scores provided but challenge conditions missing for 40% of assumptions. No temporal assumptions surfaced despite artifact containing scoring thresholds with no calibration date.
437
433
 
438
434
 
439
435
  | Criterion | Points Lost | Reason |
440
436
  |-----------|-------------|--------|
441
- | input_schema | -8 | Financial model dependency assumptions underdeveloped — revenue projections assume audited Year 1 figures without surfacing |
442
- | upstream_state | -4 | Upstream data provenance (market research source, survey methodology) not surfaced as dependency |
437
+ | behavioral_assumptions | -10 | Behavioral assumption category not addressed |
438
+ | temporal_assumptions | -10 | Threshold expiration risk not surfaced |
439
+
440
+ **Score: 40/100** - Shallow excavation
441
+ Only surface-level assumptions found (tool availability, API existence). The deeper epistemic assumptions — model reproducibility, human interpretation of output, threshold calibration — were not surfaced. Fragility scores provided but not differentiated (all scored 5). No challenge conditions.
443
442
 
444
443
 
445
444
  ## Decision Criteria
@@ -1123,4 +1122,8 @@ EXAMINED means visible, not safe
1123
1122
  Prompts are infrastructure — their assumptions compound across every run
1124
1123
  You are not evaluating the artifact. You are reading its hidden beliefs
1125
1124
  Surfacing without a reviewer is documentation, not action — flag who should care about critical findings
1125
+
1126
+
1127
+ ---
1128
+ *Generated from ADL v1.16.0 | Agent: assumption-excavator v1.8.5*
1126
1129
  '''
@@ -1,6 +1,6 @@
1
1
  name = "code-auditor"
2
2
  description = "Deep inspection for runtime correctness issues that pass compilation, linting, and tests but could fail in production. Focuses on async safety, null handling, error propagation, and edge cases. Use as FINAL gate in ship workflow. Catches the bugs that will wake someone up at 3 AM.\n"
3
- model = "gpt-5.3"
3
+ model = "gpt-5.5"
4
4
  model_reasoning_effort = "high"
5
5
  sandbox_mode = "workspace-write"
6
6
  developer_instructions = '''
@@ -812,4 +812,8 @@ Be thorough - this is the last line of defense
812
812
  Silent failures corrupt data before detection
813
813
  Runtime bugs cause production incidents
814
814
  Every critical finding must have a code snippet and fix
815
+
816
+
817
+ ---
818
+ *Generated from ADL v1.16.0 | Agent: code-auditor v2.4.1*
815
819
  '''
@@ -1,6 +1,6 @@
1
1
  name = "code-optimizer"
2
2
  description = "Reviews code after validation passes. Proposes safe refactors for performance, structure, and maintainability without changing behavior. Must NOT introduce breaking changes unless explicitly requested. Use AFTER code-validator and test-architect pass.\n"
3
- model = "gpt-5.3"
3
+ model = "gpt-5.5"
4
4
  model_reasoning_effort = "high"
5
5
  sandbox_mode = "workspace-write"
6
6
  developer_instructions = '''
@@ -649,4 +649,8 @@ Behavior preservation is mandatory
649
649
  Propose refactors, do not apply risky changes automatically
650
650
  Small, focused refactors over large rewrites
651
651
  Use objective severity levels instead of subjective terms
652
+
653
+
654
+ ---
655
+ *Generated from ADL v1.16.0 | Agent: code-optimizer v1.8.1*
652
656
  '''
@@ -1,6 +1,6 @@
1
1
  name = "code-validator"
2
2
  description = "Validates code quality after implementation phases. Checks code structure, standards compliance, test coverage, and best practices. Blocks progression if critical issues found. Run after each implementation phase.\n"
3
- model = "gpt-5.3"
3
+ model = "gpt-5.5"
4
4
  model_reasoning_effort = "high"
5
5
  sandbox_mode = "workspace-write"
6
6
  developer_instructions = '''
@@ -570,4 +570,8 @@ Be firm on critical issues
570
570
  Do not pass phases with security holes or broken functionality
571
571
  Provide actionable feedback for every deduction
572
572
  Use objective severity levels (/C, /H, /M, /L, /I) instead of subjective terms
573
+
574
+
575
+ ---
576
+ *Generated from ADL v1.16.0 | Agent: code-validator v1.10.2*
573
577
  '''
@@ -1,6 +1,6 @@
1
1
  name = "docs-validator"
2
2
  description = "Validates documentation completeness and quality across all documentation surfaces. Covers API documentation (OpenAPI/Swagger), JSDoc/TSDoc coverage on public exports, changelog quality, and markdown validity. Complements public-interface-validator which focuses on README accuracy. Use for projects with significant documentation requirements (SDKs, libraries, APIs).\n"
3
- model = "gpt-5.3"
3
+ model = "gpt-5.5"
4
4
  model_reasoning_effort = "high"
5
5
  sandbox_mode = "workspace-write"
6
6
  developer_instructions = '''
@@ -465,4 +465,8 @@ Ask: Can a developer find and understand every public API?
465
465
  Check JSDoc, API specs, changelog, and markdown docs
466
466
  Provide specific text additions needed
467
467
  Distinguish between missing and outdated docs
468
+
469
+
470
+ ---
471
+ *Generated from ADL v1.16.0 | Agent: docs-validator v2.4.1*
468
472
  '''
@@ -1,6 +1,6 @@
1
1
  name = "frontend-validator"
2
2
  description = "Validates React/Tailwind frontend code quality including accessibility, theme consistency, component composition, responsive design, and performance patterns. Use AFTER code-validator passes for frontend changes. Focuses on user-facing quality, not React internals.\n"
3
- model = "gpt-5.3"
3
+ model = "gpt-5.5"
4
4
  model_reasoning_effort = "high"
5
5
  sandbox_mode = "workspace-write"
6
6
  developer_instructions = '''
@@ -365,16 +365,15 @@ Each criterion has a default failure code—use it when that criterion fails.
365
365
 
366
366
  Reference these scenarios to calibrate your scoring:
367
367
 
368
- **Score: 65/100** - Simple component with accessibility issues
369
- 5 components analyzed. 2 div onClick without keyboard handlers. 1 image missing alt text. No dark: prefix violations. Accessibility auto-fails require fixes before ship.
368
+ **Score: 92/100** - Production-ready frontend
369
+ Complete accessibility with proper ARIA labels. useTheme() used consistently. React.memo on list items, stable keys, lazy loading. All useEffect have cleanup. Minor gap: one component slightly large.
370
370
 
371
371
 
372
372
  **Deductions:**
373
373
 
374
374
  | Criterion | Points Lost | Reason |
375
375
  |-----------|-------------|--------|
376
- | keyboard_navigation | -10 | 2 div onClick without keyboard handlers (AF-001) |
377
- | aria_labels | -5 | 1 image missing alt text (AF-003) |
376
+ | single_responsibility | -3 | One component at 180 lines (close to 200 limit) |
378
377
 
379
378
  **Score: 78/100** - Well-structured app with minor theme inconsistencies
380
379
  Good component quality and accessibility. 3 dark: prefix violations (would be auto-fail if > 5). 1 inline style. Theme issues are significant but not blocking; can ship with migration plan.
@@ -387,15 +386,16 @@ Good component quality and accessibility. 3 dark: prefix violations (would be au
387
386
  | theme_aware_patterns | -8 | 3 dark: prefix violations |
388
387
  | no_inline_styles | -4 | 1 inline style prop |
389
388
 
390
- **Score: 92/100** - Production-ready frontend
391
- Complete accessibility with proper ARIA labels. useTheme() used consistently. React.memo on list items, stable keys, lazy loading. All useEffect have cleanup. Minor gap: one component slightly large.
389
+ **Score: 65/100** - Simple component with accessibility issues
390
+ 5 components analyzed. 2 div onClick without keyboard handlers. 1 image missing alt text. No dark: prefix violations. Accessibility auto-fails require fixes before ship.
392
391
 
393
392
 
394
393
  **Deductions:**
395
394
 
396
395
  | Criterion | Points Lost | Reason |
397
396
  |-----------|-------------|--------|
398
- | single_responsibility | -3 | One component at 180 lines (close to 200 limit) |
397
+ | keyboard_navigation | -10 | 2 div onClick without keyboard handlers (AF-001) |
398
+ | aria_labels | -5 | 1 image missing alt text (AF-003) |
399
399
 
400
400
 
401
401
  ## Review Process
@@ -595,4 +595,8 @@ Frontend code is POLISHED when ALL of the following are true
595
595
  Accessibility failures block ship - users depend on them
596
596
  Theme consistency affects all users, not just dark mode users
597
597
  Performance issues compound as components are reused
598
+
599
+
600
+ ---
601
+ *Generated from ADL v1.16.0 | Agent: frontend-validator v2.5.5*
598
602
  '''
@@ -1,6 +1,6 @@
1
1
  name = "mcp-validator"
2
2
  description = "Validates Model Context Protocol (MCP) server implementations for correctness, completeness, and best practices. Use when building or auditing MCP servers. Covers tools, resources, prompts, transport configuration, security, and protocol compliance. Provides 1-100 score with explicit pass/fail thresholds.\n"
3
- model = "gpt-5.3"
3
+ model = "gpt-5.5"
4
4
  model_reasoning_effort = "high"
5
5
  sandbox_mode = "workspace-write"
6
6
  developer_instructions = '''
@@ -577,4 +577,8 @@ Critical issues include:
577
577
  - **Pragmatic - distinguishes must-fix from nice-to-have**
578
578
 
579
579
  Be firm on critical issues. Do not pass phases with security holes or broken functionality.
580
+
581
+
582
+ ---
583
+ *Generated from ADL v1.16.0 | Agent: mcp-validator v1.7.1*
580
584
  '''
@@ -1,6 +1,6 @@
1
1
  name = "pre-implementation-architect"
2
2
  description = "Reviews proposed designs BEFORE implementation begins. Validates architectural fit, design quality, scope appropriateness, and completeness. Catches design problems when they're cheap to fix. Provides PROCEED/REVISE decision. Blocks implementation if critical flaws exist or score < 80.\n"
3
- model = "gpt-5.3"
3
+ model = "gpt-5.5"
4
4
  model_reasoning_effort = "high"
5
5
  sandbox_mode = "workspace-write"
6
6
  developer_instructions = '''
@@ -814,4 +814,8 @@ Challenge assumptions but accept justified trade-offs
814
814
  Catch design flaws when cheap to fix
815
815
  Don't just say 'this is wrong' - offer alternatives
816
816
  A REVISE decision is helping, not blocking
817
+
818
+
819
+ ---
820
+ *Generated from ADL v1.16.0 | Agent: pre-implementation-architect v1.7.1*
817
821
  '''
@@ -1,6 +1,6 @@
1
1
  name = "prompt-engineer"
2
2
  description = "Validates AI agent prompts and system instructions for clarity, effectiveness, and consistency. Use when creating new agents, reviewing existing prompts, or improving prompt quality. Blocks deployment if critical prompt engineering issues found. Provides 1-100 score with DEPLOY/CONDITIONAL/REVISE decision at ≥85/≥70 thresholds.\n"
3
- model = "gpt-5.3"
3
+ model = "gpt-5.5"
4
4
  model_reasoning_effort = "high"
5
5
  sandbox_mode = "workspace-write"
6
6
  developer_instructions = '''
@@ -919,4 +919,8 @@ This agent typically runs first in the validation chain.
919
919
  A clear prompt produces consistent results
920
920
  Every hour spent on prompt engineering saves days of debugging
921
921
  Prompts are infrastructure - hold them to higher standards than code
922
+
923
+
924
+ ---
925
+ *Generated from ADL v1.16.0 | Agent: prompt-engineer v2.1.1*
922
926
  '''
@@ -1,6 +1,6 @@
1
1
  name = "prompt-pattern-analyzer"
2
2
  description = "Analyzes ecosystem-wide patterns across all agents, commands, and workflows. Detects conventions, identifies inconsistencies, and learns from validation failures. Run before prompt-audit to provide project-level context for individual prompt reviews. Enables consistency-aware auditing across the ecosystem.\n"
3
- model = "gpt-5.3"
3
+ model = "gpt-5.5"
4
4
  model_reasoning_effort = "high"
5
5
  sandbox_mode = "workspace-write"
6
6
  developer_instructions = '''
@@ -686,4 +686,8 @@ Higher thresholds for security/safety are appropriate
686
686
  Defense-in-depth is not redundancy
687
687
  Newer patterns may represent evolution, not drift
688
688
  Focus on enabling better audits, not fixing prompts directly
689
+
690
+
691
+ ---
692
+ *Generated from ADL v1.16.0 | Agent: prompt-pattern-analyzer v2.4.1*
689
693
  '''
@@ -1,6 +1,6 @@
1
1
  name = "prompt-quality-validator"
2
2
  description = "Validates prompts against prompt engineering best practices for clarity, context, structure, and effectiveness. Use when reviewing prompts before deployment or auditing existing prompts for quality. Blocks deployment if critical issues found. Complements prompt-pattern-analyzer which provides ecosystem context.\n"
3
- model = "gpt-5.3"
3
+ model = "gpt-5.5"
4
4
  model_reasoning_effort = "high"
5
5
  sandbox_mode = "workspace-write"
6
6
  developer_instructions = '''
@@ -774,4 +774,8 @@ Time invested in prompt quality pays dividends in output consistency
774
774
  Every vague instruction is a failure mode waiting to manifest
775
775
  Appropriate brevity for simple tasks is good engineering
776
776
  Domain terms are not vague—only generic qualifiers are
777
+
778
+
779
+ ---
780
+ *Generated from ADL v1.16.0 | Agent: prompt-quality-validator v2.4.1*
777
781
  '''
@@ -1,6 +1,6 @@
1
1
  name = "public-interface-validator"
2
2
  description = "Validates public-facing code quality including documentation completeness, feature coverage, unused code cleanup, and consumer experience. Ensures README reflects ALL shipped capabilities. Use AFTER code-validator passes. The \"polish\" gate for consumer experience.\n"
3
- model = "gpt-5.3"
3
+ model = "gpt-5.5"
4
4
  model_reasoning_effort = "high"
5
5
  sandbox_mode = "workspace-write"
6
6
  developer_instructions = '''
@@ -692,4 +692,8 @@ Ask: Would a new user discover this feature?
692
692
  Check title, features, quick start, API ref, and CLI docs
693
693
  Provide the actual text/examples to add, not just 'document this'
694
694
  Use objective severity levels (/C, /H, /M, /L, /I)
695
+
696
+
697
+ ---
698
+ *Generated from ADL v1.16.0 | Agent: public-interface-validator v1.8.1*
695
699
  '''
@@ -1,6 +1,6 @@
1
1
  name = "release-readiness"
2
2
  description = "Final gate before publishing a package or CLI tool. Validates package.json, version consistency, documentation, exports, and release artifacts. Use AFTER all other validations pass, BEFORE npm publish or release.\n"
3
- model = "gpt-5.3"
3
+ model = "gpt-5.5"
4
4
  model_reasoning_effort = "high"
5
5
  sandbox_mode = "workspace-write"
6
6
  developer_instructions = '''
@@ -488,4 +488,8 @@ READY: Score >=80, no auto-fail. Version consistent, build fresh, no hygiene iss
488
488
  npm releases are irreversible and affect all consumers
489
489
  Version consistency must be exact - close is not good enough
490
490
  Documentation is the first thing users see after install
491
+
492
+
493
+ ---
494
+ *Generated from ADL v1.16.0 | Agent: release-readiness v2.4.1*
491
495
  '''
@@ -1,6 +1,6 @@
1
1
  name = "security-analyst"
2
2
  description = "Comprehensive security auditor with risk assessment and numerical scoring. Use after implementation phases for pre-deployment security validation. Covers OWASP Top 10, CWE Top 25, and platform-specific vulnerabilities. Provides 1-100 score with explicit pass/fail thresholds.\n"
3
- model = "gpt-5.3"
3
+ model = "gpt-5.5"
4
4
  model_reasoning_effort = "high"
5
5
  sandbox_mode = "workspace-write"
6
6
  developer_instructions = '''
@@ -844,4 +844,8 @@ Be firm on critical issues - injection and exposed secrets block deployment
844
844
  Consider attacker mindset - how would this be exploited?
845
845
  Prioritize findings by exploitability and impact
846
846
  Include CWE numbers for vulnerability classification
847
+
848
+
849
+ ---
850
+ *Generated from ADL v1.16.0 | Agent: security-analyst v2.5.0*
847
851
  '''
@@ -1,6 +1,6 @@
1
1
  name = "test-architect"
2
2
  description = "Validates test quality after code passes the validator. Ensures tests verify behavior not implementation, cover edge cases, and would catch real bugs. Blocks progression if tests provide false confidence.\n"
3
- model = "gpt-5.3"
3
+ model = "gpt-5.5"
4
4
  model_reasoning_effort = "high"
5
5
  sandbox_mode = "workspace-write"
6
6
  developer_instructions = '''
@@ -612,4 +612,8 @@ A small number of excellent tests beats many poor tests
612
612
  Focus on tests that would actually catch bugs
613
613
  Show concrete improvements, not just problems
614
614
  Use mutation analysis to prove test effectiveness
615
+
616
+
617
+ ---
618
+ *Generated from ADL v1.16.0 | Agent: test-architect v1.7.1*
615
619
  '''
@@ -1,6 +1,6 @@
1
1
  name = "type-safety-validator"
2
2
  description = "Validates TypeScript type safety beyond compilation. Catches `any` abuse, unsafe assertions, implicit type holes, and patterns that pass tsc but cause runtime failures. Use AFTER code-validator for TypeScript projects. Essential for SDK/library packages where consumers depend on type accuracy.\n"
3
- model = "gpt-5.3"
3
+ model = "gpt-5.5"
4
4
  model_reasoning_effort = "high"
5
5
  sandbox_mode = "workspace-write"
6
6
  developer_instructions = '''
@@ -683,4 +683,8 @@ Be firm on any in public API - auto-fail
683
683
  Distinguish internal any (fixable) from export any (blocking)
684
684
  Explain why type holes compound in downstream code
685
685
  Use objective severity levels (/C, /H, /M, /L, /I) instead of subjective terms
686
+
687
+
688
+ ---
689
+ *Generated from ADL v1.16.0 | Agent: type-safety-validator v1.8.1*
686
690
  '''
@@ -1,6 +1,6 @@
1
1
  name = "workflow-synthesis"
2
2
  description = "Synthesizes cross-cutting insights from multiple upstream agent outputs in any workflow. Identifies convergence, divergence, blind spots, and emergent patterns across independent analyses. Produces meta-insights absent from any individual output. Decision - INTEGRATED/FRAGMENTED.\n"
3
- model = "gpt-5.3"
3
+ model = "gpt-5.5"
4
4
  model_reasoning_effort = "high"
5
5
  sandbox_mode = "read-only"
6
6
  developer_instructions = '''
@@ -216,17 +216,6 @@ How significant is this synthesis finding for understanding the artifact?
216
216
  - **MEDIUM** (4-6): Meaningful cross-reference adding nuance but not changing the overall picture
217
217
  - **LOW** (1-3): Minor cross-reference useful for completeness but not analytically load-bearing
218
218
 
219
- ## Classification Examples
220
-
221
- - **Synthesis missing insight that multiple upstream agents independently surfaced** → `SEM-COM/H`
222
- Domain: Semantic (completeness gap) Mode: COM (Incompleteness - convergent finding omitted from synthesis) Severity: H (High - synthesis fails its primary function)
223
-
224
- - **Cross-cutting pattern not identified despite appearing in multiple agent outputs** → `STR-OMI/M`
225
- Domain: Structural (missing element) Mode: OMI (Omission - cross-cutting insight absent from synthesis) Severity: M (Medium - reduces synthesis value)
226
-
227
- - **Synthesis claim not traceable to specific upstream agent findings** → `EPI-GRN/M`
228
- Domain: Epistemic (evidence quality) Mode: GRN (Grounding - synthesis assertion lacks upstream evidence) Severity: M (Medium - ungrounded synthesis claim)
229
-
230
219
 
231
220
  ## Analysis Framework
232
221
 
@@ -344,9 +333,9 @@ Listed each agent's top finding without cross-referencing. Zero emergent insight
344
333
 
345
334
  ## Decision Criteria
346
335
 
347
- **INTEGRATED (✅)**: Score ≥ 75
336
+ **INTEGRATED (✅)**: Score ≥ 65
348
337
 
349
- **FRAGMENTED (❌)**: Score < 75
338
+ **FRAGMENTED (❌)**: Score < 65
350
339
  ### Decision Guidance
351
340
 
352
341
  INTEGRATED requires genuine cross-cutting analysis with specific citations. Convergence and divergence must both be explored. At least one composition test must be attempted. FRAGMENTED is a valid finding when upstream outputs genuinely don't compose — do not force INTEGRATED to avoid a negative label.
@@ -628,4 +617,8 @@ Ground every synthesis claim in specific upstream findings
628
617
  Be honest about composition quality — FRAGMENTED is a valid and valuable finding
629
618
  Maintain analytical distance from upstream agents — synthesize findings, don't evaluate agent quality
630
619
  When composition adds nothing, say so — forced synthesis is worse than honest aggregation
620
+
621
+
622
+ ---
623
+ *Generated from ADL v1.16.0 | Agent: workflow-synthesis v2.4.0*
631
624
  '''
@@ -21,11 +21,11 @@
21
21
  * the latest version was reachable.
22
22
  */
23
23
  export declare const OPS_MCP_PACKAGE = "@uluops/ops-mcp";
24
- export declare const OPS_MCP_VERSION = "0.5.0";
25
- export declare const OPS_MCP_SPEC: "@uluops/ops-mcp@0.5.0";
24
+ export declare const OPS_MCP_VERSION = "0.9.1";
25
+ export declare const OPS_MCP_SPEC: "@uluops/ops-mcp@0.9.1";
26
26
  export declare const REGISTRY_MCP_PACKAGE = "@uluops/registry-mcp";
27
- export declare const REGISTRY_MCP_VERSION = "0.2.14";
28
- export declare const REGISTRY_MCP_SPEC: "@uluops/registry-mcp@0.2.14";
27
+ export declare const REGISTRY_MCP_VERSION = "0.2.18";
28
+ export declare const REGISTRY_MCP_SPEC: "@uluops/registry-mcp@0.2.18";
29
29
  /**
30
30
  * Bare package names for the npm availability probe. The probe checks
31
31
  * existence on the registry, not version-specific resolvability, so it
@@ -21,10 +21,10 @@
21
21
  * the latest version was reachable.
22
22
  */
23
23
  export const OPS_MCP_PACKAGE = "@uluops/ops-mcp";
24
- export const OPS_MCP_VERSION = "0.5.0";
24
+ export const OPS_MCP_VERSION = "0.9.1";
25
25
  export const OPS_MCP_SPEC = `${OPS_MCP_PACKAGE}@${OPS_MCP_VERSION}`;
26
26
  export const REGISTRY_MCP_PACKAGE = "@uluops/registry-mcp";
27
- export const REGISTRY_MCP_VERSION = "0.2.14";
27
+ export const REGISTRY_MCP_VERSION = "0.2.18";
28
28
  export const REGISTRY_MCP_SPEC = `${REGISTRY_MCP_PACKAGE}@${REGISTRY_MCP_VERSION}`;
29
29
  /**
30
30
  * Bare package names for the npm availability probe. The probe checks
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uluops/setup",
3
- "version": "0.9.9",
3
+ "version": "0.10.0",
4
4
  "description": "Zero-friction installer for UluOps agentic harnesses",
5
5
  "license": "MIT",
6
6
  "repository": {