specrails-desktop 2.44.2 → 2.46.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -0
- package/client/dist/assets/{ActivityFeedPage-CL1BUXoj.js → ActivityFeedPage-CsHQPpvl.js} +1 -1
- package/client/dist/assets/AgentBrowserCapture-BVcCK7x0.js +1 -0
- package/client/dist/assets/AgentModeAnalyticsPane-C5F3m29G.js +2 -0
- package/client/dist/assets/AgentModeCodePane-C5QuKsu0.js +2 -0
- package/client/dist/assets/AgentModeJobsPane-rnD-EPMd.js +1 -0
- package/client/dist/assets/{AgentsPage-DbLezyI6.js → AgentsPage-DiyJWZcU.js} +3 -3
- package/client/dist/assets/{AnalyticsPage-DTsg-N5z.js → AnalyticsPage-Dkmb0dnd.js} +1 -1
- package/client/dist/assets/CodePage-C0dH99-t.js +3 -0
- package/client/dist/assets/{DesktopAnalyticsPage-Y0Z1ryQr.js → DesktopAnalyticsPage-DuTjr2n8.js} +1 -1
- package/client/dist/assets/{DocsDialog-KeCOxvGH.js → DocsDialog-BVZJZHVR.js} +1 -1
- package/client/dist/assets/{DocsPage-Bt1PgYPO.js → DocsPage-D6KHTv8V.js} +1 -1
- package/client/dist/assets/{ExportDropdown-DczgU4at.js → ExportDropdown-pWZGYCv3.js} +1 -1
- package/client/dist/assets/InteractiveJobComposer-Ck3PQxjN.js +19 -0
- package/client/dist/assets/JobDetailModal-DkoMH3ie.js +1 -0
- package/client/dist/assets/JobDetailPage-CBVivcl2.js +1 -0
- package/client/dist/assets/JobsPage-CzxEeca9.js +1 -0
- package/client/dist/assets/{LoopBuilderPage-CqC4hbXb.js → LoopBuilderPage-CtmOE4-b.js} +2 -2
- package/client/dist/assets/{LoopPreviewModal-T0zuFv6y.js → LoopPreviewModal-BYeLjQbr.js} +1 -1
- package/client/dist/assets/{LoopsPage-DrQGcZv2.js → LoopsPage-6iW7g25k.js} +1 -1
- package/client/dist/assets/{MinimizedChatsContext-C843Y13h.js → MinimizedChatsContext-B6wVevPq.js} +1 -1
- package/client/dist/assets/PluginsPage-Bf1z3WPF.js +2 -0
- package/client/dist/assets/ProjectSettingsDialog-DsCRIc4v.js +1 -0
- package/client/dist/assets/{RepositoryDeliveries-bgIHE8ep.js → RepositoryDeliveries-BQm_3TO8.js} +1 -1
- package/client/dist/assets/{RepositoryScopeSelector-CZ_U_R3U.js → RepositoryScopeSelector-DnAFtjTz.js} +1 -1
- package/client/dist/assets/{ReviewPacketPage-DbKO6pT8.js → ReviewPacketPage-CaXAwtR7.js} +1 -1
- package/client/dist/assets/TemplatePreviewModal-9aRFpJ_J.js +1 -0
- package/client/dist/assets/{TicketDetailModalContext-BbRA6cb7.js → TicketDetailModalContext-B1X5glzV.js} +1 -1
- package/client/dist/assets/{Trans-mjFZpHw1.js → Trans-BwLgX6n7.js} +1 -1
- package/client/dist/assets/agentRuntime-BUagD2I_.js +1 -0
- package/client/dist/assets/agentRuntime-BkL70WtX.js +1 -0
- package/client/dist/assets/agentRuntime-CWXVqgqE.js +1 -0
- package/client/dist/assets/agentRuntime-CvNBdUAe.js +1 -0
- package/client/dist/assets/agentRuntime-D4RmeuNY.js +1 -0
- package/client/dist/assets/agentRuntime-DTidBWKg.js +1 -0
- package/client/dist/assets/agentRuntime-DhKYfyaU.js +1 -0
- package/client/dist/assets/agentRuntime-LL8_GLx9.js +1 -0
- package/client/dist/assets/format-duration-NJVPOPBG.js +1 -0
- package/client/dist/assets/{formatDistanceToNow-BIL-HxGy.js → formatDistanceToNow-CeZv93iR.js} +1 -1
- package/client/dist/assets/{getTimezoneOffsetInMilliseconds-Cs9z2sdO.js → getTimezoneOffsetInMilliseconds-DxAkjvJb.js} +1 -1
- package/client/dist/assets/index-Cfh9aaep.css +2 -0
- package/client/dist/assets/index-DYoyDWim.js +76 -0
- package/client/dist/assets/{jira-api-B2BBZQTx.js → jira-api-ChjVGJx_.js} +1 -1
- package/client/dist/assets/narration-4OSg0b5-.js +1 -0
- package/client/dist/assets/narration-BPMOQPFe.js +1 -0
- package/client/dist/assets/narration-BSldQfjd.js +1 -0
- package/client/dist/assets/narration-CvQdWdR5.js +1 -0
- package/client/dist/assets/narration-D-e9ssxA.js +1 -0
- package/client/dist/assets/narration-DKLjXIFI.js +1 -0
- package/client/dist/assets/narration-DnrKC3mF.js +1 -0
- package/client/dist/assets/narration-Dypjxeel.js +1 -0
- package/client/dist/assets/{project-repositories-xSqimlaj.js → project-repositories-BvBj9XH9.js} +1 -1
- package/client/dist/assets/{provider-capabilities-B-wOHfLP.js → provider-capabilities-CXipwrXJ.js} +1 -1
- package/client/dist/assets/{settings-2Yx_9yJM.js → settings-B3rWEwOx.js} +1 -1
- package/client/dist/assets/{settings-BLOsrXEw.js → settings-BcVoiMyw.js} +1 -1
- package/client/dist/assets/settings-BlYvaQ7w.js +1 -0
- package/client/dist/assets/{settings-DPjNiUsS.js → settings-CD5uEBiL.js} +1 -1
- package/client/dist/assets/{settings-CpKwTew1.js → settings-CyhS0DLJ.js} +1 -1
- package/client/dist/assets/{settings-FIqwhTup.js → settings-DCMFUGDs.js} +1 -1
- package/client/dist/assets/{settings-C6s9PtRD.js → settings-DEgcA0jd.js} +1 -1
- package/client/dist/assets/{settings-NSWQjY5B.js → settings-DmbYKyMJ.js} +1 -1
- package/client/dist/assets/{spending-DXOcv1tn.js → spending-DNM6NAoA.js} +1 -1
- package/client/dist/assets/{useDesktop-gVGoWWiY.js → useDesktop-mxHVQOdx.js} +1 -1
- package/client/dist/assets/useSharedWebSocket-DnxHj8jP.js +2 -0
- package/client/dist/index.html +19 -17
- package/docs/README.md +1 -0
- package/docs/internals/README.md +2 -0
- package/docs/internals/agent-runtime-framework-evaluation.md +208 -0
- package/docs/internals/programmatic-agent-runtime-validation.md +72 -0
- package/docs/internals/programmatic-agent-runtime.md +140 -0
- package/package.json +1 -1
- package/server/dist/agent-mcp-config.js +5 -0
- package/server/dist/agent-operator-prompt.js +2 -2
- package/server/dist/agent-runtime-accounting.js +73 -0
- package/server/dist/agent-runtime-bridge.js +231 -0
- package/server/dist/agent-runtime-controls-router.js +86 -0
- package/server/dist/agent-runtime-controls.js +341 -0
- package/server/dist/agent-runtime-loader.js +91 -0
- package/server/dist/agent-runtime-metrics.js +46 -0
- package/server/dist/agent-runtime-paths.js +12 -0
- package/server/dist/agent-runtime-repositories.js +37 -0
- package/server/dist/agent-runtime-settings-router.js +61 -0
- package/server/dist/agent-runtime-settings.js +239 -0
- package/server/dist/agent-runtime-settlement.js +133 -0
- package/server/dist/agent-runtime-verification-suggestions.js +119 -0
- package/server/dist/contract-refine-runner.js +7 -7
- package/server/dist/core-completion.js +8 -5
- package/server/dist/core-execution.js +17 -8
- package/server/dist/core-node-runtime.js +12 -0
- package/server/dist/desktop-router.js +26 -0
- package/server/dist/job-listing.js +7 -3
- package/server/dist/loop-command-catalog.js +2 -2
- package/server/dist/loop-executors.js +44 -2
- package/server/dist/loop-factory.js +1 -1
- package/server/dist/loop-run-manager.js +62 -28
- package/server/dist/mcp/tools/specs.js +7 -4
- package/server/dist/multi-repo-execution.js +1 -0
- package/server/dist/project-registry.js +9 -0
- package/server/dist/project-router-jobs.js +8 -2
- package/server/dist/project-router-settings.js +4 -0
- package/server/dist/project-router-tickets.js +4 -4
- package/server/dist/providers/runtime.js +6 -0
- package/server/dist/rail-isolated-launch.js +18 -0
- package/server/dist/rails-router.js +4 -1
- package/server/dist/runtime-role-prompts-router.js +25 -0
- package/server/dist/schemas/agent-runtime.schema.json +60 -0
- package/server/dist/worktree-overlay.js +32 -1
- package/client/dist/assets/AgentBrowserCapture-BQem50z4.js +0 -1
- package/client/dist/assets/AgentModeAnalyticsPane-BC_xbT_l.js +0 -2
- package/client/dist/assets/AgentModeCodePane-DK-q3K6Y.js +0 -2
- package/client/dist/assets/AgentModeJobsPane-C7ELTPC1.js +0 -1
- package/client/dist/assets/CodePage-DwgbeqQ_.js +0 -3
- package/client/dist/assets/InteractiveJobComposer-BcR6xlHU.js +0 -19
- package/client/dist/assets/JobDetailModal-DKA_DUHe.js +0 -1
- package/client/dist/assets/JobDetailPage-vaOcE-cI.js +0 -1
- package/client/dist/assets/JobsPage-Bwoe-Fb0.js +0 -1
- package/client/dist/assets/PluginsPage-CTUHX_Lu.js +0 -2
- package/client/dist/assets/ProjectSettingsDialog-DeL8md1s.js +0 -1
- package/client/dist/assets/TemplatePreviewModal-_CgCCV7L.js +0 -1
- package/client/dist/assets/index-Bfj8HvX1.css +0 -2
- package/client/dist/assets/index-UnaJTMIy.js +0 -76
- package/client/dist/assets/narration-B-ynWTpn.js +0 -1
- package/client/dist/assets/narration-Bj_laZwg.js +0 -1
- package/client/dist/assets/narration-BxHpmMv_.js +0 -1
- package/client/dist/assets/narration-C1Px0-0j.js +0 -1
- package/client/dist/assets/narration-CAXhGbzM.js +0 -1
- package/client/dist/assets/narration-DQ1l_Kq6.js +0 -1
- package/client/dist/assets/narration-XISXwqZd.js +0 -1
- package/client/dist/assets/narration-kwzfglQo.js +0 -1
- package/client/dist/assets/settings-CCNZN5Q-.js +0 -1
- package/client/dist/assets/useSharedWebSocket-DbGL_YOt.js +0 -2
|
@@ -0,0 +1,208 @@
|
|
|
1
|
+
# Evolución de Specrails hacia un runtime de agentes
|
|
2
|
+
|
|
3
|
+
Análisis del 11 de septiembre de 2026. Recomendación arquitectónica basada en el código local y documentación oficial de los proyectos comparados.
|
|
4
|
+
|
|
5
|
+
**Recomendación**
|
|
6
|
+
|
|
7
|
+
Sí merece la pena pasar a una ejecución programática: Specrails debe decidir qué agente se ejecuta, con qué contexto, proveedor, herramientas, límites y evidencia de finalización. Recomiendo un runtime TypeScript compartido entre Core, Desktop y futuros workers. Conservaría el conocimiento y los controles de Specrails, y utilizaría componentes existentes para la infraestructura genérica.
|
|
8
|
+
|
|
9
|
+
LangGraph.js es mi candidato preferido para el motor de workflows. Lo validaría con un piloto frente a un dispatcher TypeScript mínimo que reutilice el journal actual de Core. La adopción se justifica si simplifica de forma demostrable reanudación, interrupciones y coordinación de ramas; añadir otra máquina de estados para ejecutar las mismas seis fases no sería suficiente.
|
|
10
|
+
|
|
11
|
+
Usaría `ai-agent-dev` como referencia de patrones y de un ejecutor Claude. No lo convertiría directamente en el motor general. Tampoco introduciría simultáneamente LangGraph, Mastra y otro supervisor de agentes. El objetivo es tener una autoridad clara para ejecutar el workflow.
|
|
12
|
+
|
|
13
|
+
**Alcance y confianza del análisis**
|
|
14
|
+
|
|
15
|
+
Se revisaron código, contratos, manifiestos, lockfiles, tests existentes y documentación de tres repositorios. `specrails-web` no se auditó funcionalmente: la decisión afecta principalmente a los tres componentes siguientes.
|
|
16
|
+
|
|
17
|
+
| Repositorio | Versión local | Commit de referencia | Situación |
|
|
18
|
+
| --- | --- | --- | --- |
|
|
19
|
+
| ai-agent-dev | 0.40.4 | `2949748` | Sin cambios locales detectados al iniciar |
|
|
20
|
+
| specrails-core | 5.1.0 | `2a8265e9` | Con cambios locales previos, incluidos runtime y su test |
|
|
21
|
+
| specrails-desktop | 2.43.1 | `8bc84921` | Con cambios locales previos en varias áreas |
|
|
22
|
+
|
|
23
|
+
Las conclusiones describen el checkout actual, no necesariamente los paquetes publicados. No se instalaron dependencias, ejecutaron modelos, consumieron API ni desplegaron servicios. En Core se ejecutó un subconjunto offline: `pipeline-state.test.ts`, `profile-schema.test.ts` y `profile-cli-validation.test.ts`, con 47 pruebas aprobadas en tres archivos. En Desktop pasaron otras 74 pruebas en cuatro archivos: `loop-graph.test.ts`, `loop-runs-store.test.ts`, `providers/registry.test.ts` y `providers/runtime.test.ts`. Son 121 pruebas aprobadas, no las suites completas. En ai-agent-dev se inspeccionaron las pruebas sin ejecutarlas: no tenía dependencias instaladas. Este documento es el único archivo creado por el análisis. Las ventajas de rendimiento propuestas son hipótesis para medir, no resultados de un benchmark de agentes.
|
|
24
|
+
|
|
25
|
+
**El punto de partida real**
|
|
26
|
+
|
|
27
|
+
Los tres repositorios ya contienen ejecución programática, con responsabilidades repartidas:
|
|
28
|
+
|
|
29
|
+
| Componente | Lo que ya aporta | Lo que falta para el objetivo |
|
|
30
|
+
| --- | --- | --- |
|
|
31
|
+
| Core | Roles, OpenSpec, perfiles, contexto de repositorios, journal, gates y recibos reales de verificación | Un ejecutor común que lance agentes y controle el avance sin depender del coordinador LLM |
|
|
32
|
+
| Desktop | Adaptadores, procesos, streaming, colas, loops, SQLite, costes, worktrees, MCP y entrega | Control homogéneo por fase; reanudación durable del grafo; política de proveedor por rol |
|
|
33
|
+
| ai-agent-dev | Pipeline TypeScript con Claude Agent SDK, contexto preparado, JSON de análisis, herramientas y checks | Generalización, proveedores intercambiables, persistencia durable y garantías de ejecución |
|
|
34
|
+
|
|
35
|
+
En Desktop, el pipeline principal sigue empaquetado en un único paso de IA. Su propio código lo explica: architect → developer → reviewer ocurre dentro de la invocación completa de Core. Eso limita qué puede observar y controlar el host en cada fase. [Grafo de Implement](/Users/javi/repos/specrails-desktop/server/loop-factory.ts:42).
|
|
36
|
+
|
|
37
|
+
La migración debe conservar los prompts que expresan conocimiento de arquitectura, desarrollo y revisión. Las decisiones mecánicas —routing, siguiente fase, reintento, validación, presupuesto— deben tener una implementación ejecutable única.
|
|
38
|
+
|
|
39
|
+
**Qué aporta ai-agent-dev**
|
|
40
|
+
|
|
41
|
+
El servicio NestJS recibe eventos y crea Jobs Kubernetes. Hay tres variantes de agente para Android/múltiples ecosistemas, iOS y web. La variante se selecciona con reglas específicas de repositorios de Busuu. El inventario contiene 67 archivos fuente TypeScript, aproximadamente 13.875 líneas, y 33 archivos `.spec.ts`. [Dispatch del servicio](/Users/javi/repos/ai-agent-dev/service/src/webhook/webhook.service.ts:129).
|
|
42
|
+
|
|
43
|
+
El flujo de la variante web es: ticket → clonación → mapa del repositorio → scout opcional → análisis estructurado → decisión por confianza → rama y OpenSpec → ejecución secuencial de tareas → pasos obligatorios → archivo de OpenSpec → checks y reparaciones → commit/PR → comentario. [Entrada](/Users/javi/repos/ai-agent-dev/agent-web/src/index.ts:223), [implementación](/Users/javi/repos/ai-agent-dev/agent-web/src/implement.ts:559).
|
|
44
|
+
|
|
45
|
+
Reutilizaría estos patrones:
|
|
46
|
+
|
|
47
|
+
- Preparar contexto con código antes de invocar IA: mapa de servicios y explorador económico cuando el tamaño lo justifica. [Mapa](/Users/javi/repos/ai-agent-dev/agent-web/src/repo-map.ts:1), [selección del scout](/Users/javi/repos/ai-agent-dev/agent-web/src/agent.ts:110).
|
|
48
|
+
- Separar análisis y solicitud de aclaración en resultados estructurados. Añadiría validación estricta en runtime: el cast de TypeScript actual no valida el contenido recibido. [Schema](/Users/javi/repos/ai-agent-dev/agent-web/src/analysis-schema.ts:1), [consumo del resultado](/Users/javi/repos/ai-agent-dev/agent-web/src/index.ts:272).
|
|
49
|
+
- Elegir modelos por fase y dar contexto acotado a cada tarea. Es selección estática; no demuestra optimización dinámica ni ahorro. [Opciones de análisis](/Users/javi/repos/ai-agent-dev/agent-web/src/agent.ts:206), [ejecutor de tareas](/Users/javi/repos/ai-agent-dev/agent-web/src/implement.ts:289).
|
|
50
|
+
- Instrumentar fases con OpenTelemetry y asociar tokens y costes a las invocaciones. [Instrumentación](/Users/javi/repos/ai-agent-dev/agent-web/src/instrumentation.ts:21).
|
|
51
|
+
- Extraer los adaptadores de toolchains y los checks útiles, eliminando dependencias de Jira, repositorios y releases internos.
|
|
52
|
+
|
|
53
|
+
Hay limitaciones concretas que impiden tratarlo como un framework general terminado:
|
|
54
|
+
|
|
55
|
+
| Hallazgo en el código | Implicación para Specrails |
|
|
56
|
+
| --- | --- |
|
|
57
|
+
| Claude Agent SDK declarado como `^0.3.148`; lockfile resuelve `0.3.217` | Acoplamiento a una implementación de agente, sin interfaz neutral de ejecución. [Lockfile](/Users/javi/repos/ai-agent-dev/agent/package-lock.json:36) |
|
|
58
|
+
| `persistSession: false` en las consultas inspeccionadas | La continuación usa principalmente Git, artefactos y Jira; no hay checkpoint transaccional de fases. [Opciones](/Users/javi/repos/ai-agent-dev/agent-web/src/agent.ts:212) |
|
|
59
|
+
| Al haber diez Jobs activos se omite crear otro | No hay cola durable ni reserva atómica; se puede perder trabajo. [Límite](/Users/javi/repos/ai-agent-dev/service/src/webhook/webhook.service.ts:166) |
|
|
60
|
+
| Se captura coste, pero no existe un presupuesto global homogéneo | Turnos y timeout limitan ejecución, no garantizan gasto máximo. Hay una excepción puntual de presupuesto en Maestro. [Timeout](/Users/javi/repos/ai-agent-dev/agent-web/src/query-timeout.ts:10), [Maestro](/Users/javi/repos/ai-agent-dev/agent/src/maestro.ts:204) |
|
|
61
|
+
| El caller de `fixQualityFailures` toma el coste e ignora el resultado final | Puede continuar hacia commit/PR con checks fallidos; además archiva OpenSpec antes de ejecutarlos. [Caller](/Users/javi/repos/ai-agent-dev/agent-web/src/implement.ts:682), [resultado del fixer](/Users/javi/repos/ai-agent-dev/agent-web/src/quality/index.ts:316) |
|
|
62
|
+
| La terminación de tareas no discrimina todos los estados de error del SDK antes de marcarlas hechas | «Terminó la invocación» no equivale a «se cumplió la tarea». [Resultado](/Users/javi/repos/ai-agent-dev/agent-web/src/implement.ts:317), [marcado](/Users/javi/repos/ai-agent-dev/agent-web/src/implement.ts:669) |
|
|
63
|
+
| Análisis con Bash, bypass de permisos y todo `process.env` | La intención de solo lectura del prompt no impone una frontera real; conservar el enforcement más fuerte de Desktop. [Configuración](/Users/javi/repos/ai-agent-dev/agent-web/src/agent.ts:206) |
|
|
64
|
+
|
|
65
|
+
Otros hallazgos: el scout no entra en el total visible publicado del pipeline; `--branch` intenta consultar historial antes de clonar y la investigación puede empezar en la rama por defecto; hay módulos duplicados entre las tres variantes. Los checks iOS devuelven verificación no disponible fuera de macOS. Son motivos para extraer conceptos y módulos seleccionados, en vez de copiar toda la aplicación. [Scout](/Users/javi/repos/ai-agent-dev/agent/src/scout.ts:94), [continuación](/Users/javi/repos/ai-agent-dev/agent-web/src/index.ts:243), [checkout](/Users/javi/repos/ai-agent-dev/agent-web/src/implement.ts:575), [checks iOS](/Users/javi/repos/ai-agent-dev/agent-ios/src/quality/ios.ts:4).
|
|
66
|
+
|
|
67
|
+
**Qué conservar de Core y Desktop**
|
|
68
|
+
|
|
69
|
+
Core ya proporciona un contrato valioso: identidad de ejecución, specs congeladas, repositorios seleccionados y ownership de Git, backlog y worktrees. Su runtime escribe un journal atómico, ejecuta comprobaciones reales y vincula los recibos al contenido del candidato y al entorno. También determina qué fase ha quedado invalidada. Un framework genérico no sustituye estas reglas. [Contexto](/Users/javi/repos/specrails-core/src/installer/runtime/pipeline-state.ts:6), [gates y transiciones](/Users/javi/repos/specrails-core/src/installer/runtime/pipeline-state.ts:335), [reanudación lógica](/Users/javi/repos/specrails-core/src/installer/runtime/pipeline-state.ts:427).
|
|
70
|
+
|
|
71
|
+
Pero este helper no ejecuta agentes: la coordinación principal sigue en plantillas distintas para Claude, Codex y Gemini; Kimi tiene un runner específico considerable. Existe incluso deriva entre el perfil validado y el prompt: schema/CLI permiten routing vacío o sin default, mientras el prompt Claude exige routing y exactamente un default. Centralizar esa decisión en código elimina una fuente de divergencia. [Operaciones del helper](/Users/javi/repos/specrails-core/src/installer/runtime/pipeline-state.ts:670), [schema](/Users/javi/repos/specrails-core/schemas/profile.v1.json:74), [validador CLI](/Users/javi/repos/specrails-core/bin/specrails-core.mjs:516), [prompt](/Users/javi/repos/specrails-core/templates/commands/specrails/implement.md:93).
|
|
72
|
+
|
|
73
|
+
Desktop ya tiene contratos de capacidades y eventos para Claude, Codex, Gemini y Kimi. Hay transporte Codex app-server y Claude stream-json, además de ejecución por procesos. No hace falta reemplazar de golpe estos transportes para conseguir control programático. [ProviderAdapter](/Users/javi/repos/specrails-desktop/server/providers/types.ts:213), [ciclo común](/Users/javi/repos/specrails-desktop/server/spawn-lifecycle.ts:94), [sesiones vivas](/Users/javi/repos/specrails-desktop/server/providers/live-session.ts:5).
|
|
74
|
+
|
|
75
|
+
Sus límites actuales ayudan a definir el piloto:
|
|
76
|
+
|
|
77
|
+
- El motor de loops sigue un único sucesor; no implementa fan-out/join interno, aunque sí existe paralelismo entre ejecuciones aisladas. [Validación del grafo](/Users/javi/repos/specrails-desktop/server/loop-graph.ts:208).
|
|
78
|
+
- Al reiniciar, los loops activos o pausados se marcan fallidos. La recuperación contable y del outbox es útil, pero no reanuda la continuación en memoria. [Reconciliación](/Users/javi/repos/specrails-desktop/server/loop-runs-store.ts:545).
|
|
79
|
+
- En loops manda el proveedor/modelo del rail para todos los nodos. Core admite un proveedor global y modelos por agente dentro de él. Permitir proveedor distinto por rol requiere un contrato nuevo explícito. [Política del rail](/Users/javi/repos/specrails-desktop/server/loop-run-manager.ts:1465), [perfil v1](/Users/javi/repos/specrails-core/schemas/profile.v1.json:26).
|
|
80
|
+
- El cap de coste del loop se comprueba entre pasos: puede excederse por un paso completo. Si el paso contiene todo Implement, el control es muy grueso. [Contrato del cap](/Users/javi/repos/specrails-desktop/server/loop-graph.ts:57).
|
|
81
|
+
- El ledger ya diferencia consumo y estimaciones; el desglose por fase es principalmente reconstrucción de eventos Claude. Hay que medir desde cada invocación independiente. [Ledger](/Users/javi/repos/specrails-desktop/server/ai-invocations.ts:37), [desglose](/Users/javi/repos/specrails-desktop/server/job-phase-breakdown.ts:3).
|
|
82
|
+
|
|
83
|
+
Mantendría worktrees, locks, alcance multirrepo, validación final de Core, MCP y entrega de PRs. [Preparación y finalización](/Users/javi/repos/specrails-desktop/server/core-execution.ts:48), [aislamiento](/Users/javi/repos/specrails-desktop/server/rail-isolation.ts:28), [control de herramientas](/Users/javi/repos/specrails-desktop/server/mcp/tools/types.ts:280).
|
|
84
|
+
|
|
85
|
+
**Comparativa de alternativas**
|
|
86
|
+
|
|
87
|
+
La tabla expresa mi valoración de encaje con este código, no una clasificación universal ni un benchmark de rendimiento.
|
|
88
|
+
|
|
89
|
+
| Alternativa | Aportación y licencia comprobada | Encaje en Specrails |
|
|
90
|
+
| --- | --- | --- |
|
|
91
|
+
| LangGraph.js | Grafos con estado, checkpoints, streaming e interrupciones; biblioteca MIT | Primera opción para probar como motor embebido. Permite mantener nodos propios y ejecutores actuales. Exige diseñar almacenamiento, recovery y efectos externos. [Overview](https://docs.langchain.com/oss/javascript/langgraph/overview), [licencia](https://github.com/langchain-ai/langgraphjs/blob/main/LICENSE) |
|
|
92
|
+
| Mastra | Framework TypeScript con agentes, workflows y suspensión/reanudación; núcleo Apache-2.0, directorios `ee/` con licencia distinta | Segunda opción. Atractivo para una aplicación nueva que necesite muchas piezas integradas; aquí existe solapamiento con producto y runtime de Desktop. [Workflows](https://mastra.ai/docs/workflows/overview), [licencias](https://github.com/mastra-ai/mastra/blob/main/LICENSE.md) |
|
|
93
|
+
| Vercel AI SDK | Capa TypeScript multiproveedor, resultados estructurados y bucles de herramientas; Apache-2.0 | Útil si necesitamos llamadas API directas. Por sí solo no acredita reanudación durable de un workflow completo. Puede usar proveedores directos; no exige contratar el gateway. Revisar versión: el README actual exige Node 22+, mientras Specrails declara 20.19+. [Repositorio](https://github.com/vercel/ai), [licencia](https://github.com/vercel/ai/blob/main/LICENSE) |
|
|
94
|
+
| OpenAI Agents SDK | Agentes, herramientas, handoffs y control del bucle en TS/Python; admite adapters para modelos externos | Viable, especialmente con inversión en OpenAI. No lo elegiría como autoridad de dominio; tampoco equivale a Codex. El host sigue siendo responsable de storage y decisiones de aprobación. [SDK](https://developers.openai.com/api/docs/guides/agents/sdk), [proveedores](https://developers.openai.com/api/docs/guides/agents/models) |
|
|
95
|
+
| CrewAI | Crews y Flows en Python; MIT | Ofrece tanto colaboración autónoma como control explícito. No compensa introducir otro runtime/lenguaje para sustituir las piezas TS que ya existen. [Repositorio y licencia](https://github.com/crewAIInc/crewAI) |
|
|
96
|
+
| Microsoft Agent Framework | Workflows y agentes, con foco documentado en .NET/Python; MIT | Más natural en un stack Microsoft. No es mi primera opción para el paquete npm y el sidecar actuales. Para un desarrollo nuevo no partiría del AutoGen anterior: su repo dirige a migración. [Framework](https://github.com/microsoft/agent-framework), [AutoGen](https://github.com/microsoft/autogen) |
|
|
97
|
+
|
|
98
|
+
LangGraph puede usarse como biblioteca sin adoptar la plataforma alojada de LangSmith. Su persistencia necesita un checkpointer durable: el de memoria pierde estado al reiniciar. La documentación presenta SQLite para uso local/desarrollo y PostgreSQL como alternativa; la adecuación de SQLite al producto de escritorio debe verificarse con concurrencia, WAL, bloqueo y pruebas de caída. [Persistencia](https://docs.langchain.com/oss/javascript/langgraph/persistence).
|
|
99
|
+
|
|
100
|
+
Mastra también persiste snapshots de workflows y puede guardar estado de suspensión en almacenamiento configurado. No lo descartaría por carecer de estas capacidades. Lo situaría detrás de LangGraph por la cantidad de responsabilidades que Specrails ya tiene resueltas. [Snapshots de Mastra](https://mastra.ai/docs/workflows/snapshots).
|
|
101
|
+
|
|
102
|
+
**El ejecutor de programación es una decisión distinta**
|
|
103
|
+
|
|
104
|
+
Un cliente de modelos devuelve texto o llamadas a herramientas. Un ejecutor de programación también administra archivos, edición, shell, herramientas, contexto, permisos y sesiones. Reemplazar un CLI por una llamada a un modelo no reproduce automáticamente ese comportamiento.
|
|
105
|
+
|
|
106
|
+
| Ejecutor posible | Valor para Specrails | Decisión recomendada |
|
|
107
|
+
| --- | --- | --- |
|
|
108
|
+
| Adaptadores actuales + Claude Agent SDK / Codex SDK o app-server | Aprovechan herramientas y sesiones de programación ya conocidas | Primera etapa. Exponerlos detrás de `AgentExecutor`; modernizar transporte donde aporte una mejora concreta. [Claude SDK](https://code.claude.com/docs/en/agent-sdk/overview), [Codex SDK](https://learn.chatgpt.com/docs/codex-sdk) |
|
|
109
|
+
| OpenCode | Coding agent MIT, proveedores múltiples y cliente JS/TS para controlar un servidor | Buen candidato a ejecutor neutral posterior. Probar calidad, permisos, lifecycle del proceso y empaquetado antes de convertirlo en dependencia. [Licencia](https://github.com/anomalyco/opencode), [SDK](https://opencode.ai/docs/sdk/), [proveedores](https://opencode.ai/docs/providers/) |
|
|
110
|
+
| Deep Agents | Harness sobre LangGraph, con contexto, herramientas y subagentes, y ejemplos multiproveedor | Alternativa si queremos una implementación de agente más controlada por Specrails. Añade otra migración: probar separadamente del motor de fases. [Documentación JS](https://docs.langchain.com/oss/javascript/deepagents/overview) |
|
|
111
|
+
| OpenHands Software Agent SDK | SDK MIT con APIs Python, TypeScript y REST; workspace local o Agent Server con entornos efímeros | Interesante para workers de desarrollo aislados. Evaluar su arquitectura/operación; no descartarlo por asumir que solo ofrece Python. [Repositorio](https://github.com/OpenHands/software-agent-sdk) |
|
|
112
|
+
|
|
113
|
+
LiteLLM resuelve otra capa: gateway de APIs, routing, balanceo y alternativas de proveedor. Lo consideraría si Specrails opera workers compartidos y necesita centralizar credenciales y límites. No lo impondría a cada instalación de Desktop ni lo usaría para orquestar fases. Su código base tiene licencia MIT con partes sujetas a licencia comercial. [Gateway](https://github.com/BerriAI/litellm), [failover](https://docs.litellm.ai/docs/proxy/reliability), [licencia](https://github.com/BerriAI/litellm/blob/litellm_internal_staging/LICENSE).
|
|
114
|
+
|
|
115
|
+
**Arquitectura propuesta**
|
|
116
|
+
|
|
117
|
+
```mermaid
|
|
118
|
+
flowchart TD
|
|
119
|
+
Entry[Desktop / CLI / worker] --> Runtime[Runtime compartido TypeScript]
|
|
120
|
+
Core[Core: roles, specs, reglas y gates] --> Runtime
|
|
121
|
+
Runtime --> Workflow[Motor de ejecución reemplazable]
|
|
122
|
+
Workflow --> Executors[AgentExecutor por fase]
|
|
123
|
+
Executors --> Native[Claude / Codex / Gemini / Kimi]
|
|
124
|
+
Executors --> Neutral[Ejecutor neutral o herramientas + API]
|
|
125
|
+
Runtime --> Store[Estado durable y eventos]
|
|
126
|
+
Runtime --> Policy[Presupuesto, permisos y capacidades]
|
|
127
|
+
Runtime --> Host[Servicios del host: worktrees, checks y entrega]
|
|
128
|
+
Store --> UI[Proyecciones para la UI y analítica]
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
Empezaría como módulo o paquete dentro del ecosistema actual, sin crear obligatoriamente un cuarto producto o repositorio. Debe poder consumirse sin Electron/Tauri, React o un Desktop arrancado. Core standalone conserva su CLI; Desktop actúa como host. Jira/Kubernetes de `ai-agent-dev` serían adaptadores de entrada y ejecución remota opcionales.
|
|
132
|
+
|
|
133
|
+
Los contratos clave serían:
|
|
134
|
+
|
|
135
|
+
- `AgentDefinition`: rol, versión de instrucciones, schemas de entrada/salida y capacidades requeridas.
|
|
136
|
+
- `AgentExecutor`: ejecutar, cancelar, consultar capacidades y reanudar si el transporte lo soporta; emite eventos normalizados.
|
|
137
|
+
- `ProviderPolicy`: proveedor/modelo por rol, credencial referenciada, esfuerzo y alternativas permitidas. Separar proveedor de inferencia del tipo de ejecutor.
|
|
138
|
+
- `RunContext`: scope congelado, repositorios, referencias a artefactos, ownership actual de Core y versiones de workflow, Core y definiciones de agentes.
|
|
139
|
+
- `StepResult`: estado explícito (`succeeded`, `failed`, `blocked`, `cancelled`, `needs_recovery`), resultado validado, recibos y consumo.
|
|
140
|
+
- `BudgetPolicy`: tiempo, tokens, turnos, coste y concurrencia. Lo desconocido se registra como desconocido.
|
|
141
|
+
- `HostServices`: filesystem/worktrees, runner de comprobaciones, aprobaciones, publicación y backlog.
|
|
142
|
+
|
|
143
|
+
No trasladaría un historial interno de Claude a Codex como si fuera una sesión portable. Un cambio de proveedor entrega un paquete neutral de artefactos, resultados y estado del workspace. El executor solo anuncia resume nativo cuando existe.
|
|
144
|
+
|
|
145
|
+
**Cómo evitar dos máquinas de estado en conflicto**
|
|
146
|
+
|
|
147
|
+
Core debe seguir definiendo las reglas de validez del trabajo. El motor elegido decide scheduling y continuidad; los ejecutores solo ejecutan un rol. Desktop conserva UI, admisión de jobs, concesión de recursos y servicios del host; deja de planificar las fases del camino migrado. El runtime solo puede repartir workers dentro de la concurrencia y presupuesto concedidos por el host. Durante el piloto, cada ejecución pertenece a un único motor: el nuevo camino invoca roles acotados y nunca ejecuta `/implement` completo dentro de otro workflow.
|
|
148
|
+
|
|
149
|
+
En el piloto conservador, el journal de Core es la autoridad: cada inicio o reanudación consulta `inspectPipeline`, y ningún checkpoint puede saltar sus gates. Si se adopta un motor durable como destino, se extraen los gates como políticas y la persistencia del motor controla la ejecución; el journal público pasa a ser una proyección compatible. El helper CLI debe delegar en ese coordinador para los runs migrados, en lugar de escribir el mismo estado por su cuenta.
|
|
150
|
+
|
|
151
|
+
Un coordinador único valida el gate de Core, registra el recibo de la fase y publica un evento. Las tablas de estado que usa Desktop son proyecciones. Si la transición mantiene temporalmente varios almacenes, se necesitan revisiones monotónicas, correlación de eventos y reconciliación explícita ante una caída entre escrituras. Sincronizar dos JSON después de cada nodo no resuelve el problema. Además, la reanudación siempre debe comprobar que el candidato y sus evidencias siguen vigentes.
|
|
152
|
+
|
|
153
|
+
Para efectos externos —PRs, commits, cambios de estado— usar identidad estable por ejecución/paso y comprobar qué ocurrió antes de repetir. Conservar y ampliar el patrón de outbox existente. Un checkpoint no revierte cambios en Git, no resucita un proceso y no garantiza ejecución exactamente una vez de una API remota.
|
|
154
|
+
|
|
155
|
+
La pausa humana debe sobrevivir al reinicio. Una fase interrumpida a mitad de edición requiere comprobar el worktree y el último recibo: reanudar la sesión cuando sea posible o pasar a recuperación explícita. No basta con repetir ciegamente el nodo. Los reintentos del motor son infraestructura, no prueba de que una operación sea segura para repetir. [Políticas de retry y timeout](https://docs.langchain.com/oss/javascript/langgraph/fault-tolerance).
|
|
156
|
+
|
|
157
|
+
Actualizar Desktop o Core no debe hacer que un checkpoint antiguo continúe sobre un grafo incompatible. Cada run congela versiones; el runtime exige compatibilidad, migración explícita o recuperación. Al reanudar también se revalidan los permisos actuales: mantener la identidad de una aprobación no restaura capacidades revocadas o caducadas.
|
|
158
|
+
|
|
159
|
+
**De dónde puede venir la eficiencia**
|
|
160
|
+
|
|
161
|
+
1. Quitar del contexto LLM validaciones de perfil, routing mecánico, preparación de ramas y decisiones de control ya expresables en código.
|
|
162
|
+
2. Entregar a cada rol el contexto que necesita: specs y mapa para arquitectura; tareas y archivos concretos para desarrollo; diff, criterios y recibos para revisión.
|
|
163
|
+
3. Reutilizar resultados cuya identidad de candidato/contexto siga vigente. Invalidarlos cuando cambie el código, la spec o el entorno relevante.
|
|
164
|
+
4. Elegir inicialmente modelos por rol mediante reglas simples y perfiles versionados. Escalar capacidad tras un fallo clasificado, con límite; no introducir un agente supervisor que decida cada transición trivial.
|
|
165
|
+
5. Paralelizar solo trabajo independiente, en workspaces compatibles. Dos desarrolladores editando los mismos archivos necesitan coordinación y validación del resultado integrado.
|
|
166
|
+
6. Reservar presupuesto antes de despachar y reconciliar consumo real después. Un límite global requiere coordinar invocaciones concurrentes. Con APIs se puede limitar mejor cada llamada; con algunos CLIs solo se ve el total al terminar, de modo que no prometería un cap monetario exacto.
|
|
167
|
+
7. Instrumentar todas las invocaciones, incluidos scout, retries, reparaciones y síntesis. Separar costes reportados, estimados y no disponibles.
|
|
168
|
+
|
|
169
|
+
La métrica principal debe ser coste por spec aceptada, acompañada de calidad y tiempo humano. Menos tokens con más errores no es una mejora; más agentes tampoco implica más eficiencia. Cambiar SDK y política de modelos a la vez impediría atribuir el resultado a una causa.
|
|
170
|
+
|
|
171
|
+
**Plan de migración y criterio de decisión**
|
|
172
|
+
|
|
173
|
+
| Etapa | Entregable | Criterio para avanzar |
|
|
174
|
+
| --- | --- | --- |
|
|
175
|
+
| 1. Contratos y línea base | Ejecutores, eventos, errores y ownership; métricas actuales sobre un conjunto de tareas | Poder observar invocaciones y fases sin depender de inferencias del texto |
|
|
176
|
+
| 2. Piloto de ejecución | Un flujo explícito arquitectura → desarrollo → verificación → revisión → corrección limitada → archivo; entrega del host | Mismas garantías de Core y pruebas de caída/reanudación |
|
|
177
|
+
| 3. Comparar motores | Implementar el flujo con dispatcher mínimo y con LangGraph, usando los mismos ejecutores y fixtures | Elegir LangGraph si reduce código especial de recovery/coordinación sin degradar empaquetado y compatibilidad |
|
|
178
|
+
| 4. Migración gradual | Flag por ejecución/proyecto; Core CLI y Desktop consumen el mismo runtime; los jobs antiguos terminan en su motor | Ninguna ejecución es reclamada por dos motores; rollback por nuevas ejecuciones |
|
|
179
|
+
| 5. Proveedores por rol | Perfil nuevo con capacidades, límites y alternativas; SDKs donde aporten valor | Cambio de proveedor comprobado sin perder scope, permisos, trazabilidad ni calidad |
|
|
180
|
+
| 6. Escala opcional | Workers, gateway y almacenamiento remoto si el producto lo necesita | Demanda operativa demostrada; no requisito para uso local |
|
|
181
|
+
|
|
182
|
+
El piloto debe pasar, al menos, estas pruebas verificables:
|
|
183
|
+
|
|
184
|
+
- Matar el proceso después de desarrollo y continuar en verificación/revisión sin regenerar arquitectura ni repetir una fase válida.
|
|
185
|
+
- Matarlo durante una edición y recuperar el workspace de forma explícita; no certificar como terminado lo que solo tiene logs parciales.
|
|
186
|
+
- Pausar para aprobación, reiniciar y resolver esa misma aprobación sin perder su identidad.
|
|
187
|
+
- Actualizar el runtime con un run pausado y detectar incompatibilidad de versiones; revocar un permiso antes de reanudar y bloquear el siguiente efecto que lo requiera.
|
|
188
|
+
- Simular 429, timeout y credencial inválida: diferenciar fallo transitorio, límite de cuenta y error permanente; aplicar retry acotado. Un fallback tras resultado ambiguo debe reconciliar efectos.
|
|
189
|
+
- Unir dos trabajos independientes y verificar el candidato integrado; rechazar escrituras fuera de los repositorios seleccionados.
|
|
190
|
+
- Impedir archivo/finalización si los checks obligatorios fallan o si su recibo ya no describe el candidato actual.
|
|
191
|
+
- Repetir un callback y reiniciar entre intención/confirmación de PR sin duplicar publicación ni contabilizar dos veces.
|
|
192
|
+
- Comprobar cancelación del árbol de procesos, bloqueo de SQLite, empaquetado y compatibilidad Windows/macOS/Linux, incluyendo Node y dependencias nativas.
|
|
193
|
+
|
|
194
|
+
Después haría una evaluación pequeña y emparejada, por ejemplo 15–20 specs representativas, ejecutadas desde el mismo commit base. Esa cifra es una propuesta de trabajo, no un tamaño estadístico que garantice una conclusión. Primero mantener modelo y herramientas constantes para medir el cambio de orquestación; después probar modelos por rol. Registrar éxito aceptado, regresiones, coste total, duración, reintentos, contexto repetido, checks redundantes y minutos de intervención humana. No se ha realizado esa evaluación en esta investigación.
|
|
195
|
+
|
|
196
|
+
Si el dispatcher existente consigue los mismos resultados con menos complejidad, lo mantendría. La frontera `WorkflowEngine` permitiría introducir LangGraph más adelante. Lo que recomiendo adoptar desde el principio es el control explícito y compartido de ejecución.
|
|
197
|
+
|
|
198
|
+
**Qué significa gratuito y open source en esta decisión**
|
|
199
|
+
|
|
200
|
+
Es posible ejecutar el núcleo MIT de LangGraph localmente sin contratar una plataforma de orquestación. Las otras alternativas de la tabla tienen licencias y componentes propios; los productos cloud o Enterprise no deben confundirse con sus bibliotecas abiertas. Las licencias se han consultado como metadatos técnicos; no constituyen un análisis jurídico.
|
|
201
|
+
|
|
202
|
+
Las llamadas a modelos siguen consumiendo API, cuotas de una suscripción o recursos de hardware. Un modelo local evita facturación por token a un proveedor, pero requiere infraestructura y validar su calidad. Tampoco convierte un framework abierto en un stack completamente abierto si depende de un ejecutor propietario.
|
|
203
|
+
|
|
204
|
+
En particular, que `ai-agent-dev` declare MIT no convierte Claude Agent SDK en MIT: su lockfile remite a otra licencia y la documentación oficial remite a términos comerciales. Puede ser un adaptador opcional de un runtime abierto. [Lockfile](/Users/javi/repos/ai-agent-dev/agent/package-lock.json:36), [licencia y términos del SDK](https://code.claude.com/docs/en/agent-sdk/overview#license-and-terms).
|
|
205
|
+
|
|
206
|
+
Para autenticación, recomiendo modelar ambos caminos —cuenta local del usuario y API— sin asumir que son intercambiables. La documentación de Claude contiene un matiz relevante: el artículo de soporte indica que se pausó el cambio anunciado de consumo de suscripción, mientras la documentación del SDK limita ofrecer login/límites de claude.ai en productos de terceros sin aprobación previa. Eso no permite prometer a usuarios de Specrails acceso programático incluido por defecto. Mantener los transportes existentes no resuelve por sí mismo la elegibilidad del producto. [Actualización del plan](https://support.claude.com/en/articles/15036540-use-the-claude-agent-sdk-with-your-claude-plan), [integración del SDK](https://code.claude.com/docs/en/agent-sdk/overview).
|
|
207
|
+
|
|
208
|
+
La decisión inmediata propuesta es invertir en un piloto de runtime compartido, tomando LangGraph.js como candidato externo principal y el journal actual como base de comparación. Aprovechar los patrones de `ai-agent-dev`, los contratos de Core y los servicios de Desktop permite avanzar con una migración acotada y medir antes de comprometer una reescritura.
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
# Programmatic agent runtime: implementation verification
|
|
2
|
+
|
|
3
|
+
Validated locally on **2026-09-11**, on macOS with Node 25.9.0. The complete paired execution was also validated with Desktop's bundled **Node 22.23.1**. This records the paired source implementation in Specrails Core and Desktop; it is not a published release or a Windows runner result.
|
|
4
|
+
|
|
5
|
+
## Result
|
|
6
|
+
|
|
7
|
+
Core owns a typed LangGraph workflow for architecture, development, deterministic verification, review, bounded corrections and archive. Provider adapters execute individual roles. Desktop admits the configured workflow into its existing scoped worktree/rail execution and exposes configuration, progress, durable continuation, approval and cancellation.
|
|
8
|
+
|
|
9
|
+
The supported built-in transports are Claude, Codex, Gemini, Kimi and an OpenAI-compatible HTTP tool loop. A programmatic executor registry supports additional providers without changing the workflow. The new orchestration infrastructure is free and open source; existing model services and local hardware retain their own costs and terms.
|
|
10
|
+
|
|
11
|
+
## Automated checks
|
|
12
|
+
|
|
13
|
+
| Check | Result |
|
|
14
|
+
| --- | --- |
|
|
15
|
+
| Core full coverage suite | 50 files passed; 750 tests passed, 1 skipped |
|
|
16
|
+
| Core coverage | 90.86% lines, 86.44% statements, 79.63% branches, 93.23% functions |
|
|
17
|
+
| Desktop server/CLI full coverage suite | 336 files passed, 2 skipped; 8,276 tests passed, 7 skipped |
|
|
18
|
+
| Desktop server/CLI coverage | 89.36% lines, 86.67% statements, 79.66% branches, 90.64% functions |
|
|
19
|
+
| Desktop client full coverage suite | 390 files passed; 4,877 tests passed |
|
|
20
|
+
| Desktop client coverage | 90.04% lines/statements, 84.28% branches, 76% functions |
|
|
21
|
+
| TypeScript | Core and Desktop typechecks passed |
|
|
22
|
+
| Builds | Core build; Desktop server, client, CLI and MCP bridge builds passed |
|
|
23
|
+
| Script regressions | Core 19 and Desktop 75 tests passed |
|
|
24
|
+
| Core npm package | Actual tarball installed in an unrelated consumer; both CLI entries, four provider assemblies, runtime export/declarations and a LangGraph execution passed |
|
|
25
|
+
| Desktop npm package | Actual tarball installed in an unrelated consumer; CLI, assets, schemas and compiled CommonJS-to-external-Core-CLI negotiation/validation passed, including simulated pkg execution without bundled Node |
|
|
26
|
+
| Paired source assembly | Locked Core production dependencies staged in `src-tauri/core`; offline workflow smoke passed |
|
|
27
|
+
| Complete paired workflow | Compiled Desktop bridge → real Core CLI → localhost HTTP tool fixture → real verification → approval → archive passed on Node 25.9.0 and bundled Node 22.23.1; two resumes preserved completed roles, usage and Git ownership |
|
|
28
|
+
| Sidecar bundle | Server esbuild bundle using the native build's CJS/Node 22 options passed |
|
|
29
|
+
| New Desktop portability job | Workflow passed `actionlint`; its exact eight-suite command passed 111 tests locally on macOS |
|
|
30
|
+
| Existing Core compatibility | Desktop 2.43.1 and the locally selected installed Core 5.2.3 passed the existing contract check |
|
|
31
|
+
|
|
32
|
+
All existing coverage thresholds remained enabled. The client suite initially hit an unrelated AddProjectDialog timeout under concurrent heavy builds; its isolated suite and a complete rerun with two workers passed. Server tests that bind localhost were run with the required execution permissions after sandbox-only `EPERM` failures.
|
|
33
|
+
|
|
34
|
+
The Desktop full coverage figures above precede the final Node-launcher fallback fix. After that fix, the eight affected suites passed 105 tests, typecheck and the server build passed again, and the final installed-package plus paired Node 22 checks passed. The resolver preserves the running interpreter in ordinary Node, prefers a bundled interpreter when available, and uses PATH `node` from a packaged server whose bundled Node is missing. The new portability job also includes its regression suite.
|
|
35
|
+
|
|
36
|
+
The installed Core compatibility check preserves the legacy contract. The new API was separately exercised against the **paired source Core** through its installed package, assembled bundle and Desktop's compiled loader; an older published package is not evidence that API 1 is present.
|
|
37
|
+
|
|
38
|
+
## Live provider runs (2026-09-11, macOS, Claude Sonnet via the installed CLI)
|
|
39
|
+
|
|
40
|
+
Two end-to-end runs on a throwaway Node repository (`node --test`) using the source Core CLI directly, before and after the second round of changes:
|
|
41
|
+
|
|
42
|
+
| Run | Configuration | Outcome |
|
|
43
|
+
| --- | --- | --- |
|
|
44
|
+
| live-01 (previous runtime) | Claude for all roles, verification configured | Failed after 2 min and $0.18: the developer had no shell, left a "run the tests" task unchecked, and the verify step aborted the workflow with "Required implementation tasks remain incomplete" |
|
|
45
|
+
| live-02 (current runtime) | Claude for all roles, **no verification configured** | Succeeded in 72 s for $0.33: the architect proposed `npm test`, the developer ran the tests in its sandbox and ticked all six tasks, verification ran the proposed command, the reviewer approved with score 96, archive completed |
|
|
46
|
+
|
|
47
|
+
The same failure shape was reproduced by a user on a greenfield HTML project (a manual browser checklist task the developer could not tick). The changes that address it: developer tool parity with the legacy Implement step, optional/architect-proposed verification with admitted unverified repositories, unchecked tasks routed back as developer feedback, architect instructions that forbid manual or delivery tasks, session reuse on correction passes, native structured output with one in-session repair turn, readable phase notes instead of raw JSON in the log, and a fresh attempt budget on explicit resume.
|
|
48
|
+
|
|
49
|
+
## Behavioral evidence
|
|
50
|
+
|
|
51
|
+
- Offline workflow tests exercise success, correction limits, configuration identity, exclusive leases, persisted approvals, process interruption, explicit recovery, evidence invalidation and archive crash reconciliation.
|
|
52
|
+
- Core rejects invalid verification repository IDs/paths and known unsupported provider limits before any role executes. Injected executors declare their own limit capabilities.
|
|
53
|
+
- Provider tests exercise structured CLI events, old Kimi ACP plan mode, scoped tools, cancellation, incomplete responses, Windows argument handling and a real localhost OpenAI-compatible coding fixture. A local Kimi 0.27 capability probe initialized ACP and selected plan mode without submitting a model prompt.
|
|
54
|
+
- Gemini and Kimi ACP return the final assistant turn separately from tool-progress commentary. Gemini's native policy engine was also checked locally across 36 decisions with plan disabled and existing broad user approvals: only the designated read-only tools remained allowed. System policies that disable the per-run policy cause an explicit capability failure.
|
|
55
|
+
- Desktop integration tests execute real child processes and verification commands, validate final Core identity/evidence, preserve interrupted worktrees and retain original scope when resuming.
|
|
56
|
+
- `node scripts/smoke-agent-runtime-pair.mjs` additionally exercises the actual compiled Desktop/Core boundary against a deterministic local HTTP model fixture. The fixture edits only its temporary repository through Core tools and validates the complete approval/archive flow.
|
|
57
|
+
- Desktop adopts Core's paused-question contract: `pendingQuestion`, `traceId` and step `visits` from `status --compact`, `resume --answer`, tighten-only review thresholds (70/75/60 floors) and `architect.onLowConfidence`, plus tolerant parsing of `span` events. Unit tests cover the `answer_required` gate, the answer argv, the bridge's question-aware pause message and the memoized API probe.
|
|
58
|
+
- Continuation accounting reuses the existing durable recovery ledger. Replayed events do not duplicate usage; unavailable cost or token values remain unknown. Recovery also handles interrupted continuations before accepting another one.
|
|
59
|
+
- Core's existing CI matrix runs the new tests on macOS, Windows and Linux with Node 20.19.0, 22 and 24. Desktop adds focused runtime integration coverage on macOS and Windows. These remote jobs have not been dispatched from this session.
|
|
60
|
+
|
|
61
|
+
## Remaining release and operational limits
|
|
62
|
+
|
|
63
|
+
1. **Changes are local source, not a published release.** The production Desktop registry lock still pins Core 5.1.1; the local Core checkout identifies itself as 5.1.0. Publish and pin a reviewed paired Core release before distributing Desktop. Do not downgrade an activated managed Core installation to try this source feature; use the documented explicit development runtime override.
|
|
64
|
+
2. **Native Windows and live paid-provider execution remain unverified here.** Windows paths, process shims and protocol contracts have automated coverage, but native Windows execution must pass on its runner. No paid inference was invoked during implementation.
|
|
65
|
+
3. **Enable projects explicitly.** The runtime applies to implementation rail steps. Mission chat and other unrelated AI features keep their existing transports. A role still needs central model instructions; workflow ordering, gates, retries and archive are code-owned.
|
|
66
|
+
4. **Resume preserves host delivery boundaries.** A resumed Core workflow finishes in the original worktree. It does not automatically restart the former rail's PR/backlog delivery phase. The normal uninterrupted rail retains its existing delivery flow.
|
|
67
|
+
5. **Capabilities differ by provider.** Native dollar caps are available only for the built-in Claude executor. Kimi has no authoritative usage counters, and old ACP versions cannot expose every multi-repository read-only scope. Gemini read-only roles require enforceable native admin policies. Unsupported configurations fail explicitly. Local endpoints need function tool support.
|
|
68
|
+
6. **Archive uses complete reviewed specifications.** It replaces the corresponding main specification documents; it does not merge partial OpenSpec deltas.
|
|
69
|
+
|
|
70
|
+
A production dependency audit also found two high-severity advisories in the pre-existing `fast-uri` 3.1.0 and `js-yaml` 4.1.1 entries. Those exact versions were already in Core's original lockfile; the new runtime dependencies did not introduce them. This implementation does not claim a clean baseline dependency audit.
|
|
71
|
+
|
|
72
|
+
See [setup and recovery](programmatic-agent-runtime.md), [architecture evaluation](agent-runtime-framework-evaluation.md), and the paired Core checkout's `docs/agent-runtime.md` for configuration and programmatic APIs.
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
# Programmatic agent runtime
|
|
2
|
+
|
|
3
|
+
Desktop can hand implementation to Core's **runtime API 1**. Core runs architect, developer, deterministic verification, reviewer and archive as separate LangGraph phases. Desktop keeps project/worktree selection, rail lifecycle, logs, accounting and delivery ownership.
|
|
4
|
+
|
|
5
|
+
This is the only implementation engine in the current source tree. It applies to implementation rail steps and their Core completion check. Mission chat and unrelated AI features keep their existing transports. Provider-native implementation prompts and skills are not invoked inside the programmatic phases.
|
|
6
|
+
|
|
7
|
+
## Build the paired source
|
|
8
|
+
|
|
9
|
+
Use Node **20.19.0+** for Core; Node **22.22.3** matches Desktop's native CI runtime. Provider requirements can be higher. Both repositories' dependencies must be installed.
|
|
10
|
+
|
|
11
|
+
```sh
|
|
12
|
+
cd ../specrails-core
|
|
13
|
+
npm ci
|
|
14
|
+
npm run build
|
|
15
|
+
npm run check:package
|
|
16
|
+
cd ../specrails-desktop
|
|
17
|
+
npm ci
|
|
18
|
+
npm ci --prefix client
|
|
19
|
+
node scripts/assemble-bundled-core.mjs --source ../specrails-core
|
|
20
|
+
npm run dev
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
The assembly command stages the built Core checkout into `src-tauri/core`, installs its locked production dependency closure, and runs an offline workflow smoke test. It can download npm dependencies; it does not invoke an AI provider. Build Core first and repeat assembly after changing its runtime. Native development uses `npm run dev:desktop` instead of `npm run dev`; stop the existing app first and follow the [native development instructions](../../README.md#develop-from-source).
|
|
24
|
+
|
|
25
|
+
To verify the complete paired execution boundary without paid model calls:
|
|
26
|
+
|
|
27
|
+
```sh
|
|
28
|
+
npm run build:server
|
|
29
|
+
node scripts/smoke-agent-runtime-pair.mjs
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
This runs the compiled Desktop bridge and real bundled Core against a temporary localhost model fixture. It checks tool writes, a real verification subprocess, archive approval, resume, usage and host Git ownership. It uses a temporary repository and deletes it afterward. Pass `--core /absolute/path/to/Core/dist/agent-runtime/index.js` to exercise another built Core checkout.
|
|
33
|
+
|
|
34
|
+
For web development, an explicit runtime override can point to the built module:
|
|
35
|
+
|
|
36
|
+
```sh
|
|
37
|
+
# macOS shell, from specrails-desktop
|
|
38
|
+
export SPECRAILS_CORE_RUNTIME_PATH="$PWD/../specrails-core/dist/agent-runtime/index.js"
|
|
39
|
+
npm run dev
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
```powershell
|
|
43
|
+
# PowerShell, from specrails-desktop
|
|
44
|
+
$env:SPECRAILS_CORE_RUNTIME_PATH = (Resolve-Path ../specrails-core/dist/agent-runtime/index.js).Path
|
|
45
|
+
npm run dev
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
Use this override when an activated managed or globally installed Core is newer than the source bundle and does not yet expose API 1. Merely assembling an older-version checkout does not override Desktop's selected installation. Keep its installation/lifecycle selection intact while testing the explicitly selected execution module.
|
|
49
|
+
|
|
50
|
+
The override identifies **index.js**, not the package directory or CLI file. Runtime discovery checks it first, then the existing Core resolver's selected installation, including activated managed updates. An authoritative managed, override or bundled installation is not silently replaced when its runtime is missing or incompatible. Production does not fall back to a sibling checkout. Development can also resolve an installed `specrails-core/agent-runtime` export or a sibling Core build when no authoritative runtime is available.
|
|
51
|
+
|
|
52
|
+
Desktop negotiates `runtime api` and sends configuration to `runtime validate --stdin` through its bundled/system Node interpreter. This keeps the Core ESM runtime outside Desktop's CommonJS/pkg process and avoids relying on unsupported dynamic imports inside the native sidecar.
|
|
53
|
+
|
|
54
|
+
`SPECRAILS_CORE_RUNTIME_PATH` selects this execution module. The existing `SPECRAILS_CORE_BIN` controls the installation/lifecycle resolver and is a different setting. Prefer a paired source bundle when testing the complete installation and execution flow.
|
|
55
|
+
|
|
56
|
+
## Configure a project
|
|
57
|
+
|
|
58
|
+
1. Open **Project settings → Agent runtime** (its own section in the project dialog).
|
|
59
|
+
2. Choose a provider per role. The model dropdown lists each CLI's catalog with the default marked; turns, attempts and timeout show their defaults in the fields. Claude, Codex, Gemini and Kimi remain available; open **General settings → Specrails Agents → Provider connections** to add an OpenAI-compatible endpoint for a local or remote model (local endpoints do not require a key; API providers need an explicit model).
|
|
60
|
+
3. Verification commands are optional. A project that never saved runtime settings is prefilled with the checks Desktop detects offline (`package.json` test/type-check/lint scripts with the right package manager, Cargo, Go, pytest, Gradle, Maven, .NET, Make); **Detect project checks** re-runs that detection. Each row is a repository plus one command line (`npm run test -- --strict`; quotes group arguments). Leave the list empty and the architect proposes the project's own checks on each run; repositories with no automated check are still reviewed and recorded as unverified in Core's receipt.
|
|
61
|
+
4. **Review gate** and **Architect confidence** are optional. Review thresholds (overall score and the five aspects: type correctness, pattern adherence, test coverage, security, architectural alignment) can only *tighten* Core's own gate: the floors are 70 overall, 75 for security and 60 for every other aspect, and Desktop rejects lower values before saving (`Review threshold review.minScore must be at least 70 (Core's own review gate)`). Empty fields omit the key so Core's defaults apply. `architect.onLowConfidence` decides what happens when, after one autonomous investigation pass, the architect's design is still low in confidence: `ask` (default) pauses the run with the architect's question; `proceed` continues on stated assumptions.
|
|
62
|
+
5. Save the project settings. Implementation always uses the agent runtime.
|
|
63
|
+
6. Start an implementation through the normal rail flow.
|
|
64
|
+
|
|
65
|
+
Settings are saved at `<project execution .specrails directory>/agent-runtime.json`. Missing configuration uses the default agent runtime. The retired enabled flag cannot select another engine. Malformed configuration blocks admission; it is not ignored. Saving verifies that Core exposes the expected API. Connections are stored globally in `~/.specrails/runtime-providers.json`; project files retain role references, models, limits and verification. Existing embedded connections migrate once, with stable disambiguated IDs on endpoint conflicts. New runs freeze the resolved configuration, while saved runs retain their original snapshot.
|
|
66
|
+
|
|
67
|
+
Core owns role instructions and permissions. The developer role edits and runs commands inside its CLI sandbox (the same autonomy as the legacy Implement step); architect and reviewer are read-only. A legacy rail profile/model selection does not override the runtime's per-role provider configuration. The JSON schema is [server/schemas/agent-runtime.schema.json](../../server/schemas/agent-runtime.schema.json), mirrored from Core. For a complete configuration, custom executor examples, Kimi capabilities and API tooling details, see [Core's runtime guide](https://github.com/fjpulidop/specrails-core/blob/main/docs/agent-runtime.md) in the paired revision.
|
|
68
|
+
|
|
69
|
+
The built-in Claude adapter supports its native dollar cap. Built-in Codex, Gemini, Kimi and OpenAI-compatible adapters reject `maxCostUsd`; remove that limit for mixed-provider/local runs. Kimi also rejects token caps because its usage is unavailable. Unknown cost/tokens remain unknown in accounting, rather than becoming zero. Attempt, timeout and tool limits remain available. Existing provider services, licenses and inference costs are separate from the free open-source orchestration runtime.
|
|
70
|
+
|
|
71
|
+
Gemini architect/reviewer roles require native `--admin-policy` support. Core applies a temporary read-only tool allowlist that remains effective if user settings disable plan mode. If system policies prevent per-run enforcement, Core rejects that role with an actionable capability error; it does not replace managed policies. The developer role retains its normal editing transport.
|
|
72
|
+
|
|
73
|
+
## State, approvals and continuation
|
|
74
|
+
|
|
75
|
+
The rail creates a frozen Core execution context for its original repository/worktree paths. State is kept below the project's execution `.specrails/pipeline/<runId>/` directory:
|
|
76
|
+
|
|
77
|
+
| File | Purpose |
|
|
78
|
+
| --- | --- |
|
|
79
|
+
| `desktop-context.json` | Selected repositories, original worktrees, frozen scope and ownership |
|
|
80
|
+
| `desktop-runtime-config.json` | Admission snapshot of project settings with verification commands restricted to the selected repositories; project settings remain unchanged |
|
|
81
|
+
| `desktop-runtime-host.json` | Allowlisted host settings needed to reconstruct execution |
|
|
82
|
+
| `agent-runtime-request.json` | Core's frozen configuration and change name |
|
|
83
|
+
| `state.json`, `receipts/` | Core gates and verification receipts |
|
|
84
|
+
| `agent-workflow/<runId>/checkpoint.json` | Durable phases, attempts, approvals and usage |
|
|
85
|
+
|
|
86
|
+
A run pauses (Core exit code 2) either on an **approval** (`pendingApproval`, for example before archive) or on a **question** (`pendingQuestion: { stepId, requestedAt, question }`) when the architect is configured to ask on low confidence. A pending question can only be resumed together with an answer: the runs panel shows the question with a textarea and an **Answer and resume** action, which posts `{ "answer": "…" }` (nonempty, at most 20,000 characters). Resuming without an answer while a question is open is rejected with `400 answer_required`. Once Core records `answeredAt`, the question is history and ordinary resume applies again. The bridge reports a paused run's reason in the job log: awaiting approval, or the pending question text.
|
|
87
|
+
|
|
88
|
+
**Implementation cards and job detail** expose resume, archive approval, question answering, interrupted-step recovery and continuation cancellation actions next to the work. Cards query their latest job by original rail identity; job detail queries its exact run. Legacy jobs render no runtime panel. **Jobs → Saved executions** retains the cross-run history. Project settings contain runtime configuration only. A continuation resumes from the phase shown, in the original worktree, and writes its progress into that job's log (a `[runtime] continuation started from phase …` banner, tool activity, phase notes and the final outcome). Resuming a run that stopped at the developer attempt limit grants a fresh attempt budget. Wait for the original rail execution to settle before resuming. Active rail jobs are stopped through their job controls; the continuation's Cancel action owns only continuations started from this panel.
|
|
89
|
+
|
|
90
|
+
Resume retains valid completed phases and rechecks Core evidence. Changed code or environment requires fresh verification/review. An ambiguous interrupted write requires an explicit recovery action after inspecting partial changes. A changed frozen config/identity requires a new run. Missing original worktrees or mismatched execution manifests block recovery; the controller never invents a replacement worktree.
|
|
91
|
+
|
|
92
|
+
Saved executions link directly to the original job log. Continuations broadcast readable logs and raw runtime events to that job, publish `runtime.continuation` lifecycle updates, and reserve the original implementation card until settlement. Jobs list/detail project the continuation as running while it is active (including the running filter); the failed attempt remains in history. Job and card cancellation route to the continuation process. After a successful continuation, Desktop revalidates Core receipts, commits the exact preserved worktrees locally, and reconciles the job and repository delivery cards to completed / ready for review. A new completion event supersedes the earlier failed summary while retaining the failed attempt and its usage. The activity reservation then clears.
|
|
93
|
+
|
|
94
|
+
**A continuation prepares local delivery for review after Core succeeds; it does not create a PR or merge code.** Use the normal delivery card actions to review and publish. Older successful continuations expose **Prepare delivery**, which performs the same receipt validation and local settlement without invoking a model. This also applies when approval is granted after the original rail has settled. Core archive success is not a claim that a PR was created. The initial uninterrupted successful rail retains its normal host delivery flow.
|
|
95
|
+
|
|
96
|
+
Disabling project runtime settings affects future admission. Existing runs retain their frozen runtime request and remain available for explicit continuation; they do not switch back to a platform prompt.
|
|
97
|
+
|
|
98
|
+
## API and logs
|
|
99
|
+
|
|
100
|
+
All routes are under `/api/projects/:projectId`:
|
|
101
|
+
|
|
102
|
+
| Method/path | Result |
|
|
103
|
+
| --- | --- |
|
|
104
|
+
| `GET /agent-runtime/config` | Saved/default config and runtime availability |
|
|
105
|
+
| `PUT /agent-runtime/config` | Validate and atomically save configuration |
|
|
106
|
+
| `GET /agent-runtime/runs` | Recent run status, `traceId`, pending approval or question, recoverable phases and available controls |
|
|
107
|
+
| `POST /agent-runtime/runs/:runId/resume` | Accept `{}`, `{ "approve": ["archive"] }`, `{ "recover": ["developer"] }`, explicit `invalidate` phase IDs, or `{ "answer": "…" }` for a pending question (required while one is open) |
|
|
108
|
+
| `POST /agent-runtime/runs/:runId/cancel` | Cancel a continuation owned by this controller |
|
|
109
|
+
|
|
110
|
+
Resume responds `202` after admission and continues asynchronously. The valid phase IDs are `architect`, `developer`, `verify`, `reviewer` and `archive`. Core's lease remains the cross-process concurrency guard. A run cannot be resumed while its original Desktop execution is active.
|
|
111
|
+
|
|
112
|
+
Desktop launches `node <Core>/dist/agent-runtime/cli.js` with structured argv (`--approve`, `--recover`, `--invalidate`, `--answer <text>` on resume) and consumes JSON lines for phase events (`workflow-event`), agent output (`agent-event`), verification output, trace spans (`span`: `{ traceId, spanId, name, stepId, attempt, visit, startedAt, endedAt, status, usage?, error? }`) and the terminal `runtime-result`. Spans are stored verbatim as job events for diagnostics and are not narrated in the log. `runtime status --compact` returns `state.traceId`, `pendingApproval`, `pendingQuestion` and per-step `{ status, visits }`. Desktop probes `runtime api` once per Core CLI file revision (path plus mtime/size); `runtime validate --stdin` runs on every save. The job log shows phase transitions (`[runtime] step_started: developer`), live tool activity per role (`[developer] Read src/app.ts`, `[developer] Bash npm test`) and Core's own phase notes (architecture written, verification passed, review approved or corrections requested); the final JSON of architect and reviewer is not echoed. The narrated view (Relato) derives its milestones from the same events: each runtime phase, the tools used, correction loops and a stopped workflow with Core's structural reason. Accounting uses the invocation's new attempts, so resuming a completed phase does not bill its cumulative history twice. Status queries use `--compact`, are read-only and do not invoke providers; accumulated logs stay in Core's checkpoint instead of overflowing the process status response.
|
|
113
|
+
|
|
114
|
+
Provider invocation/cancellation supports native macOS processes and Windows executables/npm shims. Actual provider behavior still depends on the installed CLI version and capabilities. The offline tests cover fake CLI/ACP frames, Windows argv rules, a local HTTP coding fixture and real verification subprocesses; live provider smoke tests and Windows CI remain separate validation.
|
|
115
|
+
|
|
116
|
+
## Rollout and release pairing
|
|
117
|
+
|
|
118
|
+
The committed registry bundle lock and `CORE_BUNDLE_VERSION` currently pin **Core 5.1.1**, a previously published package. That pin does not incorporate these source changes. Source assembly is the supported development route until the paired Core runtime release is available.
|
|
119
|
+
|
|
120
|
+
A production release must:
|
|
121
|
+
|
|
122
|
+
1. Publish a reviewed Core package containing API 1, `dist/agent-runtime/` and its production dependencies.
|
|
123
|
+
2. Update `scripts/assemble-bundled-core.lock.json` and `CORE_BUNDLE_VERSION` together to that exact release, capturing the full dependency integrity closure.
|
|
124
|
+
3. Run Core package checks, Desktop compatibility/package checks and both macOS/Windows native validation before packaging the paired app.
|
|
125
|
+
|
|
126
|
+
Do not relabel an old 5.1.1 bundle or copy only `dist/agent-runtime`: LangGraph and the complete runtime dependency closure are required. Source assembly writes `source-bundle.json` with the Core version, runtime API and lock hash for traceability; it does not publish Core or update the production registry lock.
|
|
127
|
+
|
|
128
|
+
Verify role outputs, delivery ownership and saved-run recovery before releasing the paired app. Implementation has no legacy fallback; profile v1 remains only for other workflows. Core's programmatic archive writes reviewed **complete specification replacements**; it does not merge partial OpenSpec delta snippets. Preserve unchanged requirements in the architect's output and inspect that behavior during the pilot.
|
|
129
|
+
|
|
130
|
+
See [Core runtime selection and recovery](core-runtime-updates.md) for the separate framework-update lifecycle and [the original evaluation](agent-runtime-framework-evaluation.md) for the architecture rationale.
|
|
131
|
+
|
|
132
|
+
The [implementation verification record](programmatic-agent-runtime-validation.md) lists the completed checks and outstanding release validation.
|
|
133
|
+
|
|
134
|
+
## Efficiency metrics
|
|
135
|
+
|
|
136
|
+
Core's additive runtime metrics v1 are passed from compact status to each saved run's optional `metrics` field. The contextual run panels and Jobs history render a collapsed **Usage and time** panel with reported cost, active execution/agent time, calls and token/cache totals, plus per-phase attempts, duration, provider calls and cost. All eight locales include the panel labels.
|
|
137
|
+
|
|
138
|
+
Older Core versions continue to work without the panel. The server validates numeric fields and known phase IDs, drops unsupported/malformed metrics and projects only the supported fields; it never forwards arbitrary transcripts from the metrics object. Missing billing is displayed as unavailable rather than zero. Cache tokens are already included in input tokens, and agent duration already includes native tool work. The panel does not estimate savings or measure implementation quality.
|
|
139
|
+
|
|
140
|
+
The paired `agent-runtime-efficiency` OpenSpec change in specrails-core documents verification ownership, efficient API tools and the measurement contract. Core exposes the same report through `runtime status` / `runtime-result`, so a fixed set of real tasks can be compared without a Desktop database migration.
|
package/package.json
CHANGED
|
@@ -266,6 +266,11 @@ function codexMcpOverrides(entry) {
|
|
|
266
266
|
const args = [
|
|
267
267
|
...c(`mcp_servers.specrails.command=${JSON.stringify(entry.command)}`),
|
|
268
268
|
...c(`mcp_servers.specrails.args=[${entry.args.map((a) => JSON.stringify(a)).join(', ')}]`),
|
|
269
|
+
// This invocation owns a server-minted capability. Specrails authorizes
|
|
270
|
+
// each action against its mission tier; Codex's tool-level prompt cannot
|
|
271
|
+
// distinguish facade reads from writes and fails under approvalPolicy=never.
|
|
272
|
+
// Keep this scoped to the internal bridge, never external MCP servers.
|
|
273
|
+
...c('mcp_servers.specrails.default_tools_approval_mode="approve"'),
|
|
269
274
|
];
|
|
270
275
|
for (const [k, v] of Object.entries(entry.env)) {
|
|
271
276
|
args.push(...c(`mcp_servers.specrails.env.${k}=${JSON.stringify(v)}`));
|
|
@@ -79,8 +79,8 @@ repository members, and runs coordinated AI coding pipelines over their selected
|
|
|
79
79
|
routing). Supported by Claude and Kimi; forced to null on Codex/Gemini rails.
|
|
80
80
|
- **Provider / engine** — claude, codex, gemini or kimi. A project installs one
|
|
81
81
|
or more; AI-spawning actions may pick any installed one. Claude and Kimi
|
|
82
|
-
support profiles and Freestyle; Contract Refine
|
|
83
|
-
|
|
82
|
+
support profiles and Freestyle; Contract Refine supports Claude and Codex
|
|
83
|
+
(read-only). SMASH and persistent interactive jobs remain Claude-only.
|
|
84
84
|
Cost is authoritative on Claude, estimated (~) on Codex/Gemini, and
|
|
85
85
|
unavailable when Kimi does not report it.
|
|
86
86
|
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.parseProgrammaticUsage = parseProgrammaticUsage;
|
|
4
|
+
const SETTLED_STEPS = new Set(['step_succeeded', 'step_failed', 'step_blocked', 'step_paused', 'step_interrupted']);
|
|
5
|
+
const object = (value) => value !== null && typeof value === 'object' && !Array.isArray(value);
|
|
6
|
+
const known = (value) => typeof value === 'number' && Number.isFinite(value) && value >= 0 ? value : undefined;
|
|
7
|
+
/** Recover one Core invocation's accounting from raw stdout event payloads.
|
|
8
|
+
* A terminal invocation total already includes its phase events and wins over
|
|
9
|
+
* them. Without that frame, an unfinished attempt makes inclusive totals
|
|
10
|
+
* unknown: costs from completed phases alone would be misleadingly precise. */
|
|
11
|
+
function parseProgrammaticUsage(rawRows) {
|
|
12
|
+
let recognized = false;
|
|
13
|
+
let terminal;
|
|
14
|
+
const seen = new Set();
|
|
15
|
+
const started = new Set();
|
|
16
|
+
const settled = new Set();
|
|
17
|
+
const usages = [];
|
|
18
|
+
for (const row of rawRows) {
|
|
19
|
+
let frame;
|
|
20
|
+
try {
|
|
21
|
+
frame = JSON.parse(row);
|
|
22
|
+
}
|
|
23
|
+
catch {
|
|
24
|
+
continue;
|
|
25
|
+
}
|
|
26
|
+
if (!object(frame))
|
|
27
|
+
continue;
|
|
28
|
+
if (frame.type === 'runtime-result') {
|
|
29
|
+
recognized = true;
|
|
30
|
+
terminal = frame;
|
|
31
|
+
continue;
|
|
32
|
+
}
|
|
33
|
+
if (frame.type === 'agent-event') {
|
|
34
|
+
recognized = true;
|
|
35
|
+
continue;
|
|
36
|
+
}
|
|
37
|
+
if (frame.type !== 'workflow-event' || !object(frame.event))
|
|
38
|
+
continue;
|
|
39
|
+
const event = frame.event;
|
|
40
|
+
if (typeof event.id !== 'string' || typeof event.type !== 'string')
|
|
41
|
+
continue;
|
|
42
|
+
recognized = true;
|
|
43
|
+
if (seen.has(event.id))
|
|
44
|
+
continue;
|
|
45
|
+
seen.add(event.id);
|
|
46
|
+
const attempt = typeof event.attemptId === 'string' ? event.attemptId : typeof event.stepId === 'string' ? event.stepId : event.id;
|
|
47
|
+
if (event.type === 'step_started')
|
|
48
|
+
started.add(attempt);
|
|
49
|
+
if (SETTLED_STEPS.has(event.type)) {
|
|
50
|
+
settled.add(attempt);
|
|
51
|
+
usages.push(object(event.usage) ? event.usage : {});
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
if (!recognized)
|
|
55
|
+
return undefined;
|
|
56
|
+
let usage = {};
|
|
57
|
+
if (object(terminal?.invocationUsage)) {
|
|
58
|
+
usage = terminal.invocationUsage;
|
|
59
|
+
}
|
|
60
|
+
else if (usages.length > 0 && [...started].every((attempt) => settled.has(attempt))) {
|
|
61
|
+
for (const key of ['costUsd', 'inputTokens', 'outputTokens']) {
|
|
62
|
+
const values = usages.map((item) => known(item[key]));
|
|
63
|
+
if (values.every((value) => value !== undefined))
|
|
64
|
+
usage[key] = values.reduce((sum, value) => sum + value, 0);
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
const cost = known(usage.costUsd), tokensIn = known(usage.inputTokens), tokensOut = known(usage.outputTokens);
|
|
68
|
+
return {
|
|
69
|
+
provider: 'agent-runtime', model: 'per-role', cost, tokensIn, tokensOut,
|
|
70
|
+
tokens: tokensIn === undefined || tokensOut === undefined ? undefined : known(tokensIn + tokensOut),
|
|
71
|
+
estimated: cost === undefined, failed: terminal?.status !== 'succeeded',
|
|
72
|
+
};
|
|
73
|
+
}
|