@hecer/yoke 1.22.0 → 1.23.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/.codex-plugin/plugin.json +1 -1
- package/CHANGELOG.md +21 -0
- package/README.md +3 -1
- package/TODOS.md +6 -0
- package/canon/manifest.yaml +1 -1
- package/canon/skills/visual-verification/SKILL.md +25 -2
- package/canon/tools/codex-rtk-hook.mjs +6 -16
- package/dist/cli.js +79 -0
- package/dist/code-intelligence/adapters/mcp.js +1 -0
- package/dist/code-intelligence/coordinator.js +3 -1
- package/dist/code-intelligence/index.js +1 -0
- package/dist/code-intelligence/mcp-client.js +15 -4
- package/dist/code-intelligence/mcp-server.js +4 -1
- package/dist/code-intelligence/preflight.js +71 -0
- package/dist/loop/cache-isolation.js +36 -0
- package/dist/loop/loop.js +24 -4
- package/dist/loop/parallel-adapters.js +35 -2
- package/dist/loop/parallel-command.js +17 -2
- package/dist/loop/proof-retention.js +70 -0
- package/dist/loop/reporter.js +1 -1
- package/dist/loop/run-command.js +6 -1
- package/dist/loop/runner.js +1 -1
- package/dist/loop/worker.js +7 -0
- package/dist/observability/history.js +1 -0
- package/dist/observability/local-report.js +120 -0
- package/dist/observability/usage.js +2 -0
- package/dist/retrofit/config.js +3 -1
- package/dist/retrofit/gitignore.js +10 -0
- package/dist/retrofit/planners/codex.js +20 -20
- package/dist/routing/router.js +2 -0
- package/dist/smoke/command.js +87 -13
- package/dist/update/check.js +1 -1
- package/docs/DELIVERY-JOURNEYS.md +8 -1
- package/docs/RELEASE-VALIDATION-1.23.0.md +39 -0
- package/docs/benchmarks/2026-10-04-efficiency/ANALYSE.md +182 -0
- package/docs/benchmarks/2026-10-04-efficiency/compare-help.py +55 -0
- package/docs/benchmarks/2026-10-04-efficiency/manifest.json +125 -0
- package/docs/benchmarks/2026-10-04-efficiency/provenance-analysis.json +90 -0
- package/docs/benchmarks/2026-10-04-efficiency/provenance-design.json +90 -0
- package/docs/benchmarks/2026-10-04-efficiency/provenance-original-report.json +90 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/DEVELOPMENT_ANALYSIS.md +142 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/RESULT.md +21 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/commands.jsonl +26 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/environment.json +31 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/final-yoke-smoke.json +40 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/model-purpose-hints.csv +19 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/observations.jsonl +21 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/observer-command-phases.csv +12 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/roles.csv +5 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/shell-categories.csv +8 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/stories.csv +8 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/summary.json +469 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-history.jsonl +104 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-1.log +58 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-2.log +29 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-3.log +5 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-4.log +12 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-phases.csv +10 -0
- package/docs/benchmarks/2026-10-04-efficiency/regression-comparison.json +104 -0
- package/docs/superpowers/plans/2026-10-04-yoke-1.23-efficiency-prd.json +11 -0
- package/docs/superpowers/plans/2026-10-04-yoke-1.23-efficiency.md +83 -0
- package/docs/superpowers/specs/2026-10-04-yoke-1.23-efficiency-design.md +120 -0
- package/gemini-extension.json +1 -1
- package/package.json +1 -1
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
{
|
|
2
|
+
"answers": {
|
|
3
|
+
"located": "NO",
|
|
4
|
+
"scan_complete": "YES",
|
|
5
|
+
"trusted": "UNKNOWN",
|
|
6
|
+
"verified": "UNKNOWN"
|
|
7
|
+
},
|
|
8
|
+
"asset": "G:\\NN-Developed\\Worktrees\\yoke-1.23-efficiency\\docs\\benchmarks\\2026-10-04-efficiency\\raw\\DEVELOPMENT_ANALYSIS.md",
|
|
9
|
+
"asset_sha256": "f1f7ba9f901376809bddf43bfb4a3f0b2f9fe9d8f4893aa1fd8335aac6bd5dc0",
|
|
10
|
+
"components": [
|
|
11
|
+
{
|
|
12
|
+
"detail": "Retain ABSENT as a bounded local observation for this fully inspected file.",
|
|
13
|
+
"ran": true,
|
|
14
|
+
"skill": "inspect-content-provenance",
|
|
15
|
+
"summary": "presence=ABSENT"
|
|
16
|
+
},
|
|
17
|
+
{
|
|
18
|
+
"detail": "No verifier was supplied; cryptographic questions cannot be answered.",
|
|
19
|
+
"ran": false,
|
|
20
|
+
"skill": "verify-content-credentials",
|
|
21
|
+
"summary": "not run"
|
|
22
|
+
},
|
|
23
|
+
{
|
|
24
|
+
"detail": "Format TEXT has no supported metadata parser in this build, so no privacy conclusion can be drawn.",
|
|
25
|
+
"ran": true,
|
|
26
|
+
"skill": "audit-metadata-privacy",
|
|
27
|
+
"summary": "risk=UNKNOWN"
|
|
28
|
+
},
|
|
29
|
+
{
|
|
30
|
+
"detail": null,
|
|
31
|
+
"ran": true,
|
|
32
|
+
"skill": "detect-text-watermark",
|
|
33
|
+
"summary": "status=NO_SIGNAL_OBSERVED"
|
|
34
|
+
}
|
|
35
|
+
],
|
|
36
|
+
"evidence": {
|
|
37
|
+
"integrity": "UNKNOWN",
|
|
38
|
+
"manifest": {
|
|
39
|
+
"actions": [],
|
|
40
|
+
"assertion_labels": [],
|
|
41
|
+
"claim_generator": null,
|
|
42
|
+
"format": null,
|
|
43
|
+
"ingredient_count": 0,
|
|
44
|
+
"signature_alg": null,
|
|
45
|
+
"signature_issuer": null,
|
|
46
|
+
"signature_time": null,
|
|
47
|
+
"title": null
|
|
48
|
+
},
|
|
49
|
+
"manifest_presence": "ABSENT",
|
|
50
|
+
"markers": [],
|
|
51
|
+
"signer_trust": "UNKNOWN"
|
|
52
|
+
},
|
|
53
|
+
"format": "TEXT",
|
|
54
|
+
"limitations": [
|
|
55
|
+
"This is a composition of the underlying analyzers; it performs no new analysis.",
|
|
56
|
+
"No answer here is an authorship classification.",
|
|
57
|
+
"A trust answer is only meaningful relative to the named trust policy."
|
|
58
|
+
],
|
|
59
|
+
"privacy_risk": "UNKNOWN",
|
|
60
|
+
"reason": null,
|
|
61
|
+
"schema_version": "2.0",
|
|
62
|
+
"text_signals": {
|
|
63
|
+
"did_not_run": [
|
|
64
|
+
"anthropic-official",
|
|
65
|
+
"kgw-research",
|
|
66
|
+
"synthid-text"
|
|
67
|
+
],
|
|
68
|
+
"scan_complete": true,
|
|
69
|
+
"status": "NO_SIGNAL_OBSERVED"
|
|
70
|
+
},
|
|
71
|
+
"tool": "audit-provenance",
|
|
72
|
+
"trust_policy": null,
|
|
73
|
+
"unknowns": [
|
|
74
|
+
{
|
|
75
|
+
"next_step": "Re-run with --c2patool pointing at c2patool 0.20.0 or newer.",
|
|
76
|
+
"question": "verified",
|
|
77
|
+
"why": "No conforming verifier was supplied."
|
|
78
|
+
},
|
|
79
|
+
{
|
|
80
|
+
"next_step": "Re-run with --c2patool pointing at c2patool 0.20.0 or newer.",
|
|
81
|
+
"question": "trusted",
|
|
82
|
+
"why": "No conforming verifier was supplied."
|
|
83
|
+
},
|
|
84
|
+
{
|
|
85
|
+
"next_step": "None available. Do not infer authorship from this.",
|
|
86
|
+
"question": "text_watermark",
|
|
87
|
+
"why": "Keyed model-level watermarks cannot be checked without the provider's key."
|
|
88
|
+
}
|
|
89
|
+
]
|
|
90
|
+
}
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
# NEXUS: Entwicklungsanalyse mit Yoke
|
|
2
|
+
|
|
3
|
+
Messstand: 2026-10-04T09:27:32.214737+00:00 (UTC). Ein realer Benchmark-Lauf; keine Vergleichsmessung gegen direkten Codex. Rohdaten enthalten beobachtbare Metadaten und Messwerte, keine verborgenen Reasoning-Inhalte.
|
|
4
|
+
|
|
5
|
+
## Kompakte Auswertung
|
|
6
|
+
|
|
7
|
+
7/7 Stories bestanden; acht Implementierungsversuche, maximal zwei parallele Worker. Summierte Yoke-Loop-Laufzeit 56,4 Minuten. Implementierungsworker: 11.234.592 Input-Tokens einschließlich 10.621.952 Cache-Tokens, 82.193 Output-Tokens. Alle gemessenen Rollen zusammen zum Cutoff: 28,094,098 Input, davon 26,848,896 Cache-Lesen; 1,245,202 Input ohne Cache und 195,930 Output. Die große kumulierte Input-Zahl zählt wiederholten Kontext bei jedem Aufruf; sie entspricht nicht der Menge einmalig gelesener Inhalte.
|
|
8
|
+
|
|
9
|
+
Die wichtigsten belegten Bremsen waren Einrichtungs-/Kompatibilitätsprobleme, wiederholte Prüfung und Browserarbeit, eine Observer-verursachte Integrationswiederholung und ein erst unabhängig entdeckter Layoutfehler. Shell-Prüfungen der Implementierungsworker beanspruchten summiert 461,3 Prozesssekunden; diese Zeit ist teilweise in den Implementierungsphasen enthalten. Eine belastbare Aufteilung jeder Sekunde in Modelllatenz, Denken und Tool-Wartezeit ist mit den verfügbaren Daten nicht möglich.
|
|
10
|
+
|
|
11
|
+
## Messverfahren und Grenzen
|
|
12
|
+
|
|
13
|
+
- Provider-Tokens stammen aus Yokes History und nativen Codex-Tokenereignissen. Beide Ansichten werden nicht addiert. Eingabetokens enthalten Cache-Lese-Tokens; Reasoning-Ausgabetokens sind eine Teilmenge der Ausgabetokens.
|
|
14
|
+
- Ausgabebytes sind serialisierte Tool- bzw. Shell-Ausgaben, keine Provider-Tokens. Tool-Bytes können Bilddaten und Metadaten enthalten; daraus lässt sich keine Textkontextgröße ableiten. Gezählt werden eindeutige Usage-Ereignisse, keine unabhängig verifizierten HTTP-Requests.
|
|
15
|
+
- Aufteilung nach Story/Rolle ist direkt beobachtbar. Zweck einzelner Modellaufrufe in model-calls.jsonl und Shell-Kategorien ist eine Heuristik aus sichtbaren Tool-Aufrufen; gemischte Befehle sind nicht exakt zerlegbar.
|
|
16
|
+
- Shell-Zeiten sind Prozesslaufzeiten. Hintergrundserver, parallele Worker und Prüfungen können sich überlappen; ihre Summe ist keine Entwicklungs-Gesamtzeit. Verschachtelte Yoke-Phasen ebenfalls nicht doppelt addieren.
|
|
17
|
+
- RTK-Savings und Code-Intelligence-Tokenbudgets sind Schätzungen des Tools, keine zusätzlich gemessenen Provider-Tokens oder Geldbeträge. Geldkosten fehlen und werden als unbekannt geführt.
|
|
18
|
+
- Coordinator enthält Einrichtung, laufende Kommunikation, Instrumentierung und eigenständige Kontrolle. Dieser zusätzliche Messaufwand wird separat ausgewiesen. Werte enden am Messstand; spätere Aufrufe und Schlussantwort sind nicht enthalten.
|
|
19
|
+
|
|
20
|
+
## Messumfang, Umgebung und Instrumentierungsaufwand
|
|
21
|
+
|
|
22
|
+
Seit Start des 10-Sekunden-Observers bis zum Messstand: 91.7 Minuten. Das ist die erfasste verstrichene Zeit einschließlich Einrichtung, Unterbrechungen, Kontrolle und Analyse; kein reiner Produktentwicklungswert. Der Observer wurde vor der finalen Auswertung gestoppt. Sein letzter CPU-/RSS-Snapshot steht in observer-resources-final.txt.
|
|
23
|
+
|
|
24
|
+
Yoke 1.22.0; codex-cli 0.160.0; rtk 0.51.0; Runner gpt-6.1-sol / medium. Isolierte Worktrees, automatische Parallelität und Entscheidungen aktiv. Kein --explore. Code-Intelligence-Facade in ACTIVE: MCP-Handshake erfolgreich, tatsächliche semantische Backends graft/graphify/serena fehlen. Ein semantischer Effizienzgewinn wurde deshalb nicht nachgewiesen.
|
|
25
|
+
|
|
26
|
+
RTK-Datenbank für Root und sämtliche projektbezogenen Worktrees: 42 Befehle; geschätzte Input-/Output-Tokens 13.752/11.635; geschätzte Einsparung 2.117 (15.4 %). Diese lokalen Schätzungen betreffen registrierte RTK-Befehle und beweisen keinen prozentualen Rückgang des gesamten Modellverbrauchs. Native automatische Umschreibung verschachtelter Code-Mode-Aufrufe war nicht nachweisbar; explizite RTK-Nutzung ist ab den letzten Workern belegt.
|
|
27
|
+
|
|
28
|
+
| Instrumentierte Befehlsphase | Befehle | Fehler | Wall s | Child CPU s |
|
|
29
|
+
|---|---:|---:|---:|---:|
|
|
30
|
+
| environment | 4 | 1 | 0.9 | 0.6 |
|
|
31
|
+
| code-intelligence | 3 | 0 | 2.0 | 1.3 |
|
|
32
|
+
| planning | 5 | 1 | 61.4 | 11.0 |
|
|
33
|
+
| dependency-setup | 2 | 1 | 10.0 | 6.8 |
|
|
34
|
+
| yoke-smoke | 1 | 1 | 0.4 | 0.3 |
|
|
35
|
+
| execution | 4 | 2 | 3384.7 | 1277.9 |
|
|
36
|
+
| recovery | 1 | 0 | 0.5 | 0.3 |
|
|
37
|
+
| dev-server | 1 | 1 | 0.8 | 0.5 |
|
|
38
|
+
| browser-inspection | 2 | 0 | 14.3 | 5.4 |
|
|
39
|
+
| final-validation | 1 | 0 | 20.0 | 30.0 |
|
|
40
|
+
| final-yoke-smoke | 2 | 2 | 5.5 | 4.2 |
|
|
41
|
+
|
|
42
|
+
Dies umfasst nur über measure.py gestartete Prozesse. Child CPU kann überlappende Unterprozesse einschließen; Messskript-Ausführung, Tool-Roundtrips und Provider-Latenz sind nicht vollständig getrennt. Der Coordinator-Tokenverbrauch umfasst sowohl nötige Orchestrierung als auch zusätzliche Messarbeit; diese sind rückwirkend nicht exakt auseinanderzurechnen.
|
|
43
|
+
|
|
44
|
+
## Tokenverbrauch nach beobachtetem Zweck
|
|
45
|
+
|
|
46
|
+
Die folgende Aufteilung ist ausdrücklich eine Heuristik anhand des zuletzt sichtbaren Tool-Aufrufs. Ein Aufruf kann Lesen, Editieren und Prüfen verbinden. Werte sind keine exakte Trennung zwischen Denkzeit, Schreiben und Tool-Ergebnisverarbeitung. Vollständige Rolle/Zweck-Matrix: model-purpose-hints.csv.
|
|
47
|
+
|
|
48
|
+
| Implementierungsworker: Zweckhinweis | Usage-Ereignisse | Input | Cache | Output |
|
|
49
|
+
|---|---:|---:|---:|---:|
|
|
50
|
+
| discovery | 35 | 2.015.491 | 1.824.896 | 6.669 |
|
|
51
|
+
| checks | 63 | 3.377.368 | 3.130.624 | 44.163 |
|
|
52
|
+
| code_intelligence | 6 | 263.056 | 252.160 | 785 |
|
|
53
|
+
| dependency_setup | 6 | 270.273 | 244.480 | 2.947 |
|
|
54
|
+
| dev_server_or_mixed | 12 | 809.495 | 797.184 | 4.029 |
|
|
55
|
+
| waiting_or_polling | 26 | 1.668.311 | 1.642.752 | 1.869 |
|
|
56
|
+
| browser_validation | 41 | 2.830.598 | 2.729.856 | 21.731 |
|
|
57
|
+
|
|
58
|
+
### Direkt belegte Nacharbeit
|
|
59
|
+
|
|
60
|
+
Eine zusätzliche STORY-5-Implementierungsphase entstand durch den Observer-Commit während einer Integration. STORY-7 ist eine zusätzliche Reparatur innerhalb des ursprünglichen Produktscopes: Die unabhängige Prüfung fand trotz bestandener erster Gates 938px Desktop-Höhe bei 900px Viewport. Nach Reparatur misst der korrekt zugeordnete Root-Server 900px; Eventpanel-Unterkante 884px. Beide Messstände und Server-Provenienz sind archiviert. Fehlmessungen gegen den fremden Server wurden ausdrücklich invalidiert.
|
|
61
|
+
|
|
62
|
+
Guardian-Sessions sind separat ausgewiesen, weil sie in der Yoke-Story-Tokenansicht nicht enthalten waren. Für Kapazitäts-/Kostenplanung muss Yoke diese Kontrollkosten zusätzlich sichtbar machen. Keine erfundenen Geldkosten; keine Gegenrechnung von Cache-Lesetokens als kostenlos.
|
|
63
|
+
|
|
64
|
+
## Tokens nach Rolle
|
|
65
|
+
|
|
66
|
+
| Rolle | Sessions gemessen/gesamt | Modellaufrufe | Input inkl. Cache | Cache gelesen | Input ohne Cache | Output |
|
|
67
|
+
|---|---:|---:|---:|---:|---:|---:|
|
|
68
|
+
| guardian | 7/7 | 48 | 1.281.675 | 1.073.152 | 208.523 | 5.404 |
|
|
69
|
+
| implementation | 8/8 | 189 | 11.234.592 | 10.621.952 | 612.640 | 82.193 |
|
|
70
|
+
| coordinator | 1/1 | 127 | 15.525.113 | 15.139.456 | 385.657 | 107.118 |
|
|
71
|
+
| planner_or_review | 2/2 | 2 | 52.718 | 14.336 | 38.382 | 1.215 |
|
|
72
|
+
|
|
73
|
+
## Stories und Zeit
|
|
74
|
+
|
|
75
|
+
| Story | bestanden | Implementierungsaufrufe | Implementierung s | Gate-Prüfung s | Integration s | Input | Cache | Output |
|
|
76
|
+
|---|---|---:|---:|---:|---:|---:|---:|---:|
|
|
77
|
+
| STORY-1 | True | 1 | 565.9 | 3.1 | 6.3 | 1.435.585 | 1.314.176 | 12.725 |
|
|
78
|
+
| STORY-2 | True | 1 | 367.4 | 2.8 | 8.7 | 698.268 | 643.968 | 8.663 |
|
|
79
|
+
| STORY-3 | True | 1 | 696.3 | 5.2 | 9.2 | 2.080.455 | 1.994.752 | 15.192 |
|
|
80
|
+
| STORY-4 | True | 1 | 505.8 | 4.1 | 11.8 | 921.789 | 857.088 | 12.190 |
|
|
81
|
+
| STORY-5 | True | 2 | 566.1 | 11.6 | 35.7 | 1.347.301 | 1.245.568 | 11.899 |
|
|
82
|
+
| STORY-6 | True | 1 | 683.2 | 28.8 | 28.3 | 2.950.211 | 2.853.888 | 13.068 |
|
|
83
|
+
| STORY-7 | True | 1 | 484.2 | 18.4 | 0.0 | 1.800.983 | 1.712.512 | 8.456 |
|
|
84
|
+
|
|
85
|
+
Abgeschlossene Loop-Prozesse: 4, davon fehlgeschlagen: 2. Summierte Loop-Laufzeit: 3384.7 s. Summierte Implementierungsphasen: 3868.9 s; Vereinigungsdauer dieser Intervalle: 3184.7 s. Beobachtete überlappende Worker-Zeit: 684.2 s. Dies ist keine gemessene Beschleunigung gegenüber einem seriellen Kontrolllauf.
|
|
86
|
+
Parallelitätsstichproben: 530, Intervall 10 s, beobachtete Spitze 2 Implementierungsworker und 2 gemeinsame Einheiten. Kurze Spitzen zwischen Stichproben können fehlen.
|
|
87
|
+
|
|
88
|
+
## Shell-Arbeit der Implementierungsworker
|
|
89
|
+
|
|
90
|
+
| Kategorie (heuristisch) | Aufrufe | Exit != 0 / abgebrochen | Prozesslaufzeit s | Ausgabebytes |
|
|
91
|
+
|---|---:|---:|---:|---:|
|
|
92
|
+
| discovery | 73 | 12 | 32.2 | 452.893 |
|
|
93
|
+
| dependency_setup | 10 | 5 | 114.5 | 6.349 |
|
|
94
|
+
| checks | 92 | 25 | 461.3 | 420.588 |
|
|
95
|
+
| browser_validation | 23 | 9 | 103.3 | 109.393 |
|
|
96
|
+
| dev_server_or_mixed | 12 | 12 | 677.8 | 4.534 |
|
|
97
|
+
| other | 17 | 3 | 86.7 | 4.748 |
|
|
98
|
+
| file_editing | 1 | 0 | 0.3 | 0 |
|
|
99
|
+
|
|
100
|
+
Fehlgeschlagene Testbefehle umfassen bewusst rote TDD-Tests. Sie sind nicht automatisch Produktfehler oder zusätzliche Yoke-Story-Versuche. dev_server_or_mixed enthält langlebige bzw. gemischte Befehle.
|
|
101
|
+
|
|
102
|
+
## Konkrete Findings und Verbesserungen
|
|
103
|
+
|
|
104
|
+
| ID | Beobachtung | Auswirkung / Verbesserung |
|
|
105
|
+
|---|---|---|
|
|
106
|
+
| CLI-HELP-MUTATION | yoke setup --help created 95 files and enabled loop; yoke retrofit --help then disabled loop | Read-only discovery mutated configuration and required explicit correction Handle help before command dispatch; reject unknown flags before mutation |
|
|
107
|
+
| RTK-INIT-FLAGS | rtk init --codex --auto-patch exits 1: cannot be combined | One failed setup call Document per-runner incompatible flags and expose compatibility validation |
|
|
108
|
+
| CI-WORKSPACE-ID | code_context rejects absolute path and configured codeIntelligence.workspaceId; MCP server computes hash workspace ID and ignores configured workspaceId | Two failed probes before reading implementation to determine identifier Expose workspace identifier in initialization/tool listing, honor config or document discovery method |
|
|
109
|
+
| CI-BACKENDS-MISSING | graft, graphify and serena executables absent on PATH | Facade can be enabled but semantic/structural backend functionality unavailable Add preflight that distinguishes configured active mode from operational backend coverage and provides pinned installation guidance |
|
|
110
|
+
| RTK-DUPLICATE-HOOKS | Yoke legacy hook and rtk init native Codex hook both present; rtk hook check is supported (exit 0) | Double hook registration observed; rewrite effectiveness with actual Codex tool schema needs separate measurement Use idempotent native Codex hook integration and provide actual rewrite/adoption telemetry |
|
|
111
|
+
| COMPACT-STATUS-MISSING | yoke loop status --compact prints story=none phase=undefined while loop-status.json reports active STORY-1 implementing | Compact progress unsuitable for reliable supervision; detailed status file required Summarize parallel workers in compact status and avoid undefined values |
|
|
112
|
+
| WRITE-SCOPE-DRIFT | STORY-2 declared src/simulation and tests/simulation.test.ts but commit writes src/simulation.ts, src/useSimulation.ts, src/types.ts, tests/simulation.test.tsx; scheduler scopes are advisory | Declared non-overlap does not prove actual non-overlap; shared type file changed during parallel topology work Compare candidate changes against writes declarations, surface deviations and recheck collision risk before integration |
|
|
113
|
+
| SMOKE-MISDIAGNOSIS | Yoke flow-smoke reports Playwright not found while node_modules/playwright/package.json exists; launchPlaywright catches both import and chromium.launch errors and returns null | Package absence and browser launch/sandbox failures have the same actionable error; worker reads harness source to diagnose Return structured import/browser-binary/sandbox/launch failure causes and preserve underlying error |
|
|
114
|
+
| EPHEMERAL-PROOF | After isolated STORY-1/2 integration, target .yoke/proof directory absent; earlier worker smoke screenshot was in removed worktree | Intermediate manual proof lost on cleanup; root must preserve required screenshots in tracked artifact path Copy and content-bind selected verification artifacts to target before deleting isolated worktrees |
|
|
115
|
+
| RTK-LEGACY-ALLOW | Yoke-generated rtk.mjs emits updatedInput without permissionDecision allow; official RTK Codex documentation requires allow for replacement to take effect | Legacy transparent command rewriting incompatible with documented Codex response contract; native RTK integration is also registered but worker adoption not observed Use current native rtk hook codex; verify an actual verbose Codex command records compressed output and RTK history |
|
|
116
|
+
| RTK-NATIVE-CORRECTION | Removed legacy Yoke hook registration; native rtk hook codex remains sole active hook and manual exact-schema probe rewrites git status correctly | Future workers receive native integration without duplicate response; effectiveness measured separately |
|
|
117
|
+
| SHARED-DEPS-SANDBOX | STORY-5 Vitest fails EPERM mkdir at target root node_modules/.vite-temp while executing in isolated worker; tests succeed after worker retries | Dependency reuse caused additional startup failure and approval/escalation path Share immutable package download cache, provide per-worktree writable dependency/runtime cache or explicit allowed paths |
|
|
118
|
+
| NETWORK-RETRY-CACHE | Target npm ci fails ECONNRESET downloading why-is-node-running; npm ci --offline --cache /private/tmp/nexus-npm-cache then installs 105 packages in 4s | At least one failed installation and repeated dependency setup; cached recovery successful Preflight and reuse exact lockfile package cache, classify network failures separately from code failures; n=1 does not establish general speedup |
|
|
119
|
+
| GUARDIAN-USAGE-GAP | Native Codex sessions with source.subagent.other=guardian report token usage separately; Yoke worker token event matches implementation session and excludes those sessions | Harness totals omit observable approval-model usage; independent observer reports it separately Expose approval/guardian call coverage and distinguish implementation usage from full provider-session cost; unknown charge remains unknown |
|
|
120
|
+
| OBSERVER-HEAD-RACE | Observer committed AGENTS.md during STORY-5 integrated gates; Yoke rejected changed target HEAD, retained candidate, and stopped loop | One failed integration and explicit recovery; prior source preserved; recovery invalidates proof and reruns implementation/gates Observer should make harness changes only between loop batches; CLI should expose integration busy state and avoid premature integration complete message |
|
|
121
|
+
| VISUAL-COMPOSITION | INVALIDATED: first observer screenshots came from another pre-existing NEXUS on port 4173; expected root Vite process exited with port-in-use error | Those visual conclusions are excluded from this project assessment Verify owning process cwd, successful server startup and source identity before collecting browser evidence |
|
|
122
|
+
| OBSERVER-WRONG-SERVER | First browser probe navigated occupied port 4173 before checking Vite process had started. Different NEXUS app names and styles exposed mismatch. Target process exit 1 confirmed port conflict. Existing listener left untouched. | One invalid browser run; screenshots and prior visual finding invalidated; its time/tokens remain observer overhead Require startup success plus bound process cwd and served source signature; use unique port and record provenance |
|
|
123
|
+
| RUNTIME-IGNORE-GAP | yoke prd assess creates .yoke/supervision/<uuid>.json but retrofit .gitignore lacks this directory. New loop blocked as dirty before any iteration. Initial planner supervision file was already tracked. | One zero-iteration loop failure; runtime ignore corrected and tracked runtime file untracked without deleting it Include all watchdog/supervision runtime state in generated ignore rules and distinguish harness-owned dirtiness from user code |
|
|
124
|
+
| COMPLETION-DIRTY-PROOFS | After loop complete 6/6, overview.png and selected-machine.png are modified because completion reexecutes screenshot-writing tests after story commit | Next loop starts dirty and requires explicit artifact commit even though previous loop reported completion Store rerun proofs outside tracked final assets, promote deterministic final proof once, or finalize generated artifact commit after completion verification |
|
|
125
|
+
| VERIFIED-VISUAL-GAP | Correctly bound app on port 6317 has document scrollHeight 938 for viewport 900; event panel bottom 914. All original six stories and design-scan passed. | Full-screen layout still incomplete; finite original-scope repair STORY-7 added, preserving existing functionality and tests Assert panel bottom bounds and populated incident/event states, not only horizontal overflow or panel top visibility; use independent visually bound review |
|
|
126
|
+
| OBSERVER-FINAL-SMOKE-DIRTY | Two final root smoke attempts had 1/1 successful browser flows but source evidence was rejected: first because observer config was uncommitted; second because observer report generation created an untracked root file during the gate. | Observer interference, not product failure. Root smoke must run only after artifact writes and commit are finished. Treat integrated source gates as a write lock for all coordinator tasks; keep measurement generation outside that interval. |
|
|
127
|
+
|
|
128
|
+
Priorität für Yoke: (1) sichere CLI-Hilfe und klare Preflight-Prüfung tatsächlicher RTK/CI-Funktion, (2) korrekte Zuordnung von Provider-, Approval-, Kontroll- und Integrationsfehlern, (3) konkrete Ursachen bei Browserfehlern und wiederverwendbare Paket-Caches mit isolierten Schreibverzeichnissen, (4) überprüfbare Schreibbereiche und dauerhafte Proof-Artefakte, (5) verlässliche kompakte Live-Status- und Tokeninformationen.
|
|
129
|
+
|
|
130
|
+
Die HEAD-Race wurde vom Observer verursacht. Yokes Ablehnung war eine richtige Sicherheitsentscheidung; die dadurch entstandene Zeit und Wiederholung dürfen nicht als spontanes Modellversagen interpretiert werden. Der erhaltene Kandidat wurde über Yokes Recovery-APIs in eine erneute Implementierungsprüfung überführt; Integrationsnachweise wurden nicht als weiter gültig ausgegeben.
|
|
131
|
+
|
|
132
|
+
Der native RTK-Codex-Vertrag wurde mit der [offiziellen RTK-Dokumentation](https://github.com/rtk-ai/rtk/blob/develop/hooks/codex/README.md) abgeglichen. Alle übrigen konkreten Findings stammen aus lokalen Befehlen, Laufzeitdateien und der installierten Yoke-Implementierung 1.22.0.
|
|
133
|
+
|
|
134
|
+
## Daten für Folgeanalysen
|
|
135
|
+
|
|
136
|
+
measurements/summary.json, stories.csv, roles.csv, yoke-phases.csv, shell-categories.csv, model-calls.jsonl, shell-metrics.jsonl, sessions.json, yoke-history.jsonl, commands.jsonl, status-samples.jsonl, observations.jsonl und Code-Intelligence-/Recovery-Probes. Screenshots liegen separat unter screenshots/. Observer- und Auswerteskripte liegen unter tools/.
|
|
137
|
+
|
|
138
|
+
Für belastbare Verbesserungsnachweise: gleiche Aufgabe und Testgates mit/ohne Änderung, gleiche Modellversion, mehrere Wiederholungen, frischer und warmer Cache, kontrollierte Netzwerk-/Sandboxbedingungen sowie getrennte Erfassung des Beobachtungsaufwands. Dieser Lauf liefert konkrete Fehlerbelege und Messdaten, aber keine kausale Aussage über allgemeine Yoke-Effizienz.
|
|
139
|
+
|
|
140
|
+
## Abschlussprüfung nach dem Mess-Cutoff
|
|
141
|
+
|
|
142
|
+
Yoke Flow-Smoke gegen den unveränderten, committed Root-Server bestand am 2026-10-04T09:28:21.392Z: 1/1, 2,156 Sekunden Gesamtzeit, stabile Source-Fingerprints. Bericht in measurements/final-yoke-smoke.json. Das bestätigt, dass die zuvor korrekt abgelehnten Observer-Prüfungen nach Ende aller Schreibvorgänge funktionieren. Dieses zusätzliche Gate und die anschließende Artefaktverpackung sind nicht in den oben eingefrorenen Token-/Zeitaggregaten enthalten.
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
# NEXUS — benchmark result
|
|
2
|
+
|
|
3
|
+
**Implementation complete.** React/TypeScript/Vite application runs at http://127.0.0.1:6317. Start independently with `npm ci` and `npm run dev`.
|
|
4
|
+
|
|
5
|
+
Completed: 12 typed systems; interactive animated SVG topology with selection, pan/zoom/reset/focus; deterministic one-second telemetry and pause/resume; detailed inspector and operator commands; injected/acknowledged incidents; timestamped events and KPIs; searchable keyboard command palette; Fleet search/filter/sort/context selection; responsive desktop/mobile layout. Other navigation modules deliberately identify their unavailable benchmark state.
|
|
6
|
+
|
|
7
|
+
## Validation
|
|
8
|
+
|
|
9
|
+
- Final root `npm run verify` passed: TypeScript, 25 unit/component tests, six Chromium browser journeys, production build. No lint configuration exists.
|
|
10
|
+
- Browser journeys exercise simulation, incidents, inspector, topology, palette keyboard/focus behavior, Fleet, navigation and responsive operations; zero console/page errors asserted.
|
|
11
|
+
- Independent root-server inspection verified listener cwd and served source signature: 1440×900 document fits without scrolling; event panel bottom 884px. Mobile 390×844 has no horizontal overflow; vertical scrolling exposes inspector/events.
|
|
12
|
+
- Five required exact-viewport screenshots in `screenshots/`: overview, selected-machine, incident, Fleet and mobile. Additional `running-overview.png` shows live telemetry and a timestamped event. Final screenshots were visually inspected.
|
|
13
|
+
- Yoke 1.22.0: 7/7 stories pass, eight implementation attempts (STORY-5 retried following an observer-caused HEAD change), isolated worktrees, automatic scheduling/decisions, peak two implementation workers. Continuous exploration was never enabled.
|
|
14
|
+
- Final integrated root Yoke flow-smoke passed 1/1 on committed source `edd9292`, with matching before/after fingerprints. Report: `measurements/final-yoke-smoke.json` (this artifact packaging follows the measurement cutoff).
|
|
15
|
+
- Yoke acceptance gates passed; final story design-scan score 0 against budget 4; story flow-smoke 1/1 passed. Two later observer smoke attempts ran 1/1 successful flows but correctly rejected source evidence while observer files changed; these are recorded separately from product checks.
|
|
16
|
+
|
|
17
|
+
## Environment and limitations
|
|
18
|
+
|
|
19
|
+
Codex CLI 0.160.0; worker model gpt-6.1-sol / medium; RTK 0.51.0 native Codex integration and explicit commands verified. Code Intelligence configured **ACTIVE** and MCP facade functional, but graft/graphify/serena semantic backends unavailable: this requested capability could not deliver semantic analysis. No known failing product acceptance criteria remain. Simulation is local and deterministic; there is no backend.
|
|
20
|
+
|
|
21
|
+
Detailed measured times, provider token usage by story/role, retries, RTK estimates, observer overhead and improvement findings: [DEVELOPMENT_ANALYSIS.md](DEVELOPMENT_ANALYSIS.md). Machine-readable tables and metadata: [measurements/](measurements/). Measurement cutoff is recorded there; later report packaging and the final reply are excluded. Monetary cost is unknown. No hidden reasoning content is included.
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
{"schema": 1, "phase": "environment", "command": ["yoke", "setup", ".", "--yes", "--host=codex", "--code-intelligence=active"], "started_at": "2026-10-04T07:50:34.579136+00:00", "ended_at": "2026-10-04T07:50:35.052705+00:00", "wall_seconds": 0.4736, "child_user_seconds": 0.2617, "child_system_seconds": 0.0812, "exit_code": 0, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
2
|
+
{"schema": 1, "phase": "environment", "command": ["rtk", "init", "--codex", "--auto-patch"], "started_at": "2026-10-04T07:50:35.109289+00:00", "ended_at": "2026-10-04T07:50:35.124643+00:00", "wall_seconds": 0.0154, "child_user_seconds": 0.0032, "child_system_seconds": 0.0043, "exit_code": 1, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
3
|
+
{"schema": 1, "phase": "environment", "command": ["rtk", "init", "--codex"], "started_at": "2026-10-04T07:50:50.958600+00:00", "ended_at": "2026-10-04T07:50:50.994874+00:00", "wall_seconds": 0.0363, "child_user_seconds": 0.0039, "child_system_seconds": 0.0094, "exit_code": 0, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
4
|
+
{"schema": 1, "phase": "environment", "command": ["yoke", "loop", "on", "."], "started_at": "2026-10-04T07:50:51.044200+00:00", "ended_at": "2026-10-04T07:50:51.433251+00:00", "wall_seconds": 0.3891, "child_user_seconds": 0.217, "child_system_seconds": 0.0657, "exit_code": 0, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
5
|
+
{"schema": 1, "phase": "code-intelligence", "command": ["node", "benchmark-artifacts/tools/code-intelligence-probe.mjs"], "started_at": "2026-10-04T07:52:04.747837+00:00", "ended_at": "2026-10-04T07:52:05.474266+00:00", "wall_seconds": 0.7264, "child_user_seconds": 0.3273, "child_system_seconds": 0.1002, "exit_code": 0, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
6
|
+
{"schema": 1, "phase": "code-intelligence", "command": ["node", "benchmark-artifacts/tools/code-intelligence-probe.mjs"], "started_at": "2026-10-04T07:54:15.939646+00:00", "ended_at": "2026-10-04T07:54:16.280657+00:00", "wall_seconds": 0.341, "child_user_seconds": 0.2623, "child_system_seconds": 0.0429, "exit_code": 0, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
7
|
+
{"schema": 1, "phase": "planning", "command": ["yoke", "prd", "check", "."], "started_at": "2026-10-04T07:54:16.333374+00:00", "ended_at": "2026-10-04T07:54:16.539325+00:00", "wall_seconds": 0.206, "child_user_seconds": 0.2085, "child_system_seconds": 0.0264, "exit_code": 1, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
8
|
+
{"schema": 1, "phase": "planning", "command": ["yoke", "prd", "assess", "."], "started_at": "2026-10-04T07:54:28.869074+00:00", "ended_at": "2026-10-04T07:55:10.087293+00:00", "wall_seconds": 41.2182, "child_user_seconds": 3.4678, "child_system_seconds": 1.9825, "exit_code": 0, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
9
|
+
{"schema": 1, "phase": "planning", "command": ["yoke", "prd", "check", "."], "started_at": "2026-10-04T07:55:10.178127+00:00", "ended_at": "2026-10-04T07:55:10.494997+00:00", "wall_seconds": 0.3169, "child_user_seconds": 0.2128, "child_system_seconds": 0.0474, "exit_code": 0, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
10
|
+
{"schema": 1, "phase": "code-intelligence", "command": ["node", "benchmark-artifacts/tools/code-intelligence-probe.mjs"], "started_at": "2026-10-04T07:55:46.432377+00:00", "ended_at": "2026-10-04T07:55:47.351798+00:00", "wall_seconds": 0.9194, "child_user_seconds": 0.3523, "child_system_seconds": 0.1662, "exit_code": 0, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
11
|
+
{"schema": 1, "phase": "dependency-setup", "command": ["rtk", "summary", "npm", "ci", "--no-audit", "--no-fund", "--fetch-retries=0", "--fetch-timeout=20000"], "started_at": "2026-10-04T08:06:36.488499+00:00", "ended_at": "2026-10-04T08:06:42.107381+00:00", "wall_seconds": 5.6189, "child_user_seconds": 1.6001, "child_system_seconds": 1.868, "exit_code": 1, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
12
|
+
{"schema": 1, "phase": "dependency-setup", "command": ["rtk", "summary", "npm", "ci", "--offline", "--no-audit", "--no-fund", "--cache", "/private/tmp/nexus-npm-cache"], "started_at": "2026-10-04T08:07:51.872878+00:00", "ended_at": "2026-10-04T08:07:56.303096+00:00", "wall_seconds": 4.4302, "child_user_seconds": 1.4637, "child_system_seconds": 1.8639, "exit_code": 0, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
13
|
+
{"schema": 1, "phase": "yoke-smoke", "command": ["yoke", "flow-smoke", ".yoke/worktrees/bfe61bdcdbbcaea9c0b2c12f", "--label=observer-topology"], "started_at": "2026-10-04T08:17:46.589515+00:00", "ended_at": "2026-10-04T08:17:47.012070+00:00", "wall_seconds": 0.4226, "child_user_seconds": 0.2277, "child_system_seconds": 0.043, "exit_code": 2, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
14
|
+
{"schema": 1, "phase": "execution", "command": ["yoke", "loop", "run", ".", "--runner=codex", "--isolate", "--parallel=auto", "--decision-policy=auto", "--max=12", "--timeout=30"], "started_at": "2026-10-04T07:56:11.479061+00:00", "ended_at": "2026-10-04T08:29:09.630333+00:00", "wall_seconds": 1978.1513, "child_user_seconds": 439.897, "child_system_seconds": 158.2676, "exit_code": 1, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
15
|
+
{"schema": 1, "phase": "recovery", "command": ["node", "benchmark-artifacts/measurements/reconcile-recovery.mjs"], "started_at": "2026-10-04T08:32:18.181448+00:00", "ended_at": "2026-10-04T08:32:18.712772+00:00", "wall_seconds": 0.5313, "child_user_seconds": 0.2001, "child_system_seconds": 0.1451, "exit_code": 0, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
16
|
+
{"schema": 1, "phase": "dev-server", "command": ["npm", "run", "dev", "--", "--host", "127.0.0.1", "--port", "4173", "--strictPort"], "started_at": "2026-10-04T08:36:29.869135+00:00", "ended_at": "2026-10-04T08:36:30.694902+00:00", "wall_seconds": 0.8258, "child_user_seconds": 0.3544, "child_system_seconds": 0.1515, "exit_code": 1, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
17
|
+
{"schema": 1, "phase": "browser-inspection", "command": ["node", "benchmark-artifacts/measurements/observer-browser.mjs"], "started_at": "2026-10-04T08:36:46.535968+00:00", "ended_at": "2026-10-04T08:36:50.343066+00:00", "wall_seconds": 3.8071, "child_user_seconds": 1.4643, "child_system_seconds": 0.6258, "exit_code": 0, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
18
|
+
{"schema": 1, "phase": "execution", "command": ["yoke", "loop", "run", ".", "--runner=codex", "--isolate", "--parallel=auto", "--decision-policy=auto", "--max=8", "--timeout=30"], "started_at": "2026-10-04T08:32:18.767422+00:00", "ended_at": "2026-10-04T08:47:04.416570+00:00", "wall_seconds": 885.6494, "child_user_seconds": 356.0493, "child_system_seconds": 103.5048, "exit_code": 0, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
19
|
+
{"schema": 1, "phase": "browser-inspection", "command": ["node", "benchmark-artifacts/measurements/observer-browser.mjs"], "started_at": "2026-10-04T08:52:24.384590+00:00", "ended_at": "2026-10-04T08:52:34.917468+00:00", "wall_seconds": 10.5329, "child_user_seconds": 2.44, "child_system_seconds": 0.9022, "exit_code": 0, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
20
|
+
{"schema": 1, "phase": "planning", "command": ["yoke", "prd", "assess", ".", "--story=STORY-7"], "started_at": "2026-10-04T08:58:03.354016+00:00", "ended_at": "2026-10-04T08:58:22.747547+00:00", "wall_seconds": 19.3936, "child_user_seconds": 3.1061, "child_system_seconds": 1.7403, "exit_code": 0, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
21
|
+
{"schema": 1, "phase": "planning", "command": ["yoke", "prd", "check", "."], "started_at": "2026-10-04T08:58:22.841808+00:00", "ended_at": "2026-10-04T08:58:23.136397+00:00", "wall_seconds": 0.2946, "child_user_seconds": 0.2052, "child_system_seconds": 0.0429, "exit_code": 0, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
22
|
+
{"schema": 1, "phase": "execution", "command": ["yoke", "loop", "run", ".", "--runner=codex", "--isolate", "--parallel=auto", "--decision-policy=auto", "--max=4", "--timeout=30"], "started_at": "2026-10-04T08:58:23.381604+00:00", "ended_at": "2026-10-04T08:58:23.813100+00:00", "wall_seconds": 0.4315, "child_user_seconds": 0.3126, "child_system_seconds": 0.1054, "exit_code": 1, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
23
|
+
{"schema": 1, "phase": "execution", "command": ["yoke", "loop", "run", ".", "--runner=codex", "--isolate", "--parallel=auto", "--decision-policy=auto", "--max=4", "--timeout=30"], "started_at": "2026-10-04T09:03:18.113999+00:00", "ended_at": "2026-10-04T09:11:58.621802+00:00", "wall_seconds": 520.5078, "child_user_seconds": 171.3881, "child_system_seconds": 48.3344, "exit_code": 0, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
24
|
+
{"schema": 1, "phase": "final-validation", "command": ["rtk", "proxy", "npm", "run", "verify"], "started_at": "2026-10-04T09:24:26.434839+00:00", "ended_at": "2026-10-04T09:24:46.431823+00:00", "wall_seconds": 19.997, "child_user_seconds": 24.9152, "child_system_seconds": 5.0867, "exit_code": 0, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
25
|
+
{"schema": 1, "phase": "final-yoke-smoke", "command": ["rtk", "proxy", "yoke", "flow-smoke", ".", "--url=http://127.0.0.1:6317", "--label=FINAL-INTEGRATED"], "started_at": "2026-10-04T09:25:21.438466+00:00", "ended_at": "2026-10-04T09:25:24.549999+00:00", "wall_seconds": 3.1115, "child_user_seconds": 1.3225, "child_system_seconds": 0.9601, "exit_code": 1, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
26
|
+
{"schema": 1, "phase": "final-yoke-smoke", "command": ["rtk", "proxy", "yoke", "flow-smoke", ".", "--url=http://127.0.0.1:6317", "--label=FINAL-INTEGRATED"], "started_at": "2026-10-04T09:26:39.498381+00:00", "ended_at": "2026-10-04T09:26:41.899985+00:00", "wall_seconds": 2.4016, "child_user_seconds": 1.1687, "child_system_seconds": 0.7318, "exit_code": 1, "cwd": "/Users/hecer/orca/workspaces/Yoke-Test/moonfish"}
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
{
|
|
2
|
+
"captured_at": "2026-10-04T09:25:22.622378+00:00",
|
|
3
|
+
"yoke": "1.22.0",
|
|
4
|
+
"codex": "codex-cli 0.160.0",
|
|
5
|
+
"rtk": "rtk 0.51.0",
|
|
6
|
+
"packages": {
|
|
7
|
+
"react": "19.3.0",
|
|
8
|
+
"typescript": "7.0.2",
|
|
9
|
+
"vite": "8.3.2",
|
|
10
|
+
"vitest": "5.0.3",
|
|
11
|
+
"playwright": "1.63.0"
|
|
12
|
+
},
|
|
13
|
+
"code_intelligence": {
|
|
14
|
+
"mode": "active",
|
|
15
|
+
"facade_handshake": "passed",
|
|
16
|
+
"semantic_backends": "graft, graphify, serena unavailable",
|
|
17
|
+
"semantic_benefit_measured": false
|
|
18
|
+
},
|
|
19
|
+
"runner": {
|
|
20
|
+
"model": "gpt-6.1-sol",
|
|
21
|
+
"reasoning_effort": "medium"
|
|
22
|
+
},
|
|
23
|
+
"rtk_hook": "native Codex integration plus explicit commands",
|
|
24
|
+
"continuous_exploration": false,
|
|
25
|
+
"app_url": "http://127.0.0.1:6317",
|
|
26
|
+
"viewport_checks": {
|
|
27
|
+
"desktop": "1440x900",
|
|
28
|
+
"mobile": "390x844"
|
|
29
|
+
},
|
|
30
|
+
"lint": "not configured"
|
|
31
|
+
}
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
{
|
|
2
|
+
"version": 1,
|
|
3
|
+
"startedAt": "2026-10-04T09:28:19.236Z",
|
|
4
|
+
"generatedAt": "2026-10-04T09:28:21.392Z",
|
|
5
|
+
"durationMs": 2156,
|
|
6
|
+
"status": "passed",
|
|
7
|
+
"sourceFingerprint": "b75398d707bf14c97f8032a13a8be4bea197a6f0ea6169bad7b9c9fd8cc1c821",
|
|
8
|
+
"afterSourceFingerprint": "b75398d707bf14c97f8032a13a8be4bea197a6f0ea6169bad7b9c9fd8cc1c821",
|
|
9
|
+
"sourceStable": true,
|
|
10
|
+
"configDigest": "82535fa082b63b11ba8c0cb012f4f5e5775cd6d1bc2d062fb04a6b8cbc51eb1c",
|
|
11
|
+
"environment": {
|
|
12
|
+
"platform": "darwin",
|
|
13
|
+
"architecture": "arm64",
|
|
14
|
+
"nodeVersion": "v22.23.2",
|
|
15
|
+
"driver": "playwright-chromium",
|
|
16
|
+
"browserVersion": "153.0.8010.12",
|
|
17
|
+
"baseOrigin": "http://127.0.0.1:6317",
|
|
18
|
+
"viewport": {
|
|
19
|
+
"width": 1280,
|
|
20
|
+
"height": 720
|
|
21
|
+
}
|
|
22
|
+
},
|
|
23
|
+
"flows": [
|
|
24
|
+
{
|
|
25
|
+
"name": "foundation",
|
|
26
|
+
"status": "passed",
|
|
27
|
+
"durationMs": 846,
|
|
28
|
+
"navigation": {
|
|
29
|
+
"status": "passed",
|
|
30
|
+
"durationMs": 183
|
|
31
|
+
},
|
|
32
|
+
"landmark": {
|
|
33
|
+
"status": "passed",
|
|
34
|
+
"durationMs": 182
|
|
35
|
+
},
|
|
36
|
+
"steps": [],
|
|
37
|
+
"screenshot": "foundation.png"
|
|
38
|
+
}
|
|
39
|
+
]
|
|
40
|
+
}
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
role_purpose_hint,usage_events,input_tokens,cached_input_tokens,output_tokens
|
|
2
|
+
guardian/unknown,48,1281675,1073152,5404
|
|
3
|
+
implementation/discovery,35,2015491,1824896,6669
|
|
4
|
+
implementation/checks,63,3377368,3130624,44163
|
|
5
|
+
implementation/code_intelligence,6,263056,252160,785
|
|
6
|
+
implementation/dependency_setup,6,270273,244480,2947
|
|
7
|
+
implementation/dev_server_or_mixed,12,809495,797184,4029
|
|
8
|
+
implementation/waiting_or_polling,26,1668311,1642752,1869
|
|
9
|
+
implementation/browser_validation,41,2830598,2729856,21731
|
|
10
|
+
coordinator/discovery,26,2839894,2756096,14891
|
|
11
|
+
coordinator/browser_validation,24,3122621,3013248,27865
|
|
12
|
+
coordinator/checks,11,1031473,1002752,5642
|
|
13
|
+
coordinator/code_intelligence,4,295093,281344,3248
|
|
14
|
+
coordinator/dev_server_or_mixed,5,696003,682752,10074
|
|
15
|
+
coordinator/waiting_or_polling,35,4593089,4546688,31499
|
|
16
|
+
coordinator/other,14,1765447,1753856,468
|
|
17
|
+
coordinator/file_editing,7,1080599,1003392,11569
|
|
18
|
+
coordinator/dependency_setup,1,100894,99328,1862
|
|
19
|
+
planner_or_review/unknown,2,52718,14336,1215
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
{"id": "CLI-HELP-MUTATION", "severity": "high", "phase": "environment", "observed": "yoke setup --help created 95 files and enabled loop; yoke retrofit --help then disabled loop", "impact": "Read-only discovery mutated configuration and required explicit correction", "evidence": "CLI output and generated .yoke/config.yaml", "suggestion": "Handle help before command dispatch; reject unknown flags before mutation"}
|
|
2
|
+
{"id": "RTK-INIT-FLAGS", "severity": "low", "phase": "environment", "observed": "rtk init --codex --auto-patch exits 1: cannot be combined", "impact": "One failed setup call", "suggestion": "Document per-runner incompatible flags and expose compatibility validation"}
|
|
3
|
+
{"id": "CI-WORKSPACE-ID", "severity": "medium", "phase": "environment", "observed": "code_context rejects absolute path and configured codeIntelligence.workspaceId; MCP server computes hash workspace ID and ignores configured workspaceId", "impact": "Two failed probes before reading implementation to determine identifier", "suggestion": "Expose workspace identifier in initialization/tool listing, honor config or document discovery method"}
|
|
4
|
+
{"id": "CI-BACKENDS-MISSING", "severity": "medium", "phase": "environment", "observed": "graft, graphify and serena executables absent on PATH", "impact": "Facade can be enabled but semantic/structural backend functionality unavailable", "suggestion": "Add preflight that distinguishes configured active mode from operational backend coverage and provides pinned installation guidance"}
|
|
5
|
+
{"id": "RTK-DUPLICATE-HOOKS", "severity": "medium", "phase": "environment", "observed": "Yoke legacy hook and rtk init native Codex hook both present; rtk hook check is supported (exit 0)", "impact": "Double hook registration observed; rewrite effectiveness with actual Codex tool schema needs separate measurement", "suggestion": "Use idempotent native Codex hook integration and provide actual rewrite/adoption telemetry"}
|
|
6
|
+
{"id": "COMPACT-STATUS-MISSING", "severity": "medium", "phase": "execution", "observed": "yoke loop status --compact prints story=none phase=undefined while loop-status.json reports active STORY-1 implementing", "impact": "Compact progress unsuitable for reliable supervision; detailed status file required", "suggestion": "Summarize parallel workers in compact status and avoid undefined values"}
|
|
7
|
+
{"id": "WRITE-SCOPE-DRIFT", "severity": "medium", "phase": "execution", "observed": "STORY-2 declared src/simulation and tests/simulation.test.ts but commit writes src/simulation.ts, src/useSimulation.ts, src/types.ts, tests/simulation.test.tsx; scheduler scopes are advisory", "impact": "Declared non-overlap does not prove actual non-overlap; shared type file changed during parallel topology work", "suggestion": "Compare candidate changes against writes declarations, surface deviations and recheck collision risk before integration"}
|
|
8
|
+
{"id": "SMOKE-MISDIAGNOSIS", "severity": "medium", "phase": "verification", "observed": "Yoke flow-smoke reports Playwright not found while node_modules/playwright/package.json exists; launchPlaywright catches both import and chromium.launch errors and returns null", "impact": "Package absence and browser launch/sandbox failures have the same actionable error; worker reads harness source to diagnose", "suggestion": "Return structured import/browser-binary/sandbox/launch failure causes and preserve underlying error"}
|
|
9
|
+
{"id": "EPHEMERAL-PROOF", "severity": "medium", "phase": "integration", "observed": "After isolated STORY-1/2 integration, target .yoke/proof directory absent; earlier worker smoke screenshot was in removed worktree", "impact": "Intermediate manual proof lost on cleanup; root must preserve required screenshots in tracked artifact path", "suggestion": "Copy and content-bind selected verification artifacts to target before deleting isolated worktrees"}
|
|
10
|
+
{"id": "RTK-LEGACY-ALLOW", "severity": "high", "phase": "environment", "observed": "Yoke-generated rtk.mjs emits updatedInput without permissionDecision allow; official RTK Codex documentation requires allow for replacement to take effect", "impact": "Legacy transparent command rewriting incompatible with documented Codex response contract; native RTK integration is also registered but worker adoption not observed", "suggestion": "Use current native rtk hook codex; verify an actual verbose Codex command records compressed output and RTK history", "reference": "https://github.com/rtk-ai/rtk/blob/develop/hooks/codex/README.md"}
|
|
11
|
+
{"id": "RTK-NATIVE-CORRECTION", "timestamp": "2026-10-04T08:20:10.412392+00:00", "phase": "environment-correction", "observed": "Removed legacy Yoke hook registration; native rtk hook codex remains sole active hook and manual exact-schema probe rewrites git status correctly", "impact": "Future workers receive native integration without duplicate response; effectiveness measured separately"}
|
|
12
|
+
{"id": "SHARED-DEPS-SANDBOX", "severity": "medium", "phase": "execution", "observed": "STORY-5 Vitest fails EPERM mkdir at target root node_modules/.vite-temp while executing in isolated worker; tests succeed after worker retries", "impact": "Dependency reuse caused additional startup failure and approval/escalation path", "suggestion": "Share immutable package download cache, provide per-worktree writable dependency/runtime cache or explicit allowed paths"}
|
|
13
|
+
{"id": "NETWORK-RETRY-CACHE", "severity": "medium", "phase": "environment", "observed": "Target npm ci fails ECONNRESET downloading why-is-node-running; npm ci --offline --cache /private/tmp/nexus-npm-cache then installs 105 packages in 4s", "impact": "At least one failed installation and repeated dependency setup; cached recovery successful", "suggestion": "Preflight and reuse exact lockfile package cache, classify network failures separately from code failures; n=1 does not establish general speedup"}
|
|
14
|
+
{"id": "GUARDIAN-USAGE-GAP", "severity": "medium", "phase": "observability", "observed": "Native Codex sessions with source.subagent.other=guardian report token usage separately; Yoke worker token event matches implementation session and excludes those sessions", "impact": "Harness totals omit observable approval-model usage; independent observer reports it separately", "suggestion": "Expose approval/guardian call coverage and distinguish implementation usage from full provider-session cost; unknown charge remains unknown"}
|
|
15
|
+
{"id": "OBSERVER-HEAD-RACE", "timestamp": "2026-10-04T08:32:35.775919+00:00", "severity": "medium", "phase": "integration", "observed": "Observer committed AGENTS.md during STORY-5 integrated gates; Yoke rejected changed target HEAD, retained candidate, and stopped loop", "attribution": "observer-induced coordination error; expected safety rejection, not spontaneous implementation failure", "impact": "One failed integration and explicit recovery; prior source preserved; recovery invalidates proof and reruns implementation/gates", "suggestion": "Observer should make harness changes only between loop batches; CLI should expose integration busy state and avoid premature integration complete message"}
|
|
16
|
+
{"id": "VISUAL-COMPOSITION", "phase": "independent-visual-review", "severity": "info", "observed": "INVALIDATED: first observer screenshots came from another pre-existing NEXUS on port 4173; expected root Vite process exited with port-in-use error", "impact": "Those visual conclusions are excluded from this project assessment", "suggestion": "Verify owning process cwd, successful server startup and source identity before collecting browser evidence", "evidence": ["observer-overview.png", "observer-mobile.png", "observer-browser-initial.json"], "validity": "INVALIDATED_WRONG_SERVER"}
|
|
17
|
+
{"id": "OBSERVER-WRONG-SERVER", "timestamp": "2026-10-04T08:48:21.490940+00:00", "phase": "browser-inspection", "severity": "high", "attribution": "observer error", "observed": "First browser probe navigated occupied port 4173 before checking Vite process had started. Different NEXUS app names and styles exposed mismatch. Target process exit 1 confirmed port conflict. Existing listener left untouched.", "impact": "One invalid browser run; screenshots and prior visual finding invalidated; its time/tokens remain observer overhead", "suggestion": "Require startup success plus bound process cwd and served source signature; use unique port and record provenance", "evidence": "invalid-wrong-port-browser.json and dev-server command exit 1"}
|
|
18
|
+
{"id": "RUNTIME-IGNORE-GAP", "timestamp": "2026-10-04T09:03:17.796238+00:00", "severity": "medium", "phase": "loop-start", "observed": "yoke prd assess creates .yoke/supervision/<uuid>.json but retrofit .gitignore lacks this directory. New loop blocked as dirty before any iteration. Initial planner supervision file was already tracked.", "impact": "One zero-iteration loop failure; runtime ignore corrected and tracked runtime file untracked without deleting it", "suggestion": "Include all watchdog/supervision runtime state in generated ignore rules and distinguish harness-owned dirtiness from user code"}
|
|
19
|
+
{"id": "COMPLETION-DIRTY-PROOFS", "severity": "medium", "phase": "completion", "observed": "After loop complete 6/6, overview.png and selected-machine.png are modified because completion reexecutes screenshot-writing tests after story commit", "impact": "Next loop starts dirty and requires explicit artifact commit even though previous loop reported completion", "suggestion": "Store rerun proofs outside tracked final assets, promote deterministic final proof once, or finalize generated artifact commit after completion verification"}
|
|
20
|
+
{"id": "VERIFIED-VISUAL-GAP", "severity": "high", "phase": "independent-review", "observed": "Correctly bound app on port 6317 has document scrollHeight 938 for viewport 900; event panel bottom 914. All original six stories and design-scan passed.", "impact": "Full-screen layout still incomplete; finite original-scope repair STORY-7 added, preserving existing functionality and tests", "suggestion": "Assert panel bottom bounds and populated incident/event states, not only horizontal overflow or panel top visibility; use independent visually bound review", "evidence": "observer-browser-final.json and observer-overview.png"}
|
|
21
|
+
{"id": "OBSERVER-FINAL-SMOKE-DIRTY", "timestamp": "2026-10-04T09:27:31.989785+00:00", "observed": "Two final root smoke attempts had 1/1 successful browser flows but source evidence was rejected: first because observer config was uncommitted; second because observer report generation created an untracked root file during the gate.", "impact": "Observer interference, not product failure. Root smoke must run only after artifact writes and commit are finished.", "suggestion": "Treat integrated source gates as a write lock for all coordinator tasks; keep measurement generation outside that interval."}
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
phase,commands,failed,wall_seconds,child_cpu_seconds
|
|
2
|
+
environment,4,1,0.9144000000000001,0.6464
|
|
3
|
+
code-intelligence,3,0,1.9868000000000001,1.2511999999999999
|
|
4
|
+
planning,5,1,61.429300000000005,11.0399
|
|
5
|
+
dependency-setup,2,1,10.0491,6.7957
|
|
6
|
+
yoke-smoke,1,1,0.4226,0.2707
|
|
7
|
+
execution,4,2,3384.74,1277.8591999999999
|
|
8
|
+
recovery,1,0,0.5313,0.3452
|
|
9
|
+
dev-server,1,1,0.8258,0.5059
|
|
10
|
+
browser-inspection,2,0,14.34,5.4323
|
|
11
|
+
final-validation,1,0,19.997,30.0019
|
|
12
|
+
final-yoke-smoke,2,2,5.5131,4.1831
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
role,sessions,measured_sessions,model_calls,input_tokens,cached_input_tokens,output_tokens,tool_output_bytes,uncached_input_tokens,unknown_sessions
|
|
2
|
+
guardian,7,7,48,1281675,1073152,5404,0,208523,0
|
|
3
|
+
implementation,8,8,189,11234592,10621952,82193,6175737,612640,0
|
|
4
|
+
coordinator,1,1,127,15525113,15139456,107118,2971077,385657,0
|
|
5
|
+
planner_or_review,2,2,2,52718,14336,1215,0,38382,0
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
category,count,failed_or_cancelled,process_lifetime_seconds,output_bytes
|
|
2
|
+
discovery,73,12,32.161418747,452893
|
|
3
|
+
dependency_setup,10,5,114.452577168,6349
|
|
4
|
+
checks,92,25,461.32378954399996,420588
|
|
5
|
+
browser_validation,23,9,103.279605002,109393
|
|
6
|
+
dev_server_or_mixed,12,12,677.7817032089998,4534
|
|
7
|
+
other,17,3,86.670873458,4748
|
|
8
|
+
file_editing,1,0,0.268871041,0
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
story,title,passes,implementation_attempts,implementation_seconds,verification_seconds,integration_seconds,queue_wait_seconds,input_tokens,cached_input_tokens,output_tokens,measured_provider_calls
|
|
2
|
+
STORY-1,Scaffold runnable typed NEXUS foundation and test infrastructure,True,1,565.898,3.058,6.301,0.007,1435585,1314176,12725,1
|
|
3
|
+
STORY-2,"Implement deterministic telemetry, incidents and operator machine actions",True,1,367.382,2.816,8.723,0.011,698268,643968,8663,1
|
|
4
|
+
STORY-3,Create visually distinctive interactive digital twin topology,True,1,696.314,5.19,9.183,0.007,2080455,1994752,15192,1
|
|
5
|
+
STORY-4,"Build detailed machine inspector, Fleet and operational events/KPIs",True,1,505.759,4.124,11.815,0.004,921789,857088,12190,1
|
|
6
|
+
STORY-5,Integrate complete operations interface and searchable keyboard command palette,True,2,566.104,11.58,35.665,0.014,1347301,1245568,11899,2
|
|
7
|
+
STORY-6,"Verify integrated benchmark in browser, repair defects and save all visual proofs",True,1,683.215,28.821,28.292,0.005,2950211,2853888,13068,1
|
|
8
|
+
STORY-7,Repair verified desktop containment and mobile priority; refresh visual proof,True,1,484.22,18.435,0,0.007,1800983,1712512,8456,1
|