@hecer/yoke 1.6.2 → 1.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/.codex-plugin/plugin.json +1 -1
- package/CHANGELOG.md +42 -0
- package/README.md +61 -22
- package/canon/manifest.yaml +1 -1
- package/canon/tools/gemini-rtk-hook.mjs +25 -0
- package/dist/agents/contracts.js +2 -0
- package/dist/agents/process-streams.js +18 -4
- package/dist/agents/providers.js +32 -4
- package/dist/agents/telemetry.js +58 -10
- package/dist/check/command.js +114 -0
- package/dist/cli.js +124 -5
- package/dist/context/packet.js +32 -0
- package/dist/dashboard/analytics.js +123 -0
- package/dist/dashboard/page.js +34 -0
- package/dist/dashboard/panels.js +19 -0
- package/dist/dashboard/registry.js +68 -0
- package/dist/dashboard/server.js +177 -0
- package/dist/estimation/durations.js +20 -0
- package/dist/estimation/schedule.js +41 -0
- package/dist/execution/actions.js +24 -0
- package/dist/goals/command.js +191 -0
- package/dist/loop/dispatcher.js +19 -5
- package/dist/loop/git.js +5 -4
- package/dist/loop/loop.js +23 -3
- package/dist/loop/parallel-adapters.js +3 -2
- package/dist/loop/parallel-command.js +52 -4
- package/dist/loop/prd.js +3 -0
- package/dist/loop/recovery.js +51 -0
- package/dist/loop/reporter.js +105 -8
- package/dist/loop/run-command.js +62 -13
- package/dist/loop/runner.js +20 -16
- package/dist/loop/scheduler.js +38 -1
- package/dist/observability/events.js +72 -0
- package/dist/observability/history.js +80 -0
- package/dist/quality/candidate-comparison.js +1 -1
- package/dist/quality/command.js +16 -3
- package/dist/retrofit/config.js +11 -0
- package/dist/retrofit/gitignore.js +6 -0
- package/dist/retrofit/planners/gemini.js +11 -3
- package/dist/routing/router.js +50 -10
- package/dist/setup/command.js +3 -2
- package/dist/workspace/fingerprint.js +66 -0
- package/dist/workspace/state.js +20 -0
- package/docs/PRODUCT-DIRECTION-2026-09-05.md +199 -0
- package/docs/VERIFIED-PROJECTS-VALIDATION.md +29 -0
- package/docs/VERIFIED-PROJECTS.md +167 -0
- package/docs/superpowers/plans/2026-09-05-verified-projects.md +83 -0
- package/gemini-extension.json +1 -1
- package/hooks/bounded-gemini.mjs +70 -0
- package/package.json +1 -1
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
import { createHash } from 'node:crypto';
|
|
2
|
+
import { execFileSync } from 'node:child_process';
|
|
3
|
+
import { lstatSync, readFileSync, readdirSync, readlinkSync, realpathSync } from 'node:fs';
|
|
4
|
+
import { isAbsolute, join, relative, resolve } from 'node:path';
|
|
5
|
+
/** Content identity, including new files. Symlinks are identified, never followed. */
|
|
6
|
+
export function workspaceFingerprint(directory) {
|
|
7
|
+
const root = realpathSync(directory);
|
|
8
|
+
const hash = createHash('sha256');
|
|
9
|
+
let names;
|
|
10
|
+
const git = (args) => execFileSync('git', args, { cwd: root, encoding: 'utf8', stdio: ['ignore', 'pipe', 'pipe'], maxBuffer: 32 * 1024 * 1024 });
|
|
11
|
+
try {
|
|
12
|
+
git(['rev-parse', '--show-toplevel']);
|
|
13
|
+
}
|
|
14
|
+
catch (error) {
|
|
15
|
+
// Plain directories are supported for review fixtures and standalone checks;
|
|
16
|
+
// permission/corruption errors in actual repositories must never become success.
|
|
17
|
+
if (!String(error.stderr).includes('not a git repository'))
|
|
18
|
+
throw error;
|
|
19
|
+
const walk = (dir) => readdirSync(dir).sort().flatMap(name => {
|
|
20
|
+
if (['.git', 'node_modules', '.yoke'].includes(name))
|
|
21
|
+
return [];
|
|
22
|
+
const path = join(dir, name);
|
|
23
|
+
return lstatSync(path).isDirectory() ? walk(path) : [relative(root, path)];
|
|
24
|
+
});
|
|
25
|
+
names = walk(root);
|
|
26
|
+
return digest(names);
|
|
27
|
+
}
|
|
28
|
+
hash.update(git(['status', '--porcelain=v1', '-z', '--untracked-files=all', '--', '.', ':(exclude).yoke/**']));
|
|
29
|
+
hash.update(git(['ls-files', '--stage', '-z', '--', '.', ':(exclude).yoke/**']));
|
|
30
|
+
names = [...new Set(git(['ls-files', '--cached', '--others', '--exclude-standard', '-z', '--', '.', ':(exclude).yoke/**']).split('\0').filter(Boolean))].sort();
|
|
31
|
+
return digest(names);
|
|
32
|
+
function digest(files) {
|
|
33
|
+
// Runtime status and evidence change during checks. Executable project policy
|
|
34
|
+
// must still belong to the identity, even when globally ignored by Git.
|
|
35
|
+
for (const name of [...new Set([...files, '.yoke/acceptance.yaml', '.yoke/config.yaml', '.yoke/prd.json', '.yoke/prd.yaml'])].sort()) {
|
|
36
|
+
const full = resolve(root, name);
|
|
37
|
+
const rel = relative(root, full);
|
|
38
|
+
if (isAbsolute(rel) || rel === '..' || rel.startsWith('..\\') || rel.startsWith('../'))
|
|
39
|
+
throw new Error('Fingerprint path escapes workspace');
|
|
40
|
+
hash.update(JSON.stringify(name));
|
|
41
|
+
let stat;
|
|
42
|
+
try {
|
|
43
|
+
stat = lstatSync(full);
|
|
44
|
+
}
|
|
45
|
+
catch (error) {
|
|
46
|
+
if (error.code === 'ENOENT') {
|
|
47
|
+
hash.update('missing');
|
|
48
|
+
continue;
|
|
49
|
+
}
|
|
50
|
+
throw error;
|
|
51
|
+
}
|
|
52
|
+
if (stat.isSymbolicLink())
|
|
53
|
+
hash.update(`link:${readlinkSync(full)}`);
|
|
54
|
+
else if (stat.isFile()) {
|
|
55
|
+
if (stat.size > 64 * 1024 * 1024)
|
|
56
|
+
throw new Error(`File too large to fingerprint safely: ${name}`);
|
|
57
|
+
hash.update(`file:${stat.mode}:${stat.size}:`).update(readFileSync(full));
|
|
58
|
+
}
|
|
59
|
+
else if (stat.isDirectory())
|
|
60
|
+
throw new Error(`Cannot safely fingerprint a nested repository or directory entry: ${name}`);
|
|
61
|
+
else
|
|
62
|
+
throw new Error(`Unsupported workspace entry: ${name}`);
|
|
63
|
+
}
|
|
64
|
+
return hash.digest('hex');
|
|
65
|
+
}
|
|
66
|
+
}
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
import { lstatSync } from 'node:fs';
|
|
2
|
+
import { join } from 'node:path';
|
|
3
|
+
/** Fixed internal runtime paths only. Never follow project-controlled state links. */
|
|
4
|
+
export function statePath(root, ...parts) {
|
|
5
|
+
let path = root;
|
|
6
|
+
for (const part of ['.yoke', ...parts]) {
|
|
7
|
+
if (!part || /[\\/]/u.test(part) || part === '.' || part === '..')
|
|
8
|
+
throw new Error('Invalid state path component');
|
|
9
|
+
path = join(path, part);
|
|
10
|
+
try {
|
|
11
|
+
if (lstatSync(path).isSymbolicLink())
|
|
12
|
+
throw new Error(`Linked Yoke state is not supported: ${part}`);
|
|
13
|
+
}
|
|
14
|
+
catch (error) {
|
|
15
|
+
if (error.code !== 'ENOENT')
|
|
16
|
+
throw error;
|
|
17
|
+
}
|
|
18
|
+
}
|
|
19
|
+
return path;
|
|
20
|
+
}
|
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
# Yoke: Produktstand und Gesprächsgedächtnis
|
|
2
|
+
|
|
3
|
+
Stand: 2026-09-05. Grundlage: Repository-Analyse und anschließende Produktdiskussion mit dem Nutzer.
|
|
4
|
+
|
|
5
|
+
Dieses Dokument sichert die wesentlichen Befunde, Vorschläge und Nutzerpräferenzen der Sitzung. Die automatische claude-mem-Erinnerung meldete einen Ausfall; ihre Speicherung wurde nicht vorausgesetzt. Es ist kein Implementierungsnachweis und kein Auftrag, sämtliche Vorschläge ungefragt umzusetzen. Vor Implementierung den aktuellen Code und die Prioritäten prüfen.
|
|
6
|
+
|
|
7
|
+
## Nutzerabsicht und akzeptierte Richtung
|
|
8
|
+
|
|
9
|
+
- Yoke soll gegenüber der Konkurrenz interessanter und ein regelmäßig genutztes Entwicklerwerkzeug werden.
|
|
10
|
+
- Codex, Claude und Gemini sollen funktional gleichwertig integriert sein. Gleiche Modellintelligenz ist damit weder zugesagt noch messbar belegt.
|
|
11
|
+
- Der Nutzer begrüßte die Richtung: überprüfbare Abnahme, Ziele, Wiederaufnahme und Anbieterwechsel.
|
|
12
|
+
- Tokenverbrauch und Entwicklungszeit sollen sinken: deterministische Werkzeuge, kleine/schnelle Modelle, gezielte Eskalation und sinnvolle Parallelität.
|
|
13
|
+
- Neue Nutzeridee: bessere Zeitschätzungen für alle Tasks; bestehende ETA verbessern.
|
|
14
|
+
- Neue Nutzeridee: ein Dashboard für Yoke mit Übersicht und Detailansichten für jedes Projekt. Interesse ist festgehalten; Umfang und Gestaltung sind noch nicht beschlossen.
|
|
15
|
+
- Der Nutzer bat ausdrücklich darum, die gesamte bisherige Diskussion dauerhaft zu sichern.
|
|
16
|
+
|
|
17
|
+
## Ausgangsbefund der Prüfung
|
|
18
|
+
|
|
19
|
+
Geprüft: Yoke 1.6.2, Commit d066058. Keine Produktcodeänderungen oder neuen authentifizierten Modellbenchmarks während der Analyse.
|
|
20
|
+
|
|
21
|
+
- Gesamtsuite: 1018 bestanden, 2 übersprungen, 1 Test-Timeout von insgesamt 1021 Tests.
|
|
22
|
+
- Betroffen: tests/loop/parallel-cli.integration.test.ts, Test "does not integrate when the target rewinds during integrated gates". Gesamtlauf überschritt das 5-Sekunden-Limit; separater Lauf mit 20-Sekunden-Limit bestand in etwa 1,42 Sekunden Testzeit. Kein damit nachgewiesener Integrationsfehler; Testinstabilität untersuchen.
|
|
23
|
+
- TypeScript-Prüfung und docs:check bestanden.
|
|
24
|
+
- Vorhandene Nutzeränderungen wurden nicht angefasst: .gitignore, .omo/, .playwright-mcp/, docs/community-outreach-2026-08-20.md, docs/launch-copy-2026-08-21.md.
|
|
25
|
+
|
|
26
|
+
### Stärken
|
|
27
|
+
|
|
28
|
+
- Mechanische Prüfkommandos und strukturierte Akzeptanzkriterien; neue Standardkonfigurationen verlangen Kriteriennachweise.
|
|
29
|
+
- Gemeinsamer Canon mit nativen Skill-Paketen und Werkzeugkonfiguration für drei Anbieter.
|
|
30
|
+
- Abhängigkeiten, parallele Worker, Arbeitsbäume, Locks, Prozessüberwachung und Integrationswarteschlange.
|
|
31
|
+
- Schema-validierte Reviews und optionaler Qualitätsvergleich mit vertauschter Kandidatenreihenfolge und Konsistenzprüfung.
|
|
32
|
+
- Expliziter, versionierbarer Projektkontext und lokale Belege; kompakte Ausgabe mit Artefaktverweisen.
|
|
33
|
+
- Benchmarkdokumentation benennt fehlende Telemetrie und fehlgeschlagene Läufe.
|
|
34
|
+
|
|
35
|
+
### Konkrete Lücken und Grenzen
|
|
36
|
+
|
|
37
|
+
1. Serielles --isolate entfernt den Arbeitsbaum im finally auch bei Fehlern; GitOps verwendet worktree remove --force. Unfertige Änderungen können verloren gehen. Siehe src/loop/loop.ts und src/loop/git.ts.
|
|
38
|
+
2. Ausgeführte Tests sind nicht automatisch unabhängige Abnahmetests: Implementierer kann im untersuchten Pfad Testdateien und Testskripte ändern. Geschützte Abnahmen bzw. Kontrolle von Testabschwächungen fehlen.
|
|
39
|
+
3. Reviews, Audit, integrierte Abschlussprüfung und Browserbelege sind teils optional. README-Garantien müssen zwischen unterstützt, aktiviert und nachgewiesen unterscheiden.
|
|
40
|
+
4. flow-smoke prüft Seitenaufruf, Fehler und optionale Selektoren; kein genereller Nachweis mehrstufiger Benutzerabläufe.
|
|
41
|
+
5. repositoryFingerprint erfasst Inhalte bestehender unversionierter Dateien nicht; bei Git-Fehlern liefert es einen leeren String. Zusätzliche Reviewer-Schreibkontrolle ist damit unvollständig.
|
|
42
|
+
6. Routing nutzt grobe Kostentiers und Erfolgsquoten. Fehlende Telemetrie wird teilweise als 0 aggregiert. Vollständige Kosten für Controller, Worker, Reviews, Reparaturen und Kandidaten fehlen als verlässlicher Gesamtvertrag.
|
|
43
|
+
7. Design-Scan prüft Stilmerkmale wie Lila/Verläufe; kein allgemeiner UX-, Accessibility- oder KI-Autorschaftsnachweis.
|
|
44
|
+
8. Bisherige Benchmarks belegen keine allgemeine Überlegenheit gegenüber nativen Agenten oder Konkurrenz.
|
|
45
|
+
|
|
46
|
+
### Anbieterparität
|
|
47
|
+
|
|
48
|
+
- Codex: Skills, Aufrufrichtlinie, Rollen, JSON-Ausgabe, Modell/Reasoning, RTK-Hook. Direkte Verbindung zum nativen Goal-Zustand fehlt im untersuchten Code. Adaptives Routing deaktiviert native Multi-Agent-Funktionalität bewusst.
|
|
49
|
+
- Claude: Skills, manuelle Aufrufsteuerung, Streaming-JSON, Modell/Effort, RTK-Hook mit Plattformbedingungen. Native strukturierte Ausgabe und Teamfunktionen sind nicht durchgängig ausgenutzt.
|
|
50
|
+
- Gemini: native Skills plus Slash-Commands vorhanden. Adapter fordert kein stream-json an, obwohl Auswertung JSON-Ereignisse erwartet und Reviews berichtete Modellidentität verlangen. Gemeinsame Reasoning-/Bare-Optionen werden nicht entsprechend umgesetzt. Installer-Annahme fehlender Rewrite-Hooks ist veraltet; BeforeTool unterstützt Argumentänderungen.
|
|
51
|
+
- Native Provider-Funktionen einzeln nutzen und auf ein gemeinsames Ergebnisformat abbilden; nicht auf identische APIs aller Anbieter warten.
|
|
52
|
+
- Reale Vertragsfälle je CLI/Version/Plattform: Implementierung, Review, Ausgabe, Modellidentität, Telemetrie, Rechte, Abbruch, Wiederaufnahme und Skill-Aufruf.
|
|
53
|
+
- Lokal geprüft: codex-cli 0.153.4, Claude Code 2.1.200, Gemini CLI 0.33.1. Verfügbare CLI bedeutet keine nachgewiesene Authentifizierung oder erfolgreiche Modellaufgabe.
|
|
54
|
+
|
|
55
|
+
### Benchmarkgrenzen
|
|
56
|
+
|
|
57
|
+
bench/RESULTS.md dokumentiert eine Codex-Routingstudie mit drei Vergleichspaaren. Alle versteckten Abnahmen bestanden; Median ungefähr 33,8 % weniger Laufzeit und 11 % weniger frische Eingabetokens. Vergleich: Routing an/aus innerhalb Yokes, nicht Yoke gegen natives Codex. Andere Architekturaufgaben blieben SELF und bezahlten Controller-Overhead. Keine allgemeinen Sparprozente versprechen.
|
|
58
|
+
|
|
59
|
+
## Produktwette: täglicher Nutzen
|
|
60
|
+
|
|
61
|
+
Yoke beantwortet: "Kann ich diese Änderung übernehmen, und wie bekommen wir sie bei offenen Befunden fertig?"
|
|
62
|
+
|
|
63
|
+
Vorgeschlagene Positionierung: gemeinsame Abnahme- und Fortsetzungsschicht für Coding-Agenten. Native Agenten verfolgen Ziele; Yoke hält überprüfbaren Projektzustand, Kriterien und Abnahme stabil. Kein unnötiger zweiter Orchestrator über nativen Goals.
|
|
64
|
+
|
|
65
|
+
### Einstieg: yoke check (Vorschlag, noch kein implementierter Befehl)
|
|
66
|
+
|
|
67
|
+
- Bestehendes Repository, vorhandener Diff und konkrete Anforderung reichen für den Einstieg; vollständiger Retrofit soll nicht Voraussetzung sein.
|
|
68
|
+
- Ausgabe unterscheidet bestanden, fehlgeschlagen und nicht überprüft.
|
|
69
|
+
- Bevorzugt ausführbare Befunde: Testreproduktion, Browserablauf, Vertragsverletzung statt spekulativer Review-Kommentare.
|
|
70
|
+
- Beispielvorführung: bestehende Tests grün, Yoke reproduziert eine doppelte Bestellung bei Doppelklick, Reparatur und erneute Abnahme belegen die Behebung.
|
|
71
|
+
- Belege sind an den tatsächlich geprüften Codezustand gebunden.
|
|
72
|
+
- Kritische Abnahmetests gegebenenfalls durch gezielte Mutation prüfen: erkennt der Test den passenden absichtlich eingebauten Fehler?
|
|
73
|
+
|
|
74
|
+
### Wiederaufnahme und Anbieterwechsel
|
|
75
|
+
|
|
76
|
+
Übergabepaket enthält Ziel, Kriterien, Patch/Arbeitsstand, Umgebung, bestandene Prüfungen, offene Fehler, verworfene Ansätze, Berechtigungen und verbleibendes Budget. Änderungen und Belege bleiben bei Fehlern erhalten. Anbieterwechsel erhält überprüften Zustand und verlangt keine erneute Erklärung durch den Nutzer.
|
|
77
|
+
|
|
78
|
+
Blocker unterscheiden: Implementierungsfehler, Infrastruktur/Rate-Limit, fehlende Zugangsdaten, echte Produktentscheidung. Ein Anbieterwechsel löst nicht jede Blockade.
|
|
79
|
+
|
|
80
|
+
### Zielmodell
|
|
81
|
+
|
|
82
|
+
Gemeinsamer Zielzustand verbindet PRD, Kriteriennachweise, Änderungs-Inbox, Budget, Blocker, Wiederaufnahme und integrierte Abschlussprüfung. Native Codex Goals integrieren, keine Abschlussgarantie allein aus Modelltext ableiten. Modell darf Lösungsweg wählen; verbindliche Abnahmebedingungen bleiben nachvollziehbar.
|
|
83
|
+
|
|
84
|
+
### Zielgruppe und Differenzierung
|
|
85
|
+
|
|
86
|
+
Vorgeschlagener erster Fokus: Entwickler und kleine Teams mit bestehenden TypeScript-Webprojekten und bereits genutzten Coding-Agenten. Erst dort Einrichtung und Abnahme zuverlässig machen.
|
|
87
|
+
|
|
88
|
+
Skills, Autonomie, frischer Kontext und Zweitmeinungen sind kein exklusiver Vorsprung. Aktuelle Konkurrenz: native Codex Goals/Subagenten, Superpowers auch mit Codex/Gemini, gstack mit Codex/QA/Reviews, GSD Core mit mehreren Hosts und Phasenworkflow.
|
|
89
|
+
|
|
90
|
+
Aufbauender Vorteil: robuste Projektintegration, wiederverwendbare Abnahmefälle, zuverlässige Wiederaufnahme, echte Erfolgsdaten nach Aufgabentyp und nützliche PR-Berichte. Team-Zahlungsbereitschaft ist eine unbestätigte Hypothese.
|
|
91
|
+
|
|
92
|
+
Validierung: zehn passende Entwickler mit echten Änderungen; Zeit bis zum nützlichen Befund, Reproduzierbarkeit, Fehlalarme, eingesparte Nachprüfung und freiwillige Wiederverwendung messen.
|
|
93
|
+
|
|
94
|
+
## Token-, Kosten- und Geschwindigkeitsstrategie
|
|
95
|
+
|
|
96
|
+
Optimierungsziel: Kosten und Zeit pro unabhängig abgenommener Änderung, einschließlich Fehlversuchen und menschlicher Nacharbeit. Tokenzahl, Geldkosten und Wartezeit getrennt betrachten.
|
|
97
|
+
|
|
98
|
+
1. Deterministische Aktionen ohne Modell: Formatter/Linter, AST-/LSP-Renames, Schema-Generatoren, Logparser, Versionssynchronisierung, geprüfte Codemods und Symbolsuche. Voraussetzungen/Nachbedingungen prüfen.
|
|
99
|
+
2. Regeln vor Routing-Modell: eindeutige Aufgaben ohne Controller-Aufruf zuordnen; unklare Fälle durch Modell entscheiden lassen.
|
|
100
|
+
3. Ausführungsstufen Werkzeug / Schnell / Standard / Stark. Risiko, Testbarkeit, Umfang und beobachtete Ergebnisse bestimmen die Auswahl; Dateianzahl oder Modell-Selbstvertrauen reichen nicht.
|
|
101
|
+
4. Günstiger Erstversuch nur bei geeigneten Aufgaben; Abnahme, begrenzte Reparatur, dann Eskalation mit Patch und Fehlerbelegen. Wiederholte Fehler erkennen. Erwartete Gesamtkosten inklusive Eskalation optimieren.
|
|
102
|
+
5. Aufgabenbezogene Kontextpakete: Ziel, Kriterien, Symbole, Verträge, Tests und relevante Entscheidungen. Weitere Informationen bei Bedarf abrufen; Parent-Historie nicht standardmäßig kopieren.
|
|
103
|
+
6. Gemeinsame Exploration/Indexierung wiederverwenden und per Codezustand/Dateihash invalidieren. Keine vier identischen Repository-Erkundungen durch vier Worker.
|
|
104
|
+
7. Stabile Prompt-Präfixe, passende Modellkontinuität, gemessene Cache-Treffer. Caching reduziert nicht automatisch logischen Kontext; keine Cache-Übernahme zwischen Anbietern annehmen. CLI- und API-Fähigkeiten unterscheiden.
|
|
105
|
+
8. Parallelität nach Abhängigkeiten, Schreibbereichen, kritischem Pfad und Ressourcen. Schnittstellen zuerst; danach unabhängige Implementierung. Ein gemeinsames Limit für Yoke-Worker und native Subagenten.
|
|
106
|
+
9. Prüfungen stufenweise: schnelle deterministische Prüfungen, betroffene Tests, Integration, semantisches Review, erforderliche Gesamtprüfung. Ergebnisse nur bei passenden Code-/Umgebungs-/Konfigurationsständen wiederverwenden.
|
|
107
|
+
10. Kleine verwandte Aufgaben bündeln; sichere Build-/Paket-Caches und vorbereitete Umgebungen nutzen. Veränderliche Worker-Arbeitsstände getrennt halten.
|
|
108
|
+
11. Später direkte Modellaufrufe für eng begrenzte Klassifikation/Umformung erwägen, wenn CLI-Start unverhältnismäßig ist. Separate API-Abrechnung berücksichtigen.
|
|
109
|
+
12. Vollständige Telemetrie für alle Rollen; unbekannte Nutzung niemals als gemessene Null darstellen.
|
|
110
|
+
|
|
111
|
+
Reihenfolge vorgeschlagen: Messung, deterministische Aktionen/Router, Kontextpakete, Eskalation, besserer Scheduler, inkrementelle Prüfungen und Cache-Optimierung.
|
|
112
|
+
|
|
113
|
+
## Zeitschätzungen: neuer Schwerpunkt
|
|
114
|
+
|
|
115
|
+
### Heutiger Codebefund
|
|
116
|
+
|
|
117
|
+
src/loop/reporter.ts speichert bis zu 50 Story-Laufzeiten in .yoke/story-durations.json. Die ETA ist der arithmetische Durchschnitt abgeschlossener Stories multipliziert mit der Anzahl verbleibender Stories. Aktuelle Run-Dauern ersetzen die ältere Historie bereits nach dem ersten Abschluss. Gespeicherte StoryDuration enthält nur storyId und ms. Diese Formel berücksichtigt weder individuelle Aufgabengröße noch Modell, Ressourcen oder parallelen kritischen Pfad. Die Aussage bezieht sich auf diese ETA-Implementierung; nicht jede Parallelansicht wurde gesondert vermessen.
|
|
118
|
+
|
|
119
|
+
### Vorgeschlagene Verbesserung
|
|
120
|
+
|
|
121
|
+
- Exakte vergangene Dauer messen; zukünftige Dauer als Schätzung mit Unsicherheit anzeigen. Keine sekundengenaue Vorhersage versprechen.
|
|
122
|
+
- Phasen getrennt erfassen: Warteschlange, Kontext/Setup, Implementierung, Tests, Review, Reparatur, Integration. Aktive Ausführungszeit, Wartezeit und menschliche Blockade auseinanderhalten.
|
|
123
|
+
- Alle Versuche inklusive Fehlern und Abbrüchen erfassen; nur erfolgreiche Story-Dauern würden Wiederholungsaufwand unterschätzen.
|
|
124
|
+
- Vergleichbare Aufgaben nach Typ, Scope, Testumfang, Provider, tatsächlichem Modell, Effort, Umgebung und Parallelitätsgrad gruppieren. Mit wenigen Daten robuste gemeinsame Basis verwenden statt überfeine Gruppen.
|
|
125
|
+
- Historie und neue Beobachtungen gewichten; ein einzelner schneller Abschluss darf nicht die ganze Prognose dominieren.
|
|
126
|
+
- Zunächst Median und empirische Zeitspannen mit Stichprobenzahl; später kalibrierte Quantile, etwa P50/P80, wenn genug Daten vorliegen. Zielabdeckung und Prognosefehler messen.
|
|
127
|
+
- Projekt-ETA aus verbleibenden Aufgaben, Abhängigkeiten, freien Slots, Integrationsengpass und Ressourcen berechnen; weder einfach aufsummieren noch blind durch Workeranzahl teilen.
|
|
128
|
+
- Laufende Aufgaben anhand ihrer aktuellen Phase und verstrichenen Zeit aktualisieren. Wiederholungs-/Reparaturwahrscheinlichkeit und Modellwechsel berücksichtigen.
|
|
129
|
+
- Bei unbekannter Dauer einer Nutzerentscheidung: "wartet auf Entscheidung" und bedingte Restlaufzeit ab Wiederaufnahme; keine erfundene Fertigstellungsuhrzeit.
|
|
130
|
+
- Bei neuer Aufgabe/Modell unbekannte oder schwach gestützte Schätzung sichtbar kennzeichnen; Prognose selbst benötigt nicht zwingend einen LLM-Aufruf.
|
|
131
|
+
- Szenarien anbieten: Zeit/Kosten bei anderer Parallelität oder anderem Modell. Als Prognose ausweisen, nicht als zugesagte Einsparung.
|
|
132
|
+
|
|
133
|
+
## Dashboard pro Projekt und projektübergreifend
|
|
134
|
+
|
|
135
|
+
Status: Nutzerinteresse; folgende Ausgestaltung ist ein Vorschlag, noch keine freigegebene Implementierung.
|
|
136
|
+
|
|
137
|
+
- Lokaler Einstieg, gleiche Datenbasis wie CLI. Zunächst registrierte Projektpfade und lesende Übersicht; kein Cloudkonto als Voraussetzung.
|
|
138
|
+
- Projektübersicht: aktives Ziel, Zustand, abgenommene/offene Kriterien, laufende Worker, Blocker, Zeitspanne bis Abschluss, gemessener Verbrauch und Telemetrielücken.
|
|
139
|
+
- Projektdetail: Task-Liste und Abhängigkeitsansicht, Phasen/Zeitleiste, kritischer Pfad, aktuelle Modelle, Reviews, Fehlerreproduktionen, Artefakte und Integrationsstand.
|
|
140
|
+
- Aufmerksamkeit zuerst: Was braucht eine Entscheidung? Welcher Test blockiert? Welches Projekt ist seit wann still? Warum änderte sich die ETA?
|
|
141
|
+
- Taskdetail: ursprüngliche/aktuelle Schätzung, tatsächliche Phasendauern, Versuchshistorie, Patch, Abnahmen, Kosten und Übergaben.
|
|
142
|
+
- Spätere Steuerung: Pause/Wiederaufnahme, kritische Entscheidung beantworten, Anbieterwechsel am sicheren Übergang, Budget/Parallelität ändern. Existierende Locks und Sicherheitsgrenzen wiederverwenden; UI darf keine zweite Ausführungslogik besitzen.
|
|
143
|
+
- Metriken: Zeit/Kosten pro abgenommener Änderung, Erstversuchserfolg, Eskalationsrate, Nacharbeit, Cache-Anteil, menschliche Eingriffe, Prognosefehler und Zeitspannen-Abdeckung.
|
|
144
|
+
- Zuerst versionierte Ereignisse und vollständige Messung schaffen, dann Dashboard. Vorhandene loop-status.json, loop.log, Story-Dauern und Routing-Ereignisse sind Bausteine, aber noch keine vollständige projektübergreifende Ereignishistorie.
|
|
145
|
+
- Telemetrie standardmäßig lokal; externe Team-/Cloudfunktion und Datenumfang später ausdrücklich entwerfen.
|
|
146
|
+
|
|
147
|
+
## Empfohlene Produktabfolge
|
|
148
|
+
|
|
149
|
+
1. Fehlerbehandlung und Provider-Parität stabilisieren; vollständige Ereignisse/Verbrauch/Dauern.
|
|
150
|
+
2. Nützlichen yoke-check-Einstieg aus Review, Verify und Smoke entwickeln.
|
|
151
|
+
3. Geschützte Abnahmen, kontrollierte Reparatur, Wiederaufnahme und Anbieterwechsel.
|
|
152
|
+
4. Zeitprognosen und lesendes Projektdashboard auf derselben Ereignisbasis; anschließend gezielte Steuerung.
|
|
153
|
+
5. Gemeinsames Zielmodell und Team-/CI-Berichte ausbauen; Optimierungen durch Vergleichsläufe validieren.
|
|
154
|
+
|
|
155
|
+
Nicht bereits beschlossen: UI-Technologie, Cloudhosting, API-Providerpreise, konkrete Modellrangliste, genauer Releaseumfang, verbindlicher Zeitplan, bezahltes Produkt oder vollständige Umsetzung aller Vorschläge.
|
|
156
|
+
|
|
157
|
+
## Quellen und Wiederaufnahme
|
|
158
|
+
|
|
159
|
+
### Umsetzungsstand nach Freigabe
|
|
160
|
+
|
|
161
|
+
Der Nutzer hat anschließend ausdrücklich „ok setze alles akribisch und sicher um“ beauftragt. Die lokale Umsetzung umfasst jetzt unabhängige Checks, geschützte ausführbare Abnahmen, dauerhafte Ziele mit Anbieterwechsel und Verbrauchsgrenzen, sichere serielle Worktree-Wiederaufnahme, Gemini-Adapterkorrekturen, Ereignisse und empirische Zeitspannen, feste Routingregeln mit Eskalation, modellfreie Werkzeugaufgaben, begrenzte kontextbezogene Prompts, deklarierte Schreibbereiche und ein lokales Projektdashboard. Bedienung und Grenzen: [VERIFIED-PROJECTS.md](VERIFIED-PROJECTS.md).
|
|
162
|
+
|
|
163
|
+
Weiterhin offen sind externe Nutzer-/Wettbewerbsversuche, authentifizierte Modellvergleiche, belastbare Kalibrierung der Zeitprognosen und kommerzielle/Cloud-Entscheidungen. Selektive Testwiederverwendung und webbasierter Start/Resume sind bewusst keine behaupteten Fähigkeiten dieses lokalen Ausbaus. Qualitätskontrollen werden vollständig ausgeführt. Die ursprünglichen Produktthesen bleiben als solche dokumentiert.
|
|
164
|
+
|
|
165
|
+
Lokale Anker: src/agents/providers.ts, src/agents/telemetry.ts, src/retrofit/planners/, src/loop/loop.ts, src/loop/git.ts, src/loop/runner.ts, src/loop/reporter.ts, src/loop/scheduler.ts, src/routing/router.ts, src/routing/registry.ts, src/context/context.ts, src/smoke/command.ts, src/scan/design.ts, bench/RESULTS.md.
|
|
166
|
+
|
|
167
|
+
Am 2026-09-05 gelesene Primärquellen; vor konkreten Versions-/Preisentscheidungen erneut prüfen:
|
|
168
|
+
|
|
169
|
+
- https://learn.chatgpt.com/use-cases/follow-goals
|
|
170
|
+
- https://learn.chatgpt.com/docs/agent-configuration/subagents
|
|
171
|
+
- https://developers.openai.com/api/docs/guides/latency-optimization
|
|
172
|
+
- https://developers.openai.com/api/docs/guides/prompt-caching
|
|
173
|
+
- https://code.claude.com/docs/en/cli-reference
|
|
174
|
+
- https://code.claude.com/docs/en/agent-teams
|
|
175
|
+
- https://www.anthropic.com/engineering/effective-context-engineering-for-ai-agents
|
|
176
|
+
- https://geminicli.com/docs/cli/headless/
|
|
177
|
+
- https://geminicli.com/docs/hooks/reference/
|
|
178
|
+
- https://ai.google.dev/gemini-api/docs/caching
|
|
179
|
+
- https://github.com/obra/superpowers
|
|
180
|
+
- https://github.com/garrytan/gstack
|
|
181
|
+
- https://github.com/open-gsd/gsd-core
|
|
182
|
+
|
|
183
|
+
Provenienz der ursprünglichen README-Prüfung: kein C2PA gefunden, unterstützter Scan vollständig, Verifikation/Vertrauen/Metadatenprivatsphäre unbekannt. Unicode-Befund: Emoji-Variationszeichen, kein Nachweis eines KI-Wasserzeichens. Proprietäre Wasserzeichen nicht überprüfbar. Dieses Gesprächsdokument wurde vom KI-Assistenten aus der Sitzung zusammengefasst; es enthält keine unabhängige Bestätigung der Produktthesen.
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
## Fortsetzung am 2026-09-06: Defaults und Dashboard
|
|
187
|
+
|
|
188
|
+
Der Nutzer hat die kombinierte Umsetzung von automatischem Routing/Parallelismus und den drei Dashboardansichten Jetzt, Verbrauch & Zeit sowie Ergebnisse beauftragt. Die Änderungen liegen lokal als unveröffentlichter Ausbau vor; Paketversion und zuletzt veröffentlichtes Release bleiben 1.7.0.
|
|
189
|
+
|
|
190
|
+
Umgesetzt: Routing im asynchronen Workerpfad; neue Setups mit Routing an, Parallelität auto und Isolation an; konservativ höchstens zwei Yoke-Worker bei deklarierten Schreibbereichen; Respektierung expliziter Einstellungen; dauerhafte Messhistorie getrennt von der kurzen Aktivitätsliste; Tages-/Wochen-/Monatsauswertung in UTC; Modell- und Projektvergleich; Verbrauchsdiagramm; aktuelle Aufgaben und Phasen; Abnahmen und Aufwand pro Abnahme. Verfügbare Reviewer-, Kritiker- und Reparaturnutzung wird mit erfasst. Details und Grenzen stehen in VERIFIED-PROJECTS.md.
|
|
191
|
+
|
|
192
|
+
Weiterhin keine behaupteten Fähigkeiten: dynamische gemeinsame Nutzung von Slots durch native Subagenten (native Delegation ist für Loop-Aufrufe bei allen drei Anbietern deaktiviert), exakte Generierungsgeschwindigkeit, vollständige Rekonstruktion alter Verbrauchsdaten, automatische monatliche Archivverdichtung oder kalibrierte Zeitprognosen. Bestehende Quality-Reparaturlimits bleiben erhalten; konkurrierende Kandidaten bleiben optional.
|
|
193
|
+
|
|
194
|
+
Die Umsetzung wurde mit Tests und einer lokalen Browserprüfung geprüft; authentifizierte Modellbenchmarks und Veröffentlichung waren kein Bestandteil dieser Fortsetzung. Dieses Update wurde vom KI-Assistenten aus der laufenden Umsetzung festgehalten.
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
### Releaseauftrag am 2026-09-06
|
|
198
|
+
|
|
199
|
+
Der Nutzer hat anschließend maximal drei Worker im Automatikmodus und die Veröffentlichung der Weiterentwicklung beauftragt. Releaseziel ist 1.8.0; der frühere lokale Zwischenstand mit zwei Workern ist damit überholt. Jede neue Version muss vor Veröffentlichung einen datierten Changelogeintrag erhalten; die verbindliche Regel steht in AGENTS.md. Der tatsächliche Veröffentlichungsstatus wird über GitHub Release und npm geprüft.
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
# Local implementation validation
|
|
2
|
+
|
|
3
|
+
This record covers the September 5, 2026 source implementation, not a published release or a competitive benchmark.
|
|
4
|
+
|
|
5
|
+
## Review and behavioral checks
|
|
6
|
+
|
|
7
|
+
Independent specification and code-quality reviews identified and drove fixes for mutable acceptance, stale goal success, lost interrupted-attempt accounting, orphaned processes, incomplete workspace identity, missing final protection gates, restart-lost routing escalation, unnecessary provider installation requirements, mixed time accounting, partial cost reporting and linked pause files.
|
|
8
|
+
|
|
9
|
+
Behavioral regressions cover protected checks, source mutation during verification, truthful unmapped requirements, goal completion from real application changes, tampered tests, bounded retries, interruption recovery, source-bound worktree continuation, routing without controller calls, persisted escalation, deterministic actions, context budgets, dependency/write-scope scheduling, unknown estimates, phase/attempt evidence, local registry and HTTP security boundaries.
|
|
10
|
+
|
|
11
|
+
Final full suite: **122 test files passed; 1,098 tests passed, 2 skipped (1,100 total)** with a 20-second test timeout on Windows. This timeout accommodates the real-Git integration tests; the earlier baseline already had a 5-second timeout that passed in isolation. TypeScript build/lint, canon validation, release-metadata check and package dry run passed. The package check explicitly confirmed the compiled CLI, check/goal/dashboard modules and Gemini RTK hook are included.
|
|
12
|
+
|
|
13
|
+
## Browser and provider limits
|
|
14
|
+
|
|
15
|
+
The local dashboard was rendered and inspected in Chrome on desktop and at 390px mobile width. HTTP tests exercise registered project reads, corrupt/missing/oversized data, hostile Host/Origin headers, token rejection and authorized pause. Browser extension click automation timed out, so a complete interactive click walkthrough is not claimed.
|
|
16
|
+
|
|
17
|
+
Provider invocation/stream/parser/hook tests are local automated tests. No authenticated cross-provider development benchmark was run. Model quality parity, price superiority, exact deadline prediction and achieved real-project token savings remain unmeasured. Native structured-output adapter support does not imply a provider-native goal API.
|
|
18
|
+
|
|
19
|
+
## Provenance
|
|
20
|
+
|
|
21
|
+
Read-only `audit-provenance` scans covered README, the new usage guide and dashboard page source. No supported C2PA structure was found; supported scans completed. README Unicode findings were emoji variation selectors, not evidence of a model watermark. No content or marks were removed.
|
|
22
|
+
|
|
23
|
+
The audit's unanswered questions remain explicit:
|
|
24
|
+
|
|
25
|
+
- Verification/trust: “No conforming verifier was supplied.” Cryptographic verification and a named signer trust chain are unknown.
|
|
26
|
+
- Metadata privacy: the text format has no supported metadata parser in this analyzer; no privacy conclusion was produced.
|
|
27
|
+
- Proprietary watermark detection: “Keyed model-level watermarks cannot be checked without the provider's key.” No authorship inference follows from absence of detected marks.
|
|
28
|
+
|
|
29
|
+
The documentation and implementation were prepared with AI assistance and reviewed/tested as described above.
|
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
# Verified projects, goals and measured execution
|
|
2
|
+
|
|
3
|
+
Available in Yoke 1.7.0. These additions do not require a cloud account or replace your provider CLI. Model availability, reasoning controls, structured output and telemetry depend on the selected provider; unknown usage is never a measured zero. Live authenticated provider comparison and calibrated time/cost benchmarks remain separate validation work.
|
|
4
|
+
|
|
5
|
+
## Check an existing project
|
|
6
|
+
|
|
7
|
+
```sh
|
|
8
|
+
yoke check .
|
|
9
|
+
yoke check . --json
|
|
10
|
+
yoke check . --requirement="Guest checkout completes"
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
Checks run without retrofit. Exit codes: `0` passed, `1` failed, `2` unverified or unavailable. A free-text requirement stays unverified until you map its behavior to an executable acceptance contract. Passing a project suite does not establish arbitrary product correctness.
|
|
14
|
+
|
|
15
|
+
Create `.yoke/acceptance.yaml` with tests that inspect application behavior:
|
|
16
|
+
|
|
17
|
+
```yaml
|
|
18
|
+
version: 1
|
|
19
|
+
protected:
|
|
20
|
+
- tests/acceptance/checkout.test.mjs
|
|
21
|
+
- tests/acceptance/helpers.mjs
|
|
22
|
+
criteria:
|
|
23
|
+
- id: guest-checkout
|
|
24
|
+
text: A guest can complete checkout and receives one order confirmation.
|
|
25
|
+
commands:
|
|
26
|
+
- node --test tests/acceptance/checkout.test.mjs
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
List every local test helper, fixture and configuration file whose modification could weaken this contract. Paths must resolve inside the project. `yoke check . --protect` pins the manifest, listed files and existing package manifests/lockfiles outside the worker checkout. Autonomous goals require explicit protected test infrastructure and establish this pin automatically. Protection is change detection within the local execution model, not an OS security boundary against a hostile process using your account. Review the contract before running a goal.
|
|
30
|
+
|
|
31
|
+
After intentionally editing protected tests, use `yoke check . --protect --refresh` to approve the new baseline. Ordinary checks and retries never refresh it. Source, acceptance and configuration identity are checked before and after verification; changed inputs invalidate evidence. Gitlinks/submodule directories currently fail closed because recursive identity is not implemented. Evidence is stored in `.yoke/checks/<id>.json`; it describes that checked snapshot and is not a permanent success certificate.
|
|
32
|
+
|
|
33
|
+
## Durable goals across providers
|
|
34
|
+
|
|
35
|
+
```sh
|
|
36
|
+
yoke goal set . --objective="Finish guest checkout" --attempts=3 --minutes=30
|
|
37
|
+
yoke goal run . --runner=codex
|
|
38
|
+
yoke goal resume . --runner=claude --model=<installed-model-id>
|
|
39
|
+
yoke goal resume . --runner=gemini --model=<installed-model-id>
|
|
40
|
+
yoke goal status .
|
|
41
|
+
yoke goal handoff .
|
|
42
|
+
yoke goal pause .
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
Goals keep objective, attempts, failure context and check IDs in `.yoke/goal.json`. They share the story-loop lock. Each implementation attempt is followed by independent executable acceptance. A completed goal is checked again on a later run. Failed work stays in the project; goal execution itself does not commit or publish it. Goal handoff is readable context for native agent goal facilities; Yoke does not invent or call an undocumented native goal API.
|
|
46
|
+
|
|
47
|
+
`--minutes` limits cumulative agent execution time. Verification is measured separately and commands retain their verification timeout. `--tokens=N` is a checkpoint budget: it prevents another attempt when measured consumption is exhausted or unknown, but cannot promise a hard token cap within a single provider call. An interrupted attempt is persisted before dispatch; recovery accounts for it conservatively and reconciles recorded provider processes before continuing. Unknown process ownership blocks execution. Pause takes effect at an attempt boundary; it does not instantly kill an in-flight agent.
|
|
48
|
+
|
|
49
|
+
Explicitly extend total budgets without deleting history:
|
|
50
|
+
|
|
51
|
+
```sh
|
|
52
|
+
yoke goal budget . --attempts=5 --minutes=60
|
|
53
|
+
yoke goal budget . --tokens=200000
|
|
54
|
+
# Only if you deliberately want to remove the checkpoint token limit:
|
|
55
|
+
yoke goal budget . --clear-token-budget
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
Unknown historical token usage cannot be made known by increasing its limit. Inspect interrupted work and keep that limitation visible.
|
|
59
|
+
|
|
60
|
+
## Recover an isolated story
|
|
61
|
+
|
|
62
|
+
Failed or paused isolated story worktrees are retained. Resume the same story with:
|
|
63
|
+
|
|
64
|
+
```sh
|
|
65
|
+
yoke loop run . --isolate --resume-worktree --parallel=1 --candidates=1
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
Yoke validates the registered worktree, repository, original target commit and PRD digest. Changed target/PRD state is refused; inspect retained edits and reconcile deliberately. This flag is for serial isolated recovery; parallel workers retain their existing coordinator lifecycle. `yoke loop cleanup` remains an explicit cleanup operation; inspect its removal flags before discarding unfinished work. Existing projects should rerun retrofit to add ignore entries for checks, events and goals.
|
|
69
|
+
|
|
70
|
+
## Spend fewer model calls
|
|
71
|
+
|
|
72
|
+
Add explicit project rules to `.yoke/config.yaml`:
|
|
73
|
+
|
|
74
|
+
```yaml
|
|
75
|
+
routing:
|
|
76
|
+
enabled: true
|
|
77
|
+
strategy: cost
|
|
78
|
+
maxCandidates: 2
|
|
79
|
+
workers:
|
|
80
|
+
- id: fast
|
|
81
|
+
agent: claude
|
|
82
|
+
model: <your-fast-model-id>
|
|
83
|
+
costTier: low
|
|
84
|
+
capabilities: [docs, tests]
|
|
85
|
+
- id: strong
|
|
86
|
+
agent: codex
|
|
87
|
+
model: <your-strong-model-id>
|
|
88
|
+
costTier: high
|
|
89
|
+
capabilities: [implementation]
|
|
90
|
+
rules:
|
|
91
|
+
- area: docs
|
|
92
|
+
worker: fast
|
|
93
|
+
escalateTo: strong
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
The first matching rule bypasses the routing controller. Its optional area and story ID selectors both have to match when supplied. Independent gate failure escalates the next attempt, including after loop restart; an unavailable or unknown worker falls back to the parent. Without an explicit escalation target the parent handles the failed rule. Other stories retain adaptive routing. Gate-driven routing observations are local and bounded when read. These are measured outcomes, not guarantees that a named inexpensive model can handle every task.
|
|
97
|
+
|
|
98
|
+
For deterministic operations, configure exact executable arguments:
|
|
99
|
+
|
|
100
|
+
```yaml
|
|
101
|
+
actions:
|
|
102
|
+
- storyId: regenerate-types
|
|
103
|
+
file: node
|
|
104
|
+
args: [scripts/generate-types.mjs]
|
|
105
|
+
timeoutMs: 60000
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
Actions use no model, run without a shell, have bounded output/time and still pass the ordinary story gates before commit. They currently require serial execution with one candidate. On Windows use an executable such as `node` and its script path; shell-only `.cmd` wrappers are not automatically enabled. Commands are project-authored configuration, never free-form model-generated commands. Projects consisting entirely of configured actions do not need a model CLI unless they enable a model-based review or planning workflow.
|
|
109
|
+
|
|
110
|
+
Implementation/review prompts now select task-relevant project context deterministically within a 6,000-character budget. A stable project/glossary prefix precedes ranked historical references with source paths and content hashes. Excerpts point back to complete files. This does not claim a fixed token count or guaranteed provider cache hit. Mandatory verification always reruns; no broad result cache or selective-test bypass was introduced.
|
|
111
|
+
|
|
112
|
+
## Parallelism and time estimates
|
|
113
|
+
|
|
114
|
+
Stories can declare relative file/directory scopes:
|
|
115
|
+
|
|
116
|
+
```yaml
|
|
117
|
+
- id: api-types
|
|
118
|
+
title: Update API types
|
|
119
|
+
priority: 1
|
|
120
|
+
area: api
|
|
121
|
+
writes: [src/api, tests/api]
|
|
122
|
+
needs: []
|
|
123
|
+
acceptance: ["Replace with executable structured criteria in strict projects"]
|
|
124
|
+
passes: false
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
Overlapping declared scopes cannot run simultaneously, including while integration is pending. Dependencies, areas, explicit priorities and concurrency limits also apply. Equal-priority work that unlocks a longer dependency chain is scheduled first. Declarations are advisory scheduling input, not filesystem write enforcement; absent declarations preserve previous behavior. Final integrated gates remain mandatory.
|
|
128
|
+
|
|
129
|
+
Versioned local events record status, phase duration, attempts and available usage. Retention is bounded to 1,000 events. Estimates combine historical and current durations, preserve failed-attempt cost and expose sample counts and empirical ranges. Schedule estimates simulate dependency, scope and concurrency constraints. Predictions and later errors are attached to completed attempt events so accuracy can be evaluated. Ranges describe observed data; they are not calibrated probabilities or exact deadlines. No history means no justified time estimate.
|
|
130
|
+
|
|
131
|
+
## Local project dashboard
|
|
132
|
+
|
|
133
|
+
### Execution defaults in 1.8.0
|
|
134
|
+
|
|
135
|
+
New setups enable routing, `loop.parallel: auto` and `loop.isolate: true`. Existing explicit settings remain authoritative. At execution time, automatic routing uses configured profiles; without profiles it keeps the selected parent. Explicit `--routing` without profiles still reports a configuration error. Routing rules bypass the controller and can escalate following failed independent gates, including across worktrees and restarts. Routing now also runs asynchronously inside parallel workers. An explicit task provider affinity takes precedence over routing.
|
|
136
|
+
|
|
137
|
+
Automatic parallelism allows at most three Yoke workers when every pending task declares nonempty write scopes. Dependencies and overlapping scopes still constrain dispatch. Unknown scopes, configured tool actions and worktree recovery select serial execution. Use `--parallel=N`, `--parallel=auto`, `--no-routing` or `--no-isolate` to override defaults. Serial worktrees retain failed work and require deliberate recovery; the default does not discard an existing recovery tree. Quality repair budgets and opt-in competing candidates retain their existing policies.
|
|
138
|
+
|
|
139
|
+
Integration retains an execution slot until its candidate lands. Yoke loop runners disable native delegation so it cannot multiply the default worker budget: Codex disables multi_agent; Claude disallows Agent, Task, TeamCreate and SendMessage; Gemini uses a separate temporary system-settings copy that disables experimental agents and its always-on investigator/help overrides. Existing Gemini system policy and system-default paths are preserved; unreadable or malformed policy blocks launch. Original settings are never overwritten. The temporary copy is removed on normal exit; forced process termination may leave a private temporary directory. Explicit competing candidate counts remain a separate opt-in workload.
|
|
140
|
+
|
|
141
|
+
### Dashboard views
|
|
142
|
+
|
|
143
|
+
- **Now:** reported task and worker activity, requested provider/model, elapsed worker time, phase, integration, blockers, status age, backlog and empirical remaining-time ranges. The view refreshes every five seconds while visible and not being operated with keyboard focus. A stale status is marked rather than asserted to be live.
|
|
144
|
+
- **Usage & time:** last 24 hours, 7/30/90/365 days or custom dates, grouped by UTC day, Monday-based week or month. Displays reported input/output and cache categories, cost coverage, consumption charts, per-model time buckets, per-task usage and summed phase/call durations. The overview also compares projects for the same period.
|
|
145
|
+
- **Results:** recorded acceptances, ended attempts, explicitly successful outcomes, repair phases, rule-driven escalations and recorded tokens/time per acceptance. Saved acceptance evidence and goal attempts remain available; their timestamps may fall outside the statistics period.
|
|
146
|
+
|
|
147
|
+
The short activity list still retains at most 1,000 events. Compact measurements now also persist under `.yoke/history/YYYY-MM-DD/` independently of that retention and across runs. They contain identifiers, model/provider evidence and measurements, not prompts or full status snapshots. Runtime history is excluded from Yoke commits and added to new project ignore rules. Available reviewer, quality critic and repair usage is recorded separately; absent usage is counted as unknown. Recent and archived records are deduplicated by event ID.
|
|
148
|
+
|
|
149
|
+
Dates and buckets use UTC. The custom end date is inclusive; the API uses an exclusive upper timestamp. Usage belongs to the time the provider reports it, and durations to their end time. Tokens per elapsed minute divide recorded input/output by the full selected interval. Tokens per call minute use summed reported call duration, which can overlap across workers. Neither is measured generation speed. Cache categories are shown separately without adding them to input again.
|
|
150
|
+
|
|
151
|
+
Historical activity that was never recorded or already expired cannot be reconstructed. Queue/human waiting time that lacks measurements remains unknown; phase and attempt durations are summed worker time, not automatically elapsed project time. Queries allow at most 366 days and bounded reads (8 MiB per shard, 32 MiB overall, 50,000 records); skipped, malformed or oversized history is reported as incomplete. History currently requires local disk retention rather than automatic monthly compaction. Unknown model identity is never replaced by a requested model, and missing charges are not estimated from token counts.
|
|
152
|
+
|
|
153
|
+
```sh
|
|
154
|
+
yoke projects add /path/to/project
|
|
155
|
+
yoke projects list
|
|
156
|
+
yoke dashboard .
|
|
157
|
+
yoke dashboard --no-register --port=4100
|
|
158
|
+
yoke projects remove <opaque-project-id>
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
Open the printed loopback URL. The dashboard displays project goals, backlog, worker/status data, evidence, blockers, events, usage availability and schedule ranges. It can request a goal pause through the same service as the CLI. Start, resume and budget changes remain explicit CLI operations. Removing registration only removes the registry reference.
|
|
162
|
+
|
|
163
|
+
The HTTP service binds to `127.0.0.1`, validates Host/Origin, requires a per-session token for pause, sends a restrictive CSP and renders project strings as text. It does not serve arbitrary local files or allow remote registration. Missing/corrupt/oversized project data is shown as unavailable. Shared project registration and protected acceptance live under `YOKE_STATE_DIR` or `~/.yoke/state`; existing routing history retains its `YOKE_REGISTRY_DIR` location. The dashboard is local, not a hosted multi-user service.
|
|
164
|
+
|
|
165
|
+
## Remaining validation and product work
|
|
166
|
+
|
|
167
|
+
Live Codex/Claude/Gemini comparisons, measured competitive development-time/cost studies and estimate calibration require representative real projects and authenticated runs. Equal adapter contracts do not imply equal model capability. Cloud/team access, automatic selective test reuse, web-based start/resume controls and commercial rollout are not included in this local foundation. The complete saved direction is in [PRODUCT-DIRECTION-2026-09-05.md](PRODUCT-DIRECTION-2026-09-05.md).
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
# Verified projects implementation plan
|
|
2
|
+
|
|
3
|
+
> **For agentic workers:** Use superpowers:subagent-driven-development for bounded implementation and independent spec and quality reviews. User authorized implementation on 2026-09-05; no additional planning approval is required.
|
|
4
|
+
|
|
5
|
+
**Goal:** Deliver reliable cross-provider acceptance and continuation, measured execution with useful time estimates, efficient routing, and a local dashboard over a shared project state.
|
|
6
|
+
|
|
7
|
+
**Architecture:** Preserve existing CLI behavior unless a safety defect requires correction. Add small services for checks, goals, events, estimates, project registration and dashboard presentation. Provider capabilities stay in adapters; dashboard calls the same services as the CLI. Unknown measurements remain unknown.
|
|
8
|
+
|
|
9
|
+
**Tech Stack:** Existing TypeScript, Node built-ins, Zod, YAML and Vitest. Local HTTP dashboard with no runtime UI dependency. Existing CLI processes remain the coding executors.
|
|
10
|
+
|
|
11
|
+
## Scope and acceptance
|
|
12
|
+
|
|
13
|
+
The user approved the saved product direction. Implementation includes the concrete developer-tool capabilities; market validation with ten external users, commercial pricing, cloud hosting, promises of model equivalence and guaranteed time/cost savings are not software deliverables. Provider-native features must use verified supported interfaces; no invented goal protocol or provider prices.
|
|
14
|
+
|
|
15
|
+
## Task 1: Provider contracts and hooks
|
|
16
|
+
|
|
17
|
+
Files: src/agents/providers.ts, contracts/types/telemetry as necessary, src/retrofit/planners/gemini.ts, canon/tools/gemini-rtk-hook.mjs, focused provider/retrofit tests.
|
|
18
|
+
|
|
19
|
+
- [x] Write failing tests for Gemini streaming and model telemetry, unsupported selection handling, and BeforeTool command rewriting.
|
|
20
|
+
- [x] Run the focused tests and inspect the expected failures.
|
|
21
|
+
- [x] Add capability-aware invocation, reliable Gemini stream output and hook wiring. Retain safe/read-only profiles, validate native structured-output options where supported.
|
|
22
|
+
- [x] Run provider, process and retrofit tests plus TypeScript checks.
|
|
23
|
+
- [x] Independent spec review, then quality review; address findings.
|
|
24
|
+
|
|
25
|
+
## Task 2: Durable recovery and acceptance
|
|
26
|
+
|
|
27
|
+
Files: src/loop/loop.ts, src/loop/runner.ts, new src/check/ and src/goals/ modules, src/cli.ts, focused tests.
|
|
28
|
+
|
|
29
|
+
- [x] Reproduce lost failed worktrees and untracked reviewer mutations before changing production code.
|
|
30
|
+
- [x] Retain failed isolated work; allow intentional resume with branch/PRD identity validation. Fingerprint file content and fail closed on read failures.
|
|
31
|
+
- [x] Add yoke check that runs configured or detected verification without retrofit, reports passed/failed/unverified criteria and binds evidence to checked content.
|
|
32
|
+
- [x] Add protected-file verification and bounded repair/continuation with explicit provider selection; never silently accept changed acceptance infrastructure.
|
|
33
|
+
- [x] Add durable goals and handoff state, exposing native-agent-readable instructions instead of assuming unsupported APIs.
|
|
34
|
+
- [x] Test dirty trees, failed commands, stale evidence, missing configuration, changed protected files and recovery.
|
|
35
|
+
|
|
36
|
+
## Task 3: Measurement and estimates
|
|
37
|
+
|
|
38
|
+
Files: new src/observability/events.ts, src/estimation/, src/loop/reporter.ts, focused tests.
|
|
39
|
+
|
|
40
|
+
- [x] Test event validation, partial history, failed attempts, phase accounting and low-sample estimates.
|
|
41
|
+
- [x] Persist bounded versioned local events and preserve explicit measurement availability.
|
|
42
|
+
- [x] Combine historical and current durations robustly; expose empirical bounds and sample count.
|
|
43
|
+
- [x] Estimate dependency-constrained schedules and active work without naive division by concurrency.
|
|
44
|
+
- [x] Connect serial/parallel status to a common read model and expose estimate accuracy records.
|
|
45
|
+
|
|
46
|
+
## Task 4: Efficient execution
|
|
47
|
+
|
|
48
|
+
Files: src/routing/, src/context/, src/loop/scheduler.ts, configuration/schema and focused tests.
|
|
49
|
+
|
|
50
|
+
- [x] Test explicit deterministic routes, safe fallback and gate-driven escalation.
|
|
51
|
+
- [x] Add explicit rule-based routing to bypass unnecessary controller calls; support bounded configured tool actions and worker escalation.
|
|
52
|
+
- [x] Select relevant context within a stable prefix and content budget with source references.
|
|
53
|
+
- [x] Respect task write scopes, dependency priority and configured concurrency constraints.
|
|
54
|
+
- [x] Keep mandatory final checks; reuse only evidence with matching inputs and surface cache/usage gaps.
|
|
55
|
+
|
|
56
|
+
## Task 5: Project dashboard and shared commands
|
|
57
|
+
|
|
58
|
+
Files: new src/projects/, src/dashboard/, src/cli.ts, tests/projects/, tests/dashboard/.
|
|
59
|
+
|
|
60
|
+
- [x] Test project registration, corrupt/missing projects, task views and HTTP boundaries.
|
|
61
|
+
- [x] Add local project register/list/remove and shared state snapshots for goals, criteria, workers, events, estimates and consumption.
|
|
62
|
+
- [x] Serve a responsive dashboard on loopback only with escaped text, restrictive CSP, validated Host/Origin and opaque project IDs. Never expose arbitrary file paths as HTTP reads.
|
|
63
|
+
- [x] Provide overview, project and task details, blockers, evidence and time/cost uncertainty. Mutating controls reuse existing command boundaries and require local-session protection.
|
|
64
|
+
- [x] Verify actual HTTP responses and browser rendering, empty/error states and navigation.
|
|
65
|
+
|
|
66
|
+
## Task 6: Integration and documentation
|
|
67
|
+
|
|
68
|
+
- [x] Reconcile README claims with activation and evidence requirements; document migration and all new commands.
|
|
69
|
+
- [x] Update release metadata through the repository script.
|
|
70
|
+
- [x] Run lint, build, full tests, canon validation, docs checks and package dry run.
|
|
71
|
+
- [x] Perform independent spec and code-quality review and fix material findings.
|
|
72
|
+
- [x] Preserve original working-tree changes; provide concrete commands and remaining external verification limits.
|
|
73
|
+
|
|
74
|
+
## Verification commands
|
|
75
|
+
|
|
76
|
+
All shell calls use RTK per user instruction. Focused checks use `rtk proxy npx vitest run <test files>`. Final checks use npm scripts from package.json and `rtk proxy npx tsx src/cli.ts validate canon`. Behavioral tests must fail for the intended missing behavior before implementation. Do not call live models across user repositories for a synthetic dashboard demo.
|
|
77
|
+
|
|
78
|
+
## Execution record
|
|
79
|
+
|
|
80
|
+
- Baseline was audited earlier in this session: 1018 passed, 2 skipped, one 5-second integration-test timeout; isolated rerun passed. TypeScript and docs check passed. Work continues under the user's explicit instruction to implement and fix the recorded issues.
|
|
81
|
+
- Worktree: .worktrees/yoke-next, branch feature/verified-projects. Existing dependencies reused via junction; original user changes remain outside this worktree.
|
|
82
|
+
- Final verification: 122 files passed, 1098 tests passed and 2 platform skips. Build, typecheck, canon, docs metadata and package contents passed. Independent reviews and desktop/mobile browser inspection completed within the limits recorded in docs/VERIFIED-PROJECTS-VALIDATION.md.
|
|
83
|
+
- Scope clarification: mandatory final checks always rerun; no selective-result cache was added. Dashboard mutations are limited to shared-service pause; run/resume/budget stay CLI actions. Native provider goal integration is a supported handoff, not an invented API. Empirical prediction errors are recorded; live model/competition benchmarks and calibrated deadlines remain external validation.
|
package/gemini-extension.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "yoke",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.8.0",
|
|
4
4
|
"description": "Cross-agent coding harness: curated skill canon, mechanical safety gates, autonomous loop with proof artifacts. CLI: npm i -g @hecer/yoke",
|
|
5
5
|
"contextFileName": "GEMINI-EXTENSION.md"
|
|
6
6
|
}
|