@try-works/dsh-recursive-mode 0.4.3 → 0.4.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +20 -15
- package/lib/client/index.d.ts +3 -1
- package/lib/client/settings-view.d.ts +97 -0
- package/lib/client/settings.d.ts +25 -1
- package/lib/client.js +530 -11
- package/lib/enforcement.d.ts +37 -0
- package/lib/errors.d.ts +12 -0
- package/lib/goals-projection.d.ts +34 -4
- package/lib/index.js +594 -59
- package/lib/memory-feedback.d.ts +25 -2
- package/lib/recursive_ask.tool.d.ts +60 -0
- package/lib/recursive_init.tool.d.ts +15 -0
- package/lib/run-start.d.ts +55 -0
- package/lib/runtime.d.ts +137 -31
- package/package.json +1 -1
- package/scripts/test-recursive-mode-smoke.ts +5 -1
- package/src/client/index.ts +3 -1
- package/src/client/settings-view.ts +211 -0
- package/src/client/settings.tsx +194 -8
- package/src/client/slots.ts +15 -2
- package/src/client/styles.ts +237 -0
- package/src/enforcement.ts +74 -3
- package/src/errors.ts +12 -0
- package/src/goals-projection.ts +48 -7
- package/src/index.ts +56 -10
- package/src/memory-feedback.ts +50 -5
- package/src/phase-rules.ts +10 -1
- package/src/policy-globs.ts +25 -3
- package/src/policy-write.ts +1 -1
- package/src/recursive_ask.tool.ts +218 -17
- package/src/recursive_init.tool.ts +16 -1
- package/src/run-start.ts +123 -0
- package/src/runtime.ts +186 -14
package/src/client/styles.ts
CHANGED
|
@@ -860,6 +860,243 @@ const BOARD_CSS = `
|
|
|
860
860
|
border-color: var(--dsw-alias-state-warn-primary);
|
|
861
861
|
}
|
|
862
862
|
|
|
863
|
+
/* ===== Settings section (Settings -> Recursive): the live projection report =====
|
|
864
|
+
This panel lives INSIDE the settings shell, so it consumes the shell's own
|
|
865
|
+
--dsw-alias-* theme tokens (like .rec-badge above) instead of the board's
|
|
866
|
+
scoped --board-* paper tokens: no theme toggle, and it follows the app theme. */
|
|
867
|
+
.rec-settings {
|
|
868
|
+
display: flex;
|
|
869
|
+
flex-direction: column;
|
|
870
|
+
gap: 16px;
|
|
871
|
+
max-width: 940px;
|
|
872
|
+
padding: 2px 2px 10px;
|
|
873
|
+
color: var(--dsw-alias-label-primary);
|
|
874
|
+
font-size: 13px;
|
|
875
|
+
line-height: 1.5;
|
|
876
|
+
}
|
|
877
|
+
|
|
878
|
+
.rec-settings-header {
|
|
879
|
+
display: flex;
|
|
880
|
+
align-items: center;
|
|
881
|
+
gap: 10px;
|
|
882
|
+
}
|
|
883
|
+
|
|
884
|
+
.rec-settings-title {
|
|
885
|
+
margin: 0;
|
|
886
|
+
font-size: 16px;
|
|
887
|
+
font-weight: 600;
|
|
888
|
+
letter-spacing: -0.01em;
|
|
889
|
+
}
|
|
890
|
+
|
|
891
|
+
.rec-settings-tag {
|
|
892
|
+
flex: none;
|
|
893
|
+
padding: 2px 8px;
|
|
894
|
+
font-size: 11px;
|
|
895
|
+
font-weight: 500;
|
|
896
|
+
letter-spacing: 0.03em;
|
|
897
|
+
text-transform: uppercase;
|
|
898
|
+
color: var(--dsw-alias-label-tertiary);
|
|
899
|
+
border: 1px solid var(--dsw-alias-border-l2);
|
|
900
|
+
border-radius: 999px;
|
|
901
|
+
}
|
|
902
|
+
|
|
903
|
+
.rec-settings-close {
|
|
904
|
+
margin-left: auto;
|
|
905
|
+
padding: 5px 12px;
|
|
906
|
+
font: inherit;
|
|
907
|
+
font-size: 12px;
|
|
908
|
+
color: var(--dsw-alias-label-primary);
|
|
909
|
+
background: transparent;
|
|
910
|
+
border: 1px solid var(--dsw-alias-border-l2);
|
|
911
|
+
border-radius: 8px;
|
|
912
|
+
cursor: pointer;
|
|
913
|
+
}
|
|
914
|
+
|
|
915
|
+
.rec-settings-close:hover {
|
|
916
|
+
background: var(--dsw-alias-bg-layer-3);
|
|
917
|
+
}
|
|
918
|
+
|
|
919
|
+
.rec-settings-lede,
|
|
920
|
+
.rec-settings-hint,
|
|
921
|
+
.rec-settings-none {
|
|
922
|
+
margin: 0;
|
|
923
|
+
color: var(--dsw-alias-label-secondary);
|
|
924
|
+
}
|
|
925
|
+
|
|
926
|
+
.rec-settings-hint,
|
|
927
|
+
.rec-settings-none {
|
|
928
|
+
font-size: 12px;
|
|
929
|
+
}
|
|
930
|
+
|
|
931
|
+
.rec-settings-section,
|
|
932
|
+
.rec-settings-notcarried,
|
|
933
|
+
.rec-settings-run {
|
|
934
|
+
display: flex;
|
|
935
|
+
flex-direction: column;
|
|
936
|
+
gap: 8px;
|
|
937
|
+
padding: 12px 14px;
|
|
938
|
+
background: var(--dsw-alias-bg-layer-2);
|
|
939
|
+
border: 1px solid var(--dsw-alias-border-l2);
|
|
940
|
+
border-radius: 10px;
|
|
941
|
+
}
|
|
942
|
+
|
|
943
|
+
.rec-settings-h3 {
|
|
944
|
+
margin: 0;
|
|
945
|
+
font-size: 12px;
|
|
946
|
+
font-weight: 600;
|
|
947
|
+
letter-spacing: 0.04em;
|
|
948
|
+
text-transform: uppercase;
|
|
949
|
+
color: var(--dsw-alias-label-tertiary);
|
|
950
|
+
}
|
|
951
|
+
|
|
952
|
+
.rec-settings-h4 {
|
|
953
|
+
margin: 6px 0 0;
|
|
954
|
+
font-size: 12px;
|
|
955
|
+
font-weight: 600;
|
|
956
|
+
color: var(--dsw-alias-label-secondary);
|
|
957
|
+
}
|
|
958
|
+
|
|
959
|
+
.rec-settings-run-header {
|
|
960
|
+
display: flex;
|
|
961
|
+
align-items: center;
|
|
962
|
+
gap: 10px;
|
|
963
|
+
}
|
|
964
|
+
|
|
965
|
+
.rec-settings-run-title {
|
|
966
|
+
margin: 0;
|
|
967
|
+
font-size: 14px;
|
|
968
|
+
font-weight: 600;
|
|
969
|
+
overflow-wrap: anywhere;
|
|
970
|
+
}
|
|
971
|
+
|
|
972
|
+
.rec-settings-run-root {
|
|
973
|
+
flex: 1;
|
|
974
|
+
min-width: 0;
|
|
975
|
+
font-family: ui-monospace, SFMono-Regular, Menlo, monospace;
|
|
976
|
+
font-size: 11px;
|
|
977
|
+
color: var(--dsw-alias-label-tertiary);
|
|
978
|
+
overflow: hidden;
|
|
979
|
+
text-overflow: ellipsis;
|
|
980
|
+
white-space: nowrap;
|
|
981
|
+
}
|
|
982
|
+
|
|
983
|
+
/* Solid state pill, same vocabulary as .rec-pill but on the SHELL tokens
|
|
984
|
+
(the board's --board-* aliases are scoped to .rec-board/.rec-inspector). */
|
|
985
|
+
.rec-settings-pill {
|
|
986
|
+
flex: none;
|
|
987
|
+
padding: 3px 10px;
|
|
988
|
+
font-size: 11px;
|
|
989
|
+
font-weight: 500;
|
|
990
|
+
line-height: 1;
|
|
991
|
+
border-radius: 999px;
|
|
992
|
+
color: var(--dsw-alias-label-primary-foreground);
|
|
993
|
+
background: var(--dsw-alias-label-tertiary);
|
|
994
|
+
}
|
|
995
|
+
|
|
996
|
+
.rec-settings-pill[data-pill='locked'] { background: var(--dsw-alias-state-success-primary); }
|
|
997
|
+
.rec-settings-pill[data-pill='in-progress'] { background: var(--dsw-alias-state-business-primary); }
|
|
998
|
+
.rec-settings-pill[data-pill='paused'] { background: var(--dsw-alias-state-warn-primary); }
|
|
999
|
+
.rec-settings-pill[data-pill='blocked'] { background: var(--dsw-alias-state-error-primary); }
|
|
1000
|
+
.rec-settings-pill[data-pill='tampered'] { background: var(--dsw-alias-state-error-primary); }
|
|
1001
|
+
.rec-settings-pill[data-pill='advisory'] { background: var(--dsw-alias-state-warn-secondary); }
|
|
1002
|
+
.rec-settings-pill[data-pill='neutral'] { background: var(--dsw-alias-label-tertiary); }
|
|
1003
|
+
|
|
1004
|
+
dl.rec-settings-source,
|
|
1005
|
+
dl.rec-settings-rows {
|
|
1006
|
+
display: grid;
|
|
1007
|
+
grid-template-columns: minmax(160px, 300px) 1fr;
|
|
1008
|
+
gap: 4px 16px;
|
|
1009
|
+
margin: 0;
|
|
1010
|
+
}
|
|
1011
|
+
|
|
1012
|
+
.rec-settings-label {
|
|
1013
|
+
font-size: 12px;
|
|
1014
|
+
color: var(--dsw-alias-label-tertiary);
|
|
1015
|
+
}
|
|
1016
|
+
|
|
1017
|
+
.rec-settings-value {
|
|
1018
|
+
margin: 0;
|
|
1019
|
+
color: var(--dsw-alias-label-primary);
|
|
1020
|
+
overflow-wrap: anywhere;
|
|
1021
|
+
}
|
|
1022
|
+
|
|
1023
|
+
/* An absent value is a STATEMENT, not an empty row — always visibly marked. */
|
|
1024
|
+
.rec-settings-absent {
|
|
1025
|
+
color: var(--dsw-alias-state-warn-label);
|
|
1026
|
+
font-style: italic;
|
|
1027
|
+
}
|
|
1028
|
+
|
|
1029
|
+
.rec-settings-phases {
|
|
1030
|
+
display: flex;
|
|
1031
|
+
flex-direction: column;
|
|
1032
|
+
gap: 4px;
|
|
1033
|
+
}
|
|
1034
|
+
|
|
1035
|
+
.rec-settings-phase {
|
|
1036
|
+
display: flex;
|
|
1037
|
+
align-items: baseline;
|
|
1038
|
+
flex-wrap: wrap;
|
|
1039
|
+
gap: 10px;
|
|
1040
|
+
padding: 4px 10px;
|
|
1041
|
+
border: 1px solid var(--dsw-alias-border-l2);
|
|
1042
|
+
border-radius: 6px;
|
|
1043
|
+
}
|
|
1044
|
+
|
|
1045
|
+
.rec-settings-phase-id {
|
|
1046
|
+
flex: none;
|
|
1047
|
+
min-width: 40px;
|
|
1048
|
+
font-family: ui-monospace, SFMono-Regular, Menlo, monospace;
|
|
1049
|
+
font-size: 11px;
|
|
1050
|
+
color: var(--dsw-alias-label-secondary);
|
|
1051
|
+
}
|
|
1052
|
+
|
|
1053
|
+
.rec-settings-phase-name {
|
|
1054
|
+
flex: 1;
|
|
1055
|
+
min-width: 140px;
|
|
1056
|
+
overflow-wrap: anywhere;
|
|
1057
|
+
}
|
|
1058
|
+
|
|
1059
|
+
.rec-settings-phase-status,
|
|
1060
|
+
.rec-settings-phase-pos {
|
|
1061
|
+
flex: none;
|
|
1062
|
+
font-family: ui-monospace, SFMono-Regular, Menlo, monospace;
|
|
1063
|
+
font-size: 11px;
|
|
1064
|
+
letter-spacing: 0.02em;
|
|
1065
|
+
color: var(--dsw-alias-label-secondary);
|
|
1066
|
+
}
|
|
1067
|
+
|
|
1068
|
+
.rec-settings-phase-lock {
|
|
1069
|
+
flex: none;
|
|
1070
|
+
font-size: 11px;
|
|
1071
|
+
color: var(--dsw-alias-label-tertiary);
|
|
1072
|
+
}
|
|
1073
|
+
|
|
1074
|
+
.rec-settings-items {
|
|
1075
|
+
display: flex;
|
|
1076
|
+
flex-direction: column;
|
|
1077
|
+
gap: 4px;
|
|
1078
|
+
}
|
|
1079
|
+
|
|
1080
|
+
.rec-settings-item {
|
|
1081
|
+
display: flex;
|
|
1082
|
+
align-items: baseline;
|
|
1083
|
+
gap: 10px;
|
|
1084
|
+
font-size: 12px;
|
|
1085
|
+
}
|
|
1086
|
+
|
|
1087
|
+
.rec-settings-item-id {
|
|
1088
|
+
flex: none;
|
|
1089
|
+
min-width: 96px;
|
|
1090
|
+
font-family: ui-monospace, SFMono-Regular, Menlo, monospace;
|
|
1091
|
+
color: var(--dsw-alias-label-secondary);
|
|
1092
|
+
}
|
|
1093
|
+
|
|
1094
|
+
.rec-settings-item-text {
|
|
1095
|
+
flex: 1;
|
|
1096
|
+
min-width: 0;
|
|
1097
|
+
overflow-wrap: anywhere;
|
|
1098
|
+
}
|
|
1099
|
+
|
|
863
1100
|
@media (prefers-reduced-motion: reduce) {
|
|
864
1101
|
.rec-card, .rec-back, .rec-close, .rec-theme-toggle { transition: none; }
|
|
865
1102
|
}
|
package/src/enforcement.ts
CHANGED
|
@@ -211,7 +211,33 @@ export function currentPhaseArtifact(worktreeRoot: string, runId: string): strin
|
|
|
211
211
|
best = name
|
|
212
212
|
}
|
|
213
213
|
}
|
|
214
|
-
|
|
214
|
+
// ⚠ FIX 2 (half a) — PRESENCE IS NO LONGER EVIDENCE OF PROGRESS, AND THIS USED TO ASSUME IT WAS.
|
|
215
|
+
//
|
|
216
|
+
// The loop above returns the HIGHEST-numbered phase artifact present, on the documented assumption that "a run at
|
|
217
|
+
// phase 3 has `00`-`03` on disk". That assumption stopped being true when `recursive_init` began scaffolding ALL
|
|
218
|
+
// TWELVE artifacts, `08-memory-impact.md` included — so from turn 0 the phase-8 baseline governed every guard call.
|
|
219
|
+
// Phase 8 is the documentation phase, whose rule denies writes outside the run tree, so a run sitting at phase 3
|
|
220
|
+
// had its SOURCE EDITS denied as though it were finished and merely writing up. A live verification pass recorded
|
|
221
|
+
// five denials out of five while trying to author phase artifacts.
|
|
222
|
+
//
|
|
223
|
+
// The workflow's own notion of progress is the LOCK, so the phase in force is the LOWEST-numbered artifact that is
|
|
224
|
+
// not locked — the phase actually being worked on. Once everything is locked the run is complete, and the previous
|
|
225
|
+
// answer still stands, which keeps the old behaviour exactly where the old reasoning held. Still read from the
|
|
226
|
+
// filesystem on every call, for the same no-cache reason the directory listing is.
|
|
227
|
+
let inForce = ''
|
|
228
|
+
let inForcePhase = Number.POSITIVE_INFINITY
|
|
229
|
+
for (const name of names) {
|
|
230
|
+
if (!name.endsWith('.md')) continue
|
|
231
|
+
const phase = phaseNumberForArtifact(name)
|
|
232
|
+
if (!phase) continue
|
|
233
|
+
const value = Number(phase)
|
|
234
|
+
if (getLockStatus(join(runDir, name)) === 'LOCKED') continue
|
|
235
|
+
if (value < inForcePhase) {
|
|
236
|
+
inForcePhase = value
|
|
237
|
+
inForce = name
|
|
238
|
+
}
|
|
239
|
+
}
|
|
240
|
+
return inForce !== '' ? inForce : best
|
|
215
241
|
}
|
|
216
242
|
|
|
217
243
|
export function evaluateToolGuard(
|
|
@@ -335,10 +361,56 @@ function resolveTargetPath(target: string, worktreeRoot: string): string | null
|
|
|
335
361
|
return abs
|
|
336
362
|
}
|
|
337
363
|
|
|
364
|
+
/**
|
|
365
|
+
* The ADMISSION test for the tamper path: the resolved absolute path when
|
|
366
|
+
* `targetPath` names a run-tree `*.md` — a tamper CANDIDATE — and `null`
|
|
367
|
+
* otherwise. Pure path arithmetic on every branch (no filesystem work), so a
|
|
368
|
+
* caller may use it as a cheap shape check before paying for `existsSync`.
|
|
369
|
+
*
|
|
370
|
+
* ⚠ THE BLIND SPOT THIS FUNCTION EXISTS TO CLOSE. The admission test used to be a
|
|
371
|
+
* substring test on the target STRING alone, looking for `/.recursive/run/`. A
|
|
372
|
+
* repo-relative path has no separator before `.recursive`, so
|
|
373
|
+
* `.recursive/run/<id>/00-requirements.md` — the spelling a model actually types,
|
|
374
|
+
* and its backslash form — was rejected before ANYTHING was examined, and
|
|
375
|
+
* tampering with a locked artifact through that spelling was invisible
|
|
376
|
+
* (measured: `detectTamper` returned a record for the absolute path and `null`
|
|
377
|
+
* for the relative one, on the same file). The identical defect, in the identical
|
|
378
|
+
* spelling, was fixed one module over in `policy-globs.ts` `lockedWriteRule`; this
|
|
379
|
+
* is that fix's shape, reused rather than reinvented.
|
|
380
|
+
*
|
|
381
|
+
* So the marker is looked for on the path the target RESOLVES to as well as on
|
|
382
|
+
* the string as written. The `||` is load-bearing and the string test is KEPT
|
|
383
|
+
* rather than replaced, because a resolved-only test would SHRINK the admitted
|
|
384
|
+
* set: an absolute target that literally carries the marker but resolves away
|
|
385
|
+
* from it (`…/.recursive/run/../…`) was caught before and must stay caught. The
|
|
386
|
+
* net effect is a strict SUPERSET of the previous behaviour, so no tamper that
|
|
387
|
+
* was visible before can become invisible.
|
|
388
|
+
*
|
|
389
|
+
* EXPORTED because `src/index.ts`'s `fs/observed` listener must apply the SAME
|
|
390
|
+
* admission test before calling `detectTamper`. That listener used to carry a
|
|
391
|
+
* hand-copied mirror of this test, and a mirror is exactly what leaves half the
|
|
392
|
+
* defect behind: widening `detectTamper` alone changes nothing, because the
|
|
393
|
+
* listener rejects the spelling first. One function cannot disagree with itself.
|
|
394
|
+
*/
|
|
395
|
+
export function tamperCandidatePath(targetPath: string, worktreeRoot: string): string | null {
|
|
396
|
+
const normalized = targetPath.replace(/\\/g, '/')
|
|
397
|
+
if (!normalized.endsWith('.md')) return null
|
|
398
|
+
const abs = resolveTargetPath(normalized, worktreeRoot)
|
|
399
|
+
if (!abs) return null
|
|
400
|
+
const resolved = abs.replace(/\\/g, '/')
|
|
401
|
+
if (!normalized.includes('/.recursive/run/') && !resolved.includes('/.recursive/run/')) return null
|
|
402
|
+
return abs
|
|
403
|
+
}
|
|
404
|
+
|
|
338
405
|
/**
|
|
339
406
|
* Layer 8 - fs/observed lock-tamper detection.
|
|
340
407
|
* A locked *.md whose observed version differs from the stored LockHash is
|
|
341
408
|
* a tamper. Returns a tamper reason (or null when clean/not-applicable).
|
|
409
|
+
*
|
|
410
|
+
* The admission test lives in `tamperCandidatePath` (above), shared with the
|
|
411
|
+
* `fs/observed` listener in `src/index.ts` — see the note there for why sharing
|
|
412
|
+
* it is the point and not a tidiness preference. What this function reports is
|
|
413
|
+
* unchanged: the same record shape, carrying the target AS WRITTEN.
|
|
342
414
|
*/
|
|
343
415
|
export function detectTamper(
|
|
344
416
|
targetPath: string,
|
|
@@ -346,8 +418,7 @@ export function detectTamper(
|
|
|
346
418
|
activeRunId: string,
|
|
347
419
|
): { runId: string; path: string; reason: string } | null {
|
|
348
420
|
const normalized = targetPath.replace(/\\/g, '/')
|
|
349
|
-
|
|
350
|
-
const abs = resolveTargetPath(normalized, worktreeRoot)
|
|
421
|
+
const abs = tamperCandidatePath(normalized, worktreeRoot)
|
|
351
422
|
if (!abs || !existsSync(abs)) return null
|
|
352
423
|
const status = getLockStatus(abs)
|
|
353
424
|
if (status === 'STALE_LOCK') {
|
package/src/errors.ts
CHANGED
|
@@ -147,6 +147,18 @@ export const TOOL_ERRORS = {
|
|
|
147
147
|
|
|
148
148
|
/* 5xxx — the runtime refused an operation it understands. */
|
|
149
149
|
|
|
150
|
+
RUN_START_NO_CHANNEL: {
|
|
151
|
+
code: 'RM5502',
|
|
152
|
+
klass: 'runtime',
|
|
153
|
+
problem: 'the run-start gate needs an answer, and this composition mounts no user-questions channel to ask one directly',
|
|
154
|
+
next: 'call recursive_ask with gate: run-start and no answer to surface the question, then retry with answer: ' + '"Start run"',
|
|
155
|
+
},
|
|
156
|
+
RUN_START_UNANSWERED: {
|
|
157
|
+
code: 'RM5503',
|
|
158
|
+
klass: 'runtime',
|
|
159
|
+
problem: 'the user-questions channel mounted in this composition refused the run-start question, so no person was asked',
|
|
160
|
+
next: 'use recursive_ask without an answer to surface the question, and retry it with answer: ' + '"Start run" once the user has approved the run start',
|
|
161
|
+
},
|
|
150
162
|
RUNTIME_REFUSED: {
|
|
151
163
|
code: 'RM5501',
|
|
152
164
|
klass: 'runtime',
|
package/src/goals-projection.ts
CHANGED
|
@@ -15,6 +15,7 @@
|
|
|
15
15
|
* completed goal may be replaced, per the service contract).
|
|
16
16
|
*/
|
|
17
17
|
import type { RunState } from './lifecycle.ts'
|
|
18
|
+
import { RUN_START_NOT_APPROVED } from './run-start.ts'
|
|
18
19
|
|
|
19
20
|
/** Native goal phase (mirrors @deepseek-ai/dsh-goal GoalPhase). */
|
|
20
21
|
export type GoalPhase = 'active' | 'paused' | 'blocked' | 'complete'
|
|
@@ -97,8 +98,31 @@ function mutatePhase(service: GoalServiceLike, agent: AgentHandle, ref: GoalRefL
|
|
|
97
98
|
* Sync a run's durable goal to the requested phase. Safe: never touches a goal
|
|
98
99
|
* whose objective is not this run's marker, and never re-creates over a
|
|
99
100
|
* non-complete foreign goal.
|
|
101
|
+
*
|
|
102
|
+
* ⚠ `approved` IS THE PHASE-0 GATE, and it defaults to the SAFE direction. A goal is not a label:
|
|
103
|
+
* `create` returns an ARMED view and the harness starts driving autonomous goal rounds for the
|
|
104
|
+
* session, so creating one is starting the run. The owner's rule is that phase 0 requires explicit
|
|
105
|
+
* approval, which means the projection must be unable to arm anything on its own — hence a default of
|
|
106
|
+
* `false` and an explicit refusal in EVERY branch that would call `create`, including the two
|
|
107
|
+
* replace-a-completed-goal branches (an unapproved run cannot have reached `complete`, but "cannot
|
|
108
|
+
* happen" is what the single unguarded branch relied on too).
|
|
109
|
+
*
|
|
110
|
+
* ⚠ AND IT IS REACHED ON ORDINARY WORK, so the unapproved path is QUIET AND IDEMPOTENT: no goal is
|
|
111
|
+
* created, nothing is written, no error is thrown, and the run's artifacts are untouched. The caller
|
|
112
|
+
* reads {@link RUN_START_NOT_APPROVED} to tell "this run has not been started yet" apart from a real
|
|
113
|
+
* failure, so a normal phase step never surfaces a warning.
|
|
114
|
+
*
|
|
115
|
+
* `approved` is passed IN rather than read here because this module is pure: it takes the goal service
|
|
116
|
+
* seam and nothing else, and the plugin's own filesystem reads live in the runtime (see
|
|
117
|
+
* `RecursiveRuntime.readRunStartApproval`).
|
|
100
118
|
*/
|
|
101
|
-
export function syncRunGoal(
|
|
119
|
+
export function syncRunGoal(
|
|
120
|
+
service: GoalServiceLike | undefined | null,
|
|
121
|
+
agent: AgentHandle,
|
|
122
|
+
runId: string,
|
|
123
|
+
runState: RunState,
|
|
124
|
+
approved = false,
|
|
125
|
+
): SyncResult {
|
|
102
126
|
if (!service) return { ok: false, reason: 'no goals service' }
|
|
103
127
|
const target = RUN_TO_GOAL_PHASE[runState]
|
|
104
128
|
const current = service.get(agent)
|
|
@@ -110,6 +134,7 @@ export function syncRunGoal(service: GoalServiceLike | undefined | null, agent:
|
|
|
110
134
|
if (phase === target) return { ok: true, phase: target, ref }
|
|
111
135
|
// A completed goal is final: the contract allows it to be REPLACED, not resumed.
|
|
112
136
|
if (phase === 'complete') {
|
|
137
|
+
if (!approved) return { ok: false, reason: RUN_START_NOT_APPROVED }
|
|
113
138
|
const created = service.create(agent, { objective: goalObjective(runId, runState) })
|
|
114
139
|
return { ok: true, phase: target, ref: refOf(created), created: true }
|
|
115
140
|
}
|
|
@@ -121,29 +146,45 @@ export function syncRunGoal(service: GoalServiceLike | undefined | null, agent:
|
|
|
121
146
|
// cleared or resumed instead. Never clobber a foreign goal.
|
|
122
147
|
if (current) {
|
|
123
148
|
if (current.phase === 'complete') {
|
|
149
|
+
if (!approved) return { ok: false, reason: RUN_START_NOT_APPROVED }
|
|
124
150
|
const created = service.create(agent, { objective: goalObjective(runId, runState) })
|
|
125
151
|
return { ok: true, phase: target, ref: refOf(created), created: true }
|
|
126
152
|
}
|
|
127
153
|
return { ok: false, reason: 'a non-matching active goal exists (foreign goal not touched)' }
|
|
128
154
|
}
|
|
129
155
|
|
|
130
|
-
// 3. No current goal
|
|
156
|
+
// 3. No current goal. This is where the defect lived: scaffolding a run armed it. A run with no
|
|
157
|
+
// phase-0 approval stays goal-less — the spec exists, the run does not.
|
|
158
|
+
if (!approved) return { ok: false, reason: RUN_START_NOT_APPROVED }
|
|
131
159
|
const created = service.create(agent, { objective: goalObjective(runId, runState) })
|
|
132
160
|
return { ok: true, phase: target, ref: refOf(created), created: true }
|
|
133
161
|
}
|
|
134
162
|
|
|
135
|
-
/**
|
|
163
|
+
/**
|
|
164
|
+
* Block the current run goal (used on a gate-block). Never touches a foreign goal.
|
|
165
|
+
*
|
|
166
|
+
* ⚠ A RUN THAT WAS NEVER STARTED HAS NO GOAL TO BLOCK, so this reports the unapproved state in the
|
|
167
|
+
* same words as {@link syncRunGoal} rather than "no current goal to block": the caller's question is
|
|
168
|
+
* "why is there no goal", and the answer must not depend on which entry point happened to ask.
|
|
169
|
+
*/
|
|
136
170
|
export function blockRunGoal(service: GoalServiceLike | undefined | null, agent: AgentHandle, runId: string, reason: { code: string; message: string }): SyncResult {
|
|
137
171
|
if (!service) return { ok: false, reason: 'no goals service' }
|
|
138
172
|
const current = service.get(agent)
|
|
139
|
-
if (!current) return { ok: false, reason:
|
|
173
|
+
if (!current) return { ok: false, reason: RUN_START_NOT_APPROVED }
|
|
140
174
|
if (!isRunGoal(current, runId)) return { ok: false, reason: 'current goal is not for this run (foreign goal not touched)' }
|
|
141
175
|
const ref = refOf(current)
|
|
142
176
|
const ok = !!service.block(agent, ref, reason)
|
|
143
177
|
return ok ? { ok: true, phase: 'blocked', ref } : { ok: false, reason: 'goal block failed' }
|
|
144
178
|
}
|
|
145
179
|
|
|
146
|
-
/**
|
|
147
|
-
|
|
148
|
-
|
|
180
|
+
/**
|
|
181
|
+
* Bridge a run's blocked goal back to active (used on a reopen).
|
|
182
|
+
*
|
|
183
|
+
* ⚠ REOPEN IS NOT A BACK DOOR TO STARTING A RUN. It routes through {@link syncRunGoal}, so a reopen of
|
|
184
|
+
* an unapproved run cannot create the goal that init deliberately withheld. An APPROVED run is
|
|
185
|
+
* unaffected: its approval outlives the reopen, because the approval is a durable line in the run's
|
|
186
|
+
* own Phase 0 artifact rather than a value held in memory (verified in `tests/run-start-approval.spec.ts`).
|
|
187
|
+
*/
|
|
188
|
+
export function resumeRunGoal(service: GoalServiceLike | undefined | null, agent: AgentHandle, runId: string, approved = false): SyncResult {
|
|
189
|
+
return syncRunGoal(service, agent, runId, 'active', approved)
|
|
149
190
|
}
|
package/src/index.ts
CHANGED
|
@@ -4,6 +4,7 @@ import type { ContextFormed } from '@deepseek-ai/dsh-llm'
|
|
|
4
4
|
import { existsSync } from 'node:fs'
|
|
5
5
|
import { join } from 'node:path'
|
|
6
6
|
import { RecursiveRuntime } from './runtime.ts'
|
|
7
|
+
import type { UserQuestionsLike } from './runtime.ts'
|
|
7
8
|
import type { JobsRegistryLike } from './jobs-runner.ts'
|
|
8
9
|
import { planGateForExit } from './plan-gate.ts'
|
|
9
10
|
import { registerPhaseSkills, type SkillRegistryLike } from './skills-phase.ts'
|
|
@@ -24,7 +25,7 @@ import { createRecursiveAskTool } from './recursive_ask.tool.ts'
|
|
|
24
25
|
import { createRecursivePreviewTool } from './recursive_preview.tool.ts'
|
|
25
26
|
import type { SubagentsRuntimeLike } from './delegation.ts'
|
|
26
27
|
import { registerRecursiveCommand } from './commands.ts'
|
|
27
|
-
import { evaluateToolGuard, coerceAskToDecision, type ToolGuardDecision } from './enforcement.ts'
|
|
28
|
+
import { evaluateToolGuard, coerceAskToDecision, tamperCandidatePath, type ToolGuardDecision } from './enforcement.ts'
|
|
28
29
|
import { appendGuardDecision, appendObservedTamper, type GuardDecisionRecord } from './guard-log.ts'
|
|
29
30
|
import type { GoalServiceLike } from './goals-projection.ts'
|
|
30
31
|
import type { TeamRuntimeLike } from './teams-loop.ts'
|
|
@@ -237,6 +238,14 @@ export function apply(ctx: Context, config?: RecursiveModeConfig) {
|
|
|
237
238
|
ctx.inject(['llm'], (llmCtx: Context) => {
|
|
238
239
|
recursive.attachLlmInventory((llmCtx.get('llm') as LlmInventoryLike | undefined) ?? null)
|
|
239
240
|
})
|
|
241
|
+
// ⚠ PHASE 0 — THE HUMAN-QUESTION CHANNEL, resolved the same late-attaching way for the same reason:
|
|
242
|
+
// a composition may mount `userQuestions` after this plugin applies, and the run-start gate asks the
|
|
243
|
+
// person DIRECTLY through it before it arms a goal — which is what keeps a relayed `Start run` from
|
|
244
|
+
// being an approval while a person can actually be asked (see run-start.ts).
|
|
245
|
+
recursive.attachUserQuestions((ctx.get('userQuestions') as UserQuestionsLike | undefined) ?? null)
|
|
246
|
+
ctx.inject(['userQuestions'], (questionsCtx: Context) => {
|
|
247
|
+
recursive.attachUserQuestions((questionsCtx.get('userQuestions') as UserQuestionsLike | undefined) ?? null)
|
|
248
|
+
})
|
|
240
249
|
|
|
241
250
|
// T7 — THE SETTINGS NAMESPACE, APPLIED ON EVERY APPLY. The settings service edits the
|
|
242
251
|
// Loader entry's config and the Loader RE-APPLIES this plugin, so a toggle in the UI
|
|
@@ -307,9 +316,36 @@ export function apply(ctx: Context, config?: RecursiveModeConfig) {
|
|
|
307
316
|
ctx.tools.register(createRecursiveAskTool(recursive)),
|
|
308
317
|
// T26: the read-only view of what the enforcement contract will do, before it fires.
|
|
309
318
|
ctx.tools.register(createRecursivePreviewTool(recursive)),
|
|
310
|
-
...(agentTeams ? [ctx.tools.register(createRecursiveAuditTeamTool(agentTeams))] : []),
|
|
311
319
|
]
|
|
312
320
|
|
|
321
|
+
// ⚠ FIX 1 — THE THIRD SEAM NEEDED THE SAME LATE ATTACH AS THE OTHER TWO, AND DID NOT HAVE IT.
|
|
322
|
+
//
|
|
323
|
+
// The line that used to sit in the array above was `...(agentTeams ? [register(...)] : [])` — a ONE-SHOT
|
|
324
|
+
// `ctx.get('agentTeams')` taken at apply time. A live verification pass found the consequence: `team_task_create`
|
|
325
|
+
// worked in the same session whose tool catalog lacked `recursive_audit_team`, because the service was mounted
|
|
326
|
+
// AFTER this plugin applied. The plugin shipped 13 tool files and offered 12.
|
|
327
|
+
//
|
|
328
|
+
// `subagents` and `llm` already solve this with `ctx.inject` (above); this is that pattern, with one addition the
|
|
329
|
+
// others do not need: the tool may only be registered ONCE, because the one-shot path can already have taken it.
|
|
330
|
+
let auditTeamRegistered = agentTeams !== undefined && agentTeams !== null
|
|
331
|
+
if (auditTeamRegistered) disposers.push(ctx.tools.register(createRecursiveAuditTeamTool(agentTeams ?? null)))
|
|
332
|
+
ctx.inject(['agentTeams'], (teamCtx: Context) => {
|
|
333
|
+
if (auditTeamRegistered) return
|
|
334
|
+
const late = teamCtx.get('agentTeams') as TeamRuntimeLike | undefined
|
|
335
|
+
if (late === undefined || late === null) return
|
|
336
|
+
auditTeamRegistered = true
|
|
337
|
+
// ⚠ CORRECTED COMMENT. This registers through the OUTER plugin context (`ctx`), NOT through the
|
|
338
|
+
// injecting `teamCtx` — `teamCtx` is used on the line above only to READ the late service, and the
|
|
339
|
+
// sibling `subagents`/`llm` injects use their callback context the same way. The comment that used to
|
|
340
|
+
// sit here claimed the injecting scope owned the registration, which is not what this call does.
|
|
341
|
+
//
|
|
342
|
+
// WHAT IS NOT CLAIMED: that the fiber withdraws this registration. Nobody has observed that — no test
|
|
343
|
+
// covers the late `recursive_audit_team` being withdrawn — and the disposer returned here is not
|
|
344
|
+
// retained, unlike the eager registration above, which pushes its own onto `disposers`. Until a test
|
|
345
|
+
// observes the withdrawal, this comment promises nothing about it.
|
|
346
|
+
ctx.tools.register(createRecursiveAuditTeamTool(late))
|
|
347
|
+
})
|
|
348
|
+
|
|
313
349
|
// /recursive command (R4): preset-scoped registration, workspace-scoped dispatch.
|
|
314
350
|
const commands = ctx.get('commands') as { register: (def: unknown) => () => void } | undefined
|
|
315
351
|
if (commands) {
|
|
@@ -501,16 +537,26 @@ export function apply(ctx: Context, config?: RecursiveModeConfig) {
|
|
|
501
537
|
if (!displayPath) return
|
|
502
538
|
// Cheap shape test BEFORE any filesystem work: fs/observed fires on reads
|
|
503
539
|
// too, so enumerating runs for every observation would be a readdir per
|
|
504
|
-
// file touch.
|
|
505
|
-
//
|
|
506
|
-
//
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
// The actor is the tool execution. This event cannot await, so the root is
|
|
510
|
-
// the actor's session cwd (the same B4 sync shortcut fsPolicyIntent takes:
|
|
511
|
-
// the session cwd is authoritative, the registry path is async-only).
|
|
540
|
+
// file touch. The actor is the tool execution. This event cannot await, so
|
|
541
|
+
// the root is the actor's session cwd (the same B4 sync shortcut
|
|
542
|
+
// fsPolicyIntent takes: the session cwd is authoritative, the registry
|
|
543
|
+
// path is async-only). Resolving the cwd first is free — plain property
|
|
544
|
+
// reads — and the admission test needs it.
|
|
512
545
|
const cwd = (actor as { agent?: { session?: { header?: { cwd?: string } } } } | null)?.agent?.session?.header?.cwd ?? ''
|
|
513
546
|
if (!cwd) return
|
|
547
|
+
// ⚠ AND THIS IS `detectTamper`'s OWN ADMISSION TEST, CALLED RATHER THAN COPIED.
|
|
548
|
+
//
|
|
549
|
+
// It used to be an inline hand-copy — `endsWith('.md') && includes('/.recursive/run/')`
|
|
550
|
+
// — and a hand-copy is what made the tamper guard blind to one spelling of one
|
|
551
|
+
// path: the substring test needs a separator BEFORE `.recursive`, which a
|
|
552
|
+
// repo-relative target (`displayPath` as a model would type it) does not have, so
|
|
553
|
+
// the listener rejected the candidate here and `detectTamper` was never reached.
|
|
554
|
+
// Widening only `detectTamper` would have changed nothing observable. The test now
|
|
555
|
+
// lives in one place (`tamperCandidatePath`), so the two cannot disagree; it stays
|
|
556
|
+
// pure path arithmetic, so the "no filesystem work before admission" property the
|
|
557
|
+
// shape check exists for is preserved.
|
|
558
|
+
const normalized = displayPath.replace(/\\/g, '/')
|
|
559
|
+
if (!tamperCandidatePath(normalized, cwd)) return
|
|
514
560
|
const runId = resolveRunDir(cwd)?.runId ?? ''
|
|
515
561
|
const tamper = recursive.detectTamper(normalized, cwd, runId)
|
|
516
562
|
if (!tamper) return
|
package/src/memory-feedback.ts
CHANGED
|
@@ -19,7 +19,32 @@ import { existsSync, mkdirSync, readFileSync, writeFileSync } from 'node:fs'
|
|
|
19
19
|
import { dirname, join } from 'node:path'
|
|
20
20
|
|
|
21
21
|
/** Where the machine-owned counters live — inside the memory plane, but never a shard a person writes. */
|
|
22
|
-
|
|
22
|
+
// The path is `.recursive/memory/.feedback.json`, joined onto the REPO ROOT, which is what the
|
|
23
|
+
// doc comment above has always said and what `filterRuntimeChangedFiles` expects: a counter the
|
|
24
|
+
// memory plane writes must live inside the `.recursive` control plane, not in the product tree.
|
|
25
|
+
// Measured before this fix: a completed fixture run ended with four residual FAILs naming
|
|
26
|
+
// `memory/.feedback.json` against phases 03 and 03.5, because the file landed outside
|
|
27
|
+
// `.recursive/run/<runId>/`, survived the runtime-diff filter, and entered the run's diff only
|
|
28
|
+
// after those phases were authored - retro-invalidating them. The counter is machine-owned and
|
|
29
|
+
// disposable per the same doc comment, so a counter left at the OLD path by an earlier version is
|
|
30
|
+
// READ once (see LEGACY_FEEDBACK_FILE) rather than silently dropped.
|
|
31
|
+
export const FEEDBACK_FILE = '.recursive/memory/.feedback.json'
|
|
32
|
+
|
|
33
|
+
/**
|
|
34
|
+
* WHERE THE COUNTERS LIVED BEFORE the constant above carried its `.recursive/` prefix.
|
|
35
|
+
*
|
|
36
|
+
* ⚠ READ, BUT DELIBERATELY NEITHER MOVED NOR DELETED. Read, because evidence a previous run recorded is
|
|
37
|
+
* not the plugin's to discard, and without the fallback the first settle after the move would start the
|
|
38
|
+
* new file from an empty book — a counter lost silently, which is the one outcome ruled out. NOT moved,
|
|
39
|
+
* because a delete is the single action that could re-create the very defect the prefix fixes: a legacy
|
|
40
|
+
* file that is TRACKED and COMMITTED is absent from the run's diff while it is clean, so removing it
|
|
41
|
+
* mid-run puts `D memory/.feedback.json` into the diff of every diff-audited phase authored before the
|
|
42
|
+
* delete, which is the same retro-invalidation. Left alone, an untracked legacy file is in the diff from
|
|
43
|
+
* the run's first phase (and is therefore accounted for), while a committed one stays invisible.
|
|
44
|
+
* `readFeedback` prefers {@link FEEDBACK_FILE}, so the legacy counters are folded forward by the next
|
|
45
|
+
* settle and this file then only sits there; it is machine-owned and disposable, so delete it by hand.
|
|
46
|
+
*/
|
|
47
|
+
export const LEGACY_FEEDBACK_FILE = 'memory/.feedback.json'
|
|
23
48
|
|
|
24
49
|
/** Where a run records what it was shown. */
|
|
25
50
|
export const INJECTIONS_FILE = 'memory-injections.json'
|
|
@@ -43,10 +68,24 @@ export interface FeedbackCounter {
|
|
|
43
68
|
|
|
44
69
|
export type FeedbackBook = Record<string, FeedbackCounter>
|
|
45
70
|
|
|
46
|
-
/**
|
|
71
|
+
/**
|
|
72
|
+
* Read the counters. A missing or unreadable file is an empty book, never an error.
|
|
73
|
+
*
|
|
74
|
+
* ⚠ THE LEGACY PATH IS A FALLBACK AND ONLY A FALLBACK: it is consulted when — and only when — there is
|
|
75
|
+
* no usable file at {@link FEEDBACK_FILE} yet, which is exactly the first run after the path moved. The
|
|
76
|
+
* two are never merged, because they are two SNAPSHOTS of one counter and adding them would count a run
|
|
77
|
+
* twice. A file that exists at the current path but does not parse stays an empty book, which is what
|
|
78
|
+
* this function has always promised: a corrupt sidecar is not an invitation to read a different file.
|
|
79
|
+
*/
|
|
47
80
|
export function readFeedback(root: string, readFile: (path: string) => string | null = defaultRead): FeedbackBook {
|
|
48
|
-
const
|
|
49
|
-
if (
|
|
81
|
+
const current = readFile(join(root, FEEDBACK_FILE))
|
|
82
|
+
if (current !== null && current.trim() !== '') return parseBook(current)
|
|
83
|
+
const legacy = readFile(join(root, LEGACY_FEEDBACK_FILE))
|
|
84
|
+
return legacy === null ? {} : parseBook(legacy)
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
/** One counters file as a book: anything unreadable or unshaped is `{}`, never an error. */
|
|
88
|
+
function parseBook(text: string): FeedbackBook {
|
|
50
89
|
try {
|
|
51
90
|
const parsed = JSON.parse(text) as unknown
|
|
52
91
|
if (typeof parsed !== 'object' || parsed === null) return {}
|
|
@@ -140,7 +179,13 @@ export function settleInjections(
|
|
|
140
179
|
book[record.source] = counter
|
|
141
180
|
}
|
|
142
181
|
|
|
143
|
-
|
|
182
|
+
const feedbackPath = join(root, FEEDBACK_FILE)
|
|
183
|
+
// The counter now lives under the `.recursive` control plane, whose memory directory a fresh
|
|
184
|
+
// checkout or a scratch fixture may not have yet. Creating it here rather than assuming it
|
|
185
|
+
// exists is what a plugin owning its own control plane should do; before this, the write
|
|
186
|
+
// threw ENOENT when nothing else had already made the directory.
|
|
187
|
+
mkdirSync(dirname(feedbackPath), { recursive: true })
|
|
188
|
+
write(feedbackPath, JSON.stringify(sortBook(book), null, 2) + '\n')
|
|
144
189
|
return book
|
|
145
190
|
}
|
|
146
191
|
|
package/src/phase-rules.ts
CHANGED
|
@@ -490,7 +490,16 @@ export function resolveFrom(worktreeRoot: string, target: string): string | null
|
|
|
490
490
|
const normalized = target.replace(/\\/g, '/').trim()
|
|
491
491
|
if (!normalized) return null
|
|
492
492
|
if (/^[A-Za-z]:\//.test(normalized) || normalized.startsWith('/')) return resolve(normalized)
|
|
493
|
-
|
|
493
|
+
// ⚠ FIX 2 — THIS USED TO STRIP A LEADING DOT, WHICH IS NOT THE SAME AS STRIPPING `./`.
|
|
494
|
+
//
|
|
495
|
+
// The regex was `/^\.?\/?/`: an optional dot followed by an optional slash. Its intent was plainly to normalise a
|
|
496
|
+
// `./relative` path, but on a DOTFILE path it removed the dot and kept the name — so `.recursive/run/<id>/01.md`
|
|
497
|
+
// resolved to `<root>/recursive/run/<id>/01.md`, OUTSIDE the run tree it names. Under strict enforcement the phase
|
|
498
|
+
// guard denies writes outside the run tree, and a live verification pass recorded five denials out of five while
|
|
499
|
+
// trying to author phase artifacts — the model could not write the very files the workflow is about.
|
|
500
|
+
//
|
|
501
|
+
// Only `./` is a relative-path prefix. A bare leading dot is part of the name.
|
|
502
|
+
return resolve(join(worktreeRoot, normalized.replace(/^\.\//, '')))
|
|
494
503
|
}
|
|
495
504
|
|
|
496
505
|
/** The tool-target path of a call (same key order enforcement.ts uses). */
|