@try-works/dsh-recursive-mode 0.4.5 → 0.4.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -27,6 +27,7 @@ import {
27
27
  RUN_START_ARTIFACT,
28
28
  RUN_START_GATE,
29
29
  RUN_START_GATE_ID,
30
+ runStartSpecGuard,
30
31
  } from './run-start.ts'
31
32
 
32
33
 
@@ -294,24 +295,31 @@ export function pendingGateFor(artifactFile: string, artifactText: string | null
294
295
  export function createRecursiveAskTool(recursive: RecursiveRuntime) {
295
296
  return defineTool({
296
297
  name: 'recursive_ask',
297
- description: 'Ask a human gate as a structured decision (tdd-mode, qa-signoff, gate-block), or ASK TO START A RUN (run-start: nothing runs, and no goal exists, until this gate is approved). Call without `answer` to ask; call with it to write the answer into the artifact as a durable marker. One ask per step.',
298
+ description: 'Ask a human gate as a structured decision (tdd-mode, qa-signoff, gate-block), or ASK TO START A RUN (run-start: nothing runs, and no goal exists, until this gate is approved). Call without `answer` to ask; call with it to write the answer into the artifact as a durable marker. For run-start the mounted human channel is asked first and its own selection wins; when that channel cannot deliver the question, the refusal names the cause and `relay=true` with an explicit `answer` records the person\'s relayed approval. The run-start gate is REFUSED while the Phase 0 requirements document is still the unfilled template — the refusal quotes the placeholder lines, and there is nothing to approve until they are written. One ask per step.',
298
299
  parameters: {
299
300
  gate: { type: 'string', description: 'tdd-mode | qa-signoff | gate-block | run-start. Required. `run-start` is phase 0: approving it records the approval and arms the run goal, which is what makes the harness drive rounds.' },
300
301
  runId: { type: 'string', description: 'Run id. Required; must resolve inside the current workspace.' },
301
302
  artifact: { type: 'string', description: 'Artifact file the answer belongs in. Optional; defaults per gate (gate-block has none, so it is required for that gate; run-start is always recorded in 00-requirements.md).' },
302
303
  answer: { type: 'string', description: 'One of the gate\'s option labels. Omit to ASK. For run-start, the labels are: ' + RUN_START_GATE.options.map((option) => option.label).join(' | ') + '.' },
304
+ relay: { type: 'boolean', description: 'run-start only, and only after the person has approved in this conversation. Set relay=true when the mounted user-questions channel cannot deliver the run-start question: `answer` then stands in for the channel\'s selection and the result reports source: "relayed" instead of a direct selection. It is refused when the channel reports the question was cancelled, aborted, or timed out, and it is not needed when the person answers the card.' },
303
305
  },
304
306
  output: {
305
307
  schema: { type: 'json' },
306
308
  render: (_args, value) => [{ type: 'text', text: JSON.stringify(value, null, 2) }],
307
309
  },
308
- async execute(args: { gate?: string; runId?: string; artifact?: string; answer?: string }, exec) {
310
+ async execute(args: { gate?: string; runId?: string; artifact?: string; answer?: string; relay?: boolean }, exec) {
309
311
  const gateId = (args.gate ?? '').trim() as AskAnyGateId
310
312
  const runId = args.runId?.trim() ?? ''
311
313
  if (runId === '') return { error: toolError('MISSING_RUN_ID') } as const
312
314
  if (!askGateIds().includes(gateId)) {
313
315
  return { error: toolError('BAD_ASK_GATE', 'gate must be one of ' + askGateIds().join(' | ')) } as const
314
316
  }
317
+ // ⚠ `relay` IS RUN-START ONLY, AND SAYING SO IS THE POINT. The other three gates never consult the
318
+ // channel, so accepting the flag there would report a fallback that did not happen — the same class
319
+ // of false claim this tool was fixed for.
320
+ if (args.relay === true && !isRunStartGate(gateId)) {
321
+ return { error: toolError('RELAY_ONLY_FOR_RUN_START', 'gate is ' + gateId) } as const
322
+ }
315
323
  const root = await recursive.resolveWorkspaceRoot(exec.agent)
316
324
  if (!root) return { error: toolError('NO_WORKSPACE') } as const
317
325
 
@@ -319,6 +327,32 @@ export function createRecursiveAskTool(recursive: RecursiveRuntime) {
319
327
  // phase-0 record, and letting a caller aim the approval somewhere else is how an approval ends up
320
328
  // in a file no reader looks at. Every other gate keeps its per-gate default and its override.
321
329
  const artifact = (isRunStartGate(gateId) ? RUN_START_ARTIFACT : args.artifact ?? GATE_DEFAULT_ARTIFACT[gateId as AskGateId]).trim()
330
+
331
+ // ⚠ THE ORDERING GUARD, AND IT COMES BEFORE THE QUESTION IS PUT TO ANYBODY. A scaffolded run's Phase 0
332
+ // document is a template: placeholder requirements, unchecked lists, FAIL gates. Raising "start this run
333
+ // or hold?" over that asks a person to approve a document that is not a spec — and until this change
334
+ // nothing even SHOWED it to them. So the gate refuses while the artifact is still the template, naming
335
+ // the artifact and quoting the placeholder lines, and records nothing.
336
+ //
337
+ // It runs before BOTH branches below on purpose: asking the channel first and refusing afterwards would
338
+ // put the card in front of the person anyway, which is the defect. A refusal here does not weaken any
339
+ // part of the gate's contract — the person's own answer still wins everywhere below, a spec still
340
+ // creates no goal, and the relay rules are untouched.
341
+ if (isRunStartGate(gateId)) {
342
+ const guard = runStartSpecGuard(root, runId)
343
+ if (!guard.ok) {
344
+ return {
345
+ error: guard.reason,
346
+ gate: RUN_START_GATE_ID,
347
+ runId,
348
+ artifact: RUN_START_ARTIFACT,
349
+ // The question travels with the refusal so the caller can put the DECISION in the transcript while
350
+ // it cannot yet raise the card — the same shape the channel refusals use.
351
+ question: buildAskQuestionFor(RUN_START_GATE_ID),
352
+ } as unknown as JsonValue
353
+ }
354
+ }
355
+
322
356
  let question: AskQuestion
323
357
  try {
324
358
  question = buildAskQuestionFor(gateId)
@@ -334,7 +368,9 @@ export function createRecursiveAskTool(recursive: RecursiveRuntime) {
334
368
  // question is answered by relaying a label, so "no answer means ask". Starting a run is the decision
335
369
  // that creates the armed goal, so when this composition mounts the blocking human channel the
336
370
  // question is PUT TO THE PERSON whether or not an answer argument arrived — a caller cannot skip the
337
- // person by supplying one. Only a composition with no channel falls back to the relayed answer.
371
+ // person by supplying one. Only a composition with no channel falls back to the relayed answer, and
372
+ // a channel that FAILED is a third case: it is reported with its cause, and the relayed answer is
373
+ // taken only when the caller asks for the relay in so many words (see `recordRunStartAnswer`).
338
374
  const channelMounted = recursive.userQuestionsChannel !== null
339
375
  if (args.answer === undefined && !(isRunStartGate(gateId) && channelMounted)) {
340
376
  return { gate: gateId, marker: isRunStartGate(gateId) ? RUN_START_GATE.marker : ASK_GATES[gateId as AskGateId].marker, artifact, question } as unknown as JsonValue
@@ -355,7 +391,7 @@ export function createRecursiveAskTool(recursive: RecursiveRuntime) {
355
391
  return { error: toolError('MISSING_ASK_ARTIFACT', 'this gate needs an explicit artifact to record into') } as const
356
392
  }
357
393
  if (isRunStartGate(gateId)) {
358
- return recordRunStartAnswer(recursive, root, runId, answer, exec) as unknown as JsonValue
394
+ return recordRunStartAnswer(recursive, root, runId, answer, exec, args.relay === true) as unknown as JsonValue
359
395
  }
360
396
  const marker = answerMarker(gateId as AskGateId, answer as string)
361
397
  const written = recursive.recordAskAnswer(root, runId, artifact, marker)
@@ -367,21 +403,35 @@ export function createRecursiveAskTool(recursive: RecursiveRuntime) {
367
403
  /**
368
404
  * PHASE 0 — record the answer to the run-start gate, and start the run only if it says so.
369
405
  *
370
- * ⚠ THIS GATE NEVER ACCEPTS A RELAYED ANSWER WHILE A HUMAN CHANNEL IS MOUNTED. That is the rule that makes
371
- * an approval a human act rather than an inference: when `ctx.userQuestions` is present, the question is
372
- * PUT TO THE PERSON and nothing else can settle it — not the caller's own `answer` argument, and not a
373
- * fabrication, because `ask()` resolves only with a real selection. A person's decline is likewise final
374
- * for that call and cannot be overridden by a model that asked for `Start run` in the same breath.
406
+ * ⚠ THREE OUTCOMES, AND TELLING THEM APART IS THE FIX. The first version of this function collapsed all of
407
+ * them into one `null`: "the channel threw", "the channel resolved with something unrecognisable", and
408
+ * "nobody answered" produced the same refusal, whose text asserted a cause ("so no person was asked") the
409
+ * plugin had already thrown away. A live session paid for that: the call failed after 22.9 s, the operator
410
+ * could not be told why, and the refusal's own advice prescribed the call that had just failed. So:
411
+ *
412
+ * 1. A PERSON WAS REACHED (`unusable`): the channel resolved, so somebody answered, and their answer is
413
+ * not a label this gate offered — a skip, a custom value, several labels at once. That is a DECISION
414
+ * the gate cannot record, and it is final: neither the caller's `answer` nor `relay=true` may replace
415
+ * it. (RM5504)
416
+ * 2. THE CHANNEL FAILED (`unavailable`): no decision came back at all, and the refusal NAMES THE CAUSE
417
+ * from the error the channel threw. Here the run can still be started, because a composition whose
418
+ * channel cannot deliver the question would otherwise be unable to start any run — but only by the
419
+ * caller asking for the relay in so many words (`relay=true`), which the result reports as
420
+ * `source: "relayed"` rather than as a person's own selection. (RM5503)
421
+ * 3. NO CHANNEL IS MOUNTED: the relayed answer is the only possible source, exactly as before. (RM5502
422
+ * when there is no answer either)
375
423
  *
376
- * ⚠ AND WHEN NO CHANNEL IS MOUNTED, THE RELAYED ANSWER IS THE ONLY POSSIBLE SOURCE, so it is used — that
377
- * is the same contract the other three gates have always had, and refusing it would leave a composition
378
- * without the channel unable to start any run at all. The question is surfaced first by the ASK branch
379
- * (the card data the host renders), and the model's `answer` is that person's selection coming back.
424
+ * ⚠ AND A CANCELLED OR CLOSED QUESTION IS NEVER RELAYABLE. `ASK_CANCELLED`, `ASK_ABORTED` and
425
+ * `ASK_TIMED_OUT` are the codes that mean the question was settled from outside this gate — the card was
426
+ * dismissed, the turn was cancelled, or a foreground window ended. The relay is refused for those, so a
427
+ * question the operator stopped cannot be turned into an approval by asking again in the same breath.
428
+ * Every other failure is a composition or capability failure — the question reached nobody — which is the
429
+ * class the relay exists for.
380
430
  *
381
- * ⚠ WHAT THE GATE THEREFORE DOES *NOT* CLAIM, stated rather than implied: in a composition with no
382
- * `userQuestions` channel, a plugin cannot verify that a person was really asked, so a model could in
383
- * principle relay a label nobody gave. That is a property of the relay, not of this gate — and it is the
384
- * reason the channel is consulted in preference whenever it exists. See the header of `run-start.ts`.
431
+ * ⚠ WHAT THE GATE STILL DOES *NOT* CLAIM: a relayed approval is a relayed approval. The plugin cannot
432
+ * verify that a person gave the label, and it does not pretend otherwise — the result's `source` and
433
+ * `channel` fields say where the decision came from, and a direct selection is preferred whenever the
434
+ * channel can produce one.
385
435
  */
386
436
  export async function recordRunStartAnswer(
387
437
  recursive: RecursiveRuntime,
@@ -389,22 +439,72 @@ export async function recordRunStartAnswer(
389
439
  runId: string,
390
440
  answer: string | undefined,
391
441
  exec: { agent?: unknown; signal?: unknown; callId?: unknown },
442
+ relay = false,
392
443
  ): Promise<Record<string, unknown>> {
393
- // 1. ASK THE PERSON DIRECTLY when this composition mounts the channel. An abort or a dismissal is not
394
- // consent, so it settles nothing — and, because the channel was available, it does not hand the
395
- // decision back to the caller either (that is the refusal below).
444
+ // The question travels in every refusal: a composition whose channel cannot render a card can still put
445
+ // the exact decision to the person in the transcript, which is what makes the failure recoverable.
446
+ const question = buildAskQuestionFor(RUN_START_GATE_ID)
396
447
  const channel = recursive.userQuestionsChannel
397
- const fromChannel = channel ? await askRunStartDirectly(channel, exec) : null
398
- if (channel !== null && fromChannel === null) {
399
- // The channel exists, so a person COULD have been asked and was not: no answerer, no live root agent,
400
- // a dismissal, an abort, or a selection that is not one of this gate's labels. An approval nobody
401
- // gave is not recorded, and neither is the caller's argument.
402
- return { error: toolError('RUN_START_UNANSWERED') }
448
+
449
+ // 1. ASK THE PERSON DIRECTLY when this composition mounts the channel.
450
+ const channelOutcome: RunStartChannelOutcome | null = channel ? await askRunStartDirectly(channel, exec) : null
451
+
452
+ if (channelOutcome !== null && channelOutcome.kind === 'unusable') {
453
+ // A person WAS asked. Their answer is not an approval this gate can record, and nothing the caller
454
+ // supplies can stand in for it.
455
+ return {
456
+ error: toolError('RUN_START_ANSWER_UNUSABLE', channelOutcome.detail),
457
+ gate: RUN_START_GATE_ID,
458
+ runId,
459
+ artifact: RUN_START_ARTIFACT,
460
+ question,
461
+ }
462
+ }
463
+
464
+ if (channelOutcome !== null && channelOutcome.kind === 'unavailable') {
465
+ const blocked = !relay
466
+ ? 'the caller did not ask for the relay'
467
+ : 'the channel reports the question was cancelled, aborted, or timed out, so it is not relayable'
468
+ if (!relay || !channelOutcome.relayable) {
469
+ return {
470
+ error: toolError('RUN_START_UNANSWERED', channelOutcome.detail + ' (' + blocked + ')'),
471
+ gate: RUN_START_GATE_ID,
472
+ runId,
473
+ artifact: RUN_START_ARTIFACT,
474
+ question,
475
+ // The diagnosis, as data: a model can quote the cause, and a test can assert on it rather than on
476
+ // the prose of the sentence above.
477
+ channel: { outcome: 'unavailable', cause: channelOutcome.cause, relayable: channelOutcome.relayable },
478
+ }
479
+ }
480
+ if (answer === undefined) {
481
+ // ⚠ THE RELAY NEEDS SOMETHING TO RELAY, AND THIS IS NOT RM5502. A channel IS mounted here, so the
482
+ // "this composition mounts no user-questions channel" sentence would be false — reachable by asking
483
+ // for the relay without supplying the answer it relays.
484
+ return {
485
+ error: toolError('RUN_START_UNANSWERED', channelOutcome.detail + ' (the relay was authorised but no answer was supplied, so there is no decision to record)'),
486
+ gate: RUN_START_GATE_ID,
487
+ runId,
488
+ artifact: RUN_START_ARTIFACT,
489
+ question,
490
+ channel: { outcome: 'unavailable', cause: channelOutcome.cause, relayable: channelOutcome.relayable },
491
+ }
492
+ }
403
493
  }
494
+
495
+ const fromChannel = channelOutcome !== null && channelOutcome.kind === 'answered' ? channelOutcome.answer : null
404
496
  const final = fromChannel ?? answer
405
497
  if (final === undefined) {
406
- // No channel and no answer: there is nothing a person said, so nothing is recorded.
407
- return { error: toolError('RUN_START_NO_CHANNEL') }
498
+ // No channel is mounted and no answer was supplied — the only state left here, because an `answered`
499
+ // outcome sets `final`, an `unusable` one returned above, and an `unavailable` one either returned
500
+ // above or carried an answer through the relay. RM5502 says exactly this, and nothing is recorded.
501
+ return {
502
+ error: toolError('RUN_START_NO_CHANNEL'),
503
+ gate: RUN_START_GATE_ID,
504
+ runId,
505
+ artifact: RUN_START_ARTIFACT,
506
+ question,
507
+ }
408
508
  }
409
509
 
410
510
  // 2. Validate the decision that is about to become durable. A channel selection has already been
@@ -425,8 +525,12 @@ export async function recordRunStartAnswer(
425
525
  answer: decided,
426
526
  artifact: RUN_START_ARTIFACT,
427
527
  // Where the decision came from matters to a reader of the transcript: a direct answer is the person's
428
- // own selection; a relayed one came back through the model.
528
+ // own selection; a relayed one came back through the model. When the relay answered a FAILED channel,
529
+ // the failure travels with the result, so a relayed approval never reads as a direct selection.
429
530
  source: fromChannel === null ? 'relayed' : 'user-questions',
531
+ ...channelOutcome !== null && channelOutcome.kind === 'unavailable'
532
+ ? { channel: { outcome: 'unavailable', cause: channelOutcome.cause, relayable: channelOutcome.relayable, relayed: true } }
533
+ : {},
430
534
  path: outcome.path,
431
535
  replaced: outcome.replaced,
432
536
  // `armed` is the answer to "did the harness get a goal to drive?": the run is started only when the
@@ -438,16 +542,90 @@ export async function recordRunStartAnswer(
438
542
  }
439
543
 
440
544
  /**
441
- * Ask the run-start question through the blocking channel and return the selection, or null when the
442
- * person was not reachable (no answerer, no live root agent, a dismissal, an abort).
545
+ * PHASE 0 — WHAT THE BLOCKING CHANNEL ACTUALLY DID.
546
+ *
547
+ * ⚠ THIS TYPE EXISTS BECAUSE ITS ABSENCE WAS THE DEFECT. The first version returned a bare `null` from a
548
+ * `catch {}` for every failure and for every unrecognisable selection, so "the channel threw NO_PROVIDER",
549
+ * "the caller is not the live root agent", "the person skipped the question" and "the person typed a
550
+ * custom value" were ONE value. The refusal built from it then asserted the one cause it could not know.
551
+ * An outcome carries the cause, the raw message, and whether the failure is the class a relay may answer.
552
+ */
553
+ export type RunStartChannelOutcome =
554
+ /** The person answered, and their selection is exactly one of the gate's own labels. */
555
+ | { kind: 'answered'; answer: string }
556
+ /** The channel RESOLVED, so a person was reached — but the answer is not a label this gate offered. */
557
+ | { kind: 'unusable'; detail: string }
558
+ /** The channel THREW: no decision came back, and the cause is named rather than discarded. */
559
+ | { kind: 'unavailable'; cause: string; detail: string; relayable: boolean }
560
+
561
+ /**
562
+ * ⚠ THE CODES THAT MEAN THE QUESTION WAS CANCELLED OR CLOSED rather than never delivered: the person
563
+ * dismissed the card, their turn was cancelled, or a foreground window ended. A caller may not convert any
564
+ * of those into an approval by asking for the relay in the same breath. Every other failure means the
565
+ * question reached nobody — a composition or capability failure, which is the class the relay exists for.
566
+ */
567
+ export const NON_RELAYABLE_CHANNEL_CODES = ['ASK_CANCELLED', 'ASK_ABORTED', 'ASK_TIMED_OUT'] as const
568
+
569
+ /**
570
+ * Name the failure of one `ask()` call, without inventing anything about it.
571
+ *
572
+ * The cause is the error's own `code` when it has one (the harness's `UserQuestionError` carries
573
+ * `NO_PROVIDER`, `CALLER_NOT_LIVE`, `DELEGATED_CALLER`, `ASK_ABORTED`, …), else its `name`, else its
574
+ * JavaScript type. `detail` keeps the message verbatim so a reader sees the channel's own words rather
575
+ * than this plugin's paraphrase — the paraphrase is exactly how the previous version came to assert a
576
+ * cause nobody had.
577
+ */
578
+ export function classifyChannelFailure(err: unknown): { cause: string; detail: string; relayable: boolean } {
579
+ const code = (err as { code?: unknown } | null | undefined)?.code
580
+ const name = err instanceof Error ? err.name : typeof err
581
+ const message = err instanceof Error ? err.message : String(err)
582
+ const hasCode = typeof code === 'string' && code.trim() !== ''
583
+ const cause = hasCode ? (code as string) : name
584
+ const relayable = !(hasCode && (NON_RELAYABLE_CHANNEL_CODES as readonly string[]).includes(code as string))
585
+ return { cause, detail: 'channel threw ' + name + '[' + cause + ']: ' + message, relayable }
586
+ }
587
+
588
+ /** Describe an answer that arrived but is not a decision this gate can record. */
589
+ function describeUnusableAnswer(item: { selected: string[]; custom?: string } | undefined): string {
590
+ const offered: readonly string[] = RUN_START_GATE.options.map((option) => option.label)
591
+ const list = (values: readonly string[]): string => JSON.stringify(values.join(' | '))
592
+ if (item === undefined) {
593
+ return 'the channel resolved with no answer for question ' + JSON.stringify(RUN_START_GATE.id) + ' at all'
594
+ }
595
+ const raw = item.selected ?? []
596
+ const custom = item.custom?.trim() ?? ''
597
+ if (raw.length === 0 && custom === '') {
598
+ return 'the person skipped the question, and a skip is not an approval'
599
+ }
600
+ if (raw.length === 0) {
601
+ return 'the person answered ' + JSON.stringify(custom) + ' as free text rather than one of ' + list(offered)
602
+ }
603
+ // ⚠ THE SUBSET MATTERS. A UI can return a label the gate never offered, so "not exactly one of mine" is
604
+ // not the same statement as "the person chose something I do not know" — and the refusal says which.
605
+ const recognised = raw.filter((label) => offered.includes(label))
606
+ if (recognised.length === 0) {
607
+ return 'the person selected ' + list(raw) + ', and none of those name a label this gate offered (' + offered.join(' | ') + ')'
608
+ }
609
+ if (recognised.length === raw.length) {
610
+ return 'the person selected ' + list(raw) + ', and an approval is exactly one of ' + list(offered)
611
+ }
612
+ return 'the person selected ' + list(raw) + ', of which only ' + list(recognised) + ' name this gate\'s labels ' + list(offered)
613
+ }
614
+
615
+ /**
616
+ * Ask the run-start question through the blocking channel and report WHAT HAPPENED.
443
617
  *
444
- * The selection is filtered to the gate's OWN labels before it is returned: a question a UI answered with
445
- * a free-text custom value must not become an approval just because it arrived on the right channel.
618
+ * ⚠ THE CATCH IS THE POINT. It used to be `catch { return null }` — a blocking human question whose
619
+ * failure cause was erased at the exact moment the cause was the only thing worth knowing. Every path out
620
+ * of this function now says which path it was.
621
+ *
622
+ * The selection is filtered to the gate's OWN labels: a question a UI answered with a free-text custom
623
+ * value must not become an approval just because it arrived on the right channel.
446
624
  */
447
- async function askRunStartDirectly(
625
+ export async function askRunStartDirectly(
448
626
  channel: NonNullable<RecursiveRuntime['userQuestionsChannel']>,
449
627
  exec: { agent?: unknown; signal?: unknown; callId?: unknown },
450
- ): Promise<string | null> {
628
+ ): Promise<RunStartChannelOutcome> {
451
629
  const known = RUN_START_GATE.options.map((option) => option.label) as readonly string[]
452
630
  try {
453
631
  // The agent is passed as the LIVE handle the host gave this tool call. The real service validates it
@@ -467,11 +645,9 @@ async function askRunStartDirectly(
467
645
  } as never)
468
646
  const item = settled.answers.find((entry) => entry.id === RUN_START_GATE.id)
469
647
  const selected = item?.selected?.filter((label) => known.includes(label)) ?? []
470
- if (selected.length !== 1) return null
471
- return selected[0]
472
- } catch {
473
- // Quiet by design: the caller reports the situation, and this function's job is only to say whether a
474
- // person answered.
475
- return null
648
+ if (selected.length !== 1) return { kind: 'unusable', detail: describeUnusableAnswer(item) }
649
+ return { kind: 'answered', answer: selected[0] as string }
650
+ } catch (err) {
651
+ return { kind: 'unavailable', ...classifyChannelFailure(err) }
476
652
  }
477
653
  }
@@ -1,36 +1,54 @@
1
- import { defineTool } from '@deepseek-ai/dsh-tools'
2
- import { toolError } from './errors.ts'
3
- import type { JsonValue } from '@deepseek-ai/dsh-util-values'
4
- import type { RecursiveRuntime } from './runtime.ts'
5
-
6
- /**
7
- * `recursive_closeout` — scaffold a Phase 0-8 closeout receipt under the
8
- * SESSION's workspace only (R1 workspace-scoping invariant). The run is resolved
9
- * via the session agent's cwd -> workspace registry; a runId outside the current
10
- * workspace is rejected.
11
- */
12
- export function createRecursiveCloseoutTool(recursive: RecursiveRuntime) {
13
- return defineTool({
14
- name: 'recursive_closeout',
15
- description: 'REPORT what a closeout phase artifact is missing (Phase 4-8): it reads the artifact, lists the required sections and gates that are absent, and records a closeout receipt of its own. It NEVER writes the phase document. For a run in the CURRENT session workspace only. Workspace-scoped: refuses runIds outside the session\'s workspace. Delegates to the RecursiveRuntime service.',
16
- parameters: {
17
- phase: { type: 'string', description: 'Closeout phase to report on: 04, 05, 06, 07 or 08 (the same phase keys recursive_status prints). Phase 08 additionally fires the training trigger when it has been closed out before.' },
18
- runId: { type: 'string', description: 'Run id (e.g. 03-something). Required. Must resolve inside the current workspace.' },
19
- },
20
- output: {
21
- schema: { type: 'json' },
22
- render: (_args, value) => [{ type: 'text', text: JSON.stringify(value, null, 2) }],
23
- },
24
- async execute(args: { phase?: string; runId?: string }, exec) {
25
- if (!args.phase || !args.runId || args.runId.trim() === '') {
26
- return { error: toolError('MISSING_PHASE_AND_RUN') } as const
27
- }
28
- const root = await recursive.resolveWorkspaceRoot(exec.agent)
29
- if (!root) {
30
- return { error: toolError('NO_WORKSPACE') } as const
31
- }
32
- const result = await recursive.closeoutRun(root, args.runId.trim(), args.phase.trim(), exec.agent as { session?: { header?: { cwd?: string } } } | null)
33
- return result as unknown as JsonValue
34
- },
35
- })
1
+ import { defineTool } from '@deepseek-ai/dsh-tools'
2
+ import { toolError } from './errors.ts'
3
+ import { RUN_ID_EXAMPLES, RUN_ID_RULE, runIdProblem } from './run-id.ts'
4
+ import type { JsonValue } from '@deepseek-ai/dsh-util-values'
5
+ import type { RecursiveRuntime } from './runtime.ts'
6
+
7
+ /**
8
+ * `recursive_closeout` — scaffold a Phase 0-8 closeout receipt under the
9
+ * SESSION's workspace only (R1 workspace-scoping invariant). The run is resolved
10
+ * via the session agent's cwd -> workspace registry; a runId outside the current
11
+ * workspace is rejected.
12
+ *
13
+ * A RUN ID IS A NAME, NOT A PATH HERE TOO, and "outside the current workspace is rejected" is NOT enough
14
+ * on its own: `closeoutRun` joins the id onto the run layer and then writes a receipt under it, and its
15
+ * scoping check is `runDir.startsWith(runRoot)` — a string prefix test, not containment. A `..\` segment
16
+ * that lands on a SIBLING of the run layer passes that check whenever the sibling's name begins with
17
+ * `run`, so a path-shaped id can still receive a write. MEASURED pre-fix: `..\run-away` and `../run-away`
18
+ * were accepted and reached the report; the other shapes below were stopped only by the sibling not
19
+ * existing, which is the operator's filesystem deciding, not the tool. The rule is `run-id.ts`; this
20
+ * boundary is where the name enters, so this is where it is refused, with `recursive_init`'s refusal shape
21
+ * — `BAD_RUN_ID` (RM1107), same detail sentence.
22
+ */
23
+ export function createRecursiveCloseoutTool(recursive: RecursiveRuntime) {
24
+ return defineTool({
25
+ name: 'recursive_closeout',
26
+ description: 'REPORT what a closeout phase artifact is missing (Phase 4-8): it reads the artifact, lists the required sections and gates that are absent, and records a closeout receipt of its own. It NEVER writes the phase document. For a run in the CURRENT session workspace only. Workspace-scoped: refuses runIds outside the session\'s workspace. Delegates to the RecursiveRuntime service.',
27
+ parameters: {
28
+ phase: { type: 'string', description: 'Closeout phase to report on: 04, 05, 06, 07 or 08 (the same phase keys recursive_status prints). Phase 08 additionally fires the training trigger when it has been closed out before.' },
29
+ runId: { type: 'string', description: 'Run id — the NAME of the run directory under .recursive/run/ (e.g. 03-something), never a path: ' + RUN_ID_RULE + '. Required. Must resolve inside the current workspace.' },
30
+ },
31
+ output: {
32
+ schema: { type: 'json' },
33
+ render: (_args, value) => [{ type: 'text', text: JSON.stringify(value, null, 2) }],
34
+ },
35
+ async execute(args: { phase?: string; runId?: string }, exec) {
36
+ if (!args.phase || !args.runId || args.runId.trim() === '') {
37
+ return { error: toolError('MISSING_PHASE_AND_RUN') } as const
38
+ }
39
+ const runId = args.runId.trim()
40
+ // BEFORE `closeoutRun`, which joins the id onto the run layer and writes the closeout receipt under
41
+ // whatever directory it resolved to.
42
+ const problem = runIdProblem(runId)
43
+ if (problem !== null) {
44
+ return { error: toolError('BAD_RUN_ID', problem + ' - run ids allow ' + RUN_ID_RULE + ' - e.g. ' + RUN_ID_EXAMPLES) } as const
45
+ }
46
+ const root = await recursive.resolveWorkspaceRoot(exec.agent)
47
+ if (!root) {
48
+ return { error: toolError('NO_WORKSPACE') } as const
49
+ }
50
+ const result = await recursive.closeoutRun(root, runId, args.phase.trim(), exec.agent as { session?: { header?: { cwd?: string } } } | null)
51
+ return result as unknown as JsonValue
52
+ },
53
+ })
36
54
  }
@@ -1,5 +1,6 @@
1
1
  import { defineTool } from '@deepseek-ai/dsh-tools'
2
2
  import { codeRuntimeRefusal, toolError } from './errors.ts'
3
+ import { RUN_ID_EXAMPLES, RUN_ID_RULE, runIdProblem } from './run-id.ts'
3
4
  import type { JsonValue } from '@deepseek-ai/dsh-util-values'
4
5
  import type { RecursiveRuntime } from './runtime.ts'
5
6
 
@@ -17,13 +18,21 @@ import type { RecursiveRuntime } from './runtime.ts'
17
18
  * rather than "this call created something": re-initialising an APPROVED run reports `approved: true`
18
19
  * (and the run keeps its goal), which is what a caller re-scaffolding a run it already started needs to
19
20
  * see.
21
+ *
22
+ * AND A RUN ID IS A NAME, NOT A PATH. This is the boundary where the name enters, so it is the boundary
23
+ * that refuses a path-shaped one — loudly, and before `initRun` can mkdir anything. The check lives here
24
+ * (through the shared `run-id.ts` rule) rather than in `runtime.ts` because `initRun` is not the only
25
+ * caller and because the runtime's `join(root, '.recursive', 'run', runId)` is CORRECT for a name; what
26
+ * was missing was a gate on the name. Teaching the runtime to accept a path would silently relocate the
27
+ * run layer instead of rejecting the call. See `run-id.ts` for the rule, the evidence behind it, and the
28
+ * "do not fix this back" note.
20
29
  */
21
30
  export function createRecursiveInitTool(recursive: RecursiveRuntime) {
22
31
  return defineTool({
23
32
  name: 'recursive_init',
24
- description: 'Scaffold a new recursive-mode run directory (or ensure an existing one) with stub artifact headers. Delegates to the RecursiveRuntime service (no duplicated scaffolding logic). When createWorktree is true, a linked worktree is created first and the run is scaffolded inside it. THIS DOES NOT START THE RUN: no goal exists until the user approves phase 0 through recursive_ask gate=run-start, so the result carries runStartApproval — read it and ask.',
33
+ description: 'Scaffold a new recursive-mode run directory (or ensure an existing one) with stub artifact headers. Delegates to the RecursiveRuntime service (no duplicated scaffolding logic). When createWorktree is true, a linked worktree is created first and the run is scaffolded inside it. THIS DOES NOT START THE RUN: no goal exists until the user approves phase 0 through recursive_ask gate=run-start, so the result carries runStartApproval — read it and ask. The runId is the NAME of the run directory under .recursive/run/ and is never a path (see the parameter description): a path-shaped runId is refused before anything is written.',
25
34
  parameters: {
26
- runId: { type: 'string', description: 'Run id (e.g. 03-something). Required.' },
35
+ runId: { type: 'string', description: 'Run id — the NAME of the run directory under .recursive/run/ (e.g. 03-something, 01-calculator-lib), never a path: ' + RUN_ID_RULE + '. A run on another drive or inside a worktree is reached with recursive_worktree, not by passing a path here. Required.' },
27
36
  createWorktree: { type: 'boolean', description: 'If true, create a linked worktree at .worktrees/<runId>/ and scaffold the run inside it (default: false).' },
28
37
  baseBranch: { type: 'string', description: 'Base branch the worktree branch is cut from (default: current HEAD branch). Only used when createWorktree is true.' },
29
38
  },
@@ -32,12 +41,19 @@ export function createRecursiveInitTool(recursive: RecursiveRuntime) {
32
41
  render: (_args, value) => [{ type: 'text', text: JSON.stringify(value, null, 2) }],
33
42
  },
34
43
  async execute(args: { runId?: string; createWorktree?: boolean; baseBranch?: string }, exec) {
35
- if (!args.runId || args.runId.trim() === '') {
44
+ const runId = args.runId?.trim() ?? ''
45
+ if (runId === '') {
36
46
  return { error: toolError('MISSING_RUN_ID') } as const
37
47
  }
48
+ // BEFORE `initRun`: that call starts with `mkdirSync(runDir, { recursive: true })`, so a
49
+ // path-shaped id that reaches it has already made the operator-visible mess it was refused for.
50
+ const problem = runIdProblem(runId)
51
+ if (problem !== null) {
52
+ return { error: toolError('BAD_RUN_ID', problem + ' - run ids allow ' + RUN_ID_RULE + ' - e.g. ' + RUN_ID_EXAMPLES) } as const
53
+ }
38
54
  try {
39
55
  const result = await recursive.initRun(
40
- args.runId.trim(),
56
+ runId,
41
57
  exec.agent as { session?: { header?: { cwd?: string } } } | null,
42
58
  { createWorktree: args.createWorktree === true, baseBranch: args.baseBranch?.trim() || undefined },
43
59
  )
@@ -1,5 +1,6 @@
1
1
  import { defineTool } from '@deepseek-ai/dsh-tools'
2
2
  import { toolError } from './errors.ts'
3
+ import { RUN_ID_EXAMPLES, RUN_ID_RULE, runIdProblem } from './run-id.ts'
3
4
  import type { JsonValue } from '@deepseek-ai/dsh-util-values'
4
5
  import type { RecursiveRuntime } from './runtime.ts'
5
6
 
@@ -9,20 +10,39 @@ import type { RecursiveRuntime } from './runtime.ts'
9
10
  * (runtime.phaseRules -> phaseRulesFor) as the once-per-phase pre-step
10
11
  * reminder, so the agent can re-ask for the rules without re-injecting them on
11
12
  * every step. Returns { error } when no active phase is found.
13
+ *
14
+ * A RUN ID IS A NAME, NOT A PATH HERE TOO, INCLUDING WHEN IT IS OMITTED. Omitted and path-shaped are
15
+ * different cases and must stay different: omitted means "the latest run by mtime", which `resolveRunDir`
16
+ * answers by DISCOVERY rather than by joining anything, and that case is untouched below. A path-shaped id,
17
+ * by contrast, is joined by `phaseRules` -> `resolveRunDir` and then read, and `recordInjection` WRITES
18
+ * `memory-injections.json` under whatever directory it resolved to — with no scoping check at all on this
19
+ * path. So the gate fires only on an id that was actually supplied, and `run-id.ts` owns the rule. The
20
+ * refusal is `recursive_init`'s, `BAD_RUN_ID` (RM1107), composed identically.
12
21
  */
13
22
  export function createRecursivePhaseTool(recursive: RecursiveRuntime) {
14
23
  return defineTool({
15
24
  name: 'recursive_phase',
16
25
  description: 'Return the lint rules + instructions for the current recursive-mode phase (required sections, gates, TDD/QA notes). Call once when entering a new phase; the same rules are also auto-injected once per phase transition.',
17
26
  parameters: {
18
- runId: { type: 'string', description: 'Optional run id (defaults to the latest run by mtime)' },
27
+ runId: { type: 'string', description: 'Optional run id — the NAME of the run directory under .recursive/run/ (e.g. 03-something), never a path: ' + RUN_ID_RULE + '. Omit it for the latest run by mtime.' },
19
28
  },
20
29
  output: {
21
30
  schema: { type: 'json' },
22
31
  render: (_args, value) => [{ type: 'text', text: JSON.stringify(value, null, 2) }],
23
32
  },
24
33
  async execute(args: { runId?: string }, exec) {
25
- const result = await recursive.phaseRules(args.runId, exec.agent as { session?: { header?: { cwd?: string } } } | null)
34
+ // ABSENT IS NOT INVALID. No runId at all keeps its documented meaning — the latest run by mtime,
35
+ // resolved by discovery in `resolveRunDir` — so the gate below runs only when a name was supplied,
36
+ // while an EMPTY one is refused rather than silently read as "latest": a caller that passed `""` did
37
+ // not ask for discovery, and `resolveRunDir` would otherwise interpret their mistyped id as one.
38
+ const runId = args.runId?.trim()
39
+ if (runId !== undefined) {
40
+ const problem = runId === '' ? 'runId is empty' : runIdProblem(runId)
41
+ if (problem !== null) {
42
+ return { error: toolError('BAD_RUN_ID', problem + ' - run ids allow ' + RUN_ID_RULE + ' - e.g. ' + RUN_ID_EXAMPLES) } as const
43
+ }
44
+ }
45
+ const result = await recursive.phaseRules(runId, exec.agent as { session?: { header?: { cwd?: string } } } | null)
26
46
  if (!result) return { error: toolError('NO_PHASE') } as const
27
47
  return result as unknown as JsonValue
28
48
  },
@@ -1,5 +1,6 @@
1
1
  import { defineTool } from '@deepseek-ai/dsh-tools'
2
2
  import { toolError } from './errors.ts'
3
+ import { RUN_ID_EXAMPLES, RUN_ID_RULE, runIdProblem } from './run-id.ts'
3
4
  import type { JsonValue } from '@deepseek-ai/dsh-util-values'
4
5
  import type { RecursiveRuntime } from './runtime.ts'
5
6
 
@@ -7,6 +8,16 @@ import type { RecursiveRuntime } from './runtime.ts'
7
8
  * `recursive_scratch` — read/write/append the run-scoped disposable scratchpad
8
9
  * (R5) under the CURRENT session workspace only (R1). Scratch is git-ignored
9
10
  * and never citable as an Input.
11
+ *
12
+ * A RUN ID IS A NAME, NOT A PATH HERE TOO. `scratchRun` joins it onto the run layer, and it then WRITES
13
+ * (write/append) into the directory it landed on, so a path-shaped id does not merely fail to find a run:
14
+ * with a `..\` segment it can find and write into a SIBLING of the run layer, because the runtime's
15
+ * workspace-scoping check is `runDir.startsWith(runRoot)` — a string prefix test, not containment — and a
16
+ * sibling directory whose name begins with `run` passes it. MEASURED pre-fix: `..\run-away` was accepted
17
+ * and `scratchRun` wrote `scratch.md` into `<workspace>\.recursive\run-away\scratch\`. The rule is
18
+ * `run-id.ts`; the gate sits here, at the boundary where the name enters, rather than in `scratchRun`, for
19
+ * the same reason `recursive_init`'s does. Refusal shape is `recursive_init`'s, `BAD_RUN_ID` (RM1107),
20
+ * composed identically: one rule, one message.
10
21
  */
11
22
  export function createRecursiveScratchTool(recursive: RecursiveRuntime) {
12
23
  return defineTool({
@@ -14,7 +25,7 @@ export function createRecursiveScratchTool(recursive: RecursiveRuntime) {
14
25
  description: 'Read, write, or append the run-scoped disposable scratchpad (scratch/scratch.md or scratch/scratch.ts) for a run in the CURRENT session workspace. Workspace-scoped; scratch is git-ignored and never citable as an Input.',
15
26
  parameters: {
16
27
  action: { type: 'string', description: 'read | write | append. Required.' },
17
- runId: { type: 'string', description: 'Run id (e.g. 03-something). Required; must resolve inside the current workspace.' },
28
+ runId: { type: 'string', description: 'Run id — the NAME of the run directory under .recursive/run/ (e.g. 03-something), never a path: ' + RUN_ID_RULE + '. Required; must resolve inside the current workspace.' },
18
29
  target: { type: 'string', description: 'md | ts. Required.' },
19
30
  content: { type: 'string', description: 'Content for write/append. Optional for read.' },
20
31
  },
@@ -29,6 +40,11 @@ export function createRecursiveScratchTool(recursive: RecursiveRuntime) {
29
40
  if (!action || !runId || !target) {
30
41
  return { error: toolError('MISSING_SCRATCH_ARGS') } as const
31
42
  }
43
+ // BEFORE `scratchRun`, which joins the id onto the run layer and then writes through it.
44
+ const problem = runIdProblem(runId)
45
+ if (problem !== null) {
46
+ return { error: toolError('BAD_RUN_ID', problem + ' - run ids allow ' + RUN_ID_RULE + ' - e.g. ' + RUN_ID_EXAMPLES) } as const
47
+ }
32
48
  if (target !== 'md' && target !== 'ts') {
33
49
  return { error: toolError('BAD_TARGET') } as const
34
50
  }