@dzhechkov/harness-core 0.8.37 → 0.8.39
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.dz-manifest.json +253 -133
- package/README.md +209 -0
- package/dist/agentdb-index.d.ts +3 -3
- package/dist/agentdb-index.js +5 -5
- package/dist/agentdb-index.js.map +1 -1
- package/dist/amendment-trace.d.ts.map +1 -1
- package/dist/amendment-trace.js +4 -1
- package/dist/amendment-trace.js.map +1 -1
- package/dist/backlog.d.ts +2 -1
- package/dist/backlog.d.ts.map +1 -1
- package/dist/backlog.js +3 -2
- package/dist/backlog.js.map +1 -1
- package/dist/brain.d.ts.map +1 -1
- package/dist/brain.js +9 -2
- package/dist/brain.js.map +1 -1
- package/dist/bto-optimize.d.ts.map +1 -1
- package/dist/bto-optimize.js +9 -12
- package/dist/bto-optimize.js.map +1 -1
- package/dist/claim-check.d.ts.map +1 -1
- package/dist/claim-check.js +50 -14
- package/dist/claim-check.js.map +1 -1
- package/dist/cross-family-control.d.ts +35 -0
- package/dist/cross-family-control.d.ts.map +1 -1
- package/dist/cross-family-control.js +49 -3
- package/dist/cross-family-control.js.map +1 -1
- package/dist/experiment-assign.d.ts +110 -0
- package/dist/experiment-assign.d.ts.map +1 -0
- package/dist/experiment-assign.js +229 -0
- package/dist/experiment-assign.js.map +1 -0
- package/dist/feature-adr-checkpoints.d.ts +7 -2
- package/dist/feature-adr-checkpoints.d.ts.map +1 -1
- package/dist/feature-adr-checkpoints.js +20 -5
- package/dist/feature-adr-checkpoints.js.map +1 -1
- package/dist/feature-adr-envelope.d.ts +12 -0
- package/dist/feature-adr-envelope.d.ts.map +1 -1
- package/dist/feature-adr-envelope.js +12 -1
- package/dist/feature-adr-envelope.js.map +1 -1
- package/dist/feature-adr-routing.d.ts +4 -0
- package/dist/feature-adr-routing.d.ts.map +1 -1
- package/dist/feature-adr-routing.js +15 -1
- package/dist/feature-adr-routing.js.map +1 -1
- package/dist/index.d.ts +12 -5
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +11 -4
- package/dist/index.js.map +1 -1
- package/dist/ledger-cost-fill.d.ts +58 -0
- package/dist/ledger-cost-fill.d.ts.map +1 -0
- package/dist/ledger-cost-fill.js +78 -0
- package/dist/ledger-cost-fill.js.map +1 -0
- package/dist/loop-blobs.generated.js +2 -2
- package/dist/loop-blobs.generated.js.map +1 -1
- package/dist/mutation-gate.d.ts +29 -0
- package/dist/mutation-gate.d.ts.map +1 -1
- package/dist/mutation-gate.js +20 -0
- package/dist/mutation-gate.js.map +1 -1
- package/dist/name-check.d.ts +20 -1
- package/dist/name-check.d.ts.map +1 -1
- package/dist/name-check.js +42 -1
- package/dist/name-check.js.map +1 -1
- package/dist/no-stubs.d.ts +10 -0
- package/dist/no-stubs.d.ts.map +1 -1
- package/dist/no-stubs.js +13 -9
- package/dist/no-stubs.js.map +1 -1
- package/dist/patterns.d.ts +24 -0
- package/dist/patterns.d.ts.map +1 -1
- package/dist/patterns.js +49 -9
- package/dist/patterns.js.map +1 -1
- package/dist/publish-source-scope.d.ts +32 -0
- package/dist/publish-source-scope.d.ts.map +1 -0
- package/dist/publish-source-scope.js +53 -0
- package/dist/publish-source-scope.js.map +1 -0
- package/dist/publish.d.ts +5 -0
- package/dist/publish.d.ts.map +1 -1
- package/dist/publish.js +32 -21
- package/dist/publish.js.map +1 -1
- package/dist/qe-bridge.d.ts +12 -1
- package/dist/qe-bridge.d.ts.map +1 -1
- package/dist/qe-bridge.js +21 -11
- package/dist/qe-bridge.js.map +1 -1
- package/dist/rake-analyzer.d.ts +10 -3
- package/dist/rake-analyzer.d.ts.map +1 -1
- package/dist/rake-analyzer.js +84 -22
- package/dist/rake-analyzer.js.map +1 -1
- package/dist/recap.d.ts.map +1 -1
- package/dist/recap.js +8 -4
- package/dist/recap.js.map +1 -1
- package/dist/release.d.ts +2 -1
- package/dist/release.d.ts.map +1 -1
- package/dist/release.js +18 -4
- package/dist/release.js.map +1 -1
- package/dist/reqe.d.ts.map +1 -1
- package/dist/reqe.js +11 -1
- package/dist/reqe.js.map +1 -1
- package/dist/review-cost.d.ts +51 -0
- package/dist/review-cost.d.ts.map +1 -0
- package/dist/review-cost.js +110 -0
- package/dist/review-cost.js.map +1 -0
- package/dist/round.d.ts +136 -3
- package/dist/round.d.ts.map +1 -1
- package/dist/round.js +215 -6
- package/dist/round.js.map +1 -1
- package/dist/run-records.d.ts +51 -0
- package/dist/run-records.d.ts.map +1 -1
- package/dist/run-records.js +148 -2
- package/dist/run-records.js.map +1 -1
- package/dist/score.d.ts.map +1 -1
- package/dist/score.js +8 -1
- package/dist/score.js.map +1 -1
- package/dist/sign.d.ts +23 -0
- package/dist/sign.d.ts.map +1 -1
- package/dist/sign.js +52 -0
- package/dist/sign.js.map +1 -1
- package/dist/skills-verify.d.ts +8 -5
- package/dist/skills-verify.d.ts.map +1 -1
- package/dist/skills-verify.js +59 -7
- package/dist/skills-verify.js.map +1 -1
- package/dist/store-guard-prune.d.ts +22 -0
- package/dist/store-guard-prune.d.ts.map +1 -0
- package/dist/store-guard-prune.js +54 -0
- package/dist/store-guard-prune.js.map +1 -0
- package/dist/sweep-failure-classify.d.ts +45 -0
- package/dist/sweep-failure-classify.d.ts.map +1 -0
- package/dist/sweep-failure-classify.js +90 -0
- package/dist/sweep-failure-classify.js.map +1 -0
- package/dist/workflow-run-dispatch.d.ts +14 -2
- package/dist/workflow-run-dispatch.d.ts.map +1 -1
- package/dist/workflow-run-dispatch.js +14 -2
- package/dist/workflow-run-dispatch.js.map +1 -1
- package/package.json +1 -1
- package/sbom.json +432 -132
- package/src/agentdb-index.ts +5 -5
- package/src/amendment-trace.ts +4 -1
- package/src/backlog.ts +3 -2
- package/src/brain.ts +8 -2
- package/src/bto-optimize.ts +10 -8
- package/src/claim-check.ts +48 -12
- package/src/cross-family-control.ts +81 -3
- package/src/experiment-assign.ts +276 -0
- package/src/feature-adr-checkpoints.ts +20 -5
- package/src/feature-adr-envelope.ts +24 -1
- package/src/feature-adr-routing.ts +15 -1
- package/src/index.ts +23 -4
- package/src/loop-blobs.generated.ts +2 -2
- package/src/mutation-gate.ts +59 -0
- package/src/name-check.ts +43 -1
- package/src/no-stubs.ts +14 -5
- package/src/patterns.ts +49 -8
- package/src/publish-source-scope.ts +53 -0
- package/src/publish.ts +38 -16
- package/src/qe-bridge.ts +30 -11
- package/src/rake-analyzer.ts +75 -20
- package/src/rake-signatures.json +98 -0
- package/src/recap.ts +8 -4
- package/src/release.ts +18 -4
- package/src/reqe.ts +11 -1
- package/src/review-cost.ts +139 -0
- package/src/round.ts +327 -11
- package/src/run-records.ts +175 -1
- package/src/score.ts +8 -1
- package/src/sign.ts +53 -0
- package/src/skills-verify.ts +69 -9
- package/src/store-guard-prune.ts +81 -0
- package/src/sweep-failure-classify.ts +88 -0
- package/src/workflow-run-dispatch.ts +14 -2
package/src/run-records.ts
CHANGED
|
@@ -141,6 +141,17 @@ function priceLookup(modelId: string, table: Readonly<Record<string, LedgerPrice
|
|
|
141
141
|
return best;
|
|
142
142
|
}
|
|
143
143
|
|
|
144
|
+
// instrument-round-b fix-round-1 (Codex r1 HIGH finding 3, ADR-001 D4 amended): `--no-strict` was a
|
|
145
|
+
// BLANKET opt-out — since the one demonstrated automatic writer (the pipeline) always passed it, the
|
|
146
|
+
// new strict-by-default was operationally empty for every real caller. Replaced by a NAMED, SCOPED
|
|
147
|
+
// allowance: the caller states exactly which fields it expects to be incomplete
|
|
148
|
+
// (`allowIncomplete`) and WHY, from a closed set of reason codes — an incomplete row is written only
|
|
149
|
+
// when its actual incompleteness is a SUBSET of what was named AND the reason is recognized.
|
|
150
|
+
// Growing the incompleteness beyond what was declared (e.g. `envelope` going missing tomorrow) is
|
|
151
|
+
// refused, by construction, without anyone having to remember to tighten a flag.
|
|
152
|
+
export const INCOMPLETE_REASON_CODES = ['sandbox-metrics-unavailable', 'manual-entry'] as const;
|
|
153
|
+
export type IncompleteReasonCode = typeof INCOMPLETE_REASON_CODES[number];
|
|
154
|
+
|
|
144
155
|
export type RecordKind = 'ledger' | 'training-pair';
|
|
145
156
|
|
|
146
157
|
export type RecordVerdict =
|
|
@@ -290,6 +301,22 @@ export function decideRecordWrite(input: {
|
|
|
290
301
|
/** measurement-integrity FR-5/FR-6: rollout-log + price enrichment for a ledger row. Absent ⇒ zero
|
|
291
302
|
* behavior change (NFR-1). */
|
|
292
303
|
enrich?: LedgerEnrichInput;
|
|
304
|
+
/**
|
|
305
|
+
* instrument-round-b FR-4/A5 (ADR-001 D4), fix-round-1 (Codex r1 HIGH finding 3): an AUTO ledger
|
|
306
|
+
* row that would be written `complete:false` is refused (exit 2, before any write) UNLESS the
|
|
307
|
+
* actual incompleteness is fully covered by this NAMED, SCOPED allowance — the CLI's
|
|
308
|
+
* `--allow-incomplete <fields>`. A field this row is incomplete in that is NOT in this set still
|
|
309
|
+
* refuses, naming exactly the uncovered field(s). `--no-strict` (a blanket opt-out) is gone —
|
|
310
|
+
* see {@link INCOMPLETE_REASON_CODES}.
|
|
311
|
+
*/
|
|
312
|
+
allowIncomplete?: readonly string[];
|
|
313
|
+
/**
|
|
314
|
+
* instrument-round-b fix-round-1 (Codex r1 HIGH finding 3): WHY the row is legitimately
|
|
315
|
+
* incomplete — the CLI's `--incomplete-reason <code>`, required alongside `allowIncomplete` and
|
|
316
|
+
* validated against the closed {@link INCOMPLETE_REASON_CODES} set. An unrecognized or absent
|
|
317
|
+
* reason refuses the write even when every incomplete field IS named in `allowIncomplete`.
|
|
318
|
+
*/
|
|
319
|
+
incompleteReason?: string | null;
|
|
293
320
|
}): RecordDecision {
|
|
294
321
|
const { kind, payloadRaw, stage } = input;
|
|
295
322
|
if (kind !== 'ledger' && kind !== 'training-pair') {
|
|
@@ -350,7 +377,20 @@ export function decideRecordWrite(input: {
|
|
|
350
377
|
// carry `auto:true` too (not just gate on it transiently), so every downstream reader of the
|
|
351
378
|
// PERSISTED line keeps seeing the same signal `shapeMismatch` already gated on above. A no-op when
|
|
352
379
|
// the payload already said `auto:true` (shapeMismatch already refused any OTHER value).
|
|
353
|
-
|
|
380
|
+
// Lead delta after Codex r2 (N1 HIGH): `auto:true` used to be indistinguishable between "the
|
|
381
|
+
// CLI-level trusted marker was present" and "a hand-built payload said so", so a manual row could
|
|
382
|
+
// read as an automatic one. STRIPPING the claim was tried and rejected: the minutes-delta contract
|
|
383
|
+
// (auto + runId) legitimately rides payload-declared auto rows, and 28 existing callers write
|
|
384
|
+
// them. So the claim SURVIVES and its PROVENANCE is recorded instead — `autoSource` is the field a
|
|
385
|
+
// reader filters on when it needs trusted automatic rows only.
|
|
386
|
+
if (kind === 'ledger') {
|
|
387
|
+
if (input.auto === true) {
|
|
388
|
+
stamped['auto'] = true;
|
|
389
|
+
stamped['autoSource'] = 'cli-flag';
|
|
390
|
+
} else if (stamped['auto'] === true) {
|
|
391
|
+
stamped['autoSource'] = 'payload-claim';
|
|
392
|
+
}
|
|
393
|
+
}
|
|
354
394
|
const isGap = (v: unknown): boolean => v === null || v === undefined || (typeof v === 'string' && v.trim() === '');
|
|
355
395
|
if (input.timestamp != null && input.timestamp !== '') {
|
|
356
396
|
// An EMPTY STRING is a gap, not a value. Stamping only over null/undefined let
|
|
@@ -553,6 +593,89 @@ export function decideRecordWrite(input: {
|
|
|
553
593
|
}
|
|
554
594
|
}
|
|
555
595
|
|
|
596
|
+
// experiment-instrument FR-2/A3 (ADR-001): completeness of an AUTO ledger row. `minutes` is
|
|
597
|
+
// fill-only-null from a `wallSec` the payload carries (the workflow sandbox has a clock delta even
|
|
598
|
+
// when it has no wall clock of its own — FR-2's `wallSec` field, distinct from `minutesSincePrev`
|
|
599
|
+
// above, which needs a PREVIOUS row and a runId neither of which every auto row has). `tokens` is
|
|
600
|
+
// judged complete when it is a real number OR the row already NAMES why it is not (`tokensSource`,
|
|
601
|
+
// set above by the FR-5 rollout match, or supplied by the caller) — an unexplained non-number is the
|
|
602
|
+
// one shape that is actually incomplete. Gated on `auto` only: a MANUAL row never gains any of these
|
|
603
|
+
// three keys, so it stays byte-identical to before this feature (NFR-1).
|
|
604
|
+
if (kind === 'ledger' && stamped['auto'] === true) {
|
|
605
|
+
const incompleteReasons: string[] = [];
|
|
606
|
+
if (stamped['minutes'] === null || stamped['minutes'] === undefined) {
|
|
607
|
+
const wallSec = stamped['wallSec'];
|
|
608
|
+
if (typeof wallSec === 'number' && Number.isFinite(wallSec) && wallSec >= 0) {
|
|
609
|
+
stamped['minutes'] = Math.round((wallSec / 60) * 10) / 10;
|
|
610
|
+
stamped['minutesSource'] = 'wallSec';
|
|
611
|
+
} else {
|
|
612
|
+
incompleteReasons.push('minutes');
|
|
613
|
+
}
|
|
614
|
+
}
|
|
615
|
+
// r1-10 (Codex r1 HIGH #10): completeness requires a real, finite NUMBER of tokens. Naming why
|
|
616
|
+
// tokens are missing (`tokensSource:'unavailable'`, or any other provenance) is diagnostic, never
|
|
617
|
+
// a substitute for the number itself — ADR-001 says missing tokens makes the row incomplete, full
|
|
618
|
+
// stop. The old check (`tokensSource === undefined`) treated a NAMED absence as if it were data.
|
|
619
|
+
if (typeof stamped['tokens'] !== 'number' || !Number.isFinite(stamped['tokens'])) {
|
|
620
|
+
incompleteReasons.push('tokens');
|
|
621
|
+
}
|
|
622
|
+
// r1-11 (Codex r1 MEDIUM #11): a payload that ALREADY declared itself incomplete (its own
|
|
623
|
+
// `complete:false` + `incompleteReasons`, e.g. a workflow-side `'artifact'` reason this module
|
|
624
|
+
// knows nothing about) must never be overwritten back to `complete:true` just because THIS
|
|
625
|
+
// module's own minutes/tokens checks both passed — that erases a true fact and leaves a
|
|
626
|
+
// contradictory row (`complete:true` alongside a stale `incompleteReasons`). Preserve and MERGE.
|
|
627
|
+
const existingCompleteRaw = stamped['complete'];
|
|
628
|
+
const existingWasIncomplete = existingCompleteRaw === false;
|
|
629
|
+
const existingReasonsRaw = stamped['incompleteReasons'];
|
|
630
|
+
const existingReasons = Array.isArray(existingReasonsRaw)
|
|
631
|
+
? existingReasonsRaw.filter((r): r is string => typeof r === 'string')
|
|
632
|
+
: [];
|
|
633
|
+
const mergedReasons = existingWasIncomplete
|
|
634
|
+
? [...new Set([...existingReasons, ...incompleteReasons])]
|
|
635
|
+
: incompleteReasons;
|
|
636
|
+
const complete = mergedReasons.length === 0 && !existingWasIncomplete;
|
|
637
|
+
// A3/A5 (круг B, ADR-001 D4), fix-round-1 (Codex r1 HIGH finding 3, finding 9): incompleteness
|
|
638
|
+
// is a REFUSAL by DEFAULT, before any write — but the escape hatch is now a NAMED, SCOPED
|
|
639
|
+
// allowance, not a blanket flag: the row writes only when EVERY name in `mergedReasons` is
|
|
640
|
+
// covered by `input.allowIncomplete` AND `input.incompleteReason` is one of the closed
|
|
641
|
+
// {@link INCOMPLETE_REASON_CODES}. This is the load-bearing condition finding 9 asked for a
|
|
642
|
+
// real discriminating mutation against — an omitted allowance (the ordinary caller who never
|
|
643
|
+
// heard of this option) refuses exactly like круг B's original default; a caller naming the
|
|
644
|
+
// wrong reason, or a wider incompleteness than it declared, ALSO refuses.
|
|
645
|
+
// Lead delta after Codex r2 (N3 MEDIUM): the closed-list check used to live INSIDE the
|
|
646
|
+
// incomplete branch, so a COMPLETE row could carry `--incomplete-reason bogus` and be written
|
|
647
|
+
// with an unrecognized code sitting in its arguments. A supplied reason is validated whenever it
|
|
648
|
+
// is supplied, complete or not.
|
|
649
|
+
const suppliedReason = input.incompleteReason ?? null;
|
|
650
|
+
if (suppliedReason !== null && !(INCOMPLETE_REASON_CODES as readonly string[]).includes(suppliedReason)) {
|
|
651
|
+
return refuse(`--incomplete-reason ${JSON.stringify(suppliedReason)} is not a recognized reason code — expected one of ${INCOMPLETE_REASON_CODES.join(', ')}`);
|
|
652
|
+
}
|
|
653
|
+
if (!complete) {
|
|
654
|
+
const allowedFields = new Set(input.allowIncomplete ?? []);
|
|
655
|
+
const reason = suppliedReason;
|
|
656
|
+
const reasonKnown = reason !== null && (INCOMPLETE_REASON_CODES as readonly string[]).includes(reason);
|
|
657
|
+
// Lead delta after Codex r2 (N2 HIGH): a payload marked `complete:false` with NO concrete
|
|
658
|
+
// reasons cannot be covered by any allowance — the subset relation has nothing to check, so
|
|
659
|
+
// `uncovered` came out empty and the row slipped through. An unexplained incompleteness is a
|
|
660
|
+
// refusal: name the fields, or do not claim incompleteness.
|
|
661
|
+
if (mergedReasons.length === 0) {
|
|
662
|
+
return refuse('auto ledger row is marked `complete:false` but names no incompleteReasons — an allowance cannot cover an unnamed gap; list the incomplete field(s) or drop the claim');
|
|
663
|
+
}
|
|
664
|
+
const uncovered = mergedReasons.filter((r) => !allowedFields.has(r));
|
|
665
|
+
if (uncovered.length > 0 || !reasonKnown) {
|
|
666
|
+
if (reason === null && allowedFields.size === 0) {
|
|
667
|
+
return refuse(`auto ledger row is incomplete (${mergedReasons.join(', ') || 'previously marked incomplete'}) — refused by default; pass --allow-incomplete <fields> and --incomplete-reason <code> to permit a scoped incomplete write`);
|
|
668
|
+
}
|
|
669
|
+
if (!reasonKnown) {
|
|
670
|
+
return refuse(`--incomplete-reason ${JSON.stringify(reason)} is not a recognized reason code — expected one of ${INCOMPLETE_REASON_CODES.join(', ')}`);
|
|
671
|
+
}
|
|
672
|
+
return refuse(`auto ledger row is incomplete in field(s) not covered by --allow-incomplete: ${uncovered.join(', ')} (incomplete: ${mergedReasons.join(', ') || 'previously marked incomplete'})`);
|
|
673
|
+
}
|
|
674
|
+
}
|
|
675
|
+
stamped['complete'] = complete;
|
|
676
|
+
if (!complete) stamped['incompleteReasons'] = mergedReasons.length > 0 ? mergedReasons : existingReasons;
|
|
677
|
+
}
|
|
678
|
+
|
|
556
679
|
let line: string;
|
|
557
680
|
try {
|
|
558
681
|
line = JSON.stringify(stamped);
|
|
@@ -604,3 +727,54 @@ export function decideReadBack(appended: string, lastLineOnDisk: string | null):
|
|
|
604
727
|
export function recordVerdictLine(kind: RecordKind, stage: string, d: RecordDecision): string {
|
|
605
728
|
return `feature-adr record (${kind}/${stage}): ${d.verdict.toUpperCase()} — ${d.reason}`;
|
|
606
729
|
}
|
|
730
|
+
|
|
731
|
+
/** experiment-instrument FR-1/FR-3 (ADR-001): what `round.ts`'s `readOpenRoundTaskId` returns — the
|
|
732
|
+
* single source `applyTaskId` fills from. Duplicated here rather than imported so this pure module
|
|
733
|
+
* never depends on `round.ts`'s own shape; the CLI is the one holding both and wiring them together.
|
|
734
|
+
* r1-1/r1-2 (Codex r1 #1/#2): extended with `'derived-legacy'` and `'unavailable'` to stay in
|
|
735
|
+
* lockstep with `round.ts`'s own `readOpenRoundTaskId` return type. */
|
|
736
|
+
export interface TaskIdLookup {
|
|
737
|
+
readonly taskId: string | null;
|
|
738
|
+
readonly source: 'open-round' | 'derived-legacy' | 'no-open-round' | 'ambiguous' | 'unavailable';
|
|
739
|
+
}
|
|
740
|
+
|
|
741
|
+
/**
|
|
742
|
+
* experiment-instrument FR-1/FR-3/A8 (ADR-001): propagate `taskId` onto a ledger/training-pair payload
|
|
743
|
+
* BEFORE it reaches {@link decideRecordWrite} — fill-only-null, never overwritten.
|
|
744
|
+
*
|
|
745
|
+
* - The payload already names a non-empty `taskId` string ⇒ it is authoritative. When it DISAGREES
|
|
746
|
+
* with the round's own current taskId, that disagreement is a real fact worth keeping — recorded as
|
|
747
|
+
* `taskIdConflict: {payload, round}` — never silently resolved either way (A8).
|
|
748
|
+
* - The payload's `taskId` key is absent, or explicitly `null`/`undefined` ⇒ filled from `lookup`,
|
|
749
|
+
* INCLUDING the honest `null` case: no open round (A4) or two of them (A5) still stamps `taskId:
|
|
750
|
+
* null` + `taskIdSource` naming why, rather than leaving the field silently absent — absence with a
|
|
751
|
+
* named reason beats absence with none.
|
|
752
|
+
* - r1-3 (Codex r1 HIGH #3): the payload's `taskId` key is PRESENT with a value that is neither a
|
|
753
|
+
* non-empty string nor null/undefined (a number, a boolean, an object, or a blank/whitespace-only
|
|
754
|
+
* string) ⇒ that is a present-but-INVALID value, a THIRD case distinct from both of the above. It
|
|
755
|
+
* used to be treated exactly like "absent" (`typeof !== 'string'` fell through to the fill branch),
|
|
756
|
+
* silently replacing the caller's own (malformed) value with the round's — violating both
|
|
757
|
+
* fill-only-null and "a present payload value always wins". Now: the row's own value is preserved
|
|
758
|
+
* UNTOUCHED (never replaced with a guess about what the caller meant), and the problem is named in
|
|
759
|
+
* `taskIdInvalid` so a reader can see the row was neither filled nor trusted blindly.
|
|
760
|
+
*
|
|
761
|
+
* Pure: no filesystem, no clock. The CALLER (the cli) is the one that read `.dz/rounds/` to build
|
|
762
|
+
* `lookup` in the first place.
|
|
763
|
+
*/
|
|
764
|
+
export function applyTaskId(row: Record<string, unknown>, lookup: TaskIdLookup): Record<string, unknown> {
|
|
765
|
+
const hasTaskIdKey = Object.prototype.hasOwnProperty.call(row, 'taskId');
|
|
766
|
+
const rawPayloadTaskId = row['taskId'];
|
|
767
|
+
if (hasTaskIdKey && rawPayloadTaskId !== null && rawPayloadTaskId !== undefined) {
|
|
768
|
+
if (typeof rawPayloadTaskId === 'string' && rawPayloadTaskId.trim() !== '') {
|
|
769
|
+
const payloadTaskId = rawPayloadTaskId.trim();
|
|
770
|
+
if (lookup.taskId !== null && lookup.taskId !== payloadTaskId) {
|
|
771
|
+
return { ...row, taskIdConflict: { payload: payloadTaskId, round: lookup.taskId } };
|
|
772
|
+
}
|
|
773
|
+
return { ...row };
|
|
774
|
+
}
|
|
775
|
+
// r1-3: present but not a usable identity (non-string, or blank after trim) — refuse to replace
|
|
776
|
+
// it with a lookup guess; preserve it verbatim and name the problem.
|
|
777
|
+
return { ...row, taskIdInvalid: { value: rawPayloadTaskId, reason: 'taskId present but not a non-empty string' } };
|
|
778
|
+
}
|
|
779
|
+
return { ...row, taskId: lookup.taskId, taskIdSource: lookup.source };
|
|
780
|
+
}
|
package/src/score.ts
CHANGED
|
@@ -20,6 +20,7 @@
|
|
|
20
20
|
* PURE: the CLI reads the artifact files; this module only classifies.
|
|
21
21
|
*/
|
|
22
22
|
|
|
23
|
+
import { maskMarkdown } from './markdown-masker.js';
|
|
23
24
|
import { amendmentIdsIn } from './amendment-trace.js';
|
|
24
25
|
import { QE_SEVERITIES, findInvalidQeVerdictLines, parseQeFindings, readQeVerdictLines, type QeFindingsRefusedRow, type QeFindingsSummary } from './qe-findings.js';
|
|
25
26
|
|
|
@@ -103,7 +104,13 @@ export function observabilityAnswer(architectureMarkdown: string | undefined | n
|
|
|
103
104
|
// Fenced blocks are stripped FIRST: a `## Observability` inside an example fence is not a section
|
|
104
105
|
// of this document, and accepting one is a false pass anybody could write by accident
|
|
105
106
|
// (cross-family review, finding 1).
|
|
106
|
-
|
|
107
|
+
// Block scoping is DELEGATED to the canonical masker, not a local paired-fence regex. The regex
|
|
108
|
+
// it replaces required a MATCHING CLOSE, so an UNCLOSED fence stripped nothing at all and a
|
|
109
|
+
// `## Observability` quoted inside the example counted as a real section — MEASURED 2026-09-20,
|
|
110
|
+
// exactly the false pass the comment above warns about. `unclosed: 'hide'` is the CommonMark
|
|
111
|
+
// reading (an unclosed fence runs to end of document) and is the safe direction for a gate:
|
|
112
|
+
// a section that may be code must not be accepted as an answer.
|
|
113
|
+
const text = String(maskMarkdown(String(architectureMarkdown ?? '').replace(/\r\n/g, '\n'), { unclosed: 'hide' }));
|
|
107
114
|
// CommonMark: up to three leading spaces, then #s, then REQUIRED whitespace. `##Observability` is
|
|
108
115
|
// not a heading and must not pass; an indented one is, and must not be missed (finding 2).
|
|
109
116
|
const headingRe = new RegExp('^ {0,3}#{1,6}[ \t]+' + OBSERVABILITY_SECTION + '\\b.*$', 'gim');
|
package/src/sign.ts
CHANGED
|
@@ -769,3 +769,56 @@ export function decideVerifyPolicy(verdict: PackVerdict, requireSigning: boolean
|
|
|
769
769
|
? { action: 'fail', reason: 'pack is unsigned and --require-signing was passed' }
|
|
770
770
|
: { action: 'report', reason: 'pack is unsigned — not verified' };
|
|
771
771
|
}
|
|
772
|
+
|
|
773
|
+
/**
|
|
774
|
+
* Файлы, которые npm НИКОГДА не кладёт в тарбол, а рабочее дерево несёт почти всегда.
|
|
775
|
+
* Список намеренно короткий: каждая позиция — не эвристика вкуса, а правило упаковщика или
|
|
776
|
+
* общепринятый файл разработки, отсутствующий в опубликованном артефакте.
|
|
777
|
+
*/
|
|
778
|
+
const DEV_TREE_MARKERS: readonly string[] = [
|
|
779
|
+
'node_modules', // npm исключает ВСЕГДА, безусловно
|
|
780
|
+
'.git',
|
|
781
|
+
'tsconfig.json',
|
|
782
|
+
'tsconfig.tsbuildinfo',
|
|
783
|
+
'vitest.config.ts',
|
|
784
|
+
'vitest.config.mts',
|
|
785
|
+
'vitest.config.js',
|
|
786
|
+
'CHANGELOG.md',
|
|
787
|
+
'test',
|
|
788
|
+
'tests',
|
|
789
|
+
];
|
|
790
|
+
|
|
791
|
+
/**
|
|
792
|
+
* Объяснение к ПРОВАЛИВШЕЙСЯ проверке подписи, когда каталог похож на рабочее дерево.
|
|
793
|
+
*
|
|
794
|
+
* ЗАЧЕМ ЭТО СУЩЕСТВУЕТ. Манифест описывает ОПУБЛИКОВАННЫЙ ТАРБОЛ, а не рабочее дерево: при
|
|
795
|
+
* публикации package.json переписывается (`workspace:*` разворачивается, версия бампается), а
|
|
796
|
+
* файлы разработки отфильтровываются. Поэтому сверка неотфильтрованного дерева расходится
|
|
797
|
+
* ПО ПОСТРОЕНИЮ, и вывод выглядит как «подпись сломана» или «дерево подменили».
|
|
798
|
+
*
|
|
799
|
+
* ИЗМЕРЕНО 2026-09-01: обход 55 подписанных пакетов в рабочем дереве дал 54 отказа из 55, и за
|
|
800
|
+
* один день на это независимо попались ТРОЕ — ведущий агент и два роя; один внёс это в отчёт
|
|
801
|
+
* как системный дефект подписей, другой предложил завести задачу, третий чуть не поднял тревогу
|
|
802
|
+
* о взломе доверия. Инструмент, трижды за день обманувший опытного пользователя, дефектен сам.
|
|
803
|
+
*
|
|
804
|
+
* ПОЧЕМУ ЭТО ОБЪЯСНЕНИЕ, А НЕ КЛАССИФИКАЦИЯ. Соседний путь (обход установленных пакетов) считает
|
|
805
|
+
* исходным деревом всё, что лежит не под `node_modules`. Здесь так нельзя: законно распакованный
|
|
806
|
+
* тарбол в /tmp тоже лежит не под `node_modules`, и его объявили бы деревом. Поэтому функция
|
|
807
|
+
* НИЧЕГО НЕ УТВЕРЖДАЕТ о природе каталога — она НАЗЫВАЕТ НАЙДЕННЫЕ УЛИКИ и оставляет вывод
|
|
808
|
+
* человеку. Нет улик — нет и объяснения: молчание здесь честнее догадки.
|
|
809
|
+
*
|
|
810
|
+
* @param entries имена верхнего уровня в проверяемом каталоге
|
|
811
|
+
* @returns текст пояснения или null, если улик рабочего дерева не нашлось
|
|
812
|
+
*/
|
|
813
|
+
export function explainPackVerificationFailure(entries: readonly string[]): string | null {
|
|
814
|
+
const found = DEV_TREE_MARKERS.filter((marker) => entries.includes(marker));
|
|
815
|
+
if (found.length === 0) return null;
|
|
816
|
+
return [
|
|
817
|
+
'',
|
|
818
|
+
'Прежде чем читать это как подделку: манифест описывает ОПУБЛИКОВАННЫЙ ТАРБОЛ, а не рабочее дерево.',
|
|
819
|
+
`Этот каталог несёт файлы разработки, которых в тарболе нет: ${found.join(', ')}.`,
|
|
820
|
+
'При публикации package.json переписывается, а файлы разработки отфильтровываются, поэтому',
|
|
821
|
+
'сверка неотфильтрованного дерева расходится ПО ПОСТРОЕНИЮ — это не улика подделки.',
|
|
822
|
+
'Проверять надо артефакт: npm pack <имя>@<версия> → распаковать → dz verify-pack --pack <распакованное>.',
|
|
823
|
+
].join('\n');
|
|
824
|
+
}
|
package/src/skills-verify.ts
CHANGED
|
@@ -6,8 +6,8 @@
|
|
|
6
6
|
* the PROPERTY (registration). This module makes the property observable.
|
|
7
7
|
*
|
|
8
8
|
* Two layers:
|
|
9
|
-
* L1 `scanSkillsLayout` — instant, no Claude session:
|
|
10
|
-
*
|
|
9
|
+
* L1 `scanSkillsLayout` — instant, no Claude session: names whose on-disk form passes the
|
|
10
|
+
* static checks, plus known issue shapes.
|
|
11
11
|
* L2 `parseInitFacts` — the authoritative listing, parsed from the `system/init` event of
|
|
12
12
|
* + `classifyRegistration` `claude -p --output-format stream-json --verbose`. No model prose.
|
|
13
13
|
*
|
|
@@ -19,13 +19,15 @@ import { basename, isAbsolute, join, resolve } from 'node:path';
|
|
|
19
19
|
|
|
20
20
|
// ── Layer 1: static layout scan ─────────────────────────────────────
|
|
21
21
|
|
|
22
|
-
/** The
|
|
22
|
+
/** The seven static issue shapes this scanner reports. */
|
|
23
23
|
export type SkillIssueKind =
|
|
24
24
|
| 'no-skill-md' // a skill dir with no SKILL.md at depth 1 → never registers
|
|
25
25
|
| 'buried-skill-md' // a SKILL.md at depth >= 2 → the loader does not scan that deep
|
|
26
26
|
| 'plugin-manifest-trap' // .claude-plugin/plugin.json under .claude/skills → does NOT auto-register
|
|
27
27
|
| 'wildcard-allowed-tools' // `allowed-tools: *` → every tool granted; a FINDING (see below)
|
|
28
|
-
| 'empty-allowed-tools'
|
|
28
|
+
| 'empty-allowed-tools' // `allowed-tools:` with no value → looks restrictive, restricts nothing
|
|
29
|
+
| 'missing-frontmatter-fence' // measured absent from the registry without an opening `---`
|
|
30
|
+
| 'missing-description'; // measured absent from the registry without a non-empty description
|
|
29
31
|
|
|
30
32
|
export interface SkillLayoutFinding {
|
|
31
33
|
readonly dir: string;
|
|
@@ -38,7 +40,10 @@ export interface StaticScan {
|
|
|
38
40
|
readonly projectDir: string;
|
|
39
41
|
readonly skillsRoot: string;
|
|
40
42
|
readonly exists: boolean;
|
|
41
|
-
/**
|
|
43
|
+
/**
|
|
44
|
+
* Bare skill names whose on-disk form passes static layout and measured frontmatter checks.
|
|
45
|
+
* Actual registration is established only by the Layer-2 session listing.
|
|
46
|
+
*/
|
|
42
47
|
readonly registrable: readonly string[];
|
|
43
48
|
/** Load-blocking problems: these make the verdict FAIL. */
|
|
44
49
|
readonly findings: readonly SkillLayoutFinding[];
|
|
@@ -228,6 +233,39 @@ export function parseAllowedTools(markdown: string): AllowedToolsState {
|
|
|
228
233
|
return { state: 'listed', values, wildcard: values.includes('*') };
|
|
229
234
|
}
|
|
230
235
|
|
|
236
|
+
type SkillFrontmatterIssue = 'missing-frontmatter-fence' | 'missing-description';
|
|
237
|
+
|
|
238
|
+
/**
|
|
239
|
+
* Read the registration-relevant frontmatter fields, line by line and only before its closing fence.
|
|
240
|
+
* Scanning the whole file would let `description:` in skill documentation suppress a real finding,
|
|
241
|
+
* the same false-positive/false-negative boundary guarded by `parseAllowedTools` above.
|
|
242
|
+
*
|
|
243
|
+
* The two conditions are measured as absence from the registry, not refusal of invocation. Each
|
|
244
|
+
* measurement used a fresh session, so it also does not establish appearance in the same response
|
|
245
|
+
* without a restart.
|
|
246
|
+
*/
|
|
247
|
+
function findSkillFrontmatterIssue(markdown: string): SkillFrontmatterIssue | null {
|
|
248
|
+
const lines = markdown.split(/\r?\n/);
|
|
249
|
+
// NO explicit BOM strip, and that is deliberate: `String.prototype.trim()` already removes U+FEFF
|
|
250
|
+
// (it is <ZWNBSP>, part of the WhiteSpace production), so a BOM-prefixed fence compares equal to
|
|
251
|
+
// a bare one. MEASURED 2026-09-19: adding a strip left the mutant "don't strip" 70/70 green —
|
|
252
|
+
// dead code. The tolerance is pinned by a test instead, so replacing `trim()` with anything
|
|
253
|
+
// stricter goes red rather than silently rejecting every BOM-prefixed skill.
|
|
254
|
+
if (lines[0]?.trim() !== '---') return 'missing-frontmatter-fence';
|
|
255
|
+
|
|
256
|
+
let description: string | null = null;
|
|
257
|
+
for (let i = 1; i < lines.length; i++) {
|
|
258
|
+
const line = lines[i] ?? '';
|
|
259
|
+
if (line.trim() === '---') break;
|
|
260
|
+
const match = /^description:(.*)$/.exec(line);
|
|
261
|
+
if (match) {
|
|
262
|
+
description = match[1] ?? '';
|
|
263
|
+
break;
|
|
264
|
+
}
|
|
265
|
+
}
|
|
266
|
+
return description === null || description.trim() === '' ? 'missing-description' : null;
|
|
267
|
+
}
|
|
268
|
+
|
|
231
269
|
/**
|
|
232
270
|
* Privilege findings for one skill directory.
|
|
233
271
|
*
|
|
@@ -426,16 +464,38 @@ export function scanSkillsLayout(projectDir: string): StaticScan {
|
|
|
426
464
|
const isPluginContainer = hasPluginManifest(dir);
|
|
427
465
|
const bucket = isPluginContainer ? advisories : findings;
|
|
428
466
|
|
|
467
|
+
// Read under a guard, exactly like scanSkillPrivileges: an unreadable SKILL.md is another
|
|
468
|
+
// check's business, and this scanner's contract is to REPORT findings, never to throw out of
|
|
469
|
+
// the whole sweep over one bad file.
|
|
470
|
+
let frontmatterIssue: SkillFrontmatterIssue | null = null;
|
|
471
|
+
if (registers) {
|
|
472
|
+
try {
|
|
473
|
+
frontmatterIssue = findSkillFrontmatterIssue(readFileSync(join(dir, 'SKILL.md'), 'utf8'));
|
|
474
|
+
} catch {
|
|
475
|
+
frontmatterIssue = null;
|
|
476
|
+
}
|
|
477
|
+
}
|
|
478
|
+
if (frontmatterIssue) {
|
|
479
|
+
findings.push({
|
|
480
|
+
dir: name,
|
|
481
|
+
kind: frontmatterIssue,
|
|
482
|
+
detail: frontmatterIssue === 'missing-frontmatter-fence'
|
|
483
|
+
? 'SKILL.md has no opening frontmatter fence — measured absent from the registry without it'
|
|
484
|
+
: 'SKILL.md has no non-empty frontmatter description — measured absent from the registry without one',
|
|
485
|
+
});
|
|
486
|
+
}
|
|
487
|
+
|
|
429
488
|
// Права проверяются у КАЖДОГО навыка с читаемым SKILL.md, включая одно-навыковый плагин: щедрая
|
|
430
489
|
// выдача не становится безопаснее оттого, что навык лежит в контейнере.
|
|
431
490
|
if (registers) scanSkillPrivileges(dir, name, findings, advisories);
|
|
432
491
|
|
|
433
|
-
if (registers
|
|
434
|
-
|
|
435
|
-
|
|
492
|
+
if (registers) {
|
|
493
|
+
if (!isPluginContainer && !frontmatterIssue) {
|
|
494
|
+
registrable.push(name);
|
|
495
|
+
}
|
|
436
496
|
// A single-skill plugin: it has a depth-1 SKILL.md AND a manifest, so it registers NAMESPACED.
|
|
437
497
|
// Expecting the bare directory name here was a false FAIL (QE5 #2); the container check below
|
|
438
|
-
// accepts either form.
|
|
498
|
+
// accepts either form. A frontmatter finding likewise keeps the bare name out of registrable.
|
|
439
499
|
} else {
|
|
440
500
|
// A dir with no markdown was never meant to be a skill — advisory, not a failure (QE5 #7).
|
|
441
501
|
const intended = looksLikeSkillDir(dir);
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
import { sep } from 'node:path';
|
|
2
|
+
|
|
3
|
+
export type StoreGuardPruneBucket = 'stale-temp' | 'live' | 'gone-outside-tmp' | 'unreadable';
|
|
4
|
+
|
|
5
|
+
export interface StoreGuardPruneEntry {
|
|
6
|
+
readonly file: string;
|
|
7
|
+
readonly project: string | null;
|
|
8
|
+
readonly bytes: number;
|
|
9
|
+
readonly bucket: StoreGuardPruneBucket;
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
export interface StoreGuardPrunePlan {
|
|
13
|
+
readonly entries: readonly StoreGuardPruneEntry[];
|
|
14
|
+
readonly counts: Readonly<Record<StoreGuardPruneBucket, number>>;
|
|
15
|
+
readonly reclaimableBytes: number;
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
export interface StoreGuardPruneDeps {
|
|
19
|
+
readonly exists: (path: string) => boolean;
|
|
20
|
+
readonly tmpDirs: readonly string[];
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* A temp-root candidate that authorizes deletion must be a REAL directory path, never the filesystem
|
|
25
|
+
* root or an empty string. MEASURED 2026-09-20 (cross-family review, P1): `TMPDIR=/` makes
|
|
26
|
+
* `os.tmpdir()` return the single character `"/"` (node strips a trailing slash only when the path is
|
|
27
|
+
* longer than one character), and `isUnder` then matches EVERY absolute path — so every
|
|
28
|
+
* `gone-outside-tmp` mark, which FR-3 exists to protect, would be classified `stale-temp` and deleted.
|
|
29
|
+
* A degenerate candidate authorizes nothing and is dropped here, in the pure half, where it is testable.
|
|
30
|
+
*/
|
|
31
|
+
function isUsableTmpDir(tmpDir: string): boolean {
|
|
32
|
+
const trimmed = tmpDir.trim();
|
|
33
|
+
return trimmed.length > 1 && trimmed !== sep && trimmed !== '/';
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
function isUnder(project: string, tmpDir: string): boolean {
|
|
37
|
+
const prefix = tmpDir.endsWith(sep) ? tmpDir : `${tmpDir}${sep}`;
|
|
38
|
+
return project === tmpDir || project.startsWith(prefix);
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
export function planStoreGuardPrune(
|
|
42
|
+
input: readonly { readonly file: string; readonly bytes: number; readonly text: string }[],
|
|
43
|
+
deps: StoreGuardPruneDeps,
|
|
44
|
+
): StoreGuardPrunePlan {
|
|
45
|
+
const counts: Record<StoreGuardPruneBucket, number> = {
|
|
46
|
+
'stale-temp': 0,
|
|
47
|
+
live: 0,
|
|
48
|
+
'gone-outside-tmp': 0,
|
|
49
|
+
unreadable: 0,
|
|
50
|
+
};
|
|
51
|
+
let reclaimableBytes = 0;
|
|
52
|
+
|
|
53
|
+
const entries = input.map(({ file, bytes, text }): StoreGuardPruneEntry => {
|
|
54
|
+
let value: unknown;
|
|
55
|
+
try {
|
|
56
|
+
value = JSON.parse(text);
|
|
57
|
+
} catch {
|
|
58
|
+
counts.unreadable += 1;
|
|
59
|
+
return { file, project: null, bytes, bucket: 'unreadable' };
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
if (typeof value !== 'object' || value === null
|
|
63
|
+
|| typeof (value as { project?: unknown }).project !== 'string'
|
|
64
|
+
|| (value as { project: string }).project.length === 0) {
|
|
65
|
+
counts.unreadable += 1;
|
|
66
|
+
return { file, project: null, bytes, bucket: 'unreadable' };
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
const project = (value as { project: string }).project;
|
|
70
|
+
const bucket: StoreGuardPruneBucket = deps.exists(project)
|
|
71
|
+
? 'live'
|
|
72
|
+
: deps.tmpDirs.filter(isUsableTmpDir).some((tmpDir) => isUnder(project, tmpDir))
|
|
73
|
+
? 'stale-temp'
|
|
74
|
+
: 'gone-outside-tmp';
|
|
75
|
+
counts[bucket] += 1;
|
|
76
|
+
if (bucket === 'stale-temp') reclaimableBytes += bytes;
|
|
77
|
+
return { file, project, bytes, bucket };
|
|
78
|
+
});
|
|
79
|
+
|
|
80
|
+
return { entries, counts, reclaimableBytes };
|
|
81
|
+
}
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Различитель красных обхода: ЛОГИЧЕСКИЙ отказ против отказа ПОД НАГРУЗКОЙ.
|
|
3
|
+
*
|
|
4
|
+
* ЗАЧЕМ. Запись 9e48a975 установила класс: набор harness-core детерминирован В ОДИНОЧКУ и
|
|
5
|
+
* краснеет ТОЛЬКО под параллельной нагрузкой (пять последовательных прогонов — ноль красных;
|
|
6
|
+
* те же тесты под фоновым вторым набором — красные поодиночке и в разных сочетаниях).
|
|
7
|
+
* Подпись класса: тест (а) читает собранный `dist`, (б) порождает процессы, (в) несёт явный
|
|
8
|
+
* таймаут, близкий к измеренному времени.
|
|
9
|
+
*
|
|
10
|
+
* ИЗМЕРЕНО 2026-09-19, новый экземпляр того же класса: `scout` прошёл в одном обходе и упал в
|
|
11
|
+
* следующем через 17 минут — `deep-analyzer.test.ts`, `Test timed out in 5000ms`, 192 из 193
|
|
12
|
+
* зелёные; тот же файл в одиночку 5 из 5. Единственное различие между прогонами — загрузка
|
|
13
|
+
* машины (подкачка 2342 → 2671 МБ).
|
|
14
|
+
*
|
|
15
|
+
* Класс диагностирован, но МЕХАНИЗМА не имел: каждый экземпляр стоил ручного разбора. Этот
|
|
16
|
+
* модуль — чистая половина механизма: он извлекает из журнала обхода ИМЕНА упавших файлов и
|
|
17
|
+
* классифицирует результат ПЕРЕПРОГОНА В ОДИНОЧКУ. Запуск живёт в скрипте, здесь его нет.
|
|
18
|
+
*
|
|
19
|
+
* ТРЁХЗНАЧНОСТЬ ОБЯЗАТЕЛЬНА. Перепрогон, который не удалось выполнить, — это НЕ «прошёл»;
|
|
20
|
+
* прибор, который не смог посмотреть, не имеет права на вердикт.
|
|
21
|
+
*/
|
|
22
|
+
|
|
23
|
+
/** Вердикт перепрогона одного файла в одиночку. */
|
|
24
|
+
export type RerunVerdict = 'does-not-reproduce-alone' | 'reproduces-alone' | 'inconclusive';
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* Имена упавших тестовых файлов из журнала набора. Понимает ОБА формата, которые даёт обход:
|
|
28
|
+
* vitest ` FAIL test/x.test.ts > describe > it`
|
|
29
|
+
* node:test ` location: '/abs/.../test/x.test.mjs:95:1'`
|
|
30
|
+
*
|
|
31
|
+
* Пути приводятся к виду, относительному каталогу пакета, потому что перепрогон запускается
|
|
32
|
+
* из него. Порядок сохраняется, дубликаты убираются: один файл перепрогоняется один раз.
|
|
33
|
+
*/
|
|
34
|
+
export function parseFailedTestFiles(logText: string, packageDir?: string): string[] {
|
|
35
|
+
const out: string[] = [];
|
|
36
|
+
const seen = new Set<string>();
|
|
37
|
+
const add = (raw: string): void => {
|
|
38
|
+
let p = raw.trim();
|
|
39
|
+
if (packageDir !== undefined && p.startsWith(packageDir)) {
|
|
40
|
+
p = p.slice(packageDir.length).replace(/^\/+/, '');
|
|
41
|
+
}
|
|
42
|
+
// Абсолютный путь к ЧУЖОМУ пакету перепрогонять нельзя: он не относителен нашему каталогу.
|
|
43
|
+
if (p.startsWith('/')) return;
|
|
44
|
+
if (p === '' || seen.has(p)) return;
|
|
45
|
+
seen.add(p);
|
|
46
|
+
out.push(p);
|
|
47
|
+
};
|
|
48
|
+
for (const line of logText.split('\n')) {
|
|
49
|
+
const vitest = /^\s*FAIL\s+(?:\|[^|]*\|\s*)?(\S+\.(?:test|spec)\.[cm]?[jt]s)\b/.exec(line);
|
|
50
|
+
if (vitest?.[1] !== undefined) { add(vitest[1]); continue; }
|
|
51
|
+
const node = /location:\s*'([^']+\.(?:test|spec)\.[cm]?[jt]s):\d+/.exec(line);
|
|
52
|
+
if (node?.[1] !== undefined) { add(node[1]); }
|
|
53
|
+
}
|
|
54
|
+
return out;
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/**
|
|
58
|
+
* Вердикт по исходу перепрогона.
|
|
59
|
+
*
|
|
60
|
+
* @param sweepFailed упал ли файл в обходе (иначе классифицировать нечего)
|
|
61
|
+
* @param rerun результат одиночного прогона, или null если запустить не удалось
|
|
62
|
+
*/
|
|
63
|
+
export function classifyRerun(
|
|
64
|
+
sweepFailed: boolean,
|
|
65
|
+
rerun: { readonly exitCode: number; readonly ranAnyTest: boolean } | null,
|
|
66
|
+
): RerunVerdict {
|
|
67
|
+
if (!sweepFailed) return 'inconclusive';
|
|
68
|
+
if (rerun === null) return 'inconclusive';
|
|
69
|
+
// «Ни одного теста не запустилось» — это не зелёный прогон, а несостоявшаяся проба: так
|
|
70
|
+
// выглядит и опечатка в пути, и набор, который не грузится. Читать это как «прошёл» значит
|
|
71
|
+
// объявить файл нагрузочно-зависимым на основании того, что его никто не проверил.
|
|
72
|
+
if (!rerun.ranAnyTest) return 'inconclusive';
|
|
73
|
+
return rerun.exitCode === 0 ? 'does-not-reproduce-alone' : 'reproduces-alone';
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/** Человеческая расшифровка вердикта — одна строка, называющая, что делать дальше. */
|
|
77
|
+
export function describeVerdict(verdict: RerunVerdict): string {
|
|
78
|
+
if (verdict === 'does-not-reproduce-alone') {
|
|
79
|
+
// НЕ «зависит от нагрузки»: у зелёного одиночного прогона есть ДВЕ причины, и различить их
|
|
80
|
+
// перепрогоном нельзя — нагрузка обхода ЛИБО починка, внесённая уже после него. Называть
|
|
81
|
+
// одну из двух значило бы выдать догадку за измерение.
|
|
82
|
+
return 'в одиночку ЗЕЛЁНЫЙ — либо отказ зависел от нагрузки обхода, либо починка легла после него; перепрогон эти две причины не различает';
|
|
83
|
+
}
|
|
84
|
+
if (verdict === 'reproduces-alone') {
|
|
85
|
+
return 'в одиночку ТОЖЕ красный — настоящий дефект, нагрузка ни при чём';
|
|
86
|
+
}
|
|
87
|
+
return 'НЕ УСТАНОВЛЕНО — перепрогон не состоялся; это не «прошёл»';
|
|
88
|
+
}
|
|
@@ -11,6 +11,7 @@
|
|
|
11
11
|
* a convention whose reason is lost is the next thing somebody "cleans up".
|
|
12
12
|
*/
|
|
13
13
|
|
|
14
|
+
import { CODEX_EXEC_PROMPT_CEILING_CHARS } from './feature-adr-routing.js';
|
|
14
15
|
import type { Deliverable } from './loop-plan.js';
|
|
15
16
|
import { claudeProbeArgs, claudeReviewArgs, extractClaudeResult, interpretClaudeProbe, type BridgeFamily } from './qe-bridge.js';
|
|
16
17
|
import type { WfRunReason } from './workflow-run.js';
|
|
@@ -149,13 +150,24 @@ export const CODEX_SCOPING_PREFIX =
|
|
|
149
150
|
'Answer directly from this prompt text alone; no commands, no files, no tools.';
|
|
150
151
|
|
|
151
152
|
/**
|
|
152
|
-
* The REAL prompt ceiling for `codex exec
|
|
153
|
+
* The REAL prompt ceiling for `codex exec` — RE-EXPORTED, never re-declared.
|
|
153
154
|
*
|
|
154
155
|
* The folk value 1200 is refuted history: it came from an era when the prompt travelled through a
|
|
155
156
|
* fire-and-forget wrapper. Over-ceiling ⇒ a LOUD `prompt-over-ceiling`, never truncation — a
|
|
156
157
|
* truncated prompt produces a confident answer to a question nobody asked.
|
|
158
|
+
*
|
|
159
|
+
* ONE NUMBER, ONE DEFINITION. Until 2026-09-20 this module declared its own `= 24_000` beside the
|
|
160
|
+
* one in `feature-adr-routing.ts`, and NOTHING compared them: no test imported both, so the two
|
|
161
|
+
* could diverge in silence and `codexExecPlan` would route to Claude at a different threshold than
|
|
162
|
+
* `makeCodexExecDispatcher` rejects at. Measured: `feature-adr-codex-dispatch.test.ts` asserts only
|
|
163
|
+
* `> 4000`, and the mirror guard in `codex-scoped-review.test.ts` ties routing.ts to the four
|
|
164
|
+
* workflow copies but knows nothing about this file.
|
|
165
|
+
*
|
|
166
|
+
* The definition lives in `feature-adr-routing.ts` and not here because THAT module is lifted
|
|
167
|
+
* verbatim into the workflow sandbox by `scripts/gen-loop-blobs.mjs` and therefore may not import
|
|
168
|
+
* anything. The dependency can only point this way.
|
|
157
169
|
*/
|
|
158
|
-
export
|
|
170
|
+
export { CODEX_EXEC_PROMPT_CEILING_CHARS };
|
|
159
171
|
|
|
160
172
|
/**
|
|
161
173
|
* argv for one codex dispatch. The prompt travels as ONE argv element (no shell, no quoting), and
|