@dzhechkov/harness-core 0.8.37 → 0.8.38

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -290,6 +290,10 @@ export function decideRecordWrite(input: {
290
290
  /** measurement-integrity FR-5/FR-6: rollout-log + price enrichment for a ledger row. Absent ⇒ zero
291
291
  * behavior change (NFR-1). */
292
292
  enrich?: LedgerEnrichInput;
293
+ /** experiment-instrument FR-2/A3 (ADR-001): when `true`, an AUTO ledger row that would be written
294
+ * `complete:false` is refused instead (exit 2, before any write) — a круг-B default candidate, opt-
295
+ * in today so nothing that already writes incomplete auto rows starts failing underfoot (NFR-1). */
296
+ strict?: boolean;
293
297
  }): RecordDecision {
294
298
  const { kind, payloadRaw, stage } = input;
295
299
  if (kind !== 'ledger' && kind !== 'training-pair') {
@@ -553,6 +557,57 @@ export function decideRecordWrite(input: {
553
557
  }
554
558
  }
555
559
 
560
+ // experiment-instrument FR-2/A3 (ADR-001): completeness of an AUTO ledger row. `minutes` is
561
+ // fill-only-null from a `wallSec` the payload carries (the workflow sandbox has a clock delta even
562
+ // when it has no wall clock of its own — FR-2's `wallSec` field, distinct from `minutesSincePrev`
563
+ // above, which needs a PREVIOUS row and a runId neither of which every auto row has). `tokens` is
564
+ // judged complete when it is a real number OR the row already NAMES why it is not (`tokensSource`,
565
+ // set above by the FR-5 rollout match, or supplied by the caller) — an unexplained non-number is the
566
+ // one shape that is actually incomplete. Gated on `auto` only: a MANUAL row never gains any of these
567
+ // three keys, so it stays byte-identical to before this feature (NFR-1).
568
+ if (kind === 'ledger' && stamped['auto'] === true) {
569
+ const incompleteReasons: string[] = [];
570
+ if (stamped['minutes'] === null || stamped['minutes'] === undefined) {
571
+ const wallSec = stamped['wallSec'];
572
+ if (typeof wallSec === 'number' && Number.isFinite(wallSec) && wallSec >= 0) {
573
+ stamped['minutes'] = Math.round((wallSec / 60) * 10) / 10;
574
+ stamped['minutesSource'] = 'wallSec';
575
+ } else {
576
+ incompleteReasons.push('minutes');
577
+ }
578
+ }
579
+ // r1-10 (Codex r1 HIGH #10): completeness requires a real, finite NUMBER of tokens. Naming why
580
+ // tokens are missing (`tokensSource:'unavailable'`, or any other provenance) is diagnostic, never
581
+ // a substitute for the number itself — ADR-001 says missing tokens makes the row incomplete, full
582
+ // stop. The old check (`tokensSource === undefined`) treated a NAMED absence as if it were data.
583
+ if (typeof stamped['tokens'] !== 'number' || !Number.isFinite(stamped['tokens'])) {
584
+ incompleteReasons.push('tokens');
585
+ }
586
+ // r1-11 (Codex r1 MEDIUM #11): a payload that ALREADY declared itself incomplete (its own
587
+ // `complete:false` + `incompleteReasons`, e.g. a workflow-side `'artifact'` reason this module
588
+ // knows nothing about) must never be overwritten back to `complete:true` just because THIS
589
+ // module's own minutes/tokens checks both passed — that erases a true fact and leaves a
590
+ // contradictory row (`complete:true` alongside a stale `incompleteReasons`). Preserve and MERGE.
591
+ const existingCompleteRaw = stamped['complete'];
592
+ const existingWasIncomplete = existingCompleteRaw === false;
593
+ const existingReasonsRaw = stamped['incompleteReasons'];
594
+ const existingReasons = Array.isArray(existingReasonsRaw)
595
+ ? existingReasonsRaw.filter((r): r is string => typeof r === 'string')
596
+ : [];
597
+ const mergedReasons = existingWasIncomplete
598
+ ? [...new Set([...existingReasons, ...incompleteReasons])]
599
+ : incompleteReasons;
600
+ const complete = mergedReasons.length === 0 && !existingWasIncomplete;
601
+ // A3: under `--strict`, incompleteness is a REFUSAL — before any write, the target untouched —
602
+ // rather than a loudly-marked write. Without `--strict` (the default today; круг-B may flip it),
603
+ // the row is still written, just honestly marked `complete:false` with its reasons.
604
+ if (input.strict === true && !complete) {
605
+ return refuse(`auto ledger row is incomplete (${mergedReasons.join(', ') || 'previously marked incomplete'}) — refused under --strict before any write`);
606
+ }
607
+ stamped['complete'] = complete;
608
+ if (!complete) stamped['incompleteReasons'] = mergedReasons.length > 0 ? mergedReasons : existingReasons;
609
+ }
610
+
556
611
  let line: string;
557
612
  try {
558
613
  line = JSON.stringify(stamped);
@@ -604,3 +659,54 @@ export function decideReadBack(appended: string, lastLineOnDisk: string | null):
604
659
  export function recordVerdictLine(kind: RecordKind, stage: string, d: RecordDecision): string {
605
660
  return `feature-adr record (${kind}/${stage}): ${d.verdict.toUpperCase()} — ${d.reason}`;
606
661
  }
662
+
663
+ /** experiment-instrument FR-1/FR-3 (ADR-001): what `round.ts`'s `readOpenRoundTaskId` returns — the
664
+ * single source `applyTaskId` fills from. Duplicated here rather than imported so this pure module
665
+ * never depends on `round.ts`'s own shape; the CLI is the one holding both and wiring them together.
666
+ * r1-1/r1-2 (Codex r1 #1/#2): extended with `'derived-legacy'` and `'unavailable'` to stay in
667
+ * lockstep with `round.ts`'s own `readOpenRoundTaskId` return type. */
668
+ export interface TaskIdLookup {
669
+ readonly taskId: string | null;
670
+ readonly source: 'open-round' | 'derived-legacy' | 'no-open-round' | 'ambiguous' | 'unavailable';
671
+ }
672
+
673
+ /**
674
+ * experiment-instrument FR-1/FR-3/A8 (ADR-001): propagate `taskId` onto a ledger/training-pair payload
675
+ * BEFORE it reaches {@link decideRecordWrite} — fill-only-null, never overwritten.
676
+ *
677
+ * - The payload already names a non-empty `taskId` string ⇒ it is authoritative. When it DISAGREES
678
+ * with the round's own current taskId, that disagreement is a real fact worth keeping — recorded as
679
+ * `taskIdConflict: {payload, round}` — never silently resolved either way (A8).
680
+ * - The payload's `taskId` key is absent, or explicitly `null`/`undefined` ⇒ filled from `lookup`,
681
+ * INCLUDING the honest `null` case: no open round (A4) or two of them (A5) still stamps `taskId:
682
+ * null` + `taskIdSource` naming why, rather than leaving the field silently absent — absence with a
683
+ * named reason beats absence with none.
684
+ * - r1-3 (Codex r1 HIGH #3): the payload's `taskId` key is PRESENT with a value that is neither a
685
+ * non-empty string nor null/undefined (a number, a boolean, an object, or a blank/whitespace-only
686
+ * string) ⇒ that is a present-but-INVALID value, a THIRD case distinct from both of the above. It
687
+ * used to be treated exactly like "absent" (`typeof !== 'string'` fell through to the fill branch),
688
+ * silently replacing the caller's own (malformed) value with the round's — violating both
689
+ * fill-only-null and "a present payload value always wins". Now: the row's own value is preserved
690
+ * UNTOUCHED (never replaced with a guess about what the caller meant), and the problem is named in
691
+ * `taskIdInvalid` so a reader can see the row was neither filled nor trusted blindly.
692
+ *
693
+ * Pure: no filesystem, no clock. The CALLER (the cli) is the one that read `.dz/rounds/` to build
694
+ * `lookup` in the first place.
695
+ */
696
+ export function applyTaskId(row: Record<string, unknown>, lookup: TaskIdLookup): Record<string, unknown> {
697
+ const hasTaskIdKey = Object.prototype.hasOwnProperty.call(row, 'taskId');
698
+ const rawPayloadTaskId = row['taskId'];
699
+ if (hasTaskIdKey && rawPayloadTaskId !== null && rawPayloadTaskId !== undefined) {
700
+ if (typeof rawPayloadTaskId === 'string' && rawPayloadTaskId.trim() !== '') {
701
+ const payloadTaskId = rawPayloadTaskId.trim();
702
+ if (lookup.taskId !== null && lookup.taskId !== payloadTaskId) {
703
+ return { ...row, taskIdConflict: { payload: payloadTaskId, round: lookup.taskId } };
704
+ }
705
+ return { ...row };
706
+ }
707
+ // r1-3: present but not a usable identity (non-string, or blank after trim) — refuse to replace
708
+ // it with a lookup guess; preserve it verbatim and name the problem.
709
+ return { ...row, taskIdInvalid: { value: rawPayloadTaskId, reason: 'taskId present but not a non-empty string' } };
710
+ }
711
+ return { ...row, taskId: lookup.taskId, taskIdSource: lookup.source };
712
+ }