create-principles-disciple 1.133.11 → 1.133.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (90) hide show
  1. package/console/dist/ui/i18n/en.json +13 -1
  2. package/console/dist/ui/i18n/zh-CN.json +13 -1
  3. package/console/dist/ui/pages/focus/OwnerDecisionCard.js +1 -1
  4. package/console/dist/ui/utils/validators.d.ts +9 -0
  5. package/console/dist/ui/utils/validators.js +31 -0
  6. package/console/dist/web/assets/app.css +3 -0
  7. package/console/dist/web/assets/app.js +60 -3
  8. package/console/package.json +1 -1
  9. package/core/dist/runtime-v2/internalization/__tests__/artifact-summary.property.test.js +1 -1
  10. package/core/dist/runtime-v2/internalization/__tests__/artifact-summary.property.test.js.map +1 -1
  11. package/core/dist/runtime-v2/internalization/__tests__/artificer-prompt-builder.test.js +6 -4
  12. package/core/dist/runtime-v2/internalization/__tests__/artificer-prompt-builder.test.js.map +1 -1
  13. package/core/dist/runtime-v2/internalization/__tests__/evaluator-out-of-scope-governance.test.d.ts +2 -0
  14. package/core/dist/runtime-v2/internalization/__tests__/evaluator-out-of-scope-governance.test.d.ts.map +1 -0
  15. package/core/dist/runtime-v2/internalization/__tests__/evaluator-out-of-scope-governance.test.js +307 -0
  16. package/core/dist/runtime-v2/internalization/__tests__/evaluator-out-of-scope-governance.test.js.map +1 -0
  17. package/core/dist/runtime-v2/internalization/__tests__/evaluator-prompt-builder.test.js +2 -2
  18. package/core/dist/runtime-v2/internalization/__tests__/evaluator-prompt-builder.test.js.map +1 -1
  19. package/core/dist/runtime-v2/internalization/__tests__/evolution-alignment-contract.test.d.ts +18 -0
  20. package/core/dist/runtime-v2/internalization/__tests__/evolution-alignment-contract.test.d.ts.map +1 -0
  21. package/core/dist/runtime-v2/internalization/__tests__/evolution-alignment-contract.test.js +457 -0
  22. package/core/dist/runtime-v2/internalization/__tests__/evolution-alignment-contract.test.js.map +1 -0
  23. package/core/dist/runtime-v2/internalization/__tests__/owner-decision-review.test.js +88 -0
  24. package/core/dist/runtime-v2/internalization/__tests__/owner-decision-review.test.js.map +1 -1
  25. package/core/dist/runtime-v2/internalization/artifact-summary.d.ts.map +1 -1
  26. package/core/dist/runtime-v2/internalization/artifact-summary.js +8 -0
  27. package/core/dist/runtime-v2/internalization/artifact-summary.js.map +1 -1
  28. package/core/dist/runtime-v2/internalization/artificer-prompt-builder.d.ts +37 -2
  29. package/core/dist/runtime-v2/internalization/artificer-prompt-builder.d.ts.map +1 -1
  30. package/core/dist/runtime-v2/internalization/artificer-prompt-builder.js +31 -1
  31. package/core/dist/runtime-v2/internalization/artificer-prompt-builder.js.map +1 -1
  32. package/core/dist/runtime-v2/internalization/artificer-runner.d.ts +18 -1
  33. package/core/dist/runtime-v2/internalization/artificer-runner.d.ts.map +1 -1
  34. package/core/dist/runtime-v2/internalization/artificer-runner.js +60 -2
  35. package/core/dist/runtime-v2/internalization/artificer-runner.js.map +1 -1
  36. package/core/dist/runtime-v2/internalization/context-manifests.d.ts.map +1 -1
  37. package/core/dist/runtime-v2/internalization/context-manifests.js +35 -2
  38. package/core/dist/runtime-v2/internalization/context-manifests.js.map +1 -1
  39. package/core/dist/runtime-v2/internalization/evaluator-prompt-builder.d.ts +15 -2
  40. package/core/dist/runtime-v2/internalization/evaluator-prompt-builder.d.ts.map +1 -1
  41. package/core/dist/runtime-v2/internalization/evaluator-prompt-builder.js +10 -2
  42. package/core/dist/runtime-v2/internalization/evaluator-prompt-builder.js.map +1 -1
  43. package/core/dist/runtime-v2/internalization/evaluator-runner.d.ts +46 -0
  44. package/core/dist/runtime-v2/internalization/evaluator-runner.d.ts.map +1 -1
  45. package/core/dist/runtime-v2/internalization/evaluator-runner.js +352 -21
  46. package/core/dist/runtime-v2/internalization/evaluator-runner.js.map +1 -1
  47. package/core/dist/runtime-v2/internalization/index.d.ts +4 -2
  48. package/core/dist/runtime-v2/internalization/index.d.ts.map +1 -1
  49. package/core/dist/runtime-v2/internalization/index.js +2 -1
  50. package/core/dist/runtime-v2/internalization/index.js.map +1 -1
  51. package/core/dist/runtime-v2/internalization/intent-contract.d.ts +59 -0
  52. package/core/dist/runtime-v2/internalization/intent-contract.d.ts.map +1 -0
  53. package/core/dist/runtime-v2/internalization/intent-contract.js +61 -0
  54. package/core/dist/runtime-v2/internalization/intent-contract.js.map +1 -0
  55. package/core/dist/runtime-v2/internalization/owner-decision-review.d.ts +28 -0
  56. package/core/dist/runtime-v2/internalization/owner-decision-review.d.ts.map +1 -1
  57. package/core/dist/runtime-v2/internalization/owner-decision-review.js +76 -0
  58. package/core/dist/runtime-v2/internalization/owner-decision-review.js.map +1 -1
  59. package/core/dist/runtime-v2/internalization/owner-review.d.ts +8 -0
  60. package/core/dist/runtime-v2/internalization/owner-review.d.ts.map +1 -1
  61. package/core/dist/runtime-v2/internalization/owner-review.js +10 -0
  62. package/core/dist/runtime-v2/internalization/owner-review.js.map +1 -1
  63. package/core/dist/runtime-v2/internalization/pitask-metadata.d.ts +76 -0
  64. package/core/dist/runtime-v2/internalization/pitask-metadata.d.ts.map +1 -1
  65. package/core/dist/runtime-v2/internalization/pitask-metadata.js +116 -3
  66. package/core/dist/runtime-v2/internalization/pitask-metadata.js.map +1 -1
  67. package/core/dist/runtime-v2/internalization/rule-reliability-validation.d.ts +141 -1
  68. package/core/dist/runtime-v2/internalization/rule-reliability-validation.d.ts.map +1 -1
  69. package/core/dist/runtime-v2/internalization/rule-reliability-validation.js +140 -0
  70. package/core/dist/runtime-v2/internalization/rule-reliability-validation.js.map +1 -1
  71. package/core/dist/runtime-v2/internalization/scribe-output.d.ts +11 -0
  72. package/core/dist/runtime-v2/internalization/scribe-output.d.ts.map +1 -1
  73. package/core/dist/runtime-v2/internalization/scribe-output.js +13 -0
  74. package/core/dist/runtime-v2/internalization/scribe-output.js.map +1 -1
  75. package/core/dist/runtime-v2/internalization/scribe-prompt-builder.d.ts +8 -1
  76. package/core/dist/runtime-v2/internalization/scribe-prompt-builder.d.ts.map +1 -1
  77. package/core/dist/runtime-v2/internalization/scribe-prompt-builder.js +25 -1
  78. package/core/dist/runtime-v2/internalization/scribe-prompt-builder.js.map +1 -1
  79. package/core/dist/runtime-v2/runner/__tests__/base-peer-runner-failure-details.test.js +16 -1
  80. package/core/dist/runtime-v2/runner/__tests__/base-peer-runner-failure-details.test.js.map +1 -1
  81. package/core/dist/runtime-v2/runner/base-peer-runner.d.ts +9 -1
  82. package/core/dist/runtime-v2/runner/base-peer-runner.d.ts.map +1 -1
  83. package/core/dist/runtime-v2/runner/base-peer-runner.js +32 -3
  84. package/core/dist/runtime-v2/runner/base-peer-runner.js.map +1 -1
  85. package/core/package.json +1 -1
  86. package/package.json +1 -1
  87. package/plugin/dist/bundle.js +764 -724
  88. package/plugin/dist/governance-audit.js +76 -76
  89. package/plugin/dist/rulehost-evidence.js +116 -116
  90. package/release-manager/package.json +1 -1
@@ -9,9 +9,11 @@ import { EvaluatorPromptBuilder, deriveRequirementLedger } from './evaluator-pro
9
9
  import { reconcileLineageEcho } from './peer-runner-contracts.js';
10
10
  import { BasePeerRunner } from '../runner/base-peer-runner.js';
11
11
  import { EVALUATOR_STAGE1_MANIFEST, EVALUATOR_STAGE2_MANIFEST } from './context-manifests.js';
12
+ import { extractIntentContract } from './intent-contract.js';
12
13
  import { evaluateFlaggedCriteria, isForcedStage2 } from './progressive-evaluator.js';
13
14
  // PRI-426: single-round adversarial sandbox replay in succeedTask.
14
15
  import { evaluateRefinerRuleHostGate } from './refiner-rulehost-gate.js';
16
+ import { attributionFromLayer, partitionV2OutOfScopeFailures, resolveRequiresContextVersionFromArtifact } from './rule-reliability-validation.js';
15
17
  import { adversarialCasesToGoldenTrace } from './adversarial-case.js';
16
18
  import { buildGoldenTraceFromArtificer } from '../golden-trace.js';
17
19
  // PRI-485 Phase 6: auto-generate 5 v2 adversarial cases (unavailable/truncation/
@@ -595,6 +597,12 @@ export class EvaluatorRunner extends BasePeerRunner {
595
597
  sourceArtificerArtifactId: context.sourceArtificerArtifactId ?? '',
596
598
  previousEvaluation: context.previousEvaluation,
597
599
  hostToolCatalog: this.hostToolCatalog ?? undefined,
600
+ // PRI-703 Phase 1: the scribe artifact's Owner-intent contract is the
601
+ // primary intentConsistency anchor. Extracted from the FULL parsed
602
+ // scribe artifact (before any manifest narrowing — the contract is
603
+ // injected as its own block). Absent on pre-contract artifacts →
604
+ // undefined → prompt unchanged (backward compatible).
605
+ intentContract: extractIntentContract(parsedScribeArtifact) ?? undefined,
598
606
  });
599
607
  return { message, stage2Evidence: resolutionOutcome };
600
608
  }
@@ -926,13 +934,30 @@ export class EvaluatorRunner extends BasePeerRunner {
926
934
  });
927
935
  }
928
936
  }
937
+ // ── Round-2 R2 (Owner 指令 2026-09-08): 出界处置先于 intent 落库派生 ──
938
+ // deriveGovernanceEffect 只读 durable 事实 (重放已 updateRunOutput 持久化
939
+ // 的 adversarialResult + durable artificer 工件 + durable repair 轮次),
940
+ // 不依赖任何瞬时变量。派生结果随 intent 持久化 —— fresh 与 resume 由同
941
+ // 一 durable 判据得到同一处置,重启不改变已选定的治理动作 (R2 复现的
942
+ // 治理漂移断点)。
943
+ const sourceArtificerArtifactIdForGovernance = context.sourceArtificerArtifactId
944
+ ?? finalOutput.sourceArtificerArtifactId
945
+ ?? null;
946
+ const governanceEffect = await this.deriveGovernanceEffect(taskId, {
947
+ output: finalOutput,
948
+ sourceArtificerArtifactId: sourceArtificerArtifactIdForGovernance,
949
+ diagnosticReplayEvidence,
950
+ });
929
951
  // ── P0 (verdict drift): verdict + completion intent 原子落库 ──
930
952
  // 必须先于一切治理 side effect (validate bearer / seed repair / rule
931
953
  // assembly):side effect 已发生而 intent 未落 = crash 后重跑会重新问
932
954
  // LLM,新 verdict 与已发生副作用形成治理矛盾 (repair drift /
933
955
  // validation drift / validated-rule drift)。同 epoch crash/retry 重跑经
934
956
  // maybeResumePendingIntent resume,不重问。
935
- await this.recordCompletionOrThrow(taskId, runId, finalOutput.evaluation.decision);
957
+ await this.recordCompletionOrThrow(taskId, runId, {
958
+ decision: finalOutput.evaluation.decision,
959
+ governanceEffect,
960
+ });
936
961
  const ruleAssemblyInput = {
937
962
  artificerArtifact: context.artificerArtifact ?? null,
938
963
  sourceArtificerArtifactId: context.sourceArtificerArtifactId ?? null,
@@ -1056,10 +1081,14 @@ export class EvaluatorRunner extends BasePeerRunner {
1056
1081
  if (decision === 'needs_revision' && this.isRepairLoopEnabled()) {
1057
1082
  const repairOutcome = await this.maybeSeedArtificerRepair(taskId, { runId, output: finalOutput, sourceArtificerArtifactId, diagnosticReplayEvidence });
1058
1083
  if (repairOutcome.kind === 'max_iterations_reached') {
1059
- // PRI-629: budget 耗尽(decision-capable)与 seed 失败(recovery)拆分原因码
1084
+ // PRI-629: budget 耗尽(decision-capable)与 seed 失败(recovery)拆分原因码。
1085
+ // PRI-703 Phase 2: test_out_of_scope 是 FAILED_TEST 归因 — v2-context
1086
+ // case 对 v1 规则的通道设计限制,decision-capable(Owner 裁决是唯一出口)。
1060
1087
  const reasonCode = repairOutcome.detail === 'budget_exhausted'
1061
1088
  ? HUMAN_REVIEW_REASON.evaluatorRepairBudgetExhausted
1062
- : HUMAN_REVIEW_REASON.evaluatorRepairSeedFailed;
1089
+ : repairOutcome.detail === 'test_out_of_scope'
1090
+ ? HUMAN_REVIEW_REASON.evaluatorTestOutOfScope
1091
+ : HUMAN_REVIEW_REASON.evaluatorRepairSeedFailed;
1063
1092
  // Fail loud (rc-9, EP-03, ERR-002): mark the task needs_human_review
1064
1093
  // so it does NOT stay in 'leased' state (which would cause the lease
1065
1094
  // to expire and the evaluator to re-run the same verdict infinitely).
@@ -1096,6 +1125,70 @@ export class EvaluatorRunner extends BasePeerRunner {
1096
1125
  // transition decision 依据 durable runnerDecision fail-closed。
1097
1126
  return { kind: 'completed', ruleArtifactId: null };
1098
1127
  }
1128
+ /**
1129
+ * PRI-703 Phase 2: extract failed replay caseIds from the evaluator output.
1130
+ * Trust-boundary (rc-1/rc-4): adversarialResult.failedCases is untrusted
1131
+ * artifact content — validate each element's caseId shape; malformed entries
1132
+ * are skipped (they cannot inform scoping decisions).
1133
+ */
1134
+ static extractFailedCases(output) {
1135
+ // rc-1/rc-2 (ERR-001): adversarialResult.failedCases is untrusted
1136
+ // artifact content — narrow via the class's isRecord/Array guards, no `as`.
1137
+ // Round-2 R1: extract the sandbox-carried expectedDecision per case for
1138
+ // the oracle-consistency check in partitionV2OutOfScopeFailures. The
1139
+ // stored value is only ever COMPARED against the compile-time template
1140
+ // oracle — it is never the classification source of truth.
1141
+ if (!EvaluatorRunner.isRecord(output))
1142
+ return [];
1143
+ const { adversarialResult } = output;
1144
+ if (!EvaluatorRunner.isRecord(adversarialResult))
1145
+ return [];
1146
+ const { failedCases } = adversarialResult;
1147
+ if (!Array.isArray(failedCases))
1148
+ return [];
1149
+ const extracted = [];
1150
+ for (const entry of failedCases) {
1151
+ if (!EvaluatorRunner.isRecord(entry))
1152
+ continue;
1153
+ const { caseId } = entry;
1154
+ if (typeof caseId !== 'string' || caseId.trim() === '')
1155
+ continue;
1156
+ const { expectedDecision } = entry;
1157
+ extracted.push(typeof expectedDecision === 'string' && expectedDecision.trim() !== ''
1158
+ ? { caseId, expectedDecision }
1159
+ : { caseId });
1160
+ }
1161
+ return extracted;
1162
+ }
1163
+ /**
1164
+ * PRI-703 Phase 2: resolve the rule's requiresContextVersion from the
1165
+ * DURABLE artificer artifact (never from the LLM-forgible evaluator copy).
1166
+ * Three-way result (see resolveRequiresContextVersionFromArtifact):
1167
+ * 2 / undefined → resolved (v2 / v1) — the scope partition MUST run;
1168
+ * null → unresolvable (attribution stays inconclusive — fail-open to the
1169
+ * existing repair path, never blocks the loop on a read error).
1170
+ *
1171
+ * 评审 P1 修正:key-absent on a PARSED artifact is deterministically v1
1172
+ * (artificer schema only ever writes literal 2) and must reach the
1173
+ * partition — collapsing it into null made the out-of-scope routing
1174
+ * unreachable in production for exactly its target population (v1 rules).
1175
+ */
1176
+ async resolveRequiresContextVersion(artificerArtifactId, taskId) {
1177
+ try {
1178
+ const artifact = await this.artifactStore.getArtifactById(artificerArtifactId);
1179
+ if (!artifact)
1180
+ return null;
1181
+ return resolveRequiresContextVersionFromArtifact(artifact.contentJson);
1182
+ }
1183
+ catch (err) {
1184
+ this.emitEvent('attribution_scope_resolve_failed', taskId, {
1185
+ artificerArtifactId,
1186
+ reason: err instanceof Error ? err.message : String(err),
1187
+ nextAction: 'verify_artifact_store_read_for_failure_attribution',
1188
+ });
1189
+ return null;
1190
+ }
1191
+ }
1099
1192
  /**
1100
1193
  * PRI-509: Resolve the prior repair iteration by reading the dependency
1101
1194
  * artificer task's repairPayload (rc-7: written at task creation, never
@@ -1193,7 +1286,8 @@ export class EvaluatorRunner extends BasePeerRunner {
1193
1286
  * 证明 output 已 durable (updateRunOutput 在 succeedTask 最前)。
1194
1287
  * 同 epoch crash/retry 重跑经 maybeResumePendingIntent resume,不重问 LLM。
1195
1288
  */
1196
- async recordCompletionOrThrow(taskId, runId, decision) {
1289
+ async recordCompletionOrThrow(taskId, runId, decisionAndEffect) {
1290
+ const { decision, governanceEffect } = decisionAndEffect;
1197
1291
  try {
1198
1292
  const raw = await this.stateManager.getTask(taskId);
1199
1293
  if (!raw)
@@ -1210,6 +1304,12 @@ export class EvaluatorRunner extends BasePeerRunner {
1210
1304
  sourceRunId: runId,
1211
1305
  revisionEpoch: piTask.revisionCount ?? 0,
1212
1306
  status: 'pending',
1307
+ ...(governanceEffect?.selectedEffect !== undefined
1308
+ ? { selectedEffect: governanceEffect.selectedEffect }
1309
+ : {}),
1310
+ ...(governanceEffect?.effectReasonCode !== undefined
1311
+ ? { effectReasonCode: governanceEffect.effectReasonCode }
1312
+ : {}),
1213
1313
  },
1214
1314
  });
1215
1315
  await this.stateManager.updateTaskDiagnosticJson(taskId, createPITaskDiagnosticJson(merged));
@@ -1287,6 +1387,49 @@ export class EvaluatorRunner extends BasePeerRunner {
1287
1387
  decision: intent.decision,
1288
1388
  sourceRunId: intent.sourceRunId,
1289
1389
  });
1390
+ // ── Round-2 R2 (Owner 指令 2026-09-08): 选定 NHR 效果的恢复直通 ──
1391
+ // fresh 路径在 intent 落库前已把 deriveGovernanceEffect 的结论
1392
+ // (selectedEffect='needs_human_review' + effectReasonCode) 持久化;恢复
1393
+ // 执行读同一 durable intent,直接重放 NHR 效果 — 禁止重问 LLM、禁止
1394
+ // 再 seed repair、禁止重新判责 (R2 复现的治理漂移断点)。照抄
1395
+ // rollout-reviewer-runner 的 intent.effect resume 模板,零新状态源。
1396
+ if ((intent.selectedEffect ?? intent.effect) === 'needs_human_review') {
1397
+ const output2 = await this.recoverIntentOutput(taskId, intent.sourceRunId, intent.decision);
1398
+ const artifactId2 = `pi-art-${taskId}-${intent.sourceRunId}`;
1399
+ const reasonCode = intent.effectReasonCode ?? HUMAN_REVIEW_REASON.evaluatorRepairSeedFailed;
1400
+ if (reasonCode === 'evaluator_test_out_of_scope') {
1401
+ this.emitEvent('repair_loop_test_out_of_scope', taskId, {
1402
+ runId: intent.sourceRunId,
1403
+ attribution: attributionFromLayer('test'),
1404
+ source: 'resume_from_persisted_intent',
1405
+ nextAction: 'owner_decision_required_channel_upgrade_or_principle_revision',
1406
+ });
1407
+ }
1408
+ await this.markNeedsHumanReviewOrThrow(taskId, {
1409
+ runId: intent.sourceRunId,
1410
+ reasonCode,
1411
+ sourceArtifactId: artifactId2,
1412
+ });
1413
+ await this.markCompletionIntentAppliedOrThrow(taskId);
1414
+ this.emitEvent('task_needs_human_review', taskId, {
1415
+ attemptCount: leasedTask.attemptCount,
1416
+ resultRef: `${this.config.resultRefPrefix}://${intent.sourceRunId}`,
1417
+ evaluationDecision: output2.evaluation?.decision,
1418
+ evaluationScore: output2.evaluation?.score,
1419
+ ruleArtifactId: null,
1420
+ reason: `repair_loop_${reasonCode}`,
1421
+ });
1422
+ return {
1423
+ status: 'succeeded',
1424
+ taskId,
1425
+ runId: intent.sourceRunId,
1426
+ artifactId: artifactId2,
1427
+ resultRef: `${this.config.resultRefPrefix}://${intent.sourceRunId}`,
1428
+ contextHash: `resume-${intent.sourceRunId}`,
1429
+ output: output2,
1430
+ attemptCount: leasedTask.attemptCount,
1431
+ };
1432
+ }
1290
1433
  const output = await this.recoverIntentOutput(taskId, intent.sourceRunId, intent.decision);
1291
1434
  const artifactId = `pi-art-${taskId}-${intent.sourceRunId}`;
1292
1435
  const contextHash = `resume-${intent.sourceRunId}`;
@@ -1549,9 +1692,164 @@ export class EvaluatorRunner extends BasePeerRunner {
1549
1692
  * are typed as readonly string[] on EvaluatorEvaluation, so element-level
1550
1693
  * re-validation is not required here (rc-4 N/A — not unknown at this point).
1551
1694
  */
1695
+ /**
1696
+ * Round-2 R2 (Owner 指令 2026-09-08): 从 durable 证据派生本 verdict 的治理
1697
+ * 效果。fresh 与 resume 调用同一派生 —— 输入全部是 durable 事实:
1698
+ * - output: intent 落库前已由 executeDeterministicReplay 持久化的
1699
+ * adversarialResult(resume 时 recoverIntentOutput 从 run.outputPayload
1700
+ * 恢复同一对象);
1701
+ * - sourceArtificerArtifactId → durable 工件的 requiresContextVersion;
1702
+ * - priorRepairIteration → durable dependency repairPayload。
1703
+ * record 只持久化本函数的结论,不做判断 (指令要求: 禁止把复杂判断逻辑
1704
+ * 塞进 recordCompletionOrThrow)。返回 undefined = governance_transition
1705
+ * (正常效果由 applyEvaluatorDecisionEffects 执行,无需特判)。
1706
+ *
1707
+ * fresh 路径的 diagnosticReplayEvidence 仅用于验证"重放确实执行过"这一
1708
+ * 控制流事实 (P1 provenance); resume 路径不传时, 以 durable adversarialResult
1709
+ * 的形态 (passed === false 且 failedCases 非空) 为等价判据 —— intent 存在
1710
+ * 本身即证明 fresh 侧重放已执行并持久化。
1711
+ */
1712
+ /**
1713
+ * Round-2 R2: 读取当前 durable completion intent (pending 或 applied 均可读 —
1714
+ * applied 意味着效果已 materialize, 读侧只用于一致性核对, 不会改变路由)。
1715
+ * 解析失败/缺 intent 返回 null — 出界判定回退到 fresh 快路径判定
1716
+ * (fail-closed, 绝不因读不到 intent 而静默 seed)。
1717
+ */
1718
+ async readPendingOrAppliedCompletionIntent(taskId) {
1719
+ try {
1720
+ const raw = await this.stateManager.getTask(taskId);
1721
+ if (!raw)
1722
+ return null;
1723
+ const piTask = hydratePITaskRecord(raw);
1724
+ return piTask?.completionIntent ?? null;
1725
+ }
1726
+ catch (err) {
1727
+ this.emitEvent('completion_intent_read_failed', taskId, {
1728
+ errorMessage: err instanceof Error ? err.message : String(err),
1729
+ nextAction: 'out_of_scope_disposition_falls_back_to_fresh_path_derivation',
1730
+ });
1731
+ return null;
1732
+ }
1733
+ }
1734
+ async deriveGovernanceEffect(evaluatorTaskId, ctx) {
1735
+ if (ctx.output.evaluation?.decision !== 'needs_revision')
1736
+ return undefined;
1737
+ if (!this.isRepairLoopEnabled())
1738
+ return undefined;
1739
+ // P1 provenance: fresh 要求 ran:true;resume 无 live 重放时等价判据为
1740
+ // durable adversarialResult 形态 (passed:false + failedCases>0)。
1741
+ const freshReplayFailed = ctx.diagnosticReplayEvidence?.ran === true && ctx.diagnosticReplayEvidence.passed === false;
1742
+ let durableReplayFailed = false;
1743
+ if (EvaluatorRunner.isRecord(ctx.output)) {
1744
+ const { adversarialResult } = ctx.output;
1745
+ if (EvaluatorRunner.isRecord(adversarialResult)) {
1746
+ const { passed } = adversarialResult;
1747
+ const { failedCases } = adversarialResult;
1748
+ durableReplayFailed = passed === false && Array.isArray(failedCases) && failedCases.length > 0;
1749
+ }
1750
+ }
1751
+ if (!freshReplayFailed && !durableReplayFailed)
1752
+ return undefined;
1753
+ const failedCases = EvaluatorRunner.extractFailedCases(ctx.output);
1754
+ if (failedCases.length === 0)
1755
+ return undefined;
1756
+ const requiresContextVersion = ctx.sourceArtificerArtifactId
1757
+ ? await this.resolveRequiresContextVersion(ctx.sourceArtificerArtifactId, evaluatorTaskId)
1758
+ : null;
1759
+ if (requiresContextVersion === null)
1760
+ return undefined;
1761
+ const partition = partitionV2OutOfScopeFailures({
1762
+ requiresContextVersion: requiresContextVersion === 2 ? 2 : undefined,
1763
+ failedCases,
1764
+ });
1765
+ if (partition.isPureOutOfScope) {
1766
+ this.emitEvent('governance_effect_out_of_scope_selected', evaluatorTaskId, {
1767
+ outOfScopeCaseIds: [...partition.outOfScope],
1768
+ nextAction: 'selected_effect_needs_human_review_evaluator_test_out_of_scope',
1769
+ });
1770
+ return {
1771
+ selectedEffect: 'needs_human_review',
1772
+ effectReasonCode: 'evaluator_test_out_of_scope',
1773
+ };
1774
+ }
1775
+ return undefined;
1776
+ }
1552
1777
  async maybeSeedArtificerRepair(evaluatorTaskId, ctx) {
1553
1778
  const { runId: evaluatorRunId, output } = ctx;
1554
1779
  const priorRepairIteration = await this.resolvePriorRepairIteration(evaluatorTaskId);
1780
+ // ── PRI-703 Phase 2 (Owner decision 2026-09-07): deterministic failure
1781
+ // attribution BEFORE seeding a repair round. "测试失败 → 继续改 Rule" is
1782
+ // the wrong default; a needs_revision whose replay failures are ALL
1783
+ // v2-context template cases judging a v1 (action-only) rule is a
1784
+ // TEST-SCOPE problem (FAILED_TEST attribution): the v1 channel
1785
+ // structurally cannot express context semantics, so no rule repair can
1786
+ // ever satisfy those cases (Episode 001's 18/18 death-loop). Route to
1787
+ // owner review with the channel-limitation signal instead of seeding a
1788
+ // doomed repair round.
1789
+ // Provenance: adversarialResult on the output here was written by
1790
+ // executeDeterministicReplay in THIS invocation (diagnosticReplayEvidence
1791
+ // ran:true) or absent — never LLM-forged (the pre-replay LLM copy was
1792
+ // replaced before the effects phase).
1793
+ //
1794
+ // 混合场景 (in-scope ⊕ out-of-scope 并存): 修复任务照常 seed (有真实
1795
+ // Rule 缺陷可修),但归因上下文随 RepairPayload 流动,让修复 LLM 知道
1796
+ // 哪些 case 是通道限制 (不得尝试满足) — "哪里失败/为什么/改哪里" 三问
1797
+ // 在载荷层可回答 (PRI-705)。
1798
+ let repairAttribution;
1799
+ // Round-2 R2: 出界处置以持久化的 completion intent 为权威。fresh 路径在
1800
+ // recordCompletionOrThrow 之前已由 deriveGovernanceEffect 写入
1801
+ // selectedEffect='needs_human_review' —— 同一 durable 依据 (R1 oracle 校验
1802
+ // 的 partition) 派生; resume 路径经 recoverIntentOutput 后重读本 intent,
1803
+ // 无需重算也无需任何瞬时变量。瞬时 evidence 门只作 fresh 快路径与
1804
+ // 派生时的 provenance 校验, 不再是出界判定的唯一载体 (R2 漂移断点)。
1805
+ {
1806
+ const intent = await this.readPendingOrAppliedCompletionIntent(evaluatorTaskId);
1807
+ // Round-3: hydrate 将 selectedEffect 归一化为 effect —— 读侧必须两键
1808
+ // 同读,否则 fresh 分支永不命中、静默回退瞬时 evidence (ERR-024 键不对称)。
1809
+ if ((intent?.selectedEffect ?? intent?.effect) === 'needs_human_review'
1810
+ && intent?.effectReasonCode === 'evaluator_test_out_of_scope') {
1811
+ this.emitEvent('repair_loop_test_out_of_scope', evaluatorTaskId, {
1812
+ runId: evaluatorRunId,
1813
+ attribution: attributionFromLayer('test'),
1814
+ source: 'persisted_completion_intent',
1815
+ nextAction: 'owner_decision_required_channel_upgrade_or_principle_revision',
1816
+ });
1817
+ return { kind: 'max_iterations_reached', detail: 'test_out_of_scope' };
1818
+ }
1819
+ }
1820
+ if (ctx.diagnosticReplayEvidence?.ran === true && ctx.diagnosticReplayEvidence.passed === false) {
1821
+ const failedCases = EvaluatorRunner.extractFailedCases(output);
1822
+ if (failedCases.length > 0) {
1823
+ const sourceArtificerArtifactIdForScope = ctx.sourceArtificerArtifactId ?? output.sourceArtificerArtifactId;
1824
+ const requiresContextVersion = sourceArtificerArtifactIdForScope
1825
+ ? await this.resolveRequiresContextVersion(sourceArtificerArtifactIdForScope, evaluatorTaskId)
1826
+ : null;
1827
+ if (requiresContextVersion !== null) {
1828
+ const partition = partitionV2OutOfScopeFailures({
1829
+ requiresContextVersion: requiresContextVersion === 2 ? 2 : undefined,
1830
+ failedCases,
1831
+ });
1832
+ if (partition.isPureOutOfScope) {
1833
+ this.emitEvent('repair_loop_test_out_of_scope', evaluatorTaskId, {
1834
+ runId: evaluatorRunId,
1835
+ attribution: attributionFromLayer('test'),
1836
+ outOfScopeCaseIds: [...partition.outOfScope],
1837
+ failedCaseCount: ctx.diagnosticReplayEvidence.failedCaseCount,
1838
+ reason: 'all_replay_failures_are_v2_context_cases_judging_a_v1_rule',
1839
+ nextAction: 'owner_decision_required_channel_upgrade_or_principle_revision',
1840
+ });
1841
+ return { kind: 'max_iterations_reached', detail: 'test_out_of_scope' };
1842
+ }
1843
+ if (partition.outOfScope.length > 0) {
1844
+ repairAttribution = {
1845
+ attribution: attributionFromLayer('rule'),
1846
+ outOfScopeCaseIds: [...partition.outOfScope],
1847
+ reason: `replay failures mix real rule defects (in-scope cases: ${partition.inScope.join(', ')}) with test-scope v2-context cases (${partition.outOfScope.join(', ')}) that the v1 action-only channel structurally cannot express — do NOT attempt to satisfy the out-of-scope cases`,
1848
+ };
1849
+ }
1850
+ }
1851
+ }
1852
+ }
1555
1853
  // ── Slice 5: max iterations (2) reached → fail loud ──
1556
1854
  if (priorRepairIteration >= 2) {
1557
1855
  // Task state update (→ needs_human_review) is handled by the caller
@@ -1583,6 +1881,10 @@ export class EvaluatorRunner extends BasePeerRunner {
1583
1881
  //(来自 succeedTask 的 executeDeterministicReplay 调用),绝不从 LLM
1584
1882
  // 可伪造的 output.adversarialResult 反推。缺失 = 本次未执行确定性重放
1585
1883
  //(skip/error/无 live 路径),repairPayload 不包含 diagnosticReplay。
1884
+ //
1885
+ // PRI-705 / PRI-703 Phase 2: 混合场景 (非纯 out-of-scope) 下,归因与
1886
+ // out-of-scope case 清单随载荷流动 — 修复 LLM 明确知道哪些失败属于
1887
+ // 通道设计限制(不得尝试满足),哪些是真实 Rule 缺陷(必须修复)。
1586
1888
  const repairPayload = {
1587
1889
  requiredChanges: [...output.evaluation.requiredChanges],
1588
1890
  concerns: [...output.evaluation.concerns],
@@ -1591,6 +1893,7 @@ export class EvaluatorRunner extends BasePeerRunner {
1591
1893
  sourceArtificerArtifactId,
1592
1894
  sourceEvaluatorTaskId: evaluatorTaskId,
1593
1895
  ...(ctx.diagnosticReplayEvidence ? { diagnosticReplay: ctx.diagnosticReplayEvidence } : {}),
1896
+ ...(repairAttribution !== undefined ? { failureAttribution: repairAttribution } : {}),
1594
1897
  };
1595
1898
  // Resolve the dependency artificer task to inherit dependencyTaskIds
1596
1899
  // (so the repair task points to the same scribe task, preserving lineage).
@@ -2244,36 +2547,43 @@ export class EvaluatorRunner extends BasePeerRunner {
2244
2547
  const validatedAffectedTools = Array.isArray(affectedTools)
2245
2548
  ? affectedTools.filter((t) => typeof t === 'string')
2246
2549
  : [];
2247
- const ruleContent = {
2248
- implementationCode,
2249
- goldenTrace: traceBuild.trace,
2250
- goldenTraceCases,
2251
- affectedTools: validatedAffectedTools,
2252
- // adversarialResult.passed === true is the precondition for this method;
2253
- // the gate decision is therefore accepted_shadow.
2254
- ruleHostGateDecision: 'accepted_shadow',
2255
- sourceArtificerArtifactId: assemblyInput.sourceArtificerArtifactId ?? output.sourceArtificerArtifactId,
2256
- adversarialResult: output.adversarialResult,
2257
- ...(requiresContextVersion === 2 ? { requiresContextVersion } : {}),
2258
- // PRI-490: preserve evidenceRefs from Artificer artifact into rule artifact.
2259
- // Only include when the array is valid (non-empty strings) — v1 rules may omit.
2260
- ...(requiresContextVersion === 2 && Array.isArray(evidenceRefs) && evidenceRefs.every((e) => typeof e === 'string' && e.trim() !== '')
2261
- ? { evidenceRefs: evidenceRefs }
2262
- : {}),
2263
- };
2264
2550
  // P1 #7 (cross-package acceptance test discovery): resolve the scribe
2265
2551
  // principle artifact and carry forward its principle ID as
2266
2552
  // sourcePrincipleId on the rule artifact. Without this, extractPrincipleId()
2267
2553
  // in the activation dispatcher returns null for rule artifacts, causing
2268
2554
  // activateArtifact() to fail with 'invalid_artifact'/'no_principle_id'.
2269
2555
  // The rule artifact must carry lineage to the principle it enforces.
2556
+ //
2557
+ // PRI-703 Phase 1: the same resolution deterministically forwards the
2558
+ // principle's intent contract onto the rule artifact (single author of the
2559
+ // rule's intent anchor = the scribe principle; the rule echoes it, it
2560
+ // never re-derives it — no second source of truth). Resolved BEFORE the
2561
+ // ruleContent build so the contract rides in the same contentJson write.
2270
2562
  let resolvedSourcePrincipleId;
2563
+ let forwardedIntentContract;
2271
2564
  try {
2272
2565
  const principleBearerId = await this.resolvePrincipleBearerArtifact(output, taskId);
2273
2566
  if (principleBearerId) {
2274
2567
  const principleArtifact = await this.artifactStore.getArtifactById(principleBearerId);
2275
2568
  if (principleArtifact) {
2276
2569
  resolvedSourcePrincipleId = EvaluatorRunner.extractPrincipleIdFromArtifact(principleArtifact);
2570
+ let parsedPrincipleContent;
2571
+ try {
2572
+ parsedPrincipleContent = JSON.parse(principleArtifact.contentJson);
2573
+ }
2574
+ catch {
2575
+ parsedPrincipleContent = principleArtifact.contentJson;
2576
+ }
2577
+ const contract = extractIntentContract(parsedPrincipleContent);
2578
+ if (contract !== null) {
2579
+ forwardedIntentContract = contract;
2580
+ }
2581
+ else {
2582
+ this.emitEvent('intent_contract_absent_on_principle', taskId, {
2583
+ principleArtifactId: principleBearerId,
2584
+ nextAction: 'pre_contract_scribe_artifact_backward_compatible',
2585
+ });
2586
+ }
2277
2587
  }
2278
2588
  }
2279
2589
  }
@@ -2294,6 +2604,27 @@ export class EvaluatorRunner extends BasePeerRunner {
2294
2604
  });
2295
2605
  return null;
2296
2606
  }
2607
+ const ruleContent = {
2608
+ implementationCode,
2609
+ goldenTrace: traceBuild.trace,
2610
+ goldenTraceCases,
2611
+ affectedTools: validatedAffectedTools,
2612
+ // adversarialResult.passed === true is the precondition for this method;
2613
+ // the gate decision is therefore accepted_shadow.
2614
+ ruleHostGateDecision: 'accepted_shadow',
2615
+ sourceArtificerArtifactId: assemblyInput.sourceArtificerArtifactId ?? output.sourceArtificerArtifactId,
2616
+ adversarialResult: output.adversarialResult,
2617
+ ...(requiresContextVersion === 2 ? { requiresContextVersion } : {}),
2618
+ // PRI-490: preserve evidenceRefs from Artificer artifact into rule artifact.
2619
+ // Only include when the array is valid (non-empty strings) — v1 rules may omit.
2620
+ ...(requiresContextVersion === 2 && Array.isArray(evidenceRefs) && evidenceRefs.every((e) => typeof e === 'string' && e.trim() !== '')
2621
+ ? { evidenceRefs: evidenceRefs }
2622
+ : {}),
2623
+ // PRI-703 Phase 1: forward the principle's intent contract verbatim —
2624
+ // the rule artifact self-carries the intent anchor it was validated
2625
+ // against (echo of the scribe single source, never a re-derivation).
2626
+ ...(forwardedIntentContract !== undefined ? { intentContract: forwardedIntentContract } : {}),
2627
+ };
2297
2628
  const ruleArtifactId = `pi-rule-${taskId}-${runId}`;
2298
2629
  const ruleId = `rule-${taskId}`;
2299
2630
  const nowIso = new Date().toISOString();