create-principles-disciple 1.133.11 → 1.133.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/console/dist/ui/i18n/en.json +13 -1
- package/console/dist/ui/i18n/zh-CN.json +13 -1
- package/console/dist/ui/pages/focus/OwnerDecisionCard.js +1 -1
- package/console/dist/ui/utils/validators.d.ts +9 -0
- package/console/dist/ui/utils/validators.js +31 -0
- package/console/dist/web/assets/app.css +3 -0
- package/console/dist/web/assets/app.js +60 -3
- package/console/package.json +1 -1
- package/core/dist/runtime-v2/internalization/__tests__/artifact-summary.property.test.js +1 -1
- package/core/dist/runtime-v2/internalization/__tests__/artifact-summary.property.test.js.map +1 -1
- package/core/dist/runtime-v2/internalization/__tests__/artificer-prompt-builder.test.js +6 -4
- package/core/dist/runtime-v2/internalization/__tests__/artificer-prompt-builder.test.js.map +1 -1
- package/core/dist/runtime-v2/internalization/__tests__/evaluator-out-of-scope-governance.test.d.ts +2 -0
- package/core/dist/runtime-v2/internalization/__tests__/evaluator-out-of-scope-governance.test.d.ts.map +1 -0
- package/core/dist/runtime-v2/internalization/__tests__/evaluator-out-of-scope-governance.test.js +307 -0
- package/core/dist/runtime-v2/internalization/__tests__/evaluator-out-of-scope-governance.test.js.map +1 -0
- package/core/dist/runtime-v2/internalization/__tests__/evaluator-prompt-builder.test.js +2 -2
- package/core/dist/runtime-v2/internalization/__tests__/evaluator-prompt-builder.test.js.map +1 -1
- package/core/dist/runtime-v2/internalization/__tests__/evolution-alignment-contract.test.d.ts +18 -0
- package/core/dist/runtime-v2/internalization/__tests__/evolution-alignment-contract.test.d.ts.map +1 -0
- package/core/dist/runtime-v2/internalization/__tests__/evolution-alignment-contract.test.js +457 -0
- package/core/dist/runtime-v2/internalization/__tests__/evolution-alignment-contract.test.js.map +1 -0
- package/core/dist/runtime-v2/internalization/__tests__/owner-decision-review.test.js +88 -0
- package/core/dist/runtime-v2/internalization/__tests__/owner-decision-review.test.js.map +1 -1
- package/core/dist/runtime-v2/internalization/artifact-summary.d.ts.map +1 -1
- package/core/dist/runtime-v2/internalization/artifact-summary.js +8 -0
- package/core/dist/runtime-v2/internalization/artifact-summary.js.map +1 -1
- package/core/dist/runtime-v2/internalization/artificer-prompt-builder.d.ts +37 -2
- package/core/dist/runtime-v2/internalization/artificer-prompt-builder.d.ts.map +1 -1
- package/core/dist/runtime-v2/internalization/artificer-prompt-builder.js +31 -1
- package/core/dist/runtime-v2/internalization/artificer-prompt-builder.js.map +1 -1
- package/core/dist/runtime-v2/internalization/artificer-runner.d.ts +18 -1
- package/core/dist/runtime-v2/internalization/artificer-runner.d.ts.map +1 -1
- package/core/dist/runtime-v2/internalization/artificer-runner.js +60 -2
- package/core/dist/runtime-v2/internalization/artificer-runner.js.map +1 -1
- package/core/dist/runtime-v2/internalization/context-manifests.d.ts.map +1 -1
- package/core/dist/runtime-v2/internalization/context-manifests.js +35 -2
- package/core/dist/runtime-v2/internalization/context-manifests.js.map +1 -1
- package/core/dist/runtime-v2/internalization/evaluator-prompt-builder.d.ts +15 -2
- package/core/dist/runtime-v2/internalization/evaluator-prompt-builder.d.ts.map +1 -1
- package/core/dist/runtime-v2/internalization/evaluator-prompt-builder.js +10 -2
- package/core/dist/runtime-v2/internalization/evaluator-prompt-builder.js.map +1 -1
- package/core/dist/runtime-v2/internalization/evaluator-runner.d.ts +46 -0
- package/core/dist/runtime-v2/internalization/evaluator-runner.d.ts.map +1 -1
- package/core/dist/runtime-v2/internalization/evaluator-runner.js +352 -21
- package/core/dist/runtime-v2/internalization/evaluator-runner.js.map +1 -1
- package/core/dist/runtime-v2/internalization/index.d.ts +4 -2
- package/core/dist/runtime-v2/internalization/index.d.ts.map +1 -1
- package/core/dist/runtime-v2/internalization/index.js +2 -1
- package/core/dist/runtime-v2/internalization/index.js.map +1 -1
- package/core/dist/runtime-v2/internalization/intent-contract.d.ts +59 -0
- package/core/dist/runtime-v2/internalization/intent-contract.d.ts.map +1 -0
- package/core/dist/runtime-v2/internalization/intent-contract.js +61 -0
- package/core/dist/runtime-v2/internalization/intent-contract.js.map +1 -0
- package/core/dist/runtime-v2/internalization/owner-decision-review.d.ts +28 -0
- package/core/dist/runtime-v2/internalization/owner-decision-review.d.ts.map +1 -1
- package/core/dist/runtime-v2/internalization/owner-decision-review.js +76 -0
- package/core/dist/runtime-v2/internalization/owner-decision-review.js.map +1 -1
- package/core/dist/runtime-v2/internalization/owner-review.d.ts +8 -0
- package/core/dist/runtime-v2/internalization/owner-review.d.ts.map +1 -1
- package/core/dist/runtime-v2/internalization/owner-review.js +10 -0
- package/core/dist/runtime-v2/internalization/owner-review.js.map +1 -1
- package/core/dist/runtime-v2/internalization/pitask-metadata.d.ts +76 -0
- package/core/dist/runtime-v2/internalization/pitask-metadata.d.ts.map +1 -1
- package/core/dist/runtime-v2/internalization/pitask-metadata.js +116 -3
- package/core/dist/runtime-v2/internalization/pitask-metadata.js.map +1 -1
- package/core/dist/runtime-v2/internalization/rule-reliability-validation.d.ts +141 -1
- package/core/dist/runtime-v2/internalization/rule-reliability-validation.d.ts.map +1 -1
- package/core/dist/runtime-v2/internalization/rule-reliability-validation.js +140 -0
- package/core/dist/runtime-v2/internalization/rule-reliability-validation.js.map +1 -1
- package/core/dist/runtime-v2/internalization/scribe-output.d.ts +11 -0
- package/core/dist/runtime-v2/internalization/scribe-output.d.ts.map +1 -1
- package/core/dist/runtime-v2/internalization/scribe-output.js +13 -0
- package/core/dist/runtime-v2/internalization/scribe-output.js.map +1 -1
- package/core/dist/runtime-v2/internalization/scribe-prompt-builder.d.ts +8 -1
- package/core/dist/runtime-v2/internalization/scribe-prompt-builder.d.ts.map +1 -1
- package/core/dist/runtime-v2/internalization/scribe-prompt-builder.js +25 -1
- package/core/dist/runtime-v2/internalization/scribe-prompt-builder.js.map +1 -1
- package/core/dist/runtime-v2/runner/__tests__/base-peer-runner-failure-details.test.js +16 -1
- package/core/dist/runtime-v2/runner/__tests__/base-peer-runner-failure-details.test.js.map +1 -1
- package/core/dist/runtime-v2/runner/base-peer-runner.d.ts +9 -1
- package/core/dist/runtime-v2/runner/base-peer-runner.d.ts.map +1 -1
- package/core/dist/runtime-v2/runner/base-peer-runner.js +32 -3
- package/core/dist/runtime-v2/runner/base-peer-runner.js.map +1 -1
- package/core/package.json +1 -1
- package/package.json +1 -1
- package/plugin/dist/bundle.js +764 -724
- package/plugin/dist/governance-audit.js +76 -76
- package/plugin/dist/rulehost-evidence.js +116 -116
- package/release-manager/package.json +1 -1
|
@@ -9,9 +9,11 @@ import { EvaluatorPromptBuilder, deriveRequirementLedger } from './evaluator-pro
|
|
|
9
9
|
import { reconcileLineageEcho } from './peer-runner-contracts.js';
|
|
10
10
|
import { BasePeerRunner } from '../runner/base-peer-runner.js';
|
|
11
11
|
import { EVALUATOR_STAGE1_MANIFEST, EVALUATOR_STAGE2_MANIFEST } from './context-manifests.js';
|
|
12
|
+
import { extractIntentContract } from './intent-contract.js';
|
|
12
13
|
import { evaluateFlaggedCriteria, isForcedStage2 } from './progressive-evaluator.js';
|
|
13
14
|
// PRI-426: single-round adversarial sandbox replay in succeedTask.
|
|
14
15
|
import { evaluateRefinerRuleHostGate } from './refiner-rulehost-gate.js';
|
|
16
|
+
import { attributionFromLayer, partitionV2OutOfScopeFailures, resolveRequiresContextVersionFromArtifact } from './rule-reliability-validation.js';
|
|
15
17
|
import { adversarialCasesToGoldenTrace } from './adversarial-case.js';
|
|
16
18
|
import { buildGoldenTraceFromArtificer } from '../golden-trace.js';
|
|
17
19
|
// PRI-485 Phase 6: auto-generate 5 v2 adversarial cases (unavailable/truncation/
|
|
@@ -595,6 +597,12 @@ export class EvaluatorRunner extends BasePeerRunner {
|
|
|
595
597
|
sourceArtificerArtifactId: context.sourceArtificerArtifactId ?? '',
|
|
596
598
|
previousEvaluation: context.previousEvaluation,
|
|
597
599
|
hostToolCatalog: this.hostToolCatalog ?? undefined,
|
|
600
|
+
// PRI-703 Phase 1: the scribe artifact's Owner-intent contract is the
|
|
601
|
+
// primary intentConsistency anchor. Extracted from the FULL parsed
|
|
602
|
+
// scribe artifact (before any manifest narrowing — the contract is
|
|
603
|
+
// injected as its own block). Absent on pre-contract artifacts →
|
|
604
|
+
// undefined → prompt unchanged (backward compatible).
|
|
605
|
+
intentContract: extractIntentContract(parsedScribeArtifact) ?? undefined,
|
|
598
606
|
});
|
|
599
607
|
return { message, stage2Evidence: resolutionOutcome };
|
|
600
608
|
}
|
|
@@ -926,13 +934,30 @@ export class EvaluatorRunner extends BasePeerRunner {
|
|
|
926
934
|
});
|
|
927
935
|
}
|
|
928
936
|
}
|
|
937
|
+
// ── Round-2 R2 (Owner 指令 2026-09-08): 出界处置先于 intent 落库派生 ──
|
|
938
|
+
// deriveGovernanceEffect 只读 durable 事实 (重放已 updateRunOutput 持久化
|
|
939
|
+
// 的 adversarialResult + durable artificer 工件 + durable repair 轮次),
|
|
940
|
+
// 不依赖任何瞬时变量。派生结果随 intent 持久化 —— fresh 与 resume 由同
|
|
941
|
+
// 一 durable 判据得到同一处置,重启不改变已选定的治理动作 (R2 复现的
|
|
942
|
+
// 治理漂移断点)。
|
|
943
|
+
const sourceArtificerArtifactIdForGovernance = context.sourceArtificerArtifactId
|
|
944
|
+
?? finalOutput.sourceArtificerArtifactId
|
|
945
|
+
?? null;
|
|
946
|
+
const governanceEffect = await this.deriveGovernanceEffect(taskId, {
|
|
947
|
+
output: finalOutput,
|
|
948
|
+
sourceArtificerArtifactId: sourceArtificerArtifactIdForGovernance,
|
|
949
|
+
diagnosticReplayEvidence,
|
|
950
|
+
});
|
|
929
951
|
// ── P0 (verdict drift): verdict + completion intent 原子落库 ──
|
|
930
952
|
// 必须先于一切治理 side effect (validate bearer / seed repair / rule
|
|
931
953
|
// assembly):side effect 已发生而 intent 未落 = crash 后重跑会重新问
|
|
932
954
|
// LLM,新 verdict 与已发生副作用形成治理矛盾 (repair drift /
|
|
933
955
|
// validation drift / validated-rule drift)。同 epoch crash/retry 重跑经
|
|
934
956
|
// maybeResumePendingIntent resume,不重问。
|
|
935
|
-
await this.recordCompletionOrThrow(taskId, runId,
|
|
957
|
+
await this.recordCompletionOrThrow(taskId, runId, {
|
|
958
|
+
decision: finalOutput.evaluation.decision,
|
|
959
|
+
governanceEffect,
|
|
960
|
+
});
|
|
936
961
|
const ruleAssemblyInput = {
|
|
937
962
|
artificerArtifact: context.artificerArtifact ?? null,
|
|
938
963
|
sourceArtificerArtifactId: context.sourceArtificerArtifactId ?? null,
|
|
@@ -1056,10 +1081,14 @@ export class EvaluatorRunner extends BasePeerRunner {
|
|
|
1056
1081
|
if (decision === 'needs_revision' && this.isRepairLoopEnabled()) {
|
|
1057
1082
|
const repairOutcome = await this.maybeSeedArtificerRepair(taskId, { runId, output: finalOutput, sourceArtificerArtifactId, diagnosticReplayEvidence });
|
|
1058
1083
|
if (repairOutcome.kind === 'max_iterations_reached') {
|
|
1059
|
-
// PRI-629: budget 耗尽(decision-capable)与 seed 失败(recovery)
|
|
1084
|
+
// PRI-629: budget 耗尽(decision-capable)与 seed 失败(recovery)拆分原因码。
|
|
1085
|
+
// PRI-703 Phase 2: test_out_of_scope 是 FAILED_TEST 归因 — v2-context
|
|
1086
|
+
// case 对 v1 规则的通道设计限制,decision-capable(Owner 裁决是唯一出口)。
|
|
1060
1087
|
const reasonCode = repairOutcome.detail === 'budget_exhausted'
|
|
1061
1088
|
? HUMAN_REVIEW_REASON.evaluatorRepairBudgetExhausted
|
|
1062
|
-
:
|
|
1089
|
+
: repairOutcome.detail === 'test_out_of_scope'
|
|
1090
|
+
? HUMAN_REVIEW_REASON.evaluatorTestOutOfScope
|
|
1091
|
+
: HUMAN_REVIEW_REASON.evaluatorRepairSeedFailed;
|
|
1063
1092
|
// Fail loud (rc-9, EP-03, ERR-002): mark the task needs_human_review
|
|
1064
1093
|
// so it does NOT stay in 'leased' state (which would cause the lease
|
|
1065
1094
|
// to expire and the evaluator to re-run the same verdict infinitely).
|
|
@@ -1096,6 +1125,70 @@ export class EvaluatorRunner extends BasePeerRunner {
|
|
|
1096
1125
|
// transition decision 依据 durable runnerDecision fail-closed。
|
|
1097
1126
|
return { kind: 'completed', ruleArtifactId: null };
|
|
1098
1127
|
}
|
|
1128
|
+
/**
|
|
1129
|
+
* PRI-703 Phase 2: extract failed replay caseIds from the evaluator output.
|
|
1130
|
+
* Trust-boundary (rc-1/rc-4): adversarialResult.failedCases is untrusted
|
|
1131
|
+
* artifact content — validate each element's caseId shape; malformed entries
|
|
1132
|
+
* are skipped (they cannot inform scoping decisions).
|
|
1133
|
+
*/
|
|
1134
|
+
static extractFailedCases(output) {
|
|
1135
|
+
// rc-1/rc-2 (ERR-001): adversarialResult.failedCases is untrusted
|
|
1136
|
+
// artifact content — narrow via the class's isRecord/Array guards, no `as`.
|
|
1137
|
+
// Round-2 R1: extract the sandbox-carried expectedDecision per case for
|
|
1138
|
+
// the oracle-consistency check in partitionV2OutOfScopeFailures. The
|
|
1139
|
+
// stored value is only ever COMPARED against the compile-time template
|
|
1140
|
+
// oracle — it is never the classification source of truth.
|
|
1141
|
+
if (!EvaluatorRunner.isRecord(output))
|
|
1142
|
+
return [];
|
|
1143
|
+
const { adversarialResult } = output;
|
|
1144
|
+
if (!EvaluatorRunner.isRecord(adversarialResult))
|
|
1145
|
+
return [];
|
|
1146
|
+
const { failedCases } = adversarialResult;
|
|
1147
|
+
if (!Array.isArray(failedCases))
|
|
1148
|
+
return [];
|
|
1149
|
+
const extracted = [];
|
|
1150
|
+
for (const entry of failedCases) {
|
|
1151
|
+
if (!EvaluatorRunner.isRecord(entry))
|
|
1152
|
+
continue;
|
|
1153
|
+
const { caseId } = entry;
|
|
1154
|
+
if (typeof caseId !== 'string' || caseId.trim() === '')
|
|
1155
|
+
continue;
|
|
1156
|
+
const { expectedDecision } = entry;
|
|
1157
|
+
extracted.push(typeof expectedDecision === 'string' && expectedDecision.trim() !== ''
|
|
1158
|
+
? { caseId, expectedDecision }
|
|
1159
|
+
: { caseId });
|
|
1160
|
+
}
|
|
1161
|
+
return extracted;
|
|
1162
|
+
}
|
|
1163
|
+
/**
|
|
1164
|
+
* PRI-703 Phase 2: resolve the rule's requiresContextVersion from the
|
|
1165
|
+
* DURABLE artificer artifact (never from the LLM-forgible evaluator copy).
|
|
1166
|
+
* Three-way result (see resolveRequiresContextVersionFromArtifact):
|
|
1167
|
+
* 2 / undefined → resolved (v2 / v1) — the scope partition MUST run;
|
|
1168
|
+
* null → unresolvable (attribution stays inconclusive — fail-open to the
|
|
1169
|
+
* existing repair path, never blocks the loop on a read error).
|
|
1170
|
+
*
|
|
1171
|
+
* 评审 P1 修正:key-absent on a PARSED artifact is deterministically v1
|
|
1172
|
+
* (artificer schema only ever writes literal 2) and must reach the
|
|
1173
|
+
* partition — collapsing it into null made the out-of-scope routing
|
|
1174
|
+
* unreachable in production for exactly its target population (v1 rules).
|
|
1175
|
+
*/
|
|
1176
|
+
async resolveRequiresContextVersion(artificerArtifactId, taskId) {
|
|
1177
|
+
try {
|
|
1178
|
+
const artifact = await this.artifactStore.getArtifactById(artificerArtifactId);
|
|
1179
|
+
if (!artifact)
|
|
1180
|
+
return null;
|
|
1181
|
+
return resolveRequiresContextVersionFromArtifact(artifact.contentJson);
|
|
1182
|
+
}
|
|
1183
|
+
catch (err) {
|
|
1184
|
+
this.emitEvent('attribution_scope_resolve_failed', taskId, {
|
|
1185
|
+
artificerArtifactId,
|
|
1186
|
+
reason: err instanceof Error ? err.message : String(err),
|
|
1187
|
+
nextAction: 'verify_artifact_store_read_for_failure_attribution',
|
|
1188
|
+
});
|
|
1189
|
+
return null;
|
|
1190
|
+
}
|
|
1191
|
+
}
|
|
1099
1192
|
/**
|
|
1100
1193
|
* PRI-509: Resolve the prior repair iteration by reading the dependency
|
|
1101
1194
|
* artificer task's repairPayload (rc-7: written at task creation, never
|
|
@@ -1193,7 +1286,8 @@ export class EvaluatorRunner extends BasePeerRunner {
|
|
|
1193
1286
|
* 证明 output 已 durable (updateRunOutput 在 succeedTask 最前)。
|
|
1194
1287
|
* 同 epoch crash/retry 重跑经 maybeResumePendingIntent resume,不重问 LLM。
|
|
1195
1288
|
*/
|
|
1196
|
-
async recordCompletionOrThrow(taskId, runId,
|
|
1289
|
+
async recordCompletionOrThrow(taskId, runId, decisionAndEffect) {
|
|
1290
|
+
const { decision, governanceEffect } = decisionAndEffect;
|
|
1197
1291
|
try {
|
|
1198
1292
|
const raw = await this.stateManager.getTask(taskId);
|
|
1199
1293
|
if (!raw)
|
|
@@ -1210,6 +1304,12 @@ export class EvaluatorRunner extends BasePeerRunner {
|
|
|
1210
1304
|
sourceRunId: runId,
|
|
1211
1305
|
revisionEpoch: piTask.revisionCount ?? 0,
|
|
1212
1306
|
status: 'pending',
|
|
1307
|
+
...(governanceEffect?.selectedEffect !== undefined
|
|
1308
|
+
? { selectedEffect: governanceEffect.selectedEffect }
|
|
1309
|
+
: {}),
|
|
1310
|
+
...(governanceEffect?.effectReasonCode !== undefined
|
|
1311
|
+
? { effectReasonCode: governanceEffect.effectReasonCode }
|
|
1312
|
+
: {}),
|
|
1213
1313
|
},
|
|
1214
1314
|
});
|
|
1215
1315
|
await this.stateManager.updateTaskDiagnosticJson(taskId, createPITaskDiagnosticJson(merged));
|
|
@@ -1287,6 +1387,49 @@ export class EvaluatorRunner extends BasePeerRunner {
|
|
|
1287
1387
|
decision: intent.decision,
|
|
1288
1388
|
sourceRunId: intent.sourceRunId,
|
|
1289
1389
|
});
|
|
1390
|
+
// ── Round-2 R2 (Owner 指令 2026-09-08): 选定 NHR 效果的恢复直通 ──
|
|
1391
|
+
// fresh 路径在 intent 落库前已把 deriveGovernanceEffect 的结论
|
|
1392
|
+
// (selectedEffect='needs_human_review' + effectReasonCode) 持久化;恢复
|
|
1393
|
+
// 执行读同一 durable intent,直接重放 NHR 效果 — 禁止重问 LLM、禁止
|
|
1394
|
+
// 再 seed repair、禁止重新判责 (R2 复现的治理漂移断点)。照抄
|
|
1395
|
+
// rollout-reviewer-runner 的 intent.effect resume 模板,零新状态源。
|
|
1396
|
+
if ((intent.selectedEffect ?? intent.effect) === 'needs_human_review') {
|
|
1397
|
+
const output2 = await this.recoverIntentOutput(taskId, intent.sourceRunId, intent.decision);
|
|
1398
|
+
const artifactId2 = `pi-art-${taskId}-${intent.sourceRunId}`;
|
|
1399
|
+
const reasonCode = intent.effectReasonCode ?? HUMAN_REVIEW_REASON.evaluatorRepairSeedFailed;
|
|
1400
|
+
if (reasonCode === 'evaluator_test_out_of_scope') {
|
|
1401
|
+
this.emitEvent('repair_loop_test_out_of_scope', taskId, {
|
|
1402
|
+
runId: intent.sourceRunId,
|
|
1403
|
+
attribution: attributionFromLayer('test'),
|
|
1404
|
+
source: 'resume_from_persisted_intent',
|
|
1405
|
+
nextAction: 'owner_decision_required_channel_upgrade_or_principle_revision',
|
|
1406
|
+
});
|
|
1407
|
+
}
|
|
1408
|
+
await this.markNeedsHumanReviewOrThrow(taskId, {
|
|
1409
|
+
runId: intent.sourceRunId,
|
|
1410
|
+
reasonCode,
|
|
1411
|
+
sourceArtifactId: artifactId2,
|
|
1412
|
+
});
|
|
1413
|
+
await this.markCompletionIntentAppliedOrThrow(taskId);
|
|
1414
|
+
this.emitEvent('task_needs_human_review', taskId, {
|
|
1415
|
+
attemptCount: leasedTask.attemptCount,
|
|
1416
|
+
resultRef: `${this.config.resultRefPrefix}://${intent.sourceRunId}`,
|
|
1417
|
+
evaluationDecision: output2.evaluation?.decision,
|
|
1418
|
+
evaluationScore: output2.evaluation?.score,
|
|
1419
|
+
ruleArtifactId: null,
|
|
1420
|
+
reason: `repair_loop_${reasonCode}`,
|
|
1421
|
+
});
|
|
1422
|
+
return {
|
|
1423
|
+
status: 'succeeded',
|
|
1424
|
+
taskId,
|
|
1425
|
+
runId: intent.sourceRunId,
|
|
1426
|
+
artifactId: artifactId2,
|
|
1427
|
+
resultRef: `${this.config.resultRefPrefix}://${intent.sourceRunId}`,
|
|
1428
|
+
contextHash: `resume-${intent.sourceRunId}`,
|
|
1429
|
+
output: output2,
|
|
1430
|
+
attemptCount: leasedTask.attemptCount,
|
|
1431
|
+
};
|
|
1432
|
+
}
|
|
1290
1433
|
const output = await this.recoverIntentOutput(taskId, intent.sourceRunId, intent.decision);
|
|
1291
1434
|
const artifactId = `pi-art-${taskId}-${intent.sourceRunId}`;
|
|
1292
1435
|
const contextHash = `resume-${intent.sourceRunId}`;
|
|
@@ -1549,9 +1692,164 @@ export class EvaluatorRunner extends BasePeerRunner {
|
|
|
1549
1692
|
* are typed as readonly string[] on EvaluatorEvaluation, so element-level
|
|
1550
1693
|
* re-validation is not required here (rc-4 N/A — not unknown at this point).
|
|
1551
1694
|
*/
|
|
1695
|
+
/**
|
|
1696
|
+
* Round-2 R2 (Owner 指令 2026-09-08): 从 durable 证据派生本 verdict 的治理
|
|
1697
|
+
* 效果。fresh 与 resume 调用同一派生 —— 输入全部是 durable 事实:
|
|
1698
|
+
* - output: intent 落库前已由 executeDeterministicReplay 持久化的
|
|
1699
|
+
* adversarialResult(resume 时 recoverIntentOutput 从 run.outputPayload
|
|
1700
|
+
* 恢复同一对象);
|
|
1701
|
+
* - sourceArtificerArtifactId → durable 工件的 requiresContextVersion;
|
|
1702
|
+
* - priorRepairIteration → durable dependency repairPayload。
|
|
1703
|
+
* record 只持久化本函数的结论,不做判断 (指令要求: 禁止把复杂判断逻辑
|
|
1704
|
+
* 塞进 recordCompletionOrThrow)。返回 undefined = governance_transition
|
|
1705
|
+
* (正常效果由 applyEvaluatorDecisionEffects 执行,无需特判)。
|
|
1706
|
+
*
|
|
1707
|
+
* fresh 路径的 diagnosticReplayEvidence 仅用于验证"重放确实执行过"这一
|
|
1708
|
+
* 控制流事实 (P1 provenance); resume 路径不传时, 以 durable adversarialResult
|
|
1709
|
+
* 的形态 (passed === false 且 failedCases 非空) 为等价判据 —— intent 存在
|
|
1710
|
+
* 本身即证明 fresh 侧重放已执行并持久化。
|
|
1711
|
+
*/
|
|
1712
|
+
/**
|
|
1713
|
+
* Round-2 R2: 读取当前 durable completion intent (pending 或 applied 均可读 —
|
|
1714
|
+
* applied 意味着效果已 materialize, 读侧只用于一致性核对, 不会改变路由)。
|
|
1715
|
+
* 解析失败/缺 intent 返回 null — 出界判定回退到 fresh 快路径判定
|
|
1716
|
+
* (fail-closed, 绝不因读不到 intent 而静默 seed)。
|
|
1717
|
+
*/
|
|
1718
|
+
async readPendingOrAppliedCompletionIntent(taskId) {
|
|
1719
|
+
try {
|
|
1720
|
+
const raw = await this.stateManager.getTask(taskId);
|
|
1721
|
+
if (!raw)
|
|
1722
|
+
return null;
|
|
1723
|
+
const piTask = hydratePITaskRecord(raw);
|
|
1724
|
+
return piTask?.completionIntent ?? null;
|
|
1725
|
+
}
|
|
1726
|
+
catch (err) {
|
|
1727
|
+
this.emitEvent('completion_intent_read_failed', taskId, {
|
|
1728
|
+
errorMessage: err instanceof Error ? err.message : String(err),
|
|
1729
|
+
nextAction: 'out_of_scope_disposition_falls_back_to_fresh_path_derivation',
|
|
1730
|
+
});
|
|
1731
|
+
return null;
|
|
1732
|
+
}
|
|
1733
|
+
}
|
|
1734
|
+
async deriveGovernanceEffect(evaluatorTaskId, ctx) {
|
|
1735
|
+
if (ctx.output.evaluation?.decision !== 'needs_revision')
|
|
1736
|
+
return undefined;
|
|
1737
|
+
if (!this.isRepairLoopEnabled())
|
|
1738
|
+
return undefined;
|
|
1739
|
+
// P1 provenance: fresh 要求 ran:true;resume 无 live 重放时等价判据为
|
|
1740
|
+
// durable adversarialResult 形态 (passed:false + failedCases>0)。
|
|
1741
|
+
const freshReplayFailed = ctx.diagnosticReplayEvidence?.ran === true && ctx.diagnosticReplayEvidence.passed === false;
|
|
1742
|
+
let durableReplayFailed = false;
|
|
1743
|
+
if (EvaluatorRunner.isRecord(ctx.output)) {
|
|
1744
|
+
const { adversarialResult } = ctx.output;
|
|
1745
|
+
if (EvaluatorRunner.isRecord(adversarialResult)) {
|
|
1746
|
+
const { passed } = adversarialResult;
|
|
1747
|
+
const { failedCases } = adversarialResult;
|
|
1748
|
+
durableReplayFailed = passed === false && Array.isArray(failedCases) && failedCases.length > 0;
|
|
1749
|
+
}
|
|
1750
|
+
}
|
|
1751
|
+
if (!freshReplayFailed && !durableReplayFailed)
|
|
1752
|
+
return undefined;
|
|
1753
|
+
const failedCases = EvaluatorRunner.extractFailedCases(ctx.output);
|
|
1754
|
+
if (failedCases.length === 0)
|
|
1755
|
+
return undefined;
|
|
1756
|
+
const requiresContextVersion = ctx.sourceArtificerArtifactId
|
|
1757
|
+
? await this.resolveRequiresContextVersion(ctx.sourceArtificerArtifactId, evaluatorTaskId)
|
|
1758
|
+
: null;
|
|
1759
|
+
if (requiresContextVersion === null)
|
|
1760
|
+
return undefined;
|
|
1761
|
+
const partition = partitionV2OutOfScopeFailures({
|
|
1762
|
+
requiresContextVersion: requiresContextVersion === 2 ? 2 : undefined,
|
|
1763
|
+
failedCases,
|
|
1764
|
+
});
|
|
1765
|
+
if (partition.isPureOutOfScope) {
|
|
1766
|
+
this.emitEvent('governance_effect_out_of_scope_selected', evaluatorTaskId, {
|
|
1767
|
+
outOfScopeCaseIds: [...partition.outOfScope],
|
|
1768
|
+
nextAction: 'selected_effect_needs_human_review_evaluator_test_out_of_scope',
|
|
1769
|
+
});
|
|
1770
|
+
return {
|
|
1771
|
+
selectedEffect: 'needs_human_review',
|
|
1772
|
+
effectReasonCode: 'evaluator_test_out_of_scope',
|
|
1773
|
+
};
|
|
1774
|
+
}
|
|
1775
|
+
return undefined;
|
|
1776
|
+
}
|
|
1552
1777
|
async maybeSeedArtificerRepair(evaluatorTaskId, ctx) {
|
|
1553
1778
|
const { runId: evaluatorRunId, output } = ctx;
|
|
1554
1779
|
const priorRepairIteration = await this.resolvePriorRepairIteration(evaluatorTaskId);
|
|
1780
|
+
// ── PRI-703 Phase 2 (Owner decision 2026-09-07): deterministic failure
|
|
1781
|
+
// attribution BEFORE seeding a repair round. "测试失败 → 继续改 Rule" is
|
|
1782
|
+
// the wrong default; a needs_revision whose replay failures are ALL
|
|
1783
|
+
// v2-context template cases judging a v1 (action-only) rule is a
|
|
1784
|
+
// TEST-SCOPE problem (FAILED_TEST attribution): the v1 channel
|
|
1785
|
+
// structurally cannot express context semantics, so no rule repair can
|
|
1786
|
+
// ever satisfy those cases (Episode 001's 18/18 death-loop). Route to
|
|
1787
|
+
// owner review with the channel-limitation signal instead of seeding a
|
|
1788
|
+
// doomed repair round.
|
|
1789
|
+
// Provenance: adversarialResult on the output here was written by
|
|
1790
|
+
// executeDeterministicReplay in THIS invocation (diagnosticReplayEvidence
|
|
1791
|
+
// ran:true) or absent — never LLM-forged (the pre-replay LLM copy was
|
|
1792
|
+
// replaced before the effects phase).
|
|
1793
|
+
//
|
|
1794
|
+
// 混合场景 (in-scope ⊕ out-of-scope 并存): 修复任务照常 seed (有真实
|
|
1795
|
+
// Rule 缺陷可修),但归因上下文随 RepairPayload 流动,让修复 LLM 知道
|
|
1796
|
+
// 哪些 case 是通道限制 (不得尝试满足) — "哪里失败/为什么/改哪里" 三问
|
|
1797
|
+
// 在载荷层可回答 (PRI-705)。
|
|
1798
|
+
let repairAttribution;
|
|
1799
|
+
// Round-2 R2: 出界处置以持久化的 completion intent 为权威。fresh 路径在
|
|
1800
|
+
// recordCompletionOrThrow 之前已由 deriveGovernanceEffect 写入
|
|
1801
|
+
// selectedEffect='needs_human_review' —— 同一 durable 依据 (R1 oracle 校验
|
|
1802
|
+
// 的 partition) 派生; resume 路径经 recoverIntentOutput 后重读本 intent,
|
|
1803
|
+
// 无需重算也无需任何瞬时变量。瞬时 evidence 门只作 fresh 快路径与
|
|
1804
|
+
// 派生时的 provenance 校验, 不再是出界判定的唯一载体 (R2 漂移断点)。
|
|
1805
|
+
{
|
|
1806
|
+
const intent = await this.readPendingOrAppliedCompletionIntent(evaluatorTaskId);
|
|
1807
|
+
// Round-3: hydrate 将 selectedEffect 归一化为 effect —— 读侧必须两键
|
|
1808
|
+
// 同读,否则 fresh 分支永不命中、静默回退瞬时 evidence (ERR-024 键不对称)。
|
|
1809
|
+
if ((intent?.selectedEffect ?? intent?.effect) === 'needs_human_review'
|
|
1810
|
+
&& intent?.effectReasonCode === 'evaluator_test_out_of_scope') {
|
|
1811
|
+
this.emitEvent('repair_loop_test_out_of_scope', evaluatorTaskId, {
|
|
1812
|
+
runId: evaluatorRunId,
|
|
1813
|
+
attribution: attributionFromLayer('test'),
|
|
1814
|
+
source: 'persisted_completion_intent',
|
|
1815
|
+
nextAction: 'owner_decision_required_channel_upgrade_or_principle_revision',
|
|
1816
|
+
});
|
|
1817
|
+
return { kind: 'max_iterations_reached', detail: 'test_out_of_scope' };
|
|
1818
|
+
}
|
|
1819
|
+
}
|
|
1820
|
+
if (ctx.diagnosticReplayEvidence?.ran === true && ctx.diagnosticReplayEvidence.passed === false) {
|
|
1821
|
+
const failedCases = EvaluatorRunner.extractFailedCases(output);
|
|
1822
|
+
if (failedCases.length > 0) {
|
|
1823
|
+
const sourceArtificerArtifactIdForScope = ctx.sourceArtificerArtifactId ?? output.sourceArtificerArtifactId;
|
|
1824
|
+
const requiresContextVersion = sourceArtificerArtifactIdForScope
|
|
1825
|
+
? await this.resolveRequiresContextVersion(sourceArtificerArtifactIdForScope, evaluatorTaskId)
|
|
1826
|
+
: null;
|
|
1827
|
+
if (requiresContextVersion !== null) {
|
|
1828
|
+
const partition = partitionV2OutOfScopeFailures({
|
|
1829
|
+
requiresContextVersion: requiresContextVersion === 2 ? 2 : undefined,
|
|
1830
|
+
failedCases,
|
|
1831
|
+
});
|
|
1832
|
+
if (partition.isPureOutOfScope) {
|
|
1833
|
+
this.emitEvent('repair_loop_test_out_of_scope', evaluatorTaskId, {
|
|
1834
|
+
runId: evaluatorRunId,
|
|
1835
|
+
attribution: attributionFromLayer('test'),
|
|
1836
|
+
outOfScopeCaseIds: [...partition.outOfScope],
|
|
1837
|
+
failedCaseCount: ctx.diagnosticReplayEvidence.failedCaseCount,
|
|
1838
|
+
reason: 'all_replay_failures_are_v2_context_cases_judging_a_v1_rule',
|
|
1839
|
+
nextAction: 'owner_decision_required_channel_upgrade_or_principle_revision',
|
|
1840
|
+
});
|
|
1841
|
+
return { kind: 'max_iterations_reached', detail: 'test_out_of_scope' };
|
|
1842
|
+
}
|
|
1843
|
+
if (partition.outOfScope.length > 0) {
|
|
1844
|
+
repairAttribution = {
|
|
1845
|
+
attribution: attributionFromLayer('rule'),
|
|
1846
|
+
outOfScopeCaseIds: [...partition.outOfScope],
|
|
1847
|
+
reason: `replay failures mix real rule defects (in-scope cases: ${partition.inScope.join(', ')}) with test-scope v2-context cases (${partition.outOfScope.join(', ')}) that the v1 action-only channel structurally cannot express — do NOT attempt to satisfy the out-of-scope cases`,
|
|
1848
|
+
};
|
|
1849
|
+
}
|
|
1850
|
+
}
|
|
1851
|
+
}
|
|
1852
|
+
}
|
|
1555
1853
|
// ── Slice 5: max iterations (2) reached → fail loud ──
|
|
1556
1854
|
if (priorRepairIteration >= 2) {
|
|
1557
1855
|
// Task state update (→ needs_human_review) is handled by the caller
|
|
@@ -1583,6 +1881,10 @@ export class EvaluatorRunner extends BasePeerRunner {
|
|
|
1583
1881
|
//(来自 succeedTask 的 executeDeterministicReplay 调用),绝不从 LLM
|
|
1584
1882
|
// 可伪造的 output.adversarialResult 反推。缺失 = 本次未执行确定性重放
|
|
1585
1883
|
//(skip/error/无 live 路径),repairPayload 不包含 diagnosticReplay。
|
|
1884
|
+
//
|
|
1885
|
+
// PRI-705 / PRI-703 Phase 2: 混合场景 (非纯 out-of-scope) 下,归因与
|
|
1886
|
+
// out-of-scope case 清单随载荷流动 — 修复 LLM 明确知道哪些失败属于
|
|
1887
|
+
// 通道设计限制(不得尝试满足),哪些是真实 Rule 缺陷(必须修复)。
|
|
1586
1888
|
const repairPayload = {
|
|
1587
1889
|
requiredChanges: [...output.evaluation.requiredChanges],
|
|
1588
1890
|
concerns: [...output.evaluation.concerns],
|
|
@@ -1591,6 +1893,7 @@ export class EvaluatorRunner extends BasePeerRunner {
|
|
|
1591
1893
|
sourceArtificerArtifactId,
|
|
1592
1894
|
sourceEvaluatorTaskId: evaluatorTaskId,
|
|
1593
1895
|
...(ctx.diagnosticReplayEvidence ? { diagnosticReplay: ctx.diagnosticReplayEvidence } : {}),
|
|
1896
|
+
...(repairAttribution !== undefined ? { failureAttribution: repairAttribution } : {}),
|
|
1594
1897
|
};
|
|
1595
1898
|
// Resolve the dependency artificer task to inherit dependencyTaskIds
|
|
1596
1899
|
// (so the repair task points to the same scribe task, preserving lineage).
|
|
@@ -2244,36 +2547,43 @@ export class EvaluatorRunner extends BasePeerRunner {
|
|
|
2244
2547
|
const validatedAffectedTools = Array.isArray(affectedTools)
|
|
2245
2548
|
? affectedTools.filter((t) => typeof t === 'string')
|
|
2246
2549
|
: [];
|
|
2247
|
-
const ruleContent = {
|
|
2248
|
-
implementationCode,
|
|
2249
|
-
goldenTrace: traceBuild.trace,
|
|
2250
|
-
goldenTraceCases,
|
|
2251
|
-
affectedTools: validatedAffectedTools,
|
|
2252
|
-
// adversarialResult.passed === true is the precondition for this method;
|
|
2253
|
-
// the gate decision is therefore accepted_shadow.
|
|
2254
|
-
ruleHostGateDecision: 'accepted_shadow',
|
|
2255
|
-
sourceArtificerArtifactId: assemblyInput.sourceArtificerArtifactId ?? output.sourceArtificerArtifactId,
|
|
2256
|
-
adversarialResult: output.adversarialResult,
|
|
2257
|
-
...(requiresContextVersion === 2 ? { requiresContextVersion } : {}),
|
|
2258
|
-
// PRI-490: preserve evidenceRefs from Artificer artifact into rule artifact.
|
|
2259
|
-
// Only include when the array is valid (non-empty strings) — v1 rules may omit.
|
|
2260
|
-
...(requiresContextVersion === 2 && Array.isArray(evidenceRefs) && evidenceRefs.every((e) => typeof e === 'string' && e.trim() !== '')
|
|
2261
|
-
? { evidenceRefs: evidenceRefs }
|
|
2262
|
-
: {}),
|
|
2263
|
-
};
|
|
2264
2550
|
// P1 #7 (cross-package acceptance test discovery): resolve the scribe
|
|
2265
2551
|
// principle artifact and carry forward its principle ID as
|
|
2266
2552
|
// sourcePrincipleId on the rule artifact. Without this, extractPrincipleId()
|
|
2267
2553
|
// in the activation dispatcher returns null for rule artifacts, causing
|
|
2268
2554
|
// activateArtifact() to fail with 'invalid_artifact'/'no_principle_id'.
|
|
2269
2555
|
// The rule artifact must carry lineage to the principle it enforces.
|
|
2556
|
+
//
|
|
2557
|
+
// PRI-703 Phase 1: the same resolution deterministically forwards the
|
|
2558
|
+
// principle's intent contract onto the rule artifact (single author of the
|
|
2559
|
+
// rule's intent anchor = the scribe principle; the rule echoes it, it
|
|
2560
|
+
// never re-derives it — no second source of truth). Resolved BEFORE the
|
|
2561
|
+
// ruleContent build so the contract rides in the same contentJson write.
|
|
2270
2562
|
let resolvedSourcePrincipleId;
|
|
2563
|
+
let forwardedIntentContract;
|
|
2271
2564
|
try {
|
|
2272
2565
|
const principleBearerId = await this.resolvePrincipleBearerArtifact(output, taskId);
|
|
2273
2566
|
if (principleBearerId) {
|
|
2274
2567
|
const principleArtifact = await this.artifactStore.getArtifactById(principleBearerId);
|
|
2275
2568
|
if (principleArtifact) {
|
|
2276
2569
|
resolvedSourcePrincipleId = EvaluatorRunner.extractPrincipleIdFromArtifact(principleArtifact);
|
|
2570
|
+
let parsedPrincipleContent;
|
|
2571
|
+
try {
|
|
2572
|
+
parsedPrincipleContent = JSON.parse(principleArtifact.contentJson);
|
|
2573
|
+
}
|
|
2574
|
+
catch {
|
|
2575
|
+
parsedPrincipleContent = principleArtifact.contentJson;
|
|
2576
|
+
}
|
|
2577
|
+
const contract = extractIntentContract(parsedPrincipleContent);
|
|
2578
|
+
if (contract !== null) {
|
|
2579
|
+
forwardedIntentContract = contract;
|
|
2580
|
+
}
|
|
2581
|
+
else {
|
|
2582
|
+
this.emitEvent('intent_contract_absent_on_principle', taskId, {
|
|
2583
|
+
principleArtifactId: principleBearerId,
|
|
2584
|
+
nextAction: 'pre_contract_scribe_artifact_backward_compatible',
|
|
2585
|
+
});
|
|
2586
|
+
}
|
|
2277
2587
|
}
|
|
2278
2588
|
}
|
|
2279
2589
|
}
|
|
@@ -2294,6 +2604,27 @@ export class EvaluatorRunner extends BasePeerRunner {
|
|
|
2294
2604
|
});
|
|
2295
2605
|
return null;
|
|
2296
2606
|
}
|
|
2607
|
+
const ruleContent = {
|
|
2608
|
+
implementationCode,
|
|
2609
|
+
goldenTrace: traceBuild.trace,
|
|
2610
|
+
goldenTraceCases,
|
|
2611
|
+
affectedTools: validatedAffectedTools,
|
|
2612
|
+
// adversarialResult.passed === true is the precondition for this method;
|
|
2613
|
+
// the gate decision is therefore accepted_shadow.
|
|
2614
|
+
ruleHostGateDecision: 'accepted_shadow',
|
|
2615
|
+
sourceArtificerArtifactId: assemblyInput.sourceArtificerArtifactId ?? output.sourceArtificerArtifactId,
|
|
2616
|
+
adversarialResult: output.adversarialResult,
|
|
2617
|
+
...(requiresContextVersion === 2 ? { requiresContextVersion } : {}),
|
|
2618
|
+
// PRI-490: preserve evidenceRefs from Artificer artifact into rule artifact.
|
|
2619
|
+
// Only include when the array is valid (non-empty strings) — v1 rules may omit.
|
|
2620
|
+
...(requiresContextVersion === 2 && Array.isArray(evidenceRefs) && evidenceRefs.every((e) => typeof e === 'string' && e.trim() !== '')
|
|
2621
|
+
? { evidenceRefs: evidenceRefs }
|
|
2622
|
+
: {}),
|
|
2623
|
+
// PRI-703 Phase 1: forward the principle's intent contract verbatim —
|
|
2624
|
+
// the rule artifact self-carries the intent anchor it was validated
|
|
2625
|
+
// against (echo of the scribe single source, never a re-derivation).
|
|
2626
|
+
...(forwardedIntentContract !== undefined ? { intentContract: forwardedIntentContract } : {}),
|
|
2627
|
+
};
|
|
2297
2628
|
const ruleArtifactId = `pi-rule-${taskId}-${runId}`;
|
|
2298
2629
|
const ruleId = `rule-${taskId}`;
|
|
2299
2630
|
const nowIso = new Date().toISOString();
|