@yagni-app/code-staging 1.2.3-staging.1621.1 → 1.2.3-staging.1627.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/extension/pipeline/findings.js +7 -3
- package/dist/extension/pipeline/goCommand.js +1 -1
- package/dist/extension/pipeline/headlessGo.js +4 -2
- package/dist/extension/pipeline/orchestrator.d.ts +16 -1
- package/dist/extension/pipeline/orchestrator.js +15 -1
- package/dist/extension/pipeline/personas.d.ts +2 -1
- package/dist/extension/pipeline/personas.js +7 -6
- package/dist/extension/pipeline/types.d.ts +13 -4
- package/dist/extension/pipeline/types.js +6 -2
- package/package.json +2 -2
|
@@ -13,8 +13,12 @@
|
|
|
13
13
|
* after MAX_REVIEW_ROUNDS rounds, always with an explicit reason.
|
|
14
14
|
*/
|
|
15
15
|
import { MAX_REVIEW_ROUNDS } from "./types.js";
|
|
16
|
-
const SEVERITIES = ["critical", "high", "medium", "low"];
|
|
16
|
+
const SEVERITIES = ["critical", "high", "medium", "low", "needs_person"];
|
|
17
17
|
const DEFAULT_LENS = "correctness";
|
|
18
|
+
/** `NEEDS PERSON`, `needs-person` and `needs_person` all read as `needs_person`. */
|
|
19
|
+
function normalizeSeverity(value) {
|
|
20
|
+
return value.trim().toLowerCase().replace(/[\s-]+/g, "_");
|
|
21
|
+
}
|
|
18
22
|
function isSeverity(value) {
|
|
19
23
|
return SEVERITIES.includes(value);
|
|
20
24
|
}
|
|
@@ -33,7 +37,7 @@ function parsePipeLine(line, lens) {
|
|
|
33
37
|
const parts = line.split("|").map((p) => p.trim());
|
|
34
38
|
if (parts.length < 2)
|
|
35
39
|
return null;
|
|
36
|
-
const severity = parts[0]
|
|
40
|
+
const severity = normalizeSeverity(parts[0]);
|
|
37
41
|
if (!isSeverity(severity))
|
|
38
42
|
return null;
|
|
39
43
|
let finding;
|
|
@@ -81,7 +85,7 @@ function parseJsonFindings(raw, lens) {
|
|
|
81
85
|
if (!item || typeof item !== "object")
|
|
82
86
|
continue;
|
|
83
87
|
const obj = item;
|
|
84
|
-
const severity = typeof obj.severity === "string" ? obj.severity
|
|
88
|
+
const severity = typeof obj.severity === "string" ? normalizeSeverity(obj.severity) : "";
|
|
85
89
|
if (!isSeverity(severity))
|
|
86
90
|
continue;
|
|
87
91
|
const finding = {
|
|
@@ -906,7 +906,7 @@ export function registerGoCommand(pi, deps = {}) {
|
|
|
906
906
|
})()
|
|
907
907
|
: {}),
|
|
908
908
|
openPr: parsed.flags.pr,
|
|
909
|
-
remainingFindings: result.findings.filter((f) => f.severity === "medium" || f.severity === "low"),
|
|
909
|
+
remainingFindings: result.findings.filter((f) => f.severity === "medium" || f.severity === "low" || f.severity === "needs_person"),
|
|
910
910
|
...(finishBaselinePaths.length > 0 ? { baselinePaths: finishBaselinePaths } : {}),
|
|
911
911
|
signal: runSignal,
|
|
912
912
|
});
|
|
@@ -323,10 +323,12 @@ export async function runHeadlessGo(argv, deps = {}) {
|
|
|
323
323
|
},
|
|
324
324
|
onStage: (s) => {
|
|
325
325
|
// `output` carries the plan stage's full text; the stream reports the
|
|
326
|
-
// boundary, never the body.
|
|
326
|
+
// boundary, never the body. `at` is when the boundary happened, so a
|
|
327
|
+
// host that reads the line later (a reattach replays the log) still
|
|
328
|
+
// dates it right.
|
|
327
329
|
const { output: _output, ...boundary } = s;
|
|
328
330
|
if (stream)
|
|
329
|
-
emit({ type: "pipeline_stage", ...boundary });
|
|
331
|
+
emit({ type: "pipeline_stage", ...boundary, at: Date.now() });
|
|
330
332
|
},
|
|
331
333
|
logger: (event, data) => {
|
|
332
334
|
deps.logger?.(event, data);
|
|
@@ -30,7 +30,17 @@ import { type ResilienceAttemptRecord, type RunStageFn } from "./resilience.js";
|
|
|
30
30
|
import { runStage as defaultRunStage } from "./runner.js";
|
|
31
31
|
import { type VerifyOutcome, type WorkstreamCheckResult, type WorkstreamCheckTarget } from "./verify.js";
|
|
32
32
|
import { snapshotWorkspace as defaultSnapshotWorkspace } from "./workspace.js";
|
|
33
|
-
import { type CheckpointStore, type FanoutBudgetVerdict, type FanoutMode, type PipelineStage, type JsonEvent, type PipelineProgress, type PipelineResult, type ResiliencePolicy, type ResumePlan, type StageId, type StageResult, type StageTag } from "./types.js";
|
|
33
|
+
import { type CheckpointStore, type FanoutBudgetVerdict, type FanoutMode, type Finding, type PipelineStage, type JsonEvent, type PipelineProgress, type PipelineResult, type ResiliencePolicy, type ResumePlan, type StageId, type StageResult, type StageTag } from "./types.js";
|
|
34
|
+
/** Each review round's findings as the host stores them: bounded, messages clipped. */
|
|
35
|
+
export declare const MAX_ROUND_ITEMS = 20;
|
|
36
|
+
export interface RoundFindingItem {
|
|
37
|
+
severity: Finding["severity"];
|
|
38
|
+
lens: Finding["lens"];
|
|
39
|
+
message: string;
|
|
40
|
+
file?: string;
|
|
41
|
+
line?: number;
|
|
42
|
+
}
|
|
43
|
+
export declare function roundFindingItems(findings: readonly Finding[]): RoundFindingItem[];
|
|
34
44
|
export interface RunPipelineDeps {
|
|
35
45
|
/** Bounded external design preparation; never runs inside the read-only planner or approved missions. */
|
|
36
46
|
prepareDesign?: (input: {
|
|
@@ -112,6 +122,10 @@ export interface RunPipelineDeps {
|
|
|
112
122
|
* `output` carries a stage's `finalOutput` ONLY on the plan FINISH boundary, so
|
|
113
123
|
* /go can record the plan onto the work item without streaming every stage's
|
|
114
124
|
* full text. It is absent on every other boundary.
|
|
125
|
+
*
|
|
126
|
+
* `items` carries each review round's own findings on the review FINISH
|
|
127
|
+
* boundary (at most {@link MAX_ROUND_ITEMS}, messages clipped), so the host
|
|
128
|
+
* keeps every round and not only the last one.
|
|
115
129
|
*/
|
|
116
130
|
onStage?: (s: {
|
|
117
131
|
stage: StageId;
|
|
@@ -119,6 +133,7 @@ export interface RunPipelineDeps {
|
|
|
119
133
|
round?: number;
|
|
120
134
|
findings?: number;
|
|
121
135
|
output?: string;
|
|
136
|
+
items?: RoundFindingItem[];
|
|
122
137
|
}) => void;
|
|
123
138
|
/**
|
|
124
139
|
* Optional, fail-soft durable journal seam (spec: 2026-06-29-go-resilience).
|
|
@@ -35,6 +35,20 @@ import { makeRunVerify, makeWorkstreamCheck, parseChangedPaths, } from "./verify
|
|
|
35
35
|
import { builderStage, fixerStage, orchestratorStage, partitionReaskStage, PARTITION_CALLER_LABEL, REQUIRED_LENSES, REVIEW_LENSES, reaskStage, reviewStage, selectStages, synthesizerFixStage, synthesizerStage, SYNTHESIZER_CALLER_LABEL, workstreamCallerLabel, } from "./stages.js";
|
|
36
36
|
import { snapshotWorkspace as defaultSnapshotWorkspace, workspaceChanged } from "./workspace.js";
|
|
37
37
|
import { DEFAULT_RESILIENCE_POLICY, MAX_CONCURRENCY, MAX_FANOUT_CONCURRENCY, MAX_FANOUT_CONCURRENCY_ULTRA, MAX_FIX_TURNS, MIN_TOOL_CALLS_FOR_HEALTH, TOOL_ERROR_FAIL_RATE, } from "./types.js";
|
|
38
|
+
/** Each review round's findings as the host stores them: bounded, messages clipped. */
|
|
39
|
+
export const MAX_ROUND_ITEMS = 20;
|
|
40
|
+
const MAX_ROUND_ITEM_MESSAGE = 300;
|
|
41
|
+
export function roundFindingItems(findings) {
|
|
42
|
+
return findings.slice(0, MAX_ROUND_ITEMS).map((finding) => ({
|
|
43
|
+
severity: finding.severity,
|
|
44
|
+
lens: finding.lens,
|
|
45
|
+
message: finding.message.length > MAX_ROUND_ITEM_MESSAGE
|
|
46
|
+
? `${finding.message.slice(0, MAX_ROUND_ITEM_MESSAGE - 1)}…`
|
|
47
|
+
: finding.message,
|
|
48
|
+
...(finding.file ? { file: finding.file } : {}),
|
|
49
|
+
...(finding.line !== undefined ? { line: finding.line } : {}),
|
|
50
|
+
}));
|
|
51
|
+
}
|
|
38
52
|
/** Error thrown when a build stage fails; carries the partial run for the caller. */
|
|
39
53
|
export class PipelineStageError extends Error {
|
|
40
54
|
stageId;
|
|
@@ -1026,7 +1040,7 @@ export async function runPipeline(ticket, deps) {
|
|
|
1026
1040
|
...(degradedLenses ? { degradedLenses } : {}),
|
|
1027
1041
|
});
|
|
1028
1042
|
progress({ kind: "findings", round, total: findings.length, blocking: blockingCount(findings) });
|
|
1029
|
-
stageEvent("review", "finish", { round, findings: findings.length });
|
|
1043
|
+
stageEvent("review", "finish", { round, findings: findings.length, items: roundFindingItems(findings) });
|
|
1030
1044
|
// Honest health check BEFORE trusting the verdict: an aborted or crashed
|
|
1031
1045
|
// review must NOT be read as a clean review (empty output → [] findings →
|
|
1032
1046
|
// would otherwise stop 'clean'). Mirrors the build half's isFailed guard.
|
|
@@ -7,7 +7,8 @@
|
|
|
7
7
|
* - scout / planner: call `ask_yagni` before inferring a convention,
|
|
8
8
|
* - worker: call `record_decision` only for a consequential product-intent or architecture call,
|
|
9
9
|
* - reviewer (business-fit lens): call `review_business_match` and treat a
|
|
10
|
-
* conflict with a recorded decision as at least High
|
|
10
|
+
* conflict with a recorded decision as at least High, and a disagreement
|
|
11
|
+
* with the plan the person approved as needs_person.
|
|
11
12
|
*
|
|
12
13
|
* The implement diamond adds two more roles: `orchestrator` (the read-only
|
|
13
14
|
* partitioner, carrying the ```partition output contract `parsePartition` reads)
|
|
@@ -7,7 +7,8 @@
|
|
|
7
7
|
* - scout / planner: call `ask_yagni` before inferring a convention,
|
|
8
8
|
* - worker: call `record_decision` only for a consequential product-intent or architecture call,
|
|
9
9
|
* - reviewer (business-fit lens): call `review_business_match` and treat a
|
|
10
|
-
* conflict with a recorded decision as at least High
|
|
10
|
+
* conflict with a recorded decision as at least High, and a disagreement
|
|
11
|
+
* with the plan the person approved as needs_person.
|
|
11
12
|
*
|
|
12
13
|
* The implement diamond adds two more roles: `orchestrator` (the read-only
|
|
13
14
|
* partitioner, carrying the ```partition output contract `parsePartition` reads)
|
|
@@ -331,9 +332,9 @@ export const BLIND_PERSONA_BODIES = {
|
|
|
331
332
|
};
|
|
332
333
|
/** The lens-specific clause appended to the reviewer body, one per review angle. */
|
|
333
334
|
const LENS_CLAUSES = {
|
|
334
|
-
correctness: "Lens: CORRECTNESS. Hunt bugs, broken logic, unhandled edge cases, race conditions, and incorrect error handling. A
|
|
335
|
-
business_fit: "Lens: BUSINESS-FIT. This is the only-YAGNI lens. Call review_business_match and consult the decision corpus: does this change match the recorded decisions, conventions, and current priorities of this company? Right code doing the wrong thing is exactly the failure you exist to catch. A conflict with a recorded decision is at least High.",
|
|
336
|
-
does_it_hold: "Lens: DOES-IT-HOLD. Does the change actually accomplish the ticket, and does it build/test as far as read-only bash lets you verify?
|
|
335
|
+
correctness: "Lens: CORRECTNESS. Hunt bugs, broken logic, unhandled edge cases, race conditions, and incorrect error handling. A defect that breaks behavior or data at runtime is at least High. Copy, naming, dead code, and style are Medium or Low.",
|
|
336
|
+
business_fit: "Lens: BUSINESS-FIT. This is the only-YAGNI lens. Call review_business_match and consult the decision corpus: does this change match the recorded decisions, conventions, and current priorities of this company? Right code doing the wrong thing is exactly the failure you exist to catch. A conflict with a recorded decision is at least High. A disagreement with the plan the person approved is needs_person: the person decided it, so only the person can change it.",
|
|
337
|
+
does_it_hold: "Lens: DOES-IT-HOLD. Does the change actually accomplish the ticket, and does it build/test as far as read-only bash lets you verify? A change that does not do the task, or new behavior with no test at all, is at least High. Something only a person can do (a named sign-off, an account, a credential, a step outside the repository) is needs_person, never High.",
|
|
337
338
|
};
|
|
338
339
|
/**
|
|
339
340
|
* The machine-readable findings contract every reviewer must emit so the
|
|
@@ -346,7 +347,7 @@ End your review with a fenced block in EXACTLY this form, one line per finding:
|
|
|
346
347
|
SEVERITY | file:line | message
|
|
347
348
|
\`\`\`
|
|
348
349
|
|
|
349
|
-
SEVERITY is one of: critical, high, medium, low. Use \`file:line\` when you can point to a location; otherwise give a short location or omit it. Put one finding per line and nothing else inside the block. If you found no problems, emit an empty \`\`\`findings block. Only critical and high findings block the change.
|
|
350
|
+
SEVERITY is one of: critical, high, medium, low, needs_person. needs_person means only a person can do it (a named sign-off, an account, a credential, a call outside the code); it does not block, and the person sees it on the pull request. Use \`file:line\` when you can point to a location; otherwise give a short location or omit it. Put one finding per line and nothing else inside the block. If you found no problems, emit an empty \`\`\`findings block. Only critical and high findings block the change.
|
|
350
351
|
|
|
351
352
|
Be terse. Do not narrate your process, restate the diff, or quote code back at length — spend your output on the findings themselves, at most a few short paragraphs before the block. Cap the block at the 12 most important findings, most severe first; a review cut off by its own length limit helps nobody.`;
|
|
352
353
|
/**
|
|
@@ -357,7 +358,7 @@ Be terse. Do not narrate your process, restate the diff, or quote code back at l
|
|
|
357
358
|
*/
|
|
358
359
|
const BLIND_LENS_CLAUSES = {
|
|
359
360
|
correctness: LENS_CLAUSES.correctness,
|
|
360
|
-
business_fit: "Lens: BUSINESS-FIT. Does this change match the apparent product intent and the conventions visible in the code? Right code doing the wrong thing is exactly the failure you exist to catch. A clear mismatch is at least High.",
|
|
361
|
+
business_fit: "Lens: BUSINESS-FIT. Does this change match the apparent product intent and the conventions visible in the code? Right code doing the wrong thing is exactly the failure you exist to catch. A clear mismatch is at least High. A disagreement with the plan the person approved is needs_person.",
|
|
361
362
|
does_it_hold: LENS_CLAUSES.does_it_hold,
|
|
362
363
|
};
|
|
363
364
|
/**
|
|
@@ -35,8 +35,13 @@ export type StageId = "map" | "plan" | "implement" | "review" | "fix";
|
|
|
35
35
|
export type FeedStageId = StageId | "finish";
|
|
36
36
|
/** The three adversarial review angles (spec §3). */
|
|
37
37
|
export type ReviewLens = "correctness" | "business_fit" | "does_it_hold";
|
|
38
|
-
/**
|
|
39
|
-
|
|
38
|
+
/**
|
|
39
|
+
* Finding severities; only `critical` + `high` block the review→fix loop.
|
|
40
|
+
* `needs_person` is something only a person can do (a named sign-off, an
|
|
41
|
+
* account, a credential, a call outside the code): the reviewer assigns it,
|
|
42
|
+
* the fix stage never sees it, and it rides the pull request for the person.
|
|
43
|
+
*/
|
|
44
|
+
export type Severity = "critical" | "high" | "medium" | "low" | "needs_person";
|
|
40
45
|
/**
|
|
41
46
|
* One declarative stage. `model` + `tools` are the tuning surface; `taskTemplate`
|
|
42
47
|
* carries `{ticket}` / `{previous}` placeholders the invocation builder fills.
|
|
@@ -496,8 +501,12 @@ export type PipelineProgress = {
|
|
|
496
501
|
kind: "done";
|
|
497
502
|
stopReason: StopReason;
|
|
498
503
|
};
|
|
499
|
-
/**
|
|
500
|
-
|
|
504
|
+
/**
|
|
505
|
+
* Stop the review→fix loop after at most this many rounds (spec §7.3). A third
|
|
506
|
+
* round never helped: every build that reached it ended with as many findings
|
|
507
|
+
* or more, and hit the cap anyway.
|
|
508
|
+
*/
|
|
509
|
+
export declare const MAX_REVIEW_ROUNDS = 2;
|
|
501
510
|
/**
|
|
502
511
|
* Hard cap on the implement diamond's fix turns (spec decision 7). Verification
|
|
503
512
|
* may re-engage the builders that broke the tree, but only this many times: past
|
|
@@ -6,8 +6,12 @@
|
|
|
6
6
|
* scattered string literals. The stage list is a declarative contract (spec
|
|
7
7
|
* §7.2) so future complexity-lanes select a subset, never a rewrite.
|
|
8
8
|
*/
|
|
9
|
-
/**
|
|
10
|
-
|
|
9
|
+
/**
|
|
10
|
+
* Stop the review→fix loop after at most this many rounds (spec §7.3). A third
|
|
11
|
+
* round never helped: every build that reached it ended with as many findings
|
|
12
|
+
* or more, and hit the cap anyway.
|
|
13
|
+
*/
|
|
14
|
+
export const MAX_REVIEW_ROUNDS = 2;
|
|
11
15
|
/**
|
|
12
16
|
* Hard cap on the implement diamond's fix turns (spec decision 7). Verification
|
|
13
17
|
* may re-engage the builders that broke the tree, but only this many times: past
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@yagni-app/code-staging",
|
|
3
|
-
"version": "1.2.3-staging.
|
|
3
|
+
"version": "1.2.3-staging.1627.1",
|
|
4
4
|
"description": "YAGNI Code: a terminal coding agent that already knows your company. One YAGNI login routes the model and grounds the agent in your team's context.",
|
|
5
5
|
"license": "SEE LICENSE IN LICENSE.md",
|
|
6
6
|
"author": "YAGNI, Inc. <jack@yagni.app> (https://yagni.app)",
|
|
@@ -58,5 +58,5 @@
|
|
|
58
58
|
"turndown": "^7.2.4",
|
|
59
59
|
"typebox": "^1.3.15"
|
|
60
60
|
},
|
|
61
|
-
"yagniSourceSha": "
|
|
61
|
+
"yagniSourceSha": "a0b27bcdb66fa5dfeca35883d6786433cbb85e8e"
|
|
62
62
|
}
|