@doist/doistbot-cli 1.0.6 → 1.0.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/actions/auth.js +43 -11
- package/dist/actions/doctor.js +4 -2
- package/dist/actions/review.js +11 -14
- package/dist/auth.js +23 -6
- package/dist/config.js +58 -12
- package/package.json +3 -3
- package/sandbox/_node_modules/doistbot-repo-config/dist/index.d.ts +1 -1
- package/sandbox/_node_modules/doistbot-repo-config/dist/index.js +1 -1
- package/sandbox/dist/core/pi.js +31 -1
- package/sandbox/dist/core/prompt-template.js +45 -3
- package/sandbox/dist/core/repository-map.js +24 -0
- package/sandbox/dist/core/tracing.js +9 -0
- package/sandbox/dist/providers/helpers.js +51 -0
- package/sandbox/dist/providers/pi-review.js +7 -1
- package/sandbox/dist/tasks/chat/chat.js +20 -8
- package/sandbox/dist/tasks/issue-summarize/model.js +119 -48
- package/sandbox/dist/tasks/issue-summarize/prompt.js +2 -2
- package/sandbox/dist/tasks/issue-summarize/summarize.js +104 -88
- package/sandbox/dist/tasks/issue-triage/fix-attempt.js +3 -2
- package/sandbox/dist/tasks/issue-triage/hero-group-map.js +2 -0
- package/sandbox/dist/tasks/issue-triage/pr-creator.js +2 -2
- package/sandbox/dist/tasks/issue-triage/prompt.js +2 -2
- package/sandbox/dist/tasks/issue-triage/triage.js +2 -2
- package/sandbox/dist/tasks/persist-logs/llm-query-planner.js +3 -3
- package/sandbox/dist/tasks/review/engines/dedupe.js +107 -42
- package/sandbox/dist/tasks/review/engines/multi-focus.js +229 -119
- package/sandbox/dist/tasks/review/engines/shared.js +36 -1
- package/sandbox/dist/tasks/review/github-comment-format.js +12 -4
- package/sandbox/dist/tasks/review/local.js +6 -1
- package/sandbox/dist/tasks/review/review-summary.js +3 -0
- package/sandbox/dist/tasks/review/review.js +253 -186
- package/sandbox/dist/tasks/review/runner.js +10 -5
- package/sandbox/dist/tasks/review/summary-model.js +40 -18
- package/sandbox/node_modules/doistbot-repo-config/dist/index.d.ts +1 -1
- package/sandbox/node_modules/doistbot-repo-config/dist/index.js +1 -1
- package/sandbox/src/tasks/review/prompts/review-multi-focus-base-prompt.md +1 -0
|
@@ -1,8 +1,8 @@
|
|
|
1
|
-
import { fillPromptTemplate, loadPromptTemplateFile } from '../../core/prompt-template.js';
|
|
1
|
+
import { fillPromptTemplate, loadPromptTemplateFile, resolvePromptFile, } from '../../core/prompt-template.js';
|
|
2
2
|
const PROMPT_FILE = 'issue-triage-prompt.md';
|
|
3
3
|
export function loadPromptTemplate() {
|
|
4
4
|
return loadPromptTemplateFile({
|
|
5
|
-
fileName: PROMPT_FILE,
|
|
5
|
+
fileName: resolvePromptFile(import.meta.url, PROMPT_FILE),
|
|
6
6
|
promptLabel: 'issue triage',
|
|
7
7
|
});
|
|
8
8
|
}
|
|
@@ -3,7 +3,7 @@ import path from 'path';
|
|
|
3
3
|
import { SpanStatusCode } from '@opentelemetry/api';
|
|
4
4
|
import { recordTaskMetrics } from '../../core/datadog-metrics.js';
|
|
5
5
|
import { logger } from '../../core/logging.js';
|
|
6
|
-
import { loadPromptTemplateFile } from '../../core/prompt-template.js';
|
|
6
|
+
import { loadPromptTemplateFile, resolvePromptFile } from '../../core/prompt-template.js';
|
|
7
7
|
import { removeIssueCommentReactionIfPresent } from '../../core/reactions.js';
|
|
8
8
|
import { REPOS_SUBDIR, WORKSPACE_DIR } from '../../core/shared.js';
|
|
9
9
|
import { withPhase } from '../../core/tracing.js';
|
|
@@ -79,7 +79,7 @@ function buildPrompt({ ctx, issue, comments, labelNames, clonedRepositories, dec
|
|
|
79
79
|
OUTPUT_CONTRACT: outputContract,
|
|
80
80
|
FIX_ASSESSMENT_INSTRUCTIONS: enableAutoFix
|
|
81
81
|
? loadPromptTemplateFile({
|
|
82
|
-
fileName: 'fix-assessment-instructions.md',
|
|
82
|
+
fileName: resolvePromptFile(import.meta.url, 'fix-assessment-instructions.md'),
|
|
83
83
|
promptLabel: 'fix-assessment-instructions',
|
|
84
84
|
})
|
|
85
85
|
: '',
|
|
@@ -3,17 +3,17 @@ import path from 'path';
|
|
|
3
3
|
import { logger } from '../../core/logging.js';
|
|
4
4
|
import { invokePiPrompt } from '../../core/pi.js';
|
|
5
5
|
import { scrubPii } from '../../core/pii-scrubber.js';
|
|
6
|
-
import { loadPromptTemplateFile } from '../../core/prompt-template.js';
|
|
6
|
+
import { loadPromptTemplateFile, resolvePromptFile } from '../../core/prompt-template.js';
|
|
7
7
|
import { startDdProxy } from './dd-proxy.js';
|
|
8
8
|
export const PERSIST_LOGS_PROVIDER = 'openai';
|
|
9
9
|
const TIMEOUT_MS = 20 * 60 * 1000; // 20 minutes — investigation can take a while
|
|
10
10
|
function buildPrompt(params, proxyPort) {
|
|
11
11
|
const ddApiReference = loadPromptTemplateFile({
|
|
12
|
-
fileName: 'persist-logs-dd-api-reference.md',
|
|
12
|
+
fileName: resolvePromptFile(import.meta.url, 'persist-logs-dd-api-reference.md'),
|
|
13
13
|
promptLabel: 'persist-logs DD API reference',
|
|
14
14
|
});
|
|
15
15
|
const template = loadPromptTemplateFile({
|
|
16
|
-
fileName: 'persist-logs-investigation-prompt.md',
|
|
16
|
+
fileName: resolvePromptFile(import.meta.url, 'persist-logs-investigation-prompt.md'),
|
|
17
17
|
promptLabel: 'persist-logs investigation prompt',
|
|
18
18
|
});
|
|
19
19
|
const scrubbedTitle = scrubPii(params.issueTitle);
|
|
@@ -1,8 +1,10 @@
|
|
|
1
|
+
import { SpanStatusCode } from '@opentelemetry/api';
|
|
1
2
|
import { logger } from '../../../core/logging.js';
|
|
2
3
|
import { invokePiPrompt } from '../../../core/pi.js';
|
|
3
4
|
import { logPiProgressEvent } from '../../../core/pi-progress-logging.js';
|
|
5
|
+
import { genAiProviderName, withPhase } from '../../../core/tracing.js';
|
|
4
6
|
import { resolveReviewProviderCredential } from '../../../providers/credentials.js';
|
|
5
|
-
import { extractJsonFromResponse } from '../../../providers/helpers.js';
|
|
7
|
+
import { extractJsonFromResponse, readProviderErrorDetails } from '../../../providers/helpers.js';
|
|
6
8
|
import { isDedupeResponse } from '../../../providers/schemas.js';
|
|
7
9
|
const DEDUPE_TIMEOUT_MS = 5 * 60 * 1000;
|
|
8
10
|
const DEDUPE_PROVIDER = 'gemini';
|
|
@@ -14,6 +16,59 @@ const DEDUPE_THINKING_LEVEL = 'medium';
|
|
|
14
16
|
const DEDUPE_MAX_ATTEMPTS = 2;
|
|
15
17
|
const DEDUPE_RETRY_DELAY_MS = 1_000;
|
|
16
18
|
const DEDUPE_SYSTEM_PROMPT = 'You are a JSON-only duplicate-finding classifier. Use only the findings JSON in the user prompt. Do not inspect files, do not search the repository, and do not use tools. Return only valid JSON matching the requested schema.';
|
|
19
|
+
const DEDUPE_PHASE = 'review.dedupe';
|
|
20
|
+
const DEDUPE_PROVIDER_UNAVAILABLE_LOG_TYPE = 'review.dedupe.provider_unavailable';
|
|
21
|
+
function getDedupeProviderRoute(providerConfig) {
|
|
22
|
+
return providerConfig.piProviderId ?? 'google';
|
|
23
|
+
}
|
|
24
|
+
function toDedupeProviderConfig(credential, piProviderId) {
|
|
25
|
+
return {
|
|
26
|
+
provider: DEDUPE_PROVIDER,
|
|
27
|
+
apiKey: credential.apiKey,
|
|
28
|
+
...(piProviderId === 'google' ? {} : { piProviderId }),
|
|
29
|
+
modelName: DEDUPE_MODEL_BY_PI_PROVIDER[piProviderId] ?? DEDUPE_MODEL_BY_PI_PROVIDER.google,
|
|
30
|
+
};
|
|
31
|
+
}
|
|
32
|
+
function buildOpenRouterDedupeProviderConfig(credential) {
|
|
33
|
+
const resolvedCredential = resolveReviewProviderCredential(credential);
|
|
34
|
+
if (!resolvedCredential) {
|
|
35
|
+
return undefined;
|
|
36
|
+
}
|
|
37
|
+
return toDedupeProviderConfig(resolvedCredential, 'openrouter');
|
|
38
|
+
}
|
|
39
|
+
function buildDedupeProviderConfigs({ geminiCredential, openRouterCredential, }) {
|
|
40
|
+
const primaryPiProviderId = geminiCredential.piProviderId ?? 'google';
|
|
41
|
+
const primaryConfig = toDedupeProviderConfig(geminiCredential, primaryPiProviderId);
|
|
42
|
+
const openRouterConfig = primaryPiProviderId === 'openrouter'
|
|
43
|
+
? undefined
|
|
44
|
+
: buildOpenRouterDedupeProviderConfig(openRouterCredential);
|
|
45
|
+
return openRouterConfig ? [primaryConfig, openRouterConfig] : [primaryConfig];
|
|
46
|
+
}
|
|
47
|
+
function logDedupeProviderUnavailable({ providerConfig, attempt, error, shouldRetry, timedOut, }) {
|
|
48
|
+
const details = readProviderErrorDetails(error);
|
|
49
|
+
function log(message, meta) {
|
|
50
|
+
if (shouldRetry) {
|
|
51
|
+
logger.warn(message, meta);
|
|
52
|
+
}
|
|
53
|
+
else {
|
|
54
|
+
logger.error(message, meta);
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
log('Multi-focus dedupe provider unavailable', {
|
|
58
|
+
type: DEDUPE_PROVIDER_UNAVAILABLE_LOG_TYPE,
|
|
59
|
+
provider: providerConfig.provider,
|
|
60
|
+
provider_route: getDedupeProviderRoute(providerConfig),
|
|
61
|
+
model: providerConfig.modelName,
|
|
62
|
+
attempt,
|
|
63
|
+
attempts_total: DEDUPE_MAX_ATTEMPTS,
|
|
64
|
+
retrying: shouldRetry,
|
|
65
|
+
error_code: details.errorCode,
|
|
66
|
+
status: details.status,
|
|
67
|
+
reason: details.reason,
|
|
68
|
+
error: details.rawError,
|
|
69
|
+
timed_out: timedOut ?? false,
|
|
70
|
+
});
|
|
71
|
+
}
|
|
17
72
|
function buildDedupePrompt(findings) {
|
|
18
73
|
const payload = {
|
|
19
74
|
findings: findings.map((finding) => ({
|
|
@@ -110,7 +165,7 @@ export function applyDedupeGroups(findings, groups) {
|
|
|
110
165
|
});
|
|
111
166
|
return findings.filter((finding) => !dropped.has(finding.id));
|
|
112
167
|
}
|
|
113
|
-
export async function dedupeMultiFocusFindings({ findings, availableModels, workspaceDir, geminiCredential, }) {
|
|
168
|
+
export async function dedupeMultiFocusFindings({ findings, availableModels, workspaceDir, geminiCredential, openRouterCredential, }) {
|
|
114
169
|
if (findings.length <= 1) {
|
|
115
170
|
return findings;
|
|
116
171
|
}
|
|
@@ -129,45 +184,58 @@ export async function dedupeMultiFocusFindings({ findings, availableModels, work
|
|
|
129
184
|
});
|
|
130
185
|
throw new Error('Multi-focus dedupe requires a Gemini review credential');
|
|
131
186
|
}
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
187
|
+
return runDedupeAttempts({
|
|
188
|
+
findings,
|
|
189
|
+
workspaceDir,
|
|
190
|
+
providerConfigs: buildDedupeProviderConfigs({
|
|
191
|
+
geminiCredential: credential,
|
|
192
|
+
openRouterCredential,
|
|
193
|
+
}),
|
|
194
|
+
});
|
|
195
|
+
}
|
|
196
|
+
async function runDedupeAttempts({ findings, workspaceDir, providerConfigs, }) {
|
|
139
197
|
for (let attempt = 1; attempt <= DEDUPE_MAX_ATTEMPTS; attempt += 1) {
|
|
198
|
+
const providerConfig = providerConfigs[Math.min(attempt - 1, providerConfigs.length - 1)];
|
|
140
199
|
let result;
|
|
141
200
|
try {
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
201
|
+
// One span per actual invoke: a fail-then-succeed retry shows up as a
|
|
202
|
+
// separate error span and ok span, not one inflated span. Provider/
|
|
203
|
+
// model are known here, so the span carries full gen_ai.*.
|
|
204
|
+
result = await withPhase(DEDUPE_PHASE, {
|
|
205
|
+
'gen_ai.operation.name': 'chat',
|
|
206
|
+
'gen_ai.provider.name': genAiProviderName(providerConfig.provider),
|
|
207
|
+
'gen_ai.request.model': providerConfig.modelName,
|
|
208
|
+
'review.provider_route': getDedupeProviderRoute(providerConfig),
|
|
209
|
+
}, async (span) => {
|
|
210
|
+
const invocation = await invokePiPrompt({
|
|
211
|
+
provider: providerConfig.provider,
|
|
212
|
+
apiKey: providerConfig.apiKey,
|
|
213
|
+
...(providerConfig.piProviderId
|
|
214
|
+
? { piProviderId: providerConfig.piProviderId }
|
|
215
|
+
: {}),
|
|
216
|
+
prompt: buildDedupePrompt(findings),
|
|
217
|
+
cwd: workspaceDir,
|
|
218
|
+
timeoutMs: DEDUPE_TIMEOUT_MS,
|
|
219
|
+
capabilityPreset: 'read-only',
|
|
220
|
+
modelName: providerConfig.modelName,
|
|
221
|
+
thinkingLevel: DEDUPE_THINKING_LEVEL,
|
|
222
|
+
systemPrompt: DEDUPE_SYSTEM_PROMPT,
|
|
223
|
+
onProgress: (event) => {
|
|
224
|
+
logDedupeProgress(providerConfig, event);
|
|
225
|
+
},
|
|
226
|
+
});
|
|
227
|
+
if (!invocation.ok) {
|
|
228
|
+
span.setStatus({
|
|
229
|
+
code: SpanStatusCode.ERROR,
|
|
230
|
+
message: invocation.error ?? 'dedupe request failed',
|
|
231
|
+
});
|
|
232
|
+
}
|
|
233
|
+
return invocation;
|
|
158
234
|
});
|
|
159
235
|
}
|
|
160
236
|
catch (error) {
|
|
161
237
|
const shouldRetry = attempt < DEDUPE_MAX_ATTEMPTS;
|
|
162
|
-
|
|
163
|
-
? 'Multi-focus dedupe request threw, retrying'
|
|
164
|
-
: 'Multi-focus dedupe request threw', {
|
|
165
|
-
provider: providerConfig.provider,
|
|
166
|
-
model: providerConfig.modelName,
|
|
167
|
-
attempt,
|
|
168
|
-
attempts_total: DEDUPE_MAX_ATTEMPTS,
|
|
169
|
-
error: error instanceof Error ? error.message : String(error),
|
|
170
|
-
});
|
|
238
|
+
logDedupeProviderUnavailable({ providerConfig, attempt, error, shouldRetry });
|
|
171
239
|
if (shouldRetry) {
|
|
172
240
|
await waitForDedupeRetry(attempt);
|
|
173
241
|
continue;
|
|
@@ -176,15 +244,12 @@ export async function dedupeMultiFocusFindings({ findings, availableModels, work
|
|
|
176
244
|
}
|
|
177
245
|
if (!result.ok) {
|
|
178
246
|
const shouldRetry = attempt < DEDUPE_MAX_ATTEMPTS;
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
: 'Multi-focus dedupe request failed', {
|
|
182
|
-
provider: providerConfig.provider,
|
|
183
|
-
model: providerConfig.modelName,
|
|
247
|
+
logDedupeProviderUnavailable({
|
|
248
|
+
providerConfig,
|
|
184
249
|
attempt,
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
250
|
+
error: result.error ?? 'dedupe request failed',
|
|
251
|
+
shouldRetry,
|
|
252
|
+
timedOut: result.timedOut,
|
|
188
253
|
});
|
|
189
254
|
if (shouldRetry) {
|
|
190
255
|
await waitForDedupeRetry(attempt);
|
|
@@ -1,4 +1,6 @@
|
|
|
1
|
+
import { SpanStatusCode } from '@opentelemetry/api';
|
|
1
2
|
import { logger } from '../../../core/logging.js';
|
|
3
|
+
import { genAiProviderName, withPhase } from '../../../core/tracing.js';
|
|
2
4
|
import { convertFindingsToComments } from '../comment-mapper.js';
|
|
3
5
|
import { buildReviewOutputFindings } from '../finding-output.js';
|
|
4
6
|
import { buildMultiFocusPrompt, REVIEW_FOCUS_DEFINITIONS, } from '../multi-focus-prompt.js';
|
|
@@ -6,60 +8,120 @@ import { buildFallbackSummary, summarizeReviewFindings } from '../review-summary
|
|
|
6
8
|
import { parseSeverity } from '../severity.js';
|
|
7
9
|
import { buildGeminiFlashReviewSummaryModel } from '../summary-model.js';
|
|
8
10
|
import { dedupeMultiFocusFindings } from './dedupe.js';
|
|
9
|
-
import { getApiKeys, getAvailableModels, getProviderFailures, requestModelReview, } from './shared.js';
|
|
11
|
+
import { buildGeminiFlashCredential, getApiKeys, getAvailableModels, getEnabledReviewModels, getOpenRouterReviewCredential, getProviderFailures, requestModelReview, } from './shared.js';
|
|
10
12
|
const DEFAULT_REVIEW_MODELS_BY_FOCUS = {
|
|
11
|
-
general: ['
|
|
12
|
-
quality: ['
|
|
13
|
-
efficiency: ['
|
|
13
|
+
general: ['deepseek', 'zai', 'openai'],
|
|
14
|
+
quality: ['deepseek', 'zai', 'openai'],
|
|
15
|
+
efficiency: ['deepseek', 'zai', 'openai'],
|
|
14
16
|
tests: ['deepseek', 'zai'],
|
|
15
17
|
standards: ['deepseek', 'zai'],
|
|
16
18
|
};
|
|
19
|
+
// Default Gemini-flash pass on failure. tests/standards keep no fallback unless
|
|
20
|
+
// a caller opts into a wider fallback set.
|
|
21
|
+
const DEFAULT_GEMINI_FLASH_FALLBACK_FOCUSES = [
|
|
22
|
+
'general',
|
|
23
|
+
'quality',
|
|
24
|
+
'efficiency',
|
|
25
|
+
];
|
|
26
|
+
// Open models whose failed pass triggers that fallback.
|
|
27
|
+
const GEMINI_FLASH_FALLBACK_MODELS = ['deepseek', 'zai'];
|
|
17
28
|
const LOW_PRIORITY_SUMMARY_NOTICE = 'I also included a few optional follow-up notes in the details below.';
|
|
29
|
+
// Workflow phase orchestrating the council fan-out; one model span per pass.
|
|
30
|
+
const MULTI_FOCUS_PHASE = 'review.multi-focus';
|
|
31
|
+
const FOCUS_PASS_PHASE = 'review.focus-pass';
|
|
18
32
|
async function runFocusPass({ focus, model, basePrompt, standardsSection, apiKeys, workspaceDir, openRouterSessionId, onProgress, }) {
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
33
|
+
// One model span per council pass. Started concurrently inside the engine's
|
|
34
|
+
// Promise.all, but withPhase captures the active context at span start, so
|
|
35
|
+
// each parents to `review.multi-focus` without nesting under a sibling.
|
|
36
|
+
// Provider is known now; the resolved model only after the invoke (via
|
|
37
|
+
// ModelReviewResult.requestModel), so request.model is stamped on return.
|
|
38
|
+
return withPhase(FOCUS_PASS_PHASE, {
|
|
39
|
+
'gen_ai.operation.name': 'chat',
|
|
40
|
+
'gen_ai.provider.name': genAiProviderName(model),
|
|
41
|
+
'review.focus': focus.name,
|
|
42
|
+
}, async (span) => {
|
|
43
|
+
const prompt = buildMultiFocusPrompt(basePrompt, focus, standardsSection);
|
|
44
|
+
const result = await requestModelReview({
|
|
31
45
|
model,
|
|
46
|
+
prompt,
|
|
47
|
+
apiKeys,
|
|
48
|
+
workspaceDir,
|
|
49
|
+
openRouterSessionId,
|
|
50
|
+
onProgress,
|
|
32
51
|
});
|
|
52
|
+
if (result?.requestModel) {
|
|
53
|
+
span.setAttributes({ 'gen_ai.request.model': result.requestModel });
|
|
54
|
+
}
|
|
55
|
+
if (!result) {
|
|
56
|
+
const error = `No review result returned for ${model}`;
|
|
57
|
+
span.setStatus({ code: SpanStatusCode.ERROR, message: error });
|
|
58
|
+
logger.warn('Multi-focus review pass failed without a result', {
|
|
59
|
+
focus: focus.name,
|
|
60
|
+
model,
|
|
61
|
+
});
|
|
62
|
+
return {
|
|
63
|
+
status: 'failed',
|
|
64
|
+
focus: focus.name,
|
|
65
|
+
model,
|
|
66
|
+
error,
|
|
67
|
+
};
|
|
68
|
+
}
|
|
69
|
+
if (result.error) {
|
|
70
|
+
span.setStatus({ code: SpanStatusCode.ERROR, message: result.error });
|
|
71
|
+
logger.warn('Multi-focus review pass failed', {
|
|
72
|
+
focus: focus.name,
|
|
73
|
+
model,
|
|
74
|
+
error: result.error,
|
|
75
|
+
});
|
|
76
|
+
return {
|
|
77
|
+
status: 'failed',
|
|
78
|
+
focus: focus.name,
|
|
79
|
+
model,
|
|
80
|
+
error: result.error,
|
|
81
|
+
};
|
|
82
|
+
}
|
|
33
83
|
return {
|
|
34
|
-
status: '
|
|
35
|
-
focus: focus.name,
|
|
36
|
-
model,
|
|
37
|
-
error: `No review result returned for ${model}`,
|
|
38
|
-
};
|
|
39
|
-
}
|
|
40
|
-
if (result.error) {
|
|
41
|
-
logger.warn('Multi-focus review pass failed', {
|
|
42
|
-
focus: focus.name,
|
|
43
|
-
model,
|
|
44
|
-
error: result.error,
|
|
45
|
-
});
|
|
46
|
-
return {
|
|
47
|
-
status: 'failed',
|
|
84
|
+
status: 'succeeded',
|
|
48
85
|
focus: focus.name,
|
|
49
86
|
model,
|
|
50
|
-
|
|
87
|
+
summary: result.summary.trim(),
|
|
88
|
+
findings: result.findings.map((finding) => ({
|
|
89
|
+
...finding,
|
|
90
|
+
focus: focus.name,
|
|
91
|
+
})),
|
|
51
92
|
};
|
|
52
|
-
}
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
93
|
+
});
|
|
94
|
+
}
|
|
95
|
+
// Runs a focus's primary passes, then at most one Gemini-flash fallback pass for
|
|
96
|
+
// that focus when a fallback-eligible primary failed and gemini is available.
|
|
97
|
+
async function runFocusGroup({ focus, models, basePrompt, standardsSection, apiKeys, hasGeminiFallback, geminiFallbackFocuses, workspaceDir, openRouterSessionId, onProgress, }) {
|
|
98
|
+
const primaryRuns = await Promise.all(models.map((model) => runFocusPass({
|
|
99
|
+
focus,
|
|
56
100
|
model,
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
101
|
+
basePrompt,
|
|
102
|
+
standardsSection,
|
|
103
|
+
apiKeys,
|
|
104
|
+
workspaceDir,
|
|
105
|
+
openRouterSessionId,
|
|
106
|
+
onProgress,
|
|
107
|
+
})));
|
|
108
|
+
const fallbackEligibleFailed = geminiFallbackFocuses.includes(focus.name) &&
|
|
109
|
+
primaryRuns.some((run) => run.status === 'failed' && GEMINI_FLASH_FALLBACK_MODELS.includes(run.model));
|
|
110
|
+
const flashAlreadyRan = models.includes('gemini');
|
|
111
|
+
if (!fallbackEligibleFailed || flashAlreadyRan || !hasGeminiFallback) {
|
|
112
|
+
return primaryRuns;
|
|
113
|
+
}
|
|
114
|
+
const flashRun = await runFocusPass({
|
|
115
|
+
focus,
|
|
116
|
+
model: 'gemini',
|
|
117
|
+
basePrompt,
|
|
118
|
+
standardsSection,
|
|
119
|
+
apiKeys: { ...apiKeys, gemini: buildGeminiFlashCredential(apiKeys.gemini) },
|
|
120
|
+
workspaceDir,
|
|
121
|
+
openRouterSessionId,
|
|
122
|
+
onProgress,
|
|
123
|
+
});
|
|
124
|
+
return [...primaryRuns, flashRun];
|
|
63
125
|
}
|
|
64
126
|
function routeLowPriorityFindings(findings, placement = 'summary') {
|
|
65
127
|
if (placement === 'inline') {
|
|
@@ -103,9 +165,24 @@ function collectReviewObservations(focusRuns) {
|
|
|
103
165
|
}
|
|
104
166
|
return observations.slice(0, 8);
|
|
105
167
|
}
|
|
106
|
-
function getModelsForFocus(focus, availableModels) {
|
|
107
|
-
const preferredModels = DEFAULT_REVIEW_MODELS_BY_FOCUS[focus.name];
|
|
108
|
-
|
|
168
|
+
function getModelsForFocus(focus, availableModels, modelsByFocus = DEFAULT_REVIEW_MODELS_BY_FOCUS) {
|
|
169
|
+
const preferredModels = modelsByFocus[focus.name] ?? DEFAULT_REVIEW_MODELS_BY_FOCUS[focus.name];
|
|
170
|
+
const models = preferredModels.filter((model) => availableModels.includes(model));
|
|
171
|
+
// Last resort: a focus has no configured open/openai primary but Gemini is
|
|
172
|
+
// available (e.g. a Gemini-only local/CLI setup). Run Gemini so the review
|
|
173
|
+
// still produces a result instead of falling through to `review: null`.
|
|
174
|
+
if (models.length === 0 && availableModels.includes('gemini')) {
|
|
175
|
+
return ['gemini'];
|
|
176
|
+
}
|
|
177
|
+
return models;
|
|
178
|
+
}
|
|
179
|
+
function getEnabledFocusDefinitions(enabledFocuses) {
|
|
180
|
+
if (!enabledFocuses?.length) {
|
|
181
|
+
return REVIEW_FOCUS_DEFINITIONS;
|
|
182
|
+
}
|
|
183
|
+
const enabled = new Set(enabledFocuses);
|
|
184
|
+
const focusDefinitions = REVIEW_FOCUS_DEFINITIONS.filter((focus) => enabled.has(focus.name));
|
|
185
|
+
return focusDefinitions.length > 0 ? focusDefinitions : REVIEW_FOCUS_DEFINITIONS;
|
|
109
186
|
}
|
|
110
187
|
function formatSummaryOnlyFinding(finding) {
|
|
111
188
|
const severity = parseSeverity(finding.body);
|
|
@@ -134,7 +211,7 @@ function buildSummarySynthesisFindings({ inlineFindings, summaryFindings, }) {
|
|
|
134
211
|
...summaryFindings.map((finding) => formatSummaryOnlyFinding(finding)),
|
|
135
212
|
].filter(Boolean)));
|
|
136
213
|
}
|
|
137
|
-
async function buildMultiFocusReviewOutput({ inlineFindings, summaryFindings, lineLookup, prTitle, prBody, reviewObservations,
|
|
214
|
+
async function buildMultiFocusReviewOutput({ inlineFindings, summaryFindings, lineLookup, prTitle, prBody, reviewObservations, summaryModel, summaryFallbackModels, workspaceDir, }) {
|
|
138
215
|
const comments = convertFindingsToComments(inlineFindings, lineLookup);
|
|
139
216
|
const inlineCommentFindings = Array.from(new Set(comments.map((comment) => comment.body.trim()).filter(Boolean)));
|
|
140
217
|
const findings = buildSummarySynthesisFindings({ inlineFindings, summaryFindings });
|
|
@@ -145,10 +222,12 @@ async function buildMultiFocusReviewOutput({ inlineFindings, summaryFindings, li
|
|
|
145
222
|
summaryFindingsCount: summaryFindings.length,
|
|
146
223
|
});
|
|
147
224
|
}
|
|
148
|
-
const synthesizedSummary =
|
|
225
|
+
const synthesizedSummary = summaryModel == null || !hasReviewFindings
|
|
149
226
|
? null
|
|
150
227
|
: await summarizeReviewFindings({
|
|
151
|
-
credential:
|
|
228
|
+
credential: summaryModel.credential,
|
|
229
|
+
provider: summaryModel.provider,
|
|
230
|
+
...(summaryModel.modelName ? { modelName: summaryModel.modelName } : {}),
|
|
152
231
|
workspaceDir,
|
|
153
232
|
prTitle,
|
|
154
233
|
prBody,
|
|
@@ -173,10 +252,14 @@ async function buildMultiFocusReviewOutput({ inlineFindings, summaryFindings, li
|
|
|
173
252
|
}),
|
|
174
253
|
};
|
|
175
254
|
}
|
|
176
|
-
export async function runMultiFocusReviewEngine({ ctx, prompt, lineLookup, prTitle, prBody, lowPriorityFindingPlacement, standardsSection, onProgress, }) {
|
|
255
|
+
export async function runMultiFocusReviewEngine({ ctx, prompt, lineLookup, prTitle, prBody, lowPriorityFindingPlacement, standardsSection, enabledFocuses, onProgress, }) {
|
|
177
256
|
const apiKeys = getApiKeys(ctx);
|
|
178
|
-
const
|
|
257
|
+
const allAvailableModels = getAvailableModels(apiKeys);
|
|
258
|
+
const availableModels = getEnabledReviewModels(apiKeys, ctx.reviewModelsEnabled);
|
|
179
259
|
const metricModels = availableModels.length > 0 ? [...availableModels] : [...ctx.reviewModelsEnabled];
|
|
260
|
+
const focusDefinitions = getEnabledFocusDefinitions(enabledFocuses);
|
|
261
|
+
const hasGeminiFallback = allAvailableModels.includes('gemini');
|
|
262
|
+
const geminiFallbackFocuses = ctx.geminiFallbackFocuses ?? DEFAULT_GEMINI_FLASH_FALLBACK_FOCUSES;
|
|
180
263
|
if (availableModels.length === 0) {
|
|
181
264
|
return {
|
|
182
265
|
review: null,
|
|
@@ -186,81 +269,108 @@ export async function runMultiFocusReviewEngine({ ctx, prompt, lineLookup, prTit
|
|
|
186
269
|
providerFailures: [],
|
|
187
270
|
};
|
|
188
271
|
}
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
272
|
+
// Workflow phase (no gen_ai.*): it fans out N model passes and post-
|
|
273
|
+
// processes them. The per-pass model spans are the gen-AI operations.
|
|
274
|
+
return withPhase(MULTI_FOCUS_PHASE, {
|
|
275
|
+
'review.models': availableModels.length,
|
|
276
|
+
'review.support_models': allAvailableModels.length,
|
|
277
|
+
'review.focuses': focusDefinitions.length,
|
|
278
|
+
}, async (span) => {
|
|
279
|
+
// One group per focus (its primary passes plus an optional Gemini-flash
|
|
280
|
+
// fallback), flattened into a single list of passes.
|
|
281
|
+
const focusGroups = await Promise.all(focusDefinitions.map((focus) => runFocusGroup({
|
|
282
|
+
focus,
|
|
283
|
+
models: getModelsForFocus(focus, availableModels, ctx.reviewModelsByFocus),
|
|
284
|
+
basePrompt: prompt,
|
|
285
|
+
standardsSection,
|
|
286
|
+
apiKeys,
|
|
287
|
+
hasGeminiFallback,
|
|
288
|
+
geminiFallbackFocuses,
|
|
289
|
+
workspaceDir: ctx.workspaceDir,
|
|
290
|
+
openRouterSessionId: ctx.openRouterSessionId,
|
|
291
|
+
onProgress,
|
|
292
|
+
})));
|
|
293
|
+
const focusRuns = focusGroups.flat();
|
|
294
|
+
const successfulRuns = focusRuns.filter((run) => run.status === 'succeeded');
|
|
295
|
+
const providerFailures = getProviderFailures(focusRuns
|
|
296
|
+
.filter((run) => run.status === 'failed')
|
|
297
|
+
.map((run) => ({
|
|
298
|
+
model: run.model,
|
|
299
|
+
summary: '',
|
|
300
|
+
findings: [],
|
|
301
|
+
durationMs: 0,
|
|
302
|
+
error: run.error,
|
|
303
|
+
})));
|
|
304
|
+
span.setAttributes({
|
|
305
|
+
'review.passes': focusRuns.length,
|
|
306
|
+
'review.passes.succeeded': successfulRuns.length,
|
|
307
|
+
});
|
|
308
|
+
if (successfulRuns.length === 0) {
|
|
309
|
+
span.setStatus({
|
|
310
|
+
code: SpanStatusCode.ERROR,
|
|
311
|
+
message: 'all review passes failed',
|
|
312
|
+
});
|
|
313
|
+
return {
|
|
314
|
+
review: null,
|
|
315
|
+
availableModels,
|
|
316
|
+
metricModels,
|
|
317
|
+
mode: 'multi_focus',
|
|
318
|
+
providerFailures,
|
|
319
|
+
};
|
|
320
|
+
}
|
|
321
|
+
const allFindings = successfulRuns.flatMap((run) => run.findings);
|
|
322
|
+
const reviewObservations = collectReviewObservations(successfulRuns);
|
|
323
|
+
let dedupedFindings = allFindings;
|
|
324
|
+
try {
|
|
325
|
+
dedupedFindings = await dedupeMultiFocusFindings({
|
|
326
|
+
findings: allFindings,
|
|
327
|
+
availableModels: allAvailableModels,
|
|
328
|
+
workspaceDir: ctx.workspaceDir,
|
|
329
|
+
geminiCredential: apiKeys.gemini,
|
|
330
|
+
openRouterCredential: getOpenRouterReviewCredential(apiKeys, ctx.apiKeys),
|
|
331
|
+
});
|
|
332
|
+
}
|
|
333
|
+
catch (error) {
|
|
334
|
+
logger.warn('Multi-focus dedupe failed, continuing with raw findings', {
|
|
335
|
+
findingsCount: allFindings.length,
|
|
336
|
+
error: error instanceof Error ? error.message : String(error),
|
|
337
|
+
});
|
|
338
|
+
}
|
|
339
|
+
const routedFindings = routeLowPriorityFindings(dedupedFindings, lowPriorityFindingPlacement);
|
|
340
|
+
const geminiSummaryFallback = buildGeminiFlashReviewSummaryModel(apiKeys.gemini);
|
|
341
|
+
const summaryModel = ctx.reviewSummaryModel ??
|
|
342
|
+
(apiKeys.zai ? { provider: 'zai', credential: apiKeys.zai } : undefined);
|
|
343
|
+
const review = await buildMultiFocusReviewOutput({
|
|
344
|
+
inlineFindings: routedFindings.inlineFindings,
|
|
345
|
+
summaryFindings: routedFindings.summaryFindings,
|
|
346
|
+
lineLookup,
|
|
347
|
+
prTitle,
|
|
348
|
+
prBody,
|
|
349
|
+
reviewObservations,
|
|
350
|
+
summaryModel,
|
|
351
|
+
summaryFallbackModels: ctx.reviewSummaryModel == null && geminiSummaryFallback
|
|
352
|
+
? [geminiSummaryFallback]
|
|
353
|
+
: [],
|
|
354
|
+
workspaceDir: ctx.workspaceDir,
|
|
355
|
+
});
|
|
356
|
+
logger.info('Multi-focus review completed', {
|
|
357
|
+
models: availableModels,
|
|
358
|
+
supportModels: allAvailableModels,
|
|
359
|
+
focuses: focusDefinitions.map((focus) => focus.name),
|
|
360
|
+
successfulRuns: successfulRuns.length,
|
|
361
|
+
findingsCount: allFindings.length,
|
|
362
|
+
dedupedFindingsCount: dedupedFindings.length,
|
|
363
|
+
inlineFindingsCount: routedFindings.inlineFindings.length,
|
|
364
|
+
summaryOnlyFindingsCount: routedFindings.summaryFindings.length,
|
|
365
|
+
reviewObservationsCount: reviewObservations.length,
|
|
366
|
+
commentsCount: review.comments.length,
|
|
367
|
+
});
|
|
210
368
|
return {
|
|
211
|
-
review
|
|
369
|
+
review,
|
|
212
370
|
availableModels,
|
|
213
371
|
metricModels,
|
|
214
372
|
mode: 'multi_focus',
|
|
215
373
|
providerFailures,
|
|
216
374
|
};
|
|
217
|
-
}
|
|
218
|
-
const allFindings = successfulRuns.flatMap((run) => run.findings);
|
|
219
|
-
const reviewObservations = collectReviewObservations(successfulRuns);
|
|
220
|
-
let dedupedFindings = allFindings;
|
|
221
|
-
try {
|
|
222
|
-
dedupedFindings = await dedupeMultiFocusFindings({
|
|
223
|
-
findings: allFindings,
|
|
224
|
-
availableModels,
|
|
225
|
-
workspaceDir: ctx.workspaceDir,
|
|
226
|
-
geminiCredential: apiKeys.gemini,
|
|
227
|
-
});
|
|
228
|
-
}
|
|
229
|
-
catch (error) {
|
|
230
|
-
logger.warn('Multi-focus dedupe failed, continuing with raw findings', {
|
|
231
|
-
findingsCount: allFindings.length,
|
|
232
|
-
error: error instanceof Error ? error.message : String(error),
|
|
233
|
-
});
|
|
234
|
-
}
|
|
235
|
-
const routedFindings = routeLowPriorityFindings(dedupedFindings, lowPriorityFindingPlacement);
|
|
236
|
-
const geminiSummaryFallback = buildGeminiFlashReviewSummaryModel(apiKeys.gemini);
|
|
237
|
-
const review = await buildMultiFocusReviewOutput({
|
|
238
|
-
inlineFindings: routedFindings.inlineFindings,
|
|
239
|
-
summaryFindings: routedFindings.summaryFindings,
|
|
240
|
-
lineLookup,
|
|
241
|
-
prTitle,
|
|
242
|
-
prBody,
|
|
243
|
-
reviewObservations,
|
|
244
|
-
summaryCredential: apiKeys.zai,
|
|
245
|
-
summaryFallbackModels: geminiSummaryFallback ? [geminiSummaryFallback] : [],
|
|
246
|
-
workspaceDir: ctx.workspaceDir,
|
|
247
|
-
});
|
|
248
|
-
logger.info('Multi-focus review completed', {
|
|
249
|
-
models: availableModels,
|
|
250
|
-
focuses: REVIEW_FOCUS_DEFINITIONS.map((focus) => focus.name),
|
|
251
|
-
successfulRuns: successfulRuns.length,
|
|
252
|
-
findingsCount: allFindings.length,
|
|
253
|
-
dedupedFindingsCount: dedupedFindings.length,
|
|
254
|
-
inlineFindingsCount: routedFindings.inlineFindings.length,
|
|
255
|
-
summaryOnlyFindingsCount: routedFindings.summaryFindings.length,
|
|
256
|
-
reviewObservationsCount: reviewObservations.length,
|
|
257
|
-
commentsCount: review.comments.length,
|
|
258
375
|
});
|
|
259
|
-
return {
|
|
260
|
-
review,
|
|
261
|
-
availableModels,
|
|
262
|
-
metricModels,
|
|
263
|
-
mode: 'multi_focus',
|
|
264
|
-
providerFailures,
|
|
265
|
-
};
|
|
266
376
|
}
|