headlesscode 1.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/ATTRIBUTION.md +53 -0
- package/CODE_OF_CONDUCT.md +130 -0
- package/CONTRIBUTING.md +107 -0
- package/LICENSE +202 -0
- package/README.md +486 -0
- package/SECURITY.md +211 -0
- package/bin/headlesscode.mjs +83 -0
- package/package.json +63 -0
- package/shared/prompts/review-mode-prompt-short.md +93 -0
- package/shared/prompts/review-mode-prompt.md +281 -0
- package/shared/rules-code/rules.md +22 -0
- package/shared/stacks/cpp/rules.md +30 -0
- package/shared/stacks/fastapi/rules.md +30 -0
- package/shared/stacks/javascript/rules.md +37 -0
- package/shared/stacks/postgresql/rules.md +31 -0
- package/shared/stacks/python/rules.md +35 -0
- package/shared/stacks/react/rules.md +11 -0
- package/shared/stacks/typescript/rules.md +10 -0
- package/src/budget/budget.ts +221 -0
- package/src/budget/concurrency.ts +126 -0
- package/src/budget/cost.ts +309 -0
- package/src/budget/index.ts +8 -0
- package/src/checkpoints/cli.ts +256 -0
- package/src/checkpoints/service.ts +227 -0
- package/src/cli.ts +1535 -0
- package/src/cloud/docker-provider.ts +334 -0
- package/src/cloud/provider.ts +300 -0
- package/src/codeintel/call-graph.ts +78 -0
- package/src/codeintel/find-references.ts +123 -0
- package/src/codeintel/go-to-definition.ts +193 -0
- package/src/codeintel/handlers.ts +190 -0
- package/src/codeintel/import-graph.ts +173 -0
- package/src/codeintel/outline.ts +180 -0
- package/src/codeintel/position.ts +77 -0
- package/src/codeintel/program.ts +350 -0
- package/src/codeintel/rename-symbol.ts +213 -0
- package/src/codeintel/tools.ts +280 -0
- package/src/codemap/build.ts +135 -0
- package/src/codemap/cli.ts +190 -0
- package/src/codemap/extract.ts +339 -0
- package/src/codemap/files.ts +236 -0
- package/src/codemap/fingerprint.ts +65 -0
- package/src/codemap/flows.ts +62 -0
- package/src/codemap/html.ts +451 -0
- package/src/codemap/lock.ts +80 -0
- package/src/codemap/types.ts +101 -0
- package/src/codesearch/airunner-embedder.ts +185 -0
- package/src/codesearch/chunk.ts +339 -0
- package/src/codesearch/cli.ts +223 -0
- package/src/codesearch/embedder.ts +332 -0
- package/src/codesearch/files.ts +280 -0
- package/src/codesearch/index.ts +469 -0
- package/src/codesearch/ollama-embedder.ts +205 -0
- package/src/codesearch/search.ts +141 -0
- package/src/codesearch/types.ts +100 -0
- package/src/config/mode-models.ts +218 -0
- package/src/dashboard/aggregate.ts +364 -0
- package/src/dashboard/chat-thread.ts +141 -0
- package/src/dashboard/checkpoints.ts +124 -0
- package/src/dashboard/cli.ts +193 -0
- package/src/dashboard/codemap.ts +44 -0
- package/src/dashboard/files.ts +121 -0
- package/src/dashboard/page.ts +2803 -0
- package/src/dashboard/self-improvement-metrics.ts +282 -0
- package/src/dashboard/server.ts +1103 -0
- package/src/dashboard/session-launch.ts +310 -0
- package/src/dashboard/timeline.ts +273 -0
- package/src/dashboard/tool-exec.ts +107 -0
- package/src/dashboard/trend-cli.ts +141 -0
- package/src/dashboard/trend.ts +413 -0
- package/src/decision-proxy/cli.ts +261 -0
- package/src/decision-proxy/proxy.ts +569 -0
- package/src/deploy/gate-cli.ts +147 -0
- package/src/deploy/gate.ts +254 -0
- package/src/engine/condense.ts +512 -0
- package/src/engine/events.ts +428 -0
- package/src/engine/handoff.ts +71 -0
- package/src/engine/lazy-tools.ts +160 -0
- package/src/engine/local-explore.ts +653 -0
- package/src/engine/logger.ts +96 -0
- package/src/engine/loop.ts +5517 -0
- package/src/engine/parser.ts +347 -0
- package/src/engine/prompt.ts +860 -0
- package/src/engine/reports.ts +47 -0
- package/src/engine/stacks.ts +448 -0
- package/src/engine/types.ts +291 -0
- package/src/engine/usage.ts +186 -0
- package/src/github/app-auth.ts +161 -0
- package/src/github/cli.ts +448 -0
- package/src/github/installations.ts +133 -0
- package/src/github/pr.ts +321 -0
- package/src/github/provision.ts +118 -0
- package/src/github/push.ts +122 -0
- package/src/index-util.ts +50 -0
- package/src/index.ts +81 -0
- package/src/init/cli.ts +248 -0
- package/src/init/gitignore.ts +74 -0
- package/src/llm/ollama.ts +308 -0
- package/src/llm/openrouter.ts +868 -0
- package/src/llm/preflight.ts +367 -0
- package/src/llm/transcript-capture.ts +84 -0
- package/src/memory/embed.ts +110 -0
- package/src/memory/index.ts +22 -0
- package/src/memory/local.ts +259 -0
- package/src/memory/summarizer.ts +283 -0
- package/src/memory/types.ts +153 -0
- package/src/memory/uwuchat.ts +157 -0
- package/src/migrate/cli.ts +115 -0
- package/src/orchestrator/analyze-cli.ts +104 -0
- package/src/orchestrator/auto-split.ts +206 -0
- package/src/orchestrator/cleanup.ts +1003 -0
- package/src/orchestrator/cli.ts +3571 -0
- package/src/orchestrator/cost-estimate.ts +564 -0
- package/src/orchestrator/cost-history-cli.ts +242 -0
- package/src/orchestrator/cost-history.ts +397 -0
- package/src/orchestrator/git-sync.ts +250 -0
- package/src/orchestrator/index.ts +153 -0
- package/src/orchestrator/log-analysis.ts +0 -0
- package/src/orchestrator/merge-check.ts +108 -0
- package/src/orchestrator/pipeline.ts +411 -0
- package/src/orchestrator/resume.ts +1940 -0
- package/src/orchestrator/reviewer.ts +503 -0
- package/src/orchestrator/split.ts +296 -0
- package/src/orchestrator/state.ts +542 -0
- package/src/orchestrator/status.ts +697 -0
- package/src/orchestrator/verification-gate.ts +134 -0
- package/src/orchestrator/watch.ts +898 -0
- package/src/permissions/commands.ts +1083 -0
- package/src/permissions/config.ts +241 -0
- package/src/permissions/index.ts +12 -0
- package/src/permissions/protected-files.ts +96 -0
- package/src/permissions/store-protection.ts +272 -0
- package/src/project-store.ts +648 -0
- package/src/projects/cli.ts +382 -0
- package/src/qa/qa.ts +487 -0
- package/src/tools/browser/handler.ts +346 -0
- package/src/tools/browser/service.ts +406 -0
- package/src/tools/browser/smoke.ts +78 -0
- package/src/tools/browser/tool.ts +99 -0
- package/src/tools/executor.ts +2575 -0
- package/src/tools/language-detect.ts +183 -0
- package/src/tools/output-summarizer.ts +369 -0
- package/src/tools/run-tests.ts +302 -0
- package/src/tools/set-indentation-tool.ts +49 -0
- package/src/tools/test-selection.ts +160 -0
- package/src/vendor/tests/smoke.ts +103 -0
- package/src/vendor/zoo-code/VENDOR-NOTES.md +213 -0
- package/src/vendor/zoo-code/shim/anthropic.ts +71 -0
- package/src/vendor/zoo-code/shim/openai.d.ts +60 -0
- package/src/vendor/zoo-code/shim/os-name.ts +18 -0
- package/src/vendor/zoo-code/shim/strip-bom.ts +14 -0
- package/src/vendor/zoo-code/shim/vscode.ts +76 -0
- package/src/vendor/zoo-code/src/core/config/CustomModesManager.ts +1015 -0
- package/src/vendor/zoo-code/src/core/diff/strategies/multi-search-replace.ts +670 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/capabilities.ts +46 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/custom-instructions.ts +559 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/index.ts +10 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/markdown-formatting.ts +7 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/modes.ts +35 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/objective.ts +13 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/rules.ts +95 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/skills.ts +105 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/system-info.ts +30 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/tool-use-guidelines.ts +9 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/tool-use.ts +7 -0
- package/src/vendor/zoo-code/src/core/prompts/system.ts +176 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/access_mcp_resource.ts +41 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_diff.ts +40 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_patch.ts +61 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/ask_followup_question.ts +62 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/attempt_completion.ts +33 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/codebase_search.ts +43 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/converters.ts +109 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit.ts +48 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit_file.ts +72 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/execute_command.ts +54 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/generate_image.ts +51 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/index.ts +75 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/list_files.ts +41 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/mcp_server.ts +75 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/new_task.ts +39 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_command_output.ts +81 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_file.ts +169 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/run_slash_command.ts +31 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_files.ts +50 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_replace.ts +51 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/skill.ts +33 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/switch_mode.ts +31 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/update_todo_list.ts +54 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/write_to_file.ts +40 -0
- package/src/vendor/zoo-code/src/core/prompts/types.ts +12 -0
- package/src/vendor/zoo-code/src/i18n/index.ts +19 -0
- package/src/vendor/zoo-code/src/integrations/misc/extract-text.ts +81 -0
- package/src/vendor/zoo-code/src/services/checkpoints/RepoPerTaskCheckpointService.ts +15 -0
- package/src/vendor/zoo-code/src/services/checkpoints/ShadowCheckpointService.ts +553 -0
- package/src/vendor/zoo-code/src/services/checkpoints/excludes.ts +212 -0
- package/src/vendor/zoo-code/src/services/checkpoints/index.ts +3 -0
- package/src/vendor/zoo-code/src/services/checkpoints/types.ts +35 -0
- package/src/vendor/zoo-code/src/services/code-index/manager.ts +19 -0
- package/src/vendor/zoo-code/src/services/mcp/McpHub.ts +36 -0
- package/src/vendor/zoo-code/src/services/roo-config/index.ts +441 -0
- package/src/vendor/zoo-code/src/services/search/file-search.ts +143 -0
- package/src/vendor/zoo-code/src/services/skills/SkillsManager.ts +20 -0
- package/src/vendor/zoo-code/src/shared/globalFileNames.ts +9 -0
- package/src/vendor/zoo-code/src/shared/language.ts +43 -0
- package/src/vendor/zoo-code/src/shared/modes.ts +257 -0
- package/src/vendor/zoo-code/src/shared/tools.ts +385 -0
- package/src/vendor/zoo-code/src/utils/fs.ts +39 -0
- package/src/vendor/zoo-code/src/utils/globalContext.ts +22 -0
- package/src/vendor/zoo-code/src/utils/json-schema.ts +16 -0
- package/src/vendor/zoo-code/src/utils/logging.ts +21 -0
- package/src/vendor/zoo-code/src/utils/mcp-name.ts +190 -0
- package/src/vendor/zoo-code/src/utils/object.ts +18 -0
- package/src/vendor/zoo-code/src/utils/path.ts +94 -0
- package/src/vendor/zoo-code/src/utils/shell.ts +376 -0
- package/src/vendor/zoo-code/src/utils/text-normalization.ts +99 -0
- package/src/vendor/zoo-code/types/global-settings.ts +19 -0
- package/src/vendor/zoo-code/types/index.ts +22 -0
- package/src/vendor/zoo-code/types/message.ts +375 -0
- package/src/vendor/zoo-code/types/mode.ts +241 -0
- package/src/vendor/zoo-code/types/todo.ts +19 -0
- package/src/vendor/zoo-code/types/tool-params.ts +116 -0
- package/src/vendor/zoo-code/types/tool.ts +67 -0
- package/src/vendor/zoo-code/types/vscode.ts +84 -0
- package/src/vision/describe.ts +242 -0
- package/src/vision/tool.ts +91 -0
- package/src/watcher/cli.ts +369 -0
- package/src/watcher/github.ts +304 -0
- package/src/watcher/index.ts +59 -0
- package/src/watcher/state.ts +254 -0
- package/src/watcher/watch.ts +562 -0
- package/tsconfig.json +18 -0
|
@@ -0,0 +1,503 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Headless reviewer (Phase 2) — an adversarial fresh-context verification run.
|
|
3
|
+
*
|
|
4
|
+
* The reviewer is a SECOND harness session run against a worker's worktree /
|
|
5
|
+
* branch, with the review-mode checklist
|
|
6
|
+
* (`shared/prompts/review-mode-prompt.md`) used as the system prompt
|
|
7
|
+
* (`systemPromptOverride`). It reuses the reviewer's core principles
|
|
8
|
+
* verbatim in spirit:
|
|
9
|
+
*
|
|
10
|
+
* - fresh context: the reviewer did not write the code and must not give it
|
|
11
|
+
* benefit of the doubt — a report's output is a claim, not evidence,
|
|
12
|
+
* until reproduced;
|
|
13
|
+
* - non-edit discipline: the executor is READ-ONLY (no write_to_file at
|
|
14
|
+
* all), so the reviewer can inspect, re-run commands (tests, `gh`,
|
|
15
|
+
* `git diff`) and report — but can never fix anything itself;
|
|
16
|
+
* - verdict: clean, or findings (which the caller maps to "reopen issue").
|
|
17
|
+
*
|
|
18
|
+
* The verdict is parsed from the session's `attempt_completion` result (or
|
|
19
|
+
* its text-only answer) via `parseReviewResult`, which is exported for unit
|
|
20
|
+
* testing.
|
|
21
|
+
*/
|
|
22
|
+
|
|
23
|
+
import * as fs from "node:fs/promises"
|
|
24
|
+
import * as fsSync from "node:fs"
|
|
25
|
+
import * as path from "node:path"
|
|
26
|
+
import { execFileSync } from "node:child_process"
|
|
27
|
+
import { fileURLToPath } from "node:url"
|
|
28
|
+
|
|
29
|
+
import { HeadlessSession } from "../engine/loop.js"
|
|
30
|
+
import { Logger } from "../engine/logger.js"
|
|
31
|
+
import { OpenRouterClient } from "../llm/openrouter.js"
|
|
32
|
+
import { OllamaClient } from "../llm/ollama.js"
|
|
33
|
+
import { envBoolean, resolvePerModeEnv } from "../cli.js"
|
|
34
|
+
import { createReadOnlyHeadlessExecutor } from "../tools/executor.js"
|
|
35
|
+
import { getNativeTools } from "../vendor/zoo-code/src/core/prompts/tools/native-tools/index.js"
|
|
36
|
+
import { addCustomInstructions } from "../vendor/zoo-code/src/core/prompts/sections/custom-instructions.js"
|
|
37
|
+
import type { ChatTool, LlmClient, SessionResult } from "../engine/types.js"
|
|
38
|
+
import type { SessionBudget } from "../budget/budget.js"
|
|
39
|
+
|
|
40
|
+
/** Default location of the reviewer checklist, relative to the harness repo. */
|
|
41
|
+
export const DEFAULT_REVIEW_PROMPT_PATH = "shared/prompts/review-mode-prompt.md"
|
|
42
|
+
|
|
43
|
+
/** Tools the reviewer may call: read-only + command + completion/reporting. */
|
|
44
|
+
const REVIEW_TOOL_NAMES = new Set([
|
|
45
|
+
"read_file",
|
|
46
|
+
"list_files",
|
|
47
|
+
"execute_command",
|
|
48
|
+
"attempt_completion",
|
|
49
|
+
"ask_followup_question",
|
|
50
|
+
])
|
|
51
|
+
|
|
52
|
+
/** The harness repo root (parent of src/orchestrator). */
|
|
53
|
+
export const HARNESS_ROOT = fileURLToPath(new URL("../..", import.meta.url))
|
|
54
|
+
|
|
55
|
+
export interface ReviewOptions {
|
|
56
|
+
/** The worktree to review (checked out on the worker's branch). */
|
|
57
|
+
workspaceRoot: string
|
|
58
|
+
/** Mode slug for the session (default: deepseek-reviewer). */
|
|
59
|
+
mode?: string
|
|
60
|
+
/** Model id (default: env OPENROUTER_MODEL / client default). */
|
|
61
|
+
model?: string
|
|
62
|
+
/** Path to the reviewer system prompt (default: harness shared/prompts/…). */
|
|
63
|
+
reviewPromptPath?: string
|
|
64
|
+
/** LLM client; inject a fake in tests (default: OpenRouterClient). */
|
|
65
|
+
llmClient?: LlmClient
|
|
66
|
+
/** Review task text (default: built from the workspace/branch). */
|
|
67
|
+
taskText?: string
|
|
68
|
+
/**
|
|
69
|
+
* Issue number(s) this group was actually assigned, when known (the
|
|
70
|
+
* orchestrator always has this in `group.issues`). When provided (and
|
|
71
|
+
* `taskText` is not explicitly overridden), the default task text names
|
|
72
|
+
* them directly instead of asking the reviewer to discover what was
|
|
73
|
+
* worked on via `gh issue list --state closed`. 2026-08-27: verified
|
|
74
|
+
* live — a review session given the generic "figure out which issues
|
|
75
|
+
* were closed" task, against a group whose issue had NOT actually been
|
|
76
|
+
* filed to GitHub (a synthetic `--issues-json` test group, closed
|
|
77
|
+
* locally but with no real issue thread to find), stalled into repeated
|
|
78
|
+
* empty replies ("I need to review the work... let me first understand
|
|
79
|
+
* what issues were closed") across all 3 retry attempts and correctly
|
|
80
|
+
* escalated to NEEDS-HUMAN — the fail-closed path worked, but a review
|
|
81
|
+
* that already knows the issue number shouldn't have to search for it
|
|
82
|
+
* at all. This closes that gap generally, not just for the synthetic
|
|
83
|
+
* case: even in normal use, `gh issue list --state closed` can miss
|
|
84
|
+
* for other reasons (API lag, pagination, label filtering) and there's
|
|
85
|
+
* no reason to re-derive data the caller already has.
|
|
86
|
+
*/
|
|
87
|
+
issues?: number[]
|
|
88
|
+
/**
|
|
89
|
+
* Iteration ceiling — a backstop, not the primary guard (see `budget`
|
|
90
|
+
* below). Default is deliberately generous (200): a review that
|
|
91
|
+
* genuinely needs to re-run tests, re-derive baselines, and check
|
|
92
|
+
* several claims can legitimately need many tool calls, and a tight cap
|
|
93
|
+
* cutting it off mid-work is worse than a real runaway — it silently
|
|
94
|
+
* loses the review's actual findings (see `runReview`'s session-failure
|
|
95
|
+
* fallback below) rather than ending the session cleanly. A real
|
|
96
|
+
* incident: a review that had ALREADY reopened the issue and posted its
|
|
97
|
+
* findings comment (real GitHub side effects, correct) hit the OLD
|
|
98
|
+
* default of 40 right before calling `attempt_completion`, so the
|
|
99
|
+
* orchestrator only saw a synthetic "session error" instead of the
|
|
100
|
+
* real finding text.
|
|
101
|
+
*/
|
|
102
|
+
maxIterations?: number
|
|
103
|
+
/**
|
|
104
|
+
* Cost/duration backstop (SessionBudget's maxCostUsd/maxDurationMs) —
|
|
105
|
+
* the preferred way to bound a review session: a true runaway gets
|
|
106
|
+
* caught by spend or wall-clock time, not by an arbitrary tool-call
|
|
107
|
+
* count that penalizes legitimate thorough work. Omit for no budget
|
|
108
|
+
* (the harness's own defaults apply, if any).
|
|
109
|
+
*/
|
|
110
|
+
budget?: Pick<SessionBudget, "maxCostUsd" | "maxDurationMs">
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
export interface ReviewResult {
|
|
114
|
+
/** Individual findings extracted from the reviewer's final report. */
|
|
115
|
+
findings: string[]
|
|
116
|
+
/**
|
|
117
|
+
* clean = nothing wrong found; finding = reviewer reported a real problem
|
|
118
|
+
* with the WORKER's code; error = the review SESSION ITSELF failed
|
|
119
|
+
* (crashed, hit its own mistake/budget limit) before ever producing a
|
|
120
|
+
* real verdict — deliberately distinct from "finding" so the caller
|
|
121
|
+
* retries the REVIEW, not the worker's already-fine code (see
|
|
122
|
+
* runReviewWithRetries; issue caught live 2026-08-05 — a review session's
|
|
123
|
+
* own bounded-failure was being treated exactly like a code finding and
|
|
124
|
+
* triggered a pointless worker rework cycle).
|
|
125
|
+
*/
|
|
126
|
+
verdict: "clean" | "finding" | "error"
|
|
127
|
+
/** The reviewer's full final summary. */
|
|
128
|
+
summary: string
|
|
129
|
+
/**
|
|
130
|
+
* Issue #34: absolute path to the review session's complete final report
|
|
131
|
+
* (`<workspaceRoot>/.headlesscode/reports/<sessionId>.md`), when the
|
|
132
|
+
* session succeeded and the report write succeeded. The orchestrator
|
|
133
|
+
* persists this so the full reasoning behind a review verdict is one
|
|
134
|
+
* file-read away, not a re-run away. Absent for a session error (no
|
|
135
|
+
* report was ever produced).
|
|
136
|
+
*/
|
|
137
|
+
reportPath?: string
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
/** The reviewer-mode tool schemas (read-only + command, no write tools). */
|
|
141
|
+
export function reviewTools(): ChatTool[] {
|
|
142
|
+
return getNativeTools().filter(
|
|
143
|
+
(t) => t.type === "function" && REVIEW_TOOL_NAMES.has(t.function.name),
|
|
144
|
+
) as unknown as ChatTool[]
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
function currentBranch(workspaceRoot: string): string {
|
|
148
|
+
try {
|
|
149
|
+
return execFileSync("git", ["-C", workspaceRoot, "branch", "--show-current"], {
|
|
150
|
+
encoding: "utf-8",
|
|
151
|
+
timeout: 5000,
|
|
152
|
+
}).trim()
|
|
153
|
+
} catch {
|
|
154
|
+
return "current branch"
|
|
155
|
+
}
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
function defaultTaskText(workspaceRoot: string, issues?: number[]): string {
|
|
159
|
+
const targetLine =
|
|
160
|
+
issues && issues.length > 0
|
|
161
|
+
? `Review the work done in this workspace (branch: ${currentBranch(workspaceRoot)}) for ` +
|
|
162
|
+
`issue${issues.length > 1 ? "s" : ""} ${issues.map((n) => `#${n}`).join(", ")} — that is exactly ` +
|
|
163
|
+
`what this group was assigned, so start there directly (\`gh issue view ${issues[0]}\`, etc.) ` +
|
|
164
|
+
"rather than searching for it. If the issue number doesn't resolve on GitHub (e.g. a locally-" +
|
|
165
|
+
"tracked or synthetic task never filed as a real issue), that is not a review blocker — fall " +
|
|
166
|
+
"back to the worktree's own git log and diff against its base branch as the source of truth " +
|
|
167
|
+
"for what was actually done, and review that directly instead of stalling on the missing issue " +
|
|
168
|
+
"thread.\n\n"
|
|
169
|
+
: `Review the work done in this workspace (branch: ${currentBranch(workspaceRoot)}) ` +
|
|
170
|
+
"exactly as your operating procedure instructs. For each issue the worker closed, read the " +
|
|
171
|
+
"closing report, read the real diff, and re-run every checkable claim yourself.\n\n"
|
|
172
|
+
return (
|
|
173
|
+
targetLine +
|
|
174
|
+
"Any scratch " +
|
|
175
|
+
"you need (probe scripts, temp output captures) goes in `.headlesscode/scratch/` inside this " +
|
|
176
|
+
"workspace — NEVER write to `/tmp` or any other path outside the workspace. When you are " +
|
|
177
|
+
"done, call attempt_completion with a structured summary: a Findings section listing anything " +
|
|
178
|
+
"wrong with file:line evidence (omit or write 'none' if clean), and the final baseline numbers " +
|
|
179
|
+
"you personally confirmed.\n\n" +
|
|
180
|
+
"The VERY LAST LINE of your attempt_completion result must be exactly one of:\n" +
|
|
181
|
+
"VERDICT: CLEAN\n" +
|
|
182
|
+
"VERDICT: FINDING\n" +
|
|
183
|
+
"Nothing else on that line — no prose, no punctuation, no markdown formatting. This is the ONLY " +
|
|
184
|
+
"line the orchestrator parses to decide whether to trigger a rework cycle; everything else in " +
|
|
185
|
+
"your report is for a human reader. Get this exactly right even when the rest of your report " +
|
|
186
|
+
"discusses both clean and problematic findings — the verdict reflects the OVERALL outcome " +
|
|
187
|
+
"(FINDING if you reopened ANY issue, CLEAN only if none needed reopening)."
|
|
188
|
+
)
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
/**
|
|
192
|
+
* Run one headless review session against `workspaceRoot` using the reviewer
|
|
193
|
+
* checklist as the system prompt override and a read-only executor. Returns
|
|
194
|
+
* the parsed verdict/findings/summary. A failed session (LLM error, max
|
|
195
|
+
* iterations, bounded-failure mistake limit) is reported as `verdict:
|
|
196
|
+
* "error"` with a synthetic finding, so the orchestrator never mistakes an
|
|
197
|
+
* inconclusive review for a clean one — but also never mistakes it for a
|
|
198
|
+
* real code finding either (see runReviewWithRetries, which callers should
|
|
199
|
+
* generally use instead of calling this directly).
|
|
200
|
+
*/
|
|
201
|
+
export async function runReview(options: ReviewOptions): Promise<ReviewResult> {
|
|
202
|
+
const {
|
|
203
|
+
workspaceRoot,
|
|
204
|
+
mode = "deepseek-reviewer",
|
|
205
|
+
model,
|
|
206
|
+
reviewPromptPath = `${HARNESS_ROOT}${DEFAULT_REVIEW_PROMPT_PATH}`,
|
|
207
|
+
llmClient,
|
|
208
|
+
taskText,
|
|
209
|
+
issues,
|
|
210
|
+
maxIterations = 200,
|
|
211
|
+
budget,
|
|
212
|
+
} = options
|
|
213
|
+
|
|
214
|
+
// 2026-08-27: verified live — a review session couldn't find the `curlee`
|
|
215
|
+
// compiler at all ("curlee runtime is not available in this environment"),
|
|
216
|
+
// fell back to eyeballing source code instead of running real checks, and
|
|
217
|
+
// separately had no idea `.roo/rules/rules.md` (this workspace's own
|
|
218
|
+
// language/build reference — e.g. joeos's documented curlee compiler
|
|
219
|
+
// location and Curlee syntax rules) existed. Root cause: `systemPromptOverride`
|
|
220
|
+
// is used VERBATIM by loop.ts (`this.config.systemPromptOverride ?? (await
|
|
221
|
+
// buildPrompt(...))`), which means review sessions skip `buildPrompt`
|
|
222
|
+
// entirely and, with it, `addCustomInstructions` — the exact mechanism
|
|
223
|
+
// that splices `.roo/rules/` into a WORKER session's prompt. Review
|
|
224
|
+
// sessions were structurally blind to project-specific guidance that
|
|
225
|
+
// worker sessions always get. Splicing it in here closes that gap the
|
|
226
|
+
// same way buildSystemPrompt already does for workers (see prompt.ts).
|
|
227
|
+
const reviewPromptText = await fs.readFile(reviewPromptPath, "utf-8")
|
|
228
|
+
const workspaceCustomInstructions = await addCustomInstructions("", "", workspaceRoot, mode, {})
|
|
229
|
+
const systemPromptOverride = workspaceCustomInstructions
|
|
230
|
+
? `${reviewPromptText}\n\n${workspaceCustomInstructions}`
|
|
231
|
+
: reviewPromptText
|
|
232
|
+
// Issue #142 follow-up: orchestrate's review pass built its own
|
|
233
|
+
// OpenRouterClient unconditionally, so a local-backend setup (e.g. a
|
|
234
|
+
// review daemon on its own GPU) had no effect on it — only the
|
|
235
|
+
// single-session `--mode` CLI path respected HEADLESSCODE_LOCAL_BACKEND_MODES.
|
|
236
|
+
// Mirror cli.ts's gate here so `mode` (default "deepseek-reviewer")
|
|
237
|
+
// routes to the same local daemon a direct CLI invocation would.
|
|
238
|
+
const useLocalBackend =
|
|
239
|
+
process.env.HEADLESSCODE_CODE_MODE_BACKEND === "ollama" &&
|
|
240
|
+
(process.env.HEADLESSCODE_LOCAL_BACKEND_MODES ?? "code")
|
|
241
|
+
.split(",")
|
|
242
|
+
.map((s) => s.trim())
|
|
243
|
+
.filter(Boolean)
|
|
244
|
+
.includes(mode)
|
|
245
|
+
// 2026-08-27: cli.ts's own useLocalCodeBackend path learned this the hard
|
|
246
|
+
// way (see its effectiveModel doc comment) — every downstream consumer of
|
|
247
|
+
// `model` (session-start logs, cost/usage records, the `request.model`
|
|
248
|
+
// OllamaClient sends, which OllamaClient.resolveModel() prefers over its
|
|
249
|
+
// own defaultModel) must see the LOCAL model id when local backend is
|
|
250
|
+
// active, not the cloud one, or a local review session logs/tags itself
|
|
251
|
+
// as e.g. "deepseek/deepseek-v4-flash-0731" throughout even though it
|
|
252
|
+
// never touches OpenRouter.
|
|
253
|
+
const effectiveModel = useLocalBackend ? (resolvePerModeEnv("HEADLESSCODE_CODE_MODE_MODEL", mode) ?? model) : model
|
|
254
|
+
const client =
|
|
255
|
+
llmClient ??
|
|
256
|
+
(useLocalBackend
|
|
257
|
+
? new OllamaClient({
|
|
258
|
+
baseUrl: resolvePerModeEnv("HEADLESSCODE_OLLAMA_URL", mode),
|
|
259
|
+
defaultModel: effectiveModel,
|
|
260
|
+
// OllamaClient has its own independent abort timer
|
|
261
|
+
// (ollama.ts's DEFAULT_OLLAMA_TIMEOUT_MS, 300s), separate
|
|
262
|
+
// from the llmTimeoutMs passed to HeadlessSession below —
|
|
263
|
+
// verified live 2026-08-28 (cli.ts's LOCAL_LLM_TIMEOUT_MS
|
|
264
|
+
// doc comment has the full story): raising the session-
|
|
265
|
+
// level timeout alone still left review sessions dying at
|
|
266
|
+
// exactly 300000ms because this constructor never heard
|
|
267
|
+
// about it.
|
|
268
|
+
timeoutMs: useLocalBackend ? 630_000 : undefined,
|
|
269
|
+
})
|
|
270
|
+
: new OpenRouterClient({ apiKey: process.env.HEADLESSCODE_OPENROUTER_API_KEY, defaultModel: model }))
|
|
271
|
+
|
|
272
|
+
// Mirror every log line to <worktree>/review.log — the same visibility
|
|
273
|
+
// `run-worker.sh` gives a worker via harness.log (a plain `tail -f`
|
|
274
|
+
// target), which review/QA sessions never had: they run IN-PROCESS
|
|
275
|
+
// inside orchestrate rather than as a spawned subprocess with redirected
|
|
276
|
+
// stdout, so their activity was only ever visible by hand-parsing the
|
|
277
|
+
// structured `.headlesscode/events/*.jsonl` feed. Raised directly
|
|
278
|
+
// 2026-08-05: "frustrating that i can't see the review logs the same way
|
|
279
|
+
// i can the harness logs... i don't like having to hunt them down."
|
|
280
|
+
// Append-only (matches harness.log's own convention across
|
|
281
|
+
// retries/rework re-reviews) with a run-separator line per session.
|
|
282
|
+
const logFilePath = path.join(workspaceRoot, "review.log")
|
|
283
|
+
fsSync.appendFileSync(logFilePath, `\n===== headlesscode review start: ${new Date().toISOString()} =====\n`, "utf-8")
|
|
284
|
+
const logger = new Logger({ level: "info", filePath: logFilePath })
|
|
285
|
+
|
|
286
|
+
const session = new HeadlessSession({
|
|
287
|
+
workspaceRoot,
|
|
288
|
+
mode,
|
|
289
|
+
model: effectiveModel,
|
|
290
|
+
taskText: taskText ?? defaultTaskText(workspaceRoot, issues),
|
|
291
|
+
maxIterations,
|
|
292
|
+
budget,
|
|
293
|
+
systemPromptOverride,
|
|
294
|
+
tools: reviewTools(),
|
|
295
|
+
executor: createReadOnlyHeadlessExecutor(workspaceRoot),
|
|
296
|
+
llmClient: client,
|
|
297
|
+
logger,
|
|
298
|
+
// Issue #144: mirror cli.ts's local-backend cost-tracking skip — a
|
|
299
|
+
// review session on a local daemon (e.g. the 2080 review-daemon) has
|
|
300
|
+
// no real dollar cost either.
|
|
301
|
+
trackCost: !useLocalBackend,
|
|
302
|
+
// 2026-08-27: verified live — a local Qwen3.5-9B review session
|
|
303
|
+
// fabricated a fake `<tool_call>...</tool_call>` text block as its
|
|
304
|
+
// very first reply, never ran a single real verification command, and
|
|
305
|
+
// the harness's bare-text pragmatic-success fallback (see cli.ts's
|
|
306
|
+
// requireExplicitCompletion doc comment) accepted that garbage as a
|
|
307
|
+
// normal `status: "success"` result. Because the session-level result
|
|
308
|
+
// looked like an ordinary success, `parseReviewResult`'s deliberate
|
|
309
|
+
// fail-open default ("no finding marker found -> clean", see its own
|
|
310
|
+
// doc comment) then silently recorded verdict "clean" for a worker
|
|
311
|
+
// whose `git diff --stat` was completely empty — the exact false
|
|
312
|
+
// completion this review stage exists to catch. cli.ts's worker path
|
|
313
|
+
// was already hardened against this for local sessions
|
|
314
|
+
// (requireExplicitCompletion); the review path never got the same
|
|
315
|
+
// treatment, even though it runs on the same daemon and hits the same
|
|
316
|
+
// failure mode. Without this, a garbage local-reviewer reply is
|
|
317
|
+
// silently indistinguishable from a real "verified clean" verdict —
|
|
318
|
+
// forcing it here means a bare/fabricated reply becomes a
|
|
319
|
+
// non-completing mistake (retried, then a genuine session error) so
|
|
320
|
+
// runReviewWithRetries's existing fail-closed handling actually
|
|
321
|
+
// triggers instead of being bypassed.
|
|
322
|
+
requireExplicitCompletion: useLocalBackend && !envBoolean("HEADLESSCODE_ALLOW_TEXT_ONLY_COMPLETION"),
|
|
323
|
+
// Same gap as cli.ts's LOCAL_LLM_TIMEOUT_MS (see its doc comment for
|
|
324
|
+
// the full story): this session construction never set llmTimeoutMs
|
|
325
|
+
// at all, so a review session on the local daemon always used the
|
|
326
|
+
// generic DEFAULT_LLM_TIMEOUT_MS (300s) — shorter than the shim's
|
|
327
|
+
// own deliberately-raised 600s upstream patience, so the harness
|
|
328
|
+
// gives up first on a genuinely slow (not hung) local call.
|
|
329
|
+
llmTimeoutMs: useLocalBackend ? 630_000 : undefined,
|
|
330
|
+
// Same gap as cli.ts's LOCAL_MAX_TOKENS (see its doc comment for the
|
|
331
|
+
// full incident): unset here, a review session on the local daemon
|
|
332
|
+
// falls back to loop.ts's DEFAULT_MAX_TOKENS (32768) — sized for a
|
|
333
|
+
// cloud reasoning model's 128K+ window, not this daemon's real
|
|
334
|
+
// 65,536-token total context, where one runaway generation can crash
|
|
335
|
+
// the whole session outright.
|
|
336
|
+
maxTokens: useLocalBackend ? 8192 : undefined,
|
|
337
|
+
})
|
|
338
|
+
|
|
339
|
+
const result: SessionResult = await session.run()
|
|
340
|
+
if (result.status !== "success" || result.result === undefined) {
|
|
341
|
+
const error = result.error ?? "unknown review session error"
|
|
342
|
+
return {
|
|
343
|
+
findings: [`[review session error] ${error}`],
|
|
344
|
+
verdict: "error",
|
|
345
|
+
summary: `Review session failed: ${error}`,
|
|
346
|
+
}
|
|
347
|
+
}
|
|
348
|
+
return { ...parseReviewResult(result.result), reportPath: result.reportPath }
|
|
349
|
+
}
|
|
350
|
+
|
|
351
|
+
/**
|
|
352
|
+
* Run a review, retrying ONLY when the review SESSION itself failed
|
|
353
|
+
* (verdict "error" — a crash, budget stop, or bounded-failure mistake limit
|
|
354
|
+
* inside the review session), up to `maxRetries` additional attempts. A real
|
|
355
|
+
* "clean" or "finding" verdict is returned immediately, first try — this
|
|
356
|
+
* only guards against the review session's own infrastructure hiccups, not
|
|
357
|
+
* against re-litigating a real finding.
|
|
358
|
+
*
|
|
359
|
+
* Why this exists: `runReview` reports a failed session as `verdict:
|
|
360
|
+
* "error"` rather than silently treating an inconclusive review as clean —
|
|
361
|
+
* correct fail-closed behavior. But the ORIGINAL caller-side handling
|
|
362
|
+
* treated ANY non-clean verdict (including "error") as a real code finding
|
|
363
|
+
* and triggered a full worker rework cycle to "fix" it — pointlessly
|
|
364
|
+
* respawning a worker against a placeholder error message with nothing
|
|
365
|
+
* actionable in it. Caught live 2026-08-05 running issue #17's own round.
|
|
366
|
+
* After `maxRetries` failed attempts, the caller gets the final "error"
|
|
367
|
+
* result back and should escalate to needs-human (a human should look at
|
|
368
|
+
* why review sessions keep failing) rather than reworking the worker.
|
|
369
|
+
*/
|
|
370
|
+
export async function runReviewWithRetries(options: ReviewOptions, maxRetries = 2): Promise<ReviewResult> {
|
|
371
|
+
let last: ReviewResult = { findings: [], verdict: "error", summary: "no attempt made" }
|
|
372
|
+
for (let attempt = 0; attempt <= maxRetries; attempt++) {
|
|
373
|
+
last = await runReview(options)
|
|
374
|
+
if (last.verdict !== "error") {
|
|
375
|
+
return last
|
|
376
|
+
}
|
|
377
|
+
}
|
|
378
|
+
return last
|
|
379
|
+
}
|
|
380
|
+
|
|
381
|
+
/**
|
|
382
|
+
* Parse a reviewer's final report into { findings, verdict, summary }.
|
|
383
|
+
*
|
|
384
|
+
* Verdict heuristics (deterministic, tested):
|
|
385
|
+
* - "finding" if the report contains reopen/finding/problem markers, OR
|
|
386
|
+
* explicit "verdict: finding" — ALWAYS wins, even if the same report
|
|
387
|
+
* also contains clean-sounding language elsewhere (see below);
|
|
388
|
+
* - "clean" if it contains explicit clean markers ("review clean",
|
|
389
|
+
* "verified clean", "no findings", "everything checks out") AND no
|
|
390
|
+
* finding marker;
|
|
391
|
+
* - default "clean" when the report states neither (the reviewer only
|
|
392
|
+
* reopens for real problems, so an unmarked report is treated as clean).
|
|
393
|
+
*
|
|
394
|
+
* A finding marker ALWAYS overrides a clean marker, never the reverse. A
|
|
395
|
+
* real reviewer report reopening one issue while confirming several OTHER
|
|
396
|
+
* claims checked out cleanly is a completely normal report shape (e.g. a
|
|
397
|
+
* "### Verified clean" section listing what's fine, alongside a separate
|
|
398
|
+
* "### Reopened" section naming the real problem) — treating the ambient
|
|
399
|
+
* presence of "verified clean" text as grounds to override an explicit
|
|
400
|
+
* "reopened" elsewhere in the SAME report was a real production bug: it
|
|
401
|
+
* silently classified a genuinely reopened review as "clean", which would
|
|
402
|
+
* have skipped the entire rework mechanism this classification exists to
|
|
403
|
+
* trigger. Never weaken this back to "finding && !clean" — a report is
|
|
404
|
+
* either clean (nothing wrong) or it has findings; it cannot be both, and
|
|
405
|
+
* when in doubt (both markers present), a real finding must win.
|
|
406
|
+
*
|
|
407
|
+
* The finding-marker word list is deliberately narrow — NOT bare words
|
|
408
|
+
* like "failed", "wrong", "regression", "does not", or even bare
|
|
409
|
+
* "reopen(ed)" on its own. Three real production bugs came from
|
|
410
|
+
* progressively narrowing an over-broad list, each caught live in the same
|
|
411
|
+
* session:
|
|
412
|
+
* 1. A clean marker ("verified clean") anywhere overrode an explicit
|
|
413
|
+
* finding elsewhere in the same report (fixed: finding always wins).
|
|
414
|
+
* 2. Bare "failed"/"wrong"/"regression" matched ordinary baseline-
|
|
415
|
+
* reporting prose ("12 failed, 1187 passed" — pre-existing, unrelated
|
|
416
|
+
* to the change) with zero real problem present (fixed: dropped those
|
|
417
|
+
* words entirely, kept only `reopen(ed)`).
|
|
418
|
+
* 3. Bare "reopen(ed)" STILL wasn't safe: a genuinely clean report can
|
|
419
|
+
* explain that something does NOT need action using the word
|
|
420
|
+
* "reopen" itself — "(pre-existing / out of scope, no reopen)",
|
|
421
|
+
* "Staying closed. No reopening warranted." — with zero negation-
|
|
422
|
+
* detection, "reopen" appearing ANYWHERE, including inside a
|
|
423
|
+
* sentence explicitly saying it's NOT happening, still tripped the
|
|
424
|
+
* finding branch.
|
|
425
|
+
*
|
|
426
|
+
* The fix for #3 is not another negation lookbehind (that class of patch
|
|
427
|
+
* — `(?<!no\s)` — already failed once for "failed"; it only protects the
|
|
428
|
+
* EXACT phrase it names, never generalizes to different phrasing). The
|
|
429
|
+
* reliable signal instead is a STRUCTURAL one: `reopen(ed)` only counts
|
|
430
|
+
* when it is
|
|
431
|
+
* (a) a section heading on its own ("### Reopened"), or
|
|
432
|
+
* (b) explicitly tied to a specific issue number within the same
|
|
433
|
+
* sentence ("reopened #83" / "#83 ... reopened" / "issue #83
|
|
434
|
+
* REOPENED"),
|
|
435
|
+
* because the reviewer's own required report format always associates a
|
|
436
|
+
* real reopen with the specific issue number it applies to — an
|
|
437
|
+
* incidental "no reopen" aside never does. Do not go back to a bare
|
|
438
|
+
* `\breopen(ed)?\b` match, and do not add generic words back to this list
|
|
439
|
+
* without a specific report shape that needs them AND a test proving no
|
|
440
|
+
* false-positive on normal baseline/prose/aside language.
|
|
441
|
+
*/
|
|
442
|
+
export function parseReviewResult(text: string): ReviewResult {
|
|
443
|
+
const summary = text.trim()
|
|
444
|
+
|
|
445
|
+
// A real reopen: either an explicit "### Reopened" heading, or
|
|
446
|
+
// "reopen(ed)" tied to a specific issue number within ~30 chars on
|
|
447
|
+
// either side, never crossing a sentence boundary (period/newline) —
|
|
448
|
+
// see the function docstring for why bare "reopen(ed)" isn't enough.
|
|
449
|
+
const REOPENED_HEADING_RE = /(?:^|\n)#{1,6}\s*reopened?\b/i
|
|
450
|
+
const REOPENED_WITH_ISSUE_RE = /#\d+\b[^.\n]{0,30}\breopen(?:ed)?\b|\breopen(?:ed)?\b[^.\n]{0,30}#\d+\b/i
|
|
451
|
+
const VERDICT_FINDING_RE = /\bverdict\b[^.\n]*\bfinding\b/i
|
|
452
|
+
|
|
453
|
+
const isRealFindingLine = (line: string): boolean =>
|
|
454
|
+
REOPENED_HEADING_RE.test(`\n${line}`) || REOPENED_WITH_ISSUE_RE.test(line) || VERDICT_FINDING_RE.test(line)
|
|
455
|
+
|
|
456
|
+
// Extract a "## Findings" / "Findings:" section if present.
|
|
457
|
+
const findings: string[] = []
|
|
458
|
+
const section = summary.match(
|
|
459
|
+
/(?:^|\n)(?:#{1,6}\s*)?findings?\s*:?\s*\n([\s\S]*?)(?=\n#{1,6}\s|\n\s*(?:verdict|summary)\b|\n\s*(?:PR|issues?):|\s*$)/i,
|
|
460
|
+
)
|
|
461
|
+
if (section?.[1]) {
|
|
462
|
+
for (const line of section[1].split("\n")) {
|
|
463
|
+
const item = line.replace(/^[-*\d.\s)\]]+\s*/, "").trim()
|
|
464
|
+
if (item && !/^(verdict|summary)/i.test(item)) {
|
|
465
|
+
findings.push(item)
|
|
466
|
+
}
|
|
467
|
+
}
|
|
468
|
+
}
|
|
469
|
+
// Fallback: any line that names a REAL reopen (tied to an issue number
|
|
470
|
+
// or its own heading) — not just the bare word "finding"/"reopen".
|
|
471
|
+
if (findings.length === 0) {
|
|
472
|
+
for (const line of summary.split("\n")) {
|
|
473
|
+
const item = line.replace(/^[-*\d.\s)\]]+\s*/, "").trim()
|
|
474
|
+
if (isRealFindingLine(item) && item.length > 8) {
|
|
475
|
+
findings.push(item)
|
|
476
|
+
}
|
|
477
|
+
}
|
|
478
|
+
}
|
|
479
|
+
|
|
480
|
+
// Primary path: an explicit, structured "VERDICT: CLEAN"/"VERDICT: FINDING"
|
|
481
|
+
// line (required by defaultTaskText) is authoritative — exact match, no
|
|
482
|
+
// heuristics, so no future report phrasing can ever false-positive it.
|
|
483
|
+
// This exists because free-form regex heuristics on prose proved to have
|
|
484
|
+
// no ceiling on false positives: three separate real incidents (see the
|
|
485
|
+
// function docstring) each needed a NEW fix for a NEW phrasing the
|
|
486
|
+
// previous fix didn't anticipate. A rigid required line has no such
|
|
487
|
+
// ceiling — it's either present and exact, or it's absent.
|
|
488
|
+
const structuredVerdict = summary.match(/^VERDICT:\s*(CLEAN|FINDING)\s*$/im)
|
|
489
|
+
if (structuredVerdict) {
|
|
490
|
+
const verdict: "clean" | "finding" = structuredVerdict[1]!.toUpperCase() === "FINDING" ? "finding" : "clean"
|
|
491
|
+
return { findings: [...new Set(findings)], verdict, summary }
|
|
492
|
+
}
|
|
493
|
+
|
|
494
|
+
// Fallback for a session that didn't emit the required line (an older
|
|
495
|
+
// prompt, a model that ignored the instruction, or a hand-written test
|
|
496
|
+
// fixture) — the heuristic below, kept exactly as hardened by the three
|
|
497
|
+
// past incidents, but no longer the primary path.
|
|
498
|
+
const hasFindingMarker =
|
|
499
|
+
REOPENED_HEADING_RE.test(summary) || REOPENED_WITH_ISSUE_RE.test(summary) || VERDICT_FINDING_RE.test(summary)
|
|
500
|
+
|
|
501
|
+
const verdict: "clean" | "finding" = hasFindingMarker ? "finding" : "clean"
|
|
502
|
+
return { findings: [...new Set(findings)], verdict, summary }
|
|
503
|
+
}
|